Python脚本无法读取链接内文档内容,求技术解决方案
问题描述
无法让当前脚本读取抓取到的链接中的实际文档,无法提取文档内的文本内容。尝试过iframe和src,但未成功。此前从未做过此类开发,不知道还能尝试哪些方法。
原代码
from bs4 import BeautifulSoup from selenium import webdriver from urllib.parse import urlparse, parse_qs import io from PyPDF2 import PdfReader # 指定WebDriver路径 driver = webdriver.Chrome("path/to/chromedriver") # 访问目标网站 url = "https://probaterecords.shelbyal.com/shelby/search.do?indexName=opr&templateName=Main&searchQuery=richard+wygle&lq=&searchType=1&regex=%5B%5Ea-zA-Z0-9_%5D&regexwithspaces=%5B%5Ea-zA-Z0-9%5Cs_%5D&regexwithasterisks=%5B%5E*a-zA-Z0-9%5Cs_%5D&sortBy=InstrumentFilter&desc=N&searchable=DisplayName%2CLastName%2CFirstName%2CInstrument%2CRecDate%2CPartyRole%2CDocTypeDesc%2CDocType%2CBook%2CPage%2CLot%2CBlock%2CTownship%2COther%2CFreeform%2COtherName&isPhoneticSearch=&q=richard+wygle&basicSortOrder=InstrumentFilter%7CN&Instrument=&Instrument_select=AND&RecDate=&RecDate=&RecDate_select=AND&LastName=&LastName_select=AND&searchkindLast=StartsLast&FirstName=&FirstName_select=OR&FirstName2=&FirstName2_select=AND&DocTypeDesc=&DocTypeDesc_select=AND&Book=&Book_select=AND&Page=&Page_select=AND&MAPBOOK=&MAPBOOK_select=AND&MAPPAGE=&MAPPAGE_select=AND&Lot%23=&Lot%23_select=AND&Lot=&Lot_select=AND&Block=&Block_select=AND&Section=&Section_select=AND&Township=&Township_select=AND&Range=&Range_select=AND&QT=&QT_select=AND&BQT=&BQT_select=AND&LegacyNum=&LegacyNum_select=AND&advancedSortOrder=InstrumentFilter%7CN" driver.get(url) # 获取页面源码 html = driver.page_source # 解析HTML soup = BeautifulSoup(html, 'html.parser') # 查找所有class为nocolor pphoto的a标签 links = soup.select('a[class="nocolor pphoto"]') # 创建字典存储唯一链接 unique_links = {} for link in links: href = link['href'] if href.startswith('/shelby/search.do?indexName=shelbyimages&lq='): # 拼接完整链接 full_link = 'https://probaterecords.shelbyal.com' + href # 解析链接参数 parsed_url = urlparse(full_link) query_params = parse_qs(parsed_url.query) # 提取文书编号 instrument_number = query_params['lq'][0] # 提取文档类型 options = soup.select('select[name="DocTypeDesc"] option') for option in options: # 检查是否包含deed关键词 if "deeds" in option.get_text().lower(): doc_type = option.get_text() # 将链接存入字典 unique_links[instrument_number] = (full_link, doc_type) # 遍历唯一链接 for instrument_number, link_info in unique_links.items(): full_link, doc_type = link_info # 从链接打开PDF文件 response = requests.get(full_link) pdf_file = io.BytesIO(response.content) pdf_reader = PdfReader(pdf_file) # 获取页数 pages = len(pdf_reader.pages) # 初始化文本存储变量 text = "" # 遍历每页提取文本 for page in pdf_reader.pages: text += page.extract_text() # 打印结果 print("Document Type: ", doc_type) print("Instrument Number: ", instrument_number) print("Text: ", text)
解决方法
1. 补全缺失的依赖导入
代码中使用了requests库但未导入,需在开头添加:
import requests
2. 同步会话Cookie
直接用requests.get无法获取有效PDF,因为网站需要验证会话状态。需将Selenium浏览器的Cookie同步给requests会话:
# 创建requests会话并同步Selenium的Cookie session = requests.Session() for cookie in driver.get_cookies(): session.cookies.set(cookie['name'], cookie['value'])
3. 获取真实PDF链接
你抓取的full_link是跳转页面,不是直接的PDF地址。需用Selenium加载该页面后,提取实际的PDF资源地址:
# 遍历链接时替换原逻辑 for instrument_number, link_info in unique_links.items(): full_link, doc_type = link_info # 用Selenium打开跳转页面 driver.get(full_link) # 等待页面加载(建议用WebDriverWait替代sleep) import time time.sleep(2) # 提取iframe中的PDF真实链接 iframe = driver.find_element_by_tag_name('iframe') pdf_url = iframe.get_attribute('src') # 处理相对路径 if not pdf_url.startswith('http'): pdf_url = 'https://probaterecords.shelbyal.com' + pdf_url # 用带会话的requests获取PDF response = session.get(pdf_url) pdf_file = io.BytesIO(response.content) pdf_reader = PdfReader(pdf_file) # 提取文本(和原代码逻辑一致) text = "" for page in pdf_reader.pages: text += page.extract_text() print("文档类型: ", doc_type) print("文书编号: ", instrument_number) print("提取文本: ", text)
4. 优化等待逻辑(可选)
用time.sleep不够可靠,建议改用Selenium的显式等待:
from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC # 替换time.sleep(2) wait = WebDriverWait(driver, 10) iframe = wait.until(EC.presence_of_element_located((By.TAG_NAME, 'iframe'))) pdf_url = iframe.get_attribute('src')
5. 修复文档类型提取逻辑
原代码会遍历所有下拉选项,导致文档类型提取错误。应从链接所在行的对应元素提取文档类型,例如:
# 替换原文档类型提取逻辑 # 找到当前链接所在的行 row = link.find_parent('tr') # 定位行内的文档类型元素(需根据实际页面结构调整选择器) doc_type_elem = row.select_one('td:nth-child(XX)') # XX为文档类型所在列的索引 if doc_type_elem and "deeds" in doc_type_elem.get_text().lower(): doc_type = doc_type_elem.get_text() unique_links[instrument_number] = (full_link, doc_type)
内容的提问来源于stack exchange,提问作者mason
相关产品推荐
相关产品推荐

