Selenium+Scrapy下载PDF文件失败,请求排查问题原因
问题
我开发了一段基于Selenium和Scrapy的爬虫代码,预期功能是从给定的系列链接中定位并下载PDF文档。目前代码能够定位到PDF文件的链接,但无法完成下载操作。以下是我的代码实现:
class DownloaderSpider(scrapy.Spider): def __init__(self, *args, **kwargs): super(DownloaderSpider, self).__init__(*args, **kwargs) # Configure Chrome WebDriver with download preferences options = webdriver.ChromeOptions() prefs = { "download.default_directory": "c:\Users\marti\Downloads\Web Scraper\downloads", "download.prompt_for_download": False, "plugins.always_open_pdf_externally": True } options.add_experimental_option("prefs", prefs) self.driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options) def parse(self, response): query = response.meta['query'] summary = response.meta['summary'] date = response.meta['date'] deadline = response.meta['deadline'] pdf_link = None self.driver.get(response.url) # Wait until the document is fully loaded WebDriverWait(self.driver, 15).until( lambda driver: driver.execute_script("return document.readyState") == "complete" ) if not self.wait_for_stability(): self.log("Page did not stabilize in time.") return response = HtmlResponse(url=self.driver.current_url, body=self.driver.page_source, encoding='utf-8') elements = response.xpath("//tr | //div[not(div)]") self.log(f"Found {len(elements)} elements containing text.") best_match = None highest_score = 0 for element in elements: element_text = element.xpath("string(.)").get().strip() score = fuzz.partial_ratio(summary.lower(), element_text.lower()) # Accept element if it contains a matching date (or deadline) in any format if score > highest_score and (self.contains_matching_date(element_text, date) or self.contains_matching_date(element_text, deadline)): highest_score = score best_match = element if best_match and highest_score >= 0: # Adjust threshold as needed self.log(f"Best match found with score {highest_score}") pdf_link = best_match.xpath(".//a[contains(@href, '.pdf')]/@href").get() if pdf_link: self.log(f"Found PDF link: {pdf_link}") if pdf_link: pdf_link = response.urljoin(pdf_link) try: # Use Selenium to click the PDF link and trigger the download pdf_element = WebDriverWait(self.driver, 10).until( EC.element_to_be_clickable((By.XPATH, f"//a[contains(@href, '{pdf_link.split('/')[-1]}')]")) ) pdf_element.click() # Wait for the file to appear in the download directory download_dir = "c:\Users\marti\Downloads\Web Scraper\downloads" local_filename = query.replace(' ', '_') + ".pdf" local_filepath = os.path.join(download_dir, local_filename) timeout = 30 # seconds start_time = time.time() while not os.path.exists(local_filepath): if time.time() - start_time > timeout: raise Exception("Download timed out.") time.sleep(1) self.log(f"Downloaded file {local_filepath}") except Exception as e: self.log(f"Failed to download file from {pdf_link}: {e}") else: self.log("No direct PDF link found, checking for next page link.") next_page = best_match.xpath(".//a/@href").get() if best_match else None if next_page: next_page = response.urljoin(next_page) self.log(f"Following next page link: {next_page}") yield scrapy.Request(next_page, self.parse_next_page, meta={'query': query}) def parse_next_page(self, response): query = response.meta['query'] self.driver.get(response.url) WebDriverWait(self.driver, 15).until( lambda driver: driver.execute_script("return document.readyState") == "complete" ) if not self.wait_for_stability(): self.log("Page did not stabilize in time.") return response = HtmlResponse(url=self.driver.current_url, body=self.driver.page_source, encoding='utf-8') pdf_link = response.xpath("//a[contains(@href, '.pdf')]/@href").get() if pdf_link: pdf_link = response.urljoin(pdf_link) try: # Use Selenium to click the PDF link and trigger the download pdf_element = WebDriverWait(self.driver, 10).until( EC.element_to_be_clickable((By.XPATH, f"//a[contains(@href, '{pdf_link.split('/')[-1]}')]")) ) pdf_element.click() # Wait for the file to appear in the download directory download_dir = "c:\Users\marti\Downloads\Web Scraper\downloads" local_filename = query.replace(' ', '_') + ".pdf" local_filepath = os.path.join(download_dir, local_filename) timeout = 30 # seconds start_time = time.time() while not os.path.exists(local_filepath): if time.time() - start_time > timeout: raise Exception("Download timed out.") time.sleep(1) self.log(f"Downloaded file {local_filepath}") except Exception as e: self.log(f"Failed to download file from {pdf_link}: {e}") else: self.log("No PDF link found on next page.")
代码执行流程为:创建并配置Selenium驱动,遍历.csv文档中的链接,定位页面中匹配度最高的条目,若不是下载链接则跳转至新页面并定位首个下载链接,尝试下载。但执行时出现如下报错日志:
2025-03-29 17:31:14 [scrapy.downloadermiddlewares.retry] ERROR: Gave up retrying <GET https://oshanarc.gov.na/procurement> (failed 6 times): [<twisted.python.failure.Failure twisted.internet.error.ConnectionLost: Connection to the other side was lost in a non-clean fashion: Connection lost.>] 2025-03-29 17:31:14 [scrapy.core.scraper] ERROR: Error downloading <GET https://oshanarc.gov.na/procurement> Traceback (most recent call last): File "C:\Users\marti\Downloads\Web Scraper\venv\Lib\site-packages\twisted\internet\defer.py", line 2013, in _inlineCallbacks result = context.run( cast(Failure, result).throwExceptionIntoGenerator, gen ) File "C:\Users\marti\Downloads\Web Scraper\venv\Lib\site-packages\twisted\python\failure.py", line 467, in throwExceptionIntoGenerator return g.throw(self.value.with_traceback(self.tb)) ~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\marti\Downloads\Web Scraper\venv\Lib\site-packages\scrapy\core\downloader\middleware.py", line 68, in process_request return (yield download_func(request, spider)) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ twisted.web._newclient.ResponseNeverReceived: [<twisted.python.failure.Failure twisted.internet.error.ConnectionLost: Connection to the other side was lost in a non-clean fashion: Connection lost.>] 2025-03-29 17:31:14 [scrapy.core.engine] INFO: Closing spider (finished)
请问可能是什么问题导致的?
可能的问题原因及解决方向
- 目标网站反爬拦截:目标网站可能识别出Scrapy的请求特征(如默认User-Agent、高频请求),主动断开连接。可给Scrapy请求添加真实浏览器的User-Agent,配置随机User-Agent中间件;同时增加请求延迟,避免短时间内大量请求触发拦截。
- 网络环境异常:本地网络到目标网站的连接不稳定,或存在防火墙、代理干扰。可直接用浏览器访问目标链接,验证是否能正常打开;若使用代理,检查代理配置正确性,或切换网络环境测试。
- Scrapy与Selenium混用冲突:在Scrapy的parse方法中直接调用Selenium的
driver.get(),会导致Scrapy原生请求与Selenium浏览器请求并行,引发连接池混乱。建议要么全程用Selenium处理页面请求和下载,要么用Scrapy处理请求,拿到PDF链接后直接用Scrapy下载器或requests库下载,避免两种工具同时操作网络连接。 - 下载路径转义错误:代码中的Windows路径使用未转义的反斜杠
"c:\Users\marti\Downloads\Web Scraper\downloads",会导致路径识别错误,文件无法保存到指定位置。需改成转义反斜杠"c:\\Users\\marti\\Downloads\\Web Scraper\\downloads",或使用正斜杠"c:/Users/marti/Downloads/Web Scraper/downloads"。 - PDF链接定位逻辑缺陷:用
pdf_link.split('/')[-1]匹配元素href,若链接带参数或文件名重复,可能定位不到正确点击元素。可直接用完整PDF链接构造XPath,或通过之前找到的best_match元素定位子链接,而非全局搜索。
内容的提问来源于stack exchange,提问作者42WaysToAnswerThat
相关产品推荐
相关产品推荐

