You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium+Scrapy下载PDF文件失败,请求排查问题原因

问题

我开发了一段基于Selenium和Scrapy的爬虫代码,预期功能是从给定的系列链接中定位并下载PDF文档。目前代码能够定位到PDF文件的链接,但无法完成下载操作。以下是我的代码实现:

class DownloaderSpider(scrapy.Spider):

    def __init__(self, *args, **kwargs):
        super(DownloaderSpider, self).__init__(*args, **kwargs)

        # Configure Chrome WebDriver with download preferences
        options = webdriver.ChromeOptions()
        prefs = {
            "download.default_directory": "c:\Users\marti\Downloads\Web Scraper\downloads",
            "download.prompt_for_download": False,
            "plugins.always_open_pdf_externally": True
        }
        options.add_experimental_option("prefs", prefs)
        self.driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options)

    def parse(self, response):
        query = response.meta['query']
        summary = response.meta['summary']
        date = response.meta['date']
        deadline = response.meta['deadline']
        pdf_link = None

        self.driver.get(response.url)
        
        # Wait until the document is fully loaded
        WebDriverWait(self.driver, 15).until(
            lambda driver: driver.execute_script("return document.readyState") == "complete"
        )
        
        if not self.wait_for_stability():
            self.log("Page did not stabilize in time.")
            return
        
        response = HtmlResponse(url=self.driver.current_url, body=self.driver.page_source, encoding='utf-8')

        elements = response.xpath("//tr | //div[not(div)]")
        self.log(f"Found {len(elements)} elements containing text.")
        
        best_match = None
        highest_score = 0
        for element in elements:
            element_text = element.xpath("string(.)").get().strip()
            score = fuzz.partial_ratio(summary.lower(), element_text.lower())
            # Accept element if it contains a matching date (or deadline) in any format
            if score > highest_score and (self.contains_matching_date(element_text, date) or self.contains_matching_date(element_text, deadline)):
                highest_score = score
                best_match = element
        
        if best_match and highest_score >= 0:  # Adjust threshold as needed
            self.log(f"Best match found with score {highest_score}")
            pdf_link = best_match.xpath(".//a[contains(@href, '.pdf')]/@href").get()
            if pdf_link:
                self.log(f"Found PDF link: {pdf_link}")
        if pdf_link:
            pdf_link = response.urljoin(pdf_link)
            try:
                # Use Selenium to click the PDF link and trigger the download
                pdf_element = WebDriverWait(self.driver, 10).until(
                    EC.element_to_be_clickable((By.XPATH, f"//a[contains(@href, '{pdf_link.split('/')[-1]}')]"))
                )
                pdf_element.click()

                # Wait for the file to appear in the download directory
                download_dir = "c:\Users\marti\Downloads\Web Scraper\downloads"
                local_filename = query.replace(' ', '_') + ".pdf"
                local_filepath = os.path.join(download_dir, local_filename)

                timeout = 30  # seconds
                start_time = time.time()
                while not os.path.exists(local_filepath):
                    if time.time() - start_time > timeout:
                        raise Exception("Download timed out.")
                    time.sleep(1)

                self.log(f"Downloaded file {local_filepath}")
            except Exception as e:
                self.log(f"Failed to download file from {pdf_link}: {e}")
        else:
            self.log("No direct PDF link found, checking for next page link.")
            next_page = best_match.xpath(".//a/@href").get() if best_match else None
            if next_page:
                next_page = response.urljoin(next_page)
                self.log(f"Following next page link: {next_page}")
                yield scrapy.Request(next_page, self.parse_next_page, meta={'query': query})

    def parse_next_page(self, response):
        query = response.meta['query']
        
        self.driver.get(response.url)
        
        WebDriverWait(self.driver, 15).until(
            lambda driver: driver.execute_script("return document.readyState") == "complete"
        )
        
        if not self.wait_for_stability():
            self.log("Page did not stabilize in time.")
            return
        
        response = HtmlResponse(url=self.driver.current_url, body=self.driver.page_source, encoding='utf-8')
        
        pdf_link = response.xpath("//a[contains(@href, '.pdf')]/@href").get()
        if pdf_link:
            pdf_link = response.urljoin(pdf_link)
            try:
                # Use Selenium to click the PDF link and trigger the download
                pdf_element = WebDriverWait(self.driver, 10).until(
                    EC.element_to_be_clickable((By.XPATH, f"//a[contains(@href, '{pdf_link.split('/')[-1]}')]"))
                )
                pdf_element.click()

                # Wait for the file to appear in the download directory
                download_dir = "c:\Users\marti\Downloads\Web Scraper\downloads"
                local_filename = query.replace(' ', '_') + ".pdf"
                local_filepath = os.path.join(download_dir, local_filename)

                timeout = 30  # seconds
                start_time = time.time()
                while not os.path.exists(local_filepath):
                    if time.time() - start_time > timeout:
                        raise Exception("Download timed out.")
                    time.sleep(1)

                self.log(f"Downloaded file {local_filepath}")
            except Exception as e:
                self.log(f"Failed to download file from {pdf_link}: {e}")
        else:
            self.log("No PDF link found on next page.")

代码执行流程为:创建并配置Selenium驱动,遍历.csv文档中的链接,定位页面中匹配度最高的条目,若不是下载链接则跳转至新页面并定位首个下载链接,尝试下载。但执行时出现如下报错日志:

2025-03-29 17:31:14 [scrapy.downloadermiddlewares.retry] ERROR: Gave up retrying <GET https://oshanarc.gov.na/procurement> (failed 6 times): [<twisted.python.failure.Failure twisted.internet.error.ConnectionLost: Connection to the other side was lost in a non-clean fashion: Connection lost.>]
2025-03-29 17:31:14 [scrapy.core.scraper] ERROR: Error downloading <GET https://oshanarc.gov.na/procurement>
Traceback (most recent call last):
  File "C:\Users\marti\Downloads\Web Scraper\venv\Lib\site-packages\twisted\internet\defer.py", line 2013, in _inlineCallbacks
    result = context.run(
        cast(Failure, result).throwExceptionIntoGenerator, gen
    )
  File "C:\Users\marti\Downloads\Web Scraper\venv\Lib\site-packages\twisted\python\failure.py", line 467, in throwExceptionIntoGenerator
    return g.throw(self.value.with_traceback(self.tb))
           ~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\marti\Downloads\Web Scraper\venv\Lib\site-packages\scrapy\core\downloader\middleware.py", line 68, in process_request
    return (yield download_func(request, spider))
            ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
twisted.web._newclient.ResponseNeverReceived: [<twisted.python.failure.Failure twisted.internet.error.ConnectionLost: Connection to the other side was lost in a non-clean fashion: Connection lost.>]
2025-03-29 17:31:14 [scrapy.core.engine] INFO: Closing spider (finished)

请问可能是什么问题导致的?


可能的问题原因及解决方向

  • 目标网站反爬拦截:目标网站可能识别出Scrapy的请求特征(如默认User-Agent、高频请求),主动断开连接。可给Scrapy请求添加真实浏览器的User-Agent,配置随机User-Agent中间件;同时增加请求延迟,避免短时间内大量请求触发拦截。
  • 网络环境异常:本地网络到目标网站的连接不稳定,或存在防火墙、代理干扰。可直接用浏览器访问目标链接,验证是否能正常打开;若使用代理,检查代理配置正确性,或切换网络环境测试。
  • Scrapy与Selenium混用冲突:在Scrapy的parse方法中直接调用Selenium的driver.get(),会导致Scrapy原生请求与Selenium浏览器请求并行,引发连接池混乱。建议要么全程用Selenium处理页面请求和下载,要么用Scrapy处理请求,拿到PDF链接后直接用Scrapy下载器或requests库下载,避免两种工具同时操作网络连接。
  • 下载路径转义错误:代码中的Windows路径使用未转义的反斜杠"c:\Users\marti\Downloads\Web Scraper\downloads",会导致路径识别错误,文件无法保存到指定位置。需改成转义反斜杠"c:\\Users\\marti\\Downloads\\Web Scraper\\downloads",或使用正斜杠"c:/Users/marti/Downloads/Web Scraper/downloads"。
  • PDF链接定位逻辑缺陷:用pdf_link.split('/')[-1]匹配元素href,若链接带参数或文件名重复,可能定位不到正确点击元素。可直接用完整PDF链接构造XPath,或通过之前找到的best_match元素定位子链接,而非全局搜索。

内容的提问来源于stack exchange,提问作者42WaysToAnswerThat

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.13 15:15:56