You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

结合Selenium与requests爬取期刊PDF下载后为空或损坏问题求助

问题描述

我正在开发一个自动化项目,结合Selenium与requests库爬取数字图书馆的期刊类PDF资源。脚本可成功执行下载流程,但下载的PDF文件打开时会弹出报错。该脚本连接我校WiFi时运行正常(该数字资源为学校提供,需使用学校账号登录授权),且朋友在其设备上运行也正常,我的chromedriver已更新至最新版本,推测问题和授权验证或cookie处理有关。

报错信息

Adobe Acrobat error message

原代码
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.action_chains import ActionChains
from selenium.common.exceptions import TimeoutException
from selenium.common.exceptions import NoSuchElementException
import time
import requests as req


def check_need_to_sign_in():
    new_tab = driver.window_handles
    driver.switch_to.window(str(new_tab[-1]))
    url = driver.current_url
    i = 0
    try:
        WebDriverWait(driver, 5).until(EC.element_to_be_clickable(
            (By.XPATH, "//input[@class='form-control ltr_override input ext-input text-box ext-text-box']")))
        print("Sign-in necessary")
        sign_in_url = driver.current_url
        print(sign_in_url)
        for cookie in driver.get_cookies():
            print(cookie)
        cookie_list = list(map(lambda h: h.get('name')+'='+h.get('value')+'; ', driver.get_cookies()))
        cookie_string = ''.join(cookie_list)
        print(cookie_string)
        headers = {}
        headers["Cookie"] = cookie_string
        s = req.session()
        s.headers.update(headers)
        sign_in()
        response = s.get(url, verify=False)
        while i < 1:
            print(response.ok)
            if response.ok == True:
                with open(f"{article_title[13:]}.pdf", 'wb') as f:
                    f.write(response.content)
                i = + 1
    except TimeoutException:
        print("No need to sign-in")
        url = driver.current_url
        response = req.get(url, verify=False)
        while i < 1:
            print(response.ok)
            if response.ok == True:
                with open(f"{article_title[13:]}.pdf", 'wb') as f:
                    f.write(response.content)
                i = + 1


def sign_in():
    new_tab = driver.window_handles
    driver.switch_to.window(str(new_tab[-1]))
    email_fill = driver.find_element(By.XPATH, "//input[@type='email']")
    email_fill.send_keys("my email")
    email_fill.send_keys(Keys.RETURN)
    password_fill = driver.find_element(By.XPATH, "//input[@type='password']")
    password_fill.send_keys("my password")
    time.sleep(6)  # necessary sleep
    stay_signed_in = driver.find_element(By.XPATH, "//input[@type='password']")
    stay_signed_in.send_keys(Keys.RETURN)
    stay_signed_in = driver.find_element(By.XPATH, "//input[@type='submit']")  # avoid stale element error
    time.sleep(3)
    stay_signed_in = driver.find_element(By.XPATH, "//input[@type='submit']")  # avoid stale element error
    ActionChains(driver).move_to_element(stay_signed_in).click(stay_signed_in).perform()



PATH = "/Applications/chromedriver"
ser = Service(PATH)
chromeOptions = webdriver.ChromeOptions()
prefs = {"plugins.always_open_pdf_externally": True}
chromeOptions.add_experimental_option("prefs",prefs)
driver = webdriver.Chrome(service=ser,options=chromeOptions)
driver.maximize_window()
driver.implicitly_wait(20)


driver.get("https://browzine.com/libraries/1374/subjects")
print("Enter targeted Journal name:")
targeted_journal = input()
wait = WebDriverWait(driver, 10)

try:
    button = wait.until(EC.element_to_be_clickable(
        (By.XPATH, "//input[@class='hero-search ember-text-field ember-view']")))
    ActionChains(driver).move_to_element(button).click(button).perform()
    button.send_keys(targeted_journal)
    button.send_keys(Keys.RETURN)
finally:
    pass

timeout = 3
try:
    click_journal = driver.find_element(By.XPATH, "//li[@class='result journal first-result ']")
    ActionChains(driver).move_to_element(click_journal).click(click_journal).perform()
    journal_title = click_journal.find_element(By.XPATH, ".//div[@class='text']").get_attribute("title")
    print(journal_title)
    parent_tab = driver.current_window_handle
    years_available = driver.find_elements(By.XPATH, "//div[@class='year  tabindex' or @class='year selected tabindex']")
    for year in years_available:
        ActionChains(driver).move_to_element(year).click(year).perform()
        acting_on_year = year.text
        print("acting on the year " + acting_on_year)
        issues_container_block = driver.find_element(By.XPATH, "//div[@class='back-issue-items']")
        issues_available = issues_container_block.find_elements(By.XPATH, "//div[@class='issue active-override ember-view' or @class='issue ember-view']")
        for single_issue in issues_available:
            ActionChains(driver).move_to_element(single_issue).click(single_issue).perform()
            articles_in_issue = driver.find_elements(By.XPATH, "//section[@class='article-list-item-content-block ']")
            for article in articles_in_issue:
                article_title = article.get_attribute("aria-label")
                check_pdf_button = article.find_element(By.XPATH, ".//span[@class='icon fal fa-file-pdf']")
                if len(str(check_pdf_button))>0:
                    pdf_icon_of_article = article.find_element(By.XPATH,".//span[@class='icon fal fa-file-pdf']")
                    ActionChains(driver).move_to_element(pdf_icon_of_article).click(pdf_icon_of_article).perform()
                    check_need_to_sign_in()
                    driver.close()
                    driver.switch_to.window(parent_tab)
                elif NoSuchElementException:
                    print("No PDF icon")

                    pass
            continue
        continue
finally:
    pass

driver.quit()
问题原因
  • 核心问题是cookie收集时机错误:你在检测到需要登录、还未执行登录操作时就收集了当前浏览器的cookie,这部分cookie没有包含登录后的授权凭证,用这组cookie请求PDF,实际返回的是登录页面的HTML文本,存成.pdf后缀自然无法被PDF阅读器识别。
  • 请求头缺失:仅携带Cookie字段不足,大部分数字资源站点会校验User-Agent、Referer等请求头,缺失会被判定为非法请求返回错误页面。
  • 登录后无等待逻辑:执行sign_in登录操作后没有等待页面跳转、cookie更新完成就直接发请求,拿到的还是登录前的无效内容。
  • 登录后未重新获取cookie:登录完成后浏览器已经更新了授权cookie,但你没有重新读取就直接用之前收集的旧cookie发请求。
修复方案

核心代码修改(check_need_to_sign_in函数)

def check_need_to_sign_in():
    new_tab = driver.window_handles
    driver.switch_to.window(str(new_tab[-1]))
    url = driver.current_url
    headers = {
        "User-Agent": driver.execute_script("return navigator.userAgent;"),
        "Referer": driver.current_url
    }
    s = req.session()
    s.verify = False
    try:
        WebDriverWait(driver, 5).until(EC.element_to_be_clickable(
            (By.XPATH, "//input[@class='form-control ltr_override input ext-input text-box ext-text-box']")))
        print("Sign-in necessary")
        # 先执行登录
        sign_in()
        # 登录完成后等待页面跳转,再收集最新cookie
        WebDriverWait(driver, 10).until(EC.url_changes(driver.current_url))
        time.sleep(2)
    except TimeoutException:
        print("No need to sign-in")
    # 统一收集最新cookie
    cookie_list = list(map(lambda h: h.get('name')+'='+h.get('value')+'; ', driver.get_cookies()))
    cookie_string = ''.join(cookie_list)
    headers["Cookie"] = cookie_string
    s.headers.update(headers)
    response = s.get(url)
    # 校验返回内容是否为PDF
    if response.ok and 'application/pdf' in response.headers.get('Content-Type', ''):
        with open(f"{article_title[13:]}.pdf", 'wb') as f:
            f.write(response.content)
        print(f"{article_title[13:]} 下载成功")
    else:
        print(f"{article_title[13:]} 下载失败,返回内容不是PDF")

其他调整

  1. 把sign_in函数里的固定等待优化成元素等待,避免等待时间不足导致登录失败:把time.sleep(6)换成等待密码输入框可点击的逻辑,提交后等待跳转即可。
  2. 新增Content-Type校验,避免把错误页面存成PDF文件。

内容的提问来源于stack exchange,提问作者double_wizz

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.24 10:36:03