Selenium Chrome可运行但无法下载,抛出Stacktrace异常求助
问题
批量下载某网站独立页面的免费PDF资源时,用Selenium操作前几个页面能正常触发下载,但后续页面打开后无法点击下载按钮,抛出Stacktrace错误。已确认所有目标页面可正常访问,且下载按钮的绝对XPATH路径一致,怀疑被网站反爬机制拦截。
原代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time # 配置Selenium WebDriver options = webdriver.ChromeOptions() # options.add_argument('--headless') # 不需要GUI时可启用无头模式 driver = webdriver.Chrome(options=options) # 定义要遍历的testpaperid范围 start_id = 88690 end_id = 88699 # 示例结束ID,可按需调整 for testpaperid in range(start_id, end_id + 1): try: # 构造当前页面URL url = f'https://www.testpapersfree.com/show.php?testpaperid={testpaperid}' driver.get(url) # 等待下载按钮可点击并触发下载 wait = WebDriverWait(driver, 60) download_link = wait.until(EC.element_to_be_clickable((By.XPATH, '/html/body/div[1]/div/div/section/div[2]/div[2]/a'))) download_link.click() time.sleep(30) # 等待下载完成,可按需调整时长 # 可根据网站下载逻辑添加额外处理代码 except Exception as e: print(f"处理testpaperid {testpaperid}时出错: {e}") driver.quit()
报错信息
An error occurred on testpaperid 88690: Message: Stacktrace: 0 chromedriver 0x0000000100a12004 chromedriver + 4169732 1 chromedriver 0x0000000100a09ff8 chromedriver + 4136952 2 chromedriver 0x000000010065f500 chromedriver + 292096 3 chromedriver 0x00000001006a47a0 chromedriver + 575392 4 chromedriver 0x00000001006df818 chromedriver + 817176 5 chromedriver 0x00000001006985e8 chromedriver + 525800 6 chromedriver 0x00000001006994b8 chromedriver + 529592 7 chromedriver 0x00000001009d8334 chromedriver + 3932980 8 chromedriver 0x00000001009dc970 chromedriver + 3950960 9 chromedriver 0x00000001009c0774 chromedriver + 3835764 10 chromedriver 0x00000001009dd478 chromedriver + 3953784 11 chromedriver 0x00000001009b2ab4 chromedriver + 3779252 12 chromedriver 0x00000001009f9914 chromedriver + 4069652 13 chromedriver 0x00000001009f9a90 chromedriver + 4070032 14 chromedriver 0x0000000100a09c70 chromedriver + 4136048 15 libsystem_pthread.dylib 0x00000001a979bfa8 _pthread_start + 148 16 libsystem_pthread.dylib 0x00000001a9796da0 thread_start + 8
解决方案
针对网站反爬拦截问题,可从以下几个方向优化代码:
添加浏览器伪装配置
给ChromeOptions添加模拟真实浏览器的参数,避免被识别为自动化工具:options = webdriver.ChromeOptions() # 禁用自动化提示 options.add_experimental_option("excludeSwitches", ["enable-automation"]) options.add_experimental_option('useAutomationExtension', False) # 添加真实用户代理 options.add_argument("user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36") # 禁用图片加载,提升速度同时降低被检测概率 options.add_argument("--blink-settings=imagesEnabled=false")模拟人类操作行为
替换固定等待时长为随机等待,增加页面滚动、模拟真实点击等行为:import random from selenium.webdriver.common.action_chains import ActionChains # 页面加载后随机滚动 driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(random.uniform(2,5)) driver.execute_script("window.scrollTo(0, 0);") time.sleep(random.uniform(1,3)) # 等待下载按钮并模拟真实点击 download_link = wait.until(EC.element_to_be_clickable((By.XPATH, '/html/body/div[1]/div/div/section/div[2]/div[2]/a'))) ActionChains(driver).move_to_element(download_link).click().perform() # 随机等待下载触发 time.sleep(random.uniform(10,20))更换元素定位方式
避免使用易失效的绝对XPATH,改用相对定位或其他属性:# 示例:根据按钮文本定位(假设下载按钮文本为"下载PDF") download_link = wait.until(EC.element_to_be_clickable((By.LINK_TEXT, "下载PDF"))) # 或者根据class属性定位 download_link = wait.until(EC.element_to_be_clickable((By.CSS_SELECTOR, "div[class='download-box'] a")))增加请求间隔
在循环中添加随机间隔,避免短时间内频繁请求:for testpaperid in range(start_id, end_id + 1): try: # ... 原有代码 ... except Exception as e: # ... 异常处理 ... # 每次请求后随机等待5-10秒 time.sleep(random.uniform(5,10))捕获具体异常
替换泛用的Exception为Selenium特定异常,方便精准排查问题:from selenium.common.exceptions import TimeoutException, ElementClickInterceptedException, NoSuchElementException # ... 循环内 ... except TimeoutException: print(f"testpaperid {testpaperid}:下载按钮超时未加载") except ElementClickInterceptedException: print(f"testpaperid {testpaperid}:下载按钮被遮挡无法点击") except NoSuchElementException: print(f"testpaperid {testpaperid}:未找到下载按钮") except Exception as e: print(f"testpaperid {testpaperid}:未知错误 {e}")
内容的提问来源于stack exchange,提问作者SleepyDad
相关产品推荐
相关产品推荐

