使用Python Selenium无法访问网页元素下载PDF文件求助
问题描述
我无法访问目标PDF页面的任何元素:
目标PDF链接:https://www.eaton.com/content/dam/eaton/products/conduit-cable-and-wire-management/crouse-hinds/catalog-pages/crouse-hinds-locknuts-nipples-washers-reducers-plugs-rigid-catalog-page.pdf
尝试点击下载图标完成文件下载时,总是出现「未找到元素」错误;若等待元素加载则触发「超时异常」,已确认页面不存在iframe。需要帮助用Python Selenium完成该文件下载,当前脚本如下:
from selenium import webdriver import os.path # to get Employee ID import time from selenium.webdriver.chrome.options import Options # to change the download location from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.common.by import By # Get the Employee ID EmplyID = os.getenv('USER', os.getenv('USERNAME', 'user')) # Get the final download location DownloadPath = (f'C:\\Users\\{EmplyID}\\Desktop\\Catalog files\\Downloaded Files') chromeOptions = Options() chromeOptions.add_experimental_option("prefs", {"download.default_directory": DownloadPath}) # create a driver object driver = webdriver.Chrome(executable_path=f'C:\\Users\\{EmplyID}\\Desktop\\Catalog files\\chromedriver.exe',options=chromeOptions) # open the pdf driver.get("https://www.eaton.com/content/dam/eaton/products/conduit-cable-and-wire-management/crouse-hinds/catalog-pages/crouse-hinds-locknuts-nipples-washers-reducers-plugs-rigid-catalog-page.pdf") driver.maximize_window() time.sleep(10) # creating for explicit wait wait = WebDriverWait(driver,15) element = wait.until(EC.presence_of_element_located((By.ID, 'download'))) element.click() time.sleep(5)
解决方案
方法1:直接用requests下载(更高效,无需Selenium)
既然PDF有直接可访问的链接,无需通过浏览器操作,用requests库直接下载更简单高效:
import os import requests # 获取用户ID EmplyID = os.getenv('USER', os.getenv('USERNAME', 'user')) # 下载路径 DownloadPath = f'C:\\Users\\{EmplyID}\\Desktop\\Catalog files\\Downloaded Files' # 确保路径存在 os.makedirs(DownloadPath, exist_ok=True) # PDF链接 pdf_url = "https://www.eaton.com/content/dam/eaton/products/conduit-cable-and-wire-management/crouse-hinds/catalog-pages/crouse-hinds-locknuts-nipples-washers-reducers-plugs-rigid-catalog-page.pdf" # 文件名 file_name = os.path.join(DownloadPath, "crouse-hinds-catalog.pdf") # 发送请求下载 response = requests.get(pdf_url, stream=True) response.raise_for_status() # 检查请求是否成功 with open(file_name, 'wb') as f: for chunk in response.iter_content(chunk_size=8192): f.write(chunk) print(f"文件已成功下载到:{file_name}")
方法2:调整Selenium配置,让PDF直接下载(无需点击按钮)
如果必须用Selenium,修改Chrome的偏好设置,让浏览器直接下载PDF而不是在页面中预览,跳过定位下载按钮的步骤:
from selenium import webdriver import os from selenium.webdriver.chrome.options import Options # 获取用户ID EmplyID = os.getenv('USER', os.getenv('USERNAME', 'user')) # 下载路径 DownloadPath = f'C:\\Users\\{EmplyID}\\Desktop\\Catalog files\\Downloaded Files' # 确保路径存在 os.makedirs(DownloadPath, exist_ok=True) chromeOptions = Options() # 设置下载相关偏好 chromeOptions.add_experimental_option("prefs", { "download.default_directory": DownloadPath, # 禁用PDF预览,直接下载 "plugins.always_open_pdf_externally": True, "download.prompt_for_download": False, "download.directory_upgrade": True }) # 初始化驱动(注意:ChromeDriver版本需与Chrome浏览器匹配) driver = webdriver.Chrome(executable_path=f'C:\\Users\\{EmplyID}\\Desktop\\Catalog files\\chromedriver.exe', options=chromeOptions) # 访问PDF链接,浏览器会自动下载 driver.get("https://www.eaton.com/content/dam/eaton/products/conduit-cable-and-wire-management/crouse-hinds/catalog-pages/crouse-hinds-locknuts-nipples-washers-reducers-plugs-rigid-catalog-page.pdf") # 等待下载完成(可根据文件大小调整等待时间) import time time.sleep(5) driver.quit()
说明
- 原脚本失败的核心原因:Chrome内置PDF阅读器的元素ID并非
download,且不同版本的阅读器元素结构存在差异,直接定位元素稳定性极差。 - 方法2通过配置跳过PDF预览环节,让浏览器直接下载文件,从根源上避免了定位元素的问题。
内容的提问来源于stack exchange,提问作者Vishal
相关产品推荐
相关产品推荐

