如何在Selenium中获取类文件对象而无需下载到本地路径?
解决方案
方法一:复用Selenium会话,用requests流式获取文件并上传FTP
核心思路是用Selenium完成验证后,提取会话的Cookie和请求头,通过requests直接请求文件真实下载地址,获取字节流后直接上传FTP,全程不生成本地临时文件。
步骤说明
- 用Selenium完成表单验证,提取页面中的真实下载URL;
- 将Selenium的Cookie转换为
requests可识别的格式; - 带上与Selenium一致的请求头(如User-Agent、Referer),规避403拦截;
- 用
requests流式请求获取文件字节流,直接上传到FTP。
完整代码示例
import time import requests from ftplib import FTP from selenium import webdriver from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.support import expected_conditions as ec from selenium.webdriver.support.ui import WebDriverWait from webdriver_manager.chrome import ChromeDriverManager def get_file_and_upload_ftp(driver: webdriver.Chrome, url: str, ftp_host, ftp_user, ftp_pass, ftp_target_path): driver.set_page_load_timeout(40) driver.get(url=url) time.sleep(2) # 接受通知 ccc_accept = driver.find_element(By.ID, 'ccc-notify-accept') if WebDriverWait(driver, 5).until(ec.element_to_be_clickable(ccc_accept)): ccc_accept.click() # 填写表单数据 WebDriverWait(driver, 2).until(ec.presence_of_element_located((By.ID, 'agreement_form'))) driver.find_element(By.ID, 'contact_name').send_keys('Company') driver.find_element(By.ID, 'contact_title').send_keys('People') driver.find_element(By.ID, 'company').send_keys('cb') driver.find_element(By.ID, 'country').send_keys('some') # 提交表单 submit_btn = WebDriverWait(driver, 5).until( ec.element_to_be_clickable((By.XPATH, '//*[@id="doc_agreement"]/div[4]/input[1]')) ) submit_btn.click() time.sleep(2) # 获取真实下载URL(需根据实际页面调整选择器,示例假设下载链接id为download_link) download_url = WebDriverWait(driver, 10).until( ec.presence_of_element_located((By.ID, 'download_link')) ).get_attribute('href') # 转换Selenium Cookie为requests格式 cookies = {cookie['name']: cookie['value'] for cookie in driver.get_cookies()} # 获取与Selenium一致的User-Agent user_agent = driver.execute_script("return navigator.userAgent;") # 构造请求头,添加Referer避免反爬 headers = { 'User-Agent': user_agent, 'Referer': url } # 流式请求文件并直接上传FTP with requests.get(download_url, cookies=cookies, headers=headers, stream=True) as r: r.raise_for_status() with FTP(ftp_host) as ftp: ftp.login(user=ftp_user, passwd=ftp_pass) ftp.storbinary(f'STOR {ftp_target_path}', r.raw) def init_driver(): options = webdriver.ChromeOptions() options.add_argument('window-size=1920x1080') options.add_argument("disable-gpu") # 禁用PDF自动打开,避免干扰页面操作 chrome_prefs = { "plugins.always_open_pdf_externally": True, "download.open_pdf_in_system_reader": False, "profile.default_content_settings.popups": 0, } options.add_experimental_option("prefs", chrome_prefs) driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options) return driver # 调用示例 if __name__ == "__main__": target_url = "https://docs-prv.pcisecuritystandards.org/PCI%20DSS/Standard/PCI-DSS-v4_0.pdf" ftp_config = { 'host': 'your_ftp_host', 'user': 'your_ftp_user', 'pass': 'your_ftp_pass', 'target_path': '/remote/path/PCI-DSS-v4_0.pdf' } driver = init_driver() try: get_file_and_upload_ftp( driver, target_url, ftp_config['host'], ftp_config['user'], ftp_config['pass'], ftp_config['target_path'] ) finally: driver.quit()
方法二:用Chrome DevTools Protocol(CDP)拦截下载响应
如果无法获取真实下载URL,可通过CDP直接拦截Chrome的下载请求,获取文件字节数据,无需额外发起请求。
核心代码示例
import time from io import BytesIO from ftplib import FTP from selenium import webdriver from selenium.webdriver.chrome.service import Service from webdriver_manager.chrome import ChromeDriverManager def init_driver_with_cdp(): options = webdriver.ChromeOptions() options.add_argument('window-size=1920x1080') options.add_argument("disable-gpu") driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options) # 启用CDP网络监听 driver.execute_cdp_cmd('Network.enable', {}) file_content = None # 定义响应拦截回调 def handle_response(response): nonlocal file_content # 判断是否为PDF文件响应 if response['response']['mimeType'] == 'application/pdf': body = driver.execute_cdp_cmd('Network.getResponseBody', {'requestId': response['requestId']}) file_content = body['body'].encode('utf-8') if body['base64Encoded'] else body['body'] # 监听responseReceived事件 driver.execute_cdp_cmd('Network.setResponseInterception', {'patterns': [{'urlPattern': '*'}]}) driver.add_listener('Network.responseReceived', handle_response) return driver, file_content # 调用示例 if __name__ == "__main__": target_url = "https://docs-prv.pcisecuritystandards.org/PCI%20DSS/Standard/PCI-DSS-v4_0.pdf" ftp_config = { 'host': 'your_ftp_host', 'user': 'your_ftp_user', 'pass': 'your_ftp_pass', 'target_path': '/remote/path/PCI-DSS-v4_0.pdf' } driver, file_content = init_driver_with_cdp() try: # 执行表单验证逻辑(同方法一的表单填写、提交代码) # ... # 等待文件内容被捕获 while file_content is None: time.sleep(1) # 上传到FTP with FTP(ftp_config['host']) as ftp: ftp.login(user=ftp_config['user'], passwd=ftp_config['pass']) ftp.storbinary(f'STOR {ftp_config["target_path"]}', BytesIO(file_content)) finally: driver.quit()
关键注意事项
- 方法一中的下载URL选择器需根据实际页面调整,可通过开发者工具定位下载按钮元素;
- 403错误大多源于请求头不完整或Cookie格式错误,方法一的请求头配置可解决多数此类问题;
- CDP方法需Chrome版本支持,回调逻辑需根据文件Content-Type或文件名过滤,避免捕获无关响应。
内容的提问来源于stack exchange,提问作者Roman
相关产品推荐
相关产品推荐

