You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Selenium中获取类文件对象而无需下载到本地路径?

解决方案

方法一:复用Selenium会话,用requests流式获取文件并上传FTP

核心思路是用Selenium完成验证后,提取会话的Cookie和请求头,通过requests直接请求文件真实下载地址,获取字节流后直接上传FTP,全程不生成本地临时文件。

步骤说明

  1. 用Selenium完成表单验证,提取页面中的真实下载URL;
  2. 将Selenium的Cookie转换为requests可识别的格式;
  3. 带上与Selenium一致的请求头(如User-Agent、Referer),规避403拦截;
  4. 用requests流式请求获取文件字节流,直接上传到FTP。

完整代码示例

import time
import requests
from ftplib import FTP
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as ec
from selenium.webdriver.support.ui import WebDriverWait
from webdriver_manager.chrome import ChromeDriverManager

def get_file_and_upload_ftp(driver: webdriver.Chrome, url: str, ftp_host, ftp_user, ftp_pass, ftp_target_path):
    driver.set_page_load_timeout(40)
    driver.get(url=url)
    time.sleep(2)

    # 接受通知
    ccc_accept = driver.find_element(By.ID, 'ccc-notify-accept')
    if WebDriverWait(driver, 5).until(ec.element_to_be_clickable(ccc_accept)):
        ccc_accept.click()

    # 填写表单数据
    WebDriverWait(driver, 2).until(ec.presence_of_element_located((By.ID, 'agreement_form')))
    driver.find_element(By.ID, 'contact_name').send_keys('Company')
    driver.find_element(By.ID, 'contact_title').send_keys('People')
    driver.find_element(By.ID, 'company').send_keys('cb')
    driver.find_element(By.ID, 'country').send_keys('some')

    # 提交表单
    submit_btn = WebDriverWait(driver, 5).until(
        ec.element_to_be_clickable((By.XPATH, '//*[@id="doc_agreement"]/div[4]/input[1]'))
    )
    submit_btn.click()

    time.sleep(2)

    # 获取真实下载URL(需根据实际页面调整选择器,示例假设下载链接id为download_link)
    download_url = WebDriverWait(driver, 10).until(
        ec.presence_of_element_located((By.ID, 'download_link'))
    ).get_attribute('href')

    # 转换Selenium Cookie为requests格式
    cookies = {cookie['name']: cookie['value'] for cookie in driver.get_cookies()}
    # 获取与Selenium一致的User-Agent
    user_agent = driver.execute_script("return navigator.userAgent;")
    # 构造请求头,添加Referer避免反爬
    headers = {
        'User-Agent': user_agent,
        'Referer': url
    }

    # 流式请求文件并直接上传FTP
    with requests.get(download_url, cookies=cookies, headers=headers, stream=True) as r:
        r.raise_for_status()
        with FTP(ftp_host) as ftp:
            ftp.login(user=ftp_user, passwd=ftp_pass)
            ftp.storbinary(f'STOR {ftp_target_path}', r.raw)

def init_driver():
    options = webdriver.ChromeOptions()
    options.add_argument('window-size=1920x1080')
    options.add_argument("disable-gpu")
    # 禁用PDF自动打开,避免干扰页面操作
    chrome_prefs = {
        "plugins.always_open_pdf_externally": True,
        "download.open_pdf_in_system_reader": False,
        "profile.default_content_settings.popups": 0,
    }
    options.add_experimental_option("prefs", chrome_prefs)
    driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options)
    return driver

# 调用示例
if __name__ == "__main__":
    target_url = "https://docs-prv.pcisecuritystandards.org/PCI%20DSS/Standard/PCI-DSS-v4_0.pdf"
    ftp_config = {
        'host': 'your_ftp_host',
        'user': 'your_ftp_user',
        'pass': 'your_ftp_pass',
        'target_path': '/remote/path/PCI-DSS-v4_0.pdf'
    }
    driver = init_driver()
    try:
        get_file_and_upload_ftp(
            driver, target_url,
            ftp_config['host'], ftp_config['user'], ftp_config['pass'], ftp_config['target_path']
        )
    finally:
        driver.quit()

方法二:用Chrome DevTools Protocol(CDP)拦截下载响应

如果无法获取真实下载URL,可通过CDP直接拦截Chrome的下载请求,获取文件字节数据,无需额外发起请求。

核心代码示例

import time
from io import BytesIO
from ftplib import FTP
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from webdriver_manager.chrome import ChromeDriverManager

def init_driver_with_cdp():
    options = webdriver.ChromeOptions()
    options.add_argument('window-size=1920x1080')
    options.add_argument("disable-gpu")
    driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options)
    
    # 启用CDP网络监听
    driver.execute_cdp_cmd('Network.enable', {})
    file_content = None

    # 定义响应拦截回调
    def handle_response(response):
        nonlocal file_content
        # 判断是否为PDF文件响应
        if response['response']['mimeType'] == 'application/pdf':
            body = driver.execute_cdp_cmd('Network.getResponseBody', {'requestId': response['requestId']})
            file_content = body['body'].encode('utf-8') if body['base64Encoded'] else body['body']

    # 监听responseReceived事件
    driver.execute_cdp_cmd('Network.setResponseInterception', {'patterns': [{'urlPattern': '*'}]})
    driver.add_listener('Network.responseReceived', handle_response)
    
    return driver, file_content

# 调用示例
if __name__ == "__main__":
    target_url = "https://docs-prv.pcisecuritystandards.org/PCI%20DSS/Standard/PCI-DSS-v4_0.pdf"
    ftp_config = {
        'host': 'your_ftp_host',
        'user': 'your_ftp_user',
        'pass': 'your_ftp_pass',
        'target_path': '/remote/path/PCI-DSS-v4_0.pdf'
    }
    driver, file_content = init_driver_with_cdp()
    try:
        # 执行表单验证逻辑(同方法一的表单填写、提交代码)
        # ...
        
        # 等待文件内容被捕获
        while file_content is None:
            time.sleep(1)
        
        # 上传到FTP
        with FTP(ftp_config['host']) as ftp:
            ftp.login(user=ftp_config['user'], passwd=ftp_config['pass'])
            ftp.storbinary(f'STOR {ftp_config["target_path"]}', BytesIO(file_content))
    finally:
        driver.quit()

关键注意事项

  • 方法一中的下载URL选择器需根据实际页面调整,可通过开发者工具定位下载按钮元素;
  • 403错误大多源于请求头不完整或Cookie格式错误,方法一的请求头配置可解决多数此类问题;
  • CDP方法需Chrome版本支持,回调逻辑需根据文件Content-Type或文件名过滤,避免捕获无关响应。

内容的提问来源于stack exchange,提问作者Roman

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.17 13:54:55