如何用Python的Selenium从Blob URL下载CSV文件?
解决Selenium下载Blob URL对应CSV文件的问题
问题背景
我想用Python的Selenium自动化下载动态网站中Blob URL对应的CSV文件,CSV下载由点击按钮触发,但点击后生成的Blob URL无法通过传统HTTP请求直接访问,难以捕获并下载对应文件。
- 示例页面:https://snapshot.org/#/aave.eth/proposal/0x70dfd865b78c4c391e2b0729b907d152e6e8a0da683416d617d8f84782036349
- Blob URL示例:
blob:https://snapshot.org/4b2f45e9-8ca3-4105-b142-e1877e420c84
手动点击按钮能成功获取文件,但现有自动化代码无法完成下载,尝试过的代码如下:
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.chrome.service import Service from webdriver_manager.chrome import ChromeDriverManager import time # Setup ChromeDriver driver = webdriver.Chrome(service=Service(ChromeDriverManager().install())) # URL of the proposal page url = 'https://snapshot.org/#/aave.eth/proposal/0x70dfd865b78c4c391e2b0729b907d152e6e8a0da683416d617d8f84782036349' # Navigate to the page driver.get(url) try: # Wait up to 20 seconds until the expected button is found using its attributes wait = WebDriverWait(driver, 20) download_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//button[contains(.,'svg')]"))) download_button.click() print("Download initiated.") except Exception as e: print(f"Error: {e}") # Wait for the download to complete time.sleep(5) # Close the browser driver.quit()
下载按钮对应的元素代码:
<svg viewBox="0 0 24 24" width="1.2em" height="1.2em"><path fill="none" stroke="currentColor" stroke-linecap="round" stroke-linejoin="round" stroke-width="2" d="M4 16v1a3 3 0 0 0 3 3h10a3 3 0 0 0 3-3v-1m-4-4l-4 4m0 0l-4-4m4 4V4"></path></svg>
解决方案
方法一:拦截Blob构造函数,直接提取文件内容
Blob是前端在浏览器内部生成的,我们可以通过注入JavaScript脚本拦截Blob的创建过程,提取CSV内容后保存到本地。
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.chrome.service import Service from webdriver_manager.chrome import ChromeDriverManager import time # 注入拦截Blob的脚本,捕获CSV内容 intercept_blob_script = """ window.originalBlob = Blob; window.blobData = null; Blob = function(parts, options) { const blob = new window.originalBlob(parts, options); // 只拦截CSV类型的Blob if (options?.type === 'text/csv') { const reader = new FileReader(); reader.onload = function() { window.blobData = reader.result; }; reader.readAsText(blob); } return blob; }; """ # 配置Chrome选项 options = webdriver.ChromeOptions() options.add_experimental_option("prefs", { "download.prompt_for_download": False, "download.directory_upgrade": True }) # 初始化驱动并注入脚本 driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options) driver.execute_script(intercept_blob_script) # 访问目标页面 url = 'https://snapshot.org/#/aave.eth/proposal/0x70dfd865b78c4c391e2b0729b907d152e6e8a0da683416d617d8f84782036349' driver.get(url) try: # 精准定位下载按钮:通过SVG路径的d属性匹配 wait = WebDriverWait(driver, 20) download_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//button[.//path[@d='M4 16v1a3 3 0 0 0 3 3h10a3 3 0 0 0 3-3v-1m-4-4l-4 4m0 0l-4-4m4 4V4']]"))) download_button.click() print("下载已触发") # 等待数据捕获完成 time.sleep(3) blob_data = driver.execute_script("return window.blobData;") if blob_data: # 保存CSV文件 with open('proposal_votes.csv', 'w', encoding='utf-8') as f: f.write(blob_data) print("CSV文件已成功保存") else: print("未捕获到Blob数据") except Exception as e: print(f"错误:{e}") driver.quit()
方法二:配置Chrome自动下载,直接获取本地文件
通过设置Chrome的下载偏好,让文件自动保存到指定目录,再读取下载完成的文件。
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.chrome.service import Service from webdriver_manager.chrome import ChromeDriverManager import os import time from glob import glob # 设置下载目录 download_dir = os.path.abspath('./downloads') os.makedirs(download_dir, exist_ok=True) # 配置Chrome下载偏好 options = webdriver.ChromeOptions() prefs = { "download.default_directory": download_dir, "download.prompt_for_download": False, "download.directory_upgrade": True, "safebrowsing.enabled": True } options.add_experimental_option("prefs", prefs) # 初始化驱动 driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options) # 访问目标页面 url = 'https://snapshot.org/#/aave.eth/proposal/0x70dfd865b78c4c391e2b0729b907d152e6e8a0da683416d617d8f84782036349' driver.get(url) try: # 精准定位下载按钮 wait = WebDriverWait(driver, 20) download_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//button[.//path[@d='M4 16v1a3 3 0 0 0 3 3h10a3 3 0 0 0 3-3v-1m-4-4l-4 4m0 0l-4-4m4 4V4']]"))) download_button.click() print("下载已触发") # 等待下载完成 time.sleep(5) # 查找最新下载的CSV文件 csv_files = glob(os.path.join(download_dir, '*.csv')) if csv_files: latest_file = max(csv_files, key=os.path.getctime) print(f"CSV文件已下载到:{latest_file}") else: print("未找到下载的CSV文件") except Exception as e: print(f"错误:{e}") driver.quit()
关键说明
- 按钮定位问题:原代码使用
//button[contains(.,'svg')]会匹配多个带SVG的按钮,改用SVG路径的d属性可以精准定位下载按钮。 - Blob URL无法直接访问的原因:Blob URL是浏览器内部生成的临时URL,仅在当前会话有效,无法通过外部HTTP请求访问,必须从浏览器内部提取数据。
内容的提问来源于stack exchange,提问作者rischan
相关产品推荐
相关产品推荐

