You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Python抓取FCC网站下载按钮的GET请求URL实现自动化下载?

解决FCC网站Selenium点击下载按钮无文件下载的问题

问题背景

要自动从FCC宽带数据页面下载各州不同技术类型的宽带数据,当前代码可正常选择州并点击下载按钮,但按钮HTML中没有直接的下载URL,点击按钮后仅能在Chrome调试器网络面板看到实际的GET请求URL,导致点击操作无法触发实际下载。

解决方案

方案1:通过Chrome DevTools Protocol(CDP)监听网络请求捕获下载URL

利用Selenium配合CDP监听浏览器的网络请求,在点击下载按钮后捕获对应的GET请求URL,再用requests库直接下载文件,避免依赖浏览器的下载行为。

修改后的代码如下:

import time
import requests
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.support.ui import Select
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

path_to_chromedriver = r"C:\Users\stackoverflow\Documents\chromedriver_win32"
download_dir = r"C:\Your\Download\Path"  # 替换为你的本地下载目录

def main():
    chrome_options = Options()
    chrome_options.add_argument("--headless=new")  # 使用新版无头模式,稳定性更强
    chrome_options.add_experimental_option("excludeSwitches", ["enable-logging"])

    service = Service(path_to_chromedriver)
    driver = webdriver.Chrome(service=service, options=chrome_options)

    # 启用网络请求监听
    driver.execute_cdp_cmd('Network.enable', {})

    driver.get('https://broadbandmap.fcc.gov/data-download/nationwide-data?version=dec2022')
    WebDriverWait(driver, 20).until(EC.presence_of_element_located((By.ID, 'state')))

    select = Select(driver.find_element(By.ID, 'state'))

    for option in select.options[1:]:
        state_value = option.get_attribute('value')
        state_name = option.text.strip()
        select.select_by_value(state_value)
        # 等待页面完成更新,替代固定sleep
        WebDriverWait(driver, 10).until(EC.staleness_of(select.first_selected_option))
        download_files(driver, state_name)

    driver.quit()

def download_files(driver, state_name):
    technologies = ['Cable', 'Copper', 'Fiber to the Premises', 'LBR Fixed Wireless', 'Licensed Fixed Wireless',
                    'Unlicensed Fixed Wireless']

    WebDriverWait(driver, 10).until(EC.presence_of_element_located((By.CLASS_NAME, 'btn-outline-primary')))

    for tech in technologies:
        try:
            row = driver.find_element(By.XPATH, f"//td[normalize-space(.)='{tech}']/ancestor::tr")
            download_button = row.find_element(By.CSS_SELECTOR, "button.btn.btn-outline-primary.border-0")
            
            # 清空旧的网络请求记录
            driver.execute_cdp_cmd('Network.clearBrowserCache', {})
            
            # 点击按钮触发请求
            download_button.click()
            time.sleep(3)  # 等待请求发起

            # 获取所有网络请求并筛选下载URL
            requests_data = driver.execute_cdp_cmd('Network.getRequests', {})
            download_url = None
            for req in requests_data['requests']:
                if req['method'] == 'GET' and ('.zip' in req['url'] or 'get-data' in req['url']):
                    download_url = req['url']
                    break

            if download_url:
                # 用requests下载文件
                response = requests.get(download_url, stream=True)
                file_name = f"{state_name}_{tech.replace(' ', '_')}.zip"
                with open(f"{download_dir}\\{file_name}", 'wb') as f:
                    for chunk in response.iter_content(chunk_size=8192):
                        f.write(chunk)
                print(f"已完成下载:{file_name}")
            else:
                print(f"未找到{tech}对应的下载URL")

        except Exception as e:
            print(f"下载{tech}时出错:{str(e)}")

if __name__ == '__main__':
    main()

方案2:分析请求规律直接构造下载URL(更高效)

观察网络面板的请求URL,可发现其结构固定,包含州代码、技术类型标识等参数,无需点击按钮,直接构造URL后用requests下载即可,速度更快且稳定性更强。

示例代码如下:

import requests
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.support.ui import Select
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

path_to_chromedriver = r"C:\Users\stackoverflow\Documents\chromedriver_win32"
download_dir = r"C:\Your\Download\Path"

# 技术名称对应的标识code(从抓包数据中获取)
tech_code_map = {
    'Cable': 'C',
    'Copper': 'D',
    'Fiber to the Premises': 'F',
    'LBR Fixed Wireless': 'L',
    'Licensed Fixed Wireless': 'W',
    'Unlicensed Fixed Wireless': 'U'
}

def main():
    chrome_options = Options()
    chrome_options.add_argument("--headless=new")
    chrome_options.add_experimental_option("excludeSwitches", ["enable-logging"])

    service = Service(path_to_chromedriver)
    driver = webdriver.Chrome(service=service, options=chrome_options)

    driver.get('https://broadbandmap.fcc.gov/data-download/nationwide-data?version=dec2022')
    WebDriverWait(driver, 20).until(EC.presence_of_element_located((By.ID, 'state')))

    # 获取所有州的代码和名称
    select = Select(driver.find_element(By.ID, 'state'))
    state_list = [(option.get_attribute('value'), option.text.strip()) for option in select.options[1:]]

    driver.quit()  # 无需继续使用浏览器

    # 批量下载所有州的对应技术文件
    for state_code, state_name in state_list:
        download_files(state_code, state_name)

def download_files(state_code, state_name):
    base_url = "https://broadbandmap.fcc.gov/data-download/get-data"
    for tech_name, tech_code in tech_code_map.items():
        try:
            params = {
                'state': state_code,
                'tech': tech_code,
                'format': 'zip',
                'version': 'dec2022'
            }
            response = requests.get(base_url, params=params, stream=True)
            if response.status_code == 200:
                file_name = f"{state_name}_{tech_name.replace(' ', '_')}.zip"
                with open(f"{download_dir}\\{file_name}", 'wb') as f:
                    for chunk in response.iter_content(chunk_size=8192):
                        f.write(chunk)
                print(f"已完成下载:{file_name}")
            else:
                print(f"{state_name}的{tech_name}下载失败,状态码:{response.status_code}")
        except Exception as e:
            print(f"下载{state_name}的{tech_name}时出错:{str(e)}")

if __name__ == '__main__':
    main()

注意事项

  • 方案2无需依赖浏览器,速度更快且更稳定,推荐优先使用;
  • 若网站后续更新请求参数,需重新抓包更新tech_code_map或URL格式;
  • 确保已安装requests库:执行pip install requests完成安装;
  • 替换代码中的download_dir为你的本地下载目录。

内容的提问来源于stack exchange,提问作者Yash Puranik

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.19 23:07:55