You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python Selenium谷歌搜索爬取问题及结构化数据保存需求

修改后的谷歌搜索爬取代码及改动说明

问题背景

原代码仅能抓取谷歌搜索页的前4条摘要内容,无法进入目标页面提取h1、h2、p、li等结构化标签,以下是修改后的完整代码及改动说明:


修改后的完整代码

import time
import csv
from selenium.webdriver.common.by import By
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from bs4 import BeautifulSoup

# 获取用户输入:多个查询用逗号分隔
search_queries = [q.strip() for q in input("Введите запрос/запросы (через запятую):\n").split(',') if q.strip()]
limit = int(input("Введите лимит ответов на каждый запрос:\n"))

# 浏览器配置
options = webdriver.ChromeOptions()
options.add_argument("--disable-blink-features=AutomationControlled")
options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/112.0.5615.165 Safari/537.36")
options.add_argument("accept=*/*")
options.add_argument("--start-maximized")

# 初始化驱动(需替换为你的chromedriver完整路径)
s = Service(executable_path=r'C:\Program Files\google_answers\chromedriver.exe')
driver = webdriver.Chrome(service=s, options=options)

# 绕过Selenium检测
driver.execute_cdp_cmd("Page.addScriptToEvaluateOnNewDocument", {
    'source': '''
        delete window.cdc_adoQpoasnfa76pfcZLmcfl_Array;
        delete window.cdc_adoQpoasnfa76pfcZLmcfl_Promise;
        delete window.cdc_adoQpoasnfa76pfcZLmcfl_Symbol;
  '''
})

wait = WebDriverWait(driver, 10)

def extract_page_content(url):
    """提取目标页面的h1、h2、p、li标签内容"""
    try:
        # 新标签页打开链接,避免干扰搜索页
        driver.execute_script(f"window.open('{url}');")
        driver.switch_to.window(driver.window_handles[-1])
        wait.until(EC.presence_of_element_located((By.TAG_NAME, "body")))
        
        soup = BeautifulSoup(driver.page_source, "lxml")
        page_title = soup.title.string if soup.title else "无标题"
        
        # 结构化提取目标标签,过滤空文本
        content = []
        for h1 in soup.find_all('h1'):
            content.append(("h1", h1.get_text(strip=True)))
        for h2 in soup.find_all('h2'):
            content.append(("h2", h2.get_text(strip=True)))
        for p in soup.find_all('p'):
            text = p.get_text(strip=True)
            if text:
                content.append(("p", text))
        for li in soup.find_all('li'):
            text = li.get_text(strip=True)
            if text:
                content.append(("li", text))
        
        driver.close()
        driver.switch_to.window(driver.window_handles[0])
        return page_title, content
    except Exception as e:
        print(f"页面提取失败: {str(e)}")
        driver.close()
        driver.switch_to.window(driver.window_handles[0])
        return "加载失败", []

def save_to_xls(query, page_title, content):
    """将结构化数据保存到xls文件"""
    with open("answers.xls", "a+", encoding="utf-8-sig", newline='') as file:
        writer = csv.writer(file, delimiter='\t')  # 制表符分隔适配xls格式
        # 写入标识信息
        writer.writerow(["查询词", query])
        writer.writerow(["页面标题", page_title])
        writer.writerow(["标签类型", "内容"])
        # 写入标签内容
        for tag_type, text in content:
            writer.writerow([tag_type, text])
        writer.writerow([])  # 空行分隔不同页面

def search(query):
    """执行谷歌搜索并处理结果"""
    try:
        driver.get(f"https://www.google.com/search?q={query}&ie=UTF-8")
        wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, "div.g")))
        
        # 获取指定数量的有效搜索链接
        results = driver.find_elements(By.CSS_SELECTOR, "div.g a")[:limit]
        result_links = [link.get_attribute('href') for link in results if link.get_attribute('href')]
        
        for link in result_links:
            print(f"正在处理: {link}")
            page_title, content = extract_page_content(link)
            save_to_xls(query, page_title, content)
            time.sleep(2)  # 控制请求频率
    except Exception as e:
        print(f"搜索处理失败: {str(e)}")

def main():
    # 初始化文件(首次运行写入表头)
    try:
        with open("answers.xls", "r", encoding="utf-8-sig"):
            pass
    except FileNotFoundError:
        with open("answers.xls", "w", encoding="utf-8-sig", newline='') as file:
            writer = csv.writer(file, delimiter='\t')
            writer.writerow(["查询词", "页面标题", "标签类型", "内容"])
    
    for query in search_queries:
        print(f"处理查询: {query}")
        search(query)
    driver.quit()

if __name__ == "__main__":
    main()

主要改动点

  • 修正搜索词逻辑:原代码将输入拆分为单个单词搜索,现在支持多个完整查询(逗号分隔),保留查询语义
  • 补充驱动路径:完善chromedriver文件名配置,避免初始化失败
  • 添加等待机制:用WebDriverWait替代强制等待,确保元素加载完成后再操作,减少空内容抓取
  • 实现结果跳转:通过新标签页打开搜索结果链接,避免丢失搜索页状态
  • 结构化标签提取:针对h1、h2、p、li标签分别提取内容,过滤空文本保证数据有效性
  • 优化文件格式:用制表符分隔内容适配xls读取,每条记录包含查询词、页面标题、标签类型和内容,结构清晰
  • 添加异常处理:捕获页面加载、元素查找等异常,避免程序崩溃并输出错误信息
  • 控制请求频率:添加time.sleep(2)降低反爬风险
  • 初始化文件表头:首次运行自动创建文件并写入表头,避免重复写入

内容的提问来源于stack exchange,提问作者whizzkid

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.19 19:39:59