You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python实现Google/Yandex关键词搜索无结果问题求助

解决Google/Yandex搜索引擎Python搜索脚本问题

问题概述

需要编写Python脚本实现Google、Yandex等搜索引擎的关键词搜索并展示结果,现有两个方案均失败:

  • requests+BeautifulSoup方案:可正常用于DuckDuckGo搜索,但Google搜索返回“No results found”
  • Selenium方案:无法正确获取并解析搜索结果

requests+BeautifulSoup方案问题分析与修复

原问题原因

  1. Google页面结构已更新,原选择器div.g、span.st不再匹配当前结果元素
  2. User-Agent版本过旧,可能触发Google的反爬机制
  3. 搜索结果的链接是带/url?q=的跳转链接,未做提取处理

修复后的代码

import requests
from bs4 import BeautifulSoup
import re

def search_google(query):
    url = f"https://www.google.com/search?q={query}&hl=en"
    headers = {
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
        "Accept-Language": "en-US,en;q=0.5"
    }
    response = requests.get(url, headers=headers)
    response.raise_for_status()
    return response.text

def parse_google_results(html, keywords):
    soup = BeautifulSoup(html, "html.parser")
    results = soup.select("div.g")
    found_results = []

    for result in results:
        title_tag = result.select_one("h3.LC20lb")
        desc_tag = result.select_one("div.VwiC3b span")
        link_tag = result.select_one("a")

        if not (title_tag and desc_tag and link_tag):
            continue

        title = title_tag.get_text(strip=True)
        description = desc_tag.get_text(strip=True)
        raw_link = link_tag.get("href")
        
        # 提取真实链接,去掉/url?q=前缀和后续参数
        match = re.search(r"/url\?q=(.*?)&", raw_link)
        if not match:
            continue
        link = match.group(1)

        # 检查关键词匹配
        if any(keyword.lower() in title.lower() or keyword.lower() in description.lower() for keyword in keywords):
            found_results.append({
                "title": title,
                "description": description,
                "link": link
            })

    return found_results

def search_keywords(keywords, num_results=5):
    query = " ".join(keywords)
    try:
        html = search_google(query)
        results = parse_google_results(html, keywords)
    except Exception as e:
        print(f"搜索出错: {str(e)}")
        return

    if results:
        for i, result in enumerate(results[:num_results], 1):
            print(f"结果 {i}:")
            print(f"标题: {result['title']}")
            print(f"描述: {result['description']}")
            print(f"链接: {result['link']}")
            print("--------------------")
    else:
        print("未找到匹配结果。")

# 示例调用
search_keywords(["instagram"], num_results=5)

Selenium方案问题分析与修复

原问题原因

  1. 重复初始化WebDriver,造成资源浪费且逻辑混乱
  2. 使用time.sleep等待结果加载,稳定性差
  3. 页面选择器过时,无法匹配当前Google结果元素
  4. 缺失time、json模块导入
  5. parse_results中重新打开浏览器解析HTML的做法冗余

修复后的代码

from selenium import webdriver
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from webdriver_manager.chrome import ChromeDriverManager
import json
import re

def search_google(keywords):
    options = Options()
    options.headless = True
    options.add_argument("--disable-blink-features=AutomationControlled")
    options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
    
    # 使用webdriver-manager自动管理驱动,无需手动指定路径
    driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options)
    
    try:
        driver.get('https://www.google.com')
        # 等待搜索框加载
        search_input = WebDriverWait(driver, 10).until(
            EC.presence_of_element_located((By.NAME, 'q'))
        )
        search_input.send_keys(keywords)
        search_input.send_keys(Keys.ENTER)
        
        # 等待搜索结果加载
        WebDriverWait(driver, 10).until(
            EC.presence_of_element_located((By.CSS_SELECTOR, 'div.g'))
        )
        html = driver.page_source
        return html
    finally:
        driver.quit()

def parse_results(html, keywords):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    results = soup.select("div.g")
    found_results = []

    for result in results:
        title_tag = result.select_one("h3.LC20lb")
        desc_tag = result.select_one("div.VwiC3b span")
        link_tag = result.select_one("a")

        if not (title_tag and desc_tag and link_tag):
            continue

        title = title_tag.get_text(strip=True)
        description = desc_tag.get_text(strip=True)
        raw_link = link_tag.get("href")
        
        match = re.search(r"/url\?q=(.*?)&", raw_link)
        if not match:
            continue
        link = match.group(1)

        if any(keyword.lower() in title.lower() or keyword.lower() in description.lower() or keyword.lower() in link.lower() for keyword in keywords):
            found_results.append({
                "title": title,
                "description": description,
                "link": link
            })

    return found_results

def perform_search():
    keywords = ["instagram"]
    try:
        html = search_google(" ".join(keywords))
        results = parse_results(html, keywords)
    except Exception as e:
        print(f"搜索出错: {str(e)}")
        return

    if results:
        with open('results.json', 'w', encoding='utf-8') as f:
            json.dump(results, f, ensure_ascii=False, indent=2)
        print("结果已保存到results.json")
    else:
        print("未找到匹配结果。")

if __name__ == '__main__':
    perform_search()

Yandex搜索实现示例

import requests
from bs4 import BeautifulSoup

def search_yandex(query):
    url = f"https://yandex.com/search/?text={query}"
    headers = {
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
    }
    response = requests.get(url, headers=headers)
    response.raise_for_status()
    return response.text

def parse_yandex_results(html, keywords):
    soup = BeautifulSoup(html, "html.parser")
    results = soup.select("li.serp-item")
    found_results = []

    for result in results:
        title_tag = result.select_one("h2.serp-item__title a")
        desc_tag = result.select_one("div.serp-item__text")
        link_tag = result.select_one("h2.serp-item__title a")

        if not (title_tag and desc_tag and link_tag):
            continue

        title = title_tag.get_text(strip=True)
        description = desc_tag.get_text(strip=True)
        link = link_tag.get("href")

        if any(keyword.lower() in title.lower() or keyword.lower() in description.lower() for keyword in keywords):
            found_results.append({
                "title": title,
                "description": description,
                "link": link
            })

    return found_results

# 示例调用
def yandex_search_keywords(keywords, num_results=5):
    query = " ".join(keywords)
    html = search_yandex(query)
    results = parse_yandex_results(html, keywords)
    
    if results:
        for i, result in enumerate(results[:num_results], 1):
            print(f"结果 {i}:")
            print(f"标题: {result['title']}")
            print(f"描述: {result['description']}")
            print(f"链接: {result['link']}")
            print("--------------------")
    else:
        print("未找到匹配结果。")

yandex_search_keywords(["instagram"], num_results=5)

内容的提问来源于stack exchange,提问作者Orkun Koçak

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 09:24:57