Python实现Google/Yandex关键词搜索无结果问题求助
解决Google/Yandex搜索引擎Python搜索脚本问题
问题概述
需要编写Python脚本实现Google、Yandex等搜索引擎的关键词搜索并展示结果,现有两个方案均失败:
- requests+BeautifulSoup方案:可正常用于DuckDuckGo搜索,但Google搜索返回“No results found”
- Selenium方案:无法正确获取并解析搜索结果
requests+BeautifulSoup方案问题分析与修复
原问题原因
- Google页面结构已更新,原选择器
div.g、span.st不再匹配当前结果元素 - User-Agent版本过旧,可能触发Google的反爬机制
- 搜索结果的链接是带
/url?q=的跳转链接,未做提取处理
修复后的代码
import requests from bs4 import BeautifulSoup import re def search_google(query): url = f"https://www.google.com/search?q={query}&hl=en" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept-Language": "en-US,en;q=0.5" } response = requests.get(url, headers=headers) response.raise_for_status() return response.text def parse_google_results(html, keywords): soup = BeautifulSoup(html, "html.parser") results = soup.select("div.g") found_results = [] for result in results: title_tag = result.select_one("h3.LC20lb") desc_tag = result.select_one("div.VwiC3b span") link_tag = result.select_one("a") if not (title_tag and desc_tag and link_tag): continue title = title_tag.get_text(strip=True) description = desc_tag.get_text(strip=True) raw_link = link_tag.get("href") # 提取真实链接,去掉/url?q=前缀和后续参数 match = re.search(r"/url\?q=(.*?)&", raw_link) if not match: continue link = match.group(1) # 检查关键词匹配 if any(keyword.lower() in title.lower() or keyword.lower() in description.lower() for keyword in keywords): found_results.append({ "title": title, "description": description, "link": link }) return found_results def search_keywords(keywords, num_results=5): query = " ".join(keywords) try: html = search_google(query) results = parse_google_results(html, keywords) except Exception as e: print(f"搜索出错: {str(e)}") return if results: for i, result in enumerate(results[:num_results], 1): print(f"结果 {i}:") print(f"标题: {result['title']}") print(f"描述: {result['description']}") print(f"链接: {result['link']}") print("--------------------") else: print("未找到匹配结果。") # 示例调用 search_keywords(["instagram"], num_results=5)
Selenium方案问题分析与修复
原问题原因
- 重复初始化WebDriver,造成资源浪费且逻辑混乱
- 使用
time.sleep等待结果加载,稳定性差 - 页面选择器过时,无法匹配当前Google结果元素
- 缺失
time、json模块导入 parse_results中重新打开浏览器解析HTML的做法冗余
修复后的代码
from selenium import webdriver from selenium.webdriver.common.keys import Keys from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from webdriver_manager.chrome import ChromeDriverManager import json import re def search_google(keywords): options = Options() options.headless = True options.add_argument("--disable-blink-features=AutomationControlled") options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") # 使用webdriver-manager自动管理驱动,无需手动指定路径 driver = webdriver.Chrome(service=Service(ChromeDriverManager().install()), options=options) try: driver.get('https://www.google.com') # 等待搜索框加载 search_input = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.NAME, 'q')) ) search_input.send_keys(keywords) search_input.send_keys(Keys.ENTER) # 等待搜索结果加载 WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.CSS_SELECTOR, 'div.g')) ) html = driver.page_source return html finally: driver.quit() def parse_results(html, keywords): from bs4 import BeautifulSoup soup = BeautifulSoup(html, "html.parser") results = soup.select("div.g") found_results = [] for result in results: title_tag = result.select_one("h3.LC20lb") desc_tag = result.select_one("div.VwiC3b span") link_tag = result.select_one("a") if not (title_tag and desc_tag and link_tag): continue title = title_tag.get_text(strip=True) description = desc_tag.get_text(strip=True) raw_link = link_tag.get("href") match = re.search(r"/url\?q=(.*?)&", raw_link) if not match: continue link = match.group(1) if any(keyword.lower() in title.lower() or keyword.lower() in description.lower() or keyword.lower() in link.lower() for keyword in keywords): found_results.append({ "title": title, "description": description, "link": link }) return found_results def perform_search(): keywords = ["instagram"] try: html = search_google(" ".join(keywords)) results = parse_results(html, keywords) except Exception as e: print(f"搜索出错: {str(e)}") return if results: with open('results.json', 'w', encoding='utf-8') as f: json.dump(results, f, ensure_ascii=False, indent=2) print("结果已保存到results.json") else: print("未找到匹配结果。") if __name__ == '__main__': perform_search()
Yandex搜索实现示例
import requests from bs4 import BeautifulSoup def search_yandex(query): url = f"https://yandex.com/search/?text={query}" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" } response = requests.get(url, headers=headers) response.raise_for_status() return response.text def parse_yandex_results(html, keywords): soup = BeautifulSoup(html, "html.parser") results = soup.select("li.serp-item") found_results = [] for result in results: title_tag = result.select_one("h2.serp-item__title a") desc_tag = result.select_one("div.serp-item__text") link_tag = result.select_one("h2.serp-item__title a") if not (title_tag and desc_tag and link_tag): continue title = title_tag.get_text(strip=True) description = desc_tag.get_text(strip=True) link = link_tag.get("href") if any(keyword.lower() in title.lower() or keyword.lower() in description.lower() for keyword in keywords): found_results.append({ "title": title, "description": description, "link": link }) return found_results # 示例调用 def yandex_search_keywords(keywords, num_results=5): query = " ".join(keywords) html = search_yandex(query) results = parse_yandex_results(html, keywords) if results: for i, result in enumerate(results[:num_results], 1): print(f"结果 {i}:") print(f"标题: {result['title']}") print(f"描述: {result['description']}") print(f"链接: {result['link']}") print("--------------------") else: print("未找到匹配结果。") yandex_search_keywords(["instagram"], num_results=5)
内容的提问来源于stack exchange,提问作者Orkun Koçak
相关产品推荐
相关产品推荐

