You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Google Maps评论爬取超时问题求助(附Python代码)

Google Maps评论爬取超时问题解决

我编写Python脚本爬取Google Maps特定商铺的全部评论,但始终触发TimeoutException超时异常。Chrome版本为122.0.6261.113,chromedriver使用对应版本的122.0.6261.128,之前参考的博客内容已过时,原代码如下:

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException

def scrape_google_reviews(url):
    # Set up Chrome WebDriver
    chrome_options = Options()
    chrome_options.add_argument("--headless")  # Run in headless mode, i.e., without opening browser window
    chromedriver_path = 'C:/Users/Downloads/chromedriver-win64/chromedriver.exe'  # Specify path to chromedriver executable
    service = Service(chromedriver_path)
    driver = webdriver.Chrome(service=service, options=chrome_options)

    # Load the Google Maps URL
    driver.get(url)

    # Wait for the reviews to load
    try:
        WebDriverWait(driver, 120).until(EC.presence_of_element_located((By.CLASS_NAME, "ODSEW-ShBeI-content")))
    except TimeoutException as e:
        print("Timeout occurred while waiting for reviews to load:", e)
        driver.quit()
        return None
    except Exception as e:
        print("An error occurred while waiting for reviews to load:", e)
        driver.quit()
        return None

    # Extract review elements
    review_elements = driver.find_elements(By.CLASS_NAME, "ODSEW-ShBeI-content")

    # Extract review details
    reviews = []
    for review_element in review_elements:
        review_text = review_element.find_element(By.CSS_SELECTOR, ".ODSEW-ShBeI-title").text
        reviews.append(review_text)

    # Close the WebDriver
    driver.quit()

    return reviews

# Example usage
url = "https://www.google.com/maps/place/FASTECH+SOLUTIONS/@18.5165309,73.8457059,18.29z/data=!4m6!3m5!1s0x3bc2c160b5caf2dd:0x6d49235d88bd5d25!8m2!3d18.5161858!4d73.8459712!16s%2Fg%2F11t7drcv4g?entry=ttu"
reviews = scrape_google_reviews(url)
if reviews:
    for i, review in enumerate(reviews, 1):
        print(f"Review {i}: {review}")
else:
    print("Failed to scrape reviews.")

问题分析

超时主要由三个核心原因导致:

  1. 无头模式被反爬识别:默认无头模式的Chrome特征过于明显,会被Google反爬机制拦截,导致评论页面无法正常加载。
  2. 元素选择器过时:代码中使用的ODSEW-ShBeI-content等类名是Google Maps旧版DOM结构,当前页面已更新,无法定位到目标元素。
  3. 懒加载未处理:Google Maps评论采用滚动懒加载机制,仅等待初始元素无法触发全部评论加载,甚至部分页面需要先点击"查看全部评论"入口才能进入完整评论区。

修复后的代码

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException, NoSuchElementException
import time

def scrape_google_reviews(url):
    # 配置Chrome选项,模拟真实浏览器环境
    chrome_options = Options()
    chrome_options.add_argument("--headless=new")  # 新版无头模式,更接近真实浏览器特征
    chrome_options.add_argument("--window-size=1920,1080")  # 设置标准窗口尺寸
    chrome_options.add_argument("--disable-blink-features=AutomationControlled")  # 禁用自动化检测
    chrome_options.add_argument("--user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36")
    chrome_options.add_experimental_option("excludeSwitches", ["enable-automation"])
    chrome_options.add_experimental_option('useAutomationExtension', False)

    chromedriver_path = 'C:/Users/Downloads/chromedriver-win64/chromedriver.exe'
    service = Service(chromedriver_path)
    driver = webdriver.Chrome(service=service, options=chrome_options)
    # 隐藏webdriver标识,规避反爬检测
    driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})")

    try:
        driver.get(url)
        # 尝试点击"查看全部评论"入口(部分页面需要触发)
        try:
            review_tab = WebDriverWait(driver, 30).until(
                EC.element_to_be_clickable((By.XPATH, '//button[@aria-label="查看全部评论"]'))
            )
            review_tab.click()
            time.sleep(2)
        except NoSuchElementException:
            # 部分页面直接显示评论区,跳过点击逻辑
            pass

        # 循环滚动加载全部评论
        last_height = driver.execute_script("return document.body.scrollHeight")
        while True:
            driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
            time.sleep(3)  # 等待懒加载内容加载完成
            new_height = driver.execute_script("return document.body.scrollHeight")
            if new_height == last_height:
                break  # 滚动到底部,无新内容加载
            last_height = new_height

        # 等待评论容器加载完成,使用当前有效选择器
        WebDriverWait(driver, 30).until(
            EC.presence_of_all_elements_located((By.CSS_SELECTOR, 'div.jftiEf'))
        )

        # 提取评论文本内容
        review_elements = driver.find_elements(By.CSS_SELECTOR, 'div.jftiEf')
        reviews = []
        for elem in review_elements:
            try:
                review_text = elem.find_element(By.CSS_SELECTOR, 'span.wiI7pd').text
                if review_text:
                    reviews.append(review_text)
            except NoSuchElementException:
                continue

        return reviews

    except TimeoutException as e:
        print(f"加载超时: {e}")
        return None
    finally:
        driver.quit()

# 使用示例
url = "https://www.google.com/maps/place/FASTECH+SOLUTIONS/@18.5165309,73.8457059,18.29z/data=!4m6!3m5!1s0x3bc2c160b5caf2dd:0x6d49235d88bd5d25!8m2!3d18.5161858!4d73.8459712!16s%2Fg%2F11t7drcv4g?entry=ttu"
reviews = scrape_google_reviews(url)
if reviews:
    for idx, review in enumerate(reviews, 1):
        print(f"评论 {idx}: {review}")
else:
    print("未能爬取到评论")

关键修改点

  1. 反爬规避:启用新版无头模式、配置真实用户代理、禁用自动化特征检测,让浏览器行为更接近普通用户。
  2. 选择器更新:替换为Google Maps当前有效的评论容器选择器div.jftiEf和评论文本选择器span.wiI7pd。
  3. 懒加载处理:添加循环滚动逻辑,触发所有评论的懒加载。
  4. 入口处理:自动识别并点击"查看全部评论"按钮,确保进入完整评论页面。

内容的提问来源于stack exchange,提问作者Masoom Raza

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.27 19:24:52