Google Maps评论爬取超时问题求助(附Python代码)
Google Maps评论爬取超时问题解决
我编写Python脚本爬取Google Maps特定商铺的全部评论,但始终触发TimeoutException超时异常。Chrome版本为122.0.6261.113,chromedriver使用对应版本的122.0.6261.128,之前参考的博客内容已过时,原代码如下:
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.common.exceptions import TimeoutException def scrape_google_reviews(url): # Set up Chrome WebDriver chrome_options = Options() chrome_options.add_argument("--headless") # Run in headless mode, i.e., without opening browser window chromedriver_path = 'C:/Users/Downloads/chromedriver-win64/chromedriver.exe' # Specify path to chromedriver executable service = Service(chromedriver_path) driver = webdriver.Chrome(service=service, options=chrome_options) # Load the Google Maps URL driver.get(url) # Wait for the reviews to load try: WebDriverWait(driver, 120).until(EC.presence_of_element_located((By.CLASS_NAME, "ODSEW-ShBeI-content"))) except TimeoutException as e: print("Timeout occurred while waiting for reviews to load:", e) driver.quit() return None except Exception as e: print("An error occurred while waiting for reviews to load:", e) driver.quit() return None # Extract review elements review_elements = driver.find_elements(By.CLASS_NAME, "ODSEW-ShBeI-content") # Extract review details reviews = [] for review_element in review_elements: review_text = review_element.find_element(By.CSS_SELECTOR, ".ODSEW-ShBeI-title").text reviews.append(review_text) # Close the WebDriver driver.quit() return reviews # Example usage url = "https://www.google.com/maps/place/FASTECH+SOLUTIONS/@18.5165309,73.8457059,18.29z/data=!4m6!3m5!1s0x3bc2c160b5caf2dd:0x6d49235d88bd5d25!8m2!3d18.5161858!4d73.8459712!16s%2Fg%2F11t7drcv4g?entry=ttu" reviews = scrape_google_reviews(url) if reviews: for i, review in enumerate(reviews, 1): print(f"Review {i}: {review}") else: print("Failed to scrape reviews.")
问题分析
超时主要由三个核心原因导致:
- 无头模式被反爬识别:默认无头模式的Chrome特征过于明显,会被Google反爬机制拦截,导致评论页面无法正常加载。
- 元素选择器过时:代码中使用的
ODSEW-ShBeI-content等类名是Google Maps旧版DOM结构,当前页面已更新,无法定位到目标元素。 - 懒加载未处理:Google Maps评论采用滚动懒加载机制,仅等待初始元素无法触发全部评论加载,甚至部分页面需要先点击"查看全部评论"入口才能进入完整评论区。
修复后的代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.common.exceptions import TimeoutException, NoSuchElementException import time def scrape_google_reviews(url): # 配置Chrome选项,模拟真实浏览器环境 chrome_options = Options() chrome_options.add_argument("--headless=new") # 新版无头模式,更接近真实浏览器特征 chrome_options.add_argument("--window-size=1920,1080") # 设置标准窗口尺寸 chrome_options.add_argument("--disable-blink-features=AutomationControlled") # 禁用自动化检测 chrome_options.add_argument("--user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36") chrome_options.add_experimental_option("excludeSwitches", ["enable-automation"]) chrome_options.add_experimental_option('useAutomationExtension', False) chromedriver_path = 'C:/Users/Downloads/chromedriver-win64/chromedriver.exe' service = Service(chromedriver_path) driver = webdriver.Chrome(service=service, options=chrome_options) # 隐藏webdriver标识,规避反爬检测 driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})") try: driver.get(url) # 尝试点击"查看全部评论"入口(部分页面需要触发) try: review_tab = WebDriverWait(driver, 30).until( EC.element_to_be_clickable((By.XPATH, '//button[@aria-label="查看全部评论"]')) ) review_tab.click() time.sleep(2) except NoSuchElementException: # 部分页面直接显示评论区,跳过点击逻辑 pass # 循环滚动加载全部评论 last_height = driver.execute_script("return document.body.scrollHeight") while True: driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(3) # 等待懒加载内容加载完成 new_height = driver.execute_script("return document.body.scrollHeight") if new_height == last_height: break # 滚动到底部,无新内容加载 last_height = new_height # 等待评论容器加载完成,使用当前有效选择器 WebDriverWait(driver, 30).until( EC.presence_of_all_elements_located((By.CSS_SELECTOR, 'div.jftiEf')) ) # 提取评论文本内容 review_elements = driver.find_elements(By.CSS_SELECTOR, 'div.jftiEf') reviews = [] for elem in review_elements: try: review_text = elem.find_element(By.CSS_SELECTOR, 'span.wiI7pd').text if review_text: reviews.append(review_text) except NoSuchElementException: continue return reviews except TimeoutException as e: print(f"加载超时: {e}") return None finally: driver.quit() # 使用示例 url = "https://www.google.com/maps/place/FASTECH+SOLUTIONS/@18.5165309,73.8457059,18.29z/data=!4m6!3m5!1s0x3bc2c160b5caf2dd:0x6d49235d88bd5d25!8m2!3d18.5161858!4d73.8459712!16s%2Fg%2F11t7drcv4g?entry=ttu" reviews = scrape_google_reviews(url) if reviews: for idx, review in enumerate(reviews, 1): print(f"评论 {idx}: {review}") else: print("未能爬取到评论")
关键修改点
- 反爬规避:启用新版无头模式、配置真实用户代理、禁用自动化特征检测,让浏览器行为更接近普通用户。
- 选择器更新:替换为Google Maps当前有效的评论容器选择器
div.jftiEf和评论文本选择器span.wiI7pd。 - 懒加载处理:添加循环滚动逻辑,触发所有评论的懒加载。
- 入口处理:自动识别并点击"查看全部评论"按钮,确保进入完整评论页面。
内容的提问来源于stack exchange,提问作者Masoom Raza
相关产品推荐
相关产品推荐

