You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Google Maps Selenium爬虫处理第二个商家时出现元素找不到异常求助

问题描述

脚本读取Excel文件中的商家名称,在Google Maps搜索并抓取数据,第一个商家可正常抓取,第二个商家详情页抛出NoSuchElementException错误。

原代码

from selenium import webdriver
from openpyxl import load_workbook
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options as ChromeOptions
import time

import pandas as pd 
chrome_options = ChromeOptions()
Options = webdriver.ChromeOptions()
geckodriver_path = Service(r'E:\chromedriver_win32\chromedriver.exe')


driver = webdriver.Chrome(service=geckodriver_path, options=Options)

import pandas as pd

# Load the Excel workbook and select the sheet
workbook = load_workbook(r'C:\Users\muneeb\Desktop\Business name.xlsx')
sheet = workbook.active

# Create a list of business names from the first column of the sheet
business_names = [cell.value for cell in sheet['A']] 
for business_name in business_names:
    # Go to Google Maps
    driver.get('https://www.google.com/maps/@24.9349012,67.2006144,15z')
    
    # Find the search box and enter the business name
    search_box = driver.find_element(By.ID, "searchboxinput")
    search_box.send_keys(business_name)
    # Find the first result and click on it
    element = driver.find_element(By.CLASS_NAME, "mL3xi")
    element.click()
    
    # Wait for the business details to load
    time.sleep(20)
    # Scrape the business details
    name = driver.find_element(By.XPATH,"/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[2]/div[1]/div[1]/div[1]/h1/span[1]").text
    phone = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[5]/button/div[1]/div[2]/div[1]').text
    website = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[4]/a/div[1]/div[2]/div[1]').text
    reviews = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[2]/div[1]/div[1]/div[2]/div/div[1]/div[2]/span[2]/span[1]/span').text
    address = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[3]/button/div[1]/div[2]/div[1]').text

print(len(phone),len(name),len(website),len(reviews),len(address))

错误信息

Traceback (most recent call last):
  File "e:\Muneeb data\Webscraping\googlemaps_scraping.py", line 58, in <module>
    phone = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[5]/button/div[1]/div[2]/div[1]').text
            ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\muneeb\AppData\Roaming\Python\Python311\site-packages\selenium\webdriver\remote\webdriver.py", line 830, in find_element
    return self.execute(Command.FIND_ELEMENT, {"using": by, "value": value})["value"]
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\muneeb\AppData\Roaming\Python\Python311\site-packages\selenium\webdriver\remote\webdriver.py", line 440, in execute
    self.error_handler.check_response(response)
  File "C:\Users\muneeb\AppData\Roaming\Python\Python311\site-packages\selenium\webdriver\remote\errorhandler.py", line 245, in check_response
    raise exception_class(message, screen, stacktrace)
selenium.common.exceptions.NoSuchElementException: Message: no such element: Unable to locate element: {"method":"xpath","selector":"/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[5]/button/div[1]/div[2]/div[1]"}

解决方案

问题根源

  • 绝对XPath不稳定:Google Maps的DOM结构动态变化,不同商家的详情模块顺序、层级可能不同,固定的绝对XPath无法适配所有商家。
  • 固定等待不可靠:time.sleep(20)不能保证元素一定加载完成,第二个商家页面可能加载较慢,导致元素未渲染就执行抓取。
  • 未触发搜索动作:代码仅在搜索框输入名称,未点击搜索按钮或回车,第一个商家可能是缓存结果,第二个商家无法正确触发搜索。
  • 缺失异常处理:部分商家可能没有电话、网站等字段,未捕获异常会导致脚本直接中断。

修复后的代码

from selenium import webdriver
from openpyxl import load_workbook
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options as ChromeOptions
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import NoSuchElementException

# 配置Chrome选项
chrome_options = ChromeOptions()
# 可选添加无头模式:chrome_options.add_argument("--headless=new")

# 初始化驱动与显式等待
driver = webdriver.Chrome(service=Service(r'E:\chromedriver_win32\chromedriver.exe'), options=chrome_options)
wait = WebDriverWait(driver, 30)  # 最长等待30秒

# 加载Excel文件并过滤空商家名称
workbook = load_workbook(r'C:\Users\muneeb\Desktop\Business name.xlsx')
sheet = workbook.active
business_names = [cell.value for cell in sheet['A'] if cell.value is not None]

# 存储抓取结果
results = []

for business_name in business_names:
    try:
        # 打开Google Maps首页
        driver.get('https://www.google.com/maps')
        
        # 等待搜索框并输入商家名称,按回车触发搜索
        search_box = wait.until(EC.visibility_of_element_located((By.ID, "searchboxinput")))
        search_box.clear()
        search_box.send_keys(business_name)
        search_box.send_keys(Keys.ENTER)
        
        # 等待第一个搜索结果并点击
        first_result = wait.until(EC.element_to_be_clickable((By.CLASS_NAME, "mL3xi")))
        first_result.click()
        
        # 抓取商家名称
        name = wait.until(EC.visibility_of_element_located((By.XPATH, "//h1[contains(@class, 'fontHeadlineLarge')]/span"))).text
        
        # 抓取地址(处理可能缺失的情况)
        address = ""
        try:
            address = wait.until(EC.visibility_of_element_located((By.XPATH, "//button[contains(@aria-label, '地址')]/div[2]/div[1]"))).text
        except NoSuchElementException:
            pass
        
        # 抓取电话(处理可能缺失的情况)
        phone = ""
        try:
            phone = wait.until(EC.visibility_of_element_located((By.XPATH, "//button[contains(@aria-label, '电话')]/div[2]/div[1]"))).text
        except NoSuchElementException:
            pass
        
        # 抓取网站(处理可能缺失的情况)
        website = ""
        try:
            website = wait.until(EC.visibility_of_element_located((By.XPATH, "//a[contains(@aria-label, '网站')]/div[2]/div[1]"))).text
        except NoSuchElementException:
            pass
        
        # 抓取评论数(处理可能缺失的情况)
        reviews = ""
        try:
            reviews = wait.until(EC.visibility_of_element_located((By.XPATH, "//div[contains(@aria-label, '条评论')]/span[2]/span[1]/span"))).text
        except NoSuchElementException:
            pass
        
        results.append({
            "商家名称": name,
            "地址": address,
            "电话": phone,
            "网站": website,
            "评论数": reviews
        })
        print(f"已抓取:{name}")
        
    except Exception as e:
        print(f"抓取{business_name}时出错:{str(e)}")
        continue

# 打印各字段长度
for res in results:
    print(f"名称长度:{len(res['商家名称'])}, 电话长度:{len(res['电话'])}, 网站长度:{len(res['网站'])}, 评论数长度:{len(res['评论数'])}, 地址长度:{len(res['地址'])}")

# 可选:保存结果到Excel
# import pandas as pd
# pd.DataFrame(results).to_excel("商家信息.xlsx", index=False)

driver.quit()

关键改进点

  • 显式等待替代固定sleep:通过WebDriverWait等待元素可见/可点击,确保元素加载完成后再操作,避免因加载速度差异导致的错误。
  • 相对XPath定位:基于元素的aria-label属性(包含"地址"、"电话"等文本)定位,适配不同商家的DOM结构变化。
  • 触发搜索动作:输入名称后按Enter键触发搜索,确保每次搜索都是最新结果,避免依赖缓存。
  • 异常处理:对每个字段单独捕获NoSuchElementException,单个字段缺失不会导致脚本中断。
  • 过滤空值:排除Excel中为空的商家名称,避免无效搜索。

内容的提问来源于stack exchange,提问作者Muneeb Ur rehman

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.02 06:35:17