Google Maps Selenium爬虫处理第二个商家时出现元素找不到异常求助
问题描述
脚本读取Excel文件中的商家名称,在Google Maps搜索并抓取数据,第一个商家可正常抓取,第二个商家详情页抛出NoSuchElementException错误。
原代码
from selenium import webdriver from openpyxl import load_workbook from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.chrome.options import Options as ChromeOptions import time import pandas as pd chrome_options = ChromeOptions() Options = webdriver.ChromeOptions() geckodriver_path = Service(r'E:\chromedriver_win32\chromedriver.exe') driver = webdriver.Chrome(service=geckodriver_path, options=Options) import pandas as pd # Load the Excel workbook and select the sheet workbook = load_workbook(r'C:\Users\muneeb\Desktop\Business name.xlsx') sheet = workbook.active # Create a list of business names from the first column of the sheet business_names = [cell.value for cell in sheet['A']] for business_name in business_names: # Go to Google Maps driver.get('https://www.google.com/maps/@24.9349012,67.2006144,15z') # Find the search box and enter the business name search_box = driver.find_element(By.ID, "searchboxinput") search_box.send_keys(business_name) # Find the first result and click on it element = driver.find_element(By.CLASS_NAME, "mL3xi") element.click() # Wait for the business details to load time.sleep(20) # Scrape the business details name = driver.find_element(By.XPATH,"/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[2]/div[1]/div[1]/div[1]/h1/span[1]").text phone = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[5]/button/div[1]/div[2]/div[1]').text website = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[4]/a/div[1]/div[2]/div[1]').text reviews = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[2]/div[1]/div[1]/div[2]/div/div[1]/div[2]/span[2]/span[1]/span').text address = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[3]/button/div[1]/div[2]/div[1]').text print(len(phone),len(name),len(website),len(reviews),len(address))
错误信息
Traceback (most recent call last): File "e:\Muneeb data\Webscraping\googlemaps_scraping.py", line 58, in <module> phone = driver.find_element(By.XPATH,'/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[5]/button/div[1]/div[2]/div[1]').text ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\muneeb\AppData\Roaming\Python\Python311\site-packages\selenium\webdriver\remote\webdriver.py", line 830, in find_element return self.execute(Command.FIND_ELEMENT, {"using": by, "value": value})["value"] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\muneeb\AppData\Roaming\Python\Python311\site-packages\selenium\webdriver\remote\webdriver.py", line 440, in execute self.error_handler.check_response(response) File "C:\Users\muneeb\AppData\Roaming\Python\Python311\site-packages\selenium\webdriver\remote\errorhandler.py", line 245, in check_response raise exception_class(message, screen, stacktrace) selenium.common.exceptions.NoSuchElementException: Message: no such element: Unable to locate element: {"method":"xpath","selector":"/html/body/div[3]/div[9]/div[9]/div/div/div[1]/div[2]/div/div[1]/div/div/div[7]/div[5]/button/div[1]/div[2]/div[1]"}
解决方案
问题根源
- 绝对XPath不稳定:Google Maps的DOM结构动态变化,不同商家的详情模块顺序、层级可能不同,固定的绝对XPath无法适配所有商家。
- 固定等待不可靠:
time.sleep(20)不能保证元素一定加载完成,第二个商家页面可能加载较慢,导致元素未渲染就执行抓取。 - 未触发搜索动作:代码仅在搜索框输入名称,未点击搜索按钮或回车,第一个商家可能是缓存结果,第二个商家无法正确触发搜索。
- 缺失异常处理:部分商家可能没有电话、网站等字段,未捕获异常会导致脚本直接中断。
修复后的代码
from selenium import webdriver from openpyxl import load_workbook from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.chrome.options import Options as ChromeOptions from selenium.webdriver.common.keys import Keys from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.common.exceptions import NoSuchElementException # 配置Chrome选项 chrome_options = ChromeOptions() # 可选添加无头模式:chrome_options.add_argument("--headless=new") # 初始化驱动与显式等待 driver = webdriver.Chrome(service=Service(r'E:\chromedriver_win32\chromedriver.exe'), options=chrome_options) wait = WebDriverWait(driver, 30) # 最长等待30秒 # 加载Excel文件并过滤空商家名称 workbook = load_workbook(r'C:\Users\muneeb\Desktop\Business name.xlsx') sheet = workbook.active business_names = [cell.value for cell in sheet['A'] if cell.value is not None] # 存储抓取结果 results = [] for business_name in business_names: try: # 打开Google Maps首页 driver.get('https://www.google.com/maps') # 等待搜索框并输入商家名称,按回车触发搜索 search_box = wait.until(EC.visibility_of_element_located((By.ID, "searchboxinput"))) search_box.clear() search_box.send_keys(business_name) search_box.send_keys(Keys.ENTER) # 等待第一个搜索结果并点击 first_result = wait.until(EC.element_to_be_clickable((By.CLASS_NAME, "mL3xi"))) first_result.click() # 抓取商家名称 name = wait.until(EC.visibility_of_element_located((By.XPATH, "//h1[contains(@class, 'fontHeadlineLarge')]/span"))).text # 抓取地址(处理可能缺失的情况) address = "" try: address = wait.until(EC.visibility_of_element_located((By.XPATH, "//button[contains(@aria-label, '地址')]/div[2]/div[1]"))).text except NoSuchElementException: pass # 抓取电话(处理可能缺失的情况) phone = "" try: phone = wait.until(EC.visibility_of_element_located((By.XPATH, "//button[contains(@aria-label, '电话')]/div[2]/div[1]"))).text except NoSuchElementException: pass # 抓取网站(处理可能缺失的情况) website = "" try: website = wait.until(EC.visibility_of_element_located((By.XPATH, "//a[contains(@aria-label, '网站')]/div[2]/div[1]"))).text except NoSuchElementException: pass # 抓取评论数(处理可能缺失的情况) reviews = "" try: reviews = wait.until(EC.visibility_of_element_located((By.XPATH, "//div[contains(@aria-label, '条评论')]/span[2]/span[1]/span"))).text except NoSuchElementException: pass results.append({ "商家名称": name, "地址": address, "电话": phone, "网站": website, "评论数": reviews }) print(f"已抓取:{name}") except Exception as e: print(f"抓取{business_name}时出错:{str(e)}") continue # 打印各字段长度 for res in results: print(f"名称长度:{len(res['商家名称'])}, 电话长度:{len(res['电话'])}, 网站长度:{len(res['网站'])}, 评论数长度:{len(res['评论数'])}, 地址长度:{len(res['地址'])}") # 可选:保存结果到Excel # import pandas as pd # pd.DataFrame(results).to_excel("商家信息.xlsx", index=False) driver.quit()
关键改进点
- 显式等待替代固定sleep:通过
WebDriverWait等待元素可见/可点击,确保元素加载完成后再操作,避免因加载速度差异导致的错误。 - 相对XPath定位:基于元素的
aria-label属性(包含"地址"、"电话"等文本)定位,适配不同商家的DOM结构变化。 - 触发搜索动作:输入名称后按
Enter键触发搜索,确保每次搜索都是最新结果,避免依赖缓存。 - 异常处理:对每个字段单独捕获
NoSuchElementException,单个字段缺失不会导致脚本中断。 - 过滤空值:排除Excel中为空的商家名称,避免无效搜索。
内容的提问来源于stack exchange,提问作者Muneeb Ur rehman
相关产品推荐
相关产品推荐

