You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium爬Google Shopping下一页遇StaleElementReferenceException错误求助

问题描述

尝试用Selenium爬取Google Shopping下一页数据,点击下一页按钮后程序报错终止,错误信息如下:

Traceback (most recent call last):
File "c:\Users\LP\Documents\python\wedgwood\wedgwood.py", line 50, in
name = card.find_element(By.CLASS_NAME, 'OSrXXb').text.strip()

selenium.common.exceptions.StaleElementReferenceException: Message: stale element reference: element is not attached to the page document

代码实现如下:

from selenium import webdriver
import time
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
import pandas as pd



url = 'https://www.google.com.ng/search?q=list+of+all+uk+e-commerce+stores+for+buying+prada+products&hl=en&biw=946&bih=625&tbm=lcl&sxsrf=ALiCzsaIKyYpvCJVWZx_fYTwSQerSvzC6g%3A1667482905673&ei=GcVjY4fUKJeG9u8PgvGwoAE&ved=0ahUKEwjHxIvykZL7AhUXg_0HHYI4DBQQ4dUDCAk&uact=5&oq=list+of+all+uk+e-commerce+stores+for+buying+prada+products&gs_lp=Eg1nd3Mtd2l6LWxvY2FsuAED-AEBMgUQABiiBDIHEAAYHhiiBDIFEAAYogQyBRAAGKIEwgIEECMYJ0iSHFDlBliOFHAAeADIAQCQAQCYAYYDoAHxDqoBBTItMS41iAYB&sclient=gws-wiz-local#rlfi=hd:;si:;mv:[[56.121909699999996,0.16756959999999999],[51.208233299999996,-4.5053765]]'

service = Service(executable_path="C:/driver/chromedriver_win32/chromedriver.exe")

driver = webdriver.Chrome(service=service)

driver.get(url)

driver.maximize_window()

time.sleep(8)



for i in range(7):    

   site_cards = driver.find_elements(By.CLASS_NAME, 'uMdZh')
   time.sleep(4)

   site_list = []

   for card in site_cards:
      name = card.find_element(By.CLASS_NAME, 'OSrXXb').text.strip()
      submit = card.find_element(By.CLASS_NAME, 'OSrXXb')
      submit.click()
      time.sleep(4)
      try:
        more = driver.find_element(By.CLASS_NAME, 'Yy0acb').text.strip()
      except:
        print('none')
      try:
        more = driver.find_element(By.CLASS_NAME, 'mPcsfb').text.strip()
      except:
        print('none')
      time.sleep(2)
      try:
        more = driver.find_element(By.CLASS_NAME, 'YhemCb').text.strip()
      except:
        print('none')
      time.sleep(2)
      try:
        more = driver.find_element(By.CLASS_NAME, 'PQbOE').text.strip()
      except:
        print('none')
      try:
        more = driver.find_element(By.CLASS_NAME, 'Yy0acb').text.strip()
      except:
        print('none')
      try:
        more = driver.find_element(By.NAME, 'EvNWZc').text.strip()
      except:
        print('none')
      time.sleep(4)


      if ModuleNotFoundError:
        pass

      site_info = (name, more)
      site_list.append(site_info)

      col = ['Site Name', 'Site Link']
      df = pd.DataFrame([site_info], columns=col)
      df.to_csv("C:\Users\LP\Documents\python\wedgwood\prada2.csv", index=False, encoding='utf-8', mode='a+')
    
next_page = driver.find_element(By.XPATH, '//*[@id="pnnext"]')
next_page.click()
问题原因与解决方案

错误原因

StaleElementReferenceException 是因为页面刷新或跳转后,之前获取的元素对象已失效——页面DOM重新渲染,旧元素不再属于当前页面的文档结构,再调用旧元素的方法就会报错。代码中点击卡片后页面会更新,翻页操作也会完全刷新页面,之前缓存的site_cards列表里的元素都会变成过期状态。

修复方案

  1. 每次操作后重新获取元素:不要提前缓存元素列表,需要时重新查找,避免使用过期的元素引用。
  2. 替换固定等待为显式等待:time.sleep()效率低且不稳定,改用Selenium的WebDriverWait等待元素加载完成,更可靠。
  3. 调整翻页逻辑:把翻页操作放到循环内,确保每页都重新抓取数据,同时处理翻页可能出现的异常(比如最后一页没有下一页按钮)。

修改后的代码

from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import pandas as pd

url = 'https://www.google.com.ng/search?q=list+of+all+uk+e-commerce+stores+for+buying+prada+products&hl=en&biw=946&bih=625&tbm=lcl&sxsrf=ALiCzsaIKyYpvCJVWZx_fYTwSQerSvzC6g%3A1667482905673&ei=GcVjY4fUKJeG9u8PgvGwoAE&ved=0ahUKEwjHxIvykZL7AhUXg_0HHYI4DBQQ4dUDCAk&uact=5&oq=list+of+all+uk+e-commerce+stores+for+buying+prada+products&gs_lp=Eg1nd3Mtd2l6LWxvY2FsuAED-AEBMgUQABiiBDIHEAAYHhiiBDIFEAAYogQyBRAAGKIEwgIEECMYJ0iSHFDlBliOFHAAeADIAQCQAQCYAYYDoAHxDqoBBTItMS41iAYB&sclient=gws-wiz-local#rlfi=hd:;si:;mv:[[56.121909699999996,0.16756959999999999],[51.208233299999996,-4.5053765]]'

service = Service(executable_path="C:/driver/chromedriver_win32/chromedriver.exe")
driver = webdriver.Chrome(service=service)
driver.get(url)
driver.maximize_window()

# 初始化显式等待,最长等待10秒
wait = WebDriverWait(driver, 10)

# 定义csv列名,首次写入时创建表头
col = ['Site Name', 'Site Link']
first_write = True

for i in range(7):    
    try:
        # 等待卡片加载完成,重新获取当前页面的所有卡片
        site_cards = wait.until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'uMdZh')))
        site_list = []

        for card in site_cards:
            try:
                # 重新获取卡片内的名称元素,避免过期
                name_elem = card.find_element(By.CLASS_NAME, 'OSrXXb')
                name = name_elem.text.strip()
                name_elem.click()
                
                # 等待详情加载,尝试获取链接
                more = 'none'
                # 按优先级查找元素,找到后立即停止
                selectors = [
                    (By.CLASS_NAME, 'Yy0acb'),
                    (By.CLASS_NAME, 'mPcsfb'),
                    (By.CLASS_NAME, 'YhemCb'),
                    (By.CLASS_NAME, 'PQbOE'),
                    (By.NAME, 'EvNWZc')
                ]
                for by, val in selectors:
                    try:
                        elem = wait.until(EC.presence_of_element_located((by, val)))
                        more = elem.text.strip()
                        break
                    except:
                        continue
                
                site_info = (name, more)
                site_list.append(site_info)

                # 写入csv,首次写入表头
                df = pd.DataFrame([site_info], columns=col)
                df.to_csv(r"C:\Users\LP\Documents\python\wedgwood\prada2.csv", 
                          index=False, encoding='utf-8', 
                          mode='w' if first_write else 'a+', 
                          header=first_write)
                first_write = False

                # 返回上一页,准备处理下一个卡片
                driver.back()
                # 等待列表页面重新加载完成
                wait.until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'uMdZh')))
            except Exception as e:
                print(f"处理卡片出错: {str(e)}")
                driver.back()
                continue
        
        # 翻页操作,等待下一页按钮加载并点击
        try:
            next_page = wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="pnnext"]')))
            next_page.click()
            # 等待新页面加载完成
            wait.until(EC.staleness_of(site_cards[0]))  # 等待旧页面元素过期,确认页面已刷新
        except Exception as e:
            print(f"翻页出错或已到最后一页: {str(e)}")
            break
    except Exception as e:
        print(f"处理第{i+1}页出错: {str(e)}")
        break

driver.quit()

关键修改点

  • 显式等待替代固定等待:通过WebDriverWait等待元素出现或可点击,提升稳定性,避免页面加载慢导致的元素未找到问题。
  • 实时获取元素:处理每个卡片前、返回列表页后、翻页后都重新查找元素,彻底避免过期元素引用。
  • 优化链接查找逻辑:按优先级遍历选择器,找到有效元素后立即停止,减少冗余操作。
  • 修复CSV写入逻辑:首次写入时添加表头,使用原始字符串路径避免转义问题。
  • 增加异常捕获:对卡片处理、翻页操作添加异常捕获,单个错误不会导致整个程序终止。

内容的提问来源于stack exchange,提问作者Miracle

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.14 14:45:32