You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Selenium WebDriver爬取亚马逊时无法进入第二页的问题

亚马逊爬虫分页报错问题解决

我爬取亚马逊商品名称、价格、图片等信息时,需要进入第二页继续获取数据。首次运行代码一切正常,但后续多次运行均报错(报错为NoSuchElementException,无法定位分页按钮元素)。以下是我的代码,问题出在底部获取第二页元素的部分:

from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
import random
import time


URL = "https://www.amazon.com/s?k=laptop&crid=288NMI7Z5E2WR&sprefix=laptop%2Caps%2C572&ref=nb_sb_noss_1"

user_agents = [
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/94.0.4606.71 Safari/537.36",
    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/94.0.4606.71 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/93.0.4577.63 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.107 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/90.0.4430.212 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/89.0.4389.82 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/88.0.4324.104 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/87.0.4280.88 Safari/537.36",
]

service = Service()
options = webdriver.ChromeOptions()
options.add_argument(f"user-agent={random.choice(user_agents)}")
driver = webdriver.Chrome(service=service, options=options)

driver.get(URL)
time.sleep(random.uniform(2, 3))

web_page = driver.page_source
# driver.quit()

soup = BeautifulSoup(web_page, 'lxml')


image_boxes = soup.find_all("div", class_="s-product-image-container aok-relative s-text-center s-image-overlay-grey puis-image-overlay-grey s-padding-left-small s-padding-right-small puis-flex-expand-height puis puis-v25h6k2ialgxcx2aefs5gqn1u23")
boxes = soup.find_all("div", class_="puisg-col puisg-col-4-of-12 puisg-col-8-of-16 puisg-col-12-of-20 puisg-col-12-of-24 puis-list-col-right")

name_list = []
price_list = []
number_of_reviews_list = []
images_list = []

for box in boxes:
    name = box.find("span", class_="a-size-medium a-color-base a-text-normal").getText()
    price = box.find("span", class_="a-offscreen").getText()
    number_of_reviews = box.find("span", class_="a-size-base s-underline-text").getText()

    name_list.append(name)
    price_list.append(price)
    number_of_reviews_list.append(number_of_reviews)

    
for image_box in image_boxes:
    images = image_box.find("img", class_="s-image").get('src')
    images_list.append(images)


second_page = driver.find_element(By.CSS_SELECTOR, 's-pagination-item s-pagination-button')
second_page.click()

time.sleep(5)
driver.quit()

问题分析与解决方案

1. CSS选择器语法错误

原代码中定位第二页按钮的CSS选择器写法错误:'s-pagination-item s-pagination-button' 是两个类名,但CSS选择器中多个类名需要用点号分隔(表示同时拥有这两个类的元素),正确写法应为:

By.CSS_SELECTOR, '.s-pagination-item.s-pagination-button'

或者更精准定位第二页(包含文本"2"的分页按钮):

By.XPATH, '//a[@class="s-pagination-item s-pagination-button" and text()="2"]'

2. 反爬机制导致元素无法定位

亚马逊会识别频繁的爬虫行为,多次运行后可能更改页面结构、隐藏元素或触发验证。解决方法:

  • 用显式等待替代固定time.sleep,确保元素加载完成后再定位:
    先导入相关模块:
    from selenium.webdriver.support.ui import WebDriverWait
    from selenium.webdriver.support import expected_conditions as EC
    
    然后替换原有的分页按钮定位代码:
    # 等待第二页按钮可点击,最长等待10秒
    second_page = WebDriverWait(driver, 10).until(
        EC.element_to_be_clickable((By.XPATH, '//a[@class="s-pagination-item s-pagination-button" and text()="2"]'))
    )
    second_page.click()
    
  • 扩大随机延迟范围,避免固定间隔:
    time.sleep(random.uniform(3, 5))
    
  • 可选:添加--headless=new启动无头模式,减少资源占用:
    options.add_argument("--headless=new")
    

3. 其他潜在问题

  • 原代码中使用的长类名(如s-product-image-container后的动态类名)是亚马逊生成的临时类,可能随时变化,建议改用更稳定的选择器(比如通过data-component-type属性定位商品容器)。
  • 多次运行后可考虑清理浏览器缓存或更换IP,避免被封禁。

修正后的完整代码

from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import random
import time


URL = "https://www.amazon.com/s?k=laptop&crid=288NMI7Z5E2WR&sprefix=laptop%2Caps%2C572&ref=nb_sb_noss_1"

user_agents = [
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/94.0.4606.71 Safari/537.36",
    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/94.0.4606.71 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/93.0.4577.63 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.107 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/90.0.4430.212 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/89.0.4389.82 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/88.0.4324.104 Safari/537.36",
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/87.0.4280.88 Safari/537.36",
]

service = Service()
options = webdriver.ChromeOptions()
options.add_argument(f"user-agent={random.choice(user_agents)}")
# 可选:无头模式
# options.add_argument("--headless=new")
driver = webdriver.Chrome(service=service, options=options)

try:
    driver.get(URL)
    time.sleep(random.uniform(3, 5))

    web_page = driver.page_source
    soup = BeautifulSoup(web_page, 'lxml')

    # 改用更稳定的属性定位商品容器
    boxes = soup.find_all("div", {"data-component-type": "s-search-result"})
    name_list = []
    price_list = []
    number_of_reviews_list = []
    images_list = []

    for box in boxes:
        # 处理可能缺失的字段,避免报错
        name_elem = box.find("span", class_="a-size-medium a-color-base a-text-normal")
        price_elem = box.find("span", class_="a-offscreen")
        review_elem = box.find("span", class_="a-size-base s-underline-text")
        image_elem = box.find("img", class_="s-image")

        if name_elem:
            name_list.append(name_elem.getText())
        if price_elem:
            price_list.append(price_elem.getText())
        if review_elem:
            number_of_reviews_list.append(review_elem.getText())
        if image_elem:
            images_list.append(image_elem.get('src'))

    # 等待第二页按钮并点击
    second_page = WebDriverWait(driver, 10).until(
        EC.element_to_be_clickable((By.XPATH, '//a[@class="s-pagination-item s-pagination-button" and text()="2"]'))
    )
    second_page.click()
    time.sleep(random.uniform(3, 5))

    # 这里可以继续处理第二页的数据
    second_page_source = driver.page_source
    # ... 第二页数据解析逻辑

finally:
    driver.quit()

内容的提问来源于stack exchange,提问作者SolidOpt

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 05:18:12