You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何爬取动态加载页面全部商品?亚马逊爬虫仅获30个商品问题

问题:亚马逊意大利站汽车品类热销页爬虫仅抓取30个商品,如何获取全部50个?

尝试爬取亚马逊意大利站汽车品类热销页(https://www.amazon.it/gp/bestsellers/automotive/ref=zg_bs_nav_automotive_0)的商品数据,该页面为动态加载,已编写滚动页面代码,但每页应有的50个商品仅能抓取到30个。使用的代码如下:

from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
import time

def start_selenium():
    chromium_service = Service()  
    chromium_options = webdriver.ChromeOptions()
    chromium_options.add_argument("--headless")
    chromium_driver = webdriver.Chrome(service=chromium_service, options=chromium_options)
    return chromium_driver

def scroll_down(driver):
    last_height = driver.execute_script("return document.body.scrollHeight")
    while True:
        driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
        time.sleep(5)
        new_height = driver.execute_script("return document.body.scrollHeight")
        if new_height == last_height:
            break
        last_height = new_height

def get_all_deals_ids():
    deals_page = "https://www.amazon.it/gp/bestsellers/automotive/ref=zg_bs_nav_automotive_0"
    selenium_driver = start_selenium()

    try:
        selenium_driver.get(deals_page)
        deals_urls = []

        scroll_down(selenium_driver)

        deals_urls += [e.get_attribute("href") for e in selenium_driver.find_elements(By.CSS_SELECTOR, "a[class*='a-link-normal']")]

        product_ids = [extract_product_id(url) for url in deals_urls if extract_product_id(url) is not None and extract_product_id(url) != '']

        selenium_driver.quit()
        return [*set(product_ids)]

    except Exception as e:
        print(e)
        selenium_driver.quit()
        return [] 

解决方案及修改后的代码

from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
import time

def start_selenium():
    chromium_service = Service()  
    chromium_options = webdriver.ChromeOptions()
    chromium_options.add_argument("--headless")
    chromium_options.add_argument("--window-size=1920,1080")  # 设置无头模式窗口尺寸
    chromium_driver = webdriver.Chrome(service=chromium_service, options=chromium_options)
    return chromium_driver

def scroll_down(driver):
    # 分多次逐步滚动,避免一次性滚到底导致加载不完整
    for _ in range(6):
        driver.execute_script("window.scrollBy(0, window.innerHeight);")
        time.sleep(3)
    # 最后滚到底部确认
    driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
    time.sleep(4)
    # 检查是否存在"加载更多"按钮(部分热销页会有)
    try:
        load_more_btn = driver.find_element(By.CSS_SELECTOR, ".zg-load-more")
        if load_more_btn.is_displayed():
            driver.execute_script("arguments[0].click();", load_more_btn)
            time.sleep(5)
    except:
        pass

def get_all_deals_ids():
    deals_page = "https://www.amazon.it/gp/bestsellers/automotive/ref=zg_bs_nav_automotive_0"
    selenium_driver = start_selenium()

    try:
        selenium_driver.get(deals_page)
        deals_urls = []

        scroll_down(selenium_driver)

        # 使用精准选择器,仅定位商品卡片内的链接
        product_cards = selenium_driver.find_elements(By.CSS_SELECTOR, ".zg-item-immersion")
        for card in product_cards:
            link = card.find_element(By.CSS_SELECTOR, "a.a-link-normal")
            deals_urls.append(link.get_attribute("href"))

        product_ids = [extract_product_id(url) for url in deals_urls if extract_product_id(url)]

        selenium_driver.quit()
        return list(set(product_ids))

    except Exception as e:
        print(e)
        selenium_driver.quit()
        return [] 

# 补充商品ID提取逻辑(若未实现)
def extract_product_id(url):
    if "/dp/" in url:
        return url.split("/dp/")[1].split("/")[0]
    elif "/gp/product/" in url:
        return url.split("/gp/product/")[1].split("/")[0]
    return None

关键修改说明

  • 设置无头窗口尺寸:无头Chrome默认窗口过小,部分商品元素因未进入视窗不会加载,添加--window-size=1920,1080确保页面完整渲染。
  • 优化滚动逻辑:将一次性滚动改为分多次逐步滚动,每次滚动后等待加载,避免页面渲染滞后;同时检查并点击可能存在的"加载更多"按钮,确保所有商品加载完成。
  • 精准元素选择:原选择器会匹配页面中所有含a-link-normal的链接(如导航、广告),改为先定位商品卡片.zg-item-immersion,再从中提取链接,避免无关链接干扰。
  • 完善ID提取逻辑:补充亚马逊商品URL的两种常见格式解析,确保能正确提取商品ID。

内容的提问来源于stack exchange,提问作者user18786433

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.24 20:21:11