You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

亚马逊评论爬取受限:如何突破10页限制获取更多评论

亚马逊土耳其站评论爬取限制解决方法

问题背景

运行以下Python代码爬取HyperX Cloud III耳机的客户评论时,仅能获取前10页内容,第10页后“下一页”按钮被禁用,亚马逊提示需筛选评论才能继续查看:

import requests
from bs4 import BeautifulSoup
import pandas as pd
import time
import random

# HTTP headers
headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/83.0.4103.116 Safari/537.36 Edg/83.0.478.50',
    'Accept-Language': 'tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7',
    'Accept-Encoding': 'gzip, deflate, br',
    'Connection': 'keep-alive',
    'DNT': '1'
}

# LUA script
lua_script = """
function main(splash, args)
  splash:set_user_agent(args.user_agent)
  splash:go(args.url)
  splash:wait(args.wait)
  return splash:html()
end
"""

def get_soup(url):
    try:
        response = requests.post('http://localhost:8050/execute', json={
            'lua_source': lua_script,
            'url': url,
            'user_agent': headers['User-Agent'],
            'wait': random.uniform(3.0, 7.0)  # Randomize waiting time
        })
        response.raise_for_status()
        soup = BeautifulSoup(response.text, 'html.parser')
        return soup
    except requests.exceptions.RequestException as e:
        print(f"Error fetching page: {e}")
        return None

# List for reviews
reviewlist = []

# Function to extract reviews
def get_reviews(soup):
    if soup is None:
        print("Soup object is empty, skipping this page.")
        return

    reviews = soup.find_all('div', {'data-hook': 'review'})
    if not reviews:
        print("No reviews found on this page.")
        return

    for item in reviews:
        try:
            # Convert rating string to float correctly
            rating_str = item.find('i', {'data-hook': 'review-star-rating'}).text.replace('5 yıldız üzerinden', '').replace(',', '.').strip()
            rating = float(rating_str)
            
            review = {
                'product': soup.title.text.replace('Amazon.com.tr: Customer Reviews: ', '').strip(),
                'date': item.find('span', {'data-hook': 'review-date'}).text.strip(),
                'title': item.find('a', {'data-hook': 'review-title'}).text.strip(),
                'rating': rating,
                'body': item.find('span', {'data-hook': 'review-body'}).text.strip(),
            }
            reviewlist.append(review)
        except Exception as e:
            print(f"Error parsing review: {e}")

# Page loop
page_number = 1
while True:
    url = f'https://www.amazon.com.tr/HyperX-Cloud-III-Kulakl%C4%B1%C4%9F%C4%B1-USB/product-reviews/B0C3BSZ56D/ref=cm_cr_getr_d_paging_btm_prev_1?ie=UTF8&reviewerType=all_reviews&pageNumber={page_number}'
    soup = get_soup(url)
    print(f'Getting page: {page_number}')
    get_reviews(soup)
    print(len(reviewlist))

    # Check if there is a next page button
    next_page = soup.find('li', {'class': 'a-last'})
    if next_page and 'a-disabled' in next_page.get('class', []):
        break

    page_number += 1
    wait_time = random.uniform(5.0, 20.0)  # Randomize wait time
    print(f"Waiting {wait_time:.2f} seconds before fetching next page...")
    time.sleep(wait_time)

# Save results to a dataframe, then export as CSV
df = pd.DataFrame(reviewlist)
df.to_csv('hyperx-headset-reviews.csv', index=False, sep=';')
print('Successful.')

亚马逊土耳其站界面提示截图:
亚马逊土耳其站界面截图

解决方法

1. 利用筛选条件拆分爬取

亚马逊对全量评论仅开放前10页,但通过不同筛选维度(评分、排序方式等)可以解锁更多分页内容,每个筛选条件下都能获取10页评论,最后合并去重即可覆盖大部分评论。

修改后的代码示例:

import requests
from bs4 import BeautifulSoup
import pandas as pd
import time
import random

# HTTP headers
headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/83.0.4103.116 Safari/537.36 Edg/83.0.478.50',
    'Accept-Language': 'tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7',
    'Accept-Encoding': 'gzip, deflate, br',
    'Connection': 'keep-alive',
    'DNT': '1'
}

# LUA script
lua_script = """
function main(splash, args)
  splash:set_user_agent(args.user_agent)
  splash:go(args.url)
  splash:wait(args.wait)
  return splash:html()
end
"""

def get_soup(url):
    try:
        response = requests.post('http://localhost:8050/execute', json={
            'lua_source': lua_script,
            'url': url,
            'user_agent': headers['User-Agent'],
            'wait': random.uniform(3.0, 7.0)  # Randomize waiting time
        })
        response.raise_for_status()
        soup = BeautifulSoup(response.text, 'html.parser')
        return soup
    except requests.exceptions.RequestException as e:
        print(f"Error fetching page: {e}")
        return None

# List for reviews
reviewlist = []

# Function to extract reviews
def get_reviews(soup):
    if soup is None:
        print("Soup object is empty, skipping this page.")
        return

    reviews = soup.find_all('div', {'data-hook': 'review'})
    if not reviews:
        print("No reviews found on this page.")
        return

    for item in reviews:
        try:
            # Convert rating string to float correctly
            rating_str = item.find('i', {'data-hook': 'review-star-rating'}).text.replace('5 yıldız üzerinden', '').replace(',', '.').strip()
            rating = float(rating_str)
            
            review = {
                'product': soup.title.text.replace('Amazon.com.tr: Customer Reviews: ', '').strip(),
                'date': item.find('span', {'data-hook': 'review-date'}).text.strip(),
                'title': item.find('a', {'data-hook': 'review-title'}).text.strip(),
                'rating': rating,
                'body': item.find('span', {'data-hook': 'review-body'}).text.strip(),
            }
            reviewlist.append(review)
        except Exception as e:
            print(f"Error parsing review: {e}")

# 定义筛选和排序条件组合
filter_options = [
    "all_reviews", 
    "one_star", 
    "two_star", 
    "three_star", 
    "four_star", 
    "five_star"
]
sort_options = ["recent", "helpful"]

# 遍历所有条件爬取
for sort_by in sort_options:
    for filter_by in filter_options:
        page_number = 1
        while True:
            # 构造带筛选参数的请求URL
            url = f'https://www.amazon.com.tr/HyperX-Cloud-III-Kulakl%C4%B1%C4%9F%C4%B1-USB/product-reviews/B0C3BSZ56D/ref=cm_cr_getr_d_paging_btm_prev_1?ie=UTF8&reviewerType=all_reviews&filterByStar={filter_by}&sortBy={sort_by}&pageNumber={page_number}'
            soup = get_soup(url)
            print(f'爬取页面: {page_number} | 筛选条件: {filter_by} | 排序方式: {sort_by}')
            get_reviews(soup)
            print(f'累计评论数: {len(reviewlist)}')

            # 检查下一页是否可用
            next_page = soup.find('li', {'class': 'a-last'})
            if next_page and 'a-disabled' in next_page.get('class', []):
                break

            page_number += 1
            wait_time = random.uniform(10.0, 30.0)  # 延长等待时间,降低反爬检测风险
            print(f"等待 {wait_time:.2f} 秒后爬取下一页...")
            time.sleep(wait_time)

# 去重处理,避免不同筛选条件下的重复评论
df = pd.DataFrame(reviewlist).drop_duplicates(subset=['date', 'title', 'body'])
df.to_csv('hyperx-headset-reviews.csv', index=False, sep=';')
print('爬取完成,已导出CSV文件。')

2. 优化反爬规避策略

  • 轮换User-Agent:维护一个包含多款主流浏览器UA的列表,每次请求随机选择一个,避免固定UA被识别。
  • 延长随机等待时间:将等待时间调整至10-30秒区间,更贴近真人浏览间隔。
  • 使用代理IP:若爬取量较大,通过代理IP池轮换请求IP,避免单一IP被亚马逊限制访问。

3. 使用官方API(合法稳定)

亚马逊提供Product Advertising API,可合法获取商品评论数据,需提前申请开发者账号并遵守API调用频率限制。这种方式不会触发反爬机制,是长期爬取的最优方案。

内容的提问来源于stack exchange,提问作者Latte cofie

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 07:57:04