亚马逊评论爬取受限:如何突破10页限制获取更多评论
亚马逊土耳其站评论爬取限制解决方法
问题背景
运行以下Python代码爬取HyperX Cloud III耳机的客户评论时,仅能获取前10页内容,第10页后“下一页”按钮被禁用,亚马逊提示需筛选评论才能继续查看:
import requests from bs4 import BeautifulSoup import pandas as pd import time import random # HTTP headers headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/83.0.4103.116 Safari/537.36 Edg/83.0.478.50', 'Accept-Language': 'tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7', 'Accept-Encoding': 'gzip, deflate, br', 'Connection': 'keep-alive', 'DNT': '1' } # LUA script lua_script = """ function main(splash, args) splash:set_user_agent(args.user_agent) splash:go(args.url) splash:wait(args.wait) return splash:html() end """ def get_soup(url): try: response = requests.post('http://localhost:8050/execute', json={ 'lua_source': lua_script, 'url': url, 'user_agent': headers['User-Agent'], 'wait': random.uniform(3.0, 7.0) # Randomize waiting time }) response.raise_for_status() soup = BeautifulSoup(response.text, 'html.parser') return soup except requests.exceptions.RequestException as e: print(f"Error fetching page: {e}") return None # List for reviews reviewlist = [] # Function to extract reviews def get_reviews(soup): if soup is None: print("Soup object is empty, skipping this page.") return reviews = soup.find_all('div', {'data-hook': 'review'}) if not reviews: print("No reviews found on this page.") return for item in reviews: try: # Convert rating string to float correctly rating_str = item.find('i', {'data-hook': 'review-star-rating'}).text.replace('5 yıldız üzerinden', '').replace(',', '.').strip() rating = float(rating_str) review = { 'product': soup.title.text.replace('Amazon.com.tr: Customer Reviews: ', '').strip(), 'date': item.find('span', {'data-hook': 'review-date'}).text.strip(), 'title': item.find('a', {'data-hook': 'review-title'}).text.strip(), 'rating': rating, 'body': item.find('span', {'data-hook': 'review-body'}).text.strip(), } reviewlist.append(review) except Exception as e: print(f"Error parsing review: {e}") # Page loop page_number = 1 while True: url = f'https://www.amazon.com.tr/HyperX-Cloud-III-Kulakl%C4%B1%C4%9F%C4%B1-USB/product-reviews/B0C3BSZ56D/ref=cm_cr_getr_d_paging_btm_prev_1?ie=UTF8&reviewerType=all_reviews&pageNumber={page_number}' soup = get_soup(url) print(f'Getting page: {page_number}') get_reviews(soup) print(len(reviewlist)) # Check if there is a next page button next_page = soup.find('li', {'class': 'a-last'}) if next_page and 'a-disabled' in next_page.get('class', []): break page_number += 1 wait_time = random.uniform(5.0, 20.0) # Randomize wait time print(f"Waiting {wait_time:.2f} seconds before fetching next page...") time.sleep(wait_time) # Save results to a dataframe, then export as CSV df = pd.DataFrame(reviewlist) df.to_csv('hyperx-headset-reviews.csv', index=False, sep=';') print('Successful.')
亚马逊土耳其站界面提示截图:
解决方法
1. 利用筛选条件拆分爬取
亚马逊对全量评论仅开放前10页,但通过不同筛选维度(评分、排序方式等)可以解锁更多分页内容,每个筛选条件下都能获取10页评论,最后合并去重即可覆盖大部分评论。
修改后的代码示例:
import requests from bs4 import BeautifulSoup import pandas as pd import time import random # HTTP headers headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/83.0.4103.116 Safari/537.36 Edg/83.0.478.50', 'Accept-Language': 'tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7', 'Accept-Encoding': 'gzip, deflate, br', 'Connection': 'keep-alive', 'DNT': '1' } # LUA script lua_script = """ function main(splash, args) splash:set_user_agent(args.user_agent) splash:go(args.url) splash:wait(args.wait) return splash:html() end """ def get_soup(url): try: response = requests.post('http://localhost:8050/execute', json={ 'lua_source': lua_script, 'url': url, 'user_agent': headers['User-Agent'], 'wait': random.uniform(3.0, 7.0) # Randomize waiting time }) response.raise_for_status() soup = BeautifulSoup(response.text, 'html.parser') return soup except requests.exceptions.RequestException as e: print(f"Error fetching page: {e}") return None # List for reviews reviewlist = [] # Function to extract reviews def get_reviews(soup): if soup is None: print("Soup object is empty, skipping this page.") return reviews = soup.find_all('div', {'data-hook': 'review'}) if not reviews: print("No reviews found on this page.") return for item in reviews: try: # Convert rating string to float correctly rating_str = item.find('i', {'data-hook': 'review-star-rating'}).text.replace('5 yıldız üzerinden', '').replace(',', '.').strip() rating = float(rating_str) review = { 'product': soup.title.text.replace('Amazon.com.tr: Customer Reviews: ', '').strip(), 'date': item.find('span', {'data-hook': 'review-date'}).text.strip(), 'title': item.find('a', {'data-hook': 'review-title'}).text.strip(), 'rating': rating, 'body': item.find('span', {'data-hook': 'review-body'}).text.strip(), } reviewlist.append(review) except Exception as e: print(f"Error parsing review: {e}") # 定义筛选和排序条件组合 filter_options = [ "all_reviews", "one_star", "two_star", "three_star", "four_star", "five_star" ] sort_options = ["recent", "helpful"] # 遍历所有条件爬取 for sort_by in sort_options: for filter_by in filter_options: page_number = 1 while True: # 构造带筛选参数的请求URL url = f'https://www.amazon.com.tr/HyperX-Cloud-III-Kulakl%C4%B1%C4%9F%C4%B1-USB/product-reviews/B0C3BSZ56D/ref=cm_cr_getr_d_paging_btm_prev_1?ie=UTF8&reviewerType=all_reviews&filterByStar={filter_by}&sortBy={sort_by}&pageNumber={page_number}' soup = get_soup(url) print(f'爬取页面: {page_number} | 筛选条件: {filter_by} | 排序方式: {sort_by}') get_reviews(soup) print(f'累计评论数: {len(reviewlist)}') # 检查下一页是否可用 next_page = soup.find('li', {'class': 'a-last'}) if next_page and 'a-disabled' in next_page.get('class', []): break page_number += 1 wait_time = random.uniform(10.0, 30.0) # 延长等待时间,降低反爬检测风险 print(f"等待 {wait_time:.2f} 秒后爬取下一页...") time.sleep(wait_time) # 去重处理,避免不同筛选条件下的重复评论 df = pd.DataFrame(reviewlist).drop_duplicates(subset=['date', 'title', 'body']) df.to_csv('hyperx-headset-reviews.csv', index=False, sep=';') print('爬取完成,已导出CSV文件。')
2. 优化反爬规避策略
- 轮换User-Agent:维护一个包含多款主流浏览器UA的列表,每次请求随机选择一个,避免固定UA被识别。
- 延长随机等待时间:将等待时间调整至10-30秒区间,更贴近真人浏览间隔。
- 使用代理IP:若爬取量较大,通过代理IP池轮换请求IP,避免单一IP被亚马逊限制访问。
3. 使用官方API(合法稳定)
亚马逊提供Product Advertising API,可合法获取商品评论数据,需提前申请开发者账号并遵守API调用频率限制。这种方式不会触发反爬机制,是长期爬取的最优方案。
内容的提问来源于stack exchange,提问作者Latte cofie
相关产品推荐
相关产品推荐

