使用Scrapy爬取亚马逊评论时翻页CSS选择器失效问题求助
亚马逊评论爬虫无法获取全部页面问题解决
问题描述
通过ASIN编号下载亚马逊商品评论时,仅能获取第一页数据,无法爬取全部页面,核心原因是“下一页”对应的CSS选择器无法正常工作。
问题代码
import scrapy from urllib.parse import urljoin class AmazonReviewsSpider(scrapy.Spider): name = "amazon_reviews" custom_settings = { 'FEEDS': { 'data/%(name)s_%(time)s.csv': { 'format': 'csv',}} } def start_requests(self): asin_list = ['B08GKK7NMH'] for asin in asin_list: amazon_reviews_url = f'https://www.amazon.com/product-reviews/{asin}/' yield scrapy.Request(url=amazon_reviews_url, callback=self.parse_reviews, meta={'asin': asin, 'retry_count': 0}) def parse_reviews(self, response): asin = response.meta['asin'] retry_count = response.meta['retry_count'] next_page_relative_url = response.css(".a-pagination .a-last>a::attr(href)::after").get() if next_page_relative_url is not None: retry_count = 0 next_page = urljoin('https://www.amazon.com/', next_page_relative_url) yield scrapy.Request(url=next_page, callback=self.parse_reviews, meta={'asin': asin, 'retry_count': retry_count}) ## 添加重试逻辑应对亚马逊JS渲染的评论页面 elif retry_count < 3: retry_count = retry_count+1 yield scrapy.Request(url=response.url, callback=self.parse_reviews, dont_filter=True, meta={'asin': asin, 'retry_count': retry_count}) ## 解析商品评论 review_elements = response.css("#cm_cr-review_list div.review") for review_element in review_elements: yield { "asin": asin, "text": "".join(review_element.css("span[data-hook=review-body] ::text").getall()).strip(), "title": review_element.css("*[data-hook=review-title]>span::text").get(), "location_and_date": review_element.css("span[data-hook=review-date] ::text").get(), "verified": bool(review_element.css("span[data-hook=avp-badge] ::text").get()), "rating": review_element.css("[data-hook=review-star-rating] ::text").re(r"(\d+\.\d) out")[0], }
问题分析与修复方案
CSS选择器错误:原代码中
::after是伪元素选择器,用于获取元素的伪内容,但我们需要的是<a>标签的href属性值,应删除::after,正确选择器为.a-pagination .a-last>a::attr(href)。反爬应对:亚马逊会限制频繁请求,建议在配置中添加请求延迟和浏览器UA,模拟正常访问行为。
代码结构修正:原代码中
start_requests和parse_reviews方法未缩进,属于类外函数,会导致爬虫无法正常执行,需要将其缩进为类的成员方法。
修复后的完整代码
import scrapy from urllib.parse import urljoin class AmazonReviewsSpider(scrapy.Spider): name = "amazon_reviews" custom_settings = { 'FEEDS': { 'data/%(name)s_%(time)s.csv': { 'format': 'csv',}}, 'DOWNLOAD_DELAY': 2, # 添加请求延迟 'USER_AGENT': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36' # 模拟浏览器UA } def start_requests(self): asin_list = ['B08GKK7NMH'] for asin in asin_list: amazon_reviews_url = f'https://www.amazon.com/product-reviews/{asin}/' yield scrapy.Request(url=amazon_reviews_url, callback=self.parse_reviews, meta={'asin': asin, 'retry_count': 0}) def parse_reviews(self, response): asin = response.meta['asin'] retry_count = response.meta['retry_count'] # 修复下一页选择器 next_page_relative_url = response.css(".a-pagination .a-last>a::attr(href)").get() if next_page_relative_url is not None: retry_count = 0 next_page = urljoin('https://www.amazon.com/', next_page_relative_url) yield scrapy.Request(url=next_page, callback=self.parse_reviews, meta={'asin': asin, 'retry_count': retry_count}) # 重试逻辑保留,最多重试3次 elif retry_count < 3: retry_count += 1 yield scrapy.Request(url=response.url, callback=self.parse_reviews, dont_filter=True, meta={'asin': asin, 'retry_count': retry_count}) # 解析评论逻辑不变 review_elements = response.css("#cm_cr-review_list div.review") for review_element in review_elements: yield { "asin": asin, "text": "".join(review_element.css("span[data-hook=review-body] ::text").getall()).strip(), "title": review_element.css("*[data-hook=review-title]>span::text").get(), "location_and_date": review_element.css("span[data-hook=review-date] ::text").get(), "verified": bool(review_element.css("span[data-hook=avp-badge] ::text").get()), "rating": review_element.css("[data-hook=review-star-rating] ::text").re(r"(\d+\.\d) out")[0], }
内容的提问来源于stack exchange,提问作者zouarv42
相关产品推荐
相关产品推荐

