使用Scrapy爬取宜家网站无法实现分页跳转的问题求助
问题排查
- 反爬拦截:你配置的User-Agent是早已淘汰的IE7标识,宜家网站的反爬策略会直接拦截这类异常请求,所有分页请求实际返回的都是首页内容,所以你只能拿到24条首页商品数据。
- 分页逻辑错误:下一页请求的构造代码放在了商品遍历的for循环内部,会产生大量重复请求,被Scrapy默认的去重机制拦截,同时类变量count在异步请求场景下计数会出现偏差,硬编码70+条起始URL的冗余逻辑也和手动构造分页的逻辑冲突。
修复后代码
import scrapy import logging from scrapy.crawler import CrawlerProcess from scrapy.exporters import CsvItemExporter class CsvPipeline(object): def __init__(self): self.file = open('ikeaSpiderSofa.tmp', 'wb') self.exporter = CsvItemExporter(self.file, str) self.exporter.start_exporting() def close_spider(self, spider): self.exporter.finish_exporting() self.file.close() def process_item(self, item, spider): self.exporter.export_item(item) return item class ikeaSpider(scrapy.Spider): name = "ikeaSpider" # 仅保留第一个起始页,不需要硬编码全部分页 start_urls = ['https://www.ikea.com/fr/fr/cat/canapes-fu003/'] total_page = 72 custom_settings = { 'LOG_LEVEL': logging.WARNING, 'ITEM_PIPELINES': {'__main__.CsvPipeline': 1}, 'FEED_FORMAT': 'csv', 'FEED_URI': 'ikeaSpiderSofa.csv' } def parse(self, response): # 先提取当前页所有商品 for result in response.css('.range-revamp-product-compact__wrapper-link'): yield scrapy.Request(url=result.xpath('@href').extract_first(), callback=self.parse_detail) # 分页逻辑放在商品遍历外部,每页只触发一次 current_page = response.meta.get('page', 1) next_page = current_page + 1 if next_page <= self.total_page: next_url = f"https://www.ikea.com/fr/fr/cat/canapes-fu003/?page={next_page}" yield scrapy.Request(next_url, self.parse, meta={'page': next_page}) def parse_detail(self, response): label = response.css('.range-revamp-header-section__title--big.notranslate::text').get() price = response.css('.range-revamp-price__integer::text').get() description = response.css('.range-revamp-header-section__description-text::text').get() id_product = response.css('.range-revamp-product-identifier__value::text').get() arbo1 = response.css('#content > div > div.range-revamp-page-container__inner > div > div:nth-child(1) > div > nav > ol > li:nth-child(2) > a > span::text').get() arbo2 = response.css('#content > div > div.range-revamp-page-container__inner > div > div:nth-child(1) > div > nav > ol > li:nth-child(3) > a > span::text').get() arbo3 = response.css('#content > div > div.range-revamp-page-container__inner > div > div:nth-child(1) > div > nav > ol > li:nth-child(4) > a > span::text').get() arbo4 = response.css('#content > div > div.range-revamp-page-container__inner > div > div:nth-child(1) > div > nav > ol > li:nth-child(5) > a > span::text').get() arbo5 = response.css('#content > div > div.range-revamp-page-container__inner > div > div:nth-child(1) > div > nav > ol > li:nth-child(6) > a > span::text').get() producturl = response.selector.xpath('/html/head/meta[11]').get() yield { 'producturl': producturl.strip() if producturl else '', 'label': label.strip() if label else '', 'price': price.strip() if price else '', 'description': description.strip() if description else '', 'id': id_product.strip() if id_product else '', 'arbo1': arbo1.strip() if arbo1 else '', 'arbo2': arbo2.strip() if arbo2 else '', 'arbo3': arbo3.strip() if arbo3 else '', 'arbo4': arbo4.strip() if arbo4 else '', 'arbo5': arbo5.strip() if arbo5 else '' } if __name__ == "__main__": process = CrawlerProcess( # 替换为正常的浏览器User-Agent {'USER_AGENT': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36'} ) process.crawl(ikeaSpider) process.start()
优化说明
- 用meta传递当前页码,避免类变量计数混乱的问题
- 增加了字段判空逻辑,避免部分商品缺少字段时strip()报错
- 移除了冗余的硬编码起始URL,代码更简洁
- 调整了程序入口的写法,避免多进程场景下重复执行爬虫
内容的提问来源于stack exchange,提问作者PattaKempi
相关产品推荐
相关产品推荐

