如何在Scrapy中将ItemLoader输出存储为两组独立列表
解决方案
你之前的逻辑问题出在两个地方:
- Item字段用了
TakeFirst输出处理器,该处理器会丢弃列表内除第一个元素外的所有内容,自然无法得到完整列表 - 你为每个属性单独定义了字段,没有直接用目标的两个列表字段存储内容
修改后的完整代码
import scrapy from scrapy.item import Field from itemloaders.processors import TakeFirst, Identity from scrapy.crawler import CrawlerProcess from scrapy.loader import ItemLoader class EbayItem(scrapy.Item): category = Field(output_processor=TakeFirst()) name = Field(output_processor=TakeFirst()) price = Field(output_processor=TakeFirst()) product_url = Field(output_processor=TakeFirst()) # 新增两个列表字段,用Identity处理器保留所有添加的值为列表 numeral = Field(output_processor=Identity()) values = Field(output_processor=Identity()) class EbaySpider(scrapy.Spider): name = 'ebay' start_urls = { 'english': 'https://www.ebay.com/sch/i.html?_from=R40&_nkw=pokemon+cards&_sacat=2536&LH_TitleDesc=0&_sop=16&LH_All=1&rt=nc&Language=English&_dcat=183454', 'japanese':'https://www.ebay.com/sch/i.html?_from=R40&_nkw=pokemon+cards&_sacat=2536&LH_TitleDesc=0&_sop=16&LH_All=1&_oaa=1&rt=nc&Language=Japanese&_dcat=183454' } def start_requests(self): for category, url in self.start_urls.items(): yield scrapy.Request( url=url, callback=self.parse, cb_kwargs={ 'category': category } ) def parse(self, response, category): all_cards = response.xpath('//div[@class="s-item__wrapper clearfix"]') for card in all_cards: loader = ItemLoader(EbayItem(), selector=card) loader.add_value('category', category) loader.add_xpath('name', './/h3/text()') loader.add_xpath('price', './/span[@class="s-item__price"]//text()') loader.add_xpath('product_url', './/a[@class="s-item__link"]//@href') yield scrapy.Request( card.xpath('.//a[@class="s-item__link"]//@href').get(), callback=self.parse_product_details, cb_kwargs={'loader': loader} ) def parse_product_details(self, response, loader): # 属性名存入numeral列表 data11 = response.xpath("//div[@class='ux-layout-section__item ux-layout-section__item--table-view']/div[@class='ux-layout-section__row'][5]/div[@class='ux-labels-values__labels'][2]/div[@class='ux-labels-values__labels-content']/div/span//text()").get() if data11: loader.add_value('numeral', data11) data12 = response.xpath("//div[@class='ux-layout-section__item ux-layout-section__item--table-view']/div[@class='ux-layout-section__row'][6]/div[@class='ux-labels-values__labels'][2]/div[@class='ux-labels-values__labels-content']/div/span//text()").get() if data12: loader.add_value('numeral', data12) # 属性值存入values列表 val11 = response.xpath("//div[@class='ux-layout-section__item ux-layout-section__item--table-view']/div[@class='ux-layout-section__row'][5]/div[@class='ux-labels-values__values'][2]/div[@class='ux-labels-values__values-content']/div/span//text()").get() if val11: loader.add_value('values', val11) val12 = response.xpath("//div[@class='ux-layout-section__item ux-layout-section__item--table-view']/div[@class='ux-layout-section__row'][6]/div[@class='ux-labels-values__values'][2]/div[@class='ux-labels-values__values-content']/div/span//text()").get() if val12: loader.add_value('values', val12) yield loader.load_item() process = CrawlerProcess( settings={ 'FEED_URI': 'test.jl', 'FEED_FORMAT': 'jsonlines' } ) process.crawl(EbaySpider) process.start()
补充说明
如果后续需要提取更多属性对(比如nine、ten对应的内容),只需要在详情页解析逻辑中,依次把属性名添加到numeral字段、属性值添加到values字段即可,无需新增Item字段。同时代码中加了非空判断,避免字段为空时往列表里存入None值。
内容的提问来源于stack exchange,提问作者Stackbeans
相关产品推荐
相关产品推荐

