Scrapy自定义JSON管道格式修复:移除末尾对象多余逗号
修复Scrapy自定义JSON Pipeline的尾逗号问题
我用自定义Scrapy Pipeline导出商品数据成JSON,但最后一条数据末尾多了个逗号,导致JSON格式非法。
非法JSON输出(末尾多余逗号):
[{ "product_id": "11980174", "brand_id": 25354, "brand_name": "Gucci", "title": "beige and brown Dionysus GG Supreme mini canvas shoulder bag", "slug": "/shopping/gucci-beige-and-brown-dionysus-gg-supreme-mini-canvas-shoulder-bag-11980174" }, { "product_id": "17070807", "brand_id": 1168391, "brand_name": "Jonathan Adler", "title": "Clear acrylic chess set", "slug": "/shopping/jonathan-adler-clear-acrylic-chess-set-17070807" }, { "product_id": "17022890", "brand_id": 3543122, "brand_name": "Anissa Kermiche", "title": "pink, green and red Mini Jugs Jug earthenware vase set", "slug": "/shopping/anissa-kermiche-pink-green-and-red-mini-jugs-jug-earthenware-vase-set-17022890" },]
期望的合法JSON格式:
[{ "product_id": "11980174", "brand_id": 25354, "brand_name": "Gucci", "title": "beige and brown Dionysus GG Supreme mini canvas shoulder bag", "slug": "/shopping/gucci-beige-and-brown-dionysus-gg-supreme-mini-canvas-shoulder-bag-11980174" }, { "product_id": "17070807", "brand_id": 1168391, "brand_name": "Jonathan Adler", "title": "Clear acrylic chess set", "slug": "/shopping/jonathan-adler-clear-acrylic-chess-set-17070807" }, { "product_id": "17022890", "brand_id": 3543122, "brand_name": "Anissa Kermiche", "title": "pink, green and red Mini Jugs Jug earthenware vase set", "slug": "/shopping/anissa-kermiche-pink-green-and-red-mini-jugs-jug-earthenware-vase-set-17022890" }]
当前使用的自定义Pipeline代码:
from scrapy import signals import boto3 from scrapy.utils.project import get_project_settings import time import json class JsonWriterPipeline(object): def __init__(self): self.spider_time = f'{time.strftime("%Y/%G_%m/%Y.%m.%d/%Y.%m.%d")}' @classmethod def from_crawler(cls, crawler): pipeline = cls() crawler.signals.connect(pipeline.spider_opened, signals.spider_opened) crawler.signals.connect(pipeline.spider_closed, signals.spider_closed) return pipeline def spider_opened(self, spider): self.file = open("%s_items.json" % spider.name, "w") self.file.write("[") def process_item(self, item, spider): line = line = json.dumps(dict(item), indent=4) + ",\n" self.file.write(line) return item def spider_closed(self, spider): self.file.write("]") self.file.close() settings = get_project_settings() my_session = boto3.session.Session() s3 = my_session.resource( "s3", endpoint_url=settings.get("AWS_ENDPOINT_URL"), aws_access_key_id=settings.get("AWS_ACCESS_KEY_ID"), aws_secret_access_key=settings.get("AWS_SECRET_ACCESS_KEY"), ) boto_test_bucket = s3.Bucket(settings.get("AWS_STORAGE_BUCKET_NAME")) boto_test_bucket.upload_file( "%s_items.json" % spider.name, f"brownsfashion-feeds/{spider.name}_{self.spider_time}.json", )
修复方案
方案一:动态控制逗号(低内存占用,边爬边写)
适合爬取数据量较大的场景,通过标记判断是否为第一个item,避免末尾留逗号:
from scrapy import signals import boto3 from scrapy.utils.project import get_project_settings import time import json class JsonWriterPipeline(object): def __init__(self): self.spider_time = f'{time.strftime("%Y/%G_%m/%Y.%m.%d/%Y.%m.%d")}' self.first_item = True # 标记是否是第一个爬取的item @classmethod def from_crawler(cls, crawler): pipeline = cls() crawler.signals.connect(pipeline.spider_opened, signals.spider_opened) crawler.signals.connect(pipeline.spider_closed, signals.spider_closed) return pipeline def spider_opened(self, spider): self.file = open("%s_items.json" % spider.name, "w") self.file.write("[") def process_item(self, item, spider): if self.first_item: # 第一个item直接写入,不加前缀逗号 line = json.dumps(dict(item), indent=4) + "\n" self.first_item = False else: # 后续item先写逗号+换行,再写内容 line = ",\n" + json.dumps(dict(item), indent=4) self.file.write(line) return item def spider_closed(self, spider): self.file.write("]") self.file.close() settings = get_project_settings() my_session = boto3.session.Session() s3 = my_session.resource( "s3", endpoint_url=settings.get("AWS_ENDPOINT_URL"), aws_access_key_id=settings.get("AWS_ACCESS_KEY_ID"), aws_secret_access_key=settings.get("AWS_SECRET_ACCESS_KEY"), ) boto_test_bucket = s3.Bucket(settings.get("AWS_STORAGE_BUCKET_NAME")) boto_test_bucket.upload_file( "%s_items.json" % spider.name, f"brownsfashion-feeds/{spider.name}_{self.spider_time}.json", )
方案二:先收集再序列化(简洁易维护)
适合数据量不大的场景,先把所有item存入列表,最后用json.dump自动生成合法JSON:
from scrapy import signals import boto3 from scrapy.utils.project import get_project_settings import time import json class JsonWriterPipeline(object): def __init__(self): self.spider_time = f'{time.strftime("%Y/%G_%m/%Y.%m.%d/%Y.%m.%d")}' self.items = [] # 存储所有爬取到的item @classmethod def from_crawler(cls, crawler): pipeline = cls() crawler.signals.connect(pipeline.spider_opened, signals.spider_opened) crawler.signals.connect(pipeline.spider_closed, signals.spider_closed) return pipeline def spider_opened(self, spider): # 无需提前写入左括号,最后统一处理 pass def process_item(self, item, spider): self.items.append(dict(item)) # 将item转为字典存入列表 return item def spider_closed(self, spider): # 一次性序列化整个列表,自动处理逗号格式 with open("%s_items.json" % spider.name, "w") as f: json.dump(self.items, f, indent=4) settings = get_project_settings() my_session = boto3.session.Session() s3 = my_session.resource( "s3", endpoint_url=settings.get("AWS_ENDPOINT_URL"), aws_access_key_id=settings.get("AWS_ACCESS_KEY_ID"), aws_secret_access_key=settings.get("AWS_SECRET_ACCESS_KEY"), ) boto_test_bucket = s3.Bucket(settings.get("AWS_STORAGE_BUCKET_NAME")) boto_test_bucket.upload_file( "%s_items.json" % spider.name, f"brownsfashion-feeds/{spider.name}_{self.spider_time}.json", )
内容的提问来源于stack exchange,提问作者X-something
相关产品推荐
相关产品推荐

