You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Scrapy自定义JSON管道格式修复:移除末尾对象多余逗号

修复Scrapy自定义JSON Pipeline的尾逗号问题

我用自定义Scrapy Pipeline导出商品数据成JSON,但最后一条数据末尾多了个逗号,导致JSON格式非法。

非法JSON输出(末尾多余逗号):

[{
    "product_id": "11980174",
    "brand_id": 25354,
    "brand_name": "Gucci",
    "title": "beige and brown Dionysus GG Supreme mini canvas shoulder bag",
    "slug": "/shopping/gucci-beige-and-brown-dionysus-gg-supreme-mini-canvas-shoulder-bag-11980174"
},
{
    "product_id": "17070807",
    "brand_id": 1168391,
    "brand_name": "Jonathan Adler",
    "title": "Clear acrylic chess set",
    "slug": "/shopping/jonathan-adler-clear-acrylic-chess-set-17070807"
},
{
    "product_id": "17022890",
    "brand_id": 3543122,
    "brand_name": "Anissa Kermiche",
    "title": "pink, green and red Mini Jugs Jug earthenware vase set",
    "slug": "/shopping/anissa-kermiche-pink-green-and-red-mini-jugs-jug-earthenware-vase-set-17022890"
},]

期望的合法JSON格式:

[{
    "product_id": "11980174",
    "brand_id": 25354,
    "brand_name": "Gucci",
    "title": "beige and brown Dionysus GG Supreme mini canvas shoulder bag",
    "slug": "/shopping/gucci-beige-and-brown-dionysus-gg-supreme-mini-canvas-shoulder-bag-11980174"
},
{
    "product_id": "17070807",
    "brand_id": 1168391,
    "brand_name": "Jonathan Adler",
    "title": "Clear acrylic chess set",
    "slug": "/shopping/jonathan-adler-clear-acrylic-chess-set-17070807"
},
{
    "product_id": "17022890",
    "brand_id": 3543122,
    "brand_name": "Anissa Kermiche",
    "title": "pink, green and red Mini Jugs Jug earthenware vase set",
    "slug": "/shopping/anissa-kermiche-pink-green-and-red-mini-jugs-jug-earthenware-vase-set-17022890"
}]

当前使用的自定义Pipeline代码:

from scrapy import signals
import boto3
from scrapy.utils.project import get_project_settings
import time
import json


class JsonWriterPipeline(object):
    def __init__(self):
        self.spider_time = f'{time.strftime("%Y/%G_%m/%Y.%m.%d/%Y.%m.%d")}'

    @classmethod
    def from_crawler(cls, crawler):
        pipeline = cls()
        crawler.signals.connect(pipeline.spider_opened, signals.spider_opened)
        crawler.signals.connect(pipeline.spider_closed, signals.spider_closed)
        return pipeline

    def spider_opened(self, spider):
        self.file = open("%s_items.json" % spider.name, "w")
        self.file.write("[")

    def process_item(self, item, spider):
        line = line = json.dumps(dict(item), indent=4) + ",\n"
        self.file.write(line)
        return item

    def spider_closed(self, spider):
        self.file.write("]")
        self.file.close()
        settings = get_project_settings()
        my_session = boto3.session.Session()
        s3 = my_session.resource(
            "s3",
            endpoint_url=settings.get("AWS_ENDPOINT_URL"),
            aws_access_key_id=settings.get("AWS_ACCESS_KEY_ID"),
            aws_secret_access_key=settings.get("AWS_SECRET_ACCESS_KEY"),
        )
        boto_test_bucket = s3.Bucket(settings.get("AWS_STORAGE_BUCKET_NAME"))
        boto_test_bucket.upload_file(
            "%s_items.json" % spider.name,
            f"brownsfashion-feeds/{spider.name}_{self.spider_time}.json",
        )

修复方案

方案一:动态控制逗号(低内存占用,边爬边写)

适合爬取数据量较大的场景,通过标记判断是否为第一个item,避免末尾留逗号:

from scrapy import signals
import boto3
from scrapy.utils.project import get_project_settings
import time
import json


class JsonWriterPipeline(object):
    def __init__(self):
        self.spider_time = f'{time.strftime("%Y/%G_%m/%Y.%m.%d/%Y.%m.%d")}'
        self.first_item = True  # 标记是否是第一个爬取的item

    @classmethod
    def from_crawler(cls, crawler):
        pipeline = cls()
        crawler.signals.connect(pipeline.spider_opened, signals.spider_opened)
        crawler.signals.connect(pipeline.spider_closed, signals.spider_closed)
        return pipeline

    def spider_opened(self, spider):
        self.file = open("%s_items.json" % spider.name, "w")
        self.file.write("[")

    def process_item(self, item, spider):
        if self.first_item:
            # 第一个item直接写入,不加前缀逗号
            line = json.dumps(dict(item), indent=4) + "\n"
            self.first_item = False
        else:
            # 后续item先写逗号+换行,再写内容
            line = ",\n" + json.dumps(dict(item), indent=4)
        self.file.write(line)
        return item

    def spider_closed(self, spider):
        self.file.write("]")
        self.file.close()
        settings = get_project_settings()
        my_session = boto3.session.Session()
        s3 = my_session.resource(
            "s3",
            endpoint_url=settings.get("AWS_ENDPOINT_URL"),
            aws_access_key_id=settings.get("AWS_ACCESS_KEY_ID"),
            aws_secret_access_key=settings.get("AWS_SECRET_ACCESS_KEY"),
        )
        boto_test_bucket = s3.Bucket(settings.get("AWS_STORAGE_BUCKET_NAME"))
        boto_test_bucket.upload_file(
            "%s_items.json" % spider.name,
            f"brownsfashion-feeds/{spider.name}_{self.spider_time}.json",
        )

方案二:先收集再序列化(简洁易维护)

适合数据量不大的场景,先把所有item存入列表,最后用json.dump自动生成合法JSON:

from scrapy import signals
import boto3
from scrapy.utils.project import get_project_settings
import time
import json


class JsonWriterPipeline(object):
    def __init__(self):
        self.spider_time = f'{time.strftime("%Y/%G_%m/%Y.%m.%d/%Y.%m.%d")}'
        self.items = []  # 存储所有爬取到的item

    @classmethod
    def from_crawler(cls, crawler):
        pipeline = cls()
        crawler.signals.connect(pipeline.spider_opened, signals.spider_opened)
        crawler.signals.connect(pipeline.spider_closed, signals.spider_closed)
        return pipeline

    def spider_opened(self, spider):
        # 无需提前写入左括号,最后统一处理
        pass

    def process_item(self, item, spider):
        self.items.append(dict(item))  # 将item转为字典存入列表
        return item

    def spider_closed(self, spider):
        # 一次性序列化整个列表,自动处理逗号格式
        with open("%s_items.json" % spider.name, "w") as f:
            json.dump(self.items, f, indent=4)
        
        settings = get_project_settings()
        my_session = boto3.session.Session()
        s3 = my_session.resource(
            "s3",
            endpoint_url=settings.get("AWS_ENDPOINT_URL"),
            aws_access_key_id=settings.get("AWS_ACCESS_KEY_ID"),
            aws_secret_access_key=settings.get("AWS_SECRET_ACCESS_KEY"),
        )
        boto_test_bucket = s3.Bucket(settings.get("AWS_STORAGE_BUCKET_NAME"))
        boto_test_bucket.upload_file(
            "%s_items.json" % spider.name,
            f"brownsfashion-feeds/{spider.name}_{self.spider_time}.json",
        )

内容的提问来源于stack exchange,提问作者X-something

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.13 14:55:14