You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Scrapy爬虫导出CSV数据丢失问题求助

电商网站爬虫CSV数据写入缺失问题排查

我用Scrapy开发了一个电商网站爬虫,指定了5个商品页面URL,但首次运行时总有部分数据没写入product.csv(仅得到4条);再次运行时,缺失的条目还会变化。代码如下,求排查错误原因:

import scrapy
import csv
import os
import requests


class ProductSpider(scrapy.Spider):
    name = "product_spider"

    start_urls = [
        'https://softwarekaufen24.de/microsoft-office-2021-home-and-student/',
        'https://softwarekaufen24.de/windows-11-pro/',
        'https://softwarekaufen24.de/windows-11-home/',
        'https://softwarekaufen24.de/windows-10-professional/',
        'https://softwarekaufen24.de/windows-10-home/'
    ]

    def __init__(self, *args, **kwargs):
        super(ProductSpider, self).__init__(*args, **kwargs)
        self.filename = 'product.csv'
        self.logo_dir = 'logos'
        if not os.path.exists(self.logo_dir):
            os.makedirs(self.logo_dir)
        with open(self.filename, mode='w', encoding='utf-8-sig', newline='') as file:
            writer = csv.writer(file, delimiter=';')
            writer.writerow(
                ['Product Title', 'Product Price', 'Old Product Price', 'Product Description', 'Breadcrumb', 'Product Link', 'Logo Path'])

    def parse(self, response):
        product_title = response.css('h1[class="product--title"]::text').get()
        if product_title:
            product_title = product_title.strip()

        product_price = response.css(
            'meta[itemprop="price"]::attr(content)').get()

        # Old Product Price auswählen
        old_product_price = response.css(
            'span.price--line-through::text').get()

        # Beschreibung auswählen
        product_description = response.css(
            'div.product--description:nth-child(2)').extract_first()
        if product_description:
            # Unerwünschten Text entfernen
            product_description = product_description.replace(
                "SOFTWARE KAUFEN LEICHT GEMACHT!", "").strip()
            product_description = product_description.replace(
                "\n", " ").replace("\r", "")

        # Breadcrumb auswählen
        breadcrumb_items = response.css(
            'span.breadcrumb--title[itemprop="name"]')
        breadcrumb = ''
        for item in breadcrumb_items:
            breadcrumb += item.css('::text').get() + ' > '
        breadcrumb = breadcrumb[:-3]

        # Produkt-Link auswählen
        product_link = response.url

# Logo herunterladen
        logo_path = ''
        supplier_div = response.css('div[class="product--supplier"]')
        if supplier_div:
            img_url = supplier_div.css('img::attr(src)').get()
            img_alt = supplier_div.css('img::attr(alt)').get()
            if img_url and img_alt:
                img_filename = img_alt.replace(
                    ' ', '_') + os.path.splitext(img_url)[1]
                logo_path = os.path.join(self.logo_dir, img_filename)
                if not os.path.exists(logo_path):
                    self.log(f'Downloading logo: {img_url}')
                    img_data = requests.get(img_url).content
                    with open(logo_path, 'wb') as img_file:
                        img_file.write(img_data)
                    # Check if the downloaded image is SVG format
                    if os.path.splitext(img_filename)[1].lower() == '.svg':
                        # Convert SVG to PNG using the inkscape command-line tool
                        subprocess.run(
                            ['inkscape', '--export-type=png', logo_path], check=True)

        with open(self.filename, mode='a', encoding='utf-8-sig', newline='') as file:
            writer = csv.writer(file, delimiter=';')
            writer.writerow(
                [product_title, product_price, old_product_price, product_description, breadcrumb, product_link, logo_path])

        self.log(
            f'Produktname "{product_title}", Preis "{product_price}", Alter Preis "{old_product_price}", Beschreibung "{product_description}", Breadcrumb "{breadcrumb}", Link "{product_link}", Firmenname "{company_name}" und Firmenlogo "{company_logo}" wurden erfolgreich in {self.filename} gespeichert.')

错误原因分析

  • 多线程文件写入竞争:Scrapy是异步多线程框架,多个parse方法会同时执行并打开同一个CSV文件追加写入。文件的追加操作不是原子性的,多个线程的写入操作会互相干扰,导致部分数据被覆盖或丢失,这就是为什么每次缺失条目随机的原因。

  • 缺失依赖模块:代码中使用了subprocess.run()但未导入subprocess模块,当处理SVG图片时会抛出NameError,导致该条数据的处理中断,无法写入CSV。

  • 日志使用未定义变量:最后一行日志中的company_name和company_logo变量从未定义,执行到这里会抛出NameError,直接终止当前请求的处理流程,导致对应商品数据无法写入。

解决方案

1. 使用Scrapy官方Feed Exports(推荐)

Scrapy自带的Feed Exports功能会自动处理多线程下的文件写入,完全避免竞争问题,同时简化代码:

import scrapy
import os
import requests
import subprocess  # 补上缺失的模块

class ProductSpider(scrapy.Spider):
    name = "product_spider"
    start_urls = [
        'https://softwarekaufen24.de/microsoft-office-2021-home-and-student/',
        'https://softwarekaufen24.de/windows-11-pro/',
        'https://softwarekaufen24.de/windows-11-home/',
        'https://softwarekaufen24.de/windows-10-professional/',
        'https://softwarekaufen24.de/windows-10-home/'
    ]

    # 配置Feed Exports,Scrapy自动处理CSV写入
    custom_settings = {
        'FEEDS': {
            'product.csv': {
                'format': 'csv',
                'fields': ['Product Title', 'Product Price', 'Old Product Price', 'Product Description', 'Breadcrumb', 'Product Link', 'Logo Path'],
                'delimiter': ';',
                'encoding': 'utf-8-sig',
                'overwrite': True,
            }
        }
    }

    def __init__(self, *args, **kwargs):
        super(ProductSpider, self).__init__(*args, **kwargs)
        self.logo_dir = 'logos'
        if not os.path.exists(self.logo_dir):
            os.makedirs(self.logo_dir)

    def parse(self, response):
        product_title = response.css('h1[class="product--title"]::text').get()
        if product_title:
            product_title = product_title.strip()

        product_price = response.css('meta[itemprop="price"]::attr(content)').get()

        old_product_price = response.css('span.price--line-through::text').get()

        product_description = response.css('div.product--description:nth-child(2)').extract_first()
        if product_description:
            product_description = product_description.replace("SOFTWARE KAUFEN LEICHT GEMACHT!", "").strip()
            product_description = product_description.replace("\n", " ").replace("\r", "")

        breadcrumb_items = response.css('span.breadcrumb--title[itemprop="name"]')
        breadcrumb = ''
        for item in breadcrumb_items:
            breadcrumb += item.css('::text').get() + ' > '
        breadcrumb = breadcrumb[:-3]

        product_link = response.url

        logo_path = ''
        supplier_div = response.css('div[class="product--supplier"]')
        if supplier_div:
            img_url = supplier_div.css('img::attr(src)').get()
            img_alt = supplier_div.css('img::attr(alt)').get()
            if img_url and img_alt:
                img_filename = img_alt.replace(' ', '_') + os.path.splitext(img_url)[1]
                logo_path = os.path.join(self.logo_dir, img_filename)
                if not os.path.exists(logo_path):
                    self.log(f'Downloading logo: {img_url}')
                    img_data = requests.get(img_url).content
                    with open(logo_path, 'wb') as img_file:
                        img_file.write(img_data)
                    if os.path.splitext(img_filename)[1].lower() == '.svg':
                        subprocess.run(['inkscape', '--export-type=png', logo_path], check=True)

        # 返回Item,Scrapy自动写入CSV
        yield {
            'Product Title': product_title,
            'Product Price': product_price,
            'Old Product Price': old_product_price,
            'Product Description': product_description,
            'Breadcrumb': breadcrumb,
            'Product Link': product_link,
            'Logo Path': logo_path
        }

        # 修复日志变量问题
        self.log(f'Produktname "{product_title}", Preis "{product_price}", Alter Preis "{old_product_price}", Beschreibung "{product_description}", Breadcrumb "{breadcrumb}", Link "{product_link}", Firmenlogo "{logo_path}" wurden erfolgreich gespeichert.')

2. 手动处理文件时加锁(不推荐)

如果坚持自己写文件,需要用线程锁保证同一时间只有一个线程写入:

在__init__中初始化锁:

import threading
self.lock = threading.Lock()

然后写入文件时加锁:

with self.lock:
    with open(self.filename, mode='a', encoding='utf-8-sig', newline='') as file:
        writer = csv.writer(file, delimiter=';')
        writer.writerow([product_title, product_price, old_product_price, product_description, breadcrumb, product_link, logo_path])

同时必须补上subprocess导入,修复日志中的未定义变量。

内容的提问来源于stack exchange,提问作者angelableckwenn

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.23 21:47:56