Scrapy爬虫导出CSV数据丢失问题求助
电商网站爬虫CSV数据写入缺失问题排查
我用Scrapy开发了一个电商网站爬虫,指定了5个商品页面URL,但首次运行时总有部分数据没写入product.csv(仅得到4条);再次运行时,缺失的条目还会变化。代码如下,求排查错误原因:
import scrapy import csv import os import requests class ProductSpider(scrapy.Spider): name = "product_spider" start_urls = [ 'https://softwarekaufen24.de/microsoft-office-2021-home-and-student/', 'https://softwarekaufen24.de/windows-11-pro/', 'https://softwarekaufen24.de/windows-11-home/', 'https://softwarekaufen24.de/windows-10-professional/', 'https://softwarekaufen24.de/windows-10-home/' ] def __init__(self, *args, **kwargs): super(ProductSpider, self).__init__(*args, **kwargs) self.filename = 'product.csv' self.logo_dir = 'logos' if not os.path.exists(self.logo_dir): os.makedirs(self.logo_dir) with open(self.filename, mode='w', encoding='utf-8-sig', newline='') as file: writer = csv.writer(file, delimiter=';') writer.writerow( ['Product Title', 'Product Price', 'Old Product Price', 'Product Description', 'Breadcrumb', 'Product Link', 'Logo Path']) def parse(self, response): product_title = response.css('h1[class="product--title"]::text').get() if product_title: product_title = product_title.strip() product_price = response.css( 'meta[itemprop="price"]::attr(content)').get() # Old Product Price auswählen old_product_price = response.css( 'span.price--line-through::text').get() # Beschreibung auswählen product_description = response.css( 'div.product--description:nth-child(2)').extract_first() if product_description: # Unerwünschten Text entfernen product_description = product_description.replace( "SOFTWARE KAUFEN LEICHT GEMACHT!", "").strip() product_description = product_description.replace( "\n", " ").replace("\r", "") # Breadcrumb auswählen breadcrumb_items = response.css( 'span.breadcrumb--title[itemprop="name"]') breadcrumb = '' for item in breadcrumb_items: breadcrumb += item.css('::text').get() + ' > ' breadcrumb = breadcrumb[:-3] # Produkt-Link auswählen product_link = response.url # Logo herunterladen logo_path = '' supplier_div = response.css('div[class="product--supplier"]') if supplier_div: img_url = supplier_div.css('img::attr(src)').get() img_alt = supplier_div.css('img::attr(alt)').get() if img_url and img_alt: img_filename = img_alt.replace( ' ', '_') + os.path.splitext(img_url)[1] logo_path = os.path.join(self.logo_dir, img_filename) if not os.path.exists(logo_path): self.log(f'Downloading logo: {img_url}') img_data = requests.get(img_url).content with open(logo_path, 'wb') as img_file: img_file.write(img_data) # Check if the downloaded image is SVG format if os.path.splitext(img_filename)[1].lower() == '.svg': # Convert SVG to PNG using the inkscape command-line tool subprocess.run( ['inkscape', '--export-type=png', logo_path], check=True) with open(self.filename, mode='a', encoding='utf-8-sig', newline='') as file: writer = csv.writer(file, delimiter=';') writer.writerow( [product_title, product_price, old_product_price, product_description, breadcrumb, product_link, logo_path]) self.log( f'Produktname "{product_title}", Preis "{product_price}", Alter Preis "{old_product_price}", Beschreibung "{product_description}", Breadcrumb "{breadcrumb}", Link "{product_link}", Firmenname "{company_name}" und Firmenlogo "{company_logo}" wurden erfolgreich in {self.filename} gespeichert.')
错误原因分析
多线程文件写入竞争:Scrapy是异步多线程框架,多个
parse方法会同时执行并打开同一个CSV文件追加写入。文件的追加操作不是原子性的,多个线程的写入操作会互相干扰,导致部分数据被覆盖或丢失,这就是为什么每次缺失条目随机的原因。缺失依赖模块:代码中使用了
subprocess.run()但未导入subprocess模块,当处理SVG图片时会抛出NameError,导致该条数据的处理中断,无法写入CSV。日志使用未定义变量:最后一行日志中的
company_name和company_logo变量从未定义,执行到这里会抛出NameError,直接终止当前请求的处理流程,导致对应商品数据无法写入。
解决方案
1. 使用Scrapy官方Feed Exports(推荐)
Scrapy自带的Feed Exports功能会自动处理多线程下的文件写入,完全避免竞争问题,同时简化代码:
import scrapy import os import requests import subprocess # 补上缺失的模块 class ProductSpider(scrapy.Spider): name = "product_spider" start_urls = [ 'https://softwarekaufen24.de/microsoft-office-2021-home-and-student/', 'https://softwarekaufen24.de/windows-11-pro/', 'https://softwarekaufen24.de/windows-11-home/', 'https://softwarekaufen24.de/windows-10-professional/', 'https://softwarekaufen24.de/windows-10-home/' ] # 配置Feed Exports,Scrapy自动处理CSV写入 custom_settings = { 'FEEDS': { 'product.csv': { 'format': 'csv', 'fields': ['Product Title', 'Product Price', 'Old Product Price', 'Product Description', 'Breadcrumb', 'Product Link', 'Logo Path'], 'delimiter': ';', 'encoding': 'utf-8-sig', 'overwrite': True, } } } def __init__(self, *args, **kwargs): super(ProductSpider, self).__init__(*args, **kwargs) self.logo_dir = 'logos' if not os.path.exists(self.logo_dir): os.makedirs(self.logo_dir) def parse(self, response): product_title = response.css('h1[class="product--title"]::text').get() if product_title: product_title = product_title.strip() product_price = response.css('meta[itemprop="price"]::attr(content)').get() old_product_price = response.css('span.price--line-through::text').get() product_description = response.css('div.product--description:nth-child(2)').extract_first() if product_description: product_description = product_description.replace("SOFTWARE KAUFEN LEICHT GEMACHT!", "").strip() product_description = product_description.replace("\n", " ").replace("\r", "") breadcrumb_items = response.css('span.breadcrumb--title[itemprop="name"]') breadcrumb = '' for item in breadcrumb_items: breadcrumb += item.css('::text').get() + ' > ' breadcrumb = breadcrumb[:-3] product_link = response.url logo_path = '' supplier_div = response.css('div[class="product--supplier"]') if supplier_div: img_url = supplier_div.css('img::attr(src)').get() img_alt = supplier_div.css('img::attr(alt)').get() if img_url and img_alt: img_filename = img_alt.replace(' ', '_') + os.path.splitext(img_url)[1] logo_path = os.path.join(self.logo_dir, img_filename) if not os.path.exists(logo_path): self.log(f'Downloading logo: {img_url}') img_data = requests.get(img_url).content with open(logo_path, 'wb') as img_file: img_file.write(img_data) if os.path.splitext(img_filename)[1].lower() == '.svg': subprocess.run(['inkscape', '--export-type=png', logo_path], check=True) # 返回Item,Scrapy自动写入CSV yield { 'Product Title': product_title, 'Product Price': product_price, 'Old Product Price': old_product_price, 'Product Description': product_description, 'Breadcrumb': breadcrumb, 'Product Link': product_link, 'Logo Path': logo_path } # 修复日志变量问题 self.log(f'Produktname "{product_title}", Preis "{product_price}", Alter Preis "{old_product_price}", Beschreibung "{product_description}", Breadcrumb "{breadcrumb}", Link "{product_link}", Firmenlogo "{logo_path}" wurden erfolgreich gespeichert.')
2. 手动处理文件时加锁(不推荐)
如果坚持自己写文件,需要用线程锁保证同一时间只有一个线程写入:
在__init__中初始化锁:
import threading self.lock = threading.Lock()
然后写入文件时加锁:
with self.lock: with open(self.filename, mode='a', encoding='utf-8-sig', newline='') as file: writer = csv.writer(file, delimiter=';') writer.writerow([product_title, product_price, old_product_price, product_description, breadcrumb, product_link, logo_path])
同时必须补上subprocess导入,修复日志中的未定义变量。
内容的提问来源于stack exchange,提问作者angelableckwenn
相关产品推荐
相关产品推荐

