Scrapy爬虫优化:实现房源列表图片一次性批量下载
问题描述
我使用Scrapy开发了一个房产网站爬虫,用于抓取房源的文本信息(如房间数、面积等)和图片,并设置了1.5秒的全局下载延迟以避免被网站封禁。目前文本信息可以一次性抓取完成,切换页面后也能正常重复抓取对应内容,但图片却是逐个下载,且每张图片之间都间隔1.5秒。我希望调整代码实现单房源的所有图片一次性批量下载,尝试过多种方法及ChatGPT协助均未达到预期效果,以下是我的代码配置:
Spider代码
from ast import Try import scrapy from datetime import date from scrapy.loader import ItemLoader from bookscraper.items import MenzillerItem from bookscraper.itemloaders import MenzilProductLoader class BinaspiderSpider(scrapy.Spider): name = "binaspider" allowed_domains = ["website is removed deliberately"] start_urls = ["website is removed deliberately"] def parse(self, response): for href in response.css("a.item_link::attr(href)"): ev_url = href.get() yield response.follow(ev_url, callback=self.parse_listing) for a in response.css('span.next a'): yield response.follow(a, callback=self.parse) def parse_listing(self, response): # Create an item loader instance loader = ItemLoader(item=MenzillerItem(), selector=response) # Scraping text information loader.add_css('elan_adi', 'h1.product-name::text') loader.add_css('elan_nomresi', 'div.product-actions__id::text') # Extract "Yeniləndi: Dünən 12:54" loader.add_css('elan_tarixi', 'div.product-statistics__i:nth-child(1) span.product-statistics__i-text::text') # Extract "Baxışların sayı: 193" loader.add_css('baxis_sayi', 'div.product-statistics__i:nth-child(2) span.product-statistics__i-text::text') # Processing to get the full "Dünən 12:54" part from the first value elan_tarixi_list = loader.get_output_value('elan_tarixi') elan_tarixi = elan_tarixi_list[0].split(":", 1)[1].strip() if elan_tarixi_list else "" loader.replace_value('elan_tarixi', elan_tarixi) loader.add_css('unvan', 'div.product-map__left__address::text') price_parts = response.css("div.product-price__i span::text").getall() full_price = " ".join(part.strip() for part in price_parts) loader.add_value('qiymet', full_price) kateqoriya_label=response.xpath("//div[@class='product-properties__i']/label/text()")[0].get() kateqoriya_span=response.css('div.product-properties__i span::text')[0].get() kateqoriya = f"{kateqoriya_label}:{kateqoriya_span}" loader.add_value('kateqoriya', kateqoriya) baxis_value = response.css('span.product-statistics__i-text::text')[1].get()[16:] loader.add_value('baxis_sayi', baxis_value) try: name1_label = response.xpath("//div[@class='product-properties__i']/label/text()")[1].get() name1_span = response.css('div.product-properties__i span::text')[1].get().strip() name1_value = f"{name1_label}:{name1_span}" except: name1_value = '' loader.add_value('name1', name1_value) try: name2_label=response.xpath("//div[@class='product-properties__i']/label/text()")[2].get().strip() name2_span=response.css('div.product-properties__i span::text')[2].get().strip() name2_value = f"{name2_label}:{name2_span}" except: name2_value = '' loader.add_value('name2', name2_value) try: name3_label = response.xpath("//div[@class='product-properties__i']/label/text()")[3].get() name3_span = response.css('div.product-properties__i span::text')[3].get().strip() name3_value = f"{name3_label}:{name3_span}" except: name3_value = '' loader.add_value('name3', name3_value) try: name4_label = response.xpath("//div[@class='product-properties__i']/label/text()")[4].get() name4_span = response.css('div.product-properties__i span::text')[4].get().strip() name4_value = f"{name4_label}:{name4_span}" except: name4_value = '' loader.add_value('name4', name4_value) try: name5_label = response.xpath("//div[@class='product-properties__i']/label/text()")[5].get() name5_span = response.css('div.product-properties__i span::text')[5].get().strip() name5_value = f"{name5_label}:{name5_span}" except: name5_value = '' loader.add_value('name5', name5_value) try: name6_label = response.xpath("//div[@class='product-properties__i']/label/text()")[6].get() name6_span = response.css('div.product-properties__i span::text')[6].get().strip() name6_value = f"{name6_label}:{name6_span}" except: name6_value = '' loader.add_value('name6', name6_value) try: name7_label = response.xpath("//div[@class='product-properties__i']/label/text()")[7].get() name7_span = response.css('div.product-properties__i span::text')[7].get().strip() name7_value = f"{name7_label}:{name7_span}" except: name7_value = '' loader.add_value('name7', name7_value) try: name8_label = response.xpath("//div[@class='product-properties__i']/label/text()")[8].get() name8_span = response.css('div.product-properties__i span::text')[8].get().strip() name8_value = f"{name8_label}:{name8_span}" except: name8_value = '' loader.add_value('name8', name8_value) try: owner_span = response.css('div.product-owner__info-region::text').get().strip() owner_value = f"{owner_span}" except: owner_value = '' loader.add_value('owner', owner_value) try: description = response.css('div.product-description-container div.product-description__content p::text').getall() description_value = " ".join(part.strip() for part in description) except: description_value = '' loader.add_value('description', description_value) try: etraf_erazi_1_value=response.xpath("//li[@class='product-extras__i']/a/text()")[0].get() except: etraf_erazi_1_value = '' loader.add_value('etraf_erazi_1', etraf_erazi_1_value) try: etraf_erazi_2_value=response.xpath("//li[@class='product-extras__i']/a/text()")[1].get() except: etraf_erazi_2_value = '' loader.add_value('etraf_erazi_2', etraf_erazi_2_value) try: etraf_erazi_3_value=response.xpath("//li[@class='product-extras__i']/a/text()")[2].get() except: etraf_erazi_3_value = '' loader.add_value('etraf_erazi_3', etraf_erazi_3_value) try: latitude_value= response.xpath("//div[@id='item_map']/@data-lat").get() except: latitude_value = '' loader.add_value('latitude', latitude_value) try: longitude_value= response.xpath("//div[@id='item_map']/@data-lng").get() except: longitude_value = '' loader.add_value('longitude', longitude_value) try: webpage_link = response.url except: webpage_link = '' loader.add_value('webpage_link', webpage_link) # Scraping image information image_urls = response.css('img[data-stat="product-gallery-item"]::attr(src)').getall() background_image_urls = response.css('span.product-photos__slider-top-i_background::attr(style)').getall() background_image_urls = [ url.split("url('")[-1].split("')")[0] for url in background_image_urls ] image_urls.extend(background_image_urls) # 优化:直接添加整个列表,而非循环逐个add_value image_names = [f"{response.url.split('/')[-1]}_{i}.jpg" for i, _ in enumerate(image_urls, start=1)] loader.add_value('image_urls', image_urls) loader.add_value('image_names', image_names) today_date = date.today().strftime('%Y-%m-%d') loader.add_value('updated', today_date) yield loader.load_item()
Pipeline代码
# pipelines.py from scrapy.pipelines.images import ImagesPipeline import scrapy class CustomImagePipeline(ImagesPipeline): def file_path(self, request, response=None, info=None): # Extract the custom name from the request meta image_name = request.meta['image_name'] return f'{image_name}' # Save images with custom names def get_media_requests(self, item, info): # Loop through the images and names for img_url, img_name in zip(item.get('image_urls', []), item.get('image_names', [])): # 给图片请求设置0延迟,同时提升优先级让图片先批量下载 yield scrapy.Request( img_url, meta={ 'image_name': img_name, 'download_delay': 0 # 覆盖全局下载延迟 }, priority=10 # 设置更高优先级,确保图片请求优先处理 ) def item_completed(self, results, item, info): # Delete the image_names and image_urls after downloading if 'image_urls' in item: del item['image_urls'] if 'image_names' in item: del item['image_names'] # Return the item after removing image fields return item
Settings配置修改
# Scrapy settings for bookscraper project # # For simplicity, this file contains only settings considered important or # commonly used. You can find more settings by consulting the documentation: # https://docs.scrapy.org/en/latest/topics/settings.html BOT_NAME = "bookscraper" SPIDER_MODULES = ["bookscraper.spiders"] NEWSPIDER_MODULE = "bookscraper.spiders" # Crawl responsibly by identifying yourself (and your website) on the user-agent USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36" # Obey robots.txt rules ROBOTSTXT_OBEY = True # Configure a delay for requests for the same website DOWNLOAD_DELAY = 1.5 # 调整并发数,允许同时下载多个图片 CONCURRENT_REQUESTS = 16 # 全局最大并发请求数,可根据服务器性能调整 CONCURRENT_REQUESTS_PER_DOMAIN = 8 # 每个域名的并发请求数,提升这个值让图片批量下载 # Configure item pipelines ITEM_PIPELINES = { 'bookscraper.pipelines.CustomImagePipeline': 300, # Custom pipeline for images } # Set the directory for images IMAGES_STORE = r'Deliberately removed' # Adjust to your actual path # Export settings to save scraped data in CSV format FEED_FORMAT = "csv" FEED_URI = r'Deliberately removed' # Adjust the path to where you want the CSV to be saved # Set settings for Scrapy pipelines and requests REQUEST_FINGERPRINTER_IMPLEMENTATION = "2.7" TWISTED_REACTOR = "twisted.internet.asyncioreactor.AsyncioSelectorReactor" FEED_EXPORT_ENCODING = "utf-8" # 禁用AutoThrottle(如果之前开启的话),避免干扰自定义延迟设置 AUTOTHROTTLE_ENABLED = False
解决方案说明
- 核心问题:全局
DOWNLOAD_DELAY会作用于所有请求(包括图片),导致图片逐个间隔1.5秒下载。我们需要给图片请求单独设置0延迟,同时提升并发数实现批量下载。 - Spider优化:将图片URL和名称直接以列表形式添加到ItemLoader,避免循环逐个添加的冗余操作。
- Pipeline调整:在生成图片请求时,通过
meta={'download_delay':0}覆盖全局延迟,并设置更高优先级(priority=10)确保图片请求优先处理。 - Settings配置:提升
CONCURRENT_REQUESTS和CONCURRENT_REQUESTS_PER_DOMAIN的值,允许Scrapy同时发起多个图片请求,实现批量下载效果。
内容的提问来源于stack exchange,提问作者Avaz Yusibov
相关产品推荐
相关产品推荐

