Scrapy Link Extractor仅抓取部分房源,请求排查设置问题
问题描述
尝试使用Scrapy 2.6.2抓取https://www.funda.nl/koop/amsterdam/verkocht/sorteer-afmelddatum-af页面下1-617页的所有房源链接,但仅获取到前100页内容,例如第209页的“Eef Kamerbeekstraat 504 + PP”房源被遗漏。该爬虫在其他用户环境下可正常运行,推测是自身Scrapy设置导致问题,现提供爬虫脚本及配置文件请求排查。
原爬虫脚本
import scrapy from scrapy.linkextractors import LinkExtractor from scrapy.spiders import CrawlSpider, Rule class FundaSpider(CrawlSpider): name = 'funda_verkocht' allowed_domains = ['funda.nl'] start_urls = [] user_agent = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36' def start_requests(self): yield scrapy.Request(url='https://www.funda.nl/koop/amsterdam/verkocht/sorteer-afmelddatum-af/', headers={ 'User-Agent': self.user_agent }) rules = ( Rule(LinkExtractor(restrict_xpaths="//a[@data-object-url-tracking='resultlist']"), callback='parse_item', follow=True), Rule(LinkExtractor(restrict_xpaths="//a[@rel='next']"), follow = True) ) def set_user_agent(self, request): request.headers['User-Agent'] = self.user_agent return request def parse_item(self, response): yield{ 'address': response.xpath("normalize-space(//span[@class='object-header__title']/text())").get(), 'postal_code': response.xpath("normalize-space(//span[@class='object-header__subtitle fd-color-dark-3']/text())").get(), 'offered_since': response.xpath("normalize-space(//dt[.='Aangeboden sinds']/following-sibling::dd[1]/span[1]/text())").get(), 'asking-price': response.xpath("normalize-space(//div/strong[@class='object-header__price--historic']/text())").get(), 'surface': response.xpath("normalize-space(//dt[.='Wonen']/following-sibling::dd[1]/span/text())").get(), 'energy_label': response.xpath("normalize-space(//dt[.='Energielabel']/following-sibling::dd[1]/span[1]/text())").get(), 'housing_type': response.xpath("normalize-space(//dt[.='Soort appartement']/following-sibling::dd[1]/span[1]/text())").get(), 'build_year': response.xpath("normalize-space(//dt[.='Bouwjaar']/following-sibling::dd[1]/span[1]/text())").get(), 'number_rooms': response.xpath("normalize-space(//dt[.='Aantal kamers']/following-sibling::dd[1]/span[1]/text())").get(), 'number_bathrooms': response.xpath("normalize-space(//dt[.='Aantal badkamers']/following-sibling::dd[1]/span[1]/text())").get(), 'bathroom_facilities': response.xpath("normalize-space(//dt[.='Badkamervoorzieningen']/following-sibling::dd[1]/span[1]/text())").get(), 'total_floors': response.xpath("normalize-space(//dt[.='Aantal woonlagen']/following-sibling::dd[1]/span[1]/text())").get(), 'isolation': response.xpath("normalize-space(//dt[.='Isolatie']/following-sibling::dd[1]/span[1]/text())").get(), 'heating': response.xpath("normalize-space(//dt[.='Verwarming']/following-sibling::dd[1]/span[1]/text())").get(), 'warm_water': response.xpath("normalize-space(//dt[.='Warm water']/following-sibling::dd[1]/span[1]/text())").get(), 'cv_ketel': response.xpath("normalize-space(//dt[.='Cv-ketel']/following-sibling::dd[1]/span[1]/text())").get(), 'land_ownership': response.xpath("normalize-space(//dt[.='Eigendomssituatie']/following-sibling::dd[1]/span[1]/text())").get(), 'erfpacht': response.xpath("normalize-space(//dt[.='Lasten']/following-sibling::dd[1]/span[1]/text())").get(), 'location': response.xpath("normalize-space(//dt[.='Ligging']/following-sibling::dd[1]/span[1]/text())").get(), 'balcony_terrace': response.xpath("normalize-space(//dt[.='Balkon/dakterras']/following-sibling::dd[1]/span[1]/text())").get(), 'parking': response.xpath("normalize-space(//dt[.='Soort parkeergelegenheid']/following-sibling::dd[1]/span[1]/text())").get(), 'garden': response.xpath("normalize-space(//dt[.='Tuin']/following-sibling::dd[1]/span[1]/text())").get(), 'inhabitants_neighborhood': response.xpath("normalize-space(//div[.='Inwoners']/following-sibling::div[1]//text())").get(), 'families_with_kids_perc': response.xpath("normalize-space(//div[.='Gezin met kinderen']/following-sibling::div[1]//text())").get(), 'neighborhood_price_sqm': response.xpath("normalize-space(//div[.='Gem. vraagprijs / m²']/following-sibling::div[1]//text())").get() }
原配置文件
# Scrapy settings for funda project # # For simplicity, this file contains only settings considered important or # commonly used. You can find more settings consulting the documentation: # # https://docs.scrapy.org/en/latest/topics/settings.html # https://docs.scrapy.org/en/latest/topics/downloader-middleware.html # https://docs.scrapy.org/en/latest/topics/spider-middleware.html BOT_NAME = 'funda' SPIDER_MODULES = ['funda.spiders'] NEWSPIDER_MODULE = 'funda.spiders' #USER_AGENT = 'Mozilla/5.0 (X11; Linux x86_64; rv:7.0.1) Gecko/20100101 Firefox/7.7' USER_AGENT = 'Mozilla/5.0 (Windows NT 6.3; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/59.0.3071.115 Safari/537.36' # Crawl responsibly by identifying yourself (and your website) on the user-agent #USER_AGENT = 'funda (+http://www.yourdomain.com)' # Obey robots.txt rules ROBOTSTXT_OBEY = False
排查与修复方案
1. 统一User-Agent,确保所有请求应用自定义UA
原脚本中仅start_requests生成的请求设置了自定义UA,但rules中提取的房源链接和下一页请求未应用该UA,导致请求UA不一致,被网站反爬机制识别并限制页数。
修改方案:在Rule中指定process_request参数,让所有规则生成的请求都调用set_user_agent方法:
rules = ( Rule(LinkExtractor(restrict_xpaths="//a[@data-object-url-tracking='resultlist']"), callback='parse_item', follow=True, process_request='set_user_agent'), Rule(LinkExtractor(restrict_xpaths="//a[@rel='next']"), follow=True, process_request='set_user_agent') )
同时将配置文件中的USER_AGENT值改为与脚本中一致,避免冲突:
USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36'
2. 添加反爬规避设置
网站可能对高频请求的IP进行页数限制,在配置文件中添加以下参数:
# 增加下载延迟,降低请求频率 DOWNLOAD_DELAY = 3 # 限制并发请求数 CONCURRENT_REQUESTS = 2 # 启用Cookie跟踪,维持会话一致性 COOKIES_ENABLED = True
3. 确认无深度限制
Scrapy默认DEPTH_LIMIT为0(无限制),若之前修改过该值,需确保配置文件中设置:
DEPTH_LIMIT = 0
修改后的完整脚本
import scrapy from scrapy.linkextractors import LinkExtractor from scrapy.spiders import CrawlSpider, Rule class FundaSpider(CrawlSpider): name = 'funda_verkocht' allowed_domains = ['funda.nl'] start_urls = [] user_agent = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36' def start_requests(self): yield scrapy.Request(url='https://www.funda.nl/koop/amsterdam/verkocht/sorteer-afmelddatum-af/', headers={ 'User-Agent': self.user_agent }) rules = ( Rule(LinkExtractor(restrict_xpaths="//a[@data-object-url-tracking='resultlist']"), callback='parse_item', follow=True, process_request='set_user_agent'), Rule(LinkExtractor(restrict_xpaths="//a[@rel='next']"), follow=True, process_request='set_user_agent') ) def set_user_agent(self, request): request.headers['User-Agent'] = self.user_agent return request def parse_item(self, response): yield{ 'address': response.xpath("normalize-space(//span[@class='object-header__title']/text())").get(), 'postal_code': response.xpath("normalize-space(//span[@class='object-header__subtitle fd-color-dark-3']/text())").get(), 'offered_since': response.xpath("normalize-space(//dt[.='Aangeboden sinds']/following-sibling::dd[1]/span[1]/text())").get(), 'asking-price': response.xpath("normalize-space(//div/strong[@class='object-header__price--historic']/text())").get(), 'surface': response.xpath("normalize-space(//dt[.='Wonen']/following-sibling::dd[1]/span/text())").get(), 'energy_label': response.xpath("normalize-space(//dt[.='Energielabel']/following-sibling::dd[1]/span[1]/text())").get(), 'housing_type': response.xpath("normalize-space(//dt[.='Soort appartement']/following-sibling::dd[1]/span[1]/text())").get(), 'build_year': response.xpath("normalize-space(//dt[.='Bouwjaar']/following-sibling::dd[1]/span[1]/text())").get(), 'number_rooms': response.xpath("normalize-space(//dt[.='Aantal kamers']/following-sibling::dd[1]/span[1]/text())").get(), 'number_bathrooms': response.xpath("normalize-space(//dt[.='Aantal badkamers']/following-sibling::dd[1]/span[1]/text())").get(), 'bathroom_facilities': response.xpath("normalize-space(//dt[.='Badkamervoorzieningen']/following-sibling::dd[1]/span[1]/text())").get(), 'total_floors': response.xpath("normalize-space(//dt[.='Aantal woonlagen']/following-sibling::dd[1]/span[1]/text())").get(), 'isolation': response.xpath("normalize-space(//dt[.='Isolatie']/following-sibling::dd[1]/span[1]/text())").get(), 'heating': response.xpath("normalize-space(//dt[.='Verwarming']/following-sibling::dd[1]/span[1]/text())").get(), 'warm_water': response.xpath("normalize-space(//dt[.='Warm water']/following-sibling::dd[1]/span[1]/text())").get(), 'cv_ketel': response.xpath("normalize-space(//dt[.='Cv-ketel']/following-sibling::dd[1]/span[1]/text())").get(), 'land_ownership': response.xpath("normalize-space(//dt[.='Eigendomssituatie']/following-sibling::dd[1]/span[1]/text())").get(), 'erfpacht': response.xpath("normalize-space(//dt[.='Lasten']/following-sibling::dd[1]/span[1]/text())").get(), 'location': response.xpath("normalize-space(//dt[.='Ligging']/following-sibling::dd[1]/span[1]/text())").get(), 'balcony_terrace': response.xpath("normalize-space(//dt[.='Balkon/dakterras']/following-sibling::dd[1]/span[1]/text())").get(), 'parking': response.xpath("normalize-space(//dt[.='Soort parkeergelegenheid']/following-sibling::dd[1]/span[1]/text())").get(), 'garden': response.xpath("normalize-space(//dt[.='Tuin']/following-sibling::dd[1]/span[1]/text())").get(), 'inhabitants_neighborhood': response.xpath("normalize-space(//div[.='Inwoners']/following-sibling::div[1]//text())").get(), 'families_with_kids_perc': response.xpath("normalize-space(//div[.='Gezin met kinderen']/following-sibling::div[1]//text())").get(), 'neighborhood_price_sqm': response.xpath("normalize-space(//div[.='Gem. vraagprijs / m²']/following-sibling::div[1]//text())").get() }
修改后的完整配置文件
# Scrapy settings for funda project BOT_NAME = 'funda' SPIDER_MODULES = ['funda.spiders'] NEWSPIDER_MODULE = 'funda.spiders' # 统一使用自定义UA USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36' # Obey robots.txt rules ROBOTSTXT_OBEY = False # 反爬规避设置 DOWNLOAD_DELAY = 3 CONCURRENT_REQUESTS = 2 COOKIES_ENABLED = True DEPTH_LIMIT = 0
内容的提问来源于stack exchange,提问作者Lisa Herzog
相关产品推荐
相关产品推荐

