You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Scrapy Link Extractor仅抓取部分房源,请求排查设置问题

问题描述

尝试使用Scrapy 2.6.2抓取https://www.funda.nl/koop/amsterdam/verkocht/sorteer-afmelddatum-af页面下1-617页的所有房源链接,但仅获取到前100页内容,例如第209页的“Eef Kamerbeekstraat 504 + PP”房源被遗漏。该爬虫在其他用户环境下可正常运行,推测是自身Scrapy设置导致问题,现提供爬虫脚本及配置文件请求排查。

原爬虫脚本

import scrapy
from scrapy.linkextractors import LinkExtractor
from scrapy.spiders import CrawlSpider, Rule


class FundaSpider(CrawlSpider):
    name = 'funda_verkocht'
    allowed_domains = ['funda.nl']
    start_urls = []
    user_agent = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36'

    def start_requests(self):
        yield scrapy.Request(url='https://www.funda.nl/koop/amsterdam/verkocht/sorteer-afmelddatum-af/', headers={
            'User-Agent': self.user_agent
        })


    rules = (
        Rule(LinkExtractor(restrict_xpaths="//a[@data-object-url-tracking='resultlist']"), callback='parse_item', follow=True),
        Rule(LinkExtractor(restrict_xpaths="//a[@rel='next']"), follow = True)
    )

    def set_user_agent(self, request):
        request.headers['User-Agent'] = self.user_agent
        return request

    def parse_item(self, response):
        yield{
            'address': response.xpath("normalize-space(//span[@class='object-header__title']/text())").get(),
            'postal_code': response.xpath("normalize-space(//span[@class='object-header__subtitle fd-color-dark-3']/text())").get(),
            'offered_since': response.xpath("normalize-space(//dt[.='Aangeboden sinds']/following-sibling::dd[1]/span[1]/text())").get(),
            'asking-price': response.xpath("normalize-space(//div/strong[@class='object-header__price--historic']/text())").get(),
            'surface': response.xpath("normalize-space(//dt[.='Wonen']/following-sibling::dd[1]/span/text())").get(),
            'energy_label': response.xpath("normalize-space(//dt[.='Energielabel']/following-sibling::dd[1]/span[1]/text())").get(),
            'housing_type': response.xpath("normalize-space(//dt[.='Soort appartement']/following-sibling::dd[1]/span[1]/text())").get(),
            'build_year': response.xpath("normalize-space(//dt[.='Bouwjaar']/following-sibling::dd[1]/span[1]/text())").get(),
            'number_rooms': response.xpath("normalize-space(//dt[.='Aantal kamers']/following-sibling::dd[1]/span[1]/text())").get(),
            'number_bathrooms': response.xpath("normalize-space(//dt[.='Aantal badkamers']/following-sibling::dd[1]/span[1]/text())").get(),
            'bathroom_facilities': response.xpath("normalize-space(//dt[.='Badkamervoorzieningen']/following-sibling::dd[1]/span[1]/text())").get(),
            'total_floors': response.xpath("normalize-space(//dt[.='Aantal woonlagen']/following-sibling::dd[1]/span[1]/text())").get(),
            'isolation': response.xpath("normalize-space(//dt[.='Isolatie']/following-sibling::dd[1]/span[1]/text())").get(),
            'heating': response.xpath("normalize-space(//dt[.='Verwarming']/following-sibling::dd[1]/span[1]/text())").get(),
            'warm_water': response.xpath("normalize-space(//dt[.='Warm water']/following-sibling::dd[1]/span[1]/text())").get(),
            'cv_ketel': response.xpath("normalize-space(//dt[.='Cv-ketel']/following-sibling::dd[1]/span[1]/text())").get(),
            'land_ownership': response.xpath("normalize-space(//dt[.='Eigendomssituatie']/following-sibling::dd[1]/span[1]/text())").get(),
            'erfpacht': response.xpath("normalize-space(//dt[.='Lasten']/following-sibling::dd[1]/span[1]/text())").get(),
            'location': response.xpath("normalize-space(//dt[.='Ligging']/following-sibling::dd[1]/span[1]/text())").get(),
            'balcony_terrace': response.xpath("normalize-space(//dt[.='Balkon/dakterras']/following-sibling::dd[1]/span[1]/text())").get(),
            'parking': response.xpath("normalize-space(//dt[.='Soort parkeergelegenheid']/following-sibling::dd[1]/span[1]/text())").get(),
            'garden': response.xpath("normalize-space(//dt[.='Tuin']/following-sibling::dd[1]/span[1]/text())").get(),
            'inhabitants_neighborhood': response.xpath("normalize-space(//div[.='Inwoners']/following-sibling::div[1]//text())").get(),
            'families_with_kids_perc': response.xpath("normalize-space(//div[.='Gezin met kinderen']/following-sibling::div[1]//text())").get(),
            'neighborhood_price_sqm': response.xpath("normalize-space(//div[.='Gem. vraagprijs / m²']/following-sibling::div[1]//text())").get()
        }

原配置文件

# Scrapy settings for funda project
#
# For simplicity, this file contains only settings considered important or
# commonly used. You can find more settings consulting the documentation:
#
#     https://docs.scrapy.org/en/latest/topics/settings.html
#     https://docs.scrapy.org/en/latest/topics/downloader-middleware.html
#     https://docs.scrapy.org/en/latest/topics/spider-middleware.html

BOT_NAME = 'funda'

SPIDER_MODULES = ['funda.spiders']
NEWSPIDER_MODULE = 'funda.spiders'
#USER_AGENT = 'Mozilla/5.0 (X11; Linux x86_64; rv:7.0.1) Gecko/20100101 Firefox/7.7'
USER_AGENT = 'Mozilla/5.0 (Windows NT 6.3; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/59.0.3071.115 Safari/537.36'


# Crawl responsibly by identifying yourself (and your website) on the user-agent
#USER_AGENT = 'funda (+http://www.yourdomain.com)'

# Obey robots.txt rules
ROBOTSTXT_OBEY = False

排查与修复方案

1. 统一User-Agent,确保所有请求应用自定义UA

原脚本中仅start_requests生成的请求设置了自定义UA,但rules中提取的房源链接和下一页请求未应用该UA,导致请求UA不一致,被网站反爬机制识别并限制页数。

修改方案:在Rule中指定process_request参数,让所有规则生成的请求都调用set_user_agent方法:

rules = (
    Rule(LinkExtractor(restrict_xpaths="//a[@data-object-url-tracking='resultlist']"), 
         callback='parse_item', 
         follow=True,
         process_request='set_user_agent'),
    Rule(LinkExtractor(restrict_xpaths="//a[@rel='next']"), 
         follow=True,
         process_request='set_user_agent')
)

同时将配置文件中的USER_AGENT值改为与脚本中一致,避免冲突:

USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36'

2. 添加反爬规避设置

网站可能对高频请求的IP进行页数限制,在配置文件中添加以下参数:

# 增加下载延迟,降低请求频率
DOWNLOAD_DELAY = 3
# 限制并发请求数
CONCURRENT_REQUESTS = 2
# 启用Cookie跟踪,维持会话一致性
COOKIES_ENABLED = True

3. 确认无深度限制

Scrapy默认DEPTH_LIMIT为0(无限制),若之前修改过该值,需确保配置文件中设置:

DEPTH_LIMIT = 0

修改后的完整脚本

import scrapy
from scrapy.linkextractors import LinkExtractor
from scrapy.spiders import CrawlSpider, Rule


class FundaSpider(CrawlSpider):
    name = 'funda_verkocht'
    allowed_domains = ['funda.nl']
    start_urls = []
    user_agent = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36'

    def start_requests(self):
        yield scrapy.Request(url='https://www.funda.nl/koop/amsterdam/verkocht/sorteer-afmelddatum-af/', headers={
            'User-Agent': self.user_agent
        })

    rules = (
        Rule(LinkExtractor(restrict_xpaths="//a[@data-object-url-tracking='resultlist']"), 
             callback='parse_item', 
             follow=True,
             process_request='set_user_agent'),
        Rule(LinkExtractor(restrict_xpaths="//a[@rel='next']"), 
             follow=True,
             process_request='set_user_agent')
    )

    def set_user_agent(self, request):
        request.headers['User-Agent'] = self.user_agent
        return request

    def parse_item(self, response):
        yield{
            'address': response.xpath("normalize-space(//span[@class='object-header__title']/text())").get(),
            'postal_code': response.xpath("normalize-space(//span[@class='object-header__subtitle fd-color-dark-3']/text())").get(),
            'offered_since': response.xpath("normalize-space(//dt[.='Aangeboden sinds']/following-sibling::dd[1]/span[1]/text())").get(),
            'asking-price': response.xpath("normalize-space(//div/strong[@class='object-header__price--historic']/text())").get(),
            'surface': response.xpath("normalize-space(//dt[.='Wonen']/following-sibling::dd[1]/span/text())").get(),
            'energy_label': response.xpath("normalize-space(//dt[.='Energielabel']/following-sibling::dd[1]/span[1]/text())").get(),
            'housing_type': response.xpath("normalize-space(//dt[.='Soort appartement']/following-sibling::dd[1]/span[1]/text())").get(),
            'build_year': response.xpath("normalize-space(//dt[.='Bouwjaar']/following-sibling::dd[1]/span[1]/text())").get(),
            'number_rooms': response.xpath("normalize-space(//dt[.='Aantal kamers']/following-sibling::dd[1]/span[1]/text())").get(),
            'number_bathrooms': response.xpath("normalize-space(//dt[.='Aantal badkamers']/following-sibling::dd[1]/span[1]/text())").get(),
            'bathroom_facilities': response.xpath("normalize-space(//dt[.='Badkamervoorzieningen']/following-sibling::dd[1]/span[1]/text())").get(),
            'total_floors': response.xpath("normalize-space(//dt[.='Aantal woonlagen']/following-sibling::dd[1]/span[1]/text())").get(),
            'isolation': response.xpath("normalize-space(//dt[.='Isolatie']/following-sibling::dd[1]/span[1]/text())").get(),
            'heating': response.xpath("normalize-space(//dt[.='Verwarming']/following-sibling::dd[1]/span[1]/text())").get(),
            'warm_water': response.xpath("normalize-space(//dt[.='Warm water']/following-sibling::dd[1]/span[1]/text())").get(),
            'cv_ketel': response.xpath("normalize-space(//dt[.='Cv-ketel']/following-sibling::dd[1]/span[1]/text())").get(),
            'land_ownership': response.xpath("normalize-space(//dt[.='Eigendomssituatie']/following-sibling::dd[1]/span[1]/text())").get(),
            'erfpacht': response.xpath("normalize-space(//dt[.='Lasten']/following-sibling::dd[1]/span[1]/text())").get(),
            'location': response.xpath("normalize-space(//dt[.='Ligging']/following-sibling::dd[1]/span[1]/text())").get(),
            'balcony_terrace': response.xpath("normalize-space(//dt[.='Balkon/dakterras']/following-sibling::dd[1]/span[1]/text())").get(),
            'parking': response.xpath("normalize-space(//dt[.='Soort parkeergelegenheid']/following-sibling::dd[1]/span[1]/text())").get(),
            'garden': response.xpath("normalize-space(//dt[.='Tuin']/following-sibling::dd[1]/span[1]/text())").get(),
            'inhabitants_neighborhood': response.xpath("normalize-space(//div[.='Inwoners']/following-sibling::div[1]//text())").get(),
            'families_with_kids_perc': response.xpath("normalize-space(//div[.='Gezin met kinderen']/following-sibling::div[1]//text())").get(),
            'neighborhood_price_sqm': response.xpath("normalize-space(//div[.='Gem. vraagprijs / m²']/following-sibling::div[1]//text())").get()
        }

修改后的完整配置文件

# Scrapy settings for funda project

BOT_NAME = 'funda'

SPIDER_MODULES = ['funda.spiders']
NEWSPIDER_MODULE = 'funda.spiders'

# 统一使用自定义UA
USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/76.0.3809.100 Safari/537.36'

# Obey robots.txt rules
ROBOTSTXT_OBEY = False

# 反爬规避设置
DOWNLOAD_DELAY = 3
CONCURRENT_REQUESTS = 2
COOKIES_ENABLED = True
DEPTH_LIMIT = 0

内容的提问来源于stack exchange,提问作者Lisa Herzog

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.08 03:10:37