You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Scrapy爬取数据时遭遇TypeError: JSON对象非有效类型的问题排查

解决Scrapy爬虫中的TypeError: JSON对象类型错误问题

问题概述

使用Scrapy进行网页爬取时,出现TypeError: the JSON object must be str, bytes, or bytearray, not NoneType错误。虽然爬虫仍能获取大部分数据,但错误出现次数不稳定(首次运行出现56次,后续运行次数变化),需要明确错误原因、影响及解决方案。错误触发点为parse方法中payload_cursor = json.loads(x)行,其中x通过response.meta.get('dpayload')获取。

爬虫相关代码

def start_requests(self):
    # Generate requests with different payloads for check-in and check-out dates
    for payload in self.checkin_checkout():
        yield scrapy.Request(
            url=self.api_url,
            headers=self.headers,
            body=payload,
            method='POST',
            callback=self.parse,
            dont_filter=True,
            meta={'dpayload': payload}  # Pass the payload as metadata
        )

def checkin_checkout(self):
    # Generate date payloads for check-in and check-out dates
    data = json.loads(self.default_payload)
    current_date = datetime.date.today()

    # Adjust the current_date if it's a Sunday (weekday == 6)
    if current_date.weekday() == 6:
        current_date = current_date + datetime.timedelta(days=1)

    end_date = current_date + datetime.timedelta(days=30)

    checkin_dates = []
    checkout_dates = []

    # Generate check-in and check-out dates for the next 30 days
    while current_date <= end_date:
        if current_date.weekday() == 4:
            checkin_dates.append(current_date)

        if current_date.weekday() == 6:
            checkout_dates.append(current_date)

        current_date += datetime.timedelta(days=1)

    date_payloads = []

    # Generate payloads for each check-in and check-out pair
    for i in range(len(checkout_dates)):
        checkin_date = checkin_dates[i]
        checkout_date = checkout_dates[i]

        # Update the date in the payload data
        data['variables']['staysSearchRequest']['rawParams'][2]['filterValues'] = str(checkin_date)
        data['variables']['staysSearchRequest']['rawParams'][3]['filterValues'] = str(checkout_date)

        date_payload = json.dumps(data)
        date_payloads.append(date_payload)

    return date_payloads

def parse(self, response):
    data = json.loads(response.text)

    info = data['data']['presentation']['explore']['sections']['sectionIndependentData']['staysSearch']['searchResults']

    for i in range(0, 17):
        date = strftime("%d/%m/%Y")
        id_value = info[i]['listing']['id']
        name = info[i]['listing']["name"]
        rating = info[i]['listing']["avgRatingLocalized"]
        PricePerNight = info[i]['pricingQuote']["structuredStayDisplayPrice"]['primaryLine']['accessibilityLabel'].replace("$", "").replace("total", "").strip()[:2]
        totalPrice_inDollars = info[i]['pricingQuote']["structuredStayDisplayPrice"]['secondaryLine']['price'].replace("$", "").replace("total", "").strip()
        
        # Extract check-in and check-out dates from the response data
        checkIn_date = data['data']['presentation']['explore']['sections']['sectionIndependentData']['staysSearch']['filters']['filterState'][4]['value']['dateValue']
        checkOut_date = data['data']['presentation']['explore']['sections']['sectionIndependentData']['staysSearch']['filters']['filterState'][5]['value']['dateValue']
        
        url = urljoin(self.rooms_url, id_value)

        yield {
            'date': date,
            'id': id_value,
            "title": name,
            "PricePerNight": PricePerNight,
            "totalPrice_inDollars": totalPrice_inDollars,
            "checkIn_date": checkIn_date,
            'checkOut_date': checkOut_date,
            "rating": rating,
            'url': url
        }
    #pagination
    url_path = data['data']['presentation']['explore']['sections']['sectionIndependentData']['staysSearch']

    # Retrieve the original payload from metadata
    x = response.meta.get('dpayload')

    for i in range(0, 13):
        payload_cursor = json.loads(x)
        page_cursor = url_path['paginationInfo']['pageCursors'][i]
        payload_cursor['variables']['staysSearchRequest']['cursor'] = page_cursor

        next_page_payload = json.dumps(payload_cursor)

        # Send a request for the next page
        yield scrapy.Request(
            url=self.api_url,
            headers=self.headers,
            body=next_page_payload,
            method='POST',
            callback=self.parse,
            dont_filter=True
        )

错误堆栈信息

Traceback (most recent call last):
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\utils\defer.py", line 277, in iter_errback
    yield next(it)
          ^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\utils\python.py", line 350, in __next__
    return next(self.data)
           ^^^^^^^^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\utils\python.py", line 350, in __next__
    return next(self.data)
           ^^^^^^^^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\core\spidermw.py", line 106, in process_sync
    for r in iterable:
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\spidermiddlewares\offsite.py", line 28, in <genexpr>
    return (r for r in result or () if self._filter(r, spider))
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\core\spidermw.py", line 106, in process_sync
    for r in iterable:
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\spidermiddlewares\referer.py", line 352, in <genexpr>
    return (self._set_referer(r, response) for r in result or ())
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\core\spidermw.py", line 106, in process_sync
    for r in iterable:
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\spidermiddlewares\urllength.py", line 27, in <genexpr>
    return (r for r in result or () if self._filter(r, spider))
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\core\spidermw.py", line 106, in process_sync
    for r in iterable:
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\spidermiddlewares\depth.py", line 31, in <genexpr>
    return (r for r in result or () if self._filter(r, response, spider))
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\Users\Haseeb Tahir\Desktop\workk\myvenv\Lib\site-packages\scrapy\core\spidermw.py", line 106, in process_sync
    for r in iterable:
  File "C:\Users\Haseeb Tahir\Desktop\workk\travelling\travelling\spiders\airbnb2.py", line 125, in parse
    payload_cursor = json.loads(x)
                     ^^^^^^^^^^^^^
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.11_3.11.1520.0_x64__qbz5n2kfra8p0\Lib\json\__init__.py", line 339, in loads
    raise TypeError(f'the JSON object must be str, bytes or bytearray, '
TypeError: the JSON object must be str, bytes or bytearray, not NoneType

错误原因

  1. 元数据丢失:初始请求(start_requests生成)携带了dpayload元数据,但分页请求(parse方法中生成的下一页请求)未传递该元数据。当parse处理分页请求的响应时,response.meta.get('dpayload')返回None,导致json.loads无法解析。
  2. 循环逻辑缺陷:分页循环固定执行13次,若接口返回的pageCursors数量不足13,会触发索引越界,但当前错误的直接原因是x为None。
  3. payload对象复用问题:checkin_checkout方法中,所有日期payload复用同一个data对象,可能导致最终所有payload的日期都是最后一次循环的日期(隐性问题)。

对爬取过程的影响

  1. 数据丢失:分页请求的响应处理时会报错中断,导致这些分页页面的数据无法抓取,仅能获取初始请求的第一页数据。
  2. 爬虫不稳定:错误日志会干扰问题排查,频繁报错可能触发Scrapy的重试机制,浪费服务器资源,甚至被目标网站识别为恶意爬虫。
  3. 数据一致性问题:checkin_checkout方法的payload复用问题会导致所有日期的请求参数错误,爬取的数据日期与预期不符。

解决方案

1. 传递分页请求的元数据

修改parse方法中的分页请求代码,添加meta={'dpayload': x},确保每一页请求都携带dpayload:

#pagination
url_path = data['data']['presentation']['explore']['sections']['sectionIndependentData']['staysSearch']

# Retrieve the original payload from metadata
x = response.meta.get('dpayload')

# 先判断x是否存在,再处理分页
if x is not None and 'pageCursors' in url_path.get('paginationInfo', {}):
    # 根据实际pageCursors长度循环,避免固定13次导致索引越界
    for page_cursor in url_path['paginationInfo']['pageCursors']:
        try:
            payload_cursor = json.loads(x)
        except (TypeError, json.JSONDecodeError) as e:
            self.logger.error(f"解析payload失败: {str(e)}")
            continue
        
        payload_cursor['variables']['staysSearchRequest']['cursor'] = page_cursor
        next_page_payload = json.dumps(payload_cursor)

        # Send a request for the next page
        yield scrapy.Request(
            url=self.api_url,
            headers=self.headers,
            body=next_page_payload,
            method='POST',
            callback=self.parse,
            dont_filter=True,
            meta={'dpayload': x}  # 传递dpayload到下一页请求
        )

2. 修复payload对象复用问题

在checkin_checkout方法的循环中,每次重新加载默认payload,避免修改同一个对象:

# Generate payloads for each check-in and check-out pair
for i in range(len(checkout_dates)):
    # 每次循环重新加载默认payload,避免共享对象导致日期覆盖
    data = json.loads(self.default_payload)
    checkin_date = checkin_dates[i]
    checkout_date = checkout_dates[i]

    # Update the date in the payload data
    data['variables']['staysSearchRequest']['rawParams'][2]['filterValues'] = str(checkin_date)
    data['variables']['staysSearchRequest']['rawParams'][3]['filterValues'] = str(checkout_date)

    date_payload = json.dumps(data)
    date_payloads.append(date_payload)

3. 增加异常处理

在json.loads周围添加异常捕获,避免单个错误中断整个爬虫流程,同时记录错误日志便于排查。

内容的提问来源于stack exchange,提问作者haseeb tahir

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.09 18:15:56