You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

RBC文章解析器出现JSONDecodeError错误,请求排查原因

RBC文章解析器JSON解码错误排查方案

问题背景

找到一份两年前实现的RBC网站文章解析器代码,运行时触发json.decoder.JSONDecodeError: Expecting value: line 1 column 1 (char 0)错误,无法确定解析逻辑可行性及JSON错误根源,请求排查。

报错详情

\PythonSoftwareFoundation.Python.3.9_qbz5n2kfra8p0\LocalCache\local-packages\Python39\site-packages\requests\models.py", line 900, in json
    return complexjson.loads(self.text, **kwargs)
File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.9_3.9.3568.0_x64__qbz5n2kfra8p0\lib\json\__init__.py", line 346, in loads
    return _default_decoder.decode(s)
File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.9_3.9.3568.0_x64__qbz5n2kfra8p0\lib\json\decoder.py", line 337, in decode
    obj, end = self.raw_decode(s, idx=_w(s, 0).end())
File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.9_3.9.3568.0_x64__qbz5n2kfra8p0\lib\json\decoder.py", line 355, in raw_decode
    raise JSONDecodeError("Expecting value", s, err.value) from None
json.decoder.JSONDecodeError: Expecting value: line 1 column 1 (char 0)

排查步骤

  • 错误本质:该错误表示requests.Response.json()尝试解析的内容不是有效JSON,原因通常是请求返回了非JSON内容(如HTML错误页、空响应、反爬拦截页面)。原代码直接调用r.json()未做任何响应检查和异常处理。
  • 检查响应状态:在调用r.json()前,先打印响应状态码和部分内容,确认返回内容类型:
    r = rq.get(url)
    print(f"状态码: {r.status_code}")
    print(f"响应内容片段: {r.text[:500]}")
    
    大概率是RBC网站的反爬机制拦截了无请求头的请求,返回403状态码及HTML拦截页,或两年前的API端点已失效。
  • 验证API有效性:确认当前RBC搜索API的端点格式是否变更,原https://www.rbc.ru/v10/search/ajax/可能已停用或参数规则调整。
  • 补充请求头:多数网站会拦截缺失User-Agent的请求,需添加模拟浏览器的请求头绕过基础反爬。

修复后的代码

关键修改点:添加请求头、增加响应检查与异常处理、优化日期逻辑

import requests as rq
from bs4 import BeautifulSoup as bs
import pandas as pd
import numpy as np
from datetime import datetime, timedelta
from IPython import display
import time

# 模拟浏览器请求头
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36"
}

class rbc_parser:
    def __init__(self):
        pass
    def _get_url(self, param_dict: dict) -> str:
        # 去掉URL中的多余换行空格,避免参数解析错误
        url = (
            f"https://www.rbc.ru/v10/search/ajax/?"
            f"project={param_dict['project']}&"
            f"category={param_dict['category']}&"
            f"dateFrom={param_dict['dateFrom']}&"
            f"dateTo={param_dict['dateTo']}&"
            f"offset={param_dict['offset']}&"
            f"limit={param_dict['limit']}&"
            f"query={param_dict['query']}&"
            f"material={param_dict['material']}"
        )
        return url
    def _get_search_table(self, param_dict: dict,
                          includeText: bool = True) -> pd.DataFrame:
        url = self._get_url(param_dict)
        try:
            # 添加请求延迟,避免触发反爬
            time.sleep(1)
            r = rq.get(url, headers=HEADERS)
            r.raise_for_status() # 触发HTTP错误异常
            
            # 检查响应是否为JSON
            try:
                response_json = r.json()
            except ValueError:
                print(f"非JSON响应: {r.text[:500]}")
                return pd.DataFrame()
            
            if 'items' not in response_json:
                print("响应JSON中无'items'字段")
                return pd.DataFrame()
            
            search_table = pd.DataFrame(response_json['items'])
            if includeText and not search_table.empty:
                get_text = lambda x: self._get_article_data(x['fronturl'])
                search_table[['overview', 'text']] = search_table.apply(get_text,
                                                                        axis=1).tolist()
                
            return search_table.sort_values('publish_date_t', ignore_index=True)
        except rq.exceptions.RequestException as e:
            print(f"请求失败: {str(e)}")
            return pd.DataFrame()
    
    def _get_article_data(self, url: str):
        try:
            time.sleep(0.5)
            r = rq.get(url, headers=HEADERS)
            r.raise_for_status()
            soup = bs(r.text, features="lxml")
            div_overview = soup.find('div', {'class': 'article__text__overview'})
            overview = div_overview.text.replace('<br />','\n').strip() if div_overview else None
            
            p_text = soup.find_all('p')
            text = ' '.join(map(lambda x: x.text.replace('<br />','\n').strip(), p_text)) if p_text else None
            
            return overview, text 
        except rq.exceptions.RequestException as e:
            print(f"获取文章失败: {str(e)}")
            return None, None
    
    def get_articles(self,
                     param_dict,
                     time_step = 7,
                     save_every = 5,
                     save_excel = True) -> pd.DataFrame:

        param_copy = param_dict.copy()
        time_step = timedelta(days=time_step)
        dateFrom = datetime.strptime(param_copy['dateFrom'], '%d.%m.%Y')
        dateTo = datetime.strptime(param_copy['dateTo'], '%d.%m.%Y')
        if dateFrom > dateTo:
            raise ValueError('dateFrom should be less than dateTo')
        
        out = pd.DataFrame()
        save_counter = 0

        while dateFrom <= dateTo:
            current_end = dateFrom + time_step
            param_copy['dateTo'] = current_end.strftime("%d.%m.%Y") if current_end <= dateTo else dateTo.strftime("%d.%m.%Y")
            print(f'解析文章时间段: {param_copy["dateFrom"]} 至 {param_copy["dateTo"]}')
            
            batch_df = self._get_search_table(param_copy)
            if not batch_df.empty:
                out = pd.concat([out, batch_df], ignore_index=True)
            
            # 更新起始日期,避免重复抓取
            dateFrom = current_end + timedelta(days=1)
            param_copy['dateFrom'] = dateFrom.strftime("%d.%m.%Y")
            
            save_counter += 1
            if save_counter == save_every:
                display.clear_output(wait=True)
                out.to_excel("/tmp/checkpoint_table.xlsx", index=False)
                print('检查点已保存!')
                save_counter = 0
        
        if save_excel and not out.empty:
            out.to_excel(f"rbc_{param_dict['dateFrom']}_{param_dict['dateTo']}.xlsx", index=False)
        print('完成')
        return out

# 参数配置
query = 'rbc'
project = "rbcnews"
category = "TopRbcRu_economics"
material = ""
dateFrom = '2021-01-01'
dateTo = "2021-02-28"
offset = 0
limit = 100

param_dict = {
    'query'   : query, 
    'project' : project,
    'category': category,
    'dateFrom': datetime.strptime(dateFrom, '%Y-%m-%d').strftime('%d.%m.%Y'),
    'dateTo'  : datetime.strptime(dateTo, '%Y-%m-%d').strftime('%d.%m.%Y'),
    'offset'  : str(offset),
    'limit'   : str(limit),
    'material': material
}

parser = rbc_parser()
tbl = parser._get_search_table(param_dict, includeText=True)
print(f"单次查询获取文章数: {len(tbl)}")
print(tbl.head())

table = parser.get_articles(param_dict=param_dict, time_step=7, save_every=5, save_excel=True)
print(f"总获取文章数: {len(table)}")
print(table.head())

额外说明

  • 如果修复后仍无法获取有效JSON,需确认RBC当前的搜索API端点是否变更,可通过浏览器开发者工具抓包分析最新的搜索请求格式。
  • 频繁请求可能触发反爬限制,代码中已添加基础延迟,可根据实际情况调整时长。

内容的提问来源于stack exchange,提问作者Mr Atlantic

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.28 00:25:00