基于requests的Python爬虫爬取第二页时触发KeyError问题
问题:爬取2GIS迪拜酒吧列表时第二页触发KeyError
尝试用requests模块爬取迪拜区域的酒吧和餐厅名称,脚本能正常解析第一页内容,但获取第二页时抛出KeyError: 'result',而网页显示仍有更多页面未爬取。
原代码
import requests from pprint import pprint link = 'https://2gis.ae/dubai/search/Bars/rubricId/159' url = 'https://catalog.api.2gis.ru/3.0/items' headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/103.0.0.0 Safari/537.36', } params = { 'page': 1, 'page_size': 12, 'rubric_id': 159, 'fields': 'items.locale,items.flags,search_attributes,items.adm_div,items.city_alias,items.region_id,items.segment_id,items.reviews,items.point,request_type,context_rubrics,query_context,items.links,items.name_ex,items.name_back,items.org,items.group,items.external_content,items.comment,items.ads.options,items.email_for_sending.allowed,items.stat,items.description,items.geometry.centroid,items.geometry.selection,items.geometry.style,items.timezone_offset,items.context,items.address,items.is_paid,items.access,items.access_comment,items.for_trucks,items.is_incentive,items.paving_type,items.capacity,items.schedule,items.floors,dym,ad,items.rubrics,items.routes,items.reply_rate,items.purpose,items.attribute_groups,items.route_logo,items.has_goods,items.has_apartments_info,items.has_pinned_goods,items.has_realty,items.has_payments,items.is_promoted,items.delivery,items.order_with_cart,search_type,items.has_discount,items.metarubrics,broadcast,items.detailed_subtype,items.temporary_unavailable_atm_services,items.poi_category', 'key': 'rurbbn3446', 'locale': 'en_AE', 'search_device_type': 'desktop', 'search_user_hash': '7233966692562515761', 'viewpoint1': '55.09734166196474,25.248071810295556', 'viewpoint2': '55.421438338035266,25.16107818970444', 'stat[sid]': '3de514cd-c705-4753-91b6-e997c2227aa7', 'stat[user]': '1723a6dd-6008-4197-a074-c1e34e27f785', 'shv': '2023-05-02-14', 'r': '2843121634' } with requests.Session() as s: s.headers.update(headers) while True: print(f"processing page ==============> {params['page']}") res = s.get(url,params=params) try: res.json()['result']['items'] except KeyError: break for item in res.json()['result']['items']: print(item['name']) params['page']+=1
报错信息
Traceback (most recent call last): File "C:\Users\C.L\Desktop\Python basic\python scripts\demo.py", line 35, in <module> res.json()['result']['items'] KeyError: 'result'
解决方法
1. 先排查API返回的真实内容
出现KeyError时,不要直接break,先打印完整的响应JSON,明确API的返回状态(大概率是临时参数过期,API返回了错误提示)。修改异常处理部分:
try: response_data = res.json() items = response_data['result']['items'] except KeyError: print("API返回异常内容:", res.json()) break
2. 动态提取有效请求参数
你代码里的search_user_hash、stat[sid]、stat[user]、r都是页面加载时生成的临时参数,过期后请求会被拒绝。正确做法是先请求目标网页,从页面中提取这些动态参数:
import requests from bs4 import BeautifulSoup import json # 先请求网页获取动态参数 session = requests.Session() headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/103.0.0.0 Safari/537.36'} session.headers.update(headers) page_res = session.get('https://2gis.ae/dubai/search/Bars/rubricId/159') soup = BeautifulSoup(page_res.text, 'html.parser') # 从页面的__NEXT_DATA__脚本中提取参数 next_data = json.loads(soup.find('script', id='__NEXT_DATA__').text) api_key = next_data['props']['pageProps']['apiKey'] user_hash = next_data['props']['pageProps']['userHash']
3. 优化后的完整代码
import requests import time from bs4 import BeautifulSoup import json def crawl_2gis_bars(): session = requests.Session() headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/103.0.0.0 Safari/537.36', } session.headers.update(headers) # 1. 获取页面动态参数 page_url = 'https://2gis.ae/dubai/search/Bars/rubricId/159' page_res = session.get(page_url) page_res.raise_for_status() soup = BeautifulSoup(page_res.text, 'html.parser') next_data = json.loads(soup.find('script', id='__NEXT_DATA__').text) # 构建API请求参数 api_params = { 'page': 1, 'page_size': 12, 'rubric_id': 159, 'fields': 'items.name', # 只请求需要的字段,减少数据量 'key': next_data['props']['pageProps']['apiKey'], 'locale': 'en_AE', 'search_device_type': 'desktop', 'search_user_hash': next_data['props']['pageProps']['userHash'], 'viewpoint1': '55.09734166196474,25.248071810295556', 'viewpoint2': '55.421438338035266,25.16107818970444', } api_url = 'https://catalog.api.2gis.ru/3.0/items' while True: print(f"处理第 {api_params['page']} 页") time.sleep(1) # 添加延时,避免触发反爬 res = session.get(api_url, params=api_params) res.raise_for_status() response_data = res.json() # 检查返回结构 if 'result' not in response_data or not response_data['result']['items']: print("无更多内容或API返回异常") break # 打印名称 for item in response_data['result']['items']: print(item.get('name', '未获取到名称')) api_params['page'] += 1 if __name__ == '__main__': crawl_2gis_bars()
4. 注意事项
- 遵守2GIS的API使用规则,不要高频请求,避免IP被封禁
- 页面结构可能会变化,若提取参数失败,需重新检查页面的脚本内容调整提取逻辑
内容的提问来源于stack exchange,提问作者MITHU
相关产品推荐
相关产品推荐

