使用Pandas从API爬取数据:通过循环获取全部记录
批量爬取巴西众议员开支数据并合并为DataFrame
核心实现方案
遍历已有的议员ID列表,针对每个ID调用API获取全部开支记录(处理API分页逻辑),最终将所有数据合并成一个完整的DataFrame。
完整代码示例
import requests import pandas as pd import json from time import sleep # 假设你已有的议员ID数据存储在deputados_ids_df中,包含名为'id'的列 # 若从文件读取可使用:deputados_ids_df = pd.read_csv("你的议员ID文件路径.csv") def fetch_deputado_expenses(deputado_id, year=2020): """获取单个议员的所有开支记录,自动处理分页""" all_records = [] base_url = f"https://dadosabertos.camara.leg.br/api/v2/deputados/{deputado_id}/despesas" params = { "ano": year, "ordem": "ASC", "ordenarPor": "ano", "pagina": 1, "itens": 100 # 每页请求最大条目数,提升效率 } while True: try: response = requests.get(base_url, params=params) response.raise_for_status() # 捕获HTTP请求错误 data = json.loads(response.text) page_records = data["dados"] if not page_records: break # 无数据时终止循环 all_records.extend(page_records) # 检查是否存在下一页 has_next_page = any(link["rel"] == "next" for link in data["links"]) if not has_next_page: break params["pagina"] += 1 sleep(1) # 添加延迟,避免请求过于频繁触发API限制 except Exception as e: print(f"议员ID {deputado_id} 数据获取失败: {str(e)}") break return pd.DataFrame(all_records) # 初始化空DataFrame存储所有数据 all_expenses_df = pd.DataFrame() # 遍历所有议员ID批量获取数据 total_ids = len(deputados_ids_df["id"].unique()) for index, deputado_id in enumerate(deputados_ids_df["id"].unique()): print(f"处理进度: {index+1}/{total_ids} | 议员ID: {deputado_id}") single_deputado_df = fetch_deputado_expenses(deputado_id) if not single_deputado_df.empty: # 添加议员ID列,方便后续关联分析 single_deputado_df["deputado_id"] = deputado_id all_expenses_df = pd.concat([all_expenses_df, single_deputado_df], ignore_index=True) # 查看合并后的数据概况 print(f"合并后总记录数: {all_expenses_df.shape[0]}") all_expenses_df.head()
关键细节说明
- 分页处理:API默认返回条目数有限,通过
pagina参数逐页请求,直到无下一页或无数据返回 - 异常防护:捕获请求过程中的错误,避免单个议员的数据获取失败导致整个流程中断
- 请求限流:添加
sleep(1)控制请求频率,防止触发API的反爬限制 - 数据关联:给每条开支记录添加
deputado_id列,方便后续关联议员基础信息
内容的提问来源于stack exchange,提问作者RodMorais
相关产品推荐
相关产品推荐

