如何用betfairlightweight将Betfair bz2历史赔率文件转为CSV?
使用betfairlightweight转换Betfair bz2历史数据为CSV
依赖安装
先安装必要的包:
pip install betfairlightweight pandas
核心代码实现
以下代码会批量处理指定目录下的所有.bz2文件,提取Match Odds市场的关键数据(包括市场信息、runner价格变动、时间戳等)并保存为CSV:
import bz2 import os import pandas as pd from betfairlightweight import load_market def process_bz2_to_csv(bz2_file_path, output_dir): # 读取bz2文件内容 with bz2.open(bz2_file_path, 'rb') as f: market_data = f.read() # 加载市场数据 market = load_market(market_data) # 提取基础市场信息 market_info = { 'market_id': market.market_id, 'market_name': market.market_name, 'event_name': market.event.name, 'event_date': market.event.open_date } # 整理runner和价格数据 rows = [] for runner in market.runners: runner_info = { 'selection_id': runner.selection_id, 'runner_name': runner.runner_name } # 遍历所有价格更新快照 for update in runner.updates: row = {**market_info, **runner_info} row['timestamp'] = update.date # 提取最佳背价和出价(如果存在) row['best_back_price'] = update.ex.available_to_back[0].price if update.ex.available_to_back else None row['best_back_size'] = update.ex.available_to_back[0].size if update.ex.available_to_back else None row['best_lay_price'] = update.ex.available_to_lay[0].price if update.ex.available_to_lay else None row['best_lay_size'] = update.ex.available_to_lay[0].size if update.ex.available_to_lay else None rows.append(row) # 转换为DataFrame并保存为CSV df = pd.DataFrame(rows) output_filename = f"{os.path.splitext(os.path.basename(bz2_file_path))[0]}.csv" output_path = os.path.join(output_dir, output_filename) df.to_csv(output_path, index=False) print(f"已处理文件:{bz2_file_path},保存至:{output_path}") def batch_process_bz2(input_dir, output_dir): # 创建输出目录(如果不存在) os.makedirs(output_dir, exist_ok=True) # 遍历目录下所有bz2文件 for filename in os.listdir(input_dir): if filename.endswith('.bz2'): bz2_path = os.path.join(input_dir, filename) process_bz2_to_csv(bz2_path, output_dir) # 使用示例 if __name__ == "__main__": # 替换为你的bz2文件所在目录 INPUT_DIR = "./betfair_bz2_files" # 替换为输出CSV的目标目录 OUTPUT_DIR = "./betfair_csv_output" batch_process_bz2(INPUT_DIR, OUTPUT_DIR)
代码说明
load_market是betfairlightweight专门用于加载历史市场数据的方法,自动解析bz2中的原始Betfair数据流- 代码提取了每个runner的价格更新快照,包含时间戳、最佳背/出价及对应金额,和目标网站的输出逻辑一致
- 批量处理功能支持一次性转换目录下所有bz2文件,每个文件对应生成一个独立CSV
备选方案(使用betfair_parser)
如果betfairlightweight出现兼容性问题,可以使用专门的历史数据解析库betfair_parser:
pip install betfair_parser pandas
对应处理代码:
import bz2 import os import pandas as pd from betfair_parser import parse_market def process_bz2_with_parser(bz2_file_path, output_dir): with bz2.open(bz2_file_path, 'rb') as f: data = f.read() market = parse_market(data) market_info = { 'market_id': market.id, 'market_name': market.name, 'event_name': market.event.name, 'event_date': market.event.open_date } rows = [] for runner in market.runners: runner_info = {'selection_id': runner.selection_id, 'runner_name': runner.name} for snap in runner.snapshots: row = {**market_info, **runner_info} row['timestamp'] = snap.timestamp row['best_back_price'] = snap.back_prices[0].price if snap.back_prices else None row['best_back_size'] = snap.back_prices[0].size if snap.back_prices else None row['best_lay_price'] = snap.lay_prices[0].price if snap.lay_prices else None row['best_lay_size'] = snap.lay_prices[0].size if snap.lay_prices else None rows.append(row) df = pd.DataFrame(rows) output_filename = f"{os.path.splitext(os.path.basename(bz2_file_path))[0]}.csv" df.to_csv(os.path.join(output_dir, output_filename), index=False) # 批量处理逻辑可复用之前的batch_process_bz2函数
内容的提问来源于stack exchange,提问作者gilberke
相关产品推荐
相关产品推荐

