大体积JSON文件匹配与时间差计算代码优化需求
优化超大规模JSON数据匹配与处理的Python代码
需求说明
- 处理两个各含100亿条记录的JSON文件,按b、c、d、e四个字段匹配条目(注:键
a为时间字段,仅用于计算时间差,不参与匹配判断) - 条目匹配成功后,计算双方
a字段(platform_time)的时间差 - 匹配完成后,从两个文件中移除已匹配的条目
匹配规则示例:
- one[0] 与 two[1] 满足b、c、d、e字段完全一致,属于匹配条目
- one[1] 在two中找不到对应匹配字段的条目,属于未匹配条目
JSON数据示例
文件one
"one": [ { "a" : "2022-09-12 00:00:00.000", "b" : "apple", "c" : "1", "d" : "2022-09-11 23:59:59.997", "e" : 88 }, { "a" : "2022-09-12 00:00:00.000", "b" : "orange", "c" : "2", "d" : "2022-09-11 23:59:59.997", "e" : 87 }, { "a" : "2022-09-12 00:00:10.001", "b" : "apple", "c" : "6", "d" : "2022-09-11 23:59:59.997", "e" : 88 },... ]
文件two
"two": [ { "a" : "2022-09-12 00:00:30.000", "b" : "orange", "c" : "2", "d" : "2022-09-11 23:59:59.997", "e" : 87 }, { "a" : "2022-09-12 00:00:10.001", "b" : "apple", "c" : "1", "d" : "2022-09-11 23:59:59.997", "e" : 88 }, { "a" : "2022-09-12 00:00:30.000", "b" : "orange", "c" : "200", "d" : "2021-09-11 23:59:59.997", "e" : 81 },... ]
现有代码的问题
原代码采用嵌套循环遍历,时间复杂度为O(n²),面对100亿条数据时完全无法在合理时间内完成;同时直接将整个JSON文件加载到内存,会导致内存溢出,根本无法运行。此外代码中存在逻辑错误(如硬编码循环次数为100亿、错误引入未提及的amount字段)。
原代码:
import datetime import json import numpy as np import random lst_in_seconds = [] f = open('one_all.json') one = json.load(f) f.close() f1 = open('two_all.json') two = json.load(f1) f1.close() counter_one_better = 0 counter_two_better = 0 counter_the_same = 0 for k in range(10000000000): for i in range(10000000000): if one['one'][k]['b'] == two['two'][i]['b'] and one['one'][k]['e'] == two['two'][i]['e'] and one['one'][k]['amount'] == two['two'][i]['amount'] and one['one'][k]['d'] == two['two'][i]['d'] and one['one'][k]['c'] == two['two'][i]['c']: if (one['one'][k]['a']) < (two['two'][i]['a']): # one better than two delt_one = datetime.datetime.strptime((one['one'][k]['a']), '%Y-%m-%d %H:%M:%S.%f') delt_two = datetime.datetime.strptime((two['two'][i]['a']), '%Y-%m-%d %H:%M:%S.%f') delta = delt_two - delt_one diff_in_seconds = delta.total_seconds() lst_in_seconds.append(diff_in_seconds) counter_one_better += 1 two['two'][i]['b'] = random.randint(0,100000) break elif (one['one'][k]['a']) == (two['two'][i]['a']): # same delt_one = datetime.datetime.strptime((one['one'][k]['a']), '%Y-%m-%d %H:%M:%S.%f') delt_two = datetime.datetime.strptime((two['two'][i]['a']), '%Y-%m-%d %H:%M:%S.%f') delta = delt_two - delt_one diff_in_seconds = delta.total_seconds() lst_in_seconds.append(diff_in_seconds) counter_the_same += 1 two['two'][i]['b'] = random.randint(0,100000) break elif (one['one'][k]['a']) > (two['two'][i]['a']): delt_one = datetime.datetime.strptime((one['one'][k]['a']), '%Y-%m-%d %H:%M:%S.%f') delt_two = datetime.datetime.strptime((two['two'][i]['a']), '%Y-%m-%d %H:%M:%S.%f') delta = delt_one - delt_two diff_in_seconds = delta.total_seconds() diff_in_seconds_to_str = float(('-' + str(diff_in_seconds))) lst_in_seconds.append(diff_in_seconds_to_str) counter_two_better += 1 two['two'][i]['b'] = random.randint(0,100000) break #print('counter_the_same',counter_the_same,'count') #print('counter_one_better',counter_one_better,'count') #print('counter_two_better',counter_two_better,'count','\n') print('one better than two in ', round((counter_one_better / (counter_two_better+counter_one_better+counter_the_same))*100,4),'% case') print('the same ', round((counter_the_same / (counter_two_better+counter_one_better+counter_the_same))*100,4),'% case') print('two better than one in ', round((counter_two_better / (counter_two_better+counter_one_better+counter_the_same))*100,4),'% case','\n')
优化方案与代码
核心优化点
- 流式处理:使用
ijson库逐行读取JSON,避免加载整个文件到内存 - 哈希映射:将其中一个文件的条目按匹配键(b、c、d、e)分组存储,将查找时间从O(n)降到O(1)
- 预转换时间:提前将时间字符串转为时间戳,减少重复解析的开销
- 增量输出:处理过程中直接输出未匹配的条目到新文件,避免内存堆积
优化后代码
import datetime import ijson import json from collections import defaultdict # 时间格式 TIME_FORMAT = '%Y-%m-%d %H:%M:%S.%f' def str_to_timestamp(time_str): """将时间字符串转为时间戳""" return datetime.datetime.strptime(time_str, TIME_FORMAT).timestamp() def build_match_index(file_path, root_key): """构建匹配键到时间戳列表的索引""" match_index = defaultdict(list) with open(file_path, 'rb') as f: # 流式读取JSON数组中的每个元素 for item in ijson.items(f, f'{root_key}.item'): match_key = (item['b'], item['c'], item['d'], item['e']) timestamp = str_to_timestamp(item['a']) match_index[match_key].append(timestamp) return match_index def process_files(one_path, two_path, output_one_path, output_two_path): # 先构建one文件的匹配索引 match_index = build_match_index(one_path, 'one') counters = { 'one_better': 0, 'two_better': 0, 'same': 0 } time_diffs = [] # 处理two文件,同时输出未匹配条目 with open(two_path, 'rb') as f_in, open(output_two_path, 'w', encoding='utf-8') as f_out: f_out.write('{"two": [') first_item = True for item in ijson.items(f_in, 'two.item'): match_key = (item['b'], item['c'], item['d'], item['e']) two_ts = str_to_timestamp(item['a']) matched = False if match_key in match_index and match_index[match_key]: one_ts = match_index[match_key].pop() diff = two_ts - one_ts time_diffs.append(diff) if diff > 0: counters['one_better'] += 1 elif diff == 0: counters['same'] += 1 else: counters['two_better'] += 1 matched = True # 输出未匹配的条目到新文件 if not matched: if not first_item: f_out.write(',') f_out.write(json.dumps(item, ensure_ascii=False)) first_item = False f_out.write(']}') # 处理one文件,输出未匹配条目 with open(one_path, 'rb') as f_in, open(output_one_path, 'w', encoding='utf-8') as f_out: f_out.write('{"one": [') first_item = True for item in ijson.items(f_in, 'one.item'): match_key = (item['b'], item['c'], item['d'], item['e']) if not match_index[match_key]: if not first_item: f_out.write(',') f_out.write(json.dumps(item, ensure_ascii=False)) first_item = False f_out.write(']}') # 统计结果输出 total = sum(counters.values()) if total > 0: print(f"one比two快的占比: {round((counters['one_better']/total)*100,4)}%") print(f"时间相同的占比: {round((counters['same']/total)*100,4)}%") print(f"two比one快的占比: {round((counters['two_better']/total)*100,4)}%") else: print("无匹配条目") if __name__ == '__main__': # 替换为实际文件路径 process_files( one_path='one_all.json', two_path='two_all.json', output_one_path='one_unmatched.json', output_two_path='two_unmatched.json' )
补充说明
- 需要先安装
ijson库:pip install ijson - 若两个文件都极大,可先对较小的文件构建索引,减少内存占用
- 该方案时间复杂度为O(n+m),n和m分别为两个文件的记录数,可高效处理超大规模数据
内容的提问来源于stack exchange,提问作者ditrauth
相关产品推荐
相关产品推荐

