Python Pandas中KeyError: 'Total Minutes'运行时错误解决求助
KeyError: 'Total Minutes' 解决方法
问题背景
需将自定义函数open_duration或res_duration的结果存入selected_columns_df的新列['Open Resolve Duration'],运行代码时触发KeyError: 'Total Minutes'。
原始代码
import pandas as pd pd.options.mode.copy_on_write = True def calculate_totduration(sr_creation, sr_resolved): start_date = pd.to_datetime(sr_creation) end_date = pd.to_datetime(sr_resolved) total_minutes = difference.total_seconds()/60 return total_minutes def open_duration(open_time): print(type(open_time)) #sr_open_time = open_time.astype(int) sr_open_time = pd.to_numeric(open_time) if sr_open_time <=240: return('Open < 4 hours') elif sr_open_time <=480: return('Open from 4 hrs to 8 hrs') elif sr_open_time <=720: return('Open from 8 hrs to 12 hrs') elif sr_open_time <=1440: return('Open 12 hrs to 24 hrs') elif sr_open_time <=2880: return('Open 24 to 48 hrs') else: return('OPEN > 48 hrs') def res_duration(res_time): sr_res_time = pd.to_numeric(res_time) if sr_res_time <= 240: return('Within 4 hours') elif sr_res_time <=480: return('Bet 4 hrs to 8 hrs') elif sr_res_time <=720: return('8 hrs to 12 hrs') elif sr_res_time <=1440: return('12 hrs to 24 hrs') else: return('> 24 hrs') df = pd.read_csv('C:\\Users\\XXXXXX\\Desktop\\PythonCodes\\Testing10.csv',sep=',',skiprows=0,low_memory=False,encoding='utf-8') filtered_df = df[(df['Region/Circle']=='IND')] selected_columns_df = filtered_df[['Product','Customer Name','SI Number','SI Name','SR Number','SR Status','Region/Circle','Case Type','Source','SR Creation Time','Resolved Time','Total Duration SUM','Case Type','Ser Segment']] selected_columns_df['Total Minutes'] = selected_columns_df.apply(lambda row:calculate_totduration(row['SR Creation Time'],row['Resolved Time']),axis=1) if selected_columns_df['SR Status'].str.lower == 'open' or selected_columns_df['SR Status'].str.lower == 're-open': selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row:open_duration(calculate_totduration(row['SR Creation Time'],pd.to_datetime('today'))),axis=1) else: selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row :res_duration(row['Total Minutes'])) selected_columns_df.to_csv('MYoutfile4.csv') print('FILE WRITTEN') print(pd.to_datetime('today'))
错误信息
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\indexes\\base.py", line 3805, in get_loc return self._engine.get_loc(casted_key) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "index.pyx", line 167, in pandas._libs.index.IndexEngine.get_loc File "index.pyx", line 175, in pandas._libs.index.IndexEngine.get_loc File "pandas\\_libs\\index_class_helper.pxi", line 70, in pandas._libs.index.Int64Engine._check_type KeyError: 'Total Minutes' The above exception was the direct cause of the following exception: Traceback (most recent call last): File "c:\\Users\\xXXXXXx\\Desktop\\PythonCodes\\DATASETS.py", line 71, in selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row :res_duration(row['Total Minutes'])) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\frame.py", line 10374, in apply return op.apply()._finalize(self, method="apply") ^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\apply.py", line 916, in apply return self.apply_standard() ^^^^^^^^^^^^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\apply.py", line 1063, in apply_standard results, res_index = self.apply_series_generator() ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\apply.py", line 1081, in apply_series_generator results[i] = self.func(v, *self.args, **self.kwargs) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "c:\\Users\\xXXXXXx\\Desktop\\PythonCodes\\DATASETS.py", line 71, in selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row :res_duration(row['Total Minutes'])) ~~~^^^^^^^^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\series.py", line 1121, in __getitem__ return self._get_value(key) ^^^^^^^^^^^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\series.py", line 1237, in _get_value loc = self.index.get_loc(label) ^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\indexes\\base.py", line 3812, in get_loc raise KeyError(key) from err KeyError: 'Total Minutes'
问题分析及修复方案
1. 核心问题点
calculate_totduration函数未定义difference变量,导致生成Total Minutes列时直接报错,该列未被正确创建res_duration函数缩进错误,if/else代码块未包含在函数内- 全局
if/else逻辑错误:直接用Series和字符串比较会返回布尔Series,而非单个布尔值,导致判断逻辑失效,错误进入else分支引用不存在的列
2. 修复步骤
修复函数错误
- 在
calculate_totduration中添加时间差计算:difference = end_date - start_date - 修正
res_duration的缩进,将if/else代码块放入函数内
替换全局判断为逐行处理
使用apply对每一行的SR Status进行判断,避免全局Series比较的逻辑错误
修正后的完整代码
import pandas as pd pd.options.mode.copy_on_write = True def calculate_totduration(sr_creation, sr_resolved): start_date = pd.to_datetime(sr_creation) end_date = pd.to_datetime(sr_resolved) difference = end_date - start_date # 新增:计算时间差 total_minutes = difference.total_seconds() / 60 return total_minutes def open_duration(open_time): sr_open_time = pd.to_numeric(open_time) if sr_open_time <= 240: return 'Open < 4 hours' elif sr_open_time <= 480: return 'Open from 4 hrs to 8 hrs' elif sr_open_time <= 720: return 'Open from 8 hrs to 12 hrs' elif sr_open_time <= 1440: return 'Open 12 hrs to 24 hrs' elif sr_open_time <= 2880: return 'Open 24 to 48 hrs' else: return 'OPEN > 48 hrs' def res_duration(res_time): sr_res_time = pd.to_numeric(res_time) # 修正缩进:将判断逻辑放入函数内 if sr_res_time <= 240: return 'Within 4 hours' elif sr_res_time <= 480: return 'Bet 4 hrs to 8 hrs' elif sr_res_time <= 720: return '8 hrs to 12 hrs' elif sr_res_time <= 1440: return '12 hrs to 24 hrs' else: return '> 24 hrs' # 读取数据 df = pd.read_csv('C:\\Users\\XXXXXX\\Desktop\\PythonCodes\\Testing10.csv', sep=',', skiprows=0, low_memory=False, encoding='utf-8') filtered_df = df[(df['Region/Circle'] == 'IND')] selected_columns_df = filtered_df[['Product','Customer Name','SI Number','SI Name','SR Number','SR Status','Region/Circle','Case Type','Source','SR Creation Time','Resolved Time','Total Duration SUM','Case Type','Ser Segment']] # 生成Total Minutes列 selected_columns_df['Total Minutes'] = selected_columns_df.apply(lambda row: calculate_totduration(row['SR Creation Time'], row['Resolved Time']), axis=1) # 逐行判断SR Status,生成目标列 def get_duration_label(row): status = row['SR Status'].strip().lower() if status in ['open', 're-open']: # 计算当前时间与创建时间的差值 open_minutes = calculate_totduration(row['SR Creation Time'], pd.to_datetime('today')) return open_duration(open_minutes) else: return res_duration(row['Total Minutes']) selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(get_duration_label, axis=1) # 保存结果 selected_columns_df.to_csv('MYoutfile4.csv', index=False) print('FILE WRITTEN') print(pd.to_datetime('today'))
内容的提问来源于stack exchange,提问作者Jaffer
相关产品推荐
相关产品推荐

