You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python Pandas中KeyError: 'Total Minutes'运行时错误解决求助

KeyError: 'Total Minutes' 解决方法

问题背景

需将自定义函数open_duration或res_duration的结果存入selected_columns_df的新列['Open Resolve Duration'],运行代码时触发KeyError: 'Total Minutes'。

原始代码

import pandas as pd
pd.options.mode.copy_on_write = True

def calculate_totduration(sr_creation, sr_resolved):
    start_date = pd.to_datetime(sr_creation)
    end_date = pd.to_datetime(sr_resolved)
    total_minutes = difference.total_seconds()/60
    return total_minutes

def open_duration(open_time):
    print(type(open_time))
    #sr_open_time = open_time.astype(int)
    sr_open_time = pd.to_numeric(open_time)    
    if sr_open_time <=240:
        return('Open < 4 hours')    
    elif sr_open_time <=480:
        return('Open from 4 hrs to 8 hrs')    
    elif sr_open_time <=720:
        return('Open from 8 hrs to 12 hrs')    
    elif sr_open_time <=1440:
        return('Open 12 hrs to 24 hrs')    
    elif sr_open_time <=2880:
        return('Open 24 to 48 hrs')
    else:
        return('OPEN > 48 hrs')
    
def res_duration(res_time):
    sr_res_time = pd.to_numeric(res_time)  
if sr_res_time <= 240:
        return('Within 4 hours')    
elif sr_res_time <=480:
        return('Bet 4 hrs to 8 hrs')    
elif sr_res_time <=720:
        return('8 hrs to 12 hrs')    
elif sr_res_time <=1440:
        return('12 hrs to 24 hrs')    
else:
        return('> 24 hrs')

df = pd.read_csv('C:\\Users\\XXXXXX\\Desktop\\PythonCodes\\Testing10.csv',sep=',',skiprows=0,low_memory=False,encoding='utf-8')

filtered_df = df[(df['Region/Circle']=='IND')]

selected_columns_df = filtered_df[['Product','Customer Name','SI Number','SI Name','SR Number','SR Status','Region/Circle','Case Type','Source','SR Creation Time','Resolved Time','Total Duration SUM','Case Type','Ser Segment']]

selected_columns_df['Total Minutes'] = selected_columns_df.apply(lambda row:calculate_totduration(row['SR Creation Time'],row['Resolved Time']),axis=1)

if selected_columns_df['SR Status'].str.lower == 'open' or selected_columns_df['SR Status'].str.lower == 're-open':
    selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row:open_duration(calculate_totduration(row['SR Creation Time'],pd.to_datetime('today'))),axis=1)
else:
    selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row :res_duration(row['Total Minutes']))    

selected_columns_df.to_csv('MYoutfile4.csv')
print('FILE WRITTEN')
print(pd.to_datetime('today'))

错误信息

File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\indexes\\base.py", line 3805, in get_loc
return self._engine.get_loc(casted_key)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "index.pyx", line 167, in pandas._libs.index.IndexEngine.get_loc
File "index.pyx", line 175, in pandas._libs.index.IndexEngine.get_loc
File "pandas\\_libs\\index_class_helper.pxi", line 70, in pandas._libs.index.Int64Engine._check_type
KeyError: 'Total Minutes'

The above exception was the direct cause of the following exception:

Traceback (most recent call last):
File "c:\\Users\\xXXXXXx\\Desktop\\PythonCodes\\DATASETS.py", line 71, in 
selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row :res_duration(row['Total Minutes']))
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\frame.py", line 10374, in apply
return op.apply()._finalize(self, method="apply")
^^^^^^^^^^
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\apply.py", line 916, in apply
return self.apply_standard()
^^^^^^^^^^^^^^^^^^^^^
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\apply.py", line 1063, in apply_standard
results, res_index = self.apply_series_generator()
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\apply.py", line 1081, in apply_series_generator
results[i] = self.func(v, *self.args, **self.kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "c:\\Users\\xXXXXXx\\Desktop\\PythonCodes\\DATASETS.py", line 71, in 
selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(lambda row :res_duration(row['Total Minutes']))

         ~~~^^^^^^^^^^^^^^^^^ 
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\series.py", line 1121, in __getitem__
return self._get_value(key)
^^^^^^^^^^^^^^^^^^^^
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\series.py", line 1237, in _get_value
loc = self.index.get_loc(label)
^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\\Users\\xXXXXXx\\AppData\\Local\\Programs\\Python\\Python312\\Lib\\site-packages\\pandas\\core\\indexes\\base.py", line 3812, in get_loc
raise KeyError(key) from err
KeyError: 'Total Minutes'

问题分析及修复方案

1. 核心问题点

  • calculate_totduration函数未定义difference变量,导致生成Total Minutes列时直接报错,该列未被正确创建
  • res_duration函数缩进错误,if/else代码块未包含在函数内
  • 全局if/else逻辑错误:直接用Series和字符串比较会返回布尔Series,而非单个布尔值,导致判断逻辑失效,错误进入else分支引用不存在的列

2. 修复步骤

修复函数错误

  • 在calculate_totduration中添加时间差计算:difference = end_date - start_date
  • 修正res_duration的缩进,将if/else代码块放入函数内

替换全局判断为逐行处理

使用apply对每一行的SR Status进行判断,避免全局Series比较的逻辑错误

修正后的完整代码

import pandas as pd
pd.options.mode.copy_on_write = True

def calculate_totduration(sr_creation, sr_resolved):
    start_date = pd.to_datetime(sr_creation)
    end_date = pd.to_datetime(sr_resolved)
    difference = end_date - start_date  # 新增:计算时间差
    total_minutes = difference.total_seconds() / 60
    return total_minutes

def open_duration(open_time):
    sr_open_time = pd.to_numeric(open_time)    
    if sr_open_time <= 240:
        return 'Open < 4 hours'    
    elif sr_open_time <= 480:
        return 'Open from 4 hrs to 8 hrs'    
    elif sr_open_time <= 720:
        return 'Open from 8 hrs to 12 hrs'    
    elif sr_open_time <= 1440:
        return 'Open 12 hrs to 24 hrs'    
    elif sr_open_time <= 2880:
        return 'Open 24 to 48 hrs'
    else:
        return 'OPEN > 48 hrs'
    
def res_duration(res_time):
    sr_res_time = pd.to_numeric(res_time)  
    # 修正缩进:将判断逻辑放入函数内
    if sr_res_time <= 240:
        return 'Within 4 hours'    
    elif sr_res_time <= 480:
        return 'Bet 4 hrs to 8 hrs'    
    elif sr_res_time <= 720:
        return '8 hrs to 12 hrs'    
    elif sr_res_time <= 1440:
        return '12 hrs to 24 hrs'    
    else:
        return '> 24 hrs'

# 读取数据
df = pd.read_csv('C:\\Users\\XXXXXX\\Desktop\\PythonCodes\\Testing10.csv', sep=',', skiprows=0, low_memory=False, encoding='utf-8')
filtered_df = df[(df['Region/Circle'] == 'IND')]
selected_columns_df = filtered_df[['Product','Customer Name','SI Number','SI Name','SR Number','SR Status','Region/Circle','Case Type','Source','SR Creation Time','Resolved Time','Total Duration SUM','Case Type','Ser Segment']]

# 生成Total Minutes列
selected_columns_df['Total Minutes'] = selected_columns_df.apply(lambda row: calculate_totduration(row['SR Creation Time'], row['Resolved Time']), axis=1)

# 逐行判断SR Status,生成目标列
def get_duration_label(row):
    status = row['SR Status'].strip().lower()
    if status in ['open', 're-open']:
        # 计算当前时间与创建时间的差值
        open_minutes = calculate_totduration(row['SR Creation Time'], pd.to_datetime('today'))
        return open_duration(open_minutes)
    else:
        return res_duration(row['Total Minutes'])

selected_columns_df['Open Resolve Duration'] = selected_columns_df.apply(get_duration_label, axis=1)

# 保存结果
selected_columns_df.to_csv('MYoutfile4.csv', index=False)
print('FILE WRITTEN')
print(pd.to_datetime('today'))

内容的提问来源于stack exchange,提问作者Jaffer

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 06:58:12