You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

空DataFrame调用clean_list函数触发ValueError问题求助

问题:空DataFrame应用自定义函数触发ValueError错误

原代码

def replaceitem(x):
    if x in ['ORION', 'ACTION', 'ICE', 'IRIS', 'FOCUS']:
        return 'CRM Application'
    else:
        return x
    
def clean_list(row):
    new_list = sorted(set(row['APLN_NM']), key=lambda x: row['APLN_NM'].index(x))
    for idx,i in enumerate(new_list):
        new_list[idx] = replaceitem(i)
    new_list = sorted(set(new_list), key=lambda x: new_list.index(x))
    return new_list

# 应用函数到DataFrame
df_agg['APLN_NM_DISTINCT']        = df_agg.apply(clean_list, axis = 1)
df_agg_single['APLN_NM_DISTINCT'] = df_agg_single.apply(clean_list, axis = 1)

错误信息

---------------------------------------------------------------------------
KeyError                                  Traceback (most recent call last)
/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/indexes/base.py in get_loc(self, key, method, tolerance)
   2890             try:
-> 2891                 return self._engine.get_loc(casted_key)
   2892             except KeyError as err:

pandas/_libs/index.pyx in pandas._libs.index.IndexEngine.get_loc()

pandas/_libs/index.pyx in pandas._libs.index.IndexEngine.get_loc()

pandas/_libs/hashtable_class_helper.pxi in pandas._libs.hashtable.PyObjectHashTable.get_item()

pandas/_libs/hashtable_class_helper.pxi in pandas._libs.hashtable.PyObjectHashTable.get_item()

KeyError: 'APLN_NM_DISTINCT'

The above exception was the direct cause of the following exception:

KeyError                                  Traceback (most recent call last)
/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/generic.py in _set_item(self, key, value)
   3570         try:
-> 3571             loc = self._info_axis.get_loc(key)
   3572         except KeyError:

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/indexes/base.py in get_loc(self, key, method, tolerance)
   2892             except KeyError as err:
-> 2893                 raise KeyError(key) from err
   2894 

KeyError: 'APLN_NM_DISTINCT'

During handling of the above exception, another exception occurred:

ValueError                                Traceback (most recent call last)
<ipython-input-71-e8b5e8d5b514> in <module>
    431 #*********************************************************************************************************************************************
    432 df_agg['APLN_NM_DISTINCT']        = df_agg.apply(clean_list, axis = 1)
-> 433 df_agg_single['APLN_NM_DISTINCT'] = df_agg_single.apply(clean_list, axis = 1)
    434 
    435 df_agg['TOTAL_HOLD_TIME']        = df_agg_single['TOTAL_HOLD_TIME'].astype(int)

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/frame.py in __setitem__(self, key, value)
   3038         else:
   3039             # set column
-> 3040             self._set_item(key, value)
   3041 
   3042     def _setitem_slice(self, key: slice, value):

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/frame.py in _set_item(self, key, value)
   3115         self._ensure_valid_index(value)
   3116         value = self._sanitize_column(key, value)
-> 3117         NDFrame._set_item(self, key, value)
   3118 
   3119         # check if we are modifying a copy

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/generic.py in _set_item(self, key, value)
   3572         except KeyError:
   3573             # This item wasn't present, just insert at end
-> 3574             self._mgr.insert(len(self._info_axis), key, value)
   3575             return
   3576 

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/internals/managers.py in insert(self, loc, item, value, allow_duplicates)
   1187             value = _safe_reshape(value, (1,) + value.shape)
   1188 
-> 1189         block = make_block(values=value, ndim=self.ndim, placement=slice(loc, loc + 1))
   1190 
   1191         for blkno, count in _fast_count_smallints(self.blknos[loc:]):

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/internals/blocks.py in make_block(values, placement, klass, ndim, dtype)
   2717         values = DatetimeArray._simple_new(values, dtype=dtype)
   2718 
-> 2719         return klass(values, ndim=ndim, placement=placement)
   2720 
   2721 

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/internals/blocks.py in __init__(self, values, placement, ndim)
   2373             values = np.array(values, dtype=object)
   2374 
-> 2375         super().__init__(values, ndim=ndim, placement=placement)
   2376 
   2377     @property

/opt/rh/rh-python36/root/usr/lib64/python3.6/site-packages/pandas/core/internals/blocks.py in __init__(self, values, placement, ndim)
    128         if self._validate_ndim and self.ndim and len(self.mgr_locs) != len(self.values):
    129             raise ValueError(
-> 130                 f"Wrong number of items passed {len(self.values)}, "
    131                 f"placement implies {len(self.mgr_locs)}"
    132             )

ValueError: Wrong number of items passed 3, placement implies 1

问题原因

df_agg_single是空DataFrame,调用apply时自定义函数clean_list不会被执行(无行可处理),此时apply返回的结果无法匹配pandas对新列的结构要求,导致赋值时触发ValueError。

解决方案

方案1:判断DataFrame是否为空再处理

直接跳过空DataFrame的函数应用,手动创建同类型空列:

import pandas as pd

# 处理有数据的df_agg
df_agg['APLN_NM_DISTINCT'] = df_agg.apply(clean_list, axis=1)

# 仅在df_agg_single非空时应用函数,否则创建空列
if not df_agg_single.empty:
    df_agg_single['APLN_NM_DISTINCT'] = df_agg_single.apply(clean_list, axis=1)
else:
    # 创建存储列表的object类型空列
    df_agg_single['APLN_NM_DISTINCT'] = pd.Series(dtype='object')

方案2:增强自定义函数的鲁棒性

修改clean_list,处理空值或缺失列的情况,即使空DataFrame后续有数据也能兼容:

def replaceitem(x):
    if x in ['ORION', 'ACTION', 'ICE', 'IRIS', 'FOCUS']:
        return 'CRM Application'
    else:
        return x
    
def clean_list(row):
    # 先获取APLN_NM的值,不存在则返回空列表
    apln_nm = row.get('APLN_NM', [])
    # 处理空列表情况
    if not apln_nm:
        return []
    
    new_list = sorted(set(apln_nm), key=lambda x: apln_nm.index(x))
    for idx,i in enumerate(new_list):
        new_list[idx] = replaceitem(i)
    new_list = sorted(set(new_list), key=lambda x: new_list.index(x))
    return new_list

# 直接应用函数,空DataFrame会自动生成空列
df_agg['APLN_NM_DISTINCT'] = df_agg.apply(clean_list, axis=1)
df_agg_single['APLN_NM_DISTINCT'] = df_agg_single.apply(clean_list, axis=1)

方案3:强制转换为Series

将apply的结果转为指定类型的Series,确保pandas能正确识别列结构:

import pandas as pd

df_agg['APLN_NM_DISTINCT'] = df_agg.apply(clean_list, axis=1)
# 强制转为object类型Series,匹配列表存储需求
df_agg_single['APLN_NM_DISTINCT'] = pd.Series(df_agg_single.apply(clean_list, axis=1), dtype='object')

内容的提问来源于stack exchange,提问作者AshishMulupuri

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.11 16:40:39