Python向量化函数未按预期更新DataFrame权重问题排查
问题:向量化更新DataFrame权重仅部分生效
我编写了selection_update_weights函数,通过布尔掩码多条件匹配更新DataFrame中Win、DNB、O_1_5、O_2_5、U_4_5列的权重值,调用方式为df = selection_update_weights(df)。但处理大型DataFrame时,权重未按预期完全更新,逐行处理因数据集过大需耗时20分钟,寻求高效排查与解决方法。
原始函数代码
def selection_update_weights(df): # Define the selections for 'Win' selections_win = ["W & O 2.5 (both untested)", "Win (untested) & O 2.5", "Win & O 2.5 (untested)", "W & O 2.5", "W & O 1.5 (both untested)", "Win (untested) & O 1.5", "Win & O 1.5 (untested)", "W & O 1.5", "W & U 4.5 (both untested)", "Win (untested) & U 4.5", "Win & U 4.5 (untested)", "W & U 4.5", "W (untested)", "W"] # Create a boolean mask for the condition for 'Win' mask_win = (df['selection_match'] == "no match") & \ (df['selection'].isin(selections_win)) & \ (df['result_match'] == "no match") & \ (df['result'] != 'draw') # Apply the condition and update the 'Win' column df.loc[mask_win, 'Win'] = df.loc[mask_win, 'predicted_score_difference'] + 0.02 # Define the selections for 'DNB' selections_DNB = ["DNB or O 2.5 (both untested)", "DNB (untested) or O 2.5", "DNB or O 2.5 (untested)", "DNB or O 2.5", "DNB or O 1.5 (both untested)", "DNB (untested) or O 1.5", "DNB or O 1.5 (untested)", "DNB or O 1.5", "DNB (untested)", "DNB"] # Create a boolean mask for the condition for 'DNB' mask_DNB = ((df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_DNB)) & \ (df['result_match'] == "no match") & \ (df['result'] != 'draw')) # Apply the condition and update the 'DNB' column df.loc[mask_DNB, 'DNB'] = df.loc[mask_DNB, 'predicted_score_difference'] + 0.02 # Define the selections for O 1.5' selections_O_1_5 = ["W & O 1.5 (both untested)", "Win (untested) & O 1.5", "Win & O 1.5 (untested)", "W & O 1.5", "DNB or O 1.5 (both untested)", "DNB (untested) or O 1.5", "DNB or O 1.5 (untested)", "DNB or O 1.5", "O 1.5 (untested)", "O 1.5"] # Create a boolean mask for the condition for 'O 1.5' mask_O_1_5 = ((df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_O_1_5)) & \ (df['total_score'] < 2)) # Apply the condition and update the 'O 1.5' column df.loc[mask_O_1_5, 'O_1_5'] = df.loc[mask_O_1_5, 'predicted_total_score'] + 0.02 # Define the selections for O 2.5' selections_O_2_5 = ["W & O 2.5 (both untested)", "Win (untested) & O 2.5", "Win & O 2.5 (untested)", "W & O 2.5", "DNB or O 2.5 (both untested)", "DNB (untested) or O 2.5", "DNB or O 2.5 (untested)", "DNB or O 2.5", "O 2.5 (untested)", "O 2.5"] # Create a boolean mask for the condition for 'O 2.5' mask_O_2_5 = ((df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_O_2_5)) & \ (df['total_score'] < 3)) # Apply the condition and update the 'O 2.5' column df.loc[mask_O_2_5, 'O_2_5'] = df.loc[mask_O_2_5, 'predicted_total_score'] + 0.02 # Define the selections for U 4.5' selections_U_4_5 = ["W & U 4.5 (both untested)", "Win (untested) & U 4.5", "Win & U 4.5 (untested)", "W & U 4.5", "U 4.5 (untested)", "U 4.5"] # Create a boolean mask for the condition for 'O 2.5' mask_U_4_5 = ((df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_U_4_5)) & \ (df['total_score'] > 4)) # Apply the condition and update the 'O 2.5' column df.loc[mask_U_4_5, 'U_4_5'] = df.loc[mask_U_4_5, 'predicted_total_score'] - 0.02 return df
原始数据与期望结果
原始df.head()
home_score away_score total_score score_difference predicted_total_score predicted_score_difference result predicted_result result_match Win DNB O_1_5 O_2_5 U_4_5 selection selection_match 44 3 3 6 0 8.748172 8.135116 draw home no match 1.1 0.7 2.0 3.000000 4.0 W & O 2.5 (both untested) no match 50 1 0 1 1 8.605350 7.932909 home home match 1.1 0.7 2.0 8.625350 4.0 W & O 1.5 (both untested) no match 57 1 1 2 0 7.510030 7.750101 draw home no match 1.1 0.7 2.0 7.530030 4.0 W & O 1.5 (both untested) no match 62 0 1 1 1 8.895045 7.710740 away away match 1.1 0.7 2.0 8.915045 4.0 W & O 1.5 (both untested) no match 85 1 0 1 1 8.099853 7.444815 home home match 1.1 0.7 2.0 8.119853 4.0 W & O 1.5 (both untested) no match
期望更新结果
home_score away_score total_score score_difference predicted_total_score predicted_score_difference result predicted_result result_match Win DNB O_1_5 O_2_5 U_4_5 selection selection_match 3 3 6 0 8.748172 8.135116 draw home no match 8.155116 0.7 2.0 3 4.0 W & O 2.5 (both untested) no match 1 0 1 1 8.605350 7.932909 home home match 1.100000 0.7 8.625350 8.625350 4.0 W & O 1.5 (both untested) no match 1 1 2 0 7.510030 7.750101 draw home no match 7.770101 0.7 2.0 7.530030 4.0 W & O 1.5 (both untested) no match 0 1 1 1 8.895045 7.710740 away away match 1.100000 0.7 8.915045 8.915045 4.0 W & O 1.5 (both untested) no match 1 0 1 1 8.099853 7.444815 home home match 1.100000 0.7 8.119853 8.119853 4.0 W & O 1.5 (both untested) no match
高效排查步骤
- 验证掩码匹配行数:对每个掩码统计匹配行数,对比预期更新行数:
快速定位是否因掩码条件错误导致匹配行数不足。print("Win掩码匹配行数:", mask_win.sum()) print("O_1_5掩码匹配行数:", mask_O_1_5.sum()) - 检测链式赋值问题:启用Pandas链式赋值警告,排查是否操作的是DataFrame副本而非原数据:
若触发警告,说明需在函数内先复制DataFrame。import pandas as pd pd.set_option('mode.chained_assignment', 'raise') - 逐行验证条件:对未更新的目标行,单独检查每个掩码条件是否满足:
row = df.loc[44] print("selection_match是否为no match:", row['selection_match'] == 'no match') print("selection是否在selections_win:", row['selection'] in selections_win)
核心问题修复
修复后的函数代码
def selection_update_weights(df): # 复制DataFrame避免链式赋值问题 df = df.copy() # Win列相关逻辑:移除result != 'draw'条件,匹配期望结果 selections_win = ["W & O 2.5 (both untested)", "Win (untested) & O 2.5", "Win & O 2.5 (untested)", "W & O 2.5", "W & O 1.5 (both untested)", "Win (untested) & O 1.5", "Win & O 1.5 (untested)", "W & O 1.5", "W & U 4.5 (both untested)", "Win (untested) & U 4.5", "Win & U 4.5 (untested)", "W & U 4.5", "W (untested)", "W"] mask_win = (df['selection_match'] == "no match") & \ (df['selection'].isin(selections_win)) & \ (df['result_match'] == "no match") df.loc[mask_win, 'Win'] = df.loc[mask_win, 'predicted_score_difference'] + 0.02 # DNB列相关逻辑 selections_DNB = ["DNB or O 2.5 (both untested)", "DNB (untested) or O 2.5", "DNB or O 2.5 (untested)", "DNB or O 2.5", "DNB or O 1.5 (both untested)", "DNB (untested) or O 1.5", "DNB or O 1.5 (untested)", "DNB or O 1.5", "DNB (untested)", "DNB"] mask_DNB = (df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_DNB)) & \ (df['result_match'] == "no match") & \ (df['result'] != 'draw') df.loc[mask_DNB, 'DNB'] = df.loc[mask_DNB, 'predicted_score_difference'] + 0.02 # O_1_5列相关逻辑 selections_O_1_5 = ["W & O 1.5 (both untested)", "Win (untested) & O 1.5", "Win & O 1.5 (untested)", "W & O 1.5", "DNB or O 1.5 (both untested)", "DNB (untested) or O 1.5", "DNB or O 1.5 (untested)", "DNB or O 1.5", "O 1.5 (untested)", "O 1.5"] mask_O_1_5 = (df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_O_1_5)) & \ (df['total_score'] < 2) df.loc[mask_O_1_5, 'O_1_5'] = df.loc[mask_O_1_5, 'predicted_total_score'] + 0.02 # O_2_5列相关逻辑 selections_O_2_5 = ["W & O 2.5 (both untested)", "Win (untested) & O 2.5", "Win & O 2.5 (untested)", "W & O 2.5", "DNB or O 2.5 (both untested)", "DNB (untested) or O 2.5", "DNB or O 2.5 (untested)", "DNB or O 2.5", "O 2.5 (untested)", "O 2.5"] mask_O_2_5 = (df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_O_2_5)) & \ (df['total_score'] < 3) df.loc[mask_O_2_5, 'O_2_5'] = df.loc[mask_O_2_5, 'predicted_total_score'] + 0.02 # U_4_5列相关逻辑:修正注释错误 selections_U_4_5 = ["W & U 4.5 (both untested)", "Win (untested) & U 4.5", "Win & U 4.5 (untested)", "W & U 4.5", "U 4.5 (untested)", "U 4.5"] mask_U_4_5 = (df['selection_match'] == 'no match') & \ (df['selection'].isin(selections_U_4_5)) & \ (df['total_score'] > 4) df.loc[mask_U_4_5, 'U_4_5'] = df.loc[mask_U_4_5, 'predicted_total_score'] - 0.02 return df
关键修复点
- Win列掩码修正:移除
result != 'draw'条件,匹配期望结果中draw行也更新Win的需求。 - 避免链式赋值:函数开头复制DataFrame,确保修改生效。
- 注释修正:修正U_4_5掩码的错误注释,避免后续维护混淆。
内容的提问来源于stack exchange,提问作者PyNoob
相关产品推荐
相关产品推荐

