You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Sklearn XGBoost随机网格搜索获取多指标及报错解决

使用Sklearn结合XGBoost执行多指标随机网格搜索并基于F1阈值计算指标

错误原因分析

你遇到的ValueError核心问题在于使用了旧版Scikit-learn的API:sklearn.grid_search.RandomizedSearchCV是Scikit-learn 0.20版本之前的旧模块,它并不支持传入字典格式的scoring参数来实现多指标评估。只有sklearn.model_selection下的新版RandomizedSearchCV才支持多指标的字典式scoring配置。

解决方案与完整实现

下面是修正后的代码,同时实现了基于F1最优阈值的指标计算(即找到最大化F1分数的概率阈值,再用该阈值计算Precision、Recall、Accuracy):

1. 导入必要模块(替换旧版API)

from sklearn import datasets
from sklearn.metrics import precision_score, recall_score, accuracy_score, roc_auc_score, make_scorer, f1_score
from sklearn.model_selection import train_test_split, RandomizedSearchCV
import xgboost as xgb
import numpy as np

2. 定义基于F1最优阈值的自定义评估函数

这些函数会遍历多个概率阈值,找到使F1分数最高的阈值,再用该阈值计算对应指标:

def precision_at_optimal_f1(y_true, y_proba):
    # 遍历0.1到0.9之间的81个阈值(步长0.01)
    thresholds = np.linspace(0.1, 0.9, 81)
    f1_scores = []
    for thresh in thresholds:
        y_pred = (y_proba[:, 1] >= thresh).astype(int)
        f1_scores.append(f1_score(y_true, y_pred))
    
    # 找到最优阈值
    best_idx = np.argmax(f1_scores)
    best_thresh = thresholds[best_idx]
    
    # 用最优阈值计算Precision
    y_pred_best = (y_proba[:, 1] >= best_thresh).astype(int)
    return precision_score(y_true, y_pred_best)

def recall_at_optimal_f1(y_true, y_proba):
    thresholds = np.linspace(0.1, 0.9, 81)
    f1_scores = []
    for thresh in thresholds:
        y_pred = (y_proba[:, 1] >= thresh).astype(int)
        f1_scores.append(f1_score(y_true, y_pred))
    
    best_idx = np.argmax(f1_scores)
    best_thresh = thresholds[best_idx]
    
    y_pred_best = (y_proba[:, 1] >= best_thresh).astype(int)
    return recall_score(y_true, y_pred_best)

def accuracy_at_optimal_f1(y_true, y_proba):
    thresholds = np.linspace(0.1, 0.9, 81)
    f1_scores = []
    for thresh in thresholds:
        y_pred = (y_proba[:, 1] >= thresh).astype(int)
        f1_scores.append(f1_score(y_true, y_pred))
    
    best_idx = np.argmax(f1_scores)
    best_thresh = thresholds[best_idx]
    
    y_pred_best = (y_proba[:, 1] >= best_thresh).astype(int)
    return accuracy_score(y_true, y_pred_best)

3. 配置多指标评估字典

将自定义函数转换为Scikit-learn可识别的scorer,注意设置needs_proba=True,因为我们需要模型输出概率而非直接的类别标签:

scoring_evals = {
    'AUC': 'roc_auc',
    'Precision@OptimalF1': make_scorer(precision_at_optimal_f1, needs_proba=True),
    'Recall@OptimalF1': make_scorer(recall_at_optimal_f1, needs_proba=True),
    'Accuracy@OptimalF1': make_scorer(accuracy_at_optimal_f1, needs_proba=True),
    'F1_Default': make_scorer(f1_score)  # 可选:查看默认阈值(0.5)下的F1分数
}

4. 数据集与网格搜索配置

# 生成示例数据集
X, y = datasets.make_classification(n_samples=100000, n_features=20, n_informative=2, n_redundant=10, random_state=42)
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.99, random_state=42)

# XGBoost超参数搜索空间
params = {
 'min_child_weight': [0.5, 1.0, 3.0, 5.0, 7.0, 10.0],
 'gamma': [0, 0.25, 0.5, 1.0],
 'reg_lambda': [0.1, 1.0, 5.0, 10.0, 50.0, 100.0],
 "max_depth": [2,4,6,10],
 "learning_rate": [0.05,0.1, 0.2, 0.3,0.4],
 "colsample_bytree":[1, .8, .5],
 "subsample": [0.8],
 'n_estimators': [50]
}

# 网格搜索参数设置
max_models = 5
folds = 5

# 初始化XGBoost分类器(避免新版本警告)
xgb_algo = xgb.XGBClassifier(use_label_encoder=False, eval_metric='logloss', random_state=42)

# 初始化新版RandomizedSearchCV
random_search = RandomizedSearchCV(
    estimator=xgb_algo,
    param_distributions=params,
    n_iter=max_models,
    scoring=scoring_evals,
    n_jobs=4,
    cv=folds,
    verbose=1,
    random_state=2018,
    refit='AUC'  # 指定用AUC分数选择最优模型,可根据需求替换为其他指标
)

# 执行网格搜索
random_search.fit(X_train, y_train)

5. 查看结果与测试集评估

# 查看所有候选模型的交叉验证结果
print("交叉验证结果概览:")
for rank, mean_score, params in zip(random_search.cv_results_['rank_test_AUC'], 
                                   random_search.cv_results_['mean_test_AUC'],
                                   random_search.cv_results_['params']):
    print(f"排名: {rank}, 平均AUC: {mean_score:.4f}, 参数: {params}")

# 查看最优模型参数
print("\n最优模型参数:", random_search.best_params_)

# 在测试集上评估最优模型(基于F1最优阈值)
best_model = random_search.best_estimator_
y_proba = best_model.predict_proba(X_test)[:, 1]

# 重新计算测试集上的最优F1阈值
thresholds = np.linspace(0.1, 0.9, 81)
f1_scores = [f1_score(y_test, (y_proba >= thresh).astype(int)) for thresh in thresholds]
best_thresh = thresholds[np.argmax(f1_scores)]

# 输出测试集指标
print(f"\n测试集最优F1阈值: {best_thresh:.4f}")
print(f"测试集Precision@OptimalF1: {precision_score(y_test, (y_proba >= best_thresh).astype(int)):.4f}")
print(f"测试集Recall@OptimalF1: {recall_score(y_test, (y_proba >= best_thresh).astype(int)):.4f}")
print(f"测试集Accuracy@OptimalF1: {accuracy_score(y_test, (y_proba >= best_thresh).astype(int)):.4f}")
print(f"测试集AUC: {roc_auc_score(y_test, y_proba):.4f}")

关键注意点

  • API版本问题:必须使用sklearn.model_selection.RandomizedSearchCV才能支持多指标的字典式scoring。
  • 自定义scorer的needs_proba:因为我们需要模型输出概率来寻找最优阈值,所以必须设置needs_proba=True,否则Scikit-learn会传入类别标签而非概率。
  • XGBoost初始化参数:添加use_label_encoder=False和eval_metric='logloss'是为了适配新版XGBoost的要求,避免不必要的警告。
  • refit参数:多指标网格搜索时必须指定refit参数,告诉Scikit-learn用哪个指标来选择最优模型。

内容的提问来源于stack exchange,提问作者runningbirds

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.29 07:04:43