You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何为XGBoost分类器的机器学习结果计算置信区间得分?

为XGBoost分类结果计算置信区间

下面提供三种实用方法,帮你基于现有XGBoost分类结果计算置信区间,所有代码可直接对接你现有的训练流程:


方法1:Bootstrap抽样法

通过多次从训练集有放回采样、训练多个模型,对测试样本的预测概率分布取分位数得到置信区间,逻辑直观且易实现。

import numpy as np
from sklearn.utils import resample
from sklearn.metrics import classification_report, confusion_matrix
import xgboost as xgb

# 复用你原有的模型参数
xgb_params = {
    'base_score': 0.2,
    'booster': 'gbtree',
    'colsample_bylevel': 1,
    'colsample_bynode': 1,
    'colsample_bytree': 1,
    'enable_categorical': False,
    'gamma': 0,
    'gpu_id': -1,
    'importance_type': None,
    'interaction_constraints': '',
    'learning_rate': 0.300000012,
    'max_delta_step': 0,
    'max_depth': 40,
    'min_child_weight': 1,
    'monotone_constraints': '()',
    'n_estimators': 100,
    'n_jobs': 10,
    'num_parallel_tree': 1,
    'objective': 'multi:softprob',
    'predictor': 'auto',
    'random_state': 0,
    'reg_alpha': 0,
    'reg_lambda': 1,
    'scale_pos_weight': None,
    'subsample': 1,
    'tree_method': 'exact',
    'validate_parameters': 1,
    'verbosity': None
}

# 设置Bootstrap迭代次数
n_bootstrap = 100
prob_predictions = []

# 训练多组Bootstrap模型
for _ in range(n_bootstrap):
    X_resampled, y_resampled = resample(X_train_vectors_tfidf, y_train)
    model = xgb.XGBClassifier(**xgb_params)
    model.fit(X_resampled, y_resampled)
    prob_predictions.append(model.predict_proba(X_test_vectors_tfidf))

# 转换为数组并计算95%置信区间
prob_predictions = np.array(prob_predictions)
lower_ci = np.percentile(prob_predictions, 2.5, axis=0)
upper_ci = np.percentile(prob_predictions, 97.5, axis=0)

# 原模型评估
original_model = xgb.XGBClassifier(**xgb_params)
original_model.fit(X_train_vectors_tfidf, y_train)
y_predict = original_model.predict(X_test_vectors_tfidf)
y_prob = original_model.predict_proba(X_test_vectors_tfidf)[:, 1]

print(classification_report(y_test, y_predict))
print('Confusion Matrix:', confusion_matrix(y_test, y_predict))

# 示例:输出第一个样本的类别1概率及置信区间
sample_idx = 0
print(f"样本{sample_idx}类别1的预测概率: {y_prob[sample_idx]:.4f}")
print(f"95%置信区间: [{lower_ci[sample_idx, 1]:.4f}, {upper_ci[sample_idx, 1]:.4f}]")

方法2:利用XGBoost树的预测方差

基于集成模型中每棵树的原始预测值(logit值)计算方差,再通过delta方法转换为概率的置信区间,无需额外训练多组模型。

import numpy as np
import xgboost as xgb
from sklearn.metrics import classification_report, confusion_matrix

# 训练原模型
xgb_model = xgb.XGBClassifier(base_score=0.2, booster='gbtree', colsample_bylevel=1,
              colsample_bynode=1, colsample_bytree=1, enable_categorical=False,
              gamma=0, gpu_id=-1, importance_type=None,
              interaction_constraints='', learning_rate=0.300000012,
              max_delta_step=0, max_depth=40, min_child_weight=1, 
              monotone_constraints='()', n_estimators=100, n_jobs=10,
              num_parallel_tree=1, objective='multi:softprob', predictor='auto',
              random_state=0, reg_alpha=0, reg_lambda=1, scale_pos_weight=None,
              subsample=1, tree_method='exact', validate_parameters=1,
              verbosity=None)
xgb_model.fit(X_train_vectors_tfidf, y_train)

# 获取每棵树的原始logit预测值
def get_tree_predictions(model, X):
    tree_preds = []
    for i in range(model.n_estimators):
        pred = model.get_booster().predict(xgb.DMatrix(X), iteration_range=(i, i+1))
        tree_preds.append(pred)
    return np.array(tree_preds)

tree_preds = get_tree_predictions(xgb_model, X_test_vectors_tfidf)
logit_mean = np.mean(tree_preds, axis=0)
logit_var = np.var(tree_preds, axis=0)

# 将logit均值转换为概率(与predict_proba结果一致)
def softmax(x):
    exp_x = np.exp(x - np.max(x, axis=1, keepdims=True))
    return exp_x / np.sum(exp_x, axis=1, keepdims=True)

y_prob = softmax(logit_mean)[:, 1]

# 计算95%置信区间(基于正态分布近似)
z_score = 1.96
prob_var = y_prob * (1 - y_prob) * logit_var[:, 1]
prob_std = np.sqrt(prob_var)
lower_ci = np.clip(y_prob - z_score * prob_std, 0, 1)
upper_ci = np.clip(y_prob + z_score * prob_std, 0, 1)

# 输出评估结果
y_predict = xgb_model.predict(X_test_vectors_tfidf)
print(classification_report(y_test, y_predict))
print('Confusion Matrix:', confusion_matrix(y_test, y_predict))

# 示例输出
sample_idx = 0
print(f"样本{sample_idx}类别1的预测概率: {y_prob[sample_idx]:.4f}")
print(f"95%置信区间: [{lower_ci[sample_idx]:.4f}, {upper_ci[sample_idx]:.4f}]")

方法3:贝叶斯XGBoost(结合调优与Bootstrap)

通过贝叶斯优化找到最优模型参数,再用Bootstrap生成置信区间,兼顾模型性能与置信度准确性。

import numpy as np
from bayes_opt import BayesianOptimization
import xgboost as xgb
from sklearn.metrics import classification_report, confusion_matrix
from sklearn.utils import resample

# 定义贝叶斯调优的目标函数
def xgb_cv(max_depth, learning_rate, n_estimators, gamma, min_child_weight):
    params = {
        'max_depth': int(max_depth),
        'learning_rate': learning_rate,
        'n_estimators': int(n_estimators),
        'gamma': gamma,
        'min_child_weight': min_child_weight,
        'objective': 'multi:softprob',
        'random_state': 0
    }
    X_resampled, y_resampled = resample(X_train_vectors_tfidf, y_train)
    model = xgb.XGBClassifier(**params)
    model.fit(X_resampled, y_resampled)
    return -model.score(X_resampled, y_resampled)  # 最小化负准确率

# 参数搜索范围
pbounds = {
    'max_depth': (3, 50),
    'learning_rate': (0.01, 0.3),
    'n_estimators': (50, 200),
    'gamma': (0, 5),
    'min_child_weight': (1, 10)
}

# 执行贝叶斯优化
optimizer = BayesianOptimization(f=xgb_cv, pbounds=pbounds, random_state=1)
optimizer.maximize(init_points=5, n_iter=10)

# 提取最优参数并训练Bootstrap模型
best_params = optimizer.max['params']
best_params.update({
    'max_depth': int(best_params['max_depth']),
    'n_estimators': int(best_params['n_estimators']),
    'objective': 'multi:softprob',
    'random_state': 0
})

n_bootstrap = 50
prob_predictions = []
for _ in range(n_bootstrap):
    X_resampled, y_resampled = resample(X_train_vectors_tfidf, y_train)
    model = xgb.XGBClassifier(**best_params)
    model.fit(X_resampled, y_resampled)
    prob_predictions.append(model.predict_proba(X_test_vectors_tfidf))

# 计算置信区间
prob_predictions = np.array(prob_predictions)
lower_ci = np.percentile(prob_predictions, 2.5, axis=0)
upper_ci = np.percentile(prob_predictions, 97.5, axis=0)

# 训练最终模型并评估
final_model = xgb.XGBClassifier(**best_params)
final_model.fit(X_train_vectors_tfidf, y_train)
y_predict = final_model.predict(X_test_vectors_tfidf)
y_prob = final_model.predict_proba(X_test_vectors_tfidf)[:, 1]

print(classification_report(y_test, y_predict))
print('Confusion Matrix:', confusion_matrix(y_test, y_predict))

# 示例输出
sample_idx = 0
print(f"样本{sample_idx}类别1的预测概率: {y_prob[sample_idx]:.4f}")
print(f"95%置信区间: [{lower_ci[sample_idx,1]:.4f}, {upper_ci[sample_idx,1]:.4f}]")

内容的提问来源于stack exchange,提问作者AAA

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.06 02:16:14