You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R语言LightGBM贝叶斯优化报错及AUC异常问题排查

问题:R语言中LightGBM贝叶斯优化报错及最优分数异常问题

问题背景

此前使用随机搜索进行LightGBM超参数调优,现切换为贝叶斯优化以实现更精准搜索,但运行时出现高斯过程拟合报错,同时发现交叉验证中所有迭代的best_model$best_score均为1,明显不符合实际模型性能。

原代码片段

library(rBayesianOptimization)

trainModel = function(data = NULL,
                       trainIndex = NULL,
                       iteration_num = NULL,
                       add.genes = TRUE,
                       case = c(),
                       use.scale = TRUE,
                       using.Gene = FALSE,
                       algo = c()
) {

  train_set <- data[trainIndex, ]
  test_set <- data[-trainIndex, ]

  if (case != 'Response'){

    train_set = train_set[!is.na(train_set$outcome),]
    test_set = test_set[!is.na(test_set$outcome),]

  }
  
  train_set$outcome = as.factor(train_set$outcome)
  test_set$outcome = as.factor(test_set$outcome)
  
  outcome_idx = grep("outcome", colnames(train_set))
  train_x = data.matrix(train_set[, -outcome_idx])
  train_y = train_set[, outcome_idx]
  train_y = as.factor(train_y)
  test_x = data.matrix(test_set[, -outcome_idx])
  test_y = test_set[, outcome_idx]
  test_y = as.factor(test_y)

  train_x_filtered = train_x
  test_x_filtered = test_x

  k = 10
  folds = createFolds(train_y, k = k, list = TRUE, returnTrain = FALSE)

  bounds = list(
    max_depth = c(3L, 12L),
    num_leaves = c(3L, 65L),
    min_data_in_leaf = c(3L, 50L),
    feature_fraction = c(0.1, 0.9),
    bagging_fraction = c(0.1, 0.9),
    bagging_freq = c(0L, 10L),
    lambda_l1 = c(1L, 25L),
    lambda_l2 = c(1L, 40L),
    learning_rate = c(0.005, 0.1),
    min_split_gain = c(0.5, 20),
    nrounds = c(50L, 1400L))
  
  auc_score_lightgbm_bayes = function(max_depth, num_leaves, min_data_in_leaf, feature_fraction, bagging_fraction, bagging_freq, lambda_l1, lambda_l2, learning_rate, min_split_gain, nrounds) {
    max_depth = round(max_depth)
    num_leaves = round(num_leaves)
    min_data_in_leaf = round(min_data_in_leaf)
    bagging_freq = round(bagging_freq)
    nrounds = round(nrounds)
    
    params = list(
      max_depth = max_depth,
      num_leaves = num_leaves,
      min_data_in_leaf = min_data_in_leaf,
      feature_fraction = feature_fraction,
      bagging_fraction = bagging_fraction,
      bagging_freq = bagging_freq,
      lambda_l1 = lambda_l1,
      lambda_l2 = lambda_l2,
      learning_rate = learning_rate,
      min_split_gain = min_split_gain,
      nrounds = nrounds
    )
    
    auc_cv <- rep(0, k)
    for (j in 1:k) {
      
      fold_idx <- folds[[j]]
    
      dtrain <- lightgbm::lgb.Dataset(data = train_x_filtered[-fold_idx,], label = train_y[-fold_idx])
      dtest <- lightgbm::lgb.Dataset(data = train_x_filtered[fold_idx,], label = train_y[fold_idx])
      
      best_model <- lgb.train(
        data = dtrain,
        params = list(
          objective = 'binary',
          metric = 'auc',
          learning_rate = learning_rate,
          num_leaves = num_leaves,
          max_depth = max_depth,
          min_data_in_leaf = min_data_in_leaf,
          feature_fraction = feature_fraction,
          bagging_fraction = bagging_fraction,
          bagging_freq = bagging_freq,
          lambda_l1 = lambda_l1,
          lambda_l2 = lambda_l2,
          min_split_gain = min_split_gain,
          num_threads = 7
        ),
        valids = list(val = dtest),
        nrounds = nrounds,
        early_stopping_rounds = 100,
        verbose = -1
      )
      
      View(best_model$best_score)
      auc_cv[j] = best_model$best_score
    }
    return(list(Score = mean(auc_cv)))
  }
  
  optimization_result = BayesianOptimization(
    FUN = auc_score_lightgbm_bayes,
    bounds = bounds,
    init_points = 20,
    n_iter = 50, 
    acq = "ucb", 
    kappa = 2.576, 
    verbose = -1
  )
    
  best_params = optimization_result$Best_Par
  print(best_params)

  dtrain = lgb.Dataset(data = train_x_filtered, label = train_y)
  dtest = lgb.Dataset(data = test_x_filtered, label = test_y)
  best_model = best_model_lightgbm(dtrain,dtest,best_params)

  train_pred = predict(best_model, train_x_filtered)
  train_roc = roc(train_y, train_pred)
  train_auc = auc(train_roc)
  cat("Train AUC:", train_auc, "\n")

  test_pred = predict(best_model, test_x_filtered)
  test_roc = roc(test_y, test_pred)
  test_auc = auc(test_roc)
  cat("Test AUC:", test_auc, "\n")
  df = data.frame(row.names = rownames(test_x_filtered), pred = test_pred)

  return(list(df = df, best_params = best_params, auc_scores = auc_scores, train_x_filtered = train_x_filtered, test_x_filtered = test_x_filtered ,test_auc = test_auc))

}

报错信息

Error in GP_deviance(beta = row, X = X, Y = Y, nug_thres = nug_thres, :
Infinite values of the Deviance Function,
unable to find optimum parameters
9.
stop("Infinite values of the Deviance Function, \n unable to find optimum parameters \n")
8.
GP_deviance(beta = row, X = X, Y = Y, nug_thres = nug_thres,
corr = corr)
7.
FUN(newX[, i], ...)
6.
apply(X = param_init_ps, MARGIN = 1L, FUN = function(row) GP_deviance(beta = row,
X = X, Y = Y, nug_thres = nug_thres, corr = corr))
5.
GPfit::GP_fit(X = Par_Mat[Rounds_Unique, ], Y = Value_Vec[Rounds_Unique],
corr = kernel, ...)
4.
withVisible(...elt(i))
3.
utils::capture.output({
GP <- GPfit::GP_fit(X = Par_Mat[Rounds_Unique, ], Y = Value_Vec[Rounds_Unique],
corr = kernel, ...)
})
2.
BayesianOptimization(FUN = auc_score_lightgbm_bayes, bounds = bounds,
init_points = 20, n_iter = 50, acq = "ucb", kappa = 2.576,
verbose = -1) at Functions_2.R#1499
1.
trainModel(data = data_newPD, trainIndex = trainIndex, iteration_num = i,
add.genes = add_genes[i], case = "new_PD",
using.Gene = FALSE,  algo = "ALGO1")

问题分析与解决方案

核心问题定位

  1. AUC分数提取错误:best_model$best_score是嵌套列表结构(list(val = list(auc = 实际分数))),直接将其赋值给auc_cv[j]会导致存储的是列表而非数值,后续计算均值时出现异常,甚至被误解析为1。
  2. 高斯过程拟合失败:当所有初始点的分数都为1(异常值),高斯过程无法拟合出有效的响应面,进而触发Infinite values of the Deviance Function报错。

具体修复步骤

  1. 修正AUC分数提取逻辑:
    在交叉验证循环中,将auc_cv[j] = best_model$best_score替换为:

    auc_cv[j] <- best_model$best_score$val$auc
    

    同时移除View(best_model$best_score)(会中断贝叶斯优化的批量执行),改用print(best_model$best_score$val$auc)进行调试。

  2. 检查交叉验证数据分布:
    确认train_y的类别是否平衡,是否存在某折数据中仅包含单一类别(会导致AUC为1)。可以通过以下代码检查每折的类别分布:

    for (j in 1:k) {
      fold_y <- train_y[folds[[j]]]
      print(table(fold_y))
    }
    

    若存在类别单一的折,需调整交叉验证策略,确保createFolds使用分层抽样(默认已支持,但需确保train_y是正确的因子类型,可显式指定:train_y <- factor(train_y, levels = c(0, 1)))。

  3. 调整贝叶斯优化参数:

    • 先减少init_points数量(比如设为5)进行测试,避免大量异常值导致拟合失败。
    • 若仍有问题,可尝试更换获取函数(acq)为"ei"(期望提升),或调整kappa值。
  4. 修正后的核心函数片段:

    auc_score_lightgbm_bayes = function(max_depth, num_leaves, min_data_in_leaf, feature_fraction, bagging_fraction, bagging_freq, lambda_l1, lambda_l2, learning_rate, min_split_gain, nrounds) {
      max_depth = round(max_depth)
      num_leaves = round(num_leaves)
      min_data_in_leaf = round(min_data_in_leaf)
      bagging_freq = round(bagging_freq)
      nrounds = round(nrounds)
      
      auc_cv <- rep(0, k)
      for (j in 1:k) {
        
        fold_idx <- folds[[j]]
      
        dtrain <- lightgbm::lgb.Dataset(data = train_x_filtered[-fold_idx,], label = train_y[-fold_idx])
        dtest <- lightgbm::lgb.Dataset(data = train_x_filtered[fold_idx,], label = train_y[fold_idx])
        
        best_model <- lgb.train(
          data = dtrain,
          params = list(
            objective = 'binary',
            metric = 'auc',
            learning_rate = learning_rate,
            num_leaves = num_leaves,
            max_depth = max_depth,
            min_data_in_leaf = min_data_in_leaf,
            feature_fraction = feature_fraction,
            bagging_fraction = bagging_fraction,
            bagging_freq = bagging_freq,
            lambda_l1 = lambda_l1,
            lambda_l2 = lambda_l2,
            min_split_gain = min_split_gain,
            num_threads = 7
          ),
          valids = list(val = dtest),
          nrounds = nrounds,
          early_stopping_rounds = 100,
          verbose = -1
        )
        
        # 正确提取AUC分数
        current_auc <- best_model$best_score$val$auc
        print(paste("Fold", j, "AUC:", current_auc))
        auc_cv[j] <- current_auc
      }
      return(list(Score = mean(auc_cv)))
    }
    

额外检查项

  • 确认lgb.train的objective = 'binary'与任务匹配(二分类),metric = 'auc'正确生效。
  • 检查train_x_filtered是否存在异常值或缺失值,避免模型轻易拟合出完美分数。

内容的提问来源于stack exchange,提问作者Programming Noob

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 01:22:03