You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

GPBoost模型与rBayesianOptimization贝叶斯优化对接问题求助

GPBoost与rBayesianOptimization对接问题:返回值始终为0的解决方法

问题描述

我正在为GPBoost模型构建参数网格,原本采用网格搜索调参,现在尝试用rBayesianOptimization包实现贝叶斯优化,但二者难以直接对接。参数搜索可正常启动,但返回值始终为0,求指导如何正确完成二者对接,相关代码如下:

# 数据划分
Index <- createDataPartition(y = mean_tl_scale$fb_mean_tl_mm, p = 0.75, list = FALSE)
LM_train_tl_mean <- mean_tl_scale[Index, ]
LM_test_tl_mean <- mean_tl_scale[-Index, ]

# 转换为矩阵
LM_matrix_train <- data.matrix(LM_train_tl_mean)
LM_matrix_test <- data.matrix(LM_test_tl_mean)

# 提取特征数据
features_train <- LM_matrix_train[, !colnames(LM_matrix_train) %in% c("year")]
features_test <- LM_matrix_test[, !colnames(LM_matrix_test) %in% c("year")]
colnames <- colnames(features_test)[-3]

# 训练GP模型
gp_model <- GPModel(likelihood = "gaussian", cov_function = "exponential", group_data = LM_matrix_train[, c("year")])
boost_data <- gpb.Dataset(data = features_train[, colnames], label = features_train[, "fb_mean_tl_mm"])
gpb_boost <- gpb.Dataset.construct(boost_data)

# 定义贝叶斯优化参数边界
bounds <- list(
  learning_rate = c(0.05, 0.15),
  max_depth = c(5L, 7L),
  min_child_weight = c(5L, 7L),
  subsample = c(0.3, 0.5),
  colsample_bytree = c(0.5, 0.9),
  num_iterations = c(800L, 1000L),
  lambda_l2 = c(0, 5)
)

# 定义优化函数
opt_func <- function(learning_rate, max_depth, min_child_weight, subsample, colsample_bytree, num_iterations, lambda_l2) {
  params <- list(
    objective = "regression",
    learning_rate = learning_rate,
    max_depth = max_depth,
    min_child_weight = min_child_weight,
    subsample = subsample,
    colsample_bytree = colsample_bytree,
    num_iterations = num_iterations,
    lambda_l2 = lambda_l2
  )

  cv_result <- gpb.cv(
    params = params,
    data = gpb_boost,
    gp_model = gp_model,
    nrounds = 500,
    nfold = 5,
    verbose = 0,
    eval = "rmse"
  )
  
  return(list(Score = -min(cv_result$evaluation_log$test_rmse_mean), Pred = NULL))
}

# 运行贝叶斯优化
set.seed(68)
opt_result <- BayesianOptimization(
  FUN = opt_func,
  bounds = bounds,
  init_points = 10,
  n_iter = 20,
  acq = "ei",
  kappa = 2.576,
  eps = 0.0,
  verbose = TRUE
)

# 打印最优参数
opt_result

# 使用最优参数测试模型
best_params <- list(
  objective = "regression",
  learning_rate = opt_result$Best_Par["learning_rate"],
  max_depth = opt_result$Best_Par["max_depth"],
  min_child_weight = opt_result$Best_Par["min_child_weight"],
  subsample = opt_result$Best_Par["subsample"],
  colsample_bytree = opt_result$Best_Par["colsample_bytree"],
  num_iterations = opt_result$Best_Par["num_iterations"],
  lambda_l2 = opt_result$Best_Par["lambda_l2"]
)

gp_model <- GPModel(group_data = LM_matrix_train[, c("year")], likelihood = "gaussian", cov_function = "exponential")
gpboost_model <- gpboost(
  data = boost_data,
  gp_model = gp_model,
  params = best_params,
  verbose = 1
)

问题根源与修正方案

核心问题点

  1. 参数冲突:gpb.cv中同时设置了params$num_iterations和nrounds=500,GPBoost会优先使用nrounds,导致你定义的800-1000次迭代完全失效,CV结果异常。
  2. GPModel复用错误:全局gp_model基于全训练集分组数据,但CV每个折的训练数据分组不同,直接复用会导致模型计算错误。
  3. 整数参数类型不匹配:rBayesianOptimization会将整数边界参数转为浮点数(如max_depth=5.0),但GPBoost要求这类参数为整数类型。
  4. 原生gpb.cv对GPModel支持有限:原生交叉验证函数无法自动适配GPModel的分组数据更新,导致结果异常。

修正后的完整代码

library(caret)
library(gpboost)
library(rBayesianOptimization)

# 数据划分
Index <- createDataPartition(y = mean_tl_scale$fb_mean_tl_mm, p = 0.75, list = FALSE)
LM_train_tl_mean <- mean_tl_scale[Index, ]
LM_test_tl_mean <- mean_tl_scale[-Index, ]

# 转换为矩阵
LM_matrix_train <- data.matrix(LM_train_tl_mean)
LM_matrix_test <- data.matrix(LM_test_tl_mean)

# 提取特征数据
features_train <- LM_matrix_train[, !colnames(LM_matrix_train) %in% c("year")]
features_test <- LM_matrix_test[, !colnames(LM_matrix_test) %in% c("year")]
colnames <- colnames(features_test)[-3]

# 构建数据集(无需提前construct,留给CV自动处理)
boost_data <- gpb.Dataset(data = features_train[, colnames], label = features_train[, "fb_mean_tl_mm"])

# 定义贝叶斯优化参数边界
bounds <- list(
  learning_rate = c(0.05, 0.15),
  max_depth = c(5L, 7L),
  min_child_weight = c(5L, 7L),
  subsample = c(0.3, 0.5),
  colsample_bytree = c(0.5, 0.9),
  num_iterations = c(800L, 1000L),
  lambda_l2 = c(0, 5)
)

# 定义优化函数
opt_func <- function(learning_rate, max_depth, min_child_weight, subsample, colsample_bytree, num_iterations, lambda_l2) {
  # 强制转换整数参数为整数类型
  max_depth <- as.integer(max_depth)
  min_child_weight <- as.integer(min_child_weight)
  num_iterations <- as.integer(num_iterations)
  
  params <- list(
    objective = "regression",
    learning_rate = learning_rate,
    max_depth = max_depth,
    min_child_weight = min_child_weight,
    subsample = subsample,
    colsample_bytree = colsample_bytree,
    lambda_l2 = lambda_l2,
    verbose = 0
  )
  
  # 自定义CV逻辑,适配GPModel的分组数据更新
  custom_cv <- function(data, params, nfold = 5) {
    folds <- createFolds(data$get_label(), k = nfold, list = TRUE)
    rmse_scores <- numeric(nfold)
    
    for (i in 1:nfold) {
      train_idx <- unlist(folds[-i])
      val_idx <- unlist(folds[i])
      
      # 提取当前折的训练/验证数据及分组信息
      train_data <- gpb.Dataset.create.valid(data, train_idx)
      val_data <- gpb.Dataset.create.valid(data, val_idx)
      train_group <- LM_matrix_train[train_idx, "year"]
      
      # 为当前折重新初始化GPModel
      gp_model <- GPModel(likelihood = "gaussian", cov_function = "exponential", group_data = train_group)
      
      # 训练模型并记录验证集RMSE
      model <- gpboost(
        data = train_data,
        gp_model = gp_model,
        params = params,
        nrounds = num_iterations,
        valids = list(val = val_data),
        eval = "rmse",
        verbose = 0
      )
      
      rmse_scores[i] <- min(model$record_evals$val$rmse$eval)
    }
    
    # 返回5折平均RMSE
    mean(rmse_scores)
  }
  
  # 运行自定义CV
  cv_rmse <- custom_cv(data = boost_data, params = params, nfold = 5)
  
  # 返回负RMSE(贝叶斯优化默认最大化目标函数)
  return(list(Score = -cv_rmse, Pred = NULL))
}

# 运行贝叶斯优化
set.seed(68)
opt_result <- BayesianOptimization(
  FUN = opt_func,
  bounds = bounds,
  init_points = 10,
  n_iter = 20,
  acq = "ei",
  kappa = 2.576,
  eps = 0.0,
  verbose = TRUE
)

# 查看最优参数
print(opt_result)

# 使用最优参数训练最终模型
best_params <- list(
  objective = "regression",
  learning_rate = opt_result$Best_Par["learning_rate"],
  max_depth = as.integer(opt_result$Best_Par["max_depth"]),
  min_child_weight = as.integer(opt_result$Best_Par["min_child_weight"]),
  subsample = opt_result$Best_Par["subsample"],
  colsample_bytree = opt_result$Best_Par["colsample_bytree"],
  lambda_l2 = opt_result$Best_Par["lambda_l2"],
  verbose = 1
)

# 初始化最终GPModel
final_gp_model <- GPModel(group_data = LM_matrix_train[, "year"], likelihood = "gaussian", cov_function = "exponential")

# 训练最终模型
gpboost_model <- gpboost(
  data = boost_data,
  gp_model = final_gp_model,
  params = best_params,
  nrounds = as.integer(opt_result$Best_Par["num_iterations"])
)

关键修正说明

  • 移除参数冲突:将num_iterations作为gpboost的nrounds参数传入,避免与CV参数冲突。
  • 动态初始化GPModel:在CV的每个折内重新创建GPModel,确保分组数据与当前训练折匹配。
  • 整数参数转换:强制将max_depth等参数转为整数,符合GPBoost的参数要求。
  • 自定义CV逻辑:替代原生gpb.cv,解决其对GPModel的适配问题,提升调参可靠性。

内容的提问来源于stack exchange,提问作者Rocco Boyd

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 09:54:51