GPBoost模型与rBayesianOptimization贝叶斯优化对接问题求助
GPBoost与rBayesianOptimization对接问题:返回值始终为0的解决方法
问题描述
我正在为GPBoost模型构建参数网格,原本采用网格搜索调参,现在尝试用rBayesianOptimization包实现贝叶斯优化,但二者难以直接对接。参数搜索可正常启动,但返回值始终为0,求指导如何正确完成二者对接,相关代码如下:
# 数据划分 Index <- createDataPartition(y = mean_tl_scale$fb_mean_tl_mm, p = 0.75, list = FALSE) LM_train_tl_mean <- mean_tl_scale[Index, ] LM_test_tl_mean <- mean_tl_scale[-Index, ] # 转换为矩阵 LM_matrix_train <- data.matrix(LM_train_tl_mean) LM_matrix_test <- data.matrix(LM_test_tl_mean) # 提取特征数据 features_train <- LM_matrix_train[, !colnames(LM_matrix_train) %in% c("year")] features_test <- LM_matrix_test[, !colnames(LM_matrix_test) %in% c("year")] colnames <- colnames(features_test)[-3] # 训练GP模型 gp_model <- GPModel(likelihood = "gaussian", cov_function = "exponential", group_data = LM_matrix_train[, c("year")]) boost_data <- gpb.Dataset(data = features_train[, colnames], label = features_train[, "fb_mean_tl_mm"]) gpb_boost <- gpb.Dataset.construct(boost_data) # 定义贝叶斯优化参数边界 bounds <- list( learning_rate = c(0.05, 0.15), max_depth = c(5L, 7L), min_child_weight = c(5L, 7L), subsample = c(0.3, 0.5), colsample_bytree = c(0.5, 0.9), num_iterations = c(800L, 1000L), lambda_l2 = c(0, 5) ) # 定义优化函数 opt_func <- function(learning_rate, max_depth, min_child_weight, subsample, colsample_bytree, num_iterations, lambda_l2) { params <- list( objective = "regression", learning_rate = learning_rate, max_depth = max_depth, min_child_weight = min_child_weight, subsample = subsample, colsample_bytree = colsample_bytree, num_iterations = num_iterations, lambda_l2 = lambda_l2 ) cv_result <- gpb.cv( params = params, data = gpb_boost, gp_model = gp_model, nrounds = 500, nfold = 5, verbose = 0, eval = "rmse" ) return(list(Score = -min(cv_result$evaluation_log$test_rmse_mean), Pred = NULL)) } # 运行贝叶斯优化 set.seed(68) opt_result <- BayesianOptimization( FUN = opt_func, bounds = bounds, init_points = 10, n_iter = 20, acq = "ei", kappa = 2.576, eps = 0.0, verbose = TRUE ) # 打印最优参数 opt_result # 使用最优参数测试模型 best_params <- list( objective = "regression", learning_rate = opt_result$Best_Par["learning_rate"], max_depth = opt_result$Best_Par["max_depth"], min_child_weight = opt_result$Best_Par["min_child_weight"], subsample = opt_result$Best_Par["subsample"], colsample_bytree = opt_result$Best_Par["colsample_bytree"], num_iterations = opt_result$Best_Par["num_iterations"], lambda_l2 = opt_result$Best_Par["lambda_l2"] ) gp_model <- GPModel(group_data = LM_matrix_train[, c("year")], likelihood = "gaussian", cov_function = "exponential") gpboost_model <- gpboost( data = boost_data, gp_model = gp_model, params = best_params, verbose = 1 )
问题根源与修正方案
核心问题点
- 参数冲突:
gpb.cv中同时设置了params$num_iterations和nrounds=500,GPBoost会优先使用nrounds,导致你定义的800-1000次迭代完全失效,CV结果异常。 - GPModel复用错误:全局
gp_model基于全训练集分组数据,但CV每个折的训练数据分组不同,直接复用会导致模型计算错误。 - 整数参数类型不匹配:rBayesianOptimization会将整数边界参数转为浮点数(如
max_depth=5.0),但GPBoost要求这类参数为整数类型。 - 原生
gpb.cv对GPModel支持有限:原生交叉验证函数无法自动适配GPModel的分组数据更新,导致结果异常。
修正后的完整代码
library(caret) library(gpboost) library(rBayesianOptimization) # 数据划分 Index <- createDataPartition(y = mean_tl_scale$fb_mean_tl_mm, p = 0.75, list = FALSE) LM_train_tl_mean <- mean_tl_scale[Index, ] LM_test_tl_mean <- mean_tl_scale[-Index, ] # 转换为矩阵 LM_matrix_train <- data.matrix(LM_train_tl_mean) LM_matrix_test <- data.matrix(LM_test_tl_mean) # 提取特征数据 features_train <- LM_matrix_train[, !colnames(LM_matrix_train) %in% c("year")] features_test <- LM_matrix_test[, !colnames(LM_matrix_test) %in% c("year")] colnames <- colnames(features_test)[-3] # 构建数据集(无需提前construct,留给CV自动处理) boost_data <- gpb.Dataset(data = features_train[, colnames], label = features_train[, "fb_mean_tl_mm"]) # 定义贝叶斯优化参数边界 bounds <- list( learning_rate = c(0.05, 0.15), max_depth = c(5L, 7L), min_child_weight = c(5L, 7L), subsample = c(0.3, 0.5), colsample_bytree = c(0.5, 0.9), num_iterations = c(800L, 1000L), lambda_l2 = c(0, 5) ) # 定义优化函数 opt_func <- function(learning_rate, max_depth, min_child_weight, subsample, colsample_bytree, num_iterations, lambda_l2) { # 强制转换整数参数为整数类型 max_depth <- as.integer(max_depth) min_child_weight <- as.integer(min_child_weight) num_iterations <- as.integer(num_iterations) params <- list( objective = "regression", learning_rate = learning_rate, max_depth = max_depth, min_child_weight = min_child_weight, subsample = subsample, colsample_bytree = colsample_bytree, lambda_l2 = lambda_l2, verbose = 0 ) # 自定义CV逻辑,适配GPModel的分组数据更新 custom_cv <- function(data, params, nfold = 5) { folds <- createFolds(data$get_label(), k = nfold, list = TRUE) rmse_scores <- numeric(nfold) for (i in 1:nfold) { train_idx <- unlist(folds[-i]) val_idx <- unlist(folds[i]) # 提取当前折的训练/验证数据及分组信息 train_data <- gpb.Dataset.create.valid(data, train_idx) val_data <- gpb.Dataset.create.valid(data, val_idx) train_group <- LM_matrix_train[train_idx, "year"] # 为当前折重新初始化GPModel gp_model <- GPModel(likelihood = "gaussian", cov_function = "exponential", group_data = train_group) # 训练模型并记录验证集RMSE model <- gpboost( data = train_data, gp_model = gp_model, params = params, nrounds = num_iterations, valids = list(val = val_data), eval = "rmse", verbose = 0 ) rmse_scores[i] <- min(model$record_evals$val$rmse$eval) } # 返回5折平均RMSE mean(rmse_scores) } # 运行自定义CV cv_rmse <- custom_cv(data = boost_data, params = params, nfold = 5) # 返回负RMSE(贝叶斯优化默认最大化目标函数) return(list(Score = -cv_rmse, Pred = NULL)) } # 运行贝叶斯优化 set.seed(68) opt_result <- BayesianOptimization( FUN = opt_func, bounds = bounds, init_points = 10, n_iter = 20, acq = "ei", kappa = 2.576, eps = 0.0, verbose = TRUE ) # 查看最优参数 print(opt_result) # 使用最优参数训练最终模型 best_params <- list( objective = "regression", learning_rate = opt_result$Best_Par["learning_rate"], max_depth = as.integer(opt_result$Best_Par["max_depth"]), min_child_weight = as.integer(opt_result$Best_Par["min_child_weight"]), subsample = opt_result$Best_Par["subsample"], colsample_bytree = opt_result$Best_Par["colsample_bytree"], lambda_l2 = opt_result$Best_Par["lambda_l2"], verbose = 1 ) # 初始化最终GPModel final_gp_model <- GPModel(group_data = LM_matrix_train[, "year"], likelihood = "gaussian", cov_function = "exponential") # 训练最终模型 gpboost_model <- gpboost( data = boost_data, gp_model = final_gp_model, params = best_params, nrounds = as.integer(opt_result$Best_Par["num_iterations"]) )
关键修正说明
- 移除参数冲突:将
num_iterations作为gpboost的nrounds参数传入,避免与CV参数冲突。 - 动态初始化GPModel:在CV的每个折内重新创建GPModel,确保分组数据与当前训练折匹配。
- 整数参数转换:强制将
max_depth等参数转为整数,符合GPBoost的参数要求。 - 自定义CV逻辑:替代原生
gpb.cv,解决其对GPModel的适配问题,提升调参可靠性。
内容的提问来源于stack exchange,提问作者Rocco Boyd
相关产品推荐
相关产品推荐

