You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何基于多性能指标手动同步实现向后特征选择与超参数调优

多指标下嵌套向后特征选择与超参数调优(mlr3实现)

核心优化方向

针对12个特征全子集遍历的低效问题,采用向后逐步特征选择替代全组合遍历:从全特征集开始,每一轮仅评估移除单个特征后的性能,保留最优子集并重复该过程,直到达到终止条件(如性能不再提升、特征数达标)。同时在每一步嵌套多目标超参数调优,确保每个特征子集都使用最优超参数组合评估性能。

完整实现代码

library(mlr3)
library(mlr3tuning)
library(mlr3tuningspaces)
library(parallel)
library(doSNOW)
library(dplyr)

# 1. 初始化任务与学习器
task = tsk("sonar")
target_features = c("V1", "V10", "V11", "V12", "V13", "V14", "V15", "V16", "V17", "V18", "V19", "V2")
current_features = target_features  # 初始为全特征集

# 定义学习器(带调优空间)
learner_glmnet <- lts(lrn("classif.glmnet", predict_type = "prob"))
learner_rpart <- lts(lrn("classif.rpart", predict_type = "prob"))
learners <- list(glmnet = learner_glmnet, rpart = learner_rpart)

# 调优组件:多目标MBO调优
tuner <- tnr("mbo")
resampling <- rsmp("cv", folds = 3)
measures <- msrs(c("classif.sensitivity", "classif.specificity", "classif.auc"))
terminator <- trm("evals", n_evals = 5)

# 存储每一轮的结果
selection_results <- list()

# 2. 向后特征选择循环
# 终止条件:特征数≥3 或 连续2轮性能无提升
min_features = 3
performance_history = c()
no_improve_count = 0

# 启动并行集群(全局创建,避免重复开销)
cl <- makeCluster(2, outfile = "C:/Users/Downloads/output.txt")
registerDoSNOW(cl)

while(length(current_features) > min_features && no_improve_count < 2) {
  # 评估当前特征子集的最优性能
  current_perf_list <- foreach(learner_name = names(learners), .packages = c("mlr3", "mlr3tuning")) %dopar% {
    set.seed(1)
    # 创建当前特征子集的任务
    modified_task <- task$select(current_features)
    # 调优实例
    instance <- ti(
      task = modified_task,
      learner = learners[[learner_name]],
      resampling = resampling,
      measures = measures,
      terminator = terminator
    )
    # 执行调优
    tuner$optimize(instance)
    # 返回最优性能与超参数
    list(
      learner = learner_name,
      features = current_features,
      performance = instance$result[, c("classif.sensitivity", "classif.specificity", "classif.auc")],
      hyperparams = instance$result[, setdiff(colnames(instance$result), c(measures$ids(), "runtime_learners"))]
    )
  }
  
  # 计算当前子集的综合性能(这里采用加权得分,可根据需求调整)
  current_perf <- bind_rows(lapply(current_perf_list, function(x) x$performance)) %>%
    mutate(weighted_score = 0.3*classif.sensitivity + 0.3*classif.specificity + 0.4*classif.auc) %>%
    summarise(avg_weighted = mean(weighted_score)) %>%
    pull(avg_weighted)
  
  performance_history <- c(performance_history, current_perf)
  selection_results[[length(current_features)]] <- list(
    features = current_features,
    performance = current_perf,
    learner_results = current_perf_list
  )
  
  # 评估移除单个特征后的性能
  candidate_perfs <- foreach(feature_to_remove = current_features, .packages = c("mlr3", "mlr3tuning")) %dopar% {
    set.seed(1)
    candidate_features <- setdiff(current_features, feature_to_remove)
    modified_task <- task$select(candidate_features)
    
    # 对每个学习器调优后取平均性能
    learner_perfs <- lapply(learners, function(learner) {
      instance <- ti(
        task = modified_task,
        learner = learner,
        resampling = resampling,
        measures = measures,
        terminator = terminator
      )
      tuner$optimize(instance)
      instance$result[, c("classif.sensitivity", "classif.specificity", "classif.auc")]
    })
    
    # 计算候选子集的加权得分
    bind_rows(learner_perfs) %>%
      mutate(weighted_score = 0.3*classif.sensitivity + 0.3*classif.specificity + 0.4*classif.auc) %>%
      summarise(avg_weighted = mean(weighted_score)) %>%
      pull(avg_weighted)
  }
  
  # 找到移除后性能最优的特征(即移除后得分最高的)
  names(candidate_perfs) <- current_features
  best_candidate_score <- max(unlist(candidate_perfs))
  feature_to_remove <- names(which(candidate_perfs == best_candidate_score))[1]
  
  # 判断性能是否提升
  if(best_candidate_score >= current_perf) {
    current_features <- setdiff(current_features, feature_to_remove)
    no_improve_count <- 0
  } else {
    no_improve_count <- no_improve_count + 1
  }
}

stopCluster(cl)

# 3. 提取最优特征子集与结果
best_idx <- which.max(performance_history)
best_subset_result <- selection_results[[best_idx]]

cat("最优特征子集:", paste(best_subset_result$features, collapse = ", "), "\n")
cat("平均加权性能得分:", best_subset_result$performance, "\n")

关键说明

  1. 向后选择逻辑:从全特征开始,每一轮仅测试移除单个特征后的性能,避免全子集遍历的4096次计算,大幅减少工作量。
  2. 多指标处理:通过自定义加权得分(可根据生态位模型的需求调整权重)将多指标转化为可比较的综合值,也可直接基于帕累托前沿判断子集优劣(需额外实现支配关系判断)。
  3. 并行优化:全局创建一次并行集群,避免循环内重复创建销毁的开销;使用foreach并行处理不同学习器或候选特征子集。
  4. 终止条件:设置了最小特征数和连续性能无提升次数双重终止条件,避免无意义的计算。

内容的提问来源于stack exchange,提问作者Pierre Levoisin

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 11:37:14