如何为R语言glmer多级二分类逻辑回归调优阈值以最大化MCC
修复方案
核心问题定位
- 你代码中的MCC计算存在逻辑错误:使用了R中的标量逻辑运算符
&&,该运算符仅返回单个布尔值,导致你统计TP/TN/FP/FN时得到的长度恒为1,最终计算的MCC完全不符合实际值,需要替换为向量级逻辑运算符&,或直接通过混淆矩阵统计分类指标,避免手动计算出错。 - 直接读取
glmer模型内部槽@resp[[".->mu"]]获取预测值的方式兼容性差,建议用官方接口predict(MLBR_L, type = "response")获取训练集预测概率。
修正后完整代码
Df <- tibble::tribble( ~year_week,~Runner,~NewRRI,~Distance, ~HR, ~Gender,~Age,~BMI,~PreviousRRI, "2019-41", "M01" , 0, 5000, 120, "Male", 23, 18, 1, "2019-41", "M02" , 0, 6000, 125,"Female", 36, 20, 0, "2019-41", "M03" , 0, 8000, 130, "Male", 56, 21, 0, "2019-42", "M01" , 0, 5500, 122, "Male", 23, 18, 1, "2019-42", "M02" , 0, 7000, 128,"Female", 36, 20, 0, "2019-42", "M03" , 0, 15000, 132, "Male", 56, 21, 0, "2019-43", "M01" , 1, 3000, 120, "Male", 23, 18, 1, "2019-43", "M02" , 0, 9000, 127,"Female", 36, 20, 0, "2019-43", "M03" , 0, 9500, 131, "Male", 56, 21, 0, "2019-44", "M01" , 0, 15000, 125, "Male", 23, 18, 1, "2019-44", "M02" , 0, 9000, 127,"Female", 36, 20, 0, "2019-44", "M03" , 0, 9500, 131, "Male", 56, 21, 0, ) %>% mutate(Gender = as.factor(Gender), PreviousRRI = as.factor(PreviousRRI), NewRRI = as.factor(NewRRI), Runner = as.factor(Runner)) library(tidyverse) library(tidymodels) library(themis) library(caret) library(lme4) # Split data Df <- Df %>% arrange(year_week) Df_splits <- initial_time_split(Df, prop = 0.8) RunningData_train <- training(Df_splits) RunningData_test <- testing(Df_splits) # Rescale RunningData_train_norm <- preProcess(RunningData_train) RunningData_train <- predict(RunningData_train_norm, RunningData_train) RunningData_test <- predict(RunningData_train_norm, RunningData_test) # MLBLR model MLBR_L <- glmer(NewRRI ~ Distance + HR + Gender + Age + BMI + PreviousRRI + (1|Runner), data = RunningData_train, family = binomial(link = "logit"), control = glmerControl(optimizer = "bobyqa"), nAGQ = 1) # ------------------- 修正的阈值调优部分 ------------------- # 获取训练集预测概率 train_pred_prob <- predict(MLBR_L, type = "response") train_true <- as.integer(as.character(RunningData_train$NewRRI)) # 生成更细的阈值序列 cutoffs <- seq(0.01, 0.99, by = 0.01) mcc_vals <- numeric(length(cutoffs)) for (i in seq_along(cutoffs)) { # 生成预测分类 pred_class <- ifelse(train_pred_prob >= cutoffs[i], 1, 0) # 生成混淆矩阵 conf_mat <- table(True = train_true, Pred = pred_class) # 提取TP/TN/FP/FN,处理空值情况 TP <- ifelse("1" %in% rownames(conf_mat) && "1" %in% colnames(conf_mat), conf_mat["1", "1"], 0) TN <- ifelse("0" %in% rownames(conf_mat) && "0" %in% colnames(conf_mat), conf_mat["0", "0"], 0) FP <- ifelse("0" %in% rownames(conf_mat) && "1" %in% colnames(conf_mat), conf_mat["0", "1"], 0) FN <- ifelse("1" %in% rownames(conf_mat) && "0" %in% colnames(conf_mat), conf_mat["1", "0"], 0) # 计算MCC,分母为0时返回0 denominator <- sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN)) mcc_vals[i] <- ifelse(denominator == 0, 0, (TP * TN - FP * FN) / denominator) } # 绘制MCC曲线 plot(cutoffs, mcc_vals, pch = 19, type = 'b', xlab = "Cutoff Level", ylab = "MCC") # 提取最优阈值 best_cutoff <- cutoffs[which.max(mcc_vals)] cat("最优阈值为:", best_cutoff, ",对应最大MCC为:", max(mcc_vals), "\n") # 用最优阈值在测试集预测 Predictions <- RunningData_test %>% dplyr::mutate(Prediction = predict(MLBR_L, RunningData_test, type = "response"), Prediction = ifelse(Prediction > best_cutoff, 1, 0), Prediction = factor(Prediction, levels = c("0", "1")), NewRRI = factor(NewRRI, levels = c("0", "1"))) # 计算测试集MCC test_mcc <- mcc(Predictions, truth = NewRRI, estimate = Prediction) cat("测试集MCC为:", test_mcc$.estimate, "\n")
补充说明
你提供的示例数据中正例(NewRRI=1)仅有1条,属于极度不平衡的分类场景,如果你的真实数据同样存在类别不平衡问题,建议在模型训练阶段加入类别权重调整或样本重采样操作,避免最优阈值波动过大、泛化性差的问题。
内容的提问来源于stack exchange,提问作者LB.
相关产品推荐
相关产品推荐

