You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用R的glmnet包获取混淆矩阵、概率图并设置最优截断值?

使用glmnet进行Logistic Lasso回归:混淆矩阵、概率图与最优截断值设置

1. 生成训练集与测试集的预测概率

基于最优lambda值(lambda.min)生成预测概率,注意测试集自变量需与训练集保持一致格式(用model.matrix处理):

# 生成训练集预测概率
train_pred_prob <- predict(cvfit, newx = x, s = "lambda.min", type = "response")

# 处理测试集自变量
x_test <- model.matrix(asthma ~ temperature+bmi, data = test)[,-1]
# 生成测试集预测概率
test_pred_prob <- predict(cvfit, newx = x_test, s = "lambda.min", type = "response")

2. 绘制预测概率分布图

用ggplot2绘制不同真实类别下的预测概率分布,直观展示模型区分能力:

library(ggplot2)

# 整理训练集绘图数据
train_plot_data <- data.frame(
  true_label = factor(train$asthma, levels = c(0,1), labels = c("非哮喘", "哮喘")),
  pred_prob = as.vector(train_pred_prob)
)

# 绘制密度图
ggplot(train_plot_data, aes(x = pred_prob, fill = true_label)) +
  geom_density(alpha = 0.5) +
  labs(x = "预测概率", y = "密度", title = "训练集预测概率分布", fill = "真实类别") +
  theme_minimal()

3. 寻找同时最大化敏感性和特异性的截断值

通过**约登指数(Youden's J statistic)**确定最优截断值,约登指数为敏感性 + 特异性 - 1,取该指数最大值对应的阈值即可,使用pROC包实现:

library(pROC)

# 构建测试集ROC曲线对象
roc_obj <- roc(test$asthma, test_pred_prob)

# 计算约登指数最大的截断值
optimal_thresh <- coords(roc_obj, "best", ret = "threshold", best.method = "youden")
cat("最优截断值:", optimal_thresh, "\n")

4. 生成对应截断值的混淆矩阵

用最优阈值将预测概率转换为分类标签,再生成混淆矩阵:

# 基于最优阈值生成测试集预测标签
test_pred_label <- ifelse(test_pred_prob >= optimal_thresh, 1, 0)

# 基础R生成混淆矩阵
confusion_matrix <- table(真实标签 = test$asthma, 预测标签 = test_pred_label)
print(confusion_matrix)

# 若需更详细指标(可选,需加载caret包)
# library(caret)
# confusionMatrix(factor(test_pred_label), factor(test$asthma), positive = "1")

完整整合代码

将上述步骤补充到原有代码中,完整代码如下:

df<-data.frame(date = c(20120324,20120329,20121216,20130216,20130725,20130729,20130930,20131015,20131124,20131225,
                        20140324,20140530,20140613,20140721,20140630,20150102,20150214,20150312,20150316,20150329),
               temperature=c(35,36.5,34.3,37.8,39,40,34.5,35.9,35.8,36.1,37,35,36,36.3,37.8,38.1,39.2,34.5,34.9,35.2),
               bmi=c(20,23,25,27,32,24,35,21,19,29,21,32,21,22,24,25,19,18,25,26),
               asthma=c(1,1,0,1,0,0,1,1,0,0,1,1,1,0,1,0,0,1,1,0))

set.seed(101)
# 选取70%数据作为训练集
sample <- sample.int(n = nrow(df), size = floor(.7*nrow(df)), replace = F)
train <- df[sample, ]
test  <- df[-sample, ]

x<-model.matrix(asthma ~ temperature+bmi,data=train)[,-1]
y<-as.matrix(train$asthma)

library(glmnet)
glmmod <- glmnet(x, y, alpha=1, family="binomial")
cvfit = cv.glmnet(x, y)
coef_cv=coef(cvfit, s = "lambda.min")

# --- 新增步骤 ---
# 1. 生成预测概率
train_pred_prob <- predict(cvfit, newx = x, s = "lambda.min", type = "response")
x_test <- model.matrix(asthma ~ temperature+bmi, data = test)[,-1]
test_pred_prob <- predict(cvfit, newx = x_test, s = "lambda.min", type = "response")

# 2. 绘制预测概率分布图
library(ggplot2)
train_plot_data <- data.frame(
  true_label = factor(train$asthma, levels = c(0,1), labels = c("非哮喘", "哮喘")),
  pred_prob = as.vector(train_pred_prob)
)
ggplot(train_plot_data, aes(x = pred_prob, fill = true_label)) +
  geom_density(alpha = 0.5) +
  labs(x = "预测概率", y = "密度", title = "训练集预测概率分布", fill = "真实类别") +
  theme_minimal()

# 3. 寻找最优截断值
library(pROC)
roc_obj <- roc(test$asthma, test_pred_prob)
optimal_thresh <- coords(roc_obj, "best", ret = "threshold", best.method = "youden")
cat("最优截断值:", optimal_thresh, "\n")

# 4. 生成混淆矩阵
test_pred_label <- ifelse(test_pred_prob >= optimal_thresh, 1, 0)
confusion_matrix <- table(真实标签 = test$asthma, 预测标签 = test_pred_label)
print(confusion_matrix)

内容的提问来源于stack exchange,提问作者Lee

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.25 21:57:41