如何用R的glmnet包获取混淆矩阵、概率图并设置最优截断值?
使用glmnet进行Logistic Lasso回归:混淆矩阵、概率图与最优截断值设置
1. 生成训练集与测试集的预测概率
基于最优lambda值(lambda.min)生成预测概率,注意测试集自变量需与训练集保持一致格式(用model.matrix处理):
# 生成训练集预测概率 train_pred_prob <- predict(cvfit, newx = x, s = "lambda.min", type = "response") # 处理测试集自变量 x_test <- model.matrix(asthma ~ temperature+bmi, data = test)[,-1] # 生成测试集预测概率 test_pred_prob <- predict(cvfit, newx = x_test, s = "lambda.min", type = "response")
2. 绘制预测概率分布图
用ggplot2绘制不同真实类别下的预测概率分布,直观展示模型区分能力:
library(ggplot2) # 整理训练集绘图数据 train_plot_data <- data.frame( true_label = factor(train$asthma, levels = c(0,1), labels = c("非哮喘", "哮喘")), pred_prob = as.vector(train_pred_prob) ) # 绘制密度图 ggplot(train_plot_data, aes(x = pred_prob, fill = true_label)) + geom_density(alpha = 0.5) + labs(x = "预测概率", y = "密度", title = "训练集预测概率分布", fill = "真实类别") + theme_minimal()
3. 寻找同时最大化敏感性和特异性的截断值
通过**约登指数(Youden's J statistic)**确定最优截断值,约登指数为敏感性 + 特异性 - 1,取该指数最大值对应的阈值即可,使用pROC包实现:
library(pROC) # 构建测试集ROC曲线对象 roc_obj <- roc(test$asthma, test_pred_prob) # 计算约登指数最大的截断值 optimal_thresh <- coords(roc_obj, "best", ret = "threshold", best.method = "youden") cat("最优截断值:", optimal_thresh, "\n")
4. 生成对应截断值的混淆矩阵
用最优阈值将预测概率转换为分类标签,再生成混淆矩阵:
# 基于最优阈值生成测试集预测标签 test_pred_label <- ifelse(test_pred_prob >= optimal_thresh, 1, 0) # 基础R生成混淆矩阵 confusion_matrix <- table(真实标签 = test$asthma, 预测标签 = test_pred_label) print(confusion_matrix) # 若需更详细指标(可选,需加载caret包) # library(caret) # confusionMatrix(factor(test_pred_label), factor(test$asthma), positive = "1")
完整整合代码
将上述步骤补充到原有代码中,完整代码如下:
df<-data.frame(date = c(20120324,20120329,20121216,20130216,20130725,20130729,20130930,20131015,20131124,20131225, 20140324,20140530,20140613,20140721,20140630,20150102,20150214,20150312,20150316,20150329), temperature=c(35,36.5,34.3,37.8,39,40,34.5,35.9,35.8,36.1,37,35,36,36.3,37.8,38.1,39.2,34.5,34.9,35.2), bmi=c(20,23,25,27,32,24,35,21,19,29,21,32,21,22,24,25,19,18,25,26), asthma=c(1,1,0,1,0,0,1,1,0,0,1,1,1,0,1,0,0,1,1,0)) set.seed(101) # 选取70%数据作为训练集 sample <- sample.int(n = nrow(df), size = floor(.7*nrow(df)), replace = F) train <- df[sample, ] test <- df[-sample, ] x<-model.matrix(asthma ~ temperature+bmi,data=train)[,-1] y<-as.matrix(train$asthma) library(glmnet) glmmod <- glmnet(x, y, alpha=1, family="binomial") cvfit = cv.glmnet(x, y) coef_cv=coef(cvfit, s = "lambda.min") # --- 新增步骤 --- # 1. 生成预测概率 train_pred_prob <- predict(cvfit, newx = x, s = "lambda.min", type = "response") x_test <- model.matrix(asthma ~ temperature+bmi, data = test)[,-1] test_pred_prob <- predict(cvfit, newx = x_test, s = "lambda.min", type = "response") # 2. 绘制预测概率分布图 library(ggplot2) train_plot_data <- data.frame( true_label = factor(train$asthma, levels = c(0,1), labels = c("非哮喘", "哮喘")), pred_prob = as.vector(train_pred_prob) ) ggplot(train_plot_data, aes(x = pred_prob, fill = true_label)) + geom_density(alpha = 0.5) + labs(x = "预测概率", y = "密度", title = "训练集预测概率分布", fill = "真实类别") + theme_minimal() # 3. 寻找最优截断值 library(pROC) roc_obj <- roc(test$asthma, test_pred_prob) optimal_thresh <- coords(roc_obj, "best", ret = "threshold", best.method = "youden") cat("最优截断值:", optimal_thresh, "\n") # 4. 生成混淆矩阵 test_pred_label <- ifelse(test_pred_prob >= optimal_thresh, 1, 0) confusion_matrix <- table(真实标签 = test$asthma, 预测标签 = test_pred_label) print(confusion_matrix)
内容的提问来源于stack exchange,提问作者Lee
相关产品推荐
相关产品推荐

