如何在Python中绘制Logistic Regression与Random Forest的Precision-Recall对比曲线?
问题描述
我想要生成Precision-Recall曲线,在同一张图中对比Logistic Regression和Random Forest,现确认自己创建该对比曲线的步骤是否正确,感谢帮助!
我的代码
from sklearn.preprocessing import MultiLabelBinarizer as mlb import numpy as np import pandas as pd from sklearn.linear_model import LogisticRegression from sklearn.metrics import confusion_matrix from sklearn.metrics import classification_report from sklearn.datasets import make_classification from sklearn import metrics from sklearn.linear_model import LogisticRegression from sklearn.ensemble import RandomForestClassifier from sklearn.model_selection import train_test_split from sklearn.metrics import precision_recall_curve from sklearn.metrics import f1_score from sklearn.metrics import auc from matplotlib import pyplot X = df[["DIAGNOSIS_CD_Dummy"]] y = df[["TEST_RESULT_Dummy"]] # X = pd.DataFrame(df.iloc[:, -1]) # y = pd.DataFrame(df.iloc[:, :-1]) # raw confusion matrix df = pd.DataFrame(df, columns=["DIAGNOSIS_CD_Dummy", "TEST_RESULT_Dummy"]) confusion_matrix = pd.crosstab( df["TEST_RESULT_Dummy"], df["DIAGNOSIS_CD_Dummy"], rownames=["Test Result"], colnames=["Diagnosis"], ) print(confusion_matrix) # Logistic Regression Confusion Matrix from sklearn.preprocessing import MultiLabelBinarizer as mlb import numpy as np import pandas as pd from sklearn.linear_model import LogisticRegression from sklearn.metrics import confusion_matrix from sklearn.metrics import classification_report from sklearn.datasets import make_classification from sklearn import metrics # split into training and test using scikit from sklearn.model_selection import train_test_split X_train, X_test, y_train, y_test = train_test_split( X, y.values.ravel(), test_size=0.3, random_state=1, stratify=y ) log_model = LogisticRegression() log_model.fit(X_train, y_train) # use logistic regression model to make predictions y_score = log_model.predict_proba(X_test)[:, 1] y_pred = log_model.predict(X_test) y_pred = np.round(y_pred) confusion_matrix = confusion_matrix(y_test, y_pred) print("\n") print(confusion_matrix) print("\n") print(classification_report(y_test, y_pred, zero_division=0)) # calculate precision and recall precision, recall, thresholds = precision_recall_curve(y_test, y_score) # create precision recall curve fig, ax = plt.subplots() ax.plot(recall, precision, color="purple") # add axis labels to plot ax.set_title("Precision-Recall Curve") ax.set_ylabel("Precision") ax.set_xlabel("Recall") # display plot plt.show() # precision-recall curve # generate 2 class dataset X = df[["DIAGNOSIS_CD_Dummy"]] y = df[["TEST_RESULT_Dummy"]] # X = pd.DataFrame(df.iloc[:, :-1]) # y = pd.DataFrame(df.iloc[:, -1]) # split into train/test sets trainX, testX, trainy, testy = train_test_split( X, y.values.ravel(), test_size=0.3, random_state=2 ) # fit a model model = LogisticRegression(solver="lbfgs") model.fit(trainX, trainy) # predict probabilities lr_probs = model.predict_proba(testX) # probs_rf = model_rf.predict_proba(testX)[:, 1] # keep probabilities for the positive outcome only lr_probs = lr_probs[:, 1] # predict class values yhat = model.predict(testX) lr_precision, lr_recall, _ = precision_recall_curve(testy, lr_probs) lr_f1, lr_auc = f1_score(testy, yhat), auc(lr_recall, lr_precision) # precision_rf, recall_rf, _ = precision_recall_curve(testy, probs_rf) # f1_rf, auc_rf = f1_score(testy, yhat), auc(recall_rf, precision_rf) # auc_rf = auc(recall_rf, precision_rf) # summarize scores print("Logistic: f1=%.3f auc=%.3f" % (lr_f1, lr_auc)) # plot the precision-recall curves no_skill = len(testy[testy == 1]) / len(testy) pyplot.plot([0, 1], [no_skill, no_skill], linestyle="--", label="No Skill") pyplot.plot(lr_recall, lr_precision, marker=".", label="Logistic") plt.plot(lr_precision, lr_recall, label=f"AUC (Logistic Regression) = {lr_auc:.2f}") # axis labels pyplot.xlabel("Recall") pyplot.ylabel("Precision") # show the legend pyplot.legend() # show the plot pyplot.show() # Random Forest model_rf = RandomForestClassifier() model_rf.fit(trainX, trainy) # model_rf = RandomForestClassifier().fit(trainX, trainy) # predict probabilities lr_probs = model.predict_proba(testX) probs_rf = model_rf.predict_proba(testX) # keep probabilities for the positive outcome only probs_rf = probs_rf[:, 1] # predict class values yhat = model.predict(testX) precision_rf, recall_rf, _ = precision_recall_curve(testy, probs_rf) f1_rf, auc_rf = f1_score(testy, yhat), auc(recall_rf, precision_rf) auc_rf = auc(recall_rf, precision_rf) print("Random Forest: f1=%.3f auc=%.3f" % (f1_rf, auc_rf)) # plot the precision-recall curves no_skill = len(testy[testy == 1]) / len(testy) pyplot.plot([0, 1], [no_skill, no_skill], linestyle="--", label="No Skill") pyplot.plot(lr_recall, lr_precision, marker=".", label="Random Forest") plt.plot(recall_rf, precision_rf, label=f"AUC (Random Forests) = {auc_rf:.2f}") # axis labels pyplot.xlabel("Recall") pyplot.ylabel("Precision") # show the legend pyplot.legend() # show the plot pyplot.show()
代码输出
Diagnosis 0 1 Test Result 0 18385 32 1 1268 165 [[5514 11] [ 374 56]] precision recall f1-score support 0 0.94 1.00 0.97 5525 1 0.84 0.13 0.23 430 accuracy 0.94 5955 macro avg 0.89 0.56 0.60 5955 weighted avg 0.93 0.94 0.91 5955 Logistic: f1=0.193 auc=0.488 Random Forest: f1=0.193 auc=0.488
生成的图像
- Logistic Regression的Precision-Recall曲线:

- Random Forest的Precision-Recall曲线:

步骤验证与改进建议
你的核心思路方向正确,但当前代码存在几个关键问题,导致无法实现同一张图对比两个模型的目标:
1. 重复划分数据集,对比无意义
你用不同的random_state做了两次训练/测试集划分,导致两个模型在不同的测试集上评估,对比结果不具备参考性。必须保证两个模型使用同一套训练/测试数据。
2. 代码冗余,重复导入库
多次重复导入sklearn相关库(如LogisticRegression、train_test_split),完全可以只在开头导入一次。
3. 未实现同图对比
当前代码分别绘制两个模型的曲线,没有合并到同一个画布中,无法直观对比。
4. 绘图存在轴颠倒错误
在Logistic部分的绘图代码中,错误执行了plt.plot(lr_precision, lr_recall),把x轴(Recall)和y轴(Precision)搞反,导致曲线显示不符合标准定义。
修正后的代码示例
# 统一导入所需库 import numpy as np import pandas as pd from sklearn.linear_model import LogisticRegression from sklearn.ensemble import RandomForestClassifier from sklearn.model_selection import train_test_split from sklearn.metrics import precision_recall_curve, f1_score, auc, confusion_matrix, classification_report import matplotlib.pyplot as plt # 加载数据(假设df已提前加载) X = df[["DIAGNOSIS_CD_Dummy"]] y = df["TEST_RESULT_Dummy"].values.ravel() # 原始混淆矩阵 confusion_matrix_raw = pd.crosstab( df["TEST_RESULT_Dummy"], df["DIAGNOSIS_CD_Dummy"], rownames=["Test Result"], colnames=["Diagnosis"], ) print("原始混淆矩阵:") print(confusion_matrix_raw) print("\n") # 仅做一次数据划分,保证两个模型用同一套数据 X_train, X_test, y_train, y_test = train_test_split( X, y, test_size=0.3, random_state=1, stratify=y ) # -------------------------- 训练Logistic Regression -------------------------- log_model = LogisticRegression() log_model.fit(X_train, y_train) y_pred_log = log_model.predict(X_test) y_probs_log = log_model.predict_proba(X_test)[:, 1] # 输出Logistic的评估指标 print("Logistic Regression混淆矩阵:") print(confusion_matrix(y_test, y_pred_log)) print("\n分类报告:") print(classification_report(y_test, y_pred_log, zero_division=0)) # 计算Precision-Recall相关指标 lr_precision, lr_recall, _ = precision_recall_curve(y_test, y_probs_log) lr_f1 = f1_score(y_test, y_pred_log) lr_auc = auc(lr_recall, lr_precision) print(f"\nLogistic Regression: f1={lr_f1:.3f} auc={lr_auc:.3f}") # -------------------------- 训练Random Forest -------------------------- rf_model = RandomForestClassifier() rf_model.fit(X_train, y_train) y_pred_rf = rf_model.predict(X_test) y_probs_rf = rf_model.predict_proba(X_test)[:, 1] # 计算Random Forest的Precision-Recall相关指标 rf_precision, rf_recall, _ = precision_recall_curve(y_test, y_probs_rf) rf_f1 = f1_score(y_test, y_pred_rf) rf_auc = auc(rf_recall, rf_precision) print(f"Random Forest: f1={rf_f1:.3f} auc={rf_auc:.3f}") # -------------------------- 同图绘制两个模型的Precision-Recall曲线 -------------------------- plt.figure(figsize=(8, 6)) # 绘制无技能基线(随机猜测的Precision) no_skill = len(y_test[y_test == 1]) / len(y_test) plt.plot([0, 1], [no_skill, no_skill], linestyle="--", label=f"无技能基线 (Precision={no_skill:.2f})") # 绘制Logistic Regression曲线 plt.plot(lr_recall, lr_precision, marker=".", label=f"Logistic Regression (AUC={lr_auc:.2f})") # 绘制Random Forest曲线 plt.plot(rf_recall, rf_precision, marker="s", label=f"Random Forest (AUC={rf_auc:.2f})") # 设置图表属性 plt.title("Precision-Recall曲线对比") plt.xlabel("召回率 (Recall)") plt.ylabel("精确率 (Precision)") plt.legend() plt.grid(True) plt.show()
修正后的核心改进点
- 仅做一次数据划分,确保两个模型在相同的训练/测试集上评估
- 合并绘图代码,将两条曲线放在同一张图中直观对比
- 修正了绘图时x轴y轴颠倒的错误
- 简化冗余代码,统一导入库
- 补充网格线,提升图表可读性
内容的提问来源于stack exchange,提问作者kora_num14
相关产品推荐
相关产品推荐

