You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Python中绘制Logistic Regression与Random Forest的Precision-Recall对比曲线?

问题描述

我想要生成Precision-Recall曲线,在同一张图中对比Logistic Regression和Random Forest,现确认自己创建该对比曲线的步骤是否正确,感谢帮助!

我的代码

from sklearn.preprocessing import MultiLabelBinarizer as mlb
import numpy as np
import pandas as pd
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import confusion_matrix
from sklearn.metrics import classification_report
from sklearn.datasets import make_classification
from sklearn import metrics
from sklearn.linear_model import LogisticRegression
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import precision_recall_curve
from sklearn.metrics import f1_score
from sklearn.metrics import auc
from matplotlib import pyplot

X = df[["DIAGNOSIS_CD_Dummy"]]
y = df[["TEST_RESULT_Dummy"]]
# X = pd.DataFrame(df.iloc[:, -1])
# y = pd.DataFrame(df.iloc[:, :-1])

# raw confusion matrix
df = pd.DataFrame(df, columns=["DIAGNOSIS_CD_Dummy", "TEST_RESULT_Dummy"])
confusion_matrix = pd.crosstab(
    df["TEST_RESULT_Dummy"],
    df["DIAGNOSIS_CD_Dummy"],
    rownames=["Test Result"],
    colnames=["Diagnosis"],
)
print(confusion_matrix)


# Logistic Regression Confusion Matrix
from sklearn.preprocessing import MultiLabelBinarizer as mlb
import numpy as np
import pandas as pd
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import confusion_matrix
from sklearn.metrics import classification_report
from sklearn.datasets import make_classification
from sklearn import metrics


# split into training and test using scikit
from sklearn.model_selection import train_test_split

X_train, X_test, y_train, y_test = train_test_split(
    X, y.values.ravel(), test_size=0.3, random_state=1, stratify=y
)
log_model = LogisticRegression()
log_model.fit(X_train, y_train)

# use logistic regression model to make predictions
y_score = log_model.predict_proba(X_test)[:, 1]

y_pred = log_model.predict(X_test)
y_pred = np.round(y_pred)
confusion_matrix = confusion_matrix(y_test, y_pred)
print("\n")
print(confusion_matrix)
print("\n")
print(classification_report(y_test, y_pred, zero_division=0))

# calculate precision and recall
precision, recall, thresholds = precision_recall_curve(y_test, y_score)

# create precision recall curve
fig, ax = plt.subplots()
ax.plot(recall, precision, color="purple")

# add axis labels to plot
ax.set_title("Precision-Recall Curve")
ax.set_ylabel("Precision")
ax.set_xlabel("Recall")

# display plot
plt.show()

# precision-recall curve
# generate 2 class dataset
X = df[["DIAGNOSIS_CD_Dummy"]]
y = df[["TEST_RESULT_Dummy"]]

# X = pd.DataFrame(df.iloc[:, :-1])
# y = pd.DataFrame(df.iloc[:, -1])

# split into train/test sets
trainX, testX, trainy, testy = train_test_split(
    X, y.values.ravel(), test_size=0.3, random_state=2
)
# fit a model
model = LogisticRegression(solver="lbfgs")
model.fit(trainX, trainy)

# predict probabilities
lr_probs = model.predict_proba(testX)
# probs_rf = model_rf.predict_proba(testX)[:, 1]

# keep probabilities for the positive outcome only
lr_probs = lr_probs[:, 1]

# predict class values
yhat = model.predict(testX)
lr_precision, lr_recall, _ = precision_recall_curve(testy, lr_probs)
lr_f1, lr_auc = f1_score(testy, yhat), auc(lr_recall, lr_precision)

# precision_rf, recall_rf, _ = precision_recall_curve(testy, probs_rf)
# f1_rf, auc_rf = f1_score(testy, yhat), auc(recall_rf, precision_rf)
# auc_rf = auc(recall_rf, precision_rf)


# summarize scores
print("Logistic: f1=%.3f auc=%.3f" % (lr_f1, lr_auc))

# plot the precision-recall curves
no_skill = len(testy[testy == 1]) / len(testy)
pyplot.plot([0, 1], [no_skill, no_skill], linestyle="--", label="No Skill")
pyplot.plot(lr_recall, lr_precision, marker=".", label="Logistic")

plt.plot(lr_precision, lr_recall, label=f"AUC (Logistic Regression) = {lr_auc:.2f}")

# axis labels
pyplot.xlabel("Recall")
pyplot.ylabel("Precision")
# show the legend
pyplot.legend()
# show the plot
pyplot.show()



# Random Forest
model_rf = RandomForestClassifier()
model_rf.fit(trainX, trainy)
# model_rf = RandomForestClassifier().fit(trainX, trainy)

# predict probabilities
lr_probs = model.predict_proba(testX)
probs_rf = model_rf.predict_proba(testX)

# keep probabilities for the positive outcome only
probs_rf = probs_rf[:, 1]

# predict class values
yhat = model.predict(testX)
precision_rf, recall_rf, _ = precision_recall_curve(testy, probs_rf)
f1_rf, auc_rf = f1_score(testy, yhat), auc(recall_rf, precision_rf)
auc_rf = auc(recall_rf, precision_rf)

print("Random Forest: f1=%.3f auc=%.3f" % (f1_rf, auc_rf))

# plot the precision-recall curves
no_skill = len(testy[testy == 1]) / len(testy)
pyplot.plot([0, 1], [no_skill, no_skill], linestyle="--", label="No Skill")
pyplot.plot(lr_recall, lr_precision, marker=".", label="Random Forest")

plt.plot(recall_rf, precision_rf, label=f"AUC (Random Forests) = {auc_rf:.2f}")

# axis labels
pyplot.xlabel("Recall")
pyplot.ylabel("Precision")
# show the legend
pyplot.legend()
# show the plot
pyplot.show()

代码输出

Diagnosis        0    1
Test Result            
0            18385   32
1             1268  165


[[5514   11]
 [ 374   56]]


              precision    recall  f1-score   support

           0       0.94      1.00      0.97      5525
           1       0.84      0.13      0.23       430

    accuracy                           0.94      5955
   macro avg       0.89      0.56      0.60      5955
weighted avg       0.93      0.94      0.91      5955

Logistic: f1=0.193 auc=0.488
Random Forest: f1=0.193 auc=0.488

生成的图像

  • Logistic Regression的Precision-Recall曲线:
    Logistic Regression的Precision-Recall曲线
  • Random Forest的Precision-Recall曲线:
    Random Forest的Precision-Recall曲线

步骤验证与改进建议

你的核心思路方向正确,但当前代码存在几个关键问题,导致无法实现同一张图对比两个模型的目标:

1. 重复划分数据集,对比无意义

你用不同的random_state做了两次训练/测试集划分,导致两个模型在不同的测试集上评估,对比结果不具备参考性。必须保证两个模型使用同一套训练/测试数据。

2. 代码冗余,重复导入库

多次重复导入sklearn相关库(如LogisticRegression、train_test_split),完全可以只在开头导入一次。

3. 未实现同图对比

当前代码分别绘制两个模型的曲线,没有合并到同一个画布中,无法直观对比。

4. 绘图存在轴颠倒错误

在Logistic部分的绘图代码中,错误执行了plt.plot(lr_precision, lr_recall),把x轴(Recall)和y轴(Precision)搞反,导致曲线显示不符合标准定义。

修正后的代码示例

# 统一导入所需库
import numpy as np
import pandas as pd
from sklearn.linear_model import LogisticRegression
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import precision_recall_curve, f1_score, auc, confusion_matrix, classification_report
import matplotlib.pyplot as plt

# 加载数据(假设df已提前加载)
X = df[["DIAGNOSIS_CD_Dummy"]]
y = df["TEST_RESULT_Dummy"].values.ravel()

# 原始混淆矩阵
confusion_matrix_raw = pd.crosstab(
    df["TEST_RESULT_Dummy"],
    df["DIAGNOSIS_CD_Dummy"],
    rownames=["Test Result"],
    colnames=["Diagnosis"],
)
print("原始混淆矩阵:")
print(confusion_matrix_raw)
print("\n")

# 仅做一次数据划分,保证两个模型用同一套数据
X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.3, random_state=1, stratify=y
)

# -------------------------- 训练Logistic Regression --------------------------
log_model = LogisticRegression()
log_model.fit(X_train, y_train)
y_pred_log = log_model.predict(X_test)
y_probs_log = log_model.predict_proba(X_test)[:, 1]

# 输出Logistic的评估指标
print("Logistic Regression混淆矩阵:")
print(confusion_matrix(y_test, y_pred_log))
print("\n分类报告:")
print(classification_report(y_test, y_pred_log, zero_division=0))

# 计算Precision-Recall相关指标
lr_precision, lr_recall, _ = precision_recall_curve(y_test, y_probs_log)
lr_f1 = f1_score(y_test, y_pred_log)
lr_auc = auc(lr_recall, lr_precision)
print(f"\nLogistic Regression: f1={lr_f1:.3f} auc={lr_auc:.3f}")

# -------------------------- 训练Random Forest --------------------------
rf_model = RandomForestClassifier()
rf_model.fit(X_train, y_train)
y_pred_rf = rf_model.predict(X_test)
y_probs_rf = rf_model.predict_proba(X_test)[:, 1]

# 计算Random Forest的Precision-Recall相关指标
rf_precision, rf_recall, _ = precision_recall_curve(y_test, y_probs_rf)
rf_f1 = f1_score(y_test, y_pred_rf)
rf_auc = auc(rf_recall, rf_precision)
print(f"Random Forest: f1={rf_f1:.3f} auc={rf_auc:.3f}")

# -------------------------- 同图绘制两个模型的Precision-Recall曲线 --------------------------
plt.figure(figsize=(8, 6))
# 绘制无技能基线(随机猜测的Precision)
no_skill = len(y_test[y_test == 1]) / len(y_test)
plt.plot([0, 1], [no_skill, no_skill], linestyle="--", label=f"无技能基线 (Precision={no_skill:.2f})")

# 绘制Logistic Regression曲线
plt.plot(lr_recall, lr_precision, marker=".", label=f"Logistic Regression (AUC={lr_auc:.2f})")

# 绘制Random Forest曲线
plt.plot(rf_recall, rf_precision, marker="s", label=f"Random Forest (AUC={rf_auc:.2f})")

# 设置图表属性
plt.title("Precision-Recall曲线对比")
plt.xlabel("召回率 (Recall)")
plt.ylabel("精确率 (Precision)")
plt.legend()
plt.grid(True)
plt.show()

修正后的核心改进点

  • 仅做一次数据划分,确保两个模型在相同的训练/测试集上评估
  • 合并绘图代码,将两条曲线放在同一张图中直观对比
  • 修正了绘图时x轴y轴颠倒的错误
  • 简化冗余代码,统一导入库
  • 补充网格线,提升图表可读性

内容的提问来源于stack exchange,提问作者kora_num14

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.22 01:15:39