You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何解决Error: `data`与`reference`需为同水平因子的报错

解决R中confusionMatrix报错:data and reference should be factors with the same levels

运行以下R代码时遇到报错:Error: data and reference should be factors with the same levels。已尝试将预测结果转为numeric类型,也通过union统一了训练集和测试集目标变量的因子水平,但报错仍未解决。

# Load required libraries
library(caret)
library(glmnet)
library(pROC)

# Convert response variable to factor with consistent levels in training and testing data
response_levels <- union(levels(TrainingSet$target), levels(TestingSet$target))
TrainingSet$target <- factor(TrainingSet$target, levels = response_levels)
TestingSet$target <- factor(TestingSet$target, levels = response_levels)

# Create a list of formulas
formulas <- list(
  formula1 = as.formula(paste("target~. ")),
  formula2 = as.formula(paste("target ~ sex + chest.pain.type + fasting.blood.sugar + max.heart.rate + 
              exercise.angina + oldpeak + ST.slope")),
  formula3 = as.formula(paste("target~ cholesterol + sex + resting.bp.s + 
              age + fasting.blood.sugar")),
  formula4 = as.formula(paste("target~ cholesterol + sex + age + fasting.blood.sugar")),
  formula5 = as.formula(paste("target~  max.heart.rate + resting.ecg + oldpeak + ST.slope+
              chest.pain.type + exercise.angina")),
  formula6 = as.formula(paste("target~  max.heart.rate + oldpeak + ST.slope+
              chest.pain.type + exercise.angina"))
)

# Create a list of models
model_list <- list(
  logistic = list(method = "glm", family = "binomial"),
  glmnet = list(method = "glmnet", family = "binomial")
)

# Create an empty list to store the model results
results <- list()
confusion_matrices <- list()

# Loop through the models
for (model in model_list) {
  for (formula in formulas) {

   # Train the model with the current formula
    if (model$method == "glmnet") {
      # For glmnet, specify the alpha values and lambda grid
      model_fit <- train(
        as.formula(formula),
        data = TrainingSet,
        method = model$method,
        trControl = trainControl(method = "cv", number = 10),
        preProcess = c("center", "scale"),
        tuneGrid = expand.grid(alpha = 0:1, lambda = c(0.001, 0.01, 0.1, 1)),
        family = model$family
      )
    } else {
      # For other models, use default tuning parameter grid
      model_fit <- train(
        as.formula(formula),
        data = TrainingSet,
        method = model$method,
        trControl = trainControl(method = "cv", number = 10),
        preProcess = c("center", "scale"),
        family = model$family
      )
    }

    # Make predictions on the testing set
    predicted <- predict(model_fit, newdata = TestingSet)
    predicted <- as.numeric(predicted)

    # Evaluate model performance
    cm <- caret::confusionMatrix(predicted, as.factor(TestingSet$target))
    auc <- pROC::auc(roc(response = TestingSet$target, predictor = as.numeric(predicted) ))
    rmse <- sqrt(mean((predicted - as.numeric(TestingSet$target))^2))
    r2 <- cor(predicted, as.numeric(TestingSet$target))^2

   # Store results in the results list
    results[[paste(model$method, "_", names(formula), sep = "")]] <- list(
      Confusion_Matrix = cm,
      Accuracy = cm$overall["Accuracy"],
      Error_Rate = cm$byClass["Error Rate"],
      Sensitivity = cm$byClass["Sensitivity"],
      AUC = auc,
      RMSE = rmse,
      R2 = r2
    )

    # Store confusion matrix in the confusion_matrices list
    confusion_matrices[[paste(model$method, "_", names(formula), sep = "")]] <- cm$table
  }
}

# Convert results list to data frame
results_df <- do.call(rbind, lapply(results, data.frame, stringsAsFactors = FALSE))

# Print the results
print(results_df)

# Access confusion matrices
print(confusion_matrices)

问题根源与修复方案

1. 核心问题

confusionMatrix要求输入的data(预测值)和reference(真实值)必须是具有完全相同水平的因子类型。当前错误来自两个关键问题:

  • 你将原本是因子的预测结果转成了numeric类型,直接破坏了因子结构
  • 后续把真实值转成as.factor(TestingSet$target),仍与数值型的预测值类型不匹配

2. 具体修复步骤

将代码中预测与评估部分替换为以下内容:

# 生成预测结果(分类模型默认返回因子,无需转numeric)
predicted <- predict(model_fit, newdata = TestingSet)
# 强制统一预测结果的因子水平与测试集目标变量完全一致
predicted <- factor(predicted, levels = levels(TestingSet$target))

# 评估模型性能
cm <- caret::confusionMatrix(predicted, TestingSet$target)
# 仅在计算AUC、RMSE、R2时转成数值型
auc <- pROC::auc(roc(response = TestingSet$target, predictor = as.numeric(predicted)))
rmse <- sqrt(mean((as.numeric(predicted) - as.numeric(TestingSet$target))^2))
r2 <- cor(as.numeric(predicted), as.numeric(TestingSet$target))^2

3. 额外验证点

  • 运行str(TrainingSet$target)和str(TestingSet$target),确认两者都是因子且水平完全一致
  • 如果训练集中存在测试集没有的因子水平,predict()生成的结果会缺失该水平,此时手动指定levels可以补全,避免不匹配

内容的提问来源于stack exchange,提问作者SYED HASEEB UL HASSAN NAQVI

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.23 20:33:08