You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R语言recipes包step_cut报错:误判xpr为因子类型

recipes包step_cut二次prep报错的解决方法

问题描述

使用recipes包对数据框的xpo、xpr变量依次应用step_cut函数时,第二次执行prep操作报错,提示检测到因子变量xpr,但xpr实际为数值型。

可复现代码

library(recipes)

test_data <- data.frame(
  xpo = c(99, NA, NA, 100, 99, NA),
  xpr = c(90, NA, NA, 98, 86, NA),
  target = c(0, 0, 0, 0, 0, 0)
)

recipe_obj <- recipe(as.formula("target ~ ."), data = test_data) %>%
  step_naomit(all_predictors())

var_name <- 'xpo'
cutpoints <- c(9, 97, 98, 99, 101)
recipe_obj <- recipe_obj %>%
  step_cut(
    all_of(var_name),
    breaks = cutpoints,
    id = paste0("cut_", var_name)
  )

prep1 <- prep(recipe_obj, training_data=test_data)
juice1 <- juice(prep1)
bake1 <- bake(prep1, new_data=test_data)

var_name <- 'xpr'
cutpoints <- c(1, 75, 86, 99, 228)
recipe_obj2 <- recipe_obj %>%
  step_cut(
    all_of(var_name),
    breaks = cutpoints,
    id = paste0("cut_", var_name)
  )
prep2 <- prep(recipe_obj2, training_data=test_data)

错误信息

Error in `step_cut()`:
Caused by error in `prep()`:
✖ All columns selected for the step should be double or integer.
• 1 factor variable found: `xpr`

相关输出

recipe_obj2步骤信息

> tidy(recipe_obj2)
# A tibble: 3 × 6
  number operation type   trained skip  id           
   <int> <chr>     <chr>  <lgl>   <lgl> <chr>        
1      1 step      naomit FALSE   TRUE  naomit_xyvBB
2      2 step      cut    FALSE   FALSE cut_xpo      
3      3 step      cut    FALSE   FALSE cut_xpr     

juice1结构

> glimpse(juice1)
Rows: 3
Columns: 3
$ xpo    <fct> "(98,99]", "(99,101]", "(98,99]"
$ xpr    <dbl> 90, 98, 86
$ target <dbl> 0, 0, 0

版本信息

> R.version.string
[1] "R version 4.3.0 (2023-04-21)"
> packageVersion("recipes")
[1] ‘1.1.0’

解决方案

方案1:基于训练后的食谱添加新步骤

当需要查看中间步骤结果时,基于第一次训练后的食谱对象(prep1)添加第二个step_cut,可避免变量类型检测异常:

library(recipes)

test_data <- data.frame(
  xpo = c(99, NA, NA, 100, 99, NA),
  xpr = c(90, NA, NA, 98, 86, NA),
  target = c(0, 0, 0, 0, 0, 0)
)

# 初始食谱与第一次训练
recipe_obj <- recipe(target ~ ., data = test_data) %>%
  step_naomit(all_predictors())

var_name <- 'xpo'
cutpoints <- c(9, 97, 98, 99, 101)
recipe_obj <- recipe_obj %>%
  step_cut(
    all_of(var_name),
    breaks = cutpoints,
    id = paste0("cut_", var_name)
  )

prep1 <- prep(recipe_obj, training_data=test_data)
juice1 <- juice(prep1)
bake1 <- bake(prep1, new_data=test_data)

# 基于训练后的prep1添加第二个step_cut并重新训练
var_name <- 'xpr'
cutpoints <- c(1, 75, 86, 99, 228)
recipe_obj2 <- prep1 %>%
  step_cut(
    all_of(var_name),
    breaks = cutpoints,
    id = paste0("cut_", var_name)
  )
prep2 <- prep(recipe_obj2, training_data=test_data)

# 验证结果
glimpse(juice(prep2))

方案2:一次性定义所有步骤后训练

如果无需查看中间结果,直接一次性定义所有预处理步骤再执行prep,可彻底避免该问题:

library(recipes)

test_data <- data.frame(
  xpo = c(99, NA, NA, 100, 99, NA),
  xpr = c(90, NA, NA, 98, 86, NA),
  target = c(0, 0, 0, 0, 0, 0)
)

# 一次性定义所有预处理步骤
recipe_obj <- recipe(target ~ ., data = test_data) %>%
  step_naomit(all_predictors()) %>%
  step_cut(xpo, breaks = c(9, 97, 98, 99, 101), id = "cut_xpo") %>%
  step_cut(xpr, breaks = c(1, 75, 86, 99, 228), id = "cut_xpr")

# 训练食谱
prep_full <- prep(recipe_obj, training_data=test_data)
glimpse(juice(prep_full))

说明

问题根源是未训练的食谱对象在多次分步添加步骤并训练后,recipes包的变量类型跟踪机制出现异常。上述两种方案通过调整食谱构建方式,确保变量类型检测逻辑正常运行。

内容的提问来源于stack exchange,提问作者dfrankow

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.14 18:59:50