You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R tidymodels使用tune_grid调优cubist模型运行失败如何解决

问题描述

我依照操作指引依次定义模型engine、workflow与特征处理recipe,未调优的随机森林、cubist模型以及随机森林的调优流程均能正常运行,但在对cubist模型进行调优时运行失败,报错提示所有模型训练失败,具体报错为:preprocessor 1/1: Error in contrasts<-(tmp, value = contr.funs[1 + isOF[nn]]): contrasts can be applied only to factors with 2 or more levels,以下为相关代码。

可正常运行的未调优模型代码

#untuned rf
rf_model <- 
  rand_forest(trees = 1000
              # ,mtry = 30
              # ,min_n = 3
              ) %>% 
  set_engine("ranger", 
             num.threads = parallel::detectCores(), 
             importance = "permutation") %>% 
  set_mode("regression")

rf_wflow <- 
  workflow() %>% 
  add_recipe(df_recipe) %>% 
  add_model(rf_model)

system.time(rf_fit <- rf_wflow %>% fit(data = train))

# build untuned cubist model
cubist_mod <-
  cubist_rules(
    committees = 100,
    neighbors = 9
    # max_rules = integer(1)
  ) %>%
  set_engine("Cubist") %>%
  set_mode("regression")

cubist_wflow <- 
  workflow() %>% 
  add_recipe(cube_recipe) %>%
  add_model(cubist_mod)

system.time(final_cb_mod <- cubist_wflow %>% fit(data = train))

summary(final_cb_mod$fit)

可正常运行的随机森林调优代码

#tune ranger
tune_spec <- rand_forest(
  mtry = tune(),
  trees = 1000,
  min_n = tune()) %>%
  set_mode("regression") %>%
  set_engine("ranger")

tune_wf <- 
  workflow() %>%
  add_model(tune_spec) %>% 
  add_recipe(df_recipe)

set.seed(234)
trees_folds <- vfold_cv(train)

rf_grid <- grid_regular(
  mtry(range = c(10, 30)),
  min_n(range = c(2, 8)),
  levels = 5
)

set.seed(345)
system.time(
  tune_res <- tune_grid(
    tune_wf,
    resamples = trees_folds,
    grid = rf_grid,
    control =
      control_grid(#save_pred = T,
                   pkgs = c('tm', 'stringr'))
  )
)

cubist调优报错信息

Warning message:
All models failed. See the `.notes` column.

> car_tune_res$.notes[1]
[[1]]
# A tibble: 1 x 1
  .notes                                                                                                                                              
  <chr>                                                                                                                                              
1 preprocessor 1/1: Error in `contrasts<-`(`*tmp*`, value = contr.funs[1 + isOF[nn]]): contrasts can be applied only to factors with 2 or more levels

报错的cubist调优实现代码

#tune cubist
cb_grid <- expand.grid(committees = c(1, 10, 50, 100), neighbors = c(1, 5, 7, 9))

set.seed(8226)

cubist_mod <-
  cubist_rules(neighbors = tune(), committees = tune()) %>%
  set_engine("Cubist") %>% 
  set_mode("regression")

tuned_cubist_wf <- workflow() %>%
  add_model(cubist_mod) %>% 
  add_recipe(cube_recipe)

system.time(
  car_tune_res <-
    cubist_mod %>%
    tune_grid(
      price ~ .,
      resamples = trees_folds,
      grid = cb_grid,
      control =
        control_grid(#save_pred = T,
          pkgs = c('tm', 'stringr'))
    )
)

特征处理recipe参考代码

int_var <- train %>% select(where(is.integer)) %>% colnames()
int_var <- c(int_var,'geo_dist')
# excl_var <- c('url')

add_words <- 
  str_extract(train$url,'(?<=-).*(?=-)') %>% 
  str_extract_all(.,'[[:alpha:]]+') %>% 
  unlist() %>% 
  unique() %>% 
  str_to_lower()

df_recipe <-
  recipe(price ~ .,data = train) %>%
  step_geodist(lat = lat, lon = long, log = FALSE,
               ref_lat = 144.946457, ref_lon = -37.840935, # Melb CBD
               is_lat_lon = FALSE) %>%
  step_rm('suburb') %>% 
  step_rm('prop_type') %>% 
  step_rm('url') %>% 
  step_zv(all_predictors()) %>%
  # step_rm('desc') %>% 
  step_mutate(desc_raw = desc) %>%
  step_textfeature(desc_raw) %>%
  step_rename_at(
    starts_with("textfeature_"),
    fn = ~ gsub("textfeature_desc_raw_", "", .)) %>% 
  step_mutate(desc = str_to_lower(desc)) %>% 
  step_mutate(desc = removeNumbers(desc)) %>% 
  step_mutate(desc = removePunctuation(desc)) %>% 
  step_tokenize(desc) %>%  #engine = "spacyr"
  step_stopwords(desc, stopword_source = 'snowball') %>%
  step_stopwords(desc, custom_stopword_source = add_words) %>%
  step_tokenfilter(desc, max_tokens = 1e3) %>% #, max_tokens = tune()
  step_tfidf(desc) %>% #lda_models = lda_model
  step_novel(all_nominal(), -all_outcomes()) %>%
  step_YeoJohnson(all_of(!!int_var), -all_outcomes()) %>%
  step_dummy(all_nominal(), -all_outcomes(), one_hot = TRUE) %>%
  step_normalize(all_of(!!int_var))

#dimension reduction due to the sparse matrix
cube_recipe <- 
  df_recipe %>% 
  step_pca(matches("school|tfidf_desc"),threshold = .8) %>% #|lda_desc
  step_rm(starts_with("school")) %>%
  step_rm(starts_with("lda_desc"))
问题修复方案

核心问题原因

cubist调优代码中调用tune_grid时,未使用已经绑定了cube_recipe的工作流对象tuned_cubist_wf,而是直接传入裸模型对象cubist_mod和自定义公式price ~ .,整套预定义的特征处理逻辑完全没有生效。交叉验证的某一折训练集内存在仅含1个水平的分类因子,生成对比度矩阵时报错。

修复后的调优代码

仅需要修改tune_grid的调用对象,删除冗余的公式参数即可:

system.time(
  car_tune_res <-
    tuned_cubist_wf %>% # 替换为绑定了recipe的工作流对象
    tune_grid(
      resamples = trees_folds, # 删除冗余的price ~ . 定义,工作流已内置预测逻辑
      grid = cb_grid,
      control =
        control_grid(
          pkgs = c('tm', 'stringr')
        )
    )
)

可选优化

如果修改后仍偶发同类报错,可以在cube_recipe的末尾添加step_zv(all_predictors()),自动过滤交叉验证折叠中方差为0的特征,避免单水平特征引发的错误。

内容的提问来源于stack exchange,提问作者Choc_waffles

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.27 21:45:09