R tidymodels使用tune_grid调优cubist模型运行失败如何解决
问题描述
我依照操作指引依次定义模型engine、workflow与特征处理recipe,未调优的随机森林、cubist模型以及随机森林的调优流程均能正常运行,但在对cubist模型进行调优时运行失败,报错提示所有模型训练失败,具体报错为:preprocessor 1/1: Error in contrasts<-(tmp, value = contr.funs[1 + isOF[nn]]): contrasts can be applied only to factors with 2 or more levels,以下为相关代码。
可正常运行的未调优模型代码
#untuned rf rf_model <- rand_forest(trees = 1000 # ,mtry = 30 # ,min_n = 3 ) %>% set_engine("ranger", num.threads = parallel::detectCores(), importance = "permutation") %>% set_mode("regression") rf_wflow <- workflow() %>% add_recipe(df_recipe) %>% add_model(rf_model) system.time(rf_fit <- rf_wflow %>% fit(data = train)) # build untuned cubist model cubist_mod <- cubist_rules( committees = 100, neighbors = 9 # max_rules = integer(1) ) %>% set_engine("Cubist") %>% set_mode("regression") cubist_wflow <- workflow() %>% add_recipe(cube_recipe) %>% add_model(cubist_mod) system.time(final_cb_mod <- cubist_wflow %>% fit(data = train)) summary(final_cb_mod$fit)
可正常运行的随机森林调优代码
#tune ranger tune_spec <- rand_forest( mtry = tune(), trees = 1000, min_n = tune()) %>% set_mode("regression") %>% set_engine("ranger") tune_wf <- workflow() %>% add_model(tune_spec) %>% add_recipe(df_recipe) set.seed(234) trees_folds <- vfold_cv(train) rf_grid <- grid_regular( mtry(range = c(10, 30)), min_n(range = c(2, 8)), levels = 5 ) set.seed(345) system.time( tune_res <- tune_grid( tune_wf, resamples = trees_folds, grid = rf_grid, control = control_grid(#save_pred = T, pkgs = c('tm', 'stringr')) ) )
cubist调优报错信息
Warning message: All models failed. See the `.notes` column. > car_tune_res$.notes[1] [[1]] # A tibble: 1 x 1 .notes <chr> 1 preprocessor 1/1: Error in `contrasts<-`(`*tmp*`, value = contr.funs[1 + isOF[nn]]): contrasts can be applied only to factors with 2 or more levels
报错的cubist调优实现代码
#tune cubist cb_grid <- expand.grid(committees = c(1, 10, 50, 100), neighbors = c(1, 5, 7, 9)) set.seed(8226) cubist_mod <- cubist_rules(neighbors = tune(), committees = tune()) %>% set_engine("Cubist") %>% set_mode("regression") tuned_cubist_wf <- workflow() %>% add_model(cubist_mod) %>% add_recipe(cube_recipe) system.time( car_tune_res <- cubist_mod %>% tune_grid( price ~ ., resamples = trees_folds, grid = cb_grid, control = control_grid(#save_pred = T, pkgs = c('tm', 'stringr')) ) )
特征处理recipe参考代码
int_var <- train %>% select(where(is.integer)) %>% colnames() int_var <- c(int_var,'geo_dist') # excl_var <- c('url') add_words <- str_extract(train$url,'(?<=-).*(?=-)') %>% str_extract_all(.,'[[:alpha:]]+') %>% unlist() %>% unique() %>% str_to_lower() df_recipe <- recipe(price ~ .,data = train) %>% step_geodist(lat = lat, lon = long, log = FALSE, ref_lat = 144.946457, ref_lon = -37.840935, # Melb CBD is_lat_lon = FALSE) %>% step_rm('suburb') %>% step_rm('prop_type') %>% step_rm('url') %>% step_zv(all_predictors()) %>% # step_rm('desc') %>% step_mutate(desc_raw = desc) %>% step_textfeature(desc_raw) %>% step_rename_at( starts_with("textfeature_"), fn = ~ gsub("textfeature_desc_raw_", "", .)) %>% step_mutate(desc = str_to_lower(desc)) %>% step_mutate(desc = removeNumbers(desc)) %>% step_mutate(desc = removePunctuation(desc)) %>% step_tokenize(desc) %>% #engine = "spacyr" step_stopwords(desc, stopword_source = 'snowball') %>% step_stopwords(desc, custom_stopword_source = add_words) %>% step_tokenfilter(desc, max_tokens = 1e3) %>% #, max_tokens = tune() step_tfidf(desc) %>% #lda_models = lda_model step_novel(all_nominal(), -all_outcomes()) %>% step_YeoJohnson(all_of(!!int_var), -all_outcomes()) %>% step_dummy(all_nominal(), -all_outcomes(), one_hot = TRUE) %>% step_normalize(all_of(!!int_var)) #dimension reduction due to the sparse matrix cube_recipe <- df_recipe %>% step_pca(matches("school|tfidf_desc"),threshold = .8) %>% #|lda_desc step_rm(starts_with("school")) %>% step_rm(starts_with("lda_desc"))
问题修复方案
核心问题原因
cubist调优代码中调用tune_grid时,未使用已经绑定了cube_recipe的工作流对象tuned_cubist_wf,而是直接传入裸模型对象cubist_mod和自定义公式price ~ .,整套预定义的特征处理逻辑完全没有生效。交叉验证的某一折训练集内存在仅含1个水平的分类因子,生成对比度矩阵时报错。
修复后的调优代码
仅需要修改tune_grid的调用对象,删除冗余的公式参数即可:
system.time( car_tune_res <- tuned_cubist_wf %>% # 替换为绑定了recipe的工作流对象 tune_grid( resamples = trees_folds, # 删除冗余的price ~ . 定义,工作流已内置预测逻辑 grid = cb_grid, control = control_grid( pkgs = c('tm', 'stringr') ) ) )
可选优化
如果修改后仍偶发同类报错,可以在cube_recipe的末尾添加step_zv(all_predictors()),自动过滤交叉验证折叠中方差为0的特征,避免单水平特征引发的错误。
内容的提问来源于stack exchange,提问作者Choc_waffles
相关产品推荐
相关产品推荐

