在R中对比DataFrame与列表并生成标记列及权重列的技术问询
解决方案
以下是基于R语言tidyverse工具包的实现代码:
1. 加载依赖包
library(tidyverse)
2. 定义基础配置
# 目标元素列表 target_elements <- c("Maths","Science","Engg") # 权重映射 weight_map <- c(Maths = 1, Science = 2, Engg = 3)
3. 生成带Flag列的df1_soln
df1_soln <- df1 %>% rowwise() %>% mutate( # 合并当前行所有内容为单个字符串,便于检测 row_text = str_c(c_across(everything()), collapse = " "), # 检查每个目标元素是否存在 has_all = all(map_lgl(target_elements, ~str_detect(row_text, .x))), # 设置Flag值 Flag = if_else(has_all, "YES", "NO") ) %>% ungroup() %>% select(-row_text, -has_all)
4. 生成带Weightage列的df3
df3 <- df1 %>% rowwise() %>% mutate( # 处理每行的每个单元格,提取有效元素并计算最高权重 cell_weight_info = list( map(c_across(everything()), ~{ if (.x == "NA") return(NULL) # 提取单元格中包含的目标元素 matched_elems <- str_extract_all(.x, str_c(target_elements, collapse = "|"))[[1]] if (length(matched_elems) == 0) return(NULL) # 返回单元格内容和对应的最高权重 list(content = .x, max_wt = max(weight_map[matched_elems])) }) %>% discard(is.null) ), # 选出权重最高的单元格内容,无匹配则设为NA Weightage = case_when( length(cell_weight_info) == 0 ~ "NA", TRUE ~ cell_weight_info[[which.max(map_dbl(cell_weight_info, ~.x$max_wt))]]$content ) ) %>% ungroup() %>% select(-cell_weight_info)
运行上述代码后,df1_soln和df3将与预期输出完全匹配。
内容的提问来源于stack exchange,提问作者Akshi
相关产品推荐
相关产品推荐

