在R中如何根据另一列文本存在情况填充数据框指定列?
R语言数据框列填充解决方案
需求说明
当first_column存在文本时,用second_column的非空值填充对应行;first_column为空的行,保持second_column原有的空值。
问题分析
直接用fill()函数(无论向上还是向下)无法满足需求,因为fill()会批量填充所有空值(或NA),不会区分first_column是否有文本。需要先传播second_column的非空值,再根据first_column的内容过滤填充范围。
解决方案1:使用dplyr + tidyr(推荐)
通过临时列向上传播非空值,再仅对first_column非空的行应用填充值:
library(dplyr) library(tidyr) # 构造原始数据框 current_dataframe <- data.frame( first_column = c("eeer","","","sdfdfdsf", "sdfsdfdsf","faffa", "we", "", "", "", "", "", "eeee", "", "wqwq", "", "ttetxg", "", "sf", "sf","f", "dfsdf", "sqdweg", "sfsfdsdvghhh", ""), second_column = c("","", "","", "", "", "", "renato paulo", "", "", "", "", "", "carlos gomes", "", "", "", "vitor", "","","","","","", "Cassia rabelo"), stringsAsFactors = FALSE ) # 执行填充逻辑 result <- current_dataframe %>% # 将空字符串转为NA,方便填充操作 mutate(temp_col = ifelse(second_column == "", NA, second_column)) %>% # 向上传播非空值,覆盖上方的NA fill(temp_col, .direction = "up") %>% # 仅在first_column非空的行,用传播后的值替换second_column mutate(second_column = ifelse(first_column != "", temp_col, second_column)) %>% # 清理临时列,将NA转回空字符串 select(-temp_col) %>% mutate(second_column = ifelse(is.na(second_column), "", second_column)) # 验证结果是否符合预期 desired_output <- data.frame( first_column = c("eeer","","","sdfdfdsf", "sdfsdfdsf","faffa", "we", "", "", "", "", "", "eeee", "", "wqwq", "", "ttetxg", "", "sf", "sf","f", "dfsdf", "sqdweg", "sfsfdsdvghhh", ""), second_column = c("","", "","renato paulo", "renato paulo", "renato paulo", "renato paulo", "", "", "", "", "", "carlos gomes", "", "", "", "vitor", "", "Cassia rabelo","Cassia rabelo","Cassia rabelo","Cassia rabelo","Cassia rabelo","Cassia rabelo", ""), stringsAsFactors = FALSE ) all.equal(result, desired_output) # 返回TRUE表示结果正确
解决方案2:Base R实现(无需额外包)
通过循环定位second_column的非空值,再精准填充对应first_column非空的行:
# 构造原始数据框 first_column <- c("eeer","","","sdfdfdsf", "sdfsdfdsf","faffa", "we", "", "", "", "", "", "eeee", "", "wqwq", "", "ttetxg", "", "sf", "sf","f", "dfsdf", "sqdweg", "sfsfdsdvghhh", "") second_column <- c("","", "","", "", "", "", "renato paulo", "", "", "", "", "", "carlos gomes", "", "", "", "vitor", "","","","","","", "Cassia rabelo") current_dataframe <- data.frame(first_column, second_column, stringsAsFactors = FALSE) # 初始化结果列 result_col <- second_column # 找到second_column非空值的位置 non_empty_pos <- which(second_column != "") # 起始填充位置 start_pos <- 1 # 循环处理每个非空值 for (pos in non_empty_pos) { current_val <- second_column[pos] # 确定当前非空值上方的填充范围 fill_range <- start_pos:(pos - 1) # 筛选范围内first_column非空的行 target_rows <- fill_range[first_column[fill_range] != ""] # 填充对应行 result_col[target_rows] <- current_val # 更新下一次填充的起始位置 start_pos <- pos + 1 } # 生成结果数据框 result <- current_dataframe result$second_column <- result_col # 验证结果 desired_output <- data.frame( first_column = c("eeer","","","sdfdfdsf", "sdfsdfdsf","faffa", "we", "", "", "", "", "", "eeee", "", "wqwq", "", "ttetxg", "", "sf", "sf","f", "dfsdf", "sqdweg", "sfsfdsdvghhh", ""), second_column = c("","", "","renato paulo", "renato paulo", "renato paulo", "renato paulo", "", "", "", "", "", "carlos gomes", "", "", "", "vitor", "", "Cassia rabelo","Cassia rabelo","Cassia rabelo","Cassia rabelo","Cassia rabelo","Cassia rabelo", ""), stringsAsFactors = FALSE ) all.equal(result, desired_output) # 返回TRUE表示结果正确
内容的提问来源于stack exchange,提问作者Francisco1988
相关产品推荐
相关产品推荐

