如何在R语言中为循环迭代的DataFrame列名添加迭代计数?
在R的for循环中为生成的DataFrame列名添加迭代计数
问题说明
第一次用R写for循环,想给循环里生成的concat、alloc、merge、reSeq列名加上循环迭代编号——比如第一次循环后变成concat_1、alloc_1、merge_1、reSeq_1,第二次循环变成concat_2以此类推。之前试过在mutate里直接写mutate(paste0("concat_",i) = as.numeric...),但根本不生效,想知道正确的处理方式。
解决方法
dplyr的mutate不支持直接用字符串作为列名赋值,得用动态列名语法,下面给两种可行方案:
方案1:生成列时直接用动态命名(推荐)
用:=操作符结合!!(非标准求值)来绑定动态生成的列名,同时用.data或all_of()引用这些动态列名,代码修改如下:
library(dplyr) myDF1 <- data.frame( Name = c("R","R","B","R","X","X"), Group = c(0,0,0,0,1,1)) nCode <- myDF1 %>% group_by(Name) %>% mutate(nmCnt = row_number()) %>% ungroup() %>% mutate(seqBase = ifelse(Group == 0 | Group != lag(Group), nmCnt,0)) %>% mutate(seqBase = na_if(seqBase, 0)) %>% group_by(Name) %>% fill(seqBase) %>% mutate(seqBase = match(seqBase, unique(seqBase))) %>% ungroup %>% mutate(grpRnk = ifelse(Group > 0, sapply(1:n(), function(x) sum(Name[1:x]==Name[x] & Group[1:x] == Group[x])),0)) loopCntr <- nrow(unique(myDF1[myDF1$Group!=0,])) for(i in 1:loopCntr) { # 先定义带迭代号的列名 concat_col <- paste0("concat_", i) alloc_col <- paste0("alloc_", i) merge_col <- paste0("merge_", i) reSeq_col <- paste0("reSeq_", i) # 用!!和:=动态创建concat列 nCode <- nCode %>% mutate(!!concat_col := as.numeric(paste0(seqBase,".",grpRnk))) index <- filter(nCode, Group !=0) %>% select(all_of(concat_col)) %>% # 用all_of引用动态列名 distinct() %>% mutate(truncInd = trunc(.data[[concat_col]])) %>% # .data[[列名]]引用动态列 group_by(truncInd) %>% mutate(cumGrp = cur_group_id()) %>% ungroup() %>% select(-truncInd) index <- if(ifelse(loopCntr > 0, min(index[[concat_col]]), Inf) >= 2){ rbind(data.frame(!!concat_col := c(1), cumGrp=c(1)), index)}else{index} # 同理处理其他列的动态命名 nCode <- nCode %>% mutate(!!alloc_col := index[[concat_col]][index$cumGrp==1][nmCnt]) %>% mutate(!!merge_col := ifelse(is.na(.data[[alloc_col]]), seqBase, .data[[alloc_col]])) %>% group_by(Name) %>% mutate(!!reSeq_col := match(trunc(.data[[merge_col]]), unique(trunc(.data[[merge_col]])))) %>% mutate(!!reSeq_col := (.data[[reSeq_col]] + round(.data[[merge_col]]%%1 * 10,0)/10)) %>% ungroup() } print.data.frame(nCode)
关键要点:
!!会把字符串格式的列名转换成dplyr能识别的符号,:=用来完成动态列的赋值- 引用动态列时,用
.data[[col_name]](兼容所有dplyr版本)或all_of(col_name)(dplyr 1.0+推荐),避免列名识别错误
方案2:循环结束后批量重命名列
如果不想修改循环内的mutate逻辑,可以在每次循环末尾,找到刚生成的concat、alloc、merge、reSeq列,替换成带迭代号的名称:
library(dplyr) myDF1 <- data.frame( Name = c("R","R","B","R","X","X"), Group = c(0,0,0,0,1,1)) nCode <- myDF1 %>% group_by(Name) %>% mutate(nmCnt = row_number()) %>% ungroup() %>% mutate(seqBase = ifelse(Group == 0 | Group != lag(Group), nmCnt,0)) %>% mutate(seqBase = na_if(seqBase, 0)) %>% group_by(Name) %>% fill(seqBase) %>% mutate(seqBase = match(seqBase, unique(seqBase))) %>% ungroup %>% mutate(grpRnk = ifelse(Group > 0, sapply(1:n(), function(x) sum(Name[1:x]==Name[x] & Group[1:x] == Group[x])),0)) loopCntr <- nrow(unique(myDF1[myDF1$Group!=0,])) for(i in 1:loopCntr) { # 保留原循环逻辑,生成临时列名concat/alloc等 nCode <- nCode %>% mutate(concat = as.numeric(paste0(seqBase,".",grpRnk))) index <- filter(nCode, Group !=0) %>% select(concat) %>% distinct() %>% mutate(truncInd = trunc(concat)) %>% group_by(truncInd) %>% mutate(cumGrp = cur_group_id()) %>% ungroup() %>% select(-truncInd) index <- if(ifelse(loopCntr > 0, min(index$concat), Inf) >= 2){ rbind(data.frame(concat=c(1),cumGrp=c(1)),index)}else{index} nCode <- nCode %>% mutate(alloc = index$concat[index$cumGrp==1][nmCnt]) %>% mutate(merge = ifelse(is.na(alloc),seqBase,alloc)) %>% group_by(Name) %>% mutate(reSeq = match(trunc(merge), unique(trunc(merge)))) %>% mutate(reSeq = (reSeq + round(merge%%1 * 10,0)/10)) %>% ungroup() # 循环末尾重命名刚生成的4列 target_cols <- c("concat", "alloc", "merge", "reSeq") new_names <- paste0(target_cols, "_", i) colnames(nCode)[match(target_cols, colnames(nCode))] <- new_names } print.data.frame(nCode)
注意:这个方法依赖match找到最新生成的临时列,适合每次循环只生成一组临时列的场景。
内容的提问来源于stack exchange,提问作者Village.Idyot
相关产品推荐
相关产品推荐

