R语言For循环提取数据时遗漏第0行值的问题求助
问题描述
我有一个包含5列(Ref、A、B、C、D)的数据集,分为2行(1、2),其中第2行下的4列各有从0到n的多行数据。我编写了一段R语言的For循环代码,用于提取所有行的值并横向拼接,但代码遗漏了第0行的数据。
原代码
library(dplyr) # Import the data from the CSV file data <- read.csv("/file path/data.csv") EUR <- data.frame(Ref = data$Value[data$Column == "A" & data$Row == '1']) Investment_strategies <- data[data$Row == '2', ] Investment_strategies <- Investment_strategies[Investment_strategies$Line >= 0, ] line_count <- length(Investment_strategies[, "Row"]) line_num <- 0 for (line in seq_len(line_count)) { col_suffix <- paste0("_", line_num) temp_col_A <- Investment_strategies[Investment_strategies$Column == "A" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, ][match(EUR[,"Ref"], Investment_strategies[Investment_strategies$Column=="A" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, 'Ref']), 'Value'] temp_col_B <- Investment_strategies[Investment_strategies$Column == "B" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, ][match(EUR[,"Ref"], Investment_strategies[Investment_strategies$Column=="B" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, 'Ref']), 'Value'] temp_col_C <- Investment_strategies[Investment_strategies$Column == "C" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, ][match(EUR[,"Ref"], Investment_strategies[Investment_strategies$Column=="C" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, 'Ref']), 'Value'] temp_col_D <- Investment_strategies[Investment_strategies$Column == "D" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, ][match(EUR[,"Ref"], Investment_strategies[Investment_strategies$Column=="D" & Investment_strategies$Row=='2'& Investment_strategies$Line == line, 'Ref']), 'Value'] if (any(!is.na(temp_col_A)) || any(!is.na(temp_col_B)) || any(!is.na(temp_col_C)) || any(!is.na(temp_col_D))) { EUR[, paste0("Strategy", col_suffix)] <- temp_col_A EUR[, paste0("Primary", col_suffix)] <- temp_col_B EUR[, paste0("Value", col_suffix)] <- temp_col_C EUR[, paste0("Other Strategy", col_suffix)] <- temp_col_D line_num <- line_num + 1 } if (line_num > line_count) { break } } write.csv(EUR, "/file path/output.csv", row.names = FALSE)
问题根源
原循环使用seq_len(line_count)生成遍历序列,该函数会生成从1到line_count的整数,直接跳过了Line=0的行,这就是第0行数据被遗漏的核心原因。
修正后的代码
library(dplyr) # 读取数据集 data <- read.csv("/file path/data.csv") # 初始化EUR数据框,提取Row=1中Column=A的Ref值 EUR <- data.frame(Ref = data$Value[data$Column == "A" & data$Row == '1']) # 筛选Row=2且Line>=0的目标数据 Investment_strategies <- data %>% filter(Row == '2', Line >= 0) # 获取所有存在的Line值(包含0) unique_lines <- unique(Investment_strategies$Line) line_num <- 0 # 遍历每个Line值 for (line in unique_lines) { col_suffix <- paste0("_", line_num) # 提取当前Line下各列对应Ref的值 temp_col_A <- Investment_strategies %>% filter(Column == "A", Line == line) %>% slice(match(EUR$Ref, .$Ref)) %>% pull(Value) temp_col_B <- Investment_strategies %>% filter(Column == "B", Line == line) %>% slice(match(EUR$Ref, .$Ref)) %>% pull(Value) temp_col_C <- Investment_strategies %>% filter(Column == "C", Line == line) %>% slice(match(EUR$Ref, .$Ref)) %>% pull(Value) temp_col_D <- Investment_strategies %>% filter(Column == "D", Line == line) %>% slice(match(EUR$Ref, .$Ref)) %>% pull(Value) # 若当前Line下有非空值,则添加到EUR中 if (any(!is.na(c(temp_col_A, temp_col_B, temp_col_C, temp_col_D)))) { EUR[, paste0("Strategy", col_suffix)] <- temp_col_A EUR[, paste0("Primary", col_suffix)] <- temp_col_B EUR[, paste0("Value", col_suffix)] <- temp_col_C EUR[, paste0("Other Strategy", col_suffix)] <- temp_col_D line_num <- line_num + 1 } } # 输出结果文件 write.csv(EUR, "/file path/output.csv", row.names = FALSE)
修正要点
- 遍历所有Line值:改用
unique(Investment_strategies$Line)获取数据中实际存在的Line编号,确保包含0行。 - 简化数据提取:使用
dplyr的管道语法替代繁琐的子集索引,代码可读性更强。 - 优化非空判断:合并四个列的非NA值检查,逻辑更简洁。
内容的提问来源于stack exchange,提问作者user22050064
相关产品推荐
相关产品推荐

