You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何遍历数据框列表并批量执行相同数据分析?

问题背景

我导入了一个大型DataFrame,想要对原始数据的子集执行均值、标准差等分析。现有代码可生成包含目标列的新DataFrame,代码如下:

df1 <- data_clean %>%
  filter(sex=="Male" & experiment_group == "Saline")  %>%
  mutate(avg_presses = rowMeans(select(., c("total1", "total2", "total3")), na.rm=TRUE))

cohort_avg <- c()         #初始化队列均值空向量
cohort_std <- c()
for (cohort_num in 1:max(df1$cohort)) {     # 遍历所有队列
  data_build <- CI_fem_coca                         # 用原始数据初始化临时DataFrame
  for (i in 1:nrow(df1)) {                  #遍历所有行
    if (df1$cohort[i] == cohort_num) {      #如果行的队列号等于当前分组队列号
      data_build <- df1 %>%
        filter(cohort==cohort_num) %>%
        mutate(avgs = mean(avg_presses, na.rm=TRUE),    #添加队列均值列
               std = sd(avg_presses, na.rm=TRUE))   
    }
  }
  cohort_avg<- c(cohort_avg, data_build$avgs)     #将队列均值加入向量
  cohort_std <- c(cohort_std, data_build$std)
}
df1 <- df1 %>%                    #将队列均值和标准差加入原始DataFrame
  add_column(cohort_avgs = cohort_avg, cohort_sd=cohort_std ) 

df1 <- df1 %>%
  mutate(z_score = (avg_presses - cohort_avgs)/cohort_sd)

这段代码运行正常,但需要对4个DataFrame执行完全相同的分析,重复写4次代码过于繁琐。


尝试的批量处理代码及错误

循环列表版本(下标越界错误)

将4个DataFrame加入列表后循环遍历,代码如下:

CI_list <- list(df1, df2, df3, df4)

for (i in 1:length(CI_list)) {
  cohort_avg <- c()         
  cohort_std <- c()
  for (cohort_num in 1:max(CI_list[[i]]$cohort)) {     
    data_build <- CI_list[[i]]                        
    for (i in 1:nrow(CI_list[[i]])) {                 
      if (CI_list[[i]]$cohort[i] == cohort_num) {      
        data_build <- CI_list[[i]] %>%
          filter(cohort==cohort_num) %>%
          mutate(avgs = mean(avg_presses, na.rm=TRUE),    
                 std = sd(avg_presses, na.rm=TRUE))   
      }
    }
    cohort_avg<- c(cohort_avg, data_build$avgs)     
    cohort_std <- c(cohort_std, data_build$std)
  }
  CI_list[[i]] <- CI_list[[i]] %>%                    
    add_column(cohort_avgs = cohort_avg, cohort_sd=cohort_std ) 
  
  CI_list[[i]] <- CI_list[[i]] %>%
    mutate(z_score = (avg_Infusions - cohort_avgs)/cohort_sd)
}

触发**下标越界(subscript out of bounds)**错误。

函数+lapply版本(原子向量无法用$错误)

尝试用函数和lapply实现,代码如下:

find_CI_zscore <- function(df) {
  for (i in 1:length(df)) {
    cohort_avg <- c()         #初始化队列均值空向量
    cohort_std <- c()
    for (cohort_num in 1:max(df$cohort)) {     # 遍历所有队列
      data_build <- df                         # 用原始数据初始化临时DataFrame
      for (i in 1:nrow(df)) {                  #遍历所有行
        if (df$cohort[i] == cohort_num) {      #如果行的队列号等于当前分组队列号
          data_build <- df %>%
            filter(cohort==cohort_num) %>%
            mutate(avgs = mean(avg_Infusions, na.rm=TRUE),    #添加队列均值列
                   std = sd(avg_Infusions, na.rm=TRUE))   
        }
      }
      cohort_avg<- c(cohort_avg, data_build$avgs)     #将队列均值加入向量
      cohort_std <- c(cohort_std, data_build$std)
    }
    df <- df %>%                    #将队列均值和标准差加入原始DataFrame
      add_column(cohort_avgs = cohort_avg, cohort_sd=cohort_std ) 
    
    CI_list <- CI_list %>%
      mutate(z_score = (avg_Infusions - cohort_avgs)/cohort_sd)
  }
}

for (i in 1:length(CI_list)) {
  lapply(CI_list[[i]], find_CI_zscore)
}

触发错误:Error in df$cohort : $ operator is invalid for atomic vectors。


数据示例

> dput(list(df1[1:7, ], df2[1:7, ]))
list(structure(list(cohort = c(1L, 1L, 1L, 1L, 1L, 1L, 1L), avg_Infusions = c(31.3333333333333, 
32.6666666666667, 4, 20, 7, 22.6666666666667, 11.3333333333333
)), row.names = c(NA, 7L), class = "data.frame"), structure(list(
    cohort = c(1L, 1L, 1L, 1L, 1L, 1L, 1L), avg_Infusions = c(6.66666666666667, 
    17.6666666666667, 17.3333333333333, 0.333333333333333, 10, 
    8.66666666666667, 20)), row.names = c(NA, 7L), class = "data.frame"))

解决方案

问题根源

  1. 循环列表版本:内层循环重复使用变量i,覆盖外层循环的索引,导致最终访问CI_list[[i]]时i超出列表长度,触发下标越界。
  2. lapply版本:lapply(CI_list[[i]], find_CI_zscore)会把DataFrame的每一列(原子向量)传入函数,而函数里用df$cohort访问列,自然报错;同时函数内部错误修改全局变量CI_list,逻辑混乱。

优化后的批量处理代码

用dplyr分组计算替代嵌套循环,结合lapply批量处理列表中的DataFrame,代码简洁高效:

# 定义处理单个DataFrame的函数
process_df <- function(df) {
  df %>%
    # 按cohort分组,计算每组的均值和标准差并合并回原数据
    group_by(cohort) %>%
    mutate(
      cohort_avgs = mean(avg_Infusions, na.rm = TRUE),
      cohort_sd = sd(avg_Infusions, na.rm = TRUE)
    ) %>%
    ungroup() %>%
    # 计算z分数
    mutate(z_score = (avg_Infusions - cohort_avgs) / cohort_sd)
}

# 批量处理列表中的所有DataFrame
CI_list <- list(df1, df2, df3, df4)
processed_list <- lapply(CI_list, process_df)

# 可选:将处理后的结果重新赋值给原变量
# df1 <- processed_list[[1]]
# df2 <- processed_list[[2]]
# df3 <- processed_list[[3]]
# df4 <- processed_list[[4]]

代码说明

  • group_by(cohort):按队列分组,后续计算自动按组执行。
  • 分组内的mutate会将每组的均值和标准差匹配到该组的每一行,无需手动循环。
  • ungroup()取消分组,避免后续操作受分组上下文影响。
  • lapply遍历列表,对每个DataFrame调用处理函数,返回处理后的结果列表。

整合行均值计算的扩展版本

如果需要先计算avg_presses(行均值),可将该步骤整合到函数中:

process_df <- function(df, target_cols = c("total1", "total2", "total3")) {
  df %>%
    mutate(avg_presses = rowMeans(select(., all_of(target_cols)), na.rm = TRUE)) %>%
    group_by(cohort) %>%
    mutate(
      cohort_avgs = mean(avg_presses, na.rm = TRUE),
      cohort_sd = sd(avg_presses, na.rm = TRUE)
    ) %>%
    ungroup() %>%
    mutate(z_score = (avg_presses - cohort_avgs) / cohort_sd)
}

内容的提问来源于stack exchange,提问作者detective-captain42

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 02:39:25