You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

嵌套For循环生成的表格存在重复值问题求助

问题描述

我已经把所有个体划分为doy组和bdate组两类分组,想要获取每个bdate组内各doy组对应不同地点(G、E、S、D)的个体加权平均值。但运行下方嵌套For循环代码后,生成的ofig1数据框中每个doy组返回完全相同的值,期望得到包含56行的数据集,涵盖各doy/bdate组合的加权值及列求和得到的加权平均wa。

glas <- vector()
ess <- vector()
sal <- vector()
deep <- vector()
bdate_group <- vector()
doy_group <- vector()

for (bdate_group_num in 1:8) { #repeats same values for corresponding doy groups within each bdate
  for (doy_group_num in 1:7) {
    
    tot <- sum(ofig$doy_group_num == doy_group_num, na.rm = TRUE)
    
    g <- (sum(ofig$Site == "G" & ofig$doy_group_num == doy_group_num, na.rm = TRUE) / tot) * 67.43
    e <- (sum(ofig$Site == "E" & ofig$doy_group_num == doy_group_num, na.rm = TRUE) / tot) * 10.12
    s <- (sum(ofig$Site == "S" & ofig$doy_group_num == doy_group_num, na.rm = TRUE) / tot) * 16.69
    d <- (sum(ofig$Site == "D" & ofig$doy_group_num == doy_group_num, na.rm = TRUE) / tot) * 10.51
    
    glas <- c(glas, g)
    ess <- c(ess, e)
    sal <- c(sal, s)
    deep <- c(deep, d)
    bdate_group <- c(bdate_group, bdate_group_num)
    doy_group <- c(doy_group, doy_group_num)
  }
}

#create data frame
ofig1 <- data.frame(bdate_group = bdate_group, doy_group = doy_group, glas = glas, ess = ess, sal = sal, deep = deep)

#sum across columns for weighted average
ofig1$wa <- rowSums(ofig1[, c("glas", "ess", "sal", "deep")])
问题根源

循环内的计算完全未加入bdate_group_num的过滤条件,所有bdate组里的同一个doy组,计算的都是全数据集里该doy组的结果,因此每个doy组在不同bdate组中的值完全一致。

修正方案1:修复循环逻辑

在计算时加入bdate_group的过滤条件,确保是在当前bdate_group_num分组内统计数据,同时优化向量初始化和除零处理:

# 提前初始化固定长度的向量,提升效率
glas <- vector(length = 8*7)
ess <- vector(length = 8*7)
sal <- vector(length = 8*7)
deep <- vector(length = 8*7)
bdate_group <- vector(length = 8*7)
doy_group <- vector(length = 8*7)

idx <- 1
for (bdate_group_num in 1:8) {
  for (doy_group_num in 1:7) {
    # 筛选当前bdate和doy组的子集
    subset_data <- ofig[ofig$bdate_group == bdate_group_num & ofig$doy_group == doy_group_num, ]
    tot <- nrow(subset_data)
    
    # 计算各地点占比,避免空子集除零错误
    g_ratio <- if(tot == 0) 0 else sum(subset_data$Site == "G", na.rm = TRUE)/tot
    e_ratio <- if(tot == 0) 0 else sum(subset_data$Site == "E", na.rm = TRUE)/tot
    s_ratio <- if(tot == 0) 0 else sum(subset_data$Site == "S", na.rm = TRUE)/tot
    d_ratio <- if(tot == 0) 0 else sum(subset_data$Site == "D", na.rm = TRUE)/tot
    
    # 加权计算并赋值
    glas[idx] <- g_ratio * 67.43
    ess[idx] <- e_ratio * 10.12
    sal[idx] <- s_ratio * 16.69
    deep[idx] <- d_ratio * 10.51
    
    bdate_group[idx] <- bdate_group_num
    doy_group[idx] <- doy_group_num
    idx <- idx + 1
  }
}

ofig1 <- data.frame(bdate_group, doy_group, glas, ess, sal, deep)
ofig1$wa <- rowSums(ofig1[, c("glas", "ess", "sal", "deep")])
修正方案2:用dplyr分组计算(更高效简洁)

使用dplyr的分组聚合功能替代循环,代码更简洁易读,且避免手动维护索引:

library(dplyr)

ofig1 <- ofig %>%
  group_by(bdate_group, doy_group) %>%
  summarize(
    tot = n(),
    glas = (sum(Site == "G", na.rm = TRUE)/tot)*67.43,
    ess = (sum(Site == "E", na.rm = TRUE)/tot)*10.12,
    sal = (sum(Site == "S", na.rm = TRUE)/tot)*16.69,
    deep = (sum(Site == "D", na.rm = TRUE)/tot)*10.51,
    .groups = "drop" # 取消分组状态
  ) %>%
  mutate(wa = rowSums(across(c(glas, ess, sal, deep)))) %>%
  select(-tot) # 可选:移除辅助列tot

内容的提问来源于stack exchange,提问作者nick p

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.20 16:47:47