You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用for循环结合summarise_by_time()时Sum()错误计算分子为0

问题:循环汇总数值变量时Num始终显示为0

需对数据框df中指定变量列表的每个变量按季度生成汇总结果,但Num值始终错误为0。这些结果变量为仅含1和0的数值型变量。尝试过timetk::summarize_by_time()和dplyr分组汇总两种方法均无效,但非循环格式下代码可正常运行。


数据结构

df <- structure(list(Operation.Date = structure(c(1483401600, 1483401600, 
1483401600, 1483401600, 1483660800, 1483660800, 1483660800, 1483660800, 
1483660800, 1483401600, 1483747200, 1483574400, 1483574400, 1483488000, 
1483401600), tzone = "UTC", class = c("POSIXct", "POSIXt")), 
    any_morbidity = c(0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 
    0, 0), any_ssi = c(0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 
    0, 0), any_uti = c(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 
    0, 0)), row.names = c(NA, -15L), class = c("tbl_df", "tbl", 
"data.frame"))

# 初始化步骤
outcomes_unadjusted <- c("any_morbidity", "any_ssi", "any_uti" )
summaries <- list()

错误尝试1:使用timetk的summarize_by_time()

library(dplyr)
library(timetk)

for (outcome in outcomes_unadjusted) {
  df_summary <- df %>%
    summarize_by_time(
      .date_var = Operation.Date,
      .by = "quarter", 
      Num = sum(!!outcome),  # 错误:!!仅解析字符串,未正确引用数据框变量
      Denom = n(),
      Percent = (Num / Denom) * 100
    ) %>%
    mutate(mean = mean(Percent),
           outcome = paste0(outcome))
  
  summaries[[outcome]] <- df_summary
}

unadjusted_all <- bind_rows(summaries)
print(unadjusted_all)

错误尝试2:使用dplyr分组汇总

library(zoo)

for (outcome in outcomes_unadjusted) {
  df_summary <- df %>%
    mutate(quarter = as.yearqtr(as.Date(Operation.Date))) %>%
    group_by(quarter) %>%
    summarize(
      Num = sum(outcome == 1),  # 错误:outcome是字符串,不是数据框变量引用
      Denom = n(),
      Percent = (Num / Denom) * 100
    ) %>%
    mutate(mean = mean(Percent),
           outcome = paste0(outcome))
  
  summaries[[outcome]] <- df_summary
}

unadjusted_all <- bind_rows(summaries)
print(unadjusted_all)

解决方法

问题核心是循环中字符串变量名的引用方式错误,需将字符串转换为可识别的变量引用。以下是两种修正方案,以及更简洁的无循环写法:

修正timetk方法

library(dplyr)
library(timetk)
library(rlang)

for (outcome in outcomes_unadjusted) {
  outcome_sym <- sym(outcome)  # 将字符串转为符号变量
  df_summary <- df %>%
    summarize_by_time(
      .date_var = Operation.Date,
      .by = "quarter", 
      Num = sum(!!outcome_sym, na.rm = TRUE),  # 正确引用变量
      Denom = n(),
      Percent = (Num / Denom) * 100
    ) %>%
    mutate(mean = mean(Percent),
           outcome = outcome)
  
  summaries[[outcome]] <- df_summary
}

unadjusted_all <- bind_rows(summaries)
print(unadjusted_all)

修正dplyr分组方法

library(zoo)
library(dplyr)

for (outcome in outcomes_unadjusted) {
  df_summary <- df %>%
    mutate(quarter = as.yearqtr(as.Date(Operation.Date))) %>%
    group_by(quarter) %>%
    summarize(
      Num = sum(.data[[outcome]] == 1, na.rm = TRUE),  # 使用.data代词引用变量
      Denom = n(),
      Percent = (Num / Denom) * 100
    ) %>%
    mutate(mean = mean(Percent),
           outcome = outcome)
  
  summaries[[outcome]] <- df_summary
}

unadjusted_all <- bind_rows(summaries)
print(unadjusted_all)

更简洁的无循环写法(tidyverse风格)

通过pivot_longer将宽数据转为长格式,一次完成所有变量的汇总:

library(dplyr)
library(tidyr)
library(zoo)

unadjusted_all <- df %>%
  pivot_longer(cols = all_of(outcomes_unadjusted), names_to = "outcome", values_to = "value") %>%
  mutate(quarter = as.yearqtr(as.Date(Operation.Date))) %>%
  group_by(quarter, outcome) %>%
  summarize(
    Num = sum(value == 1, na.rm = TRUE),
    Denom = n(),
    Percent = (Num / Denom) * 100,
    .groups = "drop"
  ) %>%
  group_by(outcome) %>%
  mutate(mean = mean(Percent)) %>%
  ungroup()

print(unadjusted_all)

内容的提问来源于stack exchange,提问作者Cassandra

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.07 12:10:20