You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R中dplyr包group_by函数失效问题求助

问题

使用R 4.3.0与dplyr 1.1.2自定义统计函数时,group_by命令未生效。尝试过带引号/无引号传参、{{}}/[[]]/!!等多种写法,要么得到3行完全相同的全数据集汇总结果,要么仅得到1行全数据集汇总。分组变量cohort有1、2、3三个水平,期望输出3行对应各分组的统计结果。相关代码及测试数据如下:

原函数代码(补全函数定义):

run_stats <- function(inds, grouping, var, n_digit) {
  inds %>%
    dplyr::group_by(.data[[grouping]]) %>%
    dplyr::summarize(
        n       = sum     (!is.na(var)),
        nmiss   = sum     (is.na(var)),
        mean    = mean    (var, na.rm=T),
        sd      = sd      (var, na.rm=T),
        se      = sd/sqrt(n),
        lower   = mean -  as.numeric(qt(0.975, df=n-1)*se),
        upper   = mean +  as.numeric(qt(0.975, df=n-1)*se),
        min     = min     (var, na.rm=T),
        max     = max     (var, na.rm=T),
        median  = median  (var, na.rm=T),                    
        q1      = quantile(var, 0.25, na.rm=T),
        q3      = quantile(var, 0.75, na.rm=T),
        iqr     = IQR     (var, na.rm=T)) %>%
      
    dplyr::mutate(
        n     = round(n,            digits=0),
        nmiss = round(nmiss,        digits=0),
        mean  = format(round(mean,  digits=n_digit),   nsmall=n_digit),
        median= format(round(median,digits=n_digit),   nsmall=n_digit),
        sd    = format(round(sd,    digits=n_digit+1), nsmall=n_digit+1),
        se    = format(round(se,    digits=n_digit+1), nsmall=n_digit+1),
        min   = format(round(min,   digits=n_digit),   nsmall=n_digit),
        max   = format(round(max,   digits=n_digit),   nsmall=n_digit),
        q1    = format(round(q1,    digits=n_digit),   nsmall=n_digit),
        q3    = format(round(q3,    digits=n_digit),   nsmall=n_digit),
        iqr   = format(round(iqr,   digits=n_digit),   nsmall=n_digit),
        lower = format(round(lower, digits=n_digit),   nsmall=n_digit),
        upper = format(round(upper, digits=n_digit),   nsmall=n_digit)) %>%
      
    dplyr::mutate(
        range=paste0(min,  ', ', max),
        x_sd =paste0(mean, ' (', sd,')'),
        ci   =paste0(lower,', ', upper),
        q1q3 =paste0(q1,   ', ', q3))
}

ds01=run_stats(inds=mr, grouping="cohort", var=mrse, n_digit=3)
ds01

测试数据:

# 构建测试数据框
mr  <-tribble (
  ~id, ~cohort, ~mrse,
  1, 1, -0.6,
  2, 1, -0.2,
  3, 1, -0.3,
  4, 1, -0.3,
  5, 1, -0.2,
  6, 1, -1.0,
  7, 1, -2.0,
  8, 1, -1.5,
  9, 1, -0.2,
 10, 1, -0.1,
 11, 2,  0.6,
 12, 2,  2.2,
 13, 2,  0.3,
 14, 2,  3.3,
 15, 2,  1.2,
 16, 2,  1.0,
 17, 2,  4.0,
 18, 2,  1.5,
 19, 2,  0.2,
 20, 2,  0.1,
 21, 3, -0.6,
 22, 3,  0.2,
 23, 3, -0.3,
 24, 3,  0.3,
 25, 3, -0.2,
 26, 3,  1.0,
 27, 3,  0.0,
 28, 3, -0.5,
 29, 3,  0.2,
 30, 3, -0.1)

解决方法

问题核心在于整洁评估的参数传递错误,导致group_by和目标变量引用都未正确关联到分组规则和统计变量。以下是修正方案:

1. 完整修正后的函数

library(dplyr)
library(tibble)

run_stats <- function(inds, grouping, var, n_digit) {
  inds %>%
    # 用pick+all_of处理字符串类型的分组变量,确保正确分组
    dplyr::group_by(pick(all_of(grouping))) %>%
    dplyr::summarize(
      n       = sum(!is.na({{var}})),
      nmiss   = sum(is.na({{var}})),
      mean    = mean({{var}}, na.rm = TRUE),
      sd      = sd({{var}}, na.rm = TRUE),
      se      = sd/sqrt(n),
      lower   = mean - as.numeric(qt(0.975, df = n-1)*se),
      upper   = mean + as.numeric(qt(0.975, df = n-1)*se),
      min     = min({{var}}, na.rm = TRUE),
      max     = max({{var}}, na.rm = TRUE),
      median  = median({{var}}, na.rm = TRUE),                    
      q1      = quantile({{var}}, 0.25, na.rm = TRUE),
      q3      = quantile({{var}}, 0.75, na.rm = TRUE),
      iqr     = IQR({{var}}, na.rm = TRUE),
      .groups = "keep"
    ) %>%
    dplyr::mutate(
      n     = round(n, digits = 0),
      nmiss = round(nmiss, digits = 0),
      mean  = format(round(mean, digits = n_digit), nsmall = n_digit),
      median= format(round(median, digits = n_digit), nsmall = n_digit),
      sd    = format(round(sd, digits = n_digit+1), nsmall = n_digit+1),
      se    = format(round(se, digits = n_digit+1), nsmall = n_digit+1),
      min   = format(round(min, digits = n_digit), nsmall = n_digit),
      max   = format(round(max, digits = n_digit), nsmall = n_digit),
      q1    = format(round(q1, digits = n_digit), nsmall = n_digit),
      q3    = format(round(q3, digits = n_digit), nsmall = n_digit),
      iqr   = format(round(iqr, digits = n_digit), nsmall = n_digit),
      lower = format(round(lower, digits = n_digit), nsmall = n_digit),
      upper = format(round(upper, digits = n_digit), nsmall = n_digit)
    ) %>%
    dplyr::mutate(
      range = paste0(min,  ', ', max),
      x_sd  = paste0(mean, ' (', sd,')'),
      ci    = paste0(lower,', ', upper),
      q1q3  = paste0(q1,   ', ', q3)
    )
}

2. 测试运行

使用原测试数据调用修正后的函数:

ds01 <- run_stats(inds = mr, grouping = "cohort", var = mrse, n_digit = 3)
print(ds01)

运行后将得到3行对应cohort三个水平的统计结果,完全符合预期。

关键修改点说明

  • group_by(pick(all_of(grouping))):当分组变量以字符串形式传入时,all_of()将字符串转换为dplyr可识别的变量名,pick()确保保留列结构,正确触发分组逻辑。
  • {{var}}:整洁评估语法,直接引用传入的裸名变量(如mrse),避免变量解析错误,确保统计计算针对目标变量执行。
  • .groups = "keep":可选参数,确保分组上下文在summarize后保留(也可省略,dplyr会自动调整分组状态)。

内容的提问来源于stack exchange,提问作者Gerard Smits

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 05:07:27