You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用gtsummary创建含分层及双独立分组变量的tbl_summary表格

用gtsummary创建分层变量下的独立分类变量统计表格

需要创建带有分层变量a的tbl_summary表格,要求每个分层类别下展示两个独立二元变量b和c的统计结果,同时呈现变量d的n/%统计值。b和c为观测中独立的不同变量,不可合并。尝试结合tbl_summary、tbl_strata和tbl_merge实现时遇到以下问题:

  • 首次尝试生成的表格缺少b、c的正确列及分层变量a的标识
  • 后续尝试的布局接近需求,但d的统计值在不同分层中重复显示

首次尝试代码

library(readr)
library(dplyr)
library(tidyverse)
library(gtsummary)

df <- data.frame(id=1:10,
                 a=c('red', 'blue', 'red', 'red', 'blue', 'red', 'blue', 'blue', 'blue', 'red'),
                 b=c('yes', 'yes', 'yes', 'no', 'yes', 'no', 'yes', 'yes', 'yes', 'no'),
                 c=c('cheese', 'cheese', 'steak', 'steak', 'cheese', 'steak', 'steak', 'cheese', 'steak', 'steak'),
                 d=c(22, 82, 44, 56, 27, 61, 22, 19, 38, 47)
)

df$a <- factor(df$a)
df$b <- factor(df$b)
df$c <- factor(df$c)
df$d <- factor(df$d)

t1 <- df %>%
   select(a, b, d) %>%
   mutate(a = paste("a=", a)) %>%
   mutate(b = paste("b=", b)) %>%
   tbl_strata(
      strata = a,
      .tbl_fun =
         ~ .x %>%
         tbl_summary(by = b, missing = "no"),
      .header = "**{strata}**, N = {n}"
   )

t2 <- df %>%
   select(a, c, d) %>%
   mutate(a = paste("a=", a)) %>%
   mutate(c = paste("c=", c)) %>%
   tbl_strata(
      strata = a,
      .tbl_fun =
         ~ .x %>%
         tbl_summary(by = c, missing = "no"),
      .header = "**{strata}**, N = {n}"
   )

tbl_merge(
   tbls = list(t1, t2),
   tab_spanner = c("**b**", "**c**")
)

后续尝试代码

df <- data.frame(id=1:11,
                 a=c('red', 'blue', 'red', 'red', 'blue', 'red', 'blue', 'blue', 'blue', 'red', 'blue'),
                 b=c('yes', 'yes', 'yes', 'no', 'yes', 'no', 'yes', 'yes', 'yes', 'no', 'no'),
                 x=c('cheese', 'cheese', 'steak', 'steak', 'cheese', 'steak', 'steak', 'cheese', 'steak', 'steak', 'cheese'),
                 d=c(22, 82, 44, 56, 27, 61, 22, 19, 38, 47, 38)
)

df$a <- factor(df$a)
df$b <- factor(df$b)
df$x <- factor(df$x)
df$d <- factor(df$d)

t3 <- df %>%
   select(b, d) %>%
   mutate(b = paste("b=", b)) %>%
   tbl_summary(by = b, 
               missing = "no"
   )

t4 <- df %>%
   select(x, d) %>%
   mutate(x = paste("x=", x)) %>%
   tbl_summary(by = x, 
               missing = "no"
   )

df %>% tbl_strata(
   strata = a,
   .tbl_fun =
      ~tbl_merge(
         tbls = list(t3, t4)
      ),
   .header = "**a={strata}**, N = {n}"
)

正确实现方案

核心思路是在tbl_strata的分层处理函数内,针对每个分层的子集数据,分别生成b和c的统计表格,再合并。这样每个分层的统计结果都基于当前子集,避免全局统计重复,同时保留分层标识和独立的变量列。

library(gtsummary)
library(dplyr)

# 构造并预处理数据
df <- data.frame(id=1:10,
                 a=c('red', 'blue', 'red', 'red', 'blue', 'red', 'blue', 'blue', 'blue', 'red'),
                 b=c('yes', 'yes', 'yes', 'no', 'yes', 'no', 'yes', 'yes', 'yes', 'no'),
                 c=c('cheese', 'cheese', 'steak', 'steak', 'cheese', 'steak', 'steak', 'cheese', 'steak', 'steak'),
                 d=c(22, 82, 44, 56, 27, 61, 22, 19, 38, 47)
)

df <- df %>%
  mutate(across(c(a, b, c, d), factor))

# 生成目标表格
df %>%
  tbl_strata(
    strata = a,
    .tbl_fun = function(data) {
      # 生成b变量分组的统计表格
      tbl_b <- data %>%
        select(b, d) %>%
        tbl_summary(
          by = b,
          missing = "no",
          label = list(d = "变量d"),
          statistic = list(all_categorical() = "{n} ({p}%)")
        ) %>%
        modify_header(label = "**b分组**")
      
      # 生成c变量分组的统计表格
      tbl_c <- data %>%
        select(c, d) %>%
        tbl_summary(
          by = c,
          missing = "no",
          label = list(d = "变量d"),
          statistic = list(all_categorical() = "{n} ({p}%)")
        ) %>%
        modify_header(label = "**c分组**")
      
      # 合并两个表格
      tbl_merge(tbls = list(tbl_b, tbl_c))
    },
    .header = "**a={strata}**, N = {n}"
  )

说明:

  • 在每个分层的.tbl_fun中处理当前子集数据,确保d的统计是基于当前分层的样本
  • 通过modify_header优化表格标题的可读性
  • 保留了分层变量a的标题和对应样本数,同时独立展示b和c的分组统计列

内容的提问来源于stack exchange,提问作者Edge

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.25 01:11:10