You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用gtsummary制作带自定义排序与分组缩进的描述性表格

实现带层级缩进的二分类事件数据集汇总表格

需求说明

数据集包含主事件(列名以inf_<n>标识)和子事件(列名以sub_inf_<n>_t<n>标识)两类二分类变量,要求生成的汇总表格满足:

  • 主事件按"Yes"频率降序排列,other主事件固定放在末尾
  • 子事件缩进显示在对应主事件的下方

现有代码

set.seed(123)
library(tidyverse)
library(gtsummary)
library(kableExtra)
df <- tidyr::tibble(
  id = c(1,2,3,4,5,6,7,8,9,10),
  inf_1 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "IA"),
  inf_2 = structure(factor(rep_len(0, length.out = 10), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "PNEU"),
  inf_3 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "BSI"),
  inf_4 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "other"),
  sub_inf_1_t1 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "lower"),
  sub_inf_1_t2 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "upper"),
  sub_inf_1_t3 = structure(factor(c(0,0,0,1,1,0,NA,NA,0,1), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "other"),
)

df %>%
  mutate(across(where(is.factor), ~ forcats::fct_recode(.x, "No" = "Unchecked", "Yes" = "Checked"))) %>%
  gtsummary::tbl_summary(data = ., include = -id, statistic = list(all_categorical() ~ "{n} / {N} ({p}%)")) %>%
  kable()

期望表格效果

CharacteristicN = 10
IA5 / 10 (50%)
 lower5 / 10 (50%)
 upper8 / 10 (80%)
 other3 / 8 (38%)
 Unknown2
BSI2 / 10 (20%)
PNEU0 / 10 (0%)
other3 / 10 (30%)

解决方案代码

set.seed(123)
library(tidyverse)
library(gtsummary)
library(kableExtra)

df <- tidyr::tibble(
  id = c(1,2,3,4,5,6,7,8,9,10),
  inf_1 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "IA"),
  inf_2 = structure(factor(rep_len(0, length.out = 10), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "PNEU"),
  inf_3 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "BSI"),
  inf_4 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "other"),
  sub_inf_1_t1 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "lower"),
  sub_inf_1_t2 = structure(factor(sample(seq(0,1), size = 10, replace = TRUE), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "upper"),
  sub_inf_1_t3 = structure(factor(c(0,0,0,1,1,0,NA,NA,0,1), levels = c(0,1), labels = c("Unchecked", "Checked")), label = "other"),
)

# 重新编码因子水平
df <- df %>%
  mutate(across(where(is.factor), ~ forcats::fct_recode(.x, "No" = "Unchecked", "Yes" = "Checked")))

# 1. 计算主事件频率,确定排序顺序(按Yes占比降序,other主事件放最后)
main_event_cols <- df %>% select(starts_with("inf_")) %>% colnames()
main_event_order <- df %>%
  select(all_of(main_event_cols)) %>%
  summarise(across(everything(), ~ sum(.x == "Yes", na.rm = TRUE) / n())) %>%
  pivot_longer(everything(), names_to = "col", values_to = "freq") %>%
  mutate(label = map_chr(col, ~ attr(df[[.x]], "label"))) %>%
  arrange(desc(freq), label != "other") %>%
  pull(col)

# 2. 构建完整的列顺序:主事件 + 对应子事件
column_order <- c()
for (main_col in main_event_order) {
  column_order <- c(column_order, main_col)
  # 匹配当前主事件对应的所有子事件列
  sub_col_prefix <- str_replace(main_col, "inf_", "sub_inf_")
  sub_cols <- df %>% select(starts_with(sub_col_prefix)) %>% colnames()
  if (length(sub_cols) > 0) {
    column_order <- c(column_order, sub_cols)
  }
}

# 3. 生成汇总表格并调整层级缩进和行顺序
final_tbl <- df %>%
  tbl_summary(
    include = all_of(column_order),
    statistic = list(all_categorical() ~ "{n} / {N} ({p}%)"),
    missing = "no"  # 先不自动生成缺失行,后续手动处理
  ) %>%
  # 给子事件标签添加缩进
  modify_table_body(
    ~ .x %>%
      mutate(
        label = case_when(
          str_starts(variable, "sub_inf_") ~ paste0("&emsp;", label),
          TRUE ~ label
        )
      )
  ) %>%
  # 手动添加sub_inf_1_t3的缺失值行
  modify_table_body(
    ~ bind_rows(
      .x,
      tibble(
        variable = "sub_inf_1_t3",
        label = "&emsp;Unknown",
        stat_0 = "2",
        row_type = "level",
        var_type = "categorical"
      )
    ) %>%
      # 按预设列顺序和行类型排序,确保子事件紧跟主事件
      arrange(match(variable, column_order), row_type == "label")
  ) %>%
  # 转换为kable格式并调整样式
  as_kable_extra() %>%
  kable_styling(full_width = FALSE)

print(final_tbl)

关键步骤说明

  1. 主事件排序:计算每个主事件的"Yes"占比,按占比降序排列,同时将other主事件固定到末尾
  2. 列顺序构建:遍历排序后的主事件,将对应子事件列紧跟在主事件列之后
  3. 层级缩进:通过modify_table_body给子事件的标签添加HTML缩进符&emsp;
  4. 缺失值处理:手动添加子事件的缺失值统计行,匹配期望的表格格式

内容的提问来源于stack exchange,提问作者devster

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.15 17:44:54