You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

在R中循环执行成对比较时出现pull()报错求助

R代码循环中pairwise_t_test报错问题解析

问题描述

运行R代码时,anova_test与get_anova_table函数可正常执行,但pairwise_t_test函数抛出如下错误:

Error in pull():
! Can't extract columns that don't exist.
✖ Column column doesn't exist.

单独针对特定列执行非循环版本代码时无此问题,疑问:为何pairwise_t_test的列指定方式与anova的dv参数不一致会引发错误?

相关代码

library(rstatix)

nnames <- names(df)[unlist(lapply(df, is.numeric))]
res.aov <- list()
aov_tab <- list()
pc <- list()
pc1 <- list()

for (column in nnames) {
  res.aov[[column]] <- anova_test(data = df, dv = column, 
                                  wid = `Subject`, within = `Timepoint`, between = `Genotype`)
  aov_tab[[column]] <- get_anova_table(res.aov[[column]])
  
  pc[[column]]<- df %>% pairwise_t_test(column ~`Timepoint`, paired=TRUE, p.adjust.method = "holm")
  pc[[column]]<- pc[[column]] %>% add_xy_position(x="Timepoint")
  
  pc1[[column]]<- df %>% group_by(Timepoint) %>% pairwise_t_test(column ~ `Genotype`)
  pc1[[column]]<- pc1[[column]] %>% add_xy_position(x= "Timepoint")
  } 

数据结构

dput(df)
structure(list(Subject = c("ASCVD002", "ASCVD002", "ASCVD002", 
"ASCVD003", "ASCVD003", "ASCVD003", "ASCVD004", "ASCVD004", "ASCVD004", 
"ASCVD005", "ASCVD005", "ASCVD005", "ASCVD006", "ASCVD006", "ASCVD006", 
"ASCVD008", "ASCVD008", "ASCVD008", "ASCVD009", "ASCVD009", "ASCVD009", 
"ASCVD010", "ASCVD010", "ASCVD010", "ASCVD011", "ASCVD011", "ASCVD011"
), Timepoint = c("0", "0.25", "0.5", "0", "0.25", "0.5", "0", 
"0.25", "0.5", "0", "0.25", "0.5", "0", "0.25", "0.5", "0", "0.25", 
"0.5", "0", "0.25", "0.5", "0", "0.25", "0.5", "0", "0.25", "0.5"
), Genotype = c("Heterozygote", "Heterozygote", "Heterozygote", 
"Heterozygote", "Heterozygote", "Heterozygote", "Heterozygote", 
"Heterozygote", "Heterozygote", "GG", "GG", "GG", "AA", "AA", 
"AA", "GG", "GG", "GG", "AA", "AA", "AA", "AA", "AA", "AA", "GG", 
"GG", "GG"), `Tregs CD127lo CD25+` = c(2702, 2175, 2651, 1672.8, 
3762, 4264, 1975, 3208, 3285, 3457, 3383, 2619.9, 11872, 16101, 
13443, 3935, 1894, 2297, 7385, 8901, 9522, 7100, 8789, 9309, 
371, 379, 514), `Monocytes % of Live by Size` = c(1.38, 2.66, 
4.74, 5.83, 3.9, 5.06, 6.36, 3.45, 2.64, 6.33, 10.7, 9.41, 3.42, 
3.46, 2.73, 2.38, 3.12, 4.44, 5.31, 3.59, 4.91, 1.53, 6.54, 4.85, 
6.87, 3.66, 5.07), `NK cells` = c(90.62, 153.6, 159.8, 88, 118, 
159, 74, 82, 64, 30, 344, 73, 29, 198, 79, 145, 258, 307, 30, 
74.4, 0, 47.3, 32, 0, 52.6, 95.3, 51.7)), class = c("tbl_df", 
"tbl", "data.frame"), row.names = c(NA, -27L))

错误原因

anova_test的dv参数支持直接传入字符串形式的列名,内部会自动映射到数据中对应的列;但pairwise_t_test的公式参数遵循标准R公式解析逻辑,会将column当作字面列名处理——循环中column是存储实际列名的变量(比如"Tregs CD127lo CD25+"),但公式column ~ Timepoint会被解析为寻找名为column的列,而数据中不存在该列,因此触发报错。

解决方案

提供两种可靠的修改方式,实现动态列名的正确引用:

方式1:用reformulate动态生成公式

通过reformulate函数根据循环变量动态创建公式对象,替代硬编码的公式:

library(rstatix)

nnames <- names(df)[unlist(lapply(df, is.numeric))]
res.aov <- list()
aov_tab <- list()
pc <- list()
pc1 <- list()

for (column in nnames) {
  res.aov[[column]] <- anova_test(data = df, dv = column, 
                                  wid = `Subject`, within = `Timepoint`, between = `Genotype`)
  aov_tab[[column]] <- get_anova_table(res.aov[[column]])
  
  # 动态构建Timepoint配对t检验的公式
  formula_pc <- reformulate("Timepoint", response = column)
  pc[[column]] <- df %>% pairwise_t_test(formula_pc, paired=TRUE, p.adjust.method = "holm") %>%
    add_xy_position(x="Timepoint")
  
  # 动态构建Genotype组间t检验的公式
  formula_pc1 <- reformulate("Genotype", response = column)
  pc1[[column]] <- df %>% group_by(Timepoint) %>% 
    pairwise_t_test(formula_pc1) %>%
    add_xy_position(x= "Timepoint")
}

方式2:用.data代词引用列

使用tidyverse的.data代词直接引用循环变量对应的列:

library(rstatix)

nnames <- names(df)[unlist(lapply(df, is.numeric))]
res.aov <- list()
aov_tab <- list()
pc <- list()
pc1 <- list()

for (column in nnames) {
  res.aov[[column]] <- anova_test(data = df, dv = column, 
                                  wid = `Subject`, within = `Timepoint`, between = `Genotype`)
  aov_tab[[column]] <- get_anova_table(res.aov[[column]])
  
  pc[[column]] <- df %>% pairwise_t_test(.data[[column]] ~ `Timepoint`, paired=TRUE, p.adjust.method = "holm") %>%
    add_xy_position(x="Timepoint")
  
  pc1[[column]] <- df %>% group_by(Timepoint) %>% 
    pairwise_t_test(.data[[column]] ~ `Genotype`) %>%
    add_xy_position(x= "Timepoint")
}

补充说明

anova_test的dv参数设计为兼容字符串输入,是函数内部做了特殊处理;而pairwise_t_test依赖标准公式接口,必须显式处理动态列名的引用,这就是两者表现差异的核心原因。

内容的提问来源于stack exchange,提问作者swats15

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.08 23:45:34