使用facet_wrap时如何为分组箱线图正确添加Tukey标记字母
问题:箱线图Tukey显著性标记字母无法对应到处理组上方
我正在绘制光合数据可视化图:
- x轴为植物发育阶段,y轴为平均产量
- 通过
facet_wrap(~Line.x)生成4张子图,分别展示4种植物品系的产量变化 - 每个品系在每个发育阶段设置5种处理(
Day.Night.C.x),已完成ANOVA和Tukey多重比较,需为箱线图添加显著性标记字母
当前已成功让Tukey字母显示在对应子图和发育阶段区域,但字母未定位到对应处理的箱线图正上方,需要调整geom_text代码逻辑。
当前代码
ggplot(data_new, aes(DevelopmentStage, A.µmol.m...s..., fill=Day.Night.C.x)) + geom_boxplot() + labs(x="Line (plant), development stage, and treatment", y="Average yeild") + theme_bw() + theme(panel.grid.major = element_blank(), panel.grid.minor = element_blank(), text=element_text(size=20)) + geom_text(data = A.line.tukey.allstages, aes(x = DevelopmentStage, y = quant, label = Tukey), size = 5, vjust=-1.25, hjust =-1)+ facet_wrap(~Line.x)+ labs(title='Effect of treatments on net yield' , caption = 'Groups sharing the same letters are not statistically different from one another at the p=0.05 level', fill= 'Treatment')
数据示例子集
structure(list(Yield = c(8.151375452, 8.382517873, 8.44283403, 10.31881757, 10.3557304, 9.928674142, 10.33451677, 10.69507697, 10.57664686, 11.16575551, 11.04700581, 11.23337648, 6.543935774, 6.572356108, 7.345439724, 12.53070112, 12.25825808, 12.3404997, 13.23889684, 13.2798362, 12.49735692, 11.87736011, 11.85216308, 11.59254363, 9.284810253, 9.011053873, 9.173928237, 14.05942045, 13.78539868, 14.09753387, 13.79196207, 13.27845743, 13.42134734, 13.41913518, 15.27662039, 15.10059174, 11.15805372, 12.39770032, 12.45950119, 11.94798457, 12.61388394, 12.14116668, 12.21064804, 6.69254258, 12.52026321, 12.88836323, 13.60539121, 13.22586637), Days.after.sow.x = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 5L, 5L, 5L, 5L, 5L, 5L, 5L, 5L, 5L, 5L, 5L, 5L), .Label = c("21", "25", "28", "32", "35", "42"), class = "factor"), Day.Night.C.x = structure(c(2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L), .Label = c("25/15", "20/25", "25/25", "35/20", "35/25"), class = "factor"), Rep = structure(c(1L, 2L, 3L, 1L, 2L, 3L, 1L, 2L, 3L, 1L, 2L, 3L, 2L, 3L, 4L, 5L, 6L, 7L, 17L, 18L, 19L, 2L, 3L, 4L, 1L, 2L, 3L, 4L, 5L, 6L, 1L, 2L, 3L, 9L, 10L, 11L, 3L, 4L, 5L, 6L, 7L, 8L, 1L, 2L, 3L, 1L, 2L, 3L), .Label = c("1", "2", "3", "4", "5", "6", "7", "8", "9", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "28", "29", "30", "31", "32", "33", "34", "35", "36"), class = "factor"), Line.x = structure(c(4L, 4L, 4L, 2L, 2L, 2L, 3L, 3L, 3L, 1L, 1L, 1L, 4L, 4L, 4L, 2L, 2L, 2L, 3L, 3L, 3L, 1L, 1L, 1L, 4L, 4L, 4L, 2L, 2L, 2L, 3L, 3L, 3L, 1L, 1L, 1L, 4L, 4L, 4L, 2L, 2L, 2L, 3L, 3L, 3L, 1L, 1L, 1L), .Label = c("Manatee", "60183", "Bambino", "60179"), class = "factor"), DevelopmentStage = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 4L, 4L, 4L, 4L, 4L, 4L, 4L, 4L, 4L, 4L, 4L, 4L), .Label = c("Seedling", "Juvenile", "Mature", "Harvest"), class = "factor")), row.names = c(1L, 2L, 3L, 37L, 38L, 39L, 73L, 74L, 75L, 109L, 110L, 111L, 146L, 147L, 148L, 185L, 186L, 187L, 233L, 234L, 235L, 254L, 255L, 256L, 289L, 290L, 291L, 328L, 329L, 330L, 361L, 362L, 363L, 405L, 406L, 407L, 435L, 436L, 437L, 474L, 475L, 476L, 505L, 506L, 507L, 541L, 542L, 543L), class = "data.frame")
解决方案
核心原因
当前geom_text未绑定Day.Night.C.x处理变量,无法识别同一发育阶段下的不同处理分组,导致字母位置偏移。
调整步骤
确保Tukey数据包含完整分组信息:
A.line.tukey.allstages必须包含Line.x、DevelopmentStage、Day.Night.C.x三个分组变量,以及每个处理组对应的y轴定位值(箱线图最大值+偏移量)。匹配箱线图的分组偏移:使用
position_dodge让geom_text和geom_boxplot的分组位置对齐,代码调整如下:
ggplot(data_new, aes(DevelopmentStage, A.µmol.m...s..., fill=Day.Night.C.x)) + # 设置箱线图的宽度和偏移量,后续geom_text要匹配此值 geom_boxplot(width = 0.7, position = position_dodge(width = 0.8)) + labs(x="品系、发育阶段与处理", y="平均产量") + theme_bw() + theme(panel.grid.major = element_blank(), panel.grid.minor = element_blank(), text=element_text(size=20)) + # 关键调整:添加group=Day.Night.C.x,并用position_dodge匹配箱线图偏移 geom_text(data = A.line.tukey.allstages, aes(x = DevelopmentStage, y = quant, label = Tukey, group=Day.Night.C.x), size = 5, vjust=-0.5, position = position_dodge(width = 0.8)) + # 和geom_boxplot的偏移宽度一致 facet_wrap(~Line.x)+ labs(title='处理对净产量的影响' , caption = '共享相同字母的组在p=0.05水平上无统计学差异', fill= '处理')
补充:计算精准的y轴定位值
如果你的quant是全局最大值,建议重新计算每个处理组的最大值加偏移,避免字母与箱线图重叠:
library(dplyr) # 计算每个分组的产量最大值,合并到Tukey结果中 A.line.tukey.allstages <- A.line.tukey.allstages %>% left_join( data_new %>% group_by(Line.x, DevelopmentStage, Day.Night.C.x) %>% summarise(max_y = max(A.µmol.m...s...), .groups="drop"), by = c("Line.x", "DevelopmentStage", "Day.Night.C.x") ) %>% mutate(quant = max_y + 0.5) # 0.5为偏移量,可根据y轴范围调整
内容的提问来源于stack exchange,提问作者Hannah Mather
相关产品推荐
相关产品推荐

