You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R语言箱线图的X轴仅重排指定Disease类别?

问题与解决方案

问题描述

现有一个名为test的DataFrame,结构如下:

dput(test)

structure(list(Groups = c("Group1", "Group2", "Group3", "Group4", 
"Group5", "Group6", "Group7", "Group8", "Group9", "Group10", 
"Group11", "Group12", "Group13", "Group14", "Group15", "Group16", 
"Group17", "Group18", "Group19", "Group20", "Group21", "Group22"
), Disease = c("Brain", "Brain", "Brain", "Blood", "Blood", "Esophagus", 
"Esophagus", "Esophagus", "Brain", "Brain", "OE", "control", 
"OE", "control", "OE", "control", "PE", "PE_X", "PE_X", "PE", 
"PE_X", "PE"), variable = c("Name", "Name", "Name", "Name", "Name", 
"Name", "Name", "Name", "Name", "Name", "Name", "Name", "Name", "Name", 
"Name", "Name", "Name", "Name", "Name", "Name", "Name", "Name"), 
value = c(1.079825876, 0.236961206, 0.286498286, 0.374978442, 
3.620160544, 1.876875376, 0.293402656, 0.176208121, 0.622282653, 
1.373705338, 9.235592994, 1.437889832, 8.70900915, 1.772903362, 
9.070885831, 1.792899823, 10.29580836, 1.373281466, 4.210242765, 
0, 7.331498976, 14.11415563)), class = "data.frame", row.names = c(NA, 
-22L))

已通过以下代码绘制箱线图:

ggplot(data= subset(test, variable == "Name")) + 
  geom_boxplot(aes(x=Disease, y=value, fill=Disease), outlier.shape=NA) +
  geom_jitter(aes(x=Disease, y=value, fill=Disease), position=position_dodge(0.2)) +
  theme_classic(base_size = 12) + xlab("") + ylab("value")

此前尝试过按value列重排X轴的Disease类别:

ggplot(data= test) + 
  geom_boxplot(aes(x=reorder(Disease,value, na.rm=TRUE), y=value, fill=Disease), outlier.shape=NA) +
  geom_jitter(aes(x=Disease, y=value, fill=Disease), position=position_dodge(0.2)) +
  theme_classic(base_size = 12) + xlab("") + ylab("value")

需求:仅对Blood、Brain、Esophagus这几类Disease按value重排X轴顺序,其余类别保留在图表右侧。

解决方案

方法一:用forcats包自动调整因子顺序

通过标记分组、计算统计量,再重新设定因子水平:

# 加载tidyverse(包含dplyr和forcats)
library(tidyverse)

# 处理数据,调整Disease的因子顺序
test <- test %>%
  # 标记需要排序的类别和其余类别
  mutate(sort_flag = ifelse(Disease %in% c("Blood", "Brain", "Esophagus"), "to_sort", "others")) %>%
  # 计算每个Disease类别的value均值,作为排序依据
  group_by(Disease) %>%
  mutate(mean_val = mean(value, na.rm = TRUE)) %>%
  ungroup() %>%
  # 重新设定因子水平:先排需要排序的类别(按均值),再排其余类别
  mutate(Disease = fct_reorder2(Disease, sort_flag, mean_val, .desc = c(FALSE, TRUE)))

# 绘制图表
ggplot(data= subset(test, variable == "Name")) + 
  geom_boxplot(aes(x=Disease, y=value, fill=Disease), outlier.shape=NA) +
  geom_jitter(aes(x=Disease, y=value, fill=Disease), position=position_dodge(0.2)) +
  theme_classic(base_size = 12) + 
  xlab("") + 
  ylab("value")

方法二:手动指定因子水平(更直观)

手动筛选并排序目标类别,再合并其余类别:

# 加载dplyr
library(dplyr)

# 提取需要排序的类别,按value均值排序
sorted_cats <- test %>%
  filter(Disease %in% c("Blood", "Brain", "Esophagus")) %>%
  group_by(Disease) %>%
  summarise(mean_val = mean(value, na.rm = TRUE)) %>%
  arrange(mean_val) %>%  # 改为arrange(desc(mean_val))可实现降序
  pull(Disease)

# 提取其余类别,保留原始唯一顺序
other_cats <- test %>%
  filter(!Disease %in% sorted_cats) %>%
  pull(Disease) %>%
  unique()

# 将Disease转为因子,指定顺序:先排排序后的目标类别,再排其余类别
test$Disease <- factor(test$Disease, levels = c(sorted_cats, other_cats))

# 绘制图表
ggplot(data= subset(test, variable == "Name")) + 
  geom_boxplot(aes(x=Disease, y=value, fill=Disease), outlier.shape=NA) +
  geom_jitter(aes(x=Disease, y=value, fill=Disease), position=position_dodge(0.2)) +
  theme_classic(base_size = 12) + 
  xlab("") + 
  ylab("value")

说明

  • 两种方法都能实现需求:目标类别按value的均值排序后放在左侧,其余类别在右侧。
  • 若需要按中位数或其他统计量排序,只需将代码中的mean(value, na.rm = TRUE)替换为median(value, na.rm = TRUE)等即可。
  • 方法二更灵活,可直接调整sorted_cats的排序逻辑(升序/降序),也可手动修改other_cats的顺序。

内容的提问来源于stack exchange,提问作者beginner

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.15 17:17:32