You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R ggplot2双分面图表中按因子水平值排序X轴?

问题描述

我正在制作堆叠条形图,展示20名参与者完成两项相同任务(task)时,使用三种策略(CA.presence:Presence、Possible presence、Absence)的时间占比(Duration),并按task进行分面。使用以下代码生成了图表:

CA_presence %>% ggplot(aes(fill=CA.presence, y= Duration, x= participant, label =    scales::percent(Duration))) + 
geom_bar(width = .7, position="fill", stat="identity") +
scale_y_continuous(labels = scales::label_percent()) +
facet_grid(~task) +
ggtitle("Presence of CA in the dataset") +
theme(plot.title = element_text(size = 15)) +
theme(plot.title = element_text(hjust = 0.5)) +
theme(legend.text = element_text(size = 10)) +
theme(strip.text.x = element_text(size = 13)) +
labs(x = "Participant",
     y = "Proportion of discourse time",
     fill = "CA presence") +
theme(axis.text = element_text(size = 10)) +
theme(axis.title = element_text(size = 10)) +
scale_fill_brewer(palette = "Greens") +
coord_flip()

数据集前几行的dput()输出:

structure(
  list(
    participant = c("L001", "L001", "L002", "L002", "L016", "L016"),
    task = c("T05", "T12", "T05", "T12", "T05", "T12"),
    language = c("French", "French", "French", "French", "French", "French"),
    Duration = c(8823, 46275, 2459, 38193, 20488, 160970),
    CA.presence = c("Presence", "Presence", "Presence", "Presence", "Presence", "Presence")
  ),
  class = c("grouped_df", "tbl_df", "tbl", "data.frame"),
  row.names = c(NA, -6L), groups = structure(
    list(
      participant = c("L001", "L001", "L002", "L002", "L016", "L016"),
      task = c("T05", "T12", "T05", "T12", "T05", "T12"),
      .rows = structure(list(
        1L, 2L, 3L, 4L, 5L, 6L
      ), ptype = integer(0), class = c(
        "vctrs_list_of",
        "vctrs_vctr", "list"
      ))
    ),
    class = c("tbl_df", "tbl", "data.frame"),
    row.names = c(NA, -6L), .drop = TRUE
  )
)

核心需求:如何在该双分面图表中按匹配因子水平的数值对X轴(participant)进行排序?


解决方案

要实现分面图表中参与者的有序排列,需要先将participant转换为有序因子,排序依据可根据需求选择(如某任务下特定策略的时长、总时长等),分两种场景处理:

场景1:所有分面使用统一排序(基于某一任务的数值)

比如我们选择T05任务中"Presence"策略的Duration作为排序依据,步骤如下:

1. 预处理数据,生成有序因子

library(dplyr)
library(ggplot2)

# 取消原数据的分组状态
CA_presence <- CA_presence %>% ungroup()

# 计算排序参考值:提取T05任务中Presence策略的参与者时长,按时长升序排列
sort_ref <- CA_presence %>%
  filter(task == "T05", CA.presence == "Presence") %>%
  arrange(Duration)  # 若需降序,改为arrange(desc(Duration))

# 将participant转换为有序因子,顺序匹配sort_ref中的排列
CA_presence <- CA_presence %>%
  mutate(participant = factor(participant, levels = sort_ref$participant, ordered = TRUE))

2. 绘制图表(复用原有代码)

此时participant已按指定顺序排序,直接运行原有绘图代码即可:

CA_presence %>% 
  ggplot(aes(fill = CA.presence, y = Duration, x = participant, label = scales::percent(Duration))) + 
  geom_bar(width = .7, position = "fill", stat = "identity") +
  scale_y_continuous(labels = scales::label_percent()) +
  facet_grid(~task) +
  ggtitle("Presence of CA in the dataset") +
  theme(plot.title = element_text(size = 15, hjust = 0.5),
        legend.text = element_text(size = 10),
        strip.text.x = element_text(size = 13),
        axis.text = element_text(size = 10),
        axis.title = element_text(size = 10)) +
  labs(x = "Participant",
       y = "Proportion of discourse time",
       fill = "CA presence") +
  scale_fill_brewer(palette = "Greens") +
  coord_flip()

场景2:每个分面独立排序(按对应任务的数值)

若需要T05和T12分面内的参与者各自按该任务下的数值排序,需借助ggh4x包实现自由刻度的分面:

1. 安装并加载依赖包

install.packages("ggh4x")
library(ggh4x)

2. 计算每个分面内的排序顺序

# 按任务分组,计算每个参与者在该任务下的总时长,生成排序序号
sort_by_task <- CA_presence %>%
  group_by(task, participant) %>%
  summarise(total_duration = sum(Duration), .groups = "drop") %>%
  group_by(task) %>%
  arrange(total_duration) %>%  # 按总时长升序,降序用desc(total_duration)
  mutate(order = row_number())

# 将排序序号合并到原数据
CA_presence_ordered <- CA_presence %>%
  left_join(sort_by_task, by = c("task", "participant"))

3. 绘制分面独立排序的图表

CA_presence_ordered %>%
  ggplot(aes(fill = CA.presence, y = Duration, x = order, label = scales::percent(Duration))) + 
  geom_bar(width = .7, position = "fill", stat = "identity") +
  scale_y_continuous(labels = scales::label_percent()) +
  # 使用facet_grid2实现自由x轴刻度
  facet_grid2(~task, scales = "free_x", space = "free_x") +
  # 将x轴刻度替换为参与者编号
  scale_x_continuous(breaks = sort_by_task$order, labels = sort_by_task$participant) +
  ggtitle("Presence of CA in the dataset") +
  theme(plot.title = element_text(size = 15, hjust = 0.5),
        legend.text = element_text(size = 10),
        strip.text.x = element_text(size = 13),
        axis.text = element_text(size = 10),
        axis.title = element_text(size = 10)) +
  labs(x = "Participant",
       y = "Proportion of discourse time",
       fill = "CA presence") +
  scale_fill_brewer(palette = "Greens") +
  coord_flip()

内容的提问来源于stack exchange,提问作者Sébastien

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 22:06:14