如何在R的ggplot2中按指定类别排序李克特量表图
问题
我在R中有一个名为df的李克特量表响应数据框,已经用ggplot2绘制了对应的李克特图。现在需要调整图表排序:将图中的item先按「Very Dissatisfied」类别占比从低到高排序,若该占比相同则按「Dissatisfied」类别占比从低到高排序。
数据预览
df # A tibble: 90 × 3 # Groups: item [18] item Response Percentage <chr> <fct> <dbl> 1 A Very Dissatisfied 25 2 A Dissatisfied 25 3 A Average 33.3 4 A Satisfied 11.1 5 A Very Satisfied 55.6 6 B Very Dissatisfied 25 7 B Dissatisfied 25 8 B Average 44.4 9 B Satisfied 25 10 B Very Satisfied 55.6 # ℹ 80 more rows # ℹ Use `print(n = ...)` to see more rows
现有绘图代码
response_mapping <- c("Very Dissatisfied" = 1, "Dissatisfied" = 2, "Average" = 3, "Satisfied" = 4, "Very Satisfied" = 5) # Apply the mapping and calculate the sign data_f_sum <- df %>% ungroup() %>% mutate(res.sgn = sign(response_mapping[as.character(Response)] - 3)) %>% summarise(sum.prcnt = sum(Percentage), .by = c(item, res.sgn)) data_f_sum likert_levels = c("Very Dissatisfied", "Dissatisfied" , "Average" , "Satisfied", "Very Satisfied") df = df%>% mutate(Response = factor(Response , levels = likert_levels)) ggplot(data = df, aes(Percentage, item, fill = Response)) + geom_col(position = position_likert()) + scale_x_continuous(labels = ggstats::label_percent_abs()) + geom_label(data = data_f_sum, aes(label = sprintf("%.1f", sum.prcnt), y = item, x = res.sgn), alpha = 0.3, inherit.aes = FALSE) + coord_cartesian(xlim = c(-1, 1)) + scale_fill_brewer(type = "div", palette = "RdYlGn") + theme_bw()+ theme(legend.position = "bottom")
数据结构
structure(list(item = c("A", "A", "A", "A", "A", "B", "B", "B", "B", "B", "C", "C", "C", "C", "C", "D", "D", "D", "D", "D", "E", "E", "E", "E", "E", "F", "F", "F", "F", "F", "G", "G", "G", "G", "G", "H", "H", "H", "H", "H", "I", "I", "I", "I", "I", "J", "J", "J", "J", "J", "K", "K", "K", "K", "K", "L", "L", "L", "L", "L", "M", "M", "M", "M", "M", "N", "N", "N", "N", "N", "O", "O", "O", "O", "O", "P", "P", "P", "P", "P", "Q", "Q", "Q", "Q", "Q", "R", "R", "R", "R", "R"), Response = structure(c(1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L, 1L, 2L, 3L, 4L, 5L), levels = c("Very Dissatisfied", "Dissatisfied", "Average", "Satisfied", "Very Satisfied"), class = "factor"), Percentage = c(25, 25, 33.3, 11.1, 55.6, 25, 25, 44.4, 25, 55.6, 25, 25, 22.2, 33.3, 44.4, 25, 25, 33.3, 11.1, 55.6, 25, 22.2, 11.1, 11.1, 55.6, 25, 25, 44.4, 11.1, 44.4, 25, 25, 11.1, 33.3, 55.6, 25, 25, 33.3, 22.2, 44.4, 25, 25, 11.1, 33.3, 55.6, 25, 25, 22.2, 22.2, 55.6, 25, 25, 11.1, 11.1, 77.8, 25, 25, 11.1, 33.3, 55.6, 25, 25, 33.3, 25, 66.7, 25, 25, 33.3, 11.1, 55.6, 25, 11.1, 25, 33.3, 55.6, 25, 25, 22.2, 22.2, 55.6, 25, 22.2, 22.2, 11.1, 44.4, 25, 11.1, 22.2, 11.1, 55.6)), class = c("grouped_df", "tbl_df", "tbl", "data.frame" ), row.names = c(NA, -90L), groups = structure(list(item = c("A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R"), .rows = structure(list(1:5, 6:10, 11:15, 16:20, 21:25, 26:30, 31:35, 36:40, 41:45, 46:50, 51:55, 56:60, 61:65, 66:70, 71:75, 76:80, 81:85, 86:90), ptype = integer(0), class = c("vctrs_list_of", "vctrs_vctr", "list"))), class = c("tbl_df", "tbl", "data.frame" ), row.names = c(NA, -18L), .drop = TRUE))
解决方案
要实现指定的排序逻辑,核心是把item转换为有序因子,按照「Very Dissatisfied」占比升序、「Dissatisfied」占比升序的规则定义因子水平顺序。具体步骤如下:
- 从原数据中提取每个
item对应的「Very Dissatisfied」和「Dissatisfied」的占比 - 按照排序规则对
item进行排序,得到目标顺序 - 将原数据中的
item转换为因子,指定排序后的水平 - 用修改后的数据绘图即可
修改后的完整代码
library(tidyverse) library(ggplot2) library(ggstats) # 定义响应映射和李克特水平 response_mapping <- c("Very Dissatisfied" = 1, "Dissatisfied" = 2, "Average" = 3, "Satisfied" = 4, "Very Satisfied" = 5) likert_levels = c("Very Dissatisfied", "Dissatisfied" , "Average" , "Satisfied", "Very Satisfied") # 处理数据:提取排序所需的占比,重新排序item为有序因子 df_sorted <- df %>% ungroup() %>% mutate(Response = factor(Response, levels = likert_levels)) %>% # 提取每个item的Very Dissatisfied和Dissatisfied占比 pivot_wider(names_from = Response, values_from = Percentage) %>% # 按照指定规则排序:先Very Dissatisfied升序,再Dissatisfied升序 arrange(`Very Dissatisfied`, `Dissatisfied`) %>% # 提取排序后的item顺序 pull(item) %>% # 将原数据的item转换为因子,指定排序后的水平 {mutate(df, item = factor(item, levels = .))} %>% # 重新分组(可选,保持原数据结构) group_by(item) # 计算数据总和标签(原逻辑不变) data_f_sum <- df_sorted %>% ungroup() %>% mutate(res.sgn = sign(response_mapping[as.character(Response)] - 3)) %>% summarise(sum.prcnt = sum(Percentage), .by = c(item, res.sgn)) # 绘制排序后的李克特图 ggplot(data = df_sorted, aes(Percentage, item, fill = Response)) + geom_col(position = position_likert()) + scale_x_continuous(labels = ggstats::label_percent_abs()) + geom_label(data = data_f_sum, aes(label = sprintf("%.1f", sum.prcnt), y = item, x = res.sgn), alpha = 0.3, inherit.aes = FALSE) + coord_cartesian(xlim = c(-1, 1)) + scale_fill_brewer(type = "div", palette = "RdYlGn") + theme_bw()+ theme(legend.position = "bottom")
关键部分解释
- 使用
pivot_wider把每个item的各类响应占比转换为宽格式,方便提取「Very Dissatisfied」和「Dissatisfied」的数值 arrange(Very Dissatisfied,Dissatisfied)实现了先按前者升序、再按后者升序的排序逻辑- 通过
factor(item, levels = .)把item转换为有序因子,ggplot会严格按照因子水平的顺序绘制y轴项目
内容的提问来源于stack exchange,提问作者Homer Jay Simpson
相关产品推荐
相关产品推荐

