如何正确绘制geom_errorbar()以匹配分组的最小值/最大值
问题
现有包含预测值与实际结果的数据集,预测值存在数值范围,已通过stat_summary绘制出均值。当前使用geom_errorbar()只能展示全组的数值范围,需求为:
- 仅为
Type为Predictions的预测值显示误差棒 - 误差棒需展示**对应日期+对应分区(Zona)**的最小值与最大值
当前代码:
x %>% ggplot(aes(x = Fecha, y = MC_Clientes, color = Type, group = Zona)) + geom_point() + stat_summary(geom = "point", fun = "mean", col = "black", size = 3, shape = 24, fill = "red") + geom_errorbar(aes(ymin = min(MC_Clientes), ymax = max(MC_Clientes)), width = 0.2) + #geom_hline(aes(yintercept = mean(MC_Clientes)), color="blue") + facet_wrap(~ Zona)
数据集:
x = structure(list(Fecha = structure(c(18993, 19448, 19723, 19631, 19997, 19083, 19631, 19174, 19631, 19539, 19905, 18993, 19083, 19266, 19448, 19083, 19358, 19723, 19539, 19358, 19358, 19814, 19266, 17987, 19723, 19723, 19083, 19723, 17987, 18993, 18993, 19174, 19814, 19905, 19997, 19539, 19266, 19358, 19905, 18993, 19723, 18993, 19174, 19814, 19448, 19358, 18262, 19266, 19083, 18536, 18628, 19448, 19083, 19997, 19997, 19723, 19448, 19723, 19997, 19266), class = "Date"), MC_Clientes = c(`10` = 3.30914630982938, `163` = 3.40705102004165, `148` = 3.27210950513308, `225` = 3.30737187089665, `114` = 3.27216488807394, `156` = 3.29952915488204, `110` = 3.39960342129031, `65` = 3.44020887082005, `226` = 3.3477486021078, `72` = 3.37828428718909, `89` = 3.35889902349877, `130` = 3.24841711499344, `33` = 3.39557631661649, `96` = 3.30867139996011, `48` = 3.29220186390159, `39` = 3.38775370812839, `11` = 3.33638404617856, `28` = 3.34128523097663, `196` = 3.32642364225741, `20` = 3.30914630982938, `15` = 3.32250928717776, `59` = 3.38775370812839, `215` = 3.30737187089665, 3.23830671159728, `143` = 3.262723490499, `144` = 3.28580416922959, `154` = 3.19317306859439, `146` = 3.22387486690392, 3.50997312704934, `125` = 3.2916234598844, `128` = 3.27210950513308, `68` = 3.42813783609873, `57` = 3.36530584016227, `204` = 3.43236203716387, `118` = 3.3861339757903, `192` = 3.4510934126958, `211` = 3.26617681913074, `140` = 3.24841711499344, `87` = 3.43574132492634, `2` = 3.30010088087279, `27` = 3.33539993355129, `129` = 3.27777616345452, `184` = 3.43236203716387, `55` = 3.31622658677549, `42` = 3.34351801788172, `132` = 3.28169122393979, 3.18803129400015, `92` = 3.35020429586366, `157` = 3.39074839564108, 3.11563497015824, 3.40541198110811, `166` = 3.29952915488204, `38` = 3.29220186390159, `111` = 3.25522990668431, `113` = 3.26178511373003, `150` = 3.24841711499344, `167` = 3.39074839564108, `22` = 3.30010088087279, `233` = 3.40771653641682, `214` = 3.26936263234189), Zona = c("A", "B", "B", "B", "A", "B", "A", "A", "B", "A", "A", "B", "A", "A", "A", "A", "A", "A", "B", "A", "A", "A", "B", "B", "B", "B", "B", "B", "A", "B", "B", "A", "A", "B", "A", "B", "B", "B", "A", "A", "A", "B", "B", "A", "A", "B", "B", "A", "B", "B", "A", "B", "A", "A", "A", "B", "B", "A", "B", "B"), Type = c("Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Actual", "Predictions", "Predictions", "Predictions", "Predictions", "Actual", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Actual", "Predictions", "Predictions", "Actual", "Actual", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions", "Predictions")), row.names = c(NA, -60L), class = c("tbl_df", "tbl", "data.frame"))
解决方案
方法1:预先汇总数据(逻辑清晰,推荐)
先过滤出预测值数据,按Zona和Fecha分组计算每组的最小值、最大值和均值,再用于绘图:
library(dplyr) library(ggplot2) # 汇总预测值的分组统计数据 pred_summary <- x %>% filter(Type == "Predictions") %>% group_by(Zona, Fecha) %>% summarise( mean_mc = mean(MC_Clientes), min_mc = min(MC_Clientes), max_mc = max(MC_Clientes), .groups = "drop" ) # 绘图 x %>% ggplot(aes(x = Fecha, y = MC_Clientes, color = Type)) + geom_point() + # 添加与误差棒匹配的均值点 geom_point(data = pred_summary, aes(y = mean_mc), col = "black", size = 3, shape = 24, fill = "red") + # 仅为预测值添加分组误差棒 geom_errorbar(data = pred_summary, aes(y = mean_mc, ymin = min_mc, ymax = max_mc), width = 0.2, color = "black") + facet_wrap(~ Zona)
方法2:直接用stat_summary生成误差棒
无需预先处理数据,通过stat_summary指定计算逻辑,同时过滤预测值:
x %>% ggplot(aes(x = Fecha, y = MC_Clientes, color = Type)) + geom_point() + # 绘制所有数据的均值点(如需仅展示预测值均值,可添加filter逻辑) stat_summary(geom = "point", fun = "mean", col = "black", size = 3, shape = 24, fill = "red") + # 仅为预测值生成分组误差棒 stat_summary(data = . %>% filter(Type == "Predictions"), geom = "errorbar", fun.min = min, fun.max = max, width = 0.2, color = "black") + facet_wrap(~ Zona)
关键说明
- 过滤预测值:通过
filter(Type == "Predictions")确保误差棒仅针对预测值数据 - 分组计算:按
Zona和Fecha分组,保证误差棒是对应日期+分区的数值范围,而非全组全局范围 - 数据匹配:方法1中用汇总数据的均值点和误差棒配对,避免出现均值与误差棒不匹配的情况
内容的提问来源于stack exchange,提问作者user113156
相关产品推荐
相关产品推荐

