You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何正确绘制geom_errorbar()以匹配分组的最小值/最大值

问题

现有包含预测值与实际结果的数据集,预测值存在数值范围,已通过stat_summary绘制出均值。当前使用geom_errorbar()只能展示全组的数值范围,需求为:

  • 仅为Type为Predictions的预测值显示误差棒
  • 误差棒需展示**对应日期+对应分区(Zona)**的最小值与最大值

当前代码:

x %>%
  ggplot(aes(x = Fecha, y = MC_Clientes, color = Type, group = Zona)) +
  geom_point() +
  stat_summary(geom = "point", fun = "mean", col = "black", size = 3, shape = 24, fill = "red") +
  geom_errorbar(aes(ymin = min(MC_Clientes), ymax = max(MC_Clientes)), width = 0.2) +
  #geom_hline(aes(yintercept = mean(MC_Clientes)), color="blue") +
  facet_wrap(~ Zona)

数据集:

x = structure(list(Fecha = structure(c(18993, 19448, 19723, 19631, 
19997, 19083, 19631, 19174, 19631, 19539, 19905, 18993, 19083, 
19266, 19448, 19083, 19358, 19723, 19539, 19358, 19358, 19814, 
19266, 17987, 19723, 19723, 19083, 19723, 17987, 18993, 18993, 
19174, 19814, 19905, 19997, 19539, 19266, 19358, 19905, 18993, 
19723, 18993, 19174, 19814, 19448, 19358, 18262, 19266, 19083, 
18536, 18628, 19448, 19083, 19997, 19997, 19723, 19448, 19723, 
19997, 19266), class = "Date"), MC_Clientes = c(`10` = 3.30914630982938, 
`163` = 3.40705102004165, `148` = 3.27210950513308, `225` = 3.30737187089665, 
`114` = 3.27216488807394, `156` = 3.29952915488204, `110` = 3.39960342129031, 
`65` = 3.44020887082005, `226` = 3.3477486021078, `72` = 3.37828428718909, 
`89` = 3.35889902349877, `130` = 3.24841711499344, `33` = 3.39557631661649, 
`96` = 3.30867139996011, `48` = 3.29220186390159, `39` = 3.38775370812839, 
`11` = 3.33638404617856, `28` = 3.34128523097663, `196` = 3.32642364225741, 
`20` = 3.30914630982938, `15` = 3.32250928717776, `59` = 3.38775370812839, 
`215` = 3.30737187089665, 3.23830671159728, `143` = 3.262723490499, 
`144` = 3.28580416922959, `154` = 3.19317306859439, `146` = 3.22387486690392, 
3.50997312704934, `125` = 3.2916234598844, `128` = 3.27210950513308, 
`68` = 3.42813783609873, `57` = 3.36530584016227, `204` = 3.43236203716387, 
`118` = 3.3861339757903, `192` = 3.4510934126958, `211` = 3.26617681913074, 
`140` = 3.24841711499344, `87` = 3.43574132492634, `2` = 3.30010088087279, 
`27` = 3.33539993355129, `129` = 3.27777616345452, `184` = 3.43236203716387, 
`55` = 3.31622658677549, `42` = 3.34351801788172, `132` = 3.28169122393979, 
3.18803129400015, `92` = 3.35020429586366, `157` = 3.39074839564108, 
3.11563497015824, 3.40541198110811, `166` = 3.29952915488204, 
`38` = 3.29220186390159, `111` = 3.25522990668431, `113` = 3.26178511373003, 
`150` = 3.24841711499344, `167` = 3.39074839564108, `22` = 3.30010088087279, 
`233` = 3.40771653641682, `214` = 3.26936263234189), Zona = c("A", 
"B", "B", "B", "A", "B", "A", "A", "B", "A", "A", "B", "A", "A", 
"A", "A", "A", "A", "B", "A", "A", "A", "B", "B", "B", "B", "B", 
"B", "A", "B", "B", "A", "A", "B", "A", "B", "B", "B", "A", "A", 
"A", "B", "B", "A", "A", "B", "B", "A", "B", "B", "A", "B", "A", 
"A", "A", "B", "B", "A", "B", "B"), Type = c("Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Actual", "Predictions", "Predictions", "Predictions", 
"Predictions", "Actual", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Actual", 
"Predictions", "Predictions", "Actual", "Actual", "Predictions", 
"Predictions", "Predictions", "Predictions", "Predictions", "Predictions", 
"Predictions", "Predictions", "Predictions")), row.names = c(NA, 
-60L), class = c("tbl_df", "tbl", "data.frame"))
解决方案

方法1:预先汇总数据(逻辑清晰,推荐)

先过滤出预测值数据,按Zona和Fecha分组计算每组的最小值、最大值和均值,再用于绘图:

library(dplyr)
library(ggplot2)

# 汇总预测值的分组统计数据
pred_summary <- x %>%
  filter(Type == "Predictions") %>%
  group_by(Zona, Fecha) %>%
  summarise(
    mean_mc = mean(MC_Clientes),
    min_mc = min(MC_Clientes),
    max_mc = max(MC_Clientes),
    .groups = "drop"
  )

# 绘图
x %>%
  ggplot(aes(x = Fecha, y = MC_Clientes, color = Type)) +
  geom_point() +
  # 添加与误差棒匹配的均值点
  geom_point(data = pred_summary, aes(y = mean_mc), 
             col = "black", size = 3, shape = 24, fill = "red") +
  # 仅为预测值添加分组误差棒
  geom_errorbar(data = pred_summary, aes(y = mean_mc, ymin = min_mc, ymax = max_mc),
                width = 0.2, color = "black") +
  facet_wrap(~ Zona)

方法2:直接用stat_summary生成误差棒

无需预先处理数据,通过stat_summary指定计算逻辑,同时过滤预测值:

x %>%
  ggplot(aes(x = Fecha, y = MC_Clientes, color = Type)) +
  geom_point() +
  # 绘制所有数据的均值点(如需仅展示预测值均值,可添加filter逻辑)
  stat_summary(geom = "point", fun = "mean", col = "black", size = 3, shape = 24, fill = "red") +
  # 仅为预测值生成分组误差棒
  stat_summary(data = . %>% filter(Type == "Predictions"),
               geom = "errorbar",
               fun.min = min,
               fun.max = max,
               width = 0.2,
               color = "black") +
  facet_wrap(~ Zona)

关键说明

  1. 过滤预测值:通过filter(Type == "Predictions")确保误差棒仅针对预测值数据
  2. 分组计算:按Zona和Fecha分组,保证误差棒是对应日期+分区的数值范围,而非全组全局范围
  3. 数据匹配:方法1中用汇总数据的均值点和误差棒配对,避免出现均值与误差棒不匹配的情况

内容的提问来源于stack exchange,提问作者user113156

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.16 04:01:19