You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R的ggplot2中用长格式数据通过geom_errorbar绘制误差棒柱状图

问题:长格式数据绘制带误差棒的柱状图失败

我用宽格式数据框coun2b能成功绘制带误差棒的柱状图,但完整数据集是长格式(示例为samp1),绘制Camden县的图时,geom_errorbar里用filter指定ymin和ymax报错:

Error in `geom_errorbar()`:
! Problem while computing aesthetics.
ℹ Error occurred in the 2nd layer.
Caused by error in `check_aesthetics()`:
! Aesthetics must be either length 1 or the same as the data (9)
✖ Fix the following mappings: `ymin` and `ymax`
Run `rlang::last_error()` to see where the error occurred.
Warning message:
In geom_errorbar(data = samp1 %>% filter(county == "Camden"), aes(ymin = samp1 %>%  :
  Ignoring unknown aesthetics: position

需要解决:用长格式数据samp1生成和宽格式一致的柱状图,且后续支持多县并列柱状图。


宽格式数据coun2b结构

> dput(coun2b)
structure(list(Camden = c(13.9933481152993, 17.5410199556541, 
26.0055432372506, 19.1064301552106, 9.05764966740577, 17.5321507760532
), Guilford = c(24.674715261959, 27.5097949886105, 25.4646924829157, 
22.2637813211845, 7.60227790432802, 17.9681093394077), years = 2012:2017, 
    Camden_ymin = c(12.4514939737261, 15.4927722105436, 22.5744436662436, 
    16.8415649174844, 7.45264839077184, 15.6645677387521), Guilford_ymin = c(23.2136204848819, 
    26.3627764588421, 23.8076842636931, 20.383805927254, 5.58799564906578, 
    16.2548749333076), Camden_ymax = c(15.5352022568726, 19.5892677007646, 
    29.4366428082575, 21.3712953929369, 10.6626509440397, 19.3997338133543
    ), Guilford_ymax = c(26.1358100390361, 28.6568135183788, 
    27.1217007021384, 24.143756715115, 9.61656015959026, 19.6813437455079
    )), class = "data.frame", row.names = c(NA, -6L))

宽格式绘图代码

library(tidyverse)

ggplot(coun2b, aes(x=years, Guilford, group=years)) + 
  labs(title = "Counts in Guilford, N.C.", 
       y="Number of Days", x="Year" ) + geom_col( position = "dodge") +
  geom_errorbar(aes(ymin=Guilford_ymin, ymax=Guilford_ymax), position="dodge") + 
  theme(axis.text.x = element_text(face="bold"), axis.title.x = element_text(size=14), 
        axis.text.y = element_text(face="bold"), axis.title.y = element_text(size=14), 
        title = element_text(size=12)) +
  scale_x_continuous("Year", labels = plotscalex, breaks=plotscalex) +
  geom_hline(aes(yintercept = mean(Guilford[years %in% 2012:2016]),
                 linetype='Mean for 2012-2016')) +
  scale_linetype_manual(name="Legend", values=c("Mean for 2012-2016"=1) )

长格式数据samp1结构

> dput(samp1)
structure(list(years = c(2012L, 2012L, 2012L, 2013L, 2013L, 2013L, 
2014L, 2014L, 2014L, 2012L, 2012L, 2012L, 2013L, 2013L, 2013L, 
2014L, 2014L, 2014L), valu = c("mean", "ymin", "ymax", "mean", 
"ymin", "ymax", "mean", "ymin", "ymax", "mean", "ymin", "ymax", 
"mean", "ymin", "ymax", "mean", "ymin", "ymax"), name = c("Camden", 
"Camden", "Camden", "Camden", "Camden", "Camden", "Camden", "Camden", 
"Camden", "Guilford", "Guilford", "Guilford", "Guilford", "Guilford", 
"Guilford", "Guilford", "Guilford", "Guilford"), value = c(13.9933481152993, 
12.4514939737261, 15.5352022568726, 17.5410199556541, 15.4927722105436, 
19.5892677007646, 26.0055432372506, 22.5744436662436, 29.4366428082575, 
24.674715261959, 23.2136204848819, 26.1358100390361, 27.5097949886105, 
26.3627764588421, 28.6568135183788, 25.4646924829157, 23.8076842636931, 
27.1217007021384), county = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L, 2L), levels = c("Camden", 
"Guilford", "Pasquotank", "Wake"), class = "factor")), row.names = c(NA, 
-18L), class = c("tbl_df", "tbl", "data.frame"))

尝试的错误代码

samp1 %>% filter(county == "Camden") %>% 
    ggplot( aes(x=years, y=value, group=years)) + 
    labs(title = "Number of Days in April-August with Suitable Weather for\nLate Blight Sporulation in Camden, N.C.", y="Number of Days", x="Year" ) + 
    geom_col(data=samp1 %>% filter(county=="Camden", valu=="mean"), aes(x=years, 
                                                                         y=value), position = "dodge") +
    geom_errorbar(data=samp1 %>% filter(county=="Camden"), 
                  aes(ymin=samp1 %>% filter(valu=="ymin"), ymax=samp1 %>% filter(valu=="ymax"), position="dodge")) + 
    theme(axis.text.x = element_text(face="bold"), axis.title.x = element_text(size=14), 
          axis.text.y = element_text(face="bold"), axis.title.y = element_text(size=14), 
          title = element_text(size=12)) +
    scale_x_continuous("Year", labels = plotscalex, breaks=plotscalex) +
    geom_hline(aes(yintercept = mean(Camden[years %in% 2012:2016]),
                   linetype='Mean for 2012-2016'))+
    scale_linetype_manual(name="Legend", values=c("Mean for 2012-2016"=1) )

解决方案

1. 错误原因

  • geom_errorbar中直接用samp1 %>% filter(valu=="ymin")传递给ymin/ymax是错误的:这会返回整个过滤后的数据集,而非对应每个年份的数值,导致美学映射长度不匹配。
  • position参数不属于aes()映射,应放在geom_errorbar()外层。
  • 绘制均值水平线时,错误引用了宽格式列名Camden,长格式数据需重新计算均值。

2. 正确实现代码

单县(Camden)绘图

先将长格式数据转换为按年份分组的宽格式,对齐宽格式绘图逻辑:

library(tidyverse)

# 处理数据:长格式转宽格式
camden_data <- samp1 %>%
  filter(county == "Camden") %>%
  pivot_wider(names_from = valu, values_from = value)

# 计算2012-2016年的均值
camden_mean <- camden_data %>%
  filter(years %in% 2012:2016) %>%
  pull(mean) %>%
  mean()

# 绘图
ggplot(camden_data, aes(x = years, y = mean)) +
  labs(title = "北卡罗来纳州Camden县4-8月晚疫病孢子萌发适宜天数",
       y = "天数", x = "年份") +
  geom_col(position = "dodge") +
  geom_errorbar(aes(ymin = ymin, ymax = ymax), 
                width = 0.2, position = position_dodge(0.9)) +
  theme(axis.text.x = element_text(face="bold"), 
        axis.title.x = element_text(size=14), 
        axis.text.y = element_text(face="bold"), 
        axis.title.y = element_text(size=14), 
        title = element_text(size=12)) +
  scale_x_continuous("年份", labels = unique(camden_data$years), breaks = unique(camden_data$years)) +
  geom_hline(aes(yintercept = camden_mean, linetype = "2012-2016年均值")) +
  scale_linetype_manual(name = "图例", values = c("2012-2016年均值" = 1))

多县并列柱状图

保留所有县数据,通过fill = county实现并列:

# 处理所有县的数据
all_county_data <- samp1 %>%
  pivot_wider(names_from = valu, values_from = value)

# 计算每个县2012-2016年的均值
county_means <- all_county_data %>%
  filter(years %in% 2012:2016) %>%
  group_by(county) %>%
  summarize(mean_val = mean(mean))

# 绘图
ggplot(all_county_data, aes(x = years, y = mean, fill = county)) +
  labs(title = "北卡罗来纳州各县4-8月晚疫病孢子萌发适宜天数",
       y = "天数", x = "年份", fill = "县") +
  geom_col(position = position_dodge(0.9)) +
  geom_errorbar(aes(ymin = ymin, ymax = ymax), 
                width = 0.2, position = position_dodge(0.9)) +
  theme(axis.text.x = element_text(face="bold"), 
        axis.title.x = element_text(size=14), 
        axis.text.y = element_text(face="bold"), 
        axis.title.y = element_text(size=14), 
        title = element_text(size=12)) +
  scale_x_continuous("年份", labels = unique(all_county_data$years), breaks = unique(all_county_data$years)) +
  geom_hline(data = county_means, aes(yintercept = mean_val, color = county, linetype = "2012-2016年均值")) +
  scale_linetype_manual(name = "图例", values = c("2012-2016年均值" = 1)) +
  scale_color_discrete(name = "县")

3. 关键说明

  • 用pivot_wider将长格式转换为「每行对应一个年份+县,列包含mean/ymin/ymax」的结构,避免美学映射错误。
  • position_dodge(0.9)需与geom_col参数一致,确保误差棒与柱子对齐。
  • 均值计算需基于长格式数据重新提取,不能直接引用宽格式列名。

内容的提问来源于stack exchange,提问作者John Polo

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.02 08:15:31