You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R语言热图中压缩图例:对计数结果分箱处理

问题背景与需求

我有如下格式的数据集:

heights <- rnorm(10000, mean=170, sd=10) 
weights <- rnorm(10000, mean=65, sd=15) 

data <- data.frame(heights, weights)
   heights  weights
1 164.0554 75.21385
2 167.8416 80.20245
3 170.8382 64.86342
4 175.3897 73.40080
5 177.6491 42.82188
6 169.2133 79.28145

我使用以下R代码统计每个5x5分箱中的数据点数量:

max_height <- max(data$heights)
max_weight <- max(data$weights)

height_breaks <- seq(0, max_height, by=5)
weight_breaks <- seq(0, max_weight, by=5)

combinations <- expand.grid(height = seq_along(height_breaks)[-length(height_breaks)],
                            weight = seq_along(weight_breaks)[-length(weight_breaks)])

interval_label <- function(breaks, index) {
  paste0("(", breaks[index], "-", breaks[index + 1], ")")
}

combinations$height_interval <- mapply(interval_label, list(height_breaks), combinations$height)
combinations$weight_interval <- mapply(interval_label, list(weight_breaks), combinations$weight)

height_weight_boxes <- combinations[, c("height_interval", "weight_interval")]
 
count_points_in_box <- function(min_height, max_height, min_weight, max_weight, data) {
    data %>%
        filter(height >= min_height, height <= max_height,
               weight >= min_weight, weight <= max_weight) %>%
        nrow()
}

library(dplyr)
transformed_df <- height_weight_boxes %>%
    mutate(
        box_number = row_number(), 
        min_height = as.numeric(sub("\((.*?)-.*", "\1", height_interval)), 
        max_height = as.numeric(sub(".*-(.*)\)", "\1", height_interval)), 
        min_weight = as.numeric(sub("\((.*?)-.*", "\1", weight_interval)), 
        max_weight = as.numeric(sub(".*-(.*)\)", "\1", weight_interval)) 
    ) 

count_points_in_box <- function(min_height, max_height, min_weight, max_weight, data) {
    data %>%
        filter(heights >= min_height, heights < max_height,
               weights >= min_weight, weights < max_weight) %>%
        nrow()
}

final <- transformed_df %>%
    rowwise() %>%
    mutate(count = count_points_in_box(min_height, max_height, min_weight, max_weight, data))

基于此数据生成热图的代码如下:

library(ggplot2)
library(viridisLite)

distinct_counts <- length(unique(final$count))

color_palette <- magma(distinct_counts)

final$color <- factor(final$count)

ggplot(final, aes(x = min_weight, y = min_height, fill = final$color)) +
    geom_tile() +
    scale_fill_manual(values = color_palette, guide = guide_legend(title = "Count")) +
    labs(x = "Minimum Weight", y = "Minimum Height", title = "2D Heatmap of Counts") +
    theme_minimal() +
    theme(legend.position = "right")

热图整体效果良好,但图例显示了所有计数对应的颜色,过长且冗余。我希望压缩图例,通过对计数结果按相似范围分箱实现,请问是否可行?

注:实际场景中我处理的是超大数据集,先通过SQL的CASE WHEN语句完成计数聚合,再将结果导入R生成热图,上述R数据处理代码模拟了SQL的处理逻辑。生成SQL语句的R代码如下:

sql_query <- "SELECT *, CASE"
for (i in 1:nrow(df)) {
  sql_query <- paste0(sql_query, " WHEN height BETWEEN ", df$min_height[i], " AND ", df$max_height[i],
                      " AND weight BETWEEN ", df$min_weight[i], " AND ", df$max_weight[i],
                      " THEN ", df$box_number[i])
}
sql_query <- paste0(sql_query, " END AS box_num FROM my_table;")
解决方案

完全可行,核心是对count字段进行分箱分组,用分组后的变量替代原始计数映射填充色,就能大幅简化图例。以下是几种实用实现方式:

方法1:手动指定分箱区间

如果对计数分布有明确认知,可直接设定分箱阈值,灵活性最高:

library(dplyr)
library(ggplot2)
library(viridisLite)

# 手动定义计数分箱区间,可根据实际数据调整
final <- final %>%
  mutate(count_bin = case_when(
    count == 0 ~ "0",
    count > 0 & count <= 20 ~ "1-20",
    count > 20 & count <= 50 ~ "21-50",
    count > 50 & count <= 100 ~ "51-100",
    count > 100 ~ ">100"
  )) %>%
  # 设定分箱顺序,保证图例逻辑正确
  mutate(count_bin = factor(count_bin, levels = c("0", "1-20", "21-50", "51-100", ">100")))

# 生成对应分箱数量的调色板
bin_count <- length(unique(final$count_bin))
color_palette <- magma(bin_count)

# 绘制简化图例的热图
ggplot(final, aes(x = min_weight, y = min_height, fill = count_bin)) +
  geom_tile() +
  scale_fill_manual(values = color_palette, guide = guide_legend(title = "Count Range")) +
  labs(x = "Minimum Weight", y = "Minimum Height", title = "2D Heatmap of Counts (Binned)") +
  theme_minimal() +
  theme(legend.position = "right")

方法2:自动分箱(适合无预设认知的场景)

不想手动设定区间时,可用cut()函数自动分箱,支持两种常见逻辑:

分位数分箱(适配偏态分布)

按计数的分位数划分区间,避免少数高计数区间挤占图例:

final <- final %>%
  # 按四分位数分箱,probs参数可调整分箱数量(比如seq(0,1,0.1)生成10个分箱)
  mutate(count_bin = cut(count, breaks = quantile(count, probs = seq(0, 1, 0.2)), include.lowest = TRUE))

# 绘制热图,直接用viridis离散调色板
ggplot(final, aes(x = min_weight, y = min_height, fill = count_bin)) +
  geom_tile() +
  scale_fill_viridis_d(option = "magma", guide = guide_legend(title = "Count Range")) +
  labs(x = "Minimum Weight", y = "Minimum Height", title = "2D Heatmap of Counts (Quantile Binned)") +
  theme_minimal() +
  theme(legend.position = "right")

等宽分箱(适配均匀分布)

按固定宽度划分计数区间,逻辑直观:

final <- final %>%
  # 指定分箱数量,比如breaks=5生成5个等宽区间
  mutate(count_bin = cut(count, breaks = 5, include.lowest = TRUE))

# 绘制热图
ggplot(final, aes(x = min_weight, y = min_height, fill = count_bin)) +
  geom_tile() +
  scale_fill_viridis_d(option = "magma", guide = guide_legend(title = "Count Range")) +
  labs(x = "Minimum Weight", y = "Minimum Height", title = "2D Heatmap of Counts (Equal Width Binned)") +
  theme_minimal() +
  theme(legend.position = "right")

适配SQL场景的优化

针对超大数据集,可直接在SQL阶段完成计数分箱,减少R端处理量:

SELECT 
  box_num,
  min_height,
  max_height,
  min_weight,
  max_weight,
  count,
  CASE
    WHEN count = 0 THEN '0'
    WHEN count > 0 AND count <= 20 THEN '1-20'
    WHEN count > 20 AND count <= 50 THEN '21-50'
    WHEN count > 50 AND count <= 100 THEN '51-100'
    WHEN count > 100 THEN '>100'
  END AS count_bin
FROM (
  -- 替换为你的原有计数聚合逻辑
  SELECT 
    box_num,
    min_height,
    max_height,
    min_weight,
    max_weight,
    COUNT(*) AS count
  FROM my_table
  GROUP BY box_num, min_height, max_height, min_weight, max_weight
) AS aggregated_data;

导入R后直接用count_bin字段映射填充色即可,无需额外处理。


内容的提问来源于stack exchange,提问作者stats_noob

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 07:34:54