You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

在R中合并多个树状图与热图:报错排查及实现方案

问题:热图合并层次聚类树状图报错处理

数据背景

拥有一个包含染色体区域和样本数据的大型数据框,预览如下:

head(joined_df_sorted)
# A tibble: 6 × 24
  chrom   start     end EE85756 EE85757 EE87786 EE87787 EE87788 EE87789 EE87790 EE87811 EE87812 EE87813 EE87814 EE87815 EE87893 EE87894 EE87895 EE87896 EE87897
  <chr>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>   <dbl>
1 chr1  1000001 2000000       3       4       3       3       3       3       3       3       3       3       3       3       3       3       3       3       3
2 chr1  3000001 4000000       3       4       8       3       3       3       3       3       3       3       3       3       3       3       3       3       3
3 chr1  4000001 5000000       3       4      13       3       3       8       3       3       3       3       3       3       3       6       3       3       3       3
4 chr1  5000001 6000000       3       4      10       3       3       7       3       3       3       3       3       3       3       3       3       3       3       3
5 chr1  6000001 7000000       3       4       3       3       3       3       3       3       3       3       3       3       3       3       3       3       3       3
6 chr1  7000001 8000000       3       4       7       3       3       3       3       3       3       3       3       3       3       3       3       3       3       3
# ℹ 1 more variable: chromid <chr>

已执行代码

# 合并前两列生成唯一行名
joined_df_sorted$chromid <- paste0(joined_df_sorted$chrom, "_", joined_df_sorted$start)

# 保留所需列
joined_df_sorted2 <- as.data.frame(joined_df_sorted[,c(24,4:23)])

# 将第一列设为行名
joined_df_sorted3 <- joined_df_sorted2
joined_df_sorted3X <- joined_df_sorted3[,-1]
rownames(joined_df_sorted3X) <- joined_df_sorted3[,1]

# 转置数据框
install.packages("sjmisc")
library(sjmisc)

joined_df_sorted3X_t <- joined_df_sorted3X %>% 
  rotate_df(cn = FALSE)

CanCohortDat <- joined_df_sorted3X_t

# 定义样本分组
BileDuct  <- c("EE87786",   "EE87787",  "EE87788",  "EE87789",  "EE87790")
Breast <-  c("EE87811", "EE87812",  "EE87813",  "EE87814",  "EE87815")
Gastric <- c("EE87893", "EE87894",  "EE87895",  "EE87896",  "EE87897")
Healthy <- c("EE85756", "EE85757",  "EE85770","EE85775",    "EE85784")

# 对每个分组单独聚类
h1 <- hclust(dist(CanCohortDat[BileDuct,]))
h2 <- hclust(dist(CanCohortDat[Breast,]))
h3 <- hclust(dist(CanCohortDat[Gastric,]))
h4 <- hclust(dist(CanCohortDat[Healthy,]))

# 合并四个聚类树
hc <- as.hclust(merge(merge(merge(
  as.dendrogram(h1), as.dendrogram(h2)), as.dendrogram(h3)), 
  as.dendrogram(h4)))

# 按分组顺序提取数据
CanCoh <- CanCohortDat[c(BileDuct, Breast, Gastric, Healthy),]

# 创建行注释
cohort_annotation <- data.frame(Region = c(rep("BileDuct", length(BileDuct)), 
                                           rep("Breast", length(Breast)),
                                           rep("Gastric", length(Gastric)),
                                           rep("Healthy", length(Healthy))),
                                row.names = c(BileDuct, Breast, Gastric, Healthy))

# 绘制热图
pheatmap(CanCoh , cluster_rows = hc, 
         annotation_row = cohort_annotation)

报错信息

`use_raster` is automatically set to TRUE for a matrix with more than 2000
columns You can control `use_raster` argument by explicitly setting TRUE/FALSE
to it.

Set `ht_opt$message = FALSE` to turn off this message.
'magick' package is suggested to install to give better rasterization.

Set `ht_opt$message = FALSE` to turn off this message.
Error in hclust(get_dist(t(submat), distance), method = method) : 
  NA/NaN/Inf in foreign function call (arg 10)
In addition: Warning message:
The input is a data frame, convert it to the matrix. 

期望输出:Y轴为按合并树状图分组的EE*样本,X轴显示染色体位置。


问题分析与解决方案

1. 核心报错原因

  • 数据类型不匹配:pheatmap要求输入为矩阵,传入数据框会导致距离计算异常
  • 异常值干扰:数据中存在NA、NaN或Inf值,导致聚类计算失败
  • 聚类合并逻辑偏差:直接合并树状图时,未确保聚类结构与数据行顺序完全匹配

2. 分步解决方法

(1)转换数据为矩阵并清理异常值

# 转换为矩阵,替换异常值
CanCoh_matrix <- as.matrix(CanCoh)
# 将NA/Inf替换为0(可根据业务逻辑调整)
CanCoh_matrix[is.na(CanCoh_matrix) | is.infinite(CanCoh_matrix)] <- 0

(2)优化聚类合并逻辑

使用dendextend包更灵活地合并树状图,确保结构与数据匹配:

install.packages("dendextend")
library(dendextend)

# 转换为dendrogram对象
d1 <- as.dendrogram(h1)
d2 <- as.dendrogram(h2)
d3 <- as.dendrogram(h3)
d4 <- as.dendrogram(h4)

# 分层合并树状图,height值可根据实际聚类距离调整
merged_dend <- merge(merge(merge(d1, d2, height = 100), d3, height = 200), d4, height = 300)
# 转回hclust对象
hc_merged <- as.hclust(merged_dend)

(3)调整pheatmap参数

# 可选:安装magick优化渲染
install.packages("magick")

# 关闭提示消息
ht_opt$message <- FALSE

# 绘制热图,关闭列聚类以保留染色体位置顺序
pheatmap(CanCoh_matrix, 
         cluster_rows = hc_merged, 
         cluster_cols = FALSE,
         annotation_row = cohort_annotation,
         use_raster = TRUE)

(4)验证数据一致性

确保矩阵行顺序与聚类标签完全匹配:

all(rownames(CanCoh_matrix) == hc_merged$labels)

若结果为FALSE,需重新调整数据行顺序或聚类合并逻辑。


内容的提问来源于stack exchange,提问作者Debajyoti Kabiraj

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 22:05:01