如何在R的ComplexUpset中获取intersection size对应的成员名称?
解决ComplexUpset交集成员名称提取问题
问题修正与分析
你原代码中存在一处关键笔误:distinct_genes_table <- data.frame(Genes_table, b= 0, c= 0, ...)里的Genes_table未定义,需替换为distinct_genes_table,这是导致后续数据异常、无法提取成员的核心原因之一。以下是修正后的完整解决方案:
方法一:借助ComplexUpset内置数据提取交集成员
先通过upset_data获取图中展示的交集集合,再批量提取对应基因:
library(ComplexUpset) library(readxl) library(dplyr) library(tibble) # 修正原代码笔误,重新构建数据框 setwd("/Users/hasancandemirci/Desktop//data") hcd <- read_xlsx("a.xlsx") hcd <- as.data.frame(hcd) Genes <- c(hcd$b, hcd$c, hcd$d, hcd$e, hcd$f, hcd$g) distinct_genes_table <- tibble(Genes) # 替换未定义的Genes_table为distinct_genes_table distinct_genes_table <- data.frame(distinct_genes_table, b=0, c=0, d=0, e=0, f=0, g=0) distinct_genes_table$b <- ifelse(distinct_genes_table$Genes %in% hcd$b, 1, 0) distinct_genes_table$c <- ifelse(distinct_genes_table$Genes %in% hcd$c, 1, 0) distinct_genes_table$d <- ifelse(distinct_genes_table$Genes %in% hcd$d, 1, 0) distinct_genes_table$e <- ifelse(distinct_genes_table$Genes %in% hcd$e, 1, 0) distinct_genes_table$f <- ifelse(distinct_genes_table$Genes %in% hcd$f, 1, 0) distinct_genes_table$g <- ifelse(distinct_genes_table$Genes %in% hcd$g, 1, 0) distinct_genes_table <- na.omit(distinct_genes_table) tf <- colnames(distinct_genes_table)[2:7] # 同步修正此处的对象名 # 获取图中展示的交集数据(匹配绘图时的min_size参数) upset_data_obj <- upset_data(distinct_genes_table, tf, min_size=15) target_intersections <- unique(upset_data_obj$plot_intersections_subset) # 批量提取每个交集的基因成员 intersection_results <- lapply(target_intersections, function(intersect_obj) { # 获取当前交集对应的组名 group_names <- names(intersect_obj)[intersect_obj] # 筛选同时属于这些组的基因 genes_list <- distinct_genes_table %>% filter(if_all(all_of(group_names), ~ . == 1)) %>% pull(Genes) # 返回交集标识和对应基因 list( intersection = paste(group_names, collapse = " & "), genes = genes_list, size = length(genes_list) ) }) # 转换为数据框方便查看 intersection_df <- bind_rows(intersection_results) print(intersection_df)
方法二:直接从原始数据框筛选指定交集
如果需要单独提取某个特定交集的基因,可直接用筛选语句:
# 示例:提取同时属于b、c、d三个组的基因 bcd_genes <- distinct_genes_table %>% filter(b == 1 & c == 1 & d == 1) %>% pull(Genes) print(bcd_genes)
若要批量提取所有满足min_size≥15的交集,可生成所有组组合后逐一验证:
# 生成所有非空组组合 all_group_combs <- lapply(1:length(tf), function(k) { combn(tf, k, simplify = FALSE) }) %>% unlist(recursive = FALSE) # 筛选出大小≥15的交集并提取基因 valid_intersections <- lapply(all_group_combs, function(comb) { gene_count <- distinct_genes_table %>% filter(if_all(all_of(comb), ~ . == 1)) %>% nrow() if(gene_count >= 15) { genes <- distinct_genes_table %>% filter(if_all(all_of(comb), ~ . == 1)) %>% pull(Genes) list( intersection = paste(comb, collapse = " & "), size = gene_count, genes = genes ) } else { NULL } }) %>% Filter(Negate(is.null), .) # 转换为数据框查看 valid_intersection_df <- bind_rows(valid_intersections) print(valid_intersection_df)
内容的提问来源于stack exchange,提问作者Hasan Can Demirci
相关产品推荐
相关产品推荐

