You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用combn函数实现分组下的逐行成对比较定制功能

问题描述

数据集片段

dat2 <- read.table(text = "
   nodepair  V1  V2  V3  V4  V5  V6  V7  V8  V9 ES   
 1 A1_A1        0    21     0     0     0     0     0     0    78 45   
 2 A2_A1        0     0     0     0     0     0     0     0    99 45   
 3 A2_A2        0     1     0     0     0     0     0     0    98 45   
 4 A3_A1        0     0     0     0     0     6     1     3    89 45   
 5 A3_A2        0     0     0     0     0     0     0     0    99 45   
 6 A1_A1        0    20     0     0     0     0     0     0    65 46   
 7 A2_A1        0     0     0     0     0     0     0     0    85 46   
 8 A2_A2        0     1     0     0     0     0     0     0    84 46   
 9 A3_A1        0     0     0     0     2     6     3     3    71 46   
 10 A3_A2        0     0     0     0     0     0     0     0    85 46   
 11 A1_A1        0    25     0     0     0     0     0     0    45 47   
 12 A2_A1        0     0     0     0     0     0     0     0    70 47   
 13 A2_A2        0     1     0     0     0     0     0     0    69 47   
 14 A3_A1        0     0     0     0     0     8     0     1    61 47   
 15 A3_A2        0     0     0     0     0     0     0     0    70 47   
 16 A1_A1        0    37     0     0     0     0     0     0    77 48   
 17 A2_A1        0     0     0     0     0     0     0     0   114 48   
 18 A2_A2        0     0     0     0     0     0     0     0   114 48   
 19 A3_A1        0     0     0     0     2     9     0     3   100 48   
 20 A3_A2        0     0     0     0     0     0     0     0   114 48   
 ", header = TRUE)

需求

按nodepair分组,同一组内的不同ES行两两配对比较:对每一对行的V1到V9列,若两行对应列数值都大于0,则标记为1,否则为0。输出格式示例如下:

dat3 <- read.table(text = "
    nodepair1 nodepair2  V1  V2  V3  V4  V5  V6  V7  V8  V9    
    A1_A1(45) A1_A1(46)   0     0    1     0     0     0     0     0     1        
  ", header = TRUE)

现有代码(未完成)

dat2 <- dat2 %>%
   group_by(nodepair) %>%
   col2 = t(combn(nodepair,2)))

解决方案

使用dplyr分组处理,结合combn生成组内两两行组合,逐列判断条件并生成结果:

library(dplyr)

# 处理单个nodepair分组的函数
process_group <- function(group_data) {
  # 获取组内行索引,生成两两组合
  row_indices <- seq_len(nrow(group_data))
  combn_pairs <- combn(row_indices, 2, simplify = FALSE)
  
  # 遍历每个组合生成结果行
  result_rows <- lapply(combn_pairs, function(pair) {
    row1 <- group_data[pair[1], ]
    row2 <- group_data[pair[2], ]
    
    # 构造配对名称:nodepair(ES)
    nodepair1 <- paste0(row1$nodepair, "(", row1$ES, ")")
    nodepair2 <- paste0(row2$nodepair, "(", row2$ES, ")")
    
    # 判断V1-V9列是否都>0,生成标记值
    v1 <- row1 %>% select(V1:V9) %>% as.numeric()
    v2 <- row2 %>% select(V1:V9) %>% as.numeric()
    compare_result <- as.integer(v1 > 0 & v2 > 0)
    
    # 整合成结果行
    data.frame(
      nodepair1 = nodepair1,
      nodepair2 = nodepair2,
      t(compare_result),
      stringsAsFactors = FALSE
    )
  })
  
  # 合并所有结果行
  bind_rows(result_rows)
}

# 应用到所有分组并合并结果
dat3 <- dat2 %>%
  group_by(nodepair) %>%
  group_modify(~ process_group(.x)) %>%
  ungroup()

# 查看结果片段
head(dat3)

代码说明

  1. process_group函数:针对单个nodepair分组,用combn生成所有行的两两组合,对每对行的V1-V9列做条件判断,生成符合格式的结果行。
  2. group_modify:将处理函数批量应用到每个分组,自动合并各组结果。
  3. 最终dat3包含所有nodepair组内的两两行比较结果,V1-V9列标记对应位置是否满足两行值都>0的条件。

内容的提问来源于stack exchange,提问作者Hard_Course

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.14 04:28:11