You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用R自动化处理待积分区间:解决重叠区间切片瓶颈

自动化NMR积分区间处理方案(基于tidyverse)

问题说明

需要自动化生成NMR积分区间:对初始区间的From和To分别加减指定偏移量(如160),处理偏移后产生的区间重叠问题,同时保留每个区间对应的Component归属信息。此前用绑定含NA表格的方法无法实现无信息丢失的重叠切片。

初始数据表:

int_NMR <- data.frame(
  "Component" = c("A", "B", "C", "D", "E", "F", "G", "H"),
  "From" = c(0.0, 45.0, 60.0, 95.0, 110.0, 145.0, 165.0, 190),
  "To" = c(45.0, 60.0, 95.0, 110.0, 145.0, 165.0, 190.0, 215.0)
)

原有尝试代码(无法满足需求):

na_frame <- NULL
int_NMR_all<- NULL

int_NMR_high <- data.frame(setNames(lapply(int_NMR[1], function(x) paste("ssb_high", x, sep="_")),"Component_ssb"),int_NMR[2:3]+sb_ofset)

int_NMR_low <- data.frame(setNames(lapply(int_NMR[1], function(x) paste("ssb_low", x, sep="_")),"Component_ssb"),int_NMR[2:3]-sb_ofset)

int_NMR_all <- rbind(int_NMR_low,int_NMR_high)

na_frame <- as.data.frame(matrix(NA, nrow = nrow(int_NMR), ncol = 3))
names(na_frame) <- names(int_NMR_all)
int_NMR_all <- rbind(int_NMR_all, na_frame)

na_frame <- as.data.frame(matrix(NA, nrow = nrow(int_NMR_all), ncol = 1))
names(na_frame) <- c("Component")
int_NMR_all<- cbind(int_NMR_all, na_frame)

na_frame <- as.data.frame(matrix(NA, nrow = nrow(int_NMR), ncol = 1))
names(na_frame) <- c("Component_ssb")
int_NMR<- cbind(int_NMR, na_frame)

na_frame <- as.data.frame(matrix(NA, nrow = 2*nrow(int_NMR), ncol = 4))

names(na_frame) <- names(int_NMR_all)

int_NMR <- rbind(int_NMR, na_frame)

int_NMR_all <- rbind(int_NMR,int_NMR_all)

int_NMR_all <- int_NMR_all[!is.na(int_NMR_all$From),] 

tidyverse解决方案

使用dplyr和tidyr工具链可以高效处理区间偏移、重叠切片并完整保留归属信息:

1. 加载依赖包

library(tidyverse)

2. 生成所有偏移后的区间

定义偏移量,整合原区间、减偏移量的区间、加偏移量的区间,并标记组件类型:

sb_offset <- 160 # 设置偏移量

all_intervals <- int_NMR %>%
  pivot_longer(cols = c(From, To), names_to = "bound", values_to = "value") %>%
  expand_grid(shift_type = c("original", "low", "high")) %>%
  mutate(
    value = case_when(
      shift_type == "low" ~ value - sb_offset,
      shift_type == "high" ~ value + sb_offset,
      TRUE ~ value
    ),
    component_id = case_when(
      shift_type == "low" ~ paste0("ssb_low_", Component),
      shift_type == "high" ~ paste0("ssb_high_", Component),
      TRUE ~ Component
    )
  ) %>%
  pivot_wider(names_from = "bound", values_from = "value") %>%
  select(component_id, From, To)

3. 提取并排序所有边界点

收集所有区间的边界值,去重后排序得到切片的分割点:

all_bounds <- all_intervals %>%
  pivot_longer(cols = c(From, To), values_to = "bound") %>%
  pull(bound) %>%
  unique() %>%
  sort()

4. 切片重叠区间并匹配归属

生成分割后的新区间,为每个区间匹配对应的组件:

sliced_intervals <- tibble(
  From = head(all_bounds, -1),
  To = tail(all_bounds, -1)
) %>%
  rowwise() %>%
  mutate(
    components = list(
      all_intervals %>%
        filter(From <= !!cur_data()$From & To >= !!cur_data()$To) %>%
        pull(component_id)
    )
  ) %>%
  ungroup() %>%
  unnest(components) %>%
  rename(Component = components)

最终的sliced_intervals数据框包含所有切片后的区间,每个区间都明确对应所属的Component(包括原组件和偏移后的组件),无归属信息丢失。

内容的提问来源于stack exchange,提问作者LuisCol8

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.14 04:36:03