You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R语言循环if过滤鸟类音频数据时的行索引异常问题

鸟类音频检测数据集的鸣唱事件合并问题

数据集与需求

  • 数据集:由BirdNET生成的鸟类音频检测片段,每个片段3秒、重叠2秒;涵盖31个公园、10台记录仪、124种鸟类,共超1900万条观测。数据字段包括id、park_abbr、am_no、sci_name、start_s、end_s、conf。
  • 核心需求:将同一公园、同一记录仪、时间间隔≤20秒的同物种片段合并为单个鸣唱事件:
    • 保留事件首行的id、park_abbr、am_no、sci_name、start_s数据
    • 更新end_s为该事件最后一个片段的end_s
    • 计算事件内所有片段conf的平均值

现有代码

示例数据集

library(tidyverse)

# 示例数据
blackbird <- structure(list(id = c(5801137L, 5801138L, 5801139L, 5801146L, 
5801147L, 5801148L, 5801149L, 5801150L, 5801154L, 5801155L, 5801156L, 
5801157L, 5801158L, 5801168L, 5801178L, 5801188L, 5801189L, 5801190L, 
5801191L, 5801192L, 5801193L, 5801194L, 5801195L, 5801196L, 5801197L, 
5801198L, 5801200L, 5801202L, 5801203L, 5801204L, 5801206L, 5801208L, 
5801211L, 5801215L, 5801217L, 5801219L, 5801220L, 5801221L, 5801222L, 
5801223L, 5801224L, 5801230L, 5801231L, 5801232L, 5801235L, 10399185L, 
10399202L, 13435015L, 13435017L, 13435018L, 13435019L, 13435020L, 
13435021L, 13435022L, 13435023L, 13435024L, 13435025L, 13435026L, 
13435027L, 13435028L, 13435030L, 13435032L, 13435033L, 14544238L, 
14544245L), park_abbr = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 2L, 2L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 
3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 4L, 4L), levels = c("GV", "NH", 
"RO", "TE"), class = "factor"), am_no = structure(c(1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L), levels = "A1", class = "factor"), 
    sci_name = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
    1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
    1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
    1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 
    1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L), levels = "Aegithalos caudatus", class = "factor"), 
    start_s = c(0L, 1L, 2L, 6L, 7L, 8L, 9L, 10L, 14L, 15L, 16L, 
    17L, 18L, 25L, 32L, 42L, 43L, 44L, 45L, 46L, 47L, 48L, 49L, 
    50L, 51L, 52L, 53L, 54L, 55L, 56L, 57L, 58L, 59L, 61L, 62L, 
    63L, 64L, 65L, 66L, 67L, 68L, 73L, 74L, 75L, 76L, 103L, 122L, 
    732L, 733L, 734L, 742L, 743L, 744L, 745L, 746L, 747L, 748L, 
    749L, 750L, 751L, 752L, 753L, 754L, 339L, 495L), end_s = c(3L, 
    4L, 5L, 9L, 10L, 11L, 12L, 13L, 17L, 18L, 19L, 20L, 21L, 
    28L, 35L, 45L, 46L, 47L, 48L, 49L, 50L, 51L, 52L, 53L, 54L, 
    55L, 56L, 57L, 58L, 59L, 60L, 61L, 62L, 64L, 65L, 66L, 67L, 
    68L, 69L, 70L, 71L, 76L, 77L, 78L, 79L, 106L, 125L, 735L, 
    736L, 737L, 745L, 746L, 747L, 748L, 749L, 750L, 751L, 752L, 
    753L, 754L, 755L, 756L, 757L, 342L, 498L), conf = c(0.2234, 
    0.5614, 0.4781, 0.1463, 0.3758, 0.5148, 0.7357, 0.5954, 0.3679, 
    0.8198, 0.5869, 0.82, 0.3683, 0.3678, 0.1484, 0.9014, 0.9899, 
    0.9958, 0.9964, 0.9957, 0.9467, 0.9939, 0.9917, 0.9915, 0.9533, 
    0.7444, 0.8523, 0.9729, 0.9253, 0.6424, 0.8775, 0.5058, 0.3696, 
    0.8907, 0.8253, 0.9798, 0.8965, 0.9547, 0.7425, 0.9277, 0.7885, 
    0.9165, 0.5399, 0.5519, 0.1256, 0.1927, 0.2104, 0.546, 0.1157, 
    0.1264, 0.4881, 0.8465, 0.5787, 0.9218, 0.7392, 0.9863, 0.9783, 
    0.9544, 0.2077, 0.1305, 0.6694, 0.1298, 0.8092, 0.1887, 0.3975
    )), row.names = c(NA, -65L), class = "data.frame")

处理函数

# 原处理函数
clean.dataset <- function(df) {
  cleaned_df <- data.frame()
  last_park <- NULL
  last_am_no <- NULL
  last_start <- NULL
  last_end <- NULL
  last_conf <- NULL
  keeping_count <- NULL
  for (i in 1:nrow(df)) {
    park <- df$park_abbr[i]
    am_no <- df$am_no[i]
    start <- df$start_s[i]
    end <- df$end_s[i]
    conf <- df$conf[i]
    if (
      is.null(last_start)
    ) {
      cleaned_df <- rbind(cleaned_df, df[i, ])
      last_park <- park
      last_am_no <- am_no
      last_start <- start
      last_end <- end
      last_conf <- conf
      keeping_count <- 1
    } else {
      if (
        start - last_end < 20 && last_park == park && last_am_no == am_no
      ) {
        last_end <- end
        last_conf <- last_conf + conf
        keeping_count <- keeping_count + 1
      } else {
        if (park != last_park || last_am_no != am_no) {
          cleaned_df <- rbind(cleaned_df, df[i, ])
          last_conf <- last_conf / keeping_count
          cleaned_df[i - keeping_count, 6] <- last_end # 6 was 8 to match full data
          cleaned_df[i - keeping_count, 7] <- last_conf # 7 was 9 to match full data
          last_park <- park
          last_am_no <- am_no
          last_start <- start
          last_end <- end
          last_conf <- conf
          keeping_count <- 1
        } else {
          cleaned_df <- rbind(cleaned_df, df[i, ])
          last_conf <- last_conf / keeping_count
          cleaned_df[i - keeping_count, 6] <- last_end # 6 was 8 to match full data
          cleaned_df[i - keeping_count, 7] <- last_conf # 7 was 9 to match full data
          last_park <- park
          last_am_no <- am_no
          last_start <- start
          last_end <- end
          last_conf <- conf
          keeping_count <- 1
        }
      }
    }
  }
  return(cleaned_df)
}

# 调用函数
clean.dataset(blackbird)

问题现象

函数输出异常:

  • 目标行的end_s和conf未正确更新
  • 出现大量NA行
  • 生成命名为x.1的额外行(仅包含正确的end_s和conf值)

问题原因

  1. 行索引逻辑错误:循环中用i - keeping_count定位事件首行的逻辑不成立,因为cleaned_df的行数与原df不一致(合并时仅保留事件首行),导致索引越界或定位错误。
  2. 数据类型不一致:反复使用rbind拼接空数据框和因子类型数据时,会引发列名冲突(如x.1),同时导致NA值出现。
  3. 未处理最后一个事件:循环结束后,未对最后一个正在跟踪的事件进行end_s更新和conf平均值计算。

解决方法

方法1:修正循环函数

clean.dataset <- function(df) {
  # 先按分组和时间排序,确保数据顺序正确
  df <- df %>% arrange(park_abbr, am_no, start_s)
  
  # 初始化结果数据框,保留原数据结构
  cleaned_df <- df[0, ]
  # 初始化当前事件为第一行数据
  current_event <- df[1, ]
  current_conf_sum <- current_event$conf
  current_count <- 1
  
  # 从第二行开始遍历
  for (i in 2:nrow(df)) {
    row <- df[i, ]
    # 判断是否属于同一事件:同公园、同记录仪、间隔≤20秒
    same_event <- (row$park_abbr == current_event$park_abbr) &&
                  (row$am_no == current_event$am_no) &&
                  (row$start_s - current_event$end_s <= 20)
    
    if (same_event) {
      # 更新当前事件的结束时间、conf总和及计数
      current_event$end_s <- row$end_s
      current_conf_sum <- current_conf_sum + row$conf
      current_count <- current_count + 1
    } else {
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.12 11:16:01