R语言循环if过滤鸟类音频数据时的行索引异常问题
鸟类音频检测数据集的鸣唱事件合并问题
数据集与需求
- 数据集:由BirdNET生成的鸟类音频检测片段,每个片段3秒、重叠2秒;涵盖31个公园、10台记录仪、124种鸟类,共超1900万条观测。数据字段包括
id、park_abbr、am_no、sci_name、start_s、end_s、conf。 - 核心需求:将同一公园、同一记录仪、时间间隔≤20秒的同物种片段合并为单个鸣唱事件:
- 保留事件首行的
id、park_abbr、am_no、sci_name、start_s数据 - 更新
end_s为该事件最后一个片段的end_s - 计算事件内所有片段
conf的平均值
- 保留事件首行的
现有代码
示例数据集
library(tidyverse) # 示例数据 blackbird <- structure(list(id = c(5801137L, 5801138L, 5801139L, 5801146L, 5801147L, 5801148L, 5801149L, 5801150L, 5801154L, 5801155L, 5801156L, 5801157L, 5801158L, 5801168L, 5801178L, 5801188L, 5801189L, 5801190L, 5801191L, 5801192L, 5801193L, 5801194L, 5801195L, 5801196L, 5801197L, 5801198L, 5801200L, 5801202L, 5801203L, 5801204L, 5801206L, 5801208L, 5801211L, 5801215L, 5801217L, 5801219L, 5801220L, 5801221L, 5801222L, 5801223L, 5801224L, 5801230L, 5801231L, 5801232L, 5801235L, 10399185L, 10399202L, 13435015L, 13435017L, 13435018L, 13435019L, 13435020L, 13435021L, 13435022L, 13435023L, 13435024L, 13435025L, 13435026L, 13435027L, 13435028L, 13435030L, 13435032L, 13435033L, 14544238L, 14544245L), park_abbr = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 2L, 2L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 3L, 4L, 4L), levels = c("GV", "NH", "RO", "TE"), class = "factor"), am_no = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L), levels = "A1", class = "factor"), sci_name = structure(c(1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L, 1L), levels = "Aegithalos caudatus", class = "factor"), start_s = c(0L, 1L, 2L, 6L, 7L, 8L, 9L, 10L, 14L, 15L, 16L, 17L, 18L, 25L, 32L, 42L, 43L, 44L, 45L, 46L, 47L, 48L, 49L, 50L, 51L, 52L, 53L, 54L, 55L, 56L, 57L, 58L, 59L, 61L, 62L, 63L, 64L, 65L, 66L, 67L, 68L, 73L, 74L, 75L, 76L, 103L, 122L, 732L, 733L, 734L, 742L, 743L, 744L, 745L, 746L, 747L, 748L, 749L, 750L, 751L, 752L, 753L, 754L, 339L, 495L), end_s = c(3L, 4L, 5L, 9L, 10L, 11L, 12L, 13L, 17L, 18L, 19L, 20L, 21L, 28L, 35L, 45L, 46L, 47L, 48L, 49L, 50L, 51L, 52L, 53L, 54L, 55L, 56L, 57L, 58L, 59L, 60L, 61L, 62L, 64L, 65L, 66L, 67L, 68L, 69L, 70L, 71L, 76L, 77L, 78L, 79L, 106L, 125L, 735L, 736L, 737L, 745L, 746L, 747L, 748L, 749L, 750L, 751L, 752L, 753L, 754L, 755L, 756L, 757L, 342L, 498L), conf = c(0.2234, 0.5614, 0.4781, 0.1463, 0.3758, 0.5148, 0.7357, 0.5954, 0.3679, 0.8198, 0.5869, 0.82, 0.3683, 0.3678, 0.1484, 0.9014, 0.9899, 0.9958, 0.9964, 0.9957, 0.9467, 0.9939, 0.9917, 0.9915, 0.9533, 0.7444, 0.8523, 0.9729, 0.9253, 0.6424, 0.8775, 0.5058, 0.3696, 0.8907, 0.8253, 0.9798, 0.8965, 0.9547, 0.7425, 0.9277, 0.7885, 0.9165, 0.5399, 0.5519, 0.1256, 0.1927, 0.2104, 0.546, 0.1157, 0.1264, 0.4881, 0.8465, 0.5787, 0.9218, 0.7392, 0.9863, 0.9783, 0.9544, 0.2077, 0.1305, 0.6694, 0.1298, 0.8092, 0.1887, 0.3975 )), row.names = c(NA, -65L), class = "data.frame")
处理函数
# 原处理函数 clean.dataset <- function(df) { cleaned_df <- data.frame() last_park <- NULL last_am_no <- NULL last_start <- NULL last_end <- NULL last_conf <- NULL keeping_count <- NULL for (i in 1:nrow(df)) { park <- df$park_abbr[i] am_no <- df$am_no[i] start <- df$start_s[i] end <- df$end_s[i] conf <- df$conf[i] if ( is.null(last_start) ) { cleaned_df <- rbind(cleaned_df, df[i, ]) last_park <- park last_am_no <- am_no last_start <- start last_end <- end last_conf <- conf keeping_count <- 1 } else { if ( start - last_end < 20 && last_park == park && last_am_no == am_no ) { last_end <- end last_conf <- last_conf + conf keeping_count <- keeping_count + 1 } else { if (park != last_park || last_am_no != am_no) { cleaned_df <- rbind(cleaned_df, df[i, ]) last_conf <- last_conf / keeping_count cleaned_df[i - keeping_count, 6] <- last_end # 6 was 8 to match full data cleaned_df[i - keeping_count, 7] <- last_conf # 7 was 9 to match full data last_park <- park last_am_no <- am_no last_start <- start last_end <- end last_conf <- conf keeping_count <- 1 } else { cleaned_df <- rbind(cleaned_df, df[i, ]) last_conf <- last_conf / keeping_count cleaned_df[i - keeping_count, 6] <- last_end # 6 was 8 to match full data cleaned_df[i - keeping_count, 7] <- last_conf # 7 was 9 to match full data last_park <- park last_am_no <- am_no last_start <- start last_end <- end last_conf <- conf keeping_count <- 1 } } } } return(cleaned_df) } # 调用函数 clean.dataset(blackbird)
问题现象
函数输出异常:
- 目标行的
end_s和conf未正确更新 - 出现大量NA行
- 生成命名为
x.1的额外行(仅包含正确的end_s和conf值)
问题原因
- 行索引逻辑错误:循环中用
i - keeping_count定位事件首行的逻辑不成立,因为cleaned_df的行数与原df不一致(合并时仅保留事件首行),导致索引越界或定位错误。 - 数据类型不一致:反复使用
rbind拼接空数据框和因子类型数据时,会引发列名冲突(如x.1),同时导致NA值出现。 - 未处理最后一个事件:循环结束后,未对最后一个正在跟踪的事件进行
end_s更新和conf平均值计算。
解决方法
方法1:修正循环函数
clean.dataset <- function(df) { # 先按分组和时间排序,确保数据顺序正确 df <- df %>% arrange(park_abbr, am_no, start_s) # 初始化结果数据框,保留原数据结构 cleaned_df <- df[0, ] # 初始化当前事件为第一行数据 current_event <- df[1, ] current_conf_sum <- current_event$conf current_count <- 1 # 从第二行开始遍历 for (i in 2:nrow(df)) { row <- df[i, ] # 判断是否属于同一事件:同公园、同记录仪、间隔≤20秒 same_event <- (row$park_abbr == current_event$park_abbr) && (row$am_no == current_event$am_no) && (row$start_s - current_event$end_s <= 20) if (same_event) { # 更新当前事件的结束时间、conf总和及计数 current_event$end_s <- row$end_s current_conf_sum <- current_conf_sum + row$conf current_count <- current_count + 1 } else {
相关产品推荐
相关产品推荐

