如何在R语言中识别采样日期是否处于风暴事件±3天范围内
识别采样日期是否处于风暴事件±3天范围内的解决方案
我有两个数据框,一个包含采样日期,另一个包含风暴(AR)事件日期,需要识别每个采样日期是否处于任意风暴事件的±3天范围内。以下是示例数据:
采样数据框
structure(list(Date = structure(c(7319, 7378, 7439, 7500, 7531, 7562, 7592, 7623, 7653, 7684), class = "Date"), NO3 = c(2.37, 3.42, 3.13, 2.24, 1.97, 2.22, 2.58, 2.15, 2.05, 3.09), cumP = c(122.8, 104, 19.9, 20.4, 0, 8.8, 134.3, 232.8, 168.3, 171.3), Season = c("R", "R", "F", "I", "H", "R", "R", "R", "R", "R")), row.names = c(NA, 10L), class = "data.frame")
风暴事件数据框
structure(list(Date = structure(c(7311, 7313, 7316, 7329, 7338, 7345, 7355, 7451, 7458, 7474, 7580, 7581, 7586, 7598, 7601, 7602, 7615, 7617, 7618, 7619, 7620, 7621, 7630, 7631, 7632, 7637, 7641, 7642, 7646, 7647, 7655), class = "Date")), row.names = c(NA, -31L), class = "data.frame")
方法一:使用dplyr + lubridate包
这种方法代码简洁易读,适合处理数据框操作:
# 加载依赖包 library(dplyr) library(lubridate) # 读取示例数据(数据已加载可跳过此步) sampling_df <- structure(list(Date = structure(c(7319, 7378, 7439, 7500, 7531, 7562, 7592, 7623, 7653, 7684), class = "Date"), NO3 = c(2.37, 3.42, 3.13, 2.24, 1.97, 2.22, 2.58, 2.15, 2.05, 3.09), cumP = c(122.8, 104, 19.9, 20.4, 0, 8.8, 134.3, 232.8, 168.3, 171.3), Season = c("R", "R", "F", "I", "H", "R", "R", "R", "R", "R")), row.names = c(NA, 10L), class = "data.frame") ar_events_df <- structure(list(Date = structure(c(7311, 7313, 7316, 7329, 7338, 7345, 7355, 7451, 7458, 7474, 7580, 7581, 7586, 7598, 7601, 7602, 7615, 7617, 7618, 7619, 7620, 7621, 7630, 7631, 7632, 7637, 7641, 7642, 7646, 7647, 7655), class = "Date")), row.names = c(NA, -31L), class = "data.frame") # 生成所有风暴事件±3天的日期集合,去重避免重复判断 ar_date_range <- ar_events_df %>% rowwise() %>% mutate(date_range = list(seq(Date - days(3), Date + days(3), by = "day"))) %>% pull(date_range) %>% unlist() %>% unique() %>% as.Date(origin = "1970-01-01") # 为采样数据框添加标记列,判断是否在风暴影响范围内 sampling_df <- sampling_df %>% mutate(near_AR = Date %in% ar_date_range) # 查看最终结果 print(sampling_df)
方法二:基础R实现
如果不想加载额外包,可使用基础R代码完成:
# 读取示例数据(数据已加载可跳过此步) sampling_df <- structure(list(Date = structure(c(7319, 7378, 7439, 7500, 7531, 7562, 7592, 7623, 7653, 7684), class = "Date"), NO3 = c(2.37, 3.42, 3.13, 2.24, 1.97, 2.22, 2.58, 2.15, 2.05, 3.09), cumP = c(122.8, 104, 19.9, 20.4, 0, 8.8, 134.3, 232.8, 168.3, 171.3), Season = c("R", "R", "F", "I", "H", "R", "R", "R", "R", "R")), row.names = c(NA, 10L), class = "data.frame") ar_events_df <- structure(list(Date = structure(c(7311, 7313, 7316, 7329, 7338, 7345, 7355, 7451, 7458, 7474, 7580, 7581, 7586, 7598, 7601, 7602, 7615, 7617, 7618, 7619, 7620, 7621, 7630, 7631, 7632, 7637, 7641, 7642, 7646, 7647, 7655), class = "Date")), row.names = c(NA, -31L), class = "data.frame") # 生成所有风暴事件±3天的日期序列,去重 ar_dates <- ar_events_df$Date ar_range <- unlist(lapply(ar_dates, function(x) seq(x - 3, x + 3, by = "day"))) ar_range <- unique(ar_range) # 标记采样日期是否在风暴影响范围内 sampling_df$near_AR <- sampling_df$Date %in% ar_range # 查看最终结果 print(sampling_df)
结果说明
运行代码后,采样数据框会新增一列near_AR:
TRUE表示该采样日期处于任意风暴事件的±3天范围内FALSE表示该采样日期不在任何风暴事件的±3天范围内
内容的提问来源于stack exchange,提问作者arcticmermaid
相关产品推荐
相关产品推荐

