如何用R追踪年度观测新增情况?优化data.table实现方案
更简洁的data.table实现连续年份颜色分类统计
原始数据集
library(data.table) library(ggplot2) set.seed(123) years <- 2010:2020 max_colors <- 50 data <- data.frame() for (year in years) { n_colors <- sample(5:max_colors, 1) color_pool <- 1:max_colors colors <- sample(color_pool, n_colors, replace = TRUE) year_data <- data.frame( color = colors, year = year ) data <- rbind(data, year_data) } setDT(data) myt <- data
统计需求
针对每对连续年份(如2010-2011、2011-2012等),需统计三类颜色的数量:
- 全新颜色:从未在之前任何年份出现过的颜色
- 来自前一年的颜色:上一年出现过的颜色
- 复现颜色:曾在更早年份出现,但未出现在上一年的颜色
原始循环实现
目前通过手动循环实现的代码如下:
analyze_consecutive_years_dt <- function(df) { setDT(df) years <- sort(unique(df$year)) results <- data.table() for (i in 1:(length(years) - 1)) { year_i <- years[i] year_i_plus_1 <- years[i + 1] colors_i <- df[year == year_i, unique(color)] colors_i_plus_1 <- df[year == year_i_plus_1, unique(color)] if (i == 1) { colors_before_i <- character(0) } else { colors_before_i <- df[year < year_i, unique(color)] } colors_from_i_in_i_plus_1 <- length(intersect(colors_i, colors_i_plus_1)) never_seen_before <- length(setdiff(colors_i_plus_1, df[year < year_i_plus_1, unique(color)])) not_in_i_but_in_earlier <- length(intersect(setdiff(colors_i_plus_1, colors_i), colors_before_i)) results <- rbind(results, data.table( year_pair = paste0(year_i, "-", year_i_plus_1), total_colors_i_plus_1 = length(colors_i_plus_1), colors_from_i_in_i_plus_1 = colors_from_i_in_i_plus_1, never_seen_before = never_seen_before, not_in_i_but_in_earlier = not_in_i_but_in_earlier )) } return(results) } results <- analyze_consecutive_years_dt(myt)
更简洁的data.table实现
利用data.table的分组、滚动计算和集合操作特性,可以完全避免循环,同时保持高效性适配大数据集:
# 按颜色分组,记录每个颜色首次出现的年份 color_first_occur <- myt[, .(first_year = min(year)), by = color] # 按年份分组,得到每年的唯一颜色集合,并按年份排序 year_colors <- myt[, .(colors = list(unique(color))), by = year] setorder(year_colors, year) # 生成前一年的颜色集合、以及截至前一年所有出现过的颜色集合 year_colors[, `:=`( prev_year_colors = shift(colors), all_prev_colors = shift(Reduce(union, colors, accumulate = TRUE)) )] # 计算各项统计指标 results_clean <- year_colors[-1, .( year_pair = paste0(year - 1, "-", year), total_colors_i_plus_1 = length(unlist(colors)), colors_from_i_in_i_plus_1 = length(intersect(unlist(colors), unlist(prev_year_colors))), never_seen_before = length(setdiff(unlist(colors), unlist(all_prev_colors))), not_in_i_but_in_earlier = length(setdiff(unlist(colors), union(unlist(prev_year_colors), setdiff(unlist(colors), unlist(all_prev_colors))))) )]
可视化代码
以下是生成统计结果堆叠柱状图的代码:
plot_data <- results_clean[, .( year = year, `全新颜色` = never_seen_before, `来自前一年` = colors_from_i_in_i_plus_1, `曾出现但非上年` = not_in_i_but_in_earlier )] first_year <- min(myt$year) first_year_colors <- myt[year == first_year, length(unique(color))] first_year_row <- data.table( year = first_year, `全新颜色` = first_year_colors, `来自前一年` = 0, `曾出现但非上年` = 0 ) plot_data <- rbind(first_year_row, plot_data) plot_long <- melt(plot_data, id.vars = "year", variable.name = "类别", value.name = "数量") p <- ggplot(plot_long, aes(x = factor(year), y = 数量, fill = 类别)) + geom_bar(stat = "identity") + scale_fill_brewer(type = "qual", palette = "Set1") + labs( title = "年度颜色分类统计", x = "年份", y = "颜色数量", fill = "类别" ) + theme_minimal() print(p)

内容的提问来源于stack exchange,提问作者stats_noob
相关产品推荐
相关产品推荐

