R语言中5×5网格数据点统计的计数偏差修正求助
问题原因
你的计数偏差是因为区间上限未包含数据最大值,且过滤条件使用<而非针对最后一个区间的<=,导致等于最大值的点被排除。同时,从0开始生成区间会产生大量空网格,完全没必要。
通过验证可以确认漏点数量:
set.seed(42) heights <- rnorm(100, mean=170, sd=10) weights <- rnorm(100, mean=65, sd=15) data <- data.frame(heights, weights) sum(data$heights == max(data$heights)) # 1个点等于最大身高 sum(data$weights == max(data$weights)) # 2个点等于最大体重 # 1+2=3,正好对应100-97的差值
解决方法
方法一:使用cut函数分箱(推荐,简洁高效)
cut函数可以自动处理区间边界,确保所有数据点被包含,且代码更简洁:
library(dplyr) library(ggplot2) set.seed(42) heights <- rnorm(100, mean=170, sd=10) weights <- rnorm(100, mean=65, sd=15) data <- data.frame(heights, weights) # 生成5cm/5kg间隔的区间,include.lowest确保包含最小值 data <- data %>% mutate( height_bin = cut(heights, breaks = seq(floor(min(heights)), ceiling(max(heights)), by = 5), include.lowest = TRUE), weight_bin = cut(weights, breaks = seq(floor(min(weights)), ceiling(max(weights)), by = 5), include.lowest = TRUE) ) # 统计每个网格的点数 final <- data %>% count(height_bin, weight_bin, name = "count") %>% # 可选:转换为区间极值,方便后续分析 mutate( min_height = as.numeric(sub("\\[(.*?)-.*", "\\1", height_bin)), max_height = as.numeric(sub(".*-(.*)\\]", "\\1", height_bin)), min_weight = as.numeric(sub("\\[(.*?)-.*", "\\1", weight_bin)), max_weight = as.numeric(sub(".*-(.*)\\]", "\\1", weight_bin)) ) # 验证总数 sum(final$count) # 输出100 nrow(data) # 输出100
方法二:修正原有代码逻辑
如果坚持使用原有流程,需要调整区间范围和过滤条件:
library(dplyr) library(ggplot2) set.seed(42) heights <- rnorm(100, mean=170, sd=10) weights <- rnorm(100, mean=65, sd=15) data <- data.frame(heights, weights) # 1. 修正边界:覆盖所有数据点,起点取最小值向下取整到5的倍数,终点取最大值向上取整到5的倍数 min_height <- floor(min(data$heights)/5)*5 max_height <- ceiling(max(data$heights)/5)*5 min_weight <- floor(min(data$weights)/5)*5 max_weight <- ceiling(max(data$weights)/5)*5 height_breaks <- seq(min_height, max_height, by=5) weight_breaks <- seq(min_weight, max_weight, by=5) # 2. 生成区间标签,最后一个区间使用闭区间 combinations <- expand.grid(height = seq_along(height_breaks)[-length(height_breaks)], weight = seq_along(weight_breaks)[-length(weight_breaks)]) interval_label <- function(breaks, index) { if(index == length(breaks)-1){ paste0("[", breaks[index], "-", breaks[index + 1], "]") } else { paste0("[", breaks[index], "-", breaks[index + 1], ")") } } combinations$height_interval <- mapply(interval_label, list(height_breaks), combinations$height) combinations$weight_interval <- mapply(interval_label, list(weight_breaks), combinations$weight) height_weight_boxes <- combinations[, c("height_interval", "weight_interval")] # 3. 提取区间极值 transformed_df <- height_weight_boxes %>% mutate( box_number = row_number(), min_height = as.numeric(sub("\\[(.*?)-.*", "\\1", height_interval)), max_height = as.numeric(sub(".*-(.*)\\]?", "\\1", height_interval)), min_weight = as.numeric(sub("\\[(.*?)-.*", "\\1", weight_interval)), max_weight = as.numeric(sub(".*-(.*)\\]?", "\\1", weight_interval)) ) %>% select(box_number, min_height, max_height, min_weight, max_weight) # 4. 修正计数函数,区分最后一个区间的过滤条件 count_points_in_box <- function(min_h, max_h, min_w, max_w, data, is_last_h, is_last_w) { data %>% filter( heights >= min_h, if(is_last_h) heights <= max_h else heights < max_h, weights >= min_w, if(is_last_w) weights <= max_w else weights < max_w ) %>% nrow() } # 标记是否为最后一个区间 transformed_df <- transformed_df %>% mutate( is_last_height = max_height == !!max_height, is_last_weight = max_weight == !!max_weight ) # 统计每个网格点数 final <- transformed_df %>% rowwise() %>% mutate(count = count_points_in_box(min_height, max_height, min_weight, max_weight, data, is_last_height, is_last_weight)) # 验证总数 sum(final$count) # 输出100 nrow(data) # 输出100
内容的提问来源于stack exchange,提问作者stats_noob
相关产品推荐
相关产品推荐

