You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R语言中5×5网格数据点统计的计数偏差修正求助

问题原因

你的计数偏差是因为区间上限未包含数据最大值,且过滤条件使用<而非针对最后一个区间的<=,导致等于最大值的点被排除。同时,从0开始生成区间会产生大量空网格,完全没必要。

通过验证可以确认漏点数量:

set.seed(42) 
heights <- rnorm(100, mean=170, sd=10) 
weights <- rnorm(100, mean=65, sd=15) 
data <- data.frame(heights, weights)

sum(data$heights == max(data$heights)) # 1个点等于最大身高
sum(data$weights == max(data$weights)) # 2个点等于最大体重
# 1+2=3,正好对应100-97的差值
解决方法

方法一:使用cut函数分箱(推荐,简洁高效)

cut函数可以自动处理区间边界,确保所有数据点被包含,且代码更简洁:

library(dplyr)
library(ggplot2)

set.seed(42) 
heights <- rnorm(100, mean=170, sd=10) 
weights <- rnorm(100, mean=65, sd=15) 
data <- data.frame(heights, weights)

# 生成5cm/5kg间隔的区间,include.lowest确保包含最小值
data <- data %>%
  mutate(
    height_bin = cut(heights, breaks = seq(floor(min(heights)), ceiling(max(heights)), by = 5), include.lowest = TRUE),
    weight_bin = cut(weights, breaks = seq(floor(min(weights)), ceiling(max(weights)), by = 5), include.lowest = TRUE)
  )

# 统计每个网格的点数
final <- data %>%
  count(height_bin, weight_bin, name = "count") %>%
  # 可选:转换为区间极值,方便后续分析
  mutate(
    min_height = as.numeric(sub("\\[(.*?)-.*", "\\1", height_bin)),
    max_height = as.numeric(sub(".*-(.*)\\]", "\\1", height_bin)),
    min_weight = as.numeric(sub("\\[(.*?)-.*", "\\1", weight_bin)),
    max_weight = as.numeric(sub(".*-(.*)\\]", "\\1", weight_bin))
  )

# 验证总数
sum(final$count) # 输出100
nrow(data)       # 输出100

方法二:修正原有代码逻辑

如果坚持使用原有流程,需要调整区间范围和过滤条件:

library(dplyr)
library(ggplot2)

set.seed(42) 
heights <- rnorm(100, mean=170, sd=10) 
weights <- rnorm(100, mean=65, sd=15) 
data <- data.frame(heights, weights)

# 1. 修正边界:覆盖所有数据点,起点取最小值向下取整到5的倍数,终点取最大值向上取整到5的倍数
min_height <- floor(min(data$heights)/5)*5
max_height <- ceiling(max(data$heights)/5)*5
min_weight <- floor(min(data$weights)/5)*5
max_weight <- ceiling(max(data$weights)/5)*5

height_breaks <- seq(min_height, max_height, by=5)
weight_breaks <- seq(min_weight, max_weight, by=5)

# 2. 生成区间标签,最后一个区间使用闭区间
combinations <- expand.grid(height = seq_along(height_breaks)[-length(height_breaks)],
                            weight = seq_along(weight_breaks)[-length(weight_breaks)])

interval_label <- function(breaks, index) {
  if(index == length(breaks)-1){
    paste0("[", breaks[index], "-", breaks[index + 1], "]")
  } else {
    paste0("[", breaks[index], "-", breaks[index + 1], ")")
  }
}

combinations$height_interval <- mapply(interval_label, list(height_breaks), combinations$height)
combinations$weight_interval <- mapply(interval_label, list(weight_breaks), combinations$weight)

height_weight_boxes <- combinations[, c("height_interval", "weight_interval")]

# 3. 提取区间极值
transformed_df <- height_weight_boxes %>%
  mutate(
    box_number = row_number(), 
    min_height = as.numeric(sub("\\[(.*?)-.*", "\\1", height_interval)), 
    max_height = as.numeric(sub(".*-(.*)\\]?", "\\1", height_interval)), 
    min_weight = as.numeric(sub("\\[(.*?)-.*", "\\1", weight_interval)), 
    max_weight = as.numeric(sub(".*-(.*)\\]?", "\\1", weight_interval)) 
  ) %>%
  select(box_number, min_height, max_height, min_weight, max_weight)

# 4. 修正计数函数,区分最后一个区间的过滤条件
count_points_in_box <- function(min_h, max_h, min_w, max_w, data, is_last_h, is_last_w) {
  data %>%
    filter(
      heights >= min_h,
      if(is_last_h) heights <= max_h else heights < max_h,
      weights >= min_w,
      if(is_last_w) weights <= max_w else weights < max_w
    ) %>%
    nrow()
}

# 标记是否为最后一个区间
transformed_df <- transformed_df %>%
  mutate(
    is_last_height = max_height == !!max_height,
    is_last_weight = max_weight == !!max_weight
  )

# 统计每个网格点数
final <- transformed_df %>%
  rowwise() %>%
  mutate(count = count_points_in_box(min_height, max_height, min_weight, max_weight, data, is_last_height, is_last_weight))

# 验证总数
sum(final$count) # 输出100
nrow(data)       # 输出100

内容的提问来源于stack exchange,提问作者stats_noob

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 07:34:57