You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R语言Keras中1D卷积网络预测输出维度异常问题求助

多变量时间序列CNN模型预测输出长度异常问题解决

问题描述

在R语言中使用Keras处理多变量时间序列数据时,DNN、GRU、LSTM模型均能正常训练并生成符合长度要求的预测,但构建含1D卷积层的模型(如CNN-GRU/CNN-LSTM)时,训练过程正常,测试集预测输出的向量长度却远小于预期。例如预留1000个时间步的测试集,模型仅返回约300个预测结果,且只要使用layer_conv_1d()就会出现该问题。

可复现代码如下:

数据生成与生成器实现

library(tidyverse)
library(keras)
library(reticulate)

# 生成样本数据
generate_data = function(n_samples) {
  set.seed(123)
  
  time = seq(1, n_samples)
  covariate1 = rnorm(n_samples, mean = 0, sd = 1)
  covariate2 = rnorm(n_samples, mean = 5, sd = 2)
  covariate3 = rnorm(n_samples, mean = -3, sd = 3)
  target = sin(seq(1, n_samples) * 0.1) + rnorm(n_samples, mean = 0, sd = 0.2)
  
  data = tibble(Time = time, Covariate1 = covariate1, Covariate2 = covariate2, Covariate3 = covariate3, Target = target)
  return(data)
}

nsamp = 5000
sample_data = as.matrix(generate_data(nsamp))

# 时间序列数据生成器
generator <- function(data, lookback, delay, min_index, max_index,
                      shuffle = FALSE, batch_size, step, 
                      predseries) {
  
  if (is.null(max_index)) max_index <- nrow(data) - delay - 1
  i <- min_index + lookback
  function() {

    if (shuffle) {
      rows <- sample(c((min_index+lookback):max_index), size = batch_size)
    } else {
      if (i + batch_size >= max_index)
        i <<- min_index + lookback
        rows <- c(i:min(i+batch_size, max_index))
        i <<- i + length(rows)
    }

    samples <- array(0, dim = c(length(rows),
                                lookback / step,
                                dim(data)[[-1]]))

    targets <- array(0, dim = c(length(rows)))
    
    for (j in 1:length(rows)) {
      indices <- seq(rows[[j]] - lookback, rows[[j]],
                     length.out = dim(samples)[[2]])
      samples[j,,] <- data[indices,]
      targets[[j]] <- data[rows[[j]] + delay,predseries]
    }
    list(samples, targets)
  }
}

# 生成器参数配置
lookback = 10
step = 1
delay = 1
batch_size = 20
predser = 1 # 目标变量索引

# 数据集划分
min_train = 1
max_train = floor(nsamp*2/3)
min_val = max_train+1
max_val = min_val + floor(0.5*(nsamp-max_train))
min_test = max_val+1
max_test = NULL

# 计算各数据集的steps
val_steps = floor( (max_val - min_val - lookback) / batch_size )
test_steps = floor( (nrow(sample_data) - max_val - lookback) / batch_size)
train_steps = floor( (max_train - min_train - lookback) / batch_size )

# 创建各数据集生成器
train_gen = generator(
  sample_data,
  lookback = lookback,
  delay = delay,
  min_index = min_train,
  max_index = max_train,
  step = step,
  batch_size = batch_size,
  predseries = predser  
)

val_gen = generator(
  sample_data,
  lookback = lookback,
  delay = delay,
  min_index = min_val,
  max_index = max_val,
  step = step,
  batch_size = batch_size,
  predseries = predser  
)

test_gen = generator(
  sample_data,
  lookback = lookback,
  delay = delay,
  min_index = min_test,
  max_index = NULL,
  step = step,
  batch_size = batch_size,
  predseries = predser    
)

存在问题的模型结构

build_and_compile_model = function() {
  model = keras_model_sequential() %>%
      layer_conv_1d(
          filters=64, 
          kernel_size=2, 
          activation="relu",
          input_shape = list(NULL, dim(sample_data)[[-1]])
        ) %>%
      layer_max_pooling_1d(pool_size=3) %>%
      layer_dense(64, activation = 'relu') %>% 
      layer_dense(units = 1)

   model %>% compile(
      loss = 'mean_absolute_error',
      optimizer = optimizer_adam()
    )

    model
  }

model1  = build_and_compile_model()

# 训练模型(原代码存在steps参数错误)
model1 %>% fit(
  train_gen,
  steps_per_epoch = train_steps,
  epochs = 20,
  validation_data = val_gen,
  validation_steps = val_steps
)

问题原因分析

  1. 时间维度未正确压缩:

    • 输入序列长度为lookback=10,经过kernel_size=2的1D卷积后,时间维度变为10 - 2 + 1 = 9
    • 再经过pool_size=3的最大池化后,时间维度进一步变为9 / 3 = 3
    • 直接接入layer_dense会保留这个时间维度,模型输出形状为(batch_size, 3, 1),而生成器提供的目标形状是(batch_size, 1)。训练时Keras会自动广播匹配,但预测时每个输入样本会输出3个预测值,最终总预测数为测试样本数除以3(向下取整),导致833个测试样本仅返回277个结果。
  2. 训练参数错误:原代码中fit函数使用steps_per_epoch = test_steps,导致训练时仅遍历测试集量级的数据,而非完整训练集,虽不直接影响预测长度,但会降低模型性能。

解决方案

修改模型结构,在卷积池化后加入全局池化层,将压缩后的序列维度压缩为单个向量,确保每个输入样本对应一个预测值:

build_and_compile_model_fixed = function() {
  model = keras_model_sequential() %>%
      layer_conv_1d(
          filters=64, 
          kernel_size=2, 
          activation="relu",
          input_shape = list(NULL, dim(sample_data)[[-1]])
        ) %>%
      layer_max_pooling_1d(pool_size=3) %>%
      # 新增全局平均池化层,将时间维度压缩为单个向量
      layer_global_average_pooling_1d() %>%
      layer_dense(64, activation = 'relu') %>% 
      layer_dense(units = 1)

   model %>% compile(
      loss = 'mean_absolute_error',
      optimizer = optimizer_adam()
    )

    model
  }

# 训练修正后的模型
model_fixed = build_and_compile_model_fixed()
model_fixed %>% fit(
  train_gen,
  steps_per_epoch = train_steps,
  epochs = 20,
  validation_data = val_gen,
  validation_steps = val_steps
)

验证效果

使用修正后的模型预测,输出长度将与测试样本数一致:

test_pred = model_fixed %>% predict(test_gen, steps = test_steps)
length(test_pred) # 应等于测试样本总数,与预期的833匹配

内容的提问来源于stack exchange,提问作者Jacob Usinowicz

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 11:34:55