You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用GroupNormalizer创建TimeSeriesDataSet时触发KeyError:0求助

问题描述

尝试基于PyTorch Forecasting构建Temporal Fusion Transformer时,始终触发KeyError: 0错误,数据中并无0值。切换为最小测试数据集、简化代码后错误仍存在,且仅在TimeSeriesDataSet中使用GroupNormalizer时触发。运行环境为PaperSpace Jupyter Notebook的Python 3.9.13容器。

测试代码

!pip install numpy==1.24.4
!pip install pandas
!pip install --ignore-installed PyYAML==5.4.1
!pip install torch
!pip install pytorch-lightning
!pip install pytorch-forecasting
!pip install statsmodels
!pip install scipy
!pip install matplotlib

import pandas as pd
from pytorch_forecasting import TimeSeriesDataSet
from pytorch_forecasting.data.encoders import NaNLabelEncoder, GroupNormalizer

# Create a small sample dataframe
sample_data = pd.DataFrame({
    'symbol': ['AAPL', 'AAPL', 'AAPL', 'GOOG', 'GOOG', 'GOOG'],
    'period': [1, 2, 3, 1, 2, 3],
    'close': [150.0, 152.0, 151.0, 2800.0, 2820.0, 2810.0],
    'day_of_epoch': [19844, 19844, 19844, 19845, 19845, 19845]
})

# Ensure 'period' column is treated as integer index
sample_data['period'] = sample_data['period'].astype(int)

# Print the dataframe to verify
print(sample_data)

# Define the maximum prediction length
max_prediction_length = 1
max_encoder_length = 2

# Define the training dataset with minimal configuration
training = TimeSeriesDataSet(
    sample_data,
    time_idx="period",
    target="close",
    group_ids=["symbol", "day_of_epoch"],
    min_encoder_length=1,
    max_encoder_length=max_encoder_length,
    min_prediction_length=1,
    max_prediction_length=max_prediction_length,
    static_categoricals=["symbol"],
    time_varying_known_categoricals=[],
    time_varying_known_reals=["period", "day_of_epoch"],
    time_varying_unknown_categoricals=[],
    time_varying_unknown_reals=["close"],
    target_normalizer=GroupNormalizer(
        groups=["symbol", "day_of_epoch"], transformation="softplus"
    ),
    add_relative_time_idx=True,
    add_target_scales=True,
    add_encoder_length=True,
    categorical_encoders={
        'symbol': NaNLabelEncoder(add_nan=True)
    },
)

报错输出

symbol  period   close  day_of_epoch
0   AAPL       1   150.0         19844
1   AAPL       2   152.0         19844
2   AAPL       3   151.0         19844
3   GOOG       1  2800.0         19845
4   GOOG       2  2820.0         19845
5   GOOG       3  2810.0         19845

---------------------------------------------------------------------------
KeyError                                  Traceback (most recent call last)
File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/encoders.py:331, in NaNLabelEncoder.transform(self, y, return_norm, target_scale, ignore_na)
    330 try:
--> 331     encoded = [self.classes_[v] for v in y]
    332 except KeyError as e:

File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/encoders.py:331, in <listcomp>(.0)
    330 try:
--> 331     encoded = [self.classes_[v] for v in y]
    332 except KeyError as e:

KeyError: 0

During handling of the above exception, another exception occurred:

KeyError                                  Traceback (most recent call last)
Input In [1], in <cell line: 34>()
     31 max_encoder_length = 2
     33 # Define the training dataset with minimal configuration
---> 34 training = TimeSeriesDataSet(
     35     sample_data,
     36     time_idx="period",
     37     target="close",
     38     group_ids=["symbol", "day_of_epoch"],
     39     min_encoder_length=1,
     40     max_encoder_length=max_encoder_length,
     41     min_prediction_length=1,
     42     max_prediction_length=max_prediction_length,
     43     static_categoricals=["symbol"],
     44     time_varying_known_categoricals=[],
     45     time_varying_known_reals=["period", "day_of_epoch"],
     46     time_varying_unknown_categoricals=[],
     47     time_varying_unknown_reals=["close"],
     48     target_normalizer=GroupNormalizer(
     49         groups=["symbol", "day_of_epoch"], transformation="softplus"
     50     ),
     51     add_relative_time_idx=True,
     52     add_target_scales=True,
     53     add_encoder_length=True,
     54     categorical_encoders={
     55         'symbol': NaNLabelEncoder(add_nan=True)
     56     },
     57 )

File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/timeseries.py:476, in TimeSeriesDataSet.__init__(self, data, time_idx, target, group_ids, weight, max_encoder_length, min_encoder_length, min_prediction_idx, min_prediction_length, max_prediction_length, static_categoricals, static_reals, time_varying_known_categoricals, time_varying_known_reals, time_varying_unknown_categoricals, time_varying_unknown_reals, variable_groups, constant_fill_strategy, allow_missing_timesteps, lags, add_relative_time_idx, add_target_scales, add_encoder_length, target_normalizer, categorical_encoders, scalers, randomize_length, predict_mode)
    473 data = data.sort_values(self.group_ids + [self.time_idx])
    475 # preprocess data
--> 476 data = self._preprocess_data(data)
    477 for target in self.target_names:
    478     assert target not in self.scalers, "Target normalizer is separate and not in scalers."

File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/timeseries.py:837, in TimeSeriesDataSet._preprocess_data(self, data)
    831     transformer = self.get_transformer(name)
    832     if (
    833         name not in self.target_names
    834         and transformer is not None
    835         and not isinstance(transformer, EncoderNormalizer)
    836     ):
--> 837         data[name] = self.transform_values(name, data[name], data=data, inverse=False)
    839 # encode lagged categorical targets
    840 for name in self.lagged_targets:
    841     # normalizer only now available

File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/timeseries.py:935, in TimeSeriesDataSet.transform_values(self, name, values, data, inverse, group_id, **kwargs)
    933 # remaining categories
    934 if name in self.flat_categoricals + self.group_ids + self._group_ids:
--> 935     return transform(values, **kwargs)
    937 # reals
    938 elif name in self.reals:

File /usr/local/lib/python3.9/dist-packages/sklearn/utils/_set_output.py:313, in _wrap_method_output.<locals>.wrapped(self, X, *args, **kwargs)
    311 @wraps(f)
    312 def wrapped(self, X, *args, **kwargs):
--> 313     data_to_wrap = f(self, X, *args, **kwargs)
    314     if isinstance(data_to_wrap, tuple):
    315         # only wrap the first output for cross decomposition
    316         return_tuple = (
    317             _wrap_data_with_container(method, data_to_wrap[0], X, self),
    318             *data_to_wrap[1:],
    319         )

File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/encoders.py:333, in NaNLabelEncoder.transform(self, y, return_norm, target_scale, ignore_na)
    331             encoded = [self.classes_[v] for v in y]
    332         except KeyError as e:
--> 333             raise KeyError(
    334                 f"Unknown category '{e.args[0]}' encountered. Set `add_nan=True` to allow unknown categories"
    335             )
    337 if isinstance(y, torch.Tensor):
    338     encoded = torch.tensor(encoded, dtype=torch.long, device=y.device)

KeyError: "Unknown category '0' encountered. Set `add_nan=True` to allow unknown categories"

解决方案

错误根源

  1. day_of_epoch是每个分组的固定值,却被标记为time_varying_known_reals(随时间变化的实数),属性定义错误导致预处理逻辑混乱。
  2. 分组变量day_of_epoch为整数类型,未转为分类变量也未配置对应编码器,导致GroupNormalizer处理分组时,错误将数值索引传入symbol的编码器,触发KeyError:0。

修复后的代码

!pip install numpy==1.24.4
!pip install pandas
!pip install --ignore-installed PyYAML==5.4.1
!pip install torch
!pip install pytorch-lightning
!pip install pytorch-forecasting
!pip install statsmodels
!pip install scipy
!pip install matplotlib

import pandas as pd
from pytorch_forecasting import TimeSeriesDataSet
from pytorch_forecasting.data.encoders import NaNLabelEncoder, GroupNormalizer

# Create a small sample dataframe
sample_data = pd.DataFrame({
    'symbol': ['AAPL', 'AAPL', 'AAPL', 'GOOG', 'GOOG', 'GOOG'],
    'period': [1, 2, 3, 1, 2, 3],
    'close': [150.0, 152.0, 151.0, 2800.0, 2820.0, 2810.0],
    'day_of_epoch': [19844, 19844, 19844, 19845, 19845, 19845]
})

# Ensure 'period' column is treated as integer index
sample_data['period'] = sample_data['period'].astype(int)
# 将day_of_epoch转为字符串,作为分类变量处理
sample_data['day_of_epoch'] = sample_data['day_of_epoch'].astype(str)

# Print the dataframe to verify
print(sample_data)

# Define the maximum prediction length
max_prediction_length = 1
max_encoder_length = 2

# Define the training dataset with corrected configuration
training = TimeSeriesDataSet(
    sample_data,
    time_idx="period",
    target="close",
    group_ids=["symbol", "day_of_epoch"],
    min_encoder_length=1,
    max_encoder_length=max_encoder_length,
    min_prediction_length=1,
    max_prediction_length=max_prediction_length,
    static_categoricals=["symbol", "day_of_epoch"],  # day_of_epoch作为静态分类变量
    time_varying_known_categoricals=[],
    time_varying_known_reals=["period"],  # 移除day_of_epoch,它不随时间变化
    time_varying_unknown_categoricals=[],
    time_varying_unknown_reals=["close"],
    target_normalizer=GroupNormalizer(
        groups=["symbol", "day_of_epoch"], transformation="softplus"
    ),
    add_relative_time_idx=True,
    add_target_scales=True,
    add_encoder_length=True,
    categorical_encoders={
        'symbol': NaNLabelEncoder(add_nan=True),
        'day_of_epoch': NaNLabelEncoder(add_nan=True)  # 为day_of_epoch添加分类编码器
    },
)

关键修改点

  • 将day_of_epoch转为字符串类型,作为分类分组变量。
  • 把day_of_epoch从time_varying_known_reals移至static_categoricals,匹配其静态属性。
  • 为day_of_epoch添加NaNLabelEncoder,确保所有分类分组变量都有对应的编码器。

内容的提问来源于stack exchange,提问作者Michael Ferrier

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 20:27:02