使用GroupNormalizer创建TimeSeriesDataSet时触发KeyError:0求助
问题描述
尝试基于PyTorch Forecasting构建Temporal Fusion Transformer时,始终触发KeyError: 0错误,数据中并无0值。切换为最小测试数据集、简化代码后错误仍存在,且仅在TimeSeriesDataSet中使用GroupNormalizer时触发。运行环境为PaperSpace Jupyter Notebook的Python 3.9.13容器。
测试代码
!pip install numpy==1.24.4 !pip install pandas !pip install --ignore-installed PyYAML==5.4.1 !pip install torch !pip install pytorch-lightning !pip install pytorch-forecasting !pip install statsmodels !pip install scipy !pip install matplotlib import pandas as pd from pytorch_forecasting import TimeSeriesDataSet from pytorch_forecasting.data.encoders import NaNLabelEncoder, GroupNormalizer # Create a small sample dataframe sample_data = pd.DataFrame({ 'symbol': ['AAPL', 'AAPL', 'AAPL', 'GOOG', 'GOOG', 'GOOG'], 'period': [1, 2, 3, 1, 2, 3], 'close': [150.0, 152.0, 151.0, 2800.0, 2820.0, 2810.0], 'day_of_epoch': [19844, 19844, 19844, 19845, 19845, 19845] }) # Ensure 'period' column is treated as integer index sample_data['period'] = sample_data['period'].astype(int) # Print the dataframe to verify print(sample_data) # Define the maximum prediction length max_prediction_length = 1 max_encoder_length = 2 # Define the training dataset with minimal configuration training = TimeSeriesDataSet( sample_data, time_idx="period", target="close", group_ids=["symbol", "day_of_epoch"], min_encoder_length=1, max_encoder_length=max_encoder_length, min_prediction_length=1, max_prediction_length=max_prediction_length, static_categoricals=["symbol"], time_varying_known_categoricals=[], time_varying_known_reals=["period", "day_of_epoch"], time_varying_unknown_categoricals=[], time_varying_unknown_reals=["close"], target_normalizer=GroupNormalizer( groups=["symbol", "day_of_epoch"], transformation="softplus" ), add_relative_time_idx=True, add_target_scales=True, add_encoder_length=True, categorical_encoders={ 'symbol': NaNLabelEncoder(add_nan=True) }, )
报错输出
symbol period close day_of_epoch 0 AAPL 1 150.0 19844 1 AAPL 2 152.0 19844 2 AAPL 3 151.0 19844 3 GOOG 1 2800.0 19845 4 GOOG 2 2820.0 19845 5 GOOG 3 2810.0 19845 --------------------------------------------------------------------------- KeyError Traceback (most recent call last) File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/encoders.py:331, in NaNLabelEncoder.transform(self, y, return_norm, target_scale, ignore_na) 330 try: --> 331 encoded = [self.classes_[v] for v in y] 332 except KeyError as e: File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/encoders.py:331, in <listcomp>(.0) 330 try: --> 331 encoded = [self.classes_[v] for v in y] 332 except KeyError as e: KeyError: 0 During handling of the above exception, another exception occurred: KeyError Traceback (most recent call last) Input In [1], in <cell line: 34>() 31 max_encoder_length = 2 33 # Define the training dataset with minimal configuration ---> 34 training = TimeSeriesDataSet( 35 sample_data, 36 time_idx="period", 37 target="close", 38 group_ids=["symbol", "day_of_epoch"], 39 min_encoder_length=1, 40 max_encoder_length=max_encoder_length, 41 min_prediction_length=1, 42 max_prediction_length=max_prediction_length, 43 static_categoricals=["symbol"], 44 time_varying_known_categoricals=[], 45 time_varying_known_reals=["period", "day_of_epoch"], 46 time_varying_unknown_categoricals=[], 47 time_varying_unknown_reals=["close"], 48 target_normalizer=GroupNormalizer( 49 groups=["symbol", "day_of_epoch"], transformation="softplus" 50 ), 51 add_relative_time_idx=True, 52 add_target_scales=True, 53 add_encoder_length=True, 54 categorical_encoders={ 55 'symbol': NaNLabelEncoder(add_nan=True) 56 }, 57 ) File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/timeseries.py:476, in TimeSeriesDataSet.__init__(self, data, time_idx, target, group_ids, weight, max_encoder_length, min_encoder_length, min_prediction_idx, min_prediction_length, max_prediction_length, static_categoricals, static_reals, time_varying_known_categoricals, time_varying_known_reals, time_varying_unknown_categoricals, time_varying_unknown_reals, variable_groups, constant_fill_strategy, allow_missing_timesteps, lags, add_relative_time_idx, add_target_scales, add_encoder_length, target_normalizer, categorical_encoders, scalers, randomize_length, predict_mode) 473 data = data.sort_values(self.group_ids + [self.time_idx]) 475 # preprocess data --> 476 data = self._preprocess_data(data) 477 for target in self.target_names: 478 assert target not in self.scalers, "Target normalizer is separate and not in scalers." File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/timeseries.py:837, in TimeSeriesDataSet._preprocess_data(self, data) 831 transformer = self.get_transformer(name) 832 if ( 833 name not in self.target_names 834 and transformer is not None 835 and not isinstance(transformer, EncoderNormalizer) 836 ): --> 837 data[name] = self.transform_values(name, data[name], data=data, inverse=False) 839 # encode lagged categorical targets 840 for name in self.lagged_targets: 841 # normalizer only now available File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/timeseries.py:935, in TimeSeriesDataSet.transform_values(self, name, values, data, inverse, group_id, **kwargs) 933 # remaining categories 934 if name in self.flat_categoricals + self.group_ids + self._group_ids: --> 935 return transform(values, **kwargs) 937 # reals 938 elif name in self.reals: File /usr/local/lib/python3.9/dist-packages/sklearn/utils/_set_output.py:313, in _wrap_method_output.<locals>.wrapped(self, X, *args, **kwargs) 311 @wraps(f) 312 def wrapped(self, X, *args, **kwargs): --> 313 data_to_wrap = f(self, X, *args, **kwargs) 314 if isinstance(data_to_wrap, tuple): 315 # only wrap the first output for cross decomposition 316 return_tuple = ( 317 _wrap_data_with_container(method, data_to_wrap[0], X, self), 318 *data_to_wrap[1:], 319 ) File /usr/local/lib/python3.9/dist-packages/pytorch_forecasting/data/encoders.py:333, in NaNLabelEncoder.transform(self, y, return_norm, target_scale, ignore_na) 331 encoded = [self.classes_[v] for v in y] 332 except KeyError as e: --> 333 raise KeyError( 334 f"Unknown category '{e.args[0]}' encountered. Set `add_nan=True` to allow unknown categories" 335 ) 337 if isinstance(y, torch.Tensor): 338 encoded = torch.tensor(encoded, dtype=torch.long, device=y.device) KeyError: "Unknown category '0' encountered. Set `add_nan=True` to allow unknown categories"
解决方案
错误根源
day_of_epoch是每个分组的固定值,却被标记为time_varying_known_reals(随时间变化的实数),属性定义错误导致预处理逻辑混乱。- 分组变量
day_of_epoch为整数类型,未转为分类变量也未配置对应编码器,导致GroupNormalizer处理分组时,错误将数值索引传入symbol的编码器,触发KeyError:0。
修复后的代码
!pip install numpy==1.24.4 !pip install pandas !pip install --ignore-installed PyYAML==5.4.1 !pip install torch !pip install pytorch-lightning !pip install pytorch-forecasting !pip install statsmodels !pip install scipy !pip install matplotlib import pandas as pd from pytorch_forecasting import TimeSeriesDataSet from pytorch_forecasting.data.encoders import NaNLabelEncoder, GroupNormalizer # Create a small sample dataframe sample_data = pd.DataFrame({ 'symbol': ['AAPL', 'AAPL', 'AAPL', 'GOOG', 'GOOG', 'GOOG'], 'period': [1, 2, 3, 1, 2, 3], 'close': [150.0, 152.0, 151.0, 2800.0, 2820.0, 2810.0], 'day_of_epoch': [19844, 19844, 19844, 19845, 19845, 19845] }) # Ensure 'period' column is treated as integer index sample_data['period'] = sample_data['period'].astype(int) # 将day_of_epoch转为字符串,作为分类变量处理 sample_data['day_of_epoch'] = sample_data['day_of_epoch'].astype(str) # Print the dataframe to verify print(sample_data) # Define the maximum prediction length max_prediction_length = 1 max_encoder_length = 2 # Define the training dataset with corrected configuration training = TimeSeriesDataSet( sample_data, time_idx="period", target="close", group_ids=["symbol", "day_of_epoch"], min_encoder_length=1, max_encoder_length=max_encoder_length, min_prediction_length=1, max_prediction_length=max_prediction_length, static_categoricals=["symbol", "day_of_epoch"], # day_of_epoch作为静态分类变量 time_varying_known_categoricals=[], time_varying_known_reals=["period"], # 移除day_of_epoch,它不随时间变化 time_varying_unknown_categoricals=[], time_varying_unknown_reals=["close"], target_normalizer=GroupNormalizer( groups=["symbol", "day_of_epoch"], transformation="softplus" ), add_relative_time_idx=True, add_target_scales=True, add_encoder_length=True, categorical_encoders={ 'symbol': NaNLabelEncoder(add_nan=True), 'day_of_epoch': NaNLabelEncoder(add_nan=True) # 为day_of_epoch添加分类编码器 }, )
关键修改点
- 将
day_of_epoch转为字符串类型,作为分类分组变量。 - 把
day_of_epoch从time_varying_known_reals移至static_categoricals,匹配其静态属性。 - 为
day_of_epoch添加NaNLabelEncoder,确保所有分类分组变量都有对应的编码器。
内容的提问来源于stack exchange,提问作者Michael Ferrier
相关产品推荐
相关产品推荐

