时间序列预测中LSTM验证损失波动问题咨询
时间序列预测中LSTM验证损失异常波动的问题咨询
本人开展时间序列预测研究数周,采用ARIMA与LSTM模型进行预测对比实验:
- 实验结果图(图1)包含4个子图:左上为ARIMA训练数据与拟合点,右上为ARIMA测试与预测点,左下为LSTM训练数据与预测拟合点(可忽略),右下为LSTM测试与预测点
- 计算RMSE、MSE指标后,LSTM误差更低,与文献中“LSTM优于ARIMA”的结论一致
但绘制LSTM的损失与验证损失曲线(图2)时,发现验证损失呈现异常波动模式。本人推测原因是时间序列存在大量异常值/离群点,导致划分的验证集无法有效反映模型学习效果,但缺乏文献支撑,特此咨询专业见解。
实验代码
import pandas as pd import matplotlib.pyplot as plt from statsmodels.tsa.stattools import adfuller from statsmodels.graphics.tsaplots import plot_acf,plot_pacf import statsmodels.api as sm from statsmodels.tsa.arima.model import ARIMA from sklearn.metrics import r2_score,mean_squared_error,mean_absolute_percentage_error,mean_absolute_percentage_error from statsmodels.tsa.seasonal import STL import numpy as np from pandas import Series, DataFrame from scipy import stats from statsmodels.tsa.stattools import adfuller import statsmodels from statsmodels.tsa.seasonal import seasonal_decompose from pandas.plotting import register_matplotlib_converters import pmdarima as pm register_matplotlib_converters() import warnings import time from numpy import array from keras.models import Sequential from keras.layers import LSTM from keras.layers import Dense from numpy import array import keras_tuner as kt import tensorflow as tf print(tf.__version__) from numpy import array from tensorflow import keras import keras_tuner as kt from sklearn.preprocessing import MinMaxScaler from keras.layers import Bidirectional from keras.models import Sequential from keras.preprocessing.sequence import TimeseriesGenerator from keras.layers import Bidirectional from tensorflow.keras import initializers import random as rn np.random.seed(123) rn.seed(123) tf.random.set_seed(123) tf.keras.utils.set_random_seed(123) keras.utils.set_random_seed(123) warnings.filterwarnings('ignore') df3 = pd.read_csv('favorita_train.csv') ## 1 - Get TS and do STL print("TS lenbgth : "+str(len(df3))) results = seasonal_decompose(df3['unit_sales'],period=30) results.plot(); train_all = df3.iloc[:int(len(df3)*0.8)] train = df3.iloc[:int(len(df3)*0.6)] val = df3.iloc[int(len(df3)*0.6):int(len(df3)*0.8)] test = df3.iloc[int(len(df3)*0.8):] scaler = MinMaxScaler() scaler.fit(train_all) scaled_all = scaler.transform(df3) scaled_train = scaler.transform(train) scaled_train_all = scaler.transform(train_all) scaled_val = scaler.transform(val) scaled_test = scaler.transform(test) # We do the same thing, but now instead for 12 months n_features = 1 n_input =5 train_generator_all = TimeseriesGenerator(scaled_train_all, scaled_train_all, length=n_input, batch_size=1,shuffle=True) train_generator = TimeseriesGenerator(scaled_train, scaled_train, length=n_input, batch_size=1,shuffle=True) val_generator = TimeseriesGenerator(scaled_val, scaled_val, length=n_input, batch_size=1,shuffle=True) adfPValue = adfuller(scaled_all) adfPValue=adfPValue[1] adi = len(scaled_all)/((scaled_all != 0).sum()) sd=scaled_all.std() mean=scaled_all.mean() cv2 = np.square(sd/mean) print("CV2 (describe magnitude of demande variability <0.5 is good):"+str(cv2)) print("SD (-2,2 is good | mean data variance is low):"+str(sd)) print("ADI (1.3 or smaller means smooth ts):"+str(adi)) print("Stationarity test (stationary if <0.05):"+str(adfPValue)) def model_builder(hp): model = keras.Sequential() hp_units = hp.Int('units', min_value=1, max_value=50, step=1) hp_layers = hp.Int('layers', min_value=1, max_value=3, step=1) if hp_layers==1 : model.add(Bidirectional(LSTM(hp_units,activation='relu'), input_shape=(n_input, n_features))) elif hp_layers==2: model.add(Bidirectional(LSTM(hp_units, activation='relu', return_sequences=True), input_shape=(n_input, n_features))) model.add(Bidirectional(LSTM(hp_units, activation='relu'))) else: model.add(Bidirectional(LSTM(hp_units, activation='relu', return_sequences=True), input_shape=(n_input, n_features))) for i in range(hp_layers-2): model.add(Bidirectional(LSTM(hp_units, activation='relu', return_sequences=True))) model.add(Bidirectional(LSTM(hp_units, activation='relu'))) model.add(Dense(1)) hp_learning_rate = hp.Choice('learning_rate', values=[1e-2, 1e-3, 1e-4]) model.compile(optimizer=keras.optimizers.Adam(learning_rate=hp_learning_rate), loss='mse',metrics=['accuracy']) return model tuner = kt.Hyperband(model_builder, objective='val_loss', max_epochs=300, factor=3, directory='499', project_name='949', seed=123) stop_early = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=30) tuner.search(train_generator, epochs=300, validation_data=val_generator, shuffle=True, callbacks=[stop_early], batch_size=len(train_generator)) best_hps=tuner.get_best_hyperparameters(num_trials=1)[0] print(best_hps.get('units')) print(best_hps.get('layers')) print(best_hps.get('window')) print(best_hps.get('learning_rate')) best_hps=tuner.get_best_hyperparameters(num_trials=1)[0] model = tuner.hypermodel.build(best_hps) history = model.fit(img_train, label_train, epochs=50, validation_split=0.2) val_acc_per_epoch = history.history['val_accuracy'] best_epoch = val_acc_per_epoch.index(max(val_acc_per_epoch)) + 1 print('Best epoch: %d' % (best_epoch,))
实验结果图
图1:ARIMA与LSTM模型拟合/预测对比

图2:LSTM训练损失与验证损失曲线

内容的提问来源于stack exchange,提问作者Bilel Abderrahmane BENZIANE
相关产品推荐
相关产品推荐

