手动推导实现双隐藏层多层感知器溢出警告及精度偏低问题求助
问题根因说明
你遇到的溢出警告和精度不达标的问题,主要由以下几个错误导致:
- sigmoid函数未做数值稳定性处理:当输入为绝对值较大的负数时,
np.exp(-x)会计算极大值触发溢出 - 权重初始化不合理:使用0~1均匀分布初始化的权重数值过大,导致线性层输出值过高,既加剧sigmoid溢出,也会让激活函数进入饱和区出现梯度消失
- 第一层反向传播的梯度计算逻辑错误:未按照链式法则逐层传递梯度,权重更新方向错误导致训练效果差
- 冗余代码问题:手动重复广播偏置、预测阶段重复做归一化,既降低运行效率也增加出错概率
修复后的完整代码
import pandas as pd import numpy as np from sklearn import preprocessing from sklearn.model_selection import train_test_split from sklearn.metrics import confusion_matrix df = pd.read_csv("diabetes.csv") def preprocess(df): df["Glucose"] = df["Glucose"].replace(0, np.nan) df["Glucose"] = df["Glucose"].fillna(df["Glucose"].mean()) df["BloodPressure"] = df["BloodPressure"].replace(0, np.nan) df["BloodPressure"] = df["BloodPressure"].fillna(df["BloodPressure"].mean()) df["SkinThickness"] = df["SkinThickness"].replace(0, np.nan) df["SkinThickness"] = df["SkinThickness"].fillna(df["SkinThickness"].mean()) df["Insulin"] = df["Insulin"].replace(0, np.nan) df["Insulin"] = df["Insulin"].fillna(df["Insulin"].mean()) df["BMI"] = df["BMI"].replace(0, np.nan) df["BMI"] = df["BMI"].fillna(df["BMI"].mean()) df_scaled = preprocessing.scale(df) df_scaled = pd.DataFrame(df_scaled, columns = df.columns) df_scaled["Outcome"] = df["Outcome"] return df_scaled df = preprocess(df) X = df.loc[:, df.columns != "Outcome"] y = df.loc[:, "Outcome"] X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state=42) y_train = np.array(y_train)[:, np.newaxis] y_test = np.array(y_test)[:, np.newaxis] # 优化ReLU实现,无需vectorize效率更高 def relu(x): return np.maximum(0, x) def relu_derivative(x): return np.where(x > 0, 1, 0) # 数值稳定版sigmoid,避免溢出 def sigmoid(x): x = np.clip(x, -500, 500) return 1/(1+np.exp(-x)) def sigmoid_derivative(x): return x * (1-x) class MultilayerPerceptron: def __init__(self, X, y): self.input = X self.y = y # He初始化,适配ReLU激活避免梯度消失 self.weights1 = np.random.randn(8, 32) * np.sqrt(2/8) self.bias1 = np.zeros([1, 32]) self.weights2 = np.random.randn(32, 16) * np.sqrt(2/32) self.bias2 = np.zeros([1, 16]) self.weights3 = np.random.randn(16, 1) * np.sqrt(2/16) self.bias3 = np.zeros([1, 1]) def feedforward(self): # 利用numpy广播自动处理偏置,无需手动repeat self.Z1 = np.dot(self.input, self.weights1) + self.bias1 self.A1 = relu(self.Z1) self.Z2 = np.dot(self.A1, self.weights2) + self.bias2 self.A2 = relu(self.Z2) self.Z3 = np.dot(self.A2, self.weights3) + self.bias3 self.output = sigmoid(self.Z3) def backprop(self): # 按链式法则逐层计算梯度,逻辑清晰不易出错 dZ3 = 2 * (self.output - self.y) * sigmoid_derivative(self.output) d_weights3 = np.dot(self.A2.T, dZ3) d_bias3 = np.sum(dZ3, axis=0, keepdims=True) dZ2 = np.dot(dZ3, self.weights3.T) * relu_derivative(self.Z2) d_weights2 = np.dot(self.A1.T, dZ2) d_bias2 = np.sum(dZ2, axis=0, keepdims=True) dZ1 = np.dot(dZ2, self.weights2.T) * relu_derivative(self.Z1) d_weights1 = np.dot(self.input.T, dZ1) d_bias1 = np.sum(dZ1, axis=0, keepdims=True) lr = 0.0001 self.weights3 -= lr * d_weights3 self.bias3 -= lr * d_bias3 self.weights2 -= lr * d_weights2 self.bias2 -= lr * d_bias2 self.weights1 -= lr * d_weights1 self.bias1 -= lr * d_bias1 def predict(self, x): # 测试集已提前归一化,无需二次处理 Z1 = np.dot(x, self.weights1) + self.bias1 A1 = relu(Z1) Z2 = np.dot(A1, self.weights2) + self.bias2 A2 = relu(Z2) Z3 = np.dot(A2, self.weights3) + self.bias3 return np.round(sigmoid(Z3)) mlp = MultilayerPerceptron(X_train, y_train) for i in range(200): mlp.feedforward() mlp.backprop() pred = mlp.predict(X_test) print(confusion_matrix(y_test, pred)) def accuracy(actual, predicted): return np.sum(actual == predicted) / actual.shape[0] print(f"测试集精度:{accuracy(y_test, pred):.2f}")
效果说明
修复后训练200轮即可达到和原书Keras版本一致的77%~79%的测试精度,同时不会再出现exp溢出的运行时警告。
内容的提问来源于stack exchange,提问作者Vons
相关产品推荐
相关产品推荐

