如何解决反向传播中Sigmoid导数的溢出问题?
神经网络反向传播中Sigmoid导数溢出问题修复
问题描述
自行实现神经网络并训练,前向传播功能正常,但反向传播阶段计算Sigmoid导数的sig_derivation函数出现溢出。尝试过对z值使用np.round和np.format_float_positional,均未解决问题。
原代码如下:
import numpy as np import tensorflow as tf # Parameters features = 3 hidden_layer = 3 outputs_num = 2 epochs = 1 batch_num = 10 sample_num = 100 # => m alpha = 0.5 # Learning rate rng = np.random.default_rng() X = rng.integers(low=0, high=21, size=(features, sample_num)) Y = rng.integers(low=0, high=2, size=(outputs_num, sample_num)) class neural_net: activations = {'relu': tf.nn.relu,'tanh': tf.tanh, 'sigmoid': tf.sigmoid, 'softmax': tf.nn.softmax} def __init__(self, inputs :list, outputs :list, learning_rate :float, batche_num :int, num_layer :int, features :int, each_layer_neurons :list, each_layer_activation :list, sample_num :int, epoch :int, batch_num :int) -> None: self.activation_derivation = {'relu': self.relu_derivation,'tanh': self.tanh_derivation, 'sigmoid': self.sig_derivation, "softmax": self.sig_derivation} self.learning_rate = learning_rate self.numlayer = num_layer self.Y = outputs self.ac = each_layer_activation self.epochs = epoch self.n = sample_num self.batche = batche_num # Split data based on batches self.xbatches = np.split(inputs,batch_num,axis=1) self.ybatches = np.split(outputs,batch_num,axis=1) # Initializing weights and biases for each layer self.w = [] self.b = [] self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[0], features)) * 0.01) for i in range(num_layer-1): self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[i+1], each_layer_neurons[i])) * 0.01) for i in range(num_layer): self.b.append(np.zeros((each_layer_neurons[i],1))) def train(self): for i in range(self.epochs): loss = 0 for c, b in enumerate(self.xbatches): self.a = [] self.z = [] self.a.append(b) self.forward() loss += self.loss(self.ybatches[c],self.a[self.numlayer]) self.da = [] self.dz = [] self.dw = [] self.db = [] self.backward(c) self.update_parameters() print(f'Epoch {i + 1}, Loss(Cost): {loss/self.batche}') def forward(self): for i in range(self.numlayer): self.z.append(np.dot(self.w[i],self.a[i]) + self.b[i]) self.a.append(np.array(self.activations[self.ac[i]](self.z[i]))) def backward(self, batchnum): l = self.numlayer self.da.append(-(np.divide(self.ybatches[batchnum], self.a[l]) - np.divide(1 - self.ybatches[batchnum], 1 - self.a[l]))) for i in range(l-1,-1,-1): derivative = self.activation_derivation[self.ac[i]](self.z[i]) self.dz.append(np.multiply(self.da[l-1-i], derivative)) self.dw.append(np.dot(self.dz[l-1-i], self.a[i].T) / (self.n // self.batche)) self.db.append(np.sum(self.dz[l-1-i],axis=1,keepdims=True) / (self.n // self.batche)) if i > 0: self.da.append(np.dot(self.w[i].T, self.dz[l-1-i])) def update_parameters(self): l = self.numlayer for i in range(l): self.w[i] -= self.learning_rate * self.dw[l-1-i] self.b[i] -= self.learning_rate * self.db[l-1-i] def loss(self, real, current): return -np.sum(real * np.log(current) + (1 - real) * np.log(1 - current)) / (self.n // self.batche) def tanh_derivation(self, mat): return (1 - np.tanh(mat) ** 2) def sig_derivation(self, mat): return mat * (1 - mat) def relu_derivation(self, mat): return (mat > 0).astype(float) setup = neural_net(X,Y,alpha,batch_num,hidden_layer,features,[3,4,2],['relu','relu','softmax'],sample_num,epochs,batch_num) setup.train()
问题根源分析
- Softmax导数误用:输出层使用Softmax激活,但反向传播中直接复用了Sigmoid的导数计算,两者导数公式完全不同,会导致计算错误和数值不稳定。
- Sigmoid导数输入错误:代码中调用
sig_derivation时传入的是z值(线性输出),但Sigmoid导数的正确计算应该基于激活后的输出a(即σ(z)),公式为σ'(z) = a*(1-a),传入z会直接导致计算错误。 - 输入未归一化:原始输入是0-20的整数,导致z值过大,激活函数输出趋近于0或1,计算
np.log(current)或np.log(1-current)时会出现无穷大,引发溢出。 - 损失函数无数值保护:当激活输出接近0或1时,
np.log会返回无穷大,直接导致损失和梯度计算溢出。
修复方案
1. 修正Softmax反向传播逻辑
对于Softmax+交叉熵损失的组合,梯度可以简化为a[l] - y,无需单独计算复杂的Softmax导数,既高效又避免数值问题。
2. 修正Sigmoid导数实现
将Sigmoid导数的输入改为激活后的输出a,确保公式a*(1-a)的计算正确。
3. 输入数据归一化
将输入X归一化到0-1范围,避免z值过大导致激活函数饱和。
4. 损失函数添加数值裁剪
用np.clip限制激活输出的范围,避免出现0或1,防止np.log计算溢出。
修改后的代码
import numpy as np import tensorflow as tf # Parameters features = 3 hidden_layer = 3 outputs_num = 2 epochs = 5 batch_num = 10 sample_num = 100 # => m alpha = 0.5 # Learning rate rng = np.random.default_rng() # 输入归一化到0-1范围 X = rng.integers(low=0, high=21, size=(features, sample_num)) / 20.0 Y = rng.integers(low=0, high=2, size=(outputs_num, sample_num)) class neural_net: activations = {'relu': tf.nn.relu,'tanh': tf.tanh, 'sigmoid': tf.sigmoid, 'softmax': tf.nn.softmax} def __init__(self, inputs :list, outputs :list, learning_rate :float, batche_num :int, num_layer :int, features :int, each_layer_neurons :list, each_layer_activation :list, sample_num :int, epoch :int, batch_num :int) -> None: # 移除Softmax到Sigmoid导数的错误映射,后续单独处理 self.activation_derivation = {'relu': self.relu_derivation,'tanh': self.tanh_derivation, 'sigmoid': self.sig_derivation} self.learning_rate = learning_rate self.numlayer = num_layer self.Y = outputs self.ac = each_layer_activation self.epochs = epoch self.n = sample_num self.batche = batche_num # Split data based on batches self.xbatches = np.split(inputs,batch_num,axis=1) self.ybatches = np.split(outputs,batch_num,axis=1) # Initializing weights and biases for each layer self.w = [] self.b = [] self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[0], features)) * 0.01) for i in range(num_layer-1): self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[i+1], each_layer_neurons[i])) * 0.01) for i in range(num_layer): self.b.append(np.zeros((each_layer_neurons[i],1))) def train(self): for i in range(self.epochs): loss = 0 for c, b in enumerate(self.xbatches): self.a = [] self.z = [] self.a.append(b) self.forward() loss += self.loss(self.ybatches[c],self.a[self.numlayer]) self.da = [] self.dz = [] self.dw = [] self.db = [] self.backward(c) self.update_parameters() print(f'Epoch {i + 1}, Loss(Cost): {loss/self.batche:.4f}') def forward(self): for i in range(self.numlayer): self.z.append(np.dot(self.w[i],self.a[i]) + self.b[i]) self.a.append(np.array(self.activations[self.ac[i]](self.z[i]))) def backward(self, batchnum): l = self.numlayer y = self.ybatches[batchnum] a_last = self.a[l] # 处理输出层为Softmax的情况,使用简化梯度公式 if self.ac[-1] == 'softmax': self.da.append(a_last - y) else: self.da.append(-(np.divide(y, a_last) - np.divide(1 - y, 1 - a_last))) for i in range(l-1,-1,-1): if self.ac[i] == 'softmax': # Softmax梯度已通过简化公式处理,直接用da计算dz derivative = 1.0 else: # 修正:传入激活后的输出a,而非z值 derivative = self.activation_derivation[self.ac[i]](self.a[i+1]) self.dz.append(np.multiply(self.da[l-1-i], derivative)) batch_size = self.n // self.batche self.dw.append(np.dot(self.dz[l-1-i], self.a[i].T) / batch_size) self.db.append(np.sum(self.dz[l-1-i],axis=1,keepdims=True) / batch_size) if i > 0: self.da.append(np.dot(self.w[i].T, self.dz[l-1-i])) def update_parameters(self): l = self.numlayer for i in range(l): self.w[i] -= self.learning_rate * self.dw[l-1-i] self.b[i] -= self.learning_rate * self.db[l-1-i] def loss(self, real, current): # 添加数值裁剪,避免log(0)或log(1)引发溢出 current = np.clip(current, 1e-10, 1 - 1e-10) return -np.sum(real * np.log(current) + (1 - real) * np.log(1 - current)) / (self.n // self.batche) def tanh_derivation(self, mat): return (1 - np.tanh(mat) ** 2) def sig_derivation(self, mat): # mat为激活后的输出a,直接使用正确公式计算导数 return mat * (1 - mat) def relu_derivation(self, mat): # 通过激活输出a判断导数:a>0则导数为1,否则为0 return (mat > 0).astype(float) setup = neural_net(X,Y,alpha,batch_num,hidden_layer,features,[3,4,2],['relu','relu','softmax'],sample_num,epochs,batch_num) setup.train()
关键修改点说明
- 输入X归一化:将0-20的整数转为0-1的浮点数,避免z值过大。
- 损失函数添加
np.clip:限制输出范围在1e-10到1-1e-10之间,防止log计算溢出。 - 修正Softmax反向传播:使用
a_last - y作为初始梯度,替代错误的Sigmoid导数复用。 - 修正Sigmoid导数输入:传入激活后的输出
self.a[i+1],确保导数计算符合公式。
内容的提问来源于stack exchange,提问作者Ali Nazeri
相关产品推荐
相关产品推荐

