You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何解决反向传播中Sigmoid导数的溢出问题?

神经网络反向传播中Sigmoid导数溢出问题修复

问题描述

自行实现神经网络并训练,前向传播功能正常,但反向传播阶段计算Sigmoid导数的sig_derivation函数出现溢出。尝试过对z值使用np.round和np.format_float_positional,均未解决问题。

原代码如下:

import numpy as np
import tensorflow as tf

# Parameters
features = 3
hidden_layer = 3
outputs_num = 2
epochs = 1
batch_num = 10
sample_num = 100 # => m
alpha = 0.5  # Learning rate

rng = np.random.default_rng()
X = rng.integers(low=0, high=21, size=(features, sample_num))
Y = rng.integers(low=0, high=2, size=(outputs_num, sample_num))

class neural_net:
    activations = {'relu': tf.nn.relu,'tanh': tf.tanh, 'sigmoid': tf.sigmoid, 'softmax': tf.nn.softmax}
    
    def __init__(self, inputs :list, outputs :list, learning_rate :float, batche_num :int, num_layer :int, features :int, each_layer_neurons :list, each_layer_activation :list, sample_num :int, epoch :int, batch_num :int) -> None:
        self.activation_derivation = {'relu': self.relu_derivation,'tanh': self.tanh_derivation, 'sigmoid': self.sig_derivation, "softmax": self.sig_derivation}
        self.learning_rate = learning_rate
        self.numlayer = num_layer
        self.Y = outputs
        self.ac = each_layer_activation
        self.epochs = epoch
        self.n = sample_num

        self.batche = batche_num
        # Split data based on batches
        self.xbatches = np.split(inputs,batch_num,axis=1)
        self.ybatches = np.split(outputs,batch_num,axis=1)
        # Initializing weights and biases for each layer
        self.w = []
        self.b = []
        self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[0], features)) * 0.01)
        for i in range(num_layer-1):
            self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[i+1], each_layer_neurons[i])) * 0.01)

        for i in range(num_layer):
            self.b.append(np.zeros((each_layer_neurons[i],1)))
    
    def train(self):
        for i in range(self.epochs):
            loss = 0
            for c, b in enumerate(self.xbatches):
                self.a = []
                self.z = []
                self.a.append(b)
                self.forward()
                loss += self.loss(self.ybatches[c],self.a[self.numlayer])
                self.da = []
                self.dz = []
                self.dw = []
                self.db = []
                self.backward(c)
                self.update_parameters()
            print(f'Epoch {i + 1}, Loss(Cost): {loss/self.batche}')
    
    def forward(self):
        for i in range(self.numlayer):
            self.z.append(np.dot(self.w[i],self.a[i]) + self.b[i])
            self.a.append(np.array(self.activations[self.ac[i]](self.z[i])))

           
    def backward(self, batchnum):
        l = self.numlayer
        self.da.append(-(np.divide(self.ybatches[batchnum], self.a[l]) - np.divide(1 - self.ybatches[batchnum], 1 - self.a[l])))
        for i in range(l-1,-1,-1):
            derivative = self.activation_derivation[self.ac[i]](self.z[i])
            self.dz.append(np.multiply(self.da[l-1-i], derivative))
            self.dw.append(np.dot(self.dz[l-1-i], self.a[i].T) / (self.n // self.batche))
            self.db.append(np.sum(self.dz[l-1-i],axis=1,keepdims=True) / (self.n // self.batche))
            if i > 0:
                self.da.append(np.dot(self.w[i].T, self.dz[l-1-i]))

    def update_parameters(self):
        l = self.numlayer
        for i in range(l):
            self.w[i] -= self.learning_rate * self.dw[l-1-i]
            self.b[i] -= self.learning_rate * self.db[l-1-i]

    def loss(self, real, current):
        return -np.sum(real * np.log(current) + (1 - real) * np.log(1 - current)) / (self.n // self.batche)

    def tanh_derivation(self, mat):
        return (1 - np.tanh(mat) ** 2)
    
    def sig_derivation(self, mat):
        return mat * (1 - mat)
    
    def relu_derivation(self, mat):
        return (mat > 0).astype(float)
    
setup = neural_net(X,Y,alpha,batch_num,hidden_layer,features,[3,4,2],['relu','relu','softmax'],sample_num,epochs,batch_num)
setup.train()

问题根源分析

  1. Softmax导数误用:输出层使用Softmax激活,但反向传播中直接复用了Sigmoid的导数计算,两者导数公式完全不同,会导致计算错误和数值不稳定。
  2. Sigmoid导数输入错误:代码中调用sig_derivation时传入的是z值(线性输出),但Sigmoid导数的正确计算应该基于激活后的输出a(即σ(z)),公式为σ'(z) = a*(1-a),传入z会直接导致计算错误。
  3. 输入未归一化:原始输入是0-20的整数,导致z值过大,激活函数输出趋近于0或1,计算np.log(current)或np.log(1-current)时会出现无穷大,引发溢出。
  4. 损失函数无数值保护:当激活输出接近0或1时,np.log会返回无穷大,直接导致损失和梯度计算溢出。

修复方案

1. 修正Softmax反向传播逻辑

对于Softmax+交叉熵损失的组合,梯度可以简化为a[l] - y,无需单独计算复杂的Softmax导数,既高效又避免数值问题。

2. 修正Sigmoid导数实现

将Sigmoid导数的输入改为激活后的输出a,确保公式a*(1-a)的计算正确。

3. 输入数据归一化

将输入X归一化到0-1范围,避免z值过大导致激活函数饱和。

4. 损失函数添加数值裁剪

用np.clip限制激活输出的范围,避免出现0或1,防止np.log计算溢出。

修改后的代码

import numpy as np
import tensorflow as tf

# Parameters
features = 3
hidden_layer = 3
outputs_num = 2
epochs = 5
batch_num = 10
sample_num = 100 # => m
alpha = 0.5  # Learning rate

rng = np.random.default_rng()
# 输入归一化到0-1范围
X = rng.integers(low=0, high=21, size=(features, sample_num)) / 20.0
Y = rng.integers(low=0, high=2, size=(outputs_num, sample_num))

class neural_net:
    activations = {'relu': tf.nn.relu,'tanh': tf.tanh, 'sigmoid': tf.sigmoid, 'softmax': tf.nn.softmax}
    
    def __init__(self, inputs :list, outputs :list, learning_rate :float, batche_num :int, num_layer :int, features :int, each_layer_neurons :list, each_layer_activation :list, sample_num :int, epoch :int, batch_num :int) -> None:
        # 移除Softmax到Sigmoid导数的错误映射,后续单独处理
        self.activation_derivation = {'relu': self.relu_derivation,'tanh': self.tanh_derivation, 'sigmoid': self.sig_derivation}
        self.learning_rate = learning_rate
        self.numlayer = num_layer
        self.Y = outputs
        self.ac = each_layer_activation
        self.epochs = epoch
        self.n = sample_num

        self.batche = batche_num
        # Split data based on batches
        self.xbatches = np.split(inputs,batch_num,axis=1)
        self.ybatches = np.split(outputs,batch_num,axis=1)
        # Initializing weights and biases for each layer
        self.w = []
        self.b = []
        self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[0], features)) * 0.01)
        for i in range(num_layer-1):
            self.w.append(rng.integers(low=0, high=11, size=(each_layer_neurons[i+1], each_layer_neurons[i])) * 0.01)

        for i in range(num_layer):
            self.b.append(np.zeros((each_layer_neurons[i],1)))
    
    def train(self):
        for i in range(self.epochs):
            loss = 0
            for c, b in enumerate(self.xbatches):
                self.a = []
                self.z = []
                self.a.append(b)
                self.forward()
                loss += self.loss(self.ybatches[c],self.a[self.numlayer])
                self.da = []
                self.dz = []
                self.dw = []
                self.db = []
                self.backward(c)
                self.update_parameters()
            print(f'Epoch {i + 1}, Loss(Cost): {loss/self.batche:.4f}')
    
    def forward(self):
        for i in range(self.numlayer):
            self.z.append(np.dot(self.w[i],self.a[i]) + self.b[i])
            self.a.append(np.array(self.activations[self.ac[i]](self.z[i])))

           
    def backward(self, batchnum):
        l = self.numlayer
        y = self.ybatches[batchnum]
        a_last = self.a[l]
        # 处理输出层为Softmax的情况,使用简化梯度公式
        if self.ac[-1] == 'softmax':
            self.da.append(a_last - y)
        else:
            self.da.append(-(np.divide(y, a_last) - np.divide(1 - y, 1 - a_last)))
        
        for i in range(l-1,-1,-1):
            if self.ac[i] == 'softmax':
                # Softmax梯度已通过简化公式处理,直接用da计算dz
                derivative = 1.0
            else:
                # 修正:传入激活后的输出a,而非z值
                derivative = self.activation_derivation[self.ac[i]](self.a[i+1])
            
            self.dz.append(np.multiply(self.da[l-1-i], derivative))
            batch_size = self.n // self.batche
            self.dw.append(np.dot(self.dz[l-1-i], self.a[i].T) / batch_size)
            self.db.append(np.sum(self.dz[l-1-i],axis=1,keepdims=True) / batch_size)
            if i > 0:
                self.da.append(np.dot(self.w[i].T, self.dz[l-1-i]))

    def update_parameters(self):
        l = self.numlayer
        for i in range(l):
            self.w[i] -= self.learning_rate * self.dw[l-1-i]
            self.b[i] -= self.learning_rate * self.db[l-1-i]

    def loss(self, real, current):
        # 添加数值裁剪,避免log(0)或log(1)引发溢出
        current = np.clip(current, 1e-10, 1 - 1e-10)
        return -np.sum(real * np.log(current) + (1 - real) * np.log(1 - current)) / (self.n // self.batche)

    def tanh_derivation(self, mat):
        return (1 - np.tanh(mat) ** 2)
    
    def sig_derivation(self, mat):
        # mat为激活后的输出a,直接使用正确公式计算导数
        return mat * (1 - mat)
    
    def relu_derivation(self, mat):
        # 通过激活输出a判断导数:a>0则导数为1,否则为0
        return (mat > 0).astype(float)
    
setup = neural_net(X,Y,alpha,batch_num,hidden_layer,features,[3,4,2],['relu','relu','softmax'],sample_num,epochs,batch_num)
setup.train()

关键修改点说明

  • 输入X归一化:将0-20的整数转为0-1的浮点数,避免z值过大。
  • 损失函数添加np.clip:限制输出范围在1e-10到1-1e-10之间,防止log计算溢出。
  • 修正Softmax反向传播:使用a_last - y作为初始梯度,替代错误的Sigmoid导数复用。
  • 修正Sigmoid导数输入:传入激活后的输出self.a[i+1],确保导数计算符合公式。

内容的提问来源于stack exchange,提问作者Ali Nazeri

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 14:52:05