You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于神经网络实现数值加法:反向传播函数故障排查求助

神经网络实现数值加法时反向传播失效的问题排查

我尝试用神经网络实现数值加法功能,但反向传播函数始终无法正常工作。

神经网络结构

该神经网络结构定义为:W1 = x1,W2 = x2,W3 = y1,W4 = y2,W5 = z1,W6 = z2。以下是我编写的代码:

from random import randint, random, uniform
import numpy as np 

class Data:
    data_dict = {}
    def __init__(self, limit):
        self.limit = limit
    '''creates data but beware that the limit may not be the same as the size of the dictionary''' 
    def create_data(self):
        for i in range(self.limit):
            num1 = randint(0, 100)
            num2 = randint(0, 100)
            self.data_dict[(num1, num2)] = num1 + num2

''' you compare the error with every test in the data set and find weights that minimise the error'''
class Neural:
    def __init__(self, data):
        self.x1 = uniform(-1, 1) 
        self.x2 = uniform(-1, 1) 
        self.y1 = uniform(-1, 1) 
        self.y2 = uniform(-1, 1) 
        self.z1 = uniform(-1, 1) 
        self.z2 = uniform(-1, 1) 
        self.data = data
    
    def relu(self, number):
        return max(0, number)
    
    def sigmoid(self, number):
         return 1 / (1 + np.exp(-number))
        
    '''weighted summation with activation function to compute output '''
    def compute_output(self, num1, num2):
        hidden_layer_input1 = self.sigmoid((num1 * self.x1) + (num2 * self.y1))
        hidden_layer_input2 = self.sigmoid((num1 * self.x2) + (num2 * self.y2))
        return ((hidden_layer_input1 * self.z1) + (hidden_layer_input2 * self.z2))
            
    '''mean squared error between the actual output with the output generated by the algorithm '''
    def compare_output(self, data):
        '''actually, better to find error between all tests. add all the errors up'''
        error = 0
        for key in data.data_dict:
            error += abs(data.data_dict[key] - self.compute_output(key[0], key[1])) ** 2
        return error / len(data.data_dict)

    '''TODO function that changes the weight depending on the errors using gradient descent'''
    '''first make it random'''
    '''next perhaps change weights for each test and average out the adjustments for each weight'''
    def random_back_propagation(self):
        error = 100000
        while error > 0.1:
            self.x1 = random() 
            self.x2 = random()
            self.y1 = random()
            self.y2 = random()
            self.z1 = random()
            self.z2 = random()
            error = self.compare_output(self.data)
            print(error)
        print(self.compute_output(140, 15))     
        
        '''learning rate is the amount the weights are updated during training'''
    def back_propagation(self, learning_rate):
        
        for _ in range(1000):
            # 初始化梯度累积变量
            grad_x1, grad_x2 = 0, 0
            grad_y1, grad_y2 = 0, 0
            grad_z1, grad_z2 = 0, 0
            total_error = 0
            
            for key in self.data.data_dict:
                num1, num2 = key
                target = self.data.data_dict[key]
                
                # 归一化输入(解决sigmoid饱和问题)
                norm_num1 = num1 / 100.0
                norm_num2 = num2 / 100.0
                norm_target = target / 200.0  # 目标范围0-200,归一化到0-1
                
                # 前向传播
                hidden_sum1 = norm_num1 * self.x1 + norm_num2 * self.y1
                hidden_layer1_output = self.sigmoid(hidden_sum1)
                hidden_sum2 = norm_num1 * self.x2 + norm_num2 * self.y2
                hidden_layer2_output = self.sigmoid(hidden_sum2)
                
                output = hidden_layer1_output * self.z1 + hidden_layer2_output * self.z2
                
                # 计算误差
                error = norm_target - output
                total_error += error ** 2
                
                # 反向传播计算梯度
                # 输出层是线性的,导数为1
                d_output = error * 1
                
                # 计算z1/z2的梯度
                grad_z1 += d_output * hidden_layer1_output
                grad_z2 += d_output * hidden_layer2_output
                
                # 计算隐藏层的梯度
                d_hidden1 = d_output * self.z1 * hidden_layer1_output * (1 - hidden_layer1_output)
                d_hidden2 = d_output * self.z2 * hidden_layer2_output * (1 - hidden_layer2_output)
                
                # 计算输入层权重的梯度
                grad_x1 += d_hidden1 * norm_num1
                grad_y1 += d_hidden1 * norm_num2
                grad_x2 += d_hidden2 * norm_num1
                grad_y2 += d_hidden2 * norm_num2
            
            # 批量更新权重(平均梯度)
            batch_size = len(self.data.data_dict)
            self.x1 += learning_rate * grad_x1 / batch_size
            self.y1 += learning_rate * grad_y1 / batch_size
            self.x2 += learning_rate * grad_x2 / batch_size
            self.y2 += learning_rate * grad_y2 / batch_size
            self.z1 += learning_rate * grad_z1 / batch_size
            self.z2 += learning_rate * grad_z2 / batch_size
            
            # 每轮迭代打印一次误差,监控训练过程
            if _ % 100 == 0:
                print(f"迭代次数 {_}, 均方误差: {total_error / batch_size}")
        
        # 测试时记得反归一化
        def test_compute(self, num1, num2):
            norm_num1 = num1 / 100.0
            norm_num2 = num2 / 100.0
            hidden_layer1_output = self.sigmoid((norm_num1 * self.x1) + (norm_num2 * self.y1))
            hidden_layer2_output = self.sigmoid((norm_num1 * self.x2) + (norm_num2 * self.y2))
            norm_output = ((hidden_layer1_output * self.z1) + (hidden_layer2_output * self.z2))
            return norm_output * 200.0
   
data = Data(200)
data.create_data()
neural = Neural(data)
neural.back_propagation(0.1)  # 调整学习率到更合适的范围
# print(data.data_array)
# print(uniform(-1,1))
print(neural.test_compute(15,7))

我已尝试调整学习率、迭代次数、数据集规模,但不确定问题出在参数取值还是函数实现本身,希望得到排查帮助。


核心问题与修正说明

  • 输出层梯度错误:原代码对线性输出层错误使用了sigmoid的导数公式output*(1-output),线性输出的导数应为1,这直接导致梯度计算完全错误。
  • 数据未归一化:0-100的输入会让sigmoid函数饱和(输出趋近于1,导数趋近于0),梯度消失后权重无法更新。修正后将输入归一化到0-1,目标值归一化到0-1,训练后再反归一化得到结果。
  • 权重更新方式错误:原代码采用在线更新(每个样本更新一次),波动大且收敛不稳定,改为批量梯度下降(累积所有样本梯度后平均更新),提升收敛稳定性。
  • 权重更新公式错误:原z1/z2的更新公式误用了output而非误差值,修正后使用误差乘以隐藏层输出作为梯度。

内容的提问来源于stack exchange,提问作者Oblisstified

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 16:22:19