You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

MNIST手写数字识别神经网络反向传播训练异常排查求助

神经网络反向传播故障排查:MNIST识别准确率异常问题

基于GitHub项目实现的MNIST手写数字识别神经网络出现反向传播异常,模型仅能成功识别部分数字(如0),其余数字识别完全失败。测试集准确率随训练轮次变化的曲线显示,准确率卡在40%左右且波动幅度极大:

测试集准确率随训练轮次变化曲线

Layer模块代码

import numpy as np

def ReLU_calculate(inputs):
    return np.maximum(0, inputs)


def Softmax_calculate(inputs):
    exp_values = np.exp(inputs - np.max(inputs, axis=1, keepdims=True))
    probabilities = exp_values / np.sum(exp_values, axis=1, keepdims=True)
    return probabilities

def ReLU_derivative(input):
    if input > 0:
        return 1
    return 0


def Softmax_derivative(inputs, index):
    expSum = 0
    for input in inputs:
        expSum = expSum + np.exp(input)

    ex = np.exp(inputs[index])

    return (ex * expSum - ex * ex) / (expSum * expSum)

def cost_derivative(output, y):
    if output == 0 or output == 1:
        return 0
    else:
        return (-1 * output + y) / (output * (output - 1))

class Layer:
    def __init__(self, numNodesIn, numNodes):
        self.weights = 0.1 * np.random.randn(numNodesIn, numNodes)
        self.biases = np.zeros((1, numNodes))
        self.inputs = None
        self.weighted_inputs = None
        self.activations = None
        self.nodes_in = numNodesIn
        self.nodes_out = numNodes
        self.cost_gradientW = None
        self.cost_gradientB = None

    def forward(self, inputs):
        self.inputs = inputs
        self.weighted_inputs = np.dot(inputs, self.weights) + self.biases

    def hidden_activation(self):
        self.activations = ReLU_calculate(self.weighted_inputs)

    def output_activation(self):
        self.activations = Softmax_calculate(self.weighted_inputs)
    def CalculateOutputLayerNodeValues(self, data, expectedOutputs):
        nodeValues = []
        for i in range(data.batch_size):
            current_nodeValues = []
            for x in range(len(expectedOutputs[i])):
                costDerivative = cost_derivative(self.activations[i][x], expectedOutputs[i][x])
                activationDerivative = Softmax_derivative(self.weighted_inputs[i], x)
                current_nodeValues.append(costDerivative * activationDerivative)
            nodeValues.append(current_nodeValues)
        return nodeValues

    def UpdateGradients(self, data, nodeValues):
        cost_gradientW = [[0 for x in range(self.nodes_out)] for j in range(self.nodes_in)]
        cost_gradientB = [0 for x in range(self.nodes_out)]
        for i in range(data.batch_size):
            for nodeOut in range(self.nodes_out):
                nodeValue = nodeValues[i][nodeOut]
                for nodeIn in range(self.nodes_in):
                    derivativeCostWrtWeight = self.inputs[i][nodeIn] * nodeValue
                    cost_gradientW[nodeIn][nodeOut] += derivativeCostWrtWeight

                derivativeCostWrtBias = 1 * nodeValues[i][nodeOut]
                cost_gradientB[nodeOut] += derivativeCostWrtBias

        self.cost_gradientW = cost_gradientW
        self.cost_gradientB = cost_gradientB

    def CalculateHiddenLayerNodeValues(self, data, oldLayer, oldNodeValues):
        newNodeValues = []
        for batch in range(data.batch_size):
            smallerNodeValues = []
            for i in range(self.nodes_out):
                newNodeValue = 0
                for j in range(len(oldNodeValues)):
                    newNodeValue += oldLayer.weights[i][j] * oldNodeValues[batch][j]

                newNodeValue = newNodeValue * ReLU_derivative(self.weighted_inputs[batch][i])
                smallerNodeValues.append(newNodeValue)
            newNodeValues.append(smallerNodeValues)

        return newNodeValues

    def ApplyGradients(self, learnRate):
        for nodeOut in range(self.nodes_out):
            for nodeIn in range(self.nodes_in):
                self.weights[nodeIn][nodeOut] -= self.cost_gradientW[nodeIn][nodeOut] * learnRate

            self.biases[0][nodeOut] -= self.cost_gradientB[nodeOut] * learnRate

        self.cost_gradientB = None
        self.cost_gradientW = None

Network模块代码

import numpy as np
import matplotlib.pyplot as plt

from Layer import Layer
from keras.datasets import mnist


class Loss:
    def calculate(self, output, y):
        sample_losses = self.forward(output, y)
        data_loss = np.mean(sample_losses)
        return data_loss

class Loss_CategoricalCrossentropy(Loss):
    def forward(self, y_pred, y_true):
        samples = len(y_pred)
        y_pred_clipped = np.clip(y_pred, 1e-7, 1-1e-7)

        if len(y_true.shape) == 1:
            correct_confidences = y_pred_clipped[range(samples), y_true]

        elif len(y_true.shape) == 2:
            correct_confidences = np.sum(y_pred_clipped*y_true, axis=1)

        negative_log_likelihoods = -np.log(correct_confidences)
        return negative_log_likelihoods

class Data:
    def __init__(self):
        self.batch_size = 0
        self.inputs = []
        self.activations = []
        self.weighted_inputs = []
        self.nodeValues = []


class Network:
    def __init__(self, size):
        self.length = len(size) - 1
        self.network = []
        self.data = Data()
        for i in range(len(size) - 1):
            layer = Layer(size[i], size[i + 1])
            self.network.append(layer)

    # assuming that there must be at least 2 layers in the network
    def forward(self, X):
        # create new data object to store data for the current pass
        self.data = Data()
        self.data.batch_size = len(X)
        # set first output to the input values passed into forward
        current_output = X
        self.data.activations.append(current_output)
        # Pass forward through the neural network
        for i in range(self.length - 1):
            self.network[i].forward(current_output)
            self.data.weighted_inputs.append(self.network[i].weighted_inputs)
            self.network[i].hidden_activation()
            current_output = (self.network[i].activations)
            self.data.activations.append(current_output)

        self.network[self.length - 1].forward(current_output)
        self.data.weighted_inputs.append(self.network[self.length - 1].weighted_inputs)
        self.network[self.length - 1].output_activation()
        final_output = self.network[self.length - 1].activations
        self.data.activations.append(final_output)
        return final_output

    def UpdateAllGradients(self, inputs, expectedOutputs, learnRate):
        self.forward(inputs)
        outputLayer = self.network[self.length - 1]
        nodeValues = outputLayer.CalculateOutputLayerNodeValues(self.data, expectedOutputs)
        outputLayer.UpdateGradients(self.data, nodeValues)

        for i in reversed(range(self.length - 1)):
            hiddenLayer = self.network[i]
            nodeValues = hiddenLayer.CalculateHiddenLayerNodeValues(self.data, self.network[i + 1], nodeValues)
            hiddenLayer.UpdateGradients(self.data, nodeValues)

        for layer in network.network:
            layer.ApplyGradients(learnRate)

    def test_accuracy(self, inputs, expected):
        total = 0
        size = len(inputs)
        inputs_flat = []
        for x in range(len(inputs)):
            inputs_flat.append(inputs[x].flatten())
        outputs = self.forward(inputs_flat)
        for sample in range(len(outputs)):
            index = np.where(outputs[sample] == max(outputs[sample]))[0][0]
            if index == expected[sample]:
                total += 1
        return total / size

(train_X, y_train), (test_X, y_test) = mnist.load_data()

print('X_train: ' + str(train_X.shape))
print('Y_train: ' + str(y_train.shape))
print('X_test:  ' + str(test_X.shape))
print('Y_test:  ' + str(y_test.shape))

x_train = train_X.astype("float32") / 255
x_test = test_X.astype("float32") / 255

network = Network([784, 50, 16, 10])

cost_function = Loss_CategoricalCrossentropy()
costs = []
accuracy = []
for i in range(10000):
    print("Pass: ", i, "\n")
    inputs = [x_train[i].flatten()]
    outputs = np.zeros((1, 10))
    outputs[0][y_train[i]] = 1
    network.UpdateAllGradients(inputs, outputs, 0.1)
    print(np.array(network.data.activations[-1]), outputs)
    accuracy.append(network.test_accuracy(x_test, y_test))
    costs.append(cost_function.calculate(np.array(network.data.activations[-1]),
                                outputs))
    print("\n")

plt.plot(accuracy)
plt.show()

核心问题排查与修复建议

1. 输出层梯度计算错误

你使用了分类交叉熵损失,但自定义的cost_derivative和Softmax_derivative组合与交叉熵+Softmax的梯度公式不匹配。实际上,这个组合有简化的梯度计算方式,直接返回self.activations - expectedOutputs即可,无需单独计算两个导数相乘,这是导致梯度失效的核心原因。

2. ReLU导数未向量化实现

当前ReLU_derivative仅支持标量输入,处理numpy数组时会返回单个标量,导致隐藏层梯度计算完全错误。修改为向量化实现:

def ReLU_derivative(input):
    return np.where(input > 0, 1, 0)

3. 梯度未做批量平均

UpdateGradients中直接累加整个批次的梯度,但未除以批次大小,导致梯度值被放大,学习率实际效果远超预期,引发训练震荡。需在梯度赋值时添加平均操作:

self.cost_gradientW = np.array(cost_gradientW) / data.batch_size
self.cost_gradientB = np.array(cost_gradientB) / data.batch_size

4. 单样本训练+未打乱数据

训练循环每次仅用一个样本,且按原始顺序训练,导致模型偏向先出现的样本,无法学习全部分类。建议改为小批量训练(如batch_size=32),并在每个epoch前打乱训练数据。

5. 隐藏层节点值权重索引错误

CalculateHiddenLayerNodeValues中,权重索引oldLayer.weights[i][j]顺序错误,权重矩阵维度为(nodes_in, nodes_out),正确索引应为oldLayer.weights[j][i]。

6. 学习率过高

当前学习率0.1对于SGD或小批量训练来说过高,建议降低至0.01或更小,避免训练不稳定。

内容的提问来源于stack exchange,提问作者Jonpaco23

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.29 23:10:54