MNIST手写数字识别神经网络反向传播训练异常排查求助
基于GitHub项目实现的MNIST手写数字识别神经网络出现反向传播异常,模型仅能成功识别部分数字(如0),其余数字识别完全失败。测试集准确率随训练轮次变化的曲线显示,准确率卡在40%左右且波动幅度极大:

Layer模块代码
import numpy as np def ReLU_calculate(inputs): return np.maximum(0, inputs) def Softmax_calculate(inputs): exp_values = np.exp(inputs - np.max(inputs, axis=1, keepdims=True)) probabilities = exp_values / np.sum(exp_values, axis=1, keepdims=True) return probabilities def ReLU_derivative(input): if input > 0: return 1 return 0 def Softmax_derivative(inputs, index): expSum = 0 for input in inputs: expSum = expSum + np.exp(input) ex = np.exp(inputs[index]) return (ex * expSum - ex * ex) / (expSum * expSum) def cost_derivative(output, y): if output == 0 or output == 1: return 0 else: return (-1 * output + y) / (output * (output - 1)) class Layer: def __init__(self, numNodesIn, numNodes): self.weights = 0.1 * np.random.randn(numNodesIn, numNodes) self.biases = np.zeros((1, numNodes)) self.inputs = None self.weighted_inputs = None self.activations = None self.nodes_in = numNodesIn self.nodes_out = numNodes self.cost_gradientW = None self.cost_gradientB = None def forward(self, inputs): self.inputs = inputs self.weighted_inputs = np.dot(inputs, self.weights) + self.biases def hidden_activation(self): self.activations = ReLU_calculate(self.weighted_inputs) def output_activation(self): self.activations = Softmax_calculate(self.weighted_inputs) def CalculateOutputLayerNodeValues(self, data, expectedOutputs): nodeValues = [] for i in range(data.batch_size): current_nodeValues = [] for x in range(len(expectedOutputs[i])): costDerivative = cost_derivative(self.activations[i][x], expectedOutputs[i][x]) activationDerivative = Softmax_derivative(self.weighted_inputs[i], x) current_nodeValues.append(costDerivative * activationDerivative) nodeValues.append(current_nodeValues) return nodeValues def UpdateGradients(self, data, nodeValues): cost_gradientW = [[0 for x in range(self.nodes_out)] for j in range(self.nodes_in)] cost_gradientB = [0 for x in range(self.nodes_out)] for i in range(data.batch_size): for nodeOut in range(self.nodes_out): nodeValue = nodeValues[i][nodeOut] for nodeIn in range(self.nodes_in): derivativeCostWrtWeight = self.inputs[i][nodeIn] * nodeValue cost_gradientW[nodeIn][nodeOut] += derivativeCostWrtWeight derivativeCostWrtBias = 1 * nodeValues[i][nodeOut] cost_gradientB[nodeOut] += derivativeCostWrtBias self.cost_gradientW = cost_gradientW self.cost_gradientB = cost_gradientB def CalculateHiddenLayerNodeValues(self, data, oldLayer, oldNodeValues): newNodeValues = [] for batch in range(data.batch_size): smallerNodeValues = [] for i in range(self.nodes_out): newNodeValue = 0 for j in range(len(oldNodeValues)): newNodeValue += oldLayer.weights[i][j] * oldNodeValues[batch][j] newNodeValue = newNodeValue * ReLU_derivative(self.weighted_inputs[batch][i]) smallerNodeValues.append(newNodeValue) newNodeValues.append(smallerNodeValues) return newNodeValues def ApplyGradients(self, learnRate): for nodeOut in range(self.nodes_out): for nodeIn in range(self.nodes_in): self.weights[nodeIn][nodeOut] -= self.cost_gradientW[nodeIn][nodeOut] * learnRate self.biases[0][nodeOut] -= self.cost_gradientB[nodeOut] * learnRate self.cost_gradientB = None self.cost_gradientW = None
Network模块代码
import numpy as np import matplotlib.pyplot as plt from Layer import Layer from keras.datasets import mnist class Loss: def calculate(self, output, y): sample_losses = self.forward(output, y) data_loss = np.mean(sample_losses) return data_loss class Loss_CategoricalCrossentropy(Loss): def forward(self, y_pred, y_true): samples = len(y_pred) y_pred_clipped = np.clip(y_pred, 1e-7, 1-1e-7) if len(y_true.shape) == 1: correct_confidences = y_pred_clipped[range(samples), y_true] elif len(y_true.shape) == 2: correct_confidences = np.sum(y_pred_clipped*y_true, axis=1) negative_log_likelihoods = -np.log(correct_confidences) return negative_log_likelihoods class Data: def __init__(self): self.batch_size = 0 self.inputs = [] self.activations = [] self.weighted_inputs = [] self.nodeValues = [] class Network: def __init__(self, size): self.length = len(size) - 1 self.network = [] self.data = Data() for i in range(len(size) - 1): layer = Layer(size[i], size[i + 1]) self.network.append(layer) # assuming that there must be at least 2 layers in the network def forward(self, X): # create new data object to store data for the current pass self.data = Data() self.data.batch_size = len(X) # set first output to the input values passed into forward current_output = X self.data.activations.append(current_output) # Pass forward through the neural network for i in range(self.length - 1): self.network[i].forward(current_output) self.data.weighted_inputs.append(self.network[i].weighted_inputs) self.network[i].hidden_activation() current_output = (self.network[i].activations) self.data.activations.append(current_output) self.network[self.length - 1].forward(current_output) self.data.weighted_inputs.append(self.network[self.length - 1].weighted_inputs) self.network[self.length - 1].output_activation() final_output = self.network[self.length - 1].activations self.data.activations.append(final_output) return final_output def UpdateAllGradients(self, inputs, expectedOutputs, learnRate): self.forward(inputs) outputLayer = self.network[self.length - 1] nodeValues = outputLayer.CalculateOutputLayerNodeValues(self.data, expectedOutputs) outputLayer.UpdateGradients(self.data, nodeValues) for i in reversed(range(self.length - 1)): hiddenLayer = self.network[i] nodeValues = hiddenLayer.CalculateHiddenLayerNodeValues(self.data, self.network[i + 1], nodeValues) hiddenLayer.UpdateGradients(self.data, nodeValues) for layer in network.network: layer.ApplyGradients(learnRate) def test_accuracy(self, inputs, expected): total = 0 size = len(inputs) inputs_flat = [] for x in range(len(inputs)): inputs_flat.append(inputs[x].flatten()) outputs = self.forward(inputs_flat) for sample in range(len(outputs)): index = np.where(outputs[sample] == max(outputs[sample]))[0][0] if index == expected[sample]: total += 1 return total / size (train_X, y_train), (test_X, y_test) = mnist.load_data() print('X_train: ' + str(train_X.shape)) print('Y_train: ' + str(y_train.shape)) print('X_test: ' + str(test_X.shape)) print('Y_test: ' + str(y_test.shape)) x_train = train_X.astype("float32") / 255 x_test = test_X.astype("float32") / 255 network = Network([784, 50, 16, 10]) cost_function = Loss_CategoricalCrossentropy() costs = [] accuracy = [] for i in range(10000): print("Pass: ", i, "\n") inputs = [x_train[i].flatten()] outputs = np.zeros((1, 10)) outputs[0][y_train[i]] = 1 network.UpdateAllGradients(inputs, outputs, 0.1) print(np.array(network.data.activations[-1]), outputs) accuracy.append(network.test_accuracy(x_test, y_test)) costs.append(cost_function.calculate(np.array(network.data.activations[-1]), outputs)) print("\n") plt.plot(accuracy) plt.show()
核心问题排查与修复建议
1. 输出层梯度计算错误
你使用了分类交叉熵损失,但自定义的cost_derivative和Softmax_derivative组合与交叉熵+Softmax的梯度公式不匹配。实际上,这个组合有简化的梯度计算方式,直接返回self.activations - expectedOutputs即可,无需单独计算两个导数相乘,这是导致梯度失效的核心原因。
2. ReLU导数未向量化实现
当前ReLU_derivative仅支持标量输入,处理numpy数组时会返回单个标量,导致隐藏层梯度计算完全错误。修改为向量化实现:
def ReLU_derivative(input): return np.where(input > 0, 1, 0)
3. 梯度未做批量平均
UpdateGradients中直接累加整个批次的梯度,但未除以批次大小,导致梯度值被放大,学习率实际效果远超预期,引发训练震荡。需在梯度赋值时添加平均操作:
self.cost_gradientW = np.array(cost_gradientW) / data.batch_size self.cost_gradientB = np.array(cost_gradientB) / data.batch_size
4. 单样本训练+未打乱数据
训练循环每次仅用一个样本,且按原始顺序训练,导致模型偏向先出现的样本,无法学习全部分类。建议改为小批量训练(如batch_size=32),并在每个epoch前打乱训练数据。
5. 隐藏层节点值权重索引错误
CalculateHiddenLayerNodeValues中,权重索引oldLayer.weights[i][j]顺序错误,权重矩阵维度为(nodes_in, nodes_out),正确索引应为oldLayer.weights[j][i]。
6. 学习率过高
当前学习率0.1对于SGD或小批量训练来说过高,建议降低至0.01或更小,避免训练不稳定。
内容的提问来源于stack exchange,提问作者Jonpaco23

