调试神经网络前向传播:如何传递前层输出解决维度不匹配问题
神经网络前向传播维度不匹配问题修复
问题现象
实现神经网络前向传播时触发维度不匹配错误:
ValueError: shapes (4,4) and (2,1) not aligned: 4 (dim 1) != 2 (dim 0)
核心原因是train方法中所有层都传入原始输入x.T,而非前一层的输出,导致隐藏层矩阵乘法的维度无法对应。
原始代码问题点
错误的train方法
def train(self, x, y): for layer in self.layers: prediction = layer.forward_prop(x.T) # column vector
forward_prop方法(逻辑本身正确,但输入传递错误)
def forward_prop(self, x): if self.is_input: self.out = x else: # bug here: x has to be updated according to previous layer's output! w_sum = np.dot(self.weight, x) + self.bias self.out = self.activation(w_sum) return self.out
期望的前向传播流程
- 第0层(输入层):返回
z^(0) = x - 第1层(第一个隐藏层):返回
z^(1) = sig(W^(1) @ z^(0) + b^(1)) - 第2层(第二个隐藏层):返回
z^(2) = sig(W^(2) @ z^(1) + b^(2)) - 第3层(输出层):返回
z^(3) = sig(W^(3) @ z^(2) + b^(3)) - ...
- 第i层:返回
z^(i) = sig(W^(i) @ z^(i-1) + b^(i))
修复方案
在train方法中维护一个中间变量,逐层传递前一层的输出:
- 初始将输入
x.T传入输入层,得到第一层输出 - 后续每一层都把前一层的输出作为当前层的输入
- 最终输出即为最后一层的计算结果
修正后的完整代码
import numpy as np def sigmoid(x): return 1 / (1 + np.exp(-x)) class NeuralNetwork: def __init__(self): self.layers = [] def add_layer(self, layer): self.layers.append(layer) def create(self): for i, layer in enumerate(self.layers): if i == 0: layer.is_input = True else: layer.init_parameters(self.layers[i - 1].neurons) def summary(self): for i, layer in enumerate(self.layers): print("Layer", i) print("neurons:", layer.neurons) print("is_input:", layer.is_input) print("act:", layer.activation) print("weight:", np.shape(layer.weight)) print(layer.weight) print("bias:", np.shape(layer.bias)) print(layer.bias) print("") def train(self, x, y): # 维护中间输出,逐层传递 current_input = x.T for layer in self.layers: current_input = layer.forward_prop(current_input) prediction = current_input # 最终输出为最后一层的结果 class Layer: def __init__(self, neurons, is_input=False, activation=None): self.out = None self.weight = None self.bias = None self.neurons = neurons self.is_input = is_input self.activation = activation def init_parameters(self, prev_layer_neurons): self.weight = np.asmatrix(np.random.normal(0, 0.5, (self.neurons, prev_layer_neurons))) self.bias = np.asmatrix(np.random.normal(0, 0.5, self.neurons)).T # column vector if self.activation is None: self.activation = sigmoid def forward_prop(self, x): if self.is_input: self.out = x else: w_sum = np.dot(self.weight, x) + self.bias self.out = self.activation(w_sum) return self.out if __name__ == '__main__': net = NeuralNetwork() d = 2 # input dimension c = 1 # output for m in (d, 4, 4, c): layer = Layer(m) net.add_layer(layer) net.create() # net.summary() # dbg # Training set X = np.asmatrix([ [0, 0], [0, 1], [1, 0], [1, 1] ]) y = np.asarray([0, 0, 1, 0]) net.train(X[2], y[2])
内容的提问来源于stack exchange,提问作者tail
相关产品推荐
相关产品推荐

