Autograd实现中点积调用意外触发__mul__致广播错误
问题:调用dot方法时意外触发__mul__导致维度不匹配错误
我从零实现numpy版Autograd和神经网络时遇到一个奇怪问题——调用dot方法执行矩阵点积,却触发了__mul__元素级乘法,导致维度不匹配报错。
grad.py代码
from typing import Self import numpy as np class Variable: def __init__(self, value: np.ndarray=None): self.value = value if isinstance(value, np.ndarray) else np.asarray(value) self.prev = None def _variablify(self, x) -> Self: if not isinstance(x, Variable): x = Variable(x) return x def __add__(self, x) -> Self: x = self._variablify(x) y = Variable(self.value + x.value) return y def __mul__(self, x) -> Self: x = self._variablify(x) y = Variable(self.value * x.value) return y __radd__ = __add__ __rmul__ = __mul__ def dot(self, x): x = self._variablify(x) y = Variable(self.value.dot(x.value)) return y def __lt__(self, other): return self.value < other def __gt__(self, other): return self.value > other def dot(a: Variable, b: Variable): return a.dot(b)
main.py代码
from typing import Self import numpy as np from grad import Variable import grad class Layer: def __init__(self, neurons: int): self.n_size = neurons self.activation = Variable(0) def previous(self, layer: Self): self.previous_layer = layer self.previous_layer.next_layer = self def next(self, layer: Self): self.next_layer = layer self.next_layer.previous_layer = self def initialise(self): self.weight_matrix = Variable(np.random.normal(0, 0.01, (self.n_size, self.next_layer.n_size))) self.bias_vector = Variable(np.random.normal(0, 0.01, (1, self.next_layer.n_size))) self.next_layer.x = grad.dot(self.activation, self.weight_matrix) + self.bias_vector self.next_layer.activation = np.where(self.next_layer.x > 0, self.next_layer.x, 0.01*self.next_layer.x) # Using LeakyReLU if __name__ == "__main__": input_layer = Layer(5) input_layer.activation = Variable(np.random.randint(1, 5, (1,5))) h1 = Layer(3) h1.previous(input_layer) output = Layer(2) output.previous(h1) input_layer.initialise() h1.initialise() print(input_layer.activation, h1.activation, output.activation)
错误信息
Traceback (most recent call last): File ".../main.py", line 62, in <module> h1.initialise() File ".../main.py", line 40, in initialise self.next_layer.x = grad.dot(self.activation, self.weight_matrix) + self.bias_vector ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File ".../grad.py", line 191, in dot return a.dot(b) ^^^^^^^^ File ".../grad.py", line 49, in __mul__ y = Variable(self.value * x.value) ~~~~~~~~~~~^~~~~~~~~ ValueError: operands could not be broadcast together with shapes (1,3) (3,2)
原因分析
核心问题是**h1.activation被赋值为numpy数组,而非Variable对象**:
- 在
input_layer.initialise()执行时,通过np.where给h1.activation赋值,但np.where无法识别自定义的Variable类,最终返回的是numpy数组,而非Variable实例。 - 当执行
h1.initialise()时,self.activation是numpy数组,调用grad.dot(self.activation, self.weight_matrix)时,第一个参数是numpy数组,而非Variable。 - 此时
grad.dot中调用的是numpy数组的dot方法,numpy在处理Variable类型的第二个参数时,错误触发了Variable的__mul__方法(元素级乘法),而矩阵维度(1,3)和(3,2)无法进行元素级乘法,导致报错。
解决办法
修改Layer.initialise方法中对next_layer.activation的赋值逻辑,先提取Variable的value进行numpy操作,再将结果包装为Variable对象:
def initialise(self): self.weight_matrix = Variable(np.random.normal(0, 0.01, (self.n_size, self.next_layer.n_size))) self.bias_vector = Variable(np.random.normal(0, 0.01, (1, self.next_layer.n_size))) self.next_layer.x = grad.dot(self.activation, self.weight_matrix) + self.bias_vector # 先提取value做LeakyReLU运算,再包装成Variable x_value = self.next_layer.x.value activation_value = np.where(x_value > 0, x_value, 0.01 * x_value) self.next_layer.activation = Variable(activation_value)
这样h1.activation就会是Variable实例,后续调用grad.dot时会正确触发Variable类的dot方法,执行矩阵点积而非元素级乘法。
内容的提问来源于stack exchange,提问作者random_hooman
相关产品推荐
相关产品推荐

