You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

转置卷积层的前向与反向传播如何正确实现?

转置卷积层前向与反向传播的正确实现问题

先向认为本文偏数学而非编程的读者致歉。神经网络兼具数学与编程属性,本次问题聚焦于编程实现。我已用C++从零实现了可正常运行的CNN,因此确认自己实现的卷积(convolution)和全卷积(full convolution)函数是正确的。

基础CNN卷积层实现

前向传播

Matrix<float> cnn_forward(Matrix<float> weight, Matrix<float> prev){
    Matrix<float> output = prev.convolute(weight);
    return output;
}

反向传播(无偏置或激活函数)

cnn_back cnn_backward(Matrix<float> a_prev, Matrix<float> dz, Matrix<float> kernel){
    Matrix<float> rotated = kernel.rotate_180();
    Matrix<float> dx = dz.convolute_full(rotated);
    Matrix<float> dw = a_prev.convolute(dz);
    cnn_back output;
    output.dw = std::move(dw);
    output.dx = std::move(dx);
    return output;  
}

转置卷积层的尝试实现

我了解到转置卷积层是卷积层的逆操作,因此尝试按如下方式实现转置卷积层的前向与反向传播,目标是用二维矩阵实现PyTorch中的torch.nn.ConvTranspose2d,并与上述基础卷积公式对应。

前向传播

//forward
Matrix<float> fcn_forward(Matrix<float> weight, Matrix<float> prev){
    Matrix<float> output = prev.convolute_full(weight.rotate_180());
    return output;
} 

反向传播(无偏置或激活函数)

//backward
fcn_back fcn_backward(Matrix<float> a_prev, Matrix<float> dz, Matrix<float> kernel){
    Matrix<float> dx = dz.convolute(kernel);
    Matrix<float> dw = dz.convolute(a_prev);
    fcn_back output;
    output.dw = std::move(dw);
    output.dx = std::move(dx);
    return output;
}

Python+NumPy复现代码

为了调试,我用Python+NumPy复现了上述逻辑:

def convolute(X, W, strides=(1,1)):
    new_row = (int)((X.shape[0] - W.shape[0])/strides[0] +1)
    new_col = (int)((X.shape[1] - W.shape[1])/strides[1] +1)
    out = np.zeros((new_row, new_col), dtype=float)
    x_last = 0
    y_last = 0
    for x in range(0, X.shape[0]-(W.shape[0] - 1), strides[0]):
        for y in range(0, X.shape[1]-(W.shape[1] - 1), strides[1]):
            amt = 0.0
            for i in range(0, W.shape[0]):
                for j in range(0, W.shape[1]):
                    amt += W[i][j] * X[x+i][y+j]
            out[x_last][y_last] = amt
            y_last += 1
        x_last += 1
        y_last = 0
    return out


def convolute_full(X, W, strides=(1, 1)):
    row_num = (X.shape[0] - 1) * strides[0] + W.shape[0]
    col_num = (X.shape[1] - 1) * strides[1] + W.shape[1]
    output = np.zeros([row_num, col_num])
    for i in range(0, X.shape[0]):
        i_prime = i * strides[0] 
        for j in range(0, X.shape[1]):
            j_prime = j * strides[1]
            for k_row in range(W.shape[0]):
                for k_col in range(W.shape[1]):
                    output[i_prime+k_row, j_prime+k_col] += W[k_row, k_col] * X[i, j]
    return output


def get_errors(predicted, label):
    return label - predicted

def fcn_forward(weight, prev):
    rotated = np.rot90(np.rot90(weight)) 
    output = convolute_full(prev, rotated)
    return output

def fcn_backward(a_prev, dz, kernel):
    dx = convolute(dz, kernel)
    dw = convolute(dz, a_prev)
    dx = np.clip(dx, 10, -10)
    return dx, dw


def forward(weights, X_init):
    values = []
    values.append(X_init)
    predicted = fcn_forward(weights[0], X_init)
    values.append(predicted)
    predicted = fcn_forward(weights[1], predicted)
    values.append(predicted)
    return values

def backward(weights, values, label, learningRate=0.001):
    dz = get_errors(values[-1], label)
    dx, dw = fcn_backward(values[-2], dz, weights[-1])
    weights[-1] = weights[-1] - learningRate*dw
    dz = dx
    dx, dw = fcn_backward(values[-3], dz, weights[-2])
    weights[-2] = weights[-2] - learningRate*dw
    return weights


def train_example():
    epoch = int(input("enter epoch: "))
    #creating a random input
    inp = np.random.randn(10,10)
    #creating the weight matricies
    weights = [np.random.randn(3,3), np.random.randn(3,3)]
    #creating the wanted output
    label = np.random.randn(14,14)
    for i in range(0, epoch):
        values = forward(weights, inp)
        if(i == 0 or i == 1):
            errors = get_errors(values[-1], label)
            print("errors:")
            print(errors)
            print("error sum: ", np.sum(errors))
        weights = backward(weights, values, label)
    print("current prediction:")
    print(values[-1])
    print("label: ")
    print(label)
    errors = get_errors(values[-1], label)
    print("errors:")
    print(errors)
    print("error sum at end of training: ", np.sum(errors))

但该实现无法正常工作,权重未得到正确更新,误差持续增大。请问转置卷积层的前向与反向传播的正确实现方式是什么?


修正后的代码

感谢@Bob的解答,以下是修正后的代码:

def convolute(X, W, strides=(1,1)):
    new_row = (int)((X.shape[0] - W.shape[0])/strides[0] +1)
    new_col = (int)((X.shape[1] - W.shape[1])/strides[1] +1)
    out = np.zeros((new_row, new_col), dtype=float)
    x_last = 0
    y_last = 0
    for x in range(0, X.shape[0]-(W.shape[0] - 1), strides[0]):
        for y in range(0, X.shape[1]-(W.shape[1] - 1), strides[1]):
            amt = 0.0
            for i in range(0, W.shape[0]):
                for j in range(0, W.shape[1]):
                    amt += W[i][j] * X[x+i][y+j]
            out[x_last][y_last] = amt
            y_last += 1
        x_last += 1
        y_last = 0
    return out


#this is the same result as scipy.signal.convolute2d
def convolute_full(X, W, strides=(1, 1)):
    row_num = (X.shape[0] - 1) * strides[0] + W.shape[0]
    col_num = (X.shape[1] - 1) * strides[1] + W.shape[1]
    output = np.zeros([row_num, col_num])
    for i in range(0, X.shape[0]):
        i_prime = i * strides[0] 
        for j in range(0, X.shape[1]):
            j_prime = j * strides[1]
            for k_row in range(W.shape[0]):
                for k_col in range(W.shape[1]):
                    output[i_prime+k_row, j_prime+k_col] += W[k_row, k_col] * X[i, j]
    return output

def convolute_full_backward(X, dZ, dW, strides=(1, 1)):
    for i in range(0, X.shape[0]):
        i_prime = i * strides[0] 
        for j in range(0, X.shape[1]):
            j_prime = j * strides[1]
            for k_row in range(dW.shape[0]):
                for k_col in range(dW.shape[1]):
                    dW[k_row, k_col] += dZ[i_prime+k_row, j_prime+k_col] * X[i, j]
    return dW


def get_errors(predicted, label):
    return label - predicted

def fcn_forward(W, X):
    rotated = np.rot90(np.rot90(W))
    output = convolute_full(X, rotated)
    return output

def fcn_backward(X, dZ, kernel):
    dw = np.zeros(kernel.shape)
    dw = convolute_full_backward(X, dZ, dw)
    dw = np.rot90(np.rot90(dw))
    dx = convolute(dZ, np.rot90(np.rot90(kernel)))
    np.clip(dx, 10, -10)
    return dx, dw


def forward(weights, X):
    values = []
    values.append(X)
    predicted = fcn_forward(weights[0], X)
    values.append(predicted)
    predicted = fcn_forward(weights[1], predicted)
    values.append(predicted)
    return values

def backward(weights, values, label, learningRate=0.001):
    dz = get_errors(values[-1], label)
    dx, dw = fcn_backward(values[-2], dz, weights[-1])
    weights[-1] = weights[-1] + learningRate*dw
    dz = dx
    dx, dw = fcn_backward(values[-3], dz, weights[-2])
    #new apply dw:
    weights[-2] = weights[-2] + learningRate*dw
    return weights
    

def train_example():
    epoch = int(input("please enter epoch: "))
    inp = np.random.randn(10,10)
    weights = [np.random.randn(3,3), np.random.randn(3,3)]
    label = np.random.randn(14,14)
    for i in range(0, epoch):
        values = forward(weights, inp)
        errors = get_errors(values[-1], label)
        print("error sum at {} is: {}".format(i, np.sum(errors)))
        weights = backward(weights, values, label)

    errors = get_errors(values[-1], label)
    print("error sum at end of training: ", np.sum(errors))

内容的提问来源于stack exchange,提问作者Sam Moldenha

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.23 07:45:29