You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

L层深度神经网络训练异常求助:成本下降但预测输出降低、精度无变化

深度神经网络训练异常排查

我正在构建一个包含L层的深度神经网络,训练时发现:训练初期cost有所下降,但predicted output持续降低,且accuracy始终没有变化。已尝试以下方案但问题仍未解决,恳请帮忙排查代码问题:

  • 调整不同learning rate
  • 检查各元素维度

代码实现

import numpy as np
np.random.seed(1)

input_0 = np.round(np.random.rand(4,50),2)
ground_truth = np.random.randint(low=0, high = 1,size = (1,50))

# Determine the number of training, dev, and test examples
num_train = int(0.5 * input_0.shape[1])
num_dev = int(0.2 * input_0.shape[1])
num_test = input_0.shape[1] - num_train - num_dev


# Create a random permutation of the indices
perm = np.random.permutation(input_0.shape[1])

# Split the input and ground truth data into training, dev, and test sets
train_input = input_0[:, perm[:num_train]]
train_ground_truth = ground_truth[:, perm[:num_train]]

dev_input = input_0[:, perm[num_train:num_train+num_dev]]
dev_ground_truth = ground_truth[:, perm[num_train:num_train+num_dev]]

test_input = input_0[:, perm[num_train+num_dev:]]
test_ground_truth = ground_truth[:, perm[num_train+num_dev:]]

num_layers_hid_out = 5

def initializing(num_layers,input_0):
    neurons = []
    weights = {}
    biases = {}
    for i in range(1,num_layers+1): #1,2,3,4,.....29,30,31
        # appending all number of neurons for each layer
        if i == num_layers:
            neurons.append(1)
        elif i%2 == 0:
            neurons.append(8)
        else:
            neurons.append(5)

        # initializing weight and bias for each layer
        if i == 1:
            weights[f"w_layer{i}"] = np.round(np.random.randn(neurons[i-1],input_0.shape[0]),2)
        else:
            weights[f"w_layer{i}"] = np.round(np.random.randn(neurons[i-1],neurons[i-2]),2)

        biases[f"b_layer{i}"] = np.zeros((neurons[i-1],1))
    return weights,biases

weights,biases = initializing(num_layers=num_layers_hid_out,input_0=input_0)

def sigmoid(z):
    return 1/(1+np.exp(-z))
def sigmoid_deriv(z):
    return z*(1-z)


def forward_prop(input_data,weights,biases,num_layer):
    A = input_data
    forward_cache = {}
    for layer in range(1,num_layer+1):
        A = sigmoid(np.dot(weights[f"w_layer{layer}"], A) + biases[f"b_layer{layer}"])
        forward_cache[f"A_layer{layer}"] = A
    return A,forward_cache


def compute_avg_cost(predicted_output,ground_truth):
    m = ground_truth.shape[1]

    cost = (-1/m)*np.sum((ground_truth*np.log(predicted_output)) + (1-ground_truth)*np.log(1-predicted_output))
    
    return cost

def accuracy_compute(predicted,ground_truth):
    m = ground_truth.shape[1]

    mask = predicted > 0.5
    # Use the boolean mask to convert True values to 1 and False values to 0
    binary_array = mask.astype(int)
    

    num_correct = np.sum(binary_array == ground_truth)
    return num_correct*100 / m

def backward_prop(input_data,weights,biases,num_layer,forward_cache,ground_truth,learning_rate,predicted_output):
    dZ = {}
    dW = {}
    db = {}
    m = ground_truth.shape[1]
    
    for i in reversed(range(1,num_layer+1)): #  31,30,29,.....3,2,1
        
        if i == num_layer:
            dZ[f"dZ_layer{i}"] = predicted_output - ground_truth
            dW[f"dW_layer{i}"] = (1/m)*np.dot(dZ[f"dZ_layer{i}"],forward_cache[f"A_layer{i-1}"].T)
            db[f"db_layer{i}"] = (1/m)*np.sum(dZ[f"dZ_layer{i}"],axis=1,keepdims=True)
        
        elif i == 1:
            dZ[f"dZ_layer{i}"] = np.dot(weights[f"w_layer{i+1}"].T,dZ[f"dZ_layer{i+1}"])*sigmoid_deriv(forward_cache[f"A_layer{i}"])            
            dW[f"dW_layer{i}"] = (1/m)*np.dot(dZ[f"dZ_layer{i}"],input_data.T)
            db[f"db_layer{i}"] = (1/m)*np.sum(dZ[f"dZ_layer{i}"],axis=1,keepdims=True)
        
        else:
            dZ[f"dZ_layer{i}"] = np.dot(weights[f"w_layer{i+1}"].T,dZ[f"dZ_layer{i+1}"])*sigmoid_deriv(forward_cache[f"A_layer{i}"])            
            dW[f"dW_layer{i}"] = (1/m)*np.dot(dZ[f"dZ_layer{i}"],forward_cache[f"A_layer{i-1}"].T)
            db[f"db_layer{i}"] = (1/m)*np.sum(dZ[f"dZ_layer{i}"],axis=1,keepdims=True)

    # for i in range(1,num_layer+1):      
        weights[f"w_layer{i}"] = weights[f"w_layer{i}"] - (learning_rate*dW[f"dW_layer{i}"])
        biases[f"b_layer{i}"] = biases[f"b_layer{i}"] - (learning_rate*db[f"db_layer{i}"])
    return weights, biases

epoch = 100000
y_loss = []
train_loops = []
learning_rate = 0.0002
prev_loss = -1
for i in range(epoch):
    this_output, forward_cache = forward_prop(input_data = train_input,weights=weights,biases=biases,num_layer=num_layers_hid_out)
    cost = compute_avg_cost(predicted_output=this_output,ground_truth=train_ground_truth)
    current_acc = accuracy_compute(predicted=this_output,ground_truth=train_ground_truth)
    # print(f"Current loss={cost} Current Accuracy={current_acc}")
    print(this_output)
    weights, biases = backward_prop(input_data=train_input,
                                    weights=weights,
                                    biases=biases,
                                    num_layer=num_layers_hid_out,
                                    forward_cache=forward_cache,
                                    ground_truth=train_ground_truth,
                                    learning_rate=learning_rate,
                                    predicted_output=this_output)

内容的提问来源于stack exchange,提问作者w.chaiwat

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.25 13:09:52