L层深度神经网络训练异常求助:成本下降但预测输出降低、精度无变化
深度神经网络训练异常排查
我正在构建一个包含L层的深度神经网络,训练时发现:训练初期cost有所下降,但predicted output持续降低,且accuracy始终没有变化。已尝试以下方案但问题仍未解决,恳请帮忙排查代码问题:
- 调整不同learning rate
- 检查各元素维度
代码实现
import numpy as np np.random.seed(1) input_0 = np.round(np.random.rand(4,50),2) ground_truth = np.random.randint(low=0, high = 1,size = (1,50)) # Determine the number of training, dev, and test examples num_train = int(0.5 * input_0.shape[1]) num_dev = int(0.2 * input_0.shape[1]) num_test = input_0.shape[1] - num_train - num_dev # Create a random permutation of the indices perm = np.random.permutation(input_0.shape[1]) # Split the input and ground truth data into training, dev, and test sets train_input = input_0[:, perm[:num_train]] train_ground_truth = ground_truth[:, perm[:num_train]] dev_input = input_0[:, perm[num_train:num_train+num_dev]] dev_ground_truth = ground_truth[:, perm[num_train:num_train+num_dev]] test_input = input_0[:, perm[num_train+num_dev:]] test_ground_truth = ground_truth[:, perm[num_train+num_dev:]] num_layers_hid_out = 5 def initializing(num_layers,input_0): neurons = [] weights = {} biases = {} for i in range(1,num_layers+1): #1,2,3,4,.....29,30,31 # appending all number of neurons for each layer if i == num_layers: neurons.append(1) elif i%2 == 0: neurons.append(8) else: neurons.append(5) # initializing weight and bias for each layer if i == 1: weights[f"w_layer{i}"] = np.round(np.random.randn(neurons[i-1],input_0.shape[0]),2) else: weights[f"w_layer{i}"] = np.round(np.random.randn(neurons[i-1],neurons[i-2]),2) biases[f"b_layer{i}"] = np.zeros((neurons[i-1],1)) return weights,biases weights,biases = initializing(num_layers=num_layers_hid_out,input_0=input_0) def sigmoid(z): return 1/(1+np.exp(-z)) def sigmoid_deriv(z): return z*(1-z) def forward_prop(input_data,weights,biases,num_layer): A = input_data forward_cache = {} for layer in range(1,num_layer+1): A = sigmoid(np.dot(weights[f"w_layer{layer}"], A) + biases[f"b_layer{layer}"]) forward_cache[f"A_layer{layer}"] = A return A,forward_cache def compute_avg_cost(predicted_output,ground_truth): m = ground_truth.shape[1] cost = (-1/m)*np.sum((ground_truth*np.log(predicted_output)) + (1-ground_truth)*np.log(1-predicted_output)) return cost def accuracy_compute(predicted,ground_truth): m = ground_truth.shape[1] mask = predicted > 0.5 # Use the boolean mask to convert True values to 1 and False values to 0 binary_array = mask.astype(int) num_correct = np.sum(binary_array == ground_truth) return num_correct*100 / m def backward_prop(input_data,weights,biases,num_layer,forward_cache,ground_truth,learning_rate,predicted_output): dZ = {} dW = {} db = {} m = ground_truth.shape[1] for i in reversed(range(1,num_layer+1)): # 31,30,29,.....3,2,1 if i == num_layer: dZ[f"dZ_layer{i}"] = predicted_output - ground_truth dW[f"dW_layer{i}"] = (1/m)*np.dot(dZ[f"dZ_layer{i}"],forward_cache[f"A_layer{i-1}"].T) db[f"db_layer{i}"] = (1/m)*np.sum(dZ[f"dZ_layer{i}"],axis=1,keepdims=True) elif i == 1: dZ[f"dZ_layer{i}"] = np.dot(weights[f"w_layer{i+1}"].T,dZ[f"dZ_layer{i+1}"])*sigmoid_deriv(forward_cache[f"A_layer{i}"]) dW[f"dW_layer{i}"] = (1/m)*np.dot(dZ[f"dZ_layer{i}"],input_data.T) db[f"db_layer{i}"] = (1/m)*np.sum(dZ[f"dZ_layer{i}"],axis=1,keepdims=True) else: dZ[f"dZ_layer{i}"] = np.dot(weights[f"w_layer{i+1}"].T,dZ[f"dZ_layer{i+1}"])*sigmoid_deriv(forward_cache[f"A_layer{i}"]) dW[f"dW_layer{i}"] = (1/m)*np.dot(dZ[f"dZ_layer{i}"],forward_cache[f"A_layer{i-1}"].T) db[f"db_layer{i}"] = (1/m)*np.sum(dZ[f"dZ_layer{i}"],axis=1,keepdims=True) # for i in range(1,num_layer+1): weights[f"w_layer{i}"] = weights[f"w_layer{i}"] - (learning_rate*dW[f"dW_layer{i}"]) biases[f"b_layer{i}"] = biases[f"b_layer{i}"] - (learning_rate*db[f"db_layer{i}"]) return weights, biases epoch = 100000 y_loss = [] train_loops = [] learning_rate = 0.0002 prev_loss = -1 for i in range(epoch): this_output, forward_cache = forward_prop(input_data = train_input,weights=weights,biases=biases,num_layer=num_layers_hid_out) cost = compute_avg_cost(predicted_output=this_output,ground_truth=train_ground_truth) current_acc = accuracy_compute(predicted=this_output,ground_truth=train_ground_truth) # print(f"Current loss={cost} Current Accuracy={current_acc}") print(this_output) weights, biases = backward_prop(input_data=train_input, weights=weights, biases=biases, num_layer=num_layers_hid_out, forward_cache=forward_cache, ground_truth=train_ground_truth, learning_rate=learning_rate, predicted_output=this_output)
内容的提问来源于stack exchange,提问作者w.chaiwat
相关产品推荐
相关产品推荐

