基于Python Keras的多输入矩阵预测输出矩阵及形状报错解决
错误原因
- 形状不匹配:当前模型输出维度为(10,16),但标签维度为(10,2,2)(展开后为每个样本4个值),二者维度完全不对应。
- 损失函数选择错误:你使用的
SparseCategoricalCrossentropy适用于单分类任务(每个样本对应1个类别标签),但你的需求是每个样本输出2x2共4个0/1值,属于逐元素二分类任务,损失函数不匹配。 - 卷积参数不合理:输入是2x2的小尺寸矩阵,第二层卷积使用3x3的卷积核且无padding,会导致卷积输出尺寸为0,特征提取完全失效。
- 输出层设置错误:最后一层输出16个神经元用softmax激活,不符合输出4个二分类值的需求。
修改方案
1. 调整模型结构
- 卷积层增加
padding='same'保证输出尺寸和输入一致,同时缩小卷积核尺寸适配小输入 - 最后一层全连接层设置为4个神经元,激活函数改为
sigmoid适配二分类输出 - 输出最后reshape为(2,2)和标签形状对齐
2. 调整损失函数
替换为BinaryCrossentropy适配逐元素二分类任务
3. 调整准确率计算逻辑
改为逐元素对比预测值和标签的匹配度
修改后的完整代码
import warnings import sys if not sys.warnoptions: warnings.simplefilter("ignore") import numpy as np import tensorflow as tf from tensorflow import keras from IPython.display import clear_output class model(keras.Model): def __init__(self): super().__init__() # 调整卷积层,增加padding保证小尺寸输入有有效输出 self.Conv2D_1 = tf.keras.layers.Conv2D(filters=32, kernel_size=(1, 1), strides=(1, 1), padding='same' ) self.Conv2D_2 = tf.keras.layers.Conv2D(filters=32, kernel_size=(2, 2), strides=(1, 1), padding='same' ) # 调整输出层,输出4个值对应2x2矩阵的每个元素 self.Combined_dense_1 = tf.keras.layers.Dense(units=32, activation="relu") self.Combined_dense_2 = tf.keras.layers.Dense(units=4, activation="sigmoid") def call(self, input_image_one, input_image_two): # 处理第一个输入矩阵 I = self.Conv2D_1(input_image_one) I = self.Conv2D_2(I) I = tf.keras.layers.Flatten()(I) # 处理第二个输入矩阵 N = self.Conv2D_1(input_image_two) N = self.Conv2D_2(N) N = tf.keras.layers.Flatten()(N) # 特征融合 x = tf.concat([N, I], 1) x = self.Combined_dense_1(x) x = self.Combined_dense_2(x) # 输出reshape为2x2和标签形状对齐 x = tf.reshape(x, (-1, 2, 2)) return x network = model() optimizer = tf.keras.optimizers.Adam() # 替换为二分类交叉熵损失 loss_function = tf.keras.losses.BinaryCrossentropy() def train_step(model, optimizer, loss_function, images_one_batch, images_two_batch, labels): with tf.GradientTape() as tape: model_output = model(images_one_batch, images_two_batch) loss = loss_function(labels, model_output) grads = tape.gradient(loss, model.trainable_variables) optimizer.apply_gradients(zip(grads, model.trainable_variables)) return loss def train(model, optimizer, loss_function, epochs, images_one_batch, images_two_batch, labels): loss_array = [] for epoch in range(epochs): loss = train_step(model, optimizer, loss_function, images_one_batch, images_two_batch, labels) loss_array.append(loss) if ((epoch + 1) % 20 == 0): # 调整准确率计算逻辑,逐元素对比 network_output = network(images_one_batch, images_two_batch) preds = np.round(network_output) acc = np.mean(preds == labels) * 100 print(" loss:", loss.numpy(), " Accuracy: ", acc, "%") clear_output(wait=True) NumberofVars = 2; width= NumberofVars; height = NumberofVars NumberOfComputationSets = 10 CM_MatrixArr1 = [] CM_MatrixArr2 = [] for j in range(NumberOfComputationSets): Theta1 = list(np.reshape(np.random.randint(2, size=4), (1,4))[0]) Theta1 = list(np.float_(Theta1)) CM_MatrixArr1.append(Theta1) Theta2 = list(np.reshape(np.random.randint(2, size=4), (1,4))[0]) Theta2 = list(np.float_(Theta2)) CM_MatrixArr2.append(Theta2) combinedCM_MatrixArr = [] for x,y in zip(CM_MatrixArr1, CM_MatrixArr2): combinedCM = [] for a,b in zip(x,y): LogVal = (a == b) combinedCM.append(float(LogVal == True)) combinedCM_MatrixArr.append(combinedCM) combinedCM_MatrixArr = np.array(combinedCM_MatrixArr) combinedCM_MatrixArr = combinedCM_MatrixArr.reshape(NumberOfComputationSets,2,2) CM_MatrixArr1 = np.array(CM_MatrixArr1) CM_MatrixArr1 = CM_MatrixArr1.reshape(NumberOfComputationSets,2,2) CM_MatrixArr1 = CM_MatrixArr1.reshape(NumberOfComputationSets, 2,2,1) CM_MatrixArr2 = np.array(CM_MatrixArr2) CM_MatrixArr2 = CM_MatrixArr2.reshape(NumberOfComputationSets,2,2) CM_MatrixArr2 = CM_MatrixArr2.reshape(NumberOfComputationSets, 2,2,1) train(network,optimizer,loss_function,300,CM_MatrixArr1,CM_MatrixArr2,combinedCM_MatrixArr)
内容的提问来源于stack exchange,提问作者Brian Droncheff
相关产品推荐
相关产品推荐

