You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

TensorFlow/Keras自定义损失层报错:无法计算输出张量的解决方法

问题:自定义损失层模型训练触发AssertionError,无法计算输出KerasTensor

问题描述

基于TensorFlow/Keras构建带有自定义损失层的模型,运行1个epoch后出现AssertionError,提示无法计算out_layer的输出KerasTensor。已确认训练和验证数据格式正确,附上报错信息与相关代码,请求解决。

报错信息

AssertionError: Exception encountered when calling layer 'model_21' (type Functional).

Could not compute output KerasTensor(type_spec=TensorSpec(shape=(None, 3, 1), dtype=tf.float32, name=None), name='Placeholder_2:0', description="created by layer 'out_layer'")

Call arguments received by layer 'model_21' (type Functional):
  • inputs=tf.Tensor(shape=(1, 72, 72, 28), dtype=float32)
  • training=False
  • mask=None

相关代码

import tensorflow as tf
from tensorflow.keras import layers, models
from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint
from tensorflow.keras import backend as K

# loss function for angle loss
def angle_loss(y_true, y_pred):

    y_true = tf.squeeze(y_true, axis=-1)  
    y_pred = tf.squeeze(y_pred, axis=-1)

    y_true_normalized = tf.math.l2_normalize(y_true, axis=-1)
    y_pred_normalized = tf.math.l2_normalize(y_pred, axis=-1)

    cosine_similarity = tf.reduce_sum(tf.multiply(y_true_normalized, y_pred_normalized), axis=-1)

    angle_loss = 1 - cosine_similarity
    return angle_loss

# loss function for position loss
def position_loss(n_true, pos_true, y_pred):
    norm_n_true = tf.linalg.norm(n_true, axis=1)
    
    v = pos_true - y_pred
    
    abs_dot_v_n = K.abs(K.sum(tf.multiply(n_true, v), axis=1))
    
    distance = tf.divide(abs_dot_v_n, norm_n_true)
    
    distance_avg = K.mean(distance)
    
    return distance_avg

class CustomLossLayer(layers.Layer):
    def __init__(self, normal_loss, position_loss, **kwargs):
        super(CustomLossLayer, self).__init__(**kwargs)
        self.normal_loss_weight = self.add_weight(name='normal_loss_weight', initializer='ones', trainable=True)
        self.position_loss_weight = self.add_weight(name='position_loss_weight', initializer='ones', trainable=True)
        self.normal_loss = normal_loss
        self.position_loss = position_loss

    def call(self, inputs):
        y_true, y_pred = inputs
        normal_loss = self.normal_loss(y_true[0], y_pred[0])
        position_loss = self.position_loss(y_true[0], y_true[1], y_pred[1])
        loss = self.normal_loss_weight * normal_loss + self.position_loss_weight * position_loss
        self.add_loss(loss, inputs=inputs)
        
        self.add_metric(normal_loss, name='angle_loss')
        self.add_metric(position_loss, name='position_loss')
        
        return y_pred
    
def build_model(input_shape):
    input_layer = layers.Input(shape=input_shape, name="input_1")
    
    true_labels = [
        tf.keras.Input(shape=(3, 1), name="true_labels_pos"),
        tf.keras.Input(shape=(3, 1), name="true_labels_normal"),
    ]
    
    conv1 = layers.Conv3D(32, kernel_size=(3, 3, 3), activation='relu', padding='same')(input_layer)
    maxpool1 = layers.MaxPooling3D(pool_size=(2, 2, 2))(conv1)
    conv2 = layers.Conv3D(64, kernel_size=(3, 3, 3), activation='relu', padding='same')(maxpool1)
    maxpool2 = layers.MaxPooling3D(pool_size=(2, 2, 2))(conv2)
    flatten = layers.Flatten()(maxpool2)
    dense = layers.Dense(128, activation='relu')(flatten)
    
    # Normal head
    normal_output = layers.Dense(3, name='normal_head')(dense)
    normal_output = layers.Reshape((3, 1), name='normal_output')(normal_output) 
    
    # Position head
    position_output = layers.Dense(3, name='position_head')(dense)
    position_output = layers.Reshape((3, 1), name='position_output')(position_output) 
    
    pred_labels = [normal_output, position_output]
    
    out = CustomLossLayer(normal_loss=angle_loss, position_loss=position_loss, name="out_layer")([true_labels, pred_labels])
    
    model = models.Model([input_layer, true_labels], out)
 
    # Compile the model
    model.compile(loss=None, optimizer='adam', weighted_metrics=[])

    return model


# Build and train the model
model = build_model((72, 72, 28, 1))

early_stopping = EarlyStopping(monitor='val_loss', patience=3, verbose=1, restore_best_weights=True)
model_checkpoint = ModelCheckpoint('best_model.h5', monitor='val_loss', save_best_only=True, verbose=1)

# Train the model
history = model.fit([train_scans, train_position, train_normal], 
                    validation_data=([val_scans, val_position, val_normal]),
                    epochs=10, batch_size=1, callbacks=[early_stopping, model_checkpoint])

解决方法

1. 修正自定义损失层的标签与预测对应关系

在CustomLossLayer的call方法中,当前用位置标签计算法线损失,属于对应错误。需调整标签与预测的匹配关系:

def call(self, inputs):
    y_true, y_pred = inputs
    # 修正:法线标签对应法线预测
    normal_loss = self.normal_loss(y_true[1], y_pred[0])
    # 修正:位置损失参数为法线标签、位置标签、位置预测
    position_loss = self.position_loss(y_true[1], y_true[0], y_pred[1])
    loss = self.normal_loss_weight * normal_loss + self.position_loss_weight * position_loss
    self.add_loss(loss, inputs=inputs)
    
    self.add_metric(normal_loss, name='angle_loss')
    self.add_metric(position_loss, name='position_loss')
    
    return y_pred

2. 修复模型fit的validation_data格式

validation_data需要传入(验证输入, 验证标签),由于使用自定义损失层,验证标签可设为None:

history = model.fit(
    x=[train_scans, train_position, train_normal], 
    validation_data=([val_scans, val_position, val_normal], None),
    epochs=10, batch_size=1, callbacks=[early_stopping, model_checkpoint]
)

3. 拆分训练模型与推理模型

当前训练模型需要传入真实标签,推理时无法提供这些标签,需单独构建推理模型:

def build_model(input_shape):
    input_layer = layers.Input(shape=input_shape, name="input_1")
    
    # 共享特征提取网络
    conv1 = layers.Conv3D(32, kernel_size=(3, 3, 3), activation='relu', padding='same')(input_layer)
    maxpool1 = layers.MaxPooling3D(pool_size=(2, 2, 2))(conv1)
    conv2 = layers.Conv3D(64, kernel_size=(3, 3, 3), activation='relu', padding='same')(maxpool1)
    maxpool2 = layers.MaxPooling3D(pool_size=(2, 2, 2))(conv2)
    flatten = layers.Flatten()(maxpool2)
    dense = layers.Dense(128, activation='relu')(flatten)
    
    # 预测头
    normal_output = layers.Dense(3, name='normal_head')(dense)
    normal_output = layers.Reshape((3, 1), name='normal_output')(normal_output) 
    position_output = layers.Dense(3, name='position_head')(dense)
    position_output = layers.Reshape((3, 1), name='position_output')(position_output) 
    pred_labels = [normal_output, position_output]
    
    # 推理模型(仅接收输入数据)
    inference_model = models.Model(input_layer, pred_labels)
    
    # 训练模型(需传入真实标签)
    true_labels = [
        tf.keras.Input(shape=(3, 1), name="true_labels_pos"),
        tf.keras.Input(shape=(3, 1), name="true_labels_normal"),
    ]
    out = CustomLossLayer(normal_loss=angle_loss, position_loss=position_loss, name="out_layer")([true_labels, pred_labels])
    train_model = models.Model([input_layer, true_labels], out)
    train_model.compile(loss=None, optimizer='adam', weighted_metrics=[])
    
    return train_model, inference_model

# 使用示例
train_model, inference_model = build_model((72, 72, 28, 1))

# 训练
history = train_model.fit(
    x=[train_scans, train_position, train_normal], 
    validation_data=([val_scans, val_position, val_normal], None),
    epochs=10, batch_size=1, callbacks=[early_stopping, model_checkpoint]
)

# 推理
predictions = inference_model.predict(val_scans)

4. 修正position_loss的维度计算

输入数据形状为(None,3,1),计算点积时需沿最后一个维度求和,同时避免除以0:

def position_loss(n_true, pos_true, y_pred):
    # 压缩最后一个维度简化计算
    n_true = tf.squeeze(n_true, axis=-1)
    pos_true = tf.squeeze(pos_true, axis=-1)
    y_pred = tf.squeeze(y_pred, axis=-1)
    
    norm_n_true = tf.linalg.norm(n_true, axis=-1)
    
    v = pos_true - y_pred
    
    # 沿最后一个维度计算点积
    abs_dot_v_n = K.abs(K.sum(tf.multiply(n_true, v), axis=-1))
    
    # 加小值避免除以0
    distance = tf.divide(abs_dot_v_n, norm_n_true + 1e-8)
    
    distance_avg = K.mean(distance)
    
    return distance_avg

内容的提问来源于stack exchange,提问作者Juna Santos

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.29 09:09:56