You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

相机内参估计模型自定义多输入损失函数梯度报错排查

相机内参估计模型自定义损失函数报错排查与修复

问题场景

构建相机内参估计模型时,自定义损失逻辑如下:利用模型输出的相机内参对图像做去畸变处理,将去畸变后的图像与同位置激光雷达生成的深度图对比计算损失。运行时出现梯度形状不匹配的报错。

报错信息

Node: 'gradient_tape/model/pose_loss_layer/undistort_layer/map/while/gradients/model/pose_loss_layer/undistort_layer/map/while/strided_slice_2_grad/StridedSliceGrad'
shape of dy was [] instead of [5]
         [[{{node gradient_tape/model/pose_loss_layer/undistort_layer/map/while/gradients/model/pose_loss_layer/undistort_layer/map/while/strided_slice_2_grad/StridedSliceGrad}}]] [Op:__inference_train_function_2727]

复现代码

import tensorflow as tf
from tensorflow.keras.layers import Layer, Input, Lambda, Concatenate, GlobalAveragePooling2D, Dense, Model
import cv2
import numpy as np

class UndistortLayer(Layer):
    def __init__(self):
        super(UndistortLayer, self).__init__()

    def call(self, inputs):
        camera_images, intrinsic_params, distortion_coeffs = inputs

        def undistort_image(camera_image, intrinsic_params, distortion_coeffs):
            # Create intrinsic matrix
            intrinsic_matrix = create_intrinsic_matrix(intrinsic_params)

            # Undistort image using OpenCV
            undistorted_image = cv2.undistort(
                camera_image.numpy().astype(np.uint8),
                intrinsic_matrix,
                np.array(distortion_coeffs, dtype=np.float32)
            )
            return undistorted_image.astype(np.float32)

        # Apply the undistortion function using tf.py_function for each image
        undistorted_images = tf.map_fn(
            lambda i: tf.py_function(
                undistort_image,
                [camera_images[i], intrinsic_params[i], distortion_coeffs[i]],
                Tout=tf.float32
            ),
            tf.range(tf.shape(camera_images)[0]),  # Loop over the batch
            dtype=tf.float32
        )
        undistorted_images.set_shape([None, 480, 640, 3])
        return undistorted_images


class PoseLossLayer(tf.keras.layers.Layer):
    def call(self, inputs):
        output_intrinsic, camera_input, depth_input = inputs
        
        camera_images = camera_input  # Shape: (batch_size, 480, 640, 3)
        depth_images = depth_input    # Shape: (batch_size, 480, 640)

        # Retrieve intrinsic parameters and distortion coefficients from predictions
        intrinsic_params = output_intrinsic[..., :3]  # fx, cx, cy
        distortion_coeffs = output_intrinsic[..., 3:]  # distortion coefficients

        # Ensure UndistortLayer supports backpropagation
        undistorted_camera_images = UndistortLayer()([camera_images, intrinsic_params, distortion_coeffs])

        # Convert to grayscale correctly
        undistorted_camera_images_gray = tf.image.rgb_to_grayscale(undistorted_camera_images)

        # Calculate SSIM between undistorted camera images and depth images
        ssim_loss = tf.image.ssim(undistorted_camera_images_gray, depth_images, max_val=1.0)

        # Compute the mean SSIM loss to ensure a scalar is returned
        mean_ssim_loss = tf.reduce_mean(ssim_loss)

        # Return a scalar loss value for backpropagation
        return 1 - mean_ssim_loss


def create_intrinsic_matrix(params):
    # 假设params是[fx, cx, cy],生成3x3内参矩阵
    return np.array([
        [params[0], 0, params[1]],
        [0, params[0], params[2]],  # 假设fy=fx,若不是则需要调整
        [0, 0, 1]
    ], dtype=np.float32)


def create_model():
    # Define flexible input layers
    camera_input = Input(shape=(480, 640, 3), name='camera_image')
    depth_input = Input(shape=(480, 640, 1), name='depth_image')

    # Process depth image to have 3 channels
    depth_input_3ch = Lambda(lambda x: tf.image.grayscale_to_rgb(x))(depth_input)

    # Concatenate camera and depth images
    concatenated_features = Concatenate(name='RGB_Depth_concatenate')([camera_input, depth_input_3ch])
    
    # Add Global Average Pooling to reduce spatial dimensions
    pooled_features = GlobalAveragePooling2D()(concatenated_features)

    # Dense layers
    x = Dense(64, activation='relu')(pooled_features)
    x = Dense(32, activation='relu')(x)
    output_intrinsic = Dense(8, activation='linear', name='intrinsic_parameters')(x)

    # PoseLossLayer to compute loss inside the graph
    pose_loss = PoseLossLayer()([output_intrinsic, camera_input, depth_input])

    # Define the model
    model = Model(inputs=[camera_input, depth_input], outputs=output_intrinsic)

    # Add loss to model
    model.add_loss(pose_loss)

    # Compile the model (without loss, as it's added via add_loss)
    model.compile(optimizer='adam')

    return model

# 假设以下是模拟数据
camera_images_array = np.random.rand(10, 480, 640, 3).astype(np.float32) *255
depth_images_array = np.random.rand(10, 480, 640, 1).astype(np.float32)
Parameters_Short = np.random.rand(10,8).astype(np.float32)

model = create_model()
model.fit(
    x=[camera_images_array, depth_images_array],
    y=Parameters_Short, 
    epochs=1,
    batch_size=5,
    # callbacks=[CustomCallback()]
)

问题根源

  1. tf.py_function+OpenCV操作不支持自动微分:cv2.undistort是基于numpy的纯CPU操作,无法被TensorFlow的梯度磁带(GradientTape)跟踪,导致反向传播时梯度信息丢失或形状异常。
  2. tf.map_fn的梯度传播缺陷:在循环处理batch时,外部函数的输出形状没有被TensorFlow正确识别,导致梯度计算时出现dy形状不匹配的错误。

修复方案

替换OpenCV的去畸变操作,改用TensorFlow原生实现的可微分去畸变逻辑,确保所有操作都在TensorFlow计算图内完成,支持自动微分。

修改后的核心代码

纯TensorFlow可微分去畸变层

class UndistortLayer(Layer):
    def __init__(self, image_height=480, image_width=640):
        super(UndistortLayer, self).__init__()
        self.image_height = image_height
        self.image_width = image_width
        
        # 预生成图像网格坐标(归一化前的像素坐标)
        x = tf.range(image_width, dtype=tf.float32)
        y = tf.range(image_height, dtype=tf.float32)
        x_grid, y_grid = tf.meshgrid(x, y)
        # 展平为一维,方便批量处理
        self.x_grid_flat = tf.reshape(x_grid, [-1])
        self.y_grid_flat = tf.reshape(y_grid, [-1])

    def call(self, inputs):
        camera_images, intrinsic_params, distortion_coeffs = inputs
        batch_size = tf.shape(camera_images)[0]
        
        # 将网格坐标扩展到batch维度
        x_grid_batch = tf.tile(tf.expand_dims(self.x_grid_flat, 0), [batch_size, 1])
        y_grid_batch = tf.tile(tf.expand_dims(self.y_grid_flat, 0), [batch_size, 1])
        
        # 提取内参:fx, cx, cy
        fx = intrinsic_params[..., 0]
        cx = intrinsic_params[..., 1]
        cy = intrinsic_params[..., 2]
        
        # 计算归一化坐标 (x - cx)/fx, (y - cy)/fy(假设fy=fx,若需区分可修改)
        x_norm = (x_grid_batch - tf.expand_dims(cx, axis=1)) / tf.expand_dims(fx, axis=1)
        y_norm = (y_grid_batch - tf.expand_dims(cy, axis=1)) / tf.expand_dims(fx, axis=1)
        
        # 提取畸变系数:k1, k2, p1, p2, k3(对应OpenCV的畸变参数顺序)
        k1 = distortion_coeffs[..., 0]
        k2 = distortion_coeffs[..., 1]
        p1 = distortion_coeffs[..., 2]
        p2 = distortion_coeffs[..., 3]
        k3 = distortion_coeffs[..., 4] if tf.shape(distortion_coeffs)[-1] >=5 else tf.zeros_like(k1)
        
        # 计算畸变后的归一化坐标
        r_sq = x_norm **2 + y_norm **2
        x_distorted = x_norm * (1 + k1*r_sq + k2*r_sq**2 + k3*r_sq**3) + 2*p1*x_norm*y_norm + p2*(r_sq + 2*x_norm**2)
        y_distorted = y_norm * (1 + k1*r_sq + k2*r_sq**2 + k3*r_sq**3) + p1*(r_sq + 2*y_norm**2) + 2*p2*x_norm*y_norm
        
        # 转换回像素坐标
        x_undistorted = x_distorted * tf.expand_dims(fx, axis=1) + tf.expand_dims(cx, axis=1)
        y_undistorted = y_distorted * tf.expand_dims(fx, axis=1) + tf.expand_dims(cy, axis=1)
        
        # 整理坐标为 [batch, num_pixels, 2],并限制在图像范围内
        coords = tf.stack([y_undistorted, x_undistorted], axis=-1)
        coords = tf.clip_by_value(coords, 0.0, tf.cast(tf.stack([self.image_height-1, self.image_width-1]), tf.float32))
        
        # 重塑为 [batch, height, width, 2],用于双线性插值采样
        coords_reshaped = tf.reshape(coords, [batch_size, self.image_height, self.image_width, 2])
        # 使用TensorFlow原生插值获取去畸变图像
        undistorted_images = tf.image.interpolate_bilinear(camera_images, coords_reshaped)
        
        return undistorted_images

调整损失层的输入匹配

class PoseLossLayer(tf.keras.layers.Layer):
    def call(self, inputs):
        output_intrinsic, camera_input, depth_input = inputs
        
        camera_images = camera_input  # Shape: (batch_size, 480, 640, 3)
        # 确保深度图是4D张量(匹配灰度图形状)
        depth_images = tf.expand_dims(depth_input, axis=-1) if tf.rank(depth_input) ==3 else depth_input

        # 提取内参和畸变系数
        intrinsic_params = output_intrinsic[..., :3]  # fx, cx, cy
        distortion_coeffs = output_intrinsic[..., 3:]  # 畸变系数
        
        # 使用可微分的去畸变层
        undistorted_camera_images = UndistortLayer()([camera_images, intrinsic_params, distortion_coeffs])

        # 转灰度并归一化到[0,1]
        undistorted_camera_gray = tf.image.rgb_to_grayscale(undistorted_camera_images)
        undistorted_camera_gray = tf.cast(undistorted_camera_gray, tf.float32) / 255.0
        
        # 深度图归一化到[0,1]
        depth_images_norm = tf.cast(depth_images, tf.float32) / tf.reduce_max(depth_images, axis=[1,2,3], keepdims=True)

        # 计算SSIM损失
        ssim_loss = tf.image.ssim(undistorted_camera_gray, depth_images_norm, max_val=1.0)
        mean_ssim_loss = tf.reduce_mean(ssim_loss)
        
        return 1 - mean_ssim_loss

额外注意事项

  1. 固定输入形状:将模型输入从(None, None, 3)改为固定的(480, 640, 3),避免动态形状导致的网格生成问题。
  2. 归一化处理:SSIM计算要求输入在[0, max_val]范围内,需确保图像和深度图都做了归一化。
  3. 监督与自监督结合:如果是有监督任务,可保留y=Parameters_Short并添加MSE损失,与SSIM损失加权结合,提升训练稳定性。

内容的提问来源于stack exchange,提问作者Rob Binnenmars

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.17 14:09:50