You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++与Python运行TensorFlow分割模型结果不一致问题排查

TensorFlow Lite C++与Python在Raspberry Pi 4(AARCH64)上输出结果不一致问题

在Raspberry Pi 4(AARCH64)设备上,使用相同的MobileNetv3-Cityscapes模型和输入图像,分别通过C++和Python版本的TensorFlow Lite运行,输出结果存在明显小数差异。已确认输入数据仅因打印精度导致末尾数字略有不同,实际调试值完全一致,但输出差异示例如下:

C++Python
-1.85958-1.9500647
-4.14567-4.2847333
8.413678.452963

两边代码均使用Float32类型,且已通过C++调试确认输出值真实存在差异,求分析原因。


Python实现代码

import cv2
import numpy as np
import tensorflow as tf

if __name__ == '__main__':
    image = cv2.imread('../input_videos/photo2.jpeg')
    model_path = '../models/lite-model_deeplabv3-mobilenetv3-cityscapes_1_default_2.tflite'

    # preprocess data
    frame = image
    input_size = (2049, 1025)
    resized_frame = cv2.resize(frame, input_size)
    resized_frame = cv2.cvtColor(resized_frame, cv2.COLOR_BGR2RGB)
    frame_for_prediction = np.asarray(resized_frame).astype(np.float32)
    frame_for_prediction = np.expand_dims(frame_for_prediction, 0)
    frame_for_prediction = frame_for_prediction / 127.5 - 1
    print(frame_for_prediction.shape)
    print(f"input_data: {frame_for_prediction}")

    # load model
    interpreter = tf.lite.Interpreter(model_path=model_path)
    input_details = interpreter.get_input_details()
    interpreter.allocate_tensors()
    interpreter.set_tensor(input_details[0]['index'], frame_for_prediction)
    interpreter.invoke()
    raw_prediction = interpreter.tensor(interpreter.get_output_details()[0]['index'])()
    print(raw_prediction.shape)
    print(f"raw_prediction: {raw_prediction[0, 0, 0, :]}")

C++实现代码

#include <iostream>
#include "opencv2/opencv.hpp"
#include "tensorflow/lite/interpreter.h"
#include "tensorflow/lite/model_builder.h"
#include "tensorflow/lite/interpreter_builder.h"
#include "tensorflow/lite/core/shims/cc/kernels/register.h"

using namespace std;
using namespace tflite;

template<typename ElemType>
inline void printVecElems(const std::vector<ElemType> &mat, size_t elemCnt, const char *matName)
{
    size_t cnt{};
    bool brk = false;
    std::cout << matName << ":" << std::endl;
    std::cout << "[ ";
    for (uint32_t row = 0; row < mat.size(); ++row)
    {
        std::cout << +mat.at(row);

        if (row + 1 < mat.size())
            std::cout << ", ";

        ++cnt;
        if (cnt > elemCnt)
        {
            brk = true;
            break;
        }
    }
    if (brk)
        std::cout << "..." << std::endl;
    else
        std::cout << " ]" << std::endl;
}

static int runTfTest()
{
    cv::Mat image = cv::imread("~/input_videos/photo2.jpeg");
    if(image.empty())
    {
        cout << "Can not read input file" << endl;
        return -1;
    }

    cv::Mat dst;
    cv::resize(image, dst, cv::Size2i(2049, 1025), 0, 0, cv::INTER_LINEAR);
    if(dst.empty())
        return -1;
    cv::cvtColor(dst, dst, cv::COLOR_BGR2RGB);
    cv::Mat flat = dst.reshape(1, dst.total()*dst.channels());
    std::vector<float> frameForPrediction = dst.isContinuous()? flat : flat.clone();
    for(auto& pt : frameForPrediction)
        pt = (pt / 127.5) - 1.0;
    printVecElems<float>(frameForPrediction, 100, "input_data");

    // Create model from file. Note that the model instance must outlive the
    // interpreter instance.
    auto model = tflite::FlatBufferModel::BuildFromFile("~/models/lite-model_deeplabv3-mobilenetv3-cityscapes_1_default_2.tflite");
    if (model == nullptr)
        return -1;

    // Create an Interpreter with an InterpreterBuilder.
    std::unique_ptr<Interpreter> interpreter;
    tflite::ops::builtin::BuiltinOpResolver resolver;
    if (InterpreterBuilder(*model, resolver)(&interpreter) != kTfLiteOk)
        return -1;

    if (interpreter->AllocateTensors() != kTfLiteOk)
        return -1;

    auto inputData = interpreter->typed_input_tensor<float>(0);
    for(uint32_t i=0; i<frameForPrediction.size(); ++i)
    {
        inputData[i] = frameForPrediction[i];     // TODO: Pass input data to input tensor without copy
    }

    if(interpreter->Invoke() != kTfLiteOk)
        return -1;

    auto outputData = interpreter->typed_output_tensor<float>(0);
    const std::vector<int>& outputs = interpreter.get()->outputs();
    TfLiteTensor* outDetails = interpreter.get()->tensor(outputs[0]);
    std::vector<float> rawPrediction;
    for(uint32_t i=0; i< outDetails->bytes / sizeof(float); ++i)
    {
        rawPrediction.push_back(outputData[i]);     // TODO: Pass output tensor to output data without copy
    }

    printVecElems<float>(rawPrediction, 100, "raw_prediction");

    return 0;
}

int main()
{
    int ret {};

    ret = runTfTest();
    if(ret != 0)
        cout << "Run failed." << endl;

    return ret;
}

可能的原因分析

  • TFLite版本差异:Python和C++依赖的TensorFlow Lite库版本可能不一致,不同版本的算子实现、ARM NEON优化策略存在差异,导致计算结果偏差。
  • 数据拷贝与内存对齐问题:C++中cv::Mat转std::vector<float>时,若内存不连续可能导致数据拷贝错误,可尝试用flat.copyTo(frameForPrediction)替代当前赋值逻辑,确保数据完全一致。
  • 算子注册与优化配置差异:C使用的BuiltinOpResolver可能未启用部分优化算子,或Python版本默认开启了NNAPI/GPU加速,而C未配置。可在C++中显式关闭NNAPI,或检查算子注册是否完整。
  • 预处理细节差异:OpenCV在Python和C++中的resize插值精度、颜色转换后的数值存储可能存在细微差异,可逐像素对比预处理后的输入数据,确认完全一致。
  • Interpreter配置差异:Python的TFLite Interpreter默认线程数、内存分配器类型可能与C不同,可在C中设置相同线程数(如interpreter->SetNumThreads(4)),或统一内存分配策略。

内容的提问来源于stack exchange,提问作者killdaclick

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 22:20:26