You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

YOLOv8 TFLite C++推理异常:检测框与类别预测问题求助

YOLOv8 TFLite C++推理检测框坐标异常问题

我尝试使用基于COCO数据集训练的YOLOv8模型,通过TFLite的C++ API进行目标检测推理。该场景几乎没有官方文档,参考Ultralytics的示例代码适配后,推理输出完全异常:类别、检测框数量及位置均不正确。

最初检查detections向量时,发现类别预测错误,且检测框数量始终为1。预处理和后处理步骤均参考官方实现适配,TFLite模型已通过Python API正确导出,当时无从下手排查。

经过调试,修复了输出形状解码问题,类别预测有所改善,但NMS仍无法正常工作。经排查,核心问题在于模型输出的x、y、w、h值范围为(0,1),而非预期的(0,图像尺寸)。我已确认以下几点:

  • 预处理步骤与Python实现完全一致(letterbox格式转换、归一化到0-1、通道转置)
  • 输出张量形状符合预期((1, 84, 8400)),转置后得到(8400, 84)的张量格式正确
  • Python端使用该TFLite模型推理结果正常

仅检测框坐标值异常,特此求助排查思路。

所用代码

#include "tensorflow/lite/model.h"
#include "tensorflow/lite/interpreter.h"
#include "tensorflow/lite/kernels/register.h"
#include <iostream>
#include <opencv2/imgproc.hpp>
#include <opencv2/highgui.hpp>
#include <opencv2/opencv.hpp>
#include <random>

#define NUM_CLASSES_COCO 80

// Convert to letterbox format (copied from official implementation)
cv::Mat formatToSquare(const cv::Mat &source)
{
    int col = source.cols;
    int row = source.rows;
    int _max = MAX(col, row);
    cv::Mat result = cv::Mat::zeros(_max, _max, CV_8UC3);
    source.copyTo(result(cv::Rect(0, 0, col, row)));
    return result;
}

// Preprocessing function (adapted from official)
cv::Mat preprocess(cv::Mat &inputImage, int model_input_height, int model_input_width) {

    cv::Mat modelInputImage = formatToSquare(inputImage);
    cv::Mat blob;        

    //model height and width are same so doesn't matter in this case               
    cv::dnn::blobFromImage(modelInputImage, blob, 1.0 / 255.0, cv::Size(model_input_height, model_input_width), cv::Scalar(), true, false);

    return blob;
}

int main()
{
    cv::Mat inputImage = //Assume loaded correctly;
    float modelConfidenceThreshold = 0.25;
    float modelScoreThreshold = 0.45;
    float modelNMSThreshold = 0.5;

    // path to a yolov8n tflite file trained on the coco dataset
    std::unique_ptr<tflite::FlatBufferModel> model = tflite::FlatBufferModel::BuildFromFile(model_path.c_str());

    tflite::ops::builtin::BuiltinOpResolver resolver;
    std::unique_ptr<tflite::Interpreter> interpreter;
    tflite::InterpreterBuilder(*model, resolver)(&interpreter);

     // Allocate tensors
    if (interpreter->AllocateTensors() != kTfLiteOk)
    {
        std::cerr << "Failed to allocate tensors." << std::endl;
        return;
    }

    int input = interpreter->inputs()[0];
    TfLiteIntArray *dims = interpreter->tensor(input)->dims; 

    cv::Mat preprocessed_image_blob = preprocess(inputImage, model_input_width, model_input_height);

    float *input_data = interpreter->typed_tensor<float>(input);
    std::memcpy(input_data, preprocessed_image_blob.data, sizeof(float) * model_input_width * model_input_height * model_input_channels);

    // Run inference
    if (interpreter->Invoke() != kTfLiteOk)
    {
        std::cerr << "Failed to invoke tflite interpreter." << std::endl;
        return;
    }

    TfLiteTensor *output_tensor = interpreter->tensor(interpreter->outputs()[0]); //dims are (1, 84, 8400) as expected

    int rows, dimensions;
    rows = output_shape->data[2];
    dimensions = output_shape->data[1];

    float* data = output_tensor->data.f;

    // This is post processing part which is rather tricky.
    // Following the official implementation to reshape the initial 
    // tensor from (1, 84, 8400) to (8400, 84) and then proceed to 
    // flatten it. 

    cv::Mat temp(dimensions, rows, CV_32F, data); // need this to transpose to shape (8400, 84)

    cv::transpose(temp, temp);

    float* new_data = (float*) temp.data;

    float x_factor = image.cols/model_input_width;
    float y_factor = image.rows/model_input_height;
    
    std::vector<int> class_ids;
    std::vector<float> confidences;
    std::vector<cv::Rect> boxes;

    for (int i = 0; i < rows; i++) {
        float *classes_scores = new_data + 4;

        cv::Mat scores(1, NUM_CLASSES_COCO, CV_32FC1, classes_scores);
        cv::Point class_id;
        double maxClassScore;

        cv::minMaxLoc(scores, 0, &maxClassScore, 0, &class_id);
        if (maxClassScore > modelScoreThreshold) 
        {
            confidences.push_back(maxClassScore);
            class_ids.push_back(class_id.x);

            float x = new_data[0];
            float y = new_data[1];
            float w = new_data[2];
            float h = new_data[3];

            int left = int((x - 0.5 * w) * x_factor);
            int top = int((y - 0.5 * h) * y_factor);

            int width = int(w * x_factor);
            int height = int(h * y_factor);

            boxes.push_back(cv::Rect(left, top, width, height));
        }

        new_data += dimensions;
    }
     
    // Perform NMS over the bounding boxes
    std::vector<int> nms_result;
    cv::dnn::NMSBoxes(boxes, confidences, modelScoreThreshold, modelNMSThreshold, nms_result);

    std::cout << "after nms: " << nms_result.size() << std::endl;

    std::vector<Detection> detections{};
    for (unsigned long i = 0; i < nms_result.size(); ++i)
    {
        int idx = nms_result[i];

        Detection result;
        result.class_id = class_ids[idx];
        result.confidence = confidences[idx];

        std::random_device rd;
        std::mt19937 gen(rd());
        std::uniform_int_distribution<int> dis(100, 255);
        result.color = cv::Scalar(dis(gen),
                                  dis(gen),
                                  dis(gen));

        result.box = boxes[idx];

        detections.push_back(result);
    }
}

内容的提问来源于stack exchange,提问作者Aditya

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.19 09:59:52