YOLOv8 TFLite C++推理异常:检测框与类别预测问题求助
YOLOv8 TFLite C++推理检测框坐标异常问题
我尝试使用基于COCO数据集训练的YOLOv8模型,通过TFLite的C++ API进行目标检测推理。该场景几乎没有官方文档,参考Ultralytics的示例代码适配后,推理输出完全异常:类别、检测框数量及位置均不正确。
最初检查detections向量时,发现类别预测错误,且检测框数量始终为1。预处理和后处理步骤均参考官方实现适配,TFLite模型已通过Python API正确导出,当时无从下手排查。
经过调试,修复了输出形状解码问题,类别预测有所改善,但NMS仍无法正常工作。经排查,核心问题在于模型输出的x、y、w、h值范围为(0,1),而非预期的(0,图像尺寸)。我已确认以下几点:
- 预处理步骤与Python实现完全一致(letterbox格式转换、归一化到0-1、通道转置)
- 输出张量形状符合预期((1, 84, 8400)),转置后得到(8400, 84)的张量格式正确
- Python端使用该TFLite模型推理结果正常
仅检测框坐标值异常,特此求助排查思路。
所用代码
#include "tensorflow/lite/model.h" #include "tensorflow/lite/interpreter.h" #include "tensorflow/lite/kernels/register.h" #include <iostream> #include <opencv2/imgproc.hpp> #include <opencv2/highgui.hpp> #include <opencv2/opencv.hpp> #include <random> #define NUM_CLASSES_COCO 80 // Convert to letterbox format (copied from official implementation) cv::Mat formatToSquare(const cv::Mat &source) { int col = source.cols; int row = source.rows; int _max = MAX(col, row); cv::Mat result = cv::Mat::zeros(_max, _max, CV_8UC3); source.copyTo(result(cv::Rect(0, 0, col, row))); return result; } // Preprocessing function (adapted from official) cv::Mat preprocess(cv::Mat &inputImage, int model_input_height, int model_input_width) { cv::Mat modelInputImage = formatToSquare(inputImage); cv::Mat blob; //model height and width are same so doesn't matter in this case cv::dnn::blobFromImage(modelInputImage, blob, 1.0 / 255.0, cv::Size(model_input_height, model_input_width), cv::Scalar(), true, false); return blob; } int main() { cv::Mat inputImage = //Assume loaded correctly; float modelConfidenceThreshold = 0.25; float modelScoreThreshold = 0.45; float modelNMSThreshold = 0.5; // path to a yolov8n tflite file trained on the coco dataset std::unique_ptr<tflite::FlatBufferModel> model = tflite::FlatBufferModel::BuildFromFile(model_path.c_str()); tflite::ops::builtin::BuiltinOpResolver resolver; std::unique_ptr<tflite::Interpreter> interpreter; tflite::InterpreterBuilder(*model, resolver)(&interpreter); // Allocate tensors if (interpreter->AllocateTensors() != kTfLiteOk) { std::cerr << "Failed to allocate tensors." << std::endl; return; } int input = interpreter->inputs()[0]; TfLiteIntArray *dims = interpreter->tensor(input)->dims; cv::Mat preprocessed_image_blob = preprocess(inputImage, model_input_width, model_input_height); float *input_data = interpreter->typed_tensor<float>(input); std::memcpy(input_data, preprocessed_image_blob.data, sizeof(float) * model_input_width * model_input_height * model_input_channels); // Run inference if (interpreter->Invoke() != kTfLiteOk) { std::cerr << "Failed to invoke tflite interpreter." << std::endl; return; } TfLiteTensor *output_tensor = interpreter->tensor(interpreter->outputs()[0]); //dims are (1, 84, 8400) as expected int rows, dimensions; rows = output_shape->data[2]; dimensions = output_shape->data[1]; float* data = output_tensor->data.f; // This is post processing part which is rather tricky. // Following the official implementation to reshape the initial // tensor from (1, 84, 8400) to (8400, 84) and then proceed to // flatten it. cv::Mat temp(dimensions, rows, CV_32F, data); // need this to transpose to shape (8400, 84) cv::transpose(temp, temp); float* new_data = (float*) temp.data; float x_factor = image.cols/model_input_width; float y_factor = image.rows/model_input_height; std::vector<int> class_ids; std::vector<float> confidences; std::vector<cv::Rect> boxes; for (int i = 0; i < rows; i++) { float *classes_scores = new_data + 4; cv::Mat scores(1, NUM_CLASSES_COCO, CV_32FC1, classes_scores); cv::Point class_id; double maxClassScore; cv::minMaxLoc(scores, 0, &maxClassScore, 0, &class_id); if (maxClassScore > modelScoreThreshold) { confidences.push_back(maxClassScore); class_ids.push_back(class_id.x); float x = new_data[0]; float y = new_data[1]; float w = new_data[2]; float h = new_data[3]; int left = int((x - 0.5 * w) * x_factor); int top = int((y - 0.5 * h) * y_factor); int width = int(w * x_factor); int height = int(h * y_factor); boxes.push_back(cv::Rect(left, top, width, height)); } new_data += dimensions; } // Perform NMS over the bounding boxes std::vector<int> nms_result; cv::dnn::NMSBoxes(boxes, confidences, modelScoreThreshold, modelNMSThreshold, nms_result); std::cout << "after nms: " << nms_result.size() << std::endl; std::vector<Detection> detections{}; for (unsigned long i = 0; i < nms_result.size(); ++i) { int idx = nms_result[i]; Detection result; result.class_id = class_ids[idx]; result.confidence = confidences[idx]; std::random_device rd; std::mt19937 gen(rd()); std::uniform_int_distribution<int> dis(100, 255); result.color = cv::Scalar(dis(gen), dis(gen), dis(gen)); result.box = boxes[idx]; detections.push_back(result); } }
内容的提问来源于stack exchange,提问作者Aditya
相关产品推荐
相关产品推荐

