You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++端TFLite部署YOLOv5 FP16模型推理异常求助

问题:YOLOv5 FP16 TFLite模型推理时输入类型与内存规格不匹配导致输出混乱

在C++环境运行YOLOv5 FP16格式的TFLite模型推理时,发现模型第一层输入类型为F32,但TFLite分配的内存却是FP16规格,最终导致输出张量数据混乱。

模型加载代码

tflite::ops::builtin::BuiltinOpResolver resolver;
tflite::InterpreterBuilder builder(*mModel, resolver);
builder(&mInterpreter);
TfLiteStatus status = mInterpreter->AllocateTensors();
_input = mInterpreter->inputs()[0];
TfLiteIntArray *dims = mInterpreter->tensor(_input)->dims;
_in_height = dims->data[1];
_in_width = dims->data[2];
_in_channels = dims->data[3];
_in_type = mInterpreter->tensor(_input)->type;
// yolo v5 input tensor which is kTFLiteFloat32
_input_8 = mInterpreter->typed_input_tensor<float>(_input);

推理读取代码

int _out = mInterpreter->outputs()[0];
TfLiteIntArray *_out_dims = mInterpreter->tensor(_out)->dims;
int _out_row   = _out_dims->data[1];   
int _out_colum = _out_dims->data[2];   
/*  model with
 *{
 *   "output_size": [1, 6300, 85],
 *   "input_size":  [320, 320]
 *}
 */
TfLiteTensor *pOutputTensor = mInterpreter->tensor(mInterpreter->outputs()[0]);

std::vector<std::vector<float>> predV{};
tensor2Vector2d(pOutputTensor, predV, _out_row, _out_colum);

std::vector<int> indices;
std::vector<int> classIds;
std::vector<float> confidences;
std::vector<cv::Rect> boxes;

nonMaximumSupprition(predV,
    boxes,
    confidences,
    classIds,
    indices,
    _out_row,
    _out_colum);

void FaceDetector::tensor2Vector2d(
    const TfLiteTensor *tensor,
    std::vector<std::vector<float>> &predV,
    const int row,
    const int col) {

    for (int32_t i = 0; i < row; i++){
        std::vector<float> _tem;

        for (int j = 0; j < col; j++){
            float val_float = tensor->data.f[i * col + j];
            _tem.push_back(val_float);
        }
        predV.push_back(_tem);
    }
}

void FaceDetector::nonMaximumSupprition(std::vector<std::vector<float>>& predV,
    std::vector<cv::Rect> &boxes,
    std::vector<float> &confidences,
    std::vector<int> &classIds,
    std::vector<int> &indices,
    const int &row,
    const int &colum) {
    std::vector<cv::Rect> boxesNMS;
    std::vector<float> scores;
    double confidence;
    cv::Point classId;

    for (int i = 0; i < row; i++){
        if (predV[i][4] > confThreshold){
            // height--> image.rows,  width--> image.cols;
            int left = (predV[i][0] - predV[i][2] / 2) * _img_width;
            int top = (predV[i][1] - predV[i][3] / 2) * _img_height;
            int w = predV[i][2] * _img_width;
            int h = predV[i][3] * _img_height;

            for (int j = 5; j < colum; j++)
            {
                // conf = obj_conf * cls_conf
                scores.push_back(predV[i][j] * predV[i][4]);
            }

            cv::minMaxLoc(
                scores,
                0,
                &confidence,
                0,
                &classId);

            if (confidence > confThreshold) {
                boxes.push_back(cv::Rect(left, top, w, h));
                confidences.push_back(confidence);
                classIds.push_back(classId.x % 80);
                boxesNMS.push_back(cv::Rect(left, top, w, h));
            }
        }
    }

    cv::dnn::NMSBoxes(
        boxesNMS,
        confidences,
        confThreshold,
        nmsThreshold,
        indices);
}

预处理代码

// copy img data to input tensor
template <typename T>
void FaceDetector::fill(T *in, cv::Mat &src) {
    int n = 0, nc = src.channels(), ne = src.elemSize();

    if (src.isContinuous()){
        memcpy(in, src.data, nc * src.cols * src.rows);
        return;
    }
    
    for (int y = 0; y < src.rows; ++y)
        for (int x = 0; x < src.cols; ++x)
            for (int c = 0; c < nc; ++c)
                in[n++] = src.data[y * src.step + x * ne + c];
}

// preProcess
void FaceDetector::preProcess(cv::Mat &img) {
    cv::resize(
        img,
        img,
        cv::Size(
            _in_width,
            _in_height),
        cv::INTER_CUBIC);
        // convert img from uchar to f32 
        img.convertTo(img, CV_32FC3);
        // normalize
        img /= 255;
}

曾尝试将输入张量指针转为int16(误将其等同于float16),并将图像转为CV_16FC3格式,但问题仍未解决,求可行的解决方案。

内容的提问来源于stack exchange,提问作者Ice Wind

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.03 23:40:58