You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

YOLOv8模型从.pt转ONNX后输出处理困惑求助

问题:YOLOv8 ONNX模型输出处理错误导致检测框位置异常

我把基于自定义数据集迁移训练的YOLOv8 .pt模型转成了ONNX格式,要部署到没法加载Ultralytics版本YOLOv8的边缘设备。转换后的ONNX模型能加载运行预测,但输出数据处理不对,检测框位置完全错了。我写了两个测试脚本对比同一张图的.pt和ONNX模型输出,求解决办法,相关输出和脚本如下:


.pt模型输出

results: [ultralytics.engine.results.Results object with attributes:

boxes: ultralytics.engine.results.Boxes object
keypoints: None
masks: None
names: {0: 'trolley'}
orig_img: array([[[142, 144, 158],
        [173, 175, 188],
        [192, 194, 207],
        ...,
        [ 45,  53,  72],
        [ 44,  52,  71],
        [ 43,  51,  70]],

       [[138, 140, 153],
        [142, 144, 157],
        [145, 146, 159],
        ...,
        [ 45,  53,  72],
        [ 45,  53,  72],
        [ 45,  53,  72]],

       [[136, 137, 150],
        [161, 163, 176],
        [138, 140, 152],
        ...,
        [ 44,  52,  71],
        [ 46,  54,  73],
        [ 48,  56,  75]],

       ...,

       [[ 75,  81,  97],
        [ 77,  83,  99],
        [ 74,  80,  96],
        ...,
        [ 65,  67,  79],
        [ 65,  67,  79],
        [ 65,  67,  79]],

       [[ 77,  83,  99],
        [ 79,  85, 101],
        [ 76,  82,  98],
        ...,
        [ 64,  66,  78],
        [ 64,  66,  78],
        [ 64,  66,  78]],

       [[ 77,  83,  99],
        [ 80,  85, 101],
        [ 76,  82,  98],
        ...,
        [ 63,  65,  77],
        [ 63,  65,  77],
        [ 63,  65,  77]]], dtype=uint8)
orig_shape: (640, 640)
path: 'image0.jpg'
probs: None
save_dir: None
speed: {'preprocess': 0.0, 'inference': 81.86841011047363, 'postprocess': 48.83885383605957}]
0: 640x640 1 trolley, 81.9ms
Speed: 0.0ms preprocess, 81.9ms inference, 48.8ms postprocess per image at shape (1, 3, 640, 640)

.onnx模型输出

Number of output tensors: 1
Output predictions: [array([[[6.74059772e+00, 1.68045845e+01, 2.33795547e+01, ...,
         5.81766541e+02, 5.95164062e+02, 6.02357910e+02],
        [1.48594885e+01, 1.79399471e+01, 1.95681229e+01, ...,
         6.07668945e+02, 6.05005066e+02, 5.90472534e+02],
        [1.38631535e+01, 3.30112076e+01, 4.58652878e+01, ...,
         1.03839111e+02, 8.46026001e+01, 8.07698364e+01],
        [2.82973385e+01, 3.45392494e+01, 3.80951614e+01, ...,
         6.51624146e+01, 6.59660645e+01, 9.91400146e+01],
        [1.19209290e-07, 0.00000000e+00, 0.00000000e+00, ...,
         5.96046448e-08, 1.75833702e-06, 1.34110451e-06]]], dtype=float32)]
Output tensor shape: (1, 5, 8400)
Sample output values (first 10):
[ 6.7405977 16.804585  23.379555  25.867968  34.81388   40.286224
 45.366837  55.982033  70.706924  76.8924   ]
Number of detections per image: 8400
Sample detection values (first set of 5 values):
[6.74059772e+00 1.48594885e+01 1.38631535e+01 2.82973385e+01
 1.19209290e-07]
Slot 1: Min value: 2.1957478523254395, Max value: 636.6444702148438, Mean value: 318.2766418457031
Slot 2: Min value: 4.1632208824157715, Max value: 636.5177001953125, Mean value: 321.9232177734375
Slot 3: Min value: 5.034980773925781, Max value: 206.54994201660156, Mean value: 60.202857971191406
Slot 4: Min value: 6.93548583984375, Max value: 306.9089660644531, Mean value: 90.83543395996094
Slot 5: Min value: 0.0, Max value: 0.8410309553146362, Mean value: 0.0009538077283650637

Inspecting patterns in the first 10 detections for each slot:
Slot 1 first 10 values: [ 6.7405977 16.804585  23.379555  25.867968  34.81388   40.286224
 45.366837  55.982033  70.706924  76.8924   ]
Slot 2 first 10 values: [14.8594885 17.939947  19.568123  18.928112  12.157583   9.730759
  9.018981  11.631983   9.608661   7.939432 ]
Slot 3 first 10 values: [13.863153 33.011208 45.865288 47.089123 36.74875  32.579952 29.31453
 29.247162 27.974869 25.54161 ]
Slot 4 first 10 values: [28.297338 34.53925  38.09516  36.430298 23.29528  18.743835 17.69494
 23.052486 18.969786 15.53459 ]
Slot 5 first 10 values: [1.1920929e-07 0.0000000e+00 0.0000000e+00 0.0000000e+00 0.0000000e+00
 0.0000000e+00 0.0000000e+00 0.0000000e+00 0.0000000e+00 0.0000000e+00]
Slots possibly containing confidence scores (0-100 range): [4]

.pt模型检测脚本

import torch
from ultralytics import YOLO
import cv2


DEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'

model_path = 'Trolley_Detect-02.pt'
image_path = '13.jpg'
model = YOLO(model_path)
model.to(DEVICE)


def preprocess_image(image, width, height):
    resized_frame = cv2.resize(image, (width, height))
    resized_frame = cv2.cvtColor(resized_frame, cv2.COLOR_BGR2RGB)
    img_tensor = torch.from_numpy(resized_frame).unsqueeze(0).permute(0, 3, 1, 2).float().div(255.0).to(DEVICE)
    return img_tensor


frame = cv2.imread(image_path)
img_tensor = preprocess_image(frame, 640, 640)

with torch.no_grad():
    results = model(img_tensor)

print(f'results: {results}')

.onnx模型检测脚本

import cv2
import numpy as np
import onnxruntime as ort
from PIL import Image

THRESHOLD = 0.5
NMS_THRESHOLD = 0.5
CLASSES = ["trolley"]

def preprocess_image(image_path, input_size=(640, 640)):
    image = Image.open(image_path).convert("RGB")
    image = image.resize(input_size)
    image_np = np.array(image).astype(np.float32)
    image_np = np.transpose(image_np, (2, 0, 1))
    image_np /= 255.0
    image_np = np.expand_dims(image_np, axis=0)
    return image_np

def infer(model_path, input_tensor):
    session = ort.InferenceSession(model_path)
    input_name = session.get_inputs()[0].name
    predictions = session.run(None, {input_name: input_tensor})
    return predictions

def inspect_output_tensor(predictions):
    print(f"Number of output tensors: {len(predictions)}")
    print(f"Output predictions: {predictions}")
    # Assuming there's only one output
    output_tensor = predictions[0]
    print(f"Output tensor shape: {output_tensor.shape}")

    # Print the first 10 values in the output tensor for inspection
    print("Sample output values (first 10):")
    print(output_tensor.flatten()[:10])

    # Assuming the output tensor shape is (1, 5, N), where N is the number of detections
    num_detections = output_tensor.shape[2]
    print(f"Number of detections per image: {num_detections}")

    # Sample the first detection set (first set of 5 values) to inspect
    sample_detection = output_tensor[0, :, 0]
    print("Sample detection values (first set of 5 values):")
    print(sample_detection)

    # Inspect the range of values in each of the 5 slots across all detections
    for i in range(5):
        slot_values = output_tensor[0, i, :]
        print(
            f"Slot {i + 1}: Min value: {slot_values.min()}, Max value: {slot_values.max()}, Mean value: {slot_values.mean()}")

    print("\nInspecting patterns in the first 10 detections for each slot:")
    for i in range(5):
        slot_values_first_10 = output_tensor[0, i, :10]
        print(f"Slot {i + 1} first 10 values: {slot_values_first_10}")

    # If confidence scores are a range from 0 to 100 as suspected,
    # identify which slot(s) might contain these values
    possible_confidence_slots = [i for i in range(5) if
                                 (output_tensor[0, i, :].min() >= 0 and output_tensor[0, i, :].max() <= 100)]
    print(f"Slots possibly containing confidence scores (0-100 range): {possible_confidence_slots}")


image_path = "13.jpg"
input_tensor = preprocess_image(image_path)

model_path = "opset13/Trolley_Detect-02.onnx"
predictions = infer(model_path, input_tensor)

inspect_output_tensor(predictions)

模型转换脚本

from ultralytics import YOLO

# Load a model
model = YOLO('Trolley_Detect-02.pt')  # load a custom trained model

# Export the model
model.export(format='onnx')

解决方案

1. 明确ONNX输出格式

YOLOv8的ONNX输出是**(1, 5, 8400)**的张量,其中5个维度对应的是:[x_center, y_center, width, height, confidence]。你之前对输出维度的对应关系判断错误,这是检测框位置异常的核心原因:

  • Slot1 = x_center(中心点x坐标)
  • Slot2 = y_center(中心点y坐标)
  • Slot3 = width(框宽度)
  • Slot4 = height(框高度)
  • Slot5 = confidence(置信度,范围0-1)

2. 后处理步骤修正

需要将中心点坐标转换为左上角和右下角坐标,同时进行置信度过滤和NMS(非极大值抑制),步骤如下:

def postprocess(output_tensor, input_shape, orig_image_shape, conf_threshold=0.5, nms_threshold=0.5):
    # 提取数据:(1,5,8400) -> (5,8400) -> (8400,5)
    predictions
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.30 08:17:21