如何将TensorFlow输出的边界框调整适配到原始图像尺寸
解决方案
核心原理
你当前使用的TFLite目标检测模型输出的边界框为[y_min, x_min, y_max, x_max]格式的归一化数值,取值范围为0~1,是相对于输入模型的512x512尺寸图像的比例值。要将边界框映射回原始尺寸图像,只需要将四个归一化值分别乘以原图的实际高度、宽度即可,不需要基于resize后的512尺寸计算。
代码修改说明
- 读取原始图像后先记录原图的宽高参数
- 边界框计算时使用原图宽高代替resize后的512尺寸参数
- 如需裁剪对应区域,直接基于计算出的原图坐标切片即可保存
完整修改后代码
import tensorflow as tf import numpy as np import cv2 import pathlib import os from silence_tensorflow import silence_tensorflow from PIL import Image silence_tensorflow() interpreter = tf.lite.Interpreter(model_path="model.tflite") input_details = interpreter.get_input_details() output_details = interpreter.get_output_details() interpreter.allocate_tensors() def draw_rect(image, box): h, w, c = image.shape y_min = int(max(1, (box[0] * h))) x_min = int(max(1, (box[1] * w))) y_max = int(min(h, (box[2] * h))) x_max = int(min(w, (box[3] * w))) # draw a rectangle on the image cv2.rectangle(image, (x_min, y_min), (x_max, y_max), (13, 13, 13), 2) # 返回坐标方便后续裁剪 return x_min, y_min, x_max, y_max for file in pathlib.Path('./').iterdir(): if file.suffix != '.jpeg' and file.suffix != '.png': continue img = cv2.imread(r"{}".format(file.resolve())) # 记录原图宽高 orig_h, orig_w = img.shape[:2] print(f'[Converting] {file.resolve()}') new_img = cv2.resize(img, (512, 512)) interpreter.set_tensor(input_details[0]['index'], [new_img]) interpreter.invoke() rects = interpreter.get_tensor( output_details[0]['index']) scores = interpreter.get_tensor( output_details[2]['index']) for index, score in enumerate(scores[0]): if index == 0: print('[Quantity]') print(index,score) if index == 1: print('[Barcode]') print(index,score) if score > 0.2: # 如需在resize后的图上画框保留这行 draw_rect(new_img,rects[0][index]) # 计算原图坐标、画框、裁剪 x_min, y_min, x_max, y_max = draw_rect(img, rects[0][index]) # 裁剪对应区域 crop_region = img[y_min:y_max, x_min:x_max] # 保存裁剪结果 cv2.imwrite(f"crop_{file.stem}_{index}.png", crop_region) print(f"裁剪区域坐标:{x_min, y_min, x_max, y_max}") # 展示带框的原始图像可替换为img cv2.imshow("resized image", new_img) cv2.imshow("original image", img) cv2.waitKey(0) cv2.destroyAllWindows()
补充说明
如果你的原始图像宽高比不等于1:1,直接resize到512x512会导致图像拉伸,进而引起边界框偏移。如果需要更高的坐标精度,建议预处理图像时采用等比例缩放+边缘补零的方式调整到512x512,映射坐标时再减去对应的补零偏移量即可。
内容的提问来源于stack exchange,提问作者Muneeb Ahmad Khurram
相关产品推荐
相关产品推荐

