如何让YOLO仅在指定ROI区域内进行目标检测?
问题描述
我通过绘制多边形ROI裁剪了YOLO数据集的图像,仅保留ROI区域内的内容(用遮罩将ROI外变黑),但使用YOLO训练后,模型仍会标记原图像中ROI外的区域(即现在图像的黑色区域)。以下是我的数据集处理代码和训练代码:
数据集处理代码
import os import cv2 import numpy as np import yaml x1, y1, x2, y2 = 250, 229, 738, 511 # 示例坐标 input_base_path = 'yolo_converted' output_base_path = 'roi' input_images_train_path = os.path.join(input_base_path, 'images/train') input_images_val_path = os.path.join(input_base_path, 'images/val') input_labels_train_path = os.path.join(input_base_path, 'labels/train') input_labels_val_path = os.path.join(input_base_path, 'labels/val') output_images_train_path = os.path.join(output_base_path, 'images/train') output_images_val_path = os.path.join(output_base_path, 'images/val') output_labels_train_path = os.path.join(output_base_path, 'labels/train') output_labels_val_path = os.path.join(output_base_path, 'labels/val') os.makedirs(output_images_train_path, exist_ok=True) os.makedirs(output_images_val_path, exist_ok=True) os.makedirs(output_labels_train_path, exist_ok=True) os.makedirs(output_labels_val_path, exist_ok=True) def apply_roi(image_path, label_path, output_image_path, output_label_path, points): image = cv2.imread(image_path) if image is None: print(f"错误:无法读取图像 {image_path}") return mask = np.zeros(image.shape[:2], dtype=np.uint8) cv2.fillPoly(mask, [np.array(points)], 255) masked_image = cv2.bitwise_and(image, image, mask=mask) cv2.imwrite(output_image_path, masked_image) print(f"图像已保存:{output_image_path}") new_label_lines = [] if os.path.exists(label_path): with open(label_path, 'r') as label_file: lines = label_file.readlines() for line in lines: cls, x, y, w, h = map(float, line.strip().split()) abs_x, abs_y = x * image.shape[1], y * image.shape[0] abs_w, abs_h = w * image.shape[1], h * image.shape[0] box_x1 = abs_x - abs_w / 2 box_y1 = abs_y - abs_h / 2 box_x2 = abs_x + abs_w / 2 box_y2 = abs_y + abs_h / 2 if (x1 <= box_x1 <= x2 and y1 <= box_y1 <= y2 and x1 <= box_x2 <= x2 and y1 <= box_y2 <= y2): new_x = (abs_x - x1) / (x2 - x1) new_y = (abs_y - y1) / (y2 - y1) new_w = abs_w / (x2 - x1) new_h = abs_h / (y2 - y1) new_label_lines.append(f"{cls} {new_x:.6f} {new_y:.6f} {new_w:.6f} {new_h:.6f}\n") with open(output_label_path, 'w') as new_label_file: new_label_file.writelines(new_label_lines) print(f"标签已保存:{output_label_path}") else: print(f"警告:未找到标签文件 {label_path}") def process_folder(input_images_path, input_labels_path, output_images_path, output_labels_path, points): for file_name in os.listdir(input_images_path): image_path = os.path.join(input_images_path, file_name) label_path = os.path.join(input_labels_path, os.path.splitext(file_name)[0] + '.txt') output_image_path = os.path.join(output_images_path, file_name) output_label_path = os.path.join(output_labels_path, os.path.splitext(file_name)[0] + '.txt') if os.path.isfile(image_path): print(f"正在处理:{image_path}") apply_roi(image_path, label_path, output_image_path, output_label_path, points) else: print(f"警告:无效文件 {file_name}") points = [(250, 253), (330, 229), (738, 444), (601, 511)] print("正在处理Train文件夹...") process_folder(input_images_train_path, input_labels_train_path, output_images_train_path, output_labels_train_path, points) print("正在处理Val文件夹...") process_folder(input_images_val_path, input_labels_val_path, output_images_val_path, output_labels_val_path, points) train_yaml_path = os.path.join(input_base_path, 'train.yaml') with open(train_yaml_path, 'r') as file: train_yaml_data = yaml.safe_load(file) base_path = '/home/ubuntu/yolo/folders/performance_metrics/' train_yaml_data['train'] = os.path.join(base_path, output_images_train_path) train_yaml_data['val'] = os.path.join(base_path, output_images_val_path) output_yaml_path = os.path.join(output_base_path, 'train.yaml') with open(output_yaml_path, 'w') as yaml_file: yaml.dump(train_yaml_data, yaml_file)
训练代码
from ultralytics import YOLO model = YOLO('yolov8n.pt') # 使用YOLOv8则用'yolov8n.pt',YOLOv5则用'yolov5s.pt' results = model.train(data='roi/train.yaml', epochs=20, imgsz=640, batch=4)
问题原因
- 仅遮罩未实际裁剪图像:当前代码只是用遮罩把ROI外区域变黑,但图像尺寸仍保留原大小,模型训练时会看到整个图像(包括黑色区域),推理时也会对全图进行检测,从而产生ROI外的误检。
- 标签过滤条件过于严格:代码中要求目标框的四个角都在ROI内才保留,会丢失部分仅部分区域在ROI内的有效目标,导致训练数据不完整。
- 坐标转换基准错误:使用
x1,y1,x2,y2作为ROI的边界,但实际ROI是多边形,这两个值并非多边形的最小/最大边界,导致标签坐标转换不准确。
解决方案
方案1:实际裁剪图像到ROI区域(推荐)
修改数据集处理代码,将图像裁剪为ROI多边形的最小外接矩形尺寸,彻底移除ROI外的区域,让模型只关注有效区域:
修改后的apply_roi函数
def apply_roi(image_path, label_path, output_image_path, output_label_path, points): image = cv2.imread(image_path) if image is None: print(f"错误:无法读取图像 {image_path}") return # 转换为numpy数组 roi_polygon = np.array(points, np.int32) # 获取ROI的最小外接矩形边界 x_min, y_min = np.min(roi_polygon, axis=0) x_max, y_max = np.max(roi_polygon, axis=0) # 创建遮罩并提取ROI区域 mask = np.zeros(image.shape[:2], dtype=np.uint8) cv2.fillPoly(mask, [roi_polygon], 255) masked_image = cv2.bitwise_and(image, image, mask=mask) # 裁剪到ROI的最小外接矩形 cropped_image = masked_image[y_min:y_max, x_min:x_max] cv2.imwrite(output_image_path, cropped_image) print(f"图像已保存:{output_image_path}") new_label_lines = [] if os.path.exists(label_path): with open(label_path, 'r') as label_file: lines = label_file.readlines() for line in lines: cls, x, y, w, h = map(float, line.strip().split()) abs_x = x * image.shape[1] abs_y = y * image.shape[0] abs_w = w * image.shape[1] abs_h = h * image.shape[0] box_x1 = abs_x - abs_w / 2 box_y1 = abs_y - abs_h / 2 box_x2 = abs_x + abs_w / 2 box_y2 = abs_y + abs_h / 2 # 放宽过滤条件:只要目标中心在ROI内,或与ROI交集超过50% # 检查中心是否在ROI内 center_in_roi = cv2.pointPolygonTest(roi_polygon, (abs_x, abs_y), False) >= 0 # 计算框与ROI的交集面积 box_polygon = np.array([[box_x1, box_y1], [box_x2, box_y1], [box_x2, box_y2], [box_x1, box_y2]], np.int32) intersection_area = cv2.intersectConvexConvex(roi_polygon, box_polygon)[0] box_area = abs_w * abs_h intersection_ratio = intersection_area / box_area if box_area > 0 else 0 if center_in_roi or intersection_ratio > 0.5: # 基于裁剪后的图像尺寸转换坐标 new_x = (abs_x - x_min) / (x_max - x_min) new_y = (abs_y - y_min) / (y_max - y_min) new_w = abs_w / (x_max - x_min) new_h = abs_h / (y_max - y_min) # 确保坐标在[0,1]范围内(避免浮点误差) new_x = np.clip(new_x, 0, 1) new_y = np.clip(new_y, 0, 1) new_w = np.clip(new_w, 0, 1) new_h = np.clip(new_h, 0, 1) new_label_lines.append(f"{cls} {new_x:.6f} {new_y:.6f} {new_w:.6f} {new_h:.6f}\n") with open(output_label_path, 'w') as new_label_file: new_label_file.writelines(new_label_lines) print(f"标签已保存:{output_label_path}") else: print(f"警告:未找到标签文件 {label_path}")
修改后重新生成数据集并训练,模型将只基于ROI区域的图像进行学习,推理时自然不会检测原ROI外的区域。
方案2:推理阶段过滤ROI外的检测结果
如果不想重新训练数据集,可以在推理时手动过滤掉ROI外的检测框:
推理过滤代码
from ultralytics import YOLO import cv2 import numpy as np # 加载训练好的模型 model = YOLO('runs/detect/train/weights/best.pt') # 原ROI多边形坐标 roi_points = [(250, 253), (330, 229), (738, 444), (601, 511)] roi_polygon = np.array(roi_points, np.int32) # 处理单张测试图像 test_img_path = 'test_image.jpg' img = cv2.imread(test_img_path) results = model(img) # 过滤检测结果 for result in results: filtered_boxes = [] for box in result.boxes: # 获取框的中心坐标 x_center, y_center = box.xywh[0][0].item(), box.xywh[0][1].item() # 判断中心是否在ROI内 if cv2.pointPolygonTest(roi_polygon, (x_center, y_center), False) >= 0: filtered_boxes.append(box) # 更新结果中的框 result.boxes = filtered_boxes # 绘制过滤后的结果 annotated_img = results[0].plot() cv2.imwrite('filtered_result.jpg', annotated_img) cv2.imshow('Filtered Detection', annotated_img) cv2.waitKey(0) cv2.destroyAllWindows()
这种方法不需要修改训练数据,直接在推理阶段过滤掉无效检测,适合快速验证。
方案3:训练时添加ROI约束(进阶)
可以通过修改YOLO的损失函数,对ROI外的预测框施加惩罚,或者自定义数据增强策略,强制模型只关注ROI区域。但这种方法需要修改YOLO的源码,复杂度较高,仅适合有一定代码基础的用户。
内容的提问来源于stack exchange,提问作者Seher Elbasan
相关产品推荐
相关产品推荐

