You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何让YOLO仅在指定ROI区域内进行目标检测?

问题描述

我通过绘制多边形ROI裁剪了YOLO数据集的图像,仅保留ROI区域内的内容(用遮罩将ROI外变黑),但使用YOLO训练后,模型仍会标记原图像中ROI外的区域(即现在图像的黑色区域)。以下是我的数据集处理代码和训练代码:

数据集处理代码

import os
import cv2
import numpy as np
import yaml

x1, y1, x2, y2 = 250, 229, 738, 511  # 示例坐标

input_base_path = 'yolo_converted'
output_base_path = 'roi'

input_images_train_path = os.path.join(input_base_path, 'images/train')
input_images_val_path = os.path.join(input_base_path, 'images/val')
input_labels_train_path = os.path.join(input_base_path, 'labels/train')
input_labels_val_path = os.path.join(input_base_path, 'labels/val')

output_images_train_path = os.path.join(output_base_path, 'images/train')
output_images_val_path = os.path.join(output_base_path, 'images/val')
output_labels_train_path = os.path.join(output_base_path, 'labels/train')
output_labels_val_path = os.path.join(output_base_path, 'labels/val')

os.makedirs(output_images_train_path, exist_ok=True)
os.makedirs(output_images_val_path, exist_ok=True)
os.makedirs(output_labels_train_path, exist_ok=True)
os.makedirs(output_labels_val_path, exist_ok=True)

def apply_roi(image_path, label_path, output_image_path, output_label_path, points):
    image = cv2.imread(image_path)
    if image is None:
        print(f"错误:无法读取图像 {image_path}")
        return

    mask = np.zeros(image.shape[:2], dtype=np.uint8)
    cv2.fillPoly(mask, [np.array(points)], 255)
    masked_image = cv2.bitwise_and(image, image, mask=mask)

    cv2.imwrite(output_image_path, masked_image)
    print(f"图像已保存:{output_image_path}")

    new_label_lines = []
    if os.path.exists(label_path):
        with open(label_path, 'r') as label_file:
            lines = label_file.readlines()
            for line in lines:
                cls, x, y, w, h = map(float, line.strip().split())
                abs_x, abs_y = x * image.shape[1], y * image.shape[0]
                abs_w, abs_h = w * image.shape[1], h * image.shape[0]

                box_x1 = abs_x - abs_w / 2
                box_y1 = abs_y - abs_h / 2
                box_x2 = abs_x + abs_w / 2
                box_y2 = abs_y + abs_h / 2

                if (x1 <= box_x1 <= x2 and y1 <= box_y1 <= y2 and
                    x1 <= box_x2 <= x2 and y1 <= box_y2 <= y2):
                    new_x = (abs_x - x1) / (x2 - x1)
                    new_y = (abs_y - y1) / (y2 - y1)
                    new_w = abs_w / (x2 - x1)
                    new_h = abs_h / (y2 - y1)
                    new_label_lines.append(f"{cls} {new_x:.6f} {new_y:.6f} {new_w:.6f} {new_h:.6f}\n")

        with open(output_label_path, 'w') as new_label_file:
            new_label_file.writelines(new_label_lines)
        print(f"标签已保存:{output_label_path}")
    else:
        print(f"警告:未找到标签文件 {label_path}")

def process_folder(input_images_path, input_labels_path, output_images_path, output_labels_path, points):
    for file_name in os.listdir(input_images_path):
        image_path = os.path.join(input_images_path, file_name)
        label_path = os.path.join(input_labels_path, os.path.splitext(file_name)[0] + '.txt')
        output_image_path = os.path.join(output_images_path, file_name)
        output_label_path = os.path.join(output_labels_path, os.path.splitext(file_name)[0] + '.txt')

        if os.path.isfile(image_path):
            print(f"正在处理:{image_path}")
            apply_roi(image_path, label_path, output_image_path, output_label_path, points)
        else:
            print(f"警告:无效文件 {file_name}")

points = [(250, 253), (330, 229), (738, 444), (601, 511)]

print("正在处理Train文件夹...")
process_folder(input_images_train_path, input_labels_train_path, output_images_train_path, output_labels_train_path, points)
print("正在处理Val文件夹...")
process_folder(input_images_val_path, input_labels_val_path, output_images_val_path, output_labels_val_path, points)

train_yaml_path = os.path.join(input_base_path, 'train.yaml')
with open(train_yaml_path, 'r') as file:
    train_yaml_data = yaml.safe_load(file)

base_path = '/home/ubuntu/yolo/folders/performance_metrics/'
train_yaml_data['train'] = os.path.join(base_path, output_images_train_path)
train_yaml_data['val'] = os.path.join(base_path, output_images_val_path)

output_yaml_path = os.path.join(output_base_path, 'train.yaml')
with open(output_yaml_path, 'w') as yaml_file:
    yaml.dump(train_yaml_data, yaml_file)

训练代码

from ultralytics import YOLO
model = YOLO('yolov8n.pt')  # 使用YOLOv8则用'yolov8n.pt',YOLOv5则用'yolov5s.pt'
results = model.train(data='roi/train.yaml', epochs=20, imgsz=640, batch=4)
问题原因
  • 仅遮罩未实际裁剪图像:当前代码只是用遮罩把ROI外区域变黑,但图像尺寸仍保留原大小,模型训练时会看到整个图像(包括黑色区域),推理时也会对全图进行检测,从而产生ROI外的误检。
  • 标签过滤条件过于严格:代码中要求目标框的四个角都在ROI内才保留,会丢失部分仅部分区域在ROI内的有效目标,导致训练数据不完整。
  • 坐标转换基准错误:使用x1,y1,x2,y2作为ROI的边界,但实际ROI是多边形,这两个值并非多边形的最小/最大边界,导致标签坐标转换不准确。
解决方案

方案1:实际裁剪图像到ROI区域(推荐)

修改数据集处理代码,将图像裁剪为ROI多边形的最小外接矩形尺寸,彻底移除ROI外的区域,让模型只关注有效区域:

修改后的apply_roi函数

def apply_roi(image_path, label_path, output_image_path, output_label_path, points):
    image = cv2.imread(image_path)
    if image is None:
        print(f"错误:无法读取图像 {image_path}")
        return

    # 转换为numpy数组
    roi_polygon = np.array(points, np.int32)
    # 获取ROI的最小外接矩形边界
    x_min, y_min = np.min(roi_polygon, axis=0)
    x_max, y_max = np.max(roi_polygon, axis=0)
    
    # 创建遮罩并提取ROI区域
    mask = np.zeros(image.shape[:2], dtype=np.uint8)
    cv2.fillPoly(mask, [roi_polygon], 255)
    masked_image = cv2.bitwise_and(image, image, mask=mask)
    # 裁剪到ROI的最小外接矩形
    cropped_image = masked_image[y_min:y_max, x_min:x_max]
    
    cv2.imwrite(output_image_path, cropped_image)
    print(f"图像已保存:{output_image_path}")

    new_label_lines = []
    if os.path.exists(label_path):
        with open(label_path, 'r') as label_file:
            lines = label_file.readlines()
            for line in lines:
                cls, x, y, w, h = map(float, line.strip().split())
                abs_x = x * image.shape[1]
                abs_y = y * image.shape[0]
                abs_w = w * image.shape[1]
                abs_h = h * image.shape[0]

                box_x1 = abs_x - abs_w / 2
                box_y1 = abs_y - abs_h / 2
                box_x2 = abs_x + abs_w / 2
                box_y2 = abs_y + abs_h / 2

                # 放宽过滤条件:只要目标中心在ROI内,或与ROI交集超过50%
                # 检查中心是否在ROI内
                center_in_roi = cv2.pointPolygonTest(roi_polygon, (abs_x, abs_y), False) >= 0
                # 计算框与ROI的交集面积
                box_polygon = np.array([[box_x1, box_y1], [box_x2, box_y1], [box_x2, box_y2], [box_x1, box_y2]], np.int32)
                intersection_area = cv2.intersectConvexConvex(roi_polygon, box_polygon)[0]
                box_area = abs_w * abs_h
                intersection_ratio = intersection_area / box_area if box_area > 0 else 0

                if center_in_roi or intersection_ratio > 0.5:
                    # 基于裁剪后的图像尺寸转换坐标
                    new_x = (abs_x - x_min) / (x_max - x_min)
                    new_y = (abs_y - y_min) / (y_max - y_min)
                    new_w = abs_w / (x_max - x_min)
                    new_h = abs_h / (y_max - y_min)
                    # 确保坐标在[0,1]范围内(避免浮点误差)
                    new_x = np.clip(new_x, 0, 1)
                    new_y = np.clip(new_y, 0, 1)
                    new_w = np.clip(new_w, 0, 1)
                    new_h = np.clip(new_h, 0, 1)
                    new_label_lines.append(f"{cls} {new_x:.6f} {new_y:.6f} {new_w:.6f} {new_h:.6f}\n")

        with open(output_label_path, 'w') as new_label_file:
            new_label_file.writelines(new_label_lines)
        print(f"标签已保存:{output_label_path}")
    else:
        print(f"警告:未找到标签文件 {label_path}")

修改后重新生成数据集并训练,模型将只基于ROI区域的图像进行学习,推理时自然不会检测原ROI外的区域。

方案2:推理阶段过滤ROI外的检测结果

如果不想重新训练数据集,可以在推理时手动过滤掉ROI外的检测框:

推理过滤代码

from ultralytics import YOLO
import cv2
import numpy as np

# 加载训练好的模型
model = YOLO('runs/detect/train/weights/best.pt')
# 原ROI多边形坐标
roi_points = [(250, 253), (330, 229), (738, 444), (601, 511)]
roi_polygon = np.array(roi_points, np.int32)

# 处理单张测试图像
test_img_path = 'test_image.jpg'
img = cv2.imread(test_img_path)
results = model(img)

# 过滤检测结果
for result in results:
    filtered_boxes = []
    for box in result.boxes:
        # 获取框的中心坐标
        x_center, y_center = box.xywh[0][0].item(), box.xywh[0][1].item()
        # 判断中心是否在ROI内
        if cv2.pointPolygonTest(roi_polygon, (x_center, y_center), False) >= 0:
            filtered_boxes.append(box)
    # 更新结果中的框
    result.boxes = filtered_boxes

# 绘制过滤后的结果
annotated_img = results[0].plot()
cv2.imwrite('filtered_result.jpg', annotated_img)
cv2.imshow('Filtered Detection', annotated_img)
cv2.waitKey(0)
cv2.destroyAllWindows()

这种方法不需要修改训练数据,直接在推理阶段过滤掉无效检测,适合快速验证。

方案3:训练时添加ROI约束(进阶)

可以通过修改YOLO的损失函数,对ROI外的预测框施加惩罚,或者自定义数据增强策略,强制模型只关注ROI区域。但这种方法需要修改YOLO的源码,复杂度较高,仅适合有一定代码基础的用户。


内容的提问来源于stack exchange,提问作者Seher Elbasan

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 20:07:01