You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

在PyQt中基于BB坐标实现bounding box点击选中以标注对象名称

实现方案

核心修改点

  1. 调整数据存储逻辑,将同一张图片的所有bounding box(以下简称bbox)按路径分组,解决单图多bbox的展示问题
  2. 给显示图片的Label添加点击事件监听,实现点击坐标到图片实际像素坐标的转换
  3. 实现bbox碰撞检测逻辑,判断点击坐标落在哪个bbox范围内
  4. 新增选中状态高亮,选中的bbox用不同颜色绘制
  5. 新增命名按钮绑定逻辑,支持给选中的bbox赋值名称

完整修改后代码

from PyQt5 import QtGui, QtWidgets, QtCore
from PyQt5.QtWidgets import QFileDialog, QInputDialog, QApplication
import csv
from pygui import Ui_MainWindow
from collections import defaultdict
import sys
import cv2

# 给Row新增可选的name字段
class Row:
    def __init__(self, image_path, x, y, w, h, name=None):
        self.image_path = image_path
        self.x = int(x)
        self.y = int(y)
        self.w = int(w)
        self.h = int(h)
        self.name = name

class mainProgram(QtWidgets.QMainWindow, Ui_MainWindow):
    def __init__(self, parent=None):
        super(mainProgram, self).__init__(parent)
        self.setupUi(self)
        # 初始化存储变量
        self.image_groups = defaultdict(list) # 按图片路径分组存储所有bbox
        self.image_list = [] # 所有不重复的图片路径列表,用于切换
        self.current_img_index = 0 # 当前显示的图片索引
        self.current_bboxes = [] # 当前图片的所有bbox
        self.selected_bbox = None # 当前选中的bbox
        # 给label安装事件过滤器,监听点击事件
        self.label.installEventFilter(self)

    def all_callbacks(self):
        self.Upload.clicked.connect(self.on_click_upload)
        self.Next.clicked.connect(self.on_click_next)
        self.Previous.clicked.connect(self.on_click_previous)
        # 新增命名按钮的回调,替换为你UI里命名按钮的实际对象名
        self.NameBtn.clicked.connect(self.on_click_name)

    def eventFilter(self, watched, event):
        # 监听label的鼠标点击事件
        if watched == self.label and event.type() == QtCore.QEvent.MouseButtonPress:
            if not self.label.pixmap() or len(self.current_bboxes) == 0:
                return super().eventFilter(watched, event)
            # 坐标转换:label点击坐标转图片实际像素坐标
            pixmap = self.label.pixmap()
            label_w = self.label.width()
            label_h = self.label.height()
            pixmap_w = pixmap.width()
            pixmap_h = pixmap.height()
            scale_x = pixmap_w / label_w
            scale_y = pixmap_h / label_h
            click_x = event.x() * scale_x
            click_y = event.y() * scale_y
            # 碰撞检测:判断点击落在哪个bbox里
            self.selected_bbox = None
            # 倒序遍历,优先选中最上层的bbox(如果有重叠)
            for bbox in reversed(self.current_bboxes):
                x1, y1 = bbox.x, bbox.y
                x2, y2 = bbox.x + bbox.w, bbox.y + bbox.h
                if x1 <= click_x <= x2 and y1 <= click_y <= y2:
                    self.selected_bbox = bbox
                    break
            # 重绘当前图片,高亮选中的bbox
            self.draw_current_image()
            return True
        return super().eventFilter(watched, event)

    def convert_cv_image_to_qt(self, cv_img):
        rgb_image = cv2.cvtColor(cv_img, cv2.COLOR_BGR2RGB)
        h, w, ch = rgb_image.shape
        bytes_per_line = ch * w
        convert_to_Qt_format = QtGui.QImage(rgb_image.data, w, h, bytes_per_line, QtGui.QImage.Format_RGB888)
        return QtGui.QPixmap.fromImage(convert_to_Qt_format)

    def draw_current_image(self):
        if len(self.current_bboxes) == 0:
            return
        image_path = self.current_bboxes[0].image_path
        image = cv2.imread(image_path)
        # 先画所有未选中的bbox,用红色
        for bbox in self.current_bboxes:
            if bbox != self.selected_bbox:
                cv2.rectangle(image, (bbox.x, bbox.y), (bbox.x+bbox.w, bbox.y+bbox.h), (0,0,255), 2)
                # 如果有名称,显示名称
                if bbox.name:
                    cv2.putText(image, bbox.name, (bbox.x, bbox.y-10), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (0,0,255), 2)
        # 再画选中的bbox,用绿色,层级最高
        if self.selected_bbox:
            bbox = self.selected_bbox
            cv2.rectangle(image, (bbox.x, bbox.y), (bbox.x+bbox.w, bbox.y+bbox.h), (0,255,0), 2)
            text = bbox.name if bbox.name else "未命名"
            cv2.putText(image, text, (bbox.x, bbox.y-10), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (0,255,0), 2)
        qimage = self.convert_cv_image_to_qt(image)
        self.label.setPixmap(qimage)
        self.label.show()

    def on_click_upload(self):
        dialog = QFileDialog()
        csv_file = dialog.getOpenFileName(None, "Import CSV", "", "CSV data files (*.csv)")
        if not csv_file[0]:
            return
        try:
            with open(csv_file[0], encoding='utf-8') as fp:
                reader = csv.reader(fp, delimiter=',')
                self.image_groups.clear()
                for r in reader:
                    if len(r) <5:
                        continue
                    row = Row(*r[:5])
                    self.image_groups[row.image_path].append(row)
        except PermissionError:
            print("你没有权限打开该文件")
            return
        if len(self.image_groups) ==0:
            print("文件为空,请选择其他文件")
            return
        self.image_list = list(self.image_groups.keys())
        self.current_img_index = 0
        self.current_bboxes = self.image_groups[self.image_list[self.current_img_index]]
        self.selected_bbox = None
        self.draw_current_image()

    def next_image(self, offset=1):
        if len(self.image_list) ==0:
            return
        self.current_img_index = (self.current_img_index + offset) % len(self.image_list)
        self.current_bboxes = self.image_groups[self.image_list[self.current_img_index]]
        self.selected_bbox = None
        self.draw_current_image()

    def on_click_next(self):
        self.next_image(offset=1)

    def on_click_previous(self):
        self.next_image(offset=-1)

    def on_click_name(self):
        # 点击命名按钮的逻辑
        if not self.selected_bbox:
            print("请先选中一个bounding box")
            return
        name, ok = QInputDialog.getText(self, "输入名称", "请输入目标对象名称:")
        if ok and name.strip():
            self.selected_bbox.name = name.strip()
            self.draw_current_image()

def execute_pipeline():
    app = QApplication(sys.argv)
    annotationGui = mainProgram()
    annotationGui.show()
    annotationGui.all_callbacks()
    sys.exit(app.exec_())

if __name__ == "__main__":
    execute_pipeline()

注意事项

  • 代码中的NameBtn需要替换为你自己UI文件中命名按钮的实际对象名
  • 导入的CSV文件每行格式仍为图片路径,x,y,w,h即可,名称会存在程序运行内存中,如需持久化存储可自行添加导出CSV的逻辑
  • 若bbox存在重叠,点击重叠区域会优先选中后绘制的bbox(即列表中靠后的bbox)

内容的提问来源于stack exchange,提问作者iamkk

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.06 07:48:03