You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将图像中每个分割实例导出为独立图像?

基于Detectron2/DETR的网球实例分割与独立贴纸导出脚本(Colab可用)

方案一:使用Detectron2实现

1. 安装依赖

在Colab代码单元格中运行以下命令:

!pip install pyyaml==5.1
!pip install detectron2 -f https://dl.fbaipublicfiles.com/detectron2/wheels/cu111/torch1.9/index.html

2. 完整运行脚本

import os
import cv2
import numpy as np
from PIL import Image
from detectron2 import model_zoo
from detectron2.engine import DefaultPredictor
from detectron2.config import get_cfg

# 配置实例分割模型
cfg = get_cfg()
cfg.merge_from_file(model_zoo.get_config_file("COCO-InstanceSegmentation/mask_rcnn_R_50_FPN_3x.yaml"))
cfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.5  # 可根据需求调整置信度阈值
cfg.MODEL.WEIGHTS = model_zoo.get_checkpoint_url("COCO-InstanceSegmentation/mask_rcnn_R_50_FPN_3x.yaml")
predictor = DefaultPredictor(cfg)

def export_instance_stickers(image_path, save_dir="./instance_stickers"):
    # 创建保存目录
    os.makedirs(save_dir, exist_ok=True)
    
    # 读取输入图像
    img = cv2.imread(image_path)
    if img is None:
        print(f"无法读取图像:{image_path}")
        return
    
    # 执行模型预测
    outputs = predictor(img)
    instances = outputs["instances"].to("cpu")
    
    # COCO数据集类别名称映射
    class_names = ["person", "bicycle", "car", "motorcycle", "airplane", "bus", "train", "truck", "boat", "traffic light",
                   "fire hydrant", "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", "sheep", "cow",
                   "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella", "handbag", "tie", "suitcase", "frisbee",
                   "skis", "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", "skateboard", "surfboard",
                   "tennis racket", "bottle", "wine glass", "cup", "fork", "knife", "spoon", "bowl", "banana", "apple",
                   "sandwich", "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair", "couch",
                   "potted plant", "bed", "dining table", "toilet", "tv", "laptop", "mouse", "remote", "keyboard", "cell phone",
                   "microwave", "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", "scissors", "teddy bear",
                   "hair drier", "toothbrush"]
    
    # 遍历每个检测到的实例
    for idx in range(len(instances)):
        # 提取实例掩码、类别和置信度
        mask = instances.pred_masks[idx].numpy()
        class_idx = instances.pred_classes[idx].item()
        class_name = class_names[class_idx]
        score = instances.scores[idx].item()
        
        # 将原图转为RGBA格式,非掩码区域设为透明
        rgba_img = cv2.cvtColor(img, cv2.COLOR_BGR2RGBA)
        rgba_img[~mask] = [0, 0, 0, 0]
        
        # 裁剪到实例最小边界框,去除多余空白
        y_indices, x_indices = np.where(mask)
        x_min, x_max = x_indices.min(), x_indices.max()
        y_min, y_max = y_indices.min(), y_indices.max()
        cropped_rgba = rgba_img[y_min:y_max+1, x_min:x_max+1]
        
        # 生成保存路径并导出图像
        save_name = f"{class_name}_instance_{idx}_score_{score:.2f}.png"
        save_path = os.path.join(save_dir, save_name)
        Image.fromarray(cropped_rgba).save(save_path)
        print(f"已导出:{save_path}")

# --------------------------
# 使用示例
# --------------------------
# 1. 先通过Colab左侧文件面板上传网球比赛图像(如tennis_match.jpg)
# 2. 调用函数导出实例贴纸
export_instance_stickers("tennis_match.jpg")

方案二:使用DETR实现(可选)

如果偏好DETR模型,可使用以下脚本:

1. 安装依赖

!pip install transformers pillow torch

2. 完整运行脚本

import os
from PIL import Image
from transformers import DetrImageProcessor, DetrForSegmentation
import torch

# 加载DETR全景分割模型
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50-panoptic")
model = DetrForSegmentation.from_pretrained("facebook/detr-resnet-50-panoptic")

def export_detr_stickers(image_path, save_dir="./detr_instance_stickers"):
    os.makedirs(save_dir, exist_ok=True)
    img = Image.open(image_path).convert("RGB")
    
    # 预处理图像并执行预测
    inputs = processor(images=img, return_tensors="pt")
    outputs = model(**inputs)
    
    # 后处理获取实例分割结果
    result = processor.post_process_panoptic(outputs, target_sizes=[img.size[::-1]])[0]
    panoptic_seg = result["segmentation"]
    segments_info = result["segments_info"]
    
    # 遍历每个物体实例(跳过背景)
    for seg_info in segments_info:
        if seg_info["isthing"]:
            class_id = seg_info["category_id"]
            class_name = model.config.id2label[class_id]
            mask = panoptic_seg == seg_info["id"]
            
            # 创建透明背景的实例图像
            rgba_img = Image.new("RGBA", img.size, (0,0,0,0))
            img_rgba = img.convert("RGBA")
            new_data = []
            for i, pixel in enumerate(img_rgba.getdata()):
                new_data.append(pixel if mask.flatten()[i] else (0,0,0,0))
            rgba_img.putdata(new_data)
            
            # 裁剪到实例边界框
            bbox = seg_info["bbox"]
            cropped = rgba_img.crop((bbox[0], bbox[1], bbox[0]+bbox[2], bbox[1]+bbox[3]))
            
            # 保存图像
            save_name = f"{class_name}_instance_{seg_info['id']}.png"
            save_path = os.path.join(save_dir, save_name)
            cropped.save(save_path)
            print(f"已导出:{save_path}")

# --------------------------
# 使用示例
# --------------------------
export_detr_stickers("tennis_match.jpg")

注意事项

  • 运行前需在Colab中启用GPU:代码执行程序 → 更改运行时类型 → 硬件加速器选择GPU
  • 若需读取Google Drive中的图像,可先挂载Drive:
from google.colab import drive
drive.mount('/content/drive')
# 调用时使用Drive路径,如export_instance_stickers("/content/drive/MyDrive/tennis_match.jpg")

内容的提问来源于stack exchange,提问作者Matthew Gerges

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.09 23:35:22