You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为CrowdCounting-P2PNet添加掩码区域目标检测时运行报错求助

特定区域内密集人群计数问题排查与修复

我基于P2PNet密集人群识别源码,尝试在detect.py中添加自定义多边形区域掩码实现仅统计特定区域内的人数,但运行代码时出现报错。以下是原代码及修复方案:


原报错代码

import argparse
import datetime
import random
import time
from pathlib import Path

import torch
import torchvision.transforms as standard_transforms
import numpy as np

from PIL import Image
import cv2
from crowd_datasets import build_dataset
from engine import *
from models import build_model
import os
import warnings
warnings.filterwarnings('ignore')

def get_args_parser():
    parser = argparse.ArgumentParser('Set parameters for P2PNet evaluation', add_help=False)

    # * Backbone
    parser.add_argument('--backbone', default='vgg16_bn', type=str,
                        help="name of the convolutional backbone to use")

    parser.add_argument('--row', default=2, type=int,
                        help="row number of anchor points")
    parser.add_argument('--line', default=2, type=int,
                        help="line number of anchor points")

    parser.add_argument('--output_dir', default='',
                        help='path where to save')
    parser.add_argument('--weight_path', default='',
                        help='path where the trained weights saved')

    parser.add_argument('--gpu_id', default=0, type=int, help='the gpu used for evaluation')

    return parser

def main(args, debug=False):
    os.environ["CUDA_VISIBLE_DEVICES"] = '{}'.format(args.gpu_id)

    print(args)
    device = torch.device('cuda')
    # get the P2PNet
    model = build_model(args)
    # move to GPU
    model.to(device)
    # load trained model
    if args.weight_path is not None:
        checkpoint = torch.load(args.weight_path, map_location='cpu')
        model.load_state_dict(checkpoint['model'])
    # convert to eval mode
    model.eval()
    # create the pre-processing transform
    transform = standard_transforms.Compose([
        standard_transforms.ToTensor(), 
        standard_transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
    ])

    hl1 = 203 / 907  # 监测区域高度距离图片顶部比例
    wl1 = 303 / 1603  # 监测区域高度距离图片左部比例
    hl2 = 81 / 907  # 监测区域高度距离图片顶部比例
    wl2 = 872 / 1603  # 监测区域高度距离图片左部比例
    hl3 = 415 / 907  # 监测区域高度距离图片顶部比例
    wl3 = 1037 / 1603  # 监测区域高度距离图片左部比例
    hl4 = 704 / 907  # 监测区域高度距离图片顶部比例
    wl4 = 770 / 1603  # 监测区域高度距离图片左部比例
    hl5 = 685 / 907  # 监测区域高度距离图片顶部比例
    wl5 = 307 / 1603  # 监测区域高度距离图片左部比例

    # set your image path here
    img_paths = r"E:\vedio\test"
    output_dir = r"E:\vedio\test"
    for img in os.listdir(img_paths):
        img_path = img_paths+"/"+img
        print(f"{img_path}开始识别")
        # load the images
        img_raw = Image.open(img_path).convert('RGB')
        # round the size
        width, height = img_raw.size
        new_width = width // 128 * 128
        new_height = height // 128 * 128
        img_raw = img_raw.resize((new_width, new_height), Image.ANTIALIAS)
        # pre-processing
        im0 = transform(img_raw)

        mask = np.zeros([im0.shape[1], im0.shape[2]], dtype=np.uint8)
        pts = np.array([[int(im0.shape[2] * wl1), int(im0.shape[1] * hl1)],  # pts1
                        [int(im0.shape[2] * wl2), int(im0.shape[1] * hl2)],  # pts2
                        [int(im0.shape[2] * wl3), int(im0.shape[1] * hl3)],  # pts3
                        [int(im0.shape[2] * wl4), int(im0.shape[1] * hl4)],  # pts4
                        [int(im0.shape[2] * wl5), int(im0.shape[1] * hl5)]], np.int32)
        mask = cv2.fillPoly(mask, [pts], (255, 255, 255))
        print(mask)
        img = im0.transpose((1, 2, 0))
        img = cv2.add(img, np.zeros(np.shape(img), dtype=np.uint8), mask=mask)
        img = img.transpose((2, 0, 1))
        cv2.polylines(im0, [pts], True, (255, 255, 0), 3)

        print(img)
        samples = torch.Tensor(img).unsqueeze(0)
        samples = samples.to(device)

        # run inference
        outputs = model(samples)
        outputs_scores = torch.nn.functional.softmax(outputs['pred_logits'], -1)[:, :, 1][0]

        outputs_points = outputs['pred_points'][0]

        threshold = 0.5
        # filter the predictions
        points = outputs_points[outputs_scores > threshold].detach().cpu().numpy().tolist()
        predict_cnt = int((outputs_scores > threshold).sum())

        outputs_scores = torch.nn.functional.softmax(outputs['pred_logits'], -1)[:, :, 1][0]

        outputs_points = outputs['pred_points'][0]
        # draw the predictions
        size = 2
        img_to_draw = cv2.cvtColor(np.array(img_raw), cv2.COLOR_RGB2BGR)
        for p in points:
            img_to_draw = cv2.circle(img_to_draw, (int(p[0]), int(p[1])), size, (0, 0, 255), -1)
        # save the visualized image
        cv2.imwrite(os.path.join(output_dir, '{}-pred{}.jpg'.format(img_path.split(".")[0],predict_cnt)), img_to_draw)

if __name__ == '__main__':
    parser = argparse.ArgumentParser('P2PNet evaluation script', parents=[get_args_parser()])
    args = parser.parse_args()
    main(args)

问题分析

  1. 数据类型不兼容:归一化后的Tensor图像(值范围[-1,1])无法直接与OpenCV的uint8掩码进行运算,会导致数据异常。
  2. 输入图像修改错误:修改归一化后的输入图像会破坏模型要求的输入分布,影响预测精度。
  3. 绘制函数参数错误:cv2.polylines无法直接处理Tensor类型的图像,需转为numpy数组。

修复后代码

import argparse
import datetime
import random
import time
from pathlib import Path

import torch
import torchvision.transforms as standard_transforms
import numpy as np

from PIL import Image
import cv2
from crowd_datasets import build_dataset
from engine import *
from models import build_model
import os
import warnings
warnings.filterwarnings('ignore')

def get_args_parser():
    parser = argparse.ArgumentParser('Set parameters for P2PNet evaluation', add_help=False)

    # * Backbone
    parser.add_argument('--backbone', default='vgg16_bn', type=str,
                        help="name of the convolutional backbone to use")

    parser.add_argument('--row', default=2, type=int,
                        help="row number of anchor points")
    parser.add_argument('--line', default=2, type=int,
                        help="line number of anchor points")

    parser.add_argument('--output_dir', default='',
                        help='path where to save')
    parser.add_argument('--weight_path', default='',
                        help='path where the trained weights saved')

    parser.add_argument('--gpu_id', default=0, type=int, help='the gpu used for evaluation')

    return parser

def main(args, debug=False):
    os.environ["CUDA_VISIBLE_DEVICES"] = '{}'.format(args.gpu_id)

    print(args)
    device = torch.device('cuda')
    # get the P2PNet
    model = build_model(args)
    # move to GPU
    model.to(device)
    # load trained model
    if args.weight_path is not None:
        checkpoint = torch.load(args.weight_path, map_location='cpu')
        model.load_state_dict(checkpoint['model'])
    # convert to eval mode
    model.eval()
    # create the pre-processing transform
    transform = standard_transforms.Compose([
        standard_transforms.ToTensor(), 
        standard_transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
    ])

    # 监测区域点的比例(基于原始图像尺寸907x1603)
    hl1 = 203 / 907  
    wl1 = 303 / 1603  
    hl2 = 81 / 907  
    wl2 = 872 / 1603  
    hl3 = 415 / 907  
    wl3 = 1037 / 1603  
    hl4 = 704 / 907  
    wl4 = 770 / 1603  
    hl5 = 685 / 907  
    wl5 = 307 / 1603  

    # set your image path here
    img_paths = r"E:\vedio\test"
    output_dir = r"E:\vedio\test"
    for img in os.listdir(img_paths):
        img_path = os.path.join(img_paths, img)
        print(f"{img_path}开始识别")
        # load the images
        img_raw = Image.open(img_path).convert('RGB')
        # round the size
        width, height = img_raw.size
        new_width = width // 128 * 128
        new_height = height // 128 * 128
        img_raw = img_raw.resize((new_width, new_height), Image.ANTIALIAS)
        # pre-processing
        im0 = transform(img_raw).unsqueeze(0).to(device)

        # 计算当前图像尺寸下的多边形区域点
        pts = np.array([
            [int(new_width * wl1), int(new_height * hl1)],
            [int(new_width * wl2), int(new_height * hl2)],
            [int(new_width * wl3), int(new_height * hl3)],
            [int(new_width * wl4), int(new_height * hl4)],
            [int(new_width * wl5), int(new_height * hl5)]
        ], np.int32)

        # run inference
        with torch.no_grad():
            outputs = model(im0)
        outputs_scores = torch.nn.functional.softmax(outputs['pred_logits'], -1)[:, :, 1][0]
        outputs_points = outputs['pred_points'][0]

        threshold = 0.5
        # 先过滤置信度符合要求的点
        valid_mask = outputs_scores > threshold
        points = outputs_points[valid_mask].detach().cpu().numpy()

        # 过滤位于多边形区域内的点
        in_area_points = []
        for p in points:
            # cv2.pointPolygonTest判断点是否在多边形内,返回值>0表示在内部
            if cv2.pointPolygonTest(pts, (p[0], p[1]), False) > 0:
                in_area_points.append(p)
        predict_cnt = len(in_area_points)

        # 绘制结果:先画多边形区域,再画区域内的检测点
        img_to_draw = cv2.cvtColor(np.array(img_raw), cv2.COLOR_RGB2BGR)
        # 绘制多边形区域
        cv2.polylines(img_to_draw, [pts], True, (255, 255, 0), 3)
        # 绘制区域内的检测点
        size = 2
        for p in in_area_points:
            img_to_draw = cv2.circle(img_to_draw, (int(p[0]), int(p[1])), size, (0, 0, 255), -1)
        
        # 保存可视化图像
        save_name = f"{os.path.splitext(img)[0]}-pred{predict_cnt}.jpg"
        cv2.imwrite(os.path.join(output_dir, save_name), img_to_draw)

if __name__ == '__main__':
    parser = argparse.ArgumentParser('P2PNet evaluation script', parents=[get_args_parser()])
    args = parser.parse_args()
    main(args)

关键修改说明

  1. 移除输入图像修改逻辑:不再对归一化后的输入图像应用掩码,避免破坏模型输入要求。
  2. 后处理阶段过滤点:先通过置信度过滤有效检测点,再用cv2.pointPolygonTest判断点是否在目标多边形区域内,仅统计区域内的点数。
  3. 修复图像绘制问题:直接在原始图像(转为OpenCV格式)上绘制多边形和检测点,避免Tensor与numpy格式不兼容的问题。
  4. 路径拼接优化:使用os.path.join处理路径,避免手动拼接出现的分隔符问题。

内容的提问来源于stack exchange,提问作者starhu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.07 02:27:04