基于Python的非均匀/不完整目标检测及视频跟踪实现咨询
非均匀、不完整目标检测+视频跟踪技术方案及Python实现
针对非均匀(纹理/灰度分布不一致)、不完整(存在遮挡、边缘残缺、仅局部可见)的粒子类目标,核心思路是避开依赖完整轮廓、全局特征的常规检测跟踪逻辑,用局部分割+局部特征匹配的路线降低漏检、跟丢概率。
核心技术方案选型
- 检测阶段:采用双分支互补检测结构
- 轻量实例分割分支:用MobileSAM抓取有明确语义的残缺、非均匀目标,不需要目标完整出现在视野内,只要局部特征匹配即可检出
- 自适应局部阈值分割分支:通过邻域高斯阈值分割+连通域分析,补充分割模型漏检的弱特征、灰度分布不均的小粒子,避免全局阈值对非均匀目标失效的问题
- 两个分支的结果通过IOU做去重,过滤噪点
- 跟踪阶段:采用卡尔曼运动预测+ORB局部特征匹配+匈牙利匹配的关联逻辑
- ORB是局部点特征,哪怕目标仅露出1/3区域,只要有可匹配的角点就能完成跨帧关联,比光流、全局表观特征更适配不完整目标
- 卡尔曼滤波预测目标运动轨迹,补全短时间遮挡、完全消失帧的目标位置
- 轨迹维护逻辑:设置目标消失缓冲帧数,超过阈值未匹配到检测结果再删除轨迹,避免短暂遮挡导致ID频繁切换
Python具体实现
依赖安装
pip install opencv-python numpy ultralytics filterpy scipy
1. 检测模块代码
import cv2 import numpy as np from ultralytics import SAM from filterpy.kalman import KalmanFilter from scipy.optimize import linear_sum_assignment # 加载轻量分割模型,首次运行会自动拉取对应权重 sam = SAM("mobile_sam.pt") def calc_iou(box1, box2): x1 = max(box1[0], box2[0]) y1 = max(box1[1], box2[1]) x2 = min(box1[2], box2[2]) y2 = min(box1[3], box2[3]) inter_area = max(0, x2 - x1) * max(0, y2 - y1) area1 = (box1[2] - box1[0]) * (box1[3] - box1[1]) area2 = (box2[2] - box2[0]) * (box2[3] - box2[1]) return inter_area / (area1 + area2 - inter_area + 1e-6) def detect_targets(frame): h, w = frame.shape[:2] detect_boxes = [] detect_masks = [] # 分支1:MobileSAM分割检测 sam_res = sam(frame, verbose=False)[0] if sam_res.masks is not None: for mask, box in zip(sam_res.masks.data.cpu().numpy(), sam_res.boxes.xyxy.cpu().numpy()): x1, y1, x2, y2 = box.astype(int) # 过滤过小噪点,阈值根据实际粒子尺寸调整 if (x2 - x1) * (y2 - y1) < 10: continue detect_boxes.append([x1, y1, x2, y2]) detect_masks.append(mask.astype(np.uint8)) # 分支2:自适应局部阈值分割补漏 gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY) # 邻域窗口大小设为目标平均直径的1.2~1.5倍 local_bin = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY_INV, 31, 5) # 形态学去噪 kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (3,3)) local_bin = cv2.morphologyEx(local_bin, cv2.MORPH_OPEN, kernel) # 连通域分析提取目标 n_labels, labels, stats, _ = cv2.connectedComponentsWithStats(local_bin, connectivity=8) for i in range(1, n_labels): x, y, bw, bh, area = stats[i] # 过滤噪点和过大的背景块 if area < 10 or area > 0.3 * h * w: continue x1, y1, x2, y2 = x, y, x + bw, y + bh # 和已有检测结果去重 repeat = False for db in detect_boxes: if calc_iou([x1,y1,x2,y2], db) > 0.3: repeat = True break if not repeat: detect_boxes.append([x1, y1, x2, y2]) detect_masks.append((labels == i).astype(np.uint8)) return detect_boxes, detect_masks
2. 跟踪模块代码
# 初始化ORB局部特征提取器 orb = cv2.ORB_create(nfeatures=200) # 轨迹存储字典 tracks = dict() track_id = 0 max_disappear_frames = 10 # 目标最大允许消失帧数 def init_kalman(): # 状态量:[x,y,w,h,vx,vy,vw,vh],观测量:[x,y,w,h] kf = KalmanFilter(dim_x=8, dim_z=4) kf.F = np.array([[1,0,0,0,1,0,0,0], [0,1,0,0,0,1,0,0], [0,0,1,0,0,0,1,0], [0,0,0,1,0,0,0,1], [0,0,0,0,1,0,0,0], [0,0,0,0,0,1,0,0], [0,0,0,0,0,0,1,0], [0,0,0,0,0,0,0,1]], np.float32) kf.H = np.array([[1,0,0,0,0,0,0,0], [0,1,0,0,0,0,0,0], [0,0,1,0,0,0,0,0], [0,0,0,1,0,0,0,0]], np.float32) kf.R[2:,2:] *= 10 kf.P[4:,4:] *= 1000 kf.P *= 10 kf.Q[-1,-1] *= 0.01 kf.Q[4:,4:] *= 0.01 return kf def update_tracks(detect_boxes, frame): global track_id # 所有轨迹先做卡尔曼预测 for tid in tracks: tracks[tid]['kf'].predict() if len(tracks) > 0 and len(detect_boxes) > 0: track_list = list(tracks.items()) cost_matrix = np.zeros((len(track_list), len(detect_boxes)), np.float32) # 计算匹配代价:0.7权重位置IOU代价 + 0.3权重特征匹配代价 for i, (tid, t_data) in enumerate(track_list): px, py, pw, ph = t_data['kf'].x[:4].flatten() pred_box = [px, py, px+pw, py+ph] for j, d_box in enumerate(detect_boxes): iou_cost = 1 - calc_iou(pred_box, d_box) # 计算ORB特征匹配度 x1,y1,x2,y2 = d_box roi = frame[y1:y2, x1:x2] _, des2 = orb.detectAndCompute(roi, None) feat_cost = 1.0 if des2 is not None and t_data['des'] is not None: bf = cv2.BFMatcher(cv2.NORM_HAMMING, crossCheck=True) matches = bf.match(t_data['des'], des2) feat_cost = 1 - len(matches) / max(len(t_data['kp']), 1) cost_matrix[i,j] = 0.7 * iou_cost + 0.3 * feat_cost # 匈牙利匹配 row_ind, col_ind = linear_sum_assignment(cost_matrix) matched_tids = set() matched_det_ids = set() for r, c in zip(row_ind, col_ind): if cost_matrix[r,c] < 0.7: tid = track_list[r][0] x1,y1,x2,y2 = detect_boxes[c] # 更新卡尔曼状态 tracks[tid]['kf'].update(np.array([x1,y1,x2-x1,y2-y1], np.float32)) # 更新特征库 roi = frame[y1:y2, x1:x2] kp, des = orb.detectAndCompute(roi, None) if des is not None: tracks[tid]['kp'] = kp tracks[tid]['des'] = des tracks[tid]['disappear'] = 0 tracks[tid]['box'] = [x1,y1,x2,y2] matched_tids.add(tid) matched_det_ids.add(c) # 未匹配轨迹计数+1 for tid in tracks: if tid not in matched_tids: tracks[tid]['disappear'] += 1 # 清理消失过久的轨迹 [tracks.pop(tid) for tid in list(tracks.keys()) if tracks[tid]['disappear'] > max_disappear_frames] # 未匹配检测新建轨迹 for j in range(len(detect_boxes)): if j not in matched_det_ids: x1,y1,x2,y2 = detect_boxes[j] kf = init_kalman() kf.x[:4] = np.array([x1,y1,x2-x1,y2-y1], np.float32).reshape((4,1)) roi = frame[y1:y2, x1:x2] kp, des = orb.detectAndCompute(roi, None) tracks[track_id] = { 'kf':kf, 'kp':kp, 'des':des, 'box':[x1,y1,x2,y2], 'disappear':0 } track_id +=1 else: # 无现有轨迹时所有检测新建轨迹 for d_box in detect_boxes: x1,y1,x2,y2 = d_box kf = init_kalman() kf.x[:4] = np.array([x1,y1,x2-x1,y2-y1], np.float32).reshape((4,1)) roi = frame[y1:y2, x1:x2] kp, des = orb.detectAndCompute(roi, None) tracks[track_id] = { 'kf':kf, 'kp':kp, 'des':des, 'box':[x1,y1,x2,y2], 'disappear':0 } track_id +=1 # 返回当前帧有效跟踪结果 return [(tid, t['box']) for tid, t in tracks.items() if t['disappear'] == 0]
3. 视频处理主逻辑
def process_video(input_path, output_path=None): cap = cv2.VideoCapture(input_path) writer = None while cap.isOpened(): ret, frame = cap.read() if not ret: break boxes, _ = detect_targets(frame) track_res = update_tracks(boxes, frame) # 绘制结果 for tid, (x1,y1,x2,y2) in track_res: cv2.rectangle(frame, (x1,y1), (x2,y2), (0,255,0), 2) cv2.putText(frame, f"ID:{tid}", (x1, y1-5), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0,255,0), 2) # 保存或实时显示 if output_path: if writer is None: h,w = frame.shape[:2] writer = cv2.VideoWriter(output_path, cv2.VideoWriter_fourcc(*'mp4v'), 30, (w,h)) writer.write(frame) else: cv2.imshow("track_res", frame) if cv2.waitKey(1) == ord('q'): break cap.release() if writer: writer.release() cv2.destroyAllWindows() if __name__ == "__main__": process_video("your_test_video.mp4", "track_output.mp4")
调参说明
- 检测阶段的面积阈值、自适应阈值窗口大小,需要根据视频中目标的实际尺寸调整,窗口大小设为目标平均直径的1.2~1.5倍时分割效果最优
- 如果是特定类别的粒子目标,可以用少量标注的局部目标框(不需要标完整目标)微调YOLOv8-seg替换通用MobileSAM,检测精度会明显提升
- 目标运动速度快、遮挡少的场景可以把
max_disappear_frames设为35,遮挡多、运动慢的场景可以设为1520
内容的提问来源于stack exchange,提问作者BhargavN
相关产品推荐
相关产品推荐

