You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Mediapipe中为RealSense流设置running_mode=VIDEO/LIVE_STREAM?

切换MediaPipe HandLandmarker至VIDEO或LIVE_STREAM模式实现更优跟踪

核心修改要点

MediaPipe的VIDEO和LIVE_STREAM模式依赖帧时间戳实现连续跟踪,相比IMAGE模式能减少重复检测、提升跟踪连贯性。以下是针对两种模式的具体实现方案:


方案一:切换至VIDEO模式

VIDEO模式适用于固定帧率的实时流(如RealSense的30FPS流),需要为每一帧传递微秒级时间戳,并使用detect_for_video()方法替代detect()。

修改后的完整代码:

import cv2
import numpy as np
import pyrealsense2 as rs
import mediapipe as mp
from mediapipe.tasks import python
from mediapipe.tasks.python import vision
from mediapipe import solutions
from mediapipe.framework.formats import landmark_pb2

def draw_landmarks_on_image(rgb_image, detection_result):
    hand_landmarks_list = detection_result.hand_landmarks
    annotated_image = np.copy(rgb_image)
    
    for idx in range(len(hand_landmarks_list)):
        hand_landmarks = hand_landmarks_list[idx]
        hand_landmarks_proto = landmark_pb2.NormalizedLandmarkList()
        hand_landmarks_proto.landmark.extend([
            landmark_pb2.NormalizedLandmark(x=landmark.x, y=landmark.y, z=landmark.z) for landmark in hand_landmarks
        ])
        solutions.drawing_utils.draw_landmarks(
            annotated_image,
            hand_landmarks_proto,
            solutions.hands.HAND_CONNECTIONS,
            solutions.drawing_styles.get_default_hand_landmarks_style(),
            solutions.drawing_styles.get_default_hand_connections_style())
    return annotated_image

def hand_detection_realsense_video():
    # 替换为你的模型路径
    model_path_full = "path/to/your/hand_landmarker.task"
    
    VisionRunningMode = mp.tasks.vision.RunningMode
    options = vision.HandLandmarkerOptions(
        base_options=python.BaseOptions(model_asset_path=model_path_full),
        running_mode=VisionRunningMode.VIDEO,  # 切换为VIDEO模式
        num_hands=2,
        min_hand_detection_confidence=0.5,
        min_hand_presence_confidence=0.5,
        min_tracking_confidence=0.5
    )
    detector = vision.HandLandmarker.create_from_options(options)

    # RealSense流配置
    pipeline = rs.pipeline()
    config = rs.config()
    pipeline_wrapper = rs.pipeline_wrapper(pipeline)
    pipeline_profile = config.resolve(pipeline_wrapper)
    device = pipeline_profile.get_device()
    device_product_line = str(device.get_info(rs.camera_info.product_line))

    config.enable_stream(rs.stream.depth, 640, 480, rs.format.z16, 30)
    config.enable_stream(rs.stream.color, 640, 480, rs.format.bgr8, 30)

    pipeline.start(config)
    fps = 30
    frame_timestamp = 0  # 初始时间戳(微秒)

    try:
        while True:
            frames = pipeline.wait_for_frames()
            depth_frame = frames.get_depth_frame()
            color_frame = frames.get_color_frame()
            if not depth_frame or not color_frame:
                continue

            color_image = np.asanyarray(color_frame.get_data())
            mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=color_image)
            
            # 计算当前帧时间戳:每帧间隔1e6/fps微秒
            frame_timestamp += int(1e6 / fps)
            # 使用VIDEO模式检测方法
            detection_result = detector.detect_for_video(mp_image, frame_timestamp)
            
            annotated_image = draw_landmarks_on_image(mp_image.numpy_view(), detection_result)
            images = np.hstack((color_image, annotated_image))
            cv2.namedWindow('RealSense', cv2.WINDOW_AUTOSIZE)
            cv2.imshow('RealSense', images)
            
            if cv2.waitKey(1) & 0xFF == ord('q'):
                break
    finally:
        pipeline.stop()
        detector.close()  # 释放检测器资源

if __name__ == "__main__":
    hand_detection_realsense_video()

方案二:切换至LIVE_STREAM模式

LIVE_STREAM模式适用于低延迟实时流场景,采用异步检测+回调函数处理结果,需要:

  1. 定义结果回调函数
  2. 使用detect_async()方法传递帧和时间戳
  3. 保证时间戳严格递增

修改后的完整代码:

import cv2
import numpy as np
import pyrealsense2 as rs
import mediapipe as mp
from mediapipe.tasks import python
from mediapipe.tasks.python import vision
from mediapipe import solutions
from mediapipe.framework.formats import landmark_pb2
import threading

# 线程安全传递检测结果
result_lock = threading.Lock()
latest_detection_result = None

def draw_landmarks_on_image(rgb_image, detection_result):
    if detection_result is None:
        return rgb_image
    hand_landmarks_list = detection_result.hand_landmarks
    annotated_image = np.copy(rgb_image)
    
    for idx in range(len(hand_landmarks_list)):
        hand_landmarks = hand_landmarks_list[idx]
        hand_landmarks_proto = landmark_pb2.NormalizedLandmarkList()
        hand_landmarks_proto.landmark.extend([
            landmark_pb2.NormalizedLandmark(x=landmark.x, y=landmark.y, z=landmark.z) for landmark in hand_landmarks
        ])
        solutions.drawing_utils.draw_landmarks(
            annotated_image,
            hand_landmarks_proto,
            solutions.hands.HAND_CONNECTIONS,
            solutions.drawing_styles.get_default_hand_landmarks_style(),
            solutions.drawing_styles.get_default_hand_connections_style())
    return annotated_image

# LIVE_STREAM模式结果回调函数
def result_callback(result: vision.HandLandmarkerResult, output_image: mp.Image, timestamp_ms: int):
    global latest_detection_result
    with result_lock:
        latest_detection_result = result

def hand_detection_realsense_live():
    # 替换为你的模型路径
    model_path_full = "path/to/your/hand_landmarker.task"
    
    VisionRunningMode = mp.tasks.vision.RunningMode
    options = vision.HandLandmarkerOptions(
        base_options=python.BaseOptions(model_asset_path=model_path_full),
        running_mode=VisionRunningMode.LIVE_STREAM,  # 切换为LIVE_STREAM模式
        result_callback=result_callback,  # 设置回调函数
        num_hands=2,
        min_hand_detection_confidence=0.5,
        min_hand_presence_confidence=0.5,
        min_tracking_confidence=0.5
    )
    detector = vision.HandLandmarker.create_from_options(options)

    # RealSense流配置
    pipeline = rs.pipeline()
    config = rs.config()
    pipeline_wrapper = rs.pipeline_wrapper(pipeline)
    pipeline_profile = config.resolve(pipeline_wrapper)
    device = pipeline_profile.get_device()
    device_product_line = str(device.get_info(rs.camera_info.product_line))

    config.enable_stream(rs.stream.depth, 640, 480, rs.format.z16, 30)
    config.enable_stream(rs.stream.color, 640, 480, rs.format.bgr8, 30)

    pipeline.start(config)
    fps = 30
    frame_timestamp = 0  # 微秒级时间戳

    try:
        while True:
            frames = pipeline.wait_for_frames()
            depth_frame = frames.get_depth_frame()
            color_frame = frames.get_color_frame()
            if not depth_frame or not color_frame:
                continue

            color_image = np.asanyarray(color_frame.get_data())
            mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=color_image)
            
            frame_timestamp += int(1e6 / fps)
            # 异步提交检测任务
            detector.detect_async(mp_image, frame_timestamp)
            
            # 获取最新检测结果并绘制
            with result_lock:
                current_result = latest_detection_result
            annotated_image = draw_landmarks_on_image(mp_image.numpy_view(), current_result)
            
            images = np.hstack((color_image, annotated_image))
            cv2.namedWindow('RealSense Live', cv2.WINDOW_AUTOSIZE)
            cv2.imshow('RealSense Live', images)
            
            if cv2.waitKey(1) & 0xFF == ord('q'):
                break
    finally:
        pipeline.stop()
        detector.close()  # 必须释放检测器资源

if __name__ == "__main__":
    hand_detection_realsense_live()

关键注意事项

  • 时间戳要求:两种模式都需要传递严格递增的微秒级时间戳,MediaPipe依靠时间戳判断帧顺序,避免跟踪混乱。
  • 资源释放:使用完检测器后必须调用detector.close(),防止内存泄漏。
  • 回调线程安全:LIVE_STREAM模式的回调在单独线程执行,需用锁保证结果访问的线程安全。

内容的提问来源于stack exchange,提问作者Rémi.T

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.07 19:50:33