如何在Mediapipe中为RealSense流设置running_mode=VIDEO/LIVE_STREAM?
切换MediaPipe HandLandmarker至VIDEO或LIVE_STREAM模式实现更优跟踪
核心修改要点
MediaPipe的VIDEO和LIVE_STREAM模式依赖帧时间戳实现连续跟踪,相比IMAGE模式能减少重复检测、提升跟踪连贯性。以下是针对两种模式的具体实现方案:
方案一:切换至VIDEO模式
VIDEO模式适用于固定帧率的实时流(如RealSense的30FPS流),需要为每一帧传递微秒级时间戳,并使用detect_for_video()方法替代detect()。
修改后的完整代码:
import cv2 import numpy as np import pyrealsense2 as rs import mediapipe as mp from mediapipe.tasks import python from mediapipe.tasks.python import vision from mediapipe import solutions from mediapipe.framework.formats import landmark_pb2 def draw_landmarks_on_image(rgb_image, detection_result): hand_landmarks_list = detection_result.hand_landmarks annotated_image = np.copy(rgb_image) for idx in range(len(hand_landmarks_list)): hand_landmarks = hand_landmarks_list[idx] hand_landmarks_proto = landmark_pb2.NormalizedLandmarkList() hand_landmarks_proto.landmark.extend([ landmark_pb2.NormalizedLandmark(x=landmark.x, y=landmark.y, z=landmark.z) for landmark in hand_landmarks ]) solutions.drawing_utils.draw_landmarks( annotated_image, hand_landmarks_proto, solutions.hands.HAND_CONNECTIONS, solutions.drawing_styles.get_default_hand_landmarks_style(), solutions.drawing_styles.get_default_hand_connections_style()) return annotated_image def hand_detection_realsense_video(): # 替换为你的模型路径 model_path_full = "path/to/your/hand_landmarker.task" VisionRunningMode = mp.tasks.vision.RunningMode options = vision.HandLandmarkerOptions( base_options=python.BaseOptions(model_asset_path=model_path_full), running_mode=VisionRunningMode.VIDEO, # 切换为VIDEO模式 num_hands=2, min_hand_detection_confidence=0.5, min_hand_presence_confidence=0.5, min_tracking_confidence=0.5 ) detector = vision.HandLandmarker.create_from_options(options) # RealSense流配置 pipeline = rs.pipeline() config = rs.config() pipeline_wrapper = rs.pipeline_wrapper(pipeline) pipeline_profile = config.resolve(pipeline_wrapper) device = pipeline_profile.get_device() device_product_line = str(device.get_info(rs.camera_info.product_line)) config.enable_stream(rs.stream.depth, 640, 480, rs.format.z16, 30) config.enable_stream(rs.stream.color, 640, 480, rs.format.bgr8, 30) pipeline.start(config) fps = 30 frame_timestamp = 0 # 初始时间戳(微秒) try: while True: frames = pipeline.wait_for_frames() depth_frame = frames.get_depth_frame() color_frame = frames.get_color_frame() if not depth_frame or not color_frame: continue color_image = np.asanyarray(color_frame.get_data()) mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=color_image) # 计算当前帧时间戳:每帧间隔1e6/fps微秒 frame_timestamp += int(1e6 / fps) # 使用VIDEO模式检测方法 detection_result = detector.detect_for_video(mp_image, frame_timestamp) annotated_image = draw_landmarks_on_image(mp_image.numpy_view(), detection_result) images = np.hstack((color_image, annotated_image)) cv2.namedWindow('RealSense', cv2.WINDOW_AUTOSIZE) cv2.imshow('RealSense', images) if cv2.waitKey(1) & 0xFF == ord('q'): break finally: pipeline.stop() detector.close() # 释放检测器资源 if __name__ == "__main__": hand_detection_realsense_video()
方案二:切换至LIVE_STREAM模式
LIVE_STREAM模式适用于低延迟实时流场景,采用异步检测+回调函数处理结果,需要:
- 定义结果回调函数
- 使用
detect_async()方法传递帧和时间戳 - 保证时间戳严格递增
修改后的完整代码:
import cv2 import numpy as np import pyrealsense2 as rs import mediapipe as mp from mediapipe.tasks import python from mediapipe.tasks.python import vision from mediapipe import solutions from mediapipe.framework.formats import landmark_pb2 import threading # 线程安全传递检测结果 result_lock = threading.Lock() latest_detection_result = None def draw_landmarks_on_image(rgb_image, detection_result): if detection_result is None: return rgb_image hand_landmarks_list = detection_result.hand_landmarks annotated_image = np.copy(rgb_image) for idx in range(len(hand_landmarks_list)): hand_landmarks = hand_landmarks_list[idx] hand_landmarks_proto = landmark_pb2.NormalizedLandmarkList() hand_landmarks_proto.landmark.extend([ landmark_pb2.NormalizedLandmark(x=landmark.x, y=landmark.y, z=landmark.z) for landmark in hand_landmarks ]) solutions.drawing_utils.draw_landmarks( annotated_image, hand_landmarks_proto, solutions.hands.HAND_CONNECTIONS, solutions.drawing_styles.get_default_hand_landmarks_style(), solutions.drawing_styles.get_default_hand_connections_style()) return annotated_image # LIVE_STREAM模式结果回调函数 def result_callback(result: vision.HandLandmarkerResult, output_image: mp.Image, timestamp_ms: int): global latest_detection_result with result_lock: latest_detection_result = result def hand_detection_realsense_live(): # 替换为你的模型路径 model_path_full = "path/to/your/hand_landmarker.task" VisionRunningMode = mp.tasks.vision.RunningMode options = vision.HandLandmarkerOptions( base_options=python.BaseOptions(model_asset_path=model_path_full), running_mode=VisionRunningMode.LIVE_STREAM, # 切换为LIVE_STREAM模式 result_callback=result_callback, # 设置回调函数 num_hands=2, min_hand_detection_confidence=0.5, min_hand_presence_confidence=0.5, min_tracking_confidence=0.5 ) detector = vision.HandLandmarker.create_from_options(options) # RealSense流配置 pipeline = rs.pipeline() config = rs.config() pipeline_wrapper = rs.pipeline_wrapper(pipeline) pipeline_profile = config.resolve(pipeline_wrapper) device = pipeline_profile.get_device() device_product_line = str(device.get_info(rs.camera_info.product_line)) config.enable_stream(rs.stream.depth, 640, 480, rs.format.z16, 30) config.enable_stream(rs.stream.color, 640, 480, rs.format.bgr8, 30) pipeline.start(config) fps = 30 frame_timestamp = 0 # 微秒级时间戳 try: while True: frames = pipeline.wait_for_frames() depth_frame = frames.get_depth_frame() color_frame = frames.get_color_frame() if not depth_frame or not color_frame: continue color_image = np.asanyarray(color_frame.get_data()) mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=color_image) frame_timestamp += int(1e6 / fps) # 异步提交检测任务 detector.detect_async(mp_image, frame_timestamp) # 获取最新检测结果并绘制 with result_lock: current_result = latest_detection_result annotated_image = draw_landmarks_on_image(mp_image.numpy_view(), current_result) images = np.hstack((color_image, annotated_image)) cv2.namedWindow('RealSense Live', cv2.WINDOW_AUTOSIZE) cv2.imshow('RealSense Live', images) if cv2.waitKey(1) & 0xFF == ord('q'): break finally: pipeline.stop() detector.close() # 必须释放检测器资源 if __name__ == "__main__": hand_detection_realsense_live()
关键注意事项
- 时间戳要求:两种模式都需要传递严格递增的微秒级时间戳,MediaPipe依靠时间戳判断帧顺序,避免跟踪混乱。
- 资源释放:使用完检测器后必须调用
detector.close(),防止内存泄漏。 - 回调线程安全:LIVE_STREAM模式的回调在单独线程执行,需用锁保证结果访问的线程安全。
内容的提问来源于stack exchange,提问作者Rémi.T
相关产品推荐
相关产品推荐

