如何为Intel Realsense D415的Python+TensorFlow目标识别代码添加测距功能
实现方案
你使用的是Intel RealSense深度相机,本身自带深度测量能力,只需要按以下步骤修改代码就能实现识别物体同时显示距离的效果:
必要修改点
- 新增深度流配置,同时添加深度-彩色帧对齐模块,保证两个画面的坐标完全匹配,避免测距位置偏移
- 读取帧数据时同步获取深度帧
- 计算识别框的中心坐标,调用RealSense内置方法获取该点的距离值
- 调用OpenCV的文字绘制接口将距离显示在识别框上方
修改后完整代码
import pyrealsense2 as rs import numpy as np import cv2 import tensorflow as tf # 配置深度和彩色流 pipeline = rs.pipeline() config = rs.config() # 原有彩色流配置保留 config.enable_stream(rs.stream.color, 1280, 720, rs.format.bgr8, 30) # ==========新增:开启深度流========== config.enable_stream(rs.stream.depth, 1280, 720, rs.format.z16, 30) # ==========新增:创建对齐模块,将深度帧对齐到彩色帧========== align_to = rs.stream.color align = rs.align(align_to) print("[INFO] Starting streaming...") pipeline.start(config) print("[INFO] Camera ready.") print("[INFO] Loading model...") PATH_TO_CKPT = "frozen_inference_graph_coco.pb" detection_graph = tf.Graph() with detection_graph.as_default(): od_graph_def = tf.compat.v1.GraphDef() with tf.compat.v1.gfile.GFile(PATH_TO_CKPT, 'rb') as fid: serialized_graph = fid.read() od_graph_def.ParseFromString(serialized_graph) tf.compat.v1.import_graph_def(od_graph_def, name='') sess = tf.compat.v1.Session(graph=detection_graph) image_tensor = detection_graph.get_tensor_by_name('image_tensor:0') detection_boxes = detection_graph.get_tensor_by_name('detection_boxes:0') detection_scores = detection_graph.get_tensor_by_name('detection_scores:0') detection_classes = detection_graph.get_tensor_by_name('detection_classes:0') num_detections = detection_graph.get_tensor_by_name('num_detections:0') print("[INFO] Model loaded.") colors_hash = {} while True: frames = pipeline.wait_for_frames() # ==========新增:对齐深度和彩色帧========== aligned_frames = align.process(frames) # ==========修改:从对齐后的帧里取彩色和深度帧========== color_frame = aligned_frames.get_color_frame() depth_frame = aligned_frames.get_depth_frame() color_image = np.asanyarray(color_frame.get_data()) scaled_size = (color_frame.width, color_frame.height) image_expanded = np.expand_dims(color_image, axis=0) (boxes, scores, classes, num) = sess.run([detection_boxes, detection_scores, detection_classes, num_detections], feed_dict={image_tensor: image_expanded}) boxes = np.squeeze(boxes) classes = np.squeeze(classes).astype(np.int32) scores = np.squeeze(scores) for idx in range(int(num)): class_ = classes[idx] score = scores[idx] box = boxes[idx] if class_ not in colors_hash: colors_hash[class_] = tuple(np.random.choice(range(256), size=3)) if score > 0.6: left = int(box[1] * color_frame.width) top = int(box[0] * color_frame.height) right = int(box[3] * color_frame.width) bottom = int(box[2] * color_frame.height) p1 = (left, top) p2 = (right, bottom) r, g, b = colors_hash[class_] cv2.rectangle(color_image, p1, p2, (int(r), int(g), int(b)), 2, 1) # ==========新增:计算识别框中心,获取距离========== center_x = (left + right) // 2 center_y = (top + bottom) // 2 # get_distance返回的单位是米 distance = depth_frame.get_distance(center_x, center_y) # 处理无效深度值的情况 if distance == 0: dis_text = "Distance: No valid data" else: # 保留两位小数显示 dis_text = f"Distance: {distance:.2f}m" # ==========新增:把距离文字画到框的上方========== cv2.putText(color_image, dis_text, (left, top-10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (int(r), int(g), int(b)), 2) cv2.namedWindow('RealSense', cv2.WINDOW_AUTOSIZE) cv2.imshow('RealSense', color_image) # 按q键可以退出窗口 if cv2.waitKey(1) & 0xFF == ord('q'): break print("[INFO] stop streaming ...") pipeline.stop() # 销毁所有OpenCV窗口 cv2.destroyAllWindows()
小提示
- 如果想要显示厘米单位,把
distance乘以100,同时把文字里的m改成cm即可 - 若出现大量无有效深度的提示,检查物体是否在你所用RealSense型号的测距范围内,以及摄像头深度镜头有没有被遮挡
- 代码已经添加了按
q键退出的逻辑,避免无法正常关闭窗口的问题
内容的提问来源于stack exchange,提问作者Wassim Labidi
相关产品推荐
相关产品推荐

