MoveNet实时摄像头姿态检测关键点定位不准问题求助
MoveNet关键点定位偏差问题
- 在搭载M2芯片的13英寸MacBook Pro上,使用MoveNet处理摄像头实时画面时出现以下定位偏差:
- 仅拍摄人脸及肩部上方区域时,肩部关键点位置过高
- 受试者后移至画面完整显示上半身时,肩部关键点恢复正常,但眼部位置偏低,且手臂关键点仅能识别到肘部,无法延伸至手腕
- 尝试修改绘图与预处理函数后,有时会出现关键点直接偏移至屏幕左上角,完全无法对应人体区域的情况
相关代码
import numpy as np from matplotlib import pyplot as plt import cv2 import tensorflow as tf EDGES = { (0, 1): 'm', (0, 2): 'c', (1, 3): 'm', (2, 4): 'c', (0, 5): 'm', (0, 6): 'c', (5, 7): 'm', (7, 9): 'm', (6, 8): 'c', (8, 10): 'c', (5, 6): 'y', (5, 11): 'm', (6, 12): 'c', (11, 12): 'y', (11, 13): 'm', (13, 15): 'm', (12, 14): 'c', (14, 16): 'c' }# 定义关键点之间的连接边 def draw_keypoints(frame, keypoints, confidence_threshold): y, x, c = frame.shape shaped = np.squeeze(np.multiply(keypoints, [y,x,1])) for kp in shaped: ky, kx, kp_conf = kp if kp_conf > confidence_threshold: cv2.circle(frame, (int(kx), int(ky)), 4, (0,255,0), -1) def draw_connections(frame, keypoints, edges, confidence_threshold): y, x, c = frame.shape shaped = np.squeeze(np.multiply(keypoints, [y,x,1])) for edge, color in edges.items(): p1, p2 = edge y1, x1, c1 = shaped[p1] y2, x2, c2 = shaped[p2] if (c1 > confidence_threshold) & (c2 > confidence_threshold): cv2.line(frame, (int(x1), int(y1)), (int(x2), int(y2)), (0,0,255), 2) def preprocess_image(frame): # 定义目标尺寸 target_size = 256 # 计算原始画面的宽高比 orig_height, orig_width, _ = frame.shape aspect_ratio = orig_width / orig_height # 调整画面尺寸 if aspect_ratio >= 1: # 宽大于等于高 new_width = target_size new_height = round(target_size / aspect_ratio) else: # 高大于宽 new_height = target_size new_width = round(target_size * aspect_ratio) frame = cv2.resize(frame, (new_width, new_height)) # 填充画面 pad_top = (target_size - new_height) // 2 pad_bottom = target_size - new_height - pad_top pad_left = (target_size - new_width) // 2 pad_right = target_size - new_width - pad_left frame = cv2.copyMakeBorder(frame, pad_top, pad_bottom, pad_left, pad_right, cv2.BORDER_CONSTANT) return frame interpreter = tf.lite.Interpreter(model_path='lite-model_movenet_singlepose_thunder_3.tflite') # 加载模型 interpreter.allocate_tensors() # 为模型分配内存 img = any cap = cv2.VideoCapture(0) while cap.isOpened(): ret, frame = cap.read() # 重塑图像 img = frame.copy() img = preprocess_image(img) # 转换为float32并添加batch维度 input_image = np.expand_dims(img.astype(np.float32), axis=0) # 设置输入输出 input_details = interpreter.get_input_details() output_details = interpreter.get_output_details() # 执行预测 interpreter.set_tensor(input_details[0]['index'], input_image) interpreter.invoke() keypoints_with_scores = interpreter.get_tensor(output_details[0]['index']) # 渲染关键点和连接边 draw_connections(frame, keypoints_with_scores, EDGES, 0.1) draw_keypoints(frame, keypoints_with_scores, 0.1) cv2.imshow('MoveNet Lightning', frame) if cv2.waitKey(10) & 0xFF==ord('q'): break cap.release() cv2.destroyAllWindows() plt.imshow(img) # 显示预处理后的图像 print(img.shape) right_hand = keypoints_with_scores[0][0][9] # 获取右手关键点 left_hand = keypoints_with_scores[0][0][10] # 获取左手关键点 px_cordinates = np.array(left_hand[:2]*[720,1280]).astype(int) # 转换为像素坐标
内容的提问来源于stack exchange,提问作者Agam Aneja
相关产品推荐
相关产品推荐

