基于面部关键点的头部姿态估计:无dlib实现视线向量偏移问题求助
问题说明
我需要实现从鼻尖出发、与用户视线方向一致的向量。目前找到的无需dlib、基于面部关键点的实现示例均无法正常运行,本机无法安装dlib也没有时间排查相关安装问题,当前获取的面部关键点全部准确,问题应出在其他环节。我期望实现对应效果,但现有代码生成的视线向量偏移非常严重,原代码如下:
import numpy as np import mediapipe as mp def x_element(elem): return elem[0] def y_element(elem): return elem[1] cap = cv2.VideoCapture(0) pTime = 0 faceXY = [] mpDraw = mp.solutions.drawing_utils mpFaceMesh = mp.solutions.face_mesh faceMesh = mpFaceMesh.FaceMesh(max_num_faces=5, min_detection_confidence=.9, min_tracking_confidence=.01) drawSpec = mpDraw.DrawingSpec(0,1,1) success, img = cap.read() height, width = img.shape[:2] size = img.shape # 3D model points. face3Dmodel = np.array([ (0.0, 0.0, 0.0), # Nose tip (0.0, -330.0, -65.0), # Chin (-225.0, 170.0, -135.0), # Left eye left corner (225.0, 170.0, -135.0), # Right eye right corne (-150.0, -150.0, -125.0), # Left Mouth corner (150.0, -150.0, -125.0) # Right mouth corner ],dtype=np.float64) dist_coeffs = np.zeros((4, 1)) # Assuming no lens distortion focal_length = size[1] center = (size[1] / 2, size[0] / 2) camera_matrix = np.array( [[focal_length, 0, center[0]], [0, focal_length, center[1]], [0, 0, 1]], dtype="double" ) while True: success, img = cap.read() imgRGB = cv2.cvtColor(img, cv2.COLOR_BGR2RGB) results = faceMesh.process(imgRGB) if results.multi_face_landmarks: # if faces found dist=[] for faceNum, faceLms in enumerate(results.multi_face_landmarks): # loop through all matches mpDraw.draw_landmarks(img, faceLms, landmark_drawing_spec=drawSpec) # draw every match faceXY = [] for id,lm in enumerate(faceLms.landmark): # loop over all land marks of one face ih, iw, _ = img.shape x,y = int(lm.x*iw), int(lm.y*ih) # print(lm) faceXY.append((x, y)) # put all xy points in neat array image_points = np.array([ faceXY[1], faceXY[175], faceXY[446], faceXY[226], faceXY[57], faceXY[287] ], dtype="double") for i in image_points: cv2.circle(img,(int(i[0]),int(i[1])),4,(255,0,0),-1) maxXY = max(faceXY, key=x_element)[0], max(faceXY, key=y_element)[1] minXY = min(faceXY, key=x_element)[0], min(faceXY, key=y_element)[1] xcenter = (maxXY[0] + minXY[0]) / 2 ycenter = (maxXY[1] + minXY[1]) / 2 dist.append((faceNum, (int(((xcenter-width/2)**2+(ycenter-height/2)**2)**.4)), maxXY, minXY)) # faceID, distance, maxXY, minXY print(image_points) (success, rotation_vector, translation_vector) = cv2.solvePnP(face3Dmodel, image_points, camera_matrix, dist_coeffs) (nose_end_point2D, jacobian) = cv2.projectPoints(np.array([(0.0, 0.0, 1000.0)]), rotation_vector, translation_vector, camera_matrix, dist_coeffs) p1 = (int(image_points[0][0]), int(image_points[0][1])) p2 = (int(nose_end_point2D[0][0][0]), int(nose_end_point2D[0][0][1])) cv2.line(img, p1, p2, (255, 0, 0), 2) dist.sort(key=y_element) # print(dist) for i,faceLms in enumerate(results.multi_face_landmarks): if i == 0: cv2.rectangle(img,dist[i][2],dist[i][3],(0,255,0),2) else: cv2.rectangle(img, dist[i][2], dist[i][3], (0, 0, 255), 2) cv2.imshow("Image", img) cv2.waitKey(1)
修复方案
导致偏移的核心问题是关键点索引匹配错误、坐标系方向搞反、求解算法未指定,具体修改点如下:
- 修正Mediapipe面部关键点与3D通用人脸模型的对应索引,原代码中的175、446等索引和预设的3D人脸点完全不匹配,是偏移的主要原因
- 修正视线向量的z轴方向,3D通用人脸模型中z轴负方向为朝向相机的方向,原代码使用z正方向投影会导致向量方向完全相反
- 为solvePnP指定稳定的求解算法,默认算法容易出现异常解
- 补充缺失的cv2导入声明
修正后可运行代码
import cv2 import numpy as np import mediapipe as mp def x_element(elem): return elem[0] def y_element(elem): return elem[1] cap = cv2.VideoCapture(0) pTime = 0 faceXY = [] mpDraw = mp.solutions.drawing_utils mpFaceMesh = mp.solutions.face_mesh faceMesh = mpFaceMesh.FaceMesh(max_num_faces=5, min_detection_confidence=.9, min_tracking_confidence=.01) drawSpec = mpDraw.DrawingSpec(0,1,1) success, img = cap.read() height, width = img.shape[:2] size = img.shape # 3D通用人脸模型点 face3Dmodel = np.array([ (0.0, 0.0, 0.0), # 鼻尖 (0.0, -330.0, -65.0), # 下巴 (-225.0, 170.0, -135.0), # 左眼左角 (225.0, 170.0, -135.0), # 右眼右角 (-150.0, -150.0, -125.0), # 左嘴角 (150.0, -150.0, -125.0) # 右嘴角 ],dtype=np.float64) dist_coeffs = np.zeros((4, 1)) # 假设无镜头畸变 focal_length = size[1] center = (size[1] / 2, size[0] / 2) camera_matrix = np.array( [[focal_length, 0, center[0]], [0, focal_length, center[1]], [0, 0, 1]], dtype="double" ) while True: success, img = cap.read() if not success: break imgRGB = cv2.cvtColor(img, cv2.COLOR_BGR2RGB) results = faceMesh.process(imgRGB) if results.multi_face_landmarks: dist=[] for faceNum, faceLms in enumerate(results.multi_face_landmarks): mpDraw.draw_landmarks(img, faceLms, landmark_drawing_spec=drawSpec) faceXY = [] for id,lm in enumerate(faceLms.landmark): ih, iw, _ = img.shape x,y = int(lm.x*iw), int(lm.y*ih) faceXY.append((x, y)) # 修正关键点索引匹配 image_points = np.array([ faceXY[1], # 鼻尖 faceXY[152], # 下巴 faceXY[33], # 左眼左角 faceXY[263], # 右眼右角 faceXY[61], # 左嘴角 faceXY[291] # 右嘴角 ], dtype="double") for i in image_points: cv2.circle(img,(int(i[0]),int(i[1])),4,(255,0,0),-1) maxXY = max(faceXY, key=x_element)[0], max(faceXY, key=y_element)[1] minXY = min(faceXY, key=x_element)[0], min(faceXY, key=y_element)[1] xcenter = (maxXY[0] + minXY[0]) / 2 ycenter = (maxXY[1] + minXY[1]) / 2 dist.append((faceNum, (int(((xcenter-width/2)**2+(ycenter-height/2)**2)**.4)), maxXY, minXY)) # 指定求解算法提升稳定性 (success, rotation_vector, translation_vector) = cv2.solvePnP(face3Dmodel, image_points, camera_matrix, dist_coeffs, flags=cv2.SOLVEPNP_ITERATIVE) # 修正z轴方向,投影指向用户前方的点 (nose_end_point2D, jacobian) = cv2.projectPoints(np.array([(0.0, 0.0, -1000.0)]), rotation_vector, translation_vector, camera_matrix, dist_coeffs) p1 = (int(image_points[0][0]), int(image_points[0][1])) p2 = (int(nose_end_point2D[0][0][0]), int(nose_end_point2D[0][0][1])) cv2.line(img, p1, p2, (255, 0, 0), 2) dist.sort(key=y_element) for i,faceLms in enumerate(results.multi_face_landmarks): if i == 0: cv2.rectangle(img,dist[i][2],dist[i][3],(0,255,0),2) else: cv2.rectangle(img, dist[i][2], dist[i][3], (0, 0, 255), 2) cv2.imshow("Image", img) if cv2.waitKey(1) & 0xFF == ord('q'): break cap.release() cv2.destroyAllWindows()
内容的提问来源于stack exchange,提问作者Carlos Cuartas
相关产品推荐
相关产品推荐

