You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于面部关键点的头部姿态估计:无dlib实现视线向量偏移问题求助

问题说明

我需要实现从鼻尖出发、与用户视线方向一致的向量。目前找到的无需dlib、基于面部关键点的实现示例均无法正常运行,本机无法安装dlib也没有时间排查相关安装问题,当前获取的面部关键点全部准确,问题应出在其他环节。我期望实现对应效果,但现有代码生成的视线向量偏移非常严重,原代码如下:

import numpy as np
import mediapipe as mp

def x_element(elem):
    return elem[0]
def y_element(elem):
    return elem[1]

cap = cv2.VideoCapture(0)
pTime = 0
faceXY = []
mpDraw = mp.solutions.drawing_utils
mpFaceMesh = mp.solutions.face_mesh
faceMesh = mpFaceMesh.FaceMesh(max_num_faces=5, min_detection_confidence=.9, min_tracking_confidence=.01)
drawSpec = mpDraw.DrawingSpec(0,1,1)

success, img = cap.read()
height, width = img.shape[:2]
size = img.shape

# 3D model points.
face3Dmodel = np.array([
    (0.0, 0.0, 0.0),  # Nose tip
    (0.0, -330.0, -65.0),  # Chin
    (-225.0, 170.0, -135.0),  # Left eye left corner
    (225.0, 170.0, -135.0),  # Right eye right corne
    (-150.0, -150.0, -125.0),  # Left Mouth corner
    (150.0, -150.0, -125.0)  # Right mouth corner
],dtype=np.float64)


dist_coeffs = np.zeros((4, 1))  # Assuming no lens distortion
focal_length = size[1]
center = (size[1] / 2, size[0] / 2)
camera_matrix = np.array(
    [[focal_length, 0, center[0]],
     [0, focal_length, center[1]],
     [0, 0, 1]], dtype="double"
)

while True:
    success, img = cap.read()
    imgRGB = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
    results = faceMesh.process(imgRGB)
    if results.multi_face_landmarks:                                            # if faces found
        dist=[]
        for faceNum, faceLms in enumerate(results.multi_face_landmarks):                            # loop through all matches
            mpDraw.draw_landmarks(img, faceLms, landmark_drawing_spec=drawSpec) # draw every match
            faceXY = []
            for id,lm in enumerate(faceLms.landmark):                           # loop over all land marks of one face
                ih, iw, _ = img.shape
                x,y = int(lm.x*iw), int(lm.y*ih)
                # print(lm)
                faceXY.append((x, y))                                           # put all xy points in neat array
            image_points = np.array([
                faceXY[1],
                faceXY[175],
                faceXY[446],
                faceXY[226],
                faceXY[57],
                faceXY[287]
            ], dtype="double")
            for i in image_points:
                cv2.circle(img,(int(i[0]),int(i[1])),4,(255,0,0),-1)
            maxXY = max(faceXY, key=x_element)[0], max(faceXY, key=y_element)[1]
            minXY = min(faceXY, key=x_element)[0], min(faceXY, key=y_element)[1]

            xcenter = (maxXY[0] + minXY[0]) / 2
            ycenter = (maxXY[1] + minXY[1]) / 2

            dist.append((faceNum, (int(((xcenter-width/2)**2+(ycenter-height/2)**2)**.4)), maxXY, minXY))     # faceID, distance, maxXY, minXY

            print(image_points)

            (success, rotation_vector, translation_vector) = cv2.solvePnP(face3Dmodel, image_points,  camera_matrix, dist_coeffs)
            (nose_end_point2D, jacobian) = cv2.projectPoints(np.array([(0.0, 0.0, 1000.0)]), rotation_vector, translation_vector, camera_matrix, dist_coeffs)

            p1 = (int(image_points[0][0]), int(image_points[0][1]))
            p2 = (int(nose_end_point2D[0][0][0]), int(nose_end_point2D[0][0][1]))

            cv2.line(img, p1, p2, (255, 0, 0), 2)

        dist.sort(key=y_element)
        # print(dist)

        for i,faceLms in enumerate(results.multi_face_landmarks):
            if i == 0:
                cv2.rectangle(img,dist[i][2],dist[i][3],(0,255,0),2)
            else:
                cv2.rectangle(img, dist[i][2], dist[i][3], (0, 0, 255), 2)


    cv2.imshow("Image", img)
    cv2.waitKey(1)
修复方案

导致偏移的核心问题是关键点索引匹配错误、坐标系方向搞反、求解算法未指定,具体修改点如下:

  • 修正Mediapipe面部关键点与3D通用人脸模型的对应索引,原代码中的175、446等索引和预设的3D人脸点完全不匹配,是偏移的主要原因
  • 修正视线向量的z轴方向,3D通用人脸模型中z轴负方向为朝向相机的方向,原代码使用z正方向投影会导致向量方向完全相反
  • 为solvePnP指定稳定的求解算法,默认算法容易出现异常解
  • 补充缺失的cv2导入声明
修正后可运行代码
import cv2
import numpy as np
import mediapipe as mp

def x_element(elem):
    return elem[0]
def y_element(elem):
    return elem[1]

cap = cv2.VideoCapture(0)
pTime = 0
faceXY = []
mpDraw = mp.solutions.drawing_utils
mpFaceMesh = mp.solutions.face_mesh
faceMesh = mpFaceMesh.FaceMesh(max_num_faces=5, min_detection_confidence=.9, min_tracking_confidence=.01)
drawSpec = mpDraw.DrawingSpec(0,1,1)

success, img = cap.read()
height, width = img.shape[:2]
size = img.shape

# 3D通用人脸模型点
face3Dmodel = np.array([
    (0.0, 0.0, 0.0),  # 鼻尖
    (0.0, -330.0, -65.0),  # 下巴
    (-225.0, 170.0, -135.0),  # 左眼左角
    (225.0, 170.0, -135.0),  # 右眼右角
    (-150.0, -150.0, -125.0),  # 左嘴角
    (150.0, -150.0, -125.0)  # 右嘴角
],dtype=np.float64)


dist_coeffs = np.zeros((4, 1))  # 假设无镜头畸变
focal_length = size[1]
center = (size[1] / 2, size[0] / 2)
camera_matrix = np.array(
    [[focal_length, 0, center[0]],
     [0, focal_length, center[1]],
     [0, 0, 1]], dtype="double"
)

while True:
    success, img = cap.read()
    if not success:
        break
    imgRGB = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
    results = faceMesh.process(imgRGB)
    if results.multi_face_landmarks:
        dist=[]
        for faceNum, faceLms in enumerate(results.multi_face_landmarks):
            mpDraw.draw_landmarks(img, faceLms, landmark_drawing_spec=drawSpec)
            faceXY = []
            for id,lm in enumerate(faceLms.landmark):
                ih, iw, _ = img.shape
                x,y = int(lm.x*iw), int(lm.y*ih)
                faceXY.append((x, y))
            # 修正关键点索引匹配
            image_points = np.array([
                faceXY[1],    # 鼻尖
                faceXY[152],  # 下巴
                faceXY[33],   # 左眼左角
                faceXY[263],  # 右眼右角
                faceXY[61],   # 左嘴角
                faceXY[291]   # 右嘴角
            ], dtype="double")
            for i in image_points:
                cv2.circle(img,(int(i[0]),int(i[1])),4,(255,0,0),-1)
            maxXY = max(faceXY, key=x_element)[0], max(faceXY, key=y_element)[1]
            minXY = min(faceXY, key=x_element)[0], min(faceXY, key=y_element)[1]

            xcenter = (maxXY[0] + minXY[0]) / 2
            ycenter = (maxXY[1] + minXY[1]) / 2

            dist.append((faceNum, (int(((xcenter-width/2)**2+(ycenter-height/2)**2)**.4)), maxXY, minXY))

            # 指定求解算法提升稳定性
            (success, rotation_vector, translation_vector) = cv2.solvePnP(face3Dmodel, image_points, camera_matrix, dist_coeffs, flags=cv2.SOLVEPNP_ITERATIVE)
            # 修正z轴方向,投影指向用户前方的点
            (nose_end_point2D, jacobian) = cv2.projectPoints(np.array([(0.0, 0.0, -1000.0)]), rotation_vector, translation_vector, camera_matrix, dist_coeffs)

            p1 = (int(image_points[0][0]), int(image_points[0][1]))
            p2 = (int(nose_end_point2D[0][0][0]), int(nose_end_point2D[0][0][1]))

            cv2.line(img, p1, p2, (255, 0, 0), 2)

        dist.sort(key=y_element)

        for i,faceLms in enumerate(results.multi_face_landmarks):
            if i == 0:
                cv2.rectangle(img,dist[i][2],dist[i][3],(0,255,0),2)
            else:
                cv2.rectangle(img, dist[i][2], dist[i][3], (0, 0, 255), 2)


    cv2.imshow("Image", img)
    if cv2.waitKey(1) & 0xFF == ord('q'):
        break

cap.release()
cv2.destroyAllWindows()

内容的提问来源于stack exchange,提问作者Carlos Cuartas

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.05 07:18:01