如何基于MediaPipe Face Mesh计算屏幕注视位置?
基于MediaPipe Face Mesh的眼动追踪:计算屏幕视线指向
你已经通过MediaPipe Face Mesh成功获取了双眼虹膜中心坐标,接下来要实现屏幕视线指向的计算,核心是建立虹膜位置与屏幕坐标的映射关系,下面是可落地的技术方案和代码示例:
核心实现思路
目前常用两种方法,适合快速落地的是校准拟合法:
- 先让用户注视屏幕上几个已知位置(比如四角+中心),记录每个位置对应的虹膜在眼眶内的相对偏移
- 用这些数据训练一个简单的回归模型,之后就能通过实时的虹膜偏移预测视线指向的屏幕坐标
另一种是几何模型法,需要结合相机内参、面部3D关键点计算视线方向,再投影到屏幕,精度更高但实现复杂,适合对精度要求高的场景。
关键技术细节
- 选对参考关键点:除了虹膜,还要用到MediaPipe Face Mesh里的眼眶关键点(左眼:33、133等;右眼:362、263等),用来计算虹膜在眼眶内的相对位置,这个偏移量是视线映射的核心特征
- 归一化处理:把虹膜相对于眼眶的偏移归一化到[0,1]区间,避免不同用户面部大小、距离相机远近的影响
- 平滑优化:对连续帧的虹膜坐标做移动平均,减少抖动带来的预测误差
- 校准流程:至少需要3个校准点,5个点(四角+中心)能让模型更准确
整合后的代码示例
下面的代码在你现有基础上添加了校准和视线预测功能:
import cv2 import mediapipe as mp import numpy as np from tracker import Tracker from sklearn.linear_model import LinearRegression import pickle mp_face_mesh = mp.solutions.face_mesh cap = cv2.VideoCapture(0) # 替换成你的实际屏幕分辨率 SCREEN_WIDTH = 1920 SCREEN_HEIGHT = 1080 # 校准点:屏幕上需要用户注视的位置 CALIBRATION_POINTS = [ (SCREEN_WIDTH//2, SCREEN_HEIGHT//2), # 中心 (100, 100), # 左上 (SCREEN_WIDTH-100, 100), # 右上 (100, SCREEN_HEIGHT-100), # 左下 (SCREEN_WIDTH-100, SCREEN_HEIGHT-100) # 右下 ] tracker = Tracker() calib_features = [] # 存储虹膜偏移特征 calib_labels = [] # 存储对应屏幕坐标 calibrated = False model = LinearRegression() def get_iris_relative_offset(mesh_points): """计算虹膜在眼眶内的归一化偏移量(双眼取平均提高稳定性)""" # MediaPipe Face Mesh 眼眶关键点索引 LEFT_EYE_LANDMARKS = [33, 133, 157, 158, 159, 160, 161, 246] RIGHT_EYE_LANDMARKS = [362, 263, 384, 385, 386, 387, 388, 466] # 处理左眼 left_eye_box = mesh_points[LEFT_EYE_LANDMARKS] left_min_x, left_min_y = left_eye_box.min(axis=0) left_max_x, left_max_y = left_eye_box.max(axis=0) LEFT_IRIS = tracker.get_iris_points()[0] (l_cx, l_cy), _ = cv2.minEnclosingCircle(mesh_points[LEFT_IRIS]) left_offset_x = (l_cx - left_min_x) / (left_max_x - left_min_x) left_offset_y = (l_cy - left_min_y) / (left_max_y - left_min_y) # 处理右眼 right_eye_box = mesh_points[RIGHT_EYE_LANDMARKS] right_min_x, right_min_y = right_eye_box.min(axis=0) right_max_x, right_max_y = right_eye_box.max(axis=0) RIGHT_IRIS = tracker.get_iris_points()[1] (r_cx, r_cy), _ = cv2.minEnclosingCircle(mesh_points[RIGHT_IRIS]) right_offset_x = (r_cx - right_min_x) / (right_max_x - right_min_x) right_offset_y = (r_cy - right_min_y) / (right_max_y - right_min_y) # 返回双眼平均偏移 return np.array([(left_offset_x + right_offset_x)/2, (left_offset_y + right_offset_y)/2]) with tracker.get_face_mesh() as face_mesh: calib_index = 0 while cap.isOpened(): success, image = cap.read() if not success: print("Ignoring empty camera frame") continue image.flags.writeable = False image = cv2.flip(image, 1) image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) height, width = image.shape[:2] results = face_mesh.process(image) image.flags.writeable = True image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR) if results.multi_face_landmarks: mesh_points = np.array([np.multiply([p.x, p.y], [width, height]).astype(int) for p in results.multi_face_landmarks[0].landmark]) LEFT_IRIS, RIGHT_IRIS = tracker.get_iris_points() (l_cx, l_cy), l_radius = cv2.minEnclosingCircle(mesh_points[LEFT_IRIS]) (r_cx, r_cy), r_radius = cv2.minEnclosingCircle(mesh_points[RIGHT_IRIS]) center_l = np.array([l_cx, l_cy], dtype=np.int32) center_r = np.array([r_cx, r_cy], dtype=np.int32) cv2.circle(image, center_l, int(l_radius), (0, 255, 255), 1, cv2.LINE_AA) cv2.circle(image, center_r, int(r_radius), (0, 255, 255), 1, cv2.LINE_AA) # 校准流程:按空格键记录当前注视点 if not calibrated: cv2.putText(image, f"注视点 {calib_index+1}/{len(CALIBRATION_POINTS)},按空格记录", (50, 50), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 2) key = cv2.waitKey(5) & 0xFF if key == 32: offset = get_iris_relative_offset(mesh_points) calib_features.append(offset) calib_labels.append(CALIBRATION_POINTS[calib_index]) calib_index += 1 if calib_index >= len(CALIBRATION_POINTS): model.fit(calib_features, calib_labels) calibrated = True print("校准完成!") # 可选:保存模型,下次直接加载不用重复校准 with open('gaze_model.pkl', 'wb') as f: pickle.dump(model, f) else: # 实时预测视线指向的屏幕坐标 offset = get_iris_relative_offset(mesh_points) screen_x, screen_y = model.predict([offset])[0] cv2.putText(image, f"视线位置: ({int(screen_x)}, {int(screen_y)})", (50, 50), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 0, 255), 2) cv2.namedWindow("MediaPipe Face Mesh", cv2.WND_PROP_FULLSCREEN) cv2.setWindowProperty("MediaPipe Face Mesh", cv2.WND_PROP_FULLSCREEN,cv2.WINDOW_FULLSCREEN) cv2.imshow('MediaPipe Face Mesh', image) if cv2.waitKey(5) & 0xFF == 27: break cap.release() cv2.destroyAllWindows()
使用说明
- 运行代码后,按照提示依次注视屏幕上的校准点,按空格键记录每个点的虹膜位置
- 校准完成后,程序会实时显示预测的视线屏幕坐标
- 如果需要更高精度,可以增加校准点数量,或者改用多项式回归模型
内容的提问来源于stack exchange,提问作者nicesaj360
相关产品推荐
相关产品推荐

