使用OpenCV与MediaPipe实现AR墨镜时图像尺寸匹配错误排查
AR墨镜AR效果实现错误修复
问题情况
我复刻AR墨镜项目,用OpenCV实现增强现实墨镜效果,使用的图像尺寸为338×149,代码如下:
import cv2 import mediapipe as mp import numpy as np import keyboard face = mp.solutions.face_detection drawing = mp.solutions.drawing_utils address = 'images2.jpeg' cap = cv2.VideoCapture(0) with face.FaceDetection(min_detection_confidence = 0.5) as face_detection : while cap.isOpened(): success, image = cap.read() imgFront = cv2.imread(address, cv2.IMREAD_UNCHANGED) s_h, s_w, _ = imgFront.shape image_Height, image_Width, _ = image.shape results = face_detection.process(cv2.cvtColor(image, cv2.COLOR_BGR2RGB)) if results.detections: for detection in results.detections : normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.NOSE_TIP) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) Nose_tip_x = pixelCoordinatesLandmark[0] Nose_tip_y = pixelCoordinatesLandmark[1] normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.LEFT_EAR_TRAGION) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) Left_Ear_x = pixelCoordinatesLandmark[0] Left_Ear_y = pixelCoordinatesLandmark[1] normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.RIGHT_EAR_TRAGION) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) Right_Ear_x = pixelCoordinatesLandmark[0] Right_Ear_y = pixelCoordinatesLandmark[1] normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.LEFT_EYE) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) Left_Eye_x = pixelCoordinatesLandmark[0] Left_Eye_y = pixelCoordinatesLandmark[1] normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.RIGHT_EYE) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) RIGHT_Eye_x = pixelCoordinatesLandmark[0] RIGHT_Eye_y = pixelCoordinatesLandmark[1] sunglass_width = int(Left_Ear_x - Right_Ear_x + 60) sunglass_height = int((s_h/s_w)*sunglass_width) imgFront = cv2.resize(imgFront, (sunglass_width, sunglass_height), None, 0.3, 0.3) hf, wf, cf = imgFront.shape hb, wb, cb = image.shape y_adjust = int((sunglass_height/80)*80) x_adjust = int((sunglass_width/194)*100) pos = [Nose_tip_x-x_adjust,Nose_tip_y-y_adjust] hf, wf, cf = imgFront.shape hb, wb, cb = image.shape *_, mask = cv2.split(imgFront) maskBGRA = cv2.cvtColor(mask, cv2.COLOR_GRAY2BGRA) maskBGR = cv2.cvtColor(mask, cv2.COLOR_GRAY2BGR) imgRGBA = cv2.bitwise_and(imgFront, maskBGRA) imgRGB = cv2.cvtColor(imgRGBA, cv2.COLOR_BGRA2BGR) imgMaskFull = np.zeros((hb, wb, cb), np.uint8) imgMaskFull[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = imgRGB imgMaskFull2 = np.ones((hb, wb, cb), np.uint8) * 255 maskBGRInv = cv2.bitwise_not(maskBGR) imgMaskFull2[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = maskBGRInv image = cv2.bitwise_and(image, imgMaskFull2) image = cv2.bitwise_or(image, imgMaskFull) cv2.imshow('Sunglass Effect', image) if keyboard.is_pressed('q'): break cv2.waitKey(5) cap.release() cv2.destroyAllWindows()
运行时出现如下错误:
imgRGBA = cv2.bitwise_and(imgFront, maskBGRA) cv2.error: OpenCV(4.9.0) D:\a\opencv-python\opencv-python\opencv\modules\core\src\arithm.cpp:214: error: (-209:Sizes of input arguments do not match) The operation is neither 'array op array' (where arrays have the same size and type), nor 'array op scalar', nor 'scalar op array' in function 'cv::binary_op'
错误原因
- 通道数不匹配:读取的JPEG图像没有透明通道(仅3个BGR通道),但
maskBGRA是4通道的BGRA图像,导致bitwise_and操作时输入尺寸/通道数不一致。 - Resize参数冲突:
cv2.resize同时指定了目标尺寸和缩放比例(fx=0.3, fy=0.3),实际缩放后的尺寸并非计算的sunglass_width和sunglass_height,加剧尺寸不匹配问题。 - Mask分割错误:图像无alpha通道时,
*_, mask = cv2.split(imgFront)的写法会导致mask尺寸或通道数异常。
修复方案
1. 图像格式调整
将墨镜图像转换为带透明通道的PNG格式,才能正确提取透明区域作为mask。
2. 修改代码关键部分
(1)修复图像通道处理
读取图像后检查通道数,无alpha通道则手动添加:
imgFront = cv2.imread(address, cv2.IMREAD_UNCHANGED) # 检查是否有alpha通道,无则添加全白alpha通道 if len(imgFront.shape) == 3 and imgFront.shape[2] == 3: alpha = np.full((imgFront.shape[0], imgFront.shape[1]), 255, dtype=imgFront.dtype) imgFront = cv2.merge((imgFront, alpha))
(2)修复Resize参数
去掉冲突的缩放比例参数,只保留目标尺寸:
imgFront = cv2.resize(imgFront, (sunglass_width, sunglass_height))
(3)修复Mask处理逻辑
正确分割alpha通道并生成匹配的mask:
# 分割出BGRA四个通道 b, g, r, mask = cv2.split(imgFront) # 生成4通道mask,和imgFront通道数一致 maskBGRA = cv2.merge((mask, mask, mask, mask)) # 生成3通道mask,和摄像头图像通道数一致 maskBGR = cv2.merge((mask, mask, mask)) imgRGBA = cv2.bitwise_and(imgFront, maskBGRA) imgRGB = cv2.cvtColor(imgRGBA, cv2.COLOR_BGRA2BGR)
3. 完整修复后代码
import cv2 import mediapipe as mp import numpy as np import keyboard face = mp.solutions.face_detection drawing = mp.solutions.drawing_utils address = 'images2.png' # 改为PNG格式图片路径 cap = cv2.VideoCapture(0) with face.FaceDetection(min_detection_confidence = 0.5) as face_detection : while cap.isOpened(): success, image = cap.read() if not success: break imgFront = cv2.imread(address, cv2.IMREAD_UNCHANGED) # 处理图像通道,确保是4通道BGRA if len(imgFront.shape) == 3 and imgFront.shape[2] == 3: alpha = np.full((imgFront.shape[0], imgFront.shape[1]), 255, dtype=imgFront.dtype) imgFront = cv2.merge((imgFront, alpha)) s_h, s_w, _ = imgFront.shape image_Height, image_Width, _ = image.shape results = face_detection.process(cv2.cvtColor(image, cv2.COLOR_BGR2RGB)) if results.detections: for detection in results.detections : # 获取面部关键点 normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.NOSE_TIP) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) if not pixelCoordinatesLandmark: continue Nose_tip_x, Nose_tip_y = pixelCoordinatesLandmark normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.LEFT_EAR_TRAGION) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) if not pixelCoordinatesLandmark: continue Left_Ear_x, Left_Ear_y = pixelCoordinatesLandmark normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.RIGHT_EAR_TRAGION) pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height) if not pixelCoordinatesLandmark: continue Right_Ear_x, Right_Ear_y = pixelCoordinatesLandmark # 计算墨镜尺寸 sunglass_width = int(Left_Ear_x - Right_Ear_x + 60) sunglass_height = int((s_h/s_w)*sunglass_width) # 修复resize参数 imgFront = cv2.resize(imgFront, (sunglass_width, sunglass_height)) hf, wf, cf = imgFront.shape hb, wb, cb = image.shape y_adjust = int((sunglass_height/80)*80) x_adjust = int((sunglass_width/194)*100) pos = [Nose_tip_x-x_adjust,Nose_tip_y-y_adjust] # 检查位置是否超出图像范围 if pos[0] < 0 or pos[1] < 0 or pos[0]+wf > wb or pos[1]+hf > hb: continue # 修复mask处理 b, g, r, mask = cv2.split(imgFront) maskBGRA = cv2.merge((mask, mask, mask, mask)) maskBGR = cv2.merge((mask, mask, mask)) imgRGBA = cv2.bitwise_and(imgFront, maskBGRA) imgRGB = cv2.cvtColor(imgRGBA, cv2.COLOR_BGRA2BGR) imgMaskFull = np.zeros((hb, wb, cb), np.uint8) imgMaskFull[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = imgRGB imgMaskFull2 = np.ones((hb, wb, cb), np.uint8) * 255 maskBGRInv = cv2.bitwise_not(maskBGR) imgMaskFull2[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = maskBGRInv image = cv2.bitwise_and(image, imgMaskFull2) image = cv2.bitwise_or(image, imgMaskFull) cv2.imshow('Sunglass Effect', image) if keyboard.is_pressed('q'): break cv2.waitKey(5) cap.release() cv2.destroyAllWindows()
内容的提问来源于stack exchange,提问作者ACHINTYA GUPTA
相关产品推荐
相关产品推荐

