You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用OpenCV与MediaPipe实现AR墨镜时图像尺寸匹配错误排查

AR墨镜AR效果实现错误修复

问题情况

我复刻AR墨镜项目,用OpenCV实现增强现实墨镜效果,使用的图像尺寸为338×149,代码如下:

import cv2
import mediapipe as mp
import numpy as np 
import keyboard

face = mp.solutions.face_detection
drawing = mp.solutions.drawing_utils

address = 'images2.jpeg'

cap = cv2.VideoCapture(0)

with face.FaceDetection(min_detection_confidence = 0.5) as face_detection :
    while cap.isOpened():
        success, image = cap.read()
        imgFront = cv2.imread(address, cv2.IMREAD_UNCHANGED)
        
        s_h, s_w, _ = imgFront.shape
        image_Height, image_Width, _ = image.shape
        
        results = face_detection.process(cv2.cvtColor(image, cv2.COLOR_BGR2RGB)) 
        
        if results.detections:
            for detection in results.detections :
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.NOSE_TIP)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                Nose_tip_x = pixelCoordinatesLandmark[0]
                Nose_tip_y = pixelCoordinatesLandmark[1]
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.LEFT_EAR_TRAGION)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                Left_Ear_x = pixelCoordinatesLandmark[0]
                Left_Ear_y = pixelCoordinatesLandmark[1]
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.RIGHT_EAR_TRAGION)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                Right_Ear_x = pixelCoordinatesLandmark[0]
                Right_Ear_y = pixelCoordinatesLandmark[1]
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.LEFT_EYE)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                Left_Eye_x = pixelCoordinatesLandmark[0]
                Left_Eye_y = pixelCoordinatesLandmark[1]
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.RIGHT_EYE)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                RIGHT_Eye_x = pixelCoordinatesLandmark[0]
                RIGHT_Eye_y = pixelCoordinatesLandmark[1]
                
                sunglass_width = int(Left_Ear_x - Right_Ear_x + 60)
                sunglass_height = int((s_h/s_w)*sunglass_width)
                
                imgFront = cv2.resize(imgFront, (sunglass_width, sunglass_height), None, 0.3, 0.3)
                hf, wf, cf = imgFront.shape
                
                hb, wb, cb = image.shape
                
                y_adjust = int((sunglass_height/80)*80)
                x_adjust = int((sunglass_width/194)*100)
                
                pos = [Nose_tip_x-x_adjust,Nose_tip_y-y_adjust]
                
                hf, wf, cf = imgFront.shape
                hb, wb, cb = image.shape
                *_, mask = cv2.split(imgFront)
                maskBGRA = cv2.cvtColor(mask, cv2.COLOR_GRAY2BGRA)
                maskBGR = cv2.cvtColor(mask, cv2.COLOR_GRAY2BGR)
                imgRGBA = cv2.bitwise_and(imgFront, maskBGRA)
                imgRGB = cv2.cvtColor(imgRGBA, cv2.COLOR_BGRA2BGR)

                imgMaskFull = np.zeros((hb, wb, cb), np.uint8)
                imgMaskFull[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = imgRGB
                imgMaskFull2 = np.ones((hb, wb, cb), np.uint8) * 255
                maskBGRInv = cv2.bitwise_not(maskBGR)
                imgMaskFull2[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = maskBGRInv

                image = cv2.bitwise_and(image, imgMaskFull2)
                image = cv2.bitwise_or(image, imgMaskFull)
                
        cv2.imshow('Sunglass Effect', image)
        
        if keyboard.is_pressed('q'):
            break
        
        cv2.waitKey(5)      

cap.release()
cv2.destroyAllWindows()

运行时出现如下错误:

imgRGBA = cv2.bitwise_and(imgFront, maskBGRA)
cv2.error: OpenCV(4.9.0) D:\a\opencv-python\opencv-python\opencv\modules\core\src\arithm.cpp:214: error: (-209:Sizes of input arguments do not match) The operation is neither 'array op array' (where arrays have the same size and type), nor 'array op scalar', nor 'scalar op array' in function 'cv::binary_op'

错误原因

  1. 通道数不匹配:读取的JPEG图像没有透明通道(仅3个BGR通道),但maskBGRA是4通道的BGRA图像,导致bitwise_and操作时输入尺寸/通道数不一致。
  2. Resize参数冲突:cv2.resize同时指定了目标尺寸和缩放比例(fx=0.3, fy=0.3),实际缩放后的尺寸并非计算的sunglass_width和sunglass_height,加剧尺寸不匹配问题。
  3. Mask分割错误:图像无alpha通道时,*_, mask = cv2.split(imgFront)的写法会导致mask尺寸或通道数异常。

修复方案

1. 图像格式调整

将墨镜图像转换为带透明通道的PNG格式,才能正确提取透明区域作为mask。

2. 修改代码关键部分

(1)修复图像通道处理

读取图像后检查通道数,无alpha通道则手动添加:

imgFront = cv2.imread(address, cv2.IMREAD_UNCHANGED)
# 检查是否有alpha通道,无则添加全白alpha通道
if len(imgFront.shape) == 3 and imgFront.shape[2] == 3:
    alpha = np.full((imgFront.shape[0], imgFront.shape[1]), 255, dtype=imgFront.dtype)
    imgFront = cv2.merge((imgFront, alpha))

(2)修复Resize参数

去掉冲突的缩放比例参数,只保留目标尺寸:

imgFront = cv2.resize(imgFront, (sunglass_width, sunglass_height))

(3)修复Mask处理逻辑

正确分割alpha通道并生成匹配的mask:

# 分割出BGRA四个通道
b, g, r, mask = cv2.split(imgFront)
# 生成4通道mask,和imgFront通道数一致
maskBGRA = cv2.merge((mask, mask, mask, mask))
# 生成3通道mask,和摄像头图像通道数一致
maskBGR = cv2.merge((mask, mask, mask))

imgRGBA = cv2.bitwise_and(imgFront, maskBGRA)
imgRGB = cv2.cvtColor(imgRGBA, cv2.COLOR_BGRA2BGR)

3. 完整修复后代码

import cv2
import mediapipe as mp
import numpy as np 
import keyboard

face = mp.solutions.face_detection
drawing = mp.solutions.drawing_utils

address = 'images2.png'  # 改为PNG格式图片路径

cap = cv2.VideoCapture(0)

with face.FaceDetection(min_detection_confidence = 0.5) as face_detection :
    while cap.isOpened():
        success, image = cap.read()
        if not success:
            break
            
        imgFront = cv2.imread(address, cv2.IMREAD_UNCHANGED)
        # 处理图像通道,确保是4通道BGRA
        if len(imgFront.shape) == 3 and imgFront.shape[2] == 3:
            alpha = np.full((imgFront.shape[0], imgFront.shape[1]), 255, dtype=imgFront.dtype)
            imgFront = cv2.merge((imgFront, alpha))
        
        s_h, s_w, _ = imgFront.shape
        image_Height, image_Width, _ = image.shape
        
        results = face_detection.process(cv2.cvtColor(image, cv2.COLOR_BGR2RGB)) 
        
        if results.detections:
            for detection in results.detections :
                # 获取面部关键点
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.NOSE_TIP)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                if not pixelCoordinatesLandmark:
                    continue
                Nose_tip_x, Nose_tip_y = pixelCoordinatesLandmark
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.LEFT_EAR_TRAGION)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                if not pixelCoordinatesLandmark:
                    continue
                Left_Ear_x, Left_Ear_y = pixelCoordinatesLandmark
                
                normalizedLandmark = face.get_key_point(detection, face.FaceKeyPoint.RIGHT_EAR_TRAGION)
                pixelCoordinatesLandmark = drawing._normalized_to_pixel_coordinates(normalizedLandmark.x, normalizedLandmark.y, image_Width, image_Height)
                if not pixelCoordinatesLandmark:
                    continue
                Right_Ear_x, Right_Ear_y = pixelCoordinatesLandmark
                
                # 计算墨镜尺寸
                sunglass_width = int(Left_Ear_x - Right_Ear_x + 60)
                sunglass_height = int((s_h/s_w)*sunglass_width)
                
                # 修复resize参数
                imgFront = cv2.resize(imgFront, (sunglass_width, sunglass_height))
                hf, wf, cf = imgFront.shape
                
                hb, wb, cb = image.shape
                
                y_adjust = int((sunglass_height/80)*80)
                x_adjust = int((sunglass_width/194)*100)
                
                pos = [Nose_tip_x-x_adjust,Nose_tip_y-y_adjust]
                
                # 检查位置是否超出图像范围
                if pos[0] < 0 or pos[1] < 0 or pos[0]+wf > wb or pos[1]+hf > hb:
                    continue
                
                # 修复mask处理
                b, g, r, mask = cv2.split(imgFront)
                maskBGRA = cv2.merge((mask, mask, mask, mask))
                maskBGR = cv2.merge((mask, mask, mask))
                
                imgRGBA = cv2.bitwise_and(imgFront, maskBGRA)
                imgRGB = cv2.cvtColor(imgRGBA, cv2.COLOR_BGRA2BGR)

                imgMaskFull = np.zeros((hb, wb, cb), np.uint8)
                imgMaskFull[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = imgRGB
                imgMaskFull2 = np.ones((hb, wb, cb), np.uint8) * 255
                maskBGRInv = cv2.bitwise_not(maskBGR)
                imgMaskFull2[pos[1]:hf + pos[1], pos[0]:wf + pos[0], :] = maskBGRInv

                image = cv2.bitwise_and(image, imgMaskFull2)
                image = cv2.bitwise_or(image, imgMaskFull)
                
        cv2.imshow('Sunglass Effect', image)
        
        if keyboard.is_pressed('q'):
            break
        
        cv2.waitKey(5)      

cap.release()
cv2.destroyAllWindows()

内容的提问来源于stack exchange,提问作者ACHINTYA GUPTA

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 07:20:56