You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在原始图像中绘制变换后检测到的答题卡选项边界框?

答题卡自动批改:将变换后图像的选项边界框映射回原始图像

我正在编写四选项答题卡自动批改代码,需要为学生选中的涂黑选项绘制边界框。目前使用cv2.findContours、imutils.grab_contours及four_point_transform获取自定义答题卡的最大矩形区域以提取选项;已能在变换后的俯视图像中绘制选项边界框,但不知如何在原始图像中实现该操作。猜测可使用透视变换矩阵的逆矩阵来实现,但不确定是否可行及具体操作方法,相关代码如下:

find_answer函数

def find_answer(dst,):
    gray_dst = cv2.cvtColor(dst,cv2.COLOR_BGR2GRAY) 
    blurred_dst = cv2.GaussianBlur(gray_dst,(3,3),0)
    edged_dst = cv2.Canny(blurred_dst,75,200)
    black_threshold = np.sum(gray_dst[0:37,0:37])/(37*37)
    cnts = cv2.findContours(edged_dst,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)
    cnts = imutils.grab_contours(cnts)
    docCnt = None
    if len(cnts):
        cnts = sorted(cnts,key=cv2.contourArea,reverse=True)

    for c in cnts:
        peri = cv2.arcLength(c,True)
        approx = cv2.approxPolyDP(c,0.02*peri,True)
        if(len(approx))==4:
            docCnt = approx
            break
    paper =  four_point_transform(dst,docCnt.reshape(4,2))
    warped =  four_point_transform(gray_dst,docCnt.reshape(4,2)) 
    thresh = cv2.threshold(warped,0,255,cv2.THRESH_BINARY_INV|cv2.THRESH_OTSU)[1]
    cnts = cv2.findContours(thresh,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)
    cnts = imutils.grab_contours(cnts)
    cnts = sorted(cnts,key=cv2.contourArea,reverse=True)
    answers = {}
    P=True
    for i in range(len(q)):

        w = q[i][0][2]
        h = q[i][0][3]
        area = w*h
        li = []
        for j  in range(len(q[i])):
            area =  q[i][j][2]*q[i][j][3]
            y1 = q[i][j][0]
            x1 = q[i][j][1]
            y2 = q[i][j][0] + q[i][j][2]
            x2 = q[i][j][1] +q[i][j][3]
            sum = np.sum(warped[x1:x2,y1:y2])
            #print(sum)
            #print(w,h)
            if sum/area <black_threshold:
                P = False
                print('i:',i,'j:',j,'sum is:',sum)
                print('thersh is:',area*188)
                li.extend([4-j])
        answers[i+1] = li
    return answers

four_point_transform函数

def four_point_transform(image, pts):
    rect = order_points(pts)
    (tl, tr, br, bl) = rect
    # compute the width of the new image, which will be the
    # maximum distance between bottom-right and bottom-left
    # x-coordiates or the top-right and top-left x-coordinates
    widthA = np.sqrt(((br[0] - bl[0]) ** 2) + ((br[1] - bl[1]) ** 2))
    widthB = np.sqrt(((tr[0] - tl[0]) ** 2) + ((tr[1] - tl[1]) ** 2))
    maxWidth = max(int(widthA), int(widthB))
    # compute the height of the new image, which will be the
    # maximum distance between the top-right and bottom-right
    # y-coordinates or the top-left and bottom-left y-coordinates
    heightA = np.sqrt(((tr[0] - br[0]) ** 2) + ((tr[1] - br[1]) ** 2))
    heightB = np.sqrt(((tl[0] - bl[0]) ** 2) + ((tl[1] - bl[1]) ** 2))
    maxHeight = max(int(heightA), int(heightB))
    # now that we have the dimensions of the new image, construct
    # the set of destination points to obtain a "birds eye view",
    # (i.e. top-down view) of the image, again specifying points
    # in the top-left, top-right, bottom-right, and bottom-left
    # order
    dst = np.array([
        [0, 0],
        [maxWidth - 1, 0],
        [maxWidth - 1, maxHeight - 1],
        [0, maxHeight - 1]], dtype = "float32")
    # compute the perspective transform matrix and then apply it
    M = cv2.getPerspectiveTransform(rect, dst)
    warped = cv2.warpPerspective(image, M, (maxWidth, maxHeight))
    # return the warped image
    return warped

order_points函数

def order_points(pts):
    
    rect = np.zeros((4, 2), dtype = "float32")
    # the top-left point will have the smallest sum, whereas
    # the bottom-right point will have the largest sum
    s = pts.sum(axis = 1)
    rect[0] = pts[np.argmin(s)]
    rect[2] = pts[np.argmax(s)]
    # now, compute the difference between the points, the
    # top-right point will have the smallest difference,
    # whereas the bottom-left will have the largest difference
    diff = np.diff(pts, axis = 1)
    rect[1] = pts[np.argmin(diff)]
    rect[3] = pts[np.argmax(diff)]
    # return the ordered coordinates
    return rect    

解决方案:利用透视变换逆矩阵映射边界框

你的思路完全正确,通过透视变换的逆矩阵可以将变换后图像中的坐标映射回原始图像。具体实现步骤如下:

1. 修改four_point_transform函数,返回变换矩阵及逆矩阵

原函数只返回变换后的图像,我们需要让它同时返回透视变换矩阵M和其逆矩阵M_inv,后续用逆矩阵做坐标映射:

def four_point_transform(image, pts):
    rect = order_points(pts)
    (tl, tr, br, bl) = rect
    widthA = np.sqrt(((br[0] - bl[0]) ** 2) + ((br[1] - bl[1]) ** 2))
    widthB = np.sqrt(((tr[0] - tl[0]) ** 2) + ((tr[1] - tl[1]) ** 2))
    maxWidth = max(int(widthA), int(widthB))
    heightA = np.sqrt(((tr[0] - br[0]) ** 2) + ((tr[1] - br[1]) ** 2))
    heightB = np.sqrt(((tl[0] - bl[0]) ** 2) + ((tl[1] - bl[1]) ** 2))
    maxHeight = max(int(heightA), int(heightB))

    dst = np.array([
        [0, 0],
        [maxWidth - 1, 0],
        [maxWidth - 1, maxHeight - 1],
        [0, maxHeight - 1]], dtype = "float32")
    
    M = cv2.getPerspectiveTransform(rect, dst)
    # 计算逆矩阵
    M_inv = cv2.getPerspectiveTransform(dst, rect)
    warped = cv2.warpPerspective(image, M, (maxWidth, maxHeight))
    
    # 返回变换后的图像、变换矩阵、逆矩阵
    return warped, M, M_inv

2. 修改find_answer函数,添加边界框映射逻辑

在获取到逆矩阵后,对每个检测到的涂黑选项,提取其在变换后图像中的矩形顶点,通过逆矩阵映射回原始图像,再绘制边界框:

def find_answer(original_img, dst):
    gray_dst = cv2.cvtColor(dst,cv2.COLOR_BGR2GRAY) 
    blurred_dst = cv2.GaussianBlur(gray_dst,(3,3),0)
    edged_dst = cv2.Canny(blurred_dst,75,200)
    black_threshold = np.sum(gray_dst[0:37,0:37])/(37*37)
    cnts = cv2.findContours(edged_dst,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)
    cnts = imutils.grab_contours(cnts)
    docCnt = None
    if len(cnts):
        cnts = sorted(cnts,key=cv2.contourArea,reverse=True)

    for c in cnts:
        peri = cv2.arcLength(c,True)
        approx = cv2.approxPolyDP(c,0.02*peri,True)
        if(len(approx))==4:
            docCnt = approx
            break
    
    # 获取变换后的图像及逆矩阵
    paper, M, M_inv = four_point_transform(dst, docCnt.reshape(4,2))
    warped, _, _ = four_point_transform(gray_dst, docCnt.reshape(4,2)) 
    thresh = cv2.threshold(warped,0,255,cv2.THRESH_BINARY_INV|cv2.THRESH_OTSU)[1]
    cnts = cv2.findContours(thresh,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)
    cnts = imutils.grab_contours(cnts)
    cnts = sorted(cnts,key=cv2.contourArea,reverse=True)
    answers = {}
    P=True
    
    for i in range(len(q)):
        li = []
        for j  in range(len(q[i])):
            area =  q[i][j][2]*q[i][j][3]
            y1 = q[i][j][0]
            x1 = q[i][j][1]
            y2 = q[i][j][0] + q[i][j][2]
            x2 = q[i][j][1] +q[i][j][3]
            sum = np.sum(warped[x1:x2,y1:y2])
            
            if sum/area <black_threshold:
                P = False
                print('i:',i,'j:',j,'sum is:',sum)
                print('thersh is:',area*188)
                li.extend([4-j])
                
                # 提取变换后图像中选项框的四个顶点
                warped_rect = np.array([
                    [y1, x1],
                    [y2, x1],
                    [y2, x2],
                    [y1, x2]
                ], dtype="float32")
                
                # 将顶点转换为齐次坐标,用于透视变换
                warped_rect_homogeneous = np.hstack((warped_rect, np.ones((4,1), dtype="float32")))
                # 用逆矩阵映射回原始图像坐标
                original_rect_homogeneous = M_inv @ warped_rect_homogeneous.T
                # 转换回非齐次坐标(除以w分量)
                original_rect = (original_rect_homogeneous[:2] / original_rect_homogeneous[2]).T
                # 转换为整数坐标,用于绘制
                original_rect = original_rect.astype(np.int32)
                
                # 在原始图像上绘制红色边界框
                cv2.drawContours(original_img, [original_rect], -1, (0,0,255), 2)
        
        answers[i+1] = li
    
    # 返回答案和带有边界框的原始图像
    return answers, original_img

3. 调用示例

在主程序中调用find_answer时,传入原始图像和处理后的图像,最后可以保存或显示带有边界框的原始图像:

import cv2
import imutils
import numpy as np

# 假设q是预定义的选项坐标列表
q = [...]

# 读取原始图像
original_img = cv2.imread("answer_sheet.jpg")
dst = original_img.copy()

answers, marked_img = find_answer(original_img, dst)

# 显示或保存结果
cv2.imshow("Marked Original Image", marked_img)
cv2.waitKey(0)
cv2.imwrite("marked_answer_sheet.jpg", marked_img)

内容的提问来源于stack exchange,提问作者Reza shahriari

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 12:05:54