如何在原始图像中绘制变换后检测到的答题卡选项边界框?
答题卡自动批改:将变换后图像的选项边界框映射回原始图像
我正在编写四选项答题卡自动批改代码,需要为学生选中的涂黑选项绘制边界框。目前使用cv2.findContours、imutils.grab_contours及four_point_transform获取自定义答题卡的最大矩形区域以提取选项;已能在变换后的俯视图像中绘制选项边界框,但不知如何在原始图像中实现该操作。猜测可使用透视变换矩阵的逆矩阵来实现,但不确定是否可行及具体操作方法,相关代码如下:
find_answer函数
def find_answer(dst,): gray_dst = cv2.cvtColor(dst,cv2.COLOR_BGR2GRAY) blurred_dst = cv2.GaussianBlur(gray_dst,(3,3),0) edged_dst = cv2.Canny(blurred_dst,75,200) black_threshold = np.sum(gray_dst[0:37,0:37])/(37*37) cnts = cv2.findContours(edged_dst,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE) cnts = imutils.grab_contours(cnts) docCnt = None if len(cnts): cnts = sorted(cnts,key=cv2.contourArea,reverse=True) for c in cnts: peri = cv2.arcLength(c,True) approx = cv2.approxPolyDP(c,0.02*peri,True) if(len(approx))==4: docCnt = approx break paper = four_point_transform(dst,docCnt.reshape(4,2)) warped = four_point_transform(gray_dst,docCnt.reshape(4,2)) thresh = cv2.threshold(warped,0,255,cv2.THRESH_BINARY_INV|cv2.THRESH_OTSU)[1] cnts = cv2.findContours(thresh,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE) cnts = imutils.grab_contours(cnts) cnts = sorted(cnts,key=cv2.contourArea,reverse=True) answers = {} P=True for i in range(len(q)): w = q[i][0][2] h = q[i][0][3] area = w*h li = [] for j in range(len(q[i])): area = q[i][j][2]*q[i][j][3] y1 = q[i][j][0] x1 = q[i][j][1] y2 = q[i][j][0] + q[i][j][2] x2 = q[i][j][1] +q[i][j][3] sum = np.sum(warped[x1:x2,y1:y2]) #print(sum) #print(w,h) if sum/area <black_threshold: P = False print('i:',i,'j:',j,'sum is:',sum) print('thersh is:',area*188) li.extend([4-j]) answers[i+1] = li return answers
four_point_transform函数
def four_point_transform(image, pts): rect = order_points(pts) (tl, tr, br, bl) = rect # compute the width of the new image, which will be the # maximum distance between bottom-right and bottom-left # x-coordiates or the top-right and top-left x-coordinates widthA = np.sqrt(((br[0] - bl[0]) ** 2) + ((br[1] - bl[1]) ** 2)) widthB = np.sqrt(((tr[0] - tl[0]) ** 2) + ((tr[1] - tl[1]) ** 2)) maxWidth = max(int(widthA), int(widthB)) # compute the height of the new image, which will be the # maximum distance between the top-right and bottom-right # y-coordinates or the top-left and bottom-left y-coordinates heightA = np.sqrt(((tr[0] - br[0]) ** 2) + ((tr[1] - br[1]) ** 2)) heightB = np.sqrt(((tl[0] - bl[0]) ** 2) + ((tl[1] - bl[1]) ** 2)) maxHeight = max(int(heightA), int(heightB)) # now that we have the dimensions of the new image, construct # the set of destination points to obtain a "birds eye view", # (i.e. top-down view) of the image, again specifying points # in the top-left, top-right, bottom-right, and bottom-left # order dst = np.array([ [0, 0], [maxWidth - 1, 0], [maxWidth - 1, maxHeight - 1], [0, maxHeight - 1]], dtype = "float32") # compute the perspective transform matrix and then apply it M = cv2.getPerspectiveTransform(rect, dst) warped = cv2.warpPerspective(image, M, (maxWidth, maxHeight)) # return the warped image return warped
order_points函数
def order_points(pts): rect = np.zeros((4, 2), dtype = "float32") # the top-left point will have the smallest sum, whereas # the bottom-right point will have the largest sum s = pts.sum(axis = 1) rect[0] = pts[np.argmin(s)] rect[2] = pts[np.argmax(s)] # now, compute the difference between the points, the # top-right point will have the smallest difference, # whereas the bottom-left will have the largest difference diff = np.diff(pts, axis = 1) rect[1] = pts[np.argmin(diff)] rect[3] = pts[np.argmax(diff)] # return the ordered coordinates return rect
解决方案:利用透视变换逆矩阵映射边界框
你的思路完全正确,通过透视变换的逆矩阵可以将变换后图像中的坐标映射回原始图像。具体实现步骤如下:
1. 修改four_point_transform函数,返回变换矩阵及逆矩阵
原函数只返回变换后的图像,我们需要让它同时返回透视变换矩阵M和其逆矩阵M_inv,后续用逆矩阵做坐标映射:
def four_point_transform(image, pts): rect = order_points(pts) (tl, tr, br, bl) = rect widthA = np.sqrt(((br[0] - bl[0]) ** 2) + ((br[1] - bl[1]) ** 2)) widthB = np.sqrt(((tr[0] - tl[0]) ** 2) + ((tr[1] - tl[1]) ** 2)) maxWidth = max(int(widthA), int(widthB)) heightA = np.sqrt(((tr[0] - br[0]) ** 2) + ((tr[1] - br[1]) ** 2)) heightB = np.sqrt(((tl[0] - bl[0]) ** 2) + ((tl[1] - bl[1]) ** 2)) maxHeight = max(int(heightA), int(heightB)) dst = np.array([ [0, 0], [maxWidth - 1, 0], [maxWidth - 1, maxHeight - 1], [0, maxHeight - 1]], dtype = "float32") M = cv2.getPerspectiveTransform(rect, dst) # 计算逆矩阵 M_inv = cv2.getPerspectiveTransform(dst, rect) warped = cv2.warpPerspective(image, M, (maxWidth, maxHeight)) # 返回变换后的图像、变换矩阵、逆矩阵 return warped, M, M_inv
2. 修改find_answer函数,添加边界框映射逻辑
在获取到逆矩阵后,对每个检测到的涂黑选项,提取其在变换后图像中的矩形顶点,通过逆矩阵映射回原始图像,再绘制边界框:
def find_answer(original_img, dst): gray_dst = cv2.cvtColor(dst,cv2.COLOR_BGR2GRAY) blurred_dst = cv2.GaussianBlur(gray_dst,(3,3),0) edged_dst = cv2.Canny(blurred_dst,75,200) black_threshold = np.sum(gray_dst[0:37,0:37])/(37*37) cnts = cv2.findContours(edged_dst,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE) cnts = imutils.grab_contours(cnts) docCnt = None if len(cnts): cnts = sorted(cnts,key=cv2.contourArea,reverse=True) for c in cnts: peri = cv2.arcLength(c,True) approx = cv2.approxPolyDP(c,0.02*peri,True) if(len(approx))==4: docCnt = approx break # 获取变换后的图像及逆矩阵 paper, M, M_inv = four_point_transform(dst, docCnt.reshape(4,2)) warped, _, _ = four_point_transform(gray_dst, docCnt.reshape(4,2)) thresh = cv2.threshold(warped,0,255,cv2.THRESH_BINARY_INV|cv2.THRESH_OTSU)[1] cnts = cv2.findContours(thresh,cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE) cnts = imutils.grab_contours(cnts) cnts = sorted(cnts,key=cv2.contourArea,reverse=True) answers = {} P=True for i in range(len(q)): li = [] for j in range(len(q[i])): area = q[i][j][2]*q[i][j][3] y1 = q[i][j][0] x1 = q[i][j][1] y2 = q[i][j][0] + q[i][j][2] x2 = q[i][j][1] +q[i][j][3] sum = np.sum(warped[x1:x2,y1:y2]) if sum/area <black_threshold: P = False print('i:',i,'j:',j,'sum is:',sum) print('thersh is:',area*188) li.extend([4-j]) # 提取变换后图像中选项框的四个顶点 warped_rect = np.array([ [y1, x1], [y2, x1], [y2, x2], [y1, x2] ], dtype="float32") # 将顶点转换为齐次坐标,用于透视变换 warped_rect_homogeneous = np.hstack((warped_rect, np.ones((4,1), dtype="float32"))) # 用逆矩阵映射回原始图像坐标 original_rect_homogeneous = M_inv @ warped_rect_homogeneous.T # 转换回非齐次坐标(除以w分量) original_rect = (original_rect_homogeneous[:2] / original_rect_homogeneous[2]).T # 转换为整数坐标,用于绘制 original_rect = original_rect.astype(np.int32) # 在原始图像上绘制红色边界框 cv2.drawContours(original_img, [original_rect], -1, (0,0,255), 2) answers[i+1] = li # 返回答案和带有边界框的原始图像 return answers, original_img
3. 调用示例
在主程序中调用find_answer时,传入原始图像和处理后的图像,最后可以保存或显示带有边界框的原始图像:
import cv2 import imutils import numpy as np # 假设q是预定义的选项坐标列表 q = [...] # 读取原始图像 original_img = cv2.imread("answer_sheet.jpg") dst = original_img.copy() answers, marked_img = find_answer(original_img, dst) # 显示或保存结果 cv2.imshow("Marked Original Image", marked_img) cv2.waitKey(0) cv2.imwrite("marked_answer_sheet.jpg", marked_img)
内容的提问来源于stack exchange,提问作者Reza shahriari
相关产品推荐
相关产品推荐

