如何用OpenCV修复含字母框状图像透视?曲线字母适配问题
问题背景
我需要处理一张包含框内字母的图像,修复其透视畸变以适配字母识别模型。
现有预处理代码
def preprocess(image): # 转为灰度图 image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) # 降噪流程开始 # 应用CLAHE(提升图像亮度均匀性) clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8)) image = clahe.apply(image) cv2.imwrite("preprocess_stages/0_clahe.png", image) # 应用非局部均值降噪(去除噪声但会丢失部分细节) image = cv2.fastNlMeansDenoising(image, None, 10, 7, 21) cv2.imwrite("preprocess_stages/1_nl_means_denoising.png", image) # 应用高斯模糊(去除噪声但会模糊边缘) image = cv2.GaussianBlur(image, (5, 5), 0) cv2.imwrite("preprocess_stages/2_gaussian_blur.png", image) # 降噪流程结束 # 应用OTSU二值化(让字母变为白色) _, image = cv2.threshold(image, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU) cv2.imwrite("preprocess_stages/3_otsu.png", image) # 形态学腐蚀(去除噪声和小目标) kernel = np.ones((10, 10), np.uint8) image = cv2.erode(image, kernel, iterations=1) cv2.imwrite("preprocess_stages/4_morphological_erosion.png", image) return image
透视修复实现
查阅大量示例后,发现大多需要手动输入4个角点,自动计算角点的案例极少且不适用于我的场景,于是自行编写代码。将形态学开运算改为简单腐蚀后效果提升,代码如下:
def fix_perspective_convex_hull(preprocessed): # RETR_TREE检索所有轮廓并重建完整的嵌套轮廓层级 contours, _ = cv2.findContours(preprocessed, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE) # 绘制所有轮廓到新图像 image = cv2.cvtColor(preprocessed, cv2.COLOR_GRAY2BGR) cv2.drawContours(image, contours, -1, (0, 255, 0), 2) cv2.imwrite("perspective_stages/0_contours.png", image) for i, contour in enumerate(contours): hull = cv2.convexHull(contour) # 近似为四边形 epsilon = 0.1 * cv2.arcLength(hull, True) approximated = cv2.approxPolyDP(hull, epsilon, True) # 绘制近似轮廓 image = cv2.cvtColor(preprocessed, cv2.COLOR_GRAY2BGR) cv2.drawContours(image, [approximated], -1, (255, 0, 0), 1) cv2.imwrite("perspective_stages/1_approximated_contour_{}.png".format(i), image) if len(approximated) == 4: image = cv2.cvtColor(preprocessed, cv2.COLOR_GRAY2BGR) cv2.drawContours(image, [approximated], -1, (0, 0, 255), 2) cv2.imwrite("perspective_stages/2_final_contour.png", image) letter_hull = [a[0] for a in approximated] # 去除多余维度 [[x, y]] -> [x, y] break # 创建矩形点占位符 rectangle = np.zeros((4, 2), dtype="float32") # 左上角点的x+y和最小,右下角点的x+y和最大 s = np.sum(letter_hull, axis=1) rectangle[0] = letter_hull[np.argmin(s)] rectangle[2] = letter_hull[np.argmax(s)] # 右上角点的x-y差最小,左下角点的x-y差最大 d = np.diff(letter_hull, axis=1) rectangle[1] = letter_hull[np.argmin(d)] rectangle[3] = letter_hull[np.argmax(d)] # 根据点计算新图像的宽高 (top_left, top_right, bottom_right, bottom_left) = rectangle # 宽度取顶部或底部两点距离的最大值(勾股定理) width_top = np.sqrt(((top_right[0] - top_left[0]) ** 2) + ((top_right[1] - top_left[1]) ** 2)) width_bottom = np.sqrt(((bottom_right[0] - bottom_left[0]) ** 2) + ((bottom_right[1] - bottom_left[1]) ** 2)) width = max(int(width_top), int(width_bottom)) # 高度取右侧或左侧两点距离的最大值(勾股定理) height_right = np.sqrt(((top_right[0] - bottom_right[0]) ** 2) + ((top_right[1] - bottom_right[1]) ** 2)) height_left = np.sqrt(((top_left[0] - bottom_left[0]) ** 2) + ((top_left[1] - bottom_left[1]) ** 2)) height = max(int(height_right), int(height_left)) # 创建目标点 destination = np.array([ [0, 0], [width - 1, 0], [width - 1, height - 1], [0, height - 1] ], dtype="float32") # 计算透视变换矩阵 matrix = cv2.getPerspectiveTransform(rectangle, destination) warped = cv2.warpPerspective(preprocessed, matrix, (width, height)) # 添加填充 warped = cv2.copyMakeBorder(warped, 10, 10, 10, 10, cv2.BORDER_CONSTANT, value=(0, 0, 0)) # 返回矫正后的图像 return warped
现有效果(直线字母示例)
- 原始图像:

- 预处理后图像:

- 四边形检测结果:

- 透视修复后图像:

- 适配神经网络的反转与缩放图像:

存在的问题
当前代码仅对H这类由直线构成的字母有效,处理S这类曲线字母时效果很差,处理结果如下:
现寻求解决该问题的方案。
内容的提问来源于stack exchange,提问作者R1D3R175
相关产品推荐
相关产品推荐

