You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于OpenCV的扫描功能:如何修正不完整的旋转文本

文档扫描:缺失/不完整文档的透视变换解决方案

问题说明

现有基于OpenCV的文档扫描逻辑通过approxPolyDP提取4顶点轮廓实现透视变换,仅能处理完整文档图像。当文档存在缺失(如仅显示部分内容)时,无法检测到完整的4顶点轮廓,导致透视变换失效。

效果对比

  • 正常扫描效果:
    正常扫描效果示例图
  • 失效场景(文档缺失):
    失效场景示例图

解决方案思路

通过直线检测+直线延长求交点的方式,即使文档边缘不完整,也能通过提取的文档边界直线延长后计算出四个顶点,再执行透视变换。核心步骤:

  1. 边缘检测后用霍夫直线检测提取文档边界直线
  2. 将直线聚类为水平和垂直两组
  3. 每组筛选关键直线并延长,计算直线交点得到四个顶点
  4. 对顶点排序后执行透视变换

完整实现代码

import numpy as np
import argparse
import cv2
import imutils

ap = argparse.ArgumentParser()
ap.add_argument("-i", "--image", default="scan.jpg", help="输入图像路径")
args = vars(ap.parse_args())

def cv_show(name, img):
    cv2.imshow(name, img)
    cv2.waitKey(0)
    cv2.destroyWindow(name)

def order_points(pts):
    # 初始化有序坐标列表:左上、右上、右下、左下
    rect = np.zeros((4, 2), dtype="float32")
    # 左上角点横纵坐标和最小,右下角和最大
    s = pts.sum(axis=1)
    rect[0] = pts[np.argmin(s)]
    rect[2] = pts[np.argmax(s)]
    # 右上角点横纵坐标差最小,左下角差最大
    diff = np.diff(pts, axis=1)
    rect[1] = pts[np.argmin(diff)]
    rect[3] = pts[np.argmax(diff)]
    return rect

def four_point_transform(image, pts):
    rect = order_points(pts)
    (tl, tr, br, bl) = rect

    # 计算目标图像的宽高
    widthA = np.sqrt(((br[0] - bl[0])**2) + ((br[1] - bl[1])**2))
    widthB = np.sqrt(((tr[0] - tl[0])**2) + ((tr[1] - tl[1])**2))
    maxWidth = max(int(widthA), int(widthB))

    heightA = np.sqrt(((tr[0] - br[0])**2) + ((tr[1] - br[1])**2))
    heightB = np.sqrt(((tl[0] - bl[0])**2) + ((tl[1] - bl[1])**2))
    maxHeight = max(int(heightA), int(heightB))

    # 构造目标点集
    dst = np.array([
        [0, 0],
        [maxWidth - 1, 0],
        [maxWidth - 1, maxHeight - 1],
        [0, maxHeight - 1]], dtype="float32")

    # 计算透视变换矩阵并应用
    M = cv2.getPerspectiveTransform(rect, dst)
    warped = cv2.warpPerspective(image, M, (maxWidth, maxHeight))
    return warped

def get_intersection(line1, line2):
    # 计算两条直线的交点(含延长线交点)
    x1, y1, x2, y2 = line1[0]
    x3, y3, x4, y4 = line2[0]

    # 用直线一般式求解
    a1 = y2 - y1
    b1 = x1 - x2
    c1 = (y2 - y1)*x1 - (x2 - x1)*y1

    a2 = y4 - y3
    b2 = x3 - x4
    c2 = (y4 - y3)*x3 - (x4 - x3)*y3

    det = a1*b2 - a2*b1
    if det == 0:
        return None  # 平行无交点

    x = (b1*c2 - b2*c1)/det
    y = (a2*c1 - a1*c2)/det
    return (int(x), int(y))

def detect_document_corners(image):
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    gray = cv2.GaussianBlur(gray, (5,5), 0)
    edged = cv2.Canny(gray, 50, 150)

    # 霍夫直线检测
    lines = cv2.HoughLinesP(edged, rho=1, theta=np.pi/180, threshold=50, minLineLength=50, maxLineGap=10)
    if lines is None:
        return None

    # 分类水平和垂直直线(基于角度)
    horizontal_lines = []
    vertical_lines = []
    for line in lines:
        x1, y1, x2, y2 = line[0]
        angle = np.arctan2(y2 - y1, x2 - x1) * 180 / np.pi
        # 角度接近0/180为水平,接近90/270为垂直
        if abs(angle) < 10 or abs(angle - 180) < 10:
            horizontal_lines.append(line)
        elif abs(angle - 90) < 10 or abs(angle + 90) < 10:
            vertical_lines.append(line)

    if len(horizontal_lines) < 2 or len(vertical_lines) < 2:
        return None

    # 筛选关键直线:水平取最上/最下,垂直取最左/最右
    horizontal_lines.sort(key=lambda l: (l[0][1] + l[0][3])/2)
    top_h_line = horizontal_lines[0]
    bottom_h_line = horizontal_lines[-1]

    vertical_lines.sort(key=lambda l: (l[0][0] + l[0][2])/2)
    left_v_line = vertical_lines[0]
    right_v_line = vertical_lines[-1]

    # 计算四个交点
    tl = get_intersection(top_h_line, left_v_line)
    tr = get_intersection(top_h_line, right_v_line)
    br = get_intersection(bottom_h_line, right_v_line)
    bl = get_intersection(bottom_h_line, left_v_line)

    if None in [tl, tr, br, bl]:
        return None

    return np.array([tl, tr, br, bl], dtype="float32")

# 主逻辑
image = cv2.imread(args["image"])
ratio = image.shape[0] / 500.0
orig = image.copy()
resized_image = imutils.resize(image, height=500)

# 检测文档四角
corners = detect_document_corners(resized_image)
if corners is None:
    print("无法检测到文档边界")
    exit()

# 还原坐标到原始图像尺寸
corners = corners * ratio

# 执行透视变换
warped = four_point_transform(orig, corners)

# 显示结果
print("步骤1:原始图像")
cv_show("原始图像", imutils.resize(orig, height=650))
print("步骤2:检测到的文档轮廓")
cv2.drawContours(image, [corners.astype(int)], -1, (0,255,0), 2)
cv_show("轮廓", imutils.resize(image, height=650))
print("步骤3:扫描结果")
cv_show("扫描结果", imutils.resize(warped, height=650))

cv2.destroyAllWindows()

关键步骤说明

  1. 直线分类:通过霍夫直线检测提取直线后,根据角度分为水平和垂直两组,对应文档的上下左右边界。
  2. 直线筛选:取水平组中最上、最下的直线,垂直组中最左、最右的直线,模拟文档的完整边界。
  3. 交点计算:即使直线不完整,通过直线一般式求解延长线的交点,得到文档的四角顶点。
  4. 透视变换:沿用原有order_points和four_point_transform函数完成最终的校正。

内容的提问来源于stack exchange,提问作者reddish xia

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 17:44:53