You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何使用OpenCV实现OCR文字检测与识别,求Python可运行代码示例

基于OpenCV的Google Colab文字检测识别实现方案

1. 环境配置

先在Colab单元中执行以下命令安装依赖:

!apt install tesseract-ocr libtesseract-dev
!pip install opencv-python opencv-contrib-python pytesseract

如果需要识别中文,额外安装中文语言包:

!apt install tesseract-ocr-chi-sim

2. 前置准备

  • 自行获取EAST文本检测预训练模型文件frozen_east_text_detection.pb,上传到Colab文件根目录
  • 上传你需要测试的图片到Colab目录,记录图片路径

3. 完整可运行代码

import cv2
import pytesseract
import numpy as np
from google.colab.patches import cv2_imshow # Colab专用,替代本地环境的cv2.imshow

# -------------------------- 配置参数 --------------------------
EAST_MODEL_PATH = "frozen_east_text_detection.pb"
MIN_CONFIDENCE = 0.5 # 检测置信度阈值,可根据实际效果调整
INPUT_WIDTH = 320
INPUT_HEIGHT = 320
# 识别语言设置:eng=英文,chi_sim=简体中文,eng+chi_sim=同时识别中英文
RECOG_LANG = "eng"

# -------------------------- 加载检测模型 --------------------------
net = cv2.dnn.readNet(EAST_MODEL_PATH)
output_layers = [
    "feature_fusion/Conv_7/Sigmoid",
    "feature_fusion/concat_3"
]

# -------------------------- 文本检测函数 --------------------------
def text_detect(image):
    h, w = image.shape[:2]
    # 图像预处理适配模型输入
    blob = cv2.dnn.blobFromImage(image, 1.0, (INPUT_WIDTH, INPUT_HEIGHT),
                                (123.68, 116.78, 103.94), swapRB=True, crop=False)
    net.setInput(blob)
    scores, geometry = net.forward(output_layers)
    
    rects = []
    confidences = []
    for y in range(scores.shape[2]):
        scores_data = scores[0, 0, y]
        x0_data = geometry[0, 0, y]
        x1_data = geometry[0, 1, y]
        x2_data = geometry[0, 2, y]
        x3_data = geometry[0, 3, y]
        angles_data = geometry[0, 4, y]
        
        for x in range(scores.shape[3]):
            if scores_data[x] < MIN_CONFIDENCE:
                continue
            # 计算检测框坐标偏移
            offset_x = x * 4.0
            offset_y = y * 4.0
            angle = angles_data[x]
            cos = np.cos(angle)
            sin = np.sin(angle)
            h_box = x0_data[x] + x2_data[x]
            w_box = x1_data[x] + x3_data[x]
            # 计算检测框端点坐标
            end_x = int(offset_x + (cos * x1_data[x]) + (sin * x2_data[x]))
            end_y = int(offset_y - (sin * x1_data[x]) + (cos * x2_data[x]))
            start_x = int(end_x - w_box)
            start_y = int(end_y - h_box)
            
            rects.append((start_x, start_y, end_x, end_y))
            confidences.append(float(scores_data[x]))
    # 非极大值抑制去重
    indices = cv2.dnn.NMSBoxes(rects, confidences, MIN_CONFIDENCE, 0.4)
    # 坐标还原到原图尺寸
    r_w = w / float(INPUT_WIDTH)
    r_h = h / float(INPUT_HEIGHT)
    final_boxes = []
    if len(indices) > 0:
        for i in indices.flatten():
            sx, sy, ex, ey = rects[i]
            sx = int(sx * r_w)
            sy = int(sy * r_h)
            ex = int(ex * r_w)
            ey = int(ey * r_h)
            final_boxes.append((sx, sy, ex, ey))
    return final_boxes

# -------------------------- 文本识别函数 --------------------------
def text_recognize(image, boxes):
    results = []
    for (sx, sy, ex, ey) in boxes:
        # 裁剪文本区域并做预处理提升识别准确率
        text_region = image[sy:ey, sx:ex]
        gray = cv2.cvtColor(text_region, cv2.COLOR_BGR2GRAY)
        gray = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY | cv2.THRESH_OTSU)[1]
        # 执行识别
        text = pytesseract.image_to_string(gray, lang=RECOG_LANG)
        text = text.strip()
        if text:
            results.append(((sx, sy, ex, ey), text))
    return results

# -------------------------- 测试执行 --------------------------
if __name__ == "__main__":
    # 替换为你上传的测试图片路径
    image_path = "test.jpg"
    image = cv2.imread(image_path)
    # 检测文本框
    boxes = text_detect(image)
    # 识别文本内容
    results = text_recognize(image, boxes)
    # 可视化结果
    for (box, text) in results:
        sx, sy, ex, ey = box
        cv2.rectangle(image, (sx, sy), (ex, ey), (0, 255, 0), 2)
        cv2.putText(image, text, (sx, sy-10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 255, 0), 2)
    cv2_imshow(image)
    # 打印识别结果
    print("识别到的文本内容:")
    for idx, (box, text) in enumerate(results):
        print(f"{idx+1}. {text}")

4. 效果调整说明

  • 调高MIN_CONFIDENCE参数可以减少误检,调低可以减少漏检
  • 针对模糊、低对比度的图片,可以在检测前增加高斯模糊、直方图均衡化等预处理步骤
  • 针对倾斜的文本,可以在识别前增加倾斜校正逻辑

内容的提问来源于stack exchange,提问作者A.R

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.30 01:45:03