You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用EasyOCR与Tesseract检测图像中的旋转文本?

解决旋转文本的检测与识别问题

你当前用EasyOCR检测文本区域、Tesseract识别的方案无法处理旋转文本,核心原因是提取的文本区域未做旋转校正,导致Tesseract无法识别倾斜/旋转的文字。以下是修改后的实现方案:

原代码

# Text detection with easyocr and recognition with tesseract
import cv2
import easyocr
import pytesseract
pytesseract.pytesseract.tesseract_cmd = "C:\\Program Files\\Tesseract-OCR\\tesseract.exe"

# Load image
image = cv2.imread('image 18.jpg')

# Convert image to grayscale
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

# Initialize the EasyOCR text detector
reader = easyocr.Reader(['en'])

# Detect text in image
results = reader.readtext(gray)

# Image preprocessing for Tesseract
_, processed_image = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY | cv2.THRESH_OTSU)

# Recover text regions detected by EasyOCR
for (bbox, _, _) in results:
    # Extract coordinates from bounding box
    x_min, y_min = map(int, bbox[0])
    x_max, y_max = map(int, bbox[2])
    # Check image boundaries
    x_min = max(0, x_min)
    y_min = max(0, y_min)
    x_max = min(image.shape[1], x_max)
    y_max = min(image.shape[0], y_max)
    # Extract text region
    if x_max>x_min and y_max>y_min :
        text_region = processed_image[y_min:y_max, x_min:x_max]

        # Text recognition with Tesseract
        text = pytesseract.image_to_string(text_region, lang='eng')

        # Draw the bounding box on the image
        cv2.rectangle(image, (x_min, y_min), (x_max, y_max), (0, 255, 0), 2)

        # Display text detected by Tesseract
        cv2.putText(image, text, (x_min, y_min - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)


# Display image with bounding boxes and detected text
cv2.imwrite('Text Detection_2.jpg', image)

修改后的代码(支持旋转文本识别)

import cv2
import easyocr
import pytesseract
import math

pytesseract.pytesseract.tesseract_cmd = "C:\\Program Files\\Tesseract-OCR\\tesseract.exe"

# 加载图像
image = cv2.imread('image 18.jpg')
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

# 初始化EasyOCR检测器
reader = easyocr.Reader(['en'])
results = reader.readtext(gray)

# 图像预处理
_, processed_image = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY | cv2.THRESH_OTSU)

for (bbox, _, confidence) in results:
    # 获取文本框边界
    x_min, y_min = map(int, bbox[0])
    x_max, y_max = map(int, bbox[2])
    x_min = max(0, x_min)
    y_min = max(0, y_min)
    x_max = min(image.shape[1], x_max)
    y_max = min(image.shape[0], y_max)

    if x_max > x_min and y_max > y_min:
        text_region = processed_image[y_min:y_max, x_min:x_max]
        
        # 计算文本的旋转角度(基于EasyOCR返回的文本框前两个点)
        dx = bbox[1][0] - bbox[0][0]
        dy = bbox[1][1] - bbox[0][1]
        angle = math.degrees(math.atan2(dy, dx))

        # 旋转文本区域,校正为水平方向
        h, w = text_region.shape[:2]
        center = (w // 2, h // 2)
        rotation_matrix = cv2.getRotationMatrix2D(center, angle, 1.0)
        rotated_region = cv2.warpAffine(text_region, rotation_matrix, (w, h), 
                                       flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE)
        
        # 用Tesseract识别校正后的文本
        text = pytesseract.image_to_string(rotated_region, lang='eng').strip()

        # 绘制边界框和识别结果
        cv2.rectangle(image, (x_min, y_min), (x_max, y_max), (0, 255, 0), 2)
        cv2.putText(image, text, (x_min, y_min - 10), 
                   cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)

# 保存结果图像
cv2.imwrite('Text Detection_rotated.jpg', image)

关键修改说明

  • 角度计算:通过EasyOCR返回的文本框四个顶点,利用前两个点的坐标差计算文本的倾斜角度
  • 旋转校正:对每个提取的文本区域进行旋转,将倾斜文本转正为水平方向,确保Tesseract能正常识别
  • 保留预处理:维持原有的二值化预处理逻辑,叠加旋转步骤解决旋转文本识别问题

内容的提问来源于stack exchange,提问作者Kawtar Zidouh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.19 05:37:01