You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何准确检测图像中任意旋转方向的文本?

高效识别任意旋转文本的优化方案

问题背景

需要检测任意旋转方向物品上的文本,尝试过Tesseract、EasyOCR和EAST工具,其中Tesseract效果最接近预期,但旋转文本仍存在识别错误。逐角度旋转图像检测的方案耗时过长(单次运行需70小时),现有代码如下:

import os
import cv2
import pytesseract
import matplotlib.pyplot as plt
from tqdm import tqdm
import pandas as pd

# Directory containing the images
directory = 'Camera2/front'

# Ensure pytesseract can find the tesseract executable
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'  # Adjust path as necessary

# Initialize an empty list to store results
results = []

# Get the list of image files in the directory
image_files = [f for f in os.listdir(directory) if f.endswith('.jpeg') or f.endswith('.jpg')]

def preprocess_image(image):
    # Convert the image to grayscale
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    
    # Apply adaptive thresholding to preprocess the image
    binary = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2)
    
    return gray, binary

def detect_text(image):
    # Preprocess the image
    gray, binary = preprocess_image(image)
    
    # Perform OCR on the preprocessed image
    text = pytesseract.image_to_string(binary, config='--psm 3 -l eng --oem 3')  # Using page segmentation mode 3

    # Check if any text is detected
    return bool(text.strip()), text, gray

def rotate_image(image, angle):
    # Get the image dimensions
    (h, w) = image.shape[:2]
    # Calculate the center of the image
    center = (w / 2, h / 2)
    # Perform the rotation
    matrix = cv2.getRotationMatrix2D(center, angle, 1.0)
    rotated = cv2.warpAffine(image, matrix, (w, h))
    return rotated

# Iterate through each file in the directory with tqdm for progress visualization
for filename in tqdm(image_files, desc="Processing images"):
    filepath = os.path.join(directory, filename)
    
    # Load the current image
    original_image = cv2.imread(filepath)
    
    # Initialize text detection result
    has_text = False
    detected_text = ""
    gray_image = None
    
    # Rotate the image from 0 to 359 degrees
    for angle in tqdm(range(0, 360)):
        rotated_image = rotate_image(original_image, angle)
        has_text, detected_text, gray_image = detect_text(rotated_image)
        
        if has_text:
            break
    
    # Plotting the original and preprocessed images
    fig, axes = plt.subplots(1, 2, figsize=(12, 6))
    
    # Original image
    axes[0].imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))
    axes[0].set_title('Original Image')
    axes[0].axis('off')
    
    # Gray scale image
    if gray_image is not None:
        axes[1].imshow(gray_image, cmap='gray')
        axes[1].set_title('Grayscale Image with Adjusted Thresholding')
        axes[1].axis('off')
    
    plt.tight_layout()
    plt.show()
    
    if has_text:
        print(f"Text detected in {filename}:")
        print(detected_text)
        # Store text in results list if it's longer than 3 characters
        if len(detected_text) > 3:
            image_id = filename.replace('.jpeg', '').replace('.jpg', '')
            results.append({'ID': image_id, 'text': detected_text})
    else:
        print(f"No text detected in {filename}.")

results_df = pd.DataFrame(results)

优化方案

1. 利用Tesseract自带的方向检测功能

Tesseract内置OSD(方向与脚本检测)模块,可直接识别文本旋转角度,无需逐角度尝试。修改代码如下:

def detect_text_orientation(image):
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    # 使用OSD检测文本方向
    osd = pytesseract.image_to_osd(gray, config='--psm 0 -l eng')
    # 解析旋转角度
    angle = int(osd.split('Rotate: ')[1].split('\n')[0])
    return angle

# 替换原逐角度循环逻辑
for filename in tqdm(image_files, desc="Processing images"):
    filepath = os.path.join(directory, filename)
    original_image = cv2.imread(filepath)
    
    has_text = False
    detected_text = ""
    gray_image = None
    
    # 优先用OSD检测并校正
    try:
        angle = detect_text_orientation(original_image)
        rotated_image = rotate_image(original_image, -angle)  # 反向旋转校正
        has_text, detected_text, gray_image = detect_text(rotated_image)
    except:
        # OSD失败时, fallback到四个正方向尝试
        for angle in [0, 90, 180, 270]:
            rotated_image = rotate_image(original_image, angle)
            has_text, detected_text, gray_image = detect_text(rotated_image)
            if has_text:
                break
    
    # 后续绘图、结果存储逻辑保持不变

2. 优化旋转搜索策略

若OSD失效,采用分层搜索替代全角度遍历:

  • 先尝试0°、90°、180°、270°四个正方向
  • 未检测到文本时,再在相邻区间细分(如每15°尝试一次),总尝试次数从360次降至28次左右,大幅缩短耗时。

示例代码片段:

# 分层旋转检测
angles_to_try = [0, 90, 180, 270]
found = False
for angle in angles_to_try:
    rotated_image = rotate_image(original_image, angle)
    has_text, detected_text, gray_image = detect_text(rotated_image)
    if has_text:
        found = True
        break
if not found:
    # 细分0-90度区间
    for angle in range(15, 90, 15):
        rotated_image = rotate_image(original_image, angle)
        has_text, detected_text, gray_image = detect_text(rotated_image)
        if has_text:
            found = True
            break
    # 同理处理90-180、180-270、270-360区间

3. 增强图像预处理效果

针对旋转文本优化预处理步骤,提升Tesseract识别率:

  • 添加高斯模糊去噪
  • 用Otsu阈值替代自适应阈值
  • 形态学操作填补文本间隙

修改后的preprocess_image函数:

def preprocess_image(image):
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    gray = cv2.GaussianBlur(gray, (3,3), 0)  # 去噪
    _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)  # Otsu自动阈值
    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2,2))
    binary = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, kernel)  # 闭合操作增强文本连续性
    return gray, binary

4. 结合文本检测模型定位局部区域

先用EAST或YOLOv8等文本检测模型定位图像中的文本框,计算每个框的倾斜角度,单独旋转校正后再做OCR,避免处理整张图像:

  • 用EAST检测文本框并获取旋转角度
  • 裁剪文本框区域,旋转校正后单独执行Tesseract识别

内容的提问来源于stack exchange,提问作者Agura

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 15:22:17