You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用CV2+Pytesseract检测文本时误将阴影识别为文本求助

问题

在Google Colab中编写脚本,使用CV2和Pytesseract检测.mp4视频中的文本并模糊处理后输出,但眼睛、嘴巴、阴影等非文本内容被误识别并模糊。改为显示边界框后,发现几乎所有阴影都被标记为文本,调整psm配置无明显改善。尝试改用EAST模型时,遇到“qt.qpa.plugin: Could not load the Qt platform plugin "xcb" in "" even though it was found”错误(此为独立问题)。

附上当前使用的脚本:

import cv2
import numpy as np
import pytesseract

# Set the path to the tesseract executable file
pytesseract.pytesseract.tesseract_cmd = ( r'/usr/bin/tesseract' )

def process_frame(frame):
    # Convert the frame to grayscale
    gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)

    gray = cv2.medianBlur(gray, 3)
    
    # Perform adaptive thresholding to create a binary image
    binary = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY, 71, 20)

    # Apply erosion to remove small noise
    kernel = np.ones((3, 3), np.uint8)
    binary = cv2.erode(binary, kernel, iterations=1)

    # Find contours in the binary image
    contours, _ = cv2.findContours(binary, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE)
    
    # Loop over the contours
    for contour in contours:
        # Get the bounding rectangle of the contour
        x, y, w, h = cv2.boundingRect(contour)
        
        # Only process the contour if it is small (to avoid processing faces)
        if w * h < 20000:
            # Extract the region of interest (ROI) from the frame
            roi = frame[y:y+h, x:x+w]
            
            # Convert the ROI to grayscale
            gray_roi = cv2.cvtColor(roi, cv2.COLOR_BGR2GRAY)
            
            # Perform adaptive thresholding on the ROI to create a binary image
            binary_roi = cv2.adaptiveThreshold(gray_roi, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY, 71, 20)
            # Same changes as for the main binary image
            
            # Use pytesseract to recognize text in the ROI
            config = r'--psm 6 --oem 3 -c tessedit_char_whitelist=0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ --tessdata-dir "/usr/share/tesseract-ocr/4.00/tessdata"'
            text = pytesseract.image_to_string(binary_roi, config=config)
            
            # Add the text and bounding box to the output frame
            cv2.putText(frame, text, (x, y - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 0, 255), 2)
            cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 0, 255), 2)
    
    return frame


# Set the input and output file paths
input_path = ("/content/drive/Shareddrives/325.mp4")
# output_path = ("/content/" + CreativeADID + "_BoundingBoxes-model-1.mp4")
output_path = ("/content/testing_BoundingBoxes.mp4")

# Open the input video file
cap = cv2.VideoCapture(input_path)

# Get the video codec and FPS information
fourcc = cv2.VideoWriter_fourcc(*'mp4v')
fps = cap.get(cv2.CAP_PROP_FPS)

# Get the frame size of the input video
frame_width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
frame_height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))

# Create a VideoWriter object to write the output video file
out = cv2.VideoWriter(output_path, fourcc, fps, (frame_width, frame_height))

# Process each frame of the input video
while True:
    # Read a frame from the input video
    ret, frame = cap.read()

    # Break the loop if the end of the video is reached
    if not ret:
        break

    # Draw bounding boxes around the text regions of the frame
    frame = process_frame(frame)

    # Write the processed frame to the output video file
    out.write(frame)

# Release the resources
cap.release()
out.release()
cv2.destroyAllWindows()

解决方案

1. 优化轮廓筛选规则

当前仅靠面积过滤无法区分文本和阴影/面部特征,补充以下筛选条件:

  • 宽高比过滤:文本字符的宽高比通常在0.2-2之间,排除不符合的轮廓:
    aspect_ratio = w / float(h)
    if not (0.2 < aspect_ratio < 2):
        continue
    
  • 面积范围细化:设置最小面积阈值(如100),过滤极小噪声轮廓:
    if w * h < 100 or w * h > 20000:
        continue
    

2. 改用Tesseract原生文本检测API

放弃“轮廓检测+ROI OCR”的流程,直接用Tesseract的image_to_data获取带置信度的文本框,减少误判:

def process_frame(frame):
    gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
    # 配置Tesseract参数,保留白名单和psm设置
    config = r'--psm 6 --oem 3 -c tessedit_char_whitelist=0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ'
    # 获取文本检测数据,包含位置和置信度
    text_data = pytesseract.image_to_data(gray, output_type=pytesseract.Output.DICT, config=config)
    
    box_count = len(text_data['text'])
    for i in range(box_count):
        # 只保留置信度>60的结果(阈值可根据实际调整)
        if int(text_data['conf'][i]) > 60:
            x, y, w, h = text_data['left'][i], text_data['top'][i], text_data['width'][i], text_data['height'][i]
            cv2.rectangle(frame, (x, y), (x+w, y+h), (0,0,255), 2)
            cv2.putText(frame, text_data['text'][i], (x, y-5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0,0,255), 2)
    return frame

3. 调整图像预处理步骤

当前预处理过度增强了阴影对比度,修改为更适配文本检测的流程:

def process_frame(frame):
    gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
    # 用高斯模糊替代中值模糊,弱化阴影边缘
    gray = cv2.GaussianBlur(gray, (3,3), 0)
    # 改用高斯自适应阈值+反二值化,减少阴影干扰
    binary = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY_INV, 11, 2)
    # 可选:用膨胀替代腐蚀,增强文本轮廓
    kernel = np.ones((2,2), np.uint8)
    binary = cv2.dilate(binary, kernel, iterations=1)
    
    # 后续轮廓检测或Tesseract流程...
    return frame

4. EAST模型错误修复(可选)

Colab环境缺少Qt依赖导致错误,执行以下命令安装:

!apt-get install -y libxcb-xinerama0

内容的提问来源于stack exchange,提问作者Phil Hawkins

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.22 22:23:11