使用CV2+Pytesseract检测文本时误将阴影识别为文本求助
问题
在Google Colab中编写脚本,使用CV2和Pytesseract检测.mp4视频中的文本并模糊处理后输出,但眼睛、嘴巴、阴影等非文本内容被误识别并模糊。改为显示边界框后,发现几乎所有阴影都被标记为文本,调整psm配置无明显改善。尝试改用EAST模型时,遇到“qt.qpa.plugin: Could not load the Qt platform plugin "xcb" in "" even though it was found”错误(此为独立问题)。
附上当前使用的脚本:
import cv2 import numpy as np import pytesseract # Set the path to the tesseract executable file pytesseract.pytesseract.tesseract_cmd = ( r'/usr/bin/tesseract' ) def process_frame(frame): # Convert the frame to grayscale gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY) gray = cv2.medianBlur(gray, 3) # Perform adaptive thresholding to create a binary image binary = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY, 71, 20) # Apply erosion to remove small noise kernel = np.ones((3, 3), np.uint8) binary = cv2.erode(binary, kernel, iterations=1) # Find contours in the binary image contours, _ = cv2.findContours(binary, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE) # Loop over the contours for contour in contours: # Get the bounding rectangle of the contour x, y, w, h = cv2.boundingRect(contour) # Only process the contour if it is small (to avoid processing faces) if w * h < 20000: # Extract the region of interest (ROI) from the frame roi = frame[y:y+h, x:x+w] # Convert the ROI to grayscale gray_roi = cv2.cvtColor(roi, cv2.COLOR_BGR2GRAY) # Perform adaptive thresholding on the ROI to create a binary image binary_roi = cv2.adaptiveThreshold(gray_roi, 255, cv2.ADAPTIVE_THRESH_MEAN_C, cv2.THRESH_BINARY, 71, 20) # Same changes as for the main binary image # Use pytesseract to recognize text in the ROI config = r'--psm 6 --oem 3 -c tessedit_char_whitelist=0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ --tessdata-dir "/usr/share/tesseract-ocr/4.00/tessdata"' text = pytesseract.image_to_string(binary_roi, config=config) # Add the text and bounding box to the output frame cv2.putText(frame, text, (x, y - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 0, 255), 2) cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 0, 255), 2) return frame # Set the input and output file paths input_path = ("/content/drive/Shareddrives/325.mp4") # output_path = ("/content/" + CreativeADID + "_BoundingBoxes-model-1.mp4") output_path = ("/content/testing_BoundingBoxes.mp4") # Open the input video file cap = cv2.VideoCapture(input_path) # Get the video codec and FPS information fourcc = cv2.VideoWriter_fourcc(*'mp4v') fps = cap.get(cv2.CAP_PROP_FPS) # Get the frame size of the input video frame_width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) frame_height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) # Create a VideoWriter object to write the output video file out = cv2.VideoWriter(output_path, fourcc, fps, (frame_width, frame_height)) # Process each frame of the input video while True: # Read a frame from the input video ret, frame = cap.read() # Break the loop if the end of the video is reached if not ret: break # Draw bounding boxes around the text regions of the frame frame = process_frame(frame) # Write the processed frame to the output video file out.write(frame) # Release the resources cap.release() out.release() cv2.destroyAllWindows()
解决方案
1. 优化轮廓筛选规则
当前仅靠面积过滤无法区分文本和阴影/面部特征,补充以下筛选条件:
- 宽高比过滤:文本字符的宽高比通常在0.2-2之间,排除不符合的轮廓:
aspect_ratio = w / float(h) if not (0.2 < aspect_ratio < 2): continue - 面积范围细化:设置最小面积阈值(如100),过滤极小噪声轮廓:
if w * h < 100 or w * h > 20000: continue
2. 改用Tesseract原生文本检测API
放弃“轮廓检测+ROI OCR”的流程,直接用Tesseract的image_to_data获取带置信度的文本框,减少误判:
def process_frame(frame): gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY) # 配置Tesseract参数,保留白名单和psm设置 config = r'--psm 6 --oem 3 -c tessedit_char_whitelist=0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ' # 获取文本检测数据,包含位置和置信度 text_data = pytesseract.image_to_data(gray, output_type=pytesseract.Output.DICT, config=config) box_count = len(text_data['text']) for i in range(box_count): # 只保留置信度>60的结果(阈值可根据实际调整) if int(text_data['conf'][i]) > 60: x, y, w, h = text_data['left'][i], text_data['top'][i], text_data['width'][i], text_data['height'][i] cv2.rectangle(frame, (x, y), (x+w, y+h), (0,0,255), 2) cv2.putText(frame, text_data['text'][i], (x, y-5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0,0,255), 2) return frame
3. 调整图像预处理步骤
当前预处理过度增强了阴影对比度,修改为更适配文本检测的流程:
def process_frame(frame): gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY) # 用高斯模糊替代中值模糊,弱化阴影边缘 gray = cv2.GaussianBlur(gray, (3,3), 0) # 改用高斯自适应阈值+反二值化,减少阴影干扰 binary = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY_INV, 11, 2) # 可选:用膨胀替代腐蚀,增强文本轮廓 kernel = np.ones((2,2), np.uint8) binary = cv2.dilate(binary, kernel, iterations=1) # 后续轮廓检测或Tesseract流程... return frame
4. EAST模型错误修复(可选)
Colab环境缺少Qt依赖导致错误,执行以下命令安装:
!apt-get install -y libxcb-xinerama0
内容的提问来源于stack exchange,提问作者Phil Hawkins
相关产品推荐
相关产品推荐

