You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python中Tesseract OCR识别视频帧数字出错问题求助

问题:识别视频帧时间戳出错,无法计算实际录制时长

我需要计算MP4视频的实际录制时长,但部分视频传输中丢失了元数据,所以只能通过识别视频首尾帧上的时间戳来获取时长。但用pytesseract.image_to_string识别时,经常出现错误或不一致的结果,比如识别出无效时间19:12:66。我已经尝试过单独识别每个数字、调整图像二值化方式和pytesseract参数,但问题依然存在。相关代码如下:

import cv2
import pytesseract
import os

def digit_detect(image):
    text = pytesseract.image_to_string(image, config='--psm 6 --oem 3 -c tessedit_char_whitelist=0123456789:')
    return text

def resize_roi(image, x1 = 131, y1 = 11, x2 = 228, y2  = 32):
    roi = image[y1:y2, x1:x2]
    return roi

def preprocess_image(image):
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
    return binary

# def extract_time_from_image(image):

#     regions = [
#         (131, 11, 142, 31, '012'),         # Tens of hours (0-2)
#         (142, 11, 155, 31, '0123456789'),  # Units of hours (0-9)
#         (163, 11, 179, 31, '012345'),      # Tens of minutes (0-5)
#         (179, 11, 193, 31, '0123456789'),  # Units of minutes (0-9)
#         (202, 11, 215, 31, '012345'),      # Tens of seconds (0-5)
#         (215, 11, 226, 31, '0123456789')   # Units of seconds (0-9)
#     ]

#     digits = []

#     for (x1, y1, x2, y2, whitelist) in regions:

#         preprocess = preprocess_image(image)
      
#         resized_roi = resize_roi(preprocess, x1, y1, x2, y2)


#         custom_config = f'--psm 6 --oem 3 -c tessedit_char_whitelist={whitelist}'
#         digit = pytesseract.image_to_string(resized_roi, config=custom_config)
      
#         digits.append(digit)

#     return digits
    

folder_path = 'data/output_rec/rkbt/1' 
load_path = "data2"

if not os.path.isdir(folder_path):
    print(f"Error1")
    exit()

video_files = [f for f in os.listdir(folder_path) if f.endswith('.mp4')]

for video_file in video_files:
    video_path = os.path.join(folder_path, video_file)

    cap = cv2.VideoCapture(video_path)

    if not cap.isOpened():
        print(f"Error2")
        continue

    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))

    ret, first_frame = cap.read()
    if not ret:
        print(f"Error3")
        cap.release()
        continue

    cap.set(cv2.CAP_PROP_POS_FRAMES, total_frames - 1)

    ret, last_frame = cap.read()
    if not ret:
        print(f"Error3")
        cap.release()
        continue

    cap.release()

    first_frame = resize_roi(first_frame)
    last_frame = resize_roi(last_frame)

    first_frame = preprocess_image(first_frame)
    last_frame = preprocess_image(last_frame)

    # print(extract_time_from_image(first_frame))
    # print(extract_time_from_image(last_frame))

    first_frame_path = os.path.join(load_path, f"{os.path.splitext(video_file)[0]}_first_frame.jpg")
    last_frame_path = os.path.join(load_path, f"{os.path.splitext(video_file)[0]}_last_frame.jpg")

    print(f"the time is calculated from '{first_frame_path}'", digit_detect(first_frame))
    print(f"the time is calculated from '{last_frame_path}'", digit_detect(last_frame))

    cv2.imwrite(first_frame_path, first_frame)
    cv2.imwrite(last_frame_path, last_frame)

    print(f"Saved images with the first and last frames for '{video_file}'")

优化方案

1. 增强图像预处理

当前二值化适配性不足,试试这些改进:

def preprocess_image(image):
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    # 直方图均衡化提升文字与背景对比度
    gray = cv2.equalizeHist(gray)
    _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
    # 开运算清理画面小噪点
    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2,2))
    binary = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel)
    # 反转图像让文字为黑色(Tesseract对黑底白字识别精度更高)
    binary = cv2.bitwise_not(binary)
    return binary

2. 修复单字符识别逻辑

启用单字符识别模式(--psm 10),并放大ROI提升小字符清晰度,修复你注释的extract_time_from_image函数:

def extract_time_from_image(image):
    regions = [
        (131, 11, 142, 31, '012'),         # 小时十位(0-2)
        (142, 11, 155, 31, '0123456789'),  # 小时个位(0-9)
        (163, 11, 179, 31, '012345'),      # 分钟十位(0-5)
        (179, 11, 193, 31, '0123456789'),  # 分钟个位(0-9)
        (202, 11, 215, 31, '012345'),      # 秒十位(0-5)
        (215, 11, 226, 31, '0123456789')   # 秒个位(0-9)
    ]
    digits = []
    # 只预处理一次图像,避免重复计算
    preprocessed = preprocess_image(image)
    for (x1, y1, x2, y2, whitelist) in regions:
        roi = preprocessed[y1:y2, x1:x2]
        # 放大2倍提升小字符识别率
        roi = cv2.resize(roi, None, fx=2, fy=2, interpolation=cv2.INTER_CUBIC)
        # 单字符识别模式,针对性更强
        custom_config = f'--psm 10 --oem 3 -c tessedit_char_whitelist={whitelist}'
        digit = pytesseract.image_to_string(roi, config=custom_config).strip()
        # 兜底处理:识别为空时补0
        digits.append(digit if digit else '0')
    # 拼接成标准时间格式
    return f"{digits[0]}{digits[1]}:{digits[2]}{digits[3]}:{digits[4]}{digits[5]}"

3. 后处理校验修正

识别后强制校验时间合法性,修正无效值:

def validate_time(time_str):
    try:
        h, m, s = map(int, time_str.split(':'))
        # 修正超出范围的时间值
        h = max(0, h)
        m = max(0, min(m, 59))
        s = max(0, min(s, 59))
        return f"{h:02d}:{m:02d}:{s:02d}"
    except:
        return None

4. 备选:模板匹配

如果时间戳字体、位置固定,用模板匹配替代OCR稳定性更高:

  • 提前从视频帧中截取0-9和冒号的模板图像
  • 对每个数字区域遍历模板计算匹配度,取最高结果:
def match_template(roi, templates):
    max_val = 0
    best_match = '0'
    for char, template in templates.items():
        res = cv2.matchTemplate(roi, template, cv2.TM_CCOEFF_NORMED)
        _, val, _, _ = cv2.minMaxLoc(res)
        if val > max_val:
            max_val = val
            best_match = char
    # 匹配度低于0.8时返回默认值,避免错误识别
    return best_match if max_val > 0.8 else '0'

内容的提问来源于stack exchange,提问作者Ernán

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 21:22:31