Python中Tesseract OCR识别视频帧数字出错问题求助
问题:识别视频帧时间戳出错,无法计算实际录制时长
我需要计算MP4视频的实际录制时长,但部分视频传输中丢失了元数据,所以只能通过识别视频首尾帧上的时间戳来获取时长。但用pytesseract.image_to_string识别时,经常出现错误或不一致的结果,比如识别出无效时间19:12:66。我已经尝试过单独识别每个数字、调整图像二值化方式和pytesseract参数,但问题依然存在。相关代码如下:
import cv2 import pytesseract import os def digit_detect(image): text = pytesseract.image_to_string(image, config='--psm 6 --oem 3 -c tessedit_char_whitelist=0123456789:') return text def resize_roi(image, x1 = 131, y1 = 11, x2 = 228, y2 = 32): roi = image[y1:y2, x1:x2] return roi def preprocess_image(image): gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) return binary # def extract_time_from_image(image): # regions = [ # (131, 11, 142, 31, '012'), # Tens of hours (0-2) # (142, 11, 155, 31, '0123456789'), # Units of hours (0-9) # (163, 11, 179, 31, '012345'), # Tens of minutes (0-5) # (179, 11, 193, 31, '0123456789'), # Units of minutes (0-9) # (202, 11, 215, 31, '012345'), # Tens of seconds (0-5) # (215, 11, 226, 31, '0123456789') # Units of seconds (0-9) # ] # digits = [] # for (x1, y1, x2, y2, whitelist) in regions: # preprocess = preprocess_image(image) # resized_roi = resize_roi(preprocess, x1, y1, x2, y2) # custom_config = f'--psm 6 --oem 3 -c tessedit_char_whitelist={whitelist}' # digit = pytesseract.image_to_string(resized_roi, config=custom_config) # digits.append(digit) # return digits folder_path = 'data/output_rec/rkbt/1' load_path = "data2" if not os.path.isdir(folder_path): print(f"Error1") exit() video_files = [f for f in os.listdir(folder_path) if f.endswith('.mp4')] for video_file in video_files: video_path = os.path.join(folder_path, video_file) cap = cv2.VideoCapture(video_path) if not cap.isOpened(): print(f"Error2") continue total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) ret, first_frame = cap.read() if not ret: print(f"Error3") cap.release() continue cap.set(cv2.CAP_PROP_POS_FRAMES, total_frames - 1) ret, last_frame = cap.read() if not ret: print(f"Error3") cap.release() continue cap.release() first_frame = resize_roi(first_frame) last_frame = resize_roi(last_frame) first_frame = preprocess_image(first_frame) last_frame = preprocess_image(last_frame) # print(extract_time_from_image(first_frame)) # print(extract_time_from_image(last_frame)) first_frame_path = os.path.join(load_path, f"{os.path.splitext(video_file)[0]}_first_frame.jpg") last_frame_path = os.path.join(load_path, f"{os.path.splitext(video_file)[0]}_last_frame.jpg") print(f"the time is calculated from '{first_frame_path}'", digit_detect(first_frame)) print(f"the time is calculated from '{last_frame_path}'", digit_detect(last_frame)) cv2.imwrite(first_frame_path, first_frame) cv2.imwrite(last_frame_path, last_frame) print(f"Saved images with the first and last frames for '{video_file}'")
优化方案
1. 增强图像预处理
当前二值化适配性不足,试试这些改进:
def preprocess_image(image): gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) # 直方图均衡化提升文字与背景对比度 gray = cv2.equalizeHist(gray) _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) # 开运算清理画面小噪点 kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2,2)) binary = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel) # 反转图像让文字为黑色(Tesseract对黑底白字识别精度更高) binary = cv2.bitwise_not(binary) return binary
2. 修复单字符识别逻辑
启用单字符识别模式(--psm 10),并放大ROI提升小字符清晰度,修复你注释的extract_time_from_image函数:
def extract_time_from_image(image): regions = [ (131, 11, 142, 31, '012'), # 小时十位(0-2) (142, 11, 155, 31, '0123456789'), # 小时个位(0-9) (163, 11, 179, 31, '012345'), # 分钟十位(0-5) (179, 11, 193, 31, '0123456789'), # 分钟个位(0-9) (202, 11, 215, 31, '012345'), # 秒十位(0-5) (215, 11, 226, 31, '0123456789') # 秒个位(0-9) ] digits = [] # 只预处理一次图像,避免重复计算 preprocessed = preprocess_image(image) for (x1, y1, x2, y2, whitelist) in regions: roi = preprocessed[y1:y2, x1:x2] # 放大2倍提升小字符识别率 roi = cv2.resize(roi, None, fx=2, fy=2, interpolation=cv2.INTER_CUBIC) # 单字符识别模式,针对性更强 custom_config = f'--psm 10 --oem 3 -c tessedit_char_whitelist={whitelist}' digit = pytesseract.image_to_string(roi, config=custom_config).strip() # 兜底处理:识别为空时补0 digits.append(digit if digit else '0') # 拼接成标准时间格式 return f"{digits[0]}{digits[1]}:{digits[2]}{digits[3]}:{digits[4]}{digits[5]}"
3. 后处理校验修正
识别后强制校验时间合法性,修正无效值:
def validate_time(time_str): try: h, m, s = map(int, time_str.split(':')) # 修正超出范围的时间值 h = max(0, h) m = max(0, min(m, 59)) s = max(0, min(s, 59)) return f"{h:02d}:{m:02d}:{s:02d}" except: return None
4. 备选:模板匹配
如果时间戳字体、位置固定,用模板匹配替代OCR稳定性更高:
- 提前从视频帧中截取0-9和冒号的模板图像
- 对每个数字区域遍历模板计算匹配度,取最高结果:
def match_template(roi, templates): max_val = 0 best_match = '0' for char, template in templates.items(): res = cv2.matchTemplate(roi, template, cv2.TM_CCOEFF_NORMED) _, val, _, _ = cv2.minMaxLoc(res) if val > max_val: max_val = val best_match = char # 匹配度低于0.8时返回默认值,避免错误识别 return best_match if max_val > 0.8 else '0'
内容的提问来源于stack exchange,提问作者Ernán
相关产品推荐
相关产品推荐

