如何用OpenCV+pytesseract识别视频指定6个数字并实现存储与告警?
问题描述
待处理界面:
我正在使用pytesseract和OpenCV识别视频中固定位置的6个数值变量,运行程序并选择感兴趣区域(ROI)后,当前代码会识别到数字上方的无关文本,但我仅需识别界面上的6个数字(含浮点数及xx:xx格式,后者可转换为总秒数)。
感兴趣区域内的6个数字对应关系为:109=v1;00:00=v2(可存为总秒数);80=v3;1.6=v4;2.5=v5;1.1=v6。识别完成后,需每2秒将变量值存入Excel,后续还要基于这些数值实现超限告警(如v1超限时触发提示音等)功能。
我尝试了以下代码,但它会识别到数字上方的无关文本(虽可作为Excel列名,但并非当前所需):
import cv2 import pandas as pd import pytesseract from PIL import Image import time # Configuração do Tesseract (certifique-se de ter o Tesseract instalado em seu sistema) pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe' # Inicialize a planilha do Excel excel_filename = 'dados_variaveis.xlsx' df = pd.DataFrame(columns=['Tempo', 'Var1', 'Var2', 'Var3', 'Var4', 'Var5', 'Var6']) # Inicialize o objeto do vídeo video_path = 'C:\\Users\\Jayme Neto\\Desktop\\amostravideoteste.mp4' cap = cv2.VideoCapture(video_path) # Verifique se o vídeo foi aberto corretamente if not cap.isOpened(): print("Erro ao abrir o vídeo.") exit() # Função de callback para o evento de clique do mouse def selecionar_roi(event, x, y, flags, param): global selecionando, x_inicial, y_inicial, x_final, y_final if event == cv2.EVENT_LBUTTONDOWN: selecionando = True x_inicial, y_inicial = x, y elif event == cv2.EVENT_LBUTTONUP: selecionando = False x_final, y_final = x, y # Desenhe um retângulo na imagem para indicar a ROI cv2.rectangle(frame, (x_inicial, y_inicial), (x_final, y_final), (0, 255, 0), 2) cv2.imshow('Selecione a ROI', frame) # Leia o primeiro frame para exibir a imagem ret, frame = cap.read() # Verifique se o frame foi lido corretamente if not ret: print("Erro ao ler o primeiro frame.") exit() # Inicialize as variáveis de seleção da ROI selecionando = False x_inicial, y_inicial, x_final, y_final = -1, -1, -1, -1 # Crie uma janela para exibir o vídeo e configurar o evento de clique do mouse cv2.namedWindow('Selecione a ROI') cv2.setMouseCallback('Selecione a ROI', selecionar_roi) print("Clique e arraste para selecionar a região de interesse. Pressione 'ESC' quando terminar.") while True: cv2.imshow('Selecione a ROI', frame) key = cv2.waitKey(1) & 0xFF if key == 27: # Tecla 'ESC' para sair break # Libere os recursos da janela de seleção da ROI cv2.destroyAllWindows() # Continuar com a lógica original para processamento do vídeo intervalo_captura = 2 while cap.isOpened(): ret, frame = cap.read() if not ret: break # Lógica para extrair os valores das variáveis usando Tesseract OCR tela_variaveis = frame[y_inicial:y_final, x_inicial:x_final] texto_extraido = pytesseract.image_to_string(Image.fromarray(tela_variaveis), config='--psm 6') # Supondo que os valores estejam em uma linha separada por espaços valores_variaveis = [float(valor) if ':' not in valor else time.strptime(valor, '%M:%S').tm_min + time.strptime(valor, '%M:%S').tm_sec/60 for valor in texto_extraido.split()] # Adicione os valores à planilha tempo_atual = time.strftime('%H:%M:%S') df = df.append({'Tempo': tempo_atual, 'Var1': valores_variaveis[0], 'Var2': valores_variaveis[1], 'Var3': valores_variaveis[2], 'Var4': valores_variaveis[3], 'Var5': valores_variaveis[4], 'Var6': valores_variaveis[5]}, ignore_index=True) # Salve os dados no Excel a cada intervalo if float(time.time()) % intervalo_captura == 0: df.to_excel(excel_filename, index=False) # Aguarde 2 segundos (ou o intervalo desejado) antes de capturar o próximo frame time.sleep(intervalo_captura) # Libere os recursos cap.release() cv2.destroyAllWindows()
解决方案
核心优化思路
针对无关文本干扰问题,最直接的解决方式是为每个数字单独划定ROI,彻底规避无关区域;同时通过图像预处理、Tesseract配置优化提升识别准确率,额外实现超限告警功能。
修改后的完整代码
import cv2 import pandas as pd import pytesseract from PIL import Image import time import winsound # 用于超限告警提示音 # Tesseract配置:限定识别字符范围,适配单行数字识别 pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe' tesseract_config = '--psm 7 -c tessedit_char_whitelist=0123456789.:' # 初始化Excel表格 excel_filename = 'dados_variaveis.xlsx' df = pd.DataFrame(columns=['Tempo', 'Var1', 'Var2', 'Var3', 'Var4', 'Var5', 'Var6']) # 视频路径 video_path = 'C:\\Users\\Jayme Neto\\Desktop\\amostravideoteste.mp4' cap = cv2.VideoCapture(video_path) if not cap.isOpened(): print("无法打开视频。") exit() # 存储6个变量的ROI坐标 rois = [] current_roi_idx = 0 selecionando = False x_inicial, y_inicial = -1, -1 # 鼠标回调:依次选择6个数字的区域 def selecionar_rois(event, x, y, flags, param): global selecionando, x_inicial, y_inicial, current_roi_idx, frame if event == cv2.EVENT_LBUTTONDOWN: selecionando = True x_inicial, y_inicial = x, y elif event == cv2.EVENT_LBUTTONUP: selecionando = False rois.append((x_inicial, y_inicial, x, y)) cv2.rectangle(frame, (x_inicial, y_inicial), (x, y), (0, 255, 0), 2) current_roi_idx += 1 print(f"已选择第{current_roi_idx}个区域,共需选择6个。完成后按ESC") if current_roi_idx == 6: cv2.destroyWindow('Selecione as ROIs') # 读取第一帧 ret, frame = cap.read() if not ret: print("无法读取第一帧。") exit() cv2.namedWindow('Selecione as ROIs') cv2.setMouseCallback('Selecione as ROIs', selecionar_rois) print("依次拖拽选择6个数字的区域,完成后按ESC") while current_roi_idx < 6: cv2.imshow('Selecione as ROIs', frame) key = cv2.waitKey(1) & 0xFF if key == 27: break # 告警阈值示例(可根据需求修改) thresholds = {'Var1': 150} # 视频处理逻辑 intervalo_captura = 2 last_save_time = time.time() while cap.isOpened(): ret, frame = cap.read() if not ret: break valores_variaveis = [] for (x1, y1, x2, y2) in rois: # 提取单个数字的ROI roi = frame[y1:y2, x1:x2] # 图像预处理:灰度化+二值化,提升识别率 gray_roi = cv2.cvtColor(roi, cv2.COLOR_BGR2GRAY) _, thresh_roi = cv2.threshold(gray_roi, 127, 255, cv2.THRESH_BINARY_INV) # OCR识别 texto = pytesseract.image_to_string(Image.fromarray(thresh_roi), config=tesseract_config).strip() # 数值转换 if ':' in texto: mins, secs = map(int, texto.split(':')) valores_variaveis.append(mins * 60 + secs) else: try: valores_variaveis.append(float(texto)) except: valores_variaveis.append(None) # 识别失败时填充None # 超限告警 if valores_variaveis[0] is not None and valores_variaveis[0] > thresholds['Var1']: winsound.Beep(1000, 500) # 1000Hz频率,持续500ms # 添加数据到表格 tempo_atual = time.strftime('%H:%M:%S') df.loc[len(df)] = [tempo_atual] + valores_variaveis # 每2秒保存一次Excel if time.time() - last_save_time >= intervalo_captura: df.to_excel(excel_filename, index=False) last_save_time = time.time() time.sleep(intervalo_captura) cap.release() cv2.destroyAllWindows()
关键修改点说明
- 多ROI精准定位:为每个数字单独选择区域,彻底排除无关文本干扰
- 图像预处理:灰度化+二值化增强数字与背景的对比度,提升OCR识别准确率
- Tesseract优化:通过字符白名单限定仅识别数字、小数点和冒号,PSM模式适配单行数字识别场景
- 告警功能实现:使用
winsound模块实现超限提示音 - 数据保存逻辑修复:替换原代码中时间取余的精度问题,改用时间差判断触发保存
内容的提问来源于stack exchange,提问作者Jaime Neto
相关产品推荐
相关产品推荐

