You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用OpenCV+pytesseract识别视频指定6个数字并实现存储与告警?

问题描述

待处理界面:
待处理界面

我正在使用pytesseract和OpenCV识别视频中固定位置的6个数值变量,运行程序并选择感兴趣区域(ROI)后,当前代码会识别到数字上方的无关文本,但我仅需识别界面上的6个数字(含浮点数及xx:xx格式,后者可转换为总秒数)。

感兴趣区域内的6个数字对应关系为:109=v1;00:00=v2(可存为总秒数);80=v3;1.6=v4;2.5=v5;1.1=v6。识别完成后,需每2秒将变量值存入Excel,后续还要基于这些数值实现超限告警(如v1超限时触发提示音等)功能。

我尝试了以下代码,但它会识别到数字上方的无关文本(虽可作为Excel列名,但并非当前所需):

import cv2
import pandas as pd
import pytesseract
from PIL import Image
import time

# Configuração do Tesseract (certifique-se de ter o Tesseract instalado em seu sistema)
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'

# Inicialize a planilha do Excel
excel_filename = 'dados_variaveis.xlsx'
df = pd.DataFrame(columns=['Tempo', 'Var1', 'Var2', 'Var3', 'Var4', 'Var5', 'Var6'])

# Inicialize o objeto do vídeo
video_path = 'C:\\Users\\Jayme Neto\\Desktop\\amostravideoteste.mp4'
cap = cv2.VideoCapture(video_path)

# Verifique se o vídeo foi aberto corretamente
if not cap.isOpened():
    print("Erro ao abrir o vídeo.")
    exit()

# Função de callback para o evento de clique do mouse
def selecionar_roi(event, x, y, flags, param):
    global selecionando, x_inicial, y_inicial, x_final, y_final

    if event == cv2.EVENT_LBUTTONDOWN:
        selecionando = True
        x_inicial, y_inicial = x, y

    elif event == cv2.EVENT_LBUTTONUP:
        selecionando = False
        x_final, y_final = x, y

        # Desenhe um retângulo na imagem para indicar a ROI
        cv2.rectangle(frame, (x_inicial, y_inicial), (x_final, y_final), (0, 255, 0), 2)
        cv2.imshow('Selecione a ROI', frame)

# Leia o primeiro frame para exibir a imagem
ret, frame = cap.read()

# Verifique se o frame foi lido corretamente
if not ret:
    print("Erro ao ler o primeiro frame.")
    exit()

# Inicialize as variáveis de seleção da ROI
selecionando = False
x_inicial, y_inicial, x_final, y_final = -1, -1, -1, -1

# Crie uma janela para exibir o vídeo e configurar o evento de clique do mouse
cv2.namedWindow('Selecione a ROI')
cv2.setMouseCallback('Selecione a ROI', selecionar_roi)

print("Clique e arraste para selecionar a região de interesse. Pressione 'ESC' quando terminar.")

while True:
    cv2.imshow('Selecione a ROI', frame)
    key = cv2.waitKey(1) & 0xFF

    if key == 27:  # Tecla 'ESC' para sair
        break

# Libere os recursos da janela de seleção da ROI
cv2.destroyAllWindows()

# Continuar com a lógica original para processamento do vídeo
intervalo_captura = 2

while cap.isOpened():
    ret, frame = cap.read()

    if not ret:
        break

    # Lógica para extrair os valores das variáveis usando Tesseract OCR
    tela_variaveis = frame[y_inicial:y_final, x_inicial:x_final]
    texto_extraido = pytesseract.image_to_string(Image.fromarray(tela_variaveis), config='--psm 6')

    # Supondo que os valores estejam em uma linha separada por espaços
    valores_variaveis = [float(valor) if ':' not in valor else time.strptime(valor, '%M:%S').tm_min + time.strptime(valor, '%M:%S').tm_sec/60
                        for valor in texto_extraido.split()]

    # Adicione os valores à planilha
    tempo_atual = time.strftime('%H:%M:%S')
    df = df.append({'Tempo': tempo_atual, 'Var1': valores_variaveis[0], 'Var2': valores_variaveis[1],
                    'Var3': valores_variaveis[2], 'Var4': valores_variaveis[3],
                    'Var5': valores_variaveis[4], 'Var6': valores_variaveis[5]}, ignore_index=True)

    # Salve os dados no Excel a cada intervalo
    if float(time.time()) % intervalo_captura == 0:
        df.to_excel(excel_filename, index=False)

    # Aguarde 2 segundos (ou o intervalo desejado) antes de capturar o próximo frame
    time.sleep(intervalo_captura)

# Libere os recursos
cap.release()
cv2.destroyAllWindows()
解决方案

核心优化思路

针对无关文本干扰问题,最直接的解决方式是为每个数字单独划定ROI,彻底规避无关区域;同时通过图像预处理、Tesseract配置优化提升识别准确率,额外实现超限告警功能。

修改后的完整代码

import cv2
import pandas as pd
import pytesseract
from PIL import Image
import time
import winsound  # 用于超限告警提示音

# Tesseract配置:限定识别字符范围,适配单行数字识别
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'
tesseract_config = '--psm 7 -c tessedit_char_whitelist=0123456789.:'

# 初始化Excel表格
excel_filename = 'dados_variaveis.xlsx'
df = pd.DataFrame(columns=['Tempo', 'Var1', 'Var2', 'Var3', 'Var4', 'Var5', 'Var6'])

# 视频路径
video_path = 'C:\\Users\\Jayme Neto\\Desktop\\amostravideoteste.mp4'
cap = cv2.VideoCapture(video_path)

if not cap.isOpened():
    print("无法打开视频。")
    exit()

# 存储6个变量的ROI坐标
rois = []
current_roi_idx = 0
selecionando = False
x_inicial, y_inicial = -1, -1

# 鼠标回调:依次选择6个数字的区域
def selecionar_rois(event, x, y, flags, param):
    global selecionando, x_inicial, y_inicial, current_roi_idx, frame
    if event == cv2.EVENT_LBUTTONDOWN:
        selecionando = True
        x_inicial, y_inicial = x, y
    elif event == cv2.EVENT_LBUTTONUP:
        selecionando = False
        rois.append((x_inicial, y_inicial, x, y))
        cv2.rectangle(frame, (x_inicial, y_inicial), (x, y), (0, 255, 0), 2)
        current_roi_idx += 1
        print(f"已选择第{current_roi_idx}个区域,共需选择6个。完成后按ESC")
        if current_roi_idx == 6:
            cv2.destroyWindow('Selecione as ROIs')

# 读取第一帧
ret, frame = cap.read()
if not ret:
    print("无法读取第一帧。")
    exit()

cv2.namedWindow('Selecione as ROIs')
cv2.setMouseCallback('Selecione as ROIs', selecionar_rois)
print("依次拖拽选择6个数字的区域,完成后按ESC")

while current_roi_idx < 6:
    cv2.imshow('Selecione as ROIs', frame)
    key = cv2.waitKey(1) & 0xFF
    if key == 27:
        break

# 告警阈值示例(可根据需求修改)
thresholds = {'Var1': 150}

# 视频处理逻辑
intervalo_captura = 2
last_save_time = time.time()

while cap.isOpened():
    ret, frame = cap.read()
    if not ret:
        break

    valores_variaveis = []
    for (x1, y1, x2, y2) in rois:
        # 提取单个数字的ROI
        roi = frame[y1:y2, x1:x2]
        # 图像预处理:灰度化+二值化,提升识别率
        gray_roi = cv2.cvtColor(roi, cv2.COLOR_BGR2GRAY)
        _, thresh_roi = cv2.threshold(gray_roi, 127, 255, cv2.THRESH_BINARY_INV)
        # OCR识别
        texto = pytesseract.image_to_string(Image.fromarray(thresh_roi), config=tesseract_config).strip()
        # 数值转换
        if ':' in texto:
            mins, secs = map(int, texto.split(':'))
            valores_variaveis.append(mins * 60 + secs)
        else:
            try:
                valores_variaveis.append(float(texto))
            except:
                valores_variaveis.append(None)  # 识别失败时填充None

    # 超限告警
    if valores_variaveis[0] is not None and valores_variaveis[0] > thresholds['Var1']:
        winsound.Beep(1000, 500)  # 1000Hz频率,持续500ms

    # 添加数据到表格
    tempo_atual = time.strftime('%H:%M:%S')
    df.loc[len(df)] = [tempo_atual] + valores_variaveis

    # 每2秒保存一次Excel
    if time.time() - last_save_time >= intervalo_captura:
        df.to_excel(excel_filename, index=False)
        last_save_time = time.time()

    time.sleep(intervalo_captura)

cap.release()
cv2.destroyAllWindows()

关键修改点说明

  1. 多ROI精准定位:为每个数字单独选择区域,彻底排除无关文本干扰
  2. 图像预处理:灰度化+二值化增强数字与背景的对比度,提升OCR识别准确率
  3. Tesseract优化:通过字符白名单限定仅识别数字、小数点和冒号,PSM模式适配单行数字识别场景
  4. 告警功能实现:使用winsound模块实现超限提示音
  5. 数据保存逻辑修复:替换原代码中时间取余的精度问题,改用时间差判断触发保存

内容的提问来源于stack exchange,提问作者Jaime Neto

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.01 09:15:28