You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何多次调参后Tesseract仍无法识别键盘6个字母键?

键盘字母键实时识别问题及优化探索

当前采用的方案:

  • 自适应阈值处理(adaptive thresholding)
  • 按键宽高比分割(对应结果图中的绿色框)
  • 使用PSM 10模式将每个按键视为单个字符

但仍存在以下问题:部分按键无法识别、识别错误,甚至单个字符识别出两个结果(比如L键被识别为L和P)。

注:为适配平台要求裁剪了图片并重新运行识别,但裁剪前识别效果稍好(能识别更多按键,误识别更少)。

我的需求仅为识别字母键,最终目标是实现实时视频识别。

使用的Tesseract配置参数:

'-l eng --oem 1 --psm 10 -c tessedit_char_whitelist="ABCDEFGHIJKLMNOPQRSTUVWXYZ"'

我尝试过不同的图像缩放方式、单独缩放按键区域、使用开运算/闭运算等操作,但依旧无法识别所有按键。

更新内容

将图像调整为俯视视角(bird's eye)并取消白名单限制后,基本可识别所有按键(O被识别为0、I被识别为|属于可理解的情况)。想请教:

  1. 为何会出现这种情况?
  2. 如何让系统适配动态视频中的多变条件?

代码实现

import pytesseract
import numpy as np
try:
 from PIL import Image
except ImportError:
 import Image
import cv2
from tqdm import tqdm
from collections import defaultdict


def get_missing_chars(dict):
    capital_alphabet = [chr(ascii) for ascii in range(65, 91)]
    return [let for let in capital_alphabet if let not in dict]

def draw_box_and_char(img, contour_dims, c, box_col, text_col):
    x, y, w, h = contour_dims
    top_left = (x, y)
    bot_right = (x + w, y+h)
    font_offset = 3
    text_pos = (x+h//2+12, y+h-font_offset)
    img_copy = img.copy()
    cv2.rectangle(img_copy, top_left, bot_right, box_col, 2)
    cv2.putText(img_copy, c, text_pos, cv2.FONT_HERSHEY_SIMPLEX, fontScale=.5, color=text_col, thickness=1, lineType=cv2.LINE_AA)
    return img_copy

def detect_keys(img):
    scaling = .25
    img = cv2.resize(img, None, fx=scaling, fy=scaling, interpolation=cv2.INTER_AREA)
    print("img shape", img.shape)

    gray_img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)

    ratio_min = 0.7
    area_min = 1000

    nbrhood_size = 1001
    bias = 2
    # adapt to different lighting
    bin_img = cv2.adaptiveThreshold(gray_img, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C,\
                cv2.THRESH_BINARY_INV, nbrhood_size, bias)

    items = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
    contours = items[0] if len(items) == 2 else items[1]

    key_contours = []
    for c in contours:
        x, y, w, h = cv2.boundingRect(c)
        ratio = h/w
        area = cv2.contourArea(c)
        # square-like ratio, try to get character
        if ratio > ratio_min and area > area_min:
            key_contours.append(c)

    detected = defaultdict(int)
    n_kept = 0
    img_copy = cv2.cvtColor(bin_img, cv2.COLOR_GRAY2RGB)
    let_to_contour = {}
    n_contours = len(key_contours)

    # offset to get smaller square within the key segment for easier char recognition
    offset = 10
    show_each_char = False

    for _, c in tqdm(enumerate(key_contours), total=n_contours):
        x, y, w, h = cv2.boundingRect(c)
        ratio = h/w
        area = cv2.contourArea(c)
        base = np.zeros(bin_img.shape, dtype=np.uint8)
        base.fill(255)

        n_kept += 1
        new_y = y+offset
        new_x = x+offset
        new_h = h-2*offset
        new_w = w-2*offset

        base[new_y:new_y+new_h, new_x:new_x+new_w] = bin_img[new_y:new_y+new_h, new_x:new_x+new_w]
        segment = cv2.bitwise_not(base)

        # try scaling up individual keys
        # scaling = 2
        # segment = cv2.resize(segment, None, fx=scaling, fy=scaling, interpolation=cv2.INTER_CUBIC)


        # psm 10: treats the segment as a single character
        custom_config = r'-l eng --oem 1 --psm 10 -c tessedit_char_whitelist="ABCDEFGHIJKLMNOPQRSTUVWXYZ"'
        d = pytesseract.image_to_data(segment, config=custom_config, output_type='dict')
        conf = d['conf']
        c = d['text'][-1]

        if c:
            # sometimes recognizes multiple keys even though there is only 1
            for sub_c in c:
                # save character and contour to draw on image and show bounds/detection
                if sub_c not in let_to_contour or (sub_c in let_to_contour and conf > let_to_contour[sub_c]['conf']):
                    let_to_contour[sub_c] = {'conf': conf, 'cont': (new_x, new_y, new_w, new_h)}

        else:
            c = "?"
            text_col = (0, 0, 255)

        if show_each_char:
            contour_dims = (new_x, new_y, new_w, new_h)
            box_col = (0, 255, 0)
            text_col = (0, 0, 0)
            segment_with_boxes = draw_box_and_char(segment, contour_dims, c, box_col, text_col)

            cv2.imshow('segment', segment_with_boxes)
            cv2.waitKey(0)
            cv2.destroyAllWindows()

    # draw boxes around recognized keys
    for c, data in let_to_contour.items():
        box_col = (0, 255, 0)
        text_col = (0, 0, 0)
        img_copy = draw_box_and_char(img_copy, data['cont'], c, box_col, text_col)

    detected = {k: 1 for k in let_to_contour}
    for det in let_to_contour:
        print(det, let_to_contour[det])

    print("total detected: ", let_to_contour.keys())
    missing = get_missing_chars(detected)

    print(f"n_missing: {len(missing)}")
    print(f"chars missing: {missing}")
    return img_copy

if __name__ == "__main__":
    img_file = "keyboard.jpg"
    img = cv2.imread(img_file)
    img_with_detected_keys = detect_keys(img)
    cv2.imshow("detected", img_with_detected_keys)
    cv2.waitKey(0)
    cv2.destroyAllWindows()

内容的提问来源于stack exchange,提问作者macburger

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.12 03:45:35