为何多次调参后Tesseract仍无法识别键盘6个字母键?
键盘字母键实时识别问题及优化探索
当前采用的方案:
- 自适应阈值处理(adaptive thresholding)
- 按键宽高比分割(对应结果图中的绿色框)
- 使用PSM 10模式将每个按键视为单个字符
但仍存在以下问题:部分按键无法识别、识别错误,甚至单个字符识别出两个结果(比如L键被识别为L和P)。
注:为适配平台要求裁剪了图片并重新运行识别,但裁剪前识别效果稍好(能识别更多按键,误识别更少)。
我的需求仅为识别字母键,最终目标是实现实时视频识别。
使用的Tesseract配置参数:
'-l eng --oem 1 --psm 10 -c tessedit_char_whitelist="ABCDEFGHIJKLMNOPQRSTUVWXYZ"'
我尝试过不同的图像缩放方式、单独缩放按键区域、使用开运算/闭运算等操作,但依旧无法识别所有按键。
更新内容
将图像调整为俯视视角(bird's eye)并取消白名单限制后,基本可识别所有按键(O被识别为0、I被识别为|属于可理解的情况)。想请教:
- 为何会出现这种情况?
- 如何让系统适配动态视频中的多变条件?
代码实现
import pytesseract import numpy as np try: from PIL import Image except ImportError: import Image import cv2 from tqdm import tqdm from collections import defaultdict def get_missing_chars(dict): capital_alphabet = [chr(ascii) for ascii in range(65, 91)] return [let for let in capital_alphabet if let not in dict] def draw_box_and_char(img, contour_dims, c, box_col, text_col): x, y, w, h = contour_dims top_left = (x, y) bot_right = (x + w, y+h) font_offset = 3 text_pos = (x+h//2+12, y+h-font_offset) img_copy = img.copy() cv2.rectangle(img_copy, top_left, bot_right, box_col, 2) cv2.putText(img_copy, c, text_pos, cv2.FONT_HERSHEY_SIMPLEX, fontScale=.5, color=text_col, thickness=1, lineType=cv2.LINE_AA) return img_copy def detect_keys(img): scaling = .25 img = cv2.resize(img, None, fx=scaling, fy=scaling, interpolation=cv2.INTER_AREA) print("img shape", img.shape) gray_img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) ratio_min = 0.7 area_min = 1000 nbrhood_size = 1001 bias = 2 # adapt to different lighting bin_img = cv2.adaptiveThreshold(gray_img, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C,\ cv2.THRESH_BINARY_INV, nbrhood_size, bias) items = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) contours = items[0] if len(items) == 2 else items[1] key_contours = [] for c in contours: x, y, w, h = cv2.boundingRect(c) ratio = h/w area = cv2.contourArea(c) # square-like ratio, try to get character if ratio > ratio_min and area > area_min: key_contours.append(c) detected = defaultdict(int) n_kept = 0 img_copy = cv2.cvtColor(bin_img, cv2.COLOR_GRAY2RGB) let_to_contour = {} n_contours = len(key_contours) # offset to get smaller square within the key segment for easier char recognition offset = 10 show_each_char = False for _, c in tqdm(enumerate(key_contours), total=n_contours): x, y, w, h = cv2.boundingRect(c) ratio = h/w area = cv2.contourArea(c) base = np.zeros(bin_img.shape, dtype=np.uint8) base.fill(255) n_kept += 1 new_y = y+offset new_x = x+offset new_h = h-2*offset new_w = w-2*offset base[new_y:new_y+new_h, new_x:new_x+new_w] = bin_img[new_y:new_y+new_h, new_x:new_x+new_w] segment = cv2.bitwise_not(base) # try scaling up individual keys # scaling = 2 # segment = cv2.resize(segment, None, fx=scaling, fy=scaling, interpolation=cv2.INTER_CUBIC) # psm 10: treats the segment as a single character custom_config = r'-l eng --oem 1 --psm 10 -c tessedit_char_whitelist="ABCDEFGHIJKLMNOPQRSTUVWXYZ"' d = pytesseract.image_to_data(segment, config=custom_config, output_type='dict') conf = d['conf'] c = d['text'][-1] if c: # sometimes recognizes multiple keys even though there is only 1 for sub_c in c: # save character and contour to draw on image and show bounds/detection if sub_c not in let_to_contour or (sub_c in let_to_contour and conf > let_to_contour[sub_c]['conf']): let_to_contour[sub_c] = {'conf': conf, 'cont': (new_x, new_y, new_w, new_h)} else: c = "?" text_col = (0, 0, 255) if show_each_char: contour_dims = (new_x, new_y, new_w, new_h) box_col = (0, 255, 0) text_col = (0, 0, 0) segment_with_boxes = draw_box_and_char(segment, contour_dims, c, box_col, text_col) cv2.imshow('segment', segment_with_boxes) cv2.waitKey(0) cv2.destroyAllWindows() # draw boxes around recognized keys for c, data in let_to_contour.items(): box_col = (0, 255, 0) text_col = (0, 0, 0) img_copy = draw_box_and_char(img_copy, data['cont'], c, box_col, text_col) detected = {k: 1 for k in let_to_contour} for det in let_to_contour: print(det, let_to_contour[det]) print("total detected: ", let_to_contour.keys()) missing = get_missing_chars(detected) print(f"n_missing: {len(missing)}") print(f"chars missing: {missing}") return img_copy if __name__ == "__main__": img_file = "keyboard.jpg" img = cv2.imread(img_file) img_with_detected_keys = detect_keys(img) cv2.imshow("detected", img_with_detected_keys) cv2.waitKey(0) cv2.destroyAllWindows()
内容的提问来源于stack exchange,提问作者macburger
相关产品推荐
相关产品推荐

