基于Pytesseract优化图表X轴提取及识别框修正需求
图表X轴标签识别与分离实现方案
需求说明
- 使用Pytesseract生成更精准的识别框
- 提取图表X轴的数字与文本标签,实现二者分离
- 精准识别X、Y轴并分类,修复当前识别框绘制错误问题,确保数字与标签清晰分离
代码实现
import cv2 import numpy as np from pathlib import Path import pytesseract from pytesseract import Output from matplotlib import pyplot as plt from matplotlib import rcParams def getXTextFromImageArray(image, mode): image_text = [] if mode == 'x-text': # 转置图像适配横向文本识别 image = cv2.transpose(image) # Tesseract配置:英文识别+LSTM引擎+自动分页模式 config = "-l eng --oem 1 --psm 11" elif mode == 'x-labels': # Tesseract配置:仅识别数字和小数点+强制单块文本识别 config = "-l eng --oem 1 --psm 6 -c tessedit_char_whitelist=.0123456789" # 获取识别结果及位置信息 d = pytesseract.image_to_data(image, config=config, output_type=Output.DICT) n_boxes = len(d['text']) # 筛选置信度非负的有效识别结果 for i in range(n_boxes): if int(d['conf'][i]) >= 0: text = d['text'][i].strip() (x, y, w, h) = (d['left'][i], d['top'][i], d['width'][i], d['height'][i]) image_text.append((d['text'][i], (x, y, w, h))) # 去重(文本+识别框)组合 return list(set(image_text)) # 遍历目标图片目录 for path in Path(img_dir).iterdir(): if path.name.endswith(('.png', '.jpg', '.jpeg')): filepath = str(path) image = cv2.imread(filepath) image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) height, width, channels = image.shape # 检测图表的X轴、Y轴及上边界(需自行实现detectAxes函数) xaxis, yaxis, upper = detectAxes(filepath) y_text, y_labels = [], [] x_text, x_labels = [], [] # 提取轴坐标信息 x1, y1, x2, y2 = xaxis[0] xaxis = (x1, y1, x2, y2) y1_ax, y2_ax, y3_ax, y4_ax = yaxis[0] yaxis = (y1_ax, y2_ax, y3_ax, y4_ax) u1, u2, u3, u4 = upper[0] upper = (u1, u2, u3, u4) rcParams['figure.figsize'] = 15, 4 fig, ax = plt.subplots(1, 3) # 反向遍历图像,遮罩X轴上方区域以跳过刻度线干扰 gray = maskImageBackwardPassX(filepath, xaxis[1]) # 图像预处理:二值化+两次膨胀,突出文本轮廓 retX, threshX = cv2.threshold(gray, 0, 255, cv2.THRESH_OTSU | cv2.THRESH_BINARY_INV) rect_kernelX = cv2.getStructuringElement(cv2.MORPH_RECT, (1, 15)) threshX = cv2.dilate(threshX, rect_kernelX, iterations=1) rect_kernelX = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 1)) threshX = cv2.dilate(threshX, rect_kernelX, iterations=1) # 查找文本轮廓 contoursX = cv2.findContours(threshX, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) contoursX = contoursX[0] if len(contoursX) == 2 else contoursX[1] rectsX = [cv2.boundingRect(contour) for contour in contoursX] print("轮廓数量: ", len(contoursX)) # 筛选可能的X轴标签区域(需自行实现getProbableXLabels函数) image, x_labels = getProbableXLabels(image, rectsX, xaxis, yaxis, upper) white_bgX = 255 * np.ones_like(gray.copy()) # 将标签区域复制到白底图像,提升识别准确率 for (textx, texty, w, h) in x_labels: roi = gray[texty:texty + h, textx:textx + w] white_bgX[texty:texty + h, textx:textx + w] = roi # 识别X轴数字标签 x_labels_list = getXTextFromImageArray(white_bgX, 'x-labels') # 显示带识别框的图像(需自行实现display_image_with_rects函数) display_image_with_rects(image, rectsX) # 按X坐标排序数字标签,遮罩后识别文本标签 x_labels_list.sort(key=lambda item: item[1][0]) print(x_labels_list) x_labels = [] for text, (textx, texty, w, h) in x_labels_list: roi = 255 * np.ones_like(gray[texty:texty + h, textx:textx + w]) gray[texty:texty + h, textx:textx + w] = roi x_labels.append(text) # 识别X轴文本标签 x_text_list = getXTextFromImageArray(gray, 'x-text') # 按Y坐标排序文本标签 def getYFromRect(item): return item[1][1] x_text_list.sort(key=getYFromRect) for text, (textx, texty, w, h) in x_text_list: x_text.append(text) # 扫描线法筛选X轴标签:找到与扫描线相交最多的轮廓组 maxIntersection = 0 maxList = [] for i in range(y1, 1000): count = 0 current = [] for index, rect in enumerate(contoursX): # 判断扫描线与轮廓是否相交(需自行实现lineIntersectsRectY函数) if lineIntersectsRectY(i, rect): count += 1 current.append(contoursX[index]) if count > maxIntersection: maxIntersection = count maxList = current return image, maxList def maskImageBackwardPassX(filepath, end_idx): image = cv2.imread(filepath) height, width, channels = image.shape gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) # 从指定位置向下遍历,找到无内容行后停止,遮罩上方区域 while end_idx < height: if sum(gray[end_idx, :] < 200) == 0: break else: end_idx += 1 gray[0:end_idx, :] = 255 return gray
核心优化说明
- 识别精准度提升:针对数字标签启用字符白名单+PSM 6单块文本模式,避免非数字干扰;文本标签使用PSM 11自动分页模式适配横向排列。
- 标签分离逻辑:通过白底图像单独识别数字标签,再遮罩数字区域后识别文本标签,彻底规避两类标签的相互干扰。
- 识别框错误修复:结合轮廓检测+扫描线筛选,仅保留X轴区域内的标签轮廓,排除图表其他区域的干扰元素。
内容的提问来源于stack exchange,提问作者김보미
相关产品推荐
相关产品推荐

