如何用Python及图像库拆分双栏扫描文档段落与单个问题?
基于Python的扫描文档处理方案
一、前置准备
先安装所需依赖库:
pip install opencv-python pillow pytesseract
注意:需提前在系统中安装Tesseract OCR引擎,确保
pytesseract能正常调用。
二、扫描图像单问题拆分实现
核心思路
先对图像做预处理提升OCR识别精度,再提取文本,最后根据问题的编号特征(如1.、2.这类前缀)拆分出单个问题。
代码示例
import cv2 import pytesseract import re def preprocess_image(image_path): # 读取图像 img = cv2.imread(image_path) # 灰度化 gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) # 二值化(自适应阈值,处理扫描阴影) thresh = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY_INV, 11, 2) # 去噪 denoised = cv2.medianBlur(thresh, 3) return denoised def split_single_questions(image_path): processed_img = preprocess_image(image_path) # OCR提取文本 text = pytesseract.image_to_string(processed_img, lang='eng') # 按问题编号拆分(匹配类似"1. "、"2. "的前缀) question_pattern = re.compile(r'(^\d+\.\s)', re.MULTILINE) questions = re.split(question_pattern, text) # 整理成单个问题列表 formatted_questions = [] for i in range(1, len(questions), 2): formatted_questions.append(f"{questions[i]}{questions[i+1]}".strip()) return formatted_questions # 调用示例 # questions = split_single_questions("your_image_path.jpg") # for q in questions: # print("---问题分割线---") # print(q)
三、双栏扫描文档段落拆分实现
核心思路
通过分析OCR识别出的文本行坐标,聚类区分左右栏,再按栏内的行顺序合并为段落,同时处理可能的跨栏段落。
代码示例
import cv2 import pytesseract from sklearn.cluster import KMeans import numpy as np def get_text_lines_with_coords(image_path): processed_img = preprocess_image(image_path) # 获取带坐标的OCR结果 d = pytesseract.image_to_data(processed_img, lang='eng', output_type=pytesseract.Output.DICT) n_boxes = len(d['text']) text_lines = [] # 按行分组(同一y坐标区间视为一行) current_line = {"text": "", "x": [], "y": d['top'][0]} for i in range(n_boxes): if int(d['conf'][i]) > 60: # 过滤低置信度文本 # 判断是否为同一行(y坐标差小于10像素) if abs(d['top'][i] - current_line["y"]) < 10: current_line["text"] += d['text'][i] + " " current_line["x"].append(d['left'][i]) else: text_lines.append(current_line) current_line = {"text": d['text'][i] + " ", "x": [d['left'][i]], "y": d['top'][i]} text_lines.append(current_line) return text_lines def split_two_column_paragraphs(image_path): text_lines = get_text_lines_with_coords(image_path) # 提取每行的平均x坐标,用于聚类 x_coords = np.array([np.mean(line["x"]) for line in text_lines]).reshape(-1, 1) # KMeans聚类分为2类(左右栏) kmeans = KMeans(n_clusters=2, random_state=0).fit(x_coords) labels = kmeans.labels_ # 区分左右栏(按聚类中心的x坐标排序) cluster_centers = kmeans.cluster_centers_.flatten() left_col_idx = 0 if cluster_centers[0] < cluster_centers[1] else 1 right_col_idx = 1 - left_col_idx # 按栏分组 left_lines = [line["text"].strip() for line, label in zip(text_lines, labels) if label == left_col_idx] right_lines = [line["text"].strip() for line, label in zip(text_lines, labels) if label == right_col_idx] # 合并段落(按行顺序,空行分隔段落) def lines_to_paragraphs(lines): paragraphs = [] current_paragraph = [] for line in lines: if line.strip() == "": if current_paragraph: paragraphs.append(" ".join(current_paragraph)) current_paragraph = [] else: current_paragraph.append(line) if current_paragraph: paragraphs.append(" ".join(current_paragraph)) return paragraphs left_paragraphs = lines_to_paragraphs(left_lines) right_paragraphs = lines_to_paragraphs(right_lines) # 合并左右栏段落(先左后右,或按实际排版调整顺序) all_paragraphs = left_paragraphs + right_paragraphs return all_paragraphs # 调用示例 # paragraphs = split_two_column_paragraphs("your_image_path.jpg") # for p in paragraphs: # print("---段落分割线---") # print(p)
内容的提问来源于stack exchange,提问作者sibi kanagaraj
相关产品推荐
相关产品推荐

