基于OpenCV的PDF高密度页面误分类问题及优化方案问询
PDF高密度页面识别优化需求
页面示例
高密度页面示例


误分类页面示例


当前实现方案
- 计算页面平均阈值,将像素值与阈值对比
- 低于阈值的像素转为黑色(0),高于阈值的转为白色(255)
- 计算黑白像素占比,当占比>25时判定为高密度页面
现有方案局限
- 小字体多文本页面会被误判为高密度页面
- 含多张图片的页面会被误判为高密度页面
- 带背景的页面会被误判为高密度页面
现需消除上述局限,寻求更优的实现方案。
当前使用的函数代码
import fitz # PyMuPDF import cv2 import numpy as np import glob import os import time import tqdm def get_pdf_paths(directory_path): pdf_paths = glob.glob(f"{directory_path}/*.pdf") return pdf_paths def get_number_of_pages(pdf_path): try: pdf_document = fitz.open(pdf_path) number_of_pages = pdf_document.page_count pdf_document.close() return number_of_pages except Exception as e: print(f"Error: {e}") return None def convert_pdf_to_image(pdf_path, page_number=0): pdf_document = fitz.open(pdf_path) pdf_page = pdf_document.load_page(page_number) pix = pdf_page.get_pixmap() img = np.frombuffer(pix.samples, dtype=np.uint8).reshape((pix.h, pix.w, pix.n)) return cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) def calculate_density_threshold(image): average_density = np.mean(image) threshold = int(average_density) # print(threshold) return threshold def binarize_image(image, threshold): _, binary_image = cv2.threshold(image, threshold, 255, cv2.THRESH_BINARY) return binary_image def calculate_black_to_white_ratio(image): total_pixels = image.size black_pixels = np.sum(image == 0) white_pixels = total_pixels - black_pixels if white_pixels == 0: #happens when blank page white_pixels = 1 ratio = black_pixels / white_pixels return ratio#, white_pixels
执行代码
%%time directory_path = "/path/to/directory" list_of_pdf_paths = get_pdf_paths(directory_path) print(f'Number of pdfs {len(list_of_pdf_paths)}\n') p=[] for pdf_path in list_of_pdf_paths: print('----------------------------------------------------------------------------------------------') print(f'{os.path.basename(pdf_path)}\n') no_of_pages = get_number_of_pages(pdf_path) p.append(no_of_pages) print(f'No. of pages in {os.path.basename(pdf_path)} is {no_of_pages} \n') condition = True for page_number in range(0,no_of_pages): # Convert PDF page to grayscale image pdf_image = convert_pdf_to_image(pdf_path, page_number) # Calculate average density and threshold threshold = calculate_density_threshold(pdf_image) # Binarize the image using the calculated threshold binary_image = binarize_image(pdf_image, threshold) # Calculate black to white pixel ratio ratio = calculate_black_to_white_ratio(binary_image)*100 #binary_image # print(f"Black to white pixel ratio for page {page_number+1} is {ratio:.4f}") if ratio > 100: #blank page continue elif ratio > 25: condition = False print(f'Page number {page_number+1} is of high density') if condition == True: print('No high density pages found.\n')
内容的提问来源于stack exchange,提问作者Harshal Naik
相关产品推荐
相关产品推荐

