使用fitz的clip参数提取PDF文本返回空值,求解决方法
问题描述
我正在编写Python脚本将PDF转换为有声书,尝试通过设置边框(clip参数)移除页码及其他无关标题。当前使用Python 3.10.4,使用无参数的get_text可正常提取文本,但传入clip参数后返回空值。我使用的输入参数为起始页27、结束页40、边框比例0.05,代码如下:
import fitz from gtts import gTTS import os import re def extract_text_by_area(page, x0, y0, x1, y1): return page.get_text("text", clip=(x0, y0, x1, y1)) def get_text_with_area_extraction(filepath: str, start_page: int, end_page: int, x0, y0, x1, y1) -> str: with fitz.open(filepath) as doc: extracted_text = "" for page_num in range(start_page, end_page + 1): page = doc[page_num] extracted_text += extract_text_by_area(page, x0, y0, x1, y1).replace('\n', ' ').strip() standardized_text = re.sub(r'\s+', ' ', extracted_text) return standardized_text def get_pdf_dimensions(pdf_path, page_num): doc = fitz.open(pdf_path) page = doc[page_num] page_width = page.rect.width page_height = page.rect.height doc.close() return page_width, page_height def calculate_text_area(page_width, page_height, percentage): border_x = page_width * percentage border_y = page_height * percentage x0 = border_x y0 = page_height - border_y x1 = page_width - border_x y1 = border_y return x0, y0, x1, y1 filepath = 'Fooled-by-Randomness-Role-of-Chance-in-Markets-and-Life-PROPER1.pdf' start_page = int(input("Enter the starting page: ")) - 1 end_page = int(input("Enter the ending page: ")) - 1 page_num = start_page page_width, page_height = get_pdf_dimensions(filepath, page_num) percentage = float(input("Enter the percentage of the border to remove (e.g., 0.1 for 10%): ")) x0, y0, x1, y1 = calculate_text_area(page_width, page_height, percentage) extracted_text = get_text_with_area_extraction(filepath, start_page, end_page, x0, y0, x1, y1) print(extracted_text) tts = gTTS(text=extracted_text, lang='en') output_audio_path = 'extracted_text_audio.mp3' tts.save(output_audio_path) print(f"Text-to-speech audio saved as: {output_audio_path}")
解决建议
核心问题:Clip坐标顺序错误
PyMuPDF中rect的坐标规则是**(x0, y0, x1, y1)**,其中:
- x0: 左边界,x1: 右边界(x1 > x0)
- y0: 上边界,y1: 下边界(y1 > y0)
你的calculate_text_area函数把y0和y1搞反了,导致clip区域是倒置的无效区域,因此提取不到文本。修改该函数:
def calculate_text_area(page_width, page_height, percentage): border_x = page_width * percentage border_y = page_height * percentage x0 = border_x y0 = border_y # 上边界:从顶部往下留border_y距离 x1 = page_width - border_x y1 = page_height - border_y # 下边界:从底部往上留border_y距离 return x0, y0, x1, y1
其他优化建议
适配多页面尺寸差异:部分PDF的不同页面可能有不同尺寸,建议在提取每页文本时单独计算当前页的clip区域,而非仅使用起始页尺寸:
def get_text_with_area_extraction(filepath: str, start_page: int, end_page: int, percentage) -> str: with fitz.open(filepath) as doc: extracted_text = "" for page_num in range(start_page, end_page + 1): page = doc[page_num] page_width = page.rect.width page_height = page.rect.height x0, y0, x1, y1 = calculate_text_area(page_width, page_height, percentage) page_text = extract_text_by_area(page, x0, y0, x1, y1).replace('\n', ' ').strip() if page_text: extracted_text += page_text + " " standardized_text = re.sub(r'\s+', ' ', extracted_text).strip() return standardized_text调用时直接传入percentage即可,无需提前获取页面尺寸。
增加异常处理:避免文件不存在、页码超出范围、文本提取为空等情况导致脚本崩溃:
def get_text_with_area_extraction(filepath: str, start_page: int, end_page: int, percentage) -> str: if not os.path.exists(filepath): raise FileNotFoundError(f"文件 {filepath} 不存在") with fitz.open(filepath) as doc: if start_page < 0 or end_page >= len(doc) or start_page > end_page: raise ValueError("页码范围无效,请检查输入") extracted_text = "" for page_num in range(start_page, end_page + 1): page = doc[page_num] page_width = page.rect.width page_height = page.rect.height x0, y0, x1, y1 = calculate_text_area(page_width, page_height, percentage) page_text = extract_text_by_area(page, x0, y0, x1, y1).replace('\n', ' ').strip() if page_text: extracted_text += page_text + " " if not extracted_text: raise RuntimeError("未提取到任何文本,请检查clip参数或PDF内容") standardized_text = re.sub(r'\s+', ' ', extracted_text).strip() return standardized_text优化文本清理:增加对页码、特殊符号的清理,提升有声书可读性:
def clean_text(text): # 移除页码(阿拉伯数字或罗马数字) text = re.sub(r'\b\d+\b|\b[IVXLCDM]+\b', '', text) # 移除多余特殊符号 text = re.sub(r'[^\w\s.,!?;:\'-]', '', text) # 合并连续空格 text = re.sub(r'\s+', ' ', text).strip() return text在文本提取完成后调用该函数进一步处理。
分批生成音频:长文本可能导致gTTS处理失败,可拆分文本分批生成后合并音频(需安装
pydub和ffmpeg):from pydub import AudioSegment def split_text(text, chunk_size=5000): # 按句子拆分,避免截断单词 sentences = re.split(r'(?<=[.!?])\s+', text) chunks = [] current_chunk = "" for sentence in sentences: if len(current_chunk) + len(sentence) <= chunk_size: current_chunk += sentence + " " else: chunks.append(current_chunk.strip()) current_chunk = sentence + " " if current_chunk: chunks.append(current_chunk.strip()) return chunks # 使用示例 text_chunks = split_text(extracted_text) audio_segments = [] for i, chunk in enumerate(text_chunks): tts = gTTS(text=chunk, lang='en') temp_file = f"temp_chunk_{i}.mp3" tts.save(temp_file) audio_segments.append(AudioSegment.from_mp3(temp_file)) os.remove(temp_file) # 合并音频 combined_audio = sum(audio_segments) combined_audio.export(output_audio_path, format="mp3")
内容的提问来源于stack exchange,提问作者Jake DAmico
相关产品推荐
相关产品推荐

