You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用fitz的clip参数提取PDF文本返回空值,求解决方法

问题描述

我正在编写Python脚本将PDF转换为有声书,尝试通过设置边框(clip参数)移除页码及其他无关标题。当前使用Python 3.10.4,使用无参数的get_text可正常提取文本,但传入clip参数后返回空值。我使用的输入参数为起始页27、结束页40、边框比例0.05,代码如下:

import fitz
from gtts import gTTS
import os
import re

def extract_text_by_area(page, x0, y0, x1, y1):
    return page.get_text("text", clip=(x0, y0, x1, y1))

def get_text_with_area_extraction(filepath: str, start_page: int, end_page: int, x0, y0, x1, y1) -> str:
    with fitz.open(filepath) as doc:
        extracted_text = ""
        for page_num in range(start_page, end_page + 1):
            page = doc[page_num]
            extracted_text += extract_text_by_area(page, x0, y0, x1, y1).replace('\n', ' ').strip()
        
        standardized_text = re.sub(r'\s+', ' ', extracted_text)
        return standardized_text
    
def get_pdf_dimensions(pdf_path, page_num):
    doc = fitz.open(pdf_path)
    
    page = doc[page_num]
    page_width = page.rect.width
    page_height = page.rect.height
    
    doc.close()
    return page_width, page_height

def calculate_text_area(page_width, page_height, percentage):
    border_x = page_width * percentage
    border_y = page_height * percentage

    x0 = border_x
    y0 = page_height - border_y
    x1 = page_width - border_x
    y1 = border_y

    
    return x0, y0, x1, y1

filepath = 'Fooled-by-Randomness-Role-of-Chance-in-Markets-and-Life-PROPER1.pdf'
start_page = int(input("Enter the starting page: ")) - 1
end_page = int(input("Enter the ending page: ")) - 1

page_num = start_page
page_width, page_height = get_pdf_dimensions(filepath, page_num)

percentage = float(input("Enter the percentage of the border to remove (e.g., 0.1 for 10%): "))
x0, y0, x1, y1 = calculate_text_area(page_width, page_height, percentage)
	extracted_text = get_text_with_area_extraction(filepath, start_page, end_page, x0, y0, x1, y1)

print(extracted_text)

tts = gTTS(text=extracted_text, lang='en')

output_audio_path = 'extracted_text_audio.mp3'
tts.save(output_audio_path)

print(f"Text-to-speech audio saved as: {output_audio_path}")
解决建议

核心问题:Clip坐标顺序错误

PyMuPDF中rect的坐标规则是**(x0, y0, x1, y1)**,其中:

  • x0: 左边界,x1: 右边界(x1 > x0)
  • y0: 上边界,y1: 下边界(y1 > y0)

你的calculate_text_area函数把y0和y1搞反了,导致clip区域是倒置的无效区域,因此提取不到文本。修改该函数:

def calculate_text_area(page_width, page_height, percentage):
    border_x = page_width * percentage
    border_y = page_height * percentage

    x0 = border_x
    y0 = border_y  # 上边界:从顶部往下留border_y距离
    x1 = page_width - border_x
    y1 = page_height - border_y  # 下边界:从底部往上留border_y距离
    
    return x0, y0, x1, y1

其他优化建议

  • 适配多页面尺寸差异:部分PDF的不同页面可能有不同尺寸,建议在提取每页文本时单独计算当前页的clip区域,而非仅使用起始页尺寸:

    def get_text_with_area_extraction(filepath: str, start_page: int, end_page: int, percentage) -> str:
        with fitz.open(filepath) as doc:
            extracted_text = ""
            for page_num in range(start_page, end_page + 1):
                page = doc[page_num]
                page_width = page.rect.width
                page_height = page.rect.height
                x0, y0, x1, y1 = calculate_text_area(page_width, page_height, percentage)
                page_text = extract_text_by_area(page, x0, y0, x1, y1).replace('\n', ' ').strip()
                if page_text:
                    extracted_text += page_text + " "
            
            standardized_text = re.sub(r'\s+', ' ', extracted_text).strip()
            return standardized_text
    

    调用时直接传入percentage即可,无需提前获取页面尺寸。

  • 增加异常处理:避免文件不存在、页码超出范围、文本提取为空等情况导致脚本崩溃:

    def get_text_with_area_extraction(filepath: str, start_page: int, end_page: int, percentage) -> str:
        if not os.path.exists(filepath):
            raise FileNotFoundError(f"文件 {filepath} 不存在")
        with fitz.open(filepath) as doc:
            if start_page < 0 or end_page >= len(doc) or start_page > end_page:
                raise ValueError("页码范围无效,请检查输入")
            extracted_text = ""
            for page_num in range(start_page, end_page + 1):
                page = doc[page_num]
                page_width = page.rect.width
                page_height = page.rect.height
                x0, y0, x1, y1 = calculate_text_area(page_width, page_height, percentage)
                page_text = extract_text_by_area(page, x0, y0, x1, y1).replace('\n', ' ').strip()
                if page_text:
                    extracted_text += page_text + " "
            
            if not extracted_text:
                raise RuntimeError("未提取到任何文本,请检查clip参数或PDF内容")
            standardized_text = re.sub(r'\s+', ' ', extracted_text).strip()
            return standardized_text
    
  • 优化文本清理:增加对页码、特殊符号的清理,提升有声书可读性:

    def clean_text(text):
        # 移除页码(阿拉伯数字或罗马数字)
        text = re.sub(r'\b\d+\b|\b[IVXLCDM]+\b', '', text)
        # 移除多余特殊符号
        text = re.sub(r'[^\w\s.,!?;:\'-]', '', text)
        # 合并连续空格
        text = re.sub(r'\s+', ' ', text).strip()
        return text
    

    在文本提取完成后调用该函数进一步处理。

  • 分批生成音频:长文本可能导致gTTS处理失败,可拆分文本分批生成后合并音频(需安装pydub和ffmpeg):

    from pydub import AudioSegment
    
    def split_text(text, chunk_size=5000):
        # 按句子拆分,避免截断单词
        sentences = re.split(r'(?<=[.!?])\s+', text)
        chunks = []
        current_chunk = ""
        for sentence in sentences:
            if len(current_chunk) + len(sentence) <= chunk_size:
                current_chunk += sentence + " "
            else:
                chunks.append(current_chunk.strip())
                current_chunk = sentence + " "
        if current_chunk:
            chunks.append(current_chunk.strip())
        return chunks
    
    # 使用示例
    text_chunks = split_text(extracted_text)
    audio_segments = []
    for i, chunk in enumerate(text_chunks):
        tts = gTTS(text=chunk, lang='en')
        temp_file = f"temp_chunk_{i}.mp3"
        tts.save(temp_file)
        audio_segments.append(AudioSegment.from_mp3(temp_file))
        os.remove(temp_file)
    
    # 合并音频
    combined_audio = sum(audio_segments)
    combined_audio.export(output_audio_path, format="mp3")
    

内容的提问来源于stack exchange,提问作者Jake DAmico

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.13 08:20:56