You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用pdfplumber提取大型PDF表格时存在字段为空问题

问题描述

使用Python的pdfplumber库提取PDF中的表格,目标表格位于葡萄牙语短语“Quadro de Definições”第二次出现后的12页内。脚本可正常运行,但部分字段(如Custodiante、Fundo、Gestora、Escriturador等)的内容缺失。

原始提取脚本如下:

import logging
import pdfplumber
import pandas as pd
import os
import re

# 配置日志以显示信息类消息
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')

# 设置PDF文件的固定路径
PDF_PATH = 'data/prospectos/52670402000105-opd08122023v01-000566736.pdf'

def extract_tables_from_pdf(pdf_path):
    """
    从指定PDF文件中提取表格。
    
    参数:
    pdf_path (str): PDF文件的路径。
    
    返回:
    list: 提取到的表格列表,若提取失败则返回None。
    """
    # 检查PDF文件是否存在
    if not os.path.exists(pdf_path):
        logging.error(f"文件 {pdf_path} 不存在。")
        return None

    try:
        # 使用pdfplumber打开PDF文件
        with pdfplumber.open(pdf_path) as pdf:
            logging.info(f"PDF加载成功。总页数: {len(pdf.pages)}")

            # 查找"Quadro de Definições"的第二次出现位置
            page_with_second_occurrence = None
            occurrences = 0
            for page_num, page in enumerate(pdf.pages, start=1):
                if "Quadro de Definições" in page.extract_text():
                    occurrences += 1
                    if occurrences == 2:
                        page_with_second_occurrence = page_num
                        break

            # 检查是否找到第二次出现的位置
            if page_with_second_occurrence is None:
                logging.warning("未找到两次'Quadro de Definições'的出现。")
                return None
            
            logging.info(f"第二次'Quadro de Definições'出现在第 {page_with_second_occurrence} 页")

            # 定义要提取的页面范围(第二次出现位置之后的12页)
            start_page = page_with_second_occurrence
            end_page = min(start_page + 12, len(pdf.pages))

            logging.info(f"正在提取第 {start_page} 至第 {end_page} 页的表格")

            # 从指定页面范围提取表格
            tables = []
            for page in pdf.pages[start_page-1:end_page]:
                page_tables = page.extract_tables()
                if page_tables:
                    tables.extend(page_tables)

            logging.info(f"提取到的表格数量: {len(tables)}")

            return tables

    except Exception as e:
        logging.error(f"处理PDF时发生错误: {str(e)}")
        return None

def safe_strip(cell):
    """
    安全地去除单元格内容的空白字符,处理None值。
    
    参数:
    cell: 要去除空白的单元格内容。
    
    返回:
    str: 去除空白后的字符串,若cell为None则返回空字符串。
    """
    if cell is None:
        return ''
    return str(cell).strip()

def process_and_combine_tables(tables):
    """
    处理并合并所有提取到的表格为单个DataFrame。
    
    参数:
    tables (list): 从PDF中提取到的表格列表。
    
    返回:
    pandas.DataFrame: 包含所有处理后合并表格数据的DataFrame。
    """
    processed_tables = []
    for table_index, table in enumerate(tables):
        # 移除空行
        table = [row for row in table if row and any(safe_strip(cell) != '' for cell in row)]
        
        # 处理每一行
        processed_rows = []
        for row in table:
            if len(row) == 1:
                # 若行只有一列,在第一个双空格处拆分为两列
                content = safe_strip(row[0])
                split_row = re.split(r'\s{2,}', content, maxsplit=1)
                processed_rows.append(split_row if len(split_row) == 2 else [split_row[0], ''])
            else:
                # 若行有多列,取前两列
                processed_rows.append([safe_strip(cell) for cell in row[:2]])
        
        processed_tables.extend(processed_rows)
    
    # 使用所有处理后的行创建DataFrame
    df = pd.DataFrame(processed_tables, columns=['Term', 'Definition'])
    
    # 移除Term为空的行
    df = df[df['Term'] != '']
    
    # 移除重复项,保留首次出现的条目
    df = df.drop_duplicates(subset='Term', keep='first')
    
    # 将'Term'设为DataFrame的索引
    df.set_index('Term', inplace=True)
    
    return df

if __name__ == "__main__":
    # 记录表格提取过程的开始
    logging.info(f"开始从文件提取表格: {PDF_PATH}")
    
    # 从PDF中提取表格
    extracted_tables = extract_tables_from_pdf(PDF_PATH)
    
    if extracted_tables:
        try:
            # 处理并合并提取到的表格
            df_combined = process_and_combine_tables(extracted_tables)
            print(df_combined)
            
            # 将合并后的DataFrame保存为CSV文件
            csv_path = 'combined definitions framework.csv'
            df_combined.to_csv(csv_path)
            logging.info(f"合并后的DataFrame已保存至 '{csv_path}'")
        except Exception as e:
            logging.error(f"处理并合并表格时发生错误: {str(e)}")
    else:
        logging.warning("无法从PDF中提取表格。")
    
    # 记录提取和合并过程的完成
    logging.info("表格提取与合并过程已完成。")

解决策略

1. 优化pdfplumber表格提取参数

默认的表格提取参数可能无法精准识别PDF中的表格边界,导致单元格内容被拆分或丢失。可以通过自定义table_settings参数强化边界识别:

修改extract_tables的调用代码:

page_tables = page.extract_tables(table_settings={
    "vertical_strategy": "lines",  # 基于线条识别垂直边界
    "horizontal_strategy": "lines", # 基于线条识别水平边界
    "snap_tolerance": 3,  # 允许线条与单元格边缘的偏差
    "join_tolerance": 3,  # 合并相近的线条
    "edge_min_length": 10, # 忽略过短的线条
})

2. 处理跨页/跨行的单元格内容

部分字段的定义可能跨多行或跨页,当前脚本仅提取单行内容导致缺失。需要跟踪当前Term,将后续行的内容追加到对应Definition中:

重写process_and_combine_tables函数:

def process_and_combine_tables(tables):
    processed_tables = []
    current_term = None
    current_definition = []
    
    for table in tables:
        # 过滤空行
        table = [row for row in table if row and any(safe_strip(cell) != '' for cell in row)]
        
        for row in table:
            processed_cells = [safe_strip(cell) for cell in row[:2]]
            term, definition = processed_cells[0], processed_cells[1]
            
            if term:
                # 保存上一个Term的完整定义
                if current_term is not None:
                    processed_tables.append([current_term, ' '.join(current_definition)])
                current_term = term
                current_definition = [definition] if definition else []
            else:
                # 将内容追加到当前Term的定义中
                if definition and current_term is not None:
                    current_definition.append(definition)
    
    # 保存最后一条Term的定义
    if current_term is not None:
        processed_tables.append([current_term, ' '.join(current_definition)])
    
    df = pd.DataFrame(processed_tables, columns=['Term', 'Definition'])
    df = df[df['Term'] != '']
    df = df.drop_duplicates(subset='Term', keep='first')
    df.set_index('Term', inplace=True)
    
    return df

3. 基于位置拆分Term和Definition

如果PDF表格是固定布局(Term在左侧,Definition在右侧),可以通过文本的x坐标直接分割,替代双空格拆分的逻辑:

新增一个按位置提取的函数,并在提取流程中使用:

def extract_terms_by_position(page):
    # 提取页面所有单词及位置信息
    words = page.extract_words(x_tolerance=2, y_tolerance=2)
    terms = []
    current_term = None
    current_def = []
    # 自定义分割x坐标(需根据实际PDF调整,可通过page.debug_table()查看)
    split_x = 300
    
    for word in words:
        if word['x0'] < split_x:
            # 遇到新的Term,保存上一条记录
            if current_term is not None:
                terms.append([current_term, ' '.join(current_def)])
            current_term = word['text']
            current_def = []
        else:
            # 将单词追加到当前Definition
            current_def.append(word['text'])
    
    # 保存最后一条记录
    if current_term is not None:
        terms.append([current_term, ' '.join(current_def)])
    return terms

然后在extract_tables_from_pdf的页面循环中,可选择使用该方法替代extract_tables:

# 替换原有的表格提取逻辑
processed_tables = []
for page in pdf.pages[start_page-1:end_page]:
    # 优先尝试按位置提取
    position_terms = extract_terms_by_position(page)
    if position_terms:
        processed_tables.extend(position_terms)
    else:
        # fallback到表格提取
        page_tables = page.extract_tables(table_settings={
            "vertical_strategy": "lines",
            "horizontal_strategy": "lines",
            "snap_tolerance": 3,
        })
        if page_tables:
            # 转换为统一格式后加入
            for table in page_tables:
                table = [row for row in table if row and any(safe_strip(cell) != '' for cell in row)]
                for row in table:
                    processed_cells = [safe_strip(cell) for cell in row[:2]]
                    processed_tables.append(processed_cells)

4. 手动定位缺失字段提取

对于始终无法自动提取的字段,可直接定位其在PDF中的位置,单独提取该区域的文本:

def extract_specific_term(page, term_name):
    # 先定位Term所在的位置
    text = page.extract_text()
    if term_name not in text:
        return ''
    # 获取Term的坐标范围(可通过page.debug_table()可视化查看)
    # 示例:假设Custodiante在页面的特定区域
    bbox = (50, 200, 550, 300) # 自定义坐标(x0, top, x1, bottom)
    term_def = page.within_bbox(bbox).extract_text()
    # 清理文本,拆分Term和Definition
    return term_def.split(term_name, 1)[-1].strip()

5. 结合OCR处理(针对扫描版PDF)

如果目标PDF是扫描生成的图片型PDF,pdfplumber的文本提取会失效,此时需要结合OCR工具(如pytesseract)先识别文本:

import pytesseract
from pdf2image import convert_from_path

def extract_scanned_pdf(pdf_path):
    pages = convert_from_path(pdf_path)
    text_content = []
    for page in pages:
        text = pytesseract.image_to_string(page, lang='por') # 葡萄牙语识别
        text_content.append(text)
    return text_content

之后再从识别后的文本中提取表格内容。

内容的提问来源于stack exchange,提问作者Reinaldo Chaves

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.17 19:54:52