使用pdfplumber提取大型PDF表格时存在字段为空问题
问题描述
使用Python的pdfplumber库提取PDF中的表格,目标表格位于葡萄牙语短语“Quadro de Definições”第二次出现后的12页内。脚本可正常运行,但部分字段(如Custodiante、Fundo、Gestora、Escriturador等)的内容缺失。
原始提取脚本如下:
import logging import pdfplumber import pandas as pd import os import re # 配置日志以显示信息类消息 logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') # 设置PDF文件的固定路径 PDF_PATH = 'data/prospectos/52670402000105-opd08122023v01-000566736.pdf' def extract_tables_from_pdf(pdf_path): """ 从指定PDF文件中提取表格。 参数: pdf_path (str): PDF文件的路径。 返回: list: 提取到的表格列表,若提取失败则返回None。 """ # 检查PDF文件是否存在 if not os.path.exists(pdf_path): logging.error(f"文件 {pdf_path} 不存在。") return None try: # 使用pdfplumber打开PDF文件 with pdfplumber.open(pdf_path) as pdf: logging.info(f"PDF加载成功。总页数: {len(pdf.pages)}") # 查找"Quadro de Definições"的第二次出现位置 page_with_second_occurrence = None occurrences = 0 for page_num, page in enumerate(pdf.pages, start=1): if "Quadro de Definições" in page.extract_text(): occurrences += 1 if occurrences == 2: page_with_second_occurrence = page_num break # 检查是否找到第二次出现的位置 if page_with_second_occurrence is None: logging.warning("未找到两次'Quadro de Definições'的出现。") return None logging.info(f"第二次'Quadro de Definições'出现在第 {page_with_second_occurrence} 页") # 定义要提取的页面范围(第二次出现位置之后的12页) start_page = page_with_second_occurrence end_page = min(start_page + 12, len(pdf.pages)) logging.info(f"正在提取第 {start_page} 至第 {end_page} 页的表格") # 从指定页面范围提取表格 tables = [] for page in pdf.pages[start_page-1:end_page]: page_tables = page.extract_tables() if page_tables: tables.extend(page_tables) logging.info(f"提取到的表格数量: {len(tables)}") return tables except Exception as e: logging.error(f"处理PDF时发生错误: {str(e)}") return None def safe_strip(cell): """ 安全地去除单元格内容的空白字符,处理None值。 参数: cell: 要去除空白的单元格内容。 返回: str: 去除空白后的字符串,若cell为None则返回空字符串。 """ if cell is None: return '' return str(cell).strip() def process_and_combine_tables(tables): """ 处理并合并所有提取到的表格为单个DataFrame。 参数: tables (list): 从PDF中提取到的表格列表。 返回: pandas.DataFrame: 包含所有处理后合并表格数据的DataFrame。 """ processed_tables = [] for table_index, table in enumerate(tables): # 移除空行 table = [row for row in table if row and any(safe_strip(cell) != '' for cell in row)] # 处理每一行 processed_rows = [] for row in table: if len(row) == 1: # 若行只有一列,在第一个双空格处拆分为两列 content = safe_strip(row[0]) split_row = re.split(r'\s{2,}', content, maxsplit=1) processed_rows.append(split_row if len(split_row) == 2 else [split_row[0], '']) else: # 若行有多列,取前两列 processed_rows.append([safe_strip(cell) for cell in row[:2]]) processed_tables.extend(processed_rows) # 使用所有处理后的行创建DataFrame df = pd.DataFrame(processed_tables, columns=['Term', 'Definition']) # 移除Term为空的行 df = df[df['Term'] != ''] # 移除重复项,保留首次出现的条目 df = df.drop_duplicates(subset='Term', keep='first') # 将'Term'设为DataFrame的索引 df.set_index('Term', inplace=True) return df if __name__ == "__main__": # 记录表格提取过程的开始 logging.info(f"开始从文件提取表格: {PDF_PATH}") # 从PDF中提取表格 extracted_tables = extract_tables_from_pdf(PDF_PATH) if extracted_tables: try: # 处理并合并提取到的表格 df_combined = process_and_combine_tables(extracted_tables) print(df_combined) # 将合并后的DataFrame保存为CSV文件 csv_path = 'combined definitions framework.csv' df_combined.to_csv(csv_path) logging.info(f"合并后的DataFrame已保存至 '{csv_path}'") except Exception as e: logging.error(f"处理并合并表格时发生错误: {str(e)}") else: logging.warning("无法从PDF中提取表格。") # 记录提取和合并过程的完成 logging.info("表格提取与合并过程已完成。")
解决策略
1. 优化pdfplumber表格提取参数
默认的表格提取参数可能无法精准识别PDF中的表格边界,导致单元格内容被拆分或丢失。可以通过自定义table_settings参数强化边界识别:
修改extract_tables的调用代码:
page_tables = page.extract_tables(table_settings={ "vertical_strategy": "lines", # 基于线条识别垂直边界 "horizontal_strategy": "lines", # 基于线条识别水平边界 "snap_tolerance": 3, # 允许线条与单元格边缘的偏差 "join_tolerance": 3, # 合并相近的线条 "edge_min_length": 10, # 忽略过短的线条 })
2. 处理跨页/跨行的单元格内容
部分字段的定义可能跨多行或跨页,当前脚本仅提取单行内容导致缺失。需要跟踪当前Term,将后续行的内容追加到对应Definition中:
重写process_and_combine_tables函数:
def process_and_combine_tables(tables): processed_tables = [] current_term = None current_definition = [] for table in tables: # 过滤空行 table = [row for row in table if row and any(safe_strip(cell) != '' for cell in row)] for row in table: processed_cells = [safe_strip(cell) for cell in row[:2]] term, definition = processed_cells[0], processed_cells[1] if term: # 保存上一个Term的完整定义 if current_term is not None: processed_tables.append([current_term, ' '.join(current_definition)]) current_term = term current_definition = [definition] if definition else [] else: # 将内容追加到当前Term的定义中 if definition and current_term is not None: current_definition.append(definition) # 保存最后一条Term的定义 if current_term is not None: processed_tables.append([current_term, ' '.join(current_definition)]) df = pd.DataFrame(processed_tables, columns=['Term', 'Definition']) df = df[df['Term'] != ''] df = df.drop_duplicates(subset='Term', keep='first') df.set_index('Term', inplace=True) return df
3. 基于位置拆分Term和Definition
如果PDF表格是固定布局(Term在左侧,Definition在右侧),可以通过文本的x坐标直接分割,替代双空格拆分的逻辑:
新增一个按位置提取的函数,并在提取流程中使用:
def extract_terms_by_position(page): # 提取页面所有单词及位置信息 words = page.extract_words(x_tolerance=2, y_tolerance=2) terms = [] current_term = None current_def = [] # 自定义分割x坐标(需根据实际PDF调整,可通过page.debug_table()查看) split_x = 300 for word in words: if word['x0'] < split_x: # 遇到新的Term,保存上一条记录 if current_term is not None: terms.append([current_term, ' '.join(current_def)]) current_term = word['text'] current_def = [] else: # 将单词追加到当前Definition current_def.append(word['text']) # 保存最后一条记录 if current_term is not None: terms.append([current_term, ' '.join(current_def)]) return terms
然后在extract_tables_from_pdf的页面循环中,可选择使用该方法替代extract_tables:
# 替换原有的表格提取逻辑 processed_tables = [] for page in pdf.pages[start_page-1:end_page]: # 优先尝试按位置提取 position_terms = extract_terms_by_position(page) if position_terms: processed_tables.extend(position_terms) else: # fallback到表格提取 page_tables = page.extract_tables(table_settings={ "vertical_strategy": "lines", "horizontal_strategy": "lines", "snap_tolerance": 3, }) if page_tables: # 转换为统一格式后加入 for table in page_tables: table = [row for row in table if row and any(safe_strip(cell) != '' for cell in row)] for row in table: processed_cells = [safe_strip(cell) for cell in row[:2]] processed_tables.append(processed_cells)
4. 手动定位缺失字段提取
对于始终无法自动提取的字段,可直接定位其在PDF中的位置,单独提取该区域的文本:
def extract_specific_term(page, term_name): # 先定位Term所在的位置 text = page.extract_text() if term_name not in text: return '' # 获取Term的坐标范围(可通过page.debug_table()可视化查看) # 示例:假设Custodiante在页面的特定区域 bbox = (50, 200, 550, 300) # 自定义坐标(x0, top, x1, bottom) term_def = page.within_bbox(bbox).extract_text() # 清理文本,拆分Term和Definition return term_def.split(term_name, 1)[-1].strip()
5. 结合OCR处理(针对扫描版PDF)
如果目标PDF是扫描生成的图片型PDF,pdfplumber的文本提取会失效,此时需要结合OCR工具(如pytesseract)先识别文本:
import pytesseract from pdf2image import convert_from_path def extract_scanned_pdf(pdf_path): pages = convert_from_path(pdf_path) text_content = [] for page in pages: text = pytesseract.image_to_string(page, lang='por') # 葡萄牙语识别 text_content.append(text) return text_content
之后再从识别后的文本中提取表格内容。
内容的提问来源于stack exchange,提问作者Reinaldo Chaves
相关产品推荐
相关产品推荐

