如何将Eurostat目录PDF提取为关联标题、链接的可检索结构化JSON
欧盟统计局目录PDF结构化提取方案
核心实现思路
你当前的代码已经覆盖了链接提取、文本提取、文件下载三个独立功能,缺少的核心环节是链接与对应标题、章节层级的匹配关联,可以通过「坐标匹配」方案实现:所有PDF内的链接标注和对应的标题文本都有重合/邻近的坐标范围,逐页同步提取两类元素的坐标后,即可完成匹配;再通过文本的缩进量判断章节层级,最终拼装为结构化数据。
前置依赖安装
pip install pypdf2 requests
完整整合代码
import json import requests import PyPDF2 from PyPDF2.generic import NameObject # 全局配置 PDF_PATH = "table_of_contents_en.pdf" OUTPUT_JSON = "eurostat_toc.json" DOWNLOAD_TSV = False # 可配置是否下载tsv.gz文件 TSV_SAVE_DIR = "./tsv_files/" # 坐标匹配阈值:链接矩形和文本矩形的重合度超过该值则判定为关联 COORD_THRESHOLD = 0.6 def get_text_blocks(page): """提取单页所有带坐标的文本块""" text_blocks = [] if hasattr(page, 'extract_text_with_layout'): layout = page.extract_text_with_layout() for block in layout: if block['text'].strip(): text_blocks.append({ "text": block['text'].strip(), "x0": block['x0'], "y0": block['y0'], "x1": block['x1'], "y1": block['y1'], "indent": block['x0'] # 用x坐标作为缩进量判断章节层级 }) else: # 兼容旧版PyPDF2,若坐标提取不准可替换为pdfplumber库 raw_text = page.extract_text() lines = [l.strip() for l in raw_text.split('\n') if l.strip()] # 简单模拟缩进,实际使用建议换pdfplumber for idx, line in enumerate(lines): indent = len(line) - len(line.lstrip()) text_blocks.append({ "text": line.lstrip(), "indent": indent, "y_order": idx # 用行顺序匹配 }) return text_blocks def get_link_annotations(page): """提取单页所有带坐标的链接""" links = [] key = NameObject('/Annots') uri_key = NameObject('/URI') a_key = NameObject('/A') rect_key = NameObject('/Rect') if key not in page: return links for ann in page[key]: ann_obj = ann.get_object() if a_key not in ann_obj or rect_key not in ann_obj: continue a_obj = ann_obj[a_key] if uri_key not in a_obj: continue uri = a_obj[uri_key] rect = ann_obj[rect_key] links.append({ "url": uri, "x0": rect[0], "y0": rect[1], "x1": rect[2], "y1": rect[3] }) return links def calc_overlap(rect1, rect2): """计算两个矩形的重合度""" x_left = max(rect1['x0'], rect2['x0']) y_bottom = max(rect1['y0'], rect2['y0']) x_right = min(rect1['x1'], rect2['x1']) y_top = min(rect1['y1'], rect2['y1']) if x_right < x_left or y_top < y_bottom: return 0 overlap_area = (x_right - x_left) * (y_top - y_bottom) rect1_area = (rect1['x1'] - rect1['x0']) * (rect1['y1'] - rect1['y0']) return overlap_area / rect1_area if rect1_area > 0 else 0 def main(): pdf_file = open(PDF_PATH, 'rb') pdf_reader = PyPDF2.PdfFileReader(pdf_file) total_pages = pdf_reader.getNumPages() all_items = [] chapter_stack = [] # 维护当前章节路径栈 last_indent = -1 for page_num in range(total_pages): print(f"处理第 {page_num+1}/{total_pages} 页") page = pdf_reader.getPage(page_num) # 同时提取文本块和链接 text_blocks = get_text_blocks(page) links = get_link_annotations(page) for text_block in text_blocks: # 匹配关联链接 block_links = [] for link in links: overlap = calc_overlap(text_block, link) if overlap >= COORD_THRESHOLD: block_links.append(link['url']) # 判定章节层级 current_indent = text_block['indent'] if current_indent > last_indent: chapter_stack.append(text_block['text']) elif current_indent < last_indent: pop_count = (last_indent - current_indent) // 10 # 按缩进差计算退栈层数,可根据实际PDF调整 for _ in range(pop_count): if chapter_stack: chapter_stack.pop() chapter_stack.append(text_block['text']) else: if chapter_stack: chapter_stack.pop() chapter_stack.append(text_block['text']) last_indent = current_indent # 区分3类链接 link_map = {"tsv_gz": None, "html": None, "other": []} code = None for url in block_links: if url.endswith(".tsv.gz"): link_map["tsv_gz"] = url code = url.split("/")[-1].replace(".tsv.gz", "") # 可选下载tsv文件 if DOWNLOAD_TSV and link_map["tsv_gz"]: r = requests.get(link_map["tsv_gz"], allow_redirects=True) with open(f"{TSV_SAVE_DIR}{code}.tsv.gz", "wb") as f: f.write(r.content) elif url.endswith(".htm") or url.endswith(".html"): link_map["html"] = url else: link_map["other"].append(url) # 存入结果 all_items.append({ "chapter_path": chapter_stack.copy(), "title": text_block['text'], "dataset_code": code, "links": link_map }) # 输出JSON with open(OUTPUT_JSON, "w", encoding="utf-8") as f: json.dump(all_items, f, ensure_ascii=False, indent=2) pdf_file.close() print(f"处理完成,结果已保存至 {OUTPUT_JSON}") if __name__ == "__main__": import os if DOWNLOAD_TSV and not os.path.exists(TSV_SAVE_DIR): os.makedirs(TSV_SAVE_DIR) main()
输出JSON结构说明
每个条目包含完整的章节归属、标题、数据集编码、三类关联链接,示例如下:
{ "chapter_path": [ "General and regional statistics", "Population and social conditions", "Population" ], "title": "Population on 1 January by age and sex", "dataset_code": "demo_pjan", "links": { "tsv_gz": "[对应数据压缩包地址]", "html": "[对应元数据说明页地址]", "other": [] } }
常见优化点
- 若坐标匹配准确率低,可将PyPDF2替换为
pdfplumber库,文本和坐标提取精度更高 - 章节缩进的判断阈值可根据你下载的PDF实际排版调整,避免层级识别错误
- 可以增加去重逻辑,过滤非叶子节点的章节标题条目
内容的提问来源于stack exchange,提问作者alma korte
相关产品推荐
相关产品推荐

