You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将Eurostat目录PDF提取为关联标题、链接的可检索结构化JSON

欧盟统计局目录PDF结构化提取方案

核心实现思路

你当前的代码已经覆盖了链接提取、文本提取、文件下载三个独立功能,缺少的核心环节是链接与对应标题、章节层级的匹配关联,可以通过「坐标匹配」方案实现:所有PDF内的链接标注和对应的标题文本都有重合/邻近的坐标范围,逐页同步提取两类元素的坐标后,即可完成匹配;再通过文本的缩进量判断章节层级,最终拼装为结构化数据。

前置依赖安装

pip install pypdf2 requests

完整整合代码

import json
import requests
import PyPDF2
from PyPDF2.generic import NameObject

# 全局配置
PDF_PATH = "table_of_contents_en.pdf"
OUTPUT_JSON = "eurostat_toc.json"
DOWNLOAD_TSV = False # 可配置是否下载tsv.gz文件
TSV_SAVE_DIR = "./tsv_files/"

# 坐标匹配阈值:链接矩形和文本矩形的重合度超过该值则判定为关联
COORD_THRESHOLD = 0.6

def get_text_blocks(page):
    """提取单页所有带坐标的文本块"""
    text_blocks = []
    if hasattr(page, 'extract_text_with_layout'):
        layout = page.extract_text_with_layout()
        for block in layout:
            if block['text'].strip():
                text_blocks.append({
                    "text": block['text'].strip(),
                    "x0": block['x0'],
                    "y0": block['y0'],
                    "x1": block['x1'],
                    "y1": block['y1'],
                    "indent": block['x0'] # 用x坐标作为缩进量判断章节层级
                })
    else:
        # 兼容旧版PyPDF2,若坐标提取不准可替换为pdfplumber库
        raw_text = page.extract_text()
        lines = [l.strip() for l in raw_text.split('\n') if l.strip()]
        # 简单模拟缩进,实际使用建议换pdfplumber
        for idx, line in enumerate(lines):
            indent = len(line) - len(line.lstrip())
            text_blocks.append({
                "text": line.lstrip(),
                "indent": indent,
                "y_order": idx # 用行顺序匹配
            })
    return text_blocks

def get_link_annotations(page):
    """提取单页所有带坐标的链接"""
    links = []
    key = NameObject('/Annots')
    uri_key = NameObject('/URI')
    a_key = NameObject('/A')
    rect_key = NameObject('/Rect')
    if key not in page:
        return links
    for ann in page[key]:
        ann_obj = ann.get_object()
        if a_key not in ann_obj or rect_key not in ann_obj:
            continue
        a_obj = ann_obj[a_key]
        if uri_key not in a_obj:
            continue
        uri = a_obj[uri_key]
        rect = ann_obj[rect_key]
        links.append({
            "url": uri,
            "x0": rect[0],
            "y0": rect[1],
            "x1": rect[2],
            "y1": rect[3]
        })
    return links

def calc_overlap(rect1, rect2):
    """计算两个矩形的重合度"""
    x_left = max(rect1['x0'], rect2['x0'])
    y_bottom = max(rect1['y0'], rect2['y0'])
    x_right = min(rect1['x1'], rect2['x1'])
    y_top = min(rect1['y1'], rect2['y1'])
    if x_right < x_left or y_top < y_bottom:
        return 0
    overlap_area = (x_right - x_left) * (y_top - y_bottom)
    rect1_area = (rect1['x1'] - rect1['x0']) * (rect1['y1'] - rect1['y0'])
    return overlap_area / rect1_area if rect1_area > 0 else 0

def main():
    pdf_file = open(PDF_PATH, 'rb')
    pdf_reader = PyPDF2.PdfFileReader(pdf_file)
    total_pages = pdf_reader.getNumPages()
    all_items = []
    chapter_stack = [] # 维护当前章节路径栈
    last_indent = -1

    for page_num in range(total_pages):
        print(f"处理第 {page_num+1}/{total_pages} 页")
        page = pdf_reader.getPage(page_num)
        # 同时提取文本块和链接
        text_blocks = get_text_blocks(page)
        links = get_link_annotations(page)

        for text_block in text_blocks:
            # 匹配关联链接
            block_links = []
            for link in links:
                overlap = calc_overlap(text_block, link)
                if overlap >= COORD_THRESHOLD:
                    block_links.append(link['url'])
            # 判定章节层级
            current_indent = text_block['indent']
            if current_indent > last_indent:
                chapter_stack.append(text_block['text'])
            elif current_indent < last_indent:
                pop_count = (last_indent - current_indent) // 10 # 按缩进差计算退栈层数,可根据实际PDF调整
                for _ in range(pop_count):
                    if chapter_stack:
                        chapter_stack.pop()
                chapter_stack.append(text_block['text'])
            else:
                if chapter_stack:
                    chapter_stack.pop()
                chapter_stack.append(text_block['text'])
            last_indent = current_indent

            # 区分3类链接
            link_map = {"tsv_gz": None, "html": None, "other": []}
            code = None
            for url in block_links:
                if url.endswith(".tsv.gz"):
                    link_map["tsv_gz"] = url
                    code = url.split("/")[-1].replace(".tsv.gz", "")
                    # 可选下载tsv文件
                    if DOWNLOAD_TSV and link_map["tsv_gz"]:
                        r = requests.get(link_map["tsv_gz"], allow_redirects=True)
                        with open(f"{TSV_SAVE_DIR}{code}.tsv.gz", "wb") as f:
                            f.write(r.content)
                elif url.endswith(".htm") or url.endswith(".html"):
                    link_map["html"] = url
                else:
                    link_map["other"].append(url)
            
            # 存入结果
            all_items.append({
                "chapter_path": chapter_stack.copy(),
                "title": text_block['text'],
                "dataset_code": code,
                "links": link_map
            })
    
    # 输出JSON
    with open(OUTPUT_JSON, "w", encoding="utf-8") as f:
        json.dump(all_items, f, ensure_ascii=False, indent=2)
    pdf_file.close()
    print(f"处理完成,结果已保存至 {OUTPUT_JSON}")

if __name__ == "__main__":
    import os
    if DOWNLOAD_TSV and not os.path.exists(TSV_SAVE_DIR):
        os.makedirs(TSV_SAVE_DIR)
    main()

输出JSON结构说明

每个条目包含完整的章节归属、标题、数据集编码、三类关联链接,示例如下:

{
  "chapter_path": [
    "General and regional statistics",
    "Population and social conditions",
    "Population"
  ],
  "title": "Population on 1 January by age and sex",
  "dataset_code": "demo_pjan",
  "links": {
    "tsv_gz": "[对应数据压缩包地址]",
    "html": "[对应元数据说明页地址]",
    "other": []
  }
}

常见优化点

  • 若坐标匹配准确率低,可将PyPDF2替换为pdfplumber库,文本和坐标提取精度更高
  • 章节缩进的判断阈值可根据你下载的PDF实际排版调整,避免层级识别错误
  • 可以增加去重逻辑,过滤非叶子节点的章节标题条目

内容的提问来源于stack exchange,提问作者alma korte

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.25 22:27:07