You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何让Python修改PDF后新增的文本显示为粗体?

解决PyMuPDF替换文本时粗体字体不生效的问题

针对PyMuPDF(fitz)中insert_text指定粗体字体但无法生效的问题,可通过以下几个方向排查和解决:

1. 预加载字体对象而非直接传字体文件路径

insert_text的fontfile参数在某些场景下可能因临时加载逻辑导致字体未正确应用,建议提前用fitz.Font()预加载字体,再传入font参数:

# 在update_pdf_text函数开头添加字体预加载逻辑
try:
    bold_font = fitz.Font(fontfile=font_path)
except Exception as e:
    print(f"字体加载失败: {e}")
    return

后续调用insert_text时替换参数:

page.insert_text((rect.x0, rect.y0),
                 "F",
                 fontsize=10,
                 font=bold_font,  # 使用预加载的字体对象
                 color=(0, 0, 0))

2. 调整文本插入坐标,避免被填充覆盖

原代码中new_y = rect.y0 + 10会将新文本放在原区域下方,可能超出视野或被遮挡。建议直接使用原文本的顶部坐标rect.y0,确保文本在被白色填充覆盖的原区域内:

new_y = rect.y0  # 替换原new_y = rect.y0 +10的逻辑

3. 验证字体文件有效性

确认你的Helvetica-Bold.ttf是真正的粗体字体文件:

  • 用系统字体查看器打开文件,检查是否显示为粗体样式
  • 替换为已知有效的粗体字体(如Windows的simhei.ttf、Mac的Helvetica Bold.ttf)测试,排除字体文件本身的问题

4. 尝试使用insert_textbox替代insert_text

insert_textbox对字体的渲染更稳定,且可确保文本限制在指定区域内,避免位置偏移:

page.insert_textbox(rect, "F", fontsize=10, fontfile=font_path, color=(0,0,0))

5. 升级PyMuPDF到最新版本

旧版本的PyMuPDF可能存在fontfile参数的bug,执行以下命令升级:

pip install --upgrade pymupdf

修改后的完整示例代码

import os
import shutil
import fitz  # PyMuPDF
import re


def rename_and_copy_files():
    base_directory = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'original')
    updated_directory = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'updated')

    if not os.path.exists(updated_directory):
        os.makedirs(updated_directory)

    font_path = "Helvetica-Bold.ttf"
    if not os.path.isfile(font_path):
        print(f"Font file not found: {font_path}")
        return

    for filename in os.listdir(base_directory):
        if '_R_' in filename:
            new_filename = filename.replace('_R_', '_F_')
            src = os.path.join(base_directory, filename)
            dst = os.path.join(updated_directory, new_filename)

            print(f"Processing file: {filename}")

            if filename.endswith('.pdf'):
                update_pdf_text(src, dst, font_path)
            else:
                shutil.copy2(src, dst)

            print(f"Copied and renamed: {filename} to {new_filename}")


def update_pdf_text(src, dst, font_path):
    document = fitz.open(src)
    
    # 预加载粗体字体
    try:
        bold_font = fitz.Font(fontfile=font_path)
    except Exception as e:
        print(f"Failed to load font: {e}")
        document.close()
        return

    for page_num in range(len(document)):
        page = document[page_num]
        text_instances = page.search_for("_R_")

        for inst in text_instances:
            rect = fitz.Rect(inst)
            full_text, start_rect, end_rect = extract_full_name(page, rect)

            if not full_text:
                continue

            updated_text = full_text.replace('_R_', '_F_')

            page.draw_rect(fitz.Rect(start_rect.x0, start_rect.y0, end_rect.x1, end_rect.y1), color=(1, 1, 1),
                           fill=(1, 1, 1))

            # 使用预加载字体,调整y坐标为原文本顶部
            page.insert_text((start_rect.x0, start_rect.y0),
                             updated_text,
                             fontsize=10,
                             font=bold_font,
                             color=(0, 0, 0))

        single_r_instances = page.search_for(" R ")
        for inst in single_r_instances:
            rect = fitz.Rect(inst)

            page.draw_rect(rect, color=(1, 1, 1), fill=(1, 1, 1))

            # 使用预加载字体,调整y坐标为原文本顶部
            page.insert_text((rect.x0, rect.y0),
                             "F",
                             fontsize=10,
                             font=bold_font,
                             color=(0, 0, 0))

    document.save(dst, garbage=4, deflate=True)
    document.close()


def extract_full_name(page, rect):
    full_text = ""
    start_rect = rect
    end_rect = rect
    words = page.get_text("words")

    name_pattern = re.compile(r'[A-Za-z0-9_\-]+')

    for word in words:
        word_text = word[4]
        if rect.intersects(fitz.Rect(word[:4])) and name_pattern.match(word_text):
            start_rect = fitz.Rect(word[:4]) if fitz.Rect(word[:4]).x0 < start_rect.x0 else start_rect
            end_rect = fitz.Rect(word[:4]) if fitz.Rect(word[:4]).x1 > end_rect.x1 else end_rect
            full_text += word_text

    return full_text, start_rect, end_rect


if __name__ == "__main__":
    rename_and_copy_files()
    print("Process finished")

内容的提问来源于stack exchange,提问作者Luka

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 18:34:50