如何让Python修改PDF后新增的文本显示为粗体?
解决PyMuPDF替换文本时粗体字体不生效的问题
针对PyMuPDF(fitz)中insert_text指定粗体字体但无法生效的问题,可通过以下几个方向排查和解决:
1. 预加载字体对象而非直接传字体文件路径
insert_text的fontfile参数在某些场景下可能因临时加载逻辑导致字体未正确应用,建议提前用fitz.Font()预加载字体,再传入font参数:
# 在update_pdf_text函数开头添加字体预加载逻辑 try: bold_font = fitz.Font(fontfile=font_path) except Exception as e: print(f"字体加载失败: {e}") return
后续调用insert_text时替换参数:
page.insert_text((rect.x0, rect.y0), "F", fontsize=10, font=bold_font, # 使用预加载的字体对象 color=(0, 0, 0))
2. 调整文本插入坐标,避免被填充覆盖
原代码中new_y = rect.y0 + 10会将新文本放在原区域下方,可能超出视野或被遮挡。建议直接使用原文本的顶部坐标rect.y0,确保文本在被白色填充覆盖的原区域内:
new_y = rect.y0 # 替换原new_y = rect.y0 +10的逻辑
3. 验证字体文件有效性
确认你的Helvetica-Bold.ttf是真正的粗体字体文件:
- 用系统字体查看器打开文件,检查是否显示为粗体样式
- 替换为已知有效的粗体字体(如Windows的
simhei.ttf、Mac的Helvetica Bold.ttf)测试,排除字体文件本身的问题
4. 尝试使用insert_textbox替代insert_text
insert_textbox对字体的渲染更稳定,且可确保文本限制在指定区域内,避免位置偏移:
page.insert_textbox(rect, "F", fontsize=10, fontfile=font_path, color=(0,0,0))
5. 升级PyMuPDF到最新版本
旧版本的PyMuPDF可能存在fontfile参数的bug,执行以下命令升级:
pip install --upgrade pymupdf
修改后的完整示例代码
import os import shutil import fitz # PyMuPDF import re def rename_and_copy_files(): base_directory = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'original') updated_directory = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'updated') if not os.path.exists(updated_directory): os.makedirs(updated_directory) font_path = "Helvetica-Bold.ttf" if not os.path.isfile(font_path): print(f"Font file not found: {font_path}") return for filename in os.listdir(base_directory): if '_R_' in filename: new_filename = filename.replace('_R_', '_F_') src = os.path.join(base_directory, filename) dst = os.path.join(updated_directory, new_filename) print(f"Processing file: {filename}") if filename.endswith('.pdf'): update_pdf_text(src, dst, font_path) else: shutil.copy2(src, dst) print(f"Copied and renamed: {filename} to {new_filename}") def update_pdf_text(src, dst, font_path): document = fitz.open(src) # 预加载粗体字体 try: bold_font = fitz.Font(fontfile=font_path) except Exception as e: print(f"Failed to load font: {e}") document.close() return for page_num in range(len(document)): page = document[page_num] text_instances = page.search_for("_R_") for inst in text_instances: rect = fitz.Rect(inst) full_text, start_rect, end_rect = extract_full_name(page, rect) if not full_text: continue updated_text = full_text.replace('_R_', '_F_') page.draw_rect(fitz.Rect(start_rect.x0, start_rect.y0, end_rect.x1, end_rect.y1), color=(1, 1, 1), fill=(1, 1, 1)) # 使用预加载字体,调整y坐标为原文本顶部 page.insert_text((start_rect.x0, start_rect.y0), updated_text, fontsize=10, font=bold_font, color=(0, 0, 0)) single_r_instances = page.search_for(" R ") for inst in single_r_instances: rect = fitz.Rect(inst) page.draw_rect(rect, color=(1, 1, 1), fill=(1, 1, 1)) # 使用预加载字体,调整y坐标为原文本顶部 page.insert_text((rect.x0, rect.y0), "F", fontsize=10, font=bold_font, color=(0, 0, 0)) document.save(dst, garbage=4, deflate=True) document.close() def extract_full_name(page, rect): full_text = "" start_rect = rect end_rect = rect words = page.get_text("words") name_pattern = re.compile(r'[A-Za-z0-9_\-]+') for word in words: word_text = word[4] if rect.intersects(fitz.Rect(word[:4])) and name_pattern.match(word_text): start_rect = fitz.Rect(word[:4]) if fitz.Rect(word[:4]).x0 < start_rect.x0 else start_rect end_rect = fitz.Rect(word[:4]) if fitz.Rect(word[:4]).x1 > end_rect.x1 else end_rect full_text += word_text return full_text, start_rect, end_rect if __name__ == "__main__": rename_and_copy_files() print("Process finished")
内容的提问来源于stack exchange,提问作者Luka
相关产品推荐
相关产品推荐

