You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于PyMuPDF与PyQt6的PDF编辑器:编辑后文本与图像的布局还原问题

基于PyMuPDF与PyQt6的PDF编辑器:编辑后文本与图像的布局还原问题

这个问题我之前做类似PDF编辑器时也踩过坑!核心问题其实是PyMuPDF(PDF标准)和Qt QGraphicsScene的坐标系统不匹配,再加上你保存时的一些细节没处理到位,比如固定字体、直接修改原文档导致重叠、图像字节流格式错误这些,下面一步步给你解决:


一、先搞懂坐标系统的核心差异

PDF的坐标系统是左下角为原点,y轴向上递增;而Qt的QGraphicsScene默认是左上角为原点,y轴向下递增。你直接把Qt item的x()、y()传给PyMuPDF的insert_text/insert_image,位置肯定完全不对!

转换公式很简单,只要拿到PDF页面的高度,就能把Qt坐标转成PyMuPDF坐标:

# 假设page是当前PyMuPDF页面对象
page_height = page.rect.height
# Qt坐标:(qt_x, qt_y) 是item的左上角位置
# PyMuPDF坐标:对应item的左下角位置(适配PDF原点)
fitz_x = qt_x
fitz_y = page_height - qt_y - item_bbox_height

二、文本插入的精准修复

你当前的文本保存逻辑有3个致命问题:

  1. 固定用12号Arial字体,和原PDF的字体、字号不匹配,导致文本宽度/行高变化,排版错位;
  2. 直接用insert_text是基于文本基线的,不是文本块的左上角,位置计算偏差;
  3. 直接修改原文档,原文本还在页面上,新插入的文本会和原内容重叠。

优化后的文本处理逻辑:

  1. 提取并保存原文本的字体信息:用get_text("dict")替代get_text("blocks"),可以拿到每个文本块的字体、字号,避免固定字体导致的错位;
  2. 用insert_textbox替代insert_text:insert_textbox可以指定文本块的完整bbox,自动排版文本,完美匹配原文本块的范围;
  3. 创建新PDF保存:避免原内容和新插入内容重叠,保证输出的是纯修改后的内容。

三、图像插入的精准修复

图像的问题除了坐标转换,还有你直接用item.pixmap().toImage().bits().asarray()生成的字节流格式不对,PyMuPDF无法正确识别;另外bbox的范围也需要转换。

优化后的图像处理逻辑:

  1. 正确转换图像bbox:把Qt的左上角bbox转成PyMuPDF的左下角bbox;
  2. 生成标准图像字节流:把QPixmap转成PNG/JPG格式的字节流,而不是直接取原始bits;
  3. 用bbox参数控制图像大小:insert_image传入转换后的bbox,PyMuPDF会自动将图像缩放到指定范围。

四、修改后的完整代码

下面是整合所有修复点的代码,关键位置都加了注释:

import fitz  # PyMuPDF
import sys
from PyQt6.QtWidgets import (
    QApplication, QGraphicsView, QGraphicsScene, QGraphicsPixmapItem,
    QGraphicsTextItem, QFileDialog, QGraphicsItem
)
from PyQt6.QtGui import QPixmap, QImage, QFont, QByteArray, QBuffer
from PyQt6.QtCore import Qt, QIODevice


class PDFEditor(QGraphicsView):
    def __init__(self, pdf_path):
        super().__init__()
        self.pdf_path = pdf_path
        self.doc = fitz.open(pdf_path)
        self.scene = QGraphicsScene()
        self.setScene(self.scene)
        self.page_details = []  # 保存每页的item引用,方便后续保存
        self.load_pdf()

    def load_pdf(self):
        for page_num in range(len(self.doc)):
            page = self.doc[page_num]
            page_rect = page.rect
            page_data = {"text_items": [], "image_items": []}
            self.render_page(page, page_rect, page_data)
            self.page_details.append(page_data)

    def render_page(self, page, page_rect, page_data):
        # 1. 处理文本:用get_text("dict")提取字体信息
        text_dict = page.get_text("dict")
        for block in text_dict["blocks"]:
            if block["type"] == 0:  # 文本块
                # 获取文本块的bbox (x0, y0, x1, y1)
                x0, y0, x1, y1 = block["bbox"]
                # 拼接文本块内容
                text_content = "".join([line["spans"][0]["text"] for line in block["lines"]])
                # 提取字体和字号(取第一个span的字体信息)
                font_info = block["lines"][0]["spans"][0]
                # 创建可编辑文本item,保存原信息
                text_item = EditableTextItem(text_content.strip(), x0, y0, page_rect.height, font_info)
                self.scene.addItem(text_item)
                page_data["text_items"].append(text_item)

        # 2. 处理图像
        for img_index, img_info in enumerate(page.get_images(full=True)):
            xref = img_info[0]
            base_img = self.doc.extract_image(xref)
            img_bytes = base_img["image"]
            img_format = base_img["ext"]
            img_qimage = QImage.fromData(img_bytes, img_format.upper())
            img_pixmap = QPixmap.fromImage(img_qimage)
            # 获取图像的原始bbox
            x, y, w, h = page.get_image_bbox(img_info)
            # 创建可编辑图像item,保存原信息
            img_item = EditableImageItem(img_pixmap, x, y, w, h, page_rect.height)
            self.scene.addItem(img_item)
            page_data["image_items"].append(img_item)

    def save_pdf(self, output_path):
        # 创建新的PDF文档,避免修改原文档导致重叠
        new_doc = fitz.open()
        for page_num, page_data in enumerate(self.page_details):
            # 复制原页面的尺寸和旋转信息
            original_page = self.doc[page_num]
            new_page = new_doc.new_page(width=original_page.rect.width, height=original_page.rect.height)
            page_height = original_page.rect.height

            # 1. 插入修改后的文本
            for text_item in page_data["text_items"]:
                qt_x = text_item.x()
                qt_y = text_item.y()
                text_bbox = text_item.boundingRect()
                text_width = text_bbox.width()
                text_height = text_bbox.height()

                # 转换坐标:Qt左上角 → PyMuPDF左下角
                fitz_x0 = qt_x
                fitz_y0 = page_height - qt_y - text_height
                fitz_x1 = qt_x + text_width
                fitz_y1 = page_height - qt_y

                # 用insert_textbox插入文本,自动匹配bbox
                new_page.insert_textbox(
                    rect=(fitz_x0, fitz_y0, fitz_x1, fitz_y1),
                    text=text_item.toPlainText(),
                    fontname=text_item.font_info["font"],
                    fontsize=text_item.font_info["size"],
                    align=Qt.AlignmentFlag.AlignLeft.value
                )

            # 2. 插入修改后的图像
            for img_item in page_data["image_items"]:
                qt_x = img_item.x()
                qt_y = img_item.y()
                img_bbox = img_item.boundingRect()
                img_width = img_bbox.width()
                img_height = img_bbox.height()

                # 转换坐标:Qt左上角 → PyMuPDF左下角
                fitz_x0 = qt_x
                fitz_y0 = page_height - qt_y - img_height
                fitz_x1 = qt_x + img_width
                fitz_y1 = page_height - qt_y

                # 生成标准PNG字节流
                pixmap = img_item.pixmap()
                img = pixmap.toImage()
                ba = QByteArray()
                buffer = QBuffer(ba)
                buffer.open(QIODevice.OpenModeFlag.WriteOnly)
                img.save(buffer, "PNG")
                img_bytes = ba.data()

                # 插入图像,用bbox控制大小和位置
                new_page.insert_image(
                    rect=(fitz_x0, fitz_y0, fitz_x1, fitz_y1),
                    stream=img_bytes
                )

        # 保存新PDF
        new_doc.save(output_path)
        new_doc.close()


class EditableTextItem(QGraphicsTextItem):
    def __init__(self, text, x, y, page_height, font_info):
        super().__init__(text)
        self.setTextInteractionFlags(Qt.TextInteractionFlag.TextEditorInteraction)
        self.setPos(x, y)
        # 用原字体和字号初始化
        self.setFont(QFont(font_info["font"], font_info["size"]))
        self.setFlag(QGraphicsItem.GraphicsItemFlag.ItemIsMovable)
        # 保存原信息,用于后续转换
        self.page_height = page_height
        self.font_info = font_info
        self.original_bbox = (x, y, x + self.boundingRect().width(), y + self.boundingRect().height())


class EditableImageItem(QGraphicsPixmapItem):
    def __init__(self, pixmap, x, y, w, h, page_height):
        super().__init__(pixmap)
        self.setPos(x, y)
        self.setFlag(QGraphicsItem.GraphicsItemFlag.ItemIsMovable)
        # 保存原信息,用于后续转换
        self.page_height = page_height
        self.original_bbox = (x, y, w, h)

    def mouseDoubleClickEvent(self, event):
        new_image_path, _ = QFileDialog.getOpenFileName(None, "选择图像", "", "Images (*.png *.jpg *.jpeg *.bmp)")
        if new_image_path:
            new_pixmap = QPixmap(new_image_path)
            # 缩放图像到原bbox大小
            scaled_pixmap = new_pixmap.scaled(
                self.original_bbox[2] - self.original_bbox[0],
                self.original_bbox[3] - self.original_bbox[1],
                Qt.AspectRatioMode.KeepAspectRatio,
                Qt.TransformationMode.SmoothTransformation
            )
            self.setPixmap(scaled_pixmap)


if __name__ == "__main__":
    app = QApplication(sys.argv)
    pdf_path, _ = QFileDialog.getOpenFileName(None, "选择PDF文件", "", "PDF Files (*.pdf)")
    if pdf_path:
        viewer = PDFEditor(pdf_path)
        viewer.show()
        sys.exit(app.exec())

五、测试注意事项

  1. 如果原PDF有特殊字体(比如中文、自定义字体),PyMuPDF可能需要手动指定字体文件路径,可以用fontname参数传入字体文件的路径,或者确保系统有该字体;
  2. 图像双击替换后,代码会自动缩放到原图像的大小,避免尺寸偏差;
  3. 保存时创建的是全新的PDF,完全基于你修改后的内容,不会有原内容重叠的问题。

备注:内容来源于stack exchange,提问作者Yousef Hashem

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.15 03:23:17