You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyPDF2.PdfFileWriter报错:多字典定义与/Prev=0问题求助

问题

我用Python编写了合并两个PDF的代码,但执行到pdf_writer = PyPDF2.PdfFileWriter(strict=False)时出现报错。错误提示包含字典中/Info键重复定义,以及/Prev=0 in the trailer - assuming there is no previous xref table的提示。完整错误信息和代码如下:

错误信息

Multiple definitions in dictionary at byte 0x1fd19 for key /Info
Multiple definitions in dictionary at byte 0x1fd25 for key /Info
Multiple definitions in dictionary at byte 0x1fd31 for key /Info
Multiple definitions in dictionary at byte 0x207d6 for key /Info
Multiple definitions in dictionary at byte 0x207e2 for key /Info
Multiple definitions in dictionary at byte 0x207ee for key /Info
/Prev=0 in the trailer - assuming there is no previous xref table
/Prev=0 in the trailer - assuming there is no previous xref table
Multiple definitions in dictionary at byte 0x23f8a for key /Info
Multiple definitions in dictionary at byte 0x23f96 for key /Info
Multiple definitions in dictionary at byte 0x23fa2 for key /Info
Multiple definitions in dictionary at byte 0x1e634 for key /Info
Multiple definitions in dictionary at byte 0x1e640 for key /Info
Multiple definitions in dictionary at byte 0x1e64c for key /Info
/Prev=0 in the trailer - assuming there is no previous xref table
/Prev=0 in the trailer - assuming there is no previous xref table
/Prev=0 in the trailer - assuming there is no previous xref table
/Prev=0 in the trailer - assuming there is no previous xref table

代码

import os
import PyPDF2
import re
import math

def main():
    # Set the directory containing the PDF files
    inputDirectory = 'input_path'
    # Set the directory where the output PDF files will be saved
    outputDirectory = 'output_path'

    # Get a list of all of the PDF filenames in the input directory
    pdfFilenames = [f for f in os.listdir(inputDirectory) if f.endswith('.pdf')]

    pdf_dict = {}
    for file in pdfFilenames:
        if file.endswith('R.pdf'):
            pdf_dict[file] = file.replace('R.pdf', 'V.pdf')
        elif file.endswith('V.pdf'):
            pdf_dict[file.replace('V.pdf', 'R.pdf')] = file
        

    err = []
    l_fin = []
    fileAct = 0
    print("Début d'execution")
    for fileR, fileV in pdf_dict.items():
        # Ouvrez les fichiers PDF
        with open('CNI-R-V//' + fileR, 'rb') as file_handle_1, open('CNI-R-V//' + fileV, 'rb') as file_handle_2:
            # Créez des objets PDF à partir des fichiers
            try:
                pdf1 = PyPDF2.PdfFileReader(file_handle_1, strict=False)
                pdf2 = PyPDF2.PdfFileReader(file_handle_2, strict=False)
                # Créez un nouvel objet PDF
                pdf_writer = PyPDF2.PdfFileWriter(strict=False)
                # Ajoutez chaque page des fichiers PDF originaux au nouveau fichier
                for page_num in range(pdf1.getNumPages()):
                    pdf_writer.addPage(pdf1.getPage(page_num))
                for page_num in range(pdf2.getNumPages()):
                    pdf_writer.addPage(pdf2.getPage(page_num))
                # Créez le fichier PDF combiné en utilisant le nom des fichiers originaux
                with open(f'output/{fileR[:-5]}_combined.pdf', 'wb') as fh:
                    pdf_writer.write(fh)
                progressBar(len(pdf_dict), fileAct, err)
                fileAct += 1
            except:
                #print('erreur', fileR, fileV)
                err.append((fileR, fileV))
                progressBar(len(pdf_dict), fileAct, err)
                fileAct += 1
                continue       

def progressBar(nbVal, actVal, err):
  percent = actVal/nbVal
  nEquals = math.floor(percent*20)
  bar = '=' * nEquals + ' ' * ( 20 - nEquals )

  print("\r", end='')
  print(f"[{bar}] {percent*100:.2f}%", end='')

  if percent == 1:
    print("\nFin d'execution")
    if err != []:
      print("Erreurs:")
      for el in err:
        print(el)

if __name__ == '__main__':
    main()
问题原因与解决方案

原因分析

  • /Info键重复定义:你要合并的PDF文件本身存在格式问题,文件内部的字典中重复定义了/Info元数据键,PyPDF2在解析这类非标准PDF时会输出警告(即便设置了strict=False也无法完全屏蔽)。
  • /Prev=0 in the trailer:PDF的trailer段/Prev字段用于指向历史交叉引用表,正常应为有效字节偏移量,而你的PDF中该值为0,PyPDF2无法找到对应表,只能假设不存在,这也是PDF格式不规范导致的。
  • 额外路径问题:代码中定义了inputDirectory但实际打开文件时硬编码了CNI-R-V//路径,可能导致文件找不到或路径解析错误,进而触发异常。

这些警告本身不一定会导致代码崩溃,但如果后续合并失败,大概率是源PDF格式问题或路径错误引发的。

解决方案

  1. 修复源PDF格式:用Adobe Acrobat或其他专业PDF工具打开源PDF并重新保存,修复内部格式错误,消除/Info重复定义和无效/Prev字段的问题。
  2. 统一文件路径:代码中使用inputDirectory拼接文件路径,避免硬编码,减少路径错误。
  3. 改用更健壮的库:如果PyPDF2处理不规范PDF效果不佳,可尝试PyMuPDF(fitz),它对损坏或非标准PDF的兼容性更强。

修改后的代码(基于PyPDF2优化)

import os
import PyPDF2
import math

def main():
    # 设置输入输出目录
    inputDirectory = 'CNI-R-V'  # 统一使用该目录作为输入路径
    outputDirectory = 'output'
    
    # 确保输出目录存在
    os.makedirs(outputDirectory, exist_ok=True)

    # 获取目录下所有PDF文件
    pdfFilenames = [f for f in os.listdir(inputDirectory) if f.lower().endswith('.pdf')]

    pdf_dict = {}
    for file in pdfFilenames:
        if file.endswith('R.pdf'):
            v_file = file.replace('R.pdf', 'V.pdf')
            if v_file in pdfFilenames:  # 检查配对文件是否存在
                pdf_dict[file] = v_file
        elif file.endswith('V.pdf'):
            r_file = file.replace('V.pdf', 'R.pdf')
            if r_file not in pdf_dict:  # 避免重复添加
                pdf_dict[r_file] = file

    err = []
    fileAct = 0
    total_files = len(pdf_dict)
    print("Début d'execution")
    
    for fileR, fileV in pdf_dict.items():
        try:
            # 使用inputDirectory拼接路径
            file1_path = os.path.join(inputDirectory, fileR)
            file2_path = os.path.join(inputDirectory, fileV)
            
            with open(file1_path, 'rb') as file_handle_1, open(file2_path, 'rb') as file_handle_2:
                pdf1 = PyPDF2.PdfFileReader(file_handle_1, strict=False)
                pdf2 = PyPDF2.PdfFileReader(file_handle_2, strict=False)
                
                pdf_writer = PyPDF2.PdfFileWriter(strict=False)
                
                # 添加第一个PDF的所有页面
                for page in pdf1.pages:
                    pdf_writer.add_page(page)
                # 添加第二个PDF的所有页面
                for page in pdf2.pages:
                    pdf_writer.add_page(page)
                
                # 生成输出路径
                output_filename = f"{fileR[:-5]}_combined.pdf"
                output_path = os.path.join(outputDirectory, output_filename)
                
                with open(output_path, 'wb') as fh:
                    pdf_writer.write(fh)
                    
        except Exception as e:
            print(f"\nErreur avec {fileR} et {fileV}: {str(e)}")
            err.append((fileR, fileV))
        finally:
            fileAct += 1
            progressBar(total_files, fileAct, err)

def progressBar(nbVal, actVal, err):
    percent = actVal / nbVal if nbVal > 0 else 0
    nEquals = math.floor(percent * 20)
    bar = '=' * nEquals + ' ' * (20 - nEquals)

    print("\r", end='')
    print(f"[{bar}] {percent*100:.2f}%", end='')

    if percent == 1:
        print("\nFin d'execution")
        if err:
            print("Erreurs:")
            for el in err:
                print(el)

if __name__ == '__main__':
    main()

使用PyMuPDF的替代方案

import os
import fitz
import math

def main():
    inputDirectory = 'CNI-R-V'
    outputDirectory = 'output'
    os.makedirs(outputDirectory, exist_ok=True)

    pdfFilenames = [f for f in os.listdir(inputDirectory) if f.lower().endswith('.pdf')]

    pdf_dict = {}
    for file in pdfFilenames:
        if file.endswith('R.pdf'):
            v_file = file.replace('R.pdf', 'V.pdf')
            if v_file in pdfFilenames:
                pdf_dict[file] = v_file
        elif file.endswith('V.pdf'):
            r_file = file.replace('V.pdf', 'R.pdf')
            if r_file not in pdf_dict:
                pdf_dict[r_file] = file

    err = []
    fileAct = 0
    total_files = len(pdf_dict)
    print("Début d'execution")
    
    for fileR, fileV in pdf_dict.items():
        try:
            file1_path = os.path.join(inputDirectory, fileR)
            file2_path = os.path.join(inputDirectory, fileV)
            
            # 创建新的PDF文档
            combined_pdf = fitz.open()
            
            # 合并第一个PDF
            with fitz.open(file1_path) as pdf1:
                combined_pdf.insert_pdf(pdf1)
            # 合并第二个PDF
            with fitz.open(file2_path) as pdf2:
                combined_pdf.insert_pdf(pdf2)
            
            # 保存合并后的PDF
            output_filename = f"{fileR[:-5]}_combined.pdf"
            output_path = os.path.join(outputDirectory, output_filename)
            combined_pdf.save(output_path)
            combined_pdf.close()
            
        except Exception as e:
            print(f"\nErreur avec {fileR} et {fileV}: {str(e)}")
            err.append((fileR, fileV))
        finally:
            fileAct += 1
            progressBar(total_files, fileAct, err)

def progressBar(nbVal, actVal, err):
    percent = actVal / nbVal if nbVal > 0 else 0
    nEquals = math.floor(percent * 20)
    bar = '=' * nEquals + ' ' * (20 - nEquals)

    print("\r", end='')
    print(f"[{bar}] {percent*100:.2f}%", end='')

    if percent == 1:
        print("\nFin d'execution")
        if err:
            print("Erreurs:")
            for el in err:
                print(el)

if __name__ == '__main__':
    main()

内容的提问来源于stack exchange,提问作者VullWen

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.06 11:55:25