You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

将PDF合并脚本从Python2.7迁移至3.10时遇未定义函数错误

问题描述

我有一个在Python2.7中可正常运行的多PDF合并脚本,迁移至Python3.10时出现错误。已完成print语句修改、注释CoreFoundation和Quartz.CoreGraphics导入等适配,但仍报以下错误:

line 85, in main
writeContext = CGPDFContextCreateWithURL(CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault,
arg, len(arg), False), None, None) NameError: name
'CGPDFContextCreateWithURL' is not defined

如果声明CGPDFContextCreateWithURL为空全局变量,错误会转移到CFURLCreateFromFileSystemRepresentation,再声明该变量则会出现kCFAllocatorDefault未定义的错误。我尝试的处理方式完全错误:

global CGPDFContextCreateWithURL CGPDFContextCreateWithURL = ""

无法理解该语句的逻辑,希望得到修正帮助,原脚本如下:

#
# join
#   Joing pages from a a collection of PDF files into a single PDF file.
#
#   join [--output <file>] [--shuffle] [--verbose]"
#
#   Parameter:
#
#   --shuffle
#   Take a page from each PDF input file in turn before taking another from each file.
#   If this option is not specified then all of the pages from a PDF file are appended
#   to the output PDF file before the next input PDF file is processed.
#
#   --verbose
#   Write information about the doings of this tool to stderr.
#
import sys
import os
import getopt
import tempfile
import shutil
# from CoreFoundation import *
# from Quartz.CoreGraphics import *

global verbose
verbose = False


def createPDFDocumentWithPath(path):
    if verbose:
        print("Creating PDF document from file %s" % (path))
    return CGPDFDocumentCreateWithURL(CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, path, len(path), False))

def writePageFromDoc(writeContext, doc, pageNum):

    page = CGPDFDocumentGetPage(doc, pageNum)
    if page:
        mediaBox = CGPDFPageGetBoxRect(page, kCGPDFMediaBox)
        if CGRectIsEmpty(mediaBox):
            mediaBox = None
            
        CGContextBeginPage(writeContext, mediaBox)
        CGContextDrawPDFPage(writeContext, page)
        CGContextEndPage(writeContext)
        if verbose:
            print("Copied page %d from %s" % (pageNum, doc))

def shufflePages(writeContext, docs, maxPages):
    
    for pageNum in xrange(1, maxPages + 1):
        for doc in docs:
            writePageFromDoc(writeContext, doc, pageNum)
                
def append(writeContext, docs, maxPages):

    for doc in docs:
        for pageNum in xrange(1, maxPages + 1) :
            writePageFromDoc(writeContext, doc, pageNum)

def main(argv):

    global verbose

    # The PDF context we will draw into to create a new PDF
    writeContext = None

    # If True then generate more verbose information
    source = None
    shuffle = False
    
    # Parse the command line options
    try:
        options, args = getopt.getopt(argv, "o:sv", ["output=", "shuffle", "verbose"])

    except getopt.GetoptError:
        usage()
        sys.exit(2)

    for option, arg in options:

        if option in ("-o", "--output") :
            if verbose:
                print("Setting %s as the destination." % (arg))
            writeContext = CGPDFContextCreateWithURL(CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, arg, len(arg), False), None, None)

        elif option in ("-s", "--shuffle") :
            if verbose :
                print("Shuffle pages to the output file.")
            shuffle = True

        elif option in ("-v", "--verbose") :
            print("Verbose mode enabled.")
            verbose = True

        else :
            print("Unknown argument: %s" % (option))
    
    if writeContext:
        # create PDFDocuments for all of the files.
        docs = map(createPDFDocumentWithPath, args)
        
        # find the maximum number of pages.
        maxPages = 0
        for doc in docs:
            if CGPDFDocumentGetNumberOfPages(doc) > maxPages:
                maxPages = CGPDFDocumentGetNumberOfPages(doc)
    
        if shuffle:
            shufflePages(writeContext, docs, maxPages)
        else:
            append(writeContext, docs, maxPages)
        
        CGPDFContextClose(writeContext)
        del writeContext
        #CGContextRelease(writeContext)
    
def usage():
    print("Usage: join [--output <file>] [--shuffle] [--verbose]")

if __name__ == "__main__":
    main(sys.argv[1:])
修正方案

1. 恢复并正确导入依赖库

你注释掉的CoreFoundation和Quartz.CoreGraphics是脚本核心依赖,这些是macOS系统框架,Python3下需要通过pyobjc库获取,先安装依赖:

pip install pyobjc-core pyobjc-framework-Quartz

然后取消注释导入语句:

from CoreFoundation import *
from Quartz.CoreGraphics import *

2. 修复Python2到3的语法差异

  • 替换xrange为range:Python3中无xrange,直接用range即可,其行为与Python2的xrange一致。
  • 处理map返回迭代器的问题:Python3中map返回迭代器,遍历一次后耗尽,需转成列表:
    docs = list(map(createPDFDocumentWithPath, args))
    
  • 优化全局变量声明:函数外无需声明global verbose,直接赋值verbose = False即可,仅在函数内部使用该变量时声明global verbose。
  • 修正错误的全局变量声明:你写的global CGPDFContextCreateWithURL CGPDFContextCreateWithURL = ""是无效语法,且完全没必要——正确导入依赖后,这些函数会自动可用。

3. 完整修正后的脚本

#
# join
#   Joins pages from a collection of PDF files into a single PDF file.
#
#   join [--output <file>] [--shuffle] [--verbose]
#
#   Parameters:
#
#   --shuffle
#   Take a page from each PDF input file in turn before taking another from each file.
#   If this option is not specified then all of the pages from a PDF file are appended
#   to the output PDF file before the next input PDF file is processed.
#
#   --verbose
#   Write information about the doings of this tool to stderr.
#
import sys
import os
import getopt
from CoreFoundation import *
from Quartz.CoreGraphics import *

verbose = False


def createPDFDocumentWithPath(path):
    if verbose:
        print(f"Creating PDF document from file {path}")
    # 处理路径编码,转为字节码传入
    path_bytes = path.encode('utf-8')
    url = CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, path_bytes, len(path_bytes), False)
    return CGPDFDocumentCreateWithURL(url)

def writePageFromDoc(writeContext, doc, pageNum):
    page = CGPDFDocumentGetPage(doc, pageNum)
    if page:
        mediaBox = CGPDFPageGetBoxRect(page, kCGPDFMediaBox)
        if CGRectIsEmpty(mediaBox):
            mediaBox = None
            
        CGContextBeginPage(writeContext, mediaBox)
        CGContextDrawPDFPage(writeContext, page)
        CGContextEndPage(writeContext)
        if verbose:
            print(f"Copied page {pageNum} from {doc}")

def shufflePages(writeContext, docs, maxPages):
    for pageNum in range(1, maxPages + 1):
        for doc in docs:
            writePageFromDoc(writeContext, doc, pageNum)
                
def append(writeContext, docs, maxPages):
    # 读取每个文档实际页数,避免无效读取
    for doc in docs:
        page_count = CGPDFDocumentGetNumberOfPages(doc)
        for pageNum in range(1, page_count + 1):
            writePageFromDoc(writeContext, doc, pageNum)

def main(argv):
    global verbose
    writeContext = None
    shuffle = False
    
    # Parse the command line options
    try:
        options, args = getopt.getopt(argv, "o:sv", ["output=", "shuffle", "verbose"])
    except getopt.GetoptError:
        usage()
        sys.exit(2)

    for option, arg in options:
        if option in ("-o", "--output"):
            if verbose:
                print(f"Setting {arg} as the destination.")
            arg_bytes = arg.encode('utf-8')
            url = CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, arg_bytes, len(arg_bytes), False)
            writeContext = CGPDFContextCreateWithURL(url, None, None)
        elif option in ("-s", "--shuffle"):
            if verbose:
                print("Shuffle pages to the output file.")
            shuffle = True
        elif option in ("-v", "--verbose"):
            print("Verbose mode enabled.")
            verbose = True
        else:
            print(f"Unknown argument: {option}")
    
    if writeContext:
        # 创建PDF文档对象并过滤无效文档
        docs = []
        for path in args:
            doc = createPDFDocumentWithPath(path)
            if doc:
                docs.append(doc)
        
        # 计算最大页数
        maxPages = 0
        for doc in docs:
            page_count = CGPDFDocumentGetNumberOfPages(doc)
            if page_count > maxPages:
                maxPages = page_count
    
        if shuffle:
            shufflePages(writeContext, docs, maxPages)
        else:
            append(writeContext, docs, maxPages)
        
        CGPDFContextClose(writeContext)
        del writeContext
    
def usage():
    print("Usage: join [--output <file>] [--shuffle] [--verbose]")

if __name__ == "__main__":
    main(sys.argv[1:])

额外说明

  • 改用Python3推荐的f-string格式化字符串,更简洁易读。
  • 处理了路径编码问题,将字符串转为字节码后传入系统框架函数,避免中文路径等编码异常。
  • 优化了append函数逻辑,仅读取每个文档实际存在的页数,避免无效的页面读取操作。

内容的提问来源于stack exchange,提问作者JAC

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.15 07:25:17