将PDF合并脚本从Python2.7迁移至3.10时遇未定义函数错误
问题描述
我有一个在Python2.7中可正常运行的多PDF合并脚本,迁移至Python3.10时出现错误。已完成print语句修改、注释CoreFoundation和Quartz.CoreGraphics导入等适配,但仍报以下错误:
line 85, in main writeContext = CGPDFContextCreateWithURL(CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, arg, len(arg), False), None, None) NameError: name 'CGPDFContextCreateWithURL' is not defined
如果声明CGPDFContextCreateWithURL为空全局变量,错误会转移到CFURLCreateFromFileSystemRepresentation,再声明该变量则会出现kCFAllocatorDefault未定义的错误。我尝试的处理方式完全错误:
global CGPDFContextCreateWithURL CGPDFContextCreateWithURL = ""
无法理解该语句的逻辑,希望得到修正帮助,原脚本如下:
# # join # Joing pages from a a collection of PDF files into a single PDF file. # # join [--output <file>] [--shuffle] [--verbose]" # # Parameter: # # --shuffle # Take a page from each PDF input file in turn before taking another from each file. # If this option is not specified then all of the pages from a PDF file are appended # to the output PDF file before the next input PDF file is processed. # # --verbose # Write information about the doings of this tool to stderr. # import sys import os import getopt import tempfile import shutil # from CoreFoundation import * # from Quartz.CoreGraphics import * global verbose verbose = False def createPDFDocumentWithPath(path): if verbose: print("Creating PDF document from file %s" % (path)) return CGPDFDocumentCreateWithURL(CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, path, len(path), False)) def writePageFromDoc(writeContext, doc, pageNum): page = CGPDFDocumentGetPage(doc, pageNum) if page: mediaBox = CGPDFPageGetBoxRect(page, kCGPDFMediaBox) if CGRectIsEmpty(mediaBox): mediaBox = None CGContextBeginPage(writeContext, mediaBox) CGContextDrawPDFPage(writeContext, page) CGContextEndPage(writeContext) if verbose: print("Copied page %d from %s" % (pageNum, doc)) def shufflePages(writeContext, docs, maxPages): for pageNum in xrange(1, maxPages + 1): for doc in docs: writePageFromDoc(writeContext, doc, pageNum) def append(writeContext, docs, maxPages): for doc in docs: for pageNum in xrange(1, maxPages + 1) : writePageFromDoc(writeContext, doc, pageNum) def main(argv): global verbose # The PDF context we will draw into to create a new PDF writeContext = None # If True then generate more verbose information source = None shuffle = False # Parse the command line options try: options, args = getopt.getopt(argv, "o:sv", ["output=", "shuffle", "verbose"]) except getopt.GetoptError: usage() sys.exit(2) for option, arg in options: if option in ("-o", "--output") : if verbose: print("Setting %s as the destination." % (arg)) writeContext = CGPDFContextCreateWithURL(CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, arg, len(arg), False), None, None) elif option in ("-s", "--shuffle") : if verbose : print("Shuffle pages to the output file.") shuffle = True elif option in ("-v", "--verbose") : print("Verbose mode enabled.") verbose = True else : print("Unknown argument: %s" % (option)) if writeContext: # create PDFDocuments for all of the files. docs = map(createPDFDocumentWithPath, args) # find the maximum number of pages. maxPages = 0 for doc in docs: if CGPDFDocumentGetNumberOfPages(doc) > maxPages: maxPages = CGPDFDocumentGetNumberOfPages(doc) if shuffle: shufflePages(writeContext, docs, maxPages) else: append(writeContext, docs, maxPages) CGPDFContextClose(writeContext) del writeContext #CGContextRelease(writeContext) def usage(): print("Usage: join [--output <file>] [--shuffle] [--verbose]") if __name__ == "__main__": main(sys.argv[1:])
修正方案
1. 恢复并正确导入依赖库
你注释掉的CoreFoundation和Quartz.CoreGraphics是脚本核心依赖,这些是macOS系统框架,Python3下需要通过pyobjc库获取,先安装依赖:
pip install pyobjc-core pyobjc-framework-Quartz
然后取消注释导入语句:
from CoreFoundation import * from Quartz.CoreGraphics import *
2. 修复Python2到3的语法差异
- 替换
xrange为range:Python3中无xrange,直接用range即可,其行为与Python2的xrange一致。 - 处理
map返回迭代器的问题:Python3中map返回迭代器,遍历一次后耗尽,需转成列表:docs = list(map(createPDFDocumentWithPath, args)) - 优化全局变量声明:函数外无需声明
global verbose,直接赋值verbose = False即可,仅在函数内部使用该变量时声明global verbose。 - 修正错误的全局变量声明:你写的
global CGPDFContextCreateWithURL CGPDFContextCreateWithURL = ""是无效语法,且完全没必要——正确导入依赖后,这些函数会自动可用。
3. 完整修正后的脚本
# # join # Joins pages from a collection of PDF files into a single PDF file. # # join [--output <file>] [--shuffle] [--verbose] # # Parameters: # # --shuffle # Take a page from each PDF input file in turn before taking another from each file. # If this option is not specified then all of the pages from a PDF file are appended # to the output PDF file before the next input PDF file is processed. # # --verbose # Write information about the doings of this tool to stderr. # import sys import os import getopt from CoreFoundation import * from Quartz.CoreGraphics import * verbose = False def createPDFDocumentWithPath(path): if verbose: print(f"Creating PDF document from file {path}") # 处理路径编码,转为字节码传入 path_bytes = path.encode('utf-8') url = CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, path_bytes, len(path_bytes), False) return CGPDFDocumentCreateWithURL(url) def writePageFromDoc(writeContext, doc, pageNum): page = CGPDFDocumentGetPage(doc, pageNum) if page: mediaBox = CGPDFPageGetBoxRect(page, kCGPDFMediaBox) if CGRectIsEmpty(mediaBox): mediaBox = None CGContextBeginPage(writeContext, mediaBox) CGContextDrawPDFPage(writeContext, page) CGContextEndPage(writeContext) if verbose: print(f"Copied page {pageNum} from {doc}") def shufflePages(writeContext, docs, maxPages): for pageNum in range(1, maxPages + 1): for doc in docs: writePageFromDoc(writeContext, doc, pageNum) def append(writeContext, docs, maxPages): # 读取每个文档实际页数,避免无效读取 for doc in docs: page_count = CGPDFDocumentGetNumberOfPages(doc) for pageNum in range(1, page_count + 1): writePageFromDoc(writeContext, doc, pageNum) def main(argv): global verbose writeContext = None shuffle = False # Parse the command line options try: options, args = getopt.getopt(argv, "o:sv", ["output=", "shuffle", "verbose"]) except getopt.GetoptError: usage() sys.exit(2) for option, arg in options: if option in ("-o", "--output"): if verbose: print(f"Setting {arg} as the destination.") arg_bytes = arg.encode('utf-8') url = CFURLCreateFromFileSystemRepresentation(kCFAllocatorDefault, arg_bytes, len(arg_bytes), False) writeContext = CGPDFContextCreateWithURL(url, None, None) elif option in ("-s", "--shuffle"): if verbose: print("Shuffle pages to the output file.") shuffle = True elif option in ("-v", "--verbose"): print("Verbose mode enabled.") verbose = True else: print(f"Unknown argument: {option}") if writeContext: # 创建PDF文档对象并过滤无效文档 docs = [] for path in args: doc = createPDFDocumentWithPath(path) if doc: docs.append(doc) # 计算最大页数 maxPages = 0 for doc in docs: page_count = CGPDFDocumentGetNumberOfPages(doc) if page_count > maxPages: maxPages = page_count if shuffle: shufflePages(writeContext, docs, maxPages) else: append(writeContext, docs, maxPages) CGPDFContextClose(writeContext) del writeContext def usage(): print("Usage: join [--output <file>] [--shuffle] [--verbose]") if __name__ == "__main__": main(sys.argv[1:])
额外说明
- 改用Python3推荐的f-string格式化字符串,更简洁易读。
- 处理了路径编码问题,将字符串转为字节码后传入系统框架函数,避免中文路径等编码异常。
- 优化了
append函数逻辑,仅读取每个文档实际存在的页数,避免无效的页面读取操作。
内容的提问来源于stack exchange,提问作者JAC
相关产品推荐
相关产品推荐

