Python批量Word转PDF性能优化:选多线程还是异步处理?
问题描述
我编写了一款Python脚本,可监控指定文件夹并捕获Word文件,将其转换为PDF格式,同时根据文件名创建对应文件夹并完成文件转移。但当待处理文件数量较多时,转换操作速度较慢。想请教针对该场景,应选择使用multithreading还是async file handling来优化性能?
原始代码
import os import time from watchdog.observers import Observer from watchdog.events import FileSystemEventHandler from win32com import client import pythoncom import shutil import asyncio from docx2pdf import convert """import aspose.words as aw """ baseAd = r"C:\inetpub\wwwroot\utkuploads" """This part is for catching errors. """ def createText(filename, filedetail): with open(r"C:\inetpub\wwwroot\utkuploads\{filename}.txt".format(filename=filename), 'w') as f: f.write(f'{filedetail}') """This doc2pdf works with WORD in backend. It opens word and converts the file to pdf. """ def doc2pdf(doc_name, pdf_name): pythoncom.CoInitialize() word = client.DispatchEx("Word.Application") if os.path.exists(pdf_name): os.remove(pdf_name) worddoc = word.Documents.Open(doc_name, ReadOnly=1) try: worddoc.SaveAs(pdf_name, FileFormat=17) except Exception as e: createText('saveasExceptionXX', f"{e}") worddoc.Close() # Quit the Word application word.Quit() pythoncom.CoUninitialize() return pdf_name """def doc2pdfX(doc_name, pdf_name): #second best convert(doc_name, pdf_name)""" """def doc2pdf2zz(doc_name, pdf_name): #best doc = aw.Document(doc_name) doc.save(pdf_name)""" class DocFileHandler(FileSystemEventHandler): def is_temporary_file(self, filename): return filename.startswith("~$") """To create folders we use the below, it takes baseAd which is defined at the beginning and folderName as parameters.""" def createFolder(self,baseAd,folderName): path = f"{baseAd}\{folderName}" isExist = os.path.exists(path) if not isExist: os.makedirs(path) return path else: return path """To create folders we use the below, it takes baseAd which is defined at the beginning and folderName as parameters.""" def createFolderAtt(self,folderName): isExist = os.path.exists(folderName) if not isExist: os.makedirs(folderName) return folderName else: return folderName def on_created(self, event): try: """The first [0] is root directory utkuploads the second is the file name with extension""" currentFileName = os.path.split(event.src_path) currentFileNameSplitted = os.path.split(event.src_path)[-1] if '.tmp' not in currentFileNameSplitted: pass print(f'File name {currentFileName}, splittted: {currentFileNameSplitted} has entered to server.') """If it is directory just pass don't do anything.""" if event.is_directory: return #"""This part needs to work for the files that needs to be converted to PDF""" #It catches DOCX files and takes their location by doc_path and creates a fake pdf_path directory elif event.event_type == 'created' and event.src_path.lower().endswith('.docx') and '@' not in currentFileNameSplitted and not self.is_temporary_file( event.src_path): doc_path = event.src_path pdf_path = os.path.splitext(doc_path)[0] + '.pdf' #print(f'Doc path: {doc_path}, \nPdf path: {pdf_path}') # If '_' in doc_path if '_' in doc_path: print(f'New Template has been detected: {doc_path}') return # If file is not temporary, not _ (template), not attachment, not TEMPLATE-REPORT elif '~$' not in doc_path and '_' not in doc_path and '@' not in doc_path and 'TEMPLATE-REPORT' not in doc_path: #print(f"File will be converted here: {doc_path}") try: if '-GENERATED-REPORT' in doc_path: # Here pdf convertion happens. doc2pdf(doc_path, pdf_path) # Create subFolder based on PDF file. createFolderPath = os.path.split(pdf_path)[-1].split(".")[0] createFolderPath = createFolderPath.replace('-GENERATED-REPORT', '') newFolderPath = self.createFolder(baseAd, createFolderPath) #print(f"New folder has been created: {newFolderPath}") pdfFileName = os.path.split(pdf_path)[-1] src_pdf = pdf_path dest_pathPdf = os.path.join(newFolderPath, pdfFileName) shutil.move(src_pdf, dest_pathPdf) #print(f"File has been moved to its destination. src: {src_pdf}, destination: {dest_pathPdf} ") #print('Doc path', doc_path) wordFileName = os.path.split(doc_path)[-1] wordPdf = wordFileName dest_pathWord = os.path.join(newFolderPath, wordPdf) shutil.move(doc_path, dest_pathWord) #print( f"Generated Rapor File has been moved to its destination. src: {doc_path}, destination: {dest_pathWord} ") elif '-IMZALIRAPOR' in doc_path: # Here pdf convertion happens. doc2pdf(doc_path, pdf_path) # Create subFolder based on PDF file. createFolderPath = os.path.split(pdf_path)[-1].split(".")[0] createFolderPath = createFolderPath.replace('-IMZALIRAPOR', '') newFolderPath = self.createFolder(baseAd, createFolderPath) #print(f"New folder has been created: {newFolderPath}") pdfFileName = os.path.split(pdf_path)[-1] src_pdf = pdf_path dest_pathPdf = os.path.join(newFolderPath, pdfFileName) shutil.move(src_pdf, dest_pathPdf) #print(f"File has been moved to its destination. src: {src_pdf}, destination: {dest_pathPdf} ") #print('Doc path', doc_path) wordFileName = os.path.split(doc_path)[-1] wordPdf = wordFileName dest_pathWord = os.path.join(newFolderPath, wordPdf) shutil.move(doc_path, dest_pathWord) #print(f"Imzali Report File has been moved to its destination. src: {doc_path}, destination: {dest_pathWord} ") elif 'GENERATED-REPORT' not in doc_path and '-IMZALIRAPOR' not in doc_path and '@' not in doc_path: #Here pdf convertion happens. doc2pdf(doc_path, pdf_path) #Create subFolder based on PDF file. createFolderPath = os.path.split(pdf_path)[-1].split(".")[0] newFolderPath = self.createFolder(baseAd, createFolderPath) #print(f"New folder has been created: {newFolderPath}") pdfFileName = os.path.split(pdf_path)[-1] src_pdf = pdf_path dest_pathPdf = os.path.join(newFolderPath, pdfFileName) shutil.move(src_pdf, dest_pathPdf) #print(f"File has been moved to its destination. src: {src_pdf}, destination: {dest_pathPdf} ") #print('Doc path', doc_path) wordFileName = os.path.split(doc_path)[-1] wordPdf = wordFileName dest_pathWord = os.path.join(newFolderPath, wordPdf) shutil.move(doc_path, dest_pathWord) #print(f"File has been moved to its destination. src: {doc_path}, destination: {dest_pathWord} ") except Exception as e: createText('exceptionHasOccured...', f'{e}') elif event.event_type == 'created' and '@' in currentFileNameSplitted and not self.is_temporary_file( event.src_path): doc_path = event.src_path folderPath = currentFileNameSplitted.split("@")[1].split(".")[0] try: baseFolderPath = os.path.split(doc_path)[:-1][0] #print(f"Attachments detected: {doc_path}, {currentFileNameSplitted}, {baseFolderPath}") dest_path = os.path.join(baseFolderPath, folderPath, currentFileNameSplitted) try: shutil.move(doc_path, dest_path) except: try: self.createFolderAtt(os.path.join(baseFolderPath, folderPath)) shutil.move(doc_path, dest_path) except Exception as e: createText('InnerAttachmentError', f'{e}') except Exception as e: createText('outerAttachmentErrorOccured', f'{e}') except Exception as e: createText('outerAllExceptionasOccured', f'{e}') if __name__ == '__main__': directory_to_watch = r"C:\inetpub\wwwroot\utkuploads" event_handler = DocFileHandler() observer = Observer() observer.schedule(event_handler, path=directory_to_watch, recursive=False) observer.start() try: while True: pass except KeyboardInterrupt: observer.stop() observer.join()
优化方案选择:优先使用多线程
为什么不选异步文件处理?
异步IO的核心优势是处理IO密集型任务(比如磁盘读写等待、网络请求),它通过在等待IO操作时切换任务来提升效率。但你的场景中,性能瓶颈是Word文档转换——这个过程是调用外部Word进程完成的,Python只是等待外部进程执行完毕,异步无法让多个Word转换任务并行,反而会增加代码复杂度,解决不了核心的速度问题。
为什么选多线程?
你的doc2pdf函数调用Win32COM启动独立的Word实例进行转换,这类调用会自动释放Python的GIL(全局解释器锁),意味着多线程可以真正并行执行多个转换任务,同时启动多个Word进程处理不同的文档,大幅提升批量文件的处理速度。
具体优化实现
可以用concurrent.futures.ThreadPoolExecutor来实现线程池,控制并发数避免系统资源过载。以下是关键修改点:
- 在DocFileHandler中初始化线程池
from concurrent.futures import ThreadPoolExecutor class DocFileHandler(FileSystemEventHandler): def __init__(self): super().__init__() # 控制最大并发数,根据系统配置调整,比如4-8 self.executor = ThreadPoolExecutor(max_workers=4)
- 把转换+移动逻辑封装为独立函数,提交到线程池
def process_doc_file(self, doc_path): pdf_path = os.path.splitext(doc_path)[0] + '.pdf' try: if '-GENERATED-REPORT' in doc_path: doc2pdf(doc_path, pdf_path) createFolderPath = os.path.split(pdf_path)[-1].split(".")[0].replace('-GENERATED-REPORT', '') newFolderPath = self.createFolder(baseAd, createFolderPath) # 移动PDF和Word文件 pdfFileName = os.path.split(pdf_path)[-1] shutil.move(pdf_path, os.path.join(newFolderPath, pdfFileName)) wordFileName = os.path.split(doc_path)[-1] shutil.move(doc_path, os.path.join(newFolderPath, wordFileName)) elif '-IMZALIRAPOR' in doc_path: doc2pdf(doc_path, pdf_path) createFolderPath = os.path.split(pdf_path)[-1].split(".")[0].replace('-IMZALIRAPOR', '') newFolderPath = self.createFolder(baseAd, createFolderPath) pdfFileName = os.path.split(pdf_path)[-1] shutil.move(pdf_path, os.path.join(newFolderPath, pdfFileName)) wordFileName = os.path.split(doc_path)[-1] shutil.move(doc_path, os.path.join(newFolderPath, wordFileName)) else: doc2pdf(doc_path, pdf_path) createFolderPath = os.path.split(pdf_path)[-1].split(".")[0] newFolderPath = self.createFolder(baseAd, createFolderPath) pdfFileName = os.path.split(pdf_path)[-1] shutil.move(pdf_path, os.path.join(newFolderPath, pdfFileName)) wordFileName = os.path.split(doc_path)[-1] shutil.move(doc_path, os.path.join(newFolderPath, wordFileName)) except Exception as e: createText('exceptionHasOccured...', f'{e}')
- 修改on_created事件,提交任务到线程池
在原来的处理逻辑中,把同步调用doc2pdf的部分替换为:
# 替换原来的try块内的同步逻辑 self.executor.submit(self.process_doc_file, doc_path)
注意事项
- 线程池的
max_workers不要设置过大,避免同时启动过多Word进程导致系统内存、CPU占用过高 - 保持原有的临时文件判断逻辑,避免处理未完全上传的文件
- 确保
pythoncom.CoInitialize()和CoUninitialize()在每个线程中正确调用(你的doc2pdf函数已经实现了这一点)
内容的提问来源于stack exchange,提问作者Fatih Enes
相关产品推荐
相关产品推荐

