Flask应用处理PDF后删除报错PermissionError及重复文件问题求助
问题描述
开发Flask应用实现以下流程:从Google Drive下载PDF文件,处理后重新上传回Drive,最后删除本地PDF以节省资源。目前遇到两个问题:
- 调用
delete_pdf_files函数时抛出PermissionError: [WinError 32] 另一个进程正在使用此文件,无法访问错误 - 本地生成重复PDF文件,疑似循环逻辑出错
相关代码如下:
@app.route('/run_script', methods=['GET', 'POST']) def run_script(): # Check if 'credentials' exists in session before trying to access it if 'credentials' not in session: flash('Credentials not found. Please authorize first.') return redirect(url_for('authorize')) # If 'credentials' exists in session, continue with the rest of the function credentials = google.oauth2.credentials.Credentials( **session['credentials']) drive_service = build('drive', 'v3', credentials=credentials) # Call the Drive v3 API to list all files folder_id = '1MX4lFFCUG45Qn_QIKaDamPRzFdsM05wJ' query = f"'{folder_id}' in parents and mimeType='application/pdf'" results = drive_service.files().list( q=query, pageSize=100, fields="nextPageToken, files(id, name)", includeItemsFromAllDrives=True, supportsAllDrives=True ).execute() items = results.get('files', []) # Download each file for item in items: request = drive_service.files().get_media(fileId=item['id']) with io.FileIO(item['name'], 'wb') as fh: downloader = MediaIoBaseDownload(fh, request) done = False while done is False: status, done = downloader.next_chunk() for filename in os.listdir(): if filename.endswith(".pdf"): pdf_path = Path(filename) pdf = PdfReader(str(pdf_path)) number_of_pages = len(pdf.pages) #print("Number of pages:", number_of_pages) # Convert PDF to a list of PIL Images using pdf2image images = pdf2image.convert_from_path(str(pdf_path)) # Extract text from each image using pytesseract text_list = [] for image in images: text = image_to_string(image, config="--psm 6") text_list.append(text) # Combine all text from the list into a single string pdf_text = "\n".join(text_list) #print("Text extracted from the PDF:") #print(pdf_text) # Regiser a new font pdfmetrics.registerFont(TTFont('Cambria', 'cambria.ttc')) pdfmetrics.registerFont(TTFont('Cambria-Bold', 'cambriab.ttf')) pdfmetrics.registerFont(TTFont('Arial-Bold', 'arialbd.ttf')) custom_color = Color(.7, .3, 0) def add_text_to_page_combined(input_pdf_path, output_pdf_path, text1, text2, text3, x, y1, y2, y3, font1_size, font1_color, font2_size, font2_color, page_number=0): packet = io.BytesIO() can = canvas.Canvas(packet, pagesize=letter) # Set different properties for text1 on the first page and other pages if page_number == 0: can.setFont('Cambria-Bold', 16) can.setFillColor(custom_color) else: can.setFont('Arial-Bold', font1_size) can.setFillColor(font1_color) text1_width = stringWidth(text1, 'Cambria-Bold', can._fontsize) can.drawString(x - text1_width, y1, text1) # Add underline to text1 if page_number == 0: underline_thickness = 1.5 underline_position = y1 - 5 can.setLineWidth(underline_thickness) can.setStrokeColor(can._fillColorObj) can.line(x - text1_width, underline_position, x, underline_position) else: underline_thickness = 0 # Draw text2 and text3 with custom font size and color can.setFont('Cambria', font2_size) can.setFillColor(font2_color) text2_width = stringWidth(text2, 'Cambria', can._fontsize) text3_width = stringWidth(text3, 'Cambria', can._fontsize) can.drawString(x - text2_width, y2, text2) can.drawString(x - text3_width, y3, text3) can.save() packet.seek(0) new_pdf_page = PdfReader(packet).pages[0] # Use 'with' to properly manage the file resource with open(input_pdf_path, "rb") as file: existing_pdf = PdfReader(file) output = PdfWriter() for i, page in enumerate(existing_pdf.pages): if i == page_number: page.merge_page(new_pdf_page) output.add_page(page) with open(output_pdf_path, "wb") as output_stream: output.write(output_stream) def find_text(pdf_path, text_to_find): page_number = 0 for page_layout in extract_pages(pdf_path): for element in page_layout: if isinstance(element, LTTextContainer): for text_line in element: if text_to_find in text_line.get_text(): return (element.x0, element.y0, page_number) page_number += 1 return None # Regular expression patterns invoice_number_pattern = r"Invoice #\s*(\d+)" date_pattern = r"Date:\s*(\d{2}/\d{2}/\d{4})" services_subtotal_pattern = r"Services Subtotal\s*\$([\d,]+(\.\d{2})?)" expenses_subtotal_pattern = r"Expenses Subtotal\s*\$([\d,]+(\.\d{2})?)" subtotal_pattern = r"Subtotal\s*\$([\d,]+(\.\d{2})?)" matter_number_pattern = r"\b(\d{5}-\w+)\b" outstanding_balance_pattern = r"Total Amount Outstanding\s*\(\s*\$\s*(\d+\.\d{2})" total_amount_outstanding_pattern = r"=\s*\$([\d,]+(\.\d{2})?)" initial_retainer_pattern = r"Initial Retainer\s+[^\$]+\$(\d[\d,]*\.\d{2})" clg_trust_balance_pattern = r"CLG Trust Account Balance\s*\$([\d,]+(\.\d{2})?)" # Find and extract the values from the text invoice_number = re.search(invoice_number_pattern, pdf_text).group(1) date_match = re.search(date_pattern, pdf_text) if date_match: date = date_match.group(1).replace("/", ".") expenses_subtotal_match = re.search(expenses_subtotal_pattern, pdf_text) matter_number = re.search(matter_number_pattern, pdf_text).group(1) outstanding_balance = re.search(outstanding_balance_pattern, pdf_text).group(1) total_amount_outstanding = re.search(total_amount_outstanding_pattern, pdf_text).group(1) clg_trust_balance = re.search(clg_trust_balance_pattern, pdf_text).group(1) # Find all initial retainer payment values initial_retainer_matches = re.findall(initial_retainer_pattern, pdf_text) initial_retainer = 0 if initial_retainer_matches: initial_retainer = sum(float(match.replace(",", "")) for match in initial_retainer_matches) if expenses_subtotal_match: expenses_subtotal = expenses_subtotal_match.group(1) services_subtotal = re.search(services_subtotal_pattern, pdf_text).group(1) else: expenses_subtotal = "0" services_subtotal = re.search(subtotal_pattern, pdf_text).group(1) # Display the extracted invoice number #print("Invoice Number:", invoice_number) #print("Date:", date) #print("Services Subtotal:", services_subtotal) #print("Expenses Subtotal:", expenses_subtotal) #print("Matter Number:", matter_number) #print("Outstanding Balance:", outstanding_balance) #print("Total Amount Outstanding:", total_amount_outstanding) #print("Initial Retainer:", initial_retainer) #print("CLG Trust Account Balance:", clg_trust_balance) numeric_services_subtotal = float(services_subtotal.replace(",", "")) #print("Numeric Services Subtotal:", numeric_services_subtotal) numeric_expenses_subtotal = float(expenses_subtotal.replace(",", "")) #print("Numeric Expenses Subtotal:", numeric_expenses_subtotal) numeric_outstanding_balance = float(outstanding_balance.replace(",", "")) #print("Numeric Outstanding Balance:", numeric_outstanding_balance) numeric_total_amount_outstanding = float(total_amount_outstanding.replace(",", "")) #print("Numeric Total Amount Outstanding:", numeric_total_amount_outstanding) numeric_clg_trust_balance = float(clg_trust_balance.replace(",", "")) #print("Numeric CLG Trust Account Balance:", clg_trust_balance) replenish_retainer = initial_retainer - numeric_clg_trust_balance #print ("Amount to replenish retainer:", replenish_retainer) current_amount_owing_on_invoice = numeric_total_amount_outstanding total_payment_due = current_amount_owing_on_invoice + replenish_retainer formatted_replenish_retainer = "${:,.2f}".format(replenish_retainer) formatted_current_amount_owing_on_invoice = "${:,.2f}".format(current_amount_owing_on_invoice) formatted_total_payment_due = "${:,.2f}".format(total_payment_due) formatted_initial_retainer = "${:,.2f}".format(initial_retainer) # Define the new file name new_filename = "New_" + filename add_text_to_page_combined(str(pdf_path), new_filename, "TOTAL PAYMENT DUE: " + str(formatted_total_payment_due), "Current Amount Owing on Invoice: " + str(formatted_current_amount_owing_on_invoice), "Amount to Replenish Retainer: " + str(formatted_replenish_retainer), 560, 520, 500, 480, 16, custom_color, 12, black, 0) text_location = find_text(new_filename, "CLG Trust Account Balance") if text_location is not None: output_filename = f"{matter_number}_{date}_Invoice.pdf" add_text_to_page_combined(new_filename, output_filename, "Amount to Replenish " + formatted_initial_retainer + " Retainer " + formatted_replenish_retainer, "", "", text_location[0] + 184, text_location[1] - 15, 0, 0, 9, black, # Font size and color for text1 12, black, # Font size and color for text2 and text3 text_location[2]) # Create a MediaFileUpload object and specify the MIME type of the file media = MediaFileUpload(output_filename, mimetype='application/pdf') # Specify the metadata, including the parent folder request = drive_service.files().create( media_body=media, body={ 'name': output_filename, # use the filename as the name of the file on Google Drive 'parents': ['140n9qP4Tr5buCPEx2VZFKs-h_OlHJTIY'] # use the ID of the target folder }, supportsAllDrives=True ) # Execute the request and upload the file. # Execute the request and upload the file. request.execute() session['credentials'] = credentials_to_dict(credentials) delete_pdf_files() return "Script finished running" def delete_pdf_files(): for file in os.listdir(): if file.endswith(".pdf"): os.remove(file) def credentials_to_dict(credentials): return {'token': credentials.token, 'refresh_token': credentials.refresh_token, 'token_uri': credentials.token_uri, 'client_id': credentials.client_id, 'client_secret': credentials.client_secret, 'scopes': credentials.scopes} if __name__ == '__main__': app.run(debug=True)
问题分析
1. 文件占用错误原因
- PdfReader未显式释放资源:直接初始化
PdfReader未使用上下文管理器,部分版本的库不会自动关闭文件句柄,导致文件被锁定。 - pdf2image临时文件残留:
convert_from_path会生成临时图片文件,若未指定临时目录或清理,可能间接占用原PDF文件。 - MediaFileUpload未释放文件:上传完成后,
MediaFileUpload可能未完全关闭文件句柄,导致文件处于占用状态。 - 嵌套函数内的文件操作:
add_text_to_page_combined中虽用了with,但外层的文件引用可能未完全释放。
2. 重复PDF文件原因
当前逻辑先下载所有PDF到本地,再用os.listdir()遍历所有PDF文件处理。但处理过程中生成的New_xxx.pdf和最终的output_filename也会被识别为PDF,导致这些新生成的文件被重复处理,进而生成更多重复文件。
解决方案
1. 修复循环逻辑,避免重复处理
下载文件时记录原始文件名,仅处理这些原始文件,忽略后续生成的新PDF:
# Download each file and track original filenames original_filenames = [] for item in items: request = drive_service.files().get_media(fileId=item['id']) filename = item['name'] original_filenames.append(filename) with io.FileIO(filename, 'wb') as fh: downloader = MediaIoBaseDownload(fh, request) done = False while done is False: status, done = downloader.next_chunk() # Only process original downloaded files, not generated ones for filename in original_filenames: pdf_path = Path(filename) # ... 后续处理逻辑保持不变
2. 确保文件资源完全释放
- PdfReader使用上下文管理器:将
PdfReader的初始化放在with语句中,确保文件句柄及时关闭:# 替代直接pdf = PdfReader(str(pdf_path)) with open(str(pdf_path), 'rb') as f: pdf = PdfReader(f) number_of_pages = len(pdf.pages) - 清理pdf2image临时文件:使用临时目录存放转换后的图片,处理完成后自动删除:
import tempfile with tempfile.TemporaryDirectory() as temp_dir: images = pdf2image.convert_from_path(str(pdf_path), output_folder=temp_dir) # 处理图片... # 临时目录会自动删除 - MediaFileUpload上传后手动关闭:上传完成后调用
close()释放文件句柄:media = MediaFileUpload(output_filename, mimetype='application/pdf') request = drive_service.files().create( media_body=media, body={'name': output_filename, 'parents': ['140n9qP4Tr5buCPEx2VZFKs-h_OlHJTIY']}, supportsAllDrives=True ) request.execute() media.close() # 释放文件句柄
3. 优化文件删除逻辑
处理完单个文件后立即删除原文件和中间文件,减少资源占用和冲突:
# 在单个文件处理完成后添加删除步骤 # 上传完成后删除原文件和中间文件 for file_to_delete in [filename, new_filename, output_filename]: if os.path.exists(file_to_delete): max_retries = 3 retries = 0 while retries < max_retries: try: os.remove(file_to_delete) break except PermissionError: retries += 1 time.sleep(1) # 等待1秒后重试
4. 调整delete_pdf_files函数(可选)
添加重试机制,避免因文件未及时释放导致的删除失败:
import time def delete_pdf_files(): for file in os.listdir(): if file.endswith(".pdf"): max_retries = 3 retries = 0 while retries < max_retries: try: os.remove(file) break except PermissionError: retries += 1 time.sleep(1) # 等待1秒后重试
修改后的完整核心代码片段
@app.route('/run_script', methods=['GET', 'POST'])
相关产品推荐
相关产品推荐

