You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Flask应用处理PDF后删除报错PermissionError及重复文件问题求助

问题描述

开发Flask应用实现以下流程:从Google Drive下载PDF文件,处理后重新上传回Drive,最后删除本地PDF以节省资源。目前遇到两个问题:

  • 调用delete_pdf_files函数时抛出PermissionError: [WinError 32] 另一个进程正在使用此文件,无法访问错误
  • 本地生成重复PDF文件,疑似循环逻辑出错

相关代码如下:

@app.route('/run_script', methods=['GET', 'POST'])
def run_script():
    # Check if 'credentials' exists in session before trying to access it
    if 'credentials' not in session:
        flash('Credentials not found. Please authorize first.')
        return redirect(url_for('authorize'))

    # If 'credentials' exists in session, continue with the rest of the function
    credentials = google.oauth2.credentials.Credentials(
        **session['credentials'])

    drive_service = build('drive', 'v3', credentials=credentials)

    # Call the Drive v3 API to list all files
    folder_id = '1MX4lFFCUG45Qn_QIKaDamPRzFdsM05wJ'
    query = f"'{folder_id}' in parents and mimeType='application/pdf'"

    results = drive_service.files().list(
        q=query, 
        pageSize=100, 
        fields="nextPageToken, files(id, name)", 
        includeItemsFromAllDrives=True, 
        supportsAllDrives=True
    ).execute()

    items = results.get('files', [])


    # Download each file
    for item in items:
        request = drive_service.files().get_media(fileId=item['id'])
        with io.FileIO(item['name'], 'wb') as fh:
            downloader = MediaIoBaseDownload(fh, request)
            done = False
            while done is False:
                status, done = downloader.next_chunk()

    for filename in os.listdir():
        if filename.endswith(".pdf"):
            pdf_path = Path(filename)

            pdf = PdfReader(str(pdf_path))
            number_of_pages = len(pdf.pages)
            #print("Number of pages:", number_of_pages)

            # Convert PDF to a list of PIL Images using pdf2image
            images = pdf2image.convert_from_path(str(pdf_path))

            # Extract text from each image using pytesseract
            text_list = []
            for image in images:
                text = image_to_string(image, config="--psm 6")
                text_list.append(text)

            # Combine all text from the list into a single string
            pdf_text = "\n".join(text_list)
            #print("Text extracted from the PDF:")
            #print(pdf_text)


    # Regiser a new font

            pdfmetrics.registerFont(TTFont('Cambria', 'cambria.ttc'))
            pdfmetrics.registerFont(TTFont('Cambria-Bold', 'cambriab.ttf'))
            pdfmetrics.registerFont(TTFont('Arial-Bold', 'arialbd.ttf'))

            custom_color = Color(.7, .3, 0)

            def add_text_to_page_combined(input_pdf_path, output_pdf_path, text1, text2, text3, x, y1, y2, y3, font1_size, font1_color, font2_size, font2_color, page_number=0):
                packet = io.BytesIO()
                can = canvas.Canvas(packet, pagesize=letter)

                # Set different properties for text1 on the first page and other pages
                if page_number == 0:
                    can.setFont('Cambria-Bold', 16)
                    can.setFillColor(custom_color)
                else:
                    can.setFont('Arial-Bold', font1_size)
                    can.setFillColor(font1_color)

                text1_width = stringWidth(text1, 'Cambria-Bold', can._fontsize)
                can.drawString(x - text1_width, y1, text1)

                # Add underline to text1
                if page_number == 0:
                    underline_thickness = 1.5
                    underline_position = y1 - 5
                    can.setLineWidth(underline_thickness)
                    can.setStrokeColor(can._fillColorObj)
                    can.line(x - text1_width, underline_position, x, underline_position)
                else:
                    underline_thickness = 0

                # Draw text2 and text3 with custom font size and color
                can.setFont('Cambria', font2_size)
                can.setFillColor(font2_color)
                text2_width = stringWidth(text2, 'Cambria', can._fontsize)
                text3_width = stringWidth(text3, 'Cambria', can._fontsize)
                can.drawString(x - text2_width, y2, text2)
                can.drawString(x - text3_width, y3, text3)

                can.save()

                packet.seek(0)
                new_pdf_page = PdfReader(packet).pages[0]

                # Use 'with' to properly manage the file resource
                with open(input_pdf_path, "rb") as file:
                    existing_pdf = PdfReader(file)
                    output = PdfWriter()

                    for i, page in enumerate(existing_pdf.pages):
                        if i == page_number:
                            page.merge_page(new_pdf_page)
                        output.add_page(page)

                    with open(output_pdf_path, "wb") as output_stream:
                        output.write(output_stream)


            def find_text(pdf_path, text_to_find):
                page_number = 0
                for page_layout in extract_pages(pdf_path):
                    for element in page_layout:
                        if isinstance(element, LTTextContainer):
                            for text_line in element:
                                if text_to_find in text_line.get_text():
                                    return (element.x0, element.y0, page_number)
                    page_number += 1
                return None



            # Regular expression patterns
            invoice_number_pattern = r"Invoice #\s*(\d+)"
            date_pattern = r"Date:\s*(\d{2}/\d{2}/\d{4})"
            services_subtotal_pattern = r"Services Subtotal\s*\$([\d,]+(\.\d{2})?)"
            expenses_subtotal_pattern = r"Expenses Subtotal\s*\$([\d,]+(\.\d{2})?)"
            subtotal_pattern = r"Subtotal\s*\$([\d,]+(\.\d{2})?)"
            matter_number_pattern = r"\b(\d{5}-\w+)\b"
            outstanding_balance_pattern = r"Total Amount Outstanding\s*\(\s*\$\s*(\d+\.\d{2})"
            total_amount_outstanding_pattern = r"=\s*\$([\d,]+(\.\d{2})?)"
            initial_retainer_pattern = r"Initial Retainer\s+[^\$]+\$(\d[\d,]*\.\d{2})"
            clg_trust_balance_pattern = r"CLG Trust Account Balance\s*\$([\d,]+(\.\d{2})?)"

            # Find and extract the values from the text
            invoice_number = re.search(invoice_number_pattern, pdf_text).group(1)
            date_match = re.search(date_pattern, pdf_text)
            if date_match:
                date = date_match.group(1).replace("/", ".")
            expenses_subtotal_match = re.search(expenses_subtotal_pattern, pdf_text)
            matter_number = re.search(matter_number_pattern, pdf_text).group(1)
            outstanding_balance = re.search(outstanding_balance_pattern, pdf_text).group(1)
            total_amount_outstanding = re.search(total_amount_outstanding_pattern, pdf_text).group(1)
            clg_trust_balance = re.search(clg_trust_balance_pattern, pdf_text).group(1)
            # Find all initial retainer payment values
            initial_retainer_matches = re.findall(initial_retainer_pattern, pdf_text)
            initial_retainer = 0
            if initial_retainer_matches:
                initial_retainer = sum(float(match.replace(",", "")) for match in initial_retainer_matches)


            if expenses_subtotal_match:
                expenses_subtotal = expenses_subtotal_match.group(1)
                services_subtotal = re.search(services_subtotal_pattern, pdf_text).group(1)
            else:
                expenses_subtotal = "0"
                services_subtotal = re.search(subtotal_pattern, pdf_text).group(1)



            # Display the extracted invoice number
            #print("Invoice Number:", invoice_number)
            #print("Date:", date)
            #print("Services Subtotal:", services_subtotal)
            #print("Expenses Subtotal:", expenses_subtotal)
            #print("Matter Number:", matter_number)
            #print("Outstanding Balance:", outstanding_balance)
            #print("Total Amount Outstanding:", total_amount_outstanding)
            #print("Initial Retainer:", initial_retainer)
            #print("CLG Trust Account Balance:", clg_trust_balance)


            numeric_services_subtotal = float(services_subtotal.replace(",", ""))
            #print("Numeric Services Subtotal:", numeric_services_subtotal)

            numeric_expenses_subtotal = float(expenses_subtotal.replace(",", ""))
            #print("Numeric Expenses Subtotal:", numeric_expenses_subtotal)

            numeric_outstanding_balance = float(outstanding_balance.replace(",", ""))
            #print("Numeric Outstanding Balance:", numeric_outstanding_balance)

            numeric_total_amount_outstanding = float(total_amount_outstanding.replace(",", ""))
            #print("Numeric Total Amount Outstanding:", numeric_total_amount_outstanding)

            numeric_clg_trust_balance = float(clg_trust_balance.replace(",", ""))
            #print("Numeric CLG Trust Account Balance:", clg_trust_balance)

            replenish_retainer = initial_retainer - numeric_clg_trust_balance

            #print ("Amount to replenish retainer:", replenish_retainer)

            current_amount_owing_on_invoice = numeric_total_amount_outstanding
            total_payment_due = current_amount_owing_on_invoice + replenish_retainer

            formatted_replenish_retainer = "${:,.2f}".format(replenish_retainer)
            formatted_current_amount_owing_on_invoice = "${:,.2f}".format(current_amount_owing_on_invoice)
            formatted_total_payment_due = "${:,.2f}".format(total_payment_due)
            formatted_initial_retainer = "${:,.2f}".format(initial_retainer)




            # Define the new file name
            new_filename = "New_" + filename
            add_text_to_page_combined(str(pdf_path), new_filename, "TOTAL PAYMENT DUE: " +  str(formatted_total_payment_due), "Current Amount Owing on Invoice: " +  str(formatted_current_amount_owing_on_invoice), "Amount to Replenish Retainer: " +  str(formatted_replenish_retainer), 560, 520, 500, 480, 16, custom_color, 12, black, 0)

            text_location = find_text(new_filename, "CLG Trust Account Balance")
            if text_location is not None:
                output_filename = f"{matter_number}_{date}_Invoice.pdf"
                add_text_to_page_combined(new_filename, 
                                        output_filename, 
                                        "Amount to Replenish " + formatted_initial_retainer + " Retainer      " + formatted_replenish_retainer, "", "", 
                                        text_location[0] + 184, text_location[1] - 15, 0, 0, 
                                        9, black, # Font size and color for text1
                                        12, black, # Font size and color for text2 and text3
                                        text_location[2]) 

            # Create a MediaFileUpload object and specify the MIME type of the file
            media = MediaFileUpload(output_filename, mimetype='application/pdf')

            # Specify the metadata, including the parent folder
            request = drive_service.files().create(
                media_body=media,
                body={
                    'name': output_filename,  # use the filename as the name of the file on Google Drive
                    'parents': ['140n9qP4Tr5buCPEx2VZFKs-h_OlHJTIY']  # use the ID of the target folder
                },
                supportsAllDrives=True
            )

            # Execute the request and upload the file.
            # Execute the request and upload the file.
            request.execute()



        session['credentials'] = credentials_to_dict(credentials)

    delete_pdf_files()
    return "Script finished running"



def delete_pdf_files():
    for file in os.listdir():
        if file.endswith(".pdf"):
            os.remove(file)



def credentials_to_dict(credentials):
    return {'token': credentials.token,
            'refresh_token': credentials.refresh_token,
            'token_uri': credentials.token_uri,
            'client_id': credentials.client_id,
            'client_secret': credentials.client_secret,
            'scopes': credentials.scopes}



if __name__ == '__main__':
    app.run(debug=True)
问题分析

1. 文件占用错误原因

  • PdfReader未显式释放资源:直接初始化PdfReader未使用上下文管理器,部分版本的库不会自动关闭文件句柄,导致文件被锁定。
  • pdf2image临时文件残留:convert_from_path会生成临时图片文件,若未指定临时目录或清理,可能间接占用原PDF文件。
  • MediaFileUpload未释放文件:上传完成后,MediaFileUpload可能未完全关闭文件句柄,导致文件处于占用状态。
  • 嵌套函数内的文件操作:add_text_to_page_combined中虽用了with,但外层的文件引用可能未完全释放。

2. 重复PDF文件原因

当前逻辑先下载所有PDF到本地,再用os.listdir()遍历所有PDF文件处理。但处理过程中生成的New_xxx.pdf和最终的output_filename也会被识别为PDF,导致这些新生成的文件被重复处理,进而生成更多重复文件。

解决方案

1. 修复循环逻辑,避免重复处理

下载文件时记录原始文件名,仅处理这些原始文件,忽略后续生成的新PDF:

# Download each file and track original filenames
original_filenames = []
for item in items:
    request = drive_service.files().get_media(fileId=item['id'])
    filename = item['name']
    original_filenames.append(filename)
    with io.FileIO(filename, 'wb') as fh:
        downloader = MediaIoBaseDownload(fh, request)
        done = False
        while done is False:
            status, done = downloader.next_chunk()

# Only process original downloaded files, not generated ones
for filename in original_filenames:
    pdf_path = Path(filename)
    # ... 后续处理逻辑保持不变

2. 确保文件资源完全释放

  • PdfReader使用上下文管理器:将PdfReader的初始化放在with语句中,确保文件句柄及时关闭:
    # 替代直接pdf = PdfReader(str(pdf_path))
    with open(str(pdf_path), 'rb') as f:
        pdf = PdfReader(f)
        number_of_pages = len(pdf.pages)
    
  • 清理pdf2image临时文件:使用临时目录存放转换后的图片,处理完成后自动删除:
    import tempfile
    with tempfile.TemporaryDirectory() as temp_dir:
        images = pdf2image.convert_from_path(str(pdf_path), output_folder=temp_dir)
        # 处理图片...
    # 临时目录会自动删除
    
  • MediaFileUpload上传后手动关闭:上传完成后调用close()释放文件句柄:
    media = MediaFileUpload(output_filename, mimetype='application/pdf')
    request = drive_service.files().create(
        media_body=media,
        body={'name': output_filename, 'parents': ['140n9qP4Tr5buCPEx2VZFKs-h_OlHJTIY']},
        supportsAllDrives=True
    )
    request.execute()
    media.close()  # 释放文件句柄
    

3. 优化文件删除逻辑

处理完单个文件后立即删除原文件和中间文件,减少资源占用和冲突:

# 在单个文件处理完成后添加删除步骤
# 上传完成后删除原文件和中间文件
for file_to_delete in [filename, new_filename, output_filename]:
    if os.path.exists(file_to_delete):
        max_retries = 3
        retries = 0
        while retries < max_retries:
            try:
                os.remove(file_to_delete)
                break
            except PermissionError:
                retries += 1
                time.sleep(1)  # 等待1秒后重试

4. 调整delete_pdf_files函数(可选)

添加重试机制,避免因文件未及时释放导致的删除失败:

import time

def delete_pdf_files():
    for file in os.listdir():
        if file.endswith(".pdf"):
            max_retries = 3
            retries = 0
            while retries < max_retries:
                try:
                    os.remove(file)
                    break
                except PermissionError:
                    retries += 1
                    time.sleep(1)  # 等待1秒后重试
修改后的完整核心代码片段
@app.route('/run_script', methods=['GET', 'POST'])
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.23 05:42:26