You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用AWS Lambda转换PDF为PNG时遭遇运行时终止问题求助

AWS Lambda + Python 处理大体积PDF转PNG的解决方案

问题描述

需处理800页的大型PDF文件并转换为PNG格式,但处理50页及以上PDF时触发如下错误:

{
  "errorType": "Runtime.ExitError",
  "errorMessage": "RequestId: ddb8e3d0-0ab5-4462-ab8f-ae7797fb93ad Error: Runtime exited with error: signal: killed"
}

当前使用的Lambda代码:

import json
import boto3
import zipfile
from io import BytesIO
from PIL import Image
import logging
import os
from datetime import datetime
import fitz
import urllib.parse

# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)

# Initialize AWS clients
lambda_client = boto3.client("lambda")
dynamodb = boto3.resource("dynamodb")
s3 = boto3.client("s3")

# Specify the DynamoDB table name
TABLE_NAME = "XYZ"


# Function to update document status in DynamoDB
def update_document_status(batch_id, status, stage):
    try:
        doc_status_table_name = dynamodb.Table(TABLE_NAME)
        current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")

        update_expression = (
            "SET #status = :status, #stage = :stage, #updated_date = :updated_date"
        )
        expression_attribute_names = {
            "#status": "status",
            "#stage": "stage",
            "#updated_date": "updated_date",
        }
        expression_attribute_values = {
            ":status": status,
            ":stage": stage,
            ":updated_date": current_time,
        }

        # If the status is either completed or failed, update end_time and update_time
        if status in ["completed", "failed"]:
            update_expression += ", #end_time = :end_time, #update_time = :update_time"
            expression_attribute_names["#end_time"] = "end_time"
            expression_attribute_names["#update_time"] = "update_time"
            expression_attribute_values[":end_time"] = current_time
            expression_attribute_values[":update_time"] = current_time

        response = doc_status_table_name.update_item(
            Key={"batch_id": batch_id},
            UpdateExpression=update_expression,
            ExpressionAttributeNames=expression_attribute_names,
            ExpressionAttributeValues=expression_attribute_values,
            ReturnValues="UPDATED_NEW",
        )
        logger.info("UpdateItem succeeded: %s", response)
        return True
    except Exception as e:
        logger.error("Error updating document status: %s", e)
        raise


# Function to convert multi-page TIFF file to PNG
def convert_multi_page_tiff_to_png(tiff_file):
    try:
        png_images = []
        with Image.open(tiff_file) as img:
            if img.n_frames > 1:
                for i in range(img.n_frames):
                    img.seek(i)
                    png_bytes = BytesIO()
                    img.save(png_bytes, format="PNG")
                    png_images.append(png_bytes.getvalue())
            else:
                # For single-page TIFF, directly convert it to PNG
                png_bytes = BytesIO()
                img.save(png_bytes, format="PNG")
                png_images.append(png_bytes.getvalue())
        return png_images
    except Exception as e:
        logger.error(f"Error converting multi-page TIFF to PNG: {e}")
        raise


# Function to convert multi-page PDF file to PNG
def convert_multi_page_pdf_to_png(pdf_data):
    try:
        png_images = []
        pdf_file = fitz.open(stream=pdf_data, filetype='pdf')
        for i in range(len(pdf_file)):
            pix = pdf_file[i].get_pixmap()

            img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples)
            png_bio = BytesIO()
            img.save(png_bio, format='PNG')

            png_images.append(png_bio.getvalue())            
    except Exception as e:
        print(f"Error converting multi-page PDF to PNG: {e}")
        raise
    
    return png_images

def process_zip_file(filepath):
    with zipfile.ZipFile(filepath, 'r') as zip_ref:
        for filename in zip_ref.namelist():
            print(f"processing file: {filename}")
            if filename.endswith('.tif') or filename.endswith('.tiff'):
                tiff_data = zip_ref.read(filename)
                png_images = convert_multi_page_tiff_to_png(tiff_data)
                
                # Only needed for local testing
                save_images(png_images, filename)
            elif filename.endswith('.pdf'):
                pdf_data = zip_ref.read(filename)
                png_images = convert_multi_page_pdf_to_png(pdf_data)
                
                # Only needed for local testing
                save_images(png_images, filename)
            else:
                print(f"Skipping processing of unsupported file format: {filename}")

def lambda_handler(event, context):
    try:
        batch_id = None
        logger.info("Received Event: %s", json.dumps(event, indent=2))

        folder_name = event.get("FILENAME")

        if folder_name is None:
            logger.error("Directory name not found in the event.")
            raise ValueError("Directory name not found in the event.")

        if not isinstance(folder_name, str):
            logger.error("Directory name is not a string.")
            raise ValueError("Directory name is not a string.")

        logger.info(f"Directory Name: {folder_name}")

        bucket_name = "cnc-aws-ecs-mecc-dev-doc-processing-v2"

        response = s3.list_objects_v2(Bucket=bucket_name, Prefix=folder_name)

        json_file_key = None
        matching_zip_file_key = None

        for obj in response.get("Contents", []):
            file_key = obj["Key"]
            logger.info(f"Processing file: {file_key}")

            if file_key.endswith(".json") and file_key.startswith(folder_name):
                json_file_key = file_key
            elif file_key.endswith(".zip") and file_key.startswith(folder_name):
                matching_zip_file_key = file_key

        logger.info(f"JSON file key: {json_file_key}")
        logger.info(f"Matching ZIP file key: {matching_zip_file_key}")

        if json_file_key is not None and matching_zip_file_key is not None:
            try:
                response = s3.get_object(Bucket=bucket_name, Key=json_file_key)
                data = json.loads(response['Body'].read().decode('utf-8'))
                batch_id = data['SCANMETADATA']['batchid']
                logger.info("Batch ID: %s", batch_id)

                zip_obj = s3.get_object(Bucket=bucket_name, Key=matching_zip_file_key)

                update_document_status(batch_id, "inprogress", "Image Conversion")

                pdf_data = zip_obj['Body'].read()  # Read PDF data from the ZIP file

                process_zip_file(pdf_data)  # Pass pdf_data to the function

                update_document_status(batch_id, "completed", "Image Conversion")

                s3.delete_object(Bucket=bucket_name, Key=json_file_key)
                logger.info(
                    "JSON file removed from original location: %s", json_file_key
                )

                s3.delete_object(Bucket=bucket_name, Key=matching_zip_file_key)
                logger.info("Original ZIP file deleted: %s", matching_zip_file_key)

                if json_file_key.startswith(folder_name):
                    s3.delete_object(Bucket=bucket_name, Key=folder_name + "/")
                    logger.info("Directory removed: %s", folder_name)

                return event, folder_name

            except Exception as e:
                logger.error(f"Error processing ZIP file {matching_zip_file_key}: {e}")
                update_document_status(batch_id, "error", "Image Conversion")
                raise e
        else:
            logger.error("Both JSON and matching ZIP files are not found.")
            raise Exception("Both JSON and matching ZIP files are not found.")

    except Exception as e:
        logger.error(f"Error: {e}")
        if batch_id:
            update_document_status(batch_id, "failed", "Image Conversion")
        raise e

问题根源分析

  1. 内存资源耗尽:Lambda默认内存配置较低(如128MB),将所有转换后的PNG数据存放在内存列表中,会快速耗尽内存,导致进程被强制终止(signal: killed)。
  2. 代码逻辑缺陷:
    • 直接将ZIP字节数据传给zipfile.ZipFile,未用BytesIO包装,会引发文件操作错误。
    • 转换后的PNG全部存在内存中,未及时持久化到S3释放资源。
  3. 超时限制:默认Lambda超时时间较短(3秒),处理大文件时会超出时间限制。

解决方案

1. 调整Lambda基础配置

  • 内存配置:将Lambda内存提升至1024MB~2048MB(内存越高,CPU性能越强,处理速度越快)。
  • 超时时间:设置为最大15分钟,确保有足够时间处理大文件。
  • 存储配置:启用Lambda临时存储(/tmp),最大可设置为10GB,用于临时缓存页面数据。

2. 优化代码逻辑

核心优化点:

  • 处理一页PDF就立即上传到S3,避免内存堆积。
  • 用fitz直接生成PNG字节流,跳过PIL中转,减少内存开销。
  • 修复ZIP文件处理逻辑,用BytesIO包装字节数据。

优化后的代码示例:

import json
import boto3
import zipfile
from io import BytesIO
import logging
import os
from datetime import datetime
import fitz

# 日志配置
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)

# AWS客户端初始化
dynamodb = boto3.resource("dynamodb")
s3 = boto3.client("s3")

TABLE_NAME = "XYZ"
TARGET_BUCKET = "your-png-output-bucket"  # 替换为存储PNG的S3桶名称


def update_document_status(batch_id, status, stage):
    try:
        doc_status_table = dynamodb.Table(TABLE_NAME)
        current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")

        update_expression = "SET #status = :status, #stage = :stage, #updated_date = :updated_date"
        expression_attribute_names = {
            "#status": "status",
            "#stage": "stage",
            "#updated_date": "updated_date",
        }
        expression_attribute_values = {
            ":status": status,
            ":stage": stage,
            ":updated_date": current_time,
        }

        if status in ["completed", "failed"]:
            update_expression += ", #end_time = :end_time, #update_time = :update_time"
            expression_attribute_names["#end_time"] = "end_time"
            expression_attribute_names["#update_time"] = "update_time"
            expression_attribute_values[":end_time"] = current_time
            expression_attribute_values[":update_time"] = current_time

        doc_status_table.update_item(
            Key={"batch_id": batch_id},
            UpdateExpression=update_expression,
            ExpressionAttributeNames=expression_attribute_names,
            ExpressionAttributeValues=expression_attribute_values,
            ReturnValues="UPDATED_NEW",
        )
        logger.info("文档状态更新成功")
        return True
    except Exception as e:
        logger.error(f"更新文档状态失败: {e}")
        raise


def convert_pdf_page_to_png_and_upload(pdf_stream, batch_id, filename):
    """处理PDF单页并实时上传到S3,避免内存堆积"""
    pdf_file = fitz.open(stream=pdf_stream, filetype='pdf')
    total_pages = len(pdf_file)
    logger.info(f"开始处理PDF: {filename}, 总页数: {total_pages}")

    for page_num in range(total_pages):
        try:
            page = pdf_file[page_num]
            # 直接生成PNG字节流,无需PIL中转
            pix = page.get_pixmap()
            png_bytes = pix.tobytes("png")

            # 上传到S3,命名格式:batch_id/文件名_页码.png
            s3_key = f"{batch_id}/{os.path.splitext(filename)[0]}_page_{page_num+1}.png"
            s3.put_object(
                Bucket=TARGET_BUCKET,
                Key=s3_key,
                Body=png_bytes,
                ContentType="image/png"
            )
            logger.info(f"第{page_num+1}页上传完成: {s3_key}")
        except Exception as e:
            logger.error(f"处理第{page_num+1}页失败: {e}")
            raise
    pdf_file.close()


def process_zip_file(zip_bytes, batch_id):
    """处理ZIP文件中的PDF/TIFF"""
    with zipfile.ZipFile(BytesIO(zip_bytes), 'r') as zip_ref:
        for filename in zip_ref.namelist():
            logger.info(f"处理文件: {filename}")
            if filename.endswith('.pdf'):
                pdf_data = zip_ref.read(filename)
                convert_pdf_page_to_png_and_upload(pdf_data, batch_id, filename)
            elif filename.endswith('.tif') or filename.endswith('.tiff'):
                # 保留原TIFF转PNG逻辑,建议同样优化为逐页上传
                tiff_data = zip_ref.read(filename)
                # 此处可复用原转换逻辑并修改为逐页上传
                pass
            else:
                logger.warning(f"跳过不支持的文件格式: {filename}")


def lambda_handler(event, context):
    batch_id = None
    try:
        logger.info(f"收到事件: {json.dumps(event, indent=2)}")
        folder_name = event.get("FILENAME")

        if not folder_name or not isinstance(folder_name, str):
            raise ValueError("事件中未找到有效的目录名称")

        bucket_name = "cnc-aws-ecs-mecc-dev-doc-processing-v2"
        response = s3.list_objects_v2(Bucket=bucket_name, Prefix=folder_name)

        json_file_key = None
        zip_file_key = None
        for obj in response.get("Contents", []):
            file_key = obj["Key"]
            if file_key.endswith(".json") and file_key.startswith(folder_name):
                json_file_key = file_key
            elif file_key.endswith(".zip") and file_key.startswith(folder_name):
                zip_file_key = file_key

        if not json_file_key or not zip_file_key:
            raise Exception("未找到匹配的JSON和ZIP文件")

        # 获取batch_id
        json_obj = s3.get_object(Bucket=bucket_name, Key=json_file_key)
        data = json.loads(json_obj['Body'].read().decode('utf-8'))
        batch_id = data['SCANMETADATA']['batchid']
        logger.info(f"Batch ID: {batch_id}")

        # 更新状态为处理中
        update_document_status(batch_id, "inprogress", "Image Conversion")

        # 获取ZIP文件并处理
        zip_obj = s3.get_object(Bucket=bucket_name, Key=zip_file_key)
        zip_bytes = zip_obj['Body'].read()
        process_zip_file(zip_bytes, batch_id)

        # 更新状态为完成
        update_document_status(batch_id, "completed", "Image Conversion")

        # 清理源文件
        s3.delete_object(Bucket=bucket_name, Key=json_file_key)
        s3.delete_object(Bucket=bucket_name, Key=zip_file_key)
        s3.delete_object(Bucket=bucket_name, Key=f"{folder_name}/")
        logger.info("源文件清理完成")

        return {"status": "success", "batch_id": batch_id}

    except Exception as e:
        logger.error(f"处理失败: {e}")
        if batch_id:
            update_document_status(batch_id, "failed", "Image Conversion")
        raise e

3. 超大型文件进阶处理方案

如果PDF超过200页,建议使用AWS Step Functions拆分任务:

  • 将PDF按每20页拆分为子任务。
  • 用Step Functions并行或串行调用Lambda处理每个子任务。
  • 最后汇总所有子任务结果,更新整体状态。

关键注意事项

  • 确保Lambda执行角色拥有S3读写、DynamoDB更新的权限。
  • 监控Lambda内存使用和执行时间,根据实际情况调整配置。
  • 加密PDF需先解密再处理。

内容的提问来源于stack exchange,提问作者Gomathi

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 20:07:33