使用AWS Lambda转换PDF为PNG时遭遇运行时终止问题求助
AWS Lambda + Python 处理大体积PDF转PNG的解决方案
问题描述
需处理800页的大型PDF文件并转换为PNG格式,但处理50页及以上PDF时触发如下错误:
{ "errorType": "Runtime.ExitError", "errorMessage": "RequestId: ddb8e3d0-0ab5-4462-ab8f-ae7797fb93ad Error: Runtime exited with error: signal: killed" }
当前使用的Lambda代码:
import json import boto3 import zipfile from io import BytesIO from PIL import Image import logging import os from datetime import datetime import fitz import urllib.parse # Configure logging logging.basicConfig(level=logging.INFO) logger = logging.getLogger(__name__) # Initialize AWS clients lambda_client = boto3.client("lambda") dynamodb = boto3.resource("dynamodb") s3 = boto3.client("s3") # Specify the DynamoDB table name TABLE_NAME = "XYZ" # Function to update document status in DynamoDB def update_document_status(batch_id, status, stage): try: doc_status_table_name = dynamodb.Table(TABLE_NAME) current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S") update_expression = ( "SET #status = :status, #stage = :stage, #updated_date = :updated_date" ) expression_attribute_names = { "#status": "status", "#stage": "stage", "#updated_date": "updated_date", } expression_attribute_values = { ":status": status, ":stage": stage, ":updated_date": current_time, } # If the status is either completed or failed, update end_time and update_time if status in ["completed", "failed"]: update_expression += ", #end_time = :end_time, #update_time = :update_time" expression_attribute_names["#end_time"] = "end_time" expression_attribute_names["#update_time"] = "update_time" expression_attribute_values[":end_time"] = current_time expression_attribute_values[":update_time"] = current_time response = doc_status_table_name.update_item( Key={"batch_id": batch_id}, UpdateExpression=update_expression, ExpressionAttributeNames=expression_attribute_names, ExpressionAttributeValues=expression_attribute_values, ReturnValues="UPDATED_NEW", ) logger.info("UpdateItem succeeded: %s", response) return True except Exception as e: logger.error("Error updating document status: %s", e) raise # Function to convert multi-page TIFF file to PNG def convert_multi_page_tiff_to_png(tiff_file): try: png_images = [] with Image.open(tiff_file) as img: if img.n_frames > 1: for i in range(img.n_frames): img.seek(i) png_bytes = BytesIO() img.save(png_bytes, format="PNG") png_images.append(png_bytes.getvalue()) else: # For single-page TIFF, directly convert it to PNG png_bytes = BytesIO() img.save(png_bytes, format="PNG") png_images.append(png_bytes.getvalue()) return png_images except Exception as e: logger.error(f"Error converting multi-page TIFF to PNG: {e}") raise # Function to convert multi-page PDF file to PNG def convert_multi_page_pdf_to_png(pdf_data): try: png_images = [] pdf_file = fitz.open(stream=pdf_data, filetype='pdf') for i in range(len(pdf_file)): pix = pdf_file[i].get_pixmap() img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples) png_bio = BytesIO() img.save(png_bio, format='PNG') png_images.append(png_bio.getvalue()) except Exception as e: print(f"Error converting multi-page PDF to PNG: {e}") raise return png_images def process_zip_file(filepath): with zipfile.ZipFile(filepath, 'r') as zip_ref: for filename in zip_ref.namelist(): print(f"processing file: {filename}") if filename.endswith('.tif') or filename.endswith('.tiff'): tiff_data = zip_ref.read(filename) png_images = convert_multi_page_tiff_to_png(tiff_data) # Only needed for local testing save_images(png_images, filename) elif filename.endswith('.pdf'): pdf_data = zip_ref.read(filename) png_images = convert_multi_page_pdf_to_png(pdf_data) # Only needed for local testing save_images(png_images, filename) else: print(f"Skipping processing of unsupported file format: {filename}") def lambda_handler(event, context): try: batch_id = None logger.info("Received Event: %s", json.dumps(event, indent=2)) folder_name = event.get("FILENAME") if folder_name is None: logger.error("Directory name not found in the event.") raise ValueError("Directory name not found in the event.") if not isinstance(folder_name, str): logger.error("Directory name is not a string.") raise ValueError("Directory name is not a string.") logger.info(f"Directory Name: {folder_name}") bucket_name = "cnc-aws-ecs-mecc-dev-doc-processing-v2" response = s3.list_objects_v2(Bucket=bucket_name, Prefix=folder_name) json_file_key = None matching_zip_file_key = None for obj in response.get("Contents", []): file_key = obj["Key"] logger.info(f"Processing file: {file_key}") if file_key.endswith(".json") and file_key.startswith(folder_name): json_file_key = file_key elif file_key.endswith(".zip") and file_key.startswith(folder_name): matching_zip_file_key = file_key logger.info(f"JSON file key: {json_file_key}") logger.info(f"Matching ZIP file key: {matching_zip_file_key}") if json_file_key is not None and matching_zip_file_key is not None: try: response = s3.get_object(Bucket=bucket_name, Key=json_file_key) data = json.loads(response['Body'].read().decode('utf-8')) batch_id = data['SCANMETADATA']['batchid'] logger.info("Batch ID: %s", batch_id) zip_obj = s3.get_object(Bucket=bucket_name, Key=matching_zip_file_key) update_document_status(batch_id, "inprogress", "Image Conversion") pdf_data = zip_obj['Body'].read() # Read PDF data from the ZIP file process_zip_file(pdf_data) # Pass pdf_data to the function update_document_status(batch_id, "completed", "Image Conversion") s3.delete_object(Bucket=bucket_name, Key=json_file_key) logger.info( "JSON file removed from original location: %s", json_file_key ) s3.delete_object(Bucket=bucket_name, Key=matching_zip_file_key) logger.info("Original ZIP file deleted: %s", matching_zip_file_key) if json_file_key.startswith(folder_name): s3.delete_object(Bucket=bucket_name, Key=folder_name + "/") logger.info("Directory removed: %s", folder_name) return event, folder_name except Exception as e: logger.error(f"Error processing ZIP file {matching_zip_file_key}: {e}") update_document_status(batch_id, "error", "Image Conversion") raise e else: logger.error("Both JSON and matching ZIP files are not found.") raise Exception("Both JSON and matching ZIP files are not found.") except Exception as e: logger.error(f"Error: {e}") if batch_id: update_document_status(batch_id, "failed", "Image Conversion") raise e
问题根源分析
- 内存资源耗尽:Lambda默认内存配置较低(如128MB),将所有转换后的PNG数据存放在内存列表中,会快速耗尽内存,导致进程被强制终止(
signal: killed)。 - 代码逻辑缺陷:
- 直接将ZIP字节数据传给
zipfile.ZipFile,未用BytesIO包装,会引发文件操作错误。 - 转换后的PNG全部存在内存中,未及时持久化到S3释放资源。
- 直接将ZIP字节数据传给
- 超时限制:默认Lambda超时时间较短(3秒),处理大文件时会超出时间限制。
解决方案
1. 调整Lambda基础配置
- 内存配置:将Lambda内存提升至1024MB~2048MB(内存越高,CPU性能越强,处理速度越快)。
- 超时时间:设置为最大15分钟,确保有足够时间处理大文件。
- 存储配置:启用Lambda临时存储(/tmp),最大可设置为10GB,用于临时缓存页面数据。
2. 优化代码逻辑
核心优化点:
- 处理一页PDF就立即上传到S3,避免内存堆积。
- 用
fitz直接生成PNG字节流,跳过PIL中转,减少内存开销。 - 修复ZIP文件处理逻辑,用
BytesIO包装字节数据。
优化后的代码示例:
import json import boto3 import zipfile from io import BytesIO import logging import os from datetime import datetime import fitz # 日志配置 logging.basicConfig(level=logging.INFO) logger = logging.getLogger(__name__) # AWS客户端初始化 dynamodb = boto3.resource("dynamodb") s3 = boto3.client("s3") TABLE_NAME = "XYZ" TARGET_BUCKET = "your-png-output-bucket" # 替换为存储PNG的S3桶名称 def update_document_status(batch_id, status, stage): try: doc_status_table = dynamodb.Table(TABLE_NAME) current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S") update_expression = "SET #status = :status, #stage = :stage, #updated_date = :updated_date" expression_attribute_names = { "#status": "status", "#stage": "stage", "#updated_date": "updated_date", } expression_attribute_values = { ":status": status, ":stage": stage, ":updated_date": current_time, } if status in ["completed", "failed"]: update_expression += ", #end_time = :end_time, #update_time = :update_time" expression_attribute_names["#end_time"] = "end_time" expression_attribute_names["#update_time"] = "update_time" expression_attribute_values[":end_time"] = current_time expression_attribute_values[":update_time"] = current_time doc_status_table.update_item( Key={"batch_id": batch_id}, UpdateExpression=update_expression, ExpressionAttributeNames=expression_attribute_names, ExpressionAttributeValues=expression_attribute_values, ReturnValues="UPDATED_NEW", ) logger.info("文档状态更新成功") return True except Exception as e: logger.error(f"更新文档状态失败: {e}") raise def convert_pdf_page_to_png_and_upload(pdf_stream, batch_id, filename): """处理PDF单页并实时上传到S3,避免内存堆积""" pdf_file = fitz.open(stream=pdf_stream, filetype='pdf') total_pages = len(pdf_file) logger.info(f"开始处理PDF: {filename}, 总页数: {total_pages}") for page_num in range(total_pages): try: page = pdf_file[page_num] # 直接生成PNG字节流,无需PIL中转 pix = page.get_pixmap() png_bytes = pix.tobytes("png") # 上传到S3,命名格式:batch_id/文件名_页码.png s3_key = f"{batch_id}/{os.path.splitext(filename)[0]}_page_{page_num+1}.png" s3.put_object( Bucket=TARGET_BUCKET, Key=s3_key, Body=png_bytes, ContentType="image/png" ) logger.info(f"第{page_num+1}页上传完成: {s3_key}") except Exception as e: logger.error(f"处理第{page_num+1}页失败: {e}") raise pdf_file.close() def process_zip_file(zip_bytes, batch_id): """处理ZIP文件中的PDF/TIFF""" with zipfile.ZipFile(BytesIO(zip_bytes), 'r') as zip_ref: for filename in zip_ref.namelist(): logger.info(f"处理文件: {filename}") if filename.endswith('.pdf'): pdf_data = zip_ref.read(filename) convert_pdf_page_to_png_and_upload(pdf_data, batch_id, filename) elif filename.endswith('.tif') or filename.endswith('.tiff'): # 保留原TIFF转PNG逻辑,建议同样优化为逐页上传 tiff_data = zip_ref.read(filename) # 此处可复用原转换逻辑并修改为逐页上传 pass else: logger.warning(f"跳过不支持的文件格式: {filename}") def lambda_handler(event, context): batch_id = None try: logger.info(f"收到事件: {json.dumps(event, indent=2)}") folder_name = event.get("FILENAME") if not folder_name or not isinstance(folder_name, str): raise ValueError("事件中未找到有效的目录名称") bucket_name = "cnc-aws-ecs-mecc-dev-doc-processing-v2" response = s3.list_objects_v2(Bucket=bucket_name, Prefix=folder_name) json_file_key = None zip_file_key = None for obj in response.get("Contents", []): file_key = obj["Key"] if file_key.endswith(".json") and file_key.startswith(folder_name): json_file_key = file_key elif file_key.endswith(".zip") and file_key.startswith(folder_name): zip_file_key = file_key if not json_file_key or not zip_file_key: raise Exception("未找到匹配的JSON和ZIP文件") # 获取batch_id json_obj = s3.get_object(Bucket=bucket_name, Key=json_file_key) data = json.loads(json_obj['Body'].read().decode('utf-8')) batch_id = data['SCANMETADATA']['batchid'] logger.info(f"Batch ID: {batch_id}") # 更新状态为处理中 update_document_status(batch_id, "inprogress", "Image Conversion") # 获取ZIP文件并处理 zip_obj = s3.get_object(Bucket=bucket_name, Key=zip_file_key) zip_bytes = zip_obj['Body'].read() process_zip_file(zip_bytes, batch_id) # 更新状态为完成 update_document_status(batch_id, "completed", "Image Conversion") # 清理源文件 s3.delete_object(Bucket=bucket_name, Key=json_file_key) s3.delete_object(Bucket=bucket_name, Key=zip_file_key) s3.delete_object(Bucket=bucket_name, Key=f"{folder_name}/") logger.info("源文件清理完成") return {"status": "success", "batch_id": batch_id} except Exception as e: logger.error(f"处理失败: {e}") if batch_id: update_document_status(batch_id, "failed", "Image Conversion") raise e
3. 超大型文件进阶处理方案
如果PDF超过200页,建议使用AWS Step Functions拆分任务:
- 将PDF按每20页拆分为子任务。
- 用Step Functions并行或串行调用Lambda处理每个子任务。
- 最后汇总所有子任务结果,更新整体状态。
关键注意事项
- 确保Lambda执行角色拥有S3读写、DynamoDB更新的权限。
- 监控Lambda内存使用和执行时间,根据实际情况调整配置。
- 加密PDF需先解密再处理。
内容的提问来源于stack exchange,提问作者Gomathi
相关产品推荐
相关产品推荐

