在AWS Lambda中验证解压Zip文件遇属性错误求助
AWS Lambda处理嵌套Zip文件时出现AttributeError问题
需求是AWS Lambda函数检查Zip文件中是否存在预定义的排除扩展名,已完成步骤:验证Zip文件无违规扩展名、解压文件、将文件解压至同目录的unzipped文件夹,但代码运行时出现属性错误。
问题代码
import json import zipfile import os import boto3 from urllib.parse import unquote_plus import io import re import gzip exclude_list = [".exe", ".scr", ".vbs", ".js", ".xml", "docm", ".xps"] sns = boto3.client('sns' ) def read_nested_zip(tf, bucket, key, s3_client): print(key) print ("search for.zip:",re.search(r'\.zip', key, re.IGNORECASE)) ## need to add exception handling ##if re.search(r'\.gzip$', key, re.IGNORECASE): ## print ('gzip file found') ## fil = gzip.GzipFile(tf, mode='rb') if re.search(r'\.zip$', key, re.IGNORECASE): print ('zip file found') fil = zipfile.ZipFile(tf, "r").namelist() else: fil = () print ('no file found') print (fil) ##with fil as zipf: ##try to narrow scope - run loop else exit for file in fil: print(file) if re.search(r'(\.zip|)$', file, re.IGNORECASE): childzip = io.BytesIO(fil.read(file)) read_nested_zip(childzip, bucket, key, s3_client) else: if any(x in file.lower() for x in exclude_list): print("Binary, dont load") print(file) print(bucket) print(key) env = bucket.split('-')[2].upper() # Copy the parent zip to a separate folder and remove it from the path copy_source = {'Bucket': bucket, 'Key': key} s3_client.copy_object(Bucket=bucket, CopySource=copy_source, Key='do_not_load_'+key) s3_client.delete_object(Bucket = bucket, Key = key) sns.publish( TopicArn = 'ARN', Subject = env + ': S3 upload warning: Non standard File encountered ', Message = 'Non standard File encountered' + key + ' uploaded to bucket ' + bucket + ' The file has been moved to ' + 'do_not_load_'+key ) else: print("File in supported formats, can be loaded " + file) #folder = re.sub(r"\/[^/]+$", "",key) folder = "/".join(key.split("/", 2)[:2]) + "/unzipped" print(folder) print("Bucket is "+ bucket) print("file to copy is "+ file) buffer = io.BytesIO(fil.read(file)) s3_resource = boto3.resource('s3') s3_resource.meta.client.upload_fileobj(buffer,Bucket=bucket,Key= folder + '/' + file) s3_resource.Object(bucket, folder + '/' + file).wait_until_exists() def lambda_handler(event, context): print(event) for record in event['Records']: s3_client = boto3.client('s3') key = unquote_plus(record['s3']['object']['key']) print(key) print (type(key)) size = record['s3']['object']['size'] bucket = record['s3']['bucket']['name'] obj = s3_client.get_object(Bucket=bucket, Key=key) print(obj) putObjects = [] with io.BytesIO(obj["Body"].read()) as tf: # rewind the file #tf.seek(0) read_nested_zip(tf, bucket, key, s3_client)
错误信息
[ERROR] AttributeError: 'list' object has no attribute 'read' Traceback (most recent call last): File "/var/task/lambda_function.py", line 85, in lambda_handler read_nested_zip(tf, bucket, key, s3_client) File "/var/task/lambda_function.py", line 35, in read_nested_zip childzip = io.BytesIO(fil.read())
已尝试方案
- 修改为
childzip = io.BytesIO(fil.read(file))或childzip = io.BytesIO(fil.read()),仍失败; - 修改为
childzip = io.BytesIO(fil),出现新错误:
[ERROR] AttributeError: module 'zipfile' has no attribute 'read' Traceback (most recent call last): File "/var/task/lambda_function.py", line 85, in lambda_handler read_nested_zip(tf, bucket, key, s3_client) File "/var/task/lambda_function.py", line 25, in read_nested_zip fil = zipfile.read(tf, "r").namelist()
解决思路与方案
核心问题
代码中fil = zipfile.ZipFile(tf, "r").namelist()这行将fil赋值为文件名列表,而非ZipFile对象,导致后续调用fil.read(file)时,列表类型没有read方法,触发AttributeError。
修复步骤
1. 保留ZipFile对象而非仅文件名列表
修改判断Zip文件的代码,先创建ZipFile实例,再单独获取文件名列表:
if re.search(r'\.zip$', key, re.IGNORECASE): print ('zip file found') zip_obj = zipfile.ZipFile(tf, "r") # 保留ZipFile对象 fil = zip_obj.namelist() # 获取文件名列表 else: fil = () print ('no file found')
2. 读取文件内容时使用ZipFile对象
后续需要读取文件内容的地方,用zip_obj.read(file)替代fil.read(file):
- 处理嵌套Zip时:
childzip = io.BytesIO(zip_obj.read(file)) read_nested_zip(childzip, bucket, file, s3_client) # 递归时传入当前子Zip的文件名
- 上传合法文件时:
buffer = io.BytesIO(zip_obj.read(file))
3. 修正嵌套Zip的递归参数
递归调用read_nested_zip时,第三个参数需传递当前子Zip的文件名(file),而非原父Zip的key,确保内部判断能正确识别子Zip。
4. 补充异常处理(可选)
添加Zip文件操作的异常捕获,处理损坏的Zip文件:
try: zip_obj = zipfile.ZipFile(tf, "r") fil = zip_obj.namelist() except zipfile.BadZipFile: print(f"Invalid zip file: {key}") # 可添加SNS告警或其他处理逻辑 return
修复后的核心函数示例
def read_nested_zip(tf, bucket, key, s3_client): print(key) print ("search for.zip:",re.search(r'\.zip', key, re.IGNORECASE)) zip_obj = None if re.search(r'\.zip$', key, re.IGNORECASE): print ('zip file found') try: zip_obj = zipfile.ZipFile(tf, "r") fil = zip_obj.namelist() except zipfile.BadZipFile: print(f"Corrupted or invalid zip file: {key}") return else: fil = () print ('no file found') print (fil) for file in fil: print(file) # 修正正则,匹配以.zip结尾的文件 if re.search(r'\.zip$', file, re.IGNORECASE): childzip = io.BytesIO(zip_obj.read(file)) read_nested_zip(childzip, bucket, file, s3_client) else: if any(x in file.lower() for x in exclude_list): print("Binary, dont load") print(file) print(bucket) print(key) env = bucket.split('-')[2].upper() copy_source = {'Bucket': bucket, 'Key': key} s3_client.copy_object(Bucket=bucket, CopySource=copy_source, Key='do_not_load_'+key) s3_client.delete_object(Bucket = bucket, Key = key) sns.publish( TopicArn = 'ARN', Subject = env + ': S3 upload warning: Non standard File encountered ', Message = 'Non standard File encountered' + key + ' uploaded to bucket ' + bucket + ' The file has been moved to ' + 'do_not_load_'+key ) else: print("File in supported formats, can be loaded " + file) folder = "/".join(key.split("/", 2)[:2]) + "/unzipped" print(folder) print("Bucket is "+ bucket) print("file to copy is "+ file) buffer = io.BytesIO(zip_obj.read(file)) s3_resource = boto3.resource('s3') s3_resource.meta.client.upload_fileobj(buffer,Bucket=bucket,Key= folder + '/' + file) s3_resource.Object(bucket, folder + '/' + file).wait_until_exists()
内容的提问来源于stack exchange,提问作者raunpa
相关产品推荐
相关产品推荐

