You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

在AWS Lambda中验证解压Zip文件遇属性错误求助

AWS Lambda处理嵌套Zip文件时出现AttributeError问题

需求是AWS Lambda函数检查Zip文件中是否存在预定义的排除扩展名,已完成步骤:验证Zip文件无违规扩展名、解压文件、将文件解压至同目录的unzipped文件夹,但代码运行时出现属性错误。

问题代码

import json
import zipfile
import os
import boto3
from urllib.parse import unquote_plus
import io
import re
import gzip


exclude_list = [".exe", ".scr", ".vbs", ".js", ".xml", "docm", ".xps"]
sns = boto3.client('sns' )


def read_nested_zip(tf, bucket, key, s3_client):
        print(key)
        print ("search for.zip:",re.search(r'\.zip', key, re.IGNORECASE))
        ## need to add exception handling
        ##if re.search(r'\.gzip$', key, re.IGNORECASE):
          ##  print ('gzip file found')
        ##    fil = gzip.GzipFile(tf, mode='rb')
        if re.search(r'\.zip$', key, re.IGNORECASE):
            print ('zip file found')
            fil = zipfile.ZipFile(tf, "r").namelist()
        else:
            fil = ()
            print ('no file found')
        print (fil)
        ##with fil as zipf:
            ##try to narrow scope - run loop else exit
        for file in fil:
            print(file)
            if re.search(r'(\.zip|)$', file, re.IGNORECASE):
                childzip = io.BytesIO(fil.read(file))
                read_nested_zip(childzip, bucket, key, s3_client)
            else:
                if any(x in file.lower() for x in exclude_list):
                    print("Binary, dont load")
                    print(file)
                    print(bucket)
                    print(key)
                    env = bucket.split('-')[2].upper()
                    # Copy the parent zip to a separate folder and remove it from the path
                    copy_source = {'Bucket': bucket, 'Key': key}
                    s3_client.copy_object(Bucket=bucket, CopySource=copy_source, Key='do_not_load_'+key)
                    s3_client.delete_object(Bucket = bucket, Key = key)
                    sns.publish(
                        TopicArn = 'ARN',
                        Subject = env + ': S3 upload warning: Non standard File encountered ',
                        Message = 'Non standard File encountered' + key + ' uploaded to bucket ' + bucket + ' The file has been moved to ' + 'do_not_load_'+key
                        )
                else:
                    print("File in supported formats, can be loaded " + file)
                    #folder = re.sub(r"\/[^/]+$", "",key)
                    folder = "/".join(key.split("/", 2)[:2]) + "/unzipped"
                    print(folder)
                    print("Bucket is "+ bucket)
                    print("file to copy is "+ file)
                    buffer = io.BytesIO(fil.read(file))
                    s3_resource = boto3.resource('s3')
                    s3_resource.meta.client.upload_fileobj(buffer,Bucket=bucket,Key= folder + '/' + file)
                    s3_resource.Object(bucket, folder + '/' + file).wait_until_exists()
                
    


def lambda_handler(event, context):
    print(event)
    for record in event['Records']:
        s3_client = boto3.client('s3')
        key = unquote_plus(record['s3']['object']['key'])
        print(key)
        print (type(key))
        size = record['s3']['object']['size']
        bucket = record['s3']['bucket']['name']
        obj = s3_client.get_object(Bucket=bucket, Key=key)
        print(obj)
        putObjects = []
        with io.BytesIO(obj["Body"].read()) as tf:
            # rewind the file
            #tf.seek(0)
            read_nested_zip(tf, bucket, key, s3_client)

错误信息

[ERROR] AttributeError: 'list' object has no attribute 'read'
Traceback (most recent call last):
  File "/var/task/lambda_function.py", line 85, in lambda_handler
    read_nested_zip(tf, bucket, key, s3_client)
  File "/var/task/lambda_function.py", line 35, in read_nested_zip
    childzip = io.BytesIO(fil.read())

已尝试方案

  • 修改为childzip = io.BytesIO(fil.read(file))或childzip = io.BytesIO(fil.read()),仍失败;
  • 修改为childzip = io.BytesIO(fil),出现新错误:
[ERROR] AttributeError: module 'zipfile' has no attribute 'read'
Traceback (most recent call last):
  File "/var/task/lambda_function.py", line 85, in lambda_handler
    read_nested_zip(tf, bucket, key, s3_client)
  File "/var/task/lambda_function.py", line 25, in read_nested_zip
    fil = zipfile.read(tf, "r").namelist()

解决思路与方案

核心问题

代码中fil = zipfile.ZipFile(tf, "r").namelist()这行将fil赋值为文件名列表,而非ZipFile对象,导致后续调用fil.read(file)时,列表类型没有read方法,触发AttributeError。

修复步骤

1. 保留ZipFile对象而非仅文件名列表

修改判断Zip文件的代码,先创建ZipFile实例,再单独获取文件名列表:

if re.search(r'\.zip$', key, re.IGNORECASE):
    print ('zip file found')
    zip_obj = zipfile.ZipFile(tf, "r")  # 保留ZipFile对象
    fil = zip_obj.namelist()  # 获取文件名列表
else:
    fil = ()
    print ('no file found')

2. 读取文件内容时使用ZipFile对象

后续需要读取文件内容的地方,用zip_obj.read(file)替代fil.read(file):

  • 处理嵌套Zip时:
childzip = io.BytesIO(zip_obj.read(file))
read_nested_zip(childzip, bucket, file, s3_client)  # 递归时传入当前子Zip的文件名
  • 上传合法文件时:
buffer = io.BytesIO(zip_obj.read(file))

3. 修正嵌套Zip的递归参数

递归调用read_nested_zip时,第三个参数需传递当前子Zip的文件名(file),而非原父Zip的key,确保内部判断能正确识别子Zip。

4. 补充异常处理(可选)

添加Zip文件操作的异常捕获,处理损坏的Zip文件:

try:
    zip_obj = zipfile.ZipFile(tf, "r")
    fil = zip_obj.namelist()
except zipfile.BadZipFile:
    print(f"Invalid zip file: {key}")
    # 可添加SNS告警或其他处理逻辑
    return

修复后的核心函数示例

def read_nested_zip(tf, bucket, key, s3_client):
    print(key)
    print ("search for.zip:",re.search(r'\.zip', key, re.IGNORECASE))
    zip_obj = None
    if re.search(r'\.zip$', key, re.IGNORECASE):
        print ('zip file found')
        try:
            zip_obj = zipfile.ZipFile(tf, "r")
            fil = zip_obj.namelist()
        except zipfile.BadZipFile:
            print(f"Corrupted or invalid zip file: {key}")
            return
    else:
        fil = ()
        print ('no file found')
    print (fil)

    for file in fil:
        print(file)
        # 修正正则,匹配以.zip结尾的文件
        if re.search(r'\.zip$', file, re.IGNORECASE):
            childzip = io.BytesIO(zip_obj.read(file))
            read_nested_zip(childzip, bucket, file, s3_client)
        else:
            if any(x in file.lower() for x in exclude_list):
                print("Binary, dont load")
                print(file)
                print(bucket)
                print(key)
                env = bucket.split('-')[2].upper()
                copy_source = {'Bucket': bucket, 'Key': key}
                s3_client.copy_object(Bucket=bucket, CopySource=copy_source, Key='do_not_load_'+key)
                s3_client.delete_object(Bucket = bucket, Key = key)
                sns.publish(
                    TopicArn = 'ARN',
                    Subject = env + ': S3 upload warning: Non standard File encountered ',
                    Message = 'Non standard File encountered' + key + ' uploaded to bucket ' + bucket + ' The file has been moved to ' + 'do_not_load_'+key
                )
            else:
                print("File in supported formats, can be loaded " + file)
                folder = "/".join(key.split("/", 2)[:2]) + "/unzipped"
                print(folder)
                print("Bucket is "+ bucket)
                print("file to copy is "+ file)
                buffer = io.BytesIO(zip_obj.read(file))
                s3_resource = boto3.resource('s3')
                s3_resource.meta.client.upload_fileobj(buffer,Bucket=bucket,Key= folder + '/' + file)
                s3_resource.Object(bucket, folder + '/' + file).wait_until_exists()

内容的提问来源于stack exchange,提问作者raunpa

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.09 02:30:58