如何用Lambda+CloudWatch监控AWS FSx ONTAP存储使用率?报错求助
问题描述
首次尝试用AWS Lambda函数监控AWS中的FSx for ONTAP(FSxN)存储,目标获取StorageCapacity和StorageTier以计算filesystemUsed总使用率百分比,但执行代码时触发IndexError(列表索引越界)错误。
尝试的代码
import json import boto3 from datetime import datetime def lambda_handler(event, context): fsx = boto3.client('fsx') filesystems = fsx.describe_file_systems() table = [] for filesystem in filesystems.get('FileSystems'): status = filesystem.get('Lifecycle') filesystem_id = filesystem.get('FileSystemId') table.append(filesystem_id) cloudwatch = boto3.client('cloudwatch') result = [] for filesystem_id in table: current_time = datetime.utcnow().isoformat() response = cloudwatch.get_metric_data( MetricDataQueries=[ { 'Id': 'm1', 'MetricStat': { 'Metric': { 'Namespace': 'AWS/FSx', 'MetricName': 'StorageCapacity', 'Dimensions': [ { 'Name': 'FileSystemId', 'Value': filesystem_id }, { 'Name': 'StorageTier', 'Value': 'SSD' }, { 'Name': 'DataType', 'Value': 'All' } ] }, 'Period': 60, 'Stat': 'Sum' }, 'ReturnData': True }, { 'Id': 'm2', 'MetricStat': { 'Metric': { 'Namespace': 'AWS/FSx', 'MetricName': 'StorageUsed', 'Dimensions': [ { 'Name': 'FileSystemId', 'Value': filesystem_id }, { 'Name': 'StorageTier', 'Value': 'SSD' }, { 'Name': 'DataType', 'Value': 'All' } ] }, 'Period': 60, 'Stat': 'Sum' }, 'ReturnData': True } ], StartTime='2023-01-20T00:01:00Z', EndTime='2023-01-20T00:02:00Z' ) storage_capacity = response['MetricDataResults'][0]['Values'][0] storage_used = response['MetricDataResults'][1]['Values'][0] result.append({'filesystem_id': filesystem_id,'storage_capacity': storage_capacity, 'storage_used': storage_used}) return result
执行错误
{ "errorMessage": "list index out of range", "errorType": "IndexError", "requestId": "a09573f2-87ea-4464-afc0-8b196f669415", "stackTrace": [ " File \"/var/task/lambda_function.py\", line 75, in lambda_handler\n storage_capacity = response['MetricDataResults'][0]['Values'][0]\n" ] }
MetricDataResults示例
{'MetricDataResults': [ {'Id': 'm1', 'Label': 'StorageCapacity', 'Timestamps': [datetime.datetime(2023, 1, 20, 0, 1, tzinfo=tzlocal())], 'Values': [925308932096.0], 'StatusCode': 'Complete'}, {'Id': 'm2', 'Label': 'StorageUsed', 'Timestamps': [datetime.datetime(2023, 1, 20, 0, 1, tzinfo=tzlocal())], 'Values': [2439143424.0], 'StatusCode': 'Complete'}], 'Messages': [], 'ResponseMetadata': {'RequestId': '479a53b2-b5f5-46c0-b79d-278d803df94b', 'HTTPStatusCode': 200, 'HTTPHeaders': {'x-amzn-requestid': '479a53b2-b5f5-46c0-b79d-278d803df94b', 'content-type': 'text/xml', 'content-length': '923', 'date': 'Sat, 21 Jan 2023 18:06:35 GMT'}, 'RetryAttempts': 0}} {'MetricDataResults': [ {'Id': 'm1', 'Label': 'StorageCapacity', 'Timestamps': [datetime.datetime(2023, 1, 20, 0, 1, tzinfo=tzlocal())], 'Values': [925308932096.0], 'StatusCode': 'Complete'}, {'Id': 'm2', 'Label': 'StorageUsed', 'Timestamps': [datetime.datetime(2023, 1, 20, 0, 1, tzinfo=tzlocal())], 'Values': [2593112064.0], 'StatusCode': 'Complete'}], 'Messages': [], 'ResponseMetadata': {'RequestId': 'db9ad0a4-0a24-4f1d-be60-55bde63fd49b', 'HTTPStatusCode': 200, 'HTTPHeaders': {'x-amzn-requestid': 'db9ad0a4-0a24-4f1d-be60-55bde63fd49b', 'content-type': 'text/xml', 'content-length': '923', 'date': 'Sat, 21 Jan 2023 18:06:35 GMT'}, 'RetryAttempts': 0}}
错误原因分析
IndexError出现在response['MetricDataResults'][0]['Values'][0],核心原因是代码直接假设指标数据一定存在,实际可能触发以下场景:
- 循环中包含了非FSx for ONTAP的文件系统(比如FSx for Windows/Linux),这类系统的指标维度不匹配,导致CloudWatch返回空数据
- 硬编码了
StorageTier为SSD,但部分FSxN实例使用的是HDD存储层,请求错误维度的指标会返回空 - 固定的时间范围(
2023-01-20T00:01:00Z到2023-01-20T00:02:00Z)内没有数据点 - 文件系统处于非可用状态(比如创建中、删除中),没有生成指标数据
解决方案
1. 过滤FSx for ONTAP类型的可用实例
在获取文件系统ID时,仅保留FileSystemType为ONTAP且生命周期为AVAILABLE的实例:
for filesystem in filesystems.get('FileSystems', []): if filesystem.get('FileSystemType') != 'ONTAP' or filesystem.get('Lifecycle') != 'AVAILABLE': continue filesystem_id = filesystem.get('FileSystemId') # 同时获取实例的实际存储层 storage_tier = filesystem['OntapConfiguration']['StorageTier'] table.append({'filesystem_id': filesystem_id, 'storage_tier': storage_tier})
2. 动态使用存储层维度
不再硬编码StorageTier为SSD,改用从FSx API获取的实际存储层信息:
# 在CloudWatch请求的Dimensions中替换为动态值 { 'Name': 'StorageTier', 'Value': item['storage_tier'] }
3. 增加数据存在性安全检查
在访问Values列表前,先验证数据是否存在,避免索引越界:
# 按ID查找指标结果,避免依赖列表顺序 m1_result = next((res for res in response['MetricDataResults'] if res['Id'] == 'm1'), None) if not m1_result or not m1_result['Values']: print(f"跳过FileSystem {filesystem_id}:无StorageCapacity数据") continue storage_capacity = m1_result['Values'][0] m2_result = next((res for res in response['MetricDataResults'] if res['Id'] == 'm2'), None) if not m2_result or not m2_result['Values']: print(f"跳过FileSystem {filesystem_id}:无StorageUsed数据") continue storage_used = m2_result['Values'][0]
4. 使用动态时间范围
替换固定的历史时间为相对当前时间的范围,确保有数据:
from datetime import timedelta start_time = datetime.utcnow() - timedelta(minutes=5) end_time = datetime.utcnow() # 在get_metric_data中使用动态时间 StartTime=start_time, EndTime=end_time
完整修正后的代码
import json import boto3 from datetime import datetime, timedelta def lambda_handler(event, context): fsx = boto3.client('fsx') filesystems = fsx.describe_file_systems() table = [] # 仅保留ONTAP类型且可用的文件系统 for filesystem in filesystems.get('FileSystems', []): if filesystem.get('FileSystemType') != 'ONTAP' or filesystem.get('Lifecycle') != 'AVAILABLE': continue filesystem_id = filesystem.get('FileSystemId') # 获取ONTAP实例的实际存储层 storage_tier = filesystem['OntapConfiguration']['StorageTier'] table.append({'filesystem_id': filesystem_id, 'storage_tier': storage_tier}) cloudwatch = boto3.client('cloudwatch') result = [] # 使用最近5分钟的时间范围确保有数据 start_time = datetime.utcnow() - timedelta(minutes=5) end_time = datetime.utcnow() for item in table: filesystem_id = item['filesystem_id'] storage_tier = item['storage_tier'] response = cloudwatch.get_metric_data( MetricDataQueries=[ { 'Id': 'm1', 'MetricStat': { 'Metric': { 'Namespace': 'AWS/FSx', 'MetricName': 'StorageCapacity', 'Dimensions': [ {'Name': 'FileSystemId', 'Value': filesystem_id}, {'Name': 'StorageTier', 'Value': storage_tier}, {'Name': 'DataType', 'Value': 'All'} ] }, 'Period': 60, 'Stat': 'Sum' }, 'ReturnData': True }, { 'Id': 'm2', 'MetricStat': { 'Metric': { 'Namespace': 'AWS/FSx', 'MetricName': 'StorageUsed', 'Dimensions': [ {'Name': 'FileSystemId', 'Value': filesystem_id}, {'Name': 'StorageTier', 'Value': storage_tier}, {'Name': 'DataType', 'Value': 'All'} ] }, 'Period': 60, 'Stat': 'Sum' }, 'ReturnData': True } ], StartTime=start_time, EndTime=end_time ) # 安全获取StorageCapacity数据 m1_result = next((res for res in response['MetricDataResults'] if res['Id'] == 'm1'), None) if not m1_result or not m1_result['Values']: print(f"跳过FileSystem {filesystem_id}:无StorageCapacity数据") continue storage_capacity = m1_result['Values'][0] # 安全获取StorageUsed数据 m2_result = next((res for res in response['MetricDataResults'] if res['Id'] == 'm2'), None) if not m2_result or not m2_result['Values']: print(f"跳过FileSystem {filesystem_id}:无StorageUsed数据") continue storage_used = m2_result['Values'][0] # 计算使用率百分比 usage_percent = round((storage_used / storage_capacity) * 100, 2) result.append({ 'filesystem_id': filesystem_id, 'storage_capacity': storage_capacity, 'storage_used': storage_used, 'usage_percent': usage_percent }) return result
内容的提问来源于stack exchange,提问作者user2023
相关产品推荐
相关产品推荐

