K8s存储性能监控:fio-summary.txt更新后Prometheus指标不刷新求助
问题:Kubernetes上FIO测试文件变更后Prometheus指标未更新
我有一个运行在Kubernetes上的存储性能测试应用,测试完成后会将IOPS、带宽、延迟等统计数据写入fio-summary.txt文件。之后通过Python脚本将指标推送至Prometheus,虽然能在Prometheus上看到指标,但fio-summary.txt文件修改后指标并未更新。已尝试使用watchdog库,但问题仍未解决,需要指导如何持续监控fio-summary.txt文件变更并重新加载指标。
fio-summary.txt内容
# cat fio-summary.txt RAND_IOPS_READ,RAND_IOPS_WRITE,MIX_IOPS_READ,MIX_IOPS_WRITE,RAND_BW_READ,RAND_BW_WRITE,SEQ_BW_READ,SEQ_BW_WRITE,AVG_LATENCY_READ,AVG_LATENCY_WRITE [9999.0, 8393.5, 2266.0, 756.0, 304.0, 32.8, 311.5, 0.605, 1.655, 15.235] [84935.0, 8158.5, 1375.5, 448.5, 332.5, 32.75, 1156.5, 7.596, 30.105, 309.343] [83126.0, 6711.5, 1668.0, 551.0, 326.0, 27.6, 1081.5, 16.85, 61.481, 752.128]
现有Python脚本
import re import time from prometheus_client import CollectorRegistry, Gauge from prometheus_client.exposition import generate_latest from http.server import BaseHTTPRequestHandler, HTTPServer from watchdog.observers import Observer from watchdog.events import FileSystemEventHandler from watchdog.events import FileModifiedEvent import threading # Define a registry to hold your metrics registry = CollectorRegistry() # Function to create gauge metrics for different thread counts and metric types def create_gauge_metrics(thread_counts, metric_types): gauge_metrics = {} for metric_type in metric_types: for count in thread_counts: gauge_metrics[f'FIO_{metric_type}_threadCount_{count}'] = Gauge(f'FIO_{metric_type}_threadCount_{count}', f'FIO_{metric_type.capitalize()} for {count} Threads', registry=registry) return gauge_metrics # Example thread counts and metric types thread_counts = [1, 20, 40] metric_types = ['RAND_IOPS_READ', 'RAND_IOPS_WRITE', 'MIX_IOPS_READ', 'MIX_IOPS_WRITE', 'RAND_BW_READ', 'RAND_BW_WRITE', 'SEQ_BW_READ', 'SEQ_BW_WRITE', 'AVG_LATENCY_READ', 'AVG_LATENCY_WRITE'] # Create gauge metrics gauge_metrics = create_gauge_metrics(thread_counts, metric_types) def transpose_data(filename): transposed_data = [] with open(filename, 'r') as file: headers = file.readline().strip().split(',') for line in file: #values = re.findall(r'\[(.*?)\]', line)[0].split(', ') Working for Windows values = re.findall(r'\[(.*?)\]', line.strip())[0].split(', ') # for Linux transposed_data.append(values) return list(zip(*transposed_data)), headers def expose_data_from_file(filename, gauge_metrics): print("Exposing data from file:", filename) data, headers = transpose_data(filename) print("Data read from file:", data) print("Headers:", headers) for metric_type, values in zip(headers, data): for count, value in enumerate(values): metric_name = f'FIO_{metric_type}_threadCount_{thread_counts[count]}' try: thread_value = float(value) print("Setting metric:", metric_name, "to value:", thread_value) gauge_metrics[metric_name].set(thread_value) except Exception as e: print(f"Error processing value {value} for metric {metric_name}: {e}") # Example usage expose_data_from_file('fio-summary.txt', gauge_metrics) class FileEventHandler(FileSystemEventHandler): def on_modified(self, event): if isinstance(event, FileModifiedEvent) and event.src_path == 'fio-summary.txt': print("File modified! Reloading metrics...") expose_data_from_file('fio-summary.txt', gauge_metrics) else: print("Event type:", type(event)) # MetricsHandler class class MetricsHandler(BaseHTTPRequestHandler): def do_GET(self): print("HTTP GET request received") self.send_response(200) self.send_header('Content-Type', 'text/plain') self.end_headers() self.wfile.write(generate_latest(registry)) print("HTTP response sent") # Expose the registry containing your metrics def start_http_server(): print("Starting HTTP server...") server_address = ('', 8080) # Change port if needed httpd = HTTPServer(server_address, MetricsHandler) print('Serving metrics on port 8080') httpd.serve_forever() # File change handler def start_monitoring(): # Create an event handler event_handler = FileEventHandler() # Create an observer observer = Observer() observer.schedule(event_handler, '.', recursive=False) # Monitor only current directory # Start the observer observer.start() # Start the HTTP server in a separate thread http_thread = threading.Thread(target=start_http_server) http_thread.start() try: # Keep the observer running indefinitely observer.join() except KeyboardInterrupt: observer.stop() observer.join() if __name__ == '__main__': start_monitoring()
脚本运行日志(修改fio-summary.txt后)
# python3.6 17.py Exposing data from file: fio-summary.txt Data read from file: [('999.0', '84935.0', '83126.0'), ('8393.5', '8158.5', '6711.5'), ('2266.0', '1375.5', '1668.0'), ('756.0', '448.5', '551.0'), ('304.0', '332.5', '326.0'), ('32.8', '32.75', '27.6'), ('311.5', '1156.5', '1081.5'), ('0.605', '7.596', '16.85'), ('1.655', '30.105', '61.481'), ('15.235', '309.343', '752.128')] Headers: ['RAND_IOPS_READ', 'RAND_IOPS_WRITE', 'MIX_IOPS_READ', 'MIX_IOPS_WRITE', 'RAND_BW_READ', 'RAND_BW_WRITE', 'SEQ_BW_READ', 'SEQ_BW_WRITE', 'AVG_LATENCY_READ', 'AVG_LATENCY_WRITE'] Setting metric: FIO_RAND_IOPS_READ_threadCount_1 to value: 999.0 Setting metric: FIO_RAND_IOPS_READ_threadCount_20 to value: 84935.0 Setting metric: FIO_RAND_IOPS_READ_threadCount_40 to value: 83126.0 Setting metric: FIO_RAND_IOPS_WRITE_threadCount_1 to value: 8393.5 Setting metric: FIO_RAND_IOPS_WRITE_threadCount_20 to value: 8158.5 Setting metric: FIO_RAND_IOPS_WRITE_threadCount_40 to value: 6711.5 Setting metric: FIO_MIX_IOPS_READ_threadCount_1 to value: 2266.0 Setting metric: FIO_MIX_IOPS_READ_threadCount_20 to value: 1375.5 Setting metric: FIO_MIX_IOPS_READ_threadCount_40 to value: 1668.0 Setting metric: FIO_MIX_IOPS_WRITE_threadCount_1 to value: 756.0 Setting metric: FIO_MIX_IOPS_WRITE_threadCount_20 to value: 448.5 Setting metric: FIO_MIX_IOPS_WRITE_threadCount_40 to value: 551.0 Setting metric: FIO_RAND_BW_READ_threadCount_1 to value: 304.0 Setting metric: FIO_RAND_BW_READ_threadCount_20 to value: 332.5 Setting metric: FIO_RAND_BW_READ_threadCount_40 to value: 326.0 Setting metric: FIO_RAND_BW_WRITE_threadCount_1 to value: 32.8 Setting metric: FIO_RAND_BW_WRITE_threadCount_20 to value: 32.75 Setting metric: FIO_RAND_BW_WRITE_threadCount_40 to value: 27.6 Setting metric: FIO_SEQ_BW_READ_threadCount_1 to value: 311.5 Setting metric: FIO_SEQ_BW_READ_threadCount_20 to value: 1156.5 Setting metric: FIO_SEQ_BW_READ_threadCount_40 to value: 1081.5 Setting metric: FIO_SEQ_BW_WRITE_threadCount_1 to value: 0.605 Setting metric: FIO_SEQ_BW_WRITE_threadCount_20 to value: 7.596 Setting metric: FIO_SEQ_BW_WRITE_threadCount_40 to value: 16.85 Setting metric: FIO_AVG_LATENCY_READ_threadCount_1 to value: 1.655 Setting metric: FIO_AVG_LATENCY_READ_threadCount_20 to value: 30.105 Setting metric: FIO_AVG_LATENCY_READ_threadCount_40 to value: 61.481 Setting metric: FIO_AVG_LATENCY_WRITE_threadCount_1 to value: 15.235 Setting metric: FIO_AVG_LATENCY_WRITE_threadCount_20 to value: 309.343 Setting metric: FIO_AVG_LATENCY_WRITE_threadCount_40 to value: 752.128 Starting HTTP server... Serving metrics on port 8080 HTTP GET request received 192.168.56.1 - - [08/Apr/2024 05:42:27] "GET /metrics HTTP/1.1" 200 - HTTP response sent HTTP GET request received 192.168.56.1 - - [08/Apr/2024 05:42:27] "GET /favicon.ico HTTP/1.1" 200 - HTTP response sent Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.FileModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> Event type: <class 'watchdog.events.DirModifiedEvent'> HTTP GET request received 192.168.56.1 - - [08/Apr/2024 05:42:44] "GET /metrics HTTP/1.1" 200 - HTTP response sent HTTP GET request received 192.168.56.1 - - [08/Apr/2024 05:42:44] "GET /favicon.ico HTTP/1.1" 200 - HTTP response sent
问题分析与修复方案
从日志可以看到,watchdog确实检测到了FileModifiedEvent,但代码中event.src_path == 'fio-summary.txt'的判断逻辑存在问题:Linux系统中event.src_path会返回绝对路径,而非相对路径,导致条件不匹配,无法触发指标重新加载。
修复步骤:
修正文件路径判断逻辑
提取文件名进行匹配,避免路径格式差异:import os class FileEventHandler(FileSystemEventHandler): def on_modified(self, event): if isinstance(event, FileModifiedEvent) and os.path.basename(event.src_path) == 'fio-summary.txt': print("File modified! Reloading metrics...") expose_data_from_file('fio-summary.txt', gauge_metrics) else: print("Event type:", type(event))添加防抖机制避免重复触发
编辑器修改文件时通常会产生多次FileModifiedEvent,添加防抖逻辑减少重复加载:import os import time class FileEventHandler(FileSystemEventHandler): def __init__(self): self.last_modified = 0 def on_modified(self, event): if isinstance(event, FileModifiedEvent) and os.path.basename(event.src_path) == 'fio-summary.txt': current_time = time.time() # 1秒内只处理一次变更 if current_time - self.last_modified > 1: print("File modified! Reloading metrics...") expose_data_from_file('fio-summary.txt', gauge_metrics) self.last_modified = current_time else: print("Event type:", type(event))确保文件读取完整性
添加重试逻辑,避免读取到未写完的文件内容:def expose_data_from_file(filename, gauge_metrics): print("Exposing data from file:", filename) max_retries = 3 retry_delay = 0.5 for _ in range(max_retries): try: data, headers = transpose_data(filename) print("Data read from file:", data) print("Headers:", headers) for metric_type, values in zip(headers, data): for count, value in enumerate(values): metric_name = f'FIO_{metric_type}_threadCount_{thread_counts[count]}' try: thread_value = float(value) print("Setting metric:", metric_name, "to value:", thread_value) gauge_metrics[metric_name].set(thread_value) except Exception as e: print(f"Error processing value {value} for metric {metric_name}: {e}") return except Exception as e: print(f"Failed to read file, retrying: {e}") time.sleep(retry_delay) print("Failed to read file after multiple retries")
额外优化建议:
- 在Kubernetes环境中,将
fio-summary.txt挂载为PersistentVolumeClaim或ConfigMap,确保文件变更能被正确检测。 - 若允许轻度延迟,可考虑让Prometheus通过
file_sd_configs主动读取文件;如需实时更新,当前HTTP暴露方式更合适。
内容的提问来源于stack exchange,提问作者ratnakar reddy
相关产品推荐
相关产品推荐

