You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Spring Cloud Gateway自定义条件过滤器实现测试流量性能统计

针对指定请求头的代理层性能指标收集过滤器实现

下面按几种常见的代理服务场景,给出具体的实现方案:

1. Nginx 环境(基于Lua脚本过滤)

如果你的代理是Nginx,推荐用OpenResty的Lua模块实现。假设测试脚本会带上X-Perf-Test: true这个自定义请求头,只有存在该头的请求才会被统计性能指标。

实现步骤:

  1. 确保Nginx已集成Lua模块(直接用OpenResty最省心)
  2. 在Nginx配置中添加Lua逻辑,完成请求头判断、延迟统计和TP指标计算

示例配置片段:

http {
    # 初始化共享内存,存储过滤后的延迟数据(支持多worker进程共享)
    lua_shared_dict perf_metrics 10m;

    server {
        listen 80;

        location / {
            # 初始化请求开始时间变量
            set $start_time '';
            
            # 请求到达时检查自定义头,符合条件则记录开始时间
            access_by_lua_block {
                local test_header = ngx.req.get_headers()["X-Perf-Test"]
                if test_header and test_header == "true" then
                    ngx.var.start_time = ngx.now()
                end
            }

            # 转发请求到后端服务
            proxy_pass http://your_backend_service;

            # 请求结束后计算延迟并收集指标
            log_by_lua_block {
                local start_time = ngx.var.start_time
                if start_time ~= '' then
                    -- 计算延迟(转成毫秒)
                    local latency = (ngx.now() - start_time) * 1000
                    local metrics_dict = ngx.shared.perf_metrics

                    -- 维护滑动窗口存储最近1000条延迟数据
                    local latencies = metrics_dict:get("latencies") or "[]"
                    latencies = cjson.decode(latencies)
                    table.insert(latencies, latency)
                    if #latencies > 1000 then
                        table.remove(latencies, 1)
                    end
                    metrics_dict:set("latencies", cjson.encode(latencies))

                    -- 计算TP95/TP99指标
                    table.sort(latencies)
                    local count = #latencies
                    local tp95 = count > 0 and latencies[math.floor(count * 0.95)] or 0
                    local tp99 = count > 0 and latencies[math.floor(count * 0.99)] or 0

                    -- 发布指标:可打印到日志,或推送到Prometheus等监控系统
                    ngx.log(ngx.INFO, string.format("Perf Test Metrics - TP95: %.2fms, TP99: %.2fms, Total: %d", tp95, tp99, count))
                end
            }
        }
    }
}

2. Envoy 代理环境(自定义HTTP Filter)

如果用Envoy做代理,可以编写自定义HTTP Filter来实现需求。核心逻辑是在请求阶段识别测试头,响应阶段计算延迟并上报指标。

示例Filter核心代码片段(C++):

#include "envoy/http/filter.h"
#include "envoy/registry/registry.h"

class PerfTestFilter : public Envoy::Http::StreamFilter {
public:
  Envoy::Http::FilterHeadersStatus decodeHeaders(Envoy::Http::RequestHeaderMap& headers, bool) override {
    // 检查自定义测试请求头
    if (headers.get(Envoy::Http::LowerCaseString("x-perf-test")) != nullptr) {
      start_time_ = Envoy::MonotonicTime::now();
      track_perf_ = true;
    }
    return Envoy::Http::FilterHeadersStatus::Continue;
  }

  Envoy::Http::FilterHeadersStatus encodeHeaders(Envoy::Http::ResponseHeaderMap&, bool) override {
    if (track_perf_) {
      // 计算请求延迟(毫秒)
      auto latency = std::chrono::duration_cast<std::chrono::milliseconds>(Envoy::MonotonicTime::now() - start_time_).count();
      
      // 将延迟数据上报到Envoy指标系统,后续可由Prometheus等计算分位数
      auto& stats = callback_->streamInfo().statsScope();
      stats.distribution("perf_test.latency_ms").addValue(latency);
    }
    return Envoy::Http::FilterHeadersStatus::Continue;
  }

private:
  Envoy::MonotonicTime start_time_;
  bool track_perf_ = false;
  Envoy::Http::StreamFilterCallbacks* callback_ = nullptr;
};

// 注册自定义Filter到Envoy
static Envoy::Registry::RegisterFactory<PerfTestFilterFactory, Envoy::Server::Configuration::NamedHttpFilterConfigFactory> register_;

说明:Envoy自带的指标系统可以自动计算TP95/TP99等分位数,只需在配置中开启对应直方图的分位数统计即可。

3. Python 自定义代理(FastAPI/Flask)

如果是自研的Python代理服务,实现逻辑会更简洁。以下是FastAPI的示例代码:

from fastapi import FastAPI, Request
import httpx
import time
from collections import deque

app = FastAPI()

# 滑动窗口存储最近1000条测试请求的延迟数据
latency_window = deque(maxlen=1000)

@app.api_route("/{path:path}", methods=["GET", "POST", "PUT", "DELETE"])
async def proxy_request(request: Request, path: str):
    # 检查自定义测试请求头
    perf_test_flag = request.headers.get("X-Perf-Test") == "true"
    start_time = time.time() if perf_test_flag else None

    # 转发请求到后端服务
    async with httpx.AsyncClient() as client:
        backend_url = f"http://your_backend_service/{path}"
        response = await client.request(
            method=request.method,
            url=backend_url,
            headers=request.headers,
            content=await request.body()
        )

    # 仅对测试请求计算并发布指标
    if start_time:
        latency = (time.time() - start_time) * 1000
        latency_window.append(latency)
        
        # 计算TP95/TP99
        sorted_lats = sorted(latency_window)
        count = len(sorted_lats)
        tp95 = sorted_lats[int(count*0.95)-1] if count >=20 else 0
        tp99 = sorted_lats[int(count*0.99)-1] if count >=100 else 0
        
        # 发布指标:可打印日志、推送到监控系统
        print(f"Perf Metrics - TP95: {tp95:.2f}ms, TP99: {tp99:.2f}ms, Total Requests: {count}")

    # 返回后端响应给客户端
    return Response(
        content=response.content,
        status_code=response.status_code,
        headers=dict(response.headers)
    )

通用注意事项:

  • 请求头校验要严格:除了检查头存在,最好验证具体值(比如X-Perf-Test: true),避免误统计生产流量
  • 控制性能开销:高流量场景下,不要每次请求都计算TP指标,建议定时(比如每秒)批量计算一次
  • 指标持久化:建议将指标推送到Prometheus、InfluxDB等监控系统,方便后续可视化分析
  • 滑动窗口大小:根据测试流量QPS调整,确保窗口能反映近期性能,同时避免内存占用过高

内容的提问来源于stack exchange,提问作者user2634891

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.18 19:23:24