You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

OpenShift下Varnish+Nginx架构低吞吐量高响应时间问题排查求助

问题背景

基于OpenShift部署的服务架构为:Nginx => Varnish(2个Pod) => Node.js Web API。
使用JMeter在60秒内对同一个URL发起1000次请求,该URL理应被Varnish缓存实现快速响应,但实际测试结果不符合预期:
测试结果图
所有请求的响应头均携带X-Cache: HIT_1,确认请求确实由Varnish缓存命中返回,目前未定位到性能瓶颈根因。

配置信息

Varnish VCL配置

vcl 4.0;

import std;
import bodyaccess;

backend freshproxy {
   .host = "fresh-proxy";
   .port = "3000";
}

sub vcl_recv {
  set req.backend_hint = freshproxy;
    if (req.method == "XCGFULLBAN") {
        ban("req.http.host ~ .*");
        return (synth(200, "Full cache cleared"));
    }

  if (req.method == "GET" && ! req.url ~ "varnish_no_cache") {
    if (req.url ~ "^/api/v1/tours/favorite" || req.url ~ "^/api/v1/products/favorite") {
        return (pass);
    }

    if (
      req.url ~ "^/api/v1/products" ||
      req.url ~ "^/api/v1/tours" ||
      req.url ~ "^/api/v1/farmers" ||
      req.url ~ "^/api/v1/stories" ||
      req.url ~ "^/api/v1/recipes" ||
      req.url ~ "^/api/v1/categories/farmers" ||
      req.url ~ "^/api/v1/categories/tours" ||
      req.url ~ "^/api/v1/categories/recipes"
    ) {
      return (hash);
    }
  }

  return (pass);
}

sub vcl_backend_response {
    # We first set TTLs for most of the content we need to cache
    set beresp.ttl = 30m;
    set beresp.grace = 30m;
}

sub vcl_hash {
    # To cache POST and PUT requests
    if (req.http.X-Body-Len) {
    bodyaccess.hash_req_body();
    } else {
    hash_data("");
    }
}

sub vcl_backend_fetch {
    if (bereq.http.X-Body-Len) {
    set bereq.method = "POST";
    }
}

sub vcl_deliver {
    if (obj.hits > 0) {
    set resp.http.X-Cache = "HIT_1";
    set resp.http.X-Cache-Hits = obj.hits;
    } else {
    set resp.http.X-Cache = "MISS_1";
    }
    set resp.http.X-Cache-Expires = resp.http.Expires;
    unset resp.http.X-Varnish;
    unset resp.http.Via;
    unset resp.http.Age;
    unset resp.http.X-Purge-URL;
    unset resp.http.X-Purge-Host;
    # Remove ban-lurker friendly custom headers when delivering to client.
    unset resp.http.X-Url;
    unset resp.http.X-Host;
    # Comment these for easier Drupal cache tag debugging in development.
    unset resp.http.X-Cache-Tags;
    unset resp.http.X-Cache-Contexts;
    unset resp.http.X-Powered-By;
}

Nginx配置

server {
        listen 80;
        server_name api.my-app.ru;

        include well-known.conf;

        location /robots.txt { return 200 "User-agent: *\nDisallow: /\n"; }

        location / {
                    return 301 https://$host$request_uri;
        }
}

server {
        listen 443 ssl http2;
        server_name api.my-app.ru;

#        auth_basic "Restricted";
#        auth_basic_user_file /etc/nginx/htpasswd;

        ssl_certificate /etc/nginx/certs/my-app.ru/my-app.ru.crt;
        ssl_certificate_key /etc/nginx/certs/my-app.ru/my-app.ru.key;
        ssl_protocols TLSv1.2 TLSv1.3;
        ssl_prefer_server_ciphers On;
        ssl_ciphers ECDHE-RSA-AES128-GCM-SHA256:ECDHE-ECDSA-AES128-GCM-SHA256:ECDHE-RSA-AES256-GCM-SHA384:ECDHE-ECDSA-AES256-GCM-SHA384:DHE-RSA-AES128-GCM-SHA256:DHE-DSS-AES128-GCM-SHA256:kEDH+AESGCM:ECDHE-RSA-AES128-SHA256:ECDHE-ECDSA-AES128-SHA256:ECDHE-RSA-AES128-SHA:ECDHE-ECDSA-AES128-SHA:ECDHE-RSA-AES256-SHA384:ECDHE-ECDSA-AES256-SHA384:ECDHE-RSA-AES256-SHA:ECDHE-ECDSA-AES256-SHA:DHE-RSA-AES128-SHA256:DHE-RSA-AES128-SHA:DHE-DSS-AES128-SHA256:DHE-RSA-AES256-SHA256:DHE-DSS-AES256-SHA:DHE-RSA-AES256-SHA:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!3DES:!MD5:!PSK;
        add_header Strict-Transport-Security max-age=15768000;
        ssl_stapling on;

        include gzip.conf;
        include location_deny.conf;
        fastcgi_param                   HTTPS on;
        # To allow POST on static pages
        error_page  405     =200 $uri;
        include well-known.conf;

        access_log  /var/log/nginx/api.my-app.ru_access.log  main;
        error_log  /var/log/nginx/api.my-app.ru_error.log;        


        location /robots.txt { return 200 "User-agent: *\nDisallow: /\n"; }

        location / {
                proxy_set_header Upgrade $http_upgrade;
                proxy_set_header Connection "upgrade";
                proxy_buffering off;
                proxy_pass http://openshift-prod;
                proxy_set_header Host $host;
                proxy_set_header realip $remote_addr;
                proxy_set_header X-Real-IP  $remote_addr;
                proxy_set_header X-Forwarded-Proto https;
                proxy_set_header X-Forwarded-Port 443;
                proxy_set_header Ssl-Offloaded "https";
                proxy_set_header HTTPS "on";
                proxy_set_header X-Forwarded-For $remote_addr;
                proxy_read_timeout 1200;
                proxy_send_timeout 1200;
                proxy_connect_timeout 1200;

                ### SET GEOIP Variables ###
                proxy_set_header country $geoip_city_country_code;
                proxy_set_header region $geoip_region;
                proxy_set_header city $geoip_city;
                proxy_set_header postal $geoip_postal_code;

                proxy_set_header  X-City     $city;
                proxy_set_header  X-Country  $country;
                proxy_set_header  X-Region   $region;

                rewrite ^/pwa.html$ / permanent;
        }

        location /api {
                auth_basic off;
                proxy_buffering off;
                proxy_pass http://openshift-prod;
                proxy_set_header Host $host;
                proxy_set_header X-Forwarded-For $remote_addr;
                proxy_read_timeout 1200;
                proxy_send_timeout 1200;
                proxy_connect_timeout 1200;
        }

        location /img {
                auth_basic off;
                proxy_buffering off;
                proxy_pass http://openshift-prod;
                proxy_set_header Host $host;
                proxy_set_header X-Forwarded-For $remote_addr;
                proxy_read_timeout 1200;
                proxy_send_timeout 1200;
                proxy_connect_timeout 1200;
        }
}
排查&优化方案
  • 先检查OpenShift侧Varnish Pod的资源配额,压测时观察CPU、内存使用率,缓存命中场景下Varnish性能瓶颈基本为CPU,若单Pod CPU使用率超过80%可先调高资源配额,或扩容Pod数量。
  • 检查Varnish前端Service的会话保持配置,若开启了会话保持会导致压测流量全部打到单个Varnish Pod上,另一个Pod无流量承接,关闭会话保持即可让流量均匀分摊到2个Pod。
  • 调整Nginx的proxy_buffering配置,当前全局关闭了代理缓冲,会导致Nginx必须同步将Varnish的响应逐块转发给客户端,高并发下会占用大量连接资源大幅降低吞吐量,可将缓存接口对应的/api、/img等location的proxy_buffering改为on,调整proxy_buffer_size、proxy_buffers参数适配响应体大小。
  • 检查Varnish线程池配置,默认Varnish线程池参数较低,压测场景下极易出现请求排队,可在压测时执行varnishstat -1 MAIN.thread_queue_len查看队列长度,若数值持续大于0,需调高Varnish启动参数的thread_pool_min、thread_pool_max、thread_queue_limit值。
  • 检查Varnish的缓存对象大小限制,若响应体过大超出默认缓存阈值,会导致Varnish无法正常缓存走磁盘IO,可根据实际响应大小调整max_cache_size、storage参数。

内容的提问来源于stack exchange,提问作者Ilya Sulimanov

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.24 14:15:07