You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Varnish 4持续向异常后端发送请求问题排查求助

Varnish 4 后端轮询调度故障排查

已配置轮询调度的两个后端Server1和Server2,Server1因Apache停止运行被健康检查标记为异常,Server2状态正常,但Varnish仍持续将请求发送至异常的Server1,导致请求返回503后端获取失败。仅当移除Server1实例后,Varnish才会将流量路由至健康的Server2;新增后端后也仅会向单个后端发送请求,该后端异常时同样出现503错误。推测问题出在后端配置中,但无法定位具体原因。

主VCL配置

vcl 4.0;

import std;
import directors;

include "backends.vcl";

sub vcl_init {
    call backends_init;
}

sub vcl_recv {
    set req.http.Host = regsub(req.http.Host, ":[0-9]+", "");
    unset req.http.proxy;
    set req.url = std.querysort(req.url);
    set req.url = regsub(req.url, "\?$", "");
    set req.http.Surrogate-Capability = "key=ESI/1.0";

    if (std.healthy(req.backend_hint)) {
        #set req.grace = 10s;
    }

    if (!req.http.X-Forwarded-Proto) {
        if(std.port(server.ip) == 443 || std.port(server.ip) == 8443) {
            set req.http.X-Forwarded-Proto = "https";
        } else {
            set req.http.X-Forwarded-Proto = "https";
        }
    }

    if (req.http.Upgrade ~ "(?i)websocket") {
        return (pipe);
    }

    if (req.url ~ "(\?|&)(utm_source|utm_medium|utm_campaign|utm_content|gclid|cx|ie|cof|siteurl)=") {
        set req.url = regsuball(req.url, "&(utm_source|utm_medium|utm_campaign|utm_content|gclid|cx|ie|cof|siteurl)=([A-z0-9_\-\.%25]+)", "");
        set req.url = regsuball(req.url, "\?(utm_source|utm_medium|utm_campaign|utm_content|gclid|cx|ie|cof|siteurl)=([A-z0-9_\-\.%25]+)", "?");
        set req.url = regsub(req.url, "\?&", "?");
        set req.url = regsub(req.url, "\?$", "");
    }

    if (req.method == "PURGE") {
        if (!client.ip ~ purge) {
            return (synth(405, "Cannot purge cache from here"));
        }
        return (purge);
    }

    if (req.method != "GET" &&
        req.method != "HEAD" &&
        req.method != "PUT" &&
        req.method != "POST" &&
        req.method != "TRACE" &&
        req.method != "OPTIONS" &&
        req.method != "PATCH" &&
        req.method != "DELETE") {
        return (pipe);
    }

    if (req.method != "GET" && req.method != "HEAD") {
        return (pass);
    }

    if (req.url ~ "^[^?]*\.(7z|avi|bmp|bz2|css|csv|doc|docx|eot|flac|flv|gif|gz|ico|jpeg|jpg|js|less|mka|mkv|mov|mp3|mp4|mpeg|mpg|odt|ogg|ogm|opus|otf|pdf|png|ppt|pptx|rar|rtf|svg|svgz|swf|tar|tbz|tgz|ttf|txt|txz|wav|webm|webp|woff|woff2|xls|xlsx|xml|xz|zip)(\?.*)?$") {
        unset req.http.Cookie;
        return(hash);
    }

    # Remove all cookies except for PHP session cookie
    if (req.http.Cookie) {
        set req.http.Cookie = ";" + req.http.Cookie;
        set req.http.Cookie = regsuball(req.http.Cookie, "; +", ";");
        set req.http.Cookie = regsuball(req.http.Cookie, ";(PHPSESSID|SSESS[^=]*)=", "; \1=");
        set req.http.Cookie = regsuball(req.http.Cookie, ";[^ ][^;]*", "");
        set req.http.Cookie = regsuball(req.http.Cookie, "^[; ]+|[; ]+$", "");
 
        if (req.http.Cookie == "") {
            unset req.http.Cookie;
        }
    }

    if (req.http.cookie ~ "^\s*$") {
        unset req.http.cookie;
    }
}

sub vcl_hash {
    hash_data(req.http.X-Forwarded-Proto);
}

sub vcl_backend_response {
    if (bereq.url ~ "^[^?]*\.(7z|avi|bmp|bz2|css|csv|doc|docx|eot|flac|flv|gif|gz|ico|jpeg|jpg|js|less|mka|mkv|mov|mp3|mp4|mpeg|mpg|odt|ogg|ogm|opus|otf|pdf|png|ppt|pptx|rar|rtf|svg|svgz|swf|tar|tbz|tgz|ttf|txt|txz|wav|webm|webp|woff|woff2|xls|xlsx|xml|xz|zip)(\?.*)?$") {
        unset beresp.http.Set-Cookie;
        set beresp.ttl = 1d;
    }

    if (beresp.http.Surrogate-Control ~ "ESI/1.0") {
        unset beresp.http.Surrogate-Control;
        set beresp.do_esi = true;
    }

    set beresp.grace = 6h;
}

后端VCL配置

backend Server1 {
        # Instance: i-XXXXXXXXXXXXXXXXXX
        .host = "10.135.49.20";

        .port = "80";
        .max_connections = 300; # That's it
        .probe = {
                #.url = "/"; # short easy way (GET /)
                # We prefer to only do a HEAD /
                .request =
                        "HEAD / HTTP/1.1"
                        "Host: localhost"
                        "Connection: close"
                        "User-Agent: Varnish Health Probe";
                .interval = 10s; # check the health of each backend every 5 seconds
                .timeout = 5s; # timing out after 1 second.
                # If 3 out of the last 5 polls succeeded the backend is considered healthy, otherwise it will be marked as sick
                .window = 5;
                .threshold = 3;
                }
        .first_byte_timeout     = 90s;   # How long to wait before we receive a first byte from our backend?
        .connect_timeout        = 5s;    # How long to wait for a backend connection?
        .between_bytes_timeout  = 2s;    # How long to wait between bytes received from our backend?


}
backend Server2 {
        # Instance: i-XXXXXXXXXXXXXX
        .host = "10.135.49.137";

        .port = "80";
        .max_connections = 300; # That's it
        .probe = {
                #.url = "/"; # short easy way (GET /)
                # We prefer to only do a HEAD /
                .request =
                        "HEAD / HTTP/1.1"
                        "Host: localhost"
                        "Connection: close"
                        "User-Agent: Varnish Health Probe";
                .interval = 10s; # check the health of each backend every 5 seconds
                .timeout = 5s; # timing out after 1 second.
                # If 3 out of the last 5 polls succeeded the backend is considered healthy, otherwise it will be marked as sick
                .window = 5;
                .threshold = 3;
                }
        .first_byte_timeout     = 90s;   # How long to wait before we receive a first byte from our backend?
        .connect_timeout        = 5s;    # How long to wait for a backend connection?
        .between_bytes_timeout  = 2s;    # How long to wait between bytes received from our backend?


}


sub backends_init {
        new vdir = directors.round_robin();

        vdir.add_backend(Server1);
        vdir.add_backend(Server2);
}

acl purge {
        "localhost";
        "10.135.49.20";
        "10.135.49.137";
        "10.135.48.160";
}

后端健康检查日志

varnishlog -g raw -i backend_health
         0 Backend_health - Server1 Still sick ------- 0 3 5 0.000000 0.002185
         0 Backend_health - Server2 Still healthy 4--X-RH 5 3 5 0.001479 0.001719 HTTP/1.1 200 OK
         0 Backend_health - Server1 Still sick ------- 0 3 5 0.000000 0.002185
         0 Backend_health - Server2 Still healthy 4--X-RH 5 3 5 0.001306 0.001616 HTTP/1.1 200 OK

问题定位与修复

核心问题

虽然在backends_init中创建了轮询调度器vdir并添加了两个后端,但没有在vcl_recv中指定请求使用该调度器。Varnish不会自动使用创建的调度器,必须显式将req.backend_hint指向调度器的后端选择方法,否则请求会默认路由到某个固定后端,且不会触发调度器的健康检查筛选逻辑。

修复步骤

在vcl_recv函数的合适位置(比如处理完Host头和URL清理之后)添加以下代码:

set req.backend_hint = vdir.backend();

修改后的vcl_recv开头示例:

sub vcl_recv {
    set req.http.Host = regsub(req.http.Host, ":[0-9]+", "");
    unset req.http.proxy;
    set req.url = std.querysort(req.url);
    set req.url = regsub(req.url, "\?$", "");
    set req.http.Surrogate-Capability = "key=ESI/1.0";

    # 指定使用轮询调度器
    set req.backend_hint = vdir.backend();

    if (std.healthy(req.backend_hint)) {
        #set req.grace = 10s;
    }
    // 后续代码不变
}

原理说明

轮询调度器directors.round_robin()会自动跳过标记为不健康的后端,仅将请求分发到健康的后端实例。通过set req.backend_hint = vdir.backend(),让Varnish每次请求都通过调度器选择后端,而非直接使用单个后端定义。

内容的提问来源于stack exchange,提问作者grahamjgreen

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.25 18:27:01