You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

XDP eBPF程序性能损耗与丢包问题分析及优化问询

项目背景

我开发了一个基于XDP & eBPF的项目wag,用于在WireGuard VPN上实现基于时间的连接控制。将XDP eBPF程序挂载到WireGuard TUN设备后,出现吞吐量骤降(带eBPF时下行约20Mbps,无eBPF时约100Mbps)、Ping延迟不稳定且每600个ICMP包丢失1个的问题,且该现象在低负载(总流量低于100Mbps)时仍会发生。程序通过Cilium加载到内核,相关代码如下:

Go加载代码

// Kernel load
...
    xdpLink, err = link.AttachXDP(link.XDPOptions{
        Program:   xdpObjects.XdpProgFunc,
        Interface: iface.Index,
    })
...

eBPF内核代码

// +build ignore

#include "bpf_endian.h"
#include "common.h"

char __license[] SEC("license") = "Dual MIT/GPL";

// One /24
#define MAX_MAP_ENTRIES 256

// Inner map is a LPM tri, so we use this as the key
struct ip4_trie_key
{
    __u32 prefixlen; // first member must be u32
    __u32 addr;      // rest can are arbitrary
};

// Map of users (ipv4) to BOOTTIME uint64 timestamp denoting authorization status
struct bpf_map_def SEC("maps") sessions = {
    .type = BPF_MAP_TYPE_HASH,
    .max_entries = MAX_MAP_ENTRIES,
    .key_size = sizeof(__u32),
    .value_size = sizeof(__u64),
    .map_flags = 0,
};

// Map of users (ipv4) to BOOTTIME uint64 timestamp denoting when the last packet was recieved
struct bpf_map_def SEC("maps") last_packet_time = {
    .type = BPF_MAP_TYPE_HASH,
    .max_entries = MAX_MAP_ENTRIES,
    .key_size = sizeof(__u32),
    .value_size = sizeof(__u64),
    .map_flags = 0,
};

// A single variable in nano seconds
struct bpf_map_def SEC("maps") inactivity_timeout_minutes = {
    .type = BPF_MAP_TYPE_ARRAY,
    .max_entries = 1,
    .key_size = sizeof(__u32),
    .value_size = sizeof(__u64),
    .map_flags = 0,
};

// Two tables of the same construction
// IP to LPM trie
struct bpf_map_def SEC("maps") mfa_table = {
    .type = BPF_MAP_TYPE_HASH_OF_MAPS,
    .max_entries = MAX_MAP_ENTRIES,
    .key_size = sizeof(__u32),
    .value_size = sizeof(__u32),
    .map_flags = 0,
};

struct bpf_map_def SEC("maps") public_table = {
    .type = BPF_MAP_TYPE_HASH_OF_MAPS,
    .max_entries = MAX_MAP_ENTRIES,
    .key_size = sizeof(__u32),
    .value_size = sizeof(__u32),
    .map_flags = 0,
};

/*
Attempt to parse the IPv4 source address from the packet.
Returns 0 if there is no IPv4 header field; otherwise returns non-zero.
*/
static int parse_ip_src_dst_addr(struct xdp_md *ctx, __u32 *ip_src_addr, __u32 *ip_dst_addr)
{
    void *data_end = (void *)(long)ctx->data_end;
    void *data = (void *)(long)ctx->data;

    // As this is being attached to a wireguard interface (tun device), we dont get layer 2 frames
    // Just happy little ip packets

    // Then parse the IP header.
    struct iphdr *ip = data;
    if ((void *)(ip + 1) > data_end)
    {
        return 0;
    }

    // We dont support ipv6
    if (ip->version != 4)
    {
        return 0;
    }

    // Return the source IP address in network byte order.
    *ip_src_addr = (__u32)(ip->saddr);
    *ip_dst_addr = (__u32)(ip->daddr);

    return 1;
}

static int conntrack(__u32 *src_ip, __u32 *dst_ip)
{

    // Max lifetime of the session.
    __u64 *session_expiry = bpf_map_lookup_elem(&sessions, src_ip);
    if (!session_expiry)
    {
        return 0;
    }

    // The most recent time a valid packet was received from our a user src_ip
    __u64 *lastpacket = bpf_map_lookup_elem(&last_packet_time, src_ip);
    if (!lastpacket)
    {
        return 0;
    }

    // Our userland defined inactivity timeout
    u32 index = 0;
    __u64 *inactivity_timeout = bpf_map_lookup_elem(&inactivity_timeout_minutes, &index);
    if (!inactivity_timeout)
    {
        return 0;
    }

    __u64 currentTime = bpf_ktime_get_boot_ns();

    // The inner map must be a LPM trie
    struct ip4_trie_key key = {
        .prefixlen = 32,
        .addr = *dst_ip,
    };

    // If the inactivity timeout is not disabled and users session has timed out
    u8 isTimedOut = (*inactivity_timeout != __UINT64_MAX__ && ((currentTime - *lastpacket) >= *inactivity_timeout));

    if (isTimedOut)
    {
        u64 locked = 0;
        bpf_map_update_elem(&sessions, src_ip, &locked, BPF_EXIST);
    }

    // Order of preference is MFA -> Public, just in case someone adds multiple entries for the same route to make sure accidental exposure is less likely
    // If the key is a match for the LPM in the public table
    void *user_restricted_routes = bpf_map_lookup_elem(&mfa_table, src_ip);
    if (user_restricted_routes)
    {

        if (bpf_map_lookup_elem(user_restricted_routes, &key) &&
            // 0 indicates invalid session
            *session_expiry != 0 &&
            // If max session lifetime is disabled, or we are before the max lifetime of the session
            (*session_expiry == __UINT64_MAX__ || *session_expiry > currentTime) &&
            !isTimedOut)
        {

            // Doesnt matter if the value is not atomically set
            *lastpacket = currentTime;

            return 1;
        }
    }

    void *user_public_routes = bpf_map_lookup_elem(&public_table, src_ip);
    if (user_public_routes && bpf_map_lookup_elem(user_public_routes, &key))
    {
        // Only update the lastpacket time if we're not expired
        if (!isTimedOut)
        {
            *lastpacket = currentTime;
        }
        return 1;
    }

    return 0;
}

SEC("xdp")
int xdp_prog_func(struct xdp_md *ctx)
{
    __u32 src_ip, dst_ip;
    if (!parse_ip_src_dst_addr(ctx, &src_ip, &dst_ip))
    {
        return XDP_DROP;
    }

    if (conntrack(&src_ip, &dst_ip) || conntrack(&dst_ip, &src_ip))
    {

        return XDP_PASS;
    }

    return XDP_DROP;
}
技术问题解答

1. 如何分析eBPF程序中哪些部分是性能瓶颈?

  • 用bpftool直接 profiling:执行bpftool prog profile id <prog_id>,它会逐条统计指令的执行耗时、调用次数,直接定位到耗时最多的代码段(比如map查找、条件判断)。
  • 监控map操作统计:用bpftool map stats id <map_id>查看每个map的lookup/update操作次数、平均耗时,重点关注哈希表和嵌套表的命中率——如果命中率低,说明map的键设计或大小不合理。
  • perf追踪eBPF事件:运行perf record -e bpf:* -g,然后用perf report分析程序运行时的内核栈,看是否有频繁的锁等待、内存拷贝等开销。
  • 查看XDP网卡统计:检查/sys/class/net/<wg_iface>/statistics/xdp_*文件,比如xdp_drop、xdp_pass计数,确认是程序处理慢导致队列积压,还是逻辑上的过度丢包。

2. XDP是否存在处理时间限制,或有需遵循的最优时长标准?

XDP运行在网卡驱动的中断上下文,没有硬编码的时间上限,但必须尽可能高效:

  • 中断上下文无法被调度,处理时间过长会占用过多CPU,导致其他中断(比如其他网卡、磁盘)无法及时响应,甚至引发网卡硬件丢包。
  • 行业最优标准是单包处理时间控制在1-2微秒内(针对10Gbps网卡,线速下每秒需处理83万包左右,单包处理时间必须低于1.2微秒才能跟上流量)。如果超过这个范围,必然会出现吞吐量下降、延迟抖动的问题。
  • 内核会验证XDP程序的安全性(比如无无限循环),但不会限制单包处理时长,性能完全取决于代码实现的优化程度。

3. 我的eBPF程序实现是否合理?

当前代码存在多处影响性能的缺陷,是导致吞吐量下降的主要原因:

(1)重复的核心逻辑调用

XDP主函数中对每个包调用两次conntrack(正向、反向各一次),相当于每个包要做双倍的map查找、时间计算、LPM匹配。比如sessions、last_packet_time等map会被查找两次,嵌套的mfa_table、public_table也要被访问两次,直接翻倍了处理开销。

  • 优化:先判断IP是否属于VPN客户端网段,只对客户端IP做正向检查,反向检查只针对服务器端IP,减少不必要的重复计算。

(2)嵌套map的低效访问

mfa_table和public_table是HASH_OF_MAPS结构,每次访问需要先查外层HASH拿到内层LPM的句柄,再查内层LPM。这种双层查找的开销远大于单层map,且每个包可能触发两次。

  • 优化:如果路由规则是按用户分组,可将用户IP和目标IP合并成复合键,用单层LPM或HASH map存储;或者预加载常用用户的内层map到BPF数组中,减少外层HASH的查找次数。

(3)频繁的不必要map更新

当用户会话超时(isTimedOut为真)时,每个包都会执行bpf_map_update_elem(&sessions, src_ip, &locked, BPF_EXIST)。这会导致每个超时用户的包都触发一次map写操作,增加内核的锁竞争和内存开销。

  • 优化:记录用户的超时状态(比如在last_packet_time或单独的map中标记),只在从非超时状态变为超时状态时更新sessions map,避免重复写操作。

(4)冗余的时间获取

每次conntrack调用都会执行bpf_ktime_get_boot_ns(),同一个包的两次conntrack调用重复获取时间,增加了不必要的系统调用开销。

  • 优化:在XDP主函数中获取一次当前时间,作为参数传递给conntrack函数复用。

(5)IPv6处理不合理

代码中直接丢弃IPv6包,但TUN设备可能存在IPv6流量,这不仅会导致不必要的丢包,还会让XDP程序做无用的解析判断。

  • 优化:如果不支持IPv6,直接返回XDP_PASS让内核处理,避免XDP程序消耗资源处理不支持的协议。

内容的提问来源于stack exchange,提问作者NHAS

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 13:55:31