基于eBPF Kprobe挂钩tcp_sendmsg嗅探TCP数据包负载异常
问题:通过eBPF捕获Redis出站TCP包负载失败
我尝试用eBPF读取redis-cli发往远程Redis数据库的出站TCP数据包负载,测试eBPF的能力。当前方案是挂钩tcp_sendmsg(),通过第一个参数sock* sk过滤目标端口为19795的包,再从第二个参数msghdr的msg_iter.iov获取iov_base提取负载,但复制的数据损坏,与Wireshark捕获的真实负载不符。测试环境为WSL2上的Ubuntu 22.04,内核版本5.15.153.1-microsoft-standard-WSL2,已尝试XDP但因无关问题失败。
当前代码如下:
struct redis_sys_packet { u16 lport; u16 dport; u32 nr_segments; u32 packet_length; char packet[128]; }; struct { __uint(type, BPF_MAP_TYPE_HASH); __uint(max_entries, 1024); __type(key, struct redis_sys_packet); __type(value, u64); } redis_sys_write_calls SEC(".maps"); //... other code // kprobe handler for tcp_sendmsg SEC("kprobe/tcp_sendmsg") int kprobe__tcp_sendmsg(struct pt_regs *ctx) { if (!ctx || ctx == NULL) return 0; struct sock *sk = (struct sock *)(ctx->di); if (!sk || sk == NULL) return 0; u16 lport = 0; u16 dport = 0; bpf_probe_read_kernel(&lport, sizeof(lport), &sk->__sk_common.skc_num); bpf_probe_read_kernel(&dport, sizeof(dport), &sk->__sk_common.skc_dport); struct redis_sys_packet pack = {}; pack.lport = lport; pack.dport = bpf_ntohs(dport); if (pack.dport != 19795) return 0; // filter by port // second parameter struct msghdr* msg = (struct msghdr *)(ctx->si); if (!msg || msg == NULL) return 0; // Get the pointer to the iov (IO vector) array that holds the payload struct iov_iter *iter = &(msg->msg_iter); if (!iter || iter == NULL) return 0; bpf_probe_read(&pack.nr_segments, sizeof(unsigned long), &(iter->nr_segs)); const struct iovec *iov = (const struct iovec *)&(iter->iov); // Check if we can access the first iovec if (!iov) return 0; // Get the base address of the payload void *iovbase; bpf_probe_read(&iovbase, sizeof(void *), &iov[0].iov_base); unsigned long iov_len; // read from count instead of iov->iov_len - for some reason this always returns 1, not the true length bpf_probe_read(&iov_len, sizeof(size_t), &(iter->count)); pack.packet_length = (u32)iov_len; u32 copy_length = (u32)iov_len; if (copy_length > 128) { copy_length = 128; } else if (copy_length <= 0) { return 0; } bpf_probe_read_kernel(pack.packet, copy_length, iovbase); increment_map(&redis_sys_write_calls, &pack, 1); // returns irrelevant data return 0; }
问题分析与代码修正
1. 寄存器参数错误
x86_64架构下,kprobe遵循System V AMD64 ABI,前6个参数依次存放在rdi、rsi等64位寄存器中,你代码里用32位寄存器di、si获取参数,会导致指针截断,读取到错误的内核地址。
2. iov_iter结构访问错误
内核中iov_iter的iov并非直接存储iovec数组,且msghdr和iovec通常来自用户空间,直接强转指针会导致地址无效,需用bpf_probe_read_user读取用户空间的iovec内容。
3. 数据读取方向错误
iov_base是用户空间地址,不能用bpf_probe_read_kernel读取,必须用bpf_probe_read_user(或显式指定用户空间的bpf_probe_read)。
修正后的核心代码片段
// kprobe handler for tcp_sendmsg SEC("kprobe/tcp_sendmsg") int kprobe__tcp_sendmsg(struct pt_regs *ctx) { if (!ctx) return 0; // x86_64架构下用rdi获取第一个参数sock* sk struct sock *sk = (struct sock *)(ctx->rdi); if (!sk) return 0; u16 lport = 0; u16 dport = 0; bpf_probe_read_kernel(&lport, sizeof(lport), &sk->__sk_common.skc_num); bpf_probe_read_kernel(&dport, sizeof(dport), &sk->__sk_common.skc_dport); struct redis_sys_packet pack = {}; pack.lport = lport; pack.dport = bpf_ntohs(dport); if (pack.dport != 19795) return 0; // x86_64架构下用rsi获取第二个参数struct msghdr* msg struct msghdr* msg = (struct msghdr *)(ctx->rsi); if (!msg) return 0; struct iovec iov; // 读取用户空间的第一个iovec(msg->msg_iov指向iovec数组) if (bpf_probe_read_user(&iov, sizeof(iov), msg->msg_iov) != 0) { return 0; } unsigned long iov_len = iov.iov_len; pack.packet_length = (u32)iov_len; u32 copy_length = (u32)iov_len; if (copy_length > 128) { copy_length = 128; } else if (copy_length <= 0) { return 0; } // 从用户空间读取payload bpf_probe_read_user(pack.packet, copy_length, iov.iov_base); increment_map(&redis_sys_write_calls, &pack, 1); return 0; }
其他可选跟踪点/函数
tcp_sendmsg返回探针(kretprobe):可以验证实际发送的字节数,确认捕获的负载长度是否准确。sys_sendto/sys_sendmsg系统调用:挂钩用户态到内核态的入口,直接获取用户空间的发送数据,适合跟踪所有出站TCP/UDP包,需额外过滤TCP协议和目标端口。sk_msg_send:内核底层发送函数,适合跟踪套接字发送的所有数据,但结构依赖内核版本。- Uprobe挂钩redis-cli的
write/send函数:直接在用户态程序的发送点捕获数据,无需处理内核复杂结构,兼容性更好。
内容的提问来源于stack exchange,提问作者justsobad
相关产品推荐
相关产品推荐

