Linux eBPF 深度实战:从系统调用追踪到网络流量过滤
一、eBPF 是什么,为什么重要
eBPF(Extended Berkeley Packet Filter)是 Linux 内核中一项革命性的技术,它允许用户在不修改内核源码、不加载内核模块的情况下,在内核空间中运行自定义的沙箱程序。自 Linux 3.18 引入以来,eBPF 已经从一个简单的数据包过滤器,发展成为现代云原生基础设施的核心引擎。
Cilium、Falco、Tetragon、Pixie 等知名项目全都建立在 eBPF 之上。Facebook 用它做负载均衡(Katran),Google 用它做网络安全策略(GKE Dataplane v2),Netflix 用它做性能分析。eBPF 正在重新定义我们观测、保护和优化 Linux 系统的方式。
二、eBPF 核心架构
2.1 程序生命周期
一个 eBPF 程序的完整生命周期如下:
- 用 C(或 Rust)编写 eBPF 程序源码
- 通过 LLVM/Clang 编译为 eBPF 字节码
- 使用
bpf()系统调用加载到内核 - 内核 Verifier 进行安全验证(确保无死循环、无越界访问)
- JIT 编译器将字节码翻译为原生机器码
- 程序附着(attach)到钩子点(kprobe/tracepoint/XDP 等)
- 事件触发时执行,通过 Maps 与用户空间交换数据
2.2 eBPF Maps — 内核与用户空间的桥梁
Maps 是 eBPF 程序存储和检索数据的核心数据结构类型:
- Hash Map:键值对存储,适合计数器、连接状态表
- Array Map:固定大小的数组,适合配置和查找表
- Ring Buffer:高性能环形缓冲区,适合向用户空间流式传输事件(推荐替代 perf buffer)
- LRU Hash:自动淘汰最近最少使用的条目,适合缓存场景
- Per-CPU Maps:每 CPU 独立副本,避免锁竞争,适合高并发统计
2.3 Helper 函数
eBPF 程序不能随意调用内核函数,只能通过预定义的 Helper:
bpf_probe_read():安全读取内核内存bpf_map_lookup_elem() / bpf_map_update_elem():操作 Mapsbpf_perf_event_output() / bpf_ringbuf_output():向用户空间输出数据bpf_get_current_pid_tgid() / bpf_get_current_comm():获取当前进程信息bpf_ktime_get_ns():获取高精度时间戳
三、实战一:用 BCC 追踪系统调用延迟
3.1 安装 BCC 工具
# Ubuntu/Debian
sudo apt install bpfcc-tools linux-headers-$(uname -r)
# 或者通过 Python pip
sudo pip install bcc
3.2 使用 funclatency 测量函数延迟
BCC 提供了大量开箱即用的工具。比如测量 ext4_file_read_iter() 函数的执行延迟:
# 跟踪单个函数,输出直方图
sudo funclatency-bpfcc ext4_file_read_iter
# 跟踪带 PID 过滤的延迟
sudo funclatency-bpfcc -p 1234 ext4_file_read_iter
# 按进程聚合,每 2 秒输出一次
sudo funclatency-bpfcc -i 2 -p $(pidof mysqld) ext4_file_read_iter
输出示例:
Tracing 1 functions for "ext4_file_read_iter"... Hit Ctrl-C to end.
usecs : count distribution
0 -> 1 : 0 | |
2 -> 3 : 0 | |
4 -> 7 : 1 | |
8 -> 15 : 3 |* |
16 -> 31 : 12 |***** |
32 -> 63 : 45 |******************** |
64 -> 127 : 78 |************************************* |
128 -> 255 : 92 |****************************************|
256 -> 511 : 35 |*************** |
512 -> 1023 : 8 |*** |
1024 -> 2047 : 2 |* |
3.3 自定义 BCC Python 脚本统计 read() 系统调用延迟
#!/usr/bin/env python3
from bcc import BPF
import ctypes
# eBPF C 程序
bpf_text = """
#include <uapi/linux/ptrace.h>
#include <linux/fs.h>
// 用于存储每个请求的开始时间
BPF_HASH(start, u32, u64);
// 直方图,用户空间用来输出
BPF_HISTOGRAM(dist, u64);
// 进入 read 系统调用时记录时间戳
int trace_read_entry(struct pt_regs *ctx) {
u32 pid = bpf_get_current_pid_tgid() >> 32;
u64 ts = bpf_ktime_get_ns();
start.update(&pid, &ts);
return 0;
}
// 退出 read 系统调用时计算延迟
int trace_read_return(struct pt_regs *ctx) {
u32 pid = bpf_get_current_pid_tgid() >> 32;
u64 *tsp = start.lookup(&pid);
if (tsp == 0)
return 0;
u64 delta = bpf_ktime_get_ns() - *tsp;
// 转换为微秒
delta /= 1000;
// 按 2 的幂次分桶
u64 log2 = bpf_log2l(delta);
dist.increment(log2);
start.delete(&pid);
return 0;
}
"""
b = BPF(text=bpf_text)
b.attach_kprobe(event=b.get_syscall_fnname("read"), fn_name="trace_read_entry")
b.attach_kretprobe(event=b.get_syscall_fnname("read"), fn_name="trace_read_return")
print("Tracing read() latency... Ctrl-C to stop")
try:
sleep(10)
except KeyboardInterrupt:
pass
b["dist"].print_log2_hist("usecs")
四、实战二:用 BPFTrace 一行命令解决疑难杂症
BPFTrace 是基于 eBPF 的高级追踪语言,语法类似 awk,非常适合快速诊断。
4.1 跟踪所有 openat 调用,显示进程和文件名
sudo bpftrace -e 'tracepoint:syscalls:sys_enter_openat { printf("%s %s\n", comm, str(args->filename)); }'
4.2 统计每个进程的 read 数据量
sudo bpftrace -e 'tracepoint:syscalls:sys_exit_read /args->ret > 0/ { @bytes[comm] = sum(args->ret); }'
输出:
@bytes[chrome]: 2849321
@bytes[mysqld]: 10485760
@bytes[nginx]: 5242880
4.3 跟踪 TCP 连接建立延迟
sudo bpftrace -e '
kprobe:tcp_v4_connect {
@start[tid] = nsecs;
}
kretprobe:tcp_v4_connect /@start[tid]/ {
$lat = (nsecs - @start[tid]) / 1000;
@lat_us = hist($lat);
delete(@start[tid]);
}
'
五、实战三:XDP 高性能网络包过滤
5.1 什么是 XDP
XDP(eXpress Data Path)允许 eBPF 程序在网络卡驱动层直接处理数据包,甚至在数据包到达内核网络协议栈之前就能决定丢弃或转发。这是目前 Linux 内核中性能最高的包处理路径,单机可达 24M pps(每秒数据包数)。
5.2 编写 XDP 程序丢弃特定 IP 的流量
// xdp_drop_kern.c
#include <linux/bpf.h>
#include <linux/if_ether.h>
#include <linux/ip.h>
#include <linux/in.h>
#include "bpf_helpers.h"
SEC("xdp_drop")
int xdp_drop_prog(struct xdp_md *ctx) {
void *data_end = (void *)(long)ctx->data_end;
void *data = (void *)(long)ctx->data;
struct ethhdr *eth = data;
if ((void *)(eth + 1) > data_end)
return XDP_PASS;
// 只处理 IPv4
if (eth->h_proto != __constant_htons(ETH_P_IP))
return XDP_PASS;
struct iphdr *ip = (void *)(eth + 1);
if ((void *)(ip + 1) > data_end)
return XDP_PASS;
// 丢弃来自 10.0.0.99 的数据包
if (ip->saddr == __constant_htonl(0x0A000064)) // 10.0.0.99
return XDP_DROP;
return XDP_PASS;
}
char _license[] SEC("license") = "GPL";
5.3 加载和附着 XDP 程序
使用 iproute2 的 ip 命令直接加载:
# 编译 eBPF 字节码
clang -O2 -target bpf -c xdp_drop_kern.c -o xdp_drop_kern.o
# 加载到网卡(需要 root)
sudo ip link set dev eth0 xdp obj xdp_drop_kern.o sec xdp_drop
# 查看 XDP 程序是否已加载
sudo ip link show dev eth0
# 卸载 XDP 程序
sudo ip link set dev eth0 xdp off
六、实战四:用 libbpf 编写原生 eBPF 应用
6.1 项目结构
my_ebpf_app/
├── Makefile
├── my.bpf.c # eBPF 内核态程序
├── my.bpf.h # 头文件
├── my.c # 用户态加载程序
└── vmlinux.h # 从 bpftool 生成的内核类型定义
6.2 生成 vmlinux.h
sudo bpftool btf dump file /sys/kernel/btf/vmlinux format c > ./vmlinux.h
6.3 完整的 uprobe 示例:追踪 Go 程序中的函数调用
// trace_func.bpf.c
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
// 定义 Ring Buffer Map
struct {
__uint(type, BPF_MAP_TYPE_RINGBUF);
__uint(max_entries, 256 * 1024); // 256KB
} events SEC(".maps");
// 事件数据结构
struct event {
u32 pid;
u64 timestamp;
char comm[16];
u64 arg0;
u64 arg1;
};
SEC("uprobe//usr/bin/myapp:main.myFunc")
int trace_my_func(struct pt_regs *ctx) {
struct event *e;
// 从 Ring Buffer 中预留空间
e = bpf_ringbuf_reserve(&events, sizeof(*e), 0);
if (!e)
return 0;
// 填充事件数据
e->pid = bpf_get_current_pid_tgid() >> 32;
e->timestamp = bpf_ktime_get_ns();
bpf_get_current_comm(&e->comm, sizeof(e->comm));
// 读取函数参数(x86_64 ABI: rdi, rsi)
bpf_probe_read(&e->arg0, sizeof(u64), (void *)ctx->di);
bpf_probe_read(&e->arg1, sizeof(u64), (void *)ctx->si);
// 提交事件到用户空间
bpf_ringbuf_submit(e, 0);
return 0;
}
char LICENSE[] SEC("license") = "GPL";
6.4 用户态加载代码
// loader.c
#include <stdio.h>
#include <unistd.h>
#include <signal.h>
#include <bpf/libbpf.h>
#include "trace_func.skel.h"
static volatile bool running = true;
void sig_handler(int sig) {
running = false;
}
int handle_event(void *ctx, void *data, size_t data_sz) {
struct event *e = data;
printf("[%llu] PID=%d COMM=%s arg0=%llu arg1=%llu\n",
e->timestamp, e->pid, e->comm, e->arg0, e->arg1);
return 0;
}
int main(int argc, char **argv) {
struct trace_func_bpf *skel;
struct ring_buffer *rb = NULL;
int err;
signal(SIGINT, sig_handler);
// 打开并加载 eBPF skeleton
skel = trace_func_bpf__open_and_load();
if (!skel) {
fprintf(stderr, "Failed to open BPF skeleton\n");
return 1;
}
// Attach uprobe
err = trace_func_bpf__attach(skel);
if (err) {
fprintf(stderr, "Failed to attach BPF skeleton\n");
goto cleanup;
}
// 设置 Ring Buffer 轮询回调
rb = ring_buffer__new(bpf_map__fd(skel->maps.events), handle_event, NULL, NULL);
if (!rb) {
fprintf(stderr, "Failed to create ring buffer\n");
goto cleanup;
}
printf("Successfully started! Press Ctrl-C to stop.\n");
while (running) {
err = ring_buffer__poll(rb, 100 /* timeout_ms */);
if (err == -EINTR) break;
if (err < 0) {
fprintf(stderr, "Error polling ring buffer: %d\n", err);
break;
}
}
cleanup:
ring_buffer__free(rb);
trace_func_bpf__destroy(skel);
return 0;
}
七、eBPF 的局限与注意事项
虽然 eBPF 功能强大,但也有需要注意的限制:
- Verifier 限制:循环必须有界(Linux 5.3 前完全禁止循环),栈空间仅 512 字节,程序指令数上限 100 万(Linux 5.2+)
- 无动态内存分配:eBPF 程序不能调用 malloc,所有 Maps 必须预定义大小
- 版本兼容性:eBPF CO-RE(Compile Once, Run Everywhere)需要较新的内核和 BTF 支持
- 安全性:虽然 Verifier 确保程序不会崩溃内核,但恶意使用 eBPF 可以进行隐蔽追踪(root 权限才能加载)
- 调试困难:Verifier 报错信息不够友好
八、总结
eBPF 技术正在深刻改变 Linux 系统的可观测性、安全性和网络能力。从一行命令的 BPFTrace 诊断,到百行代码的 XDP 防火墙,再到 Cilium 构建的整个容器网络平台,eBPF 覆盖了从快速原型到生产系统的全部场景。
如果你是一名后端工程师、SRE 或安全工程师,掌握 eBPF 将让你对 Linux 系统的洞察力提升一个数量级。建议从 BPFTrace 一行命令开始,逐步深入到 BCC 脚本,最后用 libbpf 编写完整应用。这条学习曲线绝对值得投入。
核心资源:ebpf.io(官方门户)、iovisor/bcc(BCC GitHub)、Brendan Gregg 的《BPF Performance Tools》。

发表评论 取消回复