深入理解 eBPF:从内核可观测性到网络加速实战
引言
eBPF(Extended Berkeley Packet Filter)是现代 Linux 内核中最具革命性的技术之一。它允许用户在不修改内核源码、不加载内核模块的情况下,在内核空间中安全地运行自定义程序。从网络加速到安全监控,从性能追踪到可观测性,eBPF 正在重新定义我们与操作系统内核交互的方式。
本文将深入探讨 eBPF 的核心架构、编程模型、关键用例,并配合大量实战示例,帮助读者全面掌握这项技术。
1. eBPF 架构概览
1.1 从 BPF 到 eBPF 的演进
Berkeley Packet Filter 最初由 Steven McCanne 和 Van Jacobson 在 1992 年设计,用于高效网络包过滤(tcpdump 的前身)。eBPF 在 2014 年由 Alexei Starovoitov 引入 Linux 3.18,将这一概念扩展为通用的内核虚拟机。
关键演进里程碑:
- Linux 3.18 (2014):首次引入 eBPF,支持用户态挂载点
- Linux 4.1 (2015):tracepoint/kprobe 支持,开启可观测性时代
- Linux 4.7 (2016):XDP 支持诞生,网络加速成为可能
- Linux 4.15 (2018):BPF Type Format (BTF),可移植 CO-RE
- Linux 5.13+ (2021):BPF 中断上下文扩展,性能再提升
1.2 核心组件
eBPF 的架构可以分为四大核心组件:
用户态程序
│
▼
bpf() 系统调用
│
├──▶ BPF Verifier (静态分析确保安全)
│ │
│ ▼
│ JIT Compiler (编译为原生指令)
│ │
│ ▼
▼ ┌─────────────┐
BPF ◀──│ 内核执行 │
虚拟机 │ │
│ kprobe │──▶ 动/静态追踪
│ tracepoint │──▶ 内核事件采样
│ XDP/tc │──▶ 网络数据包处理
│ uprobe │──▶ 用户态函数追踪
│ LSM │──▶ 安全策略控制
└─────────────┘
│
▼
BPF Maps (共享数据结构)
│
▼
用户态读取/写入
1.3 BPF 虚拟机与寄存器模型
eBPF 虚拟机是 RISC 架构,包含 11 个 64 位通用寄存器:
R0 - 函数返回值 / 程序退出值
R1 - R5: 函数参数(最多5个参数通过寄存器传递)
R6 - R9: 被调用者保存的寄存器
R10 - 帧指针(只读),指向栈顶
eBPF 指令为 64 位定长格式,通过 JIT 编译器直接翻译为 x86/ARM 原生指令,执行效率接近原生内核代码。
2. 第一个 eBPF 程序:Hello World
2.1 使用 BCC 框架快速上手
BCC(BPF Compiler Collection)是最流行的 eBPF 开发框架,封装了 LLVM/Clang 编译链和 Python 用户态绑定:
#!/usr/bin/env python3
from bcc import BPF
# 定义 eBPF C 程序
bpf_text = u"""
#include <uapi/linux/ptrace.h>
BPF_ARRAY(call_count, u64, 1);
int trace_execve(struct pt_regs *ctx) {
u32 key = 0;
u64 *count, init_val = 1;
count = call_count.lookup_or_try_init(&key, &init_val);
if (count) {
lock_xadd(count, 1);
}
bpf_trace_printk("execve called, total=%llu\\n", *count);
return 0;
}
"""
# 加载 BPF 程序并挂载到 syscall
b = BPF(text=bpf_text)
b.attach_kprobe(event=b.get_syscall_fnname("execve"), fn_name="trace_execve")
print("开始追踪 execve 系统调用... Ctrl+C 退出")
try:
b.trace_print()
except KeyboardInterrupt:
pass
count = b["call_count"]
print(f"\n总计 execve 调用次数: {count[0].value}")
3. BPF Maps:内核与用户态的桥梁
BPF Maps 是 eBPF 程序与用户空间之间共享数据的关键机制,支持多种数据结构类型:
/* BPF_MAP_TYPE_HASH: 键值对哈希表 */
struct {
__uint(type, BPF_MAP_TYPE_HASH);
__uint(max_entries, 10240);
__type(key, u32);
__type(value, u64);
} exec_duration SEC(".maps");
/* BPF_MAP_TYPE_PERF_EVENT_ARRAY: 高性能事件流 */
struct {
__uint(type, BPF_MAP_TYPE_PERF_EVENT_ARRAY);
__uint(key_size, sizeof(u32));
__uint(value_size, sizeof(u32));
} events SEC(".maps");
/* BPF_MAP_TYPE_RINGBUF: 环形缓冲区 */
struct {
__uint(type, BPF_MAP_TYPE_RINGBUF);
__uint(max_entries, 256 * 1024);
} rb SEC(".maps");
/* BPF_MAP_TYPE_LPM_TRIE: 最长前缀匹配 */
struct {
__uint(type, BPF_MAP_TYPE_LPM_TRIE);
__uint(max_entries, 10000);
__uint(key_size, sizeof(struct bpf_lpm_trie_key) + 4);
__type(value, u32);
} ip_rules SEC(".maps");
/* BPF_MAP_TYPE_STACK_TRACE: 内核栈追踪 */
struct {
__uint(type, BPF_MAP_TYPE_STACK_TRACE);
__uint(max_entries, 1024);
__uint(key_size, sizeof(u32));
__uint(value_size, 127 * sizeof(u64));
} stacks SEC(".maps");
4. 可观测性实战
4.1 追踪内核函数执行延迟
使用 kprobe/kretprobe 挂载到任意内核函数,精确测量执行耗时:
// 追踪 ext4 文件系统 read/write 延迟分布
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
#include <bpf/bpf_core_read.h>
#define MAX_SLOTS 36
struct hist {
u32 slots[MAX_SLOTS];
};
BPF_HASH(start, u64, u64);
BPF_HISTOGRAM(dist, struct hist);
SEC("kprobe/ext4_file_read_iter")
int trace_read_start(struct pt_regs *ctx) {
u64 pid_tgid = bpf_get_current_pid_tgid();
u64 ts = bpf_ktime_get_ns();
start.update(&pid_tgid, &ts);
return 0;
}
SEC("kretprobe/ext4_file_read_iter")
int trace_read_done(struct pt_regs *ctx) {
u64 pid_tgid = bpf_get_current_pid_tgid();
u64 *tsp = start.lookup(&pid_tgid);
if (!tsp) return 0;
u64 delta_us = (bpf_ktime_get_ns() - *tsp) / 1000;
s64 slot = bpf_log2l(delta_us);
if (slot >= 0 && slot < MAX_SLOTS) {
struct hist init = {}, *histp;
histp = dist.lookup_or_try_init(&init, &init);
if (histp) {
__sync_fetch_and_add(&histp->slots[slot], 1);
}
}
start.delete(&pid_tgid);
return 0;
}
char _license[] SEC("license") = "GPL";
4.2 追踪 TCP 连接生命周期
tracepoint 比 kprobe 更稳定,不会随内核版本变化:
// 追踪 TCP 连接建立与关闭
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
#include <bpf/bpf_core_read.h>
struct event {
u32 pid;
u32 saddr, daddr;
u16 sport, dport;
u64 duration_ns;
u8 state;
};
BPF_RINGBUF_OUTPUT(events, 4096);
struct sock_info {
u64 ts;
struct sock *sk;
};
BPF_HASH(socks, u64, struct sock_info);
SEC("tracepoint/sock/inet_sock_set_state")
int tracepoint_sock_state(struct trace_event_raw_inet_sock_set_state *ctx) {
if (ctx->protocol != IPPROTO_TCP) return 0;
u64 pid_tgid = bpf_get_current_pid_tgid();
if (ctx->newstate == TCP_ESTABLISHED && ctx->oldstate == TCP_SYN_SENT) {
u64 pid = pid_tgid >> 32;
struct sock_info info = {};
info.ts = bpf_ktime_get_ns();
socks.update(&pid, &info);
struct event *e = events.ringbuf_reserve(sizeof(struct event));
if (e) {
e->state = 1;
e->pid = pid;
events.ringbuf_submit(e, 0);
}
} else if (ctx->newstate == TCP_CLOSE) {
u64 pid = pid_tgid >> 32;
struct sock_info *info = socks.lookup(&pid);
if (info) {
struct event *e = events.ringbuf_reserve(sizeof(struct event));
if (e) {
e->duration_ns = bpf_ktime_get_ns() - info->ts;
e->state = 0;
e->pid = pid;
events.ringbuf_submit(e, 0);
}
socks.delete(&pid);
}
}
return 0;
}
char _license[] SEC("license") = "GPL";
4.3 Uprobe:追踪用户态函数
eBPF 可以挂载到用户态进程的任意函数:
#!/usr/bin/env python3
from bcc import BPF
bpf_text = u"""
#include <uapi/linux/ptrace.h>
struct latency_entry {
u64 duration_us;
u32 pid;
};
BPF_HASH(start, u64, u64);
BPF_PERF_OUTPUT(latencies);
int trace_set_start(struct pt_regs *ctx) {
u64 pid_tgid = bpf_get_current_pid_tgid();
u64 ts = bpf_ktime_get_ns() / 1000;
start.update(&pid_tgid, &ts);
return 0;
}
int trace_set_return(struct pt_regs *ctx) {
u64 pid_tgid = bpf_get_current_pid_tgid();
u64 *tsp = start.lookup(&pid_tgid);
if (!tsp) return 0;
struct latency_entry entry = {};
entry.pid = pid_tgid >> 32;
entry.duration_us = bpf_ktime_get_ns() / 1000 - *tsp;
latencies.perf_submit(ctx, &entry, sizeof(entry));
start.delete(&pid_tgid);
return 0;
}
"""
b = BPF(text=bpf_text)
b.attach_uprobe(name="/usr/bin/redis-server", sym="setCommand",
fn_name="trace_set_start")
b.attach_uretprobe(name="/usr/bin/redis-server", sym="setCommand",
fn_name="trace_set_return")
latencies = []
def print_latency(cpu, data, size):
event = b["latencies"].event(data)
latencies.append(event.duration_us)
b["latencies"].open_perf_buffer(print_latency)
print("追踪 Redis setCommand 延迟...")
while True:
try:
b.perf_buffer_poll()
except KeyboardInterrupt:
break
if latencies:
latencies.sort()
avg = sum(latencies) / len(latencies)
p99 = latencies[int(len(latencies) * 0.99)]
print(f"总数: {len(latencies)}, 均值: {avg:.1f}us, P99: {p99}us")
5. XDP 高性能网络加速
5.1 XDP 原理
XDP在网卡驱动层直接处理数据包,是 Linux 中性能最高的包处理框架:
XDP_DROP ── 立即丢弃 (10Mpps+/core)
XDP_PASS ── 传递给内核协议栈
XDP_TX ── 从同源网卡发回
XDP_REDIRECT ── 定向转发到另一个网卡/CPU
5.2 DDoS 防护:SYN Flood 防护
// xdp_syn_flood.c - 在网卡入口丢弃过量 SYN 包
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_endian.h>
#define MAX_SYN_RATE 1000
#define RATE_WINDOW_NS 1000000000ULL
struct syn_counter {
u64 last_reset;
u64 count;
};
BPF_HASH(syn_stats, u32, struct syn_counter, 65536);
SEC("xdp")
int xdp_syn_filter(struct xdp_md *ctx) {
void *data = (void *)(long)ctx->data;
void *data_end = (void *)(long)ctx->data_end;
struct ethhdr *eth = data;
if ((void *)(eth + 1) > data_end) return XDP_PASS;
if (eth->h_proto != bpf_htons(ETH_P_IP)) return XDP_PASS;
struct iphdr *iph = (void *)(eth + 1);
if ((void *)(iph + 1) > data_end) return XDP_PASS;
if (iph->protocol != IPPROTO_TCP) return XDP_PASS;
u32 iph_len = iph->ihl * 4;
struct tcphdr *tcp = (void *)iph + iph_len;
if ((void *)(tcp + 1) > data_end) return XDP_PASS;
if (!tcp->syn || tcp->ack) return XDP_PASS;
u32 src_ip = iph->saddr;
u64 now = bpf_ktime_get_ns();
struct syn_counter *counter = syn_stats.lookup(&src_ip);
if (!counter) {
struct syn_counter new_counter = { now, 1 };
syn_stats.update(&src_ip, &new_counter);
return XDP_PASS;
}
if (now - counter->last_reset > RATE_WINDOW_NS) {
counter->last_reset = now;
counter->count = 1;
} else {
counter->count++;
}
if (counter->count > MAX_SYN_RATE) {
return XDP_DROP;
}
return XDP_PASS;
}
char _license[] SEC("license") = "GPL";
5.3 XDP 负载均衡
// xdp_l4_lb.c - MAC 重写负载均衡
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_endian.h>
#define VIP 0x0100007F // 127.0.0.1
#define BACKEND_COUNT 3
struct backend_info {
__u8 mac[6];
__u32 ip;
};
struct {
__uint(type, BPF_MAP_TYPE_ARRAY);
__uint(max_entries, BACKEND_COUNT);
__type(key, u32);
__type(value, struct backend_info);
} backends SEC(".maps");
SEC("xdp")
int xdp_load_balancer(struct xdp_md *ctx) {
void *data = (void *)(long)ctx->data;
void *data_end = (void *)(long)ctx->data_end;
struct ethhdr *eth = data;
if ((void *)(eth + 1) > data_end) return XDP_DROP;
struct iphdr *iph = (void *)(eth + 1);
if ((void *)(iph + 1) > data_end) return XDP_DROP;
if (iph->daddr != bpf_htonl(VIP)) return XDP_PASS;
u32 backend_idx = iph->saddr % BACKEND_COUNT;
struct backend_info *backend = backends.lookup(&backend_idx);
if (!backend) return XDP_DROP;
__builtin_memcpy(eth->h_dest, backend->mac, ETH_ALEN);
return XDP_TX;
}
char _license[] SEC("license") = "GPL";
6. BPF CO-RE:编译一次、到处运行
BPF CO-RE 结合 BTF 实现了跨平台可移植 eBPF 程序:
// CO-RE 方式读取内核结构体
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
#include <bpf/bpf_core_read.h>
SEC("kprobe/tcp_sendmsg")
int BPF_KPROBE(trace_tcp_sendmsg, struct sock *sk,
struct msghdr *msg, size_t size) {
u16 dport = BPF_CORE_READ(sk, __sk_common.skc_dport);
u32 saddr = BPF_CORE_READ(sk, __sk_common.skc_rcv_saddr);
u16 family = BPF_CORE_READ(sk, __sk_common.skc_family);
bpf_printk("tcp_sendmsg: family=%d saddr=0x%x dport=%d size=%lu\\n",
family, saddr, bpf_ntohs(dport), size);
return 0;
}
char _license[] SEC("license") = "GPL";
编译方式:
clang -g -O2 -target bpf -D__TARGET_ARCH_x86_64 \
-I/usr/include/x86_64-linux-gnu \
-o tcp_monitor.bpf.o tcp_monitor.bpf.c
bpftool gen skeleton tcp_monitor.bpf.o > tcp_monitor.skel.h
bpftool prog load tcp_monitor.bpf.o /sys/fs/bpf/tcp_monitor autoattach
7. LSM eBPF:安全策略控制
Linux 5.7+ 支持将 eBPF 附加到 LSM 钩子点:
// 阻止非特权用户提升进程优先级
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
SEC("lsm/task_setnice")
int BPF_PROG(restrict_nice, int nice, int *ret) {
if (nice < 0) {
u64 uid = bpf_get_current_uid_gid() & 0xFFFFFFFF;
if (uid != 0) {
bpf_printk("denying non-root nice elevation from uid=%d\\n", uid);
*ret = -EPERM;
return -1;
}
}
return 0;
}
char _license[] SEC("license") = "GPL";
8. eBPF 工具链生态
eBPF 拥有丰富的开源生态:
- bpftrace:类似 awk/tracer,一行命令实现复杂追踪
- BCC:Python/Lua 绑定,快速原型开发
- libbpf:C 语言原生库,生产级部署
- Cilium:基于 eBPF 的 CNI 网络方案
- Falco:运行时安全检测引擎
- Tetragon:安全可观测与运行时策略执行
- Pixie:Kubernetes 全栈可观测性平台
8.1 bpftrace 一行命令示例
# 追踪所有 openat 系统调用的文件名
bpftrace -e 'tracepoint:syscalls:sys_enter_openat { printf("%s %s\n", comm, str(args->filename)); }'
# 统计 socket 调用的时延分布
bpftrace -e 'kprobe:tcp_sendmsg { @start[tid] = nsecs; } kretprobe:tcp_sendmsg /@start[tid]/ = { @us = hist((nsecs - @start[tid]) / 1000); }'
# 按进程统计 page fault 次数
bpftrace -e 'software:page-faults: { @[comm] = count(); }'
# 追踪块设备 I/O 延迟
bpftrace -e 'kprobe:blk_mq_start_req { @start[arg0] = nsecs; } kprobe:blk_mq_end_io /@start[arg0]/ = { @io_us = hist((nsecs - @start[arg0]) / 1000); delete(@start[arg0]); }'
9. eBPF 性能优势解析
eBPF 之所以能达到接近原生内核的性能,原因在于:
- JIT 编译:eBPF 字节码由 JIT 编译器直接转换为原生 x86/ARM 指令,零解释开销
- Verifier 安全检查:运行时前通过静态分析验证无死循环、无未初始化读、无越界访问,保证安全
- 零数据拷贝:Perf Buffer / Ring Buffer 通过 mmap 实现零拷贝数据传输
- 内核内执行:避免了用户态/内核态切换开销,事件在哪里发生就在哪里处理
- 可编程直方图聚合:统计运算在内核中完成,用户态只读取聚合结果
对比传统方法:使用 iptables DDoS 防护吞吐量约 2Mpps/core,而 XDP 可达 24Mpps/core 以上。
10. 实战案例
10.1 Off-CPU 火焰图
排查 CPU 尖刺:使用 eBPF 的 offcputime 工具生成 Off-CPU 火焰图:
# 追踪进程超过 1ms 的阻塞时间
offcputime-bpfcc -p 1234 1 > out.stacks
# 生成火焰图
stackcollapse-bpfcc out.stacks > out.folded
flamegraph.pl out.folded > offcpu-flamegraph.png
# bpftrace 一行版
bpftrace -e 'kprobe:finish_task_switch /@last[tid]/ = { @[kstack, ustack, comm] = hist(nsecs - @last[tid]); } { @last[tid] = nsecs; }'
10.2 容器网络流量追踪
结合 cgroup 上下文精准追踪 Pod 流量:
#!/usr/bin/env python3
from bcc import BPF
from socket import inet_ntop, AF_INET
bpf_text = u"""
#include <uapi/linux/ptrace.h>
#include <net/sock.h>
struct flow_key {
u32 saddr, daddr;
u16 sport, dport;
u32 pid;
};
BPF_HASH(flows, struct flow_key, u64);
SEC("kprobe/tcp_sendmsg")
int trace_tcp_send(struct pt_regs *ctx, struct sock *sk,
struct msghdr *msg, size_t size) {
struct flow_key key = {};
key.saddr = sk->__sk_common.skc_rcv_saddr;
key.daddr = sk->__sk_common.skc_daddr;
key.dport = bpf_ntohs(sk->__sk_common.skc_dport);
key.sport = sk->__sk_common.skc_num;
key.pid = bpf_get_current_pid_tgid() >> 32;
if (key.dport == 0) return 0;
u64 init = size, *val;
val = flows.lookup_or_try_init(&key, &init);
if (val) lock_xadd(val, size);
return 0;
}
char _license[] SEC("license") = "GPL";
"""
b = BPF(text=bpf_text)
b.attach_kprobe(event="tcp_sendmsg", fn_name="trace_tcp_send")
import time
time.sleep(10)
flows = b["flows"]
print(f"{'SRC':>22} {'DST':>22} {'PID':>6} {'BYTES':>10}")
print("-" * 65)
for k, v in sorted(flows.items(), key=lambda x: x[1].value, reverse=True)[:20]:
src = inet_ntop(AF_INET, bytes.fromhex(f"{k.saddr:08x}"))
dst = inet_ntop(AF_INET, bytes.fromhex(f"{k.daddr:08x}"))
print(f"{src:>15}:{k.sport:<5} {dst:>15}:{k.dport:<5} {k.pid:>6} {v.value:>10}")
11. 常见陷阱与最佳实践
11.1 Verifier 拒绝你的程序?
- 无限循环:所有循环必须有界(最大 4096 次),推荐 #pragma unroll
- 未初始化读取:所有变量使用前必须显式初始化,需 bpf_probe_read
- 栈大小超限:eBPF 栈仅 512 字节,大变量放 BPF Map
- 除零风险:除法操作需 Verifier 证明除数不为零
- 指令数超限:一个函数超过百万指令需拆分为尾调用
11.2 BPF 程序调试
# 查看已加载的 BPF 程序
bpftool prog show
bpftool prog dump xlated id 123
# 查看 BPF Maps
bpftool map show
bpftool map dump id 456
# 查看挂载点
bpftool net show
bpftool perf show
# 启用动态调试
echo 'file kernel/bpf/* +p' > /sys/kernel/debug/dynamic_debug/control
# 检查 Verifier 日志
bpftool prog load prog.o /sys/fs/bpf/test 2>&1 | head -50
12. eBPF 未来趋势
- BPF for Windows:微软已在 Windows 中实现 eBPF,跨平台可观测性统一
- CO-RE Auto-Attach:程序自描述、自挂载,进一步降低使用门槛
- 可编程调度器:CFS 调度策略通过 eBPF 动态调整
- 硬件卸载:SmartNIC/DPU 原生运行 eBPF 程序(NVIDIA BlueField、Intel IPU)
- 替代部分内核模块:KernelCon 2024 讨论使用 BPF 替代部分内核模块
- eBPF + Wasm:两种字节码技术的融合互补
总结
eBPF 的核心价值在于:在不牺牲安全性和稳定性的前提下,赋予开发者编程内核的能力。它是"可编程内核"时代的基础设施,正在深刻改变网络、安全、可观测性的实现方式。
作为系统工程师,掌握 eBPF 已不再是加分项,而是刚需。从今天开始,尝试用 bpftrace 一行命令替换脚本中 strace、tcpdump 的部分场景,逐步深入,你将发现一个全新的内核世界。
参考资料:
- BPF 官方文档:ebpf.io
- 《BPF Performance Tools》 Brendan Gregg
- 《Linux Observability with BPF》 David Calavera
- BCC 项目 GitHub:iovisor/bcc
- CNCF Cilium 项目

发表评论 取消回复