eBPF CO-RE 工程化实践:libbpf 骨架、BTF 内省与跨内核版本兼容
引言
传统 eBPF 开发需要用户空间加载器手动重定位 .rodata 字段偏移、处理内核结构体差异,每次内核升级都意味着一遍痛苦的重新编译。eBPF CO-RE(Compile Once – Run Everywhere)彻底改变了这一现状:配合 BTF(BPF Type Format)和 libbpf 骨架(skeleton),我们可以像写普通 C 程序一样编写可跨内核版本运行的 eBPF 程序。
本文将从零开始构建一个完整的 CO-RE 工程,深入解析 libbpf 骨架生成机制、BTF 内省原理、重定位条目处理,以及如何在生产环境中处理不同内核版本的字段差异。
1. CO-RE 技术栈全景
CO-RE 不是单一技术,而是一组协同工作的组件链:
| 组件 | 职责 |
|---|---|
| Clang/LLVM | 生成 BTF 信息,避免 DWARF 开销 |
| BTF | 内核与 ELF 中的类型描述元数据 |
| libbpf | 运行时重定位,根据目标系统的 BTF 调整 eBPF 字节码 |
| vmlinux.h | 从当前系统 BTF 自动生成,包含所有内核类型定义 |
| bpftool | 骨架生成、BTF 提取、map/program dump |
为什么需要 CO-RE
考虑一个简单场景:要监控 execve() 调用的进程名。不同内核版本中,task_struct->comm 字段可能嵌在不同的父结构中,task_struct 本身的布局也会变化。传统方式需要在运行时通过 bpf_probe_read() 手工读取,还要维护一份巨大的 kernel header 列表。CO-RE 让你在 eBPF 代码中直接写 task->comm,libbpf 在加载时自动处理偏移量重定位。
2. 工程初始化与骨架生成
2.1 项目结构
exec_monitor/
├── Makefile
├── exec_monitor.bpf.c # eBPF 代码(运行在内核态)
├── exec_monitor.c # 用户空间加载器
└── vmlinux.h # 系统 BTF 导出
2.2 eBPF 程序
// exec_monitor.bpf.c
#include "vmlinux.h"
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
#include <bpf/bpf_core_read.h>
// 用户态可配置的过滤条件
const volatile const u32 target_pid = 0;
const volatile const char filter_comm[16] = "";
struct {
__uint(type, BPF_MAP_TYPE_RINGBUF);
__uint(max_entries, 256 * 1024) ; // 256KB ring buffer
} rb SEC(".maps");
struct event {
u32 pid;
u32 ppid;
char comm[16];
char filename[256];
};
// 声明 license(BVerifier 要求)
char LICENSE[] SEC("license") = "GPL";
SEC("tp/syscalls/sys_enter_execve")
int tracepoint__syscalls__sys_enter_execve(struct trace_event_raw_sys_enter *ctx)
{
struct event *e;
struct task_struct *task;
const char *filename;
// 使用 BPF_CORE_READ 宏进行 CO-RE 安全读取
task = (struct task_struct *)bpf_get_current_task();
u32 pid = bpf_get_current_pid_tgid() >> 32;
// PID 过滤
if (target_pid && pid != target_pid)
return 0;
// 申请 ringbuf 空间
e = bpf_ringbuf_reserve(&rb, sizeof(*e), 0);
if (!e)
return 0;
e->pid = pid;
e->ppid = BPF_CORE_READ(task, real_parent, tgid);
bpf_get_current_comm(&e->comm, sizeof(e->comm));
// 从 execve 参数中读取 filename
filename = (const char *)ctx->args[0];
bpf_probe_read_user_str(e->filename, sizeof(e->filename), filename);
// 若配置了 comm 过滤则判断
if (filter_comm[0]) {
char match = 1;
for (int i = 0; i < 15; i++) {
if (e->comm[i] != filter_comm[i]) {
match = 0;
break;
}
if (!filter_comm[i]) break;
}
if (!match) {
bpf_ringbuf_discard(e, 0);
return 0;
}
}
bpf_ringbuf_submit(e, 0);
return 0;
}
2.3 Makefile:自动化骨架生成
# MakefileAPP := exec_monitor
APP := exec_monitor
BPF_SRC := $(APP).bpf.c
USER_SRC := $(APP).c
# 检测libbpf是否以子模块形式存在(优先使用)
LIBBPF_SRC := $(abspath ../libbpf/src)
LIBBPF_OBJ := $(abspath $(OUTPUT)/libbpf.a)
CLANG ?= clang
BPFTOOL ?= bpftool
ARCH := $(shell uname -m | sed 's/x86_64/x86/' | sed 's/aarch64/arm64/')
OUTPUT := .output
BPF_CFLAGS := -target bpf -D__TARGET_ARCH_$(ARCH) -I$(OUTPUT) \
-I/usr/include -O2 -g -Wall -Werror
.PHONY: clean $(APP)
# 提取 vmlinux.h
$(OUTPUT)/vmlinux.h: $(OUTPUT)
$(BPFTOOL) btf dump file /sys/kernel/btf/vmlinux format c > $@
# 编译 BPF 字节码(嵌入 BTF)
$(OUTPUT)/$(APP).bpf.o: $(BPF_SRC) $(OUTPUT)/vmlinux.h
$(CLANG) $(BPF_CFLAGS) -c $< -o $@
# 生成骨架头文件
$(OUTPUT)/$(APP).skel.h: $(OUTPUT)/$(APP).bpf.o | $(OUTPUT)
$(BPFTOOL) gen skeleton $< > $@
# 编译用户态程序
$(USER_SRC): $(OUTPUT)/$(APP).skel.h
$(OUTPUT):
mkdir -p $@
clean:
rm -rf $(OUTPUT) $(APP)
2.4 用户空间加载器
// exec_monitor.c
#include <stdio.h>
#include <unistd.h>
#include <signal.h>
#include <sys/resource.h>
#include <bpf/libbpf.h>
#include "exec_monitor.skel.h" // 骨架头文件(自动生成的)
static volatile bool exiting = false;
static void sig_handler(int sig) {
exiting = true;
}
static int handle_event(void *ctx, void *data, size_t data_sz) {
const struct event *e = data;
printf("[%d] PID:%d PPID:%d COMM:%s FILE:%s\n",
getpid(), e->pid, e->ppid, e->comm, e->filename);
return 0;
}
static int libbpf_print_fn(enum libbpf_print_level level,
const char *format, va_list args) {
return vfprintf(stderr, format, args);
}
int main(int argc, char **argv) {
struct exec_monitor_bpf *skel;
struct ring_buffer *rb = NULL;
int err;
// 设置 libbpf 日志
libbpf_set_print(libbpf_print_fn);
// 提升 RLIMIT_MEMLOCK(eBPF map 需要锁定内存)
struct rlimit rlim_new = { RLIM_INFINITY, RLIM_INFINITY };
if (setrlimit(RLIMIT_MEMLOCK, &rlim_new)) {
fprintf(stderr, "Failed to increase RLIMIT_MEMLOCK\n");
return 1;
}
signal(SIGINT, sig_handler);
signal(SIGTERM, sig_handler);
// 1. 打开骨架
skel = exec_monitor_bpf__open();
if (!skel) {
fprintf(stderr, "Failed to open BPF skeleton\n");
return 1;
}
// 2. 用户态配置 BPF 变量
skel->rodata->target_pid = 0; // 不过滤 PID
strcpy(skel->rodata->filter_comm, "");
// 3. 加载并验证 BPF 程序(libbpf 执行 CO-RE 重定位)
err = exec_monitor_bpf__load(skel);
if (err) {
fprintf(stderr, "Failed to load BPF skeleton: %d\n", err);
goto cleanup;
}
// 4. 附加到 tracepoint
err = exec_monitor_bpf__attach(skel);
if (err) {
fprintf(stderr, "Failed to attach BPF skeleton: %d\n", err);
goto cleanup;
}
// 5. 绑定 ring buffer 回调
rb = ring_buffer__new(bpf_map__fd(skel->maps.rb), handle_event, NULL, NULL);
if (!rb) {
fprintf(stderr, "Failed to create ring buffer\n");
err = 1;
goto cleanup;
}
printf("Successfully started! Press Ctrl+C to stop.\n");
while (!exiting) {
err = ring_buffer__poll(rb, 100); // 100ms timeout
if (err == -EINTR) { err = 0; break; }
if (err < 0) {
fprintf(stderr, "Error polling ring buffer: %d\n", err);
break;
}
}
cleanup:
ring_buffer__free(rb);
exec_monitor_bpf__destroy(skel);
return err != 0;
}
3. 骨架生成的内部机制
bpftool gen skeleton 会扫描 BPF ELF 的 .rodata、.bss、.data 段以及 maps 定义,生成一个 C 头文件。骨架提供的关键 API:
打开阶段(*_open):将 BPF 字节码解析到内存,创建 map 句柄结构体,但不进行系统调用。这让你有机会在加载前覆盖 rodata 中的变量。
加载阶段(*_load):这里发生真正的重定位。libbpf 会:
1. 为每个 map 调用 bpf(BPF_MAP_CREATE)
2. 对每个 program 调用 bpf(BPF_PROG_LOAD),内核 verifier 会拒绝重定位失败的程序
3. 根据目标系统的 BTF 信息更新重定位记录中的偏移量
附加阶段(*_attach):处理 SEC() 中声明的附加类型。tp/syscalls/sys_enter_execve 会被解析并自动附加到对应的 tracepoint。
3.1 骨架生成的数据结构
生成的骨架头文件类似:
struct exec_monitor_bpf {
struct bpf_object *obj;
// Maps
struct bpf_map *maps[2];
// maps[0] = rb (ring buffer)
// Programs
struct bpf_program *programs[1];
// programs[0] = tracepoint__syscalls__sys_enter_execve
// 数据类型段
struct exec_monitor_bpf__bss {
// bss 变量
} *bss;
struct exec_monitor_bpf__rodata {
u32 target_pid;
char filter_comm[16];
} *rodata;
struct exec_monitor_bpf__data {
// data 段变量
} *data;
struct exec_monitor_bpf__elf *elf;
};
4. BTF 内省与 CO-RE 重定位原理
4.1 重定位条目如何工作
当我们写 BPF_CORE_READ(task, real_parent, tgid) 时,Clang 不会直接生成基于硬编码偏移的 ldxdw 指令。相反,它在 ELF 的重定位段(.rel.BTF / BTF.ext)中记录一条记录:
Instruction Pointer: 0x1234
Source Type: struct task_struct * (来自 vmlinux.h)
Access String: real_parent->tgid (字段访问路径)
libbpf 加载时读取宿主机 /sys/kernel/btf/vmlinux,将 target BTF(运行时的内核类型)与 local BTF(程序编译时的类型)进行匹配。如果匹配成功,calcs access string 对应的偏移量并修补指令中的字段。
4.2 BPF_CORE_READ 宏的展开
// 简化版说明
#define BPF_CORE_READ(dst, src, a) \
bpf_probe_read_kernel( \
sizeof(*(dst)) ? (dst) : (dst), \
sizeof(*(src)), \
__builtin_preserve_access_index((src).a) \
)
__builtin_preserve_access_index 是 Clang 的 CO-RE 内置函数。它不执行实际读取,而是在 ELF 中留下一条重定位记录,让 libbpf 在加载时修补偏移量。
4.3 字段兼容层的条件编译
当目标内核可能不存在某个字段时,需要手动处理:
// 示例:task_struct->cpu 字段在 5.16 之前是 sched_cpu(无名)
static __always_inline int get_task_cpu(struct task_struct *task) {
#if __has_field_in_struct(struct task_struct, cpu)
return BPF_CORE_READ(task, cpu);
#else
// 回退方案
return BPF_CORE_READ(task, recent_used_cpu);
#endif
}
更优雅的方式是使用 vmlinux.h 的 CO-RE 辅助宏:
// 显式处理不同内核版本的字段重命名
struct task_struct *task = ...;
long state;
// libbpf 提供的宏从内核 BTF 推断实际字段名(允许"field rename")
state = BPF_CORE_READ(task, __state); // 5.14 之前用 state,5.14+ 用 __state
5. 处理内核 API 差异的策略
5.1 版本守卫模式
// exec_monitor.bpf.c
#include <bpf/bpf_core_read.h>
// 通用宏定义不同内核版本的差异
#if (LINUX_VERSION_CODE < KERNEL_VERSION(5, 18, 0))
#define BPF_MAP_TYPE_RINGBUF BPF_MAP_TYPE_PERF_EVENT_ARRAY
#endif
// 使用 __builtin_preserve_type_info 检查特定内核是否存在
static __always_inline bool kernel_has_field_comm(void) {
return bpf_core_field_exists(struct task_struct, comm);
}
5.2 弱定义与默认值
利用 .bss 段的零初始化特性实现可选配置:
struct {
__uint(type, BPF_MAP_TYPE_HASH);
__uint(max_entries, 64);
__type(key, u32);
__type(value, struct rule);
} rules SEC(".maps");
// 默认行为由 bss 变量控制
const volatile bool enable_filter = false; // 用户态可修改
5.3 条件附加
在某些内核版本下,某些 tracepoint 可能不存在。可以通过 bpf_program__set_autoattach 控制:
// 在用户态代码中
if (!old_kernel) {
bpf_program__set_autoattach(skel->progs.tracepoint__tp, true);
}
或让 BPF 程序返回错误码自动跳过无效附加:
SEC("kprobe/do_unlinkat BPF")
int BPF_KPROBE(trace_unlinkat, int dfd, struct filename *name) {
// 若该 tracepoint 在此内核不存在,loader 会跳过
return 0;
}
}
6. Map 高级用法与 CO-RE
6.1 Ring Buffer vs Perf Event Array
CO-RE 项目中应当优先选择 Ring Buffer(5.8+ 内核):
| 特性 | perf event array | ring buffer |
|---|---|---|
| 内存模型 | per-CPU | 全局 |
| 数据容量 | 单页 * N CPU | 总容量可配置 |
| 事件丢失通知 | 无(需自行检测欠载) | 有(BPF_RB_NO_WAKEUP/BPF_RB_FORCE_WAKEUP) |
| 最佳场景 | 高吞吐、允许丢失 | 需要事件完整性 |
6.2 复杂 Map 类型的骨架生成示例
struct {
__uint(type, BPF_MAP_TYPE_LRU_HASH);
__uint(max_entries, 65536);
__type(key, struct flow_key); // 自定义结构体
__type(value, struct flow_stats);
} flow_table SEC(".maps");
// 在用户态中通过 skeleton 直接操作
int map_fd = bpf_map__fd(skel->maps.flow_table);
bpf_map_update_elem(map_fd, &key, &val, BPF_ANY);
7. 调试与故障排查
7.1 verifier 日志解读
verifier 拒绝 eBPF 程序是常见痛点。加载时添加详细日志:
# 增大内核 verifier 日志缓冲区
sudo sysctl -w kernel.bpf_stats_enabled=1
# 通过 libbpf 获取 verifier 日志
struct bpf_prog_load_opts opts = {
.log_level = BPF_LOG_LEVEL2
};
// 加载时传入 opts
常见错误类型:
| 错误 | 原因 | 修复 |
|---|---|---|
failed to find BTF member |
目标内核缺少结构字段 | 使用条件读取或另类字段名 |
R1 is not a scalar |
Map 访问中寄存器被“污染” | 显式重置或 range check |
unbounded loop |
verifier 不支持无法静态确定的循环 | 使用 pragma unroll 或尾调用 |
7.2 bpftool 诊断
# 查看已加载的 BPF 程序及其 map
sudo bpftool prog show
sudo bpftool map show
# 导出 eBPF 字节码为 C 伪代码(用于审计)
sudo bpftool prog dump xlated id 123 visual > prog.dot
# 追踪 verifier 决策过程
sudo bpftool prog load obj.o /sys/fs/bpf/prog -t 2
8. 生产级打包策略
8.1 多内核版本兼容
真实部署场景中,宿主机内核版本各异。策略:
- 发行版打包:用 CMake/Meson 打包时嵌入 libbpf(以静态链接方式)
- CI 交叉编译:在 CI 中使用
--auto-clang=gke-5.15链式生成多版本骨架 - 运行时检测:先探测
/sys/kernel/btf/vmlinux是否存在;若不存在,call 回退到 BCC + 动态编译
8.2 安全性加固
// 1. 启用 BPF token(6.4+ 内核)—— 允许非特权用户加载 BPF
union bpf_attr attr = { ... };
int prog_fd = bpf(BPF_PROG_LOAD, &attr, sizeof(attr));
// 2. 限制 map 的宽限期(LRU 策略替代 PERCPU_HASH)
// 3. 使用 ring buffer 替代 perf event array 减少竞态窗口
9. 完整工程实战总结
通过 CO-RE + 骨架 + BTF,eBPF 工程化流程变为:
1. 编写 vmlinux.h 导入内核类型 — 不与任何具体内核绑定
2. 编写 .bpf.c 程序代码 — 通过 BPF_CORE_READ 宏访问内核字段
3. 编译为 BPF 字节码(嵌入 BTF) — clang -target bpf
4. 生成骨架头文件 — bpftool gen skeleton
5. 编写用户态加载器 — #include skeleton.h
6. 编译链接 — 静态链接 libbpf
7. 部署执行 — 可在任何 5.15+ BTF-enabled 内核上运行
10. 总结与展望
eBPF CO-RE 将 eBPF 从“需要随内核重新编译”的实验性工具,升级为可生产性部署的可观测性与安全基础设施。libbpf 骨架隐藏了 map 创建、程序加载、事件回调绑定的繁琐细节,让开发者专注于 BPF 程序本身的业务逻辑。
随着 BPF token(6.4+)和 BPF arena(6.3+)等新特性落地,非特权加载和内核-用户空间共享内存场景将更加成熟。结合 BTF 的内省能力,eBPF 正在从“内核追踪子领域”演变为通用平台编程接口。对于任何构建云原生基础设施的团队,掌握 CO-RE 已经进入“非可选项”阶段。

发表评论 取消回复