┌─────────────────────────────────────────────────────────┐
│ Linux Kernel │
│ │
│ Container Process │
│ ┌──────────┐ seccomp filter ┌──────────────┐ │
│ │ syscall() │ ──────────────────► │ notification │ │
│ │ (阻塞) │ │ created │ │
│ └──────────┘ └──────┬───────┘ │
│ │ event fd │
│ ▼ │
│ ┌──────────────────────────────────────────────┐ │
│ │ Unix Socket (seccomp notify fd) │ │
│ └──────────────────────┬───────────────────────┘ │
└─────────────────────────┼───────────────────────────────┘
│
▼
┌─────────────────────────────────────────────────────────┐
│ Supervisor Process (用户态) │
│ │
│ 1. 读取 notification 结构体(含 syscall nr + args) │
│ 2. 安全审计 / 修改参数 │
│ 3. 自行执行 syscall(可选) │
│ 4. 通过 response 结构体返回结果 │
│ - 允许 + 返回值 │
│ - 阻断 + errno │
│ - 继续原始 syscall │
└─────────────────────────────────────────────────────────┘
关键突破在于:容器进程被挂起等待 supervisor 决策期间,supervisor 可以自由地以自身权限执行系统调用,并将结果注入容器进程。容器进程最终获得的效果如同"自己执行了 syscall",但实际执行者拥有不同的权限上下文。
二、内核实现深度剖析
2.1 数据结构关系
// include/linux/seccomp.h 核心结构体
struct seccomp_filter {
refcount_t refcnt;
u16 count; // BPF 指令条数
struct seccomp_filter *prev; // 链表前驱
struct bpf_prog *prog; // BPF 程序
struct seccomp_notif_sizes sizes; // 各结构体尺寸
};
struct seccomp_notif {
__u64 id; // 通知唯一标识
__u32 pid; // 发送方 PID
__u32 flags; // 标志位
struct seccomp_data data; // 系统调用信息
};
struct seccomp_notif_resp {
__u64 id; // 对应 notification id
__s64 val; // 返回值
__s32 error; // errno (负数)
__u32 flags; // SECCOMP_USER_NOTIF_FLAG_CONTINUE
};
2.2 内核通知分发路径
当 BPF 程序返回 SECCOMP_RET_USER_NOTIF 时,内核进入 seccomp_do_user_notification():
// kernel/seccomp.c (简化路径)
static int seccomp_do_user_notification(int this_syscall, ...)
{
// 1. 分配通知结构体并填充参数
*id = seccomp_next_notify_id(filter);
init_kiomap(&work.notif->elem, notif, sizeof(*notif));
// 2. 写入等待队列,挂起调用线程
list_add_tail(&work.list, &filter->notif->notify_lock);
// 发送事件到 eventfd (如已注册)
// 3. 等待用户态响应
wait_event_interruptible(filter->notif->wqh,
supv_responded == true);
// 4. 处理响应
if (resp.flags & SECCOMP_USER_NOTIF_FLAG_CONTINUE)
return this_syscall; // 放行原始 syscall
if (resp.error) {
// 返回错误给容器进程
syscall_set_errno(resp.error);
return resp.val;
}
return resp.val; // 注入伪造的返回值
}
值得注意的是,内核并不限制 supervisor 的执行上下文。Supervisor 以自身 PID 发起 syscall 这一事实意味着:只要 supervisor 拥有相应的能力(Capabilities)或特权,它就能完成容器进程本身无法执行的操作。
2.3 关键限制与行为约束
- 同一时间只有一个通知:同一个 seccomp filter 每次只能处理一个 notify 事件。后续 syscall 会被阻塞直到当前 notify 完成。这个设计避免了复杂的时序问题,但也意味着 notify 处理程序必须高效。
- id 生命周期管理:用户态必须通过
SECCOMP_IOCTL_NOTIF_ID_VALID ioctl 验证 id 在处理期间仍然有效(容器进程可能已退出)。
- 不可伪造的通知:通知只能通过 seccomp BPF 返回特定值触发,用户态无法伪造通知。但 supervisor 可以在响应时修改系统调用参数后替容器执行,实现 syscall 拦截与改写。
- 事件通知机制:通过
SECCOMP_IOCTL_NOTIF_SET_FLAGS + SECCOMP_USER_NOTIF_FLAG_NEW_LISTENER 可以将 notify fd 绑定到 eventfd,实现异步事件驱动模型。
三、用户态编程接口
3.1 libseccomp 封装
#include <seccomp.h>
// 创建带 notify 支持的上下文
SCMP_ACT_NOTIFY(supervisor_fd)
// 关键 API
int seccomp_notif_alloc(struct seccomp_notif **req,
struct seccomp_notif_resp **resp);
int secmat_notif_receive(int fd, struct seccomp_notif *req);
int seccomp_notif_send_resp(int fd, struct seccomp_notif_resp *resp);
int seccomp_notif_id_valid(int fd, __u64 id);
3.2 原始 ioctl 接口
如果你需要更底层控制(例如在不依赖 libseccomp 的情况下):
SECCOMP_IOCTL_NOTIF_RECV - 接收通知
SECCOMP_IOCTL_NOTIF_SEND - 发送响应
SECCOMP_IOCTL_NOTIF_ID_VALID - 验证 ID
SECCOMP_IOCTL_NOTIF_SET_FLAGS - 设置 flags
SECCOMP_IOCTL_NOTIF_ADDFD - 额外:向容器进程注入 fd(Linux 5.9+)
SECCOMP_IOCTL_NOTIF_ADDFD 是一个非常强大的功能:supervisor 可以在响应的同时打开一个 fd 并将其注入目标进程的文件描述符表。这解决了"supervisor 打开了文件但容器进程没有对应 fd"的问题。
四、代码实战:最小化 Seccomp Notify 代理
以下是一个完整的 C 语言示例,展示 supervisor 如何拦截 mount() 系统调用并安全代理执行:
4.1 容器端程序 (container.c)
#define _GNU_SOURCE
#include <seccomp.h>
#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <fcntl.h>
#include <sys/syscall.h>
int main(int argc, char **argv) {
if (argc < 2) {
fprintf(stderr, "Usage: %s <notify_fd>\n", argv[0]);
return 1;
}
int notify_fd = atoi(argv[1]);
// 初始化 seccomp 上下文
scmp_filter_ctx ctx = seccomp_init(SCMP_ACT_ALLOW);
if (!ctx) {
perror("seccomp_init");
return 1;
}
// 仅对 mount() 触发 notify
if (seccomp_rule_add(ctx, SCMP_ACT_NOTIFY(notify_fd),
SCMP_SYS(mount), 0) < 0) {
perror("seccomp_rule_add");
return 1;
}
// 加载 seccomp filter
if (seccomp_load(ctx) < 0) {
perror("seccomp_load");
return 1;
}
seccomp_release(ctx);
printf("[Container] Seccomp filter loaded (mount -> notify).\n");
printf("[Container] Attempting mount()...\n");
// 发起 mount 系统调用 - 如果没有 CAP_SYS_ADMIN,
// 传统上会失败,但 notify 将代理执行
long ret = mount("tmpfs", "/mnt/test", "tmpfs", 0, NULL);
if (ret == 0) {
printf("[Container] mount() succeeded via notify!\n");
umount("/mnt/test");
} else {
perror("[Container] mount() failed");
}
return 0;
}
4.2 Supervisor 端程序 (supervisor.c)
#define _GNU_SOURCE
#include <seccomp.h>
#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <string.h>
#include <sys/syscall.h>
#include <sys/mount.h>
#include <sys/stat.h>
#include <linux/seccomp.h>
static int process_notif(int notify_fd) {
struct seccomp_notif *req = NULL;
struct seccomp_notif_resp *resp = NULL;
int ret;
ret = seccomp_notif_alloc(&req, &resp);
if (ret < 0) {
perror("seccomp_notif_alloc");
return ret;
}
// 阻塞等待通知
while (1) {
memset(req, 0, sizeof(*req));
memset(resp, 0, sizeof(*resp));
ret = seccomp_notif_receive(notify_fd, req);
if (ret < 0) {
if (errno == EINTR) continue;
perror("seccomp_notif_receive");
break;
}
printf("[Supervisor] Got notification: pid=%u, syscall=%d\n",
req->data.pid, req->data.nr);
// 安全策略检查
if (req->data.nr == SCMP_SYS(mount)) {
// 解析 mount 参数(从 req->data.args)
char *source = (char *)(long)req->data.args[0];
char *target = (char *)(long)req->data.args[1];
char *fstype = (char *)(long)req->data.args[2];
printf("[Supervisor] mount(source=%s, target=%s, type=%s)\n",
source, target, fstype);
// 策略:仅允许 tmpfs 挂载
if (fstype && strcmp(fstype, "tmpfs") == 0) {
// 以 supervisor 权限执行 mount
long mount_ret = mount(source, target, fstype, 0, NULL);
resp->id = req->id;
resp->val = mount_ret;
resp->error = mount_ret == 0 ? 0 : -errno;
resp->flags = 0;
printf("[Supervisor] Executed mount, returning: %ld\n",
mount_ret);
} else {
// 阻断非 tmpfs 挂载请求
resp->id = req->id;
resp->val = -EPERM;
resp->error = -EPERM;
resp->flags = 0;
printf("[Supervisor] Blocked non-tmpfs mount\n");
}
} else {
// 其他放行
resp->id = req->id;
resp->val = 0;
resp->error = 0;
resp->flags = SECCOMP_USER_NOTIF_FLAG_CONTINUE;
}
// 发送响应
ret = seccomp_notif_send_resp(notify_fd, resp);
if (ret < 0) {
perror("seccomp_notif_send_resp");
break;
}
}
seccomp_notif_free(req, resp);
return ret;
}
int main(void) {
// 创建 socketpair,一端作为 notify fd
int sv[2];
if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) < 0) {
perror("socketpair");
return 1;
}
pid_t child = fork();
if (child == 0) {
// 子进程:容器
close(sv[0]);
char fd_str[16];
snprintf(fd_str, sizeof(fd_str), "%d", sv[1]);
execl("./container", "container", fd_str, NULL);
perror("execl");
return 1;
}
// 父进程:supervisor
close(sv[1]);
printf("[Supervisor] Started, fd=%d, monitoring child pid=%d\n",
sv[0], child);
process_notif(sv[0]);
return 0;
}
4.3 编译与验证
# 编译
gcc -o container container.c -lseccomp -Wall
gcc -o supervisor supervisor.c -lseccomp -Wall
# 运行 supervisor(需要 CAP_SYS_ADMIN 才能执行 mount)
sudo ./supervisor
# 预期输出:
# [Supervisor] Started, fd=3, monitoring child pid=12345
# [Container] Seccomp filter loaded (mount -> notify).
# [Container] Attempting mount()...
# [Supervisor] Got notification: pid=12345, syscall=165
# [Supervisor] mount(source=tmpfs, target=/mnt/test, type=tmpfs)
# [Supervisor] Executed mount, returning: 0
# [Container] mount() succeeded via notify!
关键观察:容器进程(即使无 CAP_SYS_ADMIN)通过 supervisor 代执行,成功完成了 tmpfs 挂载。
五、工程扩展:FUSE 文件系统代理
Seccomp notify 最典型的生产环境用途是rootless 容器中的 FUSE 文件系统支持。以下是一个 Rust 实现的高层示例,使用 tokio 异步调度 notify 处理:
// Cargo.toml
// [dependencies]
// tokio = { version = "1", features = ["full"] }
// rustix = { version = "0.38", features =["fs"] }
// seccompiler = "0.4"
use std::os::fd::RawFd;
use tokio::io::unix::AsyncFd;
use seccompiler::{SeccompNotif, SeccompNotifResponse};
struct SeccompNotifyHandler {
notify_fd: RawFd,
// 策略规则集
policy: Arc<PolicyEngine>,
}
impl SeccompNotifyHandler {
pub fn new(notify_fd: RawFd, policy: Arc<PolicyEngine>) -> Self {
Self { notify_fd, policy }
}
pub async fn run(&self) -> Result<(), Box<dyn Error>> {
let async_fd = AsyncFd::new(self.notify_fd)?;
loop {
// 等待通知就绪
let mut guard = async_fd.readable().await?;
// 读取通知
let notif = SeccompNotif::recv(self.notify_fd)?;
let pid = notif.pid;
let syscall_nr = notif.data.nr;
let args = notif.data.args;
log::debug!("[Notify] pid={}, syscall={}", pid, syscall_nr);
let resp = match syscall_nr {
// __NR_openat (257) - 文件访问代理
257 => self.handle_openat(¬if, args).await?,
// __NR_mount (165) - 挂载代理
165 => self.handle_mount(¬if, args).await?,
// __NR_io_uring_setup (425) - 高性能I/O
425 => self.handle_io_uring(¬if, args).await?,
// 默认:阻断未知 syscall
_ => SeccompNotifResponse::new(notif.id, -EPERM, 0),
};
notif.send_resp(self.notify_fd, &resp)?;
guard.clear_ready();
}
}
async fn handle_openat(&self, notif: &SeccompNotif,
args: [u64; 6]) -> Result<SeccompNotifResponse> {
let dirfd = args[0] as i32;
let pathname_cstr = unsafe {
CStr::from_ptr(args[1] as *const c_char)
};
let flags = args[2] as i32;
let mode = args[3] as u32;
// 安全策略:检查路径白名单
let path = pathname_cstr.to_str()?;
if !self.policy.is_path_allowed(path) {
return Ok(SeccompNotifResponse::new(notif.id, -EACCES, 0));
}
// 代执行 openat
let fd = unsafe {
libc::openat(dirfd, pathname_cstr.as_ptr(), flags, mode)
};
if fd < 0 {
Err(io::Error::last_os_error())
} else {
// 使用 SECCOMP_IOCTL_NOTIF_ADDFD 向容器注入 fd
SeccompNotifResponse::new_with_fd(notif.id, fd as u64, 0, fd)
}
}
}
异步架构支持多容器、多通知并发处理,因为 tokio 可以在单个线程上同时管理多个 notify fd 的就绪状态。
六、生产环境:Runc 与 Seccomp Notify 的集成
6.1 Runc 的 retuln 机制
runc(OCI 容器运行时参考实现)在 v1.0.0-rc95 之后完善了 rootless + seccomp notify 的完整支持,允许无特权用户运行需要特权的容器:
// config.json 片段
{
"process": {
"args": ["sleep", "infinity"]
},
"linux": {
"seccomp": {
"defaultAction": "SCMP_ACT_ERRNO",
"architectures": ["SCMP_ARCH_X86_64"],
"listenerPath": "/run/seccomp.sock",
"listenerMetadata": "oci-container-xyz",
"syscalls": [
{
"names": ["mount", "umount2"],
"action": "SCMP_ACT_NOTIFY"
}
]
}
}
}
关键配置项:
listenerPath:一个 Unix domain socket 路径。容器运行时通过该路径与独立的 supervisor 通信。
listenerMetadata:传递给 supervisor 的上下文信息(如容器 ID),用于区分不同容器的通知。
6.2 性能考量
在 seccomp notify 路径中,一次系统调用变为:
- 容器进程触发 syscall → 内核 seccomp 过滤 → 任务切换
- supervisor 调度 + 处理(含策略审计、影子执行)
- 用户态→内核态→用户态回写响应
- 容器进程恢复执行
实测开销约为 5-25 μs(取决于磁盘 I/O、syscall 复杂度),远低于 ptrace 拦截的 50-200 μs。对于非热路径的系统调用(如 mount、ptrace 调试操作),这一开销完全可以接受。但对于高频函数如 clock_gettime(每秒调用数百万次),notify 处理将成为严重瓶颈。
6.3 安全边界验证
Supervisor 必须验证的关键安全属性:
# 伪代码:安全校验策略
def validate_request(notif):
# 1. 确认通知发送方的 PID 在允许列表中
if notif.pid not in allowed_pids:
return DENY
# 2. 验证系统调用参数无注入攻击
if contains_injection(notif.data.args):
return DENY
# 3. 检查目标资源路径是否在容器命名空间内
if not is_within_container_root(notif.target_path):
return DENY
# 4. 审计日志
audit_log.write(notif)
return ALLOW
七、与替代技术的对比
| 技术 |
优势 |
劣势 |
适用场景 |
| Seccomp Notify |
精确控制+灵活代执行 |
异步开销大、单通知序列化 |
低频特权 syscall 代理 |
| ptrace |
通用 syscall 拦截 |
开销大(50-200μs/次)、时序复杂 |
调试器、strace |
| LSM (BPF LSM) |
同步决策、无上下文切换 |
无法修改参数、决策二值 |
安全策略强制执行 |
| Capability drop |
零开销 |
粒度粗、非黑名单无法阻止已知攻击 |
粗略隔离 |
| KVM/VM |
完全隔离 |
启动慢、资源开销大 |
多租户强隔离 |
| Landlock |
用户态沙箱、无特权 |
不支持 syscall 代理 |
文件系统权限限制 |
Seccomp notify 与 LSM BPF 形成互补关系:LSM BPF 处理高频、同步的安全决策(如文件访问检查),seccomp notify 处理低频、需要特权代执行的系统调用。
八、工程最佳实践
8.1 Supervisor 设计原则
- 最小化策略集:只将必要的 syscall 设为 notify,其余严格阻断或使用
SCMP_ACT_ALLOW 放行。
- id 验证:在处理任何通知前先用
SECCOMP_IOCTL_NOTIF_ID_VALID 检查 id 有效性,避免已退出的进程占用资源。
- 超时机制:为 notify 处理设置超时(建议 30s),防止恶意容器通过死锁 syscall 阻塞 supervisor。
- 并发安全:如果 supervisor 处理多个容器的 notify fd,可使用 epoll/eventfd 轮询。
- 响应注入 fd:对于返回 fd 的 syscall(
openat、pipe2 等),优先使用 SECCOMP_IOCTL_NOTIF_ADDFD 而非先 supervisor 执行再返回值(保持 fd 上下文隔离)。
8.2 容器编排集成模板
# Kubernetes Pod 通过 RuntimeClass 启用 seccomp notify
apiVersion: node.k8s.io/v1
kind: RuntimeClass
metadata:
name: seccomp-notify-runtime
handler: runc
scheduling:
nodeSelector:
seccomp-notify: "supported"
---
apiVersion: v1
kind: Pod
metadata:
annotations:
seccomp.notify.listener: "/var/run/supervisor.sock"
seccomp.notify.policy: "allow-mount-tmpfs-only"
spec:
runtimeClassName: seccomp-notify-runtime
containers:
- name: apps
image: myapp:latest
九、NewFd 注入与 io_uring 支持
Linux 5.9 引入的 SECCOMP_IOCTL_NOTIF_ADDFD 是 notify 机制的重大演进。它允许 supervisor 以"原子操作"方式同时完成:
- 打开 fd(如访问宿主机文件系统的文件)
- 将 fd 注入目标容器的 fd 表
// 使用 ADDFD 的响应构造
struct seccomp_notif_addfd addfd = {
.id = notif.id,
.flags = SECCOMP_ADDFD_FLAG_SEND, // 发送响应
.src_fd = supervisor_fd, // supervisor 打开的 fd
.newfd_flags = O_CLOEXEC,
// 不指定 newfd -> 内核自动分配
};
ioctl(notify_fd, SECCOMP_IOCTL_NOTIF_ADDFD, &addfd);
此特性特别适用于 io_uring 的注册文件(registered files)场景:container 需要大量文件描述符做零拷贝 IO,但 Seccomp 传统上阻止 io_uring_setup 和 io_uring_enter。通过 notify + ADDFD 流水线,容器可以在 supervisor 的协助下完成 fd 注册,享受完整的 kernel bypass 性能。
十、总结与展望
Seccomp notify 在容器安全体系中扮演着不可替代的角色:
- rootless containers:让无特权用户运行需要 mount、chroot 等特权操作的容器
- 用户态文件系统:允许容器在不能使用 syscall 时通过代理访问文件系统
- GPU 虚拟化:
nvidia-container-runtime 早期就是通过 notify 代理容器对 GPU 驱动节点的访问
- 安全审计:企业环境中将 syscall 代理层作为审计记录的强一致采集点
从 Linux 5.x 主线来看,SECCOMP_USER_NOTIF_FLAG_CONTINUE 的改进、addfd 注入的引入都表明这是一个持续演进中的核心基础设施。在可观测性方面,eBPF 已经接管了大多数监控路径;但在系统调用代理这一领域,seccomp notify 提供了 eBPF 无法实现的能力——即以不同权限上下文执行 syscall 并回写结果。
对于工程师而言,理解 seccomp notify 不仅有助于编写更安全的容器运行时,也为理解 Linux 安全模块的设计哲学提供了绝佳窗口:特权分离不是简单的"允许/拒绝"二元选择,而是可以根据上下文进行动态的、有策略的代执行。这正是零信任架构在内核层面的关键实现。
参考资源:
发表评论 取消回复