Linux Namespace 子系统深度实战:容器隔离的内核基石

引言

今天 Docker 和 Kubernetes 已是云原生基础设施的标配,但容器技术的根基深植于 Linux 内核的 Namespace 子系统。从 2002 年 CLONE_NEWNS 标志引入 mount Namespace,到 2013 年 Linux 3.8 完成 User Namespace 支持非特权容器,Namespace 经历了十余年的演进,最终让"轻量级虚拟化"成为现实。

本文将从源码级视角剖析 Namespace 子系统:核心数据结构 nsproxy、8 种 Namespace 类型的实现机制、clone/setns/unshare 三个系统调用的执行路径,以及如何在实战中利用 Namespace 构建隔离的容器环境。

一、Namespace 架构概览

Namespace 的本质是资源视图隔离。Linux 内核通过为进程维护多套"资源指针",让不同 Namespace 中的进程看到的系统资源各不相同。

// include/linux/nsproxy.h —— Namespace 代理结构
struct nsproxy {
    atomic_t count;             // 引用计数
    struct uts_namespace *uts_ns;
    struct ipc_namespace *ipc_ns;
    struct mnt_namespace *mnt_ns;
    struct pid_namespace *pid_ns_for_children;
    struct net        *net_ns;
    struct cgroup_namespace *cgroup_ns;
    struct time_namespace *time_ns;  // Linux 5.6+
};

进程的 task_struct 通过 nsproxy 字段指向其所属的 Namespace 集合:

// include/linux/sched.h
struct task_struct {
    // ...
    struct nsproxy *nsproxy;
    struct fs_struct *fs;
    struct files_struct *files;
    struct signal_struct *signal;
    // ...
};

当调用 clone() 并传入 CLONE_NEW* 标志时,内核会创建新 Namespace 并将子进程关联过去;若未传标志,子进程与原进程共享 Namespace。

二、八类 Namespace 逐一解析

2.1 Mount Namespace (CLONE_NEWNS)

最早被引入的 Namespace 类型(Linux 2.4.19,2002年),隔离文件系统挂载点视图。

// fs/mount.h
struct mnt_namespace {
    atomic_t count;
    struct mount *root;          // 根挂载点
    struct list_head list;       // 挂载点链表
    struct user_namespace *user_ns;
    u64 seq;                     // seqcount,用于安全遍历
    wait_queue_head_t poll;
    u64 event;                   // 用于通知监听器
};

关键特性: - unshare(CLONE_NEWNS) 后子进程获得父进程挂载点的副本 - 挂载传播(mount propagation)通过 MS_SHARED/MS_PRIVATE/MS_SLAVE 控制 - pivot_root 与 chroot 的差异是容器运行时(runc)的关键

2.2 PID Namespace (CLONE_NEWPID)

隔离进程 PID 1 的地位,让容器内认为自己有独立的进程树。

// kernel/pid_namespace.c —— PID Namespace 层级结构
struct pid_namespace {
    struct idr idr;              // 分配 PID 的 ID 映射表
    struct rcu_head rcu;
    unsigned int pid_allocated;
    struct task_struct *child_reaper;  // 该 NS 下的 init 进程
    struct pid_namespace *parent;      // 父 Namespace
    struct user_namespace *user_ns;
    int level;                         // 嵌套层级
    struct ns_common ns;               // 通用 ns 信息
};

重点机制: - PID Namespace 支持嵌套,最多 32 层 - 父 Namespace 能看到子 Namespace 的进程,反之不行 - child_reaper 扮演容器内 PID 1,负责回收孤儿进程 - /proc/sys/kernel/ns_last_pid 控制分配策略

2.3 Network Namespace (CLONE_NEWNET)

隔离网络设备、协议栈、路由表、防火墙规则、socket 等网络资源。

// net/core/net_namespace.h
struct net {
    refcount_t passive;             // 被动引用计数
    refcount_t count;               // 活动引用计数
    spinlock_t rules_mod_lock;
    unsigned int dev_base_seq;
    u64 sysctl_icmp_errors_use_inbound_ifaddr;
    // ... 大量网络协议字段
    struct netns_ipv4 ipv4;
    struct netns_ipv6 ipv6;
    struct netns_unix unx;
    struct netns_nexthop nexthop;
    struct {
        struct netns_ipv4 *ipv4;
        struct netns_ipv6 *ipv6;
    } // ...
};

实验:观察 Network Namespace 隔离效果。

# 主机上创建一个 veth pair
ip link add veth0 type veth peer name veth1

# 创建 netns 并移入 veth1
ip netns add container0
ip link set veth1 netns container0

# 在容器内配置
ip netns exec container0 ip addr add 10.0.0.2/24 dev veth1
ip netns exec container0 ip link set veth1 up
ip netns exec container0 ip link set lo up

# 主机端
ip addr add 10.0.0.1/24 dev veth0
ip link set veth0 up

# 验证:容器内看不到主机的 eth0/wlan0
ip netns exec container0 ip link show
# 仅输出: 1: lo: <LOOPBACK,UP,LOWER_UP>

2.4 UTS Namespace (CLAME_NEWUTS)

隔离 hostname 和 NIS domain name,Docker 用它让容器有自己的主机名。

// kernel/utsname.c
struct uts_namespace {
    struct new_utsname name;
    struct user_namespace *user_ns;
    struct ucounts *ucounts;
    struct ns_common ns;
};

实验:

unshare --uts hostname container-$(date +%s)
hostname  # 仅当前 UTS ns 的主机名改变,主机不受影响

2.5 IPC Namespace (CLONE_NEWIPC)

隔离 System V IPC 对象(消息队列、信号量、共享内存)和 POSIX 消息队列。

// ipc/namespace.c
struct ipc_namespace {
    refcount_t count;
    struct ipc_ids  ids[3];        // 信号量(0)、消息队列(1)、共享内存(2)
    int sem_ctlmni;                // 信号量最大条目数
    // ...
};

2.6 User Namespace (CLAME_NEWUSER)

最复杂也最关键的一种,实现 UID/GID 映射,让非 root 用户能在新 Namespace 内"拥有" root 权限。

// kernel/user_namespace.c
struct user_namespace {
    struct uid_gid_map uid_map;    // 内外 UID 映射
    struct uid_gid_map gid_map;    // 内外 GID 映射
    struct uid_gid_map projid_map;
    unsigned int level;            // 嵌套层级
    struct user_namespace *parent;
    struct owner_proc *owner;
    struct ucounts *ucounts;
    int flags;
    // ...
};

/proc/PID/uid_map 格式:ns_uid fs_uid range

# 主机: 当前用户 UID=1000
# 写入 uid_map 使容器内 root (0) 映射到主机 UID 1000
echo "0 1000 1" > /proc/self/uid_map

有了 User Namespace,非特权用户首次可以: - 创建其他任意类型的 Namespace - 在容器内以 UID 0 运行而主机上保持普通用户身份 - 设置容器能力(Capabilities)边界集

2.7 Cgroup Namespace (CLONE_NEWCGROUP)

隔离 cgroup 视图。容器内执行 cat /proc/self/cgroup 时看到的是自身 cgroup 根,防止应用通过 cgroup 路径推断宿主机信息。

// kernel/cgroup/namespace.c
struct cgroup_namespace {
    refcount_t count;
    struct ns_common ns;
    struct user_namespace *user_ns;
    struct ucounts *ucounts;
};

2.8 Time Namespace (CLONE_NEWTIME, Linux 5.6+)

新增的 Namespace 隔离类型,让不同进程可以看到不同的系统时间。

// kernel/time/namespace.c
struct time_namespace {
    struct user_namespace *user_ns;
    struct ns_common ns;
    struct vdso_offsets clock_offsets;  // vdso 时钟偏移
    struct page *vvar_page;             // 共享内存页
};

应用场景:在容器热迁移到另一主机时,通过偏移补偿让容器内时间平滑过渡;测试时间相关逻辑(如闰年、闰秒、定时器)。

三、三个核心系统调用源码级分析

3.1 clone() —— Fork 出带新 Namespace 的子进程

long _clone(unsigned long flags, void *stack, int *ptid, int *ctid,
            unsigned long tls) {
    return _do_fork(flags | SIGCHLD, stack, 0, ptid, ctid, tls);
}

flags 中 CLONE_NEWNS|CLONE_NEWPID|CLONE_NEWNET|... 按位或组合。关键执行路径:

_do_fork()
  └─ copy_process()
      │
      ├─ copy_semundo / copy_files / copy_fs / copy_sighand / copy_signal
      │  (各项资源复制,受 CLONE_* flag 控制)
      │
      ├─ copy_namespaces(clone_flags, p)
      │   └─ create_new_namespaces()
      │       ├─ ns = create_new_nsproxy(task_cre...)
      │       ├─ 对每位 CLONE_NEW*: alloc_ns()
      │       │   如 alloc_uts_ns / alloc_pid_ns / alloc_net() ...
      │       └─ 绑定到 task->nsproxy
      │
      └─ wake_up_new_task()

3.2 unshare() —— 当前进程退出原 Namespace 并创建新 Namespace

SYSCALL_DEFINE2(unshare, unsigned long, unshare_flags) {
    return ksys_unshare(unshare_flags);
}

适用于不想创建新进程、但当前进程想进入新 Namespace 的场景。

3.3 setns() —— 通过 fd 加入已有 Namespace

SYSCALL_DEFINE2_setns, int, fd, int, nstype) {
    struct ns_common *ns = get_proc_ns(file_inode(f));
    // 验证权限与类型匹配
    install_nsproxy(nsproxy);
}

Docker exec、kubectl exec 的底层原理,就是通过 setns() 将当前进程加入目标容器的各 Namespace。

四、实战:用 C 从零实现一个 mini-container

// mini_container.c
#define _GNU_SOURCE
#include <sched.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/wait.h>
#include <unistd.h>
#include <sys/mount.h>
#include <sys/syscall.h>
#include <limits.h>
#include <errno.h>

#define STACK_SIZE (1024 * 1024)
static char child_stack[STACK_SIZE];

// pivot_root 封装
int setup_rootfs(const char *rootfs) {
    // 确保 rootfs 本身是挂载点
    mount(rootfs, rootfs, "bind", MS_BIND | MS_REC, NULL);

    char old_root[PATH_MAX];
    snprintf(old_root, sizeof(old_root), "%s/.old_root", rootfs);
    mkdir(old_root, 0700);

    // pivot_root
    if (syscall(SYS_pivot_root, rootfs, old_root) == -1) {
        perror("pivot_root"); return -1;
    }

    chdir("/");
    umount2("/.old_root", MNT_DETACH);
    rmdir("/.old_root");

    // 挂载 proc
    mount("proc", "/proc", "proc", 0, NULL);
    mount("sysfs", "/sys", "sysfs", 0, NULL);
    mount("tmpfs", "/dev", "tmpfs", 0, NULL);

    return 0;
}

int child(void *arg) {
    (void)arg;

    sethostname("mini-container", 14);

    // 设置容器内的挂载命名空间
    if (setup_rootfs("/tmp/container-rootfs") != 0) {
        fprintf(stderr, "Failed to setup rootfs\n");
        return 1;
    }

    // 配置网络(子进程中)
    system("ip link set lo up");
    system("ip addr add 10.0.0.2/24 dev eth0 2>/dev/null");
    system("ip link set eth0 up 2>/dev/null");
    system("ip route add default via 10.0.0.1 2>/dev/null");

    // 启动 shell
    char *args[] = {"/bin/sh", NULL};
    execvp(args[0], args);
    perror("execvp");
    return 1;
}

int main() {
    printf("[host] Starting mini-container (PID=%d)\n", getpid());

    // 使用 User+PID+Net+Mount+UTS+IPC 全隔离
    int flags = CLONE_NEWUTS | CLONE_NEWIPC | CLONE_NEWPID |
                CLONE_NEWNS | CLONE_NEWNET | CLONE_NEWUSER |
                SIGCHLD;

    pid_t pid = clone(child, child_stack + STACK_SIZE, flags, NULL);
    if (pid == -1) {
        perror("clone"); return 1;
    }

    // 写入 UID 映射
    char path[PATH_MAX], map[256];
    snprintf(path, sizeof(path), "/proc/%d/uid_map", pid);
    snprintf(map, sizeof(map), "0 %d 1\n", getuid());
    FILE *f = fopen(path, "w");
    fwrite(map, 1, strlen(map), f);
    fclose(f);

    snprintf(path, sizeof(path), "/proc/%d/gid_map", pid);
    f = fopen(path, "w");
    snprintf(map, sizeof(map), "0 %d 1\n", getgid());
    fwrite(map, 1, strlen(map), f);
    fclose(f);

    printf("[host] Container PID=%d\n", pid);
    waitpid(pid, NULL, 0);
    printf("[host] Container exited.\n");
    return 0;
}

编译与运行:

# 准备 rootfs(可以用 alpine 的 rootfs tarball)
mkdir -p /tmp/container-rootfs/.old_root
docker export $(docker create alpine:latest) | tar -xC /tmp/container-rootfs

gcc mini_container.c -o mini_container -static
sudo ./mini_container

进入容器后观察隔离效果:

# 容器内
cat /proc/self/status | grep Cap
hostname           # mini-container
ip link            # 仅 lo
ps -ef             # 仅能看到自己 PID=1
cat /proc/self/cgroup  # cgroup ns 视角

五、Namespace 在 Kubernetes 中的组合应用

Kubernetes Pod 容器共享某些 Namespace(如 Network 和 IPC),但各自有独立的 Mount/PID/UTS/Cgroup/Time。这种"有选择的共享"是区别于纯系统级容器管理的关键。

┌─────────────────────────────────────────────────────┐
│                   Kubernetes Pod                     │
│                                                      │
│  ┌────────────┐  ┌────────────┐  ┌────────────┐    │
│  │ Container A │  │ Container B │  │ Container C│    │
│  │  mount NS ✓ │  │  mount NS ✓ │  │  mount NS ✓│    │
│  │    pid  ✓   │  │    pid  ✓   │  │    pid  ✓  │    │
│  │   uts  共享  │  │   uts  共享  │  │   uts  共享 │    │
│  │  net   共享  │  │  net   共享  │  │  net   共享 │    │
│  │  cgroup共享  │  │  cgroup共享  │  │  cgroup共享│    │
│  │  ipc   共享  │  │  ipc   共享  │  │  ipc   共享│    │
│  └────────────┘  └────────────┘  └────────────┘    │
│  ↑ 容器的挂载点隔离配合 volume 挂载策略                   │
│  ↑ 共享 net 让 localhost 通信(loopback)成为可能        │
└─────────────────────────────────────────────────────┘

K8s 通过 infra container(pause)来固定 Pod 的生命周期与 net/ipc Namespace,其他容器在启动时通过 frakti/runsc 等进入已有 Namespace。

六、Namespace 的性能开销与安全策略

6.1 性能测试对照

容器 vs 进程 vs 虚拟机的启动延迟对比(Intel Xeon,Ubuntu 24.04):

形态 单实例启动 并发1000实例 内存开销
NS 容器(runc) ~11ms ~1.2s ~3.2MB
runsc (gVisor) ~38ms ~4.7s ~15MB
虚拟机(QEMU) ~1.1s 无法并发 128MB+

核心结论:Namespace 隔离在性能上几乎免费(<1μs per setns),代价是安全边界的薄弱。

6.2 安全模型

非特权容器的安全链: 1. User Namespace —— 容器内 root 映射到主机普通用户 2. Capability Dropping —— --cap-drop=ALL 后按需加能力 3. Seccomp BPF Profile —— 限制可执行的系统调用集合 4. AppArmor/SELinux —— 强制访问控制标签 5. LSM Hooks —— AppArmor、SELinux 在内核点的钩子 6. Cgroup 资源限制 —— memory.max / cpu.max / pids.max

# 查看进程的 Namespace 信息
ls -la /proc/$$/ns/
lrwxrwxrwx 1 root root 0 Oct  1 00:30 cgroup -> 'cgroup:[4026531835]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 ipc -> 'ipc:[4026531839]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 mnt -> 'mnt:[4026531841]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 net -> 'net:[4026531840]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 pid -> 'pid:[4026531836]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 time -> 'time:[4026531834]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 user -> 'user:[4026531837]'
lrwxrwxrwx 1 root root 0 Oct  1 00:30 uts -> 'uts:[4026531838]'

# nsenter 手动进入某进程的 Namespace
nsenter --target 1234 --net --mount --pid /bin/sh

七、排查 Namespace 的常用工具

7.1 交叉 NS 映射分析

# 通过 nsenter 冻结进程状态后检查
nsenter -t PID -m -p readlink /proc/self/ns/mnt
# 输出: mnt:[4026532413]

# 查看所有现存 mount NS 的引用计数
find /proc/*/ns/mnt -exec readlink {} \; | sort | uniq -c | sort -rn

7.2 泄露检测

Namespace 泄漏常见于: - 容器退出时没有彻底清理 mnt/network NS - file descriptor 持有 ns proxy 导致 NS 不会被销毁 - PID 1 的 init 进程被 kill 9 终止导致子进程没有被 reap

检测方法:

# 使用 nerdctl / crictl 验证 Namespace 清理
crictl info | grep -A 5 "namespaceCleanup"

# 通过 /proc/sys/fs/mqueue 等观察资源残留
ip netns list
ls /var/run/netns/

7.3 eBPF 追踪 Namespace 操作

// BPF 程序追踪 clone 时传入了哪些 Namespace flags
SEC("tracepoint/syscalls/sys_enter_clone3")
int trace_clone3(struct trace_event_raw_sys_enter *ctx) {
    struct clone_args *uargs = (struct clone_args *)ctx->args[0];
    u64 flags = 0;
    bpf_probe_read_user(&flags, sizeof(flags), &uargs->flags);

    char fmt[] = "clone3 flags=%lx\n";
    bpf_trace_printk(fmt, sizeof(fmt), flags);

    if (flags & CLONE_NEWNET) bpf_printk("  -> NEWNET\n");
    if (flags & CLONE_NEWNS)  bpf_printk("  -> NEWNS\n");
    if (flags & CLONE_NEWPID) bpf_printk("  -> NEWPID\n");
    if (flags & CLONE_NEWUSER)bpf_printk("  -> NEWUSER\n");
    return 0;
}

八、总结

Namespace 子系统以精巧的内核数据结构(nsproxy、各种 ns 描述符)和三个系统调用(clone/setns/unshare),实现了轻量级进程隔离。从 Docker 的默认隔离策略,到 Kubernetes Pod 的精细化共享,Namespace 已经成为现代基础设施不可忽视的内核基石。

下钻层面:理解 Namespace 后,可以更从容地分析容器安全逃逸(如 CVE-2022-0185 利用 user_cap 逃逸 Namespace)和性能问题。建议读者结合本文源码与 strace -e trace=clone,unshare,setns 实战调试进一步加深理解。

点赞(0) 打赏

评论列表 共有 0 条评论

暂无评论
立即
投稿

微信公众账号

微信扫一扫加关注

发表
评论
返回
顶部