Linux 内核 Page Cache 与 VFS 架构深度工程实战:从 radix tree 到 writeback 全链路剖析

Linux 内核的虚拟文件系统(VFS)与 Page Cache 子系统是整个 I/O 栈的核心枢纽。理解这套机制,对于构建高性能存储服务、数据库引擎、文件服务器至关重要。本文将从 VFS 抽象层出发,深入剖析 address_space、xarray/radix tree 索引、readahead 预读算法、dirty page writeback 策略、mmap 内存映射以及直接 I/O 的完整实现,并结合实战案例展示性能调优的关键参数与常见陷阱。

一、VFS 虚拟文件系统抽象层

1.1 VFS 核心数据结构

VFS 通过一组抽象接口实现"一切皆文件"的设计哲学。核心数据结构构成了从系统调用到具体文件系统的桥梁:

// VFS 核心结构体层次关系
struct task_struct
  └── struct files_struct *files      // 进程的文件描述符表
        └── struct fdtable *fdtable
              └── struct file **fd_array  // 文件对象指针数组

struct file {
    struct path              f_path;       //  vfsmount   dentry
    struct inode            *f_inode;      // 关联的 inode(缓存)
    const struct file_operations *f_op;    // 文件操作函数表
    loff_t                   f_pos;        // 当前文件偏移
    struct address_space    *f_mapping;    // 页面映射(Page Cache)
    void                    *private_data;  // 文件系统私有数据
};

struct inode {
    umode_t                  i_mode;       // 文件类型和权限
    struct super_block      *i_sb;         // 所属超级块
    struct address_space    *i_mapping;    // inode 的页面映射
    const struct inode_operations *i_op;   // inode 操作
    struct timespec64_atim;               // 访问时间
    struct timespec64_mtim;               // 修改时间
    struct timespec64_ctim;               // 状态变化时间
    loff_t                   i_size;       // 文件大小
};

struct dentry {
    struct qstr              d_name;       // 目录项名称
    struct inode            *d_inode;      // 关联的 inode
    struct dentry           *d_parent;     // 父目录项
    struct hlist_bl_node     d_hash;       // 哈希表节点(dcache 查找)
    struct list_head         d_lru;        // LRU 链表(unused dentry)
    struct list_head         d_child;      // 兄弟 dentry 链表
    struct list_head         d_subdirs;    // 子目录链表
    unsigned char d_iname[DNAME_INLINE_LEN]; // 短名称内联存储
};

在这个架构中,file 对象代表一个打开的文件实例,inode 代表文件的元数据(全局唯一),dentry 代表目录项缓存(路径到 inode 的映射),super_block 代表一个已挂载的文件系统。

1.2 文件打开的完整路径

当用户调用 open() 时,内核执行以下关键步骤:

用户空间                    内核空间
─────────                  ─────────
open("/tmp/file", O_RDWR)
    │
    ▼
sys_openat()                    // 系统调用入口
    │
    ▼
do_sys_openat2()
    │
    ├── get_unused_fd_flags()   // 分配文件描述符
    ├── do_filp_open()          // 核心路径解析
    │   ├── path_openat()       // 路径遍历
    │   │   ├── link_path_walk() // 逐级解析路径组件
    │   │   │   ├── ->lookup()   // 文件系统中查找 inode
    │   │   │   └── d_lookup()   // 或从 dcache 命中
    │   │   └── complete_walk()  // 完成 lookup,持有 inode
    │   └── do_open()           // 创建 file 对象
    │       └── fops->open()    // 调用文件系统特定 open
    └── fd_install()            // 安装 fd 到进程表

其中 link_path_walk() 是整个路径解析中最复杂的部分。它需要通过 dcache(目录项缓存)逐级查找路径组件,而 dcache 使用哈希表 LRU 的方式组织,以保证快速查找与合理的内存占用。

1.3 文件描述符表与并发安全

现代多线程程序中,文件描述符表的并发访问是关键设计考量:

// 进程的文件描述符表
struct files_struct {
    atomic_t count;                // 引用计数(支持 CLONE_FILES)
    struct fdtable __rcu *fdt;     // 当前 fd 表(RCU 保护)
    struct fdtable fdtab;          // 初始内嵌表(通常 64 个)
};

struct fdtable {
    unsigned int max_fds;          // 最大 fd 数量
    struct file __rcu **fd;        // 文件对象指针数组
    unsigned long *open_fds;       // 位图:已打开的 fd
    unsigned long *close_on_exec;  // 位图:exec 时关闭的 fd
};

// fd 分配算法(select_next_fd):
// 1. 从 last_fd 1 开始扫描 open_fds 位图
// 2. 找到第一个 0 位即为空闲 fd
// 3. 使用 BSF(Bit Scan Forward)硬件指令加速
// 4. 如果超过 max_fds,触发 fdtable 扩容(翻倍)

重要:fdtable 使用 RCU(Read-Copy Update)保护。读端(如 read())可以直接解引用 fd[fd] 而不需要加锁,只有 close() 才需要 spin_lock 配合 RCU 延迟释放。这意味着高频 I/O 场景下,文件描述的获取几乎没有同步开销。

二、Page Cache:address_space 与 xarray

2.1 address_space:Page Cache 的管理中枢

address_space 是 Page Cache 的核心管理结构,每个 inode 的 i_mapping 指向它。它维护了属于该文件的所有缓存页面:

struct address_space {
    struct inode        *host;           // 所属 inode
    struct xarray        i_pages;         // 页面索引(xarray,替代旧版 radix tree)
    gfp_t                gfp_mask;        // 页面分配时的 GFP 标志
    atomic_t             i_mmap_writable; // mmap 写映射计数
    struct rb_root_cached i_mmap;        // mmap 区域的红黑树
    struct rw_semaphore  i_mmap_rwsem;    // mmap 读写信号量
    unsigned long        nrpages;         // 缓存中的页面总数
    const struct address_space_operations *a_ops; // 操作函数表
    unsigned long        flags;           // 状态标志(如 AS_EIO)
    atomic_t             private_lock;    // private 数据锁
    void                *private_data;    // 文件系统私有数据
};

struct address_space_operations {
    int (*writepage)(struct page *, struct writeback_control *);     // 写单个页面
    int (*read_folio)(struct file *, struct folio *);                 // 读取一页
    int (*writepages)(struct address_space *, struct writeback_control *); // 批量写
    int (*dirty_folio)(struct address_space *, struct folio *);       // 标记脏页
    int (*write_begin)(struct file *, struct address_space *, loff_t,
                       unsigned len, unsigned flags, struct folio **, void **);
    int (*write_end)(struct file *, struct address_space *, loff_t,
                     unsigned len, unsigned copied, struct folio *, void *);
    sector_t (*bmap)(struct address_space *, sector_t);              // 逻辑到物理块映射
    void (*invalidate_folio)(struct folio *, size_t, size_t);         // 使缓存失效
    bool (*direct_IO)(struct kiocb *, struct iov_iter *);            // 直接 I/O
    int (*migrate_folio)(struct address_space *, struct folio *, struct folio *, enum migrate_mode);
    int (*swap_activate)(struct swap_info_struct *, struct file *, sector_t *);
};

2.2 xarray:高效的页面索引结构

Linux 5.1 引入 xarray 替代旧版 radix tree,解决了 radix tree 在标记(marks/tag)扩展性方面的不足,同时保持了 O(log n) 的查找性能:

/*
 * xarray 核心原理
 * ┌─────────────────────────────────────────────────┐
 * │              根节点(XA_CHUNK_SIZE = 64)          │
 * ├────┬────┬────┬────┬────┬────┬────┬────┬────┬────┤
 * │ S0 │ S1 │ S2 │ ...│ 页面指针│ ...│ S62│ S63│
 * ├────┴────┴────┴────┼──────────┴────┴────┴────┘
 * │     Slot 存储:     │
 * │  - 页面指针 (PFN)    │  ← 文件偏移 / PAGE_SIZE
 *  │  - 指针特殊值 (NULL)  │  ← 预读占位(XA_RETRY_ENTRY)
 * │  - 标记位 (marks)    │  ← DIRTY / TOWRITE /标记
 * └─────────────────────┘
 *  高度 h 决定能索引的文件大小:
 *    高度 0:64 页 = 256 KB
 *    高度 1:4096 页 = 16 MB
 *    高度 2:262144 页 = 1 GB
 *    高度 3:~16 TB(64位系统)
 */

// xarray 查找页面
struct page *find_get_page(struct address_space *mapping,unsigned long index) {
    struct xa_state xas;
    struct page *page;

    xa_lock(                        
                    
点赞(0) 打赏

评论列表 共有 0 条评论

暂无评论
立即
投稿
网站二维码

微信公众账号

微信扫一扫加关注

发表
评论
返回
顶部