Linux 内核 Page Cache 与 VFS 架构深度工程实战:从 radix tree 到 writeback 全链路剖析
Linux 内核的虚拟文件系统(VFS)与 Page Cache 子系统是整个 I/O 栈的核心枢纽。理解这套机制,对于构建高性能存储服务、数据库引擎、文件服务器至关重要。本文将从 VFS 抽象层出发,深入剖析 address_space、xarray/radix tree 索引、readahead 预读算法、dirty page writeback 策略、mmap 内存映射以及直接 I/O 的完整实现,并结合实战案例展示性能调优的关键参数与常见陷阱。
一、VFS 虚拟文件系统抽象层
1.1 VFS 核心数据结构
VFS 通过一组抽象接口实现"一切皆文件"的设计哲学。核心数据结构构成了从系统调用到具体文件系统的桥梁:
// VFS 核心结构体层次关系
struct task_struct
└── struct files_struct *files // 进程的文件描述符表
└── struct fdtable *fdtable
└── struct file **fd_array // 文件对象指针数组
struct file {
struct path f_path; // vfsmount dentry
struct inode *f_inode; // 关联的 inode(缓存)
const struct file_operations *f_op; // 文件操作函数表
loff_t f_pos; // 当前文件偏移
struct address_space *f_mapping; // 页面映射(Page Cache)
void *private_data; // 文件系统私有数据
};
struct inode {
umode_t i_mode; // 文件类型和权限
struct super_block *i_sb; // 所属超级块
struct address_space *i_mapping; // inode 的页面映射
const struct inode_operations *i_op; // inode 操作
struct timespec64_atim; // 访问时间
struct timespec64_mtim; // 修改时间
struct timespec64_ctim; // 状态变化时间
loff_t i_size; // 文件大小
};
struct dentry {
struct qstr d_name; // 目录项名称
struct inode *d_inode; // 关联的 inode
struct dentry *d_parent; // 父目录项
struct hlist_bl_node d_hash; // 哈希表节点(dcache 查找)
struct list_head d_lru; // LRU 链表(unused dentry)
struct list_head d_child; // 兄弟 dentry 链表
struct list_head d_subdirs; // 子目录链表
unsigned char d_iname[DNAME_INLINE_LEN]; // 短名称内联存储
};
在这个架构中,file 对象代表一个打开的文件实例,inode 代表文件的元数据(全局唯一),dentry 代表目录项缓存(路径到 inode 的映射),super_block 代表一个已挂载的文件系统。
1.2 文件打开的完整路径
当用户调用 open() 时,内核执行以下关键步骤:
用户空间 内核空间
───────── ─────────
open("/tmp/file", O_RDWR)
│
▼
sys_openat() // 系统调用入口
│
▼
do_sys_openat2()
│
├── get_unused_fd_flags() // 分配文件描述符
├── do_filp_open() // 核心路径解析
│ ├── path_openat() // 路径遍历
│ │ ├── link_path_walk() // 逐级解析路径组件
│ │ │ ├── ->lookup() // 文件系统中查找 inode
│ │ │ └── d_lookup() // 或从 dcache 命中
│ │ └── complete_walk() // 完成 lookup,持有 inode
│ └── do_open() // 创建 file 对象
│ └── fops->open() // 调用文件系统特定 open
└── fd_install() // 安装 fd 到进程表
其中 link_path_walk() 是整个路径解析中最复杂的部分。它需要通过 dcache(目录项缓存)逐级查找路径组件,而 dcache 使用哈希表 LRU 的方式组织,以保证快速查找与合理的内存占用。
1.3 文件描述符表与并发安全
现代多线程程序中,文件描述符表的并发访问是关键设计考量:
// 进程的文件描述符表
struct files_struct {
atomic_t count; // 引用计数(支持 CLONE_FILES)
struct fdtable __rcu *fdt; // 当前 fd 表(RCU 保护)
struct fdtable fdtab; // 初始内嵌表(通常 64 个)
};
struct fdtable {
unsigned int max_fds; // 最大 fd 数量
struct file __rcu **fd; // 文件对象指针数组
unsigned long *open_fds; // 位图:已打开的 fd
unsigned long *close_on_exec; // 位图:exec 时关闭的 fd
};
// fd 分配算法(select_next_fd):
// 1. 从 last_fd 1 开始扫描 open_fds 位图
// 2. 找到第一个 0 位即为空闲 fd
// 3. 使用 BSF(Bit Scan Forward)硬件指令加速
// 4. 如果超过 max_fds,触发 fdtable 扩容(翻倍)
重要:fdtable 使用 RCU(Read-Copy Update)保护。读端(如
read())可以直接解引用fd[fd]而不需要加锁,只有close()才需要spin_lock配合 RCU 延迟释放。这意味着高频 I/O 场景下,文件描述的获取几乎没有同步开销。
二、Page Cache:address_space 与 xarray
2.1 address_space:Page Cache 的管理中枢
address_space 是 Page Cache 的核心管理结构,每个 inode 的 i_mapping 指向它。它维护了属于该文件的所有缓存页面:
struct address_space {
struct inode *host; // 所属 inode
struct xarray i_pages; // 页面索引(xarray,替代旧版 radix tree)
gfp_t gfp_mask; // 页面分配时的 GFP 标志
atomic_t i_mmap_writable; // mmap 写映射计数
struct rb_root_cached i_mmap; // mmap 区域的红黑树
struct rw_semaphore i_mmap_rwsem; // mmap 读写信号量
unsigned long nrpages; // 缓存中的页面总数
const struct address_space_operations *a_ops; // 操作函数表
unsigned long flags; // 状态标志(如 AS_EIO)
atomic_t private_lock; // private 数据锁
void *private_data; // 文件系统私有数据
};
struct address_space_operations {
int (*writepage)(struct page *, struct writeback_control *); // 写单个页面
int (*read_folio)(struct file *, struct folio *); // 读取一页
int (*writepages)(struct address_space *, struct writeback_control *); // 批量写
int (*dirty_folio)(struct address_space *, struct folio *); // 标记脏页
int (*write_begin)(struct file *, struct address_space *, loff_t,
unsigned len, unsigned flags, struct folio **, void **);
int (*write_end)(struct file *, struct address_space *, loff_t,
unsigned len, unsigned copied, struct folio *, void *);
sector_t (*bmap)(struct address_space *, sector_t); // 逻辑到物理块映射
void (*invalidate_folio)(struct folio *, size_t, size_t); // 使缓存失效
bool (*direct_IO)(struct kiocb *, struct iov_iter *); // 直接 I/O
int (*migrate_folio)(struct address_space *, struct folio *, struct folio *, enum migrate_mode);
int (*swap_activate)(struct swap_info_struct *, struct file *, sector_t *);
};
2.2 xarray:高效的页面索引结构
Linux 5.1 引入 xarray 替代旧版 radix tree,解决了 radix tree 在标记(marks/tag)扩展性方面的不足,同时保持了 O(log n) 的查找性能:
/*
* xarray 核心原理
* ┌─────────────────────────────────────────────────┐
* │ 根节点(XA_CHUNK_SIZE = 64) │
* ├────┬────┬────┬────┬────┬────┬────┬────┬────┬────┤
* │ S0 │ S1 │ S2 │ ...│ 页面指针│ ...│ S62│ S63│
* ├────┴────┴────┴────┼──────────┴────┴────┴────┘
* │ Slot 存储: │
* │ - 页面指针 (PFN) │ ← 文件偏移 / PAGE_SIZE
* │ - 指针特殊值 (NULL) │ ← 预读占位(XA_RETRY_ENTRY)
* │ - 标记位 (marks) │ ← DIRTY / TOWRITE /标记
* └─────────────────────┘
* 高度 h 决定能索引的文件大小:
* 高度 0:64 页 = 256 KB
* 高度 1:4096 页 = 16 MB
* 高度 2:262144 页 = 1 GB
* 高度 3:~16 TB(64位系统)
*/
// xarray 查找页面
struct page *find_get_page(struct address_space *mapping,unsigned long index) {
struct xa_state xas;
struct page *page;
xa_lock(

发表评论 取消回复