Linux 内核 VFS 虚拟文件系统深度实战:从 syscall 到 disk 的完整路径
VFS(Virtual File System)是 Linux 内核中最庞大、最核心的子系统之一。它为所有文件系统提供了统一的抽象层,使得 ext4、xfs、procfs、sysfs、tmpfs 等截然不同的文件系统能够以相同的方式被应用程序访问。本文将从系统调用入口出发,沿着 VFS 的完整调用链深入剖析 inode、dentry、file、superblock 四大核心对象的实现机制与交互关系。
一、VFS 的设计哲学与架构总览
VFS 存在的根本原因是要解决多文件系统统一访问的问题。在没有 VFS 的操作系统中,程序必须知道文件存储在什么类型的文件系统上,并调用对应的 API。而 VFS 通过引入一组通用的数据结构和函数指针,让所有文件系统遵循相同的接口。
VFS 的设计可以概括为四个核心对象 + 一组操作函数表:
| 对象 | 结构体 | 定位 | 生命周期 |
|---|---|---|---|
| 超级块 | super_block | 一个挂载的文件系统 | 挂载到卸载 |
| 索引节点 | inode | 一个文件(元数据) | 创建到删除 |
| 目录项 | dentry | 一个路径分量 | 查找后缓存 |
| 文件对象 | file | 一个打开的文件实例 | open 到 close |
这四者之间的关系可以用一句话概括:super_block 描述整个文件系统,inode 描述一个具体文件,dentry 加速路径查找,file 代表一个进程打开的文件状态。
二、超级块(SuperBlock):文件系统的骨架
super_block 代表一个已挂载的文件系统实例。每当执行 mount 操作时,内核会分配并初始化一个 super_block。
// include/linux/fs.h (简化版)
struct super_block {
struct list_head s_list; // 全局超级块链表
dev_t s_dev; // 设备标识符
unsigned long s_blocksize; // 块大小(字节)
loff_t s_maxbytes; // 最大文件大小
struct file_system_type *s_type; // 文件系统类型指针
const struct super_operations *s_op; // 超级块操作函数表
struct list_head s_inodes; // 该文件系统的所有 inode 链表
struct list_head s_dirty; // 脏 inode 链表
struct list_head s_files; // 所有打开的文件对象
void *s_fs_info; // 文件系统私有数据(如 ext4_sb_info)
// ...
};
// 超级块操作函数表(由具体文件系统实现)
struct super_operations {
struct inode *(*alloc_inode)(struct super_block *sb);
void (*destroy_inode)(struct inode *);
void (*dirty_inode)(struct inode *, int flags);
int (*write_inode)(struct inode *, struct writeback_control *wbc);
void (*evict_inode)(struct inode *);
void (*put_super)(struct super_block *);
int (*statfs)(struct dentry *, struct kstatfs *);
int (*remount_fs)(struct super_block *, int *, char *);
// ...
};
mount 的过程实际上是文件系统类型驱动中的 fill_super 回调被调用的过程:
- 用户执行 mount -t ext4 /dev/sda1 /mnt
- 内核调用 get_fs_type("ext4") 查找 file_system_type
- 分配 super_block,调用 sget() 或 mount_bdev()
- 调用具体文件系统的 fill_super() 回调,读取磁盘上的超级块信息
- 创建根 inode 和根 dentry,将 super_block.s_root 指向它们
- 将 super_block 加入全局链表 super_blocks
三、索引节点(inode):文件的元数据容器
inode 存储了文件除名称之外的所有元数据。在 VFS 中,每个文件都有一个唯一的 inode 编号(ino_t),它是文件在文件系统中的唯一标识。
// include/linux/fs.h (简化版)
struct inode {
umode_t i_mode; // 文件权限与类型(S_IFREG/S_IFDIR...)
uid_t i_uid; // 所有者 UID
gid_t i_gid; // 所属组 GID
kuid_t i_uid_k; // 命名空间感知 UID
const struct inode_operations *i_op; // inode 操作函数表
const struct file_operations *i_fop; // 文件操作函数表(缓存)
struct super_block *i_sb; // 所属超级块
struct address_space *i_mapping; // 页缓存映射
struct inode *i_hash; // inode hash 表链表
loff_t i_size; // 文件大小
struct timespec64 i_atime; // 访问时间
struct timespec64 i_mtime; // 修改时间
struct timespec64 i_ctime; // 状态变更时间
unsigned long i_ino; // inode 编号
union {
struct pipe_inode_info *i_pipe; // 管道
struct block_device *i_bdev; // 块设备
struct cdev *i_cdev; // 字符设备
};
// ...
struct mutex i_rwsem; // inode 读写信号量(保护元数据)
rwlock_t i_lock; // inode 自旋锁
};
inode_operations 定义了作用于文件元数据的操作:
struct inode_operations {
struct dentry *(*lookup)(struct inode *, struct dentry *, unsigned int);
const char *(*get_link)(struct dentry *, struct inode *, struct delayed_call *);
int (*permission)(struct inode *, int);
int (*create)(struct inode *, struct dentry *, umode_t, bool);
int (*link)(struct dentry *, struct inode *, struct dentry *);
int (*unlink)(struct inode *, struct dentry *);
int (*symlink)(struct inode *, struct dentry *, const char *);
int (*mkdir)(struct inode *, struct dentry *, umode_t);
int (*rmdir)(struct inode *, struct dentry *);
int (*mknod)(struct inode *, struct dentry *, umode_t, dev_t);
int (*rename)(struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int);
int (*setattr)(struct dentry *, struct iattr *);
// ...
};
3.1 inode 分配与缓存
分配新 inode 的过程涉及以下关键步骤:
- 调用 super_operations.alloc_inode() 分配内存(通常通过 SLAB 缓存)
- 初始化 inode 的 i_ino、i_mode、i_uid/i_gid
- 将 inode 插入全局 inode_hash_table(通过 i_hash 字段链入)
- 如果文件系统有自己的 inode 缓存(如 ext4_inode_cache),也插入其中
- 设置 inode 的 i_state 位(I_NEW 表示新创建,尚未写入磁盘)
- 调用 mark_inode_dirty() 标记为脏,等待 writeback 线程刷盘
四、目录项(dentry):路径查找的加速器
dentry(directory entry)代表路径中的一个分量。例如 /home/user/file.txt 分解为三个 dentry:home、user、file.txt。dentry 结构本身不是磁盘上的实体,而是内存中的缓存,用于加速从路径到 inode 的映射。
// include/linux/dcache.h (简化版)
struct dentry {
unsigned int d_flags; // DCACHE_* 标志
seqcount_spinlock_t d_seq; // 序列计数器(RCU 保护查找)
struct hlist_bl_node d_hash; // 全局 dentry hash 表
struct dentry *d_parent; // 父目录 dentry
struct qstr d_name; // 名称(带哈希)
struct inode *d_inode; // 关联的 inode(可能为 NULL = 负向缓存)
struct list_head d_lru; // LRU 链表(用于回收)
struct list_head d_child; // 父目录的子节点链表
struct list_head d_subdirs; // 子目录/文件的链表头
const struct dentry_operations *d_op; // dentry 操作函数表
struct super_block *d_sb; // 所属超级块
union {
struct list_head d_dentry; // 空闲 dentry 链表
struct rcu_head d_rcu; // RCU 释放
};
// ...
};
4.1 dentry 缓存(dcache)机制
Linux 将 dentry 缓存在三个位置:
- 哈希表 dentry_hashtable:由 parent_dentry_hash + name_hash 索引,O(1) 查找
- LRU 链表 dentry_unused:当 dentry 引用计数降为 0 时进入 LRU,超出限制时被回收
- 父子关系树:通过 d_parent/d_child/d_subdirs 形成树形结构
路径查找的完整流程(以 open("/home/user/file.txt", O_RDONLY) 为例):
用户空间 open() 系统调用
↓
sys_open() → do_sys_open()
↓
do_sys_open() 调用 getname() 获取路径字符串
↓
do_sys_open() 调用 get_unused_fd_flags()
↓
do_sys_openat2() 调用 do_filp_open()
↓
do_filp_open() → path_openat() 开始路径遍历
↓
[路径遍历核心] nameidata 结构保存遍历状态:
- 从 current->fs->root 或 current->fs->pwd 开始
- 对每个路径分量调用 lookup_fast() (dcache 命中时)
- 若 fast miss,调用 lookup_slow() → inode->i_op->lookup()
↓
到达最终 path 后,调用 do_open()
↓
do_open() → vfs_open() → inode->i_fop->open()
↓
将 file 对象关联到 fd,fd_install()
4.2 负向缓存(Negative Dentry)
当查找一个不存在的文件时,内核会创建一个 d_inode 为 NULL 的 dentry,标记为 DCACHE_MISS_TYPE。这样下次查找同名文件时可以直接从缓存中得到"文件不存在"的答案,避免了开销昂贵的磁盘 IO。这种机制尤其在频繁检查文件是否存在(如 ld.so 搜索共享库)的场景下效果显著。
五、文件对象(file):进程打开文件的上下文
file 结构代表一个进程打开的文件实例。同一个文件被同一个或不同进程打开多次,会产生多个独立的 file 对象。file 记录了打开时刻的状态,而非文件本身的属性。
// include/linux/fs.h (简化版)
struct file {
struct path f_path; // 包含 dentry 和 mount
struct inode *f_inode; // 关联的 inode(缓存 f_path.dentry->d_inode)
const struct file_operations *f_op; // 文件操作函数表
spinlock_t f_lock; // 保护 f_pos 等字段
atomic_long_t f_count; // 引用计数
unsigned int f_flags; // 打开标志(O_RDONLY/O_NONBLOCK/O_APPEND...)
fmode_t f_mode; // 文件访问模式(FMODE_READ/WRITE)
loff_t f_pos; // 当前读写偏移量(关键)
struct fown_struct f_owner; // 用于 SIGIO 信号投递
struct file_ra_state f_ra; // 预读状态
struct address_space *f_mapping; // 页缓存映射(通常 = inode->i_mapping)
struct list_head f_link; // 超级块的文件链表
// ...
};
最重要的区分是:inode 是"文件"的元数据,file 是"打开文件"的上下文。一个 inode 可以被多个 file 指向(多进程/多次 open),每个 file 维护独立的 f_pos 读写偏移量。
六、file_operations:设备与文件的统一接口
file_operations 是 VFS 实现多态的核心。每个字符设备驱动和块设备驱动都必须实现这套接口,使得应用层可以用相同的方式操作键盘、磁盘、串口、socket 等截然不同的实体。
struct file_operations {
struct module *owner;
loff_t (*llseek)(struct file *, loff_t, int);
ssize_t (*read)(struct file *, char __user *, size_t, loff_t *);
ssize_t (*write)(struct file *, const char __user *, size_t, loff_t *);
ssize_t (*read_iter)(struct kiocb *, struct iov_iter *);
ssize_t (*write_iter)(struct kiocb *, struct iov_iter *);
int (*iopoll)(struct kiocb *kiocb, bool spin);
int (*iterate)(struct file *, struct dir_context *);
int (*iterate_shared)(struct file *, struct dir_context *);
__poll_t (*poll)(struct file *, struct poll_table_struct *);
long (*unlocked_ioctl)(struct file *, unsigned int, unsigned long);
long (*compat_ioctl)(struct file *, unsigned int, unsigned long);
int (*mmap)(struct file *, struct vm_area_struct *);
int (*open)(struct inode *, struct file *);
int (*flush)(struct file *, fl_owner_t id);
int (*release)(struct inode *, struct file *);
int (*fsync)(struct file *, loff_t, loff_t, int datasync);
// ...
};
特别需要关注的是 read_iter / write_iter,这是 4.1+ 内核引入的异步 IO 替代接口,配合 io_uring 使用可以进一步提升性能。
七、地址空间与页缓存:读写路径的全链路分析
address_space 是 VFS 与 memory 子系统的桥梁。每个 inode 都有一个 i_mapping 指针,指向该文件在内存中的页缓存空间。
struct address_space {
struct inode *host; // 宿主 inode
struct xarray i_pages; // 页缓存树(基数树 → xarray 替换)
rwlock_t i_rwsem; // 保护 xarray 结构
unsigned long nrpages; // 缓存的总页数
const struct address_space_operations *a_ops;
// ...
};
7.1 读路径完整分析
用户空间 read(fd, buf, count)
↓
sys_read() → fdget_pos() → fdput_pos()
↓
vfs_read()
↓ (检查 f_pos + count)
↓ (检查 FMODE_READ 标志)
↓
[generic_file_read_iter()] 或具体文件系统的 read_iter
↓
[页缓存查找] 在 address_space->i_pages 中搜索 offset 对应的 page
↓
[缓存命中 (PageCache Hit)]:
- 将 kernel page 数据 copy_to_user 到 buf
- 更新 atime
- 调整 f_pos
[缓存未命中 (Page Cache Miss)]:
- 调用 page_cache_sync_readahead() 异步预读
- 从 buddy allocator 分配新 page
- 调用 a_ops->readpage() 从磁盘读取数据到 page
- wait_on_page_locked() 等待 IO 完成
- copy_to_user 到 buf
- 释放额外预读的页面
7.2 写路径完整分析
写路径更为复杂,因为涉及延迟写回(writeback)
用户空间 write(fd, buf, count)
↓
sys_write() → vfs_write()
↓
[generic_perform_write()]:
- copy_from_user() 将用户数据写入内核
- 在 address_space 中找到/分配对应 page
- 标记 page 为脏 (SetPageDirty)
- 更新 inode->i_size(若扩展了文件)
- 更新 inode->i_mtime/i_ctime
↓
[写回触发时机]:
- dirty page 超过 /proc/sys/vm/dirty_ratio
- dirty page 超过 dirty_expire_centisecs(默认 30s)
- 用户调用 sync() 或 fsync(fd)
- inode 被 evict 时(脏页必须写回才能释放)
[writeback 线程]:
- bdi_writeback 工作队列执行 wb_workfn()
- 遍历脏 inode 链表
- 调用 a_ops->writepage() 将脏页写入磁盘
- 或直接使用 writeback 的 plug 机制批量提交
八、文件描述符表:进程视角的文件管理
进程通过文件描述符(fd)引用文件对象。每个进程的 task_struct 中有一个 files_struct 结构,管理该进程所有打开的文件。
// include/linux/fdtable.h
struct fdtable {
unsigned int max_fds;
struct file __rcu **fd; // 文件对象指针数组
unsigned long *close_on_exec; // 位图:exec 时关闭
unsigned long *open_fds; // 位图:已分配的 fd
};
struct files_struct {
atomic_t count;
struct fdtable __rcu *fdt; // 当前 fdtable
struct fdtable fdtab; // 内联默认 fdtable(通常 64)
// ...
};
文件描述符分配使用最小可用 fd 算法(通过 open_fds 位图 + find_next_zero_bit())。当 fd 数量超过当前 fdt 容量时,内核会自动扩展 fdtable。
8.1 fd 与 file 的生命周期管理
- dup/dup2:增加 file 引用计数,不创建新 file 对象
- fork:子进程继承父进程 fdtable,file 引用计数加 1
- O_CLOEXEC:通过 fdt->close_on_exec 位标记,execve() 时自动关闭
- close:fd_install(NULL) 清除 fd 指向,drop file 引用计数,为 0 时调用 f_op->release()
九、页缓存与直接 IO:何时绕过缓存
并非所有场景都适合页缓存。数据库和某些高性能存储引擎倾向于使用直接 IO(O_DIRECT)来管理自己的缓存策略。
9.1 直接 IO 的优缺点
优点:
- 避免双重缓存(内核页缓存 + 应用层缓存)
- 精确控制写入时序,配合 fsync() 保证持久化
- 减少 CPU 拷贝开销(用户态 buffer 直接到设备)
缺点:
- 必须对齐:offset 和 buffer 地址必须是 block_size 的整数倍
- 失去预读(readahead)加速效果
- 增加应用复杂度,需要自行管理缓存策略
9.2 内存映射文件(mmap)
mmap 是一种将文件内容直接映射到进程虚拟地址空间的机制。它既避免了传统 read/write 的 copy_to_user,又允许将文件当作内存访问。其核心是在访问文件页面时触发 page fault,由 VFS 负责将磁盘数据加载到页缓存。
应用程序调用 mmap(NULL, len, PROT_READ, MAP_SHARED, fd, offset)
↓
内核分配一个空的 VMA(vm_area_struct),建立虚拟地址到文件的关联
↓
[不立即读取数据]
↓
应用程序访问映射区域 → 触发缺页异常(page fault)
↓
handle_mm_fault() → do_fault() → filemap_fault()
↓
在 address_space 中查找页面,未找到则从磁盘加载
↓
建立页表映射,重新执行触发 fault 的指令
十、VFS 性能调优实践
10.1 文件系统选择
- ext4:通用场景,日志开销中等,最大支持 1EB 文件系统
- xfs:大文件/高并发场景,优秀的并行 IO 性能(基于分配组)
- btrfs:需要快照/压缩/RAID 等高级功能,但稳定性仍在演进
- tmpfs纯内存文件系统,适合 /tmp、共享内存等高频读写场景
10.2 关键内核参数
| 参数 | 默认值 | 说明 |
|---|---|---|
| /proc/sys/vm/dirty_ratio | 20 | 脏页占总内存的最大百分比,超过则同步写回 |
| /proc/sys/vm/dirty_background_ratio | 10 | 后台写回线程启动阈值 |
| /proc/sys/vm/dirty_expire_centisecs | 3000(30s) | 脏页最大存活时间,超过则被写回 |
| /proc/sys/vm/dirty_writeback_centisecs | 500(5s) | writeback 线程唤醒周期 |
| /proc/sys/vm/vfs_cache_pressure | 100 | dentry/inode 回收压力(0 = 不回收) |
| /proc/sys/fs/nr_open | 1048576 | 单进程最大打开文件数 |
10.3 文件描述符与并发
- 系统级限制:/sys/fs/file-max(查看和修改最大文件句柄数)
- 进程级限制:ulimit -n(soft/hard limit)
- 连接与 fd:每个 TCP 连接消耗 1 个 fd,epoll 实例本身也消耗 1 个 fd。高并发场景下(如 C10K/C100K)必须调整 fd 上限
- ulimit 设置:在 /etc/security/limits.conf 中配置
* soft nofile 65535
十一、实战:从零实现一个简单的内存文件系统
为了更深入理解 VFS 各对象间的交互,以下展示一个最小化内存文件系统的核心框架,演示 superblock、inode、dentry、file_operations 如何协同工作。
// 最小化内存文件系统示例(教学用途,非完整实现)
#include <linux/init.h>
#include <linux/module.h>
#include <linux/fs.h>
#include <linux/pagemap.h>
#define MYFS_MAGIC 0x4D594653 // "MYFS"
// 超级块填充
static int myfs_fill_super(struct super_block *sb, void *data, int silent) {
struct inode *root_inode;
struct dentry *root_dentry;
sb->s_blocksize = PAGE_SIZE;
sb->s_blocksize_bits = PAGE_SHIFT;
sb->s_magic = MYFS_MAGIC;
sb->s_op = &myfs_super_ops;
sb->s_time_gran = 1;
// 创建根 inode
root_inode = new_inode(sb);
if (!root_inode) return -ENOMEM;
root_inode->i_ino = 1;
root_inode->i_mode = S_IFDIR | 0755;
root_inode->i_uid = current_fsuid();
root_inode->i_gid = current_fsgid();
root_inode->i_atime = root_inode->i_mtime = root_inode->i_ctime = current_time(root_inode);
root_inode->i_op = &myfs_dir_inode_ops;
root_inode->i_fop = &myfs_dir_ops;
// 创建根 dentry
root_dentry = d_make_root(root_inode);
if (!root_dentry) return -ENOMEM;
sb->s_root = root_dentry;
return 0;
}
// 挂载入口(使用 mount_nodev 用于无设备文件系统)
static struct dentry *myfs_mount(struct file_system_type *fs_type,
int flags, const char *dev_name, void *data) {
return mount_nodev(fs_type, flags, data, myfs_fill_super);
}
// inode 操作:创建文件
static int myfs_create(struct user_namespace *mnt_userns,
struct inode *dir, struct dentry *dentry,
umode_t mode, bool excl) {
struct inode *inode = new_inode(dir->i_sb);
if (!inode) return -ENOMEM;
inode->i_ino = get_next_ino();
inode->i_mode = mode;
inode->i_uid = current_fsuid();
inode->i_gid = current_fsgid();
inode->i_atime = inode->i_mtime = inode->i_ctime = current_time(inode);
inode->i_op = &myfs_file_inode_ops;
inode->i_fop = &myfs_file_ops;
inode->i_mapping->a_ops = &myfs_aops;
d_instantiate(dentry, inode);
mark_inode_dirty(inode);
return 0;
}
// 地址空间操作:读页
static int myfs_readpage(struct file *file, struct page *page) {
void *addr = kmap_local_page(page);
// 在真实文件系统中从磁盘读取数据
memset(addr, 0, PAGE_SIZE);
flush_dcache_page(page);
kunmap_local(addr);
unlock_page(page);
return 0;
}
// 地址空间操作:写页
static int myfs_writepage(struct page *page, struct writeback_control *wbc) {
void *addr = kmap_local_page(page);
// 在真实文件系统中写入磁盘
kunmap_local(addr);
return 0;
}
// 文件系统类型注册
static struct file_system_type myfs_type = {
.owner = THIS_MODULE,
.name = "myfs",
.mount = myfs_mount,
.kill_sb = kill_litter_super,
};
static int __init myfs_init(void) {
return register_filesystem(&myfs_type);
}
static void __exit myfs_exit(void) {
unregister_filesystem(&myfs_type);
}
module_init(myfs_init);
module_exit(myfs_exit);
这个最小化文件系统展示了 VFS 的核心协作模式:super_block 定义文件系统类型、alloc_inode 创建 inode、dentry 关联路径与 inode、file_operations 实现读写、address_space_operations 管理页缓存。整个框架不过百余行代码,就能实现一个可挂载的虚拟文件系统。
十二、总结
Linux VFS 的设计精妙之处在于通过四个核心对象和函数指针表的多态机制,将文件系统的"共性"与"个性"完美分离:super_block 管理全局、inode 抽象元数据、dentry 加速路径、file 记录访问状态。application 层通过统一的 syscall 接口,无需关心底层是 ext4 还是 xfs 还是 tmpfs。
理解 VFS 的完整调用链,对于系统性能诊断至关重要:
- 文件打开延迟高 → 可能是 dentry 缓存未命中导致路径查找走磁盘
- 读吞吐量低 → 检查页缓存命中率(sar -B/si/so),考虑是否 IO 调度器不合理
- 写入回写抖动 → 调整 dirty_ratio/dirty_background_ratio,平衡吞吐与数据安全
- fd 耗尽 → ulimit 限制或 fd 泄漏(lsof -p PID 排查未关闭 fd)
从 read/write syscall 到 block IO 层,VFS 经历了路径查找、缓存命中判断、页缓存交互、回写调度等环节。每一层都有其优化机制(dcache、readahead、writeback),只有理解完整链路,才能在性能问题面前做到有的放矢。

发表评论 取消回复