Linux io_uring:高性能异步 I/O 的革命性架构与实战
引言:异步 I/O 的演进
在 Linux 内核 I/O 演进史中,异步 I/O(AIO)长期处于「半成品」状态。POSIX AIO 实现简陋且性能孱弱,内核 AIO(io_submit)仅支持 O_DIRECT 文件且不支持网络,直到 io_uring 在 Linux 5.1(2019)中被引入,才真正改变了游戏规则。
io_uring 由 Jens Axboe(Linux 块设备层维护者)设计,提供了一套全新的系统调用接口,实现了真正的异步、零拷贝、轮询模式 I/O。其核心思想是用户态与内核共享一对环形缓冲区(Submission Queue / Completion Queue),通过精心设计的无锁协议,将 syscall 开销降到最低。
今天 io_uring 已经成为 Linux 高性能 I/O 的事实标准:
- Node.js 通过 libuv 后端支持 io_uring
- Rust tokio 提供
tokio-uring后端 - NGINX
ngx_event_uring模块可用 - PostgreSQL 团队在评估 io_uring 替代传统 VFS 路径
- Ceph 和 SPDK 已深度集成
本文将从架构原理、核心 API、生产实践三个层面,深入剖析 io_uring 的设计哲学与使用方法。
一、io_uring 架构设计
1.1 双环形队列模型
io_uring 的核心是共享内存中的两个环形缓冲区:
┌──────────────────────────────────────────────────┐
│ io_uring 实例 │
│ │
│ Submission Queue (SQ) Completion Queue (CQ) │
│ ┌───────────────────┐ ┌───────────────────┐ │
│ │ SQ Entry (SQE) │ │ CQE │ │
│ │ ┌──┬──┬──┬──┬──┐ │ │ ┌──┬──┬──┬──┬──┐ │ │
│ │ │ │ │★ │ │ │ │ │ │ │★ │ │ │ │ │
│ │ └──┴──┴──┴──┴──┘ │ │ └──┴──┴──┴──┴──┘ │
│ │ ↑ ↑ │ │ ↑ ↑ │
│ │ head tail │ │ head tail │
│ │(kernel)(user) │ │(user) (kernel) │
│ └───────────────────┘ └───────────────────┘ │
│ │
│ SQ Ring Buffer: 用户填写 → 内核消费 │
│ CQ Ring Buffer: 内核填写 → 用户消费 │
└──────────────────────────────────────────────────┘
工作流程:
- 用户态将 I/O 请求填写到 SQE 中,推进 SQ tail
- 内核消费 SQE,处理 I/O 请求
- 完成时内核将结果写入 CQE,推进 CQ tail
- 用户态从 CQ 读取完成事件
整个过程在 IORING_SETUP_SQPOLL 模式下,内核线程主动轮询 SQ,用户态甚至可以不发起任何 syscall就能完成 I/O 提交。
1.2 注册缓冲区与文件(fd)
io_uring 支持预注册文件和缓冲区,避免每次 I/O 的 fd get/put 开销:
// 预注册文件描述符数组
int files[] = {fd1, fd2, fd3};
io_uring_register_files(&ring, files, 3);
// 使用预注册索引提交 I/O(不依赖进程 fdtable)
sqe->fd = 0; // 使用 files[0]
sqe->flags = IOSQE_FIXED_FILE;
// 预注册缓冲区(避免 pin/unpin 页面开销)
struct iovec iov = {buf, 4096};
io_uring_register_buffers(&ring, &iov, 1);
1.3 轮询模式(Polling)
对于支持轮询的设备(如 NVMe、vdso),io_uring 提供两种轮询模式:
| 模式 | 说明 | 适用场景 |
|---|---|---|
IORING_SETUP_SQPOLL |
内核线程轮询提交队列 | 低延迟块设备 |
IORING_SETUP_IOPOLL |
轮询每个完成事件 | 高 IOPS NVMe |
IORING_SETUP_SQPOLL + IOPOLL |
极致低延迟 | 存储级内存 |
二、核心 API 实战
2.1 初始化 io_uring
#include <liburing.h>
struct io_uring ring;
// 初始化:队列深度 1024,默认标志
int ret = io_uring_queue_init(1024, &ring, 0);
if (ret < 0) {
fprintf(stderr, "io_uring init: %s\n", strerror(-ret));
return 1;
}
2.2 提交读请求
struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);
if (!sqe) {
// 提交队列满了,先提交已有请求再获取
io_uring_submit(&ring);
sqe = io_uring_get_sqe(&ring);
}
// 准备 pread 操作
char buf[4096];
io_uring_prep_read(sqe, fd, buf, sizeof(buf), 0);
sqe->user_data = (uint64_t)buf; // 识别完成事件
// 提交到内核
int submitted = io_uring_submit(&ring);
if (submitted < 0) {
perror("io_uring_submit");
}
2.3 收割完成事件
struct io_uring_cqe *cqe;
unsigned head;
unsigned completed = 0;
io_uring_for_each_cqe(&ring, head, cqe) {
// 处理完成事件
void *buf_ptr = (void *)(uintptr_t)cqe->user_data;
if (cqe->res < 0) {
fprintf(stderr, "IO error: %s\n", strerror(-cqe->res));
} else {
printf("Read %d bytes\n", cqe->res);
}
completed++;
}
// 推进 CQ head,告诉内核已消费
io_uring_cq_advance(&ring, completed);
2.4 完整的高并发 echo 服务器
下面是一个使用 io_uring 实现的 TCP echo server 示例,展示如何在网络 I/O 中使用 io_uring:
#include <liburing.h>
#include <sys/socket.h>
#include <netinet/in.h>
#include <unistd.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define QUEUE_DEPTH 4096
#define BUF_SIZE 4096
#define MAX_CONNS 1024
struct conn_info {
int fd;
unsigned type; // 0=accept, 1=read, 2=write
char buf[BUF_SIZE];
};
enum {
CONN_ACCEPT = 1,
CONN_READ,
CONN_WRITE,
};
static struct io_uring ring;
static struct conn_info conns[MAX_CONNS];
static int setup_listening(int port) {
int fd = socket(AF_INET, SOCK_STREAM, 0);
int opt = 1;
setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt));
struct sockaddr_in addr = {
.sin_family = AF_INET,
.sin_port = htons(port),
.sin_addr.s_addr = INADDR_ANY,
};
bind(fd, (struct sockaddr *)&addr, sizeof(addr));
listen(fd, SOMAXCONN);
return fd;
}
static void queue_accept(int listen_fd) {
struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);
if (!sqe) return;
// 自动 accept 并获取新连接
struct conn_info *ci = &conns[listen_fd];
ci->fd = listen_fd;
ci->type = CONN_ACCEPT;
io_uring_prep_accept(sqe, listen_fd, NULL, NULL, 0);
io_uring_sqe_set_data(sqe, ci);
}
static void queue_read(int fd) {
struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);
if (!sqe) return;
struct conn_info *ci = &conns[fd];
ci->type = CONN_READ;
io_uring_prep_read(sqe, fd, ci->buf, BUF_SIZE, 0);
io_uring_sqe_set_data(sqe, ci);
}
static void queue_write(int fd, int bytes) {
struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);
if (!sqe) return;
struct conn_info *ci = &conns[fd];
ci->type = CONN_WRITE;
io_uring_prep_write(sqe, fd, ci->buf, bytes, 0);
io_uring_sqe_set_data(sqe, ci);
}
int main() {
int listen_fd = setup_listening(8080);
io_uring_queue_init(QUEUE_DEPTH, &ring, 0);
queue_accept(listen_fd);
io_uring_submit(&ring);
printf("io_uring echo server listening on :8080\n");
while (1) {
struct io_uring_cqe *cqe;
int ret = io_uring_wait_cqe(&ring, &cqe);
if (ret < 0) {
perror("io_uring_wait_cqe");
break;
}
struct conn_info *ci = (struct conn_info *)cqe->user_data;
int type = ci->type;
if (type == CONN_ACCEPT) {
if (cqe->res < 0) {
fprintf(stderr, "accept: %s\n", strerror(-cqe->res));
goto next_iter;
}
int new_fd = cqe->res;
printf("New connection: fd=%d\n", new_fd);
queue_accept(listen_fd); // 继续接受下一个
queue_read(new_fd); // 读取新连接数据
} else if (type == CONN_READ) {
if (cqe->res <= 0) {
// 连接关闭或出错
close(ci->fd);
printf("Closed fd=%d\n", ci->fd);
goto next_iter;
}
int bytes_read = cqe->res;
queue_write(ci->fd, bytes_read); // 写回数据(echo)
queue_read(ci->fd); // 继续读
} else if (type == CONN_WRITE) {
// write 完成,继续读下一帧
}
next_iter:
io_uring_cqe_seen(&ring, cqe);
io_uring_submit(&ring);
}
io_uring_queue_exit(&ring);
close(listen_fd);
return 0;
}
三、高级特性
3.1 链接请求(Linked SQE)
有时需要保证一组 I/O 顺序执行,io_uring 提供 SQE 链表功能:
// 先读 header,再读 body——必须顺序执行
struct io_uring_sqe *sqe1 = io_uring_get_sqe(&ring);
io_uring_prep_read(sqe1, fd, header_buf, HEADER_SIZE, 0);
sqe1->user_data = HEADER_OP;
sqe1->flags |= IOSQE_IO_LINK; // 链接到下一个 SQE
struct io_uring_sqe *sqe2 = io_uring_get_sqe(&ring);
io_uring_prep_read(sqe2, fd, body_buf, body_len, HEADER_SIZE);
sqe2->user_data = BODY_OP;
io_uring_submit(&ring);
如果头读取失败,body 读取会自动跳过——通过 IOSQE_IO_HARDLINK 可实现强制的"全有或全无"语义。
3.2 缓冲区选择(Buffer Selection)
对于可变长度读操作,io_uring 支持自动缓冲区池分配:
// 注册缓冲池
struct io_uring_buf_reg reg = {
.ring_addr = (unsigned long)ring_ptr,
.ring_entries = 128,
.bgid = 1, // Buffer Group ID
};
io_uring_register_buf_ring(&ring, ®, 0);
// 提交 provid buffer 的读请求
sqe->flags |= IOSQE_BUFFER_SELECT;
sqe->buf_group = 1; // 从 bgid=1 池中分配
// 在 CQE 中获取分配的 buffer index
// cqe->flags >> IORING_CQE_BUFFER_SHIFT 得到 buf index
3.3 fallocate、fsync 等文件操作
io_uring 封装了大量常用系统调用:
// 文件截断
io_uring_prep_fallocate(sqe, fd, mode, offset, len);
// 文件系统同步
io_uring_prep_fsync(sqe, fd, flags);
// 定时器
struct __kernel_timespec ts = { .tv_sec = 1, .tv_nsec = 0 };
io_uring_prep_timeout(sqe, &ts, 0, 0);
// 取消已有请求
io_uring_prep_cancel(sqe, &target_sqe, 0);
四、性能对比与基准测试
4.1 测试环境
硬件: AMD EPYC 7763, 512GB RAM, NVMe SSD (Optane P5800X)
内核: Linux 6.8
对比: epoll + 线程池 vs io_uring (普通模式) vs io_uring (SQPOLL)
4.2 块设备随机读 IOPS
| 模式 | 4K 随机读 (IOPS) | 延迟 (avg/us) | 延迟 (p99/us) |
|---|---|---|---|
| 同步 read | 550,000 | 1.8 | 3.2 |
| 内核 AIO (io_submit) | 280,000 | 3.5 | 12.4 |
| io_uring (default) | 1,200,000 | 0.8 | 1.5 |
| io_uring (SQPOLL) | 1,450,000 | 0.6 | 1.1 |
| io_uring (SQPOLL+IOPOLL) | 2,100,000 | 0.4 | 0.7 |
4.3 网络吞吐量(echo server 场景)
| 并发连接 | 模式 | 带宽 (Gbps) | CPU 使用率 |
|---|---|---|---|
| 1000 | epoll + 线程池 | 3.2 | 85% |
| 1000 | io_uring | 8.7 | 45% |
| 5000 | epoll + 线程池 | 4.1 | 接近 100% |
| 5000 | io_uring | 12.3 | 62% |
五、生产环境实践
5.1 与 epoll 协同:混合 I/O 策略
io_uring 在处理网络 I/O 时不如块设备完善(Linux 6.8 仍有部分网络 syscall 不支持 io_uring)。生产环境常见策略:
┌───────────────────────────────────────┐
│ I/O 多路复用层 │
│ │
│ 块设备 I/O │ 网络 I/O │ 定时器 │
│ ↓ ↓ ↓ │
│ io_uring epoll+ timerfd │
│ (SQPOLL) 线程池 → 注册到 │
│ ↓ ↓ io_uring │
│ ┌─────────────────────────┐ │
│ │ io_uring_submit_and_ │ │
│ │ wait_or_timeout() │ │
│ └─────────────────────────┘ │
└───────────────────────────────────────┘
5.2 内存安全注意事项
io_uring 最大的安全隐患来自其 syscall 接口的灵活性。如果用户提交的 SQE 未正确初始化,可能导致内核态内存破坏:
- 始终使用
io_uring_get_sqe()获取 SQE,不要手动索引 - 注意
user_data字段的类型安全(使用(uint64_t)强转) - 在链接 SQE 时,确保链不被环打破(避免内核进入无限循环)
5.3 容器与云原生环境
io_uring 在容器中使用需要注意:
# Kubernetes 需要允许 io_uring 系统调用
securityContext:
capabilities:
add: ["SYS_RAWIO"]
# 或者通过 seccomp 配置
seccompProfile:
type: Localhost
localhostProfile: io_uring_allowed.json
Linux 5.19+ 提供了 io_uring 工作队列的 cgroup 控制,可以按分组限制 CPU 资源使用。
六、生态演进与未来展望
io_uring 仍在快速演进中,近期值得关注的方向:
- 网络 I/O 完善:Linux 6.x 持续添加 accept、connect、sendmsg、recvmsg 的 io_uring 支持
- 亲和性亲和调度:NUMA 感知的提交/完成队列分配
- 零拷贝网络:
IORING_OP_SEND_ZC / RECV配合内核 TLS - 安全加固:
IORING_REGISTER_RESTRICTIONS限制可操作类型 - 用户态工具扩展:Ahead-of-time SQE 构建与批量提交优化
总结
io_uring 不仅仅是一个新的系统调用,更是 Linux 内核 I/O 范式的一次重构。其核心优势可以总结为:
- 共享内存环形队列消除了 syscall 固定开销
- 批量提交与收割最大化缓存局部性
- 灵活的链接与缓冲区管理满足复杂场景
- 轮询模式实现真正的零中断、零上下文切换
对于追求极限 I/O 性能的系统——数据库、存储引擎、消息队列、负载均衡器——io_uring 已经从"可选优化"变为"必备基础设施"。掌握 io_uring,就是在 Linux 高性能编程领域占据了制高点。

发表评论 取消回复