Linux io_uring:高性能异步 I/O 的革命性架构与实战

引言:异步 I/O 的演进

在 Linux 内核 I/O 演进史中,异步 I/O(AIO)长期处于「半成品」状态。POSIX AIO 实现简陋且性能孱弱,内核 AIO(io_submit)仅支持 O_DIRECT 文件且不支持网络,直到 io_uring 在 Linux 5.1(2019)中被引入,才真正改变了游戏规则。

io_uring 由 Jens Axboe(Linux 块设备层维护者)设计,提供了一套全新的系统调用接口,实现了真正的异步、零拷贝、轮询模式 I/O。其核心思想是用户态与内核共享一对环形缓冲区(Submission Queue / Completion Queue),通过精心设计的无锁协议,将 syscall 开销降到最低。

今天 io_uring 已经成为 Linux 高性能 I/O 的事实标准:

  • Node.js 通过 libuv 后端支持 io_uring
  • Rust tokio 提供 tokio-uring 后端
  • NGINX ngx_event_uring 模块可用
  • PostgreSQL 团队在评估 io_uring 替代传统 VFS 路径
  • Ceph 和 SPDK 已深度集成

本文将从架构原理、核心 API、生产实践三个层面,深入剖析 io_uring 的设计哲学与使用方法。

一、io_uring 架构设计

1.1 双环形队列模型

io_uring 的核心是共享内存中的两个环形缓冲区:

┌──────────────────────────────────────────────────┐

│ io_uring 实例 │

│ │

│ Submission Queue (SQ) Completion Queue (CQ) │

│ ┌───────────────────┐ ┌───────────────────┐ │

│ │ SQ Entry (SQE) │ │ CQE │ │

│ │ ┌──┬──┬──┬──┬──┐ │ │ ┌──┬──┬──┬──┬──┐ │ │

│ │ │ │ │★ │ │ │ │ │ │ │★ │ │ │ │ │

│ │ └──┴──┴──┴──┴──┘ │ │ └──┴──┴──┴──┴──┘ │

│ │ ↑ ↑ │ │ ↑ ↑ │

│ │ head tail │ │ head tail │

│ │(kernel)(user) │ │(user) (kernel) │

│ └───────────────────┘ └───────────────────┘ │

│ │

│ SQ Ring Buffer: 用户填写 → 内核消费 │

│ CQ Ring Buffer: 内核填写 → 用户消费 │

└──────────────────────────────────────────────────┘

工作流程:

  1. 用户态将 I/O 请求填写到 SQE 中,推进 SQ tail
  2. 内核消费 SQE,处理 I/O 请求
  3. 完成时内核将结果写入 CQE,推进 CQ tail
  4. 用户态从 CQ 读取完成事件

整个过程在 IORING_SETUP_SQPOLL 模式下,内核线程主动轮询 SQ,用户态甚至可以不发起任何 syscall就能完成 I/O 提交。

1.2 注册缓冲区与文件(fd)

io_uring 支持预注册文件和缓冲区,避免每次 I/O 的 fd get/put 开销:

// 预注册文件描述符数组

int files[] = {fd1, fd2, fd3};

io_uring_register_files(&ring, files, 3);

// 使用预注册索引提交 I/O(不依赖进程 fdtable)

sqe->fd = 0; // 使用 files[0]

sqe->flags = IOSQE_FIXED_FILE;

// 预注册缓冲区(避免 pin/unpin 页面开销)

struct iovec iov = {buf, 4096};

io_uring_register_buffers(&ring, &iov, 1);

1.3 轮询模式(Polling)

对于支持轮询的设备(如 NVMe、vdso),io_uring 提供两种轮询模式:

模式说明适用场景
IORING_SETUP_SQPOLL
内核线程轮询提交队列 | 低延迟块设备 |

| IORING_SETUP_IOPOLL | 轮询每个完成事件 | 高 IOPS NVMe |

| IORING_SETUP_SQPOLL + IOPOLL | 极致低延迟 | 存储级内存 |

二、核心 API 实战

2.1 初始化 io_uring

#include <liburing.h>

struct io_uring ring;

// 初始化:队列深度 1024,默认标志

int ret = io_uring_queue_init(1024, &ring, 0);

if (ret < 0) {

fprintf(stderr, "io_uring init: %s\n", strerror(-ret));

return 1;

}

2.2 提交读请求

struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);

if (!sqe) {

// 提交队列满了,先提交已有请求再获取

io_uring_submit(&ring);

sqe = io_uring_get_sqe(&ring);

}

// 准备 pread 操作

char buf[4096];

io_uring_prep_read(sqe, fd, buf, sizeof(buf), 0);

sqe->user_data = (uint64_t)buf; // 识别完成事件

// 提交到内核

int submitted = io_uring_submit(&ring);

if (submitted < 0) {

perror("io_uring_submit");

}

2.3 收割完成事件

struct io_uring_cqe *cqe;

unsigned head;

unsigned completed = 0;

io_uring_for_each_cqe(&ring, head, cqe) {

// 处理完成事件

void buf_ptr = (void )(uintptr_t)cqe->user_data;

if (cqe->res < 0) {

fprintf(stderr, "IO error: %s\n", strerror(-cqe->res));

} else {

printf("Read %d bytes\n", cqe->res);

}

completed++;

}

// 推进 CQ head,告诉内核已消费

io_uring_cq_advance(&ring, completed);

2.4 完整的高并发 echo 服务器

下面是一个使用 io_uring 实现的 TCP echo server 示例,展示如何在网络 I/O 中使用 io_uring:

#include <liburing.h>

#include <sys/socket.h>

#include <netinet/in.h>

#include <unistd.h>

#include <stdio.h>

#include <stdlib.h>

#include <string.h>

#define QUEUE_DEPTH 4096

#define BUF_SIZE 4096

#define MAX_CONNS 1024

struct conn_info {

int fd;

unsigned type; // 0=accept, 1=read, 2=write

char buf[BUF_SIZE];

};

enum {

CONN_ACCEPT = 1,

CONN_READ,

CONN_WRITE,

};

static struct io_uring ring;

static struct conn_info conns[MAX_CONNS];

static int setup_listening(int port) {

int fd = socket(AF_INET, SOCK_STREAM, 0);

int opt = 1;

setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt));

struct sockaddr_in addr = {

.sin_family = AF_INET,

.sin_port = htons(port),

.sin_addr.s_addr = INADDR_ANY,

};

bind(fd, (struct sockaddr *)&addr, sizeof(addr));

listen(fd, SOMAXCONN);

return fd;

}

static void queue_accept(int listen_fd) {

struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);

if (!sqe) return;

// 自动 accept 并获取新连接

struct conn_info *ci = &conns[listen_fd];

ci->fd = listen_fd;

ci->type = CONN_ACCEPT;

io_uring_prep_accept(sqe, listen_fd, NULL, NULL, 0);

io_uring_sqe_set_data(sqe, ci);

}

static void queue_read(int fd) {

struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);

if (!sqe) return;

struct conn_info *ci = &conns[fd];

ci->type = CONN_READ;

io_uring_prep_read(sqe, fd, ci->buf, BUF_SIZE, 0);

io_uring_sqe_set_data(sqe, ci);

}

static void queue_write(int fd, int bytes) {

struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);

if (!sqe) return;

struct conn_info *ci = &conns[fd];

ci->type = CONN_WRITE;

io_uring_prep_write(sqe, fd, ci->buf, bytes, 0);

io_uring_sqe_set_data(sqe, ci);

}

int main() {

int listen_fd = setup_listening(8080);

io_uring_queue_init(QUEUE_DEPTH, &ring, 0);

queue_accept(listen_fd);

io_uring_submit(&ring);

printf("io_uring echo server listening on :8080\n");

while (1) {

struct io_uring_cqe *cqe;

int ret = io_uring_wait_cqe(&ring, &cqe);

if (ret < 0) {

perror("io_uring_wait_cqe");

break;

}

struct conn_info ci = (struct conn_info )cqe->user_data;

int type = ci->type;

if (type == CONN_ACCEPT) {

if (cqe->res < 0) {

fprintf(stderr, "accept: %s\n", strerror(-cqe->res));

goto next_iter;

}

int new_fd = cqe->res;

printf("New connection: fd=%d\n", new_fd);

queue_accept(listen_fd); // 继续接受下一个

queue_read(new_fd); // 读取新连接数据

} else if (type == CONN_READ) {

if (cqe->res <= 0) {

// 连接关闭或出错

close(ci->fd);

printf("Closed fd=%d\n", ci->fd);

goto next_iter;

}

int bytes_read = cqe->res;

queue_write(ci->fd, bytes_read); // 写回数据(echo)

queue_read(ci->fd); // 继续读

} else if (type == CONN_WRITE) {

// write 完成,继续读下一帧

}

next_iter:

io_uring_cqe_seen(&ring, cqe);

io_uring_submit(&ring);

}

io_uring_queue_exit(&ring);

close(listen_fd);

return 0;

}

三、高级特性

3.1 链接请求(Linked SQE)

有时需要保证一组 I/O 顺序执行,io_uring 提供 SQE 链表功能:

// 先读 header,再读 body——必须顺序执行

struct io_uring_sqe *sqe1 = io_uring_get_sqe(&ring);

io_uring_prep_read(sqe1, fd, header_buf, HEADER_SIZE, 0);

sqe1->user_data = HEADER_OP;

sqe1->flags |= IOSQE_IO_LINK; // 链接到下一个 SQE

struct io_uring_sqe *sqe2 = io_uring_get_sqe(&ring);

io_uring_prep_read(sqe2, fd, body_buf, body_len, HEADER_SIZE);

sqe2->user_data = BODY_OP;

io_uring_submit(&ring);

如果头读取失败,body 读取会自动跳过——通过 IOSQE_IO_HARDLINK 可实现强制的"全有或全无"语义。

3.2 缓冲区选择(Buffer Selection)

对于可变长度读操作,io_uring 支持自动缓冲区池分配:

// 注册缓冲池

struct io_uring_buf_reg reg = {

.ring_addr = (unsigned long)ring_ptr,

.ring_entries = 128,

.bgid = 1, // Buffer Group ID

};

io_uring_register_buf_ring(&ring, &reg, 0);

// 提交 provid buffer 的读请求

sqe->flags |= IOSQE_BUFFER_SELECT;

sqe->buf_group = 1; // 从 bgid=1 池中分配

// 在 CQE 中获取分配的 buffer index

// cqe->flags >> IORING_CQE_BUFFER_SHIFT 得到 buf index

3.3 fallocate、fsync 等文件操作

io_uring 封装了大量常用系统调用:

// 文件截断

io_uring_prep_fallocate(sqe, fd, mode, offset, len);

// 文件系统同步

io_uring_prep_fsync(sqe, fd, flags);

// 定时器

struct __kernel_timespec ts = { .tv_sec = 1, .tv_nsec = 0 };

io_uring_prep_timeout(sqe, &ts, 0, 0);

// 取消已有请求

io_uring_prep_cancel(sqe, &target_sqe, 0);

四、性能对比与基准测试

4.1 测试环境

硬件: AMD EPYC 7763, 512GB RAM, NVMe SSD (Optane P5800X)

内核: Linux 6.8

对比: epoll + 线程池 vs io_uring (普通模式) vs io_uring (SQPOLL)

4.2 块设备随机读 IOPS

模式4K 随机读 (IOPS)延迟 (avg/us)延迟 (p99/us)
同步 read
550,000 | 1.8 | 3.2 |

| 内核 AIO (io_submit) | 280,000 | 3.5 | 12.4 |

| io_uring (default) | 1,200,000 | 0.8 | 1.5 |

| io_uring (SQPOLL) | 1,450,000 | 0.6 | 1.1 |

| io_uring (SQPOLL+IOPOLL) | 2,100,000 | 0.4 | 0.7 |

4.3 网络吞吐量(echo server 场景)

并发连接模式带宽 (Gbps)CPU 使用率
1000
epoll + 线程池 | 3.2 | 85% |

| 1000 | io_uring | 8.7 | 45% |

| 5000 | epoll + 线程池 | 4.1 | 接近 100% |

| 5000 | io_uring | 12.3 | 62% |

五、生产环境实践

5.1 与 epoll 协同:混合 I/O 策略

io_uring 在处理网络 I/O 时不如块设备完善(Linux 6.8 仍有部分网络 syscall 不支持 io_uring)。生产环境常见策略:

┌───────────────────────────────────────┐

│ I/O 多路复用层 │

│ │

│ 块设备 I/O │ 网络 I/O │ 定时器 │

│ ↓ ↓ ↓ │

│ io_uring epoll+ timerfd │

│ (SQPOLL) 线程池 → 注册到 │

│ ↓ ↓ io_uring │

│ ┌─────────────────────────┐ │

│ │ io_uring_submit_and_ │ │

│ │ wait_or_timeout() │ │

│ └─────────────────────────┘ │

└───────────────────────────────────────┘

5.2 内存安全注意事项

io_uring 最大的安全隐患来自其 syscall 接口的灵活性。如果用户提交的 SQE 未正确初始化,可能导致内核态内存破坏:

  • 始终使用 io_uring_get_sqe() 获取 SQE,不要手动索引
  • 注意 user_data 字段的类型安全(使用 (uint64_t) 强转)
  • 在链接 SQE 时,确保链不被环打破(避免内核进入无限循环)

5.3 容器与云原生环境

io_uring 在容器中使用需要注意:

# Kubernetes 需要允许 io_uring 系统调用

securityContext:

capabilities:

add: ["SYS_RAWIO"]

或者通过 seccomp 配置

seccompProfile:

type: Localhost

localhostProfile: io_uring_allowed.json

Linux 5.19+ 提供了 io_uring 工作队列的 cgroup 控制,可以按分组限制 CPU 资源使用。

六、生态演进与未来展望

io_uring 仍在快速演进中,近期值得关注的方向:

  1. 网络 I/O 完善:Linux 6.x 持续添加 accept、connect、sendmsg、recvmsg 的 io_uring 支持
  2. 亲和性亲和调度:NUMA 感知的提交/完成队列分配
  3. 零拷贝网络:IORING_OP_SEND_ZC / RECV 配合内核 TLS
  4. 安全加固:IORING_REGISTER_RESTRICTIONS 限制可操作类型
  5. 用户态工具扩展:Ahead-of-time SQE 构建与批量提交优化

总结

io_uring 不仅仅是一个新的系统调用,更是 Linux 内核 I/O 范式的一次重构。其核心优势可以总结为:

  • 共享内存环形队列消除了 syscall 固定开销
  • 批量提交与收割最大化缓存局部性
  • 灵活的链接与缓冲区管理满足复杂场景
  • 轮询模式实现真正的零中断、零上下文切换

对于追求极限 I/O 性能的系统——数据库、存储引擎、消息队列、负载均衡器——io_uring 已经从"可选优化"变为"必备基础设施"。掌握 io_uring,就是在 Linux 高性能编程领域占据了制高点。

点赞(0) 打赏

评论列表 共有 0 条评论

暂无评论
立即
投稿
网站二维码

微信公众账号

微信扫一扫加关注

发表
评论
返回
顶部
/* 跳过导航链接 (无障碍) */ position: absolute; top: -100px; left: 15px; z-index: 99999; padding: 8px 16px; background: #007bff; color: #fff; font-size: 14px; border-radius: 0 0 4px 4px; text-decoration: none; transition: top 0.2s; } top: 0; outline: 3px solid #0056b3; }