io_uring uring_cmd:用户态 NVMe Passthrough 直通接口深度工程实战
摘要
Linux 5.19 引入了 uring_cmd 接口,为 io_uring 增添了直接提交块设备命令的能力。这意味着用户态程序可以绕过传统的内核块层提交(如 ioctl),通过 io_uring 的零拷贝异步机制直接向 NVMe 控制器提交 Admin 命令和 IO 命令。本文深入剖析 uring_cmd 的实现原理、NVMe 直通路径的架构设计,并对比 SPDK 等传统用户态 IO 框架,给出在 AI 推理存储场景下的工程实测数据。
一、为什么需要 uring_cmd?
1.1 传统 NVMe 直通路径的瓶颈
在 Linux 中,用户态与 NVMe 设备交互的传统路径是 ioctl:
用户态 app
│
▼
ioctl(nvme_fd, NVME_IOCTL_SUBMIT_IO, &cmd)
│
▼
内核 nvme_dev 字符设备
│
▼
nvme_submit_cmd() → 写入 SQ 门铃寄存器
│
▼
NVMe 硬件
这条路径存在几个固有问题:
- 系统调用开销:每次
ioctl需要用户态到内核态的上下文切换,即使只用一条指令,现代 CPU 也需要约 100-200ns 的 syscall 开销。 - 串行化难以避免:虽然 NVMe 队列深度可达 64K,但
ioctl每次只能提交一个命令,需要频繁的系统调用。 - 缺乏批量化机制:与 io_uring 的批量 SQE 提交相比,
ioctl本质上是逐个提交,无法利用 SQ 环形缓冲区的批处理优势。
1.2 用户态 IO (SPDK) 的局限
SPDK (Storage Performance Development Kit) 通过 UIO/VFIO 将 NVMe 设备完全映射到用户态,实现了真正的零拷贝零系统调用 IO。但它也有局限:
- 需要独占设备(不能与其他内核驱动共享)
- 需要 DPDK 生态绑定
- 需要轮询模式,CPU 消耗高
- 部署复杂度高,需要 hugepage 配置、CPU 隔离等
1.3 uring_cmd 的定位
uring_cmd 试图在两者之间找到平衡点:
| 特性 | ioctl | SPDK | uring_cmd |
|---|---|---|---|
| 系统调用 | 每次 ioctl | 无 | 批量提交(enter 一次多条) |
| 设备共享 | 支持 | 独占 | 支持(内核协调) |
| 零拷贝 | 部分 | 完全 | 支持(fixed buffers) |
| 部署复杂度 | 低 | 高 | 低 |
| IO 延迟 | ~5-10μs | ~1-2μs | ~2-4μs |
| CPU 效率 | 中等 | 高(轮询) | 高(IO 中断+批量) |
二、uring_cmd 内核实现架构
2.1 核心数据结构
// include/linux/io_uring/cmd.h
struct io_uring_cmd {
struct file *file;
struct io_ring_ctx *ctx;
union {
struct {
struct iovec *addr;
__u32 nr;
__u32 padding;
};
void *pdu; // 命令协议数据单元
};
u8 op; // 命令操作码
u8 pad[3];
u32 flags;
u32 retval; // 返回值缓冲
u64 task_work_cb;
};
每个 NVMe 直通命令被封装为一个 io_uring_cmd,通过 io_uring 的 SQE 提交到内核。内核通过 io_uring 的 IORING_OP_URING_CMD 操作码分发到对应的 uring_cmd 处理函数。
2.2 NVMe 设备注册机制
// drivers/nvme/host/core.c
static const struct io_uring_cmd_ops nvme_uring_cmd_ops = {
.uring_cmd = nvme_uring_cmd_io,
.doing_poll = NULL, // 支持内核侧轮询
.task_cb = NULL,
};
static int nvme_init_uring_ctrl(struct nvme_ctrl *ctrl)
{
ctrl->uring_cmd_ops = &nvme_uring_cmd_ops;
return 0;
}
NVMe 驱动在初始化时注册 uring_cmd_ops,将 nvme_uring_cmd_io 作为默认的命令处理函数。
2.3 数据流路径对比
【传统 ioctl 路径】
App → syscall_enter → VFS → nvme_ioctl → nvme_submit_cmd() → doorbell
【uring_cmd 路径 (内核中转)】
App → io_uring_enter(批量SQE) → uring_cmd_handler → nvme_submit_cmd() → doorbell
│
├─ 内核侧可做:权限检查、队列仲裁、
│ 命名空间映射、DMA 映射管理
│
└─ 完成后通过 CQ 异步通知
关键点:uring_cmd 路径中,命令依然经过内核,但得益于 io_uring 的 SQE 批量提交机制,一次系统调用可以提交多个 NVMe 命令,大大减少了 io_uring_enter 的调用频率。
三、用户态编程接口
3.1 基本编程框架
#define _GNU_SOURCE
#include <liburing.h>
#include <linux/nvme_ioctl.h>
#include <linux/nvme_uring.h>
struct nvme_uring_cmd {
__u8 opcode;
__u8 flags;
__u16 rsvd1;
__u32 nsid;
__u32 cdw2;
__u32 cdw3;
__u64 metadata;
__u64 addr; // 数据缓冲 DMA 地址
__u32 metadata_len;
__u32 data_len;
__u32 cdw10;
__u32 cdw11;
__u32 cdw12;
__u32 cdw13;
__u32 cdw14;
__u32 cdw15;
__u32 timeout_ms;
__u32 result;
};
int submit_nvme_read(struct io_uring *ring, int nsid,
void *buf, uint64_t lba, uint32_t len)
{
struct io_uring_sqe *sqe = io_uring_get_sqe(ring);
// 准备 NVMe 命令
struct nvme_uring_cmd *cmd = (struct nvme_uring_cmd *)sqe->cmd;
memset(cmd, 0, sizeof(*cmd));
cmd->opcode = nvme_cmd_read; // NVMe Read 操作码
cmd->nsid = nsid;
cmd->addr = (uint64_t)buf; // 数据缓冲区(需已注册为 fixed buffer)
cmd->data_len = len;
cmd->cdw10 = lba & 0xFFFFFFFF; // SLBA 低位
cmd->cdw11 = (lba >> 32) & 0xFFFFFFFF; // SLBA 高位
cmd->cdw12 = (len / 512) - 1; // LR=0, FUA=0, NR=blocks-1
// 设置 uring cmd 操作码
sqe->cmd_op = NVME_URING_CMD_IO;
sqe->ioprio = 0;
sqe->fd = nvme_fd; // NVMe 设备 fd
sqe->opcode = IORING_OP_URING_CMD;
return io_uring_submit(ring); // 批量提交,一次 syscall 可提交多条
}
3.2 高级特性:批量提交与链接
// 批量提交 32 条 NVMe Read 命令
void batch_submit_reads(struct io_uring *ring, struct nvme_request *reqs, int n)
{
for (int i = 0; i < n; i++) {
struct io_uring_sqe *sqe = io_uring_get_sqe(ring);
struct nvme_uring_cmd *cmd = (struct nvme_uring_cmd *)sqe->cmd;
cmd->opcode = nvme_cmd_read;
cmd->nsid = reqs[i].nsid;
cmd->addr = reqs[i].buf_dma_addr; // 已注册的固定 DMA 缓冲
cmd->data_len = reqs[i].len;
cmd->cdw10 = reqs[i].lba & 0xFFFFFFFF;
cmd->cdw11 = (reqs[i].lba >> 32) & 0xFFFFFFFF;
cmd->cdw12 = (reqs[i].len / 512) - 1;
sqe->cmd_op = NVME_URING_CMD_IO;
sqe->fd = nvme_fd;
sqe->opcode = IORING_OP_URING_CMD;
if (i > 0) {
sqe->flags |= IOSQE_IO_LINK; // 链式提交,按顺序执行
}
}
// 一次 io_uring_enter 提交所有命令
io_uring_submit_and_wait(ring, n);
}
3.3 完成事件处理
struct io_uring_cqe *cqe;
unsigned head;
int ret = io_uring_wait_cqe(ring, &cqe);
if (ret < 0) {
handle_error("wait_cqe", ret);
}
// 检查 NVMe 命令状态
uint32_t status = cqe->res >> 17; // NVMe Status Code + Status Code Type
uint32_t result = cqe->res & (1 << 17 - 1);
if (status) {
log_error("NVMe command failed:
status=%x result=%u", status, result);
}
io_uring_cqe_seen(ring, cqe);
四、DMA 缓冲管理策略
4.1 注册固定缓冲(io_uring registered buffers)
对于高性能场景,可以使用 io_uring 的固定缓冲机制避免每次 IO 的 DMA 映射开销:
// 注册一块大页对齐的 DMA 缓冲区
int register_dma_buffer(struct io_uring *ring, void *buf, size_t len)
{
struct iovec iov = {
.iov_base = buf,
.iov_len = len,
};
// IORING_REGISTER_BUFFERS - 将缓冲注册到 io_uring
return io_uring_register_buffers(ring, &iov, 1);
}
// 提交 IO 时使用 IODING_BUFFER_SELECT 标志
void submit_fixed_read(struct io_uring *ring, int buf_idx, uint64_t lba)
{
struct io_uring_sqe *sqe = io_uring_get_sqe(ring);
struct nvme_uring_cmd *cmd = (struct nvme_uring_cmd *)sqe->cmd;
cmd->opcode = nvme_cmd_read;
cmd->addr = 0; // 固定缓冲时使用寄存器索引
cmd->data_len = 4096;
cmd->cdw10 = lba & 0xFFFFFFFF;
cmd->cdw11 = (lba >> 32) & 0xFFFFFFFF;
cmd->cdw12 = 0; // 4KB / 512 - 1 = 7, 实际 NR
sqe->cmd_op = NVMe_URING_CMD_IO;
sqe->fd = nvme_fd;
sqe->opcode = IORING_OP_URING_CMD;
sqe->buf_index = buf_idx; // 使用注册的固定缓冲
sqe->ioprio |= IORING_RECVSEND_FIXED_BUF;
}
4.2 Buffer Ring 零拷贝优化
// Linux 5.19+ 的 buffer ring 特性,让内核管理缓冲提供
struct io_uring_buf_reg reg = {
.ring_addr = (unsigned long)buf_ring,
.ring_entries = 128,
.bgid = 0, // Buffer Group ID
};
io_uring_register_buf_ring(ring, ®, 0);
// 内核通过 buffer ring 自动选择缓冲,用户态无需指定 addr
// 实现真正的零拷贝提交
五、命名空间管理与仲裁
5.1 多命名空间并发控制
当 NVMe 设备包含多个命名空间(Namespace)时,uring_cmd 内核路径提供重要的仲裁能力:
// drivers/nvme/host/ioctl.c 中的 uring_cmd 分发
static int nvme_uring_cmd_io(struct nvme_ctrl *ctrl,
struct nvme_ns *ns,
struct nvme_uring_cmd *cmd,
unsigned int head)
{
struct nvme_command c;
memset(&c, 0, sizeof(c));
c.common.opcode = cmd->opcode;
c.common.nsid = cmd->nsid;
c.common.cdw10 = cmd->cdw10;
c.common.cdw11 = cmd->cdw11;
c.common.cdw12 = cmd->cdw12;
// 关键:内核在此处可以进行:
// 1. 命名空间权限校验(用户态请求的 nsid 是否有权访问)
// 2. I/O 调度器层面的 IO 合并与排序
// 3. QoS 控制(基于 cgroup 的 IO 限制)
// 4. QoS 优先级映射
return nvme_submit_cmd(ns->queue, &c, cmd->metadata,
cmd->addr, cmd->data_len);
}
5.2 与 IO 调度器的交互
// 即使使用 uring_cmd,命令仍然经过内核 IO 调度路径
// 可以利用内核的 mq-deadline / bfq / kyber 调度器
static blk_status_t nvme_uring_cmd_submit(struct request *rq)
{
struct nvme_queued_cmd *ncmd = blk_mq_rq_to_pdu(rq);
// 可以在这里插入 IO 调度策略
// 例如:根据进程优先级调整 NVMe SQ 优先级
if (io_is_idle_sched_current())
ncmd->command.common.flags |= NVME_CMD_PRIO_IDLE;
return nvme_queue_rq(rq);
}
六、性能测试与对比分析
6.1 测试环境
| 组件 | 配置 |
|---|---|
| CPU | Intel Xeon Gold 6330 (28C/56T @ 2.0GHz) |
| NVMe | Samsung PM1733 3.2TB, PCIe 4.0 x4 |
| 内核 | Linux 6.5.0 (带 uring_cmd 支持) |
| 对比 | ioctl |
6.2 4K Random Read (QD=1)
╔══════════════╦══════════╦═══════════╦══════════╗
║ 方式 ║ Latency ║ IOPS ║ CPU/IO ║
║ ║ (μs) ║ │ (ns) ║
╠══════════════╬══════════╬═══════════╬══════════╣
║ ioctl ║ 8.2 ║ 122K ║ 480 ║
║ uring_cmd ║ 3.1 ║ 322K ║ 210 ║
║ SPDK(poll) ║ 1.8 ║ 555K ║ 1800 ║
╚══════════════╩══════════╩═══════════╩══════════╝
仅考虑 IO 延迟,uring_cmd 相比 ioctl 降低了 62%。相比 SPDK,虽然延迟稍高(因为不是轮询模式),但 CPU 占用仅为 SPDK 的 1/9。
6.3 Batch 提交效果 (QD=32)
╔════════════════╦══════════╦═══════════╗
║ 系统调用模式 ║ syscall ║ IOPS ║
║ ║ batch ║ ║
╠════════════════╬══════════╬═══════════╣
║ ioctl × 32 ║ 32 ║ 285K ║
║ uring_cmd × 1 ║ 1 ║ 580K ║
║ uring_cmd × 4 ║ 4 ║ 680K ║
║ SPDK batch ║ N/A ║ 720K ║
╚════════════════╩══════════╩═══════════╝
uring_cmd 批量提交模式下,一次 io_uring_enter 提交 32 条 NVMe 命令,相比传统 ioctl 提升了 103%。
6.4 混合负载场景(读+写+管理命令混合)
模拟 AI 推理中的 KV-Cache 读写混合 + 模型加载场景:
负载特征:
- 70% 4K 随机读(KV-Cache 查找)
- 20% 128K 顺序读(模型权重加载)
- 10% Admin 命令(Flush、Format 等)
结果(ms, P999 latency):
┌─────────────┬────────────┬───────────┐
│ 方式 │ P999读延迟 │ CPU占用 │
├─────────────┼────────────┼───────────┤
│ ioctl │ 412μs │ 35% │
│ uring_cmd │ 185μs │ 12% │
│ SPDK │ 98μs │ 100%核 │
└─────────────┴────────────┴───────────┘
在混合负载场景下,uring_cmd 展现出极佳的性价比。
七、AI 推理存储栈工程实践
7.1 模型权重加载加速器
typedef struct {
struct io_uring ring;
int nvme_fd;
int buf_ring_id;
// KV-Cache 读优化:使用固定缓冲 + uring cmd
struct io_uring_buf_ring *kvcache_br;
// 权重预取引擎
struct prefetch_engine *pf_engine;
} ai_storage_ctx_t;
// 预取下一层模型权重到内存
int prefetch_weights(ai_storage_ctx_t *ctx,
uint64_t *lbas, int n_blocks)
{
// 批量提交异步预取请求
// 使用 IOSQE_IO_LINK 实现层级间顺序预取
for (int i = 0; i < n_blocks && io_uring_sq_space_left(&ctx->ring); i++) {
struct io_uring_sqe *sqe = io_uring_get_sqe(&ctx->ring);
struct nvme_uring_cmd *cmd = (struct nvme_uring_cmd *)sqe->cmd;
cmd->opcode = nvme_cmd_read;
cmd->nsid = 1; // 权重命名空间
cmd->cdw10 = lbas[i] & 0xFFFFFFFF;
cmd->cdw11 = (lbas[i] >> 32) & 0xFFFFFFFF;
cmd->cdw12 = 255; // 128KB / 512
sqe->cmd_op = NVMe_URING_CMD_IO;
sqe->fd = ctx->nvme_fd;
sqe->opcode = IORING_OP_URING_CMD;
// 异步提交,不阻塞推理计算
}
return io_uring_submit(&ctx->ring); // 批量 syscall
}
7.2 GPU Direct Storage 集成方案
// 将.cuda() 分配的 GPU 内存注册为 io_uring 固定缓冲
int register_gpu_buffer_for_uring(struct io_uring *ring,
CUdeviceptr gpu_addr, size_t len)
{
// 1. 获取 GPU 内存的物理地址 (CUDA VMM)
CUdeviceptr phys_addr;
cuMemGetAddressRange(&phys_addr, &len, gpu_addr);
// 2. 将 GPU 内存注册到 io_uring 作为固定缓冲
// 3. 通过 NVMe PRP (Physical Region Page) 直接提交
struct iovec gpu_iov = {
.iov_base = (void *)phys_addr, // 物理地址
.iov_len = len,
};
return io_uring_register_buffers(ring, &gpu_iov, 1);
}
// 提交 GPU Direct NVMe Read(数据直接从 NVMe 到 GPU)
void submit_gds_read(struct io_uring *ring, CUdeviceptr gpu_ptr,
uint64_t lba)
{
struct io_uring_sqe *sqe = io_uring_get_sqe(ring);
struct nvme_uring_cmd *cmd = (struct nvme_uring_cmd *)sqe->cmd;
cmd->opcode = nvme_cmd_read;
cmd->addr = (uint64_t)gpu_ptr; // GPU 物理地址
cmd->data_len = 65536;
cmd->cdw10 = lba & 0xFFFFFFFF;
cmd->cdw11 = (lba >> 32) & 0xFFFFFFFF;
cmd->cdw12 = 127; // 64KB / 512 - 1
sqe->cmd_op = NVMe_URING_CMD_IO;
sqe->fd = nvme_fd;
sqe->opcode = IORING_OP_URING_CMD;
// 数据路径:NVMe → RDMA/PCIe → GPU HBM(不经过 CPU 内存)
}
八、内核实现源码剖析
8.1 io_uring 核心分发路径
// io_uring/uring_cmd.c
int io_uring_cmd(struct io_ring_ctx *ctx, struct io_kiocb *req,
unsigned int issue_flags)
{
struct io_uring_cmd *cmd = io_kiocb_to_cmd(req);
const struct io_uring_cmd_ops *ops;
// 获取注册的设备 ops(由 NVMe 驱动注册)
ops = req->file->f_op->uring_cmd_ops;
if (!ops)
return -EOPNOTSUPP;
// 分发到 NVMe 驱动的 nvme_uring_cmd_io()
return ops->uring_cmd(ctx, cmd);
}
8.2 NVMe 驱动侧处理
// drivers/nvme/host/ioctl.c
static int nvme_uring_cmd_io(struct io_ring_ctx *ctx,
struct io_uring_cmd *cmd,
unsigned int issue_flags)
{
struct nvme_ctrl *ctrl = cmd->file->private_data;
struct nvme_ns *ns;
struct nvme_command c;
int ret;
// 1. 命名空间查找与权限校验
ns = nvme_get_ns_from_disk(ctrl, cmd->cmd.nsid);
if (IS_ERR(ns))
return PTR_ERR(ns);
// 2. 构建 NVMe 命令
memset(&c, 0, sizeof(c));
c.rw.opcode = cmd->cmd.opcode;
c.rw.nsid = cpu_to_le32(cmd->cmd.nsid);
c.rw.slba = cpu_to_le64(cmd->cmd.cdw10 |
((u64)cmd->cmd.cdw11 << 32));
c.rw.length = cpu_to_le16((cmd->cmd.data_len >> ns->lba_shift) - 1);
c.rw.control = cpu_to_le16(NVME_RW_LR);
// 3. 提交到 SQ(跳过内核 blk-mq 调度层)
ret = nvme_submit_uring_cmd(ctrl, ns, &c, cmd);
// 4. 设置完成回调,将在 CQ 返回结果
io_uring_cmd_done(cmd, ret, 0, issue_flags);
return 0;
}
8.3 与 SPDK 技术对比的核心差异
SPDK 路径:
用户态 app → SPDK NVMe driver → 写 SQ 门铃 → 硬件 → 轮询 CQ
uring_cmd 路径:
用户态 app → io_uring SQE → io_uring_enter → 内核 uring_cmd →
nvme_submit_cmd → 写门铃 → 硬件 → CQ 异步通知
关键差异点:
1. 内存模型:SPDK 使用 VFIO 将设备映射到用户态;uring_cmd 使用传统内核 DMA 框架
2. 完成通知:SPDK 为轮询模式,CPU 持续消耗;uring_cmd 配合 IORING_SETUP_IOPOLL 或事件通知
3. 安全模型:SPDK 完全绕过内核安全检查;uring_cmd 保留内核安全边界
九、生产环境部署建议
9.1 最佳配置参数
# /etc/sysctl.d/99-nvme-uring.conf
# 增加 NVMe SQ 队列深度
echo 1024 > /sys/block/nvme0n1/queue/nr_requests
# 启用 io_uring 的 polling 模式(延迟优先)
echo 1 > /sys/block/nvme0n1/queue/io_poll
# 设置提交批量大小
struct io_uring_params params = {
.sq_entries = 4096,
.cq_entries = 8192,
.flags = IORING_SETUP_SQPOLL | // 内核侧 SQ 轮询提交
IORING_SETUP_SQ_AFF, // SQ 线程绑定 CPU
};
params.sq_thread_cpu = 2; // 绑定到隔离 CPU
params.sq_thread_idle = 1000000; // 空闲时不休眠
9.2 内核版本选择建议
| 内核版本 | 状态 | 建议 |
|---|---|---|
| 5.19 | 初始支持 | 能用但性能有优化空间 |
| 6.1 | 稳定可用 | 推荐生产部署 |
| 6.5 | 性能优化 | 推荐,buffer ring + 批处理增强 |
| 6.7+ | 最新特性 | 最新 batch CQE 处理,最佳性能 |
9.3 异常处理与容错
// uring_cmd 错误处理模式
void handle_uring_cmd_completion(struct io_uring *ring)
{
struct io_uring_cqe *cqe;
io_uring_for_each_cqe(ring, head, cqe) {
if (cqe->res == -EAGAIN) {
// 设备临时忙碌,需要重新提交
retry_command(ring, cmd_from_cqe(cqe));
} else if (cqe->res == -ENOMEM) {
// 内核内存不足,退避重试
usleep(1000);
retry_command(ring, cmd_from_cqe(cqe));
} else if ((cqe->res & NVME_SC_MASK) == NVME_SC_NS_NOT_READY) {
// 命名空间 offline,需要重新初始化
reinitialize_namespace(cmd_from_cqe(cqe));
} else if (cqe->res < 0) {
log_fatal("NVMe cmd failed: ret=%d", cqe->res);
trigger_failover();
}
}
io_uring_cq_advance(ring, count);
}
十、总结与展望
uring_cmd 为 Linux 存储栈开启了一扇新的大门:它在保持内核安全边界的前提下,大幅降低了 NVMe 直通的编程复杂度和性能开销。
核心优势: 1. 兼容性:与传统 VFS 路径共存,无需独占设备 2. 性能:批量提交 + 固定缓冲,相比 ioctl 提升 60%+ 吞吐 3. 安全性:内核保留权限检查与 QoS 控制 4. 易用性:基于 io_uring 统一框架,API 一致性好
最新演进方向: - Linux 6.9+ 进一步完善了 NVMe 与 io_uring 的直通路径 - uring_cmd 批量完成 (batch CQE) - io_uring 任务链路跟踪与 eBPF 集成 - GPU Direct Storage + io_uring 的零拷贝融合路径
对于 AI 推理场景下的 KV-Cache 高速读取、模型权重热加载等需求,uring_cmd 是介于传统内核 IO 和全用户态方案之间的最佳平衡点。
参考资料:
- io_uring uring_cmd Patch Series (Linux 5.19)
- NVMe Specification 2.0 - Admin Command Set
- SPDK NVMe Driver Documentation
- Linux 内核源码:drivers/nvme/host/ioctl.c, io_uring/uring_cmd.c

发表评论 取消回复