Linux内核io_uring uring-cmd:用户态NVMe Passthrough实战与存储性能革命
一、存储I/O演进:从ioctl到uring-cmd
在存储性能优化的演进历程中,我们经历了三次范式革命:
- 同步I/O时代:read/write系统调用、pread/pwrite
- 异步I/O时代:POSIX AIO、libaio(通过io_submit/io_getevents)
- 革命性时代:io_uring(Linux 5.1+),统一异步I/O框架
然而,对于高性能NVMe SSD场景,即便是io_uring也有其局限。上层文件系统抽象层的开销始终存在。当我们需要直接与硬件NVMe控制器交互时——执行SMART查询、格式化 Namespace、自定义 Vendor-Specific 命令——传统做法是走NVMe ioctl(NVME_IOCTL_ADMIN_CMD / NVME_IOCTL_IO_CMD),这种同步方式与现代高性能应用格格不入。
uring-cmd(于 Linux 5.19+ 正式引入)正是填补这一空白的利器:它将NVMe命令直接嵌入io_uring的submission/completion队列,实现真正的用户态异步NVMe直通。
二、uring-cmd架构深度解析
2.1 NVMe命令结构
NVMe命令由64字节的Submission Queue Entry(SQE)定义。uring-cmd通过IORING_OP_URING_CMD操作码,将NVMe命令无缝封装到io_uring的SQE中:
// NVMe Submission Queue Entry 基础定义
struct nvme_command {
union {
struct nvme_common_command common; // 通用命令
struct nvme_rw_command rw; // 读写命令
struct nvme_identify identify; // Identify命令
struct nvme_features features; // Get/Set Features
struct nvme_format_cmd format; // Format NVM
struct nvme_sanitize_cmd sanitize; // Sanitize
struct nvme_commandEffects_log effects;
struct nvme_ns_attachment ns_attach;
struct nvme_fw_download fw_download;
struct nvme_fw_commit fw_commit;
struct nvme_download_firmware dlfirmware;
struct nvme_directive_send directive_send;
struct nvme_directive_recv directive_recv;
struct nvme_dsm_cmd dsm;
};
};
2.2 io_uring uring-cmd 数据流
用户态程序
│
▼
io_uring_enter() ────────── 批量提交 SQE
│
▼
Kernel io_uring 子系统 ──── 识别 IORING_OP_URING_CMD
│
▼
块层 NVMe 驱动 ──────────── 将 SQE 转换为 NVMe SQ Entry
│
▼
NVMe SSD 硬件控制器 ────── 执行命令
│
▼
Completion Queue (CQ) ──── 写入 CQE (Completion Queue Entry)
│
▼
用户态通过 io_uring_enter() 轮询/CQE 通知获取结果
2.3 关键io_uring接口
uring-cmd的接口设计非常优雅,核心是使用io_uring_prep_uring_cmd函数:
#include <liburing.h>
#include <linux/nvme_ioctl.h>
struct io_uring_sqe *sqe = io_uring_get_sqe(&ring);
struct nvme_uring_cmd *cmd = (struct nvme_uring_cmd *)&sqe->cmd;
// 设置操作码为 uring-cmd
sqe->opcode = IORING_OP_URING_CMD;
sqe->fd = nvme_fd; // 打开的 NVMe 字符设备 fd
// 填充 NVMe uring 命令
cmd->opcode = nvme_admin_identify; // NVMe 命令操作码
cmd->nsid = 0xFFFFFFFF; // 控制器级别命令
cmd->addr = (__u64)data_buffer; // 数据缓冲区
cmd->data_len = 4096; // 数据长度
cmd->cdw10 = 1; // Identify CNS (Controller)
io_uring_submit(&ring);
三、环境搭建与前置条件
3.1 内核要求
# 确认内核版本 >= 5.19
uname -r
# 输出应 >= 5.19.0
# 确认 io_uring 支持
grep IO_URING /boot/config-$(uname -r)
# CONFIG_IO_URING=y
# 确认 uring-cmd 支持 (Linux 5.9+)
grep BLK_URING_CMD /boot/config-$(uname -r)
3.2 安装 liburing
# Ubuntu/Debian
sudo apt install liburing-dev
# Fedora/RHEL
sudo dnf install liburing-devel
# 编译安装最新版
git clone https://github.com/axboe/liburing.git
cd liburing
./configure
make -j$(nproc)
sudo make install
3.3 NVMe 设备权限
# 确认 NVMe 设备节点
ls -la /dev/nvme*
# 输出示例:
# brw-rw---- 1 root disk 259, 0 Sep 29 10:00 /dev/nvme0
# crw-rw---- 1 root disk 243, 0 Sep 29 10:00 /dev/nvme0n1
# crw-rw---- 1 root root 10, 58 Sep 29 10:00 /dev/nvme-fabrics
# 需要root权限或属于disk组才能访问
# 或配置udev规则:
echo 'KERNEL=="nvme*[0-9]", GROUP="disk", MODE="0660"' | sudo tee /etc/udev/rules.d/99-nvme.rules
sudo udevadm control --reload-rules
sudo udevadm trigger
3.4 检查设备NVMe版本与特性
# 安装nvme-cli工具
sudo apt install nvme-cli
# 查看控制器信息
sudo nvme id-ctrl /dev/nvme0
# 查看支持的命令集
sudo nvme id-ns /dev/nvme0n1
四、核心代码实战:用户态NVMe Identify查询
下面通过完整的实战代码,展示如何利用uring-cmd异步获取NVMe控制器的Identify信息。
4.1 完整示例代码
// uring_cmd_nvme_identify.c
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <fcntl.h>
#include <unistd.h>
#include <errno.h>
#include <linux/io_uring.h>
#include <linux/nvme_ioctl.h>
#include <liburing.h>
#define QUEUE_DEPTH 4
#define DATA_LEN 4096
#define NVME_Uring_CMD_MAGIC 0x76756273 // 'nvrs'
// NVMe uring 命令结构体
// 实际上这是 struct nvme_uring_cmd,为了兼容性这里手动定义
struct my_nvme_uring_cmd {
__u8 opcode;
__u8 flags;
__u16 rsvd1;
__u32 nsid;
__u32 cdw2;
__u32 cdw3;
__u64 metadata;
__u64 addr;
__u32 metadata_len;
__u32 data_len;
__u32 cdw10;
__u32 cdw11;
__u32 cdw12;
__u32 cdw13;
__u32 cdw14;
__u32 cdw15;
__u32 timeout_ms;
__u32 result;
};
// NVMe Identify Controller 返回数据 (4096 bytes)
struct nvme_id_ctrl {
__u16 vid;
__u16 ssvid;
__u8 sn[20];
__u8 mn[40];
__u8 fr[8];
__u8 rab;
__u8 ieee[3];
__u8 cmic;
__u8 mdts;
__u16 cntlid;
__u32 ver;
__u32 rtd3r;
__u32 rtd3e;
__u8 oaes;
__u8 ctratt;
__u8 rsvd100[124];
// ... 更多字段
// 我们只展示关键字段
__u8 cass[8]; // Command Ambiguity Slot
__u16 cntrltype;
// 这是一个简化的定义,实际定义见 nvme.h
};
static int setup_io_uring(struct io_uring *ring)
{
struct io_uring_params params;
int ret;
memset(¶ms, 0, sizeof(params));
// 启用所需的特性
params.flags = 0;
ret = io_uring_queue_init_params(QUEUE_DEPTH, ring, ¶ms);
if (ret < 0) {
fprintf(stderr, "io_uring init failed: %s\n", strerror(-ret));
return ret;
}
// 检查是否支持所需的特性
if (!(params.features & IORING_FEAT_SINGLE_MMAP)) {
fprintf(stderr, "Kernel doesn't support SINGLE_MMAP\n");
ret = -ENOTSUP;
goto err;
}
return 0;
err:
io_uring_queue_exit(ring);
return ret;
}
static int submit_identify(struct io_uring *ring, int fd, void *data_buf)
{
struct io_uring_sqe *sqe;
struct my_nvme_uring_cmd *cmd;
// 获取 SQE
sqe = io_uring_get_sqe(ring);
if (!sqe) {
fprintf(stderr, "Failed to get SQE\n");
return -ENOMEM;
}
// 设置 io_uring SQE 头部
memset(sqe, 0, sizeof(*sqe));
sqe->fd = fd;
sqe->opcode = IORING_OP_URING_CMD;
sqe->cmd_op = NVME_URING_CMD_IO; // IO 队列命令
sqe->user_data = 0x12345678; // 用户数据标识
// 设置 NVMe uring 命令
cmd = (struct my_nvme_uring_cmd *)&sqe->cmd;
memset(cmd, 0, sizeof(*cmd));
// NVMe Identify 命令参数
cmd->opcode = 0x06; // admin identify 命令操作码
cmd->nsid = 0xFFFFFFFF; // 控制器级别命令
cmd->cdw10 = 1; // CNS = 1 (Controller identify)
cmd->data_len = DATA_LEN;
cmd->addr = (__u64)data_buf;
printf("[SUBMIT] 提交 NVMe Identify 命令\n");
printf(" opcode=0x%02x, nsid=0x%x, cdw10=%u, data_len=%u\n",
cmd->opcode, cmd->nsid, cmd->cdw10, cmd->data_len);
// 提交到内核
int ret = io_uring_submit(ring);
if (ret < 0) {
fprintf(stderr, "io_uring_submit failed: %s\n", strerror(-ret));
return ret;
}
return 0;
}
static int wait_completion(struct io_uring *ring, void *data_buf)
{
struct io_uring_cqe *cqe;
int ret;
// 等待 CQE
ret = io_uring_wait_cqe(ring, &cqe);
if (ret < 0) {
fprintf(stderr, "io_uring_wait_cqe failed: %s\n", strerror(-ret));
return ret;
}
// 处理 CQE
if (cqe->res < 0) {
fprintf(stderr, "命令执行失败: %s\n", strerror(-cqe->res));
} else {
printf("\n[COMPLETE] NVMe Identify 完成!\n");
printf(" 用户数据: 0x%lx, 结果: %d\n",
(unsigned long)io_uring_cqe_get_data(cqe), cqe->res);
// 解析返回的数据
struct nvme_id_ctrl *ctrl = (struct nvme_id_ctrl *)data_buf;
char sn[21] = {0};
char mn[41] = {0};
char fr[9] = {0};
memcpy(sn, ctrl->sn, 20);
memcpy(mn, ctrl->mn, 40);
memcpy(fr, ctrl->fr, 8);
printf("\n=== NVMe Controller 信息 ===\n");
printf("Vendor ID : 0x%04x\n", ctrl->vid);
printf("Serial Number : %s\n", sn);
printf("Model Number : %s\n", mn);
printf("Firmware Rev : %s\n", fr);
printf("Version : 0x%08x\n", ctrl->ver);
}
// 释放 CQE
io_uring_cqe_seen(ring, cqe);
return cqe->res;
}
int main(int argc, char *argv[])
{
struct io_uring ring;
void *data_buf;
int nvme_fd;
const char *dev_path = "/dev/nvme0";
int ret = EXIT_SUCCESS;
printf("=== uring-cmd NVMe Identify 示例 ===\n\n");
// 打开 NVMe 字符设备
nvme_fd = open(dev_path, O_RDWR);
if (nvme_fd < 0) {
fprintf(stderr, "无法打开 %s: %s\n", dev_path, strerror(errno));
return EXIT_FAILURE;
}
printf("[OK] 打开设备: %s (fd=%d)\n", dev_path, nvme_fd);
// 分配数据缓冲区 (对齐到页边界)
data_buf = aligned_alloc(4096, DATA_LEN);
if (!data_buf) {
fprintf(stderr, "分配数据缓冲区失败\n");
close(nvme_fd);
return EXIT_FAILURE;
}
// 初始化 io_uring
ret = setup_io_uring(&ring);
if (ret < 0) {
free(data_buf);
close(nvme_fd);
return EXIT_FAILURE;
}
printf("[OK] io_uring 初始化完成, depth=%d\n", QUEUE_DEPTH);
// 提交 Identify 命令
ret = submit_identify(&ring, nvme_fd, data_buf);
if (ret < 0) {
ret = EXIT_FAILURE;
goto cleanup;
}
// 等待完成
ret = wait_completion(&ring, data_buf);
if (ret < 0)
ret = EXIT_FAILURE;
cleanup:
io_uring_queue_exit(&ring);
free(data_buf);
close(nvme_fd);
printf("\n=== 示例结束 (exit code: %d) ===\n", ret);
return ret;
}
4.2 编译与运行
# 编译
gcc -o uring_cmd_nvme_identify uring_cmd_nvme_identify.c -luring -Wall -g
# 运行(需要root权限或disk组)
sudo ./uring_cmd_nvme_identify
4.3 预期输出
=== uring-cmd NVMe Identify 示例 ===
[OK] 打开设备: /dev/nvme0 (fd=3)
[OK] io_uring 初始化完成, depth=4
[SUBMIT] 提交 NVMe Identify 命令
opcode=0x06, nsid=0xffffffff, cdw10=1, data_len=4096
[COMPLETE] NVMe Identify 完成!
用户数据: 0x12345678, 结果: 0
=== NVMe Controller 信息 ===
Vendor ID : 0x144d (Samsung)
Serial Number : S6Z2NF0W123456X
Model Number : Samsung SSD 970 EVO Plus 1TB
Firmware Rev : 2B2QEXE7
Version : 0x00030000
=== 示例结束 (exit code: 0) ===
五、进阶实战:异步批量NVMe读写对比
真正的性能优势体现在批量异步I/O场景。下面我们对比三种方式执行1024次4K随机读:
| 方式 | IOPS (单核) | 延迟(μs) | CPU占用 |
|---|---|---|---|
| ioctl (同步) | ~120,000 | ~8.3 | 高(频繁 syscall) |
| io_uring 读 | ~180,000 | ~5.5 | 中(批量提交) |
| uring-cmd NVMe 直通 | ~280,000 | ~3.6 | 低(零文件系统开销) |
5.1 批量读写关键代码
// 批量提交多个NVMe命令
#define BATCH_SIZE 32
int submit_nvme_batch(struct io_uring *ring, int fd,
struct iovec *iovs, off_t *offsets, int count)
{
struct io_uring_sqe *sqe;
struct my_nvme_uring_cmd *cmd;
int i, ret;
for (i = 0; i < count; i++) {
sqe = io_uring_get_sqe(ring);
if (!sqe) {
// SQ满了,先提交已有的
ret = io_uring_submit(ring);
if (ret < 0)
return ret;
sqe = io_uring_get_sqe(ring);
}
sqe->fd = fd;
sqe->opcode = IORING_OP_URING_CMD;
sqe->cmd_op = NVME_URING_CMD_IO;
sqe->user_data = i; // 标识第几个命令
cmd = (struct my_nvme_uring_cmd *)&sqe->cmd;
memset(cmd, 0, sizeof(*cmd));
// NVMe Read 命令 (opcode=0x02)
cmd->opcode = 0x02;
cmd->nsid = 1; // Namespace 1
cmd->cdw10 = (offsets[i] / 512) & 0xFFFFFFFF; // 起始SLBA低32位
cmd->cdw11 = (offsets[i] / 512) >> 32; // SLBA高32位
cmd->cdw12 = 7; // 8个block (4K)
cmd->data_len = 4096;
cmd->addr = (__u64)iovs[i].iov_base;
}
return io_ring_submit(ring);
}
5.2 使用IORING_SETUP_SQPOLL实现真正的零系统调用
int setup_sq_poll(struct io_uring *ring)
{
struct io_uring_params params;
int ret;
memset(¶ms, 0, sizeof(params));
// 启用 SQPOLL - 内核轮询 SQ
params.flags = IORING_SETUP_SQPOLL;
params.sq_thread_idle = 2000; // 空闲2ms后睡眠
params.sq_thread_cpu = 2; // 绑定到CPU 2
ret = io_uring_queue_init_params(QUEUE_DEPTH, ring, ¶ms);
if (ret < 0) {
fprintf(stderr, "SQPOLL init failed: %s\n", strerror(-ret));
return ret;
}
// 检查是否真正启用了 SQPOLL
if (!(params.flags & IORING_SETUP_SQPOLL)) {
fprintf(stderr, "Kernel refused SQPOLL\n");
return -EINVAL;
}
printf("[OK] SQPOLL 启用 - 内核线程ID: %u\n",
params.sq_thread_pid);
return 0;
}
// 当 SQPOLL 启用时,提交命令无需系统调用!
// 只需将 SQE 放入 SQ 并写 SQ tail 指针(共享内存)
void submit_without_syscall(struct io_uring *ring, /* ... */)
{
// 填充 SQE...
// 更新 SQ tail(通过共享内存映射)
// 无需调用 io_uring_enter() !!!
// 内核线程会自动检测到新条目并处理
}
六、生产环境最佳实践
6.1 错误处理与重试策略
uring-cmd的CQE可能返回多种错误码。生产代码需要妥善处理:
static int handle_nvme_error(struct io_uring_cqe *cqe, struct nvme_id_ctrl *data)
{
int status = cqe->res;
if (status >= 0)
return 0; // 成功
// NVMe 状态码 (在 data->result 或特定字段中)
// DNR (Do Not Retry) 判断
// SCT (Status Code Type) / SC (Status Code)
switch (-status) {
case EINTR:
printf("命令被中断,建议重试\n");
return -EAGAIN;
case EIO:
printf("NVMe 操作失败,检查硬件状态\n");
// SMART 错误日志分析
return -EIO;
case ETIMEDOUT:
printf("NVMe 命令超时,可能设备故障\n");
// 检查控制器状态、SMART
return -ETIMEDOUT;
case EINVAL:
printf("参数错误,检查 NVMe 命令格式\n");
return -EINVAL;
case ENODEV:
printf("设备已移除或不可用\n");
return -ENODEV;
case EFAULT:
printf("内存地址无效\n");
return -EFAULT;
default:
printf("未知错误: %s (code=%d)\n", strerror(-status), status);
return status;
}
}
6.2 电源故障保护
NVMe设备在掉电时需要特殊处理。使用uring-cmd的NVME_URING_CMD_IO_VEC支持分散/聚集I/O,减少内存拷贝:
// 使用 RVE 模式(Repeatable Uncached Extended minimizes copying)
#define NVME_URING_CMD_IO _IOWR('N', 0x84, struct my_nvme_uring_cmd)
#define NVME_URING_CMD_IO_VEC _IOWR('N', 0x85, struct my_nvme_uring_cmd)
// 使用 iovec 避免数据拷贝
struct iovec iov[2] = {
{ .iov_base = meta_buf, .iov_len = 512 },
.iov_base = user_data, .iov_len = 4096 },
};
sqe->cmd_op = NVME_URING_CMD_IO_VEC; // 使用向量模式
cmd->addr = (__u64)iov; // iovec 数组地址
cmd->data_len = iov[0].iov_len + iov[1].iov_len;
6.3 监控与告警
// 在 completion 函数中加入监控指标
static uint64_t total_latency_ns = 0;
static uint64_t cmd_count = 0;
void completion_handler(struct io_uring_cqe *cqe, uint64_t start_ns)
{
uint4_t latency_ns = get_time_ns() - start_ns;
total_latency_ns += latency_ns;
cmd_count++;
// 5秒上报一次统计
if (cmd_count % 10000 == 0) {
printf("平均延迟: %.1f μs, 总命令: %lu\n",
(double)total_latency_ns / cmd_count / 1000.0,
cmd_count);
// P99 延迟监控
// CAS指标上报到监控系统
}
// 告警阈值检查
if (latency_ns > 100000) { // > 100ms
fprintf(stderr, "WARNING: NVMe 命令延迟过高: %.2f ms\n",
latency_ns / 1000000.0);
}
}
七、性能调优参数
# 1. 增大 NVMe SQ/CQ 深度
echo 1024 | sudo tee /sys/block/nvme0n1/queue/nr_requests
# 2. 调整 io_uring SQPOLL 线程 CPU 隔离
sudo taskset -pc 2 /sys/block/nvme0n1/io_uring/sq_thread_pid
# 3. NUMA 本地内存分配
numactl --cpunodebind=0 --membind=0 ./uring_cmd_nvme_identify
# 4. 使用 Huge Pages 减少 TLB miss
echo 20 | sudo tee /proc/sys/vm/nr_hugepages
// mmap 时使用 MAP_HUGETLB
# 5. 禁用 I/O 调度器(NVMe 不需要)
echo "none" | sudo tee /sys/block/nvme0n1/queue/scheduler
# 6. 中断亲和性优化
# 将 NVMe 中断绑定到特定CPU,避免跨NUMA
echo 4 | sudo tee /proc/irq/$(cat /proc/interrupts | grep nvme0 | head -1 | cut -d: -f1 | tr -d ' ')/smp_affinity
八、未来展望
uring-cmd 在 Linux 6.x 内核中持续演进。未来的关键发展方向包括:
- ZNS/ZBC 支持:命名空间分区管理命令直通
- TP4146 路径优化:NVMe-MI 管理接口 uring-cmd 支持
- NVMe-oF 扩展:NVMe over Fabrics的异步直通
- 安全加固:用户态 NVMe 命令白名单机制
- vfio-user 集成:将 uring-cmd 与 SPDK/vfio-user 打通,构建用户态存储虚拟化栈
io_uring uring-cmd 不是替代 io_uring 普通 I/O,而是为用户提供了一条绕过文件系统抽象层、直接与 NVMe 硬件对话的异步通道。在高性能数据库(如 TiKV、CockroachDB)、存储引擎(如 RocksDB 的 Direct I/O 增强版)、以及 SPDK 替代方案中,uring-cmd 正在成为现代存储 I/O 栈的关键一环。
参考资源:

发表评论 取消回复