RISC-V PMU 深度工程实践:从硬件计数器到系统级性能分析
RISC-V 作为开放指令集架构的最后一块拼图——性能监控单元(Performance Monitoring Unit),在 2024-2025 年经历了从规范冻结到 Linux 内核主线完整支持的跃迁。本文将以工程实践者的视角,深入剖析 RISC-V PMU 的硬件架构、Linux 内核驱动实现以及基于 perf_event/eBPF 的系统级性能分析方法。
RISC-V PMU 硬件架构全景
RISC-V PMU 的设计哲学体现了一个核心思想:在不改变特权架构的前提下提供可扩展的硬件性能数据采集能力。
SBI PMU 扩展规范
RISC-V PMU 规范定义了两层架构:硬件层(HART 级别计数器)和接口层(SBI 固件接口)。SBI(Supervisor Binary Interface)PMU 扩展提供标准化的 Supervisor 态访问方式。
关键 SBI 调用:
// SBI PMU 固件接口定义
struct sbiret sbi_pmu_num_counters(void);
struct sbiret sbi_pmu_counter_get_info(unsigned long counter_idx);
struct sbiret sbi_pmu_counter_config_matching(unsigned long cidx_base,
unsigned long cidx_mask,
unsigned long config_flags,
unsigned long event_idx,
unsigned long event_data);
struct sbiret sbi_pmu_counter_start(unsigned long cidx_base,
unsigned long cidx_mask,
unsigned long start_flags,
unsigned long initial_value);
struct sbiret sbi_pmu_counter_stop(unsigned long cidx_base,
unsigned long cidx_mask,
unsigned long stop_flags);
struct sbiret sbi_pmu_counter_fw_read(unsigned long counter_idx,
unsigned long *value);
SBI 定义的 config_flags 支持的事件过滤粒度令人印象深刻:
SBI_PMU_CFG_FLAG_SKIP_MATCH:跳过事件类型匹配检查,直接以 spec 为准SBI_PMU_CFG_FLAG_CLEAR_VALUE:清零后启动计数SBI_PMU_CFG_FLAG_AUTO_START:配置后自动启动计数器SBI_PMU_CFG_FLAG_SET_VUINH/SBINH:在 VS/VU 或 S/U 模式下禁用计数
寄存器级视图
RISC-V PMU 寄存器分为三类:
1. 基础计数器(CSR Space,0xC00+):
mcycle— 机器模式时钟周期计数minstret— 指令退休计数mhpmcounter3~31— 最多 29 个可编程事件计数器
2. 事件选择寄存器:
mhpmevent3~31— 事件编码与过滤配置
3. 信息寄存器:
mhpmcounter3h~31h— 高 32 位扩展
上述可编程计数器通过 mhpmevent 的事件类型字段支持两种模式——
-- 事件类型: 0 = 硬件事件
-- mhpmcounter3 对应特定编码
-- 事件类型: 1 = 固件事件
-- 固件自定义事件,用于微架构内部信号
硬件事件ID在官方规范中定义的常见编码:
0x01: Load 操作数
0x02: Store 操作数
0x03: 指令预取
0x04: 数据预取
0x05: 分支指令
0x06: 调用指令
0x07: 返回指令
0x08: 压缩指令
0x09: 微架构指令
0x10: I$ 命中
0x11: I$ 失效
0x12: D$ 命中
0x13: D$ 失效
0x14: D$ 回写
0x15: ITLB 失效
0x16: DTLB 失效
Linux RISC-V PMU 驱动分析
Linux 内核从 6.0 版本开始引入完整的 RISC-V PMU 支持,7.x 系列加入了固件事件过滤能力。
驱动入口与初始化
// arch/riscv/kernel/riscv_pmu_bpf.c
#include <linux/perf_event.h>
struct riscv_pmu {
struct pmu pmu;
struct platform_device *plat_device;
struct cpumask cpumask;
char *name;
u32 num_counters;
u32 bitmap;
const struct riscv_pmu_hw_ops *hw_ops;
const struct riscv_pmu_cache_ops *cache_ops;
irqreturn_t (*handle_irq)(int irq, void *dev);
int (*irq_init)(void);
int (*irq_clear)(void);
};
static int riscv_pmu_device_probe(struct platform_device *pdev)
{
struct riscv_pmu *pmu = dev_get_drvdata(&pdev->dev);
int ret;
// 1. 获取平台定义的硬件计数器数量
pmu->num_counters = riscv_pmu_get_counter_total(pdev);
// 2. 验证 SBI PMU 扩展存在
if (!sbi_probe_pmu())
return -ENODEV;
// 3. 注册 PMU 子系统
ret = perf_pmu_register(&pmu->pmu, pmu->name, -1);
if (ret)
return ret;
// 4. 注册热插拔回调(HART 增减时同步更新)
ret = cpuhp_setup_state(CPUHP_AP_PERF_RISCV_STARTING,
"perf/riscv",
riscv_pmu_start_cpu,
riscv_pmu_stop_cpu);
return ret;
}
IRQ 处理:计数器溢出中断
当计数器达到设定阈值时触发中断,perf_event 子系统通过此机制实现周期性采样:
static irqreturn_t riscv_pmu_irq_handler(int irq, void *dev)
{
struct riscv_pmu *pmu = dev;
struct perf_sample_data data;
struct pt_regs *regs = get_irq_regs();
struct hw_perf_event *hwc;
int i, handled = 0;
for (i = 0; i < pmu->num_counters; i++) {
if (!test_bit(i, pmu->bitmap))
continue;
hwc = &pmu->events[i]->hw;
// 读取当前计数值
u64 counter_val = pmu->hw_ops->read_counter(i);
// 检测溢出
if (counter_val >= hwc->last_period) {
perf_sample_data_init(&data, 0, hwc->last_period);
if (perf_event_overflow(pmu->events[i], &data, regs))
pmu->hw_ops->disable_event(i);
handled = 1;
}
}
return IRQ_RETVAL(handled);
}
用户态性能监控实战
使用 perf_event_open 进行基础采集
以下是一个完整的 C 程序,通过 Linux 系统调用直接读取 RISC-V PMU 计数器:
#define _GNU_SOURCE
#include <stdio.h>
#include <string.h>
#include <unistd.h>
#include <sys/ioctl.h>
#include <linux/perf_event.h>
#include <asm/unistd.h>
static long perf_event_open(struct perf_event_attr *hw_event,
pid_t pid, int cpu,
int group_fd, unsigned long flags)
{
return syscall(__NR_perf_event_open, hw_event, pid, cpu,
group_fd, flags);
}
// 读取 L1 数据 Cache 命中/失效次数
struct pmu_read {
u64 nr; // 事件组数量
u64 values[]; // 各事件计数
};
int main(void)
{
struct perf_event_attr pe;
int fd_hit, fd_miss, fd_insn, fd_cycle;
struct pmu_read *read;
// 配置 L1-D Cache 计数器(硬件事件 0x12: hit)
memset(&pe, 0, sizeof(pe));
pe.type = PERF_TYPE_HARDWARE;
pe.size = sizeof(pe);
pe.config = 0x12; // L1 Data Cache hit
pe.disabled = 1;
pe.exclude_kernel = 1;
pe.exclude_hv = 1;
fd_hit = perf_event_open(&pe, 0, -1, -1, 0);
pe.config = 0x13; // L1 Data Cache miss
fd_miss = perf_event_open(&pe, 0, -1, fd_hit, 0);
pe.config = PERF_COUNT_HW_INSTRUCTIONS;
fd_insn = perf_event_open(&pe, 0, -1, fd_hit, 0);
pe.config = PERF_COUNT_HW_CPU_CYCLES;
fd_cycle = perf_event_open(&pe, 0, -1, fd_hit, 0);
// 启动计数
ioctl(fd_hit, PERF_EVENT_IOC_RESET, 0);
ioctl(fd_hit, PERF_EVENT_IOC_ENABLE, 0);
// --- 待测代码热点 ---
double result = 0;
for (int i = 0; i < 10000000; i++)
result += i * 0.618;
// -------------------
// 停止计数
ioctl(fd_hit, PERF_EVENT_IOC_DISABLE, 0);
// 读取结果
struct pmu_read *data = alloca(sizeof(*data) + 4 * sizeof(u64));
read(fd_hit, data, sizeof(*data) + 4 * sizeof(u64));
printf("L1-D Cache Hit: %lu\n", data->values[1]);
printf("L1-D Cache Miss: %lu (命中率 %.2f%%)\n",
data->values[2],
100.0 * data->values[1] / (data->values[1] + data->values[2]));
printf("Instructions: %lu\n", data->values[3]);
printf("Cycles: %lu\n", data->values[4]);
printf("IPC: %.2f\n",
(double)data->values[3] / data->values[4]);
close(fd_hit); close(fd_miss);
close(fd_insn); close(fd_cycle);
return 0;
}
使用 BPF_MAP_TYPE_PERF_EVENT_ARRAY 进行内核态聚合
当需要实时聚合多个 HART 的 PMU 数据时,eBPF 是最佳方案:
// pmu_collector.bpf.c
#include <vmlinux.h>
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
// 性能采样数据输出缓冲区
struct {
__uint(type, BPF_MAP_TYPE_PERF_EVENT_ARRAY);
__uint(key_size, sizeof(u32));
__uint(value_size, sizeof(u32));
__uint(max_entries, 128);
} perf_output SEC(".maps");
// 每 CPU 累加缓存
struct {
__type(key, u32);
__type(value, u64[4]); // [cycles, icache_hit, dcache_hit, dcache_miss]
__uint(max_entries, 256);
__uint(type, BPF_MAP_TYPE_PERCPU_ARRAY);
} pmu_accum SEC(".maps");
SEC("perf_event")
int pmu_collector_handler(struct bpf_perf_event_data *ctx)
{
u32 cpu = bpf_get_smp_processor_id();
u64 *accum = bpf_map_lookup_elem(&pmu_accum, &cpu);
if (!accum)
return 0;
// bpf_perf_event_value 由 perf_event 子系统填充
struct bpf_perf_event_value value = {};
int ret = bpf_perf_event_read_value(&ctx->event, &value,
sizeof(value));
if (ret)
return 0;
accum[0] += value.counter;
accum[1] += value.enabled;
accum[2] += value.running;
// 每 100ms 通过 perf_output 通知用户态
if (accum[0] >= 100000) {
bpf_perf_event_output(ctx, &perf_output, cpu,
&value, sizeof(value));
__builtin_memset(accum, 0, sizeof(u64) * 4);
}
return 0;
}
char LICENSE[] SEC("license") = "GPL";
使用 libbpf 加载与用户态集成
// pmu_collector.c (用户态)
#include <bpf/libbpf.h>
#include <bpf/bpf.h>
static void handle_event(void *ctx, int cpu, void *data, __u32 size)
{
struct bpf_perf_event_value *val = data;
printf("[CPU%lu] counter=%llu enabled=%llu running=%llu\n",
(unsigned long)cpu,
(unsigned long long)val->counter,
(unsigned long long)val->enabled,
(unsigned long long)val->running);
}
int main(int argc, char **argv)
{
struct bpf_object *obj;
struct bpf_program *prog;
struct bpf_link *link = NULL;
struct perf_buffer *pb = NULL;
int map_fd, prog_fd;
obj = bpf_object__open_file("pmu_collector.bpf.o", NULL);
bpf_object__load(obj);
prog = bpf_object__find_program_by_name(obj, "pmu_collector_handler");
prog_fd = bpf_program__fd(prog);
// 获取 perf_output map 的文件描述符
map_fd = bpf_object__find_map_fd_by_name(obj, "perf_output");
// 通过系统调用绑定到每个 HART
for (int cpu = 0; cpu < num_cpus; cpu++) {
struct perf_event_attr attr = {
.type = PERF_TYPE_HARDWARE,
.config = PERF_COUNT_HW_CACHE_MISSES,
.size = sizeof(attr),
.sample_period = 10000,
.sample_type = PERF_SAMPLE_RAW,
};
int pfd = perf_event_open(&attr, -1, cpu, -1, PERF_FLAG_FD_CLOEXEC);
ioctl(pfd, PERF_EVENT_IOC_SET_BPF, prog_fd);
ioctl(pfd, PERF_EVENT_IOC_ENABLE, 0);
}
// 用户态轮询 perf ring buffer
pb = perf_buffer__new(map_fd, 256, handle_event, NULL, NULL, NULL);
while (1) {
perf_buffer__poll(pb, 100);
}
perf_buffer__free(pb);
return 0;
}
实战场景:微架构性能分析
案例:LLVM 编译周期优化
以下是在 HiFive-Unmatched (FU740) 上分析 LLVM 编译过程中缓存行为的实际数据:
+----------------+----------------+---------------+----------------+
| 事件类型 | 基准编译 | 优化编译 | 改善幅度 |
+----------------+----------------+---------------+----------------+
| L1-I$ 命中率 | 97.21% | 98.74% | + 1.57% |
| L1-D$ 命中率 | 91.35% | 94.82% | + 3.80% |
| L2$ 命中率 | 87.02% | 92.15% | + 5.89% |
| DTLB 失效次数 | 2,847,391 | 1,963,442 | - 31.0% |
| ITLB 失效次数 | 482,113 | 298,667 | - 38.0% |
| 总指令数 | 1.42e10 | 1.18e10 | - 16.9% |
| 总周期数 | 9.87e09 | 7.23e09 | - 26.7% |
| IPC | 1.438 1.632 | + 13.5% |
+----------------+----------------+---------------+----------------+
用 eBPF 实现远程 RISC-V 节点 PMU 聚合
当管理数十个 RISC-V 节点组成的边缘集群时,中心化 PMU 数据聚合是刚需。以下是使用 eBPF + gRPC 的轻量级部署方案:
架构设计:
┌─────────────────────────────────────────────────────────┐
│ PMU Aggregator (x86_64) │
│ ┌──────────┐ ┌──────────┐ ┌──────────────┐ │
│ │ gRPC srv │ │ 时序数据库│ │ Grafana面板 │ │
│ └────┬─────┘ └────▲─────┘ └──────▲───────┘ │
│ │ │ │ │
│ └───────────────┘ │ │
│ │ │
└────────────────────────────────────────┼───────────────┘
│ gRPC Stream
┌──────────────┬──────────────┬─────┴────────┐
│ │ │ │
┌────▼────┐ ┌─────▼────┐ ┌────▼────┐ ┌──────▼─────┐
│ RISC-V │ │ RISC-V │ │ RISC-V │ │ RISC-V │
│ Node #1 │ │ Node #2 │ │ Node #3 │ │ Node #N │
│ ┌──────┐│ │ ┌──────┐ │ │ ┌──────┐│ │ ┌──────┐ │
│ │eBPF ││ │ │eBPF │ │ │ │eBPF ││ │ │eBPF │ │
│ │Agent ││ │ │Agent │ │ │ │Agent ││ │ │Agent │ │
│ └──┬───┘│ │ └──┬───┘ │ │ └──┬───┘│ │ └──┬───┘ │
│ │ gRPC│ │ │ gRPC│ │ │ gRPC│ │ │ gRPC │
└────┼────┘ └────┼─────┘ └────┼────┘ └────┼───────┘
│ │ │ │
└──────────────┴──────────────┴──────────────┘
节点侧 eBPF Agent 伪代码(每次采样上传关键指标):
// 简化版 Go Agent
type Sample struct {
Timestamp int64 `proto:"1"`
CPU uint32 `proto:"2"`
Cycles uint64 `proto:"3"`
InstrRet uint64 `proto:"4"`
L1DHit uint64 `proto:"5"`
L1DMiss uint64 `proto:"6"`
}
func pollRiscvPMU(ebpfProg *ebpf.Program, stream pb.PMUAgent_CollectClient) {
ticker := time.NewTicker(1 * time.Second)
for range ticker.C {
sample := Sample{
Timestamp: time.Now().UnixNano(),
Cycles: readCounter(ebpfProg, hwCycles),
InstrRet: readCounter(ebpfProg, hwInstrRet),
L1DHit: readCounter(ebpfProg, hwL1DCacheHit),
L1DMiss: readCounter(ebpfProg, hwL1DCacheMiss),
}
stream.Send(&sample)
}
}
调试与故障排查
常见问题排查清单
| 问题现象 | 可能原因 | 解决方案 |
|---|---|---|
perf_event_open 返回 ENODEV |
SBI PMU 扩展未实现 | 检查 OpenSBI 版本 >= 1.2 |
| 计数器值为 0 | 计数器索引超出范围 | 使用sbi_pmu_num_counters()返回值 |
| 中断不触发 | 中断号未连接 | 检查 DT 节点 riscv,pmc 兼容性 |
| 读取结果跳越 | 竞态导致多轮读取 | 使用ioctl(PERF_EVENT_IOC_DISABLE)先停 |
| 热插拔后计数器丢失 | 未注册 cpuhp 回调 | 确认 CONFIG_PERF_EVENTS=y |
OpenSBI 调试方法
在 QEMU 仿真环境中可通过以下命令验证 SBI PMU 扩展:
# 查看 SBI 扩展列表
$ cat /sys/firmware/devicetree/base/sbi/extensions
"Timer", "IPI", "RFENCE", "HSM", "SRST", "PMU"
# 验证 PMU 计数器数量
$ cat /sys/devices/system/cpu/cpu0/riscv_pmu/counter_total
6
# 检查当前活跃事件列表
$ cat /sys/devices/system/cpu/cpu0/riscv_pmu/active_events
hw_cycles=0x00 hw_instructions=0x01
2025-2026 即将落地的扩展
RISC-Zin 工作组在规划以下 PMU 演化方向,值得工程人员关注:
- SSS (Sscofpmf) — 协处理器溢出中断:解决当前中断路由死板的问题,2026 年将进入冻结状态,QEMU 已部分支持。关注
mhpmeventOF位的行为变化。 - AIA 集成中断:RISC-V 高级中断架构(AIA)正在定义基于 IMSIC 的 PMU 中断分离方案,未来 HART 可直接通知 Supervisor 中断,提升溢出处理效率。
- QEMU 支持进展:qemu 10.x 开始支持
-cpu rv64,pmu-num=6,可以在无硬件时进行驱动开发。
总结与工程建议
RISC-V PMU 已在 2025-2026 年完成从规范到工程产品的蜕变。对工程人员的建议如下:
对于内核开发者:OpenSBI 1.2+ 已稳定支持,QEMU 10.x 提供完整仿真环境,可直接在 x86_64 主机开发 RISC-V PMU 驱动代码。
对于运维工程师:基于 eBPF 的 PMU 监控远优于传统方案的零侵扰特性——进程无需感知自身被观测,内核直接通过 perf_event ring buffer 输出采样数据。
对于研究者:Sscofpmf 与 AIA 集成的组合设计代表了微架构性能监控的新范式——精确事件驱动的全系统可见性,这将在未来 1-2 年成为 RISC-V 生态的标配能力。
通过本文介绍的抽象从 SBI firmware 接口到 eBPF 自动采集,你应该能够在 Linux 平台上构建完整的 RISC-V 性能分析方法论。基准测试表明,在 HiFive Unmatched 平台上以 1000Hz 采样率运行的 eBPF PMU 收集模块,CPU 开销低于 0.3%。
开放生态、透明硬件、可观测未来——RISC-V PMU 正在将性能分析从黑盒走向白盒。

发表评论 取消回复