RISC-V PMU 深度工程实践:从硬件计数器到系统级性能分析

RISC-V 作为开放指令集架构的最后一块拼图——性能监控单元(Performance Monitoring Unit),在 2024-2025 年经历了从规范冻结到 Linux 内核主线完整支持的跃迁。本文将以工程实践者的视角,深入剖析 RISC-V PMU 的硬件架构、Linux 内核驱动实现以及基于 perf_event/eBPF 的系统级性能分析方法。

RISC-V PMU 硬件架构全景

RISC-V PMU 的设计哲学体现了一个核心思想:在不改变特权架构的前提下提供可扩展的硬件性能数据采集能力。

SBI PMU 扩展规范

RISC-V PMU 规范定义了两层架构:硬件层(HART 级别计数器)和接口层(SBI 固件接口)。SBI(Supervisor Binary Interface)PMU 扩展提供标准化的 Supervisor 态访问方式。

关键 SBI 调用:


// SBI PMU 固件接口定义
struct sbiret sbi_pmu_num_counters(void);
struct sbiret sbi_pmu_counter_get_info(unsigned long counter_idx);
struct sbiret sbi_pmu_counter_config_matching(unsigned long cidx_base,
                                              unsigned long cidx_mask,
                                              unsigned long config_flags,
                                              unsigned long event_idx,
                                              unsigned long event_data);
struct sbiret sbi_pmu_counter_start(unsigned long cidx_base,
                                     unsigned long cidx_mask,
                                     unsigned long start_flags,
                                     unsigned long initial_value);
struct sbiret sbi_pmu_counter_stop(unsigned long cidx_base,
                                    unsigned long cidx_mask,
                                    unsigned long stop_flags);
struct sbiret sbi_pmu_counter_fw_read(unsigned long counter_idx,
                                       unsigned long *value);

SBI 定义的 config_flags 支持的事件过滤粒度令人印象深刻:

  • SBI_PMU_CFG_FLAG_SKIP_MATCH:跳过事件类型匹配检查,直接以 spec 为准
  • SBI_PMU_CFG_FLAG_CLEAR_VALUE:清零后启动计数
  • SBI_PMU_CFG_FLAG_AUTO_START:配置后自动启动计数器
  • SBI_PMU_CFG_FLAG_SET_VUINH/SBINH:在 VS/VU 或 S/U 模式下禁用计数

寄存器级视图

RISC-V PMU 寄存器分为三类:

1. 基础计数器(CSR Space,0xC00+):

  • mcycle — 机器模式时钟周期计数
  • minstret — 指令退休计数
  • mhpmcounter3~31 — 最多 29 个可编程事件计数器

2. 事件选择寄存器:

  • mhpmevent3~31 — 事件编码与过滤配置

3. 信息寄存器:

  • mhpmcounter3h~31h — 高 32 位扩展

上述可编程计数器通过 mhpmevent 的事件类型字段支持两种模式——


-- 事件类型: 0 = 硬件事件
  -- mhpmcounter3 对应特定编码
-- 事件类型: 1 = 固件事件
  -- 固件自定义事件,用于微架构内部信号

硬件事件ID在官方规范中定义的常见编码:


0x01: Load 操作数
0x02: Store 操作数
0x03: 指令预取
0x04: 数据预取
0x05: 分支指令
0x06: 调用指令
0x07: 返回指令
0x08: 压缩指令
0x09: 微架构指令
0x10: I$ 命中
0x11: I$ 失效
0x12: D$ 命中
0x13: D$ 失效
0x14: D$ 回写
0x15: ITLB 失效
0x16: DTLB 失效

Linux RISC-V PMU 驱动分析

Linux 内核从 6.0 版本开始引入完整的 RISC-V PMU 支持,7.x 系列加入了固件事件过滤能力。

驱动入口与初始化


// arch/riscv/kernel/riscv_pmu_bpf.c
#include <linux/perf_event.h>

struct riscv_pmu {
    struct pmu              pmu;
    struct platform_device  *plat_device;
    struct cpumask          cpumask;
    char                    *name;
    
    u32                     num_counters;
    u32                     bitmap;
    
    const struct riscv_pmu_hw_ops    *hw_ops;
    const struct riscv_pmu_cache_ops *cache_ops;
    
    irqreturn_t (*handle_irq)(int irq, void *dev);
    int (*irq_init)(void);
    int (*irq_clear)(void);
};

static int riscv_pmu_device_probe(struct platform_device *pdev)
{
    struct riscv_pmu *pmu = dev_get_drvdata(&pdev->dev);
    int ret;
    
    // 1. 获取平台定义的硬件计数器数量
    pmu->num_counters = riscv_pmu_get_counter_total(pdev);
    
    // 2. 验证 SBI PMU 扩展存在
    if (!sbi_probe_pmu())
        return -ENODEV;
    
    // 3. 注册 PMU 子系统
    ret = perf_pmu_register(&pmu->pmu, pmu->name, -1);
    if (ret)
        return ret;
    
    // 4. 注册热插拔回调(HART 增减时同步更新)
    ret = cpuhp_setup_state(CPUHP_AP_PERF_RISCV_STARTING,
                            "perf/riscv",
                            riscv_pmu_start_cpu,
                            riscv_pmu_stop_cpu);
    return ret;
}

IRQ 处理:计数器溢出中断

当计数器达到设定阈值时触发中断,perf_event 子系统通过此机制实现周期性采样:


static irqreturn_t riscv_pmu_irq_handler(int irq, void *dev)
{
    struct riscv_pmu *pmu = dev;
    struct perf_sample_data data;
    struct pt_regs *regs = get_irq_regs();
    struct hw_perf_event *hwc;
    int i, handled = 0;
    
    for (i = 0; i < pmu->num_counters; i++) {
        if (!test_bit(i, pmu->bitmap))
            continue;
            
        hwc = &pmu->events[i]->hw;
        
        // 读取当前计数值
        u64 counter_val = pmu->hw_ops->read_counter(i);
        
        // 检测溢出
        if (counter_val >= hwc->last_period) {
            perf_sample_data_init(&data, 0, hwc->last_period);
            
            if (perf_event_overflow(pmu->events[i], &data, regs))
                pmu->hw_ops->disable_event(i);
                
            handled = 1;
        }
    }
    
    return IRQ_RETVAL(handled);
}

用户态性能监控实战

使用 perf_event_open 进行基础采集

以下是一个完整的 C 程序,通过 Linux 系统调用直接读取 RISC-V PMU 计数器:


#define _GNU_SOURCE
#include <stdio.h>
#include <string.h>
#include <unistd.h>
#include <sys/ioctl.h>
#include <linux/perf_event.h>
#include <asm/unistd.h>

static long perf_event_open(struct perf_event_attr *hw_event,
                           pid_t pid, int cpu,
                           int group_fd, unsigned long flags)
{
    return syscall(__NR_perf_event_open, hw_event, pid, cpu,
                   group_fd, flags);
}

// 读取 L1 数据 Cache 命中/失效次数
struct pmu_read {
    u64 nr;       // 事件组数量
    u64 values[]; // 各事件计数
};

int main(void)
{
    struct perf_event_attr pe;
    int fd_hit, fd_miss, fd_insn, fd_cycle;
    struct pmu_read *read;
    
    // 配置 L1-D Cache 计数器(硬件事件 0x12: hit)
    memset(&pe, 0, sizeof(pe));
    pe.type   = PERF_TYPE_HARDWARE;
    pe.size   = sizeof(pe);
    pe.config = 0x12;  // L1 Data Cache hit
    pe.disabled       = 1;
    pe.exclude_kernel = 1;
    pe.exclude_hv     = 1;
    fd_hit = perf_event_open(&pe, 0, -1, -1, 0);
    
    pe.config = 0x13;  // L1 Data Cache miss
    fd_miss = perf_event_open(&pe, 0, -1, fd_hit, 0);
    
    pe.config = PERF_COUNT_HW_INSTRUCTIONS;
    fd_insn = perf_event_open(&pe, 0, -1, fd_hit, 0);
    
    pe.config = PERF_COUNT_HW_CPU_CYCLES;
    fd_cycle = perf_event_open(&pe, 0, -1, fd_hit, 0);
    
    // 启动计数
    ioctl(fd_hit, PERF_EVENT_IOC_RESET, 0);
    ioctl(fd_hit, PERF_EVENT_IOC_ENABLE, 0);
    
    // --- 待测代码热点 ---
    double result = 0;
    for (int i = 0; i < 10000000; i++)
        result += i * 0.618;
    // -------------------
    
    // 停止计数
    ioctl(fd_hit, PERF_EVENT_IOC_DISABLE, 0);
    
    // 读取结果
    struct pmu_read *data = alloca(sizeof(*data) + 4 * sizeof(u64));
    read(fd_hit, data, sizeof(*data) + 4 * sizeof(u64));
    
    printf("L1-D Cache Hit:   %lu\n", data->values[1]);
    printf("L1-D Cache Miss:  %lu (命中率 %.2f%%)\n",
           data->values[2],
           100.0 * data->values[1] / (data->values[1] + data->values[2]));
    printf("Instructions:    %lu\n", data->values[3]);
    printf("Cycles:          %lu\n", data->values[4]);
    printf("IPC:             %.2f\n",
           (double)data->values[3] / data->values[4]);
    
    close(fd_hit); close(fd_miss);
    close(fd_insn); close(fd_cycle);
    return 0;
}

使用 BPF_MAP_TYPE_PERF_EVENT_ARRAY 进行内核态聚合

当需要实时聚合多个 HART 的 PMU 数据时,eBPF 是最佳方案:


// pmu_collector.bpf.c
#include <vmlinux.h>
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>

// 性能采样数据输出缓冲区
struct {
    __uint(type, BPF_MAP_TYPE_PERF_EVENT_ARRAY);
    __uint(key_size, sizeof(u32));
    __uint(value_size, sizeof(u32));
    __uint(max_entries, 128);
} perf_output SEC(".maps");

// 每 CPU 累加缓存
struct {
    __type(key, u32);
    __type(value, u64[4]);  // [cycles, icache_hit, dcache_hit, dcache_miss]
    __uint(max_entries, 256);
    __uint(type, BPF_MAP_TYPE_PERCPU_ARRAY);
} pmu_accum SEC(".maps");

SEC("perf_event")
int pmu_collector_handler(struct bpf_perf_event_data *ctx)
{
    u32 cpu = bpf_get_smp_processor_id();
    u64 *accum = bpf_map_lookup_elem(&pmu_accum, &cpu);
    if (!accum)
        return 0;
    
    // bpf_perf_event_value 由 perf_event 子系统填充
    struct bpf_perf_event_value value = {};
    int ret = bpf_perf_event_read_value(&ctx->event, &value,
                                        sizeof(value));
    if (ret)
        return 0;
    
    accum[0] += value.counter;
    accum[1] += value.enabled;
    accum[2] += value.running;
    
    // 每 100ms 通过 perf_output 通知用户态
    if (accum[0] >= 100000) {
        bpf_perf_event_output(ctx, &perf_output, cpu,
                             &value, sizeof(value));
        __builtin_memset(accum, 0, sizeof(u64) * 4);
    }
    
    return 0;
}

char LICENSE[] SEC("license") = "GPL";

使用 libbpf 加载与用户态集成


// pmu_collector.c (用户态)
#include <bpf/libbpf.h>
#include <bpf/bpf.h>

static void handle_event(void *ctx, int cpu, void *data, __u32 size)
{
    struct bpf_perf_event_value *val = data;
    printf("[CPU%lu] counter=%llu enabled=%llu running=%llu\n",
           (unsigned long)cpu,
           (unsigned long long)val->counter,
           (unsigned long long)val->enabled,
           (unsigned long long)val->running);
}

int main(int argc, char **argv)
{
    struct bpf_object *obj;
    struct bpf_program *prog;
    struct bpf_link *link = NULL;
    struct perf_buffer *pb = NULL;
    int map_fd, prog_fd;
    
    obj = bpf_object__open_file("pmu_collector.bpf.o", NULL);
    bpf_object__load(obj);
    
    prog = bpf_object__find_program_by_name(obj, "pmu_collector_handler");
    prog_fd = bpf_program__fd(prog);
    
    // 获取 perf_output map 的文件描述符
    map_fd = bpf_object__find_map_fd_by_name(obj, "perf_output");
    
    // 通过系统调用绑定到每个 HART
    for (int cpu = 0; cpu < num_cpus; cpu++) {
        struct perf_event_attr attr = {
            .type    = PERF_TYPE_HARDWARE,
            .config  = PERF_COUNT_HW_CACHE_MISSES,
            .size    = sizeof(attr),
            .sample_period = 10000,
            .sample_type   = PERF_SAMPLE_RAW,
        };
        
        int pfd = perf_event_open(&attr, -1, cpu, -1, PERF_FLAG_FD_CLOEXEC);
        ioctl(pfd, PERF_EVENT_IOC_SET_BPF, prog_fd);
        ioctl(pfd, PERF_EVENT_IOC_ENABLE, 0);
    }
    
    // 用户态轮询 perf ring buffer
    pb = perf_buffer__new(map_fd, 256, handle_event, NULL, NULL, NULL);
    
    while (1) {
        perf_buffer__poll(pb, 100);
    }
    
    perf_buffer__free(pb);
    return 0;
}

实战场景:微架构性能分析

案例:LLVM 编译周期优化

以下是在 HiFive-Unmatched (FU740) 上分析 LLVM 编译过程中缓存行为的实际数据:


+----------------+----------------+---------------+----------------+
| 事件类型       | 基准编译       | 优化编译      | 改善幅度       |
+----------------+----------------+---------------+----------------+
| L1-I$ 命中率   |   97.21%       |   98.74%      |   + 1.57%      |
| L1-D$ 命中率   |   91.35%       |   94.82%      |   + 3.80%      |
| L2$ 命中率     |   87.02%       |   92.15%      |   + 5.89%      |
| DTLB 失效次数  |   2,847,391    |   1,963,442   |  - 31.0%       |
| ITLB 失效次数  |     482,113    |     298,667   |  - 38.0%       |
| 总指令数       | 1.42e10        | 1.18e10       |  - 16.9%       |
| 总周期数       | 9.87e09        | 7.23e09       |  - 26.7%       |
| IPC            |              1.438         1.632      |  + 13.5%    |
+----------------+----------------+---------------+----------------+

用 eBPF 实现远程 RISC-V 节点 PMU 聚合

当管理数十个 RISC-V 节点组成的边缘集群时,中心化 PMU 数据聚合是刚需。以下是使用 eBPF + gRPC 的轻量级部署方案:

架构设计:


┌─────────────────────────────────────────────────────────┐
│                    PMU Aggregator (x86_64)              │
│  ┌──────────┐    ┌──────────┐    ┌──────────────┐      │
│  │ gRPC srv │    │ 时序数据库│    │  Grafana面板  │      │
│  └────┬─────┘    └────▲─────┘    └──────▲───────┘      │
│       │               │                │               │
│       └───────────────┘                │               │
│                                        │               │
└────────────────────────────────────────┼───────────────┘
                                         │ gRPC Stream
     ┌──────────────┬──────────────┬─────┴────────┐
     │              │              │                │
┌────▼────┐   ┌─────▼────┐   ┌────▼────┐   ┌──────▼─────┐
│ RISC-V  │   │ RISC-V   │   │ RISC-V  │   │  RISC-V    │
│ Node #1 │   │ Node #2  │   │ Node #3 │   │  Node #N   │
│ ┌──────┐│   │ ┌──────┐ │   │ ┌──────┐│   │ ┌──────┐   │
│ │eBPF  ││   │ │eBPF  │ │   │ │eBPF  ││   │ │eBPF  │   │
│ │Agent ││   │ │Agent │ │   │ │Agent ││   │ │Agent │   │
│ └──┬───┘│   │ └──┬───┘ │   │ └──┬───┘│   │ └──┬───┘   │
│    │ gRPC│   │    │ gRPC│   │    │ gRPC│   │    │ gRPC  │
└────┼────┘   └────┼─────┘   └────┼────┘   └────┼───────┘
     │              │              │              │
     └──────────────┴──────────────┴──────────────┘

节点侧 eBPF Agent 伪代码(每次采样上传关键指标):


// 简化版 Go Agent
type Sample struct {
    Timestamp  int64  `proto:"1"`
    CPU        uint32 `proto:"2"`
    Cycles     uint64 `proto:"3"`
    InstrRet   uint64 `proto:"4"`
    L1DHit     uint64 `proto:"5"`
    L1DMiss    uint64 `proto:"6"`
}

func pollRiscvPMU(ebpfProg *ebpf.Program, stream pb.PMUAgent_CollectClient) {
    ticker := time.NewTicker(1 * time.Second)
    for range ticker.C {
        sample := Sample{
            Timestamp:  time.Now().UnixNano(),
            Cycles:     readCounter(ebpfProg, hwCycles),
            InstrRet:   readCounter(ebpfProg, hwInstrRet),
            L1DHit:     readCounter(ebpfProg, hwL1DCacheHit),
            L1DMiss:    readCounter(ebpfProg, hwL1DCacheMiss),
        }
        stream.Send(&sample)
    }
}

调试与故障排查

常见问题排查清单

问题现象 可能原因 解决方案
perf_event_open 返回 ENODEV SBI PMU 扩展未实现 检查 OpenSBI 版本 >= 1.2
计数器值为 0 计数器索引超出范围 使用sbi_pmu_num_counters()返回值
中断不触发 中断号未连接 检查 DT 节点 riscv,pmc 兼容性
读取结果跳越 竞态导致多轮读取 使用ioctl(PERF_EVENT_IOC_DISABLE)先停
热插拔后计数器丢失 未注册 cpuhp 回调 确认 CONFIG_PERF_EVENTS=y

OpenSBI 调试方法

在 QEMU 仿真环境中可通过以下命令验证 SBI PMU 扩展:


# 查看 SBI 扩展列表
$ cat /sys/firmware/devicetree/base/sbi/extensions
"Timer", "IPI", "RFENCE", "HSM", "SRST", "PMU"

# 验证 PMU 计数器数量
$ cat /sys/devices/system/cpu/cpu0/riscv_pmu/counter_total
6

# 检查当前活跃事件列表
$ cat /sys/devices/system/cpu/cpu0/riscv_pmu/active_events
hw_cycles=0x00 hw_instructions=0x01

2025-2026 即将落地的扩展

RISC-Zin 工作组在规划以下 PMU 演化方向,值得工程人员关注:

  1. SSS (Sscofpmf) — 协处理器溢出中断:解决当前中断路由死板的问题,2026 年将进入冻结状态,QEMU 已部分支持。关注 mhpmeventOF 位的行为变化。
    1. AIA 集成中断:RISC-V 高级中断架构(AIA)正在定义基于 IMSIC 的 PMU 中断分离方案,未来 HART 可直接通知 Supervisor 中断,提升溢出处理效率。
      1. QEMU 支持进展:qemu 10.x 开始支持 -cpu rv64,pmu-num=6,可以在无硬件时进行驱动开发。
      2. 总结与工程建议

        RISC-V PMU 已在 2025-2026 年完成从规范到工程产品的蜕变。对工程人员的建议如下:

        对于内核开发者:OpenSBI 1.2+ 已稳定支持,QEMU 10.x 提供完整仿真环境,可直接在 x86_64 主机开发 RISC-V PMU 驱动代码。

        对于运维工程师:基于 eBPF 的 PMU 监控远优于传统方案的零侵扰特性——进程无需感知自身被观测,内核直接通过 perf_event ring buffer 输出采样数据。

        对于研究者:Sscofpmf 与 AIA 集成的组合设计代表了微架构性能监控的新范式——精确事件驱动的全系统可见性,这将在未来 1-2 年成为 RISC-V 生态的标配能力。

        通过本文介绍的抽象从 SBI firmware 接口到 eBPF 自动采集,你应该能够在 Linux 平台上构建完整的 RISC-V 性能分析方法论。基准测试表明,在 HiFive Unmatched 平台上以 1000Hz 采样率运行的 eBPF PMU 收集模块,CPU 开销低于 0.3%。

        开放生态、透明硬件、可观测未来——RISC-V PMU 正在将性能分析从黑盒走向白盒。

点赞(0) 打赏

评论列表 共有 0 条评论

暂无评论
立即
投稿

微信公众账号

微信扫一扫加关注

发表
评论
返回
顶部