You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

求C/C++实现程序执行时各缓存层级MLP测量的代码方案

C/C++ 实现缓存层级MLP测量(基于性能计数器)

内存级并行度(MLP)反映缓存未命中时,MSHR/填充缓冲区同时持有的内存请求数量。以下是基于Linux perf_event_open系统调用的实现代码,通过硬件性能计数器统计L1、L2缓存的pending请求数,进而计算MLP。

完整实现代码

#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <sys/ioctl.h>
#include <linux/perf_event.h>
#include <asm/unistd.h>
#include <string.h>
#include <sys/mman.h>

// 封装perf_event_open系统调用(无标准库实现)
static long perf_event_open(struct perf_event_attr *hw_event, pid_t pid,
                            int cpu, int group_fd, unsigned long flags)
{
    return syscall(__NR_perf_event_open, hw_event, pid, cpu, group_fd, flags);
}

// 打开指定性能计数器
int open_perf_counter(uint64_t config, pid_t pid, int cpu)
{
    struct perf_event_attr pe;
    memset(&pe, 0, sizeof(struct perf_event_attr));
    pe.type = (config == PERF_COUNT_HW_CPU_CYCLES) ? PERF_TYPE_HARDWARE : PERF_TYPE_RAW;
    pe.size = sizeof(struct perf_event_attr);
    pe.config = config;
    pe.disabled = 1;
    pe.exclude_kernel = 1; // 排除内核空间事件
    pe.exclude_hv = 1;     // 排除虚拟化层事件

    int fd = perf_event_open(&pe, pid, cpu, -1, 0);
    if (fd == -1) {
        perror("perf_event_open failed");
        exit(EXIT_FAILURE);
    }
    return fd;
}

// 读取计数器值
uint64_t read_perf_counter(int fd)
{
    uint64_t count;
    if (read(fd, &count, sizeof(count)) == -1) {
        perror("read perf counter failed");
        exit(EXIT_FAILURE);
    }
    return count;
}

// 测试负载:制造大量缓存未命中以触发MSHR占用
void test_workload(size_t size)
{
    char *buf = malloc(size);
    if (!buf) {
        perror("malloc failed");
        exit(EXIT_FAILURE);
    }

    // 初始化内存并强制刷出缓存
    for (size_t i = 0; i < size; i++) {
        buf[i] = i % 256;
    }
    __builtin___clear_cache(buf, buf + size);

    // 按缓存行随机访问,制造连续缓存未命中
    for (size_t i = 0; i < size * 10; i++) {
        size_t idx = (rand() % (size / 64)) * 64;
        buf[idx] += 1;
    }

    free(buf);
}

int main()
{
    // Intel x86_64架构下的事件编码(需根据CPU型号调整)
    const uint64_t L1_MSHR_EVENT = 0x412E;  // L1D_PENDING_MISSES.PENDING
    const uint64_t L2_MSHR_EVENT = 0x414F;  // L2_PENDING_MISSES.PENDING
    const uint64_t CPU_CYCLES_EVENT = PERF_COUNT_HW_CPU_CYCLES;

    // 打开三个计数器:L1 MSHR、L2 MSHR、CPU总周期
    int fd_l1 = open_perf_counter(L1_MSHR_EVENT, 0, -1);
    int fd_l2 = open_perf_counter(L2_MSHR_EVENT, 0, -1);
    int fd_cycles = open_perf_counter(CPU_CYCLES_EVENT, 0, -1);

    // 重置并启动计数
    ioctl(fd_l1, PERF_EVENT_IOC_RESET, 0);
    ioctl(fd_l2, PERF_EVENT_IOC_RESET, 0);
    ioctl(fd_cycles, PERF_EVENT_IOC_RESET, 0);
    ioctl(fd_l1, PERF_EVENT_IOC_ENABLE, 0);
    ioctl(fd_l2, PERF_EVENT_IOC_ENABLE, 0);
    ioctl(fd_cycles, PERF_EVENT_IOC_ENABLE, 0);

    // 运行测试负载(64MB内存,远超L2缓存容量)
    test_workload(1024 * 1024 * 64);

    // 停止计数
    ioctl(fd_l1, PERF_EVENT_IOC_DISABLE, 0);
    ioctl(fd_l2, PERF_EVENT_IOC_DISABLE, 0);
    ioctl(fd_cycles, PERF_EVENT_IOC_DISABLE, 0);

    // 读取计数结果
    uint64_t l1_mshr_count = read_perf_counter(fd_l1);
    uint64_t l2_mshr_count = read_perf_counter(fd_l2);
    uint64_t cpu_cycles = read_perf_counter(fd_cycles);

    // 输出原始计数
    printf("L1 MSHR 总pending请求数: %lu\n", l1_mshr_count);
    printf("L2 MSHR 总pending请求数: %lu\n", l2_mshr_count);
    printf("总CPU周期数: %lu\n", cpu_cycles);

    // 计算平均MLP(假设CPU时钟频率为3GHz,替换为实际频率更准确)
    double freq_ghz = 3.0;
    double l1_mlp = (l1_mshr_count / (double)cpu_cycles) * freq_ghz * 1e9;
    double l2_mlp = (l2_mshr_count / (double)cpu_cycles) * freq_ghz * 1e9;

    printf("平均L1内存级并行度(MLP): %.2f\n", l1_mlp);
    printf("平均L2内存级并行度(MLP): %.2f\n", l2_mlp);

    // 清理资源
    close(fd_l1);
    close(fd_l2);
    close(fd_cycles);

    return 0;
}

关键说明

  1. 架构兼容性:
    • 代码默认适配Intel x86_64 CPU,事件编码需参考对应CPU的开发者手册(如Intel SDM)调整。
    • AMD CPU需替换为对应事件编码,例如L1 MSHR事件可查AMD Performance Monitor手册。
  2. 运行权限:
    • Linux 5.8+需CAP_PERFMON权限,或执行sudo sysctl -w kernel.perf_event_paranoid=0降低权限限制,也可直接用sudo运行程序。
  3. 编译命令:
    gcc -O2 -o mlp_measure mlp_measure.c -lrt
    
  4. MLP计算逻辑:
    示例中通过「总pending请求数/总CPU周期数×时钟频率」计算平均并发请求数。更准确的方式可结合内存未命中总数(如PERF_COUNT_HW_CACHE_L1D:READ:MISS事件),公式为MLP = 总pending请求数 / 总缓存未命中次数。

内容的提问来源于stack exchange,提问作者user25076706

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.24 01:27:11