求C/C++实现程序执行时各缓存层级MLP测量的代码方案
C/C++ 实现缓存层级MLP测量(基于性能计数器)
内存级并行度(MLP)反映缓存未命中时,MSHR/填充缓冲区同时持有的内存请求数量。以下是基于Linux perf_event_open系统调用的实现代码,通过硬件性能计数器统计L1、L2缓存的pending请求数,进而计算MLP。
完整实现代码
#include <stdio.h> #include <stdlib.h> #include <unistd.h> #include <sys/ioctl.h> #include <linux/perf_event.h> #include <asm/unistd.h> #include <string.h> #include <sys/mman.h> // 封装perf_event_open系统调用(无标准库实现) static long perf_event_open(struct perf_event_attr *hw_event, pid_t pid, int cpu, int group_fd, unsigned long flags) { return syscall(__NR_perf_event_open, hw_event, pid, cpu, group_fd, flags); } // 打开指定性能计数器 int open_perf_counter(uint64_t config, pid_t pid, int cpu) { struct perf_event_attr pe; memset(&pe, 0, sizeof(struct perf_event_attr)); pe.type = (config == PERF_COUNT_HW_CPU_CYCLES) ? PERF_TYPE_HARDWARE : PERF_TYPE_RAW; pe.size = sizeof(struct perf_event_attr); pe.config = config; pe.disabled = 1; pe.exclude_kernel = 1; // 排除内核空间事件 pe.exclude_hv = 1; // 排除虚拟化层事件 int fd = perf_event_open(&pe, pid, cpu, -1, 0); if (fd == -1) { perror("perf_event_open failed"); exit(EXIT_FAILURE); } return fd; } // 读取计数器值 uint64_t read_perf_counter(int fd) { uint64_t count; if (read(fd, &count, sizeof(count)) == -1) { perror("read perf counter failed"); exit(EXIT_FAILURE); } return count; } // 测试负载:制造大量缓存未命中以触发MSHR占用 void test_workload(size_t size) { char *buf = malloc(size); if (!buf) { perror("malloc failed"); exit(EXIT_FAILURE); } // 初始化内存并强制刷出缓存 for (size_t i = 0; i < size; i++) { buf[i] = i % 256; } __builtin___clear_cache(buf, buf + size); // 按缓存行随机访问,制造连续缓存未命中 for (size_t i = 0; i < size * 10; i++) { size_t idx = (rand() % (size / 64)) * 64; buf[idx] += 1; } free(buf); } int main() { // Intel x86_64架构下的事件编码(需根据CPU型号调整) const uint64_t L1_MSHR_EVENT = 0x412E; // L1D_PENDING_MISSES.PENDING const uint64_t L2_MSHR_EVENT = 0x414F; // L2_PENDING_MISSES.PENDING const uint64_t CPU_CYCLES_EVENT = PERF_COUNT_HW_CPU_CYCLES; // 打开三个计数器:L1 MSHR、L2 MSHR、CPU总周期 int fd_l1 = open_perf_counter(L1_MSHR_EVENT, 0, -1); int fd_l2 = open_perf_counter(L2_MSHR_EVENT, 0, -1); int fd_cycles = open_perf_counter(CPU_CYCLES_EVENT, 0, -1); // 重置并启动计数 ioctl(fd_l1, PERF_EVENT_IOC_RESET, 0); ioctl(fd_l2, PERF_EVENT_IOC_RESET, 0); ioctl(fd_cycles, PERF_EVENT_IOC_RESET, 0); ioctl(fd_l1, PERF_EVENT_IOC_ENABLE, 0); ioctl(fd_l2, PERF_EVENT_IOC_ENABLE, 0); ioctl(fd_cycles, PERF_EVENT_IOC_ENABLE, 0); // 运行测试负载(64MB内存,远超L2缓存容量) test_workload(1024 * 1024 * 64); // 停止计数 ioctl(fd_l1, PERF_EVENT_IOC_DISABLE, 0); ioctl(fd_l2, PERF_EVENT_IOC_DISABLE, 0); ioctl(fd_cycles, PERF_EVENT_IOC_DISABLE, 0); // 读取计数结果 uint64_t l1_mshr_count = read_perf_counter(fd_l1); uint64_t l2_mshr_count = read_perf_counter(fd_l2); uint64_t cpu_cycles = read_perf_counter(fd_cycles); // 输出原始计数 printf("L1 MSHR 总pending请求数: %lu\n", l1_mshr_count); printf("L2 MSHR 总pending请求数: %lu\n", l2_mshr_count); printf("总CPU周期数: %lu\n", cpu_cycles); // 计算平均MLP(假设CPU时钟频率为3GHz,替换为实际频率更准确) double freq_ghz = 3.0; double l1_mlp = (l1_mshr_count / (double)cpu_cycles) * freq_ghz * 1e9; double l2_mlp = (l2_mshr_count / (double)cpu_cycles) * freq_ghz * 1e9; printf("平均L1内存级并行度(MLP): %.2f\n", l1_mlp); printf("平均L2内存级并行度(MLP): %.2f\n", l2_mlp); // 清理资源 close(fd_l1); close(fd_l2); close(fd_cycles); return 0; }
关键说明
- 架构兼容性:
- 代码默认适配Intel x86_64 CPU,事件编码需参考对应CPU的开发者手册(如Intel SDM)调整。
- AMD CPU需替换为对应事件编码,例如L1 MSHR事件可查AMD Performance Monitor手册。
- 运行权限:
- Linux 5.8+需
CAP_PERFMON权限,或执行sudo sysctl -w kernel.perf_event_paranoid=0降低权限限制,也可直接用sudo运行程序。
- Linux 5.8+需
- 编译命令:
gcc -O2 -o mlp_measure mlp_measure.c -lrt - MLP计算逻辑:
示例中通过「总pending请求数/总CPU周期数×时钟频率」计算平均并发请求数。更准确的方式可结合内存未命中总数(如PERF_COUNT_HW_CACHE_L1D:READ:MISS事件),公式为MLP = 总pending请求数 / 总缓存未命中次数。
内容的提问来源于stack exchange,提问作者user25076706
相关产品推荐
相关产品推荐

