Intel平台Mac/PC下C++如何读取CPU指令数、缓存缺失等性能计数器
获取x86_64硬件性能指标的实现方案
你需要的已执行指令数、缓存缺失数、内存访问数这类指标,都可以通过CPU的**性能监控单元(PMU)**获取,下面分你的两个开发环境给出具体实现方式:
Intel Mac(Xcode)环境
macOS支持通过系统API或直接读取PMU寄存器获取这些指标,注意读取PMU需要开启权限:
- 先启用性能事件支持:
打开终端执行sudo sysctl -w kern.perf_events_enabled=1,或者在代码中通过sysctl系统调用设置。 - 代码层面可用
perf_event_open系统调用(兼容Linux/macOS)配置和读取PMU事件,示例代码片段(读取已执行指令数和L1缓存缺失数):
编译用clang:#include <stdio.h> #include <stdlib.h> #include <unistd.h> #include <sys/syscall.h> #include <linux/perf_event.h> #include <asm/unistd.h> #include <string.h> static long perf_event_open(struct perf_event_attr *hw_event, pid_t pid, int cpu, int group_fd, unsigned long flags) { return syscall(__NR_perf_event_open, hw_event, pid, cpu, group_fd, flags); } int main() { struct perf_event_attr attr_inst, attr_cache; int fd_inst, fd_cache; unsigned long long count_inst, count_cache; // 配置已执行指令数事件 memset(&attr_inst, 0, sizeof(struct perf_event_attr)); attr_inst.type = PERF_TYPE_HARDWARE; attr_inst.size = sizeof(struct perf_event_attr); attr_inst.config = PERF_COUNT_HW_INSTRUCTIONS; attr_inst.disabled = 1; attr_inst.exclude_kernel = 1; // 排除内核指令 attr_inst.exclude_hv = 1; // 排除虚拟机管理程序指令 fd_inst = perf_event_open(&attr_inst, 0, -1, -1, 0); if (fd_inst == -1) { perror("perf_event_open failed"); exit(EXIT_FAILURE); } // 配置L1缓存缺失事件 memset(&attr_cache, 0, sizeof(struct perf_event_attr)); attr_cache.type = PERF_TYPE_HW_CACHE; attr_cache.size = sizeof(struct perf_event_attr); attr_cache.config = PERF_COUNT_HW_CACHE_L1D | (PERF_COUNT_HW_CACHE_OP_READ << 8) | (PERF_COUNT_HW_CACHE_RESULT_MISS << 16); attr_cache.disabled = 1; attr_cache.exclude_kernel = 1; attr_cache.exclude_hv = 1; fd_cache = perf_event_open(&attr_cache, 0, -1, -1, 0); if (fd_cache == -1) { perror("perf_event_open cache failed"); exit(EXIT_FAILURE); } // 启动计数器 ioctl(fd_inst, PERF_EVENT_IOC_RESET, 0); ioctl(fd_inst, PERF_EVENT_IOC_ENABLE, 0); ioctl(fd_cache, PERF_EVENT_IOC_RESET, 0); ioctl(fd_cache, PERF_EVENT_IOC_ENABLE, 0); // --- 这里放你要测量的代码 --- for (int i = 0; i < 1000000; i++); // --- 测量结束 --- // 停止计数器并读取结果 ioctl(fd_inst, PERF_EVENT_IOC_DISABLE, 0); ioctl(fd_cache, PERF_EVENT_IOC_DISABLE, 0); read(fd_inst, &count_inst, sizeof(unsigned long long)); read(fd_cache, &count_cache, sizeof(unsigned long long)); printf("已执行指令数: %llu\n", count_inst); printf("L1缓存缺失数: %llu\n", count_cache); close(fd_inst); close(fd_cache); return 0; }clang -o perf_test perf_test.c,运行需sudo权限:sudo ./perf_test。 - 也可以用Xcode自带的Instruments工具(Time Profiler、Cache Misses模板)可视化这些指标,无需编写代码即可快速定位瓶颈。
Windows(Visual Studio)环境
Windows下有两种主流方式获取这些指标:
- Performance Data Helper (PDH) API:系统提供的高层API,无需特权级,适合获取进程级性能数据,示例代码片段(读取当前进程的已执行指令数和L2缓存缺失数):
#include <windows.h> #include <pdh.h> #include <stdio.h> #pragma comment(lib, "pdh.lib") int main() { PDH_HQUERY hQuery; PDH_HCOUNTER hCounterInst, hCounterCache; PDH_FMT_COUNTERVALUE counterValue; // 创建查询 if (PdhOpenQuery(NULL, 0, &hQuery) != ERROR_SUCCESS) { printf("PdhOpenQuery failed\n"); return 1; } // 添加已执行指令数计数器(当前进程) if (PdhAddCounter(hQuery, L"\\Process(%PID%)\\Instructions Retired", 0, &hCounterInst) != ERROR_SUCCESS) { printf("PdhAddCounter instructions failed\n"); PdhCloseQuery(hQuery); return 1; } // 添加L2缓存缺失计数器(当前进程) if (PdhAddCounter(hQuery, L"\\Process(%PID%)\\L2 Cache Misses", 0, &hCounterCache) != ERROR_SUCCESS) { printf("PdhAddCounter cache failed\n"); PdhRemoveCounter(hCounterInst); PdhCloseQuery(hQuery); return 1; } // 初始化计数器 PdhCollectQueryData(hQuery); // --- 这里放你要测量的代码 --- for (int i = 0; i < 1000000; i++); // --- 测量结束 --- // 获取计数器值 PdhCollectQueryData(hQuery); if (PdhGetFormattedCounterValue(hCounterInst, PDH_FMT_LARGE, NULL, &counterValue) == ERROR_SUCCESS) { printf("已执行指令数: %llu\n", counterValue.largeValue); } if (PdhGetFormattedCounterValue(hCounterCache, PDH_FMT_LARGE, NULL, &counterValue) == ERROR_SUCCESS) { printf("L2缓存缺失数: %llu\n", counterValue.largeValue); } // 清理资源 PdhRemoveCounter(hCounterInst); PdhRemoveCounter(hCounterCache); PdhCloseQuery(hQuery); return 0; }%PID%会自动替换为当前进程ID,直接用Visual Studio编译即可,无需额外配置。 - 直接读取PMU寄存器(__rdpmc指令):可获取周期级细粒度数据,但需要特权级,且需根据Intel CPU手册配置PMU事件选择寄存器,适合底层性能分析,建议优先使用PDH API。
- Visual Studio自带的**性能探查器(Performance Profiler)**可直接分析代码的指令数、缓存缺失、内存访问等指标,支持可视化和火焰图展示,适合快速排查问题。
注意事项
- 与
rdtsc结合使用时,要确保计数器读取和rdtsc调用同步,可将代码绑定到单个CPU核心减少上下文切换影响。 - 单次测量误差较大,建议多次采样取平均值。
- 不同Intel CPU型号的PMU事件编码可能略有差异,使用系统抽象API可避免兼容性问题。
内容的提问来源于stack exchange,提问作者user19179144
相关产品推荐
相关产品推荐

