You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Intel平台Mac/PC下C++如何读取CPU指令数、缓存缺失等性能计数器

获取x86_64硬件性能指标的实现方案

你需要的已执行指令数、缓存缺失数、内存访问数这类指标,都可以通过CPU的**性能监控单元(PMU)**获取,下面分你的两个开发环境给出具体实现方式:

Intel Mac(Xcode)环境

macOS支持通过系统API或直接读取PMU寄存器获取这些指标,注意读取PMU需要开启权限:

  1. 先启用性能事件支持:
    打开终端执行sudo sysctl -w kern.perf_events_enabled=1,或者在代码中通过sysctl系统调用设置。
  2. 代码层面可用perf_event_open系统调用(兼容Linux/macOS)配置和读取PMU事件,示例代码片段(读取已执行指令数和L1缓存缺失数):
    #include <stdio.h>
    #include <stdlib.h>
    #include <unistd.h>
    #include <sys/syscall.h>
    #include <linux/perf_event.h>
    #include <asm/unistd.h>
    #include <string.h>
    
    static long perf_event_open(struct perf_event_attr *hw_event, pid_t pid,
                                int cpu, int group_fd, unsigned long flags) {
        return syscall(__NR_perf_event_open, hw_event, pid, cpu, group_fd, flags);
    }
    
    int main() {
        struct perf_event_attr attr_inst, attr_cache;
        int fd_inst, fd_cache;
        unsigned long long count_inst, count_cache;
    
        // 配置已执行指令数事件
        memset(&attr_inst, 0, sizeof(struct perf_event_attr));
        attr_inst.type = PERF_TYPE_HARDWARE;
        attr_inst.size = sizeof(struct perf_event_attr);
        attr_inst.config = PERF_COUNT_HW_INSTRUCTIONS;
        attr_inst.disabled = 1;
        attr_inst.exclude_kernel = 1; // 排除内核指令
        attr_inst.exclude_hv = 1; // 排除虚拟机管理程序指令
    
        fd_inst = perf_event_open(&attr_inst, 0, -1, -1, 0);
        if (fd_inst == -1) {
            perror("perf_event_open failed");
            exit(EXIT_FAILURE);
        }
    
        // 配置L1缓存缺失事件
        memset(&attr_cache, 0, sizeof(struct perf_event_attr));
        attr_cache.type = PERF_TYPE_HW_CACHE;
        attr_cache.size = sizeof(struct perf_event_attr);
        attr_cache.config = PERF_COUNT_HW_CACHE_L1D | (PERF_COUNT_HW_CACHE_OP_READ << 8) | (PERF_COUNT_HW_CACHE_RESULT_MISS << 16);
        attr_cache.disabled = 1;
        attr_cache.exclude_kernel = 1;
        attr_cache.exclude_hv = 1;
    
        fd_cache = perf_event_open(&attr_cache, 0, -1, -1, 0);
        if (fd_cache == -1) {
            perror("perf_event_open cache failed");
            exit(EXIT_FAILURE);
        }
    
        // 启动计数器
        ioctl(fd_inst, PERF_EVENT_IOC_RESET, 0);
        ioctl(fd_inst, PERF_EVENT_IOC_ENABLE, 0);
        ioctl(fd_cache, PERF_EVENT_IOC_RESET, 0);
        ioctl(fd_cache, PERF_EVENT_IOC_ENABLE, 0);
    
        // --- 这里放你要测量的代码 ---
        for (int i = 0; i < 1000000; i++);
        // --- 测量结束 ---
    
        // 停止计数器并读取结果
        ioctl(fd_inst, PERF_EVENT_IOC_DISABLE, 0);
        ioctl(fd_cache, PERF_EVENT_IOC_DISABLE, 0);
        read(fd_inst, &count_inst, sizeof(unsigned long long));
        read(fd_cache, &count_cache, sizeof(unsigned long long));
    
        printf("已执行指令数: %llu\n", count_inst);
        printf("L1缓存缺失数: %llu\n", count_cache);
    
        close(fd_inst);
        close(fd_cache);
        return 0;
    }
    
    编译用clang:clang -o perf_test perf_test.c,运行需sudo权限:sudo ./perf_test。
  3. 也可以用Xcode自带的Instruments工具(Time Profiler、Cache Misses模板)可视化这些指标,无需编写代码即可快速定位瓶颈。

Windows(Visual Studio)环境

Windows下有两种主流方式获取这些指标:

  1. Performance Data Helper (PDH) API:系统提供的高层API,无需特权级,适合获取进程级性能数据,示例代码片段(读取当前进程的已执行指令数和L2缓存缺失数):
    #include <windows.h>
    #include <pdh.h>
    #include <stdio.h>
    
    #pragma comment(lib, "pdh.lib")
    
    int main() {
        PDH_HQUERY hQuery;
        PDH_HCOUNTER hCounterInst, hCounterCache;
        PDH_FMT_COUNTERVALUE counterValue;
    
        // 创建查询
        if (PdhOpenQuery(NULL, 0, &hQuery) != ERROR_SUCCESS) {
            printf("PdhOpenQuery failed\n");
            return 1;
        }
    
        // 添加已执行指令数计数器(当前进程)
        if (PdhAddCounter(hQuery, L"\\Process(%PID%)\\Instructions Retired", 0, &hCounterInst) != ERROR_SUCCESS) {
            printf("PdhAddCounter instructions failed\n");
            PdhCloseQuery(hQuery);
            return 1;
        }
        // 添加L2缓存缺失计数器(当前进程)
        if (PdhAddCounter(hQuery, L"\\Process(%PID%)\\L2 Cache Misses", 0, &hCounterCache) != ERROR_SUCCESS) {
            printf("PdhAddCounter cache failed\n");
            PdhRemoveCounter(hCounterInst);
            PdhCloseQuery(hQuery);
            return 1;
        }
    
        // 初始化计数器
        PdhCollectQueryData(hQuery);
    
        // --- 这里放你要测量的代码 ---
        for (int i = 0; i < 1000000; i++);
        // --- 测量结束 ---
    
        // 获取计数器值
        PdhCollectQueryData(hQuery);
        if (PdhGetFormattedCounterValue(hCounterInst, PDH_FMT_LARGE, NULL, &counterValue) == ERROR_SUCCESS) {
            printf("已执行指令数: %llu\n", counterValue.largeValue);
        }
        if (PdhGetFormattedCounterValue(hCounterCache, PDH_FMT_LARGE, NULL, &counterValue) == ERROR_SUCCESS) {
            printf("L2缓存缺失数: %llu\n", counterValue.largeValue);
        }
    
        // 清理资源
        PdhRemoveCounter(hCounterInst);
        PdhRemoveCounter(hCounterCache);
        PdhCloseQuery(hQuery);
        return 0;
    }
    
    %PID%会自动替换为当前进程ID,直接用Visual Studio编译即可,无需额外配置。
  2. 直接读取PMU寄存器(__rdpmc指令):可获取周期级细粒度数据,但需要特权级,且需根据Intel CPU手册配置PMU事件选择寄存器,适合底层性能分析,建议优先使用PDH API。
  3. Visual Studio自带的**性能探查器(Performance Profiler)**可直接分析代码的指令数、缓存缺失、内存访问等指标,支持可视化和火焰图展示,适合快速排查问题。

注意事项

  • 与rdtsc结合使用时,要确保计数器读取和rdtsc调用同步,可将代码绑定到单个CPU核心减少上下文切换影响。
  • 单次测量误差较大,建议多次采样取平均值。
  • 不同Intel CPU型号的PMU事件编码可能略有差异,使用系统抽象API可避免兼容性问题。

内容的提问来源于stack exchange,提问作者user19179144

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 15:16:05