You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++并行程序线程内存读写实时统计及确定性并行多线程实现咨询

一、实时统计每个线程的内存读写次数

1. 手动封装内存访问接口

通过给所有内存读写操作套一层封装函数,用线程本地变量统计计数,实现简单可控。

#include <iostream>
#include <thread>
#include <vector>
#include <cstdint>

// 线程本地的读写计数器
thread_local uint64_t read_count = 0;
thread_local uint64_t write_count = 0;

// 封装读操作
template<typename T>
T read(const T* ptr) {
    read_count++;
    return *ptr;
}

// 封装写操作
template<typename T>
void write(T* ptr, const T& val) {
    write_count++;
    *ptr = val;
}

// 打印当前线程计数
void print_counts() {
    std::cout << "Thread " << std::this_thread::get_id() 
              << ": Reads = " << read_count 
              << ", Writes = " << write_count << std::endl;
}

void worker() {
    int data = 0;
    for (int i = 0; i < 1000; ++i) {
        write(&data, i);
        int val = read(&data);
    }
    print_counts();
}

int main() {
    std::vector<std::thread> threads;
    for (int i = 0; i < 4; ++i) {
        threads.emplace_back(worker);
    }
    for (auto& t : threads) {
        t.join();
    }
    return 0;
}

优缺点:无需依赖外部工具,但需要修改所有内存访问代码,无法覆盖第三方库的操作。

2. 硬件性能计数器(Linux平台)

利用CPU硬件性能计数器统计内存读写,无需修改业务代码,依赖操作系统和硬件支持。

#include <iostream>
#include <thread>
#include <vector>
#include <cstdint>
#include <sys/ioctl.h>
#include <linux/perf_event.h>
#include <unistd.h>
#include <fcntl.h>
#include <cstring>
#include <cstdlib>

struct PerfCounter {
    int fd;
    uint64_t count;

    PerfCounter(uint32_t event_id) {
        perf_event_attr attr;
        memset(&attr, 0, sizeof(attr));
        attr.type = PERF_TYPE_HARDWARE;
        attr.size = sizeof(attr);
        attr.config = event_id;
        attr.disabled = 1;
        attr.exclude_kernel = 1;
        attr.exclude_hv = 1;

        fd = perf_event_open(&attr, 0, -1, -1, 0);
        if (fd == -1) {
            perror("perf_event_open failed");
            exit(1);
        }
    }

    ~PerfCounter() {
        close(fd);
    }

    void start() {
        ioctl(fd, PERF_EVENT_IOC_RESET, 0);
        ioctl(fd, PERF_EVENT_IOC_ENABLE, 0);
    }

    void stop() {
        ioctl(fd, PERF_EVENT_IOC_DISABLE, 0);
        read(fd, &count, sizeof(count));
    }
};

void worker() {
    PerfCounter load_counter(PERF_COUNT_HW_MEM_LOADS);
    PerfCounter store_counter(PERF_COUNT_HW_MEM_STORES);

    load_counter.start();
    store_counter.start();

    int data = 0;
    for (int i = 0; i < 100000; ++i) {
        data = i;
        int val = data;
    }

    load_counter.stop();
    store_counter.stop();

    std::cout << "Thread " << std::this_thread::get_id() 
              << ": Loads = " << load_counter.count 
              << ", Stores = " << store_counter.count << std::endl;
}

int main() {
    std::vector<std::thread> threads;
    for (int i = 0; i < 4; ++i) {
        threads.emplace_back(worker);
    }
    for (auto& t : threads) {
        t.join();
    }
    return 0;
}

编译运行:需链接-lrt,运行前需设置echo 0 | sudo tee /proc/sys/kernel/perf_event_paranoid或使用root权限。

3. 编译器插桩(Clang/GCC)

通过编译器插桩自动注入计数代码,无需手动修改业务代码,需掌握LLVM/GCC插桩技术。例如使用Clang编写LLVM Pass,在内存读写指令处插入计数逻辑;或用GCC的-finstrument-functions配合自定义钩子函数。

二、以内存访问次数为确定性时钟的确定性并行多线程实现

核心思路

  1. 所有内存访问通过封装函数实现,每次访问递增线程本地的"确定性时钟"计数器。
  2. 线程同步时,等待所有线程的时钟值达到指定阈值,确保执行顺序完全可控,消除调度不确定性。

代码示例

#include <iostream>
#include <thread>
#include <vector>
#include <cstdint>
#include <mutex>
#include <condition_variable>

struct SyncContext {
    std::mutex mtx;
    std::condition_variable cv;
    std::vector<uint64_t> thread_clocks;
    uint64_t global_threshold = 0;
};

thread_local uint64_t mem_access_clock = 0;
thread_local int thread_id = -1;
thread_local SyncContext* sync_ctx = nullptr;

// 封装内存访问,递增时钟
template<typename T>
T read(const T* ptr) {
    mem_access_clock++;
    return *ptr;
}

template<typename T>
void write(T* ptr, const T& val) {
    mem_access_clock++;
    *ptr = val;
}

// 同步所有线程到当前最大时钟值
void sync() {
    std::unique_lock<std::mutex> lock(sync_ctx->mtx);
    sync_ctx->thread_clocks[thread_id] = mem_access_clock;

    // 计算当前所有线程的最大时钟值
    uint64_t max_clock = 0;
    for (uint64_t clk : sync_ctx->thread_clocks) {
        max_clock = std::max(max_clock, clk);
    }

    // 等待当前线程时钟追上全局最大值
    while (mem_access_clock < max_clock) {
        sync_ctx->cv.wait(lock);
    }

    sync_ctx->global_threshold = max_clock;
    sync_ctx->cv.notify_all();
}

void worker(int tid, SyncContext& ctx) {
    thread_id = tid;
    sync_ctx = &ctx;

    int local_data = tid;
    // 第一阶段:本地内存操作
    for (int i = 0; i < 100; ++i) {
        write(&local_data, local_data + 1);
        int val = read(&local_data);
    }
    std::cout << "Thread " << tid << " reached clock " << mem_access_clock << ", syncing..." << std::endl;
    sync();
    std::cout << "Thread " << tid << " synced, global threshold: " << ctx.global_threshold << std::endl;

    // 第二阶段:跨线程共享操作
    static std::vector<int> shared_data(4, 0);
    write(&shared_data[tid], mem_access_clock);
    int shared_val = read(&shared_data[(tid+1)%4]);
    std::cout << "Thread " << tid << " read shared value from thread " << (tid+1)%4 << ": " << shared_val << std::endl;
}

int main() {
    const int num_threads = 4;
    SyncContext ctx;
    ctx.thread_clocks.resize(num_threads, 0);

    std::vector<std::thread> threads;
    for (int i = 0; i < num_threads; ++i) {
        threads.emplace_back(worker, i, std::ref(ctx));
    }
    for (auto& t : threads) {
        t.join();
    }

    return 0;
}
参考接口与手册
  1. 线程本地存储:C11及以上的thread_local关键字,参考C标准文档中"线程存储持续期"章节。
  2. Linux性能计数器:perf_event_open系统调用,查看man 2 perf_event_open手册。
  3. 编译器插桩:Clang/LLVM Pass开发文档、GCC的-finstrument-functions选项文档(man gcc)。

内容的提问来源于stack exchange,提问作者Li Jianqing

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.13 13:07:46