You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何降低双线程共享小型结构体时写入侧的性能损耗?

优化解释器性能分析器的写入侧性能损耗

问题背景

我正在为一款基于C++(C++17 x86 MSVC)实现的现有解释器添加性能分析器,需要存储3个int32_t、1个uint16_t和1个uint8_t类型的数据来表示执行帧的上下文。由于是解释型代码,无法通过内核级驱动生成快照,必须将分析逻辑加入到eval循环中,同时要尽可能避免拖慢实际计算流程。

运行逻辑如下:

  • 解释器执行每条指令前,调用小型函数上报当前帧位置
  • 采样分析器仅需每隔X毫秒获取一次该帧位置
  • 涉及两个线程共享数据:高频写入的单线程(解释器主线程)、低频读取的单线程(采样线程)
  • 核心需求:最小化写入侧的性能损耗,同时能观测到写入线程偶尔卡顿数秒的情况

基准测试代码

我编写了基准测试代码来量化开销:

#include <memory>
#include <chrono>
#include <string>
#include <iostream>
#include <thread>
#include <immintrin.h>
#include <atomic>
#include <cstring>

using namespace std;

typedef struct frame {
    int32_t a;
    int32_t b;
    uint16_t c;
    uint8_t d;
} frame;

class ProfilerBase {
public:
    virtual void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) = 0;
    virtual void Stop() = 0;
    virtual ~ProfilerBase() {}
    virtual string Name() = 0;
};

class NoOp : public ProfilerBase {
public:
    void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) override {}
    void Stop() override {}
    string Name() override { return "NoOp"; }
};

class JustStore : public ProfilerBase {
private:
    frame _current = { 0 };
public:
    string Name() override { return "OnlyStoreInMember"; }
    void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) override {
        _current.a = a;
        _current.b = b;
        _current.c = c;
        _current.d = d;
    }
    void Stop() override {
        if ((_current.a + _current.b + _current.c + _current.d) == _current.a) {
            cout << "Make sure optimizer keeps the record around";
        }
    }
};

class WithSampler : public ProfilerBase {
private:
    unique_ptr<thread> _sampling;
    atomic<bool> _keepSampling = true;
protected:
    const chrono::milliseconds _sampleEvery;
    virtual void _snap() = 0;
    virtual string _subname() = 0;
public:
    WithSampler(chrono::milliseconds sampleEvery): _sampleEvery(sampleEvery) {
        _sampling = make_unique<thread>(&WithSampler::_sampler, this);
    }
    void Stop() override {
        _keepSampling = false;
        _sampling->join();
    }

    string Name() override { 
        return _subname() + to_string(_sampleEvery.count()) + "ms"; 
    }
private:
    void _sampler() {
        auto nextTick = chrono::steady_clock::now();
        while (_keepSampling)
        {
            const auto sleepTime = nextTick - chrono::steady_clock::now();
            if (sleepTime > chrono::milliseconds(0))
            {
                this_thread::sleep_for(sleepTime);
            }
            _snap();
            nextTick += _sampleEvery;
        }

    }
};

struct checkedFrame {
    frame actual;
    int32_t check;
};


// Spinlock implementation
struct spinlock {
    std::atomic<bool> lock_ = { 0 };

    void lock() noexcept {
        for (;;) {
            if (!lock_.exchange(true, std::memory_order_acquire)) {
                return;
            }
            while (lock_.load(std::memory_order_relaxed)) {
                _mm_pause(); 
            }
        }
    }

    void unlock() noexcept {
        lock_.store(false, std::memory_order_release);
    }
};


class Spinlock : public WithSampler {
private:
    spinlock _loc;
    checkedFrame _current;
public:
    using WithSampler::WithSampler;

    string _subname() override { return "Spinlock"; }

    void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) override {
        _loc.lock();
        _current.actual.a = a;
        _current.actual.b = b;
        _current.actual.c = c;
        _current.actual.d = d;
        _current.check = a + b + c + d;
        _loc.unlock();
    }

protected:
    void _snap() override {
        _loc.lock();
        auto snap = _current;
        _loc.unlock();
        if ((snap.actual.a + snap.actual.b + snap.actual.c + snap.actual.d) != snap.check) {
            cout << "Corrupted snap!!\n";
        }
    }
};

static constexpr int32_t LOOP_MAX = 1000 * 1000 * 1000;

int measure(unique_ptr<ProfilerBase> profiler) {
    cout << "Running profiler: " << profiler->Name() << "\n ";
    cout << "\tProgress: ";
    auto start_time = std::chrono::steady_clock::now();
    int r = 0;
    for (int32_t x = 0; x < LOOP_MAX; x++)
    {
        profiler->EnterFrame(x, x + x, x & 0xFFFF, x & 0xFF);
        r += x;
        if (x % (LOOP_MAX / 1000) == 0)
        {
            this_thread::sleep_for(chrono::nanoseconds(10)); // Simulate occasional work
        }
        if (x % (LOOP_MAX / 10) == 0)
        {
            cout << static_cast<int>((static_cast<double>(x) / LOOP_MAX) * 10);
        }
        if (x % 1000 == 0) {
            _mm_pause(); // Yield to other threads
        }
        if (x == (LOOP_MAX / 2)) {
            start_time = std::chrono::steady_clock::now();
        }
    }
    cout << "\n";
    const auto done_calc = std::chrono::steady_clock::now();
    profiler->Stop();
    const auto done_writing = std::chrono::steady_clock::now();
    cout << "\tcalc: " << chrono::duration_cast<chrono::milliseconds>(done_calc - start_time).count() << "ms\n";
    cout << "\tflush: " << chrono::duration_cast<chrono::milliseconds>(done_writing - done_calc).count() << "ms\n";
    return r;
}

int main() {
    measure(make_unique<NoOp>());
    measure(make_unique<JustStore>());
    measure(make_unique<Spinlock>(chrono::milliseconds(1)));
    measure(make_unique<Spinlock>(chrono::milliseconds(10)));
    return 0;
}

测试结果

x86模式下用MSVC /O2 编译后的输出:

Running profiler: NoOp
        Progress: 0123456789
        calc: 1410ms
        flush: 0ms
Running profiler: OnlyStoreInMember
        Progress: 0123456789
        calc: 1368ms
        flush: 0ms
Running profiler: Spinlock1ms
        Progress: 0123456789
        calc: 3952ms
        flush: 4ms
Running profiler: Spinlock10ms
        Progress: 0123456789
        calc: 3985ms
        flush: 11ms

注:使用g++编译的命令为 g++ --std=c++17 -O2 -m32 -pthread -o testing small-test-case.cpp,可得到近似结果。

从结果可见,基于Spinlock的采样器相比无锁版本带来了约2.5倍的性能开销,大部分时间消耗在锁操作上,而多数情况下这些锁操作并无必要。需要可行方案降低写入侧的性能损耗。


解决方案

1. 使用无锁的原子结构体拷贝(x86专属优化)

x86架构下,对齐的、不超过64字节的内存块可以通过原子指令实现整体拷贝。你的frame结构体总大小为15字节,远小于64字节,完全符合条件。

实现思路:

  • 将frame封装为std::atomic<frame>(结构体需满足可平凡复制,你的frame完全符合)
  • 写入侧用store操作,指定memory_order_relaxed(仅需保证采样能读到最新值,无需严格内存屏障)
  • 采样侧用load操作,指定memory_order_acquire(确保读取到完整的帧数据)

修改后的采样器类示例:

class AtomicFrameSampler : public WithSampler {
private:
    std::atomic<frame> _current = {};
public:
    using WithSampler::WithSampler;

    string _subname() override { return "AtomicFrame"; }

    void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) override {
        frame new_frame = {a, b, c, d};
        _current.store(new_frame, std::memory_order_relaxed);
    }

protected:
    void _snap() override {
        frame snap = _current.load(std::memory_order_acquire);
        int32_t check = snap.a + snap.b + snap.c + snap.d;
        if ((snap.a + snap.b + snap.c + snap.d) != check) {
            cout << "Corrupted snap!!\n";
        }
    }
};

性能优势:写入侧的原子store在x86上是单个mov指令,几乎和普通内存写入一样快,完全消除锁的开销。

2. 使用双缓冲区(读写分离无锁)

维护两个独立的帧缓冲区,写入和采样操作完全分离:

  • 写入侧始终写入当前活跃的缓冲区,无需加锁
  • 采样侧读取非活跃的缓冲区,避免和写入操作冲突

实现思路:

  • 定义两个frame实例和一个原子标记位,标记当前活跃的写入缓冲区
  • 写入侧直接写入活跃缓冲区;采样侧读取另一个缓冲区,可选切换活跃标记位

示例代码:

class DoubleBufferSampler : public WithSampler {
private:
    frame _bufs[2] = {};
    std::atomic<uint8_t> _active_buf = 0;
public:
    using WithSampler::WithSampler;

    string _subname() override { return "DoubleBuffer"; }

    void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) override {
        uint8_t idx = _active_buf.load(std::memory_order_relaxed);
        _bufs[idx].a = a;
        _bufs[idx].b = b;
        _bufs[idx].c = c;
        _bufs[idx].d = d;
    }

protected:
    void _snap() override {
        uint8_t read_idx = 1 - _active_buf.load(std::memory_order_acquire);
        frame snap = _bufs[read_idx];
        // 可选:切换活跃缓冲区,让下一次采样读取新的缓冲区
        // _active_buf.store(read_idx, std::memory_order_release);
        int32_t check = snap.a + snap.b + snap.c + snap.d;
        if ((snap.a + snap.b + snap.c + snap.d) != check) {
            cout << "Corrupted snap!!\n";
        }
    }
};

性能优势:写入侧完全无锁,性能和JustStore版本一致;采样侧读取的是静态缓冲区,不会出现数据部分更新的问题。唯一的小概率问题是采样可能读到稍旧的数据,但对于性能分析场景来说完全可接受。

3. 编译器内置原子操作(精确控制内存顺序)

如果std::atomic<frame>的兼容性有顾虑,可以直接使用x86内置指令:

class BuiltinAtomicSampler : public WithSampler {
private:
    frame _current = {};
public:
    using WithSampler::WithSampler;

    string _subname() override { return "BuiltinAtomic"; }

    void EnterFrame(int32_t a, int32_t b, uint16_t c, uint8_t d) override {
        frame new_frame = {a, b, c, d};
        __atomic_store(&_current, &new_frame, __ATOMIC_RELAXED);
    }

protected:
    void _snap() override {
        frame snap;
        __atomic_load(&_current, &snap, __ATOMIC_ACQUIRE);
        int32_t check = snap.a + snap.b + snap.c + snap.d;
        if ((snap.a + snap.b + snap.c + snap.d) != check) {
            cout << "Corrupted snap!!\n";
        }
    }
};

这种方式和std::atomic实现本质一致,但可以更精确地控制内存栅栏行为。


方案对比

方案写入侧开销实现复杂度数据一致性适用场景
Spinlock高(2.5倍开销)低强一致不推荐,开销过大
原子结构体拷贝极低(接近无锁)低强一致首选方案,x86下完美适配
双缓冲区无(和JustStore一致)中最终一致(允许稍旧数据)追求极致性能,可接受采样数据稍旧

推荐优先选择原子结构体拷贝方案,它在保证数据一致性的同时,写入侧性能几乎和无锁版本一致,完全满足需求。


内容的提问来源于stack exchange,提问作者Davy Landman

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.11 08:01:35