You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

多线程为何无法提升部分CPU的内存访问速度?求优化方案

内存密集型代码多线程优化差异问题解惑

我平时使用SIMD和多线程优化C++代码,近期发现针对内存密集型代码,多线程的优化效果极差。为此编写了如下测试代码:

#include <immintrin.h>
void test_add(float* src1, float* dst, int length, int thread_num) {
    auto task = [&](int offset, int block){
        float* src1_addr = src1 + offset;
        float* dst_addr = dst + offset;
        int main_loop = block >> 3;
        int remain_loop = block - main_loop * 8;
        float coff = 0.1;
       __m256 coff_vct = _mm256_set1_ps(coff);
       for (int i = 0; i < main_loop; i++) {
           __m256 src1_vct = _mm256_loadu_ps(src1_addr);
           src1_vct = _mm256_add_ps(src1_vct, coff_vct); //The proportion of addition instruction time consumption is very small
           _mm256_storeu_ps(dst_addr, src1_vct);
           src1_addr += 8;
           dst_addr += 8;
       }
        for (int i = 0; i < remain_loop; i++) {
            *dst_addr = *src1_addr + coff;
            src1_addr += 1;
            dst_addr += 1;
        }
    };
    
    std::vector<std::future<void> > fus;
    int block_size = length / thread_num;
    for (int i = 0; i < thread_num - 1; i++) {
        int offset = i * block_size;
        fus.push_back(
            std::async(std::launch::async, [=](){ 
                task(offset, block_size);
            })
        );
    }

    int offset = (thread_num - 1) * block_size;
    task(offset, length - offset);

    for (auto& fu : fus) {
        fu.get(); 
    }
}

int test() {

    int length = 768 * 768 * 16 * 8;
    float* src1 = new float[length];
    float* dst = new float[length];

    for (int i = 0; i < length; i++) {
        src1[i] = i % 255;
        dst[i] = 0;
    }

    int thread_num = 4;
    int loop = 10;
    struct timeval start_time, end_time ;
    gettimeofday(&start_time, NULL);
    for (int i = 0; i < loop; i++) {
        test_add(src1, dst, length, thread_num);
    }
    gettimeofday(&end_time, NULL);
    printf("%d thread_num = %d cost %.2fms\n",length, thread_num, 1000.0 * (end_time.tv_sec - start_time.tv_sec) + (end_time.tv_usec - start_time.tv_usec) / 1000.0f / loop);

    for (int i = 0; i < 10; i++) {
        printf("dst = %.1f, ", dst[i]);
    }
    printf("\n");
    return 0;
}

在三台电脑上测试后,发现不同平台的多线程优化效果差异明显:

  • Mac x86 i7-9750H:
    Mac x86 i7-9750H测试结果截图
  • Win i5-10400:
    Win i5-10400测试结果截图
  • Win i7-12650H:
    Win i7-12650H测试结果截图

每个CPU核心都有独立缓存,且测试案例中不存在多线程访问冲突问题。无法理解为何Mac x86 i7-9750H和Win i5-10400上多线程无法提升内存访问速度,恳请了解CPU底层原理的人士解惑,或提供代码优化方案。

内容的提问来源于stack exchange,提问作者maofu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 01:54:54