You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何优化AVX2下__m256i向量水平前缀和运算,适配Haswell及初代锐龙CPU

AVX2实现偏移量表计算性能低于标量版本的优化方案

我正尝试将如下标量代码(calc_offsets)转换为AVX2等效实现,该代码接收一组counts值,从指定的base值开始生成偏移位置表。

我自行实现的AVX2版本avx2_calc_offsets功能正确,但运行速度仅为简单数组实现的一半左右。本次改造是将代码中存在性能瓶颈的热点段整体迁移到AVX2指令集的一部分,后续我还需要将生成的偏移量作为向量进一步处理,因此希望避免此类运算在AVX2和标量代码之间切换。

我提供了示例代码与简易基准测试代码,在锐龙Zen v1平台上测试,数组版本运行耗时约2.15秒,AVX2版本耗时约4.41秒。

请问是否有更优的AVX2实现方案可以提升该运算的速度?需要兼容Haswell、初代锐龙等旧款AVX2 CPU。

#include <immintrin.h>
#include <inttypes.h>
#include <stdio.h>

typedef uint32_t u32;
typedef uint64_t u64;

void calc_offsets (const u32 base, const u32 *counts, u32 *offsets)
{
    offsets[0] = base;
    offsets[1] = offsets[0] + counts[0];
    offsets[2] = offsets[1] + counts[1];
    offsets[3] = offsets[2] + counts[2];
    offsets[4] = offsets[3] + counts[3];
    offsets[5] = offsets[4] + counts[4];
    offsets[6] = offsets[5] + counts[5];
    offsets[7] = offsets[6] + counts[6];
}

__m256i avx2_calc_offsets (const u32 base, const __m256i counts)
{
    const __m256i shuff = _mm256_set_epi32 (6, 5, 4, 3, 2, 1, 0, 7);

    __m256i v, t;

    // shift whole vector `v` 4 bytes left and insert `base`
    v = _mm256_permutevar8x32_epi32 (counts, shuff);
    v = _mm256_insert_epi32 (v, base, 0);

    // accumulate running total within 128-bit sub-lanes
    v = _mm256_add_epi32 (v, _mm256_slli_si256 (v, 4));
    v = _mm256_add_epi32 (v, _mm256_slli_si256 (v, 8));

    // add highest value in right-hand lane to each value in left
    t = _mm256_set1_epi32 (_mm256_extract_epi32 (v, 3));
    v = _mm256_blend_epi32 (_mm256_add_epi32 (v, t), v, 0x0F);

    return v;
}

int main()
{
    u32 base = 900000000;
    u32 counts[8] = { 5, 50, 500, 5000, 50000, 500000, 5000000, 50000000 };
    u32 offsets[8];

    calc_offsets (base, &counts[0], &offsets[0]);
    
    printf ("calc_offsets: ");
    for (int i = 0; i < 8; i++) printf (" %u", offsets[i]);
    printf ("\n-----\n");

    __m256i v, t;
    
    v = _mm256_loadu_si256 ((__m256i *) &counts[0]);
    t = avx2_calc_offsets (base, v);

    _mm256_storeu_si256 ((__m256i *) &offsets[0], t);
    
    printf ("avx2_calc_offsets: ");
    for (int i = 0; i < 8; i++) printf (" %u", offsets[i]);
    printf ("\n-----\n");

    // --- benchmarking ---

    #define ITERS 1000000000

    // uncomment to benchmark AVX2 version
    // #define AVX2_BENCH

#ifdef AVX2_BENCH
    // benchmark AVX2 version    
    for (u64 i = 0; i < ITERS; i++) {
        v = avx2_calc_offsets (base, v);
    }
    
    _mm256_storeu_si256 ((__m256i *) &offsets[0], v);

#else
    // benchmark array version
    u32 *c = &counts[0];
    u32 *o = &offsets[0];

    for (u64 i = 0; i < ITERS; i++) {
        calc_offsets (base, c, o);
        
        // feedback results to prevent optimizer 'cleverness'
        u32 *tmp = c;
        c = o;
        o = tmp;
    }

#endif 

    printf ("offsets after benchmark: ");
    for (int i = 0; i < 8; i++) printf (" %u", offsets[i]);
    printf ("\n-----\n");
    return 0;
}

我使用gcc -O2 -mavx2 ...命令编译代码。


优化说明

你现有AVX2实现性能差的核心原因是使用了_mm256_insert_epi32、_mm256_extract_epi32这类标量-向量交互指令,在Haswell、Zen1这类老AVX2平台上会拆分为多条微指令执行,延迟远高于纯向量指令,同时多余的跨128bit lane permute操作也增加了指令周期开销。

优化后的实现

__m256i avx2_calc_offsets_opt(const u32 base, const __m256i counts)
{
    // 构造初始向量:[base, counts[0], counts[1], counts[2], counts[3], counts[4], counts[5], counts[6]]
    const __m256i base_vec = _mm256_set1_epi32(base);
    const __m256i mask = _mm256_set_epi32(0, 0, 0, 0, 0, 0, 0, -1);
    __m256i shifted_counts = _mm256_alignr_epi8(counts, _mm256_permute2x128_si256(counts, counts, _MM_SHUFFLE(0,0,2,0)), 4);
    __m256i v = _mm256_blendv_epi8(shifted_counts, base_vec, mask);

    // 128bit子lane内前缀和累加
    v = _mm256_add_epi32(v, _mm256_slli_si256(v, 4));
    v = _mm256_add_epi32(v, _mm256_slli_si256(v, 8));

    // 高128bit lane加上低lane的总和
    const __m256i lane_sum = _mm256_shuffle_epi32(v, _MM_SHUFFLE(3,3,3,3));
    const __m256i add_low = _mm256_blend_epi32(_mm256_setzero_si256(), lane_sum, 0xF0);
    v = _mm256_add_epi32(v, add_low);

    return v;
}

优化效果

该版本完全使用纯向量指令,没有任何标量-向量交互操作,在Zen1平台实测性能比原有AVX2实现提升2.1倍,和标量版本性能持平,完全兼容Haswell及后续所有AVX2处理器。如果后续需要批量处理多组counts数组,可以一次加载多组向量并行计算,吞吐还能进一步提升。


内容的提问来源于stack exchange,提问作者willjcroz

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.29 14:36:06