You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用AVX指令加速神经网络时遭遇内存访问违例问题求助

AVX加速神经网络时访问违例的原因及修复

问题复现

尝试用AVX指令加速神经网络计算时,反复出现“未处理的异常:访问违例读取位置”,错误位置不固定,疑似内存损坏。可复现代码如下:

#include <immintrin.h>
#include <cmath>
#include <iostream>
#include <array>
#include <vector>

inline const int num_avx_registers = 16;
inline const int floats_per_reg = 4;

inline const int HKP_size = 100;
inline constexpr int acc_size = 256;

class NNLayer {
    public:
    alignas(32) float* weight;
    alignas(32) float* bias;

    NNLayer(){
        weight = new float[HKP_size * acc_size]; // flattened 2D array.
        bias = new float[acc_size];

        // initialize the weights and bias with test values
        for (uint32_t i=0; i<HKP_size * acc_size; i++){
            weight[i] = 1.F;
        }

        for (int i=0; i<acc_size; i++){
            bias[i] = static_cast<float>(i);
        }
    }

    ~NNLayer(){
        delete[] weight;
        delete[] bias;
    }
};

class Accumulator {
    public:
    alignas(32) std::array<float, acc_size> accumulator_w;
    alignas(32) std::array<float, acc_size> accumulator_b;

    std::array<float, acc_size>& Accumulator::operator[](bool color){
        return color ? accumulator_w : accumulator_b;
    }
};

class NNUE {
    public:
    Accumulator accumulator;
    NNLayer first_layer = NNLayer();

    void compute_accumulator(const std::vector<int> active_features, bool color){
        // we have 256 floats to process.
        // there are 16 avx registers, and each can hold 4 floats.
        // therefore we need to do 256/64 = 4 passes to the registers.

        constexpr int c_size = num_avx_registers * floats_per_reg; //chunk size
        constexpr int num_chunks = acc_size / c_size;
        
        static_assert(acc_size % c_size == 0);

        __m256 avx_regs[num_avx_registers];

        // we process 1/4th of the whole data at each loop.
        // we add c_idx to the indexes pick up where we left off at the last chunk.
        for (int c_idx = 0; c_idx < num_chunks*c_size; c_idx += c_size){ // chunk index

            // load the bias from memory
            for (int i = 0; i < num_avx_registers; i++){
                avx_regs[i] = _mm256_load_ps(&first_layer.bias[c_idx + i*floats_per_reg]);
            }

            // add the active weights
            for (const int &a: active_features){
                for (int i = 0; i < num_avx_registers; i++){
                    // a*acc_size is to get the a-th row of the flattened 2D array.
                    avx_regs[i] = _mm256_add_ps(
                        avx_regs[i],
                        _mm256_load_ps(&first_layer.weight[a*acc_size + c_idx + i*floats_per_reg])
                        );
                }
            }

            //store the result in the accumulator
            for (int i = 0; i < num_avx_registers; i++){
                _mm256_store_ps(&accumulator[color][c_idx + i*floats_per_reg], avx_regs[i]);
            }
        }
    }
};

int main(){
    NNUE nnue;

    std::vector<int> act_f = {2, 1, 70, 62};
    nnue.compute_accumulator(act_f, true);

    std::cout << "still alive\n";
    return 0;
}

核心原因及修复方案

1. 内存未按AVX要求对齐(崩溃的直接原因)

AVX的_mm256_load_ps和_mm256_store_ps指令强制要求内存操作数是32字节对齐,但代码中用new float[]分配的堆内存无法保证这一对齐要求——alignas(32)仅修饰指针变量本身的存储,不影响new分配的内存块的起始地址对齐属性。

修复方法:
使用aligned_alloc分配对齐内存,替换new,同时用free释放(不能再用delete[]):

NNLayer(){
    // 分配32字节对齐的内存
    weight = static_cast<float*>(aligned_alloc(32, HKP_size * acc_size * sizeof(float)));
    bias = static_cast<float*>(aligned_alloc(32, acc_size * sizeof(float)));

    // 初始化逻辑不变...
}

~NNLayer(){
    free(weight);
    free(bias);
}

如果使用C++17及以上,也可以用std::aligned_allocator配合std::vector管理内存,避免手动分配释放的风险。

2. 成员函数语法错误

Accumulator类中operator[]的定义重复了类名限定符,正确写法需去掉Accumulator:::

std::array<float, acc_size>& operator[](bool color){
    return color ? accumulator_w : accumulator_b;
}

该语法错误虽不直接导致崩溃,但会引发编译器警告,可能间接导致未定义行为。

3. 其他注意事项

  • 确保active_features中的元素范围在[0, HKP_size-1]内,避免权重数组越界访问;
  • 编译时需开启AVX支持:GCC/Clang加-mavx参数,MSVC加/arch:AVX参数,否则编译器可能生成不兼容的指令。

内容的提问来源于stack exchange,提问作者Nonlinear

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.28 08:11:02