You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C# .Net中SIMD优化代码运行变慢的原因及相关问题咨询

C#纹理扫描线渲染的SIMD优化踩坑记录

我尝试用System.Numerics Vector对C#纹理扫描线渲染函数做SIMD优化,结果优化后函数比原函数慢3倍——原函数渲染含两个多边形的场景需0.4ms,优化后需1.7ms。排查后发现,**数组与向量间频繁的数据互转(尤其是浮点转整数时的中转操作)**是性能暴跌的核心原因。

之后改用SSE42的ConvertToVector128Int32重构代码,性能较原函数有小幅提升,但遇到了整数向量乘法的问题:

  • SSE仅支持浮点向量乘法
  • SSE2-SSE3支持双精度、无符号整数和浮点乘法
  • SSE41-SSE42的整数乘法仅返回包含2个值的Vector128<long>,且无法通过无符号整数转换规避

最终发现从SSE41开始可使用MultiplyLow实现整数向量乘法,解决了该问题。

原函数

private void RenderScanlinesTextureMapInternal(int[] minX, int[] maxX, float[] min1W, float[] max1W, System.Numerics.Vector2[] minTex, System.Numerics.Vector2[] maxTex, Texture t) {
    unsafe {
        fixed (byte* tbufferbyte = t.internalBuffer) {
            int tdimension = t.dimension;
            fixed (byte* f = backBuffer) {
                for (int y = 0; y < Height; y++) {
                    if (maxX[y] == 0 && minX[y] == 0) {
                        continue;
                    }
                    //
                    int lineLength = maxX[y] - minX[y];
                    //
                    float oneW = min1W[y];
                    float oneWDist = max1W[y] - min1W[y];
                    float dOneW = oneWDist / (float)lineLength;
                    //
                    float texX = minTex[y].X;
                    float texY = minTex[y].Y;
                    float texDistX = maxTex[y].X - minTex[y].X;
                    float texDistY = maxTex[y].Y - minTex[y].Y;
                    float dTexX = texDistX / lineLength;
                    float dTexY = texDistY / lineLength;
                    //
                    Int32* p;
                    p = (Int32*)f + ((int)y * Width) + minX[y];

                    for (int x = minX[y]; x < maxX[y]; x++, oneW += dOneW, texX += dTexX, texY += dTexY) {
                        float oneOneW = 1 / oneW;
                        float tu = texX * oneOneW;
                        float tv = texY * oneOneW;
                        float tudimensionF = tu * tdimension;
                        float tvdimensionF = tv * tdimension;
                        int tudimension = (int)tudimensionF;
                        int tvdimension = (int)tvdimensionF;
                        int u = tudimension & tdimension - 1;
                        int v = tvdimension & tdimension - 1;
                        int pos = (v * tdimension) + u;
                        *p++ = *((Int32*)tbufferbyte + pos);
                    }
                            
                }
            }
        }
    }
}

SIMD优化函数

private void RenderScanlinesTextureSIMD(int[] minX, int[] maxX, float[] min1W, float[] max1W, System.Numerics.Vector2[] minTex, System.Numerics.Vector2[] maxTex, Texture t) {
    unsafe {
        fixed (byte* tbufferbyte = t.internalBuffer) {
            int tdimension = t.dimension;
            fixed (byte* f = backBuffer) {
                int VectorFCount = Vector<float>.Count;
                //
                Vector<float> VoneOneW;
                Vector<float> Vtu;              //float tu = texX * oneOneW;
                Vector<float> Vtv;              //float tv = texY * oneOneW;
                Vector<float> VtuDimensionF;    // float tudimensionF = tu * tdimension;
                Vector<float> VtvDimensionF;    // float tvdimensionF = tv * tdimension;
                Vector<int> VtuDimension;       // int tudimension = (int)tudimensionF
                Vector<int> VtvDimension;       // int tvdimension = (int)tvdimensionF
                Vector<int> Vu;                 // int u = tudimension & tdimension - 1;
                Vector<int> Vv;                 // int v = tvdimension & tdimension - 1;
                Vector<int> Vpos;               // int pos = (v * tdimension) + u;
                Vector<float> Tfloat;

                //
                float[] Arrtdimension = new float[VectorFCount];
                int[] Arrtdimensionint = new int[VectorFCount];
                int[] Arrtdimensionminus1 = new int[VectorFCount];
                float[] ArroneW = new float[VectorFCount];
                float[] ArrtexX = new float[VectorFCount];
                float[] ArrtexY = new float[VectorFCount];
                float[] ArroneOneW = new float[VectorFCount];
                float[] ArrTu = new float[VectorFCount];
                float[] ArrTv = new float[VectorFCount];
                float[] ArrTuDimensionF = new float[VectorFCount];
                float[] ArrTvDimensionF = new float[VectorFCount];
                int[] ArrTuDimension = new int[VectorFCount];
                int[] ArrTvDimension = new int[VectorFCount];
                int[] Arru = new int[VectorFCount];
                int[] Arrv = new int[VectorFCount];
                int[] Arrpos = new int[VectorFCount];
                for (int y = 0; y < Height; y++) {
                    if (maxX[y] == 0 && minX[y] == 0) {
                        continue;
                    }
                    //
                    int lineLength = maxX[y] - minX[y];
                    //
                    float oneW = min1W[y];
                    float oneWDist = max1W[y] - min1W[y];
                    float dOneW = oneWDist / (float)lineLength;
                    //
                    float texX = minTex[y].X;
                    float texY = minTex[y].Y;
                    float texDistX = maxTex[y].X - minTex[y].X;
                    float texDistY = maxTex[y].Y - minTex[y].Y;
                    float dTexX = texDistX / lineLength;
                    float dTexY = texDistY / lineLength;
                    //
                    Int32* p;
                    p = (Int32*)f + ((int)y * Width) + minX[y];
                    int x = 0;
                    int lastBlockIndex = lineLength - (lineLength % VectorFCount);
                    for (x = minX[y]; x < minX[y]+lastBlockIndex; x+= VectorFCount) {
                        for (int b = 0; b < VectorFCount; b++) {
                            Arrtdimension[b] = tdimension;
                            Arrtdimensionint[b] = (int)tdimension;
                            Arrtdimensionminus1[b] = tdimension-1;
                            ArroneW[b] = oneW;
                            ArrtexX[b] = texX;
                            ArrtexY[b] = texY;
                            oneW += dOneW;
                            texX += dTexX;
                            texY += dTexY;
                        }

                        Tfloat = new Vector<float>(ArroneW);
                        VoneOneW = Vector<float>.One;
                        VoneOneW /= Tfloat;

                        Vtu = new Vector<float>(ArrtexX);
                        Vtu *= VoneOneW;

                        Vtv = new Vector<float>(ArrtexY);
                        Vtv *= VoneOneW;

                        VtuDimensionF = Vtu;
                        VtuDimensionF *= new Vector<float>(Arrtdimension);
                        VtuDimensionF.CopyTo(ArrTuDimensionF);

                        VtvDimensionF = Vtv;
                        VtvDimensionF *= new Vector<float>(Arrtdimension);
                        VtvDimensionF.CopyTo(ArrTvDimensionF);

                        for (int b = 0; b < VectorFCount; b++) {
                            ArrTuDimension[b] = (int)ArrTuDimensionF[b];
                            ArrTvDimension[b] = (int)ArrTvDimensionF[b];
                        }
                        VtuDimension = new Vector<int>(ArrTuDimension);
                        VtvDimension = new Vector<int>(ArrTvDimension);

                        Vu = VtuDimension;
                        Vu &= new Vector<int>(Arrtdimensionminus1);

                        Vv = VtvDimension;
                        Vv &= new Vector<int>(Arrtdimensionminus1);

                        Vpos = Vv;
                        Vpos *= new Vector<int>(Arrtdimensionint);

                        Vpos += Vu;

                        for (int b = 0; b < VectorFCount; b++) {
                            *p++ = *((Int32*)tbufferbyte + Vpos[b]);
                        }
                    }
                }
            }
        }
    }
}

内容的提问来源于stack exchange,提问作者Lasse

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 09:05:21