C# .Net中SIMD优化代码运行变慢的原因及相关问题咨询
C#纹理扫描线渲染的SIMD优化踩坑记录
我尝试用System.Numerics Vector对C#纹理扫描线渲染函数做SIMD优化,结果优化后函数比原函数慢3倍——原函数渲染含两个多边形的场景需0.4ms,优化后需1.7ms。排查后发现,**数组与向量间频繁的数据互转(尤其是浮点转整数时的中转操作)**是性能暴跌的核心原因。
之后改用SSE42的ConvertToVector128Int32重构代码,性能较原函数有小幅提升,但遇到了整数向量乘法的问题:
- SSE仅支持浮点向量乘法
- SSE2-SSE3支持双精度、无符号整数和浮点乘法
- SSE41-SSE42的整数乘法仅返回包含2个值的
Vector128<long>,且无法通过无符号整数转换规避
最终发现从SSE41开始可使用MultiplyLow实现整数向量乘法,解决了该问题。
原函数
private void RenderScanlinesTextureMapInternal(int[] minX, int[] maxX, float[] min1W, float[] max1W, System.Numerics.Vector2[] minTex, System.Numerics.Vector2[] maxTex, Texture t) { unsafe { fixed (byte* tbufferbyte = t.internalBuffer) { int tdimension = t.dimension; fixed (byte* f = backBuffer) { for (int y = 0; y < Height; y++) { if (maxX[y] == 0 && minX[y] == 0) { continue; } // int lineLength = maxX[y] - minX[y]; // float oneW = min1W[y]; float oneWDist = max1W[y] - min1W[y]; float dOneW = oneWDist / (float)lineLength; // float texX = minTex[y].X; float texY = minTex[y].Y; float texDistX = maxTex[y].X - minTex[y].X; float texDistY = maxTex[y].Y - minTex[y].Y; float dTexX = texDistX / lineLength; float dTexY = texDistY / lineLength; // Int32* p; p = (Int32*)f + ((int)y * Width) + minX[y]; for (int x = minX[y]; x < maxX[y]; x++, oneW += dOneW, texX += dTexX, texY += dTexY) { float oneOneW = 1 / oneW; float tu = texX * oneOneW; float tv = texY * oneOneW; float tudimensionF = tu * tdimension; float tvdimensionF = tv * tdimension; int tudimension = (int)tudimensionF; int tvdimension = (int)tvdimensionF; int u = tudimension & tdimension - 1; int v = tvdimension & tdimension - 1; int pos = (v * tdimension) + u; *p++ = *((Int32*)tbufferbyte + pos); } } } } } }
SIMD优化函数
private void RenderScanlinesTextureSIMD(int[] minX, int[] maxX, float[] min1W, float[] max1W, System.Numerics.Vector2[] minTex, System.Numerics.Vector2[] maxTex, Texture t) { unsafe { fixed (byte* tbufferbyte = t.internalBuffer) { int tdimension = t.dimension; fixed (byte* f = backBuffer) { int VectorFCount = Vector<float>.Count; // Vector<float> VoneOneW; Vector<float> Vtu; //float tu = texX * oneOneW; Vector<float> Vtv; //float tv = texY * oneOneW; Vector<float> VtuDimensionF; // float tudimensionF = tu * tdimension; Vector<float> VtvDimensionF; // float tvdimensionF = tv * tdimension; Vector<int> VtuDimension; // int tudimension = (int)tudimensionF Vector<int> VtvDimension; // int tvdimension = (int)tvdimensionF Vector<int> Vu; // int u = tudimension & tdimension - 1; Vector<int> Vv; // int v = tvdimension & tdimension - 1; Vector<int> Vpos; // int pos = (v * tdimension) + u; Vector<float> Tfloat; // float[] Arrtdimension = new float[VectorFCount]; int[] Arrtdimensionint = new int[VectorFCount]; int[] Arrtdimensionminus1 = new int[VectorFCount]; float[] ArroneW = new float[VectorFCount]; float[] ArrtexX = new float[VectorFCount]; float[] ArrtexY = new float[VectorFCount]; float[] ArroneOneW = new float[VectorFCount]; float[] ArrTu = new float[VectorFCount]; float[] ArrTv = new float[VectorFCount]; float[] ArrTuDimensionF = new float[VectorFCount]; float[] ArrTvDimensionF = new float[VectorFCount]; int[] ArrTuDimension = new int[VectorFCount]; int[] ArrTvDimension = new int[VectorFCount]; int[] Arru = new int[VectorFCount]; int[] Arrv = new int[VectorFCount]; int[] Arrpos = new int[VectorFCount]; for (int y = 0; y < Height; y++) { if (maxX[y] == 0 && minX[y] == 0) { continue; } // int lineLength = maxX[y] - minX[y]; // float oneW = min1W[y]; float oneWDist = max1W[y] - min1W[y]; float dOneW = oneWDist / (float)lineLength; // float texX = minTex[y].X; float texY = minTex[y].Y; float texDistX = maxTex[y].X - minTex[y].X; float texDistY = maxTex[y].Y - minTex[y].Y; float dTexX = texDistX / lineLength; float dTexY = texDistY / lineLength; // Int32* p; p = (Int32*)f + ((int)y * Width) + minX[y]; int x = 0; int lastBlockIndex = lineLength - (lineLength % VectorFCount); for (x = minX[y]; x < minX[y]+lastBlockIndex; x+= VectorFCount) { for (int b = 0; b < VectorFCount; b++) { Arrtdimension[b] = tdimension; Arrtdimensionint[b] = (int)tdimension; Arrtdimensionminus1[b] = tdimension-1; ArroneW[b] = oneW; ArrtexX[b] = texX; ArrtexY[b] = texY; oneW += dOneW; texX += dTexX; texY += dTexY; } Tfloat = new Vector<float>(ArroneW); VoneOneW = Vector<float>.One; VoneOneW /= Tfloat; Vtu = new Vector<float>(ArrtexX); Vtu *= VoneOneW; Vtv = new Vector<float>(ArrtexY); Vtv *= VoneOneW; VtuDimensionF = Vtu; VtuDimensionF *= new Vector<float>(Arrtdimension); VtuDimensionF.CopyTo(ArrTuDimensionF); VtvDimensionF = Vtv; VtvDimensionF *= new Vector<float>(Arrtdimension); VtvDimensionF.CopyTo(ArrTvDimensionF); for (int b = 0; b < VectorFCount; b++) { ArrTuDimension[b] = (int)ArrTuDimensionF[b]; ArrTvDimension[b] = (int)ArrTvDimensionF[b]; } VtuDimension = new Vector<int>(ArrTuDimension); VtvDimension = new Vector<int>(ArrTvDimension); Vu = VtuDimension; Vu &= new Vector<int>(Arrtdimensionminus1); Vv = VtvDimension; Vv &= new Vector<int>(Arrtdimensionminus1); Vpos = Vv; Vpos *= new Vector<int>(Arrtdimensionint); Vpos += Vu; for (int b = 0; b < VectorFCount; b++) { *p++ = *((Int32*)tbufferbyte + Vpos[b]); } } } } } } }
内容的提问来源于stack exchange,提问作者Lasse
相关产品推荐
相关产品推荐

