You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何高效将RGB888格式内存转换为RGB888x格式字节数组?

RGB888转RGB888x的高效实现方案

针对你在.NET Standard 2.0下(支持Span)的需求,以下是几种比逐像素循环更高效的实现方案,按性能优先级排序:

方案1:SIMD向量化处理(性能最优)

利用System.Numerics.Vector类调用CPU的SIMD指令,批量处理多个像素,大幅减少循环次数,充分利用硬件加速。

using System;
using System.Numerics;
using System.Runtime.InteropServices;
using System.Runtime.CompilerServices;

public static byte[] ConvertRgb888ToRgb888x(IntPtr rgb888Ptr, int width, int height)
{
    int pixelCount = width * height;
    byte[] result = new byte[pixelCount * 4];

    // 直接从非托管内存创建Span,避免额外拷贝
    Span<byte> source = MemoryMarshal.CreateSpan(
        ref Unsafe.AsRef<byte>(rgb888Ptr.ToPointer()),
        pixelCount * 3);
    Span<byte> destination = result.AsSpan();

    int batchSize = Vector<byte>.Count;
    int processedPixels = 0;

    // 批量处理像素,利用SIMD加速
    while (processedPixels <= pixelCount - batchSize)
    {
        // 批量提取R、G、B通道
        var rVec = new Vector<byte>(source.Slice(processedPixels * 3, batchSize));
        var gVec = new Vector<byte>(source.Slice(processedPixels * 3 + 1, batchSize));
        var bVec = new Vector<byte>(source.Slice(processedPixels * 3 + 2, batchSize));

        // 写入目标数组对应位置,第四个字节填0(可自定义值)
        rVec.CopyTo(destination.Slice(processedPixels * 4, batchSize));
        gVec.CopyTo(destination.Slice(processedPixels * 4 + 1, batchSize));
        bVec.CopyTo(destination.Slice(processedPixels * 4 + 2, batchSize));
        Vector<byte>.Zero.CopyTo(destination.Slice(processedPixels * 4 + 3, batchSize));

        processedPixels += batchSize;
    }

    // 处理剩余不足一个Vector长度的像素
    for (; processedPixels < pixelCount; processedPixels++)
    {
        destination[processedPixels * 4] = source[processedPixels * 3];
        destination[processedPixels * 4 + 1] = source[processedPixels * 3 + 1];
        destination[processedPixels * 4 + 2] = source[processedPixels * 3 + 2];
        destination[processedPixels * 4 + 3] = 0;
    }

    return result;
}

依赖说明:需要安装System.Runtime.CompilerServices.Unsafe和System.Numerics NuGet包,可通过Vector.IsHardwareAccelerated判断当前CPU是否支持SIMD加速。

方案2:分通道批量复制(平衡性能与复杂度)

通过三次独立循环复制R、G、B三个通道,利用CPU缓存的连续访问特性,比逐像素单循环效率更高,实现简单无需依赖SIMD。

using System;
using System.Runtime.InteropServices;

public static byte[] ConvertRgb888ToRgb888x(IntPtr rgb888Ptr, int width, int height)
{
    int pixelCount = width * height;
    int sourceLen = pixelCount * 3;
    byte[] sourceBuffer = new byte[sourceLen];
    Marshal.Copy(rgb888Ptr, sourceBuffer, 0, sourceLen);

    byte[] result = new byte[pixelCount * 4];
    Span<byte> src = sourceBuffer.AsSpan();
    Span<byte> dst = result.AsSpan();

    // 批量复制R通道
    for (int i = 0; i < pixelCount; i++)
    {
        dst[i * 4] = src[i * 3];
    }
    // 批量复制G通道
    for (int i = 0; i < pixelCount; i++)
    {
        dst[i * 4 + 1] = src[i * 3 + 1];
    }
    // 批量复制B通道
    for (int i = 0; i < pixelCount; i++)
    {
        dst[i * 4 + 2] = src[i * 3 + 2];
    }
    // 填充第四个字节(此处填0)
    Span<byte> alphaSpan = dst.Slice(3, pixelCount);
    for (int i = 0; i < alphaSpan.Length; i += 8)
    {
        alphaSpan.Slice(i, Math.Min(8, alphaSpan.Length - i)).Fill(0);
    }

    return result;
}

方案3:手动展开循环(兼容无SIMD环境)

手动展开循环减少分支判断开销,提升CPU指令并行度,适合无法使用SIMD的场景。

using System;
using System.Runtime.InteropServices;
using System.Runtime.CompilerServices;

public static byte[] ConvertRgb888ToRgb888x(IntPtr rgb888Ptr, int width, int height)
{
    int pixelCount = width * height;
    byte[] result = new byte[pixelCount * 4];

    Span<byte> source = MemoryMarshal.CreateSpan(
        ref Unsafe.AsRef<byte>(rgb888Ptr.ToPointer()),
        pixelCount * 3);
    Span<byte> destination = result.AsSpan();

    int i = 0;
    // 每次处理8个像素,手动展开循环
    while (i <= pixelCount - 8)
    {
        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;

        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0; i++;
    }

    // 处理剩余像素
    for (; i < pixelCount; i++)
    {
        destination[i*4] = source[i*3];
        destination[i*4+1] = source[i*3+1];
        destination[i*4+2] = source[i*3+2];
        destination[i*4+3] = 0;
    }

    return result;
}

方案选择建议

  • 优先选SIMD向量化方案,大尺寸图像下性能提升最明显;
  • 若不需要依赖额外NuGet包或兼容老CPU,选分通道复制方案;
  • 无SIMD支持的环境下,选手动展开循环方案。

内容的提问来源于stack exchange,提问作者Ragnarokkr Xia

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.19 19:14:54