You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何高效将u32中的6个5位位域扩展至u8[6]缓冲区?

位域展开的x86性能优化

问题背景

需要将一个u32变量中的6个连续5位位域,逐个提取并存储到u8[6]缓冲区中。本质是将30位的紧凑数据转换为48位的展开格式——每个原始5位段前补3个零,对应到缓冲区的一个字节:

11111 11111 11111 11111 11111 11111
                                                    ↓
00011111 00011111 00011111 00011111 00011111 00011111

原始实现与编译输出

原始C代码

void Expand(u32 x, u8 b[6]) {
    b[0] = (x >> 0) & 31;
    b[1] = (x >> 5) & 31;
    b[2] = (x >> 10) & 31;
    b[3] = (x >> 15) & 31;
    b[4] = (x >> 20) & 31;
    b[5] = (x >> 25) & 31;
}

MSVC v19.latest(/O2 /Ot /Gr)生成的汇编

@Expand@8 PROC
        mov     al, cl
        and     al, 31
        mov     BYTE PTR [edx], al
        mov     eax, ecx
        shr     eax, 5
        and     al, 31
        mov     BYTE PTR [edx+1], al
        mov     eax, ecx
        shr     eax, 10
        and     al, 31
        mov     BYTE PTR [edx+2], al
        mov     eax, ecx
        shr     eax, 15
        and     al, 31
        mov     BYTE PTR [edx+3], al
        mov     eax, ecx
        shr     eax, 20
        shr     ecx, 25
        and     al, 31
        and     cl, 31
        mov     BYTE PTR [edx+4], al
        mov     BYTE PTR [edx+5], cl
        ret     0
@Expand@8 ENDP

尝试的改进方案

后续尝试通过预填充掩码再逐位与的方式优化,代码如下:

void Expand(u32 x, u8 b[6]) {
    memset(b, 31, 6);
    b[0] &= x;
    b[1] &= x >>= 5;
    b[2] &= x >>= 5;
    b[3] &= x >>= 5;
    b[4] &= x >>= 5;
    b[5] &= x >>= 5;
}

对应的汇编输出:

@Expand@8 PROC
        mov     eax, 0x1f1f1f1f
        mov     DWORD PTR [edx], eax
        mov     WORD PTR [edx+4], ax
        and     BYTE PTR [edx], cl
        shr     ecx, 5
        and     BYTE PTR [edx+1], cl
        shr     ecx, 5
        and     BYTE PTR [edx+2], cl
        shr     ecx, 5
        and     BYTE PTR [edx+3], cl
        shr     ecx, 5
        and     BYTE PTR [edx+4], cl
        shr     ecx, 5
        and     BYTE PTR [edx+5], cl
        ret     0
@Expand@8 ENDP

高效优化方案(少于10条指令)

方案1:利用BMI2指令集的pdep(位分散)

如果目标CPU支持BMI2指令集(Intel Haswell及以上、AMD Excavator及以上),可以直接使用_pdep_u64 intrinsic,一步完成位分散:

#include <immintrin.h>

void Expand(u32 x, u8 b[6]) {
    uint64_t expanded = _pdep_u64((uint64_t)x, 0x1F1F1F1F1F1FULL);
    // 直接将64位结果写入缓冲区(需确保缓冲区至少8字节,或用memcpy复制前6字节)
    *(uint64_t*)b = expanded;
}

对应的汇编输出(MSVC):

@Expand@8 PROC
        mov     eax, ecx
        mov     rcx, 01F1F1F1F1F1Fh
        pdep    rax, rax, rcx
        mov     qword ptr [rdx], rax
        ret     0
@Expand@8 ENDP

仅4条核心指令,完全满足少于10条的要求。

方案2:无BMI2依赖的位操作魔术

如果需要兼容旧CPU,可以通过多步移位+掩码的方式实现,总指令数仍少于10条:

void Expand(u32 x, u8 b[6]) {
    uint64_t t = (uint64_t)x;
    t = (t | (t << 16)) & 0x000F800F80000ULL;
    t = (t | (t << 8)) & 0x001F001F001F0ULL;
    t = (t | (t << 4)) & 0x1F001F001F001FULL;
    t = (t | (t << 2)) & 0x1F1F1F1F1F1FULL;
    *(uint64_t*)b = t;
}

对应的汇编输出(MSVC):

@Expand@8 PROC
        mov     eax, ecx
        mov     rcx, rax
        shl     rcx, 16
        or      rax, rcx
        and     rax, 000F800F80000h
        mov     rcx, rax
        shl     rcx, 8
        or      rax, rcx
        and     rax, 001F001F001F0h
        mov     rcx, rax
        shl     rcx, 4
        or      rax, rcx
        and     rax, 1F001F001F001Fh
        mov     rcx, rax
        shl     rcx, 2
        or      rax, rcx
        and     rax, 1F1F1F1F1F1Fh
        mov     qword ptr [rdx], rax
        ret     0
@Expand@8 ENDP

核心操作共9条指令,符合要求。

内容的提问来源于stack exchange,提问作者MAP

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.18 20:25:29