You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何8字节std::array<char>与std::byte比较生成的汇编不同?

std::array<char,8>与std::arraystd::byte,8比较的汇编差异原因分析

问题现象

我发现8字节std::array的比较生成的汇编与使用std::bit_cast的结果存在差异:

  • GCC对std::array<char,8>的处理符合预期,但Clang会多生成一条mov指令——将传值的array参数从8字节寄存器溢出到红区,但仍用寄存器参数与另一个参数指向的内存进行比较。
  • 针对std::byte的std::array比较,编译器会生成8次单独的单字节cmp指令;而std::array<char,8>的比较则会被优化为一次高效的qword(8字节)比较。

代码示例

#include <array>
#include <bit>
#include <cstdint>

// 生成的汇编与另外两个函数完全不同
bool compare1(const std::array<std::byte, 8> &p, std::array<std::byte, 8> r)
{
    return p == r;
}

// 汇编与bit_cast类似,但Clang多一条指令
bool compare2(const std::array<char, 8> &p, std::array<char, 8> r)
{
    return p == r;
}

// 将byte换成char时汇编相同
bool compare3(const std::array<std::byte, 8> &p, std::array<std::byte, 8> r)
{
    return std::bit_cast<uint64_t>(p) == std::bit_cast<uint64_t>(r);
}

汇编对比

Clang生成的汇编

compare1(std::array<std::byte, 8ul>, std::array<std::byte, 8ul>):    # @compare1(std::array<std::byte, 8ul>, std::array<std::byte, 8ul>)
        cmp     dil, sil
        sete    al
        jne     .LBB0_8
        mov     eax, edi
        shr     eax, 8
        mov     ecx, esi
        shr     ecx, 8
        cmp     al, cl
        sete    al
        jne     .LBB0_8
        mov     eax, edi
        shr     eax, 16
        mov     ecx, esi
        shr     ecx, 16
        cmp     al, cl
        sete    al
        jne     .LBB0_8
        mov     eax, edi
        shr     eax, 24
        mov     ecx, esi
        shr     ecx, 24
        cmp     al, cl
        sete    al
        jne     .LBB0_8
        mov     rax, rdi
        shr     rax, 32
        mov     rcx, rsi
        shr     rcx, 32
        cmp     al, cl
        sete    al
        jne     .LBB0_8
        mov     rax, rdi
        shr     rax, 40
        mov     rcx, rsi
        shr     rcx, 40
        cmp     al, cl
        sete    al
        jne     .LBB0_8
        mov     rax, rdi
        shr     rax, 48
        mov     rcx, rsi
        shr     rcx, 48
        cmp     al, cl
        sete    al
        jne     .LBB0_8
        xor     rdi, rsi
        shr     rdi, 56
        sete    al
.LBB0_8:
        ret
compare2(std::array<char, 8ul> const&, std::array<char, 8ul>):        # @compare2(std::array<char, 8ul> const&, std::array<char, 8ul>)
        mov     qword ptr [rsp - 8], rsi
        cmp     qword ptr [rdi], rsi
        sete    al
        ret
compare3(std::array<std::byte, 8ul> const&, std::array<std::byte, 8ul>):  # @compare3(std::array<std::byte, 8ul> const&, std::array<std::byte, 8ul>)
        cmp     qword ptr [rdi], rsi
        sete    al
        ret

GCC生成的汇编

compare1(std::array<std::byte, 8ul>, std::array<std::byte, 8ul>):
        mov     rdx, rdi
        mov     rax, rsi
        cmp     sil, dil
        jne     .L9
        movzx   ecx, ah
        cmp     dh, cl
        jne     .L9
        mov     rsi, rdi
        mov     rcx, rax
        shr     rsi, 16
        shr     rcx, 16
        cmp     sil, cl
        jne     .L9
        mov     rsi, rdi
        mov     rcx, rax
        shr     rsi, 24
        shr     rcx, 24
        cmp     sil, cl
        jne     .L9
        mov     rsi, rdi
        mov     rcx, rax
        shr     rsi, 32
        shr     rcx, 32
        cmp     sil, cl
        jne     .L9
        mov     rsi, rdi
        mov     rcx, rax
        shr     rsi, 40
        shr     rcx, 40
        cmp     sil, cl
        jne     .L9
        mov     rsi, rdi
        mov     rcx, rax
        shr     rsi, 48
        shr     rcx, 48
        cmp     sil, cl
        jne     .L9
        shr     rdx, 56
        shr     rax, 56
        cmp     dl, al
        sete    al
        ret
.L9:
        xor     eax, eax
        ret
compare2(std::array<char, 8ul> const&, std::array<char, 8ul>):
        cmp     QWORD PTR [rdi], rsi
        sete    al
        ret
compare3(std::array<std::byte, 8ul> const&, std::array<std::byte, 8ul>):
        cmp     QWORD PTR [rdi], rsi
        sete    al
        ret

差异原因分析

  1. std::byte与char的类型本质差异
    std::byte在C++标准中定义为强类型枚举:enum class byte : unsigned char {};,而char是基本整数类型。编译器对std::array的默认比较是逐元素进行的,对于char这种基本类型,编译器可以安全地将连续8个char的比较优化为一次8字节的整体比较——因为逐字节比较和整体比较的结果完全一致。但对于std::byte,编译器会严格遵循枚举类型的比较语义,认为不能直接将其当作无符号整数进行整体内存比较,因此只能生成逐字节的cmp指令。

  2. Clang的额外mov指令细节
    Clang在处理std::array<char,8>传值参数时生成的额外mov qword ptr [rsp - 8], rsi指令,是x86-64系统V调用约定中红区的典型使用场景——函数栈帧下方有128字节的红区,无需调整栈指针即可访问。这条指令是Clang优化策略或ABI处理的细节,并不会影响执行效率:后续比较依然使用寄存器rsi的值,红区的内存操作成本极低,不会带来性能损失。

内容的提问来源于stack exchange,提问作者phaile

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 19:29:59