You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Apple M1的Aarch64汇编中使用w寄存器执行LDP为何崩溃?

AArch64汇编中使用w寄存器执行LDP/STP指令触发EXC_BAD_ACCESS的原因与修复

在Apple M1的AArch64汇编环境中,使用32位w寄存器配合LDP/STP指令时会触发EXC_BAD_ACCESS错误,但改用64位x寄存器则能正常运行。尽管ARM官方文档明确A64架构支持LDP搭配w寄存器使用,问题依然存在,以下是具体分析与修复方案:

可正常运行的x寄存器示例代码

.global _start

_start:
    mov x0, #1 // arg1
    mov x1, #2 // arg2

    stp x0, x1, [sp, #-16]! // push these values to the stack before we branch to another location.

    bl add_nums
    mov x2, x0   // save x0 to w2, so we don't lose once we restore the original x0 from stack

    ldp x0, x1, [sp], #16 // pop off x0, x1 with the values they were before going into add_numbs

    // Exit program
    mov x0, 0       // 0 status code
    mov x16, 1
    svc 0

add_nums:
    // stores the result back to x0, that is why we need to store the original
    // x0 on _start into the stack, so we could restore it later.
    add x0, x0, x1
    ret

触发错误的w寄存器示例代码

_start:
    mov w0, #1 // arg1
    mov w2, #2 // arg2

    stp w0, w2, [sp, #-8]! // push these values to the stack before we branch to another location.

    bl add_nums
    mov w2, w0   // save w0 to w2, so we don't lose once we restore the original w0 from stack

    ldp w0, w2, [sp], #8 // pop off w0, w2 with the values they were before going into add_numbs

    // Exit program
    mov x0, 0       // 0 status code
    mov x16, 1
    svc 0

add_nums:
    // stores the result back to w0, that is why we need to store the original
    // w0 on _start into the stack, so we could restore it later.
    add w0, w0, w2
    ret

错误信息

thread #1, queue = 'com.apple.main-thread', stop reason = EXC_BAD_ACCESS (code=259, address=0x16fdfe948)

错误原因分析

问题核心在于AArch64的栈对齐强制要求:

  1. AArch64架构规定栈指针sp必须始终保持16字节对齐。调用bl指令时,硬件会自动将8字节的返回地址压入栈,此时栈指针会减8。如果在bl前手动执行[sp, #-8]!让栈指针再减8,栈指针会变成8字节对齐,直接违反架构规则,触发内存访问错误。
  2. 虽然LDP/STP支持操作32位w寄存器,但stp w0, w2, [sp, #-8]!仅将两个32位值打包成8字节写入栈,导致栈指针偏离16字节对齐边界。而使用x寄存器时,stp x0, x1, [sp, #-16]!的操作保持了栈的16字节对齐,因此无问题。

修复方案

要解决这个问题,必须保证栈指针始终维持16字节对齐,即使操作的是32位寄存器:

方案1:保持LDP/STP操作,调整栈偏移量为16字节

.global _start

_start:
    mov w0, #1 // arg1
    mov w2, #2 // arg2

    // 栈指针减16保持16字节对齐,即使只存储两个32位值
    stp w0, w2, [sp, #-16]! 

    bl add_nums
    mov w2, w0   // 保存计算结果

    // 恢复栈指针到16字节对齐位置
    ldp w0, w2, [sp], #16 

    // 退出程序
    mov x0, 0       // 0状态码
    mov x16, 1
    svc 0

add_nums:
    add w0, w0, w2
    ret

方案2:改用单个STR/LDR指令操作32位寄存器

.global _start

_start:
    mov w0, #1 // arg1
    mov w2, #2 // arg2

    // 分别存储两个32位值,栈指针减16保持对齐
    str w0, [sp, #-16]!
    str w2, [sp, #-4]!

    bl add_nums
    mov w2, w0   // 保存计算结果

    // 恢复寄存器值和栈指针
    ldr w2, [sp], #4
    ldr w0, [sp], #16

    // 退出程序
    mov x0, 0       // 0状态码
    mov x16, 1
    svc 0

add_nums:
    add w0, w0, w2
    ret

内容的提问来源于stack exchange,提问作者maxcnunes

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 17:56:03