ARM Cortex-A7内联汇编实现点积遇段错误,求修正方案
ARM Cortex-A7 内联汇编点积段错误问题修复与分析
原始代码(含问题汇编实现)
#include <stdio.h> #include <arm_neon.h> #define ARRAY_SIZE 1024 float dot_product_c(const float *a, const float *b, int n) { float sum = 0.0f; for (int i = 0; i < n; i++) { sum += a[i] * b[i]; } return sum; } float dot_product_intrinsics(const float *a, const float *b, int n) { float32x4_t sum_vec = vdupq_n_f32(0.0f); int i; for (i = 0; i <= n - 4; i += 4) { float32x4_t a_vec = vld1q_f32(a + i); float32x4_t b_vec = vld1q_f32(b + i); sum_vec = vmlaq_f32(sum_vec, a_vec, b_vec); } float sum = vaddvq_f32(sum_vec); for (; i < n; i++) { sum += a[i] * b[i]; } return sum; } float dot_product_asm(const float *a, const float *b, int n) { float result; __asm__ volatile ( "vpush {d8}\n" // 仅保存被使用的被调用者寄存器 "mov r3, #0\n" // 初始化循环计数器 "vmov.32 s16, #0\n" // 修正:初始化累加寄存器为0 "loop:\n" "cmp r3, %2\n" "beq end_loop\n" "vldr.32 s0, [%0, r3, lsl #2]\n" "vldr.32 s1, [%1, r3, lsl #2]\n" "vmul.f32 s2, s0, s1\n" "vadd.f32 s16, s16, s2\n" "add r3, r3, #1\n" "b loop\n" "end_loop:\n" "vmov.32 %3, s16\n" "vpop {d8}\n" // 恢复被保存的寄存器 : "=r"(result) : "r"(a), "r"(n), "r"(b) : "r3", "memory", "cc", "s0", "s1", "s2", "s16" ); return result; } int main() { float a[ARRAY_SIZE], b[ARRAY_SIZE]; for (int i = 0; i < ARRAY_SIZE; i++) { a[i] = (float)i; b[i] = (float)(i * 2); } float res_c = dot_product_c(a, b, ARRAY_SIZE); float res_intr = dot_product_intrinsics(a, b, ARRAY_SIZE); float res_asm = dot_product_asm(a, b, ARRAY_SIZE); printf("C: %.2f\n", res_c); printf("Intrinsics: %.2f\n", res_intr); printf("Assembly: %.2f\n", res_asm); return 0; }
编译指令
arm-linux-gnueabihf-gcc -mfpu=neon-vfpv4 -march=armv7-a -O0 -o dotprod dotprod.c
问题分析与修复
1. vstr.32 s16, [r10] 段错误的根源与修正
原语句试图将s16的值写入r10指向的内存,但**r10未被初始化或绑定有效内存地址**,导致访问非法内存触发段错误。移除该语句后,s16未被初始化,结果为0是未定义行为下的巧合。
正确做法是直接初始化累加寄存器s16为0,将原错误语句替换为:
vmov.32 s16, #0
这条指令直接将s16清零,为后续的累加操作提供正确的初始值。
2. vpush/vpop 的必要性确认
在ARMv7-A(Cortex-A7)的EABI调用约定中:
- NEON寄存器
d0-d7属于调用者保存寄存器:函数调用时会被调用者自动保存,无需手动处理 - NEON寄存器
d8-d15属于被调用者保存寄存器:如果函数(这里是内联汇编块)使用了这些寄存器,必须手动保存并恢复,否则会破坏调用者的上下文数据
你的汇编代码中使用了s16(对应d8的低32位),属于d8寄存器组,因此必须用vpush/vpop保存和恢复。原代码中保存d0-d15过于冗余,只需保存被使用的d8即可,优化后能减少寄存器保存/恢复的开销,同时符合调用约定。
内容的提问来源于stack exchange,提问作者Fabrice Auzanneau
相关产品推荐
相关产品推荐

