You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

关于x86汇编calc_integ函数第27行起栈操作的疑惑求助

解析calc_integ函数汇编中的栈操作与FPU逻辑

原始代码

C代码

long double calc_integ(unsigned long N) {
    int a = 0;
    long double integral = 0;
    long double b = M_PI;
    long double step = (b - a) / N;
    long double point = a;
    for (size_t k = 0; k < N; k++) {
        integral += sinl(point)*expl(point) + sinl(point + step)*expl(point + step);
        point += step;
    }
    integral *= step;
    integral /= 2;
    return integral;
}

int main() {
    struct timespec sts, stf;
    unsigned long int N = 0;
    scanf("%ld", &N);
    clock_gettime(CLOCK_REALTIME, &sts);
    printf("%lf", calc_integ(N));
    clock_gettime(CLOCK_REALTIME, &stf);
    printf("Program execution time: %lf ", stf.tv_sec - sts.tv_sec + 0.000000001 * (stf.tv_nsec - sts.tv_nsec));
    return 0;
}

汇编代码(重点为.L4段)

calc_integ:
            pushq   %rbp
            movq    %rsp, %rbp
            addq    $-128, %rsp
            movq    %rdi, -88(%rbp)
            movl    $0, -44(%rbp)
            fldz
            fstpt   -16(%rbp)
            fldt    .LC1(%rip)
            fstpt   -64(%rbp)
            fildl   -44(%rbp)
            fldt    -64(%rbp)
            fsubp   %st, %st(1)
            fildq   -88(%rbp)
            cmpq    $0, -88(%rbp)
            jns     .L2
            fldt    .LC2(%rip)
            faddp   %st, %st(1)
    .L2:
            fdivrp  %st, %st(1)
            fstpt   -80(%rbp)
            fildl   -44(%rbp)
            fstpt   -32(%rbp)
            movq    $0, -40(%rbp)
            jmp     .L3
    .L4:
            pushq   -24(%rbp) ; 此处为疑惑起始位置
            pushq   -32(%rbp)
            call    sinl
            addq    $16, %rsp
            fstpt   -112(%rbp)
            pushq   -24(%rbp)
            pushq   -32(%rbp)
            call    expl
            addq    $16, %rsp
            fldt    -112(%rbp)
            fmulp   %st, %st(1)
            fstpt   -112(%rbp)
            fldt    -32(%rbp)
            fldt    -80(%rbp)
            faddp   %st, %st(1)
            leaq    -16(%rsp), %rsp
            fstpt   (%rsp)
            call    sinl
            addq    $16, %rsp
            fstpt   -128(%rbp)
            fldt    -32(%rbp)
            fldt    -80(%rbp)
            faddp   %st, %st(1)
            leaq    -16(%rsp), %rsp
            fstpt   (%rsp)
            call    expl
            addq    $16, %rsp
            fldt    -128(%rbp)
            fmulp   %st, %st(1)
            fldt    -112(%rbp)
            faddp   %st, %st(1)
            fldt    -16(%rbp)
            faddp   %st, %st(1)
            fstpt   -16(%rbp)
            fldt    -32(%rbp)
            fldt    -80(%rbp)
            faddp   %st, %st(1)
            fstpt   -32(%rbp)
            addq    $1, -40(%rbp)
    .L3:
            movq    -40(%rbp), %rax
            cmpq    -88(%rbp), %rax
            jb      .L4
            fldt    -16(%rbp)
            fldt    -80(%rbp)
            fmulp   %st, %st(1)
            fstpt   -16(%rbp)
            fldt    -16(%rbp)
            fldt    .LC3(%rip)
            fdivrp  %st, %st(1)
            fstpt   -16(%rbp)
            fldt    -16(%rbp)
            leave
            ret

用户的疑问与推理

  • 将-24(%rbp)的值压入栈顶,但根据栈结构分析,该地址未存储有效数据,应为垃圾值;
  • 随后将point的值压入栈顶;
  • 调用sinl函数,猜测其返回值存储在FPU中,但不清楚栈操作逻辑,也无法理解后续sinl与expl的返回值如何在FPU中完成乘法操作。

解析细节

1. pushq -24(%rbp)的作用:栈对齐

x86-64 System V ABI要求调用函数时,栈指针%rsp必须是16字节对齐的。进入.L4段时,当前%rsp的位置不满足对齐要求,编译器会先压入一个8字节的无关数据(-24(%rbp)属于栈上预留的未使用填充空间,值不影响),让栈满足16字节对齐,之后再压入真正的函数参数(point的long double值,在栈上占16字节,拆成两次pushq操作)。调用完函数后,addq $16, %rsp会把这两个压栈的内容一起清理,恢复栈指针。

2. 函数调用与返回值处理

  • sinl和expl都是接收long double参数、返回long double的函数,它们的返回值会存在x87 FPU的栈顶寄存器%st(0)中。
  • 调用sinl后,fstpt -112(%rbp)是把%st(0)中的返回值(即sin(point))保存到栈上的-112(%rbp)位置,方便后续使用。

3. FPU中的乘法与加法逻辑

  • 调用expl(point)后,返回值exp(point)在%st(0),此时执行fldt -112(%rbp)把之前保存的sin(point)加载到FPU栈顶,此时FPU栈结构为%st(0)=sin(point),%st(1)=exp(point)。
  • fmulp %st, %st(1)指令会执行%st(1) = %st(1) * %st(0),然后弹出栈顶,最终%st(0)中就是sin(point)*exp(point)的结果,再用fstpt -112(%rbp)把这个乘积保存起来。
  • 处理sin(point+step)*exp(point+step)的逻辑类似:先通过FPU计算point+step,把结果存到栈上作为参数调用sinl和expl,再完成乘法,最后把两个乘积相加,加到integral(存储在-16(%rbp))中。

4. 循环收尾

  • fldt -32(%rbp)加载当前point,fldt -80(%rbp)加载step,faddp完成point += step的计算,再用fstpt -32(%rbp)更新栈上的point值。
  • addq $1, -40(%rbp)是循环变量k自增,之后判断k < N,满足则回到.L4继续循环。

内容的提问来源于stack exchange,提问作者Anastasia

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.16 23:42:33