You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

浮点除法延迟低于整数除法?下溢时延迟暴增原因探究

指令延迟测试疑问解答

问题背景

在2.3GHz DigitalOcean vCPU(Intel Broadwell架构,型号79)上测试指令延迟:

  • 64位整数除法每次约耗时25个周期
  • double浮点除法每次约10个周期,尽管double的有效小数位为52位,除法周期却不到整数除法的一半
  • 浮点除法发生下溢时,延迟从常规10周期升至100-140周期,但乘法即使出现inf也无延迟变化

编译选项:-fno-tree-reassoc -O2


一、整数除法与浮点除法的延迟差异原因

1. 硬件实现逻辑不同

Intel Broadwell架构中,整数除法单元采用迭代减法试商的方式计算,每次迭代仅能处理1-2位商,64位除法需要多次迭代才能完成,官方手册标注IDIV指令延迟为22-29周期,与测试结果吻合。

而浮点除法单元采用牛顿-拉夫逊迭代法,通过乘法和加法逼近除法结果,现代CPU的浮点单元(FPU)专门针对这种迭代计算做了硬件流水线优化,Broadwell的DIVSD指令延迟为10-13周期,因此耗时远低于整数除法。

2. 数据表示的天然优势

double类型的浮点值(除非规格化数外)都是规范化的,被除数和除数可通过指数调整到相同范围,简化了初始计算步骤;而整数除法需要处理任意范围的输入,包括被除数远大于/小于除数的边界情况,硬件需要额外逻辑处理这些场景,进一步增加了延迟。

3. 硬件资源分配优先级

浮点运算在科学计算、多媒体等场景中使用频率极高,CPU厂商会投入更多晶体管优化浮点除法性能;而整数除法的使用场景相对有限,硬件资源分配优先级较低,因此延迟更高。


二、浮点下溢时除法延迟飙升,乘法无变化的原因

1. 下溢的特殊处理逻辑

当浮点除法结果小于double类型的最小规范化数(约2.2e-308)时,会进入**非规格化数(Denormal)**范围。非规格化数的隐含位为0而非1,计算时需要额外调整尾数的移位和对齐,触发FPU的特殊处理流程,导致延迟大幅增加。

而乘法产生inf时,inf是IEEE 754定义的标准特殊值,硬件可通过简单位操作直接生成结果,无需额外迭代或调整,因此延迟与常规乘法一致。

2. 硬件设计的流水线差异

浮点乘法单元对特殊值(inf、NaN)的处理与常规乘法共享大部分流水线逻辑;而除法单元处理非规格化数时,需要跳出常规的牛顿迭代流程,启用专门的处理路径,这条路径的延迟远高于常规除法。


测试代码

整数除法测试代码

#include <stdint.h>
#include <stdio.h>
#include <time.h>       // time()
#include "timecounters.h"

static const int kIterations = 100 * 1000000;

int main (int argc, const char** argv) {
  uint64_t res = 1;

  time_t t = time(NULL);    // 编译器无法预知的数值
  volatile int incr0 = t & 255;     // 未知增量 0..255
  t = time(NULL);   // 编译器无法预知的数值
  volatile int incr1 = t & 255;     // 未知增量 0..255
  t = time(NULL);   // 编译器无法预知的数值
  volatile int incr2 = t & 255;     // 未知增量 0..255
  t = time(NULL);   // 编译器无法预知的数值
  volatile int incr3 = t & 255;     // 未知增量 0..255
  t = time(NULL);   // 编译器无法预知的数值
  volatile int incr4 = t & 255;     // 未知增量 0..255

  int64_t startcy = GetCycles();
  for (int i = 0; i < kIterations; ++i) {
    res /= incr0;
    res /= incr1;
    res /= incr2;
    res /= incr3;
  }

  int64_t elapsed = GetCycles() - startcy;
  double felapsed = elapsed;

  int64_t another_startcy = GetCycles();
  for (int i = 0; i < kIterations; ++i) {
    res /= incr0;
    res /= incr1;
    res /= incr2;
    res /= incr3;
    res /= incr4;
  }

  int64_t another_elapsed = GetCycles() - another_startcy;
  double another_felapsed = another_elapsed;

  fprintf(stdout, "%d iterations, %lu cycles, %4.2f cycles/iteration\n",
          kIterations, elapsed, (another_felapsed - felapsed) / kIterations);

  fprintf(stdout, "%lu %lu\n", t, res); // 确保res不被编译器优化掉

  return 0;
}

浮点除法测试代码

#include <stdint.h>
#include <stdio.h>
#include <time.h>       // time()
#include "timecounters.h"

static const int kIterations = 100 * 1000000;

int main (int argc, const char** argv) {
  double res = 2;

  time_t t = time(NULL);    // 编译器无法预知的数值
  volatile double incr0 = 1 + 1.0 / (t & 255) / 1000000;
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr1 = 1 + 1.0 / (t & 255) / 1000000;
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr2 = 1 + 1.0 / (t & 255) / 1000000;
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr3 = 1 + 1.0 / (t & 255) / 1000000;
  fprintf(stdout, "incr3 is %4.2f\n", incr3);
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr4 = 1 + 1.0 / (t & 255) / 1000000;

  int64_t startcy = GetCycles();
  for (int i = 0; i < kIterations; ++i) {
    res /= incr0;
    res /= incr1;
    res /= incr2;
    res /= incr3;
  }

  int64_t elapsed = GetCycles() - startcy;
  double felapsed = elapsed;

  int64_t another_startcy = GetCycles();
  for (int i = 0; i < kIterations; ++i) {
    res /= incr0;
    res /= incr1;
    res /= incr2;
    res /= incr3;
    res /= incr4;
  }

  int64_t another_elapsed = GetCycles() - another_startcy;
  double another_felapsed = another_elapsed;

  fprintf(stdout, "%d iterations, %lu cycles, %4.2f cycles/iteration\n",
          kIterations, elapsed, (another_felapsed - felapsed) / kIterations);

  fprintf(stdout, "%lu %4.2f\n", t, res);   // 确保res不被编译器优化掉

  return 0;
}

浮点下溢测试代码

#include <stdint.h>
#include <stdio.h>
#include <time.h>       // time()
#include "timecounters.h"


static const int outerIter = 20;
static const int kIterations = 10000; // 100 * 1000000;

int main (int argc, const char** argv) {
  double res = 1;

  time_t t = time(NULL);    // 编译器无法预知的数值
  volatile double incr0 = 1 + 1.0 / (t & 255) / 10;
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr1 = 1 + 1.0 / (t & 255) / 10;
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr2 = 1 + 1.0 / (t & 255) / 10;
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr3 = 1 + 1.0 / (t & 255) / 10;
  fprintf(stdout, "incr3 is %10.10f\n", incr3);
  t = time(NULL);   // 编译器无法预知的数值
  volatile double incr4 = 1 + 1.0 / (t & 255) / 10;

  for (int j = 0; j < outerIter; ++j) {
    int64_t startcy = GetCycles();
    for (int i = 0; i < kIterations; ++i) {
      res /= incr0;
      res /= incr1;
      res /= incr2;
      res /= incr3;
    }

    int64_t elapsed = GetCycles() - startcy;
    double felapsed = elapsed;

    int64_t another_startcy = GetCycles();
    for (int i = 0; i < kIterations; ++i) {
      res /= incr0;
      res /= incr1;
      res /= incr2;
      res /= incr3;
      res /= incr4;
    }

    int64_t another_elapsed = GetCycles() - another_startcy;
    double another_felapsed = another_elapsed;

    fprintf(stdout, "%d iterations, %lu cycles, %4.2f cycles/iteration\n",
            kIterations, elapsed, (another_felapsed - felapsed) / kIterations);

    fprintf(stdout, "t is %lu; res is %4.310f\n", t, res);  // 确保res不被编译器优化掉
  }

  return 0;
}

CPU信息

processor : 0
vendor_id : GenuineIntel
cpu family : 6
model : 79
model name : DO-Regular
stepping : 1
microcode : 0x1
cpu MHz : 2294.608
cache size : 4096 KB

内容的提问来源于stack exchange,提问作者Zack Light

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.12 11:45:56