浮点除法延迟低于整数除法?下溢时延迟暴增原因探究
问题背景
在2.3GHz DigitalOcean vCPU(Intel Broadwell架构,型号79)上测试指令延迟:
- 64位整数除法每次约耗时25个周期
- double浮点除法每次约10个周期,尽管double的有效小数位为52位,除法周期却不到整数除法的一半
- 浮点除法发生下溢时,延迟从常规10周期升至100-140周期,但乘法即使出现inf也无延迟变化
编译选项:-fno-tree-reassoc -O2
一、整数除法与浮点除法的延迟差异原因
1. 硬件实现逻辑不同
Intel Broadwell架构中,整数除法单元采用迭代减法试商的方式计算,每次迭代仅能处理1-2位商,64位除法需要多次迭代才能完成,官方手册标注IDIV指令延迟为22-29周期,与测试结果吻合。
而浮点除法单元采用牛顿-拉夫逊迭代法,通过乘法和加法逼近除法结果,现代CPU的浮点单元(FPU)专门针对这种迭代计算做了硬件流水线优化,Broadwell的DIVSD指令延迟为10-13周期,因此耗时远低于整数除法。
2. 数据表示的天然优势
double类型的浮点值(除非规格化数外)都是规范化的,被除数和除数可通过指数调整到相同范围,简化了初始计算步骤;而整数除法需要处理任意范围的输入,包括被除数远大于/小于除数的边界情况,硬件需要额外逻辑处理这些场景,进一步增加了延迟。
3. 硬件资源分配优先级
浮点运算在科学计算、多媒体等场景中使用频率极高,CPU厂商会投入更多晶体管优化浮点除法性能;而整数除法的使用场景相对有限,硬件资源分配优先级较低,因此延迟更高。
二、浮点下溢时除法延迟飙升,乘法无变化的原因
1. 下溢的特殊处理逻辑
当浮点除法结果小于double类型的最小规范化数(约2.2e-308)时,会进入**非规格化数(Denormal)**范围。非规格化数的隐含位为0而非1,计算时需要额外调整尾数的移位和对齐,触发FPU的特殊处理流程,导致延迟大幅增加。
而乘法产生inf时,inf是IEEE 754定义的标准特殊值,硬件可通过简单位操作直接生成结果,无需额外迭代或调整,因此延迟与常规乘法一致。
2. 硬件设计的流水线差异
浮点乘法单元对特殊值(inf、NaN)的处理与常规乘法共享大部分流水线逻辑;而除法单元处理非规格化数时,需要跳出常规的牛顿迭代流程,启用专门的处理路径,这条路径的延迟远高于常规除法。
测试代码
整数除法测试代码
#include <stdint.h> #include <stdio.h> #include <time.h> // time() #include "timecounters.h" static const int kIterations = 100 * 1000000; int main (int argc, const char** argv) { uint64_t res = 1; time_t t = time(NULL); // 编译器无法预知的数值 volatile int incr0 = t & 255; // 未知增量 0..255 t = time(NULL); // 编译器无法预知的数值 volatile int incr1 = t & 255; // 未知增量 0..255 t = time(NULL); // 编译器无法预知的数值 volatile int incr2 = t & 255; // 未知增量 0..255 t = time(NULL); // 编译器无法预知的数值 volatile int incr3 = t & 255; // 未知增量 0..255 t = time(NULL); // 编译器无法预知的数值 volatile int incr4 = t & 255; // 未知增量 0..255 int64_t startcy = GetCycles(); for (int i = 0; i < kIterations; ++i) { res /= incr0; res /= incr1; res /= incr2; res /= incr3; } int64_t elapsed = GetCycles() - startcy; double felapsed = elapsed; int64_t another_startcy = GetCycles(); for (int i = 0; i < kIterations; ++i) { res /= incr0; res /= incr1; res /= incr2; res /= incr3; res /= incr4; } int64_t another_elapsed = GetCycles() - another_startcy; double another_felapsed = another_elapsed; fprintf(stdout, "%d iterations, %lu cycles, %4.2f cycles/iteration\n", kIterations, elapsed, (another_felapsed - felapsed) / kIterations); fprintf(stdout, "%lu %lu\n", t, res); // 确保res不被编译器优化掉 return 0; }
浮点除法测试代码
#include <stdint.h> #include <stdio.h> #include <time.h> // time() #include "timecounters.h" static const int kIterations = 100 * 1000000; int main (int argc, const char** argv) { double res = 2; time_t t = time(NULL); // 编译器无法预知的数值 volatile double incr0 = 1 + 1.0 / (t & 255) / 1000000; t = time(NULL); // 编译器无法预知的数值 volatile double incr1 = 1 + 1.0 / (t & 255) / 1000000; t = time(NULL); // 编译器无法预知的数值 volatile double incr2 = 1 + 1.0 / (t & 255) / 1000000; t = time(NULL); // 编译器无法预知的数值 volatile double incr3 = 1 + 1.0 / (t & 255) / 1000000; fprintf(stdout, "incr3 is %4.2f\n", incr3); t = time(NULL); // 编译器无法预知的数值 volatile double incr4 = 1 + 1.0 / (t & 255) / 1000000; int64_t startcy = GetCycles(); for (int i = 0; i < kIterations; ++i) { res /= incr0; res /= incr1; res /= incr2; res /= incr3; } int64_t elapsed = GetCycles() - startcy; double felapsed = elapsed; int64_t another_startcy = GetCycles(); for (int i = 0; i < kIterations; ++i) { res /= incr0; res /= incr1; res /= incr2; res /= incr3; res /= incr4; } int64_t another_elapsed = GetCycles() - another_startcy; double another_felapsed = another_elapsed; fprintf(stdout, "%d iterations, %lu cycles, %4.2f cycles/iteration\n", kIterations, elapsed, (another_felapsed - felapsed) / kIterations); fprintf(stdout, "%lu %4.2f\n", t, res); // 确保res不被编译器优化掉 return 0; }
浮点下溢测试代码
#include <stdint.h> #include <stdio.h> #include <time.h> // time() #include "timecounters.h" static const int outerIter = 20; static const int kIterations = 10000; // 100 * 1000000; int main (int argc, const char** argv) { double res = 1; time_t t = time(NULL); // 编译器无法预知的数值 volatile double incr0 = 1 + 1.0 / (t & 255) / 10; t = time(NULL); // 编译器无法预知的数值 volatile double incr1 = 1 + 1.0 / (t & 255) / 10; t = time(NULL); // 编译器无法预知的数值 volatile double incr2 = 1 + 1.0 / (t & 255) / 10; t = time(NULL); // 编译器无法预知的数值 volatile double incr3 = 1 + 1.0 / (t & 255) / 10; fprintf(stdout, "incr3 is %10.10f\n", incr3); t = time(NULL); // 编译器无法预知的数值 volatile double incr4 = 1 + 1.0 / (t & 255) / 10; for (int j = 0; j < outerIter; ++j) { int64_t startcy = GetCycles(); for (int i = 0; i < kIterations; ++i) { res /= incr0; res /= incr1; res /= incr2; res /= incr3; } int64_t elapsed = GetCycles() - startcy; double felapsed = elapsed; int64_t another_startcy = GetCycles(); for (int i = 0; i < kIterations; ++i) { res /= incr0; res /= incr1; res /= incr2; res /= incr3; res /= incr4; } int64_t another_elapsed = GetCycles() - another_startcy; double another_felapsed = another_elapsed; fprintf(stdout, "%d iterations, %lu cycles, %4.2f cycles/iteration\n", kIterations, elapsed, (another_felapsed - felapsed) / kIterations); fprintf(stdout, "t is %lu; res is %4.310f\n", t, res); // 确保res不被编译器优化掉 } return 0; }
CPU信息
processor : 0
vendor_id : GenuineIntel
cpu family : 6
model : 79
model name : DO-Regular
stepping : 1
microcode : 0x1
cpu MHz : 2294.608
cache size : 4096 KB
内容的提问来源于stack exchange,提问作者Zack Light

