You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Apple M1芯片分支开销实验结果对称,求技术原因解析

实验背景与疑问

我参考HPC分支流水线相关技术文档开展实验,该实验灵感源自Stack Overflow最高票问题《为什么处理排序后的数组比未排序的更快?》。原实验基于AMD Zen2芯片,我在Apple M1芯片上得到了对称的实验结果(见附图:M1芯片分支实验对称结果曲线)。

理论上当条件为真时,需要执行更多向s累加的指令,实验结果应呈不对称状态,但我将编译优化级别从-O3改为-O2、-O1及-O0后,结果仍保持对称。同时补充无分支实验结果(见附图:M1芯片无分支实验结果曲线),请问该现象的原因是什么?

实验代码

C++核心测试代码

#include <bits/stdc++.h>

#ifndef N
#define N 1'000'000
#endif

#ifndef T
#define T 1e8
#endif

#ifndef P
#define P 50
#endif

const int K = T / N;
int a[N];

int main()
{
    for (int i = 0; i < N; i++)
        a[i] = rand() % 100;

#ifdef SORT
    std::sort(a, a + N);
#endif

    clock_t start = clock();
    volatile int s = 0;

    for (int k = 0; k < K; k++)
        for (int i = 0; i < N; i++)
#ifdef CMOV
            s += (a[i] < P ? a[i] : 0);
#else
            // if (__builtin_expect(a[i] < P, false))// [[unlikely]]
            if (a[i] < P)
                s += a[i];
#endif

    float seconds = float(clock() - start) / CLOCKS_PER_SEC;
    float per_element = 1e9 * seconds / K / N;

    printf("%.4f %.4f %.4f\n", seconds, per_element, 2 * per_element);
    printf("%d\n", s);

    return 0;
}

Python绘图代码

import pickle
import matplotlib.pyplot as plt
import seaborn as sns

sns.reset_defaults()
sns.set_theme(style='whitegrid')

# benchmark
def bench(n=10**6, p=50, t=10**8, sort=False, cmov=False, unroll=False, cc='clang++'):
    res = !{cc} -std=c++17 -O3 -D N={n} -D T={t} -D P={p} {"-D CMOV" if cmov else ""} {"-D SORT" if sort else ""} {"-funroll-loops" if unroll else ""} branching.cc -o run && ./run
    print(res)
    return float(res[0].split()[-1])

ps = list(range(0, 101))
rs = [bench(p=p) for p in ps]

# plot
with open('ps.pkl', 'wb') as file:
    pickle.dump(rs, file)

with open('ps.pkl', 'rb') as file:
    rs = pickle.load(file)

plt.plot(ps, rs, c='darkred')

plt.xlabel('Branch probability (P)')
plt.ylabel('Cycles per iteration')

plt.title('for (int i = 0; i < N; i++) if (a[i] < P) s += a[i]', pad=12)

plt.ylim(bottom=0)
plt.margins(0)

fig = plt.gcf()
fig.savefig('branchy-vs-branchless.svg')
plt.show()

内容的提问来源于stack exchange,提问作者Bowen Smith

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 23:27:05