You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用GCC编译时std::execution未提升排序性能的原因排查

问题:GCC编译下std::execution::par未提升排序性能的原因

我在Windows 10环境下编写了测试代码,验证std::execution库对排序性能的提升效果,但发现GCC编译版本的并行排序没有性能提升,而MSVC版本效果显著,想请教原因。


测试代码

#include <stddef.h>
#include <stdio.h>

#include <algorithm>
#include <chrono>
#include <execution>
#include <random>
#include <ratio>
#include <vector>

using std::milli;
using std::random_device;
using std::sort;
using std::vector;
using std::chrono::duration;
using std::chrono::duration_cast;
using std::chrono::high_resolution_clock;

const size_t testSize = 1'000'000;
const int iterationCount = 5;

void print_results(                                 //
    const char* const tag,                          //
    const vector<double>& sorted,                   //
    high_resolution_clock::time_point startTime,    //
    high_resolution_clock::time_point endTime
    //
)
{
    printf("%s: Lowest: %g Highest: %g Time: %f ms\n", tag, sorted.front(), sorted.back(),
           duration_cast<duration<double, milli>>(endTime - startTime).count());
}

int main()
{
    random_device rd;

    printf("Testing with %llu doubles...\n", testSize);
    vector<double> doubles(testSize);
    for (auto& d : doubles)
    {
        d = static_cast<double>(rd());
    }

    for (size_t i = 0; i < iterationCount; ++i)
    {
        vector<double> sorted(doubles);
        const auto startTime = high_resolution_clock::now();
        sort(sorted.begin(), sorted.end());
        const auto endTime = high_resolution_clock::now();
        print_results("Serial STL", sorted, startTime, endTime);
    }

    for (size_t i = 0; i < iterationCount; ++i)
    {
        vector<double> sorted(doubles);
        const auto startTime = high_resolution_clock::now();
        std::sort(std::execution::par, sorted.begin(), sorted.end());
        const auto endTime = high_resolution_clock::now();
        print_results("Parallel STL", sorted, startTime, endTime);
    }
    return 0;
}

CMakeLists.txt配置

cmake_minimum_required(VERSION 3.14.0)
project(EXEC VERSION 0.0.1)

set(CMAKE_C_STANDARD 17)
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_CXX_STANDARD_REQUIRED ON)

add_executable(
    executionTests
    targets/executionTests.cpp
)

if(CMAKE_CXX_COMPILER_ID MATCHES "GNU")
    target_compile_options(
        executionTests
        PRIVATE
        -O3
    )
elseif(CMAKE_CXX_COMPILER_ID MATCHES "MSVC")
    STRING(REGEX REPLACE "/RTC(su|[1su])" "" CMAKE_CXX_FLAGS_DEBUG "${CMAKE_CXX_FLAGS_DEBUG}")
    STRING(REGEX REPLACE "/RTC(su|[1su])" "" CMAKE_C_FLAGS_DEBUG "${CMAKE_C_FLAGS_DEBUG}")
    target_compile_options(
        executionTests
        PRIVATE
        /O2
    )
endif()

构建脚本

# Set-Location build ; cmake .. -DCMAKE_BUILD_TYPE=Debug -G Ninja ; Set-Location ..
Set-Location build ; cmake .. -DCMAKE_BUILD_TYPE=Debug -G "Visual Studio 17 2022" ; Set-Location ..

cmake --build build --target executionTests -j 8 -v

运行结果

GCC 13.1.0(Ninja生成器)运行结果

Testing with 1000000 doubles...
Serial STL: Lowest: 9059 Highest: 4.29496e+09 Time: 75.064000 ms
Serial STL: Lowest: 9059 Highest: 4.29496e+09 Time: 78.308300 ms
Serial STL: Lowest: 9059 Highest: 4.29496e+09 Time: 77.079100 ms
Serial STL: Lowest: 9059 Highest: 4.29496e+09 Time: 77.511300 ms
Serial STL: Lowest: 9059 Highest: 4.29496e+09 Time: 76.836500 ms
Parallel STL: Lowest: 9059 Highest: 4.29496e+09 Time: 77.417900 ms
Parallel STL: Lowest: 9059 Highest: 4.29496e+09 Time: 77.452600 ms
Parallel STL: Lowest: 9059 Highest: 4.29496e+09 Time: 78.962000 ms
Parallel STL: Lowest: 9059 Highest: 4.29496e+09 Time: 80.188500 ms
Parallel STL: Lowest: 9059 Highest: 4.29496e+09 Time: 79.135000 ms

MSVC 2022运行结果

Testing with 1000000 doubles...
Serial STL: Lowest: 5059 Highest: 4.29497e+09 Time: 256.872900 ms
Serial STL: Lowest: 5059 Highest: 4.29497e+09 Time: 264.764000 ms
Serial STL: Lowest: 5059 Highest: 4.29497e+09 Time: 262.767800 ms
Serial STL: Lowest: 5059 Highest: 4.29497e+09 Time: 264.283300 ms
Serial STL: Lowest: 5059 Highest: 4.29497e+09 Time: 259.603600 ms
Parallel STL: Lowest: 5059 Highest: 4.29497e+09 Time: 86.583400 ms
Parallel STL: Lowest: 5059 Highest: 4.29497e+09 Time: 81.407500 ms
Parallel STL: Lowest: 5059 Highest: 4.29497e+09 Time: 81.962600 ms
Parallel STL: Lowest: 5059 Highest: 4.29497e+09 Time: 88.384000 ms
Parallel STL: Lowest: 5059 Highest: 4.29497e+09 Time: 84.420800 ms

编译链接日志

GCC编译链接日志

[1/2] L:\UCRT_GCC-13-1-0_x64\mingw64\bin\c++.exe   -g -O3 -std=gnu++20 -MD -MT CMakeFiles/executionTests.dir/targets/executionTests.cpp.obj -MF CMakeFiles\executionTests.dir\targets\executionTests.cpp.obj.d -o CMakeFiles/executionTests.dir/targets/executionTests.cpp.obj -c ${WorkspaceFolder}/targets/executionTests.cpp
[2/2] cmd.exe /C "cd . && L:\UCRT_GCC-13-1-0_x64\mingw64\bin\c++.exe -g  CMakeFiles/executionTests.dir/targets/executionTests.cpp.obj -o ..\${OutputDir}\executionTests.exe -Wl,--out-implib,..\${OutputDir}\libexecutionTests.dll.a -Wl,--major-image-version,0,--minor-image-version,0  -lkernel32 -luser32 -lgdi32 -lwinspool -lshell32 -lole32 -loleaut32 -luuid -lcomdlg32 -ladvapi32 && cd ."

MSVC编译链接日志

ClCompile:
     C:\Program Files\Microsoft Visual Studio\2022\Community\VC\Tools\MSVC\14.36.32532\bin\HostX64\x64\CL.exe /c /Zi /nologo /W3 /WX- /diagnostics:column /O2 /Ob0 /D _MBCS /D WIN32 /D _WINDOWS /D "CMAKE_INTDIR=\"Debug\"" /Gm- /EHsc /MDd /GS /fp:precise /Zc:wchar_t /Zc:forScope /Zc:inli
     ne /GR /std:c++20 /Fo"executionTests.dir\Debug\\" /Fd"executionTests.dir\Debug\vc143.pdb" /external:W3 /Gd /TP /errorReport:queue ${WorkspaceFolder}\targets\executionTests.cpp
     executionTests.cpp
   Link:
     C:\Program Files\Microsoft Visual Studio\2022\Community\VC\Tools\MSVC\14.36.32532\bin\HostX64\x64\link.exe /ERRORREPORT:QUEUE /OUT:"${OutputDir}\Debug\executionTests.exe" /INCREMENTAL /ILK:"executionTests
     .dir\Debug\executionTests.ilk" /NOLOGO kernel32.lib user32.lib gdi32.lib winspool.lib shell32.lib ole32.lib oleaut32.lib uuid.lib comdlg32.lib advapi32.lib /MANIFEST /MANIFESTUAC:"level='asInvoker' uiAccess='false'" /manifest:embed /DEBUG /PDB:"${OutputDir}/Debug/executionTests.pdb" /SUBSYSTEM:CONSOLE /TLBID:1 /DYNAMICBASE /NXCOMPAT /IMPLIB:"${OutputDir}/Debug/executionTests.lib" /MACHINE:X64  /machine:x64 
      executionTests.dir\Debug\executionTests.obj
     executionTests.vcxproj -> ${OutputDir}\Debug\executionTests.exe

核心疑问

为何GCC编译时std::execution::par并未带来排序性能提升?是GCC 13.1.0的STL在-O3优化下已足够高效,还是缺少必要的编译参数?


原因分析与解决方案

  • 缺少线程库链接参数:GCC的libstdc++实现的并行STL依赖POSIX线程库,必须通过-pthread编译和链接参数才能启用并行功能。你的CMake配置中仅添加了-O3,没有添加该参数,导致std::execution::par实际上退化为串行执行,因此性能和普通sort无差异。
  • 修改后的CMake配置:在GCC分支中补充-pthread参数:
if(CMAKE_CXX_COMPILER_ID MATCHES "GNU")
    target_compile_options(
        executionTests
        PRIVATE
        -O3
        -pthread
    )
    target_link_libraries(executionTests PRIVATE pthread)
endif()
  • 差异说明:MSVC的并行STL基于Windows原生线程池实现,默认无需额外链接库即可启用并行,因此能正常发挥多核性能。而GCC(包括MinGW版本)必须显式指定线程库参数,否则并行算法会自动降级为串行执行。

内容的提问来源于stack exchange,提问作者Dmytro Kovryzhenko

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 23:47:33