You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

OpenMP parallel for数组求和代码未提速反而性能更慢问题咨询

OpenMP数组求和加速测试异常问题

我正在测试OpenMP在数组求和程序中的加速效果。为避免编译器优化,使用随机数生成器生成数组元素,且设置了足够大的数组长度以体现性能差异。程序通过g++ -fopenmp -g -O0 -o main main.cpp编译,其中-g -O0用于禁用优化。但实际测试发现,OpenMP parallel for代码的运行速度显著慢于串行代码。

测试结果

线程数量为:12
填充数组
填充时间:66718888
正在运行OpenMP代码
2线程OpenMP耗时:11154095
结果: 4294903886
正在运行OpenMP代码
4线程OpenMP耗时:10832414
结果: 4294903886
正在运行OpenMP代码
6线程OpenMP耗时:11165054
结果: 4294903886
正在运行串行代码
串行耗时: 3525371
结果: 4294903886

测试代码

#include <iostream>
#include <stdio.h>
#include <omp.h>
#include <ctime>
#include <random>

using namespace std;

long long llsum(char *vec, size_t size, int threadCount) {
    long long result = 0;
    size_t i;
#pragma omp parallel for num_threads(threadCount) reduction(+: result) schedule(guided)
    for (i = 0; i < size; ++i) {
        result += vec[i];
    }
    return result;
}

int main(int argc, char **argv) {
    int threadCount = 12;
    omp_set_num_threads(threadCount);
    cout << "线程数量为:" << threadCount << endl;
    const size_t TEST_SIZE = 8000000000;
    char *testArray = new char[TEST_SIZE];
    std::mt19937 rng;
    rng.seed(std::random_device()());
    std::uniform_int_distribution<std::mt19937::result_type> dist6(0, 4);
    cout << "填充数组\n";
    auto fillingStartTime = clock();
    for (int i = 0; i < TEST_SIZE; ++i) {
        testArray[i] = dist6(rng);
    }
    auto fillingEndTime = clock();
    auto fillingTime = fillingEndTime - fillingStartTime;
    cout << "填充时间:" << fillingTime << endl;

    // 测试OpenMP耗时
    for (int i = 1; i <= 3; ++i) {
        cout << "正在运行OpenMP代码\n";
        auto ompStartTime = clock();
        auto ompResult = llsum(testArray, TEST_SIZE, i * 2);
        auto ompEndTime = clock();
        auto ompTime = ompEndTime - ompStartTime;
        cout << i * 2 << "线程OpenMP耗时:" << ompTime << endl << "结果: " << ompResult << endl;
    }

    // 测试串行求和耗时
    cout << "正在运行串行代码\n";
    auto seqStartTime = clock();
    long long expectedResult = 0;
    for (int i = 0; i < TEST_SIZE; ++i) {
        expectedResult += testArray[i];
    }
    auto seqEndTime = clock();
    auto seqTime = seqEndTime - seqStartTime;
    cout << "串行耗时: " << seqTime << endl << "结果: " << expectedResult << endl;

    delete[]testArray;
    return 0;
}

内容的提问来源于stack exchange,提问作者Name Null

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.16 07:30:52