You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何多线程(pthreads)比多进程的并行均值计算性能更优?

多进程与多线程并行均值计算的性能差异分析

实验背景与测试结果

我编写了一个演示程序,用于计算大量伪随机整数的均值,以此对比不同并发形式的性能。实现了三个版本:

  • 两个多进程版本(分别使用映射内存、管道进行IPC)
  • 一个pthreads多线程版本

所有版本均设置4个工作单元(包含父进程/主线程),测试环境为搭载Intel i7四核处理器、运行OS Monterey的MacBook Pro,输入为包含10亿(2^30)个4字节整数的二进制文件。性能对比仅针对串行与并行版本的核心差异代码——get_mean_serial和get_mean_parallel函数,通过耗时计算各版本的加速比。

测试结果显示:

  • 两个多进程版本的并行相对串行加速比约为2.33倍
  • pthreads多线程版本的加速比达3.9倍,更接近4核处理器的理论加速上限

我需要解释这一差异,尤其是多进程版本为何远未达到理想性能(以pthreads版本为对照组),目前已排查以下可能原因:

已排查的排除项

  • 单进程管理开销:将核心计算函数get_mean_over_chunk改为直接返回常量,仅保留初始化、映射内存、fork、wait及结果合并等管理代码,测得这些操作仅耗时约1毫秒,而多进程与理想加速比的差距对应耗时约0.5秒,因此该原因不成立。
  • 写时复制(Copy-on-Write)开销:多进程版本仅读取映射的输入页,且性能测量已包含IPC共享页的复制耗时,因此该额外开销可排除。
  • 缓存效应存疑:曾怀疑多进程因仅读取部分数据导致局部性低,且L3缓存共享时的块替换实际串行,导致缓存命中率低。但pthreads版本同样面临缓存局部性问题却获得近乎最优结果,因此该原因存疑。

多进程(映射内存IPC)完整实现代码

#include <stdio.h>
#include <stdlib.h>
#include <sys/time.h>
#include <sys/types.h>
#include <sys/mman.h>
#include <sys/wait.h>
#include <unistd.h>

#define NUM_WORKERS 4

FILE * fopen_checked(const char * const file, const char * const mode) {
    FILE * fp = fopen(file, mode);
    if (!fp) {
        perror("Failure to open file");
        exit(EXIT_FAILURE);
    }
    return fp;
}

void * mmap_checked(size_t size) {
    void * mem = mmap(NULL, size, PROT_READ | PROT_WRITE,
            MAP_ANONYMOUS | MAP_SHARED, -1, 0);
    if (mem == MAP_FAILED) {
        perror("Demand-zero memory allocation failure");
        exit(EXIT_FAILURE);
    }
    return mem;
}

long get_file_size(FILE * fp) {
    fseek(fp, 0L, SEEK_END);
    long size = ftell(fp);
    rewind(fp);
    return size;
}

double get_mean_over_chunk(const int * const nums, long start, long chunk) {
    double sum = 0.0;
    const int * end = nums + start + chunk;
    for (const int * ptr = nums + start; ptr < end; ++ptr) {
        sum += *ptr;
    }
    return sum / chunk;
}

double get_mean_serial(const int * const nums, const long num) {
    return get_mean_over_chunk(nums, 0, num);
}

double get_mean_parallel(const int * const nums, const long num,
        int num_workers) {
    long chunk = num / num_workers;
    double * means = mmap_checked(num_workers * sizeof(double));
    int num_children = num_workers - 1;
    int i;
    for (i = 0; i < num_children; ++i) {
        pid_t pid = fork();
        if (!pid) {
            means[i] = get_mean_over_chunk(nums, i * chunk, chunk);
            exit(EXIT_SUCCESS);
        }
    }
    means[i] = get_mean_over_chunk(nums, i * chunk, chunk);
    while (wait(NULL) > 0);
    double sum = 0.0;
    for (i = 0; i < num_workers; ++i) {
        sum += means[i];
    }
    munmap(means, num_workers * sizeof(double));
    return sum / num_workers;
}

void print_elapsed_time(struct timeval * start, struct timeval * end) {
    time_t seconds = end->tv_sec - start->tv_sec;
    suseconds_t microseconds = end->tv_usec - start->tv_usec;
    if (microseconds < 0) {
        --seconds;
        microseconds += 1000000;
    }
    printf("Elapsed time: %ld seconds, %ld microseconds\n",
            (long)seconds, (long)microseconds);
}

int main(int argc, char ** argv) {
    if (argc < 2) {
        fprintf(stderr, "Usage: %s <input-file>\n", argv[0]);
        return EXIT_FAILURE;
    }
    FILE * fp = fopen_checked(argv[1], "r");
    long size = get_file_size(fp);
    int * nums = mmap_checked(size);
    long num = size / sizeof(int);
    fread(nums, sizeof(int), num, fp);
    fclose(fp);
    struct timeval start, end;
    gettimeofday(&start, NULL);
    double mean_serial = get_mean_serial(nums, num);
    gettimeofday(&end, NULL);
    printf("Mean (serial): %f\n", mean_serial);
    print_elapsed_time(&start, &end);
    gettimeofday(&start, NULL);
    double mean_parallel = get_mean_parallel(nums, num, NUM_WORKERS);
    gettimeofday(&end, NULL);
    printf("Mean (parallel): %f\n", mean_parallel);
    print_elapsed_time(&start, &end);
    return EXIT_SUCCESS;
}

注:pthreads版本仅需修改get_mean_over_chunk函数以接收结构体指针,并行版本使用参数结构体数组,为每个pthread_create调用传递不同的结构体元素。

内容的提问来源于stack exchange,提问作者Amittai Aviram

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.19 17:53:20