You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

MPI_Gatherv()触发段错误的原因排查求助

MPI_Gatherv调用引发段错误的问题排查与修复

在并行计算任务中,仅让部分进程执行计算逻辑,主进程通过MPI_Gatherv(a_per_process, mylen_per_process, MPI_LONG_DOUBLE, a, recvcounts, displs, MPI_LONG_DOUBLE, 0, MPI_COMM_WORLD);合并结果时触发段错误。注释该语句后程序可正常运行,尝试过double/MPI_DOUBLE、long double/MPI_LONG_DOUBLE两种类型组合,问题均存在。

问题代码

/*****************************************************************************
 * DESCRIPTION:
 * This program increments every element of the array by two.
 * It extracts the averge execution time for different numbers of threads,
 * We do this in order to compare the performance of each routine.
 * Compile:
 *  $mpicc mpic.c -o mpic -fopenmp -lm -Ofast
 * Run:
 *  $mpirun -np <maxthreads> ./mpic
 ******************************************************************************/
#include <stdio.h>
#include <stdlib.h>
#include <mpi.h>

int main(int argc, char *argv[])
{
    int j = 0;
    long long int len_per_process = 0;
    long long int remainder = 0;
    long long int mylen_per_process = 0;
    int size = 0;
    int rank = 0;
    int *recvcounts, *displs;
    long double *a, *a_per_process;
    double start_comp = 0;
    double start_comm = 0;
    double end_comp = 0;
    double end_comm = 0;
    double maxtime_comp = 0;
    double maxtime_comm = 0;

    int i = 0;
    long nSamples = 10;
    long long int length = 1.0;
    int maxthreads = 0;
    int testnumber = 0;
    long long int minlength = 1;
    long long int maxlength = 1;
    int cycles = 0;
    long longlength = 0;

    MPI_Init(&argc, &argv);
    MPI_Comm_size(MPI_COMM_WORLD, &size);
    MPI_Comm_rank(MPI_COMM_WORLD, &rank);

    /*Whole array allocation in master process*/
    if (rank == 0)
    {
        a = (long double *)malloc(length * sizeof(long double));
    }

    for (length = minlength; length <= maxlength; length = length * 10)
    {
        for (i = 1; i <= size; i = i * 2)
        {
            /*Data distribution to processes*/
            len_per_process = length / i;
            remainder = length % i;
            mylen_per_process = (rank < remainder) ? (len_per_process + 1) : (len_per_process);
            recvcounts = (int *)malloc(size * sizeof(int));
            displs = (int *)malloc(size * sizeof(int));

            MPI_Allgather(&mylen_per_process, 1, MPI_INT, recvcounts, 1, MPI_INT, MPI_COMM_WORLD);

            displs[0] = 0;
            for (j = 1; j < size; j++)
            {
                displs[j] = displs[j - 1] + recvcounts[j - 1];
            }

            /*Sub-Arrays Allocation and Initialisation at each process*/
            a_per_process = (long double *)malloc(mylen_per_process * sizeof(long double));
            for (j = 0; j < mylen_per_process; j++)
            {
                a_per_process[j] = 0.0;
            }

            if (rank <= i)
            {
                /*Increment elements by 2*/
                start_comp = omp_get_wtime();
                for (j = 0; j < nSamples; j++)
                {
                    for (int k = 0; k < mylen_per_process; k++)
                    {
                        a_per_process[k] = a_per_process[k] + 2.0;
                    }
                }
                end_comp = omp_get_wtime() - start_comp;
                start_comm = omp_get_wtime();
                end_comm = omp_get_wtime() - start_comm;
            }
            // The following line causes a segfault:
MPI_Gatherv(a_per_process, mylen_per_process, MPI_LONG_DOUBLE, a, recvcounts, displs, MPI_LONG_DOUBLE, 0, MPI_COMM_WORLD);

            // Get the maximum computation and communication time
            MPI_Reduce(&end_comp, &maxtime_comp, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD);
            MPI_Reduce(&end_comm, &maxtime_comm, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD);

            MPI_Barrier(MPI_COMM_WORLD);
            free(a_per_process);
            free(recvcounts);
            free(displs);
        }
    }

    if (rank == 0)
    {
        free(a);
    }
    MPI_Finalize();
    return 0;
}

问题根源

  • 主进程数组a分配时机错误:初始仅按length=1分配内存,后续循环中length以10倍扩容,但未重新分配a的内存,导致MPI_Gatherv写入时内存越界。
  • 计数数组类型不匹配:recvcounts和displs存储的是long long int类型的元素数量,但用int*分配,当length较大时会触发整数溢出,导致位移量计算错误。
  • 进程参与逻辑错误:if (rank <= i)应为if (rank < i),进程rank从0开始,i是当前启用的进程数,比如i=2时仅rank 0、1需要执行计算,而非包含rank 2。
  • 非主进程a指针未初始化:MPI_Gatherv中,非根进程(rank≠0)的接收缓冲区参数可设为NULL,但代码中所有进程都传入a,非主进程的a是野指针,访问时触发段错误。

修复后的代码

/*****************************************************************************
 * DESCRIPTION:
 * This program increments every element of the array by two.
 * It extracts the averge execution time for different numbers of threads,
 * We do this in order to compare the performance of each routine.
 * Compile:
 *  $mpicc mpic.c -o mpic -fopenmp -lm -Ofast
 * Run:
 *  $mpirun -np <maxthreads> ./mpic
 ******************************************************************************/
#include <stdio.h>
#include <stdlib.h>
#include <mpi.h>
#include <omp.h>

int main(int argc, char *argv[])
{
    int j = 0;
    long long int len_per_process = 0;
    long long int remainder = 0;
    long long int mylen_per_process = 0;
    int size = 0;
    int rank = 0;
    long long int *recvcounts, *displs;
    long double *a = NULL, *a_per_process = NULL;
    double start_comp = 0;
    double start_comm = 0;
    double end_comp = 0;
    double end_comm = 0;
    double maxtime_comp = 0;
    double maxtime_comm = 0;

    int i = 0;
    long nSamples = 10;
    long long int length = 1.0;
    long long int minlength = 1;
    long long int maxlength = 1;

    MPI_Init(&argc, &argv);
    MPI_Comm_size(MPI_COMM_WORLD, &size);
    MPI_Comm_rank(MPI_COMM_WORLD, &rank);

    for (length = minlength; length <= maxlength; length = length * 10)
    {
        if (rank == 0)
        {
            if (a != NULL) free(a);
            a = (long double *)malloc(length * sizeof(long double));
            if (a == NULL) {
                fprintf(stderr, "Memory allocation failed for a\n");
                MPI_Abort(MPI_COMM_WORLD, 1);
            }
        }

        for (i = 1; i <= size; i = i * 2)
        {
            len_per_process = length / i;
            remainder = length % i;
            mylen_per_process = (rank < i) ? ((rank < remainder) ? (len_per_process + 1) : len_per_process) : 0;
            
            recvcounts = (long long int *)malloc(size * sizeof(long long int));
            displs = (long long int *)malloc(size * sizeof(long long int));
            if (recvcounts == NULL || displs == NULL) {
                fprintf(stderr, "Memory allocation failed for recvcounts/displs\n");
                MPI_Abort(MPI_COMM_WORLD, 1);
            }

            MPI_Allgather(&mylen_per_process, 1, MPI_LONG_LONG_INT, recvcounts, 1, MPI_LONG_LONG_INT, MPI_COMM_WORLD);

            displs[0] = 0;
            for (j = 1; j < size; j++)
            {
                displs[j] = displs[j - 1] + recvcounts[j - 1];
            }

            a_per_process = NULL;
            if (mylen_per_process > 0) {
                a_per_process = (long double *)malloc(mylen_per_process * sizeof(long double));
                if (a_per_process == NULL) {
                    fprintf(stderr, "Memory allocation failed for a_per_process\n");
                    MPI_Abort(MPI_COMM_WORLD, 1);
                }
                for (j = 0; j < mylen_per_process; j++)
                {
                    a_per_process[j] = 0.0;
                }
            }

            end_comp = 0.0;
            end_comm = 0.0;
            if (rank < i)
            {
                start_comp = omp_get_wtime();
                for (j = 0; j < nSamples; j++)
                {
                    for (int k = 0; k < mylen_per_process; k++)
                    {
                        a_per_process[k] = a_per_process[k] + 2.0;
                    }
                }
                end_comp = omp_get_wtime() - start_comp;
            }

            start_comm = omp_get_wtime();
            MPI_Gatherv(a_per_process, mylen_per_process, MPI_LONG_DOUBLE, 
                        (rank == 0) ? a : NULL, recvcounts, displs, MPI_LONG_DOUBLE, 
                        0, MPI_COMM_WORLD);
            end_comm = omp_get_wtime() - start_comm;

            MPI_Reduce(&end_comp, &maxtime_comp, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD);
            MPI_Reduce(&end_comm, &maxtime_comm, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD);

            if (rank == 0) {
                printf("Length: %lld, Processes: %d, Comp Time: %.6lf, Comm Time: %.6lf\n", 
                       length, i, maxtime_comp, maxtime_comm);
            }

            MPI_Barrier(MPI_COMM_WORLD);
            if (a_per_process != NULL) free(a_per_process);
            free(recvcounts);
            free(displs);
        }
    }

    if (rank == 0 && a != NULL)
    {
        free(a);
    }
    MPI_Finalize();
    return 0;
}

内容的提问来源于stack exchange,提问作者kalle

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.02 15:10:40