MPI_Gatherv()触发段错误的原因排查求助
MPI_Gatherv调用引发段错误的问题排查与修复
在并行计算任务中,仅让部分进程执行计算逻辑,主进程通过MPI_Gatherv(a_per_process, mylen_per_process, MPI_LONG_DOUBLE, a, recvcounts, displs, MPI_LONG_DOUBLE, 0, MPI_COMM_WORLD);合并结果时触发段错误。注释该语句后程序可正常运行,尝试过double/MPI_DOUBLE、long double/MPI_LONG_DOUBLE两种类型组合,问题均存在。
问题代码
/***************************************************************************** * DESCRIPTION: * This program increments every element of the array by two. * It extracts the averge execution time for different numbers of threads, * We do this in order to compare the performance of each routine. * Compile: * $mpicc mpic.c -o mpic -fopenmp -lm -Ofast * Run: * $mpirun -np <maxthreads> ./mpic ******************************************************************************/ #include <stdio.h> #include <stdlib.h> #include <mpi.h> int main(int argc, char *argv[]) { int j = 0; long long int len_per_process = 0; long long int remainder = 0; long long int mylen_per_process = 0; int size = 0; int rank = 0; int *recvcounts, *displs; long double *a, *a_per_process; double start_comp = 0; double start_comm = 0; double end_comp = 0; double end_comm = 0; double maxtime_comp = 0; double maxtime_comm = 0; int i = 0; long nSamples = 10; long long int length = 1.0; int maxthreads = 0; int testnumber = 0; long long int minlength = 1; long long int maxlength = 1; int cycles = 0; long longlength = 0; MPI_Init(&argc, &argv); MPI_Comm_size(MPI_COMM_WORLD, &size); MPI_Comm_rank(MPI_COMM_WORLD, &rank); /*Whole array allocation in master process*/ if (rank == 0) { a = (long double *)malloc(length * sizeof(long double)); } for (length = minlength; length <= maxlength; length = length * 10) { for (i = 1; i <= size; i = i * 2) { /*Data distribution to processes*/ len_per_process = length / i; remainder = length % i; mylen_per_process = (rank < remainder) ? (len_per_process + 1) : (len_per_process); recvcounts = (int *)malloc(size * sizeof(int)); displs = (int *)malloc(size * sizeof(int)); MPI_Allgather(&mylen_per_process, 1, MPI_INT, recvcounts, 1, MPI_INT, MPI_COMM_WORLD); displs[0] = 0; for (j = 1; j < size; j++) { displs[j] = displs[j - 1] + recvcounts[j - 1]; } /*Sub-Arrays Allocation and Initialisation at each process*/ a_per_process = (long double *)malloc(mylen_per_process * sizeof(long double)); for (j = 0; j < mylen_per_process; j++) { a_per_process[j] = 0.0; } if (rank <= i) { /*Increment elements by 2*/ start_comp = omp_get_wtime(); for (j = 0; j < nSamples; j++) { for (int k = 0; k < mylen_per_process; k++) { a_per_process[k] = a_per_process[k] + 2.0; } } end_comp = omp_get_wtime() - start_comp; start_comm = omp_get_wtime(); end_comm = omp_get_wtime() - start_comm; } // The following line causes a segfault: MPI_Gatherv(a_per_process, mylen_per_process, MPI_LONG_DOUBLE, a, recvcounts, displs, MPI_LONG_DOUBLE, 0, MPI_COMM_WORLD); // Get the maximum computation and communication time MPI_Reduce(&end_comp, &maxtime_comp, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD); MPI_Reduce(&end_comm, &maxtime_comm, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD); MPI_Barrier(MPI_COMM_WORLD); free(a_per_process); free(recvcounts); free(displs); } } if (rank == 0) { free(a); } MPI_Finalize(); return 0; }
问题根源
- 主进程数组
a分配时机错误:初始仅按length=1分配内存,后续循环中length以10倍扩容,但未重新分配a的内存,导致MPI_Gatherv写入时内存越界。 - 计数数组类型不匹配:
recvcounts和displs存储的是long long int类型的元素数量,但用int*分配,当length较大时会触发整数溢出,导致位移量计算错误。 - 进程参与逻辑错误:
if (rank <= i)应为if (rank < i),进程rank从0开始,i是当前启用的进程数,比如i=2时仅rank 0、1需要执行计算,而非包含rank 2。 - 非主进程
a指针未初始化:MPI_Gatherv中,非根进程(rank≠0)的接收缓冲区参数可设为NULL,但代码中所有进程都传入a,非主进程的a是野指针,访问时触发段错误。
修复后的代码
/***************************************************************************** * DESCRIPTION: * This program increments every element of the array by two. * It extracts the averge execution time for different numbers of threads, * We do this in order to compare the performance of each routine. * Compile: * $mpicc mpic.c -o mpic -fopenmp -lm -Ofast * Run: * $mpirun -np <maxthreads> ./mpic ******************************************************************************/ #include <stdio.h> #include <stdlib.h> #include <mpi.h> #include <omp.h> int main(int argc, char *argv[]) { int j = 0; long long int len_per_process = 0; long long int remainder = 0; long long int mylen_per_process = 0; int size = 0; int rank = 0; long long int *recvcounts, *displs; long double *a = NULL, *a_per_process = NULL; double start_comp = 0; double start_comm = 0; double end_comp = 0; double end_comm = 0; double maxtime_comp = 0; double maxtime_comm = 0; int i = 0; long nSamples = 10; long long int length = 1.0; long long int minlength = 1; long long int maxlength = 1; MPI_Init(&argc, &argv); MPI_Comm_size(MPI_COMM_WORLD, &size); MPI_Comm_rank(MPI_COMM_WORLD, &rank); for (length = minlength; length <= maxlength; length = length * 10) { if (rank == 0) { if (a != NULL) free(a); a = (long double *)malloc(length * sizeof(long double)); if (a == NULL) { fprintf(stderr, "Memory allocation failed for a\n"); MPI_Abort(MPI_COMM_WORLD, 1); } } for (i = 1; i <= size; i = i * 2) { len_per_process = length / i; remainder = length % i; mylen_per_process = (rank < i) ? ((rank < remainder) ? (len_per_process + 1) : len_per_process) : 0; recvcounts = (long long int *)malloc(size * sizeof(long long int)); displs = (long long int *)malloc(size * sizeof(long long int)); if (recvcounts == NULL || displs == NULL) { fprintf(stderr, "Memory allocation failed for recvcounts/displs\n"); MPI_Abort(MPI_COMM_WORLD, 1); } MPI_Allgather(&mylen_per_process, 1, MPI_LONG_LONG_INT, recvcounts, 1, MPI_LONG_LONG_INT, MPI_COMM_WORLD); displs[0] = 0; for (j = 1; j < size; j++) { displs[j] = displs[j - 1] + recvcounts[j - 1]; } a_per_process = NULL; if (mylen_per_process > 0) { a_per_process = (long double *)malloc(mylen_per_process * sizeof(long double)); if (a_per_process == NULL) { fprintf(stderr, "Memory allocation failed for a_per_process\n"); MPI_Abort(MPI_COMM_WORLD, 1); } for (j = 0; j < mylen_per_process; j++) { a_per_process[j] = 0.0; } } end_comp = 0.0; end_comm = 0.0; if (rank < i) { start_comp = omp_get_wtime(); for (j = 0; j < nSamples; j++) { for (int k = 0; k < mylen_per_process; k++) { a_per_process[k] = a_per_process[k] + 2.0; } } end_comp = omp_get_wtime() - start_comp; } start_comm = omp_get_wtime(); MPI_Gatherv(a_per_process, mylen_per_process, MPI_LONG_DOUBLE, (rank == 0) ? a : NULL, recvcounts, displs, MPI_LONG_DOUBLE, 0, MPI_COMM_WORLD); end_comm = omp_get_wtime() - start_comm; MPI_Reduce(&end_comp, &maxtime_comp, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD); MPI_Reduce(&end_comm, &maxtime_comm, 1, MPI_DOUBLE, MPI_MAX, 0, MPI_COMM_WORLD); if (rank == 0) { printf("Length: %lld, Processes: %d, Comp Time: %.6lf, Comm Time: %.6lf\n", length, i, maxtime_comp, maxtime_comm); } MPI_Barrier(MPI_COMM_WORLD); if (a_per_process != NULL) free(a_per_process); free(recvcounts); free(displs); } } if (rank == 0 && a != NULL) { free(a); } MPI_Finalize(); return 0; }
内容的提问来源于stack exchange,提问作者kalle
相关产品推荐
相关产品推荐

