MPI矩阵向量乘法程序malloc内存分配失败问题求助
问题概述
编写MPI并行矩阵向量乘法程序时,为最后一个进程分配额外计算行数的逻辑导致malloc内存分配失败,进而触发段错误。当N=5时错误复现,错误信息显示无法分配内存,随后进程因访问未映射地址崩溃。
核心错误分析
整数除法与ceil函数误用
代码中rows = ceil(N / numprocs)存在逻辑错误:N与numprocs均为整数,N/numprocs会先执行整数除法(截断小数部分),再传入ceil函数。例如N=5、numprocs=3时,5/3=1,ceil(1)=1,得到rows=1。后续计算extra_rows = numprocs * rows - N = 3*1-5 = -2,最终rows+extra_rows = -1,导致malloc传入负数的内存大小。由于size_t是无符号类型,负数会被转换为极大的正数,触发内存分配失败,返回NULL。MPI数据分发/收集函数不匹配
当前使用MPI_Scatter和MPI_Gather处理等长数据,但实际不同进程的任务行数不等(最后一个进程行数更少),固定的发送/接收计数会导致内存越界,加重段错误问题。内存分配失败后未终止程序
即使malloc返回NULL,程序仍继续执行,后续访问NULL指针直接触发段错误。
修复方案
正确计算进程任务行数
改用纯整数运算计算每个进程的本地行数,避免浮点运算误差:- 基础行数:
base_rows = N / numprocs - 余数:
remainder = N % numprocs - 最后一个进程的行数:
local_rows = base_rows + (rank == numprocs-1 ? remainder : 0)
- 基础行数:
使用MPI_Scatterv/MPI_Gatherv处理不等长数据
构造发送计数数组sendcounts和位移数组displs,让根进程正确分发/收集不同长度的数据块。内存分配失败后立即终止进程
当malloc返回NULL时,打印错误并调用MPI_Abort终止所有进程,避免后续非法内存访问。
完整修正代码
#define N 5 #include <stdio.h> #include <stdlib.h> #include <mpi.h> #include <sys/time.h> // 矩阵打印函数示例 void print_matrix1(int rank, int rows, float *matrix) { printf("Rank %d matrix:\n", rank); for (int i = 0; i < rows; i++) { for (int j = 0; j < N; j++) { printf("%.1f ", matrix[i*N + j]); } printf("\n"); } } int main(int argc, char *argv[]) { int i, j, rank, numprocs, base_rows, remainder, local_rows; float *matrix = NULL, *matrixAux = NULL; float vector[N]; float *result = NULL, *resultAux = NULL; int *sendcounts = NULL, *displs = NULL; MPI_Init(&argc, &argv); MPI_Comm_size(MPI_COMM_WORLD, &numprocs); MPI_Comm_rank(MPI_COMM_WORLD, &rank); // 计算每个进程的本地处理行数 base_rows = N / numprocs; remainder = N % numprocs; local_rows = base_rows + (rank == numprocs - 1 ? remainder : 0); // 分配本地内存,失败则终止所有进程 if ((matrixAux = malloc(sizeof(float) * local_rows * N)) == NULL) { perror("error 1: Failed to allocate matrixAux"); MPI_Abort(MPI_COMM_WORLD, 1); } if ((resultAux = malloc(sizeof(float) * local_rows)) == NULL) { perror("error 2: Failed to allocate resultAux"); MPI_Abort(MPI_COMM_WORLD, 1); } /* 初始化矩阵与向量(仅根进程执行) */ if (rank == 0) { if ((matrix = malloc(sizeof(float) * N * N)) == NULL) { perror("error 3: Failed to allocate matrix"); MPI_Abort(MPI_COMM_WORLD, 1); } if ((result = malloc(sizeof(float) * N)) == NULL) { perror("error 4: Failed to allocate result"); MPI_Abort(MPI_COMM_WORLD, 1); } // 分配Scatterv/Gatherv所需的计数与位移数组 sendcounts = malloc(sizeof(int) * numprocs); displs = malloc(sizeof(int) * numprocs); if (!sendcounts || !displs) { perror("error 5: Failed to allocate sendcounts/displs"); MPI_Abort(MPI_COMM_WORLD, 1); } // 填充矩阵与向量 for (int i = 0; i < N; i++) { vector[i] = i; for (int j = 0; j < N; j++) { matrix[i * N + j] = i + j; } } print_matrix1(rank, N, matrix); // 构造Scatterv的发送计数与位移 int offset = 0; for (int p = 0; p < numprocs; p++) { int p_rows = base_rows + (p == numprocs - 1 ? remainder : 0); sendcounts[p] = p_rows * N; displs[p] = offset; offset += sendcounts[p]; } } // 分发矩阵块(不等长) MPI_Scatterv(matrix, sendcounts, displs, MPI_FLOAT, matrixAux, local_rows * N, MPI_FLOAT, 0, MPI_COMM_WORLD); MPI_Bcast(vector, N, MPI_FLOAT, 0, MPI_COMM_WORLD); print_matrix1(rank, local_rows, matrixAux); // 本地矩阵向量乘法计算 for (i = 0; i < local_rows; i++) { resultAux[i] = 0; for (j = 0; j < N; j++) { resultAux[i] += matrixAux[i * N + j] * vector[j]; } } // 收集计算结果(不等长) if (rank == 0) { // 重新构造Gatherv的发送计数与位移 int offset = 0; for (int p = 0; p < numprocs; p++) { int p_rows = base_rows + (p == numprocs - 1 ? remainder : 0); sendcounts[p] = p_rows; displs[p] = offset; offset += sendcounts[p]; } } MPI_Gatherv(resultAux, local_rows, MPI_FLOAT, result, sendcounts, displs, MPI_FLOAT, 0, MPI_COMM_WORLD); /* 输出最终结果(仅根进程执行) */ if (rank == 0) { printf("Final result:\n"); for (i = 0; i < N; i++) { printf("%.2f\t", result[i]); } printf("\n"); // 释放根进程内存 free(matrix); free(result); free(sendcounts); free(displs); } // 释放本地内存 free(matrixAux); free(resultAux); MPI_Finalize(); return 0; }
内容的提问来源于stack exchange,提问作者jamondebellota

