MPI Gather调用时memcpy内存范围重叠错误的排查与解决求助
MPI图像卷积实现中的内存重叠错误分析与修复
我尝试用C语言结合MPI实现图像卷积,思路是把图像分成不同块,让不同进程处理后,在根进程用MPI_Gather汇总结果。刚接触C语言,遇到了内存重叠的错误,希望有人解释原因并提供解决方法。
代码实现
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <sys/time.h> #include <mpi.h> #define IMAGE_IN "logo.pgm" #define IMAGE_OUT "result.pgm" /* function declarations */ int * read_image (char file_name[], int *p_h, int *p_w, int *p_levels); void write_image (int *image, char file_name[], int h, int w, int levels); int* convolve(int *image_in, int *image_out, int kernel[3][3], int height, int width,int first_row, int last_row,int chunk_size) { int i, j, x, y; int sum, max_val, min_val; /* initialize max_val and min_val */ max_val = image_in[0]; min_val = image_in[0]; for (i = first_row; i < last_row; i++) { for (j = 1; j < width - 1; j++) { sum = 0; for (x = -1; x <= 1; x++) { for (y = -1; y <= 1; y++) { sum += kernel[x+1][y+1] * image_in[(i+x)*width+(j+y)]; } } /* update max_val and min_val */ if (sum > max_val) { max_val = sum; } if (sum < min_val) { min_val = sum; } image_out[i*width+j] = sum; } } printf("min is:%d\n", min_val); printf("max is: %d\n",max_val); /* normalize pixel values */ for (i = first_row; i < last_row; i++) { for (j = 1; j < width - 1; j++) { image_out[i*width+j] = 255 * (image_out[i*width+j] - min_val) / (max_val - min_val); } } return image_out; } int main(int argc, char *argv[]) { int i, j, h, w, levels ; struct timeval tdeb, tfin; int *image_in, *image_out; int my_rank; int num_procs; MPI_Init(&argc, &argv); MPI_Comm_size(MPI_COMM_WORLD, &num_procs); MPI_Comm_rank(MPI_COMM_WORLD, &my_rank); image_in = read_image (IMAGE_IN, &h, &w, &levels); int chunk_size= h/num_procs; printf ("chunk size is %d \n", chunk_size); int first_row= my_rank* chunk_size+1; int last_row= first_row+ chunk_size-1; printf ("first row is %d \n", first_row); printf ("last row is %d \n", last_row); /* image processing (just a copy in this example) */ gettimeofday(&tdeb, NULL); int kernel[3][3]={{0,-1,0}, {-1,5,-1}, {0,-1,0}}; int *local_image = malloc(chunk_size * w*sizeof(int)); image_out = malloc (h*w*sizeof(int)); local_image=convolve(image_in, image_out, kernel, chunk_size, w,first_row,last_row,chunk_size); MPI_Gather(local_image, chunk_size*w,MPI_INT,image_out, h*w, MPI_INT, 0, MPI_COMM_WORLD); gettimeofday(&tfin, NULL); printf ("computation time (microseconds): %ld\n", (tfin.tv_sec - tdeb.tv_sec)*1000000 + (tfin.tv_usec - tdeb.tv_usec)); write_image (local_image, IMAGE_OUT, h, w, levels); MPI_Finalize(); free (image_in); free (image_out); return 0; } /* * read an image from a file and dynamic memory allocation for it * param1 : name of the image file * param2 : where to store the height * param3 : where to store the weight * param4 : where to store the number of levels * return : address of the pixels */ int * read_image (char file_name[], int *p_h, int *p_w, int *p_levels){ FILE *fin; int i, j, h, w, levels ; char buffer[80]; int *image; /* open P2 image */ fin = fopen (IMAGE_IN, "r"); if (fin == NULL){ printf ("file open error\n"); exit(-1); } fscanf (fin, "%s", buffer); if (strcmp (buffer, "P2")){ printf ("the image format must be P2\n"); exit(-1); } fgets (buffer, 80, fin); fgets (buffer, 80, fin); fscanf (fin, "%d%d", &w, &h); fscanf (fin, "%d", &levels); printf ("image reading ... h = %d w = %d\n", h, w); /* dynamic memory allocation for the pixels */ image = malloc (h*w*sizeof(int)); /* pixels reading */ for (i = 0; i < h ; i++) for (j = 0; j < w; j++) fscanf (fin, "%d", image +i*w +j); fclose (fin); *p_h = h; *p_w = w; *p_levels=levels; return image; } /* * write an image in a file * param1 : address of the pixels * param2 : name of the image file * param3 : the height * param4 : the weight * param5 : the number of levels * return : void */ void write_image (int *image, char file_name[], int h, int w, int levels){ FILE *fout; int i, j; /* open the file */ fout=fopen(IMAGE_OUT,"w"); if (fout == NULL){ printf ("file opening error\n"); exit(-1); } /* header write */ fprintf(fout,"P2\n# test \n%d %d\n%d\n", w, h, levels); /* format P2, commentary, w x h points, levels */ /* pixels writing*/ for (i = 0; i < h ; i++) for (j = 0; j < w; j++) fprintf (fout, "%d\n", image[i*w+j]); fclose (fout); return; }
运行错误信息
文件src/mpi/datatype/typerep/src/typerep_yaksa_pack.c第286行断言失败:FALSE
memcpy参数内存范围重叠,dst_=0x7f56ed5be010 src_=0x7f56ed5be010 len_=1557836
节点0上终止(1):内部错误
错误原因分析
内存重叠直接触发错误:
根进程中convolve函数返回了传入的image_out指针,导致local_image和image_out指向同一块内存。调用MPI_Gather时,源缓冲区和目标缓冲区内存地址重叠,MPI执行memcpy时触发断言失败。convolve函数逻辑错误:- 函数直接修改全局的
image_out而非局部缓冲区local_image,完全违背分块处理的设计,所有进程的计算结果会互相覆盖。 - 用全局图像的第一个像素初始化
max_val和min_val,而非当前进程计算的卷积结果,导致归一化计算完全错误。
- 函数直接修改全局的
分块与边界处理问题:
- 卷积需要访问当前行的上下相邻行,当前分块逻辑未处理跨进程的边界数据共享,边缘行计算会出错。
- 所有进程都执行
write_image,多个进程同时写入同一文件会导致内容混乱。
内存泄漏:
malloc分配的local_image指针被convolve的返回值覆盖,导致这块内存无法释放。
修复步骤
1. 修正convolve函数,写入局部缓冲区
修改函数逻辑,让结果写入局部缓冲区,同时用当前进程的计算值初始化局部最大最小值:
int* convolve(int *image_in, int *local_image, int kernel[3][3], int width, int first_row, int last_row) { int i, j, x, y; int sum, max_val, min_val; // 用当前进程第一个有效计算值初始化max/min sum = 0; for (x = -1; x <= 1; x++) { for (y = -1; y <= 1; y++) { sum += kernel[x+1][y+1] * image_in[(first_row+x)*width + (1+y)]; } } max_val = sum; min_val = sum; // 计算卷积并更新局部极值 for (i = first_row; i < last_row; i++) { for (j = 1; j < width - 1; j++) { sum = 0; for (x = -1; x <= 1; x++) { for (y = -1; y <= 1; y++) { sum += kernel[x+1][y+1] * image_in[(i+x)*width + (j+y)]; } } if (sum > max_val) max_val = sum; if (sum < min_val) min_val = sum; // 写入局部缓冲区,按局部行偏移计算索引 local_image[(i - first_row)*width + j] = sum; } } // 局部归一化 for (i = first_row; i < last_row; i++) { for (j = 1; j < width - 1; j++) { int local_idx = (i - first_row)*width + j; local_image[local_idx] = 255 * (local_image[local_idx] - min_val) / (max_val - min_val); } } return local_image; }
2. 修正主函数的缓冲区与MPI调用
- 确保每个进程使用独立的局部缓冲区,根进程单独分配全局结果缓冲区
- 修正
MPI_Gather参数,避免内存重叠 - 仅让根进程执行结果写入和时间打印
- 处理图像高度无法被进程数整除的情况
int main(int argc, char *argv[]) { int i, j, h, w, levels ; struct timeval tdeb, tfin; int *image_in, *image_out = NULL; int my_rank, num_procs; MPI_Init(&argc, &argv); MPI_Comm_size(MPI_COMM_WORLD, &num_procs); MPI_Comm_rank(MPI_COMM_WORLD, &my_rank); image_in = read_image(IMAGE_IN, &h, &w, &levels); int chunk_size = h / num_procs; // 处理最后一个进程的剩余行 int first_row = my_rank * chunk_size; int last_row = (my_rank == num_procs - 1) ? h : first_row + chunk_size; int actual_chunk_rows = last_row - first_row; int kernel[3][3] = {{0,-1,0}, {-1,5,-1}, {0,-1,0}}; // 分配局部缓冲区 int *local_image = malloc(actual_chunk_rows * w * sizeof(int)); if (my_rank == 0) { image_out = malloc(h * w * sizeof(int)); // 初始化边缘行(卷积未处理的部分) memcpy(image_out, image_in, w*sizeof(int)); memcpy(image_out + (h-1)*w, image_in + (h-1)*w, w*sizeof(int)); } gettimeofday(&tdeb, NULL); // 小图像直接用全局图像数据,大图像需用MPI传递边界行 convolve(image_in, local_image, kernel, w, first_row, last_row); // 按进程顺序收集局部结果到全局缓冲区对应位置 MPI_Gather(local_image, actual_chunk_rows * w, MPI_INT, image_out + first_row * w, actual_chunk_rows * w, MPI_INT, 0, MPI_COMM_WORLD); gettimeofday(&tfin, NULL); if (my_rank == 0) { printf("computation time (microseconds): %ld\n", (tfin.tv_sec - tdeb.tv_sec)*1000000 + (tfin.tv_usec - tdeb.tv_usec)); write_image(image_out, IMAGE_OUT, h, w, levels); free(image_out); } MPI_Finalize(); free(image_in); free(local_image); return 0; }
3. 优化write_image函数
避免每个像素单独换行,减少文件体积:
void write_image(int *image, char file_name[], int h, int w, int levels) { FILE *fout; int i, j; fout = fopen(file_name, "w"); if (fout == NULL) { printf("file opening error\n"); exit(-1); } fprintf(fout, "P2\n# test\n%d %d\n%d\n", w, h, levels); for (i = 0; i < h; i++) { for (j = 0; j < w; j++) { fprintf(fout, "%d ", image[i*w + j]); } fprintf(fout, "\n"); } fclose(fout); }
4. 边界行处理补充
如果处理大图像,需要通过MPI_Send和MPI_Recv在相邻进程间传递边界行数据,确保每个进程能获取卷积所需的上下行信息。小图像可以保持当前每个进程读取完整图像的方式,简化实现。
内容的提问来源于stack exchange,提问作者Hoang Son
相关产品推荐
相关产品推荐

