You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

MPI Gather调用时memcpy内存范围重叠错误的排查与解决求助

MPI图像卷积实现中的内存重叠错误分析与修复

我尝试用C语言结合MPI实现图像卷积,思路是把图像分成不同块,让不同进程处理后,在根进程用MPI_Gather汇总结果。刚接触C语言,遇到了内存重叠的错误,希望有人解释原因并提供解决方法。

代码实现

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/time.h>
#include <mpi.h>

#define IMAGE_IN "logo.pgm"
#define IMAGE_OUT "result.pgm"

/* function declarations */
int * read_image (char file_name[], int *p_h, int *p_w, int *p_levels);
void write_image (int *image, char file_name[], int h, int w, int levels);
int* convolve(int *image_in, int *image_out, int kernel[3][3], int height, int width,int first_row, int last_row,int chunk_size) {
    int i, j, x, y;
    int sum, max_val, min_val;
   
    /* initialize max_val and min_val */
    max_val = image_in[0];
    min_val = image_in[0];

    for (i = first_row; i < last_row; i++) {
        for (j = 1; j < width - 1; j++) {
            sum = 0;
            for (x = -1; x <= 1; x++) {
                for (y = -1; y <= 1; y++) {
                    sum += kernel[x+1][y+1] * image_in[(i+x)*width+(j+y)];
                }
            }
            /* update max_val and min_val */
            if (sum > max_val) {
                max_val = sum;
            }
            if (sum < min_val) {
                min_val = sum;
            }
            image_out[i*width+j] = sum;
        }
    }
    printf("min is:%d\n", min_val);
    printf("max is: %d\n",max_val);

    /* normalize pixel values */
    for (i = first_row; i < last_row; i++) {
        for (j = 1; j < width - 1; j++) {
            image_out[i*width+j] = 255 * (image_out[i*width+j] - min_val) / (max_val - min_val);           
        }
    }
    return image_out;
  }

int main(int argc, char *argv[])
{

  int i, j, h, w, levels ;
  struct timeval tdeb, tfin;
  int *image_in, *image_out;
  int my_rank;
  int num_procs;
  
  MPI_Init(&argc, &argv);
  MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
  MPI_Comm_rank(MPI_COMM_WORLD, &my_rank);

  image_in = read_image (IMAGE_IN, &h, &w, &levels); 
  int chunk_size= h/num_procs;
  printf ("chunk size is %d \n", chunk_size);
  int first_row= my_rank* chunk_size+1;
  int last_row= first_row+ chunk_size-1;
  printf ("first row is %d \n", first_row);
  printf ("last row is %d \n", last_row);
  

  /* image processing (just a copy in this example) */
  gettimeofday(&tdeb, NULL);

  int kernel[3][3]={{0,-1,0},
                    {-1,5,-1},
                    {0,-1,0}};
                    
                    
  int *local_image =  malloc(chunk_size * w*sizeof(int));
  image_out = malloc (h*w*sizeof(int));
  
  local_image=convolve(image_in, image_out, kernel, chunk_size, w,first_row,last_row,chunk_size); 
  
  MPI_Gather(local_image, chunk_size*w,MPI_INT,image_out, h*w, MPI_INT, 0, MPI_COMM_WORLD);
  
  


  gettimeofday(&tfin, NULL);
  printf ("computation time (microseconds): %ld\n",  (tfin.tv_sec - tdeb.tv_sec)*1000000 + (tfin.tv_usec - tdeb.tv_usec));
  
  write_image (local_image, IMAGE_OUT, h, w, levels);
  
 
  MPI_Finalize(); 

  free (image_in);
  free (image_out); 

  return 0;
}

/*
* read an image from a file  and dynamic memory allocation for it
* param1 : name of the image file
* param2 : where to store the height
* param3 : where to store the weight
* param4 : where to store the number of levels
* return : address of the pixels
*/
int * read_image (char file_name[], int *p_h, int *p_w, int *p_levels){
  FILE *fin;
  int i, j, h, w, levels ;
  char buffer[80];
  int *image;

  /* open P2 image */
  fin = fopen (IMAGE_IN, "r");
  if (fin == NULL){
    printf ("file open error\n");
    exit(-1);
  } 
  fscanf (fin, "%s", buffer);
  if (strcmp (buffer, "P2")){
    printf ("the image format must be P2\n");
    exit(-1);
  }
  fgets (buffer, 80, fin);
  fgets (buffer, 80, fin);
  fscanf (fin, "%d%d", &w, &h);
  fscanf (fin, "%d", &levels);
  printf ("image reading ... h = %d w = %d\n", h, w);
  
  /* dynamic memory allocation for the pixels */
  image = malloc (h*w*sizeof(int));

  /* pixels reading */
  for (i = 0; i < h ; i++)
    for (j = 0; j < w; j++)
       fscanf (fin, "%d", image +i*w +j); 
  fclose (fin);

  *p_h = h;
  *p_w = w;
  *p_levels=levels;
  return image;
}
 

/*
* write an image in a file
* param1 : address of the pixels
* param2 : name of the image file
* param3 : the height
* param4 : the weight
* param5 : the number of levels
* return : void
*/

void write_image (int *image, char file_name[], int h, int w, int levels){
  FILE *fout;
  int i, j;

  /* open the file */
  fout=fopen(IMAGE_OUT,"w");
  if (fout == NULL){
    printf ("file opening error\n");
    exit(-1);
  }
  
  /* header write */
  fprintf(fout,"P2\n# test \n%d %d\n%d\n", w, h, levels);
  /* format P2, commentary, w x h points, levels */

  /* pixels writing*/
  for (i = 0; i < h ; i++)
    for (j = 0; j < w; j++)
       fprintf (fout, "%d\n", image[i*w+j]); 

  fclose (fout);
  return;
}

运行错误信息

文件src/mpi/datatype/typerep/src/typerep_yaksa_pack.c第286行断言失败:FALSE
memcpy参数内存范围重叠,dst_=0x7f56ed5be010 src_=0x7f56ed5be010 len_=1557836
节点0上终止(1):内部错误


错误原因分析

  1. 内存重叠直接触发错误:
    根进程中convolve函数返回了传入的image_out指针,导致local_image和image_out指向同一块内存。调用MPI_Gather时,源缓冲区和目标缓冲区内存地址重叠,MPI执行memcpy时触发断言失败。

  2. convolve函数逻辑错误:

    • 函数直接修改全局的image_out而非局部缓冲区local_image,完全违背分块处理的设计,所有进程的计算结果会互相覆盖。
    • 用全局图像的第一个像素初始化max_val和min_val,而非当前进程计算的卷积结果,导致归一化计算完全错误。
  3. 分块与边界处理问题:

    • 卷积需要访问当前行的上下相邻行,当前分块逻辑未处理跨进程的边界数据共享,边缘行计算会出错。
    • 所有进程都执行write_image,多个进程同时写入同一文件会导致内容混乱。
  4. 内存泄漏:
    malloc分配的local_image指针被convolve的返回值覆盖,导致这块内存无法释放。


修复步骤

1. 修正convolve函数,写入局部缓冲区

修改函数逻辑,让结果写入局部缓冲区,同时用当前进程的计算值初始化局部最大最小值:

int* convolve(int *image_in, int *local_image, int kernel[3][3], int width, int first_row, int last_row) {
    int i, j, x, y;
    int sum, max_val, min_val;
   
    // 用当前进程第一个有效计算值初始化max/min
    sum = 0;
    for (x = -1; x <= 1; x++) {
        for (y = -1; y <= 1; y++) {
            sum += kernel[x+1][y+1] * image_in[(first_row+x)*width + (1+y)];
        }
    }
    max_val = sum;
    min_val = sum;

    // 计算卷积并更新局部极值
    for (i = first_row; i < last_row; i++) {
        for (j = 1; j < width - 1; j++) {
            sum = 0;
            for (x = -1; x <= 1; x++) {
                for (y = -1; y <= 1; y++) {
                    sum += kernel[x+1][y+1] * image_in[(i+x)*width + (j+y)];
                }
            }
            if (sum > max_val) max_val = sum;
            if (sum < min_val) min_val = sum;
            // 写入局部缓冲区,按局部行偏移计算索引
            local_image[(i - first_row)*width + j] = sum;
        }
    }

    // 局部归一化
    for (i = first_row; i < last_row; i++) {
        for (j = 1; j < width - 1; j++) {
            int local_idx = (i - first_row)*width + j;
            local_image[local_idx] = 255 * (local_image[local_idx] - min_val) / (max_val - min_val);           
        }
    }
    return local_image;
}

2. 修正主函数的缓冲区与MPI调用

  • 确保每个进程使用独立的局部缓冲区,根进程单独分配全局结果缓冲区
  • 修正MPI_Gather参数,避免内存重叠
  • 仅让根进程执行结果写入和时间打印
  • 处理图像高度无法被进程数整除的情况
int main(int argc, char *argv[])
{
    int i, j, h, w, levels ;
    struct timeval tdeb, tfin;
    int *image_in, *image_out = NULL;
    int my_rank, num_procs;
    
    MPI_Init(&argc, &argv);
    MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
    MPI_Comm_rank(MPI_COMM_WORLD, &my_rank);

    image_in = read_image(IMAGE_IN, &h, &w, &levels); 
    int chunk_size = h / num_procs;
    // 处理最后一个进程的剩余行
    int first_row = my_rank * chunk_size;
    int last_row = (my_rank == num_procs - 1) ? h : first_row + chunk_size;
    int actual_chunk_rows = last_row - first_row;

    int kernel[3][3] = {{0,-1,0}, {-1,5,-1}, {0,-1,0}};
    
    // 分配局部缓冲区
    int *local_image = malloc(actual_chunk_rows * w * sizeof(int));
    if (my_rank == 0) {
        image_out = malloc(h * w * sizeof(int));
        // 初始化边缘行(卷积未处理的部分)
        memcpy(image_out, image_in, w*sizeof(int));
        memcpy(image_out + (h-1)*w, image_in + (h-1)*w, w*sizeof(int));
    }

    gettimeofday(&tdeb, NULL);
    
    // 小图像直接用全局图像数据,大图像需用MPI传递边界行
    convolve(image_in, local_image, kernel, w, first_row, last_row); 

    // 按进程顺序收集局部结果到全局缓冲区对应位置
    MPI_Gather(local_image, actual_chunk_rows * w, MPI_INT, 
               image_out + first_row * w, actual_chunk_rows * w, MPI_INT, 
               0, MPI_COMM_WORLD);

    gettimeofday(&tfin, NULL);

    if (my_rank == 0) {
        printf("computation time (microseconds): %ld\n", 
               (tfin.tv_sec - tdeb.tv_sec)*1000000 + (tfin.tv_usec - tdeb.tv_usec));
        write_image(image_out, IMAGE_OUT, h, w, levels);
        free(image_out);
    }

    MPI_Finalize(); 

    free(image_in);
    free(local_image); 

    return 0;
}

3. 优化write_image函数

避免每个像素单独换行,减少文件体积:

void write_image(int *image, char file_name[], int h, int w, int levels) {
    FILE *fout;
    int i, j;

    fout = fopen(file_name, "w");
    if (fout == NULL) {
        printf("file opening error\n");
        exit(-1);
    }
    
    fprintf(fout, "P2\n# test\n%d %d\n%d\n", w, h, levels);

    for (i = 0; i < h; i++) {
        for (j = 0; j < w; j++) {
            fprintf(fout, "%d ", image[i*w + j]);
        }
        fprintf(fout, "\n");
    }

    fclose(fout);
}

4. 边界行处理补充

如果处理大图像,需要通过MPI_Send和MPI_Recv在相邻进程间传递边界行数据,确保每个进程能获取卷积所需的上下行信息。小图像可以保持当前每个进程读取完整图像的方式,简化实现。


内容的提问来源于stack exchange,提问作者Hoang Son

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.24 07:07:10