You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Open MPI并行编程报错MPI_ERR_BUFFER:无效缓冲区指针求助

并行版Rabin-Karp算法MPI_ERR_BUFFER错误排查

问题场景

大学备考Open MPI并行编程考试,实现并行版Rabin-Karp算法时,主进程(rank 0)读取文件后向工作进程发送数据,触发MPI_ERR_BUFFER: invalid buffer pointer运行时错误。

错误信息

An error occurred in MPI_Recv
reported by process [724303873,11]
on communicator MPI_COMM_WORLD
MPI_ERR_BUFFER: invalid buffer pointer
MPI_ERRORS_ARE_FATAL (processes in this communicator will now abort,
and MPI will try to terminate your MPI job as well)

问题代码

//Compute the number of executor processes//
int activeCores(long long int txt_size, long long int pat_size, int cores){

    int executors = cores;

    for (int i = 0; i < cores; ++i)
    {
        if (txt_size/executors < pat_size)
        {
            executors = cores - i;
        } else {
            break;
        }
    }
    return executors;
}

int main(int argc, char *argv[])
{
    
    //SOME VARIABLES...
    int rank;
    int size;
    //long long int patlen;
    char *text;
    char *pattern;
    long long int occurrences = 0;
    long long int total_occ = 0;
    int ex_size;
    int active = 0;                         // indicates if a core is active or not. Inititally all cores are disabled
    MPI_File text_handler;
    MPI_File pattern_handler;
    MPI_Offset txt_size;    //total dimension of the file
    MPI_Offset pat_size;
    char *chunk;
    long long int chunk_length;

    MPI_Init(&argc, &argv);
    MPI_Comm_rank(MPI_COMM_WORLD, &rank);
    MPI_Comm_size(MPI_COMM_WORLD, &size);
    int flags[size];

    clock_t begin = clock();

//reading routine

    if (rank == 0)
    {
        MPI_File_open(MPI_COMM_SELF, argv[1], MPI_MODE_RDONLY, MPI_INFO_NULL, &text_handler);
        MPI_File_open(MPI_COMM_SELF, argv[2], MPI_MODE_RDONLY, MPI_INFO_NULL, &pattern_handler);
        MPI_File_get_size(pattern_handler, &pat_size);
        MPI_File_get_size(text_handler, &txt_size);

        pattern = (char *)malloc(sizeof(char)*pat_size + 1);
        text = (char *)malloc(sizeof(char)*txt_size + 1);

        MPI_File_read(pattern_handler, pattern, pat_size, MPI_CHAR, MPI_STATUS_IGNORE);
        MPI_File_read(text_handler, text, txt_size, MPI_CHAR, MPI_STATUS_IGNORE);
        
        text[txt_size] = '\0';
        pattern[pat_size] = '\0';

        MPI_File_close(&text_handler);
        MPI_File_close(&pattern_handler);

        ex_size = activeCores(txt_size, pat_size, size);

        if (ex_size == 0)
        {
            printf("The length of the pattern is bigger than the dimension of the text\n");
            MPI_Finalize();
            exit(-1);
        }

        for (int i = 0; i < size; ++i)
        {
            if (i<ex_size)
            {
                flags[i] = 1;
            } else flags[i] = 0;
        }

        chunk_length = txt_size/ex_size;

    }

    //send flag to all processes
    MPI_Scatter(flags, 1, MPI_INT, &active, 1, MPI_INT, 0, MPI_COMM_WORLD);
    MPI_Bcast(&chunk_length, 1, MPI_INT, 0, MPI_COMM_WORLD);
    MPI_Bcast(&txt_size, 1, MPI_INT, 0, MPI_COMM_WORLD);
    MPI_Bcast(&ex_size, 1, MPI_INT, 0, MPI_COMM_WORLD);

// splitting text routine

    if (rank == 0 && active == 1)
    {
        chunk = (char *)malloc(sizeof(char) * (chunk_length +1));
        chunk = strncpy(chunk, text, chunk_length);
        chunk[chunk_length]='\0';

        long long int start = 0;
        long long int stop;

        for (int i = 1; i < ex_size; ++i)
        {
            if (i<ex_size-1)
            {
                MPI_Send(&text[chunk_length*i], chunk_length, MPI_CHAR, i, 1, MPI_COMM_WORLD);
            } else if (i == ex_size-1)
            {
                MPI_Send(&text[chunk_length*i], txt_size-i*chunk_length, MPI_CHAR, i, 2, MPI_COMM_WORLD);
            }
        }
    }

    if (rank > 0 && rank < ex_size-1)
    {
        chunk = (char *)malloc(sizeof(char)*chunk_length);
        MPI_Recv(chunk, chunk_length, MPI_CHAR, 0, 1, MPI_COMM_WORLD, MPI_STATUS_IGNORE);
        chunk[chunk_length]='\0';
    }

    if (rank == ex_size-1)
    {
        chunk = (char *)malloc(sizeof(char)*(txt_size - rank*chunk_length +1));
        chunk[txt_size - rank*chunk_length]='\0';
        MPI_Recv(chunk, txt_size-rank*chunk_length, MPI_CHAR, 0, 2, MPI_COMM_WORLD, MPI_STATUS_IGNORE);
    }

    
    //CLOSE THE MPI LAYER
    MPI_Finalize();

    //STOP COUNTING CLOCK CYCLES
    clock_t end = clock();
    
    if (rank == 0)
    {
        //COMPUTING THE TIME SPENT
        double time_spent = (double)(end - begin) / CLOCKS_PER_SEC;
        printf("Total occurrences %d\n", total_occ);
        printf("time required %lf\n", time_spent);
        printf("Program executed by %d cores over %d\n", ex_size, size);

    }

    return 0;
}

错误原因分析

  1. MPI数据类型不匹配:txt_size(MPI_Offset通常为long long类型)和chunk_length是long long int,但MPI_Bcast调用时用了MPI_INT类型参数,导致非主进程接收到错误的数值。错误的数值会引发malloc分配的内存大小异常,甚至出现无效指针,触发MPI_ERR_BUFFER。
  2. 非活跃进程未正确处理:rank >= ex_size的进程active标志为0,但代码未跳过后续逻辑,这类进程的chunk指针未初始化,可能引发未定义行为。
  3. 内存越界操作:rank > 0 && rank < ex_size-1的进程中,malloc仅分配了chunk_length字节,但后续执行chunk[chunk_length]='\0',访问了超出分配范围的内存,破坏堆结构,导致指针失效。
  4. 冗余且风险的字符串拷贝:主进程中chunk = strncpy(chunk, text, chunk_length)的写法冗余,strncpy返回目标指针,此处重新赋值无意义;若text长度不足,strncpy不会自动添加终止符(虽然后续手动添加了,但写法存在风险)。

修复方案

  1. 修正MPI_Bcast的数据类型:将txt_size和chunk_length的广播类型改为MPI_LONG_LONG_INT,匹配变量类型:
    MPI_Bcast(&chunk_length, 1, MPI_LONG_LONG_INT, 0, MPI_COMM_WORLD);
    MPI_Bcast(&txt_size, 1, MPI_LONG_LONG_INT, 0, MPI_COMM_WORLD);
    
  2. 处理非活跃进程:在广播后添加判断,非活跃进程直接终止MPI并退出:
    // 广播完成后立即处理非活跃进程
    if (active == 0) {
        MPI_Finalize();
        return 0;
    }
    
  3. 修复内存越界:为需要添加终止符的chunk多分配1字节内存:
    if (rank > 0 && rank < ex_size-1)
    {
        chunk = (char *)malloc(sizeof(char)*(chunk_length + 1));
        MPI_Recv(chunk, chunk_length, MPI_CHAR, 0, 1, MPI_COMM_WORLD, MPI_STATUS_IGNORE);
        chunk[chunk_length]='\0';
    }
    
  4. 优化字符串拷贝逻辑:主进程中去掉冗余的指针赋值,直接使用strncpy:
    chunk = (char *)malloc(sizeof(char) * (chunk_length +1));
    strncpy(chunk, text, chunk_length);
    chunk[chunk_length]='\0';
    

内容的提问来源于stack exchange,提问作者VITO GIACALONE

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.29 14:07:33