You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用MPI_Type_vector实现MPI_Gather/MPI_Gatherv收集一维存储矩阵列

问题:MPI_Type_vector收集列数据时无法正确交替排列

我尝试用MPI_Type_vector把各进程中行优先一维存储的矩阵列发送到进程0,但数据没法正确交替排列。示例代码里每个进程分配等量列,用MPI_Gather实现,但输出不符合预期;实际需求是给不同进程分配任意数量的列,要用MPI_Gatherv,不确定能不能轻松实现非均匀的交替数据收集。

示例代码

#include <mpi.h>
#include <iostream>
#include <vector>
#include <string>
using namespace std;

int main(int argc, char** argv) {
    int worldRank;
    int WorldSize;
    MPI_Datatype column_send_type;
    MPI_Datatype column_send_type1;
    MPI_Datatype b_col_type;
    MPI_Datatype new_b_col_type;
    MPI_Init(NULL, NULL);
    MPI_Comm_rank(MPI_COMM_WORLD, &worldRank);
    MPI_Comm_size(MPI_COMM_WORLD, &WorldSize);
    vector <int> send (3*2);
    string stringRank = to_string(worldRank+1);
    for(int i=0; i < send.size(); i++) {
        send[i] = stoi(stringRank + to_string(i));
    }
    vector <int> rcv (3*8);
    MPI_Type_vector(3, 2, 2, MPI_INT, &column_send_type);
    MPI_Type_commit(&column_send_type);

    MPI_Type_create_resized(column_send_type, 0, 6 *sizeof(int), &column_send_type1);
    MPI_Type_commit(&column_send_type1); 

    if (worldRank == 0)
    {
        MPI_Type_vector(3, 1, 8, MPI_INT, &b_col_type);
        MPI_Type_commit(&b_col_type);

        MPI_Type_create_resized(b_col_type, 0, 1*sizeof(int), &new_b_col_type);
        MPI_Type_commit(&new_b_col_type);
    }

    MPI_Gather(send.data(), 1, column_send_type1, rcv.data(), 2, new_b_col_type, 0, MPI_COMM_WORLD);
    if (worldRank == 0) {
        for(int i =0; i< rcv.size(); i++)
        {
            cout << rcv[i] << " ";
        }
    }

    MPI_Finalize();
}

当前输出(mpirun -np 4 testGatherCol2)

10 13 20 23 30 33 40 43 11 14 21 24 31 34 41 44 12 15 22 25 32 35 42 45

预期输出

10 11 20 21 30 31 40 41 12 13 22 23 32 33 42 43 14 15 24 25 34 35 44 45
解决方案

核心问题分析

代码中发送端与接收端的自定义数据类型逻辑不匹配,导致数据排布不符合交替列需求:

  • 发送端column_send_type1设置的extent为6*sizeof(int),但接收端类型未对应这种交替排列的逻辑。
  • 接收端需要按进程交替存放列数据:即进程0的第1列、进程1的第1列、进程2的第1列、进程3的第1列,接着是进程0的第2列、进程1的第2列...以此类推。

修正后的代码(适配MPI_Gather均匀列分配)

#include <mpi.h>
#include <iostream>
#include <vector>
#include <string>
using namespace std;

int main(int argc, char** argv) {
    int worldRank;
    int worldSize;
    MPI_Datatype send_col_type;
    MPI_Datatype send_col_type_resized;
    MPI_Datatype recv_col_type;
    MPI_Datatype recv_col_type_resized;
    MPI_Init(NULL, NULL);
    MPI_Comm_rank(MPI_COMM_WORLD, &worldRank);
    MPI_Comm_size(MPI_COMM_WORLD, &worldSize);

    const int rows = 3;
    const int cols_per_proc = 2;
    const int total_cols = worldSize * cols_per_proc;

    vector<int> send(rows * cols_per_proc);
    string rank_str = to_string(worldRank + 1);
    for (int i = 0; i < send.size(); ++i) {
        send[i] = stoi(rank_str + to_string(i));
    }

    vector<int> recv(rows * total_cols);

    // 定义发送端类型:描述当前进程存储的cols_per_proc列(行优先)
    MPI_Type_vector(rows, cols_per_proc, cols_per_proc, MPI_INT, &send_col_type);
    MPI_Type_commit(&send_col_type);
    // 调整extent,确保发送时的连续逻辑
    MPI_Type_create_resized(send_col_type, 0, cols_per_proc * sizeof(int), &send_col_type_resized);
    MPI_Type_commit(&send_col_type_resized);

    if (worldRank == 0) {
        // 定义接收端类型:描述交替排列的列(每行取worldSize个元素,步长为总列数)
        MPI_Type_vector(rows, worldSize, total_cols, MPI_INT, &recv_col_type);
        MPI_Type_commit(&recv_col_type);
        // 调整extent,匹配交替排列的步长
        MPI_Type_create_resized(recv_col_type, 0, worldSize * sizeof(int), &recv_col_type_resized);
        MPI_Type_commit(&recv_col_type_resized);

        MPI_Gather(send.data(), 1, send_col_type_resized, 
                   recv.data(), cols_per_proc, recv_col_type_resized, 
                   0, MPI_COMM_WORLD);

        // 释放接收端类型
        MPI_Type_free(&recv_col_type);
        MPI_Type_free(&recv_col_type_resized);
    } else {
        MPI_Gather(send.data(), 1, send_col_type_resized, 
                   nullptr, 0, MPI_DATATYPE_NULL, 
                   0, MPI_COMM_WORLD);
    }

    if (worldRank == 0) {
        for (int val : recv) {
            cout << val << " ";
        }
        cout << endl;
    }

    // 释放发送端类型
    MPI_Type_free(&send_col_type);
    MPI_Type_free(&send_col_type_resized);

    MPI_Finalize();
    return 0;
}

扩展到MPI_Gatherv的非均匀列分配

如果要支持不同进程分配不同数量的列,只需做以下调整:

  1. 每个进程定义自己的cols_per_proc(可通过广播或预先约定)。
  2. 计算recvcounts数组:每个元素对应进程发送的自定义类型数量(即该进程的列数)。
  3. 计算displs数组:每个进程的数据在接收缓冲区中的偏移量(以自定义类型为单位)。
  4. 接收端的自定义类型需根据总列数动态调整,或针对每一列单独定义类型(更灵活)。

核心逻辑:无论列数是否均匀,接收端类型要确保每个进程的列数据被放到正确的交替位置,发送端类型要准确描述自身存储的列数据。

内容的提问来源于stack exchange,提问作者smark

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 07:39:56