如何使用MPI发送自定义结构体以并行化图像匹配任务?
问题描述
我定义了以下两个自定义结构体,需要实现每张图像的目标匹配功能,核心逻辑在findMatchesInImage函数中。现在希望用MPI并行化这个函数的调用,让每个进程独立处理一张图像,但MPI更适配结构数组而非数组结构,当前带指针的结构体给数据发送带来了困难。我能接受数据结构修改建议,但现有代码改动成本较高,想知道当前该如何发送数据,曾考虑分两批发送,但不确定是否为最优方案。
自定义结构体定义
struct dataUnit{ int id; int N; int *data; } typedef dataUnit; struct dataStruct{ double threshold; int numOfPic; int numOfObj; // int currPic; --> 可能需要 dataUnit *pictures; dataUnit *objects; } typedef dataStruct;
核心匹配函数原型
findMatchesInImage(dataUnit picture, dataUnit *objects, int numOfObj, double thresh)
最小可复现示例(MRE)
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <math.h> struct dataUnit { int id; int N; int *data; } typedef dataUnit; struct dataStruct { double threshold; int numOfPic; int numOfObj; dataUnit *pictures; dataUnit *objects; } typedef dataStruct; struct match { int objID; int i; int j; } typedef match; struct ImageMatches { match matches[3]; int numOfMatches; int picID; } typedef ImageMatches; dataUnit readDataUnit(FILE *fp) { char line[4000]; dataUnit *currDataUnit = malloc(sizeof(dataUnit)); currDataUnit->id = (int)strtol(fgets(line, 4000, fp), NULL, 10); currDataUnit->N = (int)strtol(fgets(line, 4000, fp), NULL, 10); int totalSize = currDataUnit->N * currDataUnit->N; currDataUnit->data = malloc(sizeof(int) * totalSize); for (int i = 0; i < currDataUnit->N; i++) { int j = 0; fgets(line, 4000, fp); char *token = strtok(line, " "); while (token && j < currDataUnit->N) { currDataUnit->data[i * currDataUnit->N + j] = (int)strtol(token, NULL, 10); token = strtok(NULL, " "); j++; } } return *currDataUnit; } void readData(char *path, dataStruct *data) { FILE *fp; char line[4000]; fp = fopen(path, "r"); if (fp == NULL) exit(EXIT_FAILURE); int firstLine = 1; int numOfPic = 0, numOfObj = 0; while (fgets(line, 4000, fp) != NULL) { if (firstLine) { data->threshold = strtod(line, NULL); firstLine = 0; } else if (numOfPic == 0) { numOfPic = (int)strtol(line, NULL, 10); data->numOfPic = numOfPic; data->pictures = malloc(sizeof(dataUnit) * numOfPic); for (int i = 0; i < numOfPic; i++) { data->pictures[i] = readDataUnit(fp); } } else if (numOfObj == 0) { numOfObj = (int)strtol(line, NULL, 10); data->numOfObj = numOfObj; data->objects = malloc(sizeof(dataUnit) * numOfObj); for (int i = 0; i < numOfObj; i++) { data->objects[i] = readDataUnit(fp); } } } fclose(fp); } int findMatchInArea(dataUnit *picture, dataUnit *object, int i, int j, double thresh) { double score = 0; int totalSize = (object->N) * (object->N); for (int idx = 0; idx < totalSize; idx++) { int row = idx / object->N; int col = idx % object->N; int picVal = picture->data[(row + i) * picture->N + (col + j)]; int objVal = object->data[row * object->N + col]; score += fabs((picVal - objVal) / (double)picVal); } return (score / totalSize) <= thresh ? 1 : 0; } ImageMatches *findMatchesInImage(dataUnit *picture, dataUnit *objects, int numOfObj, double thresh) { ImageMatches *imageMatches = calloc(1, sizeof(ImageMatches)); imageMatches->picID = picture->id; for (int i = 0; i < numOfObj; i++) { int windowsLength = picture->N - objects[i].N + 1; int numOfWindows = windowsLength * windowsLength; for (int j = 0; j < numOfWindows; j++) { if (findMatchInArea(picture, &(objects[i]), j / windowsLength, j % windowsLength, thresh)) { imageMatches->matches[imageMatches->numOfMatches].objID = objects[i].id; imageMatches->matches[imageMatches->numOfMatches].i = j / windowsLength; imageMatches->matches[imageMatches->numOfMatches].j = j % windowsLength; imageMatches->numOfMatches++; } } if (imageMatches->numOfMatches == 3) { return imageMatches; } } return imageMatches; } int main(int argc, char *argv[]) { dataStruct data; readData(argv[1], &data); ImageMatches imageMatches[data.numOfPic]; for (int i = 0; i < data.numOfPic; i++) { imageMatches[i] = *findMatchesInImage(&(data.pictures[i]), data.objects, data.numOfObj, data.threshold); } return 0; }
输入文件示例
0.1 2 1 5 42 68 35 1 70 34 16 40 59 5 10 17 36 52 1 77 30 69 93 26 74 71 64 69 93 2 4 85 94 95 6 97 59 87 21 29 32 42 24 6 40 52 74 1 1 3 19 79 23 10 40 66 55 37 59
输入字段说明
0.1:匹配阈值threshold2:图像总数number of images1/2:图像ID,其后为图像尺寸(5/4)及对应图像数据1:目标总数number of objects1:目标ID,其后为目标尺寸及对应目标数据
数据发送方案(MPI并行化)
因为你的dataUnit包含指针类型,MPI无法直接发送指针指向的堆内存,必须拆分数据分步骤发送,这是最直接且无需大幅修改现有代码的方案,具体步骤如下:
1. 初始化MPI,主进程读取数据
主进程(rank=0)负责读取所有数据,其他进程等待接收数据。
2. 广播公共数据给所有进程
所有进程都需要用到threshold、numOfObj以及所有objects数据,主进程先广播这些公共信息:
- 先广播
threshold(MPI_DOUBLE类型)和numOfObj(MPI_INT类型) - 对每个
object,分两步发送:- 发送该
object的id和N(两个MPI_INT) - 发送该
object的data数组(共N*N个MPI_INT)
- 发送该
- 子进程接收时,先根据
N分配内存,再接收data数组
3. 主进程分发图像数据给子进程
主进程给每个子进程分配一张图像(如果进程数多于图像数,部分进程可闲置或处理多图):
- 对目标进程,先发送该图像的
id和N(两个MPI_INT) - 再发送图像的
data数组(共N*N个MPI_INT) - 子进程接收后,分配内存存储
data,构造本地dataUnit结构体
4. 子进程执行匹配任务
子进程调用findMatchesInImage处理分配到的图像,生成ImageMatches结果。
5. 子进程返回结果给主进程
ImageMatches是固定大小的结构体(无指针),可以直接用MPI_Send发送,主进程用MPI_Recv接收并汇总结果。
关键代码片段示例
主进程发送单个dataUnit
// 发送dataUnit给指定进程 void send_dataUnit(dataUnit unit, int dest_rank) { MPI_Send(&unit.id, 1, MPI_INT, dest_rank, 0, MPI_COMM_WORLD); MPI_Send(&unit.N, 1, MPI_INT, dest_rank, 0, MPI_COMM_WORLD); MPI_Send(unit.data, unit.N*unit.N, MPI_INT, dest_rank, 0, MPI_COMM_WORLD); }
子进程接收dataUnit
// 从主进程接收dataUnit dataUnit recv_dataUnit() { dataUnit unit; MPI_Recv(&unit.id, 1, MPI_INT, 0, 0, MPI_COMM_WORLD, MPI_STATUS_IGNORE); MPI_Recv(&unit.N, 1, MPI_INT, 0, 0, MPI_COMM_WORLD, MPI_STATUS_IGNORE); unit.data = malloc(sizeof(int) * unit.N * unit.N); MPI_Recv(unit.data, unit.N*unit.N, MPI_INT, 0, 0, MPI_COMM_WORLD, MPI_STATUS_IGNORE); return unit; }
发送/接收ImageMatches
// 子进程发送结果 ImageMatches *result = findMatchesInImage(&picture, objects, numOfObj, threshold); MPI_Send(result, sizeof(ImageMatches), MPI_BYTE, 0, 1, MPI_COMM_WORLD); // 主进程接收结果 ImageMatches recv_result; MPI_Recv(&recv_result, sizeof(ImageMatches), MPI_BYTE, source_rank, 1, MPI_COMM_WORLD, MPI_STATUS_IGNORE);
方案说明
这种分步骤发送的方式完全适配你现有的数据结构,无需修改结构体定义,仅需添加MPI通信的封装函数。如果图像数量远多于进程数,主进程可以采用动态任务分配(比如用MPI_Sendrecv或MPI_Request实现非阻塞通信),但基础的分批次发送已经能满足需求,是当前场景下的最优选择之一。
内容的提问来源于stack exchange,提问作者RedYoel
相关产品推荐
相关产品推荐

