单Context下多GPU的clCreateBuffer内存分配失败问题排查
问题描述
我有两块Intel ARC A770 GPU,运行测试程序时遇到如下问题:测试程序通过clCreateBuffer分配1MB内存缓冲区,仅创建一个OpenCL Context,可选择关联1或2块GPU。单GPU模式下可成功分配超10000次,但双GPU模式下仅分配约1000次(共1GB)就触发主机内存不足错误,OpenCL错误码为-6;若为每块GPU创建独立Context则可正常分配。GPU单块显存16GB,主机内存192GB,请问问题出在哪里?
编译命令
gcc -D CL_TARGET_OPENCL_VERSION=220 -g -Wall -o OpenCLMulti OpenCLMulti.cpp -lOpenCL -lm
测试代码
OpenCLMulti.cpp
#include <sys/time.h> #include <sys/sysinfo.h> #include <sys/stat.h> #include <assert.h> #include <errno.h> #include <math.h> #include <getopt.h> #include <sys/time.h> #include <stdlib.h> #include <stdio.h> #include <string.h> #include <CL/cl.h> #include <clBLAS.h> #define MAX_PLATFORMS 1 #define MAX_GPUS 2 cl_context context; cl_program program; cl_device_id devices[MAX_GPUS] = {0}; cl_platform_id platformId[MAX_PLATFORMS]; const char* kernelFileName = "OpenCLKernels.cl"; unsigned char* getKernelCode(const char* filename, size_t* psize) { FILE* fp; size_t ret, size; unsigned char* bCode; fp = fopen(filename, "rb"); if (fp == NULL) { fprintf(stderr, "Could not open kernels source file: %s", filename); return NULL; } fseek(fp, 0, SEEK_END); size = ftell(fp); rewind(fp); bCode = (unsigned char*) malloc(size); if ((ret = fread(bCode, 1, size, fp)) != size) { fprintf(stderr, "Could not read %ld bytes, got %ld", ret, size); return NULL; } fclose(fp); *psize = size; return bCode; } const char* buildOptions = "-DOPENCL -D CL_TARGET_OPENCL_VERSION=220"; void contextCallback(const char* errInfo, const void* privateInfo, size_t cb, void* userData) { printf("contextCallback %s\n", errInfo); } cl_int setupGPU(int gpus, int* ngpus) { size_t size; cl_int err = CL_SUCCESS; char buffer[16384]; cl_uint numPlatforms, ngpusOCL; unsigned char* bCode; err = clGetPlatformIDs(MAX_PLATFORMS, &platformId[0], &numPlatforms); if (err != CL_SUCCESS) { return err; } if (numPlatforms == 0) { fprintf(stderr, "No OpenCL platforms found\n"); return -100; } if (numPlatforms > 1) { fprintf(stderr, "Found more than one OpenCL platform. Choosing first.\n"); } err = clGetPlatformInfo(platformId[0], CL_PLATFORM_PROFILE, sizeof(buffer), buffer, &size); fprintf(stderr, "Platform profile: %s\n", buffer); err = clGetPlatformInfo(platformId[0], CL_PLATFORM_VERSION, sizeof(buffer), buffer, &size); fprintf(stderr, "Platform version: %s\n", buffer); err = clGetPlatformInfo(platformId[0], CL_PLATFORM_NAME, sizeof(buffer), buffer, &size); fprintf(stderr, "Name: %s\n", buffer); cl_context_properties properties[] = {CL_CONTEXT_PLATFORM, (cl_context_properties) platformId[0], 0}; err = clGetDeviceIDs(platformId[0], CL_DEVICE_TYPE_GPU, MAX_GPUS, &devices[0], &ngpusOCL); if (err != CL_SUCCESS) { return err; } fprintf(stderr, "Found %d devices\n", ngpusOCL); bCode = getKernelCode(kernelFileName, &size); if (bCode == NULL) { fprintf(stderr, "Could not load kernel code\n"); return -1; } context = clCreateContext(properties, gpus, devices, contextCallback, NULL, &err); if (err != CL_SUCCESS) { fprintf(stderr, "Could not create GPU context: %d\n", err); return err; } fprintf(stderr, "%d: OpenCL context created successfully\n", 0); program = clCreateProgramWithSource(context, 1, (const char**) &bCode, &size, &err); if (program == NULL) { fprintf(stderr, "Could not create program from file %s, error %d\n", kernelFileName, err); return err; } fprintf(stderr, "%d: Program loaded successfully\n", 0); err = clBuildProgram(program, gpus, devices, buildOptions, NULL, NULL); if (err != CL_SUCCESS) { fprintf(stderr, "%d: Program build failed %d\n", 0, err); size = 1024 * 1024 * 4; char* output = (char*) malloc(size); err = clGetProgramBuildInfo(program, devices[0], CL_PROGRAM_BUILD_LOG, sizeof(1024 * 1024 * 4), output, &size); if (err != CL_SUCCESS) { fprintf(stderr, "%d: Program build failed. Unable to get log. Error %d\n", 0, err); return err; } fprintf(stderr, "%d: %s\n", 0, output); free(output); return err; } free(bCode); fprintf(stderr, "OpenCL kernels built successfully\n"); *ngpus = ngpusOCL; return err; } int main(int argc, char* argv[]) { int ngpus, dgpus; cl_int err; size_t asize; cl_mem mem; ngpus = 1; for (int i = 1; i < argc; i++) { if (strcmp(argv[i], "--gpus") == 0) { ngpus = atoi(argv[i + 1]); i++; } } err = setupGPU(ngpus, &dgpus); if (err != CL_SUCCESS) { exit(1); } if (dgpus < ngpus) { fprintf(stderr, "Detected %d GPUs, requested %d. Exiting\n", dgpus, ngpus); exit(1); } asize = 1 * 1024 * 1024L; for (int i = 0; i < 100000; i++) { mem = clCreateBuffer(context, CL_MEM_READ_WRITE, asize, NULL, &err); if (err != CL_SUCCESS) { fprintf(stderr, "Could not create memory, err %d", err); return err; } fprintf(stderr, "%d: Allocated %ld MB successfully\n", i, asize / (1024 * 1024)); } exit(0); exit(0); }
OpenCLKernels.cl
__kernel void dAxpy(int elements, float alpha, __global float *xData, int xOffset, int incX, __global float *yData, int yOffset, int incY) { long idx; int threadId, blockIdx, blockIdy, blockDim, gridDim; threadId = get_local_id(0); blockIdx = get_group_id(0); blockIdy = get_group_id(1); blockDim = get_local_size(0); gridDim = get_num_groups(0); idx = (blockIdx + blockIdy * gridDim) * blockDim + threadId; if (idx < elements) { yData[yOffset + idx * incY] += alpha * xData[xOffset + idx * incX]; } }
问题原因与解决方案
核心原因
当创建关联多块GPU的OpenCL Context时,调用clCreateBuffer且未指定设备绑定属性的情况下,Intel OpenCL运行时会在每块关联GPU上自动创建该缓冲区的副本,同时还会占用额外主机内存用于内存同步、管理等开销。
你的测试中每次分配1MB,双GPU模式下每次实际占用的内存为:1MB(GPU1显存)+1MB(GPU2显存)+主机内存管理开销。当分配1000次时,仅显存就占用2GB,叠加主机内存开销后触发了CL_OUT_OF_HOST_MEMORY(错误码-6)。
而单GPU模式下,缓冲区仅在一块GPU上分配副本,主机内存开销小;为每块GPU创建独立Context时,缓冲区仅绑定到对应GPU,不会跨设备复制,因此可以正常分配。
解决方案
- 明确指定缓冲区绑定设备:创建缓冲区后,通过
clEnqueueMigrateMemObjects将内存迁移到目标设备,避免自动多设备复制:// 创建命令队列(需针对目标设备) cl_command_queue queue = clCreateCommandQueue(context, devices[0], 0, &err); cl_mem mem = clCreateBuffer(context, CL_MEM_READ_WRITE, asize, NULL, &err); // 将内存迁移到指定GPU clEnqueueMigrateMemObjects(queue, 1, &mem, 0, 0, NULL, NULL); - 使用设备本地内存:如果缓冲区仅在某块GPU上使用,标记为
CL_MEM_DEVICE_LOCAL,强制在目标设备显存分配,不占用主机内存(需通过内核或显式拷贝访问):cl_mem mem = clCreateBuffer(context, CL_MEM_READ_WRITE | CL_MEM_DEVICE_LOCAL, asize, NULL, &err); - 调整运行时参数:通过环境变量禁用统一内存自动复制,例如设置
IGC_EnableUnifiedMemory=0,避免运行时自动在多设备间复制缓冲区。 - 优化内存分配策略:复用缓冲区而非频繁创建销毁,减少内存管理开销;或为每个设备单独分配缓冲区,避免跨设备的自动复制逻辑。
内容的提问来源于stack exchange,提问作者jordan
相关产品推荐
相关产品推荐

