You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

单Context下多GPU的clCreateBuffer内存分配失败问题排查

问题描述

我有两块Intel ARC A770 GPU,运行测试程序时遇到如下问题:测试程序通过clCreateBuffer分配1MB内存缓冲区,仅创建一个OpenCL Context,可选择关联1或2块GPU。单GPU模式下可成功分配超10000次,但双GPU模式下仅分配约1000次(共1GB)就触发主机内存不足错误,OpenCL错误码为-6;若为每块GPU创建独立Context则可正常分配。GPU单块显存16GB,主机内存192GB,请问问题出在哪里?

编译命令
gcc -D CL_TARGET_OPENCL_VERSION=220 -g -Wall -o OpenCLMulti OpenCLMulti.cpp -lOpenCL -lm
测试代码

OpenCLMulti.cpp

#include <sys/time.h>
#include <sys/sysinfo.h>
#include <sys/stat.h>
#include <assert.h>
#include <errno.h>
#include <math.h>
#include <getopt.h>
#include <sys/time.h>
#include <stdlib.h>
#include <stdio.h>
#include <string.h>

#include <CL/cl.h>
#include <clBLAS.h>

#define MAX_PLATFORMS 1
#define MAX_GPUS 2

cl_context context;
cl_program program;
cl_device_id devices[MAX_GPUS] = {0};
cl_platform_id platformId[MAX_PLATFORMS];

const char* kernelFileName = "OpenCLKernels.cl";

unsigned char* getKernelCode(const char* filename, size_t* psize) {
  FILE* fp;
  size_t ret, size;
  unsigned char* bCode;

  fp = fopen(filename, "rb");
  if (fp == NULL) {
    fprintf(stderr, "Could not open kernels source file: %s", filename);
    return NULL;
  }

  fseek(fp, 0, SEEK_END);
  size = ftell(fp);
  rewind(fp);

  bCode = (unsigned char*) malloc(size);

  if ((ret = fread(bCode, 1, size, fp)) != size) {
    fprintf(stderr, "Could not read %ld bytes, got %ld", ret, size);
    return NULL;
  }
  fclose(fp);
  *psize = size;
  return bCode;
}

const char* buildOptions = "-DOPENCL -D CL_TARGET_OPENCL_VERSION=220";

void contextCallback(const char* errInfo, const void* privateInfo, size_t cb, void* userData) {
  printf("contextCallback %s\n", errInfo);
}

cl_int setupGPU(int gpus, int* ngpus) {
  size_t size;
  cl_int err = CL_SUCCESS;
  char buffer[16384];
  cl_uint numPlatforms, ngpusOCL;
  unsigned char* bCode;

  err = clGetPlatformIDs(MAX_PLATFORMS, &platformId[0], &numPlatforms);
  if (err != CL_SUCCESS) {
    return err;
  }

  if (numPlatforms == 0) {
    fprintf(stderr, "No OpenCL platforms found\n");
    return -100;
  }

  if (numPlatforms > 1) {
    fprintf(stderr, "Found more than one OpenCL platform. Choosing first.\n");
  }

  err = clGetPlatformInfo(platformId[0], CL_PLATFORM_PROFILE, sizeof(buffer), buffer, &size);
  fprintf(stderr, "Platform profile: %s\n", buffer);

  err = clGetPlatformInfo(platformId[0], CL_PLATFORM_VERSION, sizeof(buffer), buffer, &size);
  fprintf(stderr, "Platform version: %s\n", buffer);

  err = clGetPlatformInfo(platformId[0], CL_PLATFORM_NAME, sizeof(buffer), buffer, &size);
  fprintf(stderr, "Name: %s\n", buffer);

  cl_context_properties properties[] = {CL_CONTEXT_PLATFORM, (cl_context_properties) platformId[0], 0};
  err = clGetDeviceIDs(platformId[0], CL_DEVICE_TYPE_GPU, MAX_GPUS, &devices[0], &ngpusOCL);
  if (err != CL_SUCCESS) {
    return err;
  }
  fprintf(stderr, "Found %d devices\n", ngpusOCL);


  bCode = getKernelCode(kernelFileName, &size);
  if (bCode == NULL) {
    fprintf(stderr, "Could not load kernel code\n");
    return -1;
  }

  context = clCreateContext(properties, gpus, devices, contextCallback, NULL, &err);
  if (err != CL_SUCCESS) {
    fprintf(stderr, "Could not create GPU context: %d\n", err);
    return err;
  }

  fprintf(stderr, "%d: OpenCL context created successfully\n", 0);

  program = clCreateProgramWithSource(context, 1, (const char**) &bCode, &size, &err);
  if (program == NULL) {
    fprintf(stderr, "Could not create program from file %s, error %d\n", kernelFileName, err);
    return err;
  }
  fprintf(stderr, "%d: Program loaded successfully\n", 0);

  err = clBuildProgram(program, gpus, devices, buildOptions, NULL, NULL);
  if (err != CL_SUCCESS) {
    fprintf(stderr, "%d: Program build failed %d\n", 0, err);
    size = 1024 * 1024 * 4;
    char* output = (char*) malloc(size);
    err = clGetProgramBuildInfo(program, devices[0], CL_PROGRAM_BUILD_LOG, sizeof(1024 * 1024 * 4), output, &size);
    if (err != CL_SUCCESS) {
      fprintf(stderr, "%d: Program build failed. Unable to get log. Error %d\n", 0, err);
      return err;
    }
    fprintf(stderr, "%d: %s\n", 0, output);
    free(output);
    return err;
  }
  free(bCode);

  fprintf(stderr, "OpenCL kernels built successfully\n");
  *ngpus = ngpusOCL;
  return err;
}

int main(int argc, char* argv[]) {
  int ngpus, dgpus;
  cl_int err;
  size_t asize;
  cl_mem mem;

  ngpus = 1;
  for (int i = 1; i < argc; i++) {
    if (strcmp(argv[i], "--gpus") == 0) {
      ngpus = atoi(argv[i + 1]);
      i++;
    }
  }
  err = setupGPU(ngpus, &dgpus);

  if (err != CL_SUCCESS) {
    exit(1);
  }

  if (dgpus < ngpus) {
    fprintf(stderr, "Detected %d GPUs, requested %d. Exiting\n", dgpus, ngpus);
    exit(1);
  }

  asize = 1 * 1024 * 1024L;
  for (int i = 0; i < 100000; i++) {
    mem = clCreateBuffer(context, CL_MEM_READ_WRITE, asize, NULL, &err);
    if (err != CL_SUCCESS) {
      fprintf(stderr, "Could not create memory, err %d", err);
      return err;
    }

    fprintf(stderr, "%d: Allocated %ld MB successfully\n", i, asize / (1024 * 1024));
  }

  exit(0);

  exit(0);
}

OpenCLKernels.cl

__kernel void dAxpy(int elements, float alpha, __global float *xData, int xOffset, int incX, __global float *yData, int yOffset, int incY) {
  long idx;
  int threadId, blockIdx, blockIdy, blockDim, gridDim;

  threadId = get_local_id(0); blockIdx = get_group_id(0); blockIdy = get_group_id(1); blockDim = get_local_size(0); gridDim = get_num_groups(0);
  idx = (blockIdx + blockIdy * gridDim) * blockDim + threadId;
  if (idx < elements) {
    yData[yOffset + idx * incY] += alpha * xData[xOffset + idx * incX];
  }
}
问题原因与解决方案

核心原因

当创建关联多块GPU的OpenCL Context时,调用clCreateBuffer且未指定设备绑定属性的情况下,Intel OpenCL运行时会在每块关联GPU上自动创建该缓冲区的副本,同时还会占用额外主机内存用于内存同步、管理等开销。

你的测试中每次分配1MB,双GPU模式下每次实际占用的内存为:1MB(GPU1显存)+1MB(GPU2显存)+主机内存管理开销。当分配1000次时,仅显存就占用2GB,叠加主机内存开销后触发了CL_OUT_OF_HOST_MEMORY(错误码-6)。

而单GPU模式下,缓冲区仅在一块GPU上分配副本,主机内存开销小;为每块GPU创建独立Context时,缓冲区仅绑定到对应GPU,不会跨设备复制,因此可以正常分配。

解决方案

  1. 明确指定缓冲区绑定设备:创建缓冲区后,通过clEnqueueMigrateMemObjects将内存迁移到目标设备,避免自动多设备复制:
    // 创建命令队列(需针对目标设备)
    cl_command_queue queue = clCreateCommandQueue(context, devices[0], 0, &err);
    cl_mem mem = clCreateBuffer(context, CL_MEM_READ_WRITE, asize, NULL, &err);
    // 将内存迁移到指定GPU
    clEnqueueMigrateMemObjects(queue, 1, &mem, 0, 0, NULL, NULL);
    
  2. 使用设备本地内存:如果缓冲区仅在某块GPU上使用,标记为CL_MEM_DEVICE_LOCAL,强制在目标设备显存分配,不占用主机内存(需通过内核或显式拷贝访问):
    cl_mem mem = clCreateBuffer(context, CL_MEM_READ_WRITE | CL_MEM_DEVICE_LOCAL, asize, NULL, &err);
    
  3. 调整运行时参数:通过环境变量禁用统一内存自动复制,例如设置IGC_EnableUnifiedMemory=0,避免运行时自动在多设备间复制缓冲区。
  4. 优化内存分配策略:复用缓冲区而非频繁创建销毁,减少内存管理开销;或为每个设备单独分配缓冲区,避免跨设备的自动复制逻辑。

内容的提问来源于stack exchange,提问作者jordan

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.22 19:24:57