You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

多GPU环境下CUDA主机与设备报告设备ID不一致问题排查

多GPU+OpenMP环境下内核仅在设备0运行的问题解决

问题描述

在HPC集群上使用OpenMP多线程绑定多GPU仿真时,主机线程已正确设置不同设备ID,但所有CUDA内核均仅在设备0上执行。以下是最小复现示例:

复现代码

#include <cuda_runtime.h>
#include <iostream>
#include <omp.h>

// Kernel to print the device ID from the GPU
__global__ void printGPUDeviceID() {
    int deviceID;
    cudaGetDevice(&deviceID);  // Get the current device ID
    printf("Device ID from the kernel: %d\n", deviceID);
}

int main() {
    // Get the number of available devices
    int num_devices;
    cudaGetDeviceCount(&num_devices);
    if (num_devices < 2) {
        std::cout << "This example requires at least two GPUs." << std::endl;
        return 1;
    }

    // Use OpenMP to create threads for each GPU
    #pragma omp parallel num_threads(num_devices)
    {
        int thread_id = omp_get_thread_num();  // Get the OpenMP thread ID
        int device_id = thread_id;             // Assign one device per thread

        // Set the current device for this thread
        cudaSetDevice(device_id);

        // Get and print the device ID from the host
        int deviceIDFromHost;
        cudaGetDevice(&deviceIDFromHost);
        printf("Device ID from the host (thread %d): %d\n", thread_id, deviceIDFromHost);

        // Launch a kernel to print the device ID from the GPU
        printGPUDeviceID<<<1, 1>>>();

        // Wait for the GPU to finish
        cudaDeviceSynchronize();

        // Check for any errors during kernel execution
        cudaError_t err = cudaGetLastError();
        if (err != cudaSuccess) {
            printf("CUDA error on device %d: %s\n", device_id, cudaGetErrorString(err));
        }
    }

    return 0;
}

预期输出

Device ID from the host (thread 0): 0
Device ID from the kernel: 0
Device ID from the host (thread 1): 1
Device ID from the kernel: 1
...

实际输出

Device ID from the host (thread 0): 0
Device ID from the kernel: 0
Device ID from the host (thread 1): 1
Device ID from the kernel: 0
...

问题原因

核心问题是CUDA上下文的延迟初始化与OpenMP线程竞争:

  • CUDA默认采用延迟初始化策略,仅当执行第一个CUDA操作(如内核启动)时,才会为当前线程的设备创建上下文。
  • 当多个OpenMP线程几乎同时启动时,可能出现竞争条件:后续线程的内核启动操作可能在cudaSetDevice的上下文完全建立前,错误复用了先启动线程已创建的设备0上下文。

解决方案

在每个线程设置设备ID后,显式触发设备上下文的初始化,避免延迟初始化带来的竞争。具体操作是在cudaSetDevice(device_id)之后添加cudaFree(0);,强制为当前设备创建独立的上下文。

修改后的代码

#include <cuda_runtime.h>
#include <iostream>
#include <omp.h>

__global__ void printGPUDeviceID() {
    int deviceID;
    cudaGetDevice(&deviceID);
    printf("Device ID from the kernel: %d\n", deviceID);
}

int main() {
    int num_devices;
    cudaGetDeviceCount(&num_devices);
    if (num_devices < 2) {
        std::cout << "This example requires at least two GPUs." << std::endl;
        return 1;
    }

    #pragma omp parallel num_threads(num_devices)
    {
        int thread_id = omp_get_thread_num();
        int device_id = thread_id;

        cudaSetDevice(device_id);
        // 强制初始化当前设备的上下文,避免延迟初始化竞争
        cudaFree(0);

        int deviceIDFromHost;
        cudaGetDevice(&deviceIDFromHost);
        printf("Device ID from the host (thread %d): %d\n", thread_id, deviceIDFromHost);

        printGPUDeviceID<<<1, 1>>>();

        cudaDeviceSynchronize();

        cudaError_t err = cudaGetLastError();
        if (err != cudaSuccess) {
            printf("CUDA error on device %d: %s\n", device_id, cudaGetErrorString(err));
        }
    }

    return 0;
}

编译注意事项

编译时需同时启用OpenMP和CUDA支持,命令示例:

nvcc -Xcompiler -fopenmp multi_gpu_test.cu -o multi_gpu_test

内容的提问来源于stack exchange,提问作者Spinor

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.18 15:07:38