You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA内核初始化的双层指针关联设备内存如何拷贝至主机

解决方法

1. 统一内存原地构造(适配现有cudaMallocManaged)

内核中放弃new,改用cudaMallocManaged分配子类对象内存,再原地构造,让对象处于主机可见的统一内存空间:

__global__ void device_init(textures** t_list, int count) {
    int idx = threadIdx.x;
    if (idx >= count) return;
    
    if (idx == 0) {
        cudaMallocManaged(&t_list[idx], sizeof(texture1));
        new(t_list[idx]) texture1(); // 原地调用构造函数
    } else {
        cudaMallocManaged(&t_list[idx], sizeof(texture2));
        new(t_list[idx]) texture2();
    }
}

内核执行后必须同步,确保设备端构造完成,主机才能安全访问:

device_init<<<1, 2>>>(t_list, 2);
cudaDeviceSynchronize();
// 此时主机可直接操作t_list指向的对象

2. 设备堆内存显式拷贝

如果必须用new在设备堆创建对象,需要手动记录对象地址和大小,再逐份拷贝到主机:

步骤1:内核中记录设备对象信息

struct ObjInfo {
    void* dev_ptr;
    size_t size;
};

__global__ void device_init(textures** t_list, ObjInfo* obj_infos, int count) {
    int idx = threadIdx.x;
    if (idx >= count) return;
    
    if (idx == 0) {
        t_list[idx] = new texture1();
        obj_infos[idx] = {t_list[idx], sizeof(texture1)};
    } else {
        t_list[idx] = new texture2();
        obj_infos[idx] = {t_list[idx], sizeof(texture2)};
    }
}

步骤2:主机端拷贝设备内存

// 分配统一内存存储对象信息
ObjInfo* obj_infos;
cudaMallocManaged(&obj_infos, 2 * sizeof(ObjInfo));

device_init<<<1, 2>>>(t_list, obj_infos, 2);
cudaDeviceSynchronize();

// 主机端分配内存并拷贝
textures** host_t_list = (textures**)malloc(2 * sizeof(textures*));
host_t_list[0] = (textures*)malloc(sizeof(texture1));
cudaMemcpy(host_t_list[0], obj_infos[0].dev_ptr, obj_infos[0].size, cudaMemcpyDeviceToHost);

host_t_list[1] = (textures*)malloc(sizeof(texture2));
cudaMemcpy(host_t_list[1], obj_infos[1].dev_ptr, obj_infos[1].size, cudaMemcpyDeviceToHost);

注意:需确保子类虚函数表在主机可见(用__host__ __device__修饰虚函数),否则主机访问虚函数会崩溃。

3. 替换双层指针为统一内存数组

避免指针嵌套问题,直接分配统一内存的对象数组,内核中原地构造:

// 取texture1和texture2的最大尺寸作为单元素大小
const size_t MAX_OBJ_SIZE = max(sizeof(texture1), sizeof(texture2));
void* t_list;
cudaMallocManaged(&t_list, 2 * MAX_OBJ_SIZE);

内核中构造对象:

__global__ void device_init(void* t_list) {
    // 构造第一个对象
    texture1* t1 = new(t_list) texture1();
    // 偏移第一个对象的大小,构造第二个
    texture2* t2 = new((char*)t_list + sizeof(texture1)) texture2();
}

主机端直接通过偏移访问:

device_init<<<1,1>>>(t_list);
cudaDeviceSynchronize();
texture1* host_t1 = (texture1*)t_list;
texture2* host_t2 = (texture2*)((char*)t_list + sizeof(texture1));

关键注意事项

  • 虚函数表:所有子类的虚函数必须用__host__ __device__修饰,确保主机和设备都能访问到虚表。
  • 内存越界:内核中严格检查线程索引范围,避免超出t_list的元素个数;分配内存时确保大小足够容纳所有对象。
  • 设备堆大小:如果用new在设备端创建大量对象,需提前设置堆大小:cudaDeviceSetLimit(cudaLimitMallocHeapSize, 1024*1024*64);(示例为64MB)

内容的提问来源于stack exchange,提问作者kratia

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.28 06:35:23