CUDA内核初始化的双层指针关联设备内存如何拷贝至主机
解决方法
1. 统一内存原地构造(适配现有cudaMallocManaged)
内核中放弃new,改用cudaMallocManaged分配子类对象内存,再原地构造,让对象处于主机可见的统一内存空间:
__global__ void device_init(textures** t_list, int count) { int idx = threadIdx.x; if (idx >= count) return; if (idx == 0) { cudaMallocManaged(&t_list[idx], sizeof(texture1)); new(t_list[idx]) texture1(); // 原地调用构造函数 } else { cudaMallocManaged(&t_list[idx], sizeof(texture2)); new(t_list[idx]) texture2(); } }
内核执行后必须同步,确保设备端构造完成,主机才能安全访问:
device_init<<<1, 2>>>(t_list, 2); cudaDeviceSynchronize(); // 此时主机可直接操作t_list指向的对象
2. 设备堆内存显式拷贝
如果必须用new在设备堆创建对象,需要手动记录对象地址和大小,再逐份拷贝到主机:
步骤1:内核中记录设备对象信息
struct ObjInfo { void* dev_ptr; size_t size; }; __global__ void device_init(textures** t_list, ObjInfo* obj_infos, int count) { int idx = threadIdx.x; if (idx >= count) return; if (idx == 0) { t_list[idx] = new texture1(); obj_infos[idx] = {t_list[idx], sizeof(texture1)}; } else { t_list[idx] = new texture2(); obj_infos[idx] = {t_list[idx], sizeof(texture2)}; } }
步骤2:主机端拷贝设备内存
// 分配统一内存存储对象信息 ObjInfo* obj_infos; cudaMallocManaged(&obj_infos, 2 * sizeof(ObjInfo)); device_init<<<1, 2>>>(t_list, obj_infos, 2); cudaDeviceSynchronize(); // 主机端分配内存并拷贝 textures** host_t_list = (textures**)malloc(2 * sizeof(textures*)); host_t_list[0] = (textures*)malloc(sizeof(texture1)); cudaMemcpy(host_t_list[0], obj_infos[0].dev_ptr, obj_infos[0].size, cudaMemcpyDeviceToHost); host_t_list[1] = (textures*)malloc(sizeof(texture2)); cudaMemcpy(host_t_list[1], obj_infos[1].dev_ptr, obj_infos[1].size, cudaMemcpyDeviceToHost);
注意:需确保子类虚函数表在主机可见(用__host__ __device__修饰虚函数),否则主机访问虚函数会崩溃。
3. 替换双层指针为统一内存数组
避免指针嵌套问题,直接分配统一内存的对象数组,内核中原地构造:
// 取texture1和texture2的最大尺寸作为单元素大小 const size_t MAX_OBJ_SIZE = max(sizeof(texture1), sizeof(texture2)); void* t_list; cudaMallocManaged(&t_list, 2 * MAX_OBJ_SIZE);
内核中构造对象:
__global__ void device_init(void* t_list) { // 构造第一个对象 texture1* t1 = new(t_list) texture1(); // 偏移第一个对象的大小,构造第二个 texture2* t2 = new((char*)t_list + sizeof(texture1)) texture2(); }
主机端直接通过偏移访问:
device_init<<<1,1>>>(t_list); cudaDeviceSynchronize(); texture1* host_t1 = (texture1*)t_list; texture2* host_t2 = (texture2*)((char*)t_list + sizeof(texture1));
关键注意事项
- 虚函数表:所有子类的虚函数必须用
__host__ __device__修饰,确保主机和设备都能访问到虚表。 - 内存越界:内核中严格检查线程索引范围,避免超出
t_list的元素个数;分配内存时确保大小足够容纳所有对象。 - 设备堆大小:如果用
new在设备端创建大量对象,需提前设置堆大小:cudaDeviceSetLimit(cudaLimitMallocHeapSize, 1024*1024*64);(示例为64MB)
内容的提问来源于stack exchange,提问作者kratia
相关产品推荐
相关产品推荐

