You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA内核传入指针为空及虚函数调用崩溃问题求助

问题排查与解决方案

核心问题1:核函数指针传递错误(值传递导致修改无效)

你的create_world核函数最初使用值传递make_list* d_world,核函数内对d_world的赋值仅修改了函数内部的局部副本,主机端持有的原始指针完全无法获取新分配对象的地址,这是导致size始终为0的根本原因。

修复方式:使用双重指针传递

将create_world的参数改为make_list** d_world,在核函数内通过*d_world赋值,才能把设备端新分配的对象地址写入主机端可访问的设备内存位置:

// 核函数参数改为双重指针
__global__ void create_world(float* d_list, make_list** d_world, int num_triangles) {
    if (threadIdx.x == 0 && blockIdx.x == 0) {
        *d_world = new make_list(&d_list, num_triangles);
    }
}

// 主机端分配存储指针的设备内存
make_list** d_world;
cudaMalloc(&d_world, sizeof(make_list*));

核心问题2:设备端对象的指针引用错误

构造make_list时传入的&d_list是核函数栈上的局部指针地址,这个地址在核函数执行结束后立即失效,其他核函数无法访问该地址指向的内容。而d_list本身已经是主机端持有的设备内存指针,直接传递即可。

修复方式:直接传递设备指针

修改make_list的成员和构造函数,直接存储设备指针:

struct make_list : public hittable {
    __device__ make_list(float* list, int n) { contents = list; size = n; };
    __device__ virtual int hit() const {
        return size;
    }
    float* contents; // 改为float*类型
    int size;
};

// 核函数构造时直接传入d_list
* d_world = new make_list(d_list, num_triangles);

核心问题3:虚函数导致的设备端崩溃

CUDA对带虚函数的类有严格要求:

  • 类的虚表必须在设备端正确生成
  • 若主机端未触发类的虚函数实例化,编译器可能不会生成设备端虚表,导致设备端调用虚函数时崩溃

修复方式:确保设备端虚表生成

  1. 所有虚函数必须添加__device__修饰
  2. 为基类添加虚析构函数,避免内存泄漏和未定义行为
  3. 若无需动态多态,尽量避免在设备端使用虚函数;必须使用时,确保类的完整定义在编译单元中可见

完整修复后的代码

#include <cuda_runtime.h>
#include <device_launch_parameters.h>
#include <vector>
#include <cstdio>

class hittable {
public:
    __device__ virtual int hit() const = 0;
    __device__ virtual ~hittable() = default; // 添加虚析构函数
};

struct make_list : public hittable {
    __device__ make_list(float* list, int n) : contents(list), size(n) {}; // 使用初始化列表更安全
    __device__ virtual int hit() const override { // 用override明确重写虚函数
        return size;
    }
    float* contents;
    int size;
};

__global__ void render(make_list** world) {
    if (threadIdx.x == 0 && blockIdx.x == 0) { // 限制单线程执行,避免冲突
        int size = (*world)->size;
        printf("size = %d\n", size);
        int new_size = (*world)->hit();
        printf("new size = %d\n", new_size);
    }
}

__global__ void create_world(float* d_list, make_list** d_world, int num_triangles) {
    if (threadIdx.x == 0 && blockIdx.x == 0) {
        *d_world = new make_list(d_list, num_triangles);
    }
}

int main() {
    float* d_list;
    make_list** d_world;
    int size = 8;
    std::vector<float> m_triangles(size, 1.0f); // 初始化测试数据

    // 分配设备内存
    cudaMalloc(&d_world, sizeof(make_list*));
    cudaMalloc((void**)&d_list, size * sizeof(float));

    // 拷贝数据到设备
    cudaMemcpy(d_list, m_triangles.data(), size * sizeof(float), cudaMemcpyHostToDevice);

    // 执行核函数创建world
    create_world <<<1, 1>>> (d_list, d_world, size);
    cudaDeviceSynchronize();
    cudaError_t err = cudaGetLastError();
    if (err != cudaSuccess) {
        printf("create_world error: %s\n", cudaGetErrorString(err));
        return 1;
    }

    // 执行渲染核函数
    render <<<1, 1>>> (d_world);
    cudaDeviceSynchronize();
    err = cudaGetLastError();
    if (err != cudaSuccess) {
        printf("render error: %s\n", cudaGetErrorString(err));
        return 1;
    }

    // 释放设备内存(先释放对象,再释放指针)
    make_list* temp_world;
    cudaMemcpy(&temp_world, d_world, sizeof(make_list*), cudaMemcpyDeviceToHost);
    cudaFree(temp_world);
    cudaFree(d_world);
    cudaFree(d_list);

    return 0;
}

额外注意事项

  • 设备端用new分配的内存必须手动释放,否则会造成设备内存泄漏
  • 核函数中初始化全局对象时,必须限制仅单线程执行,避免多线程冲突
  • 每次核函数执行后,务必检查cudaGetLastError()和cudaDeviceSynchronize(),快速定位错误

内容的提问来源于stack exchange,提问作者mrmanet

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.01 02:05:45