You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA C++虚类方法替代方案及非法内存访问问题排查

CUDA C++虚类方法替代方案(针对Shape向量场景)

原代码中调用shapes[i]->hit(ray)触发非法内存访问的根本原因有两个:

  1. 对象切片:拷贝时用sizeof(Shape)只复制了基类部分,派生类Sphere的center和radius数据完全未拷贝到设备端,导致设备端访问派生类成员时出错。
  2. 虚函数表不兼容:CUDA主机和设备的虚函数表是独立的,主机端对象的虚表指针指向主机端函数地址,拷贝到设备后,设备线程访问该地址会触发非法内存访问。

以下是适合处理Shape向量场景的替代方案:

方案1:类型标识+显式分支(简单易实现)

给每个Shape添加明确的类型标识,在设备端根据类型调用对应派生类的方法,彻底避免虚函数依赖。

步骤1:修改Shape基类,添加类型枚举和数据存储结构

// shape.cuh
#pragma once
#include "cuda_path_tracer/ray.cuh"

enum class ShapeType { SPHERE };

struct ShapeData {
    ShapeType type;
    union {
        struct { Vec3 center; float radius; } sphere;
        // 后续添加其他形状的结构体
    };
};

class Shape {
public:
    Shape() = default;
    virtual ~Shape() = default;
    // 主机端方法:将对象转换为设备可处理的ShapeData
    __host__ virtual ShapeData toDeviceData() const = 0;
};

步骤2:修改Sphere类实现转换方法

// sphere.cuh
#pragma once
#include "shape.cuh"

class Sphere : public Shape {
public:
    __host__ Sphere(const Vec3& center, float radius) : center(center), radius(radius) {}
    __host__ ShapeData toDeviceData() const override {
        return {
            .type = ShapeType::SPHERE,
            .sphere = {.center = center, .radius = radius}
        };
    }
private:
    Vec3 center;
    float radius;
};

步骤3:主机端拷贝数据到设备

不再拷贝指针数组,直接拷贝ShapeData数组:

const auto num_shapes = scene->getShapes().size();
ShapeData* d_shapes_data;
CUDA_ERROR_CHECK(cudaMalloc(&d_shapes_data, num_shapes * sizeof(ShapeData)));

// 主机端先转换所有形状为ShapeData
std::vector<ShapeData> h_shapes_data(num_shapes);
for (size_t i = 0; i < num_shapes; i++) {
    h_shapes_data[i] = scene->getShapes()[i]->toDeviceData();
}

CUDA_ERROR_CHECK(cudaMemcpy(d_shapes_data, h_shapes_data.data(), 
                            num_shapes * sizeof(ShapeData), cudaMemcpyHostToDevice));

步骤4:设备端根据类型调用对应方法

__device__ bool hitSphere(const Ray& r, const ShapeData& sphere_data) {
    Vec3 const oc = r.getOrigin() - sphere_data.sphere.center;
    float const a = r.getDirection().dot(r.getDirection());
    float const b = 2.0f * oc.dot(r.getDirection());
    float const c = oc.dot(oc) - sphere_data.sphere.radius * sphere_data.sphere.radius;
    float const discriminant = b * b - 4 * a * c;
    return discriminant > 0;
}

__device__ auto getColor(const Ray& ray, const ShapeData* shapes_data, const size_t num_shapes) -> uchar4 {
    for (size_t i = 0; i < num_shapes; i++) {
        switch (shapes_data[i].type) {
            case ShapeType::SPHERE:
                if (hitSphere(ray, shapes_data[i])) {
                    return make_uchar4(1, 0, 0, UCHAR_MAX);
                }
                break;
            // 后续添加其他形状的分支
        }
    }
    return make_uchar4(0, 0, 1, UCHAR_MAX);
}

方案2:手动实现函数表(接近虚函数的灵活性)

手动为每个形状类型创建函数表,在主机端将函数指针和数据绑定,设备端通过函数表调用方法,既保留虚函数的灵活性,又规避CUDA虚表的兼容问题。

步骤1:定义函数表结构体

// shape.cuh
#pragma once
#include "cuda_path_tracer/ray.cuh"

enum class ShapeType { SPHERE };

// 设备端可调用的函数指针类型
using HitFunc = __device__ bool(*)(const Ray&, const void*);

struct ShapeInfo {
    HitFunc hit;
    const void* data; // 指向对应形状的设备端数据
};

class Shape {
public:
    Shape() = default;
    virtual ~Shape() = default;
    // 主机端方法:返回设备端的函数指针和数据指针
    __host__ virtual ShapeInfo toDeviceInfo(void* device_data_ptr) const = 0;
    __host__ virtual size_t getDataSize() const = 0;
};

步骤2:Sphere类实现对应方法

// sphere.cuh
#pragma once
#include "shape.cuh"

struct SphereData {
    Vec3 center;
    float radius;
};

class Sphere : public Shape {
public:
    __host__ Sphere(const Vec3& center, float radius) : center(center), radius(radius) {}
    __host__ ShapeInfo toDeviceInfo(void* device_data_ptr) const override {
        SphereData* d_sphere = static_cast<SphereData*>(device_data_ptr);
        *d_sphere = {center, radius};
        return {.hit = hitSphere, .data = d_sphere};
    }
    __host__ size_t getDataSize() const override {
        return sizeof(SphereData);
    }
private:
    Vec3 center;
    float radius;
    __device__ static bool hitSphere(const Ray& r, const void* data) {
        const SphereData* sphere = static_cast<const SphereData*>(data);
        Vec3 const oc = r.getOrigin() - sphere->center;
        float const a = r.getDirection().dot(r.getDirection());
        float const b = 2.0f * oc.dot(r.getDirection());
        float const c = oc.dot(oc) - sphere->radius * sphere->radius;
        float const discriminant = b * b - 4 * a * c;
        return discriminant > 0;
    }
};

步骤3:主机端分配内存并构建函数表数组

const auto num_shapes = scene->getShapes().size();

// 1. 分配存储所有形状数据的设备内存
size_t total_data_size = 0;
for (const auto& shape : scene->getShapes()) {
    total_data_size += shape->getDataSize();
}
void* d_all_shape_data;
CUDA_ERROR_CHECK(cudaMalloc(&d_all_shape_data, total_data_size));

// 2. 构建ShapeInfo数组并拷贝到设备
ShapeInfo* d_shape_infos;
CUDA_ERROR_CHECK(cudaMalloc(&d_shape_infos, num_shapes * sizeof(ShapeInfo)));
std::vector<ShapeInfo> h_shape_infos(num_shapes);

void* current_data_ptr = d_all_shape_data;
for (size_t i = 0; i < num_shapes; i++) {
    const auto& shape = scene->getShapes()[i];
    h_shape_infos[i] = shape->toDeviceInfo(current_data_ptr);
    current_data_ptr = static_cast<char*>(current_data_ptr) + shape->getDataSize();
}

CUDA_ERROR_CHECK(cudaMemcpy(d_shape_infos, h_shape_infos.data(), 
                            num_shapes * sizeof(ShapeInfo), cudaMemcpyHostToDevice));

步骤4:设备端调用函数表中的方法

__device__ auto getColor(const Ray& ray, const ShapeInfo* shape_infos, const size_t num_shapes) -> uchar4 {
    for (size_t i = 0; i < num_shapes; i++) {
        if (shape_infos[i].hit(ray, shape_infos[i].data)) {
            return make_uchar4(1, 0, 0, UCHAR_MAX);
        }
    }
    return make_uchar4(0, 0, 1, UCHAR_MAX);
}

方案3:异构数组分组处理(性能最优)

路径追踪中同类型形状可批量处理,将相同类型的形状数据放在连续数组中,设备端按类型批量调用方法,减少循环分支开销,提升渲染性能。

核心思路

  • 主机端将不同类型的形状分组,比如所有Sphere放在一个数组,其他形状放在对应数组
  • 设备端分别处理每个类型的数组,避免循环中的分支判断

示例代码(以Sphere为例)

// 主机端分组
std::vector<SphereData> h_spheres;
for (const auto& shape : scene->getShapes()) {
    if (auto sphere = dynamic_cast<const Sphere*>(shape.get())) {
        h_spheres.push_back({sphere->center, sphere->radius});
    }
    // 处理其他类型形状
}

// 拷贝到设备
SphereData* d_spheres;
CUDA_ERROR_CHECK(cudaMalloc(&d_spheres, h_spheres.size() * sizeof(SphereData)));
CUDA_ERROR_CHECK(cudaMemcpy(d_spheres, h_spheres.data(), 
                            h_spheres.size() * sizeof(SphereData), cudaMemcpyHostToDevice));

// 设备端批量处理
__device__ auto getColor(const Ray& ray, const SphereData* spheres, size_t num_spheres) -> uchar4 {
    for (size_t i = 0; i < num_spheres; i++) {
        if (hitSphere(ray, spheres[i])) {
            return make_uchar4(1, 0, 0, UCHAR_MAX);
        }
    }
    // 处理其他类型形状数组
    return make_uchar4(0, 0, 1, UCHAR_MAX);
}

内容的提问来源于stack exchange,提问作者glowl

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.15 15:04:57