CUDA C++虚类方法替代方案及非法内存访问问题排查
CUDA C++虚类方法替代方案(针对Shape向量场景)
原代码中调用shapes[i]->hit(ray)触发非法内存访问的根本原因有两个:
- 对象切片:拷贝时用
sizeof(Shape)只复制了基类部分,派生类Sphere的center和radius数据完全未拷贝到设备端,导致设备端访问派生类成员时出错。 - 虚函数表不兼容:CUDA主机和设备的虚函数表是独立的,主机端对象的虚表指针指向主机端函数地址,拷贝到设备后,设备线程访问该地址会触发非法内存访问。
以下是适合处理Shape向量场景的替代方案:
方案1:类型标识+显式分支(简单易实现)
给每个Shape添加明确的类型标识,在设备端根据类型调用对应派生类的方法,彻底避免虚函数依赖。
步骤1:修改Shape基类,添加类型枚举和数据存储结构
// shape.cuh #pragma once #include "cuda_path_tracer/ray.cuh" enum class ShapeType { SPHERE }; struct ShapeData { ShapeType type; union { struct { Vec3 center; float radius; } sphere; // 后续添加其他形状的结构体 }; }; class Shape { public: Shape() = default; virtual ~Shape() = default; // 主机端方法:将对象转换为设备可处理的ShapeData __host__ virtual ShapeData toDeviceData() const = 0; };
步骤2:修改Sphere类实现转换方法
// sphere.cuh #pragma once #include "shape.cuh" class Sphere : public Shape { public: __host__ Sphere(const Vec3& center, float radius) : center(center), radius(radius) {} __host__ ShapeData toDeviceData() const override { return { .type = ShapeType::SPHERE, .sphere = {.center = center, .radius = radius} }; } private: Vec3 center; float radius; };
步骤3:主机端拷贝数据到设备
不再拷贝指针数组,直接拷贝ShapeData数组:
const auto num_shapes = scene->getShapes().size(); ShapeData* d_shapes_data; CUDA_ERROR_CHECK(cudaMalloc(&d_shapes_data, num_shapes * sizeof(ShapeData))); // 主机端先转换所有形状为ShapeData std::vector<ShapeData> h_shapes_data(num_shapes); for (size_t i = 0; i < num_shapes; i++) { h_shapes_data[i] = scene->getShapes()[i]->toDeviceData(); } CUDA_ERROR_CHECK(cudaMemcpy(d_shapes_data, h_shapes_data.data(), num_shapes * sizeof(ShapeData), cudaMemcpyHostToDevice));
步骤4:设备端根据类型调用对应方法
__device__ bool hitSphere(const Ray& r, const ShapeData& sphere_data) { Vec3 const oc = r.getOrigin() - sphere_data.sphere.center; float const a = r.getDirection().dot(r.getDirection()); float const b = 2.0f * oc.dot(r.getDirection()); float const c = oc.dot(oc) - sphere_data.sphere.radius * sphere_data.sphere.radius; float const discriminant = b * b - 4 * a * c; return discriminant > 0; } __device__ auto getColor(const Ray& ray, const ShapeData* shapes_data, const size_t num_shapes) -> uchar4 { for (size_t i = 0; i < num_shapes; i++) { switch (shapes_data[i].type) { case ShapeType::SPHERE: if (hitSphere(ray, shapes_data[i])) { return make_uchar4(1, 0, 0, UCHAR_MAX); } break; // 后续添加其他形状的分支 } } return make_uchar4(0, 0, 1, UCHAR_MAX); }
方案2:手动实现函数表(接近虚函数的灵活性)
手动为每个形状类型创建函数表,在主机端将函数指针和数据绑定,设备端通过函数表调用方法,既保留虚函数的灵活性,又规避CUDA虚表的兼容问题。
步骤1:定义函数表结构体
// shape.cuh #pragma once #include "cuda_path_tracer/ray.cuh" enum class ShapeType { SPHERE }; // 设备端可调用的函数指针类型 using HitFunc = __device__ bool(*)(const Ray&, const void*); struct ShapeInfo { HitFunc hit; const void* data; // 指向对应形状的设备端数据 }; class Shape { public: Shape() = default; virtual ~Shape() = default; // 主机端方法:返回设备端的函数指针和数据指针 __host__ virtual ShapeInfo toDeviceInfo(void* device_data_ptr) const = 0; __host__ virtual size_t getDataSize() const = 0; };
步骤2:Sphere类实现对应方法
// sphere.cuh #pragma once #include "shape.cuh" struct SphereData { Vec3 center; float radius; }; class Sphere : public Shape { public: __host__ Sphere(const Vec3& center, float radius) : center(center), radius(radius) {} __host__ ShapeInfo toDeviceInfo(void* device_data_ptr) const override { SphereData* d_sphere = static_cast<SphereData*>(device_data_ptr); *d_sphere = {center, radius}; return {.hit = hitSphere, .data = d_sphere}; } __host__ size_t getDataSize() const override { return sizeof(SphereData); } private: Vec3 center; float radius; __device__ static bool hitSphere(const Ray& r, const void* data) { const SphereData* sphere = static_cast<const SphereData*>(data); Vec3 const oc = r.getOrigin() - sphere->center; float const a = r.getDirection().dot(r.getDirection()); float const b = 2.0f * oc.dot(r.getDirection()); float const c = oc.dot(oc) - sphere->radius * sphere->radius; float const discriminant = b * b - 4 * a * c; return discriminant > 0; } };
步骤3:主机端分配内存并构建函数表数组
const auto num_shapes = scene->getShapes().size(); // 1. 分配存储所有形状数据的设备内存 size_t total_data_size = 0; for (const auto& shape : scene->getShapes()) { total_data_size += shape->getDataSize(); } void* d_all_shape_data; CUDA_ERROR_CHECK(cudaMalloc(&d_all_shape_data, total_data_size)); // 2. 构建ShapeInfo数组并拷贝到设备 ShapeInfo* d_shape_infos; CUDA_ERROR_CHECK(cudaMalloc(&d_shape_infos, num_shapes * sizeof(ShapeInfo))); std::vector<ShapeInfo> h_shape_infos(num_shapes); void* current_data_ptr = d_all_shape_data; for (size_t i = 0; i < num_shapes; i++) { const auto& shape = scene->getShapes()[i]; h_shape_infos[i] = shape->toDeviceInfo(current_data_ptr); current_data_ptr = static_cast<char*>(current_data_ptr) + shape->getDataSize(); } CUDA_ERROR_CHECK(cudaMemcpy(d_shape_infos, h_shape_infos.data(), num_shapes * sizeof(ShapeInfo), cudaMemcpyHostToDevice));
步骤4:设备端调用函数表中的方法
__device__ auto getColor(const Ray& ray, const ShapeInfo* shape_infos, const size_t num_shapes) -> uchar4 { for (size_t i = 0; i < num_shapes; i++) { if (shape_infos[i].hit(ray, shape_infos[i].data)) { return make_uchar4(1, 0, 0, UCHAR_MAX); } } return make_uchar4(0, 0, 1, UCHAR_MAX); }
方案3:异构数组分组处理(性能最优)
路径追踪中同类型形状可批量处理,将相同类型的形状数据放在连续数组中,设备端按类型批量调用方法,减少循环分支开销,提升渲染性能。
核心思路
- 主机端将不同类型的形状分组,比如所有Sphere放在一个数组,其他形状放在对应数组
- 设备端分别处理每个类型的数组,避免循环中的分支判断
示例代码(以Sphere为例)
// 主机端分组 std::vector<SphereData> h_spheres; for (const auto& shape : scene->getShapes()) { if (auto sphere = dynamic_cast<const Sphere*>(shape.get())) { h_spheres.push_back({sphere->center, sphere->radius}); } // 处理其他类型形状 } // 拷贝到设备 SphereData* d_spheres; CUDA_ERROR_CHECK(cudaMalloc(&d_spheres, h_spheres.size() * sizeof(SphereData))); CUDA_ERROR_CHECK(cudaMemcpy(d_spheres, h_spheres.data(), h_spheres.size() * sizeof(SphereData), cudaMemcpyHostToDevice)); // 设备端批量处理 __device__ auto getColor(const Ray& ray, const SphereData* spheres, size_t num_spheres) -> uchar4 { for (size_t i = 0; i < num_spheres; i++) { if (hitSphere(ray, spheres[i])) { return make_uchar4(1, 0, 0, UCHAR_MAX); } } // 处理其他类型形状数组 return make_uchar4(0, 0, 1, UCHAR_MAX); }
内容的提问来源于stack exchange,提问作者glowl
相关产品推荐
相关产品推荐

