CUDA设备代码无法分配2116800大小的Color数组问题求助
CUDA设备代码中分配大数组失败(错误码700)的解决方法
我定义了包含3个double变量的Color类,以及包含Color数组的Image类。在GPU设备代码中尝试分配尺寸为1960*1080的Color数组时失败,报错“couldn't allocate an array of size 2116800 on device code”,CUDA错误码为700。以下是完整代码、运行输出及CMake配置:
完整C++/CUDA代码
#include <iostream> // limited version of checkCudaErrors from helper_cuda.h in CUDA examples #define checkCudaErrors(val) check_cuda((val), #val, __FILE__, __LINE__) void check_cuda(cudaError_t result, char const* const func, const char* const file, int const line) { if (result) { std::cerr << "CUDA error = " << static_cast<unsigned int>(result) << " at " << file << ":" << line << " '" << func << "' \n"; // Make sure we call CUDA Device Reset before exiting cudaDeviceReset(); exit(-1); } } class Color { public: double r, g, b; __host__ __device__ Color() : r(0.0), g(0.0), b(0.0) { } }; class Image { public: int width = -1; int height = -1; Color* frame_buffer = nullptr; __device__ Image(int _width, int _height) : width(_width), height(_height) { frame_buffer = new Color[width * height]; } __device__ ~Image() { delete frame_buffer; } }; __global__ void init_gpu_image(Image* image, int width, int height) { printf("block id: (%d, %d, %d)\n", blockIdx.x, blockIdx.y, blockIdx.z); printf("thread id: (%d, %d, %d)\n", threadIdx.x, threadIdx.y, threadIdx.z); *image = Image(width, height); } int main() { int width = 1960; int height = 1080; printf("image dimension: %d\n", width * height); printf("image size: %d\n", sizeof(Color) * width * height); /* // works fine when allocating with cudaMallocManaged() Color* frame_buffer; checkCudaErrors(cudaMallocManaged((void **)&frame_buffer, sizeof(Color) * width * height)); checkCudaErrors(cudaGetLastError()); checkCudaErrors(cudaDeviceSynchronize()); */ Image* gpu_image; checkCudaErrors(cudaMallocManaged((void **)&gpu_image, sizeof(Image))); init_gpu_image<<<1, 1>>>(gpu_image, width, height); checkCudaErrors(cudaGetLastError()); checkCudaErrors(cudaDeviceSynchronize()); return 0; }
运行输出
image dimension: 2116800 image size: 50803200 block id: (0, 0, 0) thread id: (0, 0, 0) CUDA error = 700 at /home/wentao/Desktop/cuda-test/main.cu:68 'cudaDeviceSynchronize()'
CMakeLists.txt配置
cmake_minimum_required(VERSION 3.18 FATAL_ERROR) if (NOT CMAKE_CUDA_COMPILER) set(CMAKE_CUDA_COMPILER "/usr/local/cuda/bin/nvcc") # required by CLion endif () set(CMAKE_CXX_STANDARD 17) set(CMAKE_CXX_STANDARD_REQUIRED ON) project(cuda_test LANGUAGES CUDA CXX) add_executable(cuda_test main.cu) target_compile_options(cuda_test PRIVATE $<$<COMPILE_LANGUAGE:CUDA>: --expt-relaxed-constexpr >)
原因分析
CUDA错误码700对应cudaErrorLaunchFailure,核心问题是设备端new操作默认分配的是线程栈内存,而非全局显存。线程栈的默认大小通常只有几KB到几十KB,而你要分配的50MB数组远超过栈容量,直接导致分配失败。
解决方法
方法1:改用全局显存分配(推荐)
在主机端通过cudaMallocManaged或cudaMalloc分配全局显存,再传递给设备端的Image类使用,这是CUDA中处理大内存分配的标准方式:
修改后的代码示例:
#include <iostream> #define checkCudaErrors(val) check_cuda((val), #val, __FILE__, __LINE__) void check_cuda(cudaError_t result, char const* const func, const char* const file, int const line) { if (result) { std::cerr << "CUDA error = " << static_cast<unsigned int>(result) << " at " << file << ":" << line << " '" << func << "' \n"; cudaDeviceReset(); exit(-1); } } class Color { public: double r, g, b; __host__ __device__ Color() : r(0.0), g(0.0), b(0.0) {} }; class Image { public: int width = -1; int height = -1; Color* frame_buffer = nullptr; // 默认构造函数 __host__ __device__ Image() = default; // 带显存缓冲区的构造函数 __host__ __device__ Image(int _width, int _height, Color* buf) : width(_width), height(_height), frame_buffer(buf) {} __device__ ~Image() { // 注意:主机端分配的显存不要在设备端delete,避免双重释放 // delete frame_buffer; } }; __global__ void init_gpu_image(Image* image, int width, int height, Color* buf) { printf("block id: (%d, %d, %d)\n", blockIdx.x, blockIdx.y, blockIdx.z); printf("thread id: (%d, %d, %d)\n", threadIdx.x, threadIdx.y, threadIdx.z); *image = Image(width, height, buf); } int main() { int width = 1960; int height = 1080; size_t buf_size = sizeof(Color) * width * height; printf("image dimension: %d\n", width * height); printf("image size: %zu\n", buf_size); // 主机端分配全局可访问显存 Color* frame_buffer; checkCudaErrors(cudaMallocManaged((void**)&frame_buffer, buf_size)); Image* gpu_image; checkCudaErrors(cudaMallocManaged((void**)&gpu_image, sizeof(Image))); init_gpu_image<<<1, 1>>>(gpu_image, width, height, frame_buffer); checkCudaErrors(cudaGetLastError()); checkCudaErrors(cudaDeviceSynchronize()); // 主机端统一释放内存 checkCudaErrors(cudaFree(frame_buffer)); checkCudaErrors(cudaFree(gpu_image)); return 0; }
方法2:调整线程栈大小(不推荐)
如果一定要在设备端使用new,可以通过编译选项增大线程栈,但这种方法会占用更多GPU资源,且不同GPU的栈上限有限,仅适合小规模场景:
修改CMakeLists.txt中的CUDA编译选项:
target_compile_options(cuda_test PRIVATE $<$<COMPILE_LANGUAGE:CUDA>: --expt-relaxed-constexpr --stack-size=67108864 # 设置线程栈大小为64MB >)
关键说明
- 设备端的
new/delete默认使用线程本地栈内存,仅适合分配极小的临时数据; - 全局显存分配必须通过主机端的
cudaMalloc/cudaMallocManaged完成,或使用设备端的cudaMalloc(需注意上下文); cudaMallocManaged分配的内存可同时被主机和设备访问,简化数据管理流程。
内容的提问来源于stack exchange,提问作者Rahn
相关产品推荐
相关产品推荐

