You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA设备代码无法分配2116800大小的Color数组问题求助

CUDA设备代码中分配大数组失败(错误码700)的解决方法

我定义了包含3个double变量的Color类,以及包含Color数组的Image类。在GPU设备代码中尝试分配尺寸为1960*1080的Color数组时失败,报错“couldn't allocate an array of size 2116800 on device code”,CUDA错误码为700。以下是完整代码、运行输出及CMake配置:

完整C++/CUDA代码

#include <iostream>

// limited version of checkCudaErrors from helper_cuda.h in CUDA examples
#define checkCudaErrors(val) check_cuda((val), #val, __FILE__, __LINE__)

void check_cuda(cudaError_t result, char const* const func, const char* const file,
                int const line) {
    if (result) {
        std::cerr << "CUDA error = " << static_cast<unsigned int>(result) << " at " << file << ":"
                << line << " '" << func << "' \n";
        // Make sure we call CUDA Device Reset before exiting
        cudaDeviceReset();
        exit(-1);
    }
}

class Color {
public:
    double r, g, b;

    __host__ __device__ Color() : r(0.0), g(0.0), b(0.0) {
    }
};

class Image {
public:
    int width = -1;
    int height = -1;

    Color* frame_buffer = nullptr;

    __device__ Image(int _width, int _height) : width(_width), height(_height) {
        frame_buffer = new Color[width * height];
    }

    __device__ ~Image() {
        delete frame_buffer;
    }
};

__global__ void init_gpu_image(Image* image, int width,
                               int height) {
    printf("block id:  (%d, %d, %d)\n", blockIdx.x, blockIdx.y, blockIdx.z);
    printf("thread id: (%d, %d, %d)\n", threadIdx.x, threadIdx.y, threadIdx.z);
    *image = Image(width, height);
}

int main() {
    int width = 1960;
    int height = 1080;

    printf("image dimension: %d\n", width * height);
    printf("image size: %d\n", sizeof(Color) * width * height);

    /*
    // works fine when allocating with cudaMallocManaged()
    Color* frame_buffer;
    checkCudaErrors(cudaMallocManaged((void **)&frame_buffer, sizeof(Color) * width * height));
    checkCudaErrors(cudaGetLastError());
    checkCudaErrors(cudaDeviceSynchronize());
    */

    Image* gpu_image;
    checkCudaErrors(cudaMallocManaged((void **)&gpu_image, sizeof(Image)));
    init_gpu_image<<<1, 1>>>(gpu_image, width, height);

    checkCudaErrors(cudaGetLastError());
    checkCudaErrors(cudaDeviceSynchronize());

    return 0;
}

运行输出

image dimension: 2116800
image size: 50803200
block id:  (0, 0, 0)
thread id: (0, 0, 0)
CUDA error = 700 at /home/wentao/Desktop/cuda-test/main.cu:68 'cudaDeviceSynchronize()' 

CMakeLists.txt配置

cmake_minimum_required(VERSION 3.18 FATAL_ERROR)

if (NOT CMAKE_CUDA_COMPILER)
    set(CMAKE_CUDA_COMPILER "/usr/local/cuda/bin/nvcc")
    # required by CLion
endif ()

set(CMAKE_CXX_STANDARD 17)
set(CMAKE_CXX_STANDARD_REQUIRED ON)

project(cuda_test LANGUAGES CUDA CXX)

add_executable(cuda_test main.cu)

target_compile_options(cuda_test PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:
        --expt-relaxed-constexpr
        >)

原因分析

CUDA错误码700对应cudaErrorLaunchFailure,核心问题是设备端new操作默认分配的是线程栈内存,而非全局显存。线程栈的默认大小通常只有几KB到几十KB,而你要分配的50MB数组远超过栈容量,直接导致分配失败。

解决方法

方法1:改用全局显存分配(推荐)

在主机端通过cudaMallocManaged或cudaMalloc分配全局显存,再传递给设备端的Image类使用,这是CUDA中处理大内存分配的标准方式:

修改后的代码示例:

#include <iostream>

#define checkCudaErrors(val) check_cuda((val), #val, __FILE__, __LINE__)

void check_cuda(cudaError_t result, char const* const func, const char* const file,
                int const line) {
    if (result) {
        std::cerr << "CUDA error = " << static_cast<unsigned int>(result) << " at " << file << ":"
                << line << " '" << func << "' \n";
        cudaDeviceReset();
        exit(-1);
    }
}

class Color {
public:
    double r, g, b;
    __host__ __device__ Color() : r(0.0), g(0.0), b(0.0) {}
};

class Image {
public:
    int width = -1;
    int height = -1;
    Color* frame_buffer = nullptr;

    // 默认构造函数
    __host__ __device__ Image() = default;

    // 带显存缓冲区的构造函数
    __host__ __device__ Image(int _width, int _height, Color* buf) 
        : width(_width), height(_height), frame_buffer(buf) {}

    __device__ ~Image() {
        // 注意:主机端分配的显存不要在设备端delete,避免双重释放
        // delete frame_buffer;
    }
};

__global__ void init_gpu_image(Image* image, int width, int height, Color* buf) {
    printf("block id:  (%d, %d, %d)\n", blockIdx.x, blockIdx.y, blockIdx.z);
    printf("thread id: (%d, %d, %d)\n", threadIdx.x, threadIdx.y, threadIdx.z);
    *image = Image(width, height, buf);
}

int main() {
    int width = 1960;
    int height = 1080;
    size_t buf_size = sizeof(Color) * width * height;

    printf("image dimension: %d\n", width * height);
    printf("image size: %zu\n", buf_size);

    // 主机端分配全局可访问显存
    Color* frame_buffer;
    checkCudaErrors(cudaMallocManaged((void**)&frame_buffer, buf_size));
    
    Image* gpu_image;
    checkCudaErrors(cudaMallocManaged((void**)&gpu_image, sizeof(Image)));
    
    init_gpu_image<<<1, 1>>>(gpu_image, width, height, frame_buffer);

    checkCudaErrors(cudaGetLastError());
    checkCudaErrors(cudaDeviceSynchronize());

    // 主机端统一释放内存
    checkCudaErrors(cudaFree(frame_buffer));
    checkCudaErrors(cudaFree(gpu_image));

    return 0;
}

方法2:调整线程栈大小(不推荐)

如果一定要在设备端使用new,可以通过编译选项增大线程栈,但这种方法会占用更多GPU资源,且不同GPU的栈上限有限,仅适合小规模场景:

修改CMakeLists.txt中的CUDA编译选项:

target_compile_options(cuda_test PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:
        --expt-relaxed-constexpr
        --stack-size=67108864  # 设置线程栈大小为64MB
        >)

关键说明

  • 设备端的new/delete默认使用线程本地栈内存,仅适合分配极小的临时数据;
  • 全局显存分配必须通过主机端的cudaMalloc/cudaMallocManaged完成,或使用设备端的cudaMalloc(需注意上下文);
  • cudaMallocManaged分配的内存可同时被主机和设备访问,简化数据管理流程。

内容的提问来源于stack exchange,提问作者Rahn

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.01 23:20:28