You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用cudaMalloc复制二级double指针数组到CUDA显存不生效问题排查

错误原因

你的代码存在4个核心问题导致输出全为0:

  • cudaMalloc传参错误:cudaMalloc第一个参数要求传入存储GPU地址的主机变量的指针,原有写法cudaMalloc(cuda_src, width * sizeof(double*));直接传入了未初始化的野指针cuda_src,会导致显存分配失败,正确写法应为cudaMalloc(&cuda_src, width * sizeof(double*));,cuda_dst同理。
  • 主机端直接访问GPU指针非法:cuda_src是存储在主机内存的GPU地址,指向GPU上的一级指针数组,主机端不能直接解引用cuda_src[i],原有写法&cuda_src[i]属于非法内存访问。
  • 缺少实际数据的主机到显存拷贝:你仅将src_ptr中存储的行指针拷贝到了GPU,src_ptr[i]指向的主机侧实际double数组没有拷贝到对应GPU行地址中,内核读取到的源数据本身就是无效值。
  • 缺少实际数据的显存到主机拷贝:内核执行完成后,你仅拷贝了GPU侧的一级指针数组到主机dst_ptr,此时dst_ptr[i]存储的是GPU显存地址,主机直接访问dst_ptr[i][j]无法读取到显存中的实际数据,需要逐行把显存数据拷贝回主机内存。
修正后的二级指针实现(兼容原有写法)
#include <stdio.h>
#include <assert.h>
#include <cuda.h>
#include <cuda_runtime.h>

__global__
void copy(double **src, double **dst,  int width, int height)
{
   for (int i = 0; i < width; i++) {
        for (int j = 0; j < height; j++) {
            dst[i][j] = src[i][j];
        }
    }
}

int main(int argc, char *argv[]){
    int width = 3;
    int height = 5;

    double **src_ptr;
    double **cuda_src;
    double **dst_ptr;
    double **cuda_dst;
    // 主机侧临时存储GPU行地址的数组
    double **h_cuda_src_rows, **h_cuda_dst_rows;

    src_ptr = (double **)malloc(width * sizeof(double*));
    dst_ptr = (double **)malloc(width * sizeof(double*));
    h_cuda_src_rows = (double **)malloc(width * sizeof(double*));
    h_cuda_dst_rows = (double **)malloc(width * sizeof(double*));

    // 分配GPU侧一级指针数组
    cudaMalloc(&cuda_src, width * sizeof(double*));
    cudaMalloc(&cuda_dst, width * sizeof(double*));

    for(int i = 0; i<width;i++){
        src_ptr[i] = (double *)malloc(height * sizeof *src_ptr[i]);
        dst_ptr[i] = (double *)malloc(height * sizeof *dst_ptr[i]);
        // 分配GPU行内存,地址存在主机临时数组中
        cudaMalloc((void ** ) &h_cuda_src_rows[i], height * sizeof(double));
        cudaMalloc((void ** ) &h_cuda_dst_rows[i], height * sizeof(double));

        for(int j=0;j<height;j++){
            src_ptr[i][j] = (double)(rand() % 1000) / 1000;
            dst_ptr[i][j] = 0.;
        }
        // 拷贝每行实际数据到GPU
        cudaMemcpy(h_cuda_src_rows[i], src_ptr[i], height * sizeof(double), cudaMemcpyHostToDevice);
    }
    // 将GPU行地址数组拷贝到GPU侧一级指针数组
    cudaMemcpy(cuda_src, h_cuda_src_rows, width * sizeof(double*), cudaMemcpyHostToDevice);
    cudaMemcpy(cuda_dst, h_cuda_dst_rows, width * sizeof(double*), cudaMemcpyHostToDevice);

    copy<<<1,1>>>(cuda_src, cuda_dst, width, height);
    cudaDeviceSynchronize();

    // 逐行拷贝GPU结果回主机
    for(int i=0;i<width;i++){
        cudaMemcpy(dst_ptr[i], h_cuda_dst_rows[i], height * sizeof(double), cudaMemcpyDeviceToHost);
    }

    printf("Source: \n");
    for (int i = 0; i < width; i++) {
        for (int j = 0; j < height; j++) {
            printf("%f ", src_ptr[i][j]);
        }
        printf("\n");
    }
    printf("\n\nDest:\n");
    for (int i = 0; i < width; i++) {
        for (int j = 0; j < height; j++) {
            printf("%f ", dst_ptr[i][j]);
        }
        printf("\n");
    }

    // 内存释放省略,实际使用请补充
    return 0;
}
更推荐的连续内存实现(O(1)分配拷贝)

二级指针的锯齿数组实现需要多次分配、多次拷贝,性能开销远高于连续内存方案,你可以将二维数组展平为一维连续内存存储,用下标计算的方式访问元素,仅需1次显存分配、1次数据拷贝即可完成操作,代码更简洁性能也更高:

#include <stdio.h>
#include <cuda_runtime.h>

__global__
void copy(double *src, double *dst, int width, int height)
{
    int idx = blockIdx.x * blockDim.x + threadIdx.x;
    int total = width * height;
    if(idx < total) {
        dst[idx] = src[idx];
    }
}

int main(int argc, char *argv[]){
    int width = 3;
    int height = 5;
    int total = width * height;

    double *src_ptr = (double*)malloc(total * sizeof(double));
    double *dst_ptr = (double*)malloc(total * sizeof(double));
    double *cuda_src, *cuda_dst;

    // 初始化数据
    for(int i=0;i<width;i++){
        for(int j=0;j<height;j++){
            src_ptr[i*height + j] = (double)(rand() % 1000) / 1000;
            dst_ptr[i*height + j] = 0.;
        }
    }

    // 分配显存+拷贝数据,仅需各1次
    cudaMalloc(&cuda_src, total * sizeof(double));
    cudaMalloc(&cuda_dst, total * sizeof(double));
    cudaMemcpy(cuda_src, src_ptr, total * sizeof(double), cudaMemcpyHostToDevice);

    // 多线程并行拷贝,性能远高于单线程循环
    copy<<<(total + 255)/256, 256>>>(cuda_src, cuda_dst, width, height);
    cudaDeviceSynchronize();

    // 拷贝结果回主机
    cudaMemcpy(dst_ptr, cuda_dst, total * sizeof(double), cudaMemcpyDeviceToHost);

    printf("Source: \n");
    for (int i = 0; i < width; i++) {
        for (int j = 0; j < height; j++) {
            printf("%f ", src_ptr[i*height + j]);
        }
        printf("\n");
    }
    printf("\n\nDest:\n");
    for (int i = 0; i < width; i++) {
        for (int j = 0; j < height; j++) {
            printf("%f ", dst_ptr[i*height + j]);
        }
        printf("\n");
    }

    free(src_ptr);
    free(dst_ptr);
    cudaFree(cuda_src);
    cudaFree(cuda_dst);
    return 0;
}

内容的提问来源于stack exchange,提问作者DP_Corcoran

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.25 19:36:03