使用cudaMalloc复制二级double指针数组到CUDA显存不生效问题排查
错误原因
你的代码存在4个核心问题导致输出全为0:
cudaMalloc传参错误:cudaMalloc第一个参数要求传入存储GPU地址的主机变量的指针,原有写法cudaMalloc(cuda_src, width * sizeof(double*));直接传入了未初始化的野指针cuda_src,会导致显存分配失败,正确写法应为cudaMalloc(&cuda_src, width * sizeof(double*));,cuda_dst同理。- 主机端直接访问GPU指针非法:
cuda_src是存储在主机内存的GPU地址,指向GPU上的一级指针数组,主机端不能直接解引用cuda_src[i],原有写法&cuda_src[i]属于非法内存访问。 - 缺少实际数据的主机到显存拷贝:你仅将
src_ptr中存储的行指针拷贝到了GPU,src_ptr[i]指向的主机侧实际double数组没有拷贝到对应GPU行地址中,内核读取到的源数据本身就是无效值。 - 缺少实际数据的显存到主机拷贝:内核执行完成后,你仅拷贝了GPU侧的一级指针数组到主机
dst_ptr,此时dst_ptr[i]存储的是GPU显存地址,主机直接访问dst_ptr[i][j]无法读取到显存中的实际数据,需要逐行把显存数据拷贝回主机内存。
修正后的二级指针实现(兼容原有写法)
#include <stdio.h> #include <assert.h> #include <cuda.h> #include <cuda_runtime.h> __global__ void copy(double **src, double **dst, int width, int height) { for (int i = 0; i < width; i++) { for (int j = 0; j < height; j++) { dst[i][j] = src[i][j]; } } } int main(int argc, char *argv[]){ int width = 3; int height = 5; double **src_ptr; double **cuda_src; double **dst_ptr; double **cuda_dst; // 主机侧临时存储GPU行地址的数组 double **h_cuda_src_rows, **h_cuda_dst_rows; src_ptr = (double **)malloc(width * sizeof(double*)); dst_ptr = (double **)malloc(width * sizeof(double*)); h_cuda_src_rows = (double **)malloc(width * sizeof(double*)); h_cuda_dst_rows = (double **)malloc(width * sizeof(double*)); // 分配GPU侧一级指针数组 cudaMalloc(&cuda_src, width * sizeof(double*)); cudaMalloc(&cuda_dst, width * sizeof(double*)); for(int i = 0; i<width;i++){ src_ptr[i] = (double *)malloc(height * sizeof *src_ptr[i]); dst_ptr[i] = (double *)malloc(height * sizeof *dst_ptr[i]); // 分配GPU行内存,地址存在主机临时数组中 cudaMalloc((void ** ) &h_cuda_src_rows[i], height * sizeof(double)); cudaMalloc((void ** ) &h_cuda_dst_rows[i], height * sizeof(double)); for(int j=0;j<height;j++){ src_ptr[i][j] = (double)(rand() % 1000) / 1000; dst_ptr[i][j] = 0.; } // 拷贝每行实际数据到GPU cudaMemcpy(h_cuda_src_rows[i], src_ptr[i], height * sizeof(double), cudaMemcpyHostToDevice); } // 将GPU行地址数组拷贝到GPU侧一级指针数组 cudaMemcpy(cuda_src, h_cuda_src_rows, width * sizeof(double*), cudaMemcpyHostToDevice); cudaMemcpy(cuda_dst, h_cuda_dst_rows, width * sizeof(double*), cudaMemcpyHostToDevice); copy<<<1,1>>>(cuda_src, cuda_dst, width, height); cudaDeviceSynchronize(); // 逐行拷贝GPU结果回主机 for(int i=0;i<width;i++){ cudaMemcpy(dst_ptr[i], h_cuda_dst_rows[i], height * sizeof(double), cudaMemcpyDeviceToHost); } printf("Source: \n"); for (int i = 0; i < width; i++) { for (int j = 0; j < height; j++) { printf("%f ", src_ptr[i][j]); } printf("\n"); } printf("\n\nDest:\n"); for (int i = 0; i < width; i++) { for (int j = 0; j < height; j++) { printf("%f ", dst_ptr[i][j]); } printf("\n"); } // 内存释放省略,实际使用请补充 return 0; }
更推荐的连续内存实现(O(1)分配拷贝)
二级指针的锯齿数组实现需要多次分配、多次拷贝,性能开销远高于连续内存方案,你可以将二维数组展平为一维连续内存存储,用下标计算的方式访问元素,仅需1次显存分配、1次数据拷贝即可完成操作,代码更简洁性能也更高:
#include <stdio.h> #include <cuda_runtime.h> __global__ void copy(double *src, double *dst, int width, int height) { int idx = blockIdx.x * blockDim.x + threadIdx.x; int total = width * height; if(idx < total) { dst[idx] = src[idx]; } } int main(int argc, char *argv[]){ int width = 3; int height = 5; int total = width * height; double *src_ptr = (double*)malloc(total * sizeof(double)); double *dst_ptr = (double*)malloc(total * sizeof(double)); double *cuda_src, *cuda_dst; // 初始化数据 for(int i=0;i<width;i++){ for(int j=0;j<height;j++){ src_ptr[i*height + j] = (double)(rand() % 1000) / 1000; dst_ptr[i*height + j] = 0.; } } // 分配显存+拷贝数据,仅需各1次 cudaMalloc(&cuda_src, total * sizeof(double)); cudaMalloc(&cuda_dst, total * sizeof(double)); cudaMemcpy(cuda_src, src_ptr, total * sizeof(double), cudaMemcpyHostToDevice); // 多线程并行拷贝,性能远高于单线程循环 copy<<<(total + 255)/256, 256>>>(cuda_src, cuda_dst, width, height); cudaDeviceSynchronize(); // 拷贝结果回主机 cudaMemcpy(dst_ptr, cuda_dst, total * sizeof(double), cudaMemcpyDeviceToHost); printf("Source: \n"); for (int i = 0; i < width; i++) { for (int j = 0; j < height; j++) { printf("%f ", src_ptr[i*height + j]); } printf("\n"); } printf("\n\nDest:\n"); for (int i = 0; i < width; i++) { for (int j = 0; j < height; j++) { printf("%f ", dst_ptr[i*height + j]); } printf("\n"); } free(src_ptr); free(dst_ptr); cudaFree(cuda_src); cudaFree(cuda_dst); return 0; }
内容的提问来源于stack exchange,提问作者DP_Corcoran
相关产品推荐
相关产品推荐

