CUDA Graph示例未输出预期结果,请求问题排查
CUDA Graph内存拷贝后打印结果异常问题
场景与图结构
测试CUDA Graph时,设计的图结构如下:
实现代码
#include <cstdio> #include <cstdlib> #include <fstream> #include <iostream> #include <vector> #define NumThreads 20 #define NumBlocks 1 template <typename PtrType> __global__ void kernel1(PtrType *buffer, unsigned int numElems) { int tid = threadIdx.x + blockIdx.x * blockDim.x; buffer[tid] = (PtrType)tid; } template <typename PtrType> __global__ void kernel2(PtrType *buffer, unsigned int numElems) { int tid = threadIdx.x + blockIdx.x * blockDim.x; if(tid < numElems/2) buffer[tid] += 5; } template <typename PtrType> __global__ void kernel3(PtrType *buffer, unsigned int numElems) { int tid = threadIdx.x + blockIdx.x * blockDim.x; if(tid>=numElems/2) buffer[tid] *= 5; } template <typename PtrType> void print(void *data) { PtrType *buffer = (PtrType *)data; std::cout << "["; for (unsigned int i = 0; i < NumThreads; ++i) { std::cout << buffer[i] << ","; } std::cout << "]\n"; } void runCudaGraph(cudaGraph_t &Graph, cudaGraphExec_t &graphExec, cudaStream_t &graphStream) { cudaGraphInstantiate(&graphExec, Graph, nullptr, nullptr, 0); cudaStreamCreateWithFlags(&graphStream, cudaStreamNonBlocking); cudaGraphLaunch(graphExec, graphStream); cudaStreamSynchronize(graphStream); } void destroyCudaGraph(cudaGraph_t &Graph, cudaGraphExec_t &graphExec, cudaStream_t &graphStream) { cudaCtxResetPersistingL2Cache(); cudaGraphExecDestroy(graphExec); cudaGraphDestroy(Graph); cudaStreamDestroy(graphStream); cudaDeviceReset(); } template <typename PtrType> void createCudaGraph(cudaGraph_t &Graph, cudaGraphExec_t &graphExec, cudaStream_t &graphStream, PtrType *buffer, unsigned int numElems, PtrType *hostBuffer) { cudaGraphCreate(&Graph, 0); cudaGraphNode_t Kernel1; cudaKernelNodeParams nodeParams = {0}; memset(&nodeParams, 0, sizeof(nodeParams)); nodeParams.func = (void *)kernel1<PtrType>; nodeParams.gridDim = dim3(NumBlocks, 1, 1); nodeParams.blockDim = dim3(NumThreads/NumBlocks, 1, 1); nodeParams.sharedMemBytes = 0; void *inputs[2]; inputs[0] = (void *)&buffer; inputs[1] = (void *)&numElems; nodeParams.kernelParams = inputs; nodeParams.extra = nullptr; cudaGraphAddKernelNode(&Kernel1, Graph, nullptr, 0, &nodeParams); cudaGraphNode_t Kernel2; memset(&nodeParams, 0, sizeof(nodeParams)); nodeParams.func = (void *)kernel2<PtrType>; nodeParams.gridDim = dim3(NumBlocks, 1, 1); nodeParams.blockDim = dim3(NumThreads/NumBlocks, 1, 1); nodeParams.sharedMemBytes = 0; inputs[0] = (void *)&buffer; inputs[1] = (void *)&numElems; nodeParams.kernelParams = inputs; nodeParams.extra = NULL; cudaGraphAddKernelNode(&Kernel2, Graph, &Kernel1, 1, &nodeParams); cudaGraphNode_t Kernel3; memset(&nodeParams, 0, sizeof(nodeParams)); nodeParams.func = (void *)kernel3<PtrType>; nodeParams.gridDim = dim3(NumBlocks, 1, 1); nodeParams.blockDim = dim3(NumThreads/NumBlocks, 1, 1); nodeParams.sharedMemBytes = 0; inputs[0] = (void *)&buffer; inputs[1] = (void *)&numElems; nodeParams.kernelParams = inputs; nodeParams.extra = NULL; cudaGraphAddKernelNode(&Kernel3, Graph, &Kernel1, 1, &nodeParams); cudaGraphNode_t copyBuffer; std::vector<cudaGraphNode_t> dependencies = {Kernel2, Kernel3}; cudaGraphAddMemcpyNode1D(©Buffer, Graph,dependencies.data(),dependencies.size(),hostBuffer, buffer, numElems*sizeof(PtrType), cudaMemcpyDeviceToHost); cudaGraphNode_t Host1; cudaHostNodeParams hostNodeParams; memset(&hostNodeParams, 0, sizeof(hostNodeParams)); hostNodeParams.fn = print<PtrType>; hostNodeParams.userData = (void *)&hostBuffer; // 问题所在行 cudaGraphAddHostNode(&Host1, Graph, ©Buffer, 1, &hostNodeParams); } int main() { cudaGraph_t graph; cudaGraphExec_t graphExec; cudaStream_t graphStream; unsigned int numElems = NumThreads; unsigned int bufferSizeBytes = numElems * sizeof(unsigned int); unsigned int hostBuffer[numElems]; memset(hostBuffer, 0, bufferSizeBytes); unsigned int *deviceBuffer; cudaMalloc(&deviceBuffer, bufferSizeBytes); createCudaGraph(graph, graphExec, graphStream, deviceBuffer,numElems, hostBuffer); runCudaGraph(graph, graphExec, graphStream); destroyCudaGraph(graph, graphExec, graphStream); std::cout << "graph example done!" << std::endl; }
运行结果
实际输出(乱码):
[3593293488,22096,3561843129,22096,3561385808,22096,3593293488,22096,3598681264,22096,3561792984,22096,2687342880,0,0,0,3598597376,22096,3598599312,0,]
预期输出:
[5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 50, 55, 60, 65, 70, 75, 80, 85, 90, 95]
调试信息
使用cuda-gdb调试确认GPU端结果正确,但设备内存拷贝到主机后,打印输出出现乱码,无法定位问题根源。
问题原因与修复
问题根源
在createCudaGraph函数添加Host打印节点时,传递的参数错误:
hostNodeParams.userData = (void *)&hostBuffer;
这里传递的是hostBuffer数组的地址(即指针的指针),但print函数中直接将data转为PtrType *buffer,相当于把数组的地址当成数组首元素地址来访问,读取的是无效内存区域,导致输出乱码。
修复方法
将上述代码改为直接传递hostBuffer的首地址:
hostNodeParams.userData = (void *)hostBuffer;
验证
修复后,内存拷贝完成后,print函数能正确读取主机数组中的数据,输出符合预期结果。
内容的提问来源于stack exchange,提问作者M46f988b814
相关产品推荐
相关产品推荐

