You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA Graph示例未输出预期结果,请求问题排查

CUDA Graph内存拷贝后打印结果异常问题

场景与图结构

测试CUDA Graph时,设计的图结构如下:
CUDA Graph结构:Kernel1执行后,并行执行Kernel2和Kernel3,完成后执行设备到主机的内存拷贝,最后执行主机打印节点

实现代码

#include <cstdio>
#include <cstdlib>
#include <fstream>
#include <iostream>
#include <vector>

#define NumThreads 20
#define NumBlocks 1

template <typename PtrType>
__global__ void kernel1(PtrType *buffer, unsigned int numElems) {
  int tid = threadIdx.x + blockIdx.x * blockDim.x;
  buffer[tid] = (PtrType)tid;
}

template <typename PtrType>
__global__ void kernel2(PtrType *buffer, unsigned int numElems) {
  int tid = threadIdx.x + blockIdx.x * blockDim.x;
  if(tid < numElems/2) buffer[tid] += 5;
}

template <typename PtrType>
__global__ void kernel3(PtrType *buffer, unsigned int numElems) {
  int tid = threadIdx.x + blockIdx.x * blockDim.x;
  if(tid>=numElems/2) buffer[tid] *= 5;
}

template <typename PtrType> 
void print(void *data) {
    PtrType *buffer = (PtrType *)data;
    std::cout << "[";
    for (unsigned int i = 0; i < NumThreads; ++i) {
        std::cout << buffer[i] << ",";
    }
    std::cout << "]\n";
  }

void runCudaGraph(cudaGraph_t &Graph, cudaGraphExec_t &graphExec,
                  cudaStream_t &graphStream) {
  cudaGraphInstantiate(&graphExec, Graph, nullptr, nullptr, 0);
  cudaStreamCreateWithFlags(&graphStream, cudaStreamNonBlocking);
  cudaGraphLaunch(graphExec, graphStream);
  cudaStreamSynchronize(graphStream);
}

void destroyCudaGraph(cudaGraph_t &Graph, cudaGraphExec_t &graphExec,
                      cudaStream_t &graphStream) {
  cudaCtxResetPersistingL2Cache();

  cudaGraphExecDestroy(graphExec);
  cudaGraphDestroy(Graph);
  cudaStreamDestroy(graphStream);
  cudaDeviceReset();
}

template <typename PtrType>
void createCudaGraph(cudaGraph_t &Graph, cudaGraphExec_t &graphExec,
                     cudaStream_t &graphStream, PtrType *buffer,
                     unsigned int numElems, PtrType *hostBuffer) {
  cudaGraphCreate(&Graph, 0);

  cudaGraphNode_t Kernel1;
  cudaKernelNodeParams nodeParams = {0};
  memset(&nodeParams, 0, sizeof(nodeParams));
  nodeParams.func = (void *)kernel1<PtrType>;
  nodeParams.gridDim = dim3(NumBlocks, 1, 1);
  nodeParams.blockDim = dim3(NumThreads/NumBlocks, 1, 1);
  nodeParams.sharedMemBytes = 0;
  void *inputs[2];
  inputs[0] = (void *)&buffer;
  inputs[1] = (void *)&numElems;
  nodeParams.kernelParams = inputs;
  nodeParams.extra = nullptr;

  cudaGraphAddKernelNode(&Kernel1, Graph, nullptr, 0, &nodeParams);

  cudaGraphNode_t Kernel2;
  memset(&nodeParams, 0, sizeof(nodeParams));
  nodeParams.func = (void *)kernel2<PtrType>;
  nodeParams.gridDim = dim3(NumBlocks, 1, 1);
  nodeParams.blockDim = dim3(NumThreads/NumBlocks, 1, 1);
  nodeParams.sharedMemBytes = 0;
  inputs[0] = (void *)&buffer;
  inputs[1] = (void *)&numElems;
  nodeParams.kernelParams = inputs;
  nodeParams.extra = NULL;

  cudaGraphAddKernelNode(&Kernel2, Graph, &Kernel1, 1, &nodeParams);

  cudaGraphNode_t Kernel3;
  memset(&nodeParams, 0, sizeof(nodeParams));
  nodeParams.func = (void *)kernel3<PtrType>;
  nodeParams.gridDim = dim3(NumBlocks, 1, 1);
  nodeParams.blockDim = dim3(NumThreads/NumBlocks, 1, 1);
  nodeParams.sharedMemBytes = 0;
  inputs[0] = (void *)&buffer;
  inputs[1] = (void *)&numElems;
  nodeParams.kernelParams = inputs;
  nodeParams.extra = NULL;

  cudaGraphAddKernelNode(&Kernel3, Graph, &Kernel1, 1, &nodeParams);

  cudaGraphNode_t copyBuffer;
  std::vector<cudaGraphNode_t> dependencies = {Kernel2, Kernel3};
  cudaGraphAddMemcpyNode1D(&copyBuffer, Graph,dependencies.data(),dependencies.size(),hostBuffer, buffer, numElems*sizeof(PtrType), cudaMemcpyDeviceToHost);

  cudaGraphNode_t Host1;
  cudaHostNodeParams hostNodeParams;
  memset(&hostNodeParams, 0, sizeof(hostNodeParams));
  hostNodeParams.fn = print<PtrType>;
  hostNodeParams.userData = (void *)&hostBuffer; // 问题所在行
  cudaGraphAddHostNode(&Host1, Graph, &copyBuffer, 1,
                       &hostNodeParams);
}

int main() {
  cudaGraph_t graph;
  cudaGraphExec_t graphExec;
  cudaStream_t graphStream;

  unsigned int numElems = NumThreads;
  unsigned int bufferSizeBytes = numElems * sizeof(unsigned int);
  unsigned int hostBuffer[numElems];
  memset(hostBuffer, 0, bufferSizeBytes);
  unsigned int *deviceBuffer;
  cudaMalloc(&deviceBuffer, bufferSizeBytes);
  createCudaGraph(graph, graphExec, graphStream, deviceBuffer,numElems, hostBuffer);
  runCudaGraph(graph, graphExec, graphStream);
  destroyCudaGraph(graph, graphExec, graphStream);
  std::cout << "graph example done!" << std::endl;
}

运行结果

实际输出(乱码):

[3593293488,22096,3561843129,22096,3561385808,22096,3593293488,22096,3598681264,22096,3561792984,22096,2687342880,0,0,0,3598597376,22096,3598599312,0,]

预期输出:

[5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 50, 55, 60, 65, 70, 75, 80, 85, 90, 95]

调试信息

使用cuda-gdb调试确认GPU端结果正确,但设备内存拷贝到主机后,打印输出出现乱码,无法定位问题根源。


问题原因与修复

问题根源

在createCudaGraph函数添加Host打印节点时,传递的参数错误:

hostNodeParams.userData = (void *)&hostBuffer;

这里传递的是hostBuffer数组的地址(即指针的指针),但print函数中直接将data转为PtrType *buffer,相当于把数组的地址当成数组首元素地址来访问,读取的是无效内存区域,导致输出乱码。

修复方法

将上述代码改为直接传递hostBuffer的首地址:

hostNodeParams.userData = (void *)hostBuffer;

验证

修复后,内存拷贝完成后,print函数能正确读取主机数组中的数据,输出符合预期结果。


内容的提问来源于stack exchange,提问作者M46f988b814

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.14 13:01:16