CUDA Graph实现遇段错误求排查(附可复现代码)
CUDA Graph 段错误问题分析与修复
你的代码出现段错误的核心原因是CUDA Graph Kernel节点的参数传递方式错误,具体问题和修复方案如下:
问题根源
CUDA Graph在实例化(cudaGraphInstantiate)时会捕获kernel参数的内容,但你当前传递的是主机栈变量的地址,而非设备能访问的有效内存地址:
&d_object_n:这是主机内存中存储设备指针的变量的地址,设备端无法直接访问主机栈内存&n_ant:同样是主机栈上的标量变量地址,设备端访问会导致非法内存访问
另外,代码还有两处细节问题:
cudaGraphLaunch的错误提示字符串写错,应该对应"Failed to cudaGraphLaunch"- 没有同步设备或等待图执行完成,程序直接退出可能导致未定义行为
修复后的代码
#include <thrust/device_vector.h> #include <thrust/host_vector.h> #include <iostream> __global__ void mdl(int *n_out, const float *eigen_vals, int n) { __shared__ float shm[8]; int tid = threadIdx.x; float threshold; if (tid < n) { shm[tid] = eigen_vals[tid]; __syncthreads(); if (tid == (n-1)) { threshold = shm[tid] * 0.01; int count = 0; for (int i = 0;i < n;++i) { count += (shm[i] >= threshold) ? 1 : 0; } *n_out = count; } } } int main() { int *d_object_n; cudaMalloc((void**)&d_object_n, sizeof(int)); cudaMemset(d_object_n, 0, sizeof(int)); thrust::device_vector<float> d_eigen_values{1, 2, 3, 4, 5, 6, 7, 8}; int n_ant = 8; cudaGraphNode_t nodeMDL; cudaGraph_t m_graph; cudaGraphExec_t m_graphExec; cudaGraphCreate(&m_graph, 0); cudaKernelNodeParams nodeMDL_param = {0}; nodeMDL_param.func = (void*)mdl; nodeMDL_param.gridDim = dim3(1); nodeMDL_param.blockDim = dim3(8); nodeMDL_param.sharedMemBytes = 0; nodeMDL_param.extra = nullptr; // 修复参数传递:直接传递设备指针和标量值,而非主机变量地址 void* args[] = { d_object_n, (void*)(d_eigen_values.data().get()), (void*)&n_ant }; nodeMDL_param.kernelParams = args; cudaError_t err = cudaGraphAddKernelNode(&nodeMDL, m_graph, nullptr, 0, &nodeMDL_param); if (err != cudaSuccess) { std::cerr << "Failed to cudaGraphAddKernelNode: " << cudaGetErrorString(err) << std::endl; return 1; } err = cudaGraphInstantiate(&m_graphExec, m_graph); if (err != cudaSuccess) { std::cerr << "Failed to cudaGraphInstantiate: " << cudaGetErrorString(err) << std::endl; return 1; } err = cudaGraphLaunch(m_graphExec, 0); if (err != cudaSuccess) { std::cerr << "Failed to cudaGraphLaunch: " << cudaGetErrorString(err) << std::endl; return 1; } // 等待图执行完成 cudaDeviceSynchronize(); // 验证结果(可选) int h_result; cudaMemcpy(&h_result, d_object_n, sizeof(int), cudaMemcpyDeviceToHost); std::cout << "Result: " << h_result << std::endl; // 释放资源 cudaFree(d_object_n); cudaGraphExecDestroy(m_graphExec); cudaGraphDestroy(m_graph); return 0; }
关键修复点
- 参数传递修正:
- 设备指针参数
d_object_n直接传递其值,而非&d_object_n - 标量参数
n_ant传递其地址是可行的(CUDA会在实例化时捕获该地址的值),但要确保该变量在实例化前不会被销毁或修改
- 设备指针参数
- 添加同步与资源释放:
- 使用
cudaDeviceSynchronize()等待图执行完成,避免程序提前退出导致设备操作中断 - 新增资源销毁代码,避免内存泄漏
- 使用
- 错误提示修正:修正了
cudaGraphLaunch的错误提示字符串
内容的提问来源于stack exchange,提问作者Weimin Chan
相关产品推荐
相关产品推荐

