CUDA并行数组加法程序未生效且无报错,求问题排查
CUDA并行数组加法程序未执行加法的问题排查
我编写了一款基于CUDA的GPU并行数组加法程序,但运行后未执行数组加法操作,且无任何报错。以下是完整代码及运行结果:
原代码
#include <cuda_runtime.h> #include <cuda.h> #include <iostream> #include <stdlib.h> using namespace std; __global__ void AddInts(int *a, int *b, int count) { int id = blockIdx.x * blockDim.x + threadIdx.x; if (id < count) { a[id] += b[id]; } } int main() { srand(time(NULL)); int count = 100; int *h_a = new int[count]; int *h_b = new int[count]; for (int i = 0; i < count; i++) { h_a[i] = rand() % 1000; h_b[i] = rand() % 1000; } cout << "Prior to addition:" << endl; for (int i = 0; i < 5; i++) cout << h_a[i] << " " << h_b[i] << endl; int *d_a, *d_b; if (cudaMalloc(&d_a, sizeof(int) * count) != cudaSuccess) { cout << "Nope! No"; return 0; } if (cudaMalloc(&d_b, sizeof(int) * count) != cudaSuccess) { cout << "Nope!"; cudaFree(d_a); return 0; } if (cudaMemcpy(d_a, h_a, sizeof(int) * count, cudaMemcpyHostToDevice) != cudaSuccess) { cout << "Could not copy!" << endl; cudaFree(d_a); cudaFree(d_b); return 0; } if (cudaMemcpy(d_b, h_b, sizeof(int) * count, cudaMemcpyHostToDevice) != cudaSuccess) { cout << "Could not copy!" << endl; cudaFree(d_a); cudaFree(d_b); return 0; } AddInts <<<count / 256 + 1, 256 >>> (h_a, h_b, count); if (cudaMemcpy(h_a, h_b, sizeof(int) * count, cudaMemcpyDeviceToHost) == cudaSuccess) { delete[] h_a; delete[] h_b; cudaFree(d_a); cudaFree(d_b); cout << "Nope!" << endl; return 0; } for (int i = 0; i < 5; i++) cout << "It's " << h_a[i] << endl; cudaFree(d_a); cudaFree(d_b); delete[] h_a; delete[] h_b; return 0; }
运行结果
加法前: 188 336 489 593 706 673 330 792 329 588 结果为:188 结果为:489 结果为:706 结果为:330 结果为:329 D:\Learn\CUDA\Visual_stidio\matrxAdd\x64\Release\matrxAdd.exe (process 8468) exited with code 0. 调试停止时自动关闭控制台,请启用Tools->Options->Debugging->Automatically close the console when debugging stops. 按任意键关闭窗口...
错误原因分析
- 核函数传参错误:核函数
AddInts需要接收设备内存指针,但代码调用时传入了主机内存指针h_a和h_b。GPU无法直接访问主机内存,导致核函数未对有效数据执行加法操作。 - 内存拷贝逻辑完全错误:
- 拷贝源、目标和方向错误:应该将GPU计算后的
d_a(设备内存)拷贝回主机h_a,但代码写成了从主机h_b拷贝到h_a,完全没读取GPU计算结果。 - 条件判断颠倒:代码判断
cudaMemcpy成功就直接退出,导致后续结果输出代码未执行,正确逻辑应为拷贝失败时才报错退出。
- 拷贝源、目标和方向错误:应该将GPU计算后的
- 缺少核函数错误检查:核函数启动是异步操作,默认不会同步报错,需要手动检查才能捕获启动失败的问题。
修正后的代码
#include <cuda_runtime.h> #include <cuda.h> #include <iostream> #include <stdlib.h> using namespace std; __global__ void AddInts(int *a, int *b, int count) { int id = blockIdx.x * blockDim.x + threadIdx.x; if (id < count) { a[id] += b[id]; } } // CUDA错误检查辅助宏 #define CHECK_CUDA_ERROR(err, msg) \ if (err != cudaSuccess) { \ cerr << msg << ": " << cudaGetErrorString(err) << endl; \ exit(EXIT_FAILURE); \ } int main() { srand(time(NULL)); int count = 100; int *h_a = new int[count]; int *h_b = new int[count]; for (int i = 0; i < count; i++) { h_a[i] = rand() % 1000; h_b[i] = rand() % 1000; } cout << "Prior to addition:" << endl; for (int i = 0; i < 5; i++) cout << h_a[i] << " " << h_b[i] << endl; int *d_a, *d_b; cudaError_t err; err = cudaMalloc(&d_a, sizeof(int) * count); CHECK_CUDA_ERROR(err, "Failed to allocate d_a"); err = cudaMalloc(&d_b, sizeof(int) * count); CHECK_CUDA_ERROR(err, "Failed to allocate d_b"); err = cudaMemcpy(d_a, h_a, sizeof(int) * count, cudaMemcpyHostToDevice); CHECK_CUDA_ERROR(err, "Failed to copy h_a to d_a"); err = cudaMemcpy(d_b, h_b, sizeof(int) * count, cudaMemcpyHostToDevice); CHECK_CUDA_ERROR(err, "Failed to copy h_b to d_b"); // 核函数传入设备指针 AddInts <<<count / 256 + 1, 256 >>> (d_a, d_b, count); // 检查核函数启动错误 err = cudaGetLastError(); CHECK_CUDA_ERROR(err, "Failed to launch AddInts kernel"); // 等待核函数执行完成 err = cudaDeviceSynchronize(); CHECK_CUDA_ERROR(err, "Kernel execution failed"); // 将计算结果从设备拷贝回主机 err = cudaMemcpy(h_a, d_a, sizeof(int) * count, cudaMemcpyDeviceToHost); CHECK_CUDA_ERROR(err, "Failed to copy d_a to h_a"); cout << "\nAfter addition:" << endl; for (int i = 0; i < 5; i++) cout << "It's " << h_a[i] << endl; // 释放资源 cudaFree(d_a); cudaFree(d_b); delete[] h_a; delete[] h_b; return 0; }
修正说明
- 核函数调用时传入设备指针
d_a和d_b,确保GPU能访问正确的内存区域。 - 修正内存拷贝的源、目标和方向,将GPU计算结果从
d_a拷贝回主机h_a。 - 添加CUDA错误检查宏,覆盖内存分配、拷贝、核函数启动等所有CUDA操作,能及时捕获失败原因。
- 调整程序退出逻辑,仅在CUDA操作失败时退出,确保结果输出代码正常执行。
内容的提问来源于stack exchange,提问作者Viraj N H
相关产品推荐
相关产品推荐

