You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA并行数组加法程序未生效且无报错,求问题排查

CUDA并行数组加法程序未执行加法的问题排查

我编写了一款基于CUDA的GPU并行数组加法程序,但运行后未执行数组加法操作,且无任何报错。以下是完整代码及运行结果:

原代码

#include <cuda_runtime.h>
#include <cuda.h>
#include <iostream>
#include <stdlib.h>

using namespace std;

__global__ void AddInts(int *a, int *b, int count)
{
    int id = blockIdx.x * blockDim.x + threadIdx.x;
    if (id < count)
    {
        a[id] += b[id];
    }
}


int main() 
{
    srand(time(NULL));
    int count = 100;
    int *h_a = new int[count];
    int *h_b = new int[count];

    for (int i = 0; i < count; i++)
    {
        h_a[i] = rand() % 1000;
        h_b[i] = rand() % 1000;
    }

    cout << "Prior to addition:" << endl;
    for (int i = 0; i < 5; i++)
        cout << h_a[i] << " " << h_b[i] << endl;

    int *d_a, *d_b;

    if (cudaMalloc(&d_a, sizeof(int) * count) != cudaSuccess)
    {
        cout << "Nope! No";
        return 0;
    }

    if (cudaMalloc(&d_b, sizeof(int) * count) != cudaSuccess)
    {
        cout << "Nope!";
        cudaFree(d_a);
        return 0;
    }

    if (cudaMemcpy(d_a, h_a, sizeof(int) * count, cudaMemcpyHostToDevice) != cudaSuccess)
    {
        cout << "Could not copy!" << endl;
        cudaFree(d_a);
        cudaFree(d_b);
        return 0;
    }

    if (cudaMemcpy(d_b, h_b, sizeof(int) * count, cudaMemcpyHostToDevice) != cudaSuccess)
    {
        cout << "Could not copy!" << endl;
        cudaFree(d_a);
        cudaFree(d_b);
        return 0;
    }

    AddInts <<<count / 256 + 1, 256 >>> (h_a, h_b, count);

    if (cudaMemcpy(h_a, h_b, sizeof(int) * count, cudaMemcpyDeviceToHost) == cudaSuccess)
    {
        delete[] h_a;
        delete[] h_b;
        cudaFree(d_a);
        cudaFree(d_b);
        cout << "Nope!" << endl;
        return 0;
    }

    for (int i = 0; i < 5; i++)
        cout << "It's " << h_a[i] << endl;

    cudaFree(d_a);
    cudaFree(d_b);


    delete[] h_a;
    delete[] h_b;

    return 0;
}

运行结果

加法前:
188 336
489 593
706 673
330 792
329 588
结果为:188
结果为:489
结果为:706
结果为:330
结果为:329

D:\Learn\CUDA\Visual_stidio\matrxAdd\x64\Release\matrxAdd.exe (process
8468) exited with code 0. 调试停止时自动关闭控制台,请启用Tools->Options->Debugging->Automatically close
the console when debugging stops. 按任意键关闭窗口...

错误原因分析

  • 核函数传参错误:核函数AddInts需要接收设备内存指针,但代码调用时传入了主机内存指针h_a和h_b。GPU无法直接访问主机内存,导致核函数未对有效数据执行加法操作。
  • 内存拷贝逻辑完全错误:
    1. 拷贝源、目标和方向错误:应该将GPU计算后的d_a(设备内存)拷贝回主机h_a,但代码写成了从主机h_b拷贝到h_a,完全没读取GPU计算结果。
    2. 条件判断颠倒:代码判断cudaMemcpy成功就直接退出,导致后续结果输出代码未执行,正确逻辑应为拷贝失败时才报错退出。
  • 缺少核函数错误检查:核函数启动是异步操作,默认不会同步报错,需要手动检查才能捕获启动失败的问题。

修正后的代码

#include <cuda_runtime.h>
#include <cuda.h>
#include <iostream>
#include <stdlib.h>

using namespace std;

__global__ void AddInts(int *a, int *b, int count)
{
    int id = blockIdx.x * blockDim.x + threadIdx.x;
    if (id < count)
    {
        a[id] += b[id];
    }
}

// CUDA错误检查辅助宏
#define CHECK_CUDA_ERROR(err, msg) \
    if (err != cudaSuccess) { \
        cerr << msg << ": " << cudaGetErrorString(err) << endl; \
        exit(EXIT_FAILURE); \
    }

int main() 
{
    srand(time(NULL));
    int count = 100;
    int *h_a = new int[count];
    int *h_b = new int[count];

    for (int i = 0; i < count; i++)
    {
        h_a[i] = rand() % 1000;
        h_b[i] = rand() % 1000;
    }

    cout << "Prior to addition:" << endl;
    for (int i = 0; i < 5; i++)
        cout << h_a[i] << " " << h_b[i] << endl;

    int *d_a, *d_b;
    cudaError_t err;

    err = cudaMalloc(&d_a, sizeof(int) * count);
    CHECK_CUDA_ERROR(err, "Failed to allocate d_a");

    err = cudaMalloc(&d_b, sizeof(int) * count);
    CHECK_CUDA_ERROR(err, "Failed to allocate d_b");

    err = cudaMemcpy(d_a, h_a, sizeof(int) * count, cudaMemcpyHostToDevice);
    CHECK_CUDA_ERROR(err, "Failed to copy h_a to d_a");

    err = cudaMemcpy(d_b, h_b, sizeof(int) * count, cudaMemcpyHostToDevice);
    CHECK_CUDA_ERROR(err, "Failed to copy h_b to d_b");

    // 核函数传入设备指针
    AddInts <<<count / 256 + 1, 256 >>> (d_a, d_b, count);
    // 检查核函数启动错误
    err = cudaGetLastError();
    CHECK_CUDA_ERROR(err, "Failed to launch AddInts kernel");
    // 等待核函数执行完成
    err = cudaDeviceSynchronize();
    CHECK_CUDA_ERROR(err, "Kernel execution failed");

    // 将计算结果从设备拷贝回主机
    err = cudaMemcpy(h_a, d_a, sizeof(int) * count, cudaMemcpyDeviceToHost);
    CHECK_CUDA_ERROR(err, "Failed to copy d_a to h_a");

    cout << "\nAfter addition:" << endl;
    for (int i = 0; i < 5; i++)
        cout << "It's " << h_a[i] << endl;

    // 释放资源
    cudaFree(d_a);
    cudaFree(d_b);
    delete[] h_a;
    delete[] h_b;

    return 0;
}

修正说明

  • 核函数调用时传入设备指针d_a和d_b,确保GPU能访问正确的内存区域。
  • 修正内存拷贝的源、目标和方向,将GPU计算结果从d_a拷贝回主机h_a。
  • 添加CUDA错误检查宏,覆盖内存分配、拷贝、核函数启动等所有CUDA操作,能及时捕获失败原因。
  • 调整程序退出逻辑,仅在CUDA操作失败时退出,确保结果输出代码正常执行。

内容的提问来源于stack exchange,提问作者Viraj N H

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.26 05:24:27