You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA Thrust核能否多流并行?fill操作串行问题求助

Thrust多流并行fill操作串行执行问题排查与解决

问题描述

尝试在不同CUDA流上并行启动thrust::fill操作填充两个设备向量,但NSight Systems显示内核串行执行,且存在持续的cudaStreamSynchronize()隐式调用。自定义内核可正常并行,但希望基于Thrust API实现。

测试代码

#include <thrust/device_vector.h>
#include <thrust/fill.h>
#include <thrust/sort.h>
#include <thrust/transform.h>
#include <thrust/execution_policy.h>

#define gpuErrchk(ans)                        \
    {                                         \
        gpuAssert((ans), __FILE__, __LINE__); \
    }
inline void gpuAssert(cudaError_t code, const char* file, int line, bool abort = true)
{
    if(code != cudaSuccess)
        {
            fprintf(stderr, "GPUassert: %s %s %d\n", cudaGetErrorString(code), file, line);
            if(abort) exit(code);
        }
}

int main(void)
{
    cudaStream_t stream1, stream2;
    gpuErrchk(cudaStreamCreate(&stream1));
    gpuErrchk(cudaStreamCreate(&stream2));

    const size_t size = 10000000;

    int* d_test1_ptr;
    int* d_test2_ptr;
    gpuErrchk(cudaMalloc((void**)&d_test1_ptr, size * sizeof(int)));
    gpuErrchk(cudaMalloc((void**)&d_test2_ptr, size * sizeof(int)));

    thrust::device_ptr<int> d_test1(d_test1_ptr);
    thrust::device_ptr<int> d_test2(d_test2_ptr);

    for(int i = 0; i < 100; i++)
        {
            thrust::fill(thrust::cuda::par.on(stream1), d_test1, d_test1 + size, 2);
            thrust::fill(thrust::cuda::par.on(stream2), d_test2, d_test2 + size, 2);
        }

    gpuErrchk(cudaStreamSynchronize(stream1));
    gpuErrchk(cudaStreamSynchronize(stream2));

    gpuErrchk(cudaFree(d_test1_ptr));
    gpuErrchk(cudaFree(d_test2_ptr));

    gpuErrchk(cudaStreamDestroy(stream1));
    gpuErrchk(cudaStreamDestroy(stream2));

    std::cout << "Completed execution of dummy functions on different streams." << std::endl;

    return 0;
}

问题原因

  1. 原始CUDA指针的流绑定缺失:手动通过cudaMalloc分配的内存未与任何流绑定,Thrust在操作这类指针时,为保证内存操作安全性,会隐式插入同步操作,导致流串行化。
  2. Thrust内部的隐式同步机制:当执行策略的流与内存资源的流上下文不匹配时,Thrust会触发cudaStreamSynchronize同步默认流与指定流,造成串行执行。

解决方案

方案1:使用Thrust原生device_vector

thrust::device_vector会自动与执行策略指定的流关联,避免隐式同步。修改后的代码如下:

#include <thrust/device_vector.h>
#include <thrust/fill.h>
#include <thrust/execution_policy.h>
#include <iostream>

#define gpuErrchk(ans)                        \
    {                                         \
        gpuAssert((ans), __FILE__, __LINE__); \
    }
inline void gpuAssert(cudaError_t code, const char* file, int line, bool abort = true)
{
    if(code != cudaSuccess)
        {
            fprintf(stderr, "GPUassert: %s %s %d\n", cudaGetErrorString(code), file, line);
            if(abort) exit(code);
        }
}

int main(void)
{
    cudaStream_t stream1, stream2;
    gpuErrchk(cudaStreamCreate(&stream1));
    gpuErrchk(cudaStreamCreate(&stream2));

    const size_t size = 10000000;

    // 用Thrust device_vector替代原始指针
    thrust::device_vector<int> d_test1(size);
    thrust::device_vector<int> d_test2(size);

    for(int i = 0; i < 100; i++)
        {
            thrust::fill(thrust::cuda::par.on(stream1), d_test1.begin(), d_test1.end(), 2);
            thrust::fill(thrust::cuda::par.on(stream2), d_test2.begin(), d_test2.end(), 2);
        }

    gpuErrchk(cudaStreamSynchronize(stream1));
    gpuErrchk(cudaStreamSynchronize(stream2));

    gpuErrchk(cudaStreamDestroy(stream1));
    gpuErrchk(cudaStreamDestroy(stream2));

    std::cout << "Completed execution of dummy functions on different streams." << std::endl;

    return 0;
}

方案2:绑定原始指针到指定流(可选)

若必须使用原始CUDA指针,需通过Thrust的流感知分配器分配内存,确保内存与流绑定:

// 替换cudaMalloc为Thrust的流感知分配
thrust::device_ptr<int> d_test1 = thrust::cuda::par.on(stream1).malloc<int>(size);
thrust::device_ptr<int> d_test2 = thrust::cuda::par.on(stream2).malloc<int>(size);

// 释放时也用Thrust的接口
thrust::cuda::par.on(stream1).free(d_test1);
thrust::cuda::par.on(stream2).free(d_test2);

方案3:禁用Thrust隐式同步(谨慎使用)

设置环境变量THRUST_DEVICE_SYNC=0可禁止Thrust的隐式同步,但需自行确保所有操作的同步逻辑正确,避免数据竞争。

验证

修改后通过NSight Systems观察,cudaStreamSynchronize()仅会出现在手动调用的位置,两个流的thrust::fill内核将并行执行。

内容的提问来源于stack exchange,提问作者Nicolas Perrault

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 15:54:53