You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

RTX2080 Max-Q无法达到32bit/周期共享内存带宽问题排查

RTX 2080 Max-Q(计算能力7.5)无法达到Shared Memory 32bit/周期带宽的问题

我正在使用RTX 2080 Max-Q移动显卡,其计算能力为7.5,目前在排查为何无法达到32bit/周期的shared memory带宽。

根据NVIDIA CUDA编程指南中的描述:

Shared memory拥有32个bank,连续的32位字映射到连续的bank。每个bank的带宽为每时钟周期32位。

我编写了以下测试代码:

#include <iostream>
#include <algorithm>
#include <numeric>

using T = float;
extern __shared__ T bank[];

constexpr int warps = 8;
constexpr int pitch = 32 * warps;
constexpr int size = 32;

__managed__ long long starts[pitch];
__managed__ long long stops[pitch];
__managed__ long long clocks[pitch];

__global__ void kernel()
{
    auto* local_bank = bank + threadIdx.x;
    auto* a = local_bank;
    auto* b = a + size * pitch;

    __syncwarp();
    auto start = clock64();
    __syncwarp();

    for (int i = 0; i < size; i++)
        b[i * pitch] = a[i * pitch];

    __syncwarp();
    auto stop = clock64();
    __syncwarp();

    auto duration = stop - start;
    printf("%5lld %s", duration, threadIdx.x % 32 == 31 ? "\n" : ""); 
    __syncthreads();
    starts[threadIdx.x + blockDim.x *blockIdx.x] = start;
    stops[threadIdx.x + blockDim.x *blockIdx.x] = stop;
    clocks[threadIdx.x + blockDim.x *blockIdx.x] = duration;
}

int main()
{
    cudaDeviceSetLimit(cudaLimitStackSize, 64 * 1024);
    cudaFuncSetAttribute(kernel, cudaFuncAttributeMaxDynamicSharedMemorySize, 64*1024);
    cudaFuncSetCacheConfig(kernel, cudaFuncCachePreferShared);
    kernel<<<1, pitch, 2 * size * pitch * sizeof(T)>>>();
    cudaDeviceSynchronize();    

    auto min_clock = *std::min_element(std::begin(clocks), std::end(clocks));
    auto min_start = *std::min_element(std::begin(starts), std::end(starts));
    auto max_stop = *std::max_element(std::begin(stops), std::end(stops));
    auto avg_clock = std::accumulate(std::begin(clocks), std::end(clocks), 0) / (float)(std::end(clocks) - std::begin(clocks));
    auto avg_clock_per_access = avg_clock / (float)(warps * size * 2);
    printf("min = %lli\n", min_clock);
    printf("max = %lli\n", max_stop - min_start);
    printf("avg = %f\n", avg_clock_per_access);
}

该代码以时钟周期为单位测量每个线程的shared memory复制时长,并输出每次shared memory访问的平均周期数。我得到的结果是2周期,但理论上应该是1周期,不清楚问题出在哪里。

注:我已尝试使用float4类型,得到了相同的结果。你可以将类型T改为float4并将size减小到8,以确保使用的shared memory不超过64KB。

内容的提问来源于stack exchange,提问作者Saitama10000

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.06 01:52:52