You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

nvprof带--metrics参数对C++有效但对Fortran可执行文件无效的问题

CUDA nvprof分析Fortran代码分支效率无结果问题

我正在学习CUDA,运行nvprof命令时遇到如下问题:用C和Fortran分别编写了测试CUDA分支发散的代码,两者编译运行均无报错。执行nvprof --metrics branch_efficiency ./codeCpp.x(针对C代码)能正常获取分支效率数据;但对对应的Fortran可执行文件执行相同命令时,却无法获取数据。奇怪的是,仅执行nvprof ./codeFortran.x能输出API调用信息,但添加--metrics参数后就没有相关结果。

环境配置:Ubuntu 20系统,GPU为NVIDIA GeForce MX330,已排除GPU版本过高导致nvprof不支持的可能(C++程序可正常运行),希望有人帮忙分析此问题。


C++代码

#include "cuda_runtime.h"
#include "device_launch_parameters.h"

#include <stdio.h>
#include <stdlib.h>
#include "cuda.h"
#include "device_launch_parameters.h"
#include "cuda_common.cuh"

// kernel without divergence
__global__ void code_without_divergence(){

   // compute unique grid index
   int gid = blockIdx.x * blockDim.x + threadIdx.x;

   // define some local variables
   float a, b;
   a = b = 0.0;

   // compute the warp index
   int warp_id = gid/32;

   // conditional statement based on the warp id
   if (warp_id % 2 == 0)
   {
      a = 150.0;
      b = 50.0;
   }
   else
   {
      a = 200.0;
      b = 75.0;
   };
}

// kernel with divergence
__global__ void code_with_divergence(){

   // compute unique grid index
   int gid = blockIdx.x * blockDim.x + threadIdx.x;

   // define some local variables
   float a, b;
   a = b = 0.0;

   // conditional statement based on the gid. This will force difference
   // code branches within the same warp.
   if (gid % 2 == 0)
   {
      a = 150.0;
      b = 50.0;
   }
   else
   {
      a = 200.0;
      b = 75.0;
   };
}

int main (int argc, char** argv){

   // set the block size
   int size = 1 << 22;

   dim3 block_size(128);
   dim3 grid_size((size + block_size.x-1)/block_size.x);

   code_without_divergence <<< grid_size, block_size>>>();
   cudaDeviceSynchronize();

   code_with_divergence <<<grid_size, block_size>>>();
   cudaDeviceSynchronize();

   cudaDeviceReset();
   return EXIT_SUCCESS;

};

Fortran代码

MODULE CUDAUtils
   USE cudafor
   IMPLICIT NONE


   CONTAINS

   ! code without divergence routine
   ATTRIBUTES(GLOBAL) SUBROUTINE code_without_divergence()
      IMPLICIT NONE

      !> local variables
      INTEGER :: threadId, warpIdx
      REAL(KIND=8) :: a,b

      ! get the unique threadID
      threadId =   (blockIdx%y-1) * gridDim%x  * blockDim%x + &
                   (blockIdx%x-1) * blockDim%x + (threadIdx%x-1)

      ! adjust so that the threadId starts from 1
      threadId = threadId + 1

      ! warp index
      warpIdx = threadIdx%x/32

      ! perform the conditional statement
      IF (MOD(warpIdx,2) == 0) THEN
         a = 150.0D0
         b = 50.0D0
      ELSE
         a = 200.0D0
         b = 75.0D0
      END IF

   END SUBROUTINE code_without_divergence

   ! code with divergence routine
   ATTRIBUTES(GLOBAL) SUBROUTINE code_with_divergence()
      IMPLICIT NONE

      !> local variables
      INTEGER :: threadId, warpIdx
      REAL(KIND=8) :: a,b

      ! get the unique threadID
      threadId =   (blockIdx%y-1) * gridDim%x  * blockDim%x + &
                   (blockIdx%x-1) * blockDim%x + (threadIdx%x-1)

      ! adjust so that the threadId starts from 1
      threadId = threadId + 1

      ! perform the conditional statement
      IF (MOD(threadId,2) == 0) THEN
         a = 150.0D0
         b = 50.0D0
      ELSE
         a = 200.0D0
         b = 75.0D0
      END IF

   END SUBROUTINE code_with_divergence
END MODULE CUDAUtils

PROGRAM main
   USE CUDAUtils
   IMPLICIT NONE

   ! define the variables
   INTEGER    :: size1 = 1e20
   INTEGER    :: istat
   TYPE(DIM3) :: grid, tBlock

   ! blocksize is 42 along the 1st dimension only whereas grid is 2D
   tBlock = DIM3(128,1,1)
   grid   = DIM3((size1 + tBlock%x)/tBlock%x,1,1)

   ! just call the module
   CALL code_without_divergence<<<grid,tBlock>>>()
   istat = cudaDeviceSynchronize()

   ! just call the module
   CALL code_with_divergence<<<grid,tBlock>>>()
   istat = cudaDeviceSynchronize()


STOP
END PROGRAM main

命令输出结果

1. nvprof --metrics branch_efficiency ./codeCpp.x 输出

=6944== NVPROF is profiling process 6944, command: ./codeCpp.x
==6944== Profiling application: ./codeCpp.x
==6944== Profiling result:
==6944== Metric result:
Invocations                               Metric Name                        Metric Description         Min         Max         Avg
Device "NVIDIA GeForce MX330 (0)"
    Kernel: code_without_divergence(void)
          1                         branch_efficiency                         Branch Efficiency     100.00%     100.00%     100.00%
    Kernel: code_with_divergence(void)
          1                         branch_efficiency                         Branch Efficiency      85.71%      85.71%      85.71%

2. nvprof --metrics branch_efficiency ./codeFortran.x 输出

==6983== NVPROF is profiling process 6983, command: ./codeFortran.x
==6983== Profiling application: ./codeFortran.x
==6983== Profiling result:
No events/metrics were profiled.

3. nvprof ./codeFortran.x 输出

==7002== NVPROF is profiling process 7002, command: ./codeFortran.x
==7002== Profiling application: ./codeFortran.x
==7002== Profiling result:
No kernels were profiled.
            Type  Time(%)      Time     Calls       Avg       Min       Max  Name
      API calls:   99.82%  153.45ms         2  76.726ms     516ns  153.45ms  cudaLaunchKernel
                    0.15%  231.24us       101  2.2890us      95ns  172.81us  cuDeviceGetAttribute
                    0.01%  22.522us         1  22.522us  22.522us  22.522us  cuDeviceGetName
                    0.01%  9.1310us         1  9.1310us  9.1310us  9.1310us  cuDeviceGetPCIBusId
                    0.00%  5.4500us         2  2.7250us     876ns  4.5740us  cudaDeviceSynchronize
                    0.00%  1.3480us         3     449ns     195ns     903ns  cuDeviceGetCount
                    0.00%     611ns         1     611ns     611ns     611ns  cuModuleGetLoadingMode
                    0.00%     520ns         2     260ns     117ns     403ns  cuDeviceGet
                    0.00%     245ns         1     245ns     245ns     245ns  cuDeviceTotalMem
                    0.00%     187ns         1     187ns     187ns     187ns  cuDeviceGetUuid

内容的提问来源于stack exchange,提问作者Principio Tudisco

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.15 12:10:28