You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

SYCL内核未找到:Fortran调用SYCL代码链接阶段报错求助

SYCL内核找不到错误:No kernel named _ZTSZ7gpu_CUBEUlvE_ was found 解决思路

错误信息

terminate called after throwing an instance of 'sycl::_V1::runtime_error'
  what():  No kernel named _ZTSZ7gpu_CUBEUlvE_ was found -46 (PI_ERROR_INVALID_KERNEL_NAME)

问题背景

原有Fortran代码通过wrapper调用CUDA-C函数,现在将CUDA代码转为SYCL以适配NVIDIA GPU,用Clang++编译SYCL代码,因MPI依赖问题用ifort做链接。单独测试SYCL代码正常,但链接到Fortran项目时出现上述内核找不到错误。

复现代码

runfom.f

program runfom
  use MPI
  include 'CUB.f'

  call gpu_init()

  call gpu_CUB()

end program runfom

CUB.f

USE ISO_C_BINDING
implicit none
INTERFACE

SUBROUTINE gpu_init() bind(C, name="gpu_init")
END SUBROUTINE

SUBROUTINE gpu_CUB() bind(C, name="gpu_CUB")
END SUBROUTINE

END INTERFACE

CUB_kernel.cpp

#include <sycl/sycl.hpp>
#include <dpct/dpct.hpp>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

#include <sched.h>
#include <cmath>

extern "C"
{

void gpu_init()
{
  auto platforms = sycl::platform::get_platforms();

  printf("\n \n Checking all devices in the Node: \n");

 for (const auto & platform: platforms){
  auto devices = platform.get_devices();
  for (const auto& device : devices) {
        std::string name = device.get_info<sycl::info::device::name>();
        std::string vendor = device.get_info<sycl::info::device::vendor>();

        // Print device information
        std::cout << "Name: " << name << std::endl;
        std::cout << "Vendor: " << vendor << std::endl;
   }
  }
}

void gpu_CUB(void) {
        const int N=16;

    sycl::device dev_ct1;
    sycl::queue q_ct1(
        dev_ct1, sycl::property_list{sycl::property::queue::in_order()});
    //# Initialize vectors on host
    float A[N] = {1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1};
    float B[N] = {2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2};
    float C[N] = {0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0};

    //# Allocate memory on device
    float *d_A, *d_B, *d_C;
    d_A = sycl::malloc_device<float>(N, q_ct1);
    d_B = sycl::malloc_device<float>(N, q_ct1);
    d_C = sycl::malloc_device<float>(N, q_ct1);

    //# copy vector data from host to device
    q_ct1.memcpy(d_A, A, N * sizeof(float));
    q_ct1.memcpy(d_B, B, N * sizeof(float));

    q_ct1.single_task<>([=]() {d_C[0]=1.9;} );

    //# copy result of vector data from device to host
    q_ct1.memcpy(C, d_C, N * sizeof(float)).wait();

    //# print result on host
    for (int i = 0; i < N; i++) std::cout<< C[i] << " ";
    std::cout << "\n";

    //# free allocation on device
    sycl::free(d_A, q_ct1);
    sycl::free(d_B, q_ct1);
    sycl::free(d_C, q_ct1);
}

}

Makefile

FCOMPILE = ftn
SYCLCOMPILE = /mypath/intel/oneapi/compiler/2024.0/bin/compiler/clang++
CCOMPILE = /mypath/intel/oneapi/compiler/2024.0/bin/compiler/clang
DEFS = -DNVIDIA_GPU

FLAGS = $(DEFS) -fast
CFLAGS = $(DEFS)
NVFLAGS = $(DEFS) -fsycl -fsycl-targets=nvptx64-nvidia-cuda -I./ISO_Fortran_binding/include

FFLAGS= $(GENCMPLFLAGS) $(TARGET_ARCH) -cpp
F90FLAGS= $(GENCMPLFLAGS) $(TARGET_ARCH) -cpp

DEPS=CUB.f

LDLIBS = -L/mypath/intel/oneapi/compiler/2024.0/lib -lsycl -lstdc++

COMPILE.c = $(CCOMPILE) $(CFLAGS) -c
COMPILE.f = $(FCOMPILE) $(FFLAGS) -c
COMPILE.cpp = $(SYCLCOMPILE) $(NVFLAGS) -c
COMPILE.f90 = $(FCOMPILE) $(F90FLAGS) -c
LINK.f = $(FCOMPILE) $(FFLAGS)

#  command to compile (not link):
%.o: %.c $(DEPS)
        $(COMPILE.c) -o $@ $<

%.o: %.cpp $(DEPS)
        $(COMPILE.cpp)  -o $@ $<

#  command to compile (not link):
%.o: %.f90 $(DEPS)
        $(COMPILE.f90) -o $@ $<

#  command to compile (not link):
%.o: %.f $(DEPS)
        $(COMPILE.f) -o $@ $<

#  command to link object files:
%: %.o $(DEPS)
        $(LINK.f) $^ $(LDLIBS) -o $@

################################################################
#
# build rule
#
################################################################

runfom: runfom.o CUB_kernel.o
        $(LINK.f) $^ $(LDLIBS) -o $@

clean:  
        rm -f runfom *.o

运行命令及输出

执行命令:

SYCL_DEVICE_FILTER=cuda SYCL_PI_TRACE=1 ./runfom

输出:

SYCL_PI_TRACE[basic]: Plugin found and successfully loaded: libpi_cuda.so [ PluginVersion: 14.38.1 ]
SYCL_PI_TRACE[basic]: Plugin found and successfully loaded: libpi_unified_runtime.so [ PluginVersion: 14.37.1 ]

 
 Checking all devices in the Node: 
Name: NVIDIA A100-PCIE-40GB
Vendor: NVIDIA Corporation
SYCL_PI_TRACE[all]: Requested device_type: info::device_type::automatic
SYCL_PI_TRACE[all]: Selected device: -> final score = 500
SYCL_PI_TRACE[all]:   platform: NVIDIA CUDA BACKEND
SYCL_PI_TRACE[all]:   device: NVIDIA A100-PCIE-40GB
terminate called after throwing an instance of 'sycl::_V1::runtime_error'
  what():  No kernel named _ZTSZ7gpu_CUBEUlvE_ was found -46 (PI_ERROR_INVALID_KERNEL_NAME)

已尝试:用gfortran替代ifort,调整链接标志顺序,问题未解决。

解决思路

1. 显式指定内核名称

匿名single_task内核可能因链接流程特殊性被优化丢失,修改代码显式声明内核:
在extern "C"块外添加内核类声明:

class MyCUBKernel;

修改single_task调用:

q_ct1.single_task<class MyCUBKernel>([=]() {d_C[0]=1.9;} );

2. 调整SYCL编译选项,保留设备代码

修改NVFLAGS,添加设备代码拆分控制选项,防止设备代码被拆分后未被链接:

NVFLAGS = $(DEFS) -fsycl -fsycl-targets=nvptx64-nvidia-cuda -I./ISO_Fortran_binding/include -fsycl-device-code-split=off

3. 将SYCL代码编译为共享库后链接

先编译SYCL代码为共享库,再让Fortran编译器链接该库,避免ifort直接处理SYCL设备代码:
在Makefile中添加共享库编译规则:

CUB_kernel.so: CUB_kernel.cpp
        $(SYCLCOMPILE) $(NVFLAGS) -shared -fPIC -o $@ $< $(LDLIBS)

修改链接规则:

runfom: runfom.o CUB_kernel.so
        $(LINK.f) $^ -L. -lCUB_kernel -o $@

4. 补充链接库

确保链接时包含必要的CUDA运行时库,修改LDLIBS:

LDLIBS = -L/mypath/intel/oneapi/compiler/2024.0/lib -lsycl -lstdc++ -lcudart

5. 禁用链接符号剥离

在链接阶段添加选项防止内核符号被剥离,修改LINK.f:

LINK.f = $(FCOMPILE) $(FFLAGS) -Wl,--no-strip-all

内容的提问来源于stack exchange,提问作者alvarella

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.30 08:15:55