使用CFFI构建Python/CUDA接口时CUDA内存指针存储异常
CUDA/CFFI交互中cudaMemcpyDeviceToHost报invalid argument问题排查
问题重现
尝试用CFFI构建Python/CUDA交互接口,在数据回收阶段调用cudaMemcpyDeviceToHost时持续出现「invalid argument」错误。相关代码及输出如下:
CUDA代码(array.cu)
// array.cu #include "array.h" using namespace std; void allocate( float* host_array, float* device_array, int length ) { cout << "Allocating h_ptr (" << host_array << ") "; cout << "on device using d_ptr (" << device_array << ") "; cout << "of length=" << length << endl; CUCHK(cudaMalloc((void**) &device_array, length*sizeof(float))); CUCHK(cudaMemcpy(device_array, host_array, length*sizeof(float), cudaMemcpyHostToDevice)); } void retrieve( float* device_array, float* host_array, int length ) { cout << "Retrieving h_ptr (" << host_array << ") "; cout << "from device using d_ptr (" << device_array << ") "; cout << "of length=" << length << endl; CUCHK(cudaMemcpy(host_array, device_array, length*sizeof(float), cudaMemcpyDeviceToHost)); }
Python封装代码(cupid.py)
# cupid.py import numpy as np from cffi import FFI ffi = FFI() lib = ffi.dlopen("./cupid/src/libAlg.so") class cupid: def __init__(self, numpy_array): self._numpy_array = numpy_array self._host_array = ffi.cast("float *", np.ascontiguousarray(numpy_array, np.float32).ctypes.data) self._device_array = ffi.new("float *") self._length = numpy_array.size self._shape = numpy_array.shape self._dtype = numpy_array.dtype self.allocate() return def allocate(self): ffi.cdef( """ void allocate( float *host_array, float *device_array, int length ); """) lib.allocate(self._host_array, self._device_array, self._length) return def retrieve(self): ffi.cdef( """ void retrieve( float* device_array, float* host_array, int length ); """) lib.retrieve(self._device_array, self._host_array, self._length) self._numpy_array = np.frombuffer(ffi.buffer(self._host_array, self._length*ffi.sizeof("float")), dtype=np.float32) self._numpy_array = np.ctypeslib.as_array(self._numpy_array, shape=tuple(self._shape)) self._numpy_array = self._numpy_array.reshape(tuple(self._shape)) return self._numpy_array
调用代码(test.py)
# test.py import numpy as np from cupid import cupid from pprint import pprint numpy_array = np.zeros((5,6), dtype=np.float32) cupid_array = cupid(numpy_array) numpy_array = cupid_array.retrieve()
程序输出
Allocating h_ptr (0x559552b81fa0) on device using d_ptr (0x559552b80fe0) of length=30 Retrieving h_ptr (0x559552b81fa0) from device using d_ptr (0x559552b80fe0) of length=30 Cuda error in file 'array.cu' in line 25 : invalid argument. (<- the cudaMemcpyDeviceToHost)
错误原因
核心问题在于CUDA函数allocate的参数传递方式错误:
- 当前
device_array是以float*类型值传递给allocate函数,cudaMalloc修改的是函数内部的局部变量地址,而非Python端传入的指针本身。 - 函数执行完毕后,Python端的
self._device_array仍然保留初始的无效地址,并未获取到CUDA分配的设备内存指针,导致后续retrieve调用cudaMemcpyDeviceToHost时传入非法设备指针,触发「invalid argument」错误。
修复方案
1. 修改CUDA代码(array.cu)
将allocate函数的device_array参数改为指针的指针(float**),确保cudaMalloc分配的设备地址能传递回调用端:
// 修改后的array.cu #include "array.h" using namespace std; void allocate( float* host_array, float** device_array, // 修改为二级指针 int length ) { cout << "Allocating h_ptr (" << host_array << ") "; cout << "on device using d_ptr (" << *device_array << ") "; cout << "of length=" << length << endl; CUCHK(cudaMalloc((void**) device_array, length*sizeof(float))); // 无需取地址,直接传入二级指针 CUCHK(cudaMemcpy(*device_array, host_array, length*sizeof(float), cudaMemcpyHostToDevice)); // 解引用获取设备指针 } void retrieve( float* device_array, float* host_array, int length ) { cout << "Retrieving h_ptr (" << host_array << ") "; cout << "from device using d_ptr (" << device_array << ") "; cout << "of length=" << length << endl; CUCHK(cudaMemcpy(host_array, device_array, length*sizeof(float), cudaMemcpyDeviceToHost)); }
2. 修改Python封装代码(cupid.py)
- 调整
_device_array的类型为二级指针(float**) - 统一
ffi.cdef的位置,避免重复定义函数 - 调用
allocate时传入二级指针,调用retrieve时解引用获取实际设备指针
修改后的代码:
# 修改后的cupid.py import numpy as np from cffi import FFI ffi = FFI() # 提前定义所有需要的C函数,避免重复调用ffi.cdef ffi.cdef(""" void allocate(float *host_array, float** device_array, int length); void retrieve(float* device_array, float* host_array, int length); """) lib = ffi.dlopen("./cupid/src/libAlg.so") class cupid: def __init__(self, numpy_array): self._numpy_array = numpy_array self._host_array = ffi.cast("float *", np.ascontiguousarray(numpy_array, np.float32).ctypes.data) self._device_array = ffi.new("float**") # 改为二级指针 self._length = numpy_array.size self._shape = numpy_array.shape self._dtype = numpy_array.dtype self.allocate() def allocate(self): lib.allocate(self._host_array, self._device_array, self._length) def retrieve(self): # 解引用二级指针,获取实际的设备指针 lib.retrieve(self._device_array[0], self._host_array, self._length) self._numpy_array = np.frombuffer(ffi.buffer(self._host_array, self._length*ffi.sizeof("float")), dtype=np.float32) self._numpy_array = self._numpy_array.reshape(self._shape) return self._numpy_array
验证结果
重新编译CUDA代码生成动态库,运行test.py后,程序将正常执行,不再出现「invalid argument」错误,数据能正确从设备端拷贝回主机端。
内容的提问来源于stack exchange,提问作者TheGitPuller
相关产品推荐
相关产品推荐

