CUDA一维数组绑定纹理对象后读取数据异常求助
问题分析与修复
你的代码读取纹理返回全0,核心问题有两个:
1. 通道格式描述符错误
你用width(值为10)作为cudaCreateChannelDesc的第一个参数,但float类型是32位单精度,这里需要指定为32位,否则纹理无法正确解析内存中的float数据。
2. 未将通道格式绑定到资源描述符
创建线性纹理资源时,resDesc.res.linear.desc没有赋值,导致CUDA不知道如何解释设备内存中的数据格式。
另外,cudaFilterModeLinear用于线性插值,在使用整数索引访问非归一化纹理时,会导致采样结果异常(比如取相邻元素的平均值,但你的索引是整数,可能越界或者取到错误值),建议改为cudaFilterModePoint,除非你确实需要插值。
修复后的完整代码
#include<cuda.h> #include<cuda_runtime.h> #include<iostream> #include<stdio.h> using namespace std; __global__ void squareKernel(float* output, float *dh, int size, cudaTextureObject_t texObj) { unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; if(x >= size) return; // 原代码是x>size,会漏掉x=size的情况,改为x>=size更准确 float y = tex1D<float>(texObj, x); printf("%d, %f, %f\n", x, y, dh[x]); output[x] = y*y; } #define width 10 int main() { // 正确创建float类型的通道描述符:32位x分量,其余0,格式为float cudaChannelFormatDesc channelDesc = cudaCreateChannelDesc(32, 0, 0, 0, cudaChannelFormatKindFloat); float *hA = (float *)malloc(width*sizeof(float)); for(int i=0; i<width; i++){ hA[i] = i; } float *dA; cudaMalloc((void **)&dA, width*sizeof(float)); cudaMemcpy(dA, hA, width*sizeof(float), cudaMemcpyHostToDevice); cudaResourceDesc resDesc; memset(&resDesc, 0, sizeof(resDesc)); resDesc.resType = cudaResourceTypeLinear; resDesc.res.linear.devPtr = dA; resDesc.res.linear.sizeInBytes = width*sizeof(float); resDesc.res.linear.desc = channelDesc; // 绑定通道格式描述符 cudaTextureDesc texDesc; memset(&texDesc, 0, sizeof(texDesc)); texDesc.filterMode = cudaFilterModePoint; // 改为点采样,适合整数索引 texDesc.readMode = cudaReadModeElementType; texDesc.normalizedCoords = 0; cudaTextureObject_t texObj = 0; cudaCreateTextureObject(&texObj, &resDesc, &texDesc, NULL); float* output; cudaMalloc(&output, width * sizeof(float)); squareKernel<<<1, width>>>(output, dA, width, texObj); cudaDeviceSynchronize(); // 等待kernel执行完成,避免printf输出混乱 cudaMemcpy(hA, output, width*sizeof(float), cudaMemcpyDeviceToHost); for(int i=0; i<width; i++){ cout << i << "," << hA[i] << endl; } cudaDestroyTextureObject(texObj); cudaFree(dA); cudaFree(output); // 原代码漏掉释放output free(hA); return 0; }
关键修复点说明
- 修正
cudaCreateChannelDesc的参数,匹配float类型的32位宽度 - 把通道描述符赋值给
resDesc.res.linear.desc,让纹理知道如何解析内存数据 - 将过滤模式改为
cudaFilterModePoint,避免线性插值导致的异常采样 - 增加
cudaDeviceSynchronize()确保kernel执行完成后再拷贝结果,同时让printf输出完整 - 补充释放
output设备内存的代码,避免内存泄漏
内容的提问来源于stack exchange,提问作者alireza
相关产品推荐
相关产品推荐

