CUDA实现中值模糊时输出图像出现条纹问题求助
CUDA中值模糊异常条纹问题分析
问题现象
使用CUDA实现3x3中值模糊时,单通道处理后的图像出现不符合模糊效果的异常条纹。
原始图像:
模糊后的单通道图像:
代码实现
#include <iostream> #include <opencv2/core.hpp> #include <opencv2/imgcodecs.hpp> using namespace std; using namespace cv; #define BLOCK_SIZE 16 #define TILE_SIZE 14 #define FILTER_WIDTH 3 #define FILTER_HEIGHT 3 __device__ void sort(unsigned char* filterVector) { for (int i = 0; i < FILTER_WIDTH*FILTER_HEIGHT; i++) { for (int j = i + 1; j < FILTER_WIDTH*FILTER_HEIGHT; j++) { if (filterVector[i] > filterVector[j]) { unsigned char tmp = filterVector[i]; filterVector[i] = filterVector[j]; filterVector[j] = tmp; } } } } __global__ void medianFilter(unsigned char *srcImage, unsigned char *dstImage, unsigned int width, unsigned int height) { int x_o = TILE_SIZE * blockIdx.x + threadIdx.x; int y_o = TILE_SIZE * blockIdx.y + threadIdx.y; int x_i = x_o - (FILTER_HEIGHT / 2); int y_i = y_o - (FILTER_WIDTH / 2); __shared__ unsigned char sBuffer[BLOCK_SIZE][BLOCK_SIZE]; if ((x_i >= 0) && (x_i < width) && (y_i >= 0) && (y_i < height)) { sBuffer[threadIdx.y][threadIdx.x] = srcImage[y_i * width + x_i]; } else { sBuffer[threadIdx.y][threadIdx.x] = 0; } __syncthreads(); unsigned char filterVector[FILTER_WIDTH*FILTER_HEIGHT]; // int size_vec = sizeof(filterVector) / sizeof(filterVector[0]); // printf("%d ", size_vec); if (threadIdx.x < TILE_SIZE && threadIdx.y < TILE_SIZE) { for (int r = 0; r < FILTER_HEIGHT; r++) { for (int c = 0; c < FILTER_HEIGHT; c++) { filterVector[r*FILTER_HEIGHT+c] = sBuffer[threadIdx.y + r][threadIdx.x + c]; } } } sort(filterVector); if (x_o < width && y_o < height) { dstImage[y_o * width + x_o] = filterVector[4]; // (FILTER_WIDTH*FILTER_HEIGHT)/2 } } int main(int argc, char **argv) { std::string image_path = "./test.jpg"; cv::Mat img = imread(image_path, IMREAD_COLOR); std::string output_file = "test_gpu.jpg"; if(img.empty()) { std::cout << "Couldn't read img:" << image_path << std::endl; } Mat bgr[3]; split(img, bgr); cv::Mat dstImg (bgr[1].size(), bgr[1].type()); const int inputSize = img.cols * img.rows; const int outputSize = dstImg.cols * dstImg.rows; unsigned char *d_input, *d_output; cudaMalloc<unsigned char>(&d_input, inputSize); cudaMalloc<unsigned char>(&d_output, outputSize); cudaMemcpy(d_input, bgr[1].ptr(), inputSize, cudaMemcpyHostToDevice); const dim3 block(BLOCK_SIZE, BLOCK_SIZE); const dim3 grid((dstImg.cols + TILE_SIZE - 1)/TILE_SIZE, (dstImg.rows + TILE_SIZE - 1)/TILE_SIZE); medianFilter<<<grid,block>>>(d_input, d_output, dstImg.cols, dstImg.rows); cudaMemcpy(dstImg.ptr(), d_output, outputSize, cudaMemcpyDeviceToHost); cudaFree(d_input); cudaFree(d_output); imwrite(output_file, dstImg); }
问题原因分析
无效线程错误写入输出
每个CUDA block包含16x16线程,但仅前14x14线程对应有效输出像素(TILE_SIZE=14)。代码中所有线程都会执行排序和输出写入操作:- 对于
threadIdx.x >=14或threadIdx.y >=14的线程,filterVector未被填充有效数据(跳过了填充逻辑),内存中是未初始化的垃圾值。 - 当这些无效线程的
x_o/y_o落在图像范围内时,会将垃圾值写入输出图像,形成异常条纹。
- 对于
输出索引计算错误
x_o = TILE_SIZE * blockIdx.x + threadIdx.x的计算逻辑错误,导致:- 同一输出位置可能被多个block的线程重复处理,引发写入冲突。
- 超出TILE_SIZE的线程会计算出超出当前tile范围的
x_o/y_o,进一步加剧无效写入问题。
滤波器索引混淆(次要)
计算输入坐标x_i/y_i时,错误地将FILTER_HEIGHT用于x方向、FILTER_WIDTH用于y方向,虽然3x3滤波器下不影响结果,但会导致代码逻辑混淆,增加后续维护风险。
修复方案
限制有效线程的操作范围
将排序和输出写入逻辑放到有效线程的条件判断内:if (threadIdx.x < TILE_SIZE && threadIdx.y < TILE_SIZE) { unsigned char filterVector[FILTER_WIDTH*FILTER_HEIGHT]; for (int r = 0; r < FILTER_HEIGHT; r++) { for (int c = 0; c < FILTER_WIDTH; c++) { filterVector[r*FILTER_WIDTH+c] = sBuffer[threadIdx.y + r][threadIdx.x + c]; } } sort(filterVector); if (x_o < width && y_o < height) { dstImage[y_o * width + x_o] = filterVector[4]; } }修正输出索引计算
调整x_o/y_o的计算方式,确保每个block的有效线程对应连续的输出块:int x_o = blockIdx.x * TILE_SIZE + threadIdx.x; int y_o = blockIdx.y * TILE_SIZE + threadIdx.y;修正滤波器索引对应关系
对齐输入坐标计算与滤波器维度:int x_i = x_o - (FILTER_WIDTH / 2); int y_i = y_o - (FILTER_HEIGHT / 2);
内容的提问来源于stack exchange,提问作者craaaft
相关产品推荐
相关产品推荐

