You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CUDA实现中值模糊时输出图像出现条纹问题求助

CUDA中值模糊异常条纹问题分析

问题现象

使用CUDA实现3x3中值模糊时,单通道处理后的图像出现不符合模糊效果的异常条纹。

原始图像:
original_image

模糊后的单通道图像:
模糊后的通道图像

代码实现

#include <iostream>
#include <opencv2/core.hpp>
#include <opencv2/imgcodecs.hpp>

using namespace std;
using namespace cv;

#define BLOCK_SIZE      16
#define TILE_SIZE       14
#define FILTER_WIDTH    3
#define FILTER_HEIGHT   3


__device__ void sort(unsigned char* filterVector)
{
    for (int i = 0; i < FILTER_WIDTH*FILTER_HEIGHT; i++) {
        for (int j = i + 1; j < FILTER_WIDTH*FILTER_HEIGHT; j++) {
            if (filterVector[i] > filterVector[j]) {
                unsigned char tmp = filterVector[i];
                filterVector[i] = filterVector[j];
                filterVector[j] = tmp;
            }
        }
    }
}

__global__ void medianFilter(unsigned char *srcImage, unsigned char *dstImage, unsigned int width, unsigned int height)
{

    int x_o = TILE_SIZE * blockIdx.x + threadIdx.x;
    int y_o = TILE_SIZE * blockIdx.y + threadIdx.y;

    int x_i = x_o - (FILTER_HEIGHT / 2);
    int y_i = y_o - (FILTER_WIDTH / 2);

    __shared__ unsigned char sBuffer[BLOCK_SIZE][BLOCK_SIZE];

    if ((x_i >= 0) && (x_i < width) && (y_i >= 0) && (y_i < height)) {
        sBuffer[threadIdx.y][threadIdx.x] = srcImage[y_i * width + x_i];
    } else {
        sBuffer[threadIdx.y][threadIdx.x] = 0;
    }

    __syncthreads();

    unsigned char filterVector[FILTER_WIDTH*FILTER_HEIGHT];

    // int size_vec = sizeof(filterVector) / sizeof(filterVector[0]);

    // printf("%d 
", size_vec);

    if (threadIdx.x < TILE_SIZE && threadIdx.y < TILE_SIZE) {
        for (int r = 0; r < FILTER_HEIGHT; r++) {
            for (int c = 0; c < FILTER_HEIGHT; c++) {
                filterVector[r*FILTER_HEIGHT+c] = sBuffer[threadIdx.y + r][threadIdx.x + c];
            }
        }
    }

    sort(filterVector);

    if (x_o < width && y_o < height) {
        dstImage[y_o * width + x_o] =  filterVector[4]; // (FILTER_WIDTH*FILTER_HEIGHT)/2
    }

}

int main(int argc, char **argv)
{

    std::string image_path = "./test.jpg";
    cv::Mat img = imread(image_path, IMREAD_COLOR);
    std::string output_file = "test_gpu.jpg";

    if(img.empty())
    {
        std::cout << "Couldn't read img:" << image_path << std::endl;
    }

    Mat bgr[3];
    split(img, bgr);
    
    cv::Mat dstImg (bgr[1].size(), bgr[1].type());

    const int inputSize = img.cols * img.rows;
    const int outputSize = dstImg.cols * dstImg.rows; 
    unsigned char *d_input, *d_output;

    cudaMalloc<unsigned char>(&d_input, inputSize);
    cudaMalloc<unsigned char>(&d_output, outputSize);

    cudaMemcpy(d_input, bgr[1].ptr(), inputSize, cudaMemcpyHostToDevice);

    const dim3 block(BLOCK_SIZE, BLOCK_SIZE);
    const dim3 grid((dstImg.cols + TILE_SIZE - 1)/TILE_SIZE, (dstImg.rows + TILE_SIZE - 1)/TILE_SIZE);

    medianFilter<<<grid,block>>>(d_input, d_output, dstImg.cols, dstImg.rows);

    cudaMemcpy(dstImg.ptr(), d_output, outputSize, cudaMemcpyDeviceToHost);

    cudaFree(d_input);
    cudaFree(d_output);

    imwrite(output_file, dstImg);
}

问题原因分析

  1. 无效线程错误写入输出
    每个CUDA block包含16x16线程,但仅前14x14线程对应有效输出像素(TILE_SIZE=14)。代码中所有线程都会执行排序和输出写入操作:

    • 对于threadIdx.x >=14或threadIdx.y >=14的线程,filterVector未被填充有效数据(跳过了填充逻辑),内存中是未初始化的垃圾值。
    • 当这些无效线程的x_o/y_o落在图像范围内时,会将垃圾值写入输出图像,形成异常条纹。
  2. 输出索引计算错误
    x_o = TILE_SIZE * blockIdx.x + threadIdx.x的计算逻辑错误,导致:

    • 同一输出位置可能被多个block的线程重复处理,引发写入冲突。
    • 超出TILE_SIZE的线程会计算出超出当前tile范围的x_o/y_o,进一步加剧无效写入问题。
  3. 滤波器索引混淆(次要)
    计算输入坐标x_i/y_i时,错误地将FILTER_HEIGHT用于x方向、FILTER_WIDTH用于y方向,虽然3x3滤波器下不影响结果,但会导致代码逻辑混淆,增加后续维护风险。

修复方案

  1. 限制有效线程的操作范围
    将排序和输出写入逻辑放到有效线程的条件判断内:

    if (threadIdx.x < TILE_SIZE && threadIdx.y < TILE_SIZE) {
        unsigned char filterVector[FILTER_WIDTH*FILTER_HEIGHT];
        for (int r = 0; r < FILTER_HEIGHT; r++) {
            for (int c = 0; c < FILTER_WIDTH; c++) {
                filterVector[r*FILTER_WIDTH+c] = sBuffer[threadIdx.y + r][threadIdx.x + c];
            }
        }
        sort(filterVector);
        if (x_o < width && y_o < height) {
            dstImage[y_o * width + x_o] = filterVector[4];
        }
    }
    
  2. 修正输出索引计算
    调整x_o/y_o的计算方式,确保每个block的有效线程对应连续的输出块:

    int x_o = blockIdx.x * TILE_SIZE + threadIdx.x;
    int y_o = blockIdx.y * TILE_SIZE + threadIdx.y;
    
  3. 修正滤波器索引对应关系
    对齐输入坐标计算与滤波器维度:

    int x_i = x_o - (FILTER_WIDTH / 2);
    int y_i = y_o - (FILTER_HEIGHT / 2);
    

内容的提问来源于stack exchange,提问作者craaaft

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.22 16:27:26