You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何仅通过Dispatch在片段Tile函数中获取像素坐标?

在Metal片段Tile着色器中不依赖顶点数据获取像素坐标(无需提前绘制获取UV)

首先明确:片段Tile着色器属于图形渲染管线的扩展组件,它的执行必须依赖光栅化流程,因此无法仅通过dispatchThreads命令触发——必须配合绘制命令(如drawPrimitives)才能运行。但你可以绕过顶点阶段传递的RasterizerData,直接在片段Tile函数中自行计算像素坐标,无需提前绘制获取UV再调度。

实现方法

  1. 使用裁剪空间位置转换像素坐标
    在片段Tile函数中,通过[[position]]属性获取当前像素的裁剪空间位置,将其转换为纹理/帧缓冲区的像素坐标:
#include <metal_stdlib>
using namespace metal;

typedef struct
{
    float4 OPTexture        [[ color(0) ]];
    float4 IntermediateTex  [[ color(1) ]];
} FragmentIO;

// 简化的全屏顶点着色器(无需顶点缓冲区)
vertex float4 FullscreenVertex(uint vertexID [[vertex_id]])
{
    // 生成覆盖整个视口的三角形顶点(裁剪空间)
    const float2 positions[] = {
        {-1.0, -1.0},
        {3.0, -1.0},
        {-1.0, 3.0}
    };
    return float4(positions[vertexID], 0.0, 1.0);
}

fragment FragmentIO Unpack(float4 clipPosition [[position]],
                           texture2d<float, access::sample> srcImageTexture [[texture(0)]],
                           constant MTLRenderPassDescriptor* renderPass [[render_pass]])
{
    FragmentIO out;
    
    // 将裁剪空间坐标转换为像素坐标
    float2 ndc = clipPosition.xy / clipPosition.w; // 归一化设备坐标(-1到1)
    float2 pixelCoord = (ndc * 0.5 + 0.5) * float2(
        renderPass->colorAttachments[0].texture->width,
        renderPass->colorAttachments[0].texture->height
    );
    
    // 用像素坐标进行采样或计算
    sampler linearSampler(mag_filter::linear, min_filter::linear);
    float2 uv = pixelCoord / float2(srcImageTexture.get_width(), srcImageTexture.get_height());
    out.OPTexture = srcImageTexture.sample(linearSampler, uv);
    
    // 其他像素级计算逻辑
    out.IntermediateTex = out.OPTexture * 0.5;
    
    return out;
}
  1. 利用Tile线程索引计算坐标
    如果你的Tile渲染是基于线程组的,也可以通过线程组ID和线程索引,结合Tile的尺寸来计算当前像素的全局坐标:
#include <metal_stdlib>
using namespace metal;

typedef struct
{
    float4 OPTexture        [[ color(0) ]];
    float4 IntermediateTex  [[ color(1) ]];
} FragmentIO;

fragment FragmentIO Unpack(uint2 threadIdx [[thread_position_in_threadgroup]],
                           uint2 threadGroupIdx [[threadgroup_position_in_grid]],
                           constant MTLRenderPassDescriptor* renderPass [[render_pass]],
                           texture2d<float, access::sample> srcImageTexture [[texture(0)]])
{
    FragmentIO out;
    
    // 假设Tile尺寸为16x16(可根据你的管线配置调整)
    uint2 tileSize = uint2(16, 16);
    uint2 globalPixelCoord = threadGroupIdx * tileSize + threadIdx;
    
    // 确保坐标不超出纹理范围
    if (globalPixelCoord.x >= srcImageTexture.get_width() || 
        globalPixelCoord.y >= srcImageTexture.get_height()) {
        out.OPTexture = float4(0.0);
        out.IntermediateTex = float4(0.0);
        return out;
    }
    
    // 采样纹理
    sampler pointSampler(mag_filter::point, min_filter::point);
    out.OPTexture = srcImageTexture.sample(pointSampler, float2(globalPixelCoord) + 0.5);
    
    // 其他计算
    out.IntermediateTex = out.OPTexture;
    
    return out;
}

关键说明

  • 无论哪种方式,片段Tile着色器都需要通过绘制命令触发,但可以使用上述的全屏无缓冲顶点着色器,无需额外的顶点数据资源,实现成本极低。
  • 如果完全不想使用绘制命令,唯一的选择是使用计算管线的Tile内核函数(你已了解该实现方式),因为计算管线的调度不依赖光栅化流程。

内容的提问来源于stack exchange,提问作者Hamid Yusifli

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.13 19:53:12