You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何优化DirectX11中的网格渲染?附现有渲染命令模式实现

基于Render Command模式的D3D11渲染流程优化咨询

我当前的渲染工作流采用Render Command模式,先准备好所有渲染命令再提交执行,渲染命令按包含多子网格的Mesh对应的Object ID排序。我已经尝试通过避免重复绑定材质(纹理集合)来降低开销,但觉得当前实现还有优化空间,希望学习D3D11下的最佳渲染实践。以下是我的代码实现:

void SceneRendererD3D11::RenderScene(const glm::mat4& viewMatrix, const glm::mat4& projMatrix, bool frustumCulling, const PointLightCullingData& pointLightCullingData)
{
    auto& meshEnts = m_Scene->GetAllEntitiesWith<MeshComponent>();

    for (auto& meshEnt : meshEnts)
    {
        Entity entity = { meshEnt, m_Scene.get() };

        auto& meshComponent = m_Scene->GetRegistry().get<MeshComponent>(meshEnt);
        auto& mesh = meshComponent.Mesh;

        if (!mesh || !mesh->IsLoaded())
            continue;

        auto& meshTransform = m_Scene->GetWorldSpaceTransformMatrix(entity);
        auto& tag = entity.GetComponent<TagComponent>().Tag;

        glm::mat4 modelMatrix = glm::mat4(1.f);
        modelMatrix = meshTransform;

        const AABB& aabb = mesh->GetBounds();
        if (frustumCulling && !Intersections::OBBInFrustum(projMatrix * viewMatrix * modelMatrix, aabb))
            continue;

        std::vector<glm::mat4> boneMatrices;
        if (entity.HasComponent<AnimatorComponent>())
        {
            auto& animatorComponent = entity.GetComponent<AnimatorComponent>();
            boneMatrices = animatorComponent.Animator->GetFinalBoneMatrices();
        }

        SubmitMesh((uint32_t)meshEnt, mesh, modelMatrix, viewMatrix, projMatrix, boneMatrices, frustumCulling,
            pointLightCullingData, meshComponent.CombinedTextures, meshComponent.CombinedMaterial, tag == "Sponza" ? true : false, false, false);
    }

    FlushMeshes();
}

void SceneRendererD3D11::SubmitMesh(uint32_t objectId, std::shared_ptr<Mesh> mesh, const glm::mat4& modelMatrix, const glm::mat4& viewMatrix, const glm::mat4& projMatrix, const std::vector<glm::mat4>& boneMatrices, bool frustumCulling, PointLightCullingData pointLightCullingData, bool useCombinedMaterial, AssetHandle combinedMaterial, bool gltfMaterials, bool notTextured, bool albedoOnly)
{
    Ref<Material> combMaterial = AssetManager::GetAsset<Material>(combinedMaterial);

    auto& submeshes = mesh->GetSubmeshes();

    ID3D11Buffer* vbuff = ((DX11VertexBuffer*)mesh->m_VertexArray->GetVertexBuffer().get())->GetBuffer();
    ID3D11Buffer* ibuff = ((DX11IndexBuffer*)mesh->m_VertexArray->GetIndexBuffer().get())->GetBuffer();

    auto& viewMatrixInv = glm::inverse(viewMatrix);
    auto& projMatrixInv = glm::inverse(projMatrix);
    auto& normalMatrix = glm::transpose(glm::inverse(modelMatrix));

    for (auto& submesh : submeshes)
    {
        DrawCommand drawCommand;
        drawCommand.ObjectId = objectId;
        drawCommand.vBuffer = vbuff;
        drawCommand.iBuffer = ibuff;
        drawCommand.indexCount = submesh.IndexCount;
        drawCommand.startIndex = submesh.StartIndex;
        drawCommand.baseVertex = submesh.BaseVertex;
        drawCommand.Material = useCombinedMaterial ? combMaterial : submesh.GetMaterial();
        drawCommand.UseCombinedMaterial = useCombinedMaterial;
        drawCommand.ModelMatrix = modelMatrix;
        drawCommand.ViewMatrix = viewMatrix;
        drawCommand.ProjMatrix = projMatrix;
        drawCommand.ViewMatrixInv = viewMatrixInv;
        drawCommand.ProjMatrixInv = projMatrixInv;
        drawCommand.NormalMatrix = normalMatrix;
        drawCommand.GltfMaterials = gltfMaterials;
        drawCommand.NotTextured = notTextured;
        drawCommand.AlbedoOnly = albedoOnly;

        m_DrawCommands.push_back(drawCommand);
    }

    CopyToBoneTransformStorage(objectId, boneMatrices);
}

void SceneRendererD3D11::FlushMeshes()
{
    OV_PROFILE_FUNC("FlushMeshes");

    std::sort(m_DrawCommands.begin(), m_DrawCommands.end(), [](const DrawCommand& a, const DrawCommand& b) {
        return a.ObjectId < b.ObjectId;
        });

    ID3D11SamplerState* samplers[3]{ m_SamplerLinearWrap, m_SamplerPointClamp, m_SamplerLinearClampComparison };
    m_DX11DeviceContext->PSSetSamplers(0, 3, samplers);

    ID3D11Buffer* buffers[5]{ g_pCBMatrixes, g_pCBMaterial, g_pCBSkeletalAnimation, g_pCBLight, g_pCBPointLightShadowGen };
    m_DX11DeviceContext->VSSetConstantBuffers(0, 5, buffers);
    m_DX11DeviceContext->PSSetConstantBuffers(0, 5, buffers);

    uint32_t currentObjId = 0xFFFFFFFF;

    m_DX11DeviceContext->IASetPrimitiveTopology(D3D11_PRIMITIVE_TOPOLOGY_TRIANGLELIST);

    Material* lastBoundMaterial = nullptr;

    for (auto& command : m_DrawCommands)
    {
        MaterialCB matCb = {};
        matCb.albedoOnly = 0;
        matCb.objectId = command.ObjectId;
        matCb.sponza = command.GltfMaterials;

        const MaterialDesc& materialDesc = command.Material->GetDesc();

        matCb.emission = materialDesc.Emission;
        matCb.color = materialDesc.Color;
        matCb.roughness = materialDesc.Roughness;
        matCb.metallic = materialDesc.Metallic;
        matCb.terrain = 0;
        matCb.cubemapLod = 0.f;
        matCb.trees = 0;
        matCb.useNormalMap = materialDesc.UseNormalMap ? 1 : 0;
        matCb.invertNormalG = materialDesc.InvertNormalG ? 1 : 0;

        matCb.notTextured = command.NotTextured;
        matCb.albedoOnly = command.AlbedoOnly;

        m_DX11DeviceContext->UpdateSubresource(g_pCBMaterial, 0, NULL, &matCb, 0, 0);

        if (!command.UseCombinedMaterial)
            command.Material->Bind(nullptr);

        if (command.ObjectId != currentObjId)
        {
            MatricesCB mtxCb = {};
            mtxCb.modelMat = command.ModelMatrix;
            mtxCb.viewMat = command.ViewMatrix;
            mtxCb.projMat = command.ProjMatrix;
            mtxCb.viewMatInv = command.ViewMatrixInv;
            mtxCb.projMatInv = command.ProjMatrixInv;
            mtxCb.normalMat = command.NormalMatrix;

            m_DX11DeviceContext->UpdateSubresource(g_pCBMatrixes, 0, NULL, &mtxCb, 0, 0);

            uint32_t stride = sizeof(Vertex);
            uint32_t offset = 0;
            m_DX11DeviceContext->IASetVertexBuffers(0, 1, &command.vBuffer, &stride, &offset);
            m_DX11DeviceContext->IASetIndexBuffer(command.iBuffer, DXGI_FORMAT_R32_UINT, 0);

            SkeletalAnimationCB saCb = {};
            auto& boneMatrices = m_MeshBoneMatrices[command.ObjectId];
            //std::memset(saCb.finalBonesMatrices, 0, sizeof(saCb.finalBonesMatrices[0]) * 100);
            std::memcpy(saCb.finalBonesMatrices, boneMatrices.data(), sizeof(saCb.finalBonesMatrices[0]) * boneMatrices.size());
            m_DX11DeviceContext->UpdateSubresource(g_pCBSkeletalAnimation, 0, NULL, &saCb, 0, 0);

            if (command.UseCombinedMaterial)
                command.Material->Bind(nullptr);

            currentObjId = command.ObjectId;
        }

        m_DX11DeviceContext->DrawIndexed(command.indexCount, command.startIndex, command.baseVertex);
    }

    m_DrawCommands.clear();
}

优化建议与最佳实践

1. 调整渲染命令排序策略

当前按Object ID排序的逻辑,仅能减少同一物体的VB/IB切换开销,但材质切换的开销远大于VB/IB切换。建议优先按材质排序,再按Object ID排序,最大化减少材质绑定次数:

std::sort(m_DrawCommands.begin(), m_DrawCommands.end(), [](const DrawCommand& a, const DrawCommand& b) {
    // 先按材质指针地址排序,确保同材质的DrawCommand连续
    if (a.Material.get() != b.Material.get()) {
        return a.Material.get() < b.Material.get();
    }
    // 同材质下再按Object ID排序,减少VB/IB切换
    return a.ObjectId < b.ObjectId;
});

2. 避免重复更新材质常量缓冲区

当前代码每次执行DrawCommand都会更新g_pCBMaterial,但连续使用同一材质的命令完全可以跳过重复更新。新增lastUsedMaterial指针做判断:

Material* lastUsedMaterial = nullptr;
for (auto& command : m_DrawCommands) {
    if (command.Material.get() != lastUsedMaterial) {
        // 仅当材质变化时才更新CB并绑定材质
        MaterialCB matCb = {};
        matCb.albedoOnly = 0;
        matCb.objectId = command.ObjectId;
        matCb.sponza = command.GltfMaterials;

        const MaterialDesc& materialDesc = command.Material->GetDesc();
        matCb.emission = materialDesc.Emission;
        matCb.color = materialDesc.Color;
        matCb.roughness = materialDesc.Roughness;
        matCb.metallic = materialDesc.Metallic;
        matCb.terrain = 0;
        matCb.cubemapLod = 0.f;
        matCb.trees = 0;
        matCb.useNormalMap = materialDesc.UseNormalMap ? 1 : 0;
        matCb.invertNormalG = materialDesc.InvertNormalG ? 1 : 0;

        matCb.notTextured = command.NotTextured;
        matCb.albedoOnly = command.AlbedoOnly;

        m_DX11DeviceContext->UpdateSubresource(g_pCBMaterial, 0, NULL, &matCb, 0, 0);
        
        // 统一材质绑定逻辑,避免分支分散
        if (!command.UseCombinedMaterial) {
            command.Material->Bind(nullptr);
        } else {
            command.Material->Bind(nullptr);
        }
        lastUsedMaterial = command.Material.get();
    }

    // 后续Object ID检查与Draw逻辑不变
    // ...
}

3. 优化骨骼动画数据处理

  • 对无骨骼动画的物体,跳过g_pCBSkeletalAnimation的更新,避免空数据拷贝;
  • 预分配固定大小的骨骼矩阵缓冲区(比如支持100个骨骼),减少memcpy的动态开销;
  • 可以将骨骼矩阵合并到模型矩阵CB中,减少常量缓冲区的绑定数量。

4. 提前过滤无效子网格

当前仅对整个Mesh做视锥体剪裁,建议对子网格单独做剪裁,过滤掉Mesh中不在视锥体内的子网格,减少不必要的DrawCommand生成,降低CPU和GPU负载。

5. 消除冗余状态设置

  • 全局采样器、全局常量缓冲区(g_pCBLight、g_pCBPointLightShadowGen)、图元拓扑等固定状态,只需在FlushMeshes开头设置一次即可,无需重复执行;
  • 统一材质绑定逻辑,避免UseCombinedMaterial分支判断带来的CPU开销。

内容的提问来源于stack exchange,提问作者John Stoner

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.02 01:07:27