/* *********************************************************************************************************************** * * Copyright (c) 2014-2025 Advanced Micro Devices, Inc. All Rights Reserved. * * Permission is hereby granted, free of charge, to any person obtaining a copy * of this software and associated documentation files (the "Software"), to deal * in the Software without restriction, including without limitation the rights * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell * copies of the Software, and to permit persons to whom the Software is * furnished to do so, subject to the following conditions: * * The above copyright notice and this permission notice shall be included in all * copies or substantial portions of the Software. * * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE * SOFTWARE. * **********************************************************************************************************************/ #include "include/vk_buffer.h" #include "include/vk_cmdbuffer.h" #include "include/vk_compute_pipeline.h" #include "include/vk_conv.h" #include "include/vk_device.h" #include "include/vk_descriptor_set.h" #include "include/vk_descriptor_update_template.h" #include "include/vk_event.h" #include "include/vk_formats.h" #include "include/vk_framebuffer.h" #include "include/vk_image_view.h" #include "include/vk_render_pass.h" #include "include/vk_graphics_pipeline.h" #include "include/vk_physical_device.h" #include "include/vk_pipeline_layout.h" #include "include/vk_image.h" #include "include/vk_instance.h" #include "include/vk_utils.h" #include "include/vk_query.h" #include "include/vk_queue.h" #include "include/vk_indirect_commands_layout.h" #if VKI_RAY_TRACING #include "raytrace/vk_acceleration_structure.h" #include "raytrace/vk_ray_tracing_pipeline.h" #include "raytrace/ray_tracing_device.h" #include "raytrace/ray_tracing_util.h" #include "gpurt/gpurtLib.h" #include "gpurt/gpurtCounter.h" #endif #include "sqtt/sqtt_layer.h" #include "sqtt/sqtt_mgr.h" #include "palCmdBuffer.h" #include "palFormatInfo.h" #include "palGpuEvent.h" #include "palImage.h" #include "palQueryPool.h" #include "palSysMemory.h" #include "palDevice.h" #include "palGpuUtil.h" #include "palFormatInfo.h" #include "palVectorImpl.h" #include "palAutoBuffer.h" #include #include "devmode/devmode_mgr.h" namespace vk { namespace { constexpr Pal::BufferViewInfo EmptyVertexBufferBinding = { 0, // gpuAddr; 0, // range; 0, // stride; Pal::UndefinedSwizzledFormat, #if VKI_BUILD_GFX12 Pal::CompressionMode::Default, #endif {{0}}, // flags }; // ===================================================================================================================== // Creates a compatible PAL "clear box" structure from attachment + render area for a renderpass clear. Pal::Box BuildClearBox( const Pal::Rect& renderArea, const Framebuffer::Attachment& attachment) { Pal::Box box { }; // 2D area box.offset.x = renderArea.offset.x; box.offset.y = renderArea.offset.y; box.extent.width = renderArea.extent.width; box.extent.height = renderArea.extent.height; if (attachment.pImage->GetImageType() == VK_IMAGE_TYPE_3D) { if (attachment.pImage->Is2dArrayCompatible()) { box.offset.z = attachment.zRange.offset; box.extent.depth = attachment.zRange.extent; } else { // Whole slice range (these are offset relative to subresrange) box.offset.z = attachment.subresRange[0].startSubres.arraySlice; box.extent.depth = attachment.subresRange[0].numSlices; } } else { box.offset.z = 0; box.extent.depth = 1; } return box; } // ===================================================================================================================== // Creates a compatible PAL "clear box" structure from attachment + render area for a renderpass clear. Pal::Box BuildClearBox( const Pal::Rect& renderArea, const ImageView& imageView) { Pal::Box box{ }; // 2D area box.offset.x = renderArea.offset.x; box.offset.y = renderArea.offset.y; box.extent.width = renderArea.extent.width; box.extent.height = renderArea.extent.height; // Get the attachment image const Image* pImage = imageView.GetImage(); if (pImage->GetImageType() == VK_IMAGE_TYPE_3D) { if (pImage->Is2dArrayCompatible()) { box.offset.z = imageView.GetZRange().offset; box.extent.depth = imageView.GetZRange().extent; } else { Pal::SubresRange subresRange; imageView.GetFrameBufferAttachmentSubresRange(&subresRange); // Whole slice range (these are offset relative to subresrange) box.offset.z = subresRange.startSubres.arraySlice; box.extent.depth = subresRange.numSlices; } } else { box.offset.z = 0; box.extent.depth = 1; } return box; } // ===================================================================================================================== // Returns ranges of consecutive bits set to 1 from a bit mask. // // uint32 { 0xE47F01D6 } -> [(1, 2) (4, 1) (6, 3) (16, 7) (26, 1) (29, 3)] // // <-----> <-> <-------------> <-----> <-> <---> // +---------------------------------------------------------------+ // |1 1 1 0 0 1 0 0 0 1 1 1 1 1 1 1 0 0 0 0 0 0 0 1 1 1 0 1 0 1 1 0| // +---------------------------------------------------------------+ // // @note The implementation of RangesOfOnesInBitMask() assumes that bitMask ends with 0. // To satisfy that condition, the bitMask is promoted to uint64_t, // filled with leading zeros and looped through only relevant 33 bits. // Mentioned assumption allows avoiding edge case // in which bitMask ends in the middle of range of ones. // Util::Vector RangesOfOnesInBitMask( const uint32_t bitMask) { // Note that no allocation will be performed, so Util::Vector allocator is nullptr. Util::Vector rangesOfOnes { nullptr }; constexpr int32_t INVALID_INDEX = -1; int32_t rangeStart = INVALID_INDEX; for (int32_t bitIndex = 0; bitIndex <= 32; ++bitIndex) { const bool bitValue = (bitMask & (uint64_t { 0x1 } << bitIndex)) > 0; if (bitValue) // 1 { if (rangeStart == INVALID_INDEX) { rangeStart = bitIndex; } } else // 0 { if (rangeStart != INVALID_INDEX) { const uint32_t rangeLength = bitIndex - rangeStart; rangesOfOnes.PushBack(Pal::Range { rangeStart, rangeLength }); rangeStart = INVALID_INDEX; } } } return rangesOfOnes; } // ===================================================================================================================== // Populate a vector with PAL clear regions converted from Vulkan clear rects. // If multiview is enabled layer ranges are overridden according to viewMask. // Returns Pal::Result::Success if completed successfully. template Pal::Result CreateClearBoxes( const uint32_t rectCount, const VkClearRect* const pRects, const uint32_t viewMask, const uint32_t zOffset, const bool is3dImage, Util::Vector* const pOutClearRegions) { VK_ASSERT(pOutClearRegions != nullptr); Pal::Result palResult = Pal::Result::Success; pOutClearRegions->Clear(); // Note that it's only legal to override the default Z range in a Box if the image is 3D. If the image is not 3D // then any layer overrides must be applied to the clear's subresource range. if ((viewMask > 0) && is3dImage) { const auto layerRanges = RangesOfOnesInBitMask(viewMask); palResult = pOutClearRegions->Reserve(rectCount * layerRanges.NumElements()); if (palResult == Pal::Result::Success) { for (auto layerRangeIt = layerRanges.Begin(); layerRangeIt.IsValid(); layerRangeIt.Next()) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearRegions->PushBack(VkToPalClearBox(pRects[rectIndex], zOffset, is3dImage)); OverrideLayerRanges(pOutClearRegions->Back(), layerRangeIt.Get()); } } } } else { palResult = pOutClearRegions->Reserve(rectCount); if (palResult == Pal::Result::Success) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearRegions->PushBack(VkToPalClearBox(pRects[rectIndex], zOffset, is3dImage)); } } } return palResult; } // ===================================================================================================================== // Populate a vector with PAL clear boxes converted from Vulkan clear rects. // If multiview is enabled layer ranges are overridden according to viewMask. // Returns Pal::Result::Success if completed successfully. template Pal::Result CreateClearRegions( const uint32_t rectCount, const VkClearRect* const pRects, const uint32_t viewMask, const uint32_t zOffset, Util::Vector* const pOutClearRegions) { VK_ASSERT(pOutClearRegions != nullptr); Pal::Result palResult = Pal::Result::Success; pOutClearRegions->Clear(); if (viewMask > 0) { const auto layerRanges = RangesOfOnesInBitMask(viewMask); palResult = pOutClearRegions->Reserve(rectCount * layerRanges.NumElements()); if (palResult == Pal::Result::Success) { for (auto layerRangeIt = layerRanges.Begin(); layerRangeIt.IsValid(); layerRangeIt.Next()) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearRegions->PushBack(VkToPalClearRegion(pRects[rectIndex], zOffset)); OverrideLayerRanges(pOutClearRegions->Back(), layerRangeIt.Get()); } } } } else { palResult = pOutClearRegions->Reserve(rectCount); if (palResult == Pal::Result::Success) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearRegions->PushBack(VkToPalClearRegion(pRects[rectIndex], zOffset)); } } } return palResult; } // ===================================================================================================================== // Populate a vector with attachment's PAL subresource ranges defined by clearInfo with modified layer ranges // according to Vulkan clear rects (multiview disabled) or viewMask (multiview is enabled). // Returns Pal::Result::Success if completed successfully. template Pal::Result CreateClearSubresRanges( const vk::ImageView* pImageView, const bool is3dImage, const VkClearAttachment& clearInfo, const uint32_t rectCount, const VkClearRect* const pRects, const uint32_t viewMask, PalSubresRangeVector* const pOutClearSubresRanges) { static_assert(std::is_sameData()), Pal::SubresRange*>::value, "Wrong element type"); VK_ASSERT(pOutClearSubresRanges != nullptr); Pal::Result palResult = Pal::Result::Success; Pal::SubresRange subresRange = {}; pImageView->GetFrameBufferAttachmentSubresRange(&subresRange); pOutClearSubresRanges->Clear(); bool hasPlaneDepthAndStencil = false; if (pImageView->GetImage()->HasStencil() && pImageView->GetImage()->HasDepth()) { if (clearInfo.aspectMask == VK_IMAGE_ASPECT_STENCIL_BIT) { subresRange.startSubres.plane = 1; } else if (clearInfo.aspectMask == VK_IMAGE_ASPECT_DEPTH_BIT) { subresRange.startSubres.plane = 0; } else { hasPlaneDepthAndStencil = (clearInfo.aspectMask == (VK_IMAGE_ASPECT_STENCIL_BIT | VK_IMAGE_ASPECT_DEPTH_BIT)); } } // For 3D color images, we set up the expected subres range during ImageView::Create() call // and 3D depth/stencil images are NOT supported. // PAL expects that for all 3D images arraySlice = 0 and numSlices = 1. if (viewMask > 0) { const auto layerRanges = RangesOfOnesInBitMask(viewMask); palResult = pOutClearSubresRanges->Reserve(layerRanges.NumElements() *(hasPlaneDepthAndStencil ? 2 : 1)); if (palResult == Pal::Result::Success) { for (auto layerRangeIt = layerRanges.Begin(); layerRangeIt.IsValid(); layerRangeIt.Next()) { pOutClearSubresRanges->PushBack(subresRange); if (is3dImage == false) { pOutClearSubresRanges->Back().startSubres.arraySlice += layerRangeIt.Get().offset; pOutClearSubresRanges->Back().numSlices = layerRangeIt.Get().extent; if (hasPlaneDepthAndStencil) { subresRange.startSubres.plane = 1; pOutClearSubresRanges->PushBack(subresRange); pOutClearSubresRanges->Back().startSubres.arraySlice += layerRangeIt.Get().offset; pOutClearSubresRanges->Back().numSlices = layerRangeIt.Get().extent; } } } } } else { palResult = pOutClearSubresRanges->Reserve(rectCount *(hasPlaneDepthAndStencil ? 2 : 1)); if (palResult == Pal::Result::Success) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearSubresRanges->PushBack(subresRange); if (is3dImage == false) { pOutClearSubresRanges->Back().startSubres.arraySlice += pRects[rectIndex].baseArrayLayer; pOutClearSubresRanges->Back().numSlices = pRects[rectIndex].layerCount; if (hasPlaneDepthAndStencil) { subresRange.startSubres.plane = 1; pOutClearSubresRanges->PushBack(subresRange); pOutClearSubresRanges->Back().startSubres.arraySlice += pRects[rectIndex].baseArrayLayer; pOutClearSubresRanges->Back().numSlices = pRects[rectIndex].layerCount; } } } } } return palResult; } // ===================================================================================================================== // Populate a vector with attachment's PAL subresource ranges defined by clearInfo with modified layer ranges // according to Vulkan clear rects (multiview disabled) or viewMask (multiview is enabled). // Returns Pal::Result::Success if completed successfully. template Pal::Result CreateClearSubresRanges( const Framebuffer::Attachment& attachment, const bool is3dImage, const VkClearAttachment& clearInfo, const uint32_t rectCount, const VkClearRect* const pRects, const RenderPass& renderPass, const uint32_t subpass, PalSubresRangeVector* const pOutClearSubresRanges) { static_assert(std::is_sameData()), Pal::SubresRange*>::value, "Wrong element type"); VK_ASSERT(pOutClearSubresRanges != nullptr); Pal::Result palResult = Pal::Result::Success; const auto attachmentSubresRanges = attachment.FindSubresRanges(clearInfo.aspectMask); pOutClearSubresRanges->Clear(); // For 3D color images, we set up the expected subres range during ImageView::Create() call // and 3D depth/stencil images are NOT supported. // PAL expects that for all 3D images arraySlice = 0 and numSlices = 1. if (renderPass.IsMultiviewEnabled()) { const auto viewMask = renderPass.GetViewMask(subpass); const auto layerRanges = RangesOfOnesInBitMask(viewMask); palResult = pOutClearSubresRanges->Reserve(attachmentSubresRanges.NumElements() * layerRanges.NumElements()); if (palResult == Pal::Result::Success) { for (uint32_t rangeIndex = 0; rangeIndex < attachmentSubresRanges.NumElements(); ++rangeIndex) { for (auto layerRangeIt = layerRanges.Begin(); layerRangeIt.IsValid(); layerRangeIt.Next()) { pOutClearSubresRanges->PushBack(attachmentSubresRanges.At(rangeIndex)); if (is3dImage == false) { pOutClearSubresRanges->Back().startSubres.arraySlice += layerRangeIt.Get().offset; pOutClearSubresRanges->Back().numSlices = layerRangeIt.Get().extent; } } } } } else { palResult = pOutClearSubresRanges->Reserve(attachmentSubresRanges.NumElements() * rectCount); if (palResult == Pal::Result::Success) { for (uint32_t rangeIndex = 0; rangeIndex < attachmentSubresRanges.NumElements(); ++rangeIndex) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearSubresRanges->PushBack(attachmentSubresRanges.At(rangeIndex)); if (is3dImage == false) { pOutClearSubresRanges->Back().startSubres.arraySlice += pRects[rectIndex].baseArrayLayer; pOutClearSubresRanges->Back().numSlices = pRects[rectIndex].layerCount; } } } } } return palResult; } // ===================================================================================================================== // Returns attachment's PAL subresource ranges defined by clearInfo for LoadOp Clear. // When multiview is enabled, layer ranges are modified according active views during a renderpass. Util::Vector LoadOpClearSubresRanges( const Framebuffer::Attachment& attachment, const RPLoadOpClearInfo& clearInfo, const RenderPass& renderPass) { // Note that no allocation will be performed, so Util::Vector allocator is nullptr. Util::Vector clearSubresRanges { nullptr }; const auto attachmentSubresRanges = attachment.FindSubresRanges(clearInfo.aspect); if (renderPass.IsMultiviewEnabled()) { const auto activeViews = renderPass.GetActiveViewsBitMask(); const auto layerRanges = RangesOfOnesInBitMask(activeViews); for (uint32_t rangeIndex = 0; rangeIndex < attachmentSubresRanges.NumElements(); ++rangeIndex) { for (auto layerRangeIt = layerRanges.Begin(); layerRangeIt.IsValid(); layerRangeIt.Next()) { clearSubresRanges.PushBack(attachmentSubresRanges.At(rangeIndex)); clearSubresRanges.Back().startSubres.arraySlice += layerRangeIt.Get().offset; clearSubresRanges.Back().numSlices = layerRangeIt.Get().extent; } } } else { for (uint32_t rangeIndex = 0; rangeIndex < attachmentSubresRanges.NumElements(); ++rangeIndex) { clearSubresRanges.PushBack(attachmentSubresRanges.At(rangeIndex)); } } return clearSubresRanges; } // ===================================================================================================================== // Populate a vector with PAL rects created from Vulkan clear rects. // Returns Pal::Result::Success if completed successfully. template Pal::Result CreateClearRects( const uint32_t rectCount, const VkClearRect* const pRects, PalRectVector* const pOutClearRects) { static_assert(std::is_sameData()), Pal::Rect*>::value, "Wrong element type"); VK_ASSERT(pOutClearRects != nullptr); pOutClearRects->Clear(); const auto palResult = pOutClearRects->Reserve(rectCount); if (palResult == Pal::Result::Success) { for (uint32_t rectIndex = 0; rectIndex < rectCount; ++rectIndex) { pOutClearRects->PushBack(VkToPalRect(pRects[rectIndex].rect)); } } return palResult; } } // anonymous ns // ===================================================================================================================== CmdBuffer::CmdBuffer( Device* pDevice, CmdPool* pCmdPool, uint32_t queueFamilyIndex) : m_pDevice(pDevice), m_pCmdPool(pCmdPool), m_queueFamilyIndex(queueFamilyIndex), m_palQueueType(pDevice->GetQueueFamilyPalQueueType(queueFamilyIndex)), m_palEngineType(pDevice->GetQueueFamilyPalEngineType(queueFamilyIndex)), m_curDeviceMask(0), m_rpDeviceMask(0), m_cbBeginDeviceMask(0), m_numPalDevices(pDevice->NumPalDevices()), m_validShaderStageFlags(pDevice->VkPhysicalDevice(DefaultDeviceIndex)->GetValidShaderStages(queueFamilyIndex)), m_pStackAllocator(nullptr), m_allGpuState { }, m_flags(), m_recordingResult(VK_SUCCESS), m_pSqttState(nullptr), m_renderPassInstance(pDevice->VkInstance()->Allocator()), m_pTransformFeedbackState(nullptr), m_palDepthStencilState(pDevice->VkInstance()->Allocator()), m_palColorBlendState(pDevice->VkInstance()->Allocator()), m_palMsaaState(pDevice->VkInstance()->Allocator()), m_writtenFlippableImages(pDevice->VkInstance()->Allocator()), m_uberFetchShaderInternalDataMap(8, pDevice->VkInstance()->Allocator()), m_pUberFetchShaderTempBuffer(nullptr), m_debugPrintf(pDevice->VkInstance()->Allocator()), m_reverseThreadGroupState(false) #if VKI_RAY_TRACING , m_scratchVidMemList(pDevice->VkInstance()->Allocator()) , m_pBvhBatchState() , m_cpsCmdBufferUtil(pDevice) #endif { m_flags.wasBegun = false; m_perCmdBufDrawCallCounter = 0; m_perCmdBufDispatchCallCounter = 0; const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); m_optimizeCmdbufMode = settings.optimizeCmdbufMode; m_asyncComputeQueueMaxWavesPerCu = settings.asyncComputeQueueMaxWavesPerCu; #if VKI_ENABLE_DEBUG_BARRIERS m_dbgBarrierPreCmdMask = settings.dbgBarrierPreCmdEnable; m_dbgBarrierPostCmdMask = settings.dbgBarrierPostCmdEnable; #endif m_flags.padVertexBuffers = settings.padVertexBuffers; m_flags.prefetchCommands = settings.prefetchCommands; m_flags.prefetchShaders = settings.prefetchShaders; m_flags.disableResetReleaseResources = settings.disableResetReleaseResources; m_flags.subpassLoadOpClearsBoundAttachments = settings.subpassLoadOpClearsBoundAttachments; m_flags.preBindDefaultState = settings.preBindDefaultState; m_flags.offsetMode = pDevice->GetEnabledFeatures().robustVertexBufferExtend | pDevice->GetEnabledFeatures().pipelineRobustness; m_flags.protectFlippableImages = settings.enableBackBufferProtection & m_pDevice->VkInstance()->GetProperties().supportBlockIfFlipping; const Pal::DeviceProperties& info = m_pDevice->GetPalProperties(); m_flags.useBackupBuffer = false; memset(m_pBackupPalCmdBuffers, 0, sizeof(Pal::ICmdBuffer*) * MaxPalDevices); // If supportSplitReleaseAcquire is true, the ASIC provides split CmdRelease() and CmdAcquire() to express barrier, // and CmdReleaseThenAcquire() is still valid. This flag is currently enabled for gfx10 and above. m_flags.useReleaseAcquire = settings.useAcquireReleaseInterface; m_flags.useSplitReleaseAcquire = m_flags.useReleaseAcquire & info.queueProperties[m_palQueueType].flags.supportSplitReleaseAcquire; } // ===================================================================================================================== // Creates a new Vulkan Command Buffer object VkResult CmdBuffer::Create( Device* pDevice, const VkCommandBufferAllocateInfo* pAllocateInfo, VkCommandBuffer* pCommandBuffers) { VK_ASSERT(pAllocateInfo->sType == VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO); // Get information about the Vulkan command buffer Pal::CmdBufferCreateInfo palCreateInfo = {}; CmdPool* pCmdPool = CmdPool::ObjectFromHandle(pAllocateInfo->commandPool); uint32 queueFamilyIndex = pCmdPool->GetQueueFamilyIndex(); uint32 commandBufferCount = pAllocateInfo->commandBufferCount; palCreateInfo.pCmdAllocator = pCmdPool->PalCmdAllocator(DefaultDeviceIndex); palCreateInfo.queueType = pDevice->GetQueueFamilyPalQueueType(queueFamilyIndex); palCreateInfo.engineType = pDevice->GetQueueFamilyPalEngineType(queueFamilyIndex); palCreateInfo.flags.nested = (pAllocateInfo->level > VK_COMMAND_BUFFER_LEVEL_PRIMARY) ? 1 : 0; palCreateInfo.flags.dispatchTunneling = 1; #if VKI_BUILD_GFX12 palCreateInfo.flags.dispatchPingPongWalk = (pDevice->GetRuntimeSettings().dispatchPingPong == DispatchPingPongHw); #endif // Allocate system memory for the command buffer objects Pal::Result palResult; const uint32 numGroupedCmdBuffers = pDevice->NumPalDevices(); const size_t apiSize = sizeof(ApiCmdBuffer); const size_t perGpuSize = sizeof(PerGpuRenderState) * numGroupedCmdBuffers; const size_t palSize = pDevice->PalDevice(DefaultDeviceIndex)-> GetCmdBufferSize(palCreateInfo, &palResult) * numGroupedCmdBuffers; size_t inaccessibleSize = 0; // Accumulate the setBindingData size that will not be accessed based on available pipeline bind points. { #if VKI_RAY_TRACING static_assert(PipelineBindRayTracing + 1 == PipelineBindCount, "This code relies on the enum order!"); if (pDevice->IsExtensionEnabled(DeviceExtensions::KHR_RAY_TRACING_PIPELINE) == false) { inaccessibleSize += sizeof(uint32) * MaxBindingRegCount; static_assert(PipelineBindGraphics + 1 == PipelineBindRayTracing, "This code relies on the enum order!"); #else { #endif static_assert(PipelineBindCompute + 1 == PipelineBindGraphics, "This code relies on the enum order!"); if (palCreateInfo.queueType == Pal::QueueType::QueueTypeCompute) { inaccessibleSize += sizeof(uint32) * MaxBindingRegCount; } } } // Accumulate the setBindingData size that will not be accessed based on the dynamic descriptor data size inaccessibleSize += (MaxDynDescRegCount - (MaxDynamicDescriptors * DescriptorSetLayout::GetDynamicBufferDescDwSize(pDevice))) * sizeof(uint32); // The total object size less any inaccessible setBindingData (for the last device only to not disrupt MGPU indexing) size_t cmdBufSize = apiSize + palSize + perGpuSize - inaccessibleSize; size_t sizeDesBuf = 0; if (pDevice->IsExtensionEnabled(DeviceExtensions::EXT_DESCRIPTOR_BUFFER)) { // Descriptor buffers have a single dedicated bind point. sizeDesBuf = sizeof(DescBufBinding); cmdBufSize += sizeDesBuf; } VK_ASSERT(palResult == Pal::Result::Success); VkResult result = VK_SUCCESS; uint32 allocCount = 0; while ((result == VK_SUCCESS) && (allocCount < commandBufferCount)) { // Allocate memory for the command buffer void* pMemory = pDevice->AllocApiObject(pCmdPool->GetCmdPoolAllocator(), cmdBufSize); // Create the command buffer if (pMemory != nullptr) { void* pPalMem = Util::VoidPtrInc(pMemory, apiSize + perGpuSize - inaccessibleSize); VK_INIT_DISPATCHABLE(CmdBuffer, pMemory, (pDevice, pCmdPool, queueFamilyIndex)); pCommandBuffers[allocCount] = reinterpret_cast(pMemory); CmdBuffer* pCmdBuffer = ApiCmdBuffer::ObjectFromHandle(pCommandBuffers[allocCount]); if ((sizeDesBuf != 0) && (result == VK_SUCCESS)) { pCmdBuffer->m_allGpuState.pDescBufBinding = static_cast( Util::VoidPtrInc(pPalMem, palSize)); memset(pCmdBuffer->m_allGpuState.pDescBufBinding, 0, sizeof(DescBufBinding)); } else { pCmdBuffer->m_allGpuState.pDescBufBinding = nullptr; } result = pCmdBuffer->Initialize(pPalMem, palCreateInfo); allocCount++; } else { result = VK_ERROR_OUT_OF_HOST_MEMORY; } } if (result != VK_SUCCESS) { // Failed to create at least one command buffer; destroy any command buffers that we did succeed in creating for (uint32_t bufIdx = 0; bufIdx < commandBufferCount; ++bufIdx) { if (bufIdx < allocCount) { ApiCmdBuffer::ObjectFromHandle(pCommandBuffers[bufIdx])->Destroy(); } // No partial failures allowed for creating multiple command buffers. Update all to VK_NULL_HANDLE. pCommandBuffers[bufIdx] = VK_NULL_HANDLE; } } return result; } // ===================================================================================================================== // Initializes the command buffer. Called once during command buffer creation. VkResult CmdBuffer::Initialize( void* pPalMem, const Pal::CmdBufferCreateInfo& createInfo) { Pal::Result result = Pal::Result::Success; Pal::CmdBufferCreateInfo groupCreateInfo = createInfo; // Create the PAL command buffers size_t palMemOffset = 0; const size_t palSize = m_pDevice->PalDevice(DefaultDeviceIndex)->GetCmdBufferSize(groupCreateInfo, &result); const uint32_t numGroupedCmdBuffers = m_numPalDevices; for (uint32_t groupedIdx = 0; (groupedIdx < numGroupedCmdBuffers) && (result == Pal::Result::Success); groupedIdx++) { Pal::IDevice* const pPalDevice = m_pDevice->PalDevice(groupedIdx); groupCreateInfo.pCmdAllocator = m_pCmdPool->PalCmdAllocator(groupedIdx); result = pPalDevice->CreateCmdBuffer( groupCreateInfo, Util::VoidPtrInc(pPalMem, palMemOffset), &m_pPalCmdBuffers[groupedIdx]); if (result == Pal::Result::Success) { m_pPalCmdBuffers[groupedIdx]->SetClientData(this); palMemOffset += palSize; VK_ASSERT(palSize == pPalDevice->GetCmdBufferSize(groupCreateInfo, &result)); VK_ASSERT(result == Pal::Result::Success); } } if (result == Pal::Result::Success) { InitializeVertexBuffer(); } if (result == Pal::Result::Success) { // Register this command buffer with the pool result = m_pCmdPool->RegisterCmdBuffer(this); } if (result == Pal::Result::Success) { m_flags.is2ndLvl = groupCreateInfo.flags.nested; m_allGpuState.stencilRefMasks.flags.u8All = 0xff; // Set up the default front/back op values == 1 m_allGpuState.stencilRefMasks.frontOpValue = DefaultStencilOpValue; m_allGpuState.stencilRefMasks.backOpValue = DefaultStencilOpValue; m_allGpuState.logicOpEnable = VK_FALSE; m_allGpuState.logicOp = VK_LOGIC_OP_COPY; } // Initialize SQTT command buffer state if thread tracing support is enabled (gpuopen developer mode). if ((result == Pal::Result::Success) && (m_pDevice->GetSqttMgr() != nullptr)) { void* pSqttStorage = m_pDevice->VkInstance()->AllocMem(sizeof(SqttCmdBufferState), VK_SYSTEM_ALLOCATION_SCOPE_OBJECT); if (pSqttStorage != nullptr) { m_pSqttState = VK_PLACEMENT_NEW(pSqttStorage) SqttCmdBufferState(this); } else { result = Pal::Result::ErrorOutOfMemory; } } if (result == Pal::Result::Success) { result = m_uberFetchShaderInternalDataMap.Init(); } if ((result == Pal::Result::Success) && (createInfo.queueType == Pal::QueueType::QueueTypeDma)) { result = BackupInitialize(createInfo); } if (result == Pal::Result::Success) { m_debugPrintf.Init(m_pDevice); } #if VKI_RAY_TRACING m_pfnTraceRaysDispatchPerDevice = CmdBuffer::TraceRaysDispatchPerDevice; #endif return PalToVkResult(result); } // ===================================================================================================================== // Create backup pal cmdbuffer, only call when DMA queue cmdbuffer be created Pal::Result CmdBuffer::BackupInitialize( const Pal::CmdBufferCreateInfo& createInfo) { Pal::Result palResult = Pal::Result::Success; const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); if (m_pDevice->VkPhysicalDevice(DefaultDeviceIndex)->IsComputeEngineSupported() && settings.useBackupCmdbuffer) { for (uint32_t queuefamilyIdx = 0; queuefamilyIdx < Queue::MaxQueueFamilies; queuefamilyIdx++) { if (m_pDevice->VkPhysicalDevice(DefaultDeviceIndex)->GetQueueFamilyPalQueueType(queuefamilyIdx) == Pal::QueueType::QueueTypeCompute) { m_backupQueueFamilyIndex = queuefamilyIdx; break; } } Pal::CmdBufferCreateInfo palCreateInfo = createInfo; const VkAllocationCallbacks* pAllocCB = m_pCmdPool->GetCmdPoolAllocator(); for (uint32_t deviceIdx = 0; deviceIdx < m_pDevice->NumPalDevices(); ++deviceIdx) { palCreateInfo.pCmdAllocator = m_pCmdPool->PalCmdAllocator(deviceIdx); palCreateInfo.queueType = Pal::QueueTypeCompute; palCreateInfo.engineType = Pal::EngineTypeCompute; Pal::IDevice* const pPalDevice = m_pDevice->PalDevice(deviceIdx); const size_t palSize = pPalDevice->GetCmdBufferSize(palCreateInfo, &palResult); if (palResult == Pal::Result::Success) { void* pMemory = pAllocCB->pfnAllocation(pAllocCB->pUserData, palSize, VK_DEFAULT_MEM_ALIGN, VK_SYSTEM_ALLOCATION_SCOPE_OBJECT); if (pMemory != nullptr) { palResult = pPalDevice->CreateCmdBuffer(palCreateInfo, pMemory, &m_pBackupPalCmdBuffers[deviceIdx]); if (palResult == Pal::Result::Success) { m_pBackupPalCmdBuffers[deviceIdx]->SetClientData(this); } else { pAllocCB->pfnFree( pAllocCB->pUserData, pMemory); break; } } else { palResult = Pal::Result::ErrorOutOfMemory; } } } if (palResult != Pal::Result::Success) { for (uint32_t deviceIdx = 0; deviceIdx < m_pDevice->NumPalDevices(); ++deviceIdx) { if (m_pBackupPalCmdBuffers[deviceIdx] != nullptr) { m_pBackupPalCmdBuffers[deviceIdx]->Destroy(); pAllocCB->pfnFree( pAllocCB->pUserData, m_pBackupPalCmdBuffers[deviceIdx]); } } } } return palResult; } // ===================================================================================================================== // Will switch to use backupcmdbuffer based on m_flags.useBackupBuffer void CmdBuffer::SwitchToBackupCmdBuffer() { if ((m_flags.useBackupBuffer == false) && (m_pBackupPalCmdBuffers[0] != nullptr)) { // need to use backupbuffer set the flag m_flags.useBackupBuffer = true; uint32_t tempQueueFamilyIndex = m_queueFamilyIndex; m_queueFamilyIndex = m_backupQueueFamilyIndex; m_backupQueueFamilyIndex = tempQueueFamilyIndex; m_palQueueType = Pal::QueueType::QueueTypeCompute; m_palEngineType = Pal::EngineType::EngineTypeCompute; for (uint32_t deviceIdx = 0; deviceIdx < m_pDevice->NumPalDevices(); ++deviceIdx) { constexpr Pal::CmdBufferBuildInfo info = { }; m_pBackupPalCmdBuffers[deviceIdx]->Begin(info); Pal::ICmdBuffer* tempCmdBuffer = m_pBackupPalCmdBuffers[deviceIdx]; m_pBackupPalCmdBuffers[deviceIdx] = m_pPalCmdBuffers[deviceIdx]; m_pPalCmdBuffers[deviceIdx] = tempCmdBuffer; } } } // ===================================================================================================================== // Will restored from backupcmdbuffer based on m_flags.useBackupBuffer void CmdBuffer::RestoreFromBackupCmdBuffer() { if (m_flags.useBackupBuffer) { // need to use original palcmdbuffer uint32_t tempQueueFamilyIndex = m_queueFamilyIndex; m_queueFamilyIndex = m_backupQueueFamilyIndex; m_backupQueueFamilyIndex = tempQueueFamilyIndex; m_palQueueType = Pal::QueueType::QueueTypeDma; m_palEngineType = Pal::EngineType::EngineTypeDma; for (uint32_t deviceIdx = 0; deviceIdx < m_pDevice->NumPalDevices(); ++deviceIdx) { m_pPalCmdBuffers[deviceIdx]->End(); Pal::ICmdBuffer* tempCmdBuffer = m_pBackupPalCmdBuffers[deviceIdx]; m_pBackupPalCmdBuffers[deviceIdx] = m_pPalCmdBuffers[deviceIdx]; m_pPalCmdBuffers[deviceIdx] = tempCmdBuffer; } } } // ===================================================================================================================== Pal::Result CmdBuffer::PalCmdBufferBegin(const Pal::CmdBufferBuildInfo& cmdInfo) { Pal::Result result = Pal::Result::Success; utils::IterateMask deviceGroup(m_cbBeginDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); result = PalCmdBuffer(deviceIdx)->Begin(cmdInfo); VK_ASSERT(result == Pal::Result::Success); const Pal::IBorderColorPalette* pPalBorderColorPalette = m_pDevice->GetPalBorderColorPalette(deviceIdx); if (pPalBorderColorPalette != nullptr) { if ((m_palQueueType == Pal::QueueTypeUniversal) || (m_palQueueType == Pal::QueueTypeCompute)) { if (m_palQueueType == Pal::QueueTypeUniversal) { // Bind graphics border color palette on universal queue. PalCmdBuffer(deviceIdx)->CmdBindBorderColorPalette( Pal::PipelineBindPoint::Graphics, pPalBorderColorPalette); } PalCmdBuffer(deviceIdx)->CmdBindBorderColorPalette( Pal::PipelineBindPoint::Compute, pPalBorderColorPalette); } } } while (deviceGroup.IterateNext()); return result; } // ===================================================================================================================== Pal::Result CmdBuffer::PalCmdBufferEnd() { Pal::Result result = Pal::Result::Success; utils::IterateMask deviceGroup(m_cbBeginDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); result = PalCmdBuffer(deviceIdx)->End(); VK_ASSERT(result == Pal::Result::Success); } while (deviceGroup.IterateNext()); return result; } // ===================================================================================================================== Pal::Result CmdBuffer::PalCmdBufferReset(bool returnGpuMemory) { Pal::Result result = Pal::Result::Success; // If there was no begin, skip the reset if (m_cbBeginDeviceMask != 0) { utils::IterateMask deviceGroup(m_cbBeginDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); result = PalCmdBuffer(deviceIdx)->Reset(nullptr, returnGpuMemory); VK_ASSERT(result == Pal::Result::Success); } while (deviceGroup.IterateNext()); if (returnGpuMemory) { m_cbBeginDeviceMask = 0; } } return result; } // ===================================================================================================================== void CmdBuffer::PalCmdBufferDestroy() { for (uint32_t deviceIdx = 0; deviceIdx < VkDevice()->NumPalDevices(); deviceIdx++) { Pal::ICmdBuffer* pCmdBuffer = PalCmdBuffer(deviceIdx); if (pCmdBuffer != nullptr) { pCmdBuffer->Destroy(); } } } // ===================================================================================================================== void CmdBuffer::PalCmdBindIndexData( Buffer* pBuffer, Pal::gpusize offset, Pal::IndexType indexType, Pal::gpusize bufferSize) { uint32_t indexCount = 0; if (bufferSize == VK_WHOLE_SIZE) { indexCount = utils::BufferSizeToIndexCount(indexType, pBuffer->GetSize() - offset); } else { indexCount = utils::BufferSizeToIndexCount(indexType, bufferSize); } utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); const Pal::gpusize gpuVirtAddr = pBuffer->GpuVirtAddr(deviceIdx) + offset; PalCmdBuffer(deviceIdx)->CmdBindIndexData(gpuVirtAddr, indexCount, indexType); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdUnbindIndexData(Pal::IndexType indexType) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdBindIndexData(0, 0, indexType); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdDraw( uint32_t firstVertex, uint32_t vertexCount, uint32_t firstInstance, uint32_t instanceCount, uint32_t drawId) { // Currently only Vulkan graphics pipelines use PAL graphics pipeline bindings so there's no need to // add a delayed validation check for graphics. VK_ASSERT(PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Graphics, PipelineBindGraphics)); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdDraw(firstVertex, vertexCount, firstInstance, instanceCount, drawId); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdDrawIndexed( uint32_t firstIndex, uint32_t indexCount, int32_t vertexOffset, uint32_t firstInstance, uint32_t instanceCount, uint32_t drawId) { // Currently only Vulkan graphics pipelines use PAL graphics pipeline bindings so there's no need to // add a delayed validation check for graphics. VK_ASSERT(PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Graphics, PipelineBindGraphics)); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdDrawIndexed(firstIndex, indexCount, vertexOffset, firstInstance, instanceCount, drawId); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdDrawMeshTasks( uint32_t x, uint32_t y, uint32_t z) { utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdDispatchMesh({ x, y, z }); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== template void CmdBuffer::PalCmdDrawMeshTasksIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset) { Buffer* pBuffer = Buffer::ObjectFromHandle(buffer); // The indirect argument should be in the range of the given buffer size VK_ASSERT((stride + offset) <= pBuffer->PalMemory(DefaultDeviceIndex)->Desc().size); Pal::gpusize countVirtAddr = 0; utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); if (useBufferCount) { Buffer* pCountBuffer = Buffer::ObjectFromHandle(countBuffer); countVirtAddr = pCountBuffer->GpuVirtAddr(deviceIdx) + countOffset; } Pal::GpuVirtAddrAndStride gpuVirtAddrAndStride = { pBuffer->GpuVirtAddr(deviceIdx) + static_cast(offset), {stride}, }; PalCmdBuffer(deviceIdx)->CmdDispatchMeshIndirectMulti( gpuVirtAddrAndStride, count, countVirtAddr); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdDispatch( uint32_t x, uint32_t y, uint32_t z) { utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdDispatch({ x, y, z }, {}); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdDispatchOffset( uint32_t base_x, uint32_t base_y, uint32_t base_z, uint32_t size_x, uint32_t size_y, uint32_t size_z) { utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdDispatchOffset({ base_x, base_y, base_z }, { size_x, size_y, size_z }, { size_x, size_y, size_z }); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdDispatchIndirect( Buffer* pBuffer, Pal::gpusize offset) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); // TODO use device group dispatch offsets here. // Note: check spec to see if offset setting is applications' responsibility. PalCmdBuffer(deviceIdx)->CmdDispatchIndirect(pBuffer->GpuVirtAddr(deviceIdx) + offset); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== // Begin Vulkan command buffer VkResult CmdBuffer::Begin( const VkCommandBufferBeginInfo* pBeginInfo) { VK_ASSERT(pBeginInfo->sType == VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO); VK_ASSERT(!m_flags.isRecording); #if VKI_RAY_TRACING m_flags.hasRayTracing = false; #endif m_flags.isRenderingSuspended = false; m_flags.wasBegun = true; // Beginning a command buffer implicitly resets its state ResetState(); #if VKI_RAY_TRACING FreeRayTracingScratchVidMemory(); m_cpsCmdBufferUtil.SetCpsMemSize(0); #endif const PhysicalDevice* pPhysicalDevice = m_pDevice->VkPhysicalDevice(DefaultDeviceIndex); const Pal::DeviceProperties& deviceProps = pPhysicalDevice->PalProperties(); m_flags.useBackupBuffer = false; const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); Pal::CmdBufferBuildInfo cmdInfo = {}; RenderPass* pRenderPass = nullptr; Framebuffer* pFramebuffer = nullptr; const VkCommandBufferInheritanceRenderingInfo* pInheritanceRenderingInfo = nullptr; const VkRenderingAttachmentLocationInfoKHR* pInheritanceRenderingAttachmentLocationInfoKHR = nullptr; m_cbBeginDeviceMask = m_pDevice->GetPalDeviceMask(); cmdInfo.flags.u32All = 0; // Disabling prefetch on compute queues by default should be better since PAL's prefetch uses DMA_DATA which causes // the CP to idle and switch queues on async compute. if ((settings.enableAceShaderPrefetch) || (m_palQueueType != Pal::QueueTypeCompute)) { cmdInfo.flags.prefetchCommands = m_flags.prefetchCommands; cmdInfo.flags.prefetchShaders = m_flags.prefetchShaders; } if (IsProtected()) { cmdInfo.flags.enableTmz = 1; } Pal::InheritedStateParams inheritedStateParams = {}; uint32 currentSubPass = 0; cmdInfo.flags.optimizeOneTimeSubmit = (pBeginInfo->flags & VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT) ? 1 : 0; // To match DXCP's behavior for multiSubmitChaining, we keep the flag off unless these conditions are met if (settings.multiSubmitChaining && ((pBeginInfo->flags & VK_COMMAND_BUFFER_USAGE_SIMULTANEOUS_USE_BIT) == 0)) { cmdInfo.flags.optimizeExclusiveSubmit = 1; } switch (m_optimizeCmdbufMode) { case EnableOptimizeForRenderPassContinue: cmdInfo.flags.optimizeGpuSmallBatch = (pBeginInfo->flags & VK_COMMAND_BUFFER_USAGE_RENDER_PASS_CONTINUE_BIT) ? 1 : 0; break; case EnableOptimizeCmdbuf: cmdInfo.flags.optimizeGpuSmallBatch = 1; break; case DisableOptimizeCmdbuf: cmdInfo.flags.optimizeGpuSmallBatch = 0; break; default: cmdInfo.flags.optimizeGpuSmallBatch = (pBeginInfo->flags & VK_COMMAND_BUFFER_USAGE_RENDER_PASS_CONTINUE_BIT) ? 1 : 0; break; } if (m_flags.is2ndLvl && (pBeginInfo->pInheritanceInfo != nullptr)) { // Only provide valid inherited state pointer for 2nd level command buffers cmdInfo.pInheritedState = &inheritedStateParams; pRenderPass = RenderPass::ObjectFromHandle(pBeginInfo->pInheritanceInfo->renderPass); pFramebuffer = Framebuffer::ObjectFromHandle(pBeginInfo->pInheritanceInfo->framebuffer); currentSubPass = pBeginInfo->pInheritanceInfo->subpass; if (pBeginInfo->pInheritanceInfo->occlusionQueryEnable) { inheritedStateParams.stateFlags.occlusionQuery = 1; } const void* pNext = pBeginInfo->pInheritanceInfo->pNext; while (pNext != nullptr) { const auto* pHeader = static_cast(pNext); if (pHeader->sType == VK_STRUCTURE_TYPE_COMMAND_BUFFER_INHERITANCE_CONDITIONAL_RENDERING_INFO_EXT) { const auto* pExtInfo = static_cast(pNext); inheritedStateParams.stateFlags.predication = pExtInfo->conditionalRenderingEnable; m_flags.hasConditionalRendering = pExtInfo->conditionalRenderingEnable; } else if (pHeader->sType == VK_STRUCTURE_TYPE_COMMAND_BUFFER_INHERITANCE_RENDERING_INFO) { VK_ASSERT(m_flags.is2ndLvl); pInheritanceRenderingInfo = static_cast(pNext); inheritedStateParams.stateFlags.targetViewState = 1; } else if (pHeader->sType == VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_LOCATION_INFO_KHR) { pInheritanceRenderingAttachmentLocationInfoKHR = static_cast(pNext); } pNext = pHeader->pNext; } } const void* pNext = pBeginInfo->pNext; while (pNext != nullptr) { const auto* pHeader = static_cast(pNext); switch (static_cast(pHeader->sType)) { // Convert Vulkan flags to PAL flags. case VK_STRUCTURE_TYPE_DEVICE_GROUP_COMMAND_BUFFER_BEGIN_INFO: { const auto* pDeviceGroupInfo = static_cast(pNext); // Check that the application did not set any bits outside of our device group mask. VK_ASSERT((m_cbBeginDeviceMask & pDeviceGroupInfo->deviceMask) == pDeviceGroupInfo->deviceMask); m_cbBeginDeviceMask &= pDeviceGroupInfo->deviceMask; break; } default: // Skip any unknown extension structures break; } pNext = pHeader->pNext; } m_curDeviceMask = m_cbBeginDeviceMask; if (pRenderPass != nullptr) // secondary VkCommandBuffer will be used inside VkRenderPass { VK_ASSERT(m_flags.is2ndLvl); inheritedStateParams.stateFlags.targetViewState = 1; } Pal::Result result = PalCmdBufferBegin(cmdInfo); if (result == Pal::Result::Success) { result = m_pCmdPool->MarkCmdBufBegun(this); } if (result == Pal::Result::Success) { if (m_pStackAllocator == nullptr) { result = m_pDevice->VkInstance()->StackMgr()->AcquireAllocator(&m_pStackAllocator); } } DbgBarrierPreCmd(DbgBarrierCmdBufStart); VK_ASSERT(result == Pal::Result::Success); if (m_pSqttState != nullptr) { m_pSqttState->Begin(pBeginInfo); } if (result == Pal::Result::Success) { // If we have to resume an already started render pass then we have to do it here if (pRenderPass != nullptr) { m_allGpuState.pRenderPass = pRenderPass; m_renderPassInstance.subpass = currentSubPass; } if (pInheritanceRenderingInfo != nullptr) { m_allGpuState.dynamicRenderingInstance.viewMask = pInheritanceRenderingInfo->viewMask; m_allGpuState.dynamicRenderingInstance.colorAttachmentCount = pInheritanceRenderingInfo->colorAttachmentCount; for (uint32_t i = 0; i < m_allGpuState.dynamicRenderingInstance.colorAttachmentCount; ++i) { DynamicRenderingAttachments* pDynamicAttachment = &m_allGpuState.dynamicRenderingInstance.colorAttachments[i]; pDynamicAttachment->pImageView = nullptr; pDynamicAttachment->attachmentFormat = pInheritanceRenderingInfo->pColorAttachmentFormats[i]; pDynamicAttachment->rasterizationSamples = pInheritanceRenderingInfo->rasterizationSamples; if (pInheritanceRenderingAttachmentLocationInfoKHR != nullptr) { m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i] = pInheritanceRenderingAttachmentLocationInfoKHR->pColorAttachmentLocations[i]; } else { m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i] = i; } } m_allGpuState.dynamicRenderingInstance.depthAttachment.attachmentFormat = (pInheritanceRenderingInfo->depthAttachmentFormat != VK_FORMAT_UNDEFINED) ? pInheritanceRenderingInfo->depthAttachmentFormat : pInheritanceRenderingInfo->stencilAttachmentFormat; m_allGpuState.dynamicRenderingInstance.depthAttachment.rasterizationSamples = pInheritanceRenderingInfo->rasterizationSamples; } // if input frame buffer object pointer is NULL, it means // either this is for a primary command buffer, or this is a secondary command buffer // and the command buffer will get the frame buffer object and execution time from // beginRenderPass called in the primary command buffer if (pFramebuffer != nullptr) { m_allGpuState.pFramebuffer = pFramebuffer; } } m_flags.isRecording = true; if ((pRenderPass != nullptr) || (pInheritanceRenderingInfo != nullptr)) // secondary VkCommandBuffer will be used inside VkRenderPass { VK_ASSERT(m_flags.is2ndLvl); // In order to use secondary VkCommandBuffer inside VkRenderPass, // when vkBeginCommandBuffer() is called, the VkCommandBufferInheritanceInfo // has to specify a VkRenderPass, defining VkRenderPasses with which // the secondary VkCommandBuffer will be compatible with // and a subpass in which that secondary VkCommandBuffer will be used. // // Note that two compatible VkRenderPasses have to define // exactly the same sequence of ViewMasks. // // Therefore, ViewMask can be retrived from VkRenderPass using subpass // and baked into secondary VkCommandBuffer. // Vulkan spec guarantees that ViewMask will not have to be updated. // // Because secondary VkCommandBuffer will be called inside of a VkRenderPass // function setting ViewMask for a subpass during the VkRenderPass is called. SetViewInstanceMask(GetDeviceMask()); } if (m_palQueueType == Pal::QueueTypeUniversal) { const VkPhysicalDeviceLimits& limits = pPhysicalDevice->GetLimits(); Pal::GlobalScissorParams scissorParams = { }; scissorParams.scissorRegion.extent.width = limits.maxFramebufferWidth; scissorParams.scissorRegion.extent.height = limits.maxFramebufferHeight; { utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetGlobalScissor(scissorParams); } while (deviceGroup.IterateNext()); } m_allGpuState.staticTokens.pointLineRasterState = DynamicRenderStateToken; const Pal::PointLineRasterStateParams params = { DefaultPointSize, DefaultLineWidth, limits.pointSizeRange[0], limits.pointSizeRange[1] }; { utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetPointLineRasterState(params); } while (deviceGroup.IterateNext()); } const uint32_t supportedVrsRates = deviceProps.gfxipProperties.supportedVrsRates; // Turn variable rate shading off if it is supported. if (supportedVrsRates & (1 << static_cast(Pal::VrsShadingRate::_1x1))) { Pal::VrsCenterState centerState = {}; m_allGpuState.vrsRate = {}; Device::SetDefaultVrsRateParams(&m_allGpuState.vrsRate); utils::IterateMask deviceGroupVrs(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroupVrs.Index(); PalCmdBuffer(deviceIdx)->CmdSetVrsCenterState(centerState); // A null source image implies 1x1 shading rate for the image combiner stage. PalCmdBuffer(deviceIdx)->CmdBindSampleRateImage(nullptr); } while (deviceGroupVrs.IterateNext()); } // Reset transform feedback-related state once, in case it'll be used without first binding a valid xfb. // This is legal, but no primitives data will be generated until a valid xfb is bound in the pipeline. if (m_pDevice->IsExtensionEnabled(DeviceExtensions::EXT_TRANSFORM_FEEDBACK)) { utils::IterateMask deviceGroup(GetDeviceMask()); do { // Disable transform feedback by setting bound buffer's size and stride to 0. const uint32_t deviceIdx = deviceGroup.Index(); const Pal::BindStreamOutTargetParams nullParams = {}; PalCmdBuffer(deviceIdx)->CmdBindStreamOutTargets(nullParams); } while (deviceGroup.IterateNext()); } } // Dirty all the dynamic states, the bit should be cleared with 0 when the corresponding state is // static. m_allGpuState.dirtyGraphics.u32All = 0xFFFFFFFF; if ((m_palQueueType == Pal::QueueTypeUniversal) && m_flags.preBindDefaultState) { // Set VRS state now to avoid at bind time const uint32_t supportedVrsRates = deviceProps.gfxipProperties.supportedVrsRates; if (supportedVrsRates & (1 << static_cast(m_allGpuState.vrsRate.shadingRate))) { utils::IterateMask deviceGroupVrs(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroupVrs.Index(); PalCmdBuffer(deviceIdx)->CmdSetPerDrawVrsRate(m_allGpuState.vrsRate); } while (deviceGroupVrs.IterateNext()); } m_allGpuState.dirtyGraphics.vrs = 0; // Set default sample pattern m_allGpuState.samplePattern.sampleCount = 1; m_allGpuState.samplePattern.locations = *Device::GetDefaultQuadSamplePattern(m_allGpuState.samplePattern.sampleCount); m_allGpuState.sampleLocationsEnable = VK_FALSE; PalCmdSetMsaaQuadSamplePattern(m_allGpuState.samplePattern.sampleCount, m_allGpuState.samplePattern.locations); m_allGpuState.dirtyGraphics.samplePattern = 0; } DbgBarrierPostCmd(DbgBarrierCmdBufStart); return PalToVkResult(result); } // ===================================================================================================================== // End Vulkan command buffer VkResult CmdBuffer::End(void) { Pal::Result result; VK_ASSERT(m_flags.isRecording); DbgBarrierPreCmd(DbgBarrierCmdBufEnd); // ValidateGraphicsStates tries to update things like viewport or input assembly // only cmdBuffers specialized in graphics (universal) are going to use that state // other implementations have stub setters with PAL_NEVER_CALLED asserts if (m_palQueueType == Pal::QueueTypeUniversal) { ValidateGraphicsStates(); } if (m_pSqttState != nullptr) { m_pSqttState->End(); } DbgBarrierPostCmd(DbgBarrierCmdBufEnd); RestoreFromBackupCmdBuffer(); result = PalCmdBufferEnd(); m_flags.isRecording = false; return (m_recordingResult == VK_SUCCESS ? PalToVkResult(result) : m_recordingResult); } // ===================================================================================================================== // Resets all state PipelineState. This function is called both during vkBeginCommandBuffer (inside // CmdBuffer::ResetState()) and during vkResetCommandBuffer (inside CmdBuffer::ResetState()) and during // vkExecuteCommands. void CmdBuffer::ResetPipelineState() { m_allGpuState.boundGraphicsPipelineHash = 0; m_allGpuState.pGraphicsPipeline = nullptr; m_allGpuState.pComputePipeline = nullptr; #if VKI_RAY_TRACING m_allGpuState.pRayTracingPipeline = nullptr; #endif ResetVertexBuffer(); // Reset initial static values to "dynamic" values. This will skip initial redundancy checking because the // prior values are unknown. Since DynamicRenderStateToken is 0, this is covered by the memset above. static_assert(DynamicRenderStateToken == 0, "Unexpected value!"); memset(&m_allGpuState.staticTokens, 0u, sizeof(m_allGpuState.staticTokens)); memset(&m_allGpuState.depthStencilCreateInfo, 0u, sizeof(m_allGpuState.depthStencilCreateInfo)); memset(&m_allGpuState.samplePattern, 0u, sizeof(m_allGpuState.samplePattern)); m_allGpuState.depthClampOverride.minDepthClamp = 1.0f; m_allGpuState.depthClampOverride.maxDepthClamp = 0.0f; uint32_t bindIdx = 0; do { memset(&(m_allGpuState.pipelineState[bindIdx].userDataLayout), 0, sizeof(m_allGpuState.pipelineState[bindIdx].userDataLayout)); m_allGpuState.pipelineState[bindIdx].boundSetCount = 0; m_allGpuState.pipelineState[bindIdx].pushedConstCount = 0; m_allGpuState.pipelineState[bindIdx].dynamicBindInfo = {}; m_allGpuState.pipelineState[bindIdx].hasDynamicVertexInput = false; m_allGpuState.pipelineState[bindIdx].pVertexInputInternalData = nullptr; bindIdx++; } while (bindIdx < PipelineBindCount); auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; pDynamicState->colorWriteMask = UINT32_MAX; pDynamicState->logicOp = Pal::LogicOp::Copy; m_allGpuState.colorWriteMask = UINT32_MAX; m_allGpuState.colorWriteEnable = UINT32_MAX; m_allGpuState.logicOp = VK_LOGIC_OP_COPY; // Default MSAA state m_allGpuState.msaaCreateInfo.coverageSamples = 1; m_allGpuState.msaaCreateInfo.exposedSamples = 0; m_allGpuState.msaaCreateInfo.pixelShaderSamples = 1; m_allGpuState.msaaCreateInfo.depthStencilSamples = 1; m_allGpuState.msaaCreateInfo.shaderExportMaskSamples = 1; m_allGpuState.msaaCreateInfo.sampleMask = 1; m_allGpuState.msaaCreateInfo.sampleClusters = 1; m_allGpuState.msaaCreateInfo.alphaToCoverageSamples = 1; m_allGpuState.msaaCreateInfo.occlusionQuerySamples = 1; m_allGpuState.triangleRasterState.frontFillMode = Pal::FillMode::Solid; m_allGpuState.triangleRasterState.backFillMode = Pal::FillMode::Solid; m_allGpuState.palToApiPipeline[uint32_t(Pal::PipelineBindPoint::Compute)] = PipelineBindCompute; m_allGpuState.palToApiPipeline[uint32_t(Pal::PipelineBindPoint::Graphics)] = PipelineBindGraphics; const uint32_t numPalDevices = m_numPalDevices; uint32_t deviceIdx = 0; do { PerGpuRenderState* pPerGpuState = PerGpuState(deviceIdx); pPerGpuState->pMsaaState = nullptr; pPerGpuState->pColorBlendState = nullptr; pPerGpuState->pDepthStencilState = nullptr; pPerGpuState->scissor.count = 1; pPerGpuState->scissor.scissors[0] = {}; pPerGpuState->viewport.count = 1; pPerGpuState->viewport.viewports[0] = {}; pPerGpuState->viewport.horzClipRatio = FLT_MAX; pPerGpuState->viewport.vertClipRatio = FLT_MAX; pPerGpuState->viewport.horzDiscardRatio = 1.0f; pPerGpuState->viewport.vertDiscardRatio = 1.0f; pPerGpuState->viewport.depthRange = Pal::DepthRange::ZeroToOne; pPerGpuState->maxPipelineStackSizes = {}; pPerGpuState->dynamicPipelineStackSize = 0; deviceIdx++; } while (deviceIdx < numPalDevices); } // ===================================================================================================================== // Resets all state except for the PAL command buffer state. This function is called both during vkBeginCommandBuffer // and during vkResetCommandBuffer void CmdBuffer::ResetState() { // Memset the first section of m_allGpuState. The second section begins with pipelineState. const size_t memsetBytes = offsetof(AllGpuRenderState, pipelineState); memset(&m_allGpuState, 0, memsetBytes); ResetPipelineState(); m_curDeviceMask = InvalidPalDeviceMask; // Reset local draw call counter m_perCmdBufDrawCallCounter = 0; m_perCmdBufDispatchCallCounter = 0; m_renderPassInstance.pExecuteInfo = nullptr; m_renderPassInstance.subpass = VK_SUBPASS_EXTERNAL; m_renderPassInstance.flags.u32All = 0; m_recordingResult = VK_SUCCESS; m_flags.hasConditionalRendering = false; m_debugPrintf.Reset(m_pDevice); if (m_allGpuState.pDescBufBinding != nullptr) { memset(m_allGpuState.pDescBufBinding, 0, sizeof(DescBufBinding)); } m_writtenFlippableImages.Clear(); } // ===================================================================================================================== // Reset Vulkan command buffer VkResult CmdBuffer::Reset(VkCommandBufferResetFlags flags) { VkResult result = VK_SUCCESS; bool releaseResources = ((flags & VK_COMMAND_BUFFER_RESET_RELEASE_RESOURCES_BIT) != 0); if (m_flags.disableResetReleaseResources) { releaseResources = false; } if (m_flags.wasBegun || releaseResources) { // If the command buffer is being recorded, the stack allocator will still be around. // Make sure to free it. if (m_flags.isRecording) { End(); VK_ASSERT(!m_flags.isRecording); } if (releaseResources) { ReleaseResources(); } #if VKI_RAY_TRACING FreeRayTracingScratchVidMemory(); m_cpsCmdBufferUtil.FreePatchCpsList(m_cbBeginDeviceMask); if (m_pBvhBatchState != nullptr) { // Called here (outside of the BvhBatchLayer because Reset can be triggered // either directly on the command buffer or across the whole command pool. m_pBvhBatchState->Log("Resetting via command buffer reset.\n"); m_pBvhBatchState->Reset(); } #endif result = PalToVkResult(PalCmdBufferReset(releaseResources)); m_flags.wasBegun = false; if ((result == VK_SUCCESS) && releaseResources) { // Notify the command pool that the command buffer is reset. m_pCmdPool->UnmarkCmdBufBegun(this); } } return result; } // ===================================================================================================================== void CmdBuffer::ConvertPipelineBindPoint( VkPipelineBindPoint pipelineBindPoint, Pal::PipelineBindPoint* pPalBindPoint, PipelineBindPoint* pApiBind) { switch (pipelineBindPoint) { case VK_PIPELINE_BIND_POINT_GRAPHICS: *pPalBindPoint = Pal::PipelineBindPoint::Graphics; *pApiBind = PipelineBindGraphics; break; case VK_PIPELINE_BIND_POINT_COMPUTE: *pPalBindPoint = Pal::PipelineBindPoint::Compute; *pApiBind = PipelineBindCompute; break; #if VKI_RAY_TRACING case VK_PIPELINE_BIND_POINT_RAY_TRACING_KHR: *pPalBindPoint = Pal::PipelineBindPoint::Compute; *pApiBind = PipelineBindRayTracing; break; #endif default: VK_NEVER_CALLED(); *pPalBindPoint = Pal::PipelineBindPoint::Compute; *pApiBind = PipelineBindCompute; } } // ===================================================================================================================== // Called to rebind a currently bound pipeline of the given type to PAL. Called from vkCmdBindPipeline() but also from // various other places when it has been necessary to defer the binding of the pipeline. // // This function will also reload user data if necessary because of the pipeline switch. template void CmdBuffer::RebindPipeline() { const UserDataLayout* pNewUserDataLayout = nullptr; RebindUserDataFlags rebindFlags = 0; Pal::PipelineBindPoint palBindPoint; if (bindPoint == PipelineBindCompute) { const ComputePipeline* pPipeline = m_allGpuState.pComputePipeline; VK_ASSERT(pPipeline != nullptr); const PhysicalDevice* pPhysicalDevice = m_pDevice->VkPhysicalDevice(DefaultDeviceIndex); if ((pPhysicalDevice->GetQueueFamilyPalQueueType(m_queueFamilyIndex) == Pal::QueueTypeCompute) && (m_asyncComputeQueueMaxWavesPerCu > 0)) { Pal::DynamicComputeShaderInfo dynamicInfo = {}; dynamicInfo.maxWavesPerCu = static_cast(m_asyncComputeQueueMaxWavesPerCu); pPipeline->BindToCmdBuffer(this, dynamicInfo); } else { pPipeline->BindToCmdBuffer(this, pPipeline->GetBindInfo()); } pNewUserDataLayout = pPipeline->GetUserDataLayout(); palBindPoint = Pal::PipelineBindPoint::Compute; } else if (bindPoint == PipelineBindGraphics) { const GraphicsPipeline* pPipeline = m_allGpuState.pGraphicsPipeline; VK_ASSERT(pPipeline != nullptr); pPipeline->BindToCmdBuffer(this); if (pPipeline->ContainsStaticState(DynamicStatesInternal::VertexInputBindingStride)) { UpdateVertexBufferStrides(pPipeline); } pNewUserDataLayout = pPipeline->GetUserDataLayout(); palBindPoint = Pal::PipelineBindPoint::Graphics; // Update dynamic vertex input state and check whether need rebind uber-fetch shader internal memory PipelineBindState* pBindState = &m_allGpuState.pipelineState[PipelineBindGraphics]; if (pPipeline->ContainsDynamicState(DynamicStatesInternal::VertexInput)) { if (pBindState->hasDynamicVertexInput == false) { if (pBindState->pVertexInputInternalData != nullptr) { rebindFlags |= RebindUberFetchInternalMem; } pBindState->hasDynamicVertexInput = true; } uint32_t newUberFetchShaderUserData = GetUberFetchShaderUserData(pNewUserDataLayout); if (GetUberFetchShaderUserData(&pBindState->userDataLayout) != newUberFetchShaderUserData) { SetUberFetchShaderUserData(&pBindState->userDataLayout, newUberFetchShaderUserData); if (pBindState->pVertexInputInternalData != nullptr) { rebindFlags |= RebindUberFetchInternalMem; } } } else { pBindState->hasDynamicVertexInput = false; } } #if VKI_RAY_TRACING else if (bindPoint == PipelineBindRayTracing) { const RayTracingPipeline* pPipeline = m_allGpuState.pRayTracingPipeline; VK_ASSERT(pPipeline != nullptr); const PhysicalDevice* pPhysicalDevice = m_pDevice->VkPhysicalDevice(DefaultDeviceIndex); if ((pPhysicalDevice->GetQueueFamilyPalQueueType(m_queueFamilyIndex) == Pal::QueueTypeCompute) && (m_asyncComputeQueueMaxWavesPerCu > 0)) { Pal::DynamicComputeShaderInfo dynamicInfo = {}; dynamicInfo.maxWavesPerCu = static_cast(m_asyncComputeQueueMaxWavesPerCu); pPipeline->BindToCmdBuffer(this, dynamicInfo); } else { pPipeline->BindToCmdBuffer(this, pPipeline->GetBindInfo()); } pNewUserDataLayout = pPipeline->GetUserDataLayout(); palBindPoint = Pal::PipelineBindPoint::Compute; } #endif else { VK_NEVER_CALLED(); } VK_ASSERT(pNewUserDataLayout != nullptr); // Push Constant user data layouts are scheme-agnostic, which will always be checked and rebound if // needed. // In compact scheme, the top-level user data layout of two compatible pipeline layout may be different. // Thus, pipeline layout needs to be checked and rebound if needed. // In indirect scheme, the top-level user data layout is always the same for all the pipeline layouts built // in this scheme. So user data doesn't require to be rebound in this case. // Pipeline layouts in different scheme can never be compatible. In this case, calling vkCmdBindDescriptorSets() // to rebind descirptor sets is mandatory for user. if ((pNewUserDataLayout->scheme == m_allGpuState.pipelineState[bindPoint].userDataLayout.scheme) && (pNewUserDataLayout->scheme == PipelineLayoutScheme::Compact)) { // Update the current owner of the compute PAL pipeline binding if we bound a pipeline if ((fromBindPipeline == false) && (palBindPoint == Pal::PipelineBindPoint::Compute)) { // If the ownership of the PAL binding is changing, the current user data belongs to the old binding and must // be reloaded. if (PalPipelineBindingOwnedBy(palBindPoint, bindPoint) == false) { rebindFlags |= RebindUserDataAll; } m_allGpuState.palToApiPipeline[size_t(Pal::PipelineBindPoint::Compute)] = bindPoint; } // Graphics pipeline owner should always remain fixed, so we don't have to worry about reloading // user data (for that reason) or ownership updates. VK_ASSERT(PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Graphics, PipelineBindGraphics)); // A user data layout switch may also require some user data to be reloaded (for both gfx and compute). rebindFlags |= SwitchCompactSchemeUserDataLayouts(bindPoint, pNewUserDataLayout); } rebindFlags |= SwitchCommonUserDataLayouts(bindPoint, pNewUserDataLayout); // Cache the new user data layout information m_allGpuState.pipelineState[bindPoint].userDataLayout = *pNewUserDataLayout; // Reprogram the user data if necessary if (rebindFlags != 0) { RebindUserData(bindPoint, palBindPoint, rebindFlags); } } // ===================================================================================================================== // Bind pipeline to command buffer void CmdBuffer::BindPipeline( VkPipelineBindPoint pipelineBindPoint, VkPipeline pipeline) { DbgBarrierPreCmd(DbgBarrierBindPipeline); const Pipeline* pPipeline = Pipeline::BaseObjectFromHandle(pipeline); if (pPipeline != nullptr) { #if VKI_RAY_TRACING m_flags.hasRayTracing |= pPipeline->HasRayTracing(); #endif switch (pipelineBindPoint) { case VK_PIPELINE_BIND_POINT_COMPUTE: { m_allGpuState.pComputePipeline = static_cast(pPipeline); if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute)) { // Defer the binding by invalidating the current PAL compute binding point. This is because we // don't know what compute-based binding will be utilized until we see the work command. m_allGpuState.palToApiPipeline[size_t(Pal::PipelineBindPoint::Compute)] = PipelineBindCount; } break; } case VK_PIPELINE_BIND_POINT_GRAPHICS: { m_allGpuState.pGraphicsPipeline = static_cast(pPipeline); // Can bind the graphics pipeline immediately since only API graphics pipelines use the PAL // graphics pipeline. Note that wave limits may still defer the bind inside RebindPipeline(). VK_ASSERT(PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Graphics, PipelineBindGraphics)); RebindPipeline(); break; } #if VKI_RAY_TRACING case VK_PIPELINE_BIND_POINT_RAY_TRACING_KHR: { m_allGpuState.pRayTracingPipeline = static_cast(pPipeline); if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindRayTracing)) { // Defer the binding by invalidating the current PAL compute binding point. This is because we // don't know what compute-based binding will be utilized until we see the work command. m_allGpuState.palToApiPipeline[size_t(Pal::PipelineBindPoint::Compute)] = PipelineBindCount; } break; } #endif default: VK_NEVER_CALLED(); break; } } DbgBarrierPostCmd(DbgBarrierBindPipeline); } // ===================================================================================================================== // Called during vkCmdBindPipeline when the new pipeline's layout might be different from the previously bound layout. // This function will compare the compatibility of those layouts in compact scheme and reprogram any user data to // maintain previously-written pipeline resources to make them available in the correct locations of the new pipeline // layout. Those that are compatible with the new layout remain correctly bound. CmdBuffer::RebindUserDataFlags CmdBuffer::SwitchCompactSchemeUserDataLayouts( PipelineBindPoint apiBindPoint, const UserDataLayout* pNewUserDataLayout) { VK_ASSERT(pNewUserDataLayout != nullptr); VK_ASSERT(pNewUserDataLayout->scheme == PipelineLayoutScheme::Compact); VK_ASSERT(m_allGpuState.pipelineState[apiBindPoint].userDataLayout.scheme == PipelineLayoutScheme::Compact); PipelineBindState* pBindState = &m_allGpuState.pipelineState[apiBindPoint]; RebindUserDataFlags flags = 0; const auto& newUserDataLayout = pNewUserDataLayout->compact; const auto& curUserDataLayout = pBindState->userDataLayout.compact; // Rebind descriptor set bindings if necessary if ((newUserDataLayout.setBindingRegBase != curUserDataLayout.setBindingRegBase) | (newUserDataLayout.setBindingRegCount != curUserDataLayout.setBindingRegCount)) { flags |= RebindUserDataDescriptorSets; } return flags; } // ===================================================================================================================== // Called during vkCmdBindPipeline when the new pipeline's layout might be different from the previously bound layout. // This function will compare the compatibility of those scheme-agnostic layouts and reprogram any user data to maintain // previously-written pipeline resources to make them available in the correct locations of the new pipeline layout. // Those that are compatible with the new layout remain correctly bound. CmdBuffer::RebindUserDataFlags CmdBuffer::SwitchCommonUserDataLayouts( PipelineBindPoint apiBindPoint, const UserDataLayout* pNewUserDataLayout) { VK_ASSERT(pNewUserDataLayout != nullptr); PipelineBindState* pBindState = &m_allGpuState.pipelineState[apiBindPoint]; RebindUserDataFlags flags = 0; const auto& newUserDataLayout = pNewUserDataLayout->common; const auto& curUserDataLayout = pBindState->userDataLayout.common; // Rebind push constants if necessary if (((newUserDataLayout.pushConstRegBase != curUserDataLayout.pushConstRegBase) | (newUserDataLayout.pushConstRegCount != curUserDataLayout.pushConstRegCount)) ) { flags |= RebindUserDataPushConstants; } return flags; } // ===================================================================================================================== // Called during vkCmdBindPipeline when something requires rebinding API-provided top-level user data (descriptor // sets, push constants, etc.) void CmdBuffer::RebindUserData( PipelineBindPoint apiBindPoint, Pal::PipelineBindPoint palBindPoint, RebindUserDataFlags flags) { VK_ASSERT(flags != 0); const PipelineBindState& bindState = m_allGpuState.pipelineState[apiBindPoint]; const auto& compactuserDataLayout = bindState.userDataLayout.compact; const auto& commonUserDataLayout = bindState.userDataLayout.common; if ((flags & RebindUserDataDescriptorSets) != 0) { VK_ASSERT(bindState.userDataLayout.scheme == PipelineLayoutScheme::Compact); const uint32_t count = Util::Min(compactuserDataLayout.setBindingRegCount, bindState.boundSetCount); if (count > 0) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetUserData( palBindPoint, compactuserDataLayout.setBindingRegBase, count, PerGpuState(deviceIdx)->setBindingData[apiBindPoint]); } while (deviceGroup.IterateNext()); } } if ((flags & RebindUserDataPushConstants) != 0) { const uint32_t count = Util::Min(commonUserDataLayout.pushConstRegCount, bindState.pushedConstCount); if (count > 0) { // perDeviceStride is zero here because push constant data is replicated for all devices. // Note: There might be interesting use cases where don't want to clone this data. const uint32_t perDeviceStride = 0; PalCmdBufferSetUserData( palBindPoint, commonUserDataLayout.pushConstRegBase, count, perDeviceStride, bindState.pushConstData); } } if (((flags & RebindUberFetchInternalMem) != 0) && (bindState.pVertexInputInternalData != nullptr)) { VK_ASSERT(bindState.userDataLayout.scheme == PipelineLayoutScheme::Compact); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetUserData( palBindPoint, commonUserDataLayout.uberFetchConstBufRegBase, 2, reinterpret_cast(&bindState.pVertexInputInternalData->gpuAddress[deviceIdx])); } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== // Insert secondary command buffers into a primary command buffer void CmdBuffer::ExecuteCommands( uint32_t cmdBufferCount, const VkCommandBuffer* pCmdBuffers) { DbgBarrierPreCmd(DbgBarrierExecuteCommands); for (uint32_t i = 0; i < cmdBufferCount; i++) { CmdBuffer* pInteralCmdBuf = ApiCmdBuffer::ObjectFromHandle(pCmdBuffers[i]); // Increment per command list draw call counter m_perCmdBufDrawCallCounter += pInteralCmdBuf->GetDrawCallCount(); m_perCmdBufDispatchCallCounter += pInteralCmdBuf->GetDispatchCallCount(); #if VKI_RAY_TRACING m_flags.hasRayTracing |= pInteralCmdBuf->HasRayTracing(); #endif utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); Pal::ICmdBuffer* pPalNestedCmdBuffer = pInteralCmdBuf->PalCmdBuffer(deviceIdx); PalCmdBuffer(deviceIdx)->CmdExecuteNestedCmdBuffers(1, &pPalNestedCmdBuffer); } while (deviceGroup.IterateNext()); } // Executing secondary command buffer will clear the states of Graphic Pipeline // in that case they cannot be used after ends of execution secondary command buffer ResetPipelineState(); DbgBarrierPostCmd(DbgBarrierExecuteCommands); } // ===================================================================================================================== // Destroy a command buffer object VkResult CmdBuffer::Destroy(void) { Instance* const pInstance = m_pDevice->VkInstance(); for (uint32 i = 0; i < PipelineBindCount; ++i) { pInstance->FreeMem(m_allGpuState.pipelineState[i].pPushDescriptorSetMemory); } if (m_pSqttState != nullptr) { Util::Destructor(m_pSqttState); pInstance->FreeMem(m_pSqttState); } if (m_pTransformFeedbackState != nullptr) { pInstance->FreeMem(m_pTransformFeedbackState); } if (m_pUberFetchShaderTempBuffer != nullptr) { pInstance->FreeMem(m_pUberFetchShaderTempBuffer); } // Unregister this command buffer from the pool m_pCmdPool->UnregisterCmdBuffer(this); for (uint32_t deviceIdx = 0; deviceIdx < m_pDevice->NumPalDevices(); ++deviceIdx) { if (m_pBackupPalCmdBuffers[deviceIdx] != nullptr) { m_pBackupPalCmdBuffers[deviceIdx]->Destroy(); m_pCmdPool->GetCmdPoolAllocator()->pfnFree( m_pCmdPool->GetCmdPoolAllocator()->pUserData, m_pBackupPalCmdBuffers[deviceIdx]); } } PalCmdBufferDestroy(); ReleaseResources(); #if VKI_RAY_TRACING FreeRayTracingScratchVidMemory(); m_cpsCmdBufferUtil.FreePatchCpsList(m_pDevice->GetPalDeviceMask()); if (m_pBvhBatchState != nullptr) { // Called here (outside of the BvhBatchLayer because Destroy can be triggered // either directly on the command buffer or across the whole command pool. m_pBvhBatchState->Log("Resetting via command buffer destroy.\n"); m_pBvhBatchState->Reset(); } #endif m_debugPrintf.Reset(m_pDevice); Util::Destructor(this); m_pDevice->FreeApiObject(m_pCmdPool->GetCmdPoolAllocator(), ApiCmdBuffer::FromObject(this)); return VK_SUCCESS; } // ===================================================================================================================== void CmdBuffer::ReleaseResources() { auto pInstance = m_pDevice->VkInstance(); RenderStateCache* pRSCache = m_pDevice->GetRenderStateCache(); for (uint32_t i = 0; i < m_palDepthStencilState.NumElements(); ++i) { pRSCache->DestroyDepthStencilState( m_palDepthStencilState.At(i).pPalDepthStencil, pInstance->GetAllocCallbacks()); } m_palDepthStencilState.Clear(); for (uint32_t i = 0; i < m_palColorBlendState.NumElements(); ++i) { pRSCache->DestroyColorBlendState(m_palColorBlendState.At(i).pPalColorBlend, pInstance->GetAllocCallbacks()); } m_palColorBlendState.Clear(); for (uint32_t i = 0; i < m_palMsaaState.NumElements(); ++i) { pRSCache->DestroyMsaaState(m_palMsaaState.At(i).pPalMsaa, pInstance->GetAllocCallbacks()); } m_palMsaaState.Clear(); // Release per-attachment render pass instance memory if (m_renderPassInstance.pAttachments != nullptr) { pInstance->FreeMem(m_renderPassInstance.pAttachments); m_renderPassInstance.pAttachments = nullptr; m_renderPassInstance.maxAttachmentCount = 0; } // Release per-subpass instance memory if (m_renderPassInstance.pSamplePatterns != nullptr) { pInstance->FreeMem(m_renderPassInstance.pSamplePatterns); m_renderPassInstance.pSamplePatterns = nullptr; m_renderPassInstance.maxSubpassCount = 0; } if (m_pStackAllocator != nullptr) { pInstance->StackMgr()->ReleaseAllocator(m_pStackAllocator); m_pStackAllocator = nullptr; } } // ===================================================================================================================== template void CmdBuffer::BindDescriptorSets( VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t firstSet, uint32_t setCount, const VkDescriptorSet* pDescriptorSets, uint32_t dynamicOffsetCount, const uint32_t* pDynamicOffsets) { DbgBarrierPreCmd(DbgBarrierBindSetsPushConstants); if (setCount > 0) { Pal::PipelineBindPoint palBindPoint; PipelineBindPoint apiBindPoint; ConvertPipelineBindPoint(pipelineBindPoint, &palBindPoint, &apiBindPoint); const PipelineLayout* pLayout = PipelineLayout::ObjectFromHandle(layout); // Get user data register information from the given pipeline layout const PipelineLayout::Info& layoutInfo = pLayout->GetInfo(); // Update descriptor set binding data shadow. VK_ASSERT((firstSet + setCount) <= layoutInfo.setCount); for (uint32_t i = 0; i < setCount; ++i) { if (pDescriptorSets[i] != VK_NULL_HANDLE) { // Compute set binding point index const uint32_t setBindIdx = firstSet + i; // User data information for this set const PipelineLayout::SetUserDataLayout& setLayoutInfo = pLayout->GetSetUserData(setBindIdx); const Image* pImage = DescriptorSet::ObjectFromHandle(pDescriptorSets[i])->GetWrittenFlippableImage(); if (pImage != nullptr) { RegisterWriteToFlippableImage(pImage); } // If this descriptor set has any dynamic descriptor data then write them into the shadow. if (setLayoutInfo.dynDescCount > 0) { // NOTE: We supply patched SRDs directly in used data registers. utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); DescriptorSet::PatchedDynamicDataFromHandle( pDescriptorSets[i], deviceIdx, &(PerGpuState(deviceIdx)-> setBindingData[apiBindPoint][setLayoutInfo.dynDescDataRegOffset]), pDynamicOffsets, setLayoutInfo.dynDescCount, useCompactDescriptor); } while (deviceGroup.IterateNext()); // Skip over the already consumed dynamic offsets. pDynamicOffsets += setLayoutInfo.dynDescCount; } // If this descriptor set needs a set pointer, then write it to the shadow. if (setLayoutInfo.setPtrRegOffset != PipelineLayout::InvalidReg) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); DescriptorSet::UserDataPtrValueFromHandle( pDescriptorSets[i], deviceIdx, &(PerGpuState(deviceIdx)-> setBindingData[apiBindPoint][setLayoutInfo.setPtrRegOffset])); } while (deviceGroup.IterateNext()); } } } SetUserDataPipelineLayout(firstSet, setCount, pLayout, palBindPoint, apiBindPoint); } DbgBarrierPostCmd(DbgBarrierBindSetsPushConstants); } // ===================================================================================================================== void CmdBuffer::BindDescriptorSetsBuffers( VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t firstSet, uint32_t setCount, const DescriptorBuffers* pDescriptorBuffers) { DbgBarrierPreCmd(DbgBarrierBindSetsPushConstants); if (setCount > 0) { Pal::PipelineBindPoint palBindPoint; PipelineBindPoint apiBindPoint; ConvertPipelineBindPoint(pipelineBindPoint, &palBindPoint, &apiBindPoint); const PipelineLayout* pLayout = PipelineLayout::ObjectFromHandle(layout); // Get user data register information from the given pipeline layout const PipelineLayout::Info& layoutInfo = pLayout->GetInfo(); // Update descriptor set binding data shadow. VK_ASSERT((firstSet + setCount) <= layoutInfo.setCount); for (uint32_t i = 0; i < setCount; ++i) { // Compute set binding point index const uint32_t setBindIdx = firstSet + i; // User data information for this set const PipelineLayout::SetUserDataLayout& setLayoutInfo = pLayout->GetSetUserData(setBindIdx); // If this descriptor set needs a set pointer, then write it to the shadow. if (setLayoutInfo.setPtrRegOffset != PipelineLayout::InvalidReg) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); const DescBufBinding& bufBinding = *m_allGpuState.pDescBufBinding; PerGpuRenderState* pPerGpuState = PerGpuState(deviceIdx); Pal::gpusize bufferAddress = bufBinding.baseAddr[pDescriptorBuffers[setBindIdx].baseAddrNdx]; Pal::gpusize offset = pDescriptorBuffers[setBindIdx].offset; pPerGpuState->setBindingData[apiBindPoint][setLayoutInfo.setPtrRegOffset] = static_cast((bufferAddress + offset) & 0xFFFFFFFFull); } while (deviceGroup.IterateNext()); } } SetUserDataPipelineLayout(firstSet, setCount, pLayout, palBindPoint, apiBindPoint); } DbgBarrierPostCmd(DbgBarrierBindSetsPushConstants); } // ===================================================================================================================== void CmdBuffer::SetUserDataPipelineLayout( uint32_t firstSet, uint32_t setCount, const PipelineLayout* pLayout, const Pal::PipelineBindPoint palBindPoint, const PipelineBindPoint apiBindPoint) { VK_ASSERT(setCount > 0); // Get user data register information from the given pipeline layout const PipelineLayout::Info& layoutInfo = pLayout->GetInfo(); if (pLayout->GetScheme() == PipelineLayoutScheme::Compact) { // Get the current binding state in the command buffer PipelineBindState* pBindState = &m_allGpuState.pipelineState[apiBindPoint]; // Figure out the total range of user data registers written by this sequence of descriptor set binds const PipelineLayout::SetUserDataLayout& firstSetLayout = pLayout->GetSetUserData(firstSet); const PipelineLayout::SetUserDataLayout& lastSetLayout = pLayout->GetSetUserData(firstSet + setCount - 1); const uint32_t rangeOffsetBegin = firstSetLayout.firstRegOffset; const uint32_t rangeOffsetEnd = lastSetLayout.firstRegOffset + lastSetLayout.totalRegCount; // Update the high watermark of number of user data entries written for currently bound descriptor sets and // their dynamic offsets in the current command buffer state. pBindState->boundSetCount = Util::Max(pBindState->boundSetCount, rangeOffsetEnd); // Descriptor set with zero resource binding is allowed in spec, so we need to check this and only proceed // when there are at least 1 user data to update. const uint32_t rangeRegCount = rangeOffsetEnd - rangeOffsetBegin; if (rangeRegCount > 0) { // Program the user data register only if the current user data layout base matches that of the given // layout. Otherwise, what's happening is that the application is binding descriptor sets for a future // pipeline layout (e.g. at the top of the command buffer) and this register write will be redundant. // A future vkCmdBindPipeline will reprogram the user data register. if (PalPipelineBindingOwnedBy(palBindPoint, apiBindPoint) && (pBindState->userDataLayout.compact.setBindingRegBase == layoutInfo.userDataLayout.compact.setBindingRegBase)) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetUserData( palBindPoint, pBindState->userDataLayout.compact.setBindingRegBase + rangeOffsetBegin, rangeRegCount, &(PerGpuState(deviceIdx)->setBindingData[apiBindPoint][rangeOffsetBegin])); } while (deviceGroup.IterateNext()); } } } else if (pLayout->GetScheme() == PipelineLayoutScheme::Indirect) { const auto& userDataLayout = layoutInfo.userDataLayout.indirect; for (uint32_t setIdx = firstSet; setIdx < firstSet + setCount; ++setIdx) { const PipelineLayout::SetUserDataLayout& setLayoutInfo = pLayout->GetSetUserData(setIdx); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); if (setLayoutInfo.dynDescCount > 0) { const uint32_t dynBufferSizeDw = setLayoutInfo.dynDescCount * DescriptorSetLayout::GetDynamicBufferDescDwSize(m_pDevice); Pal::gpusize gpuAddr; void* pCpuAddr = PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData( dynBufferSizeDw, m_pDevice->GetProperties().descriptorSizes.alignmentInDwords, &gpuAddr); const uint32_t gpuAddrLow = static_cast(gpuAddr); memcpy(pCpuAddr, &(PerGpuState(deviceIdx)->setBindingData[apiBindPoint][setLayoutInfo.dynDescDataRegOffset]), dynBufferSizeDw * sizeof(uint32_t)); PalCmdBuffer(deviceIdx)->CmdSetUserData( palBindPoint, userDataLayout.setBindingPtrRegBase + 2 * setIdx * PipelineLayout::SetPtrRegCount, PipelineLayout::SetPtrRegCount, &gpuAddrLow); } if (setLayoutInfo.setPtrRegOffset != PipelineLayout::InvalidReg) { PalCmdBuffer(deviceIdx)->CmdSetUserData( palBindPoint, userDataLayout.setBindingPtrRegBase + (2 * setIdx + 1) * PipelineLayout::SetPtrRegCount, PipelineLayout::SetPtrRegCount, &(PerGpuState(deviceIdx)->setBindingData[apiBindPoint][setLayoutInfo.setPtrRegOffset])); } } while (deviceGroup.IterateNext()); } } else { VK_NEVER_CALLED(); } } // ===================================================================================================================== template VKAPI_ATTR void VKAPI_CALL CmdBuffer::CmdBindDescriptorSets( VkCommandBuffer cmdBuffer, VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t firstSet, uint32_t descriptorSetCount, const VkDescriptorSet* pDescriptorSets, uint32_t dynamicOffsetCount, const uint32_t* pDynamicOffsets) { ApiCmdBuffer::ObjectFromHandle(cmdBuffer)->BindDescriptorSets( pipelineBindPoint, layout, firstSet, descriptorSetCount, pDescriptorSets, dynamicOffsetCount, pDynamicOffsets); } // ===================================================================================================================== PFN_vkCmdBindDescriptorSets CmdBuffer::GetCmdBindDescriptorSetsFunc( const Device* pDevice) { PFN_vkCmdBindDescriptorSets pFunc = nullptr; switch (pDevice->NumPalDevices()) { case 1: pFunc = GetCmdBindDescriptorSetsFunc<1>(pDevice); break; #if (VKI_BUILD_MAX_NUM_GPUS > 1) case 2: pFunc = GetCmdBindDescriptorSetsFunc<2>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 2) case 3: pFunc = GetCmdBindDescriptorSetsFunc<3>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 3) case 4: pFunc = GetCmdBindDescriptorSetsFunc<4>(pDevice); break; #endif default: pFunc = nullptr; VK_NEVER_CALLED(); break; } return pFunc; } // ===================================================================================================================== template PFN_vkCmdBindDescriptorSets CmdBuffer::GetCmdBindDescriptorSetsFunc( const Device* pDevice) { PFN_vkCmdBindDescriptorSets pFunc = nullptr; if (pDevice->UseCompactDynamicDescriptors()) { pFunc = CmdBindDescriptorSets; } else { pFunc = CmdBindDescriptorSets; } return pFunc; } // ===================================================================================================================== template VKAPI_ATTR void VKAPI_CALL CmdBuffer::CmdPushDescriptorSet( VkCommandBuffer commandBuffer, VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t set, uint32_t descriptorWriteCount, const VkWriteDescriptorSet* pDescriptorWrites) { CmdBuffer* pCmdBuffer = ApiCmdBuffer::ObjectFromHandle(commandBuffer); pCmdBuffer->PushDescriptorSet ( pipelineBindPoint, layout, set, descriptorWriteCount, pDescriptorWrites); } // ===================================================================================================================== template PFN_vkCmdPushDescriptorSet CmdBuffer::GetCmdPushDescriptorSetFunc( const Device* pDevice) { const size_t imageDescSize = pDevice->GetProperties().descriptorSizes.imageView; const size_t samplerDescSize = pDevice->GetProperties().descriptorSizes.sampler; const size_t typedBufferDescSize = pDevice->GetProperties().descriptorSizes.typedBufferView; const size_t untypedBufferDescSize = pDevice->GetProperties().descriptorSizes.untypedBufferView; PFN_vkCmdPushDescriptorSet pFunc = nullptr; if ((imageDescSize == 32) && (samplerDescSize == 16) && (typedBufferDescSize == 16) && (untypedBufferDescSize == 16)) { pFunc = &CmdPushDescriptorSet< 32, 16, 16, 16, numPalDevices>; } else if ((imageDescSize == 32) && (samplerDescSize == 16) && (typedBufferDescSize == 24) && (untypedBufferDescSize == 16)) { pFunc = &CmdPushDescriptorSet< 32, 16, 24, 16, numPalDevices>; } else { VK_NEVER_CALLED(); } return pFunc; } // ===================================================================================================================== PFN_vkCmdPushDescriptorSet CmdBuffer::GetCmdPushDescriptorSetFunc( const Device* pDevice) { PFN_vkCmdPushDescriptorSet pFunc = nullptr; switch (pDevice->NumPalDevices()) { case 1: pFunc = GetCmdPushDescriptorSetFunc<1>(pDevice); break; #if (VKI_BUILD_MAX_NUM_GPUS > 1) case 2: pFunc = GetCmdPushDescriptorSetFunc<2>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 2) case 3: pFunc = GetCmdPushDescriptorSetFunc<3>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 3) case 4: pFunc = GetCmdPushDescriptorSetFunc<4>(pDevice); break; #endif default: VK_NEVER_CALLED(); break; } return pFunc; } // ===================================================================================================================== template VKAPI_ATTR void VKAPI_CALL CmdBuffer::CmdPushDescriptorSetWithTemplate( VkCommandBuffer commandBuffer, VkDescriptorUpdateTemplate descriptorUpdateTemplate, VkPipelineLayout layout, uint32_t set, const void* pData) { CmdBuffer* pCmdBuffer = ApiCmdBuffer::ObjectFromHandle(commandBuffer); pCmdBuffer->PushDescriptorSetWithTemplate( descriptorUpdateTemplate, layout, set, pData); } // ===================================================================================================================== PFN_vkCmdPushDescriptorSetWithTemplate CmdBuffer::GetCmdPushDescriptorSetWithTemplateFunc( const Device* pDevice) { PFN_vkCmdPushDescriptorSetWithTemplate pFunc = nullptr; switch (pDevice->NumPalDevices()) { case 1: pFunc = CmdPushDescriptorSetWithTemplate<1>; break; #if (VKI_BUILD_MAX_NUM_GPUS > 1) case 2: pFunc = CmdPushDescriptorSetWithTemplate<2>; break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 2) case 3: pFunc = CmdPushDescriptorSetWithTemplate<3>; break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 3) case 4: pFunc = CmdPushDescriptorSetWithTemplate<4>; break; #endif default: VK_NEVER_CALLED(); break; } return pFunc; } // ===================================================================================================================== void CmdBuffer::BindIndexBuffer( VkBuffer buffer, VkDeviceSize offset, VkDeviceSize size, VkIndexType indexType) { DbgBarrierPreCmd(DbgBarrierBindIndexVertexBuffer); const Pal::IndexType palIndexType = VkToPalIndexType(indexType); Buffer* pBuffer = Buffer::ObjectFromHandle(buffer); if (pBuffer != NULL) { PalCmdBindIndexData(pBuffer, offset, palIndexType, size); } else { PalCmdUnbindIndexData(palIndexType); } DbgBarrierPostCmd(DbgBarrierBindIndexVertexBuffer); } // ===================================================================================================================== // A helper to set the per-device vertex buffer binding table to an empty descriptor. void CmdBuffer::ClearVertexBufferBindings( uint32_t watermark) { for (uint32_t deviceIdx = 0; deviceIdx < m_numPalDevices; deviceIdx++) { std::fill_n( &PerGpuState(deviceIdx)->vbBindings[0], watermark, EmptyVertexBufferBinding); #if VKI_BUILD_GFX12 const Pal::CompressionMode compressionMode = m_pDevice->GetBufferViewCompressionMode(); if (compressionMode != Pal::CompressionMode::Default) { for (uint32_t i = 0; i < watermark; i++) { Pal::BufferViewInfo* const pBinding = &PerGpuState(deviceIdx)->vbBindings[i]; pBinding->compressionMode = compressionMode; } } #endif } } // ===================================================================================================================== // Initializes VB binding manager state. Should be called when the command buffer is being initialized. void CmdBuffer::InitializeVertexBuffer() { ClearVertexBufferBindings(Pal::MaxVertexBuffers); m_vbWatermark = 0; } // ===================================================================================================================== // Called to reset the state of the VB manager because the parent command buffer is being reset. void CmdBuffer::ResetVertexBuffer() { ClearVertexBufferBindings(m_vbWatermark); m_vbWatermark = 0; m_uberFetchShaderInternalDataMap.Reset(); } // ===================================================================================================================== // A part of vkCmdBindVertexBuffers implementation. void CmdBuffer::BindVertexBuffersUpdateBindingRange( uint32_t deviceIdx, Pal::BufferViewInfo* pBinding, Pal::BufferViewInfo* pEndBinding, uint32_t inputIdx, const VkBuffer* pBuffers, const VkDeviceSize* pOffsets, const VkDeviceSize* pSizes, const VkDeviceSize* pStrides) { while (pBinding != pEndBinding) { const VkBuffer buffer = pBuffers[inputIdx]; const VkDeviceSize offset = pOffsets[inputIdx]; bool padVertexBuffers = m_flags.padVertexBuffers; if (buffer != VK_NULL_HANDLE) { const Buffer* pBuffer = Buffer::ObjectFromHandle(buffer); pBinding->gpuAddr = pBuffer->GpuVirtAddr(deviceIdx) + offset; if ((pSizes != nullptr) && (pSizes[inputIdx] != VK_WHOLE_SIZE)) { pBinding->range = pSizes[inputIdx]; if ((offset != 0) && (m_flags.offsetMode == false)) { padVertexBuffers = true; } } else { pBinding->range = pBuffer->GetSize() - offset; } } else { pBinding->gpuAddr = 0; pBinding->range = 0; } if (pStrides != nullptr) { pBinding->stride = pStrides[inputIdx]; } if (padVertexBuffers && (pBinding->stride != 0)) { pBinding->range = Util::RoundUpToMultiple(pBinding->range, pBinding->stride); } #if VKI_BUILD_GFX12 const Pal::CompressionMode compressionMode = m_pDevice->GetBufferViewCompressionMode(); if (compressionMode != Pal::CompressionMode::Default) { pBinding->compressionMode = compressionMode; } #endif inputIdx++; pBinding++; } } // ===================================================================================================================== // Detects writes to flippable images and protects them during submission if necessary void CmdBuffer::RegisterWriteToFlippableImage( const Image* pImage) { if (m_flags.protectFlippableImages && pImage->IsFlippable()) { VK_ASSERT(pImage->PalMemory(DefaultDeviceIndex) != nullptr); bool imageExists = false; for (uint32_t i = 0; i < m_writtenFlippableImages.NumElements(); ++i) { if (m_writtenFlippableImages[i] == pImage) { imageExists = true; break; } } if (imageExists == false) { VK_ALERT_MSG(m_writtenFlippableImages.NumElements() > 0, "Command buffer writes to more than one swapchain image, this is unusual"); m_writtenFlippableImages.PushBack(pImage); } } } // ===================================================================================================================== // Implementation of vkCmdBindVertexBuffers void CmdBuffer::BindVertexBuffers( uint32_t firstBinding, uint32_t bindingCount, const VkBuffer* pBuffers, const VkDeviceSize* pOffsets, const VkDeviceSize* pSizes, const VkDeviceSize* pStrides) { if (bindingCount > 0) { VK_ASSERT((firstBinding + bindingCount) <= VK_ARRAY_SIZE(PerGpuRenderState::vbBindings)); DbgBarrierPreCmd(DbgBarrierBindIndexVertexBuffer); utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); Pal::BufferViewInfo* const pBinding = &PerGpuState(deviceIdx)->vbBindings[firstBinding]; BindVertexBuffersUpdateBindingRange( deviceIdx, pBinding, pBinding + bindingCount, 0, pBuffers, pOffsets, pSizes, pStrides); if (m_flags.offsetMode) { Pal::VertexBufferView vertexViews[Pal::MaxVertexBuffers] = {}; for (uint32_t idx = 0; idx < bindingCount; idx++) { vertexViews[idx].gpuva = pBinding[idx].gpuAddr; vertexViews[idx].sizeInBytes = pBinding[idx].range; vertexViews[idx].strideInBytes = pBinding[idx].stride; } const Pal::VertexBufferViews bufferViews = { .firstBuffer = firstBinding, .bufferCount = bindingCount, .offsetMode = true, .pVertexBufferViews = vertexViews }; PalCmdBuffer(deviceIdx)->CmdSetVertexBuffers(bufferViews); } else { const Pal::VertexBufferViews bufferViews = { .firstBuffer = firstBinding, .bufferCount = bindingCount, .offsetMode = false, .pBufferViewInfos = pBinding }; PalCmdBuffer(deviceIdx)->CmdSetVertexBuffers(bufferViews); } } while (deviceGroup.IterateNext()); m_vbWatermark = Util::Max(m_vbWatermark, firstBinding + bindingCount); DbgBarrierPostCmd(DbgBarrierBindIndexVertexBuffer); } } // ===================================================================================================================== void CmdBuffer::UpdateVertexBufferStrides( const GraphicsPipeline* pPipeline) { VK_ASSERT(pPipeline != nullptr); // Update strides for each binding used by the graphics pipeline. Rebuild SRD data for those bindings // whose strides changed. const bool padVertexBuffers = m_flags.padVertexBuffers; utils::IterateMask deviceGroup(GetDeviceMask()); do { const VbBindingInfo& bindingInfo = pPipeline->GetVbBindingInfo(); uint32 deviceIdx = deviceGroup.Index(); uint32 firstChanged = UINT_MAX; uint32 lastChanged = 0; uint32 count = bindingInfo.bindingCount; Pal::BufferViewInfo* pVbBindings = PerGpuState(deviceIdx)->vbBindings; for (uint32 bindex = 0; bindex < count; ++bindex) { uint32 slot = bindingInfo.bindings[bindex].slot; uint32 byteStride = bindingInfo.bindings[bindex].byteStride; Pal::BufferViewInfo* pBinding = &pVbBindings[slot]; if (pBinding->stride != byteStride) { pBinding->stride = byteStride; if (pBinding->gpuAddr != 0) { firstChanged = Util::Min(firstChanged, slot); lastChanged = Util::Max(lastChanged, slot); } if (padVertexBuffers && (pBinding->stride != 0)) { pBinding->range = Util::RoundUpToMultiple(pBinding->range, pBinding->stride); } } } if (firstChanged <= lastChanged) { auto pBinding = &PerGpuState(deviceIdx)->vbBindings[firstChanged]; if (m_flags.offsetMode) { Pal::VertexBufferView vertexViews[Pal::MaxVertexBuffers] = {}; for (uint32_t idx = 0; idx < (lastChanged - firstChanged + 1); idx++) { vertexViews[idx].gpuva = pBinding[idx].gpuAddr; vertexViews[idx].sizeInBytes = pBinding[idx].range; vertexViews[idx].strideInBytes = pBinding[idx].stride; } const Pal::VertexBufferViews bufferViews = { .firstBuffer = firstChanged, .bufferCount = (lastChanged - firstChanged) + 1, .offsetMode = true, .pVertexBufferViews = vertexViews }; PalCmdBuffer(deviceIdx)->CmdSetVertexBuffers(bufferViews); } else { const Pal::VertexBufferViews bufferViews = { .firstBuffer = firstChanged, .bufferCount = (lastChanged - firstChanged) + 1, .offsetMode = false, .pBufferViewInfos = pBinding }; PalCmdBuffer(deviceIdx)->CmdSetVertexBuffers(bufferViews); } } } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::Draw( uint32_t firstVertex, uint32_t vertexCount, uint32_t firstInstance, uint32_t instanceCount) { DbgBarrierPreCmd(DbgBarrierDrawNonIndexed); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif { PalCmdDraw(firstVertex, vertexCount, firstInstance, instanceCount, 0u); } DbgBarrierPostCmd(DbgBarrierDrawNonIndexed); } // ===================================================================================================================== void CmdBuffer::DrawIndexed( uint32_t firstIndex, uint32_t indexCount, int32_t vertexOffset, uint32_t firstInstance, uint32_t instanceCount) { DbgBarrierPreCmd(DbgBarrierDrawIndexed); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif { PalCmdDrawIndexed(firstIndex, indexCount, vertexOffset, firstInstance, instanceCount, 0u); } DbgBarrierPostCmd(DbgBarrierDrawIndexed); } // ===================================================================================================================== template void CmdBuffer::DrawIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset) { DbgBarrierPreCmd((indexed ? DbgBarrierDrawIndexed : DbgBarrierDrawNonIndexed) | DbgBarrierDrawIndirect); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif Buffer* pBuffer = Buffer::ObjectFromHandle(buffer); if ((stride + offset) <= pBuffer->PalMemory(DefaultDeviceIndex)->Desc().size) { Pal::gpusize countVirtAddr = 0; utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); Pal::GpuVirtAddrAndStride gpuVirtAddrAndStride = { pBuffer->GpuVirtAddr(deviceIdx) + static_cast(offset), {stride}, }; if (useBufferCount) { Buffer* pCountBuffer = Buffer::ObjectFromHandle(countBuffer); countVirtAddr = pCountBuffer->GpuVirtAddr(deviceIdx) + countOffset; } if (indexed == false) { PalCmdBuffer(deviceIdx)->CmdDrawIndirectMulti( gpuVirtAddrAndStride, count, countVirtAddr); } else { PalCmdBuffer(deviceIdx)->CmdDrawIndexedIndirectMulti( gpuVirtAddrAndStride, count, countVirtAddr); } } while (deviceGroup.IterateNext()); } DbgBarrierPostCmd((indexed ? DbgBarrierDrawIndexed : DbgBarrierDrawNonIndexed) | DbgBarrierDrawIndirect); } // ===================================================================================================================== template void CmdBuffer::DrawIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa) { DbgBarrierPreCmd((indexed ? DbgBarrierDrawIndexed : DbgBarrierDrawNonIndexed) | DbgBarrierDrawIndirect); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif VK_ASSERT(stride <= indirectBufferSize); Pal::GpuVirtAddrAndStride gpuVirtAddrAndStride = { indirectBufferVa, {stride} }; utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); if (indexed == false) { PalCmdBuffer(deviceIdx)->CmdDrawIndirectMulti( gpuVirtAddrAndStride, count, useBufferCount? countBufferVa : 0); } else { PalCmdBuffer(deviceIdx)->CmdDrawIndexedIndirectMulti( gpuVirtAddrAndStride, count, useBufferCount ? countBufferVa : 0); } } while (deviceGroup.IterateNext()); DbgBarrierPostCmd((indexed ? DbgBarrierDrawIndexed : DbgBarrierDrawNonIndexed) | DbgBarrierDrawIndirect); } // ===================================================================================================================== void CmdBuffer::DrawMeshTasks( uint32_t x, uint32_t y, uint32_t z) { if ((x * y * z) > 0) { DbgBarrierPreCmd(DbgBarrierDrawMeshTasks); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif PalCmdDrawMeshTasks(x, y, z); DbgBarrierPostCmd(DbgBarrierDrawMeshTasks); } } // ===================================================================================================================== template void CmdBuffer::DrawMeshTasksIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset) { DbgBarrierPreCmd(DbgBarrierDrawMeshTasksIndirect); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif PalCmdDrawMeshTasksIndirect(buffer, offset, count, stride, countBuffer, countOffset); DbgBarrierPostCmd(DbgBarrierDrawMeshTasksIndirect); } // ===================================================================================================================== template void CmdBuffer::DrawMeshTasksIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa) { DbgBarrierPreCmd(DbgBarrierDrawMeshTasksIndirect); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif VK_ASSERT(stride <= indirectBufferSize); Pal::GpuVirtAddrAndStride gpuVirtAddrAndStride = { indirectBufferVa, {stride} }; utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdDispatchMeshIndirectMulti( gpuVirtAddrAndStride, count, useBufferCount? countBufferVa : 0); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierDrawMeshTasksIndirect); } // ===================================================================================================================== void CmdBuffer::Dispatch( uint32_t x, uint32_t y, uint32_t z) { DbgBarrierPreCmd(DbgBarrierDispatch); m_perCmdBufDispatchCallCounter++; // Increment per command buffer dispatch call counter if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute) == false) { RebindPipeline(); } #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pComputePipeline, Pal::PipelineBindPoint::Compute, x, y, z, nullptr, 0, 0); #endif if (m_pDevice->GetRuntimeSettings().dispatchPingPong == DispatchPingPongSw) { BindAlternatingThreadGroupConstant(); } PalCmdDispatch(x, y, z); DbgBarrierPostCmd(DbgBarrierDispatch); } // ===================================================================================================================== void CmdBuffer::DispatchOffset( uint32_t base_x, uint32_t base_y, uint32_t base_z, uint32_t dim_x, uint32_t dim_y, uint32_t dim_z) { DbgBarrierPreCmd(DbgBarrierDispatch); m_perCmdBufDispatchCallCounter++; // Increment per command buffer dispatch call counter if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute) == false) { RebindPipeline(); } #if VKI_RAY_TRACING BindRayQueryConstants( m_allGpuState.pComputePipeline, Pal::PipelineBindPoint::Compute, dim_x, dim_y, dim_z, nullptr, 0, 0); #endif PalCmdDispatchOffset(base_x, base_y, base_z, dim_x, dim_y, dim_z); DbgBarrierPostCmd(DbgBarrierDispatch); } // ===================================================================================================================== void CmdBuffer::DispatchIndirect( VkBuffer buffer, VkDeviceSize offset) { DbgBarrierPreCmd(DbgBarrierDispatchIndirect); m_perCmdBufDispatchCallCounter++; // Increment per command buffer dispatch call counter if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute) == false) { RebindPipeline(); } Buffer* pBuffer = Buffer::ObjectFromHandle(buffer); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pComputePipeline, Pal::PipelineBindPoint::Compute, 0, 0, 0, pBuffer, offset, 0); #endif PalCmdDispatchIndirect(pBuffer, offset); DbgBarrierPostCmd(DbgBarrierDispatchIndirect); } // ===================================================================================================================== void CmdBuffer::DispatchIndirect( VkDeviceSize indirectBufferVa) { DbgBarrierPreCmd(DbgBarrierDispatchIndirect); m_perCmdBufDispatchCallCounter++; // Increment per command buffer dispatch call counter if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute) == false) { RebindPipeline(); } #if VKI_RAY_TRACING BindRayQueryConstants( m_allGpuState.pComputePipeline, Pal::PipelineBindPoint::Compute, 0, 0, 0, nullptr, 0, indirectBufferVa); #endif utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdDispatchIndirect(indirectBufferVa); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierDispatchIndirect); } // ===================================================================================================================== void CmdBuffer::ExecuteIndirect( VkBool32 isPreprocessed, const VkGeneratedCommandsInfoNV* pInfo) { IndirectCommandsLayoutNV* pLayout = IndirectCommandsLayoutNV::ObjectFromHandle(pInfo->indirectCommandsLayout); IndirectCommandsInfo info = pLayout->GetIndirectCommandsInfo(); uint64_t barrierCmd = 0; if ((info.actionType == IndirectCommandsActionType::Draw) || (info.actionType == IndirectCommandsActionType::DrawIndexed) || (info.actionType == IndirectCommandsActionType::DrawMeshTask)) { const bool isMeshTask = (info.actionType == IndirectCommandsActionType::DrawMeshTask); const bool isIndexed = (info.actionType == IndirectCommandsActionType::DrawIndexed); if (isMeshTask) { barrierCmd = DbgBarrierDrawMeshTasksIndirect; } else { barrierCmd = (isIndexed ? DbgBarrierDrawIndexed : DbgBarrierDrawNonIndexed) | DbgBarrierDrawIndirect; } DbgBarrierPreCmd(barrierCmd); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); } else if (info.actionType == IndirectCommandsActionType::Dispatch) { barrierCmd = DbgBarrierDispatchIndirect; DbgBarrierPreCmd(barrierCmd); m_perCmdBufDispatchCallCounter++; // Increment per command buffer dispatch call counter if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute) == false) { RebindPipeline(); } } else { VK_NEVER_CALLED(); } VK_ASSERT(pInfo->streamCount == 1); const Buffer* pArgumentBuffer = Buffer::ObjectFromHandle(pInfo->pStreams[0].buffer); const uint64_t argumentOffset = pInfo->pStreams[0].offset; const Buffer* pCountBuffer = Buffer::ObjectFromHandle(pInfo->sequencesCountBuffer); const uint64_t countOffset = pInfo->sequencesCountOffset; const uint32_t maxCount = pInfo->sequencesCount; utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdExecuteIndirectCmds( *pLayout->PalIndirectCmdGenerator(deviceIdx), pArgumentBuffer->GpuVirtAddr(deviceIdx) + argumentOffset, maxCount, (pCountBuffer == nullptr) ? 0 : pCountBuffer->GpuVirtAddr(deviceIdx) + countOffset); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(barrierCmd); } // ===================================================================================================================== void CmdBuffer::ExecuteIndirect( VkBool32 isPreprocessed, const VkGeneratedCommandsInfoEXT* pInfo) { IndirectCommandsLayout* pLayout = IndirectCommandsLayout::ObjectFromHandle(pInfo->indirectCommandsLayout); IndirectCommandsInfo info = pLayout->GetIndirectCommandsInfo(); uint64_t barrierCmd = 0; if ((info.actionType == IndirectCommandsActionType::Draw) || (info.actionType == IndirectCommandsActionType::DrawIndexed) || (info.actionType == IndirectCommandsActionType::DrawMeshTask)) { const bool isMeshTask = (info.actionType == IndirectCommandsActionType::DrawMeshTask); const bool isIndexed = (info.actionType == IndirectCommandsActionType::DrawIndexed); if (isMeshTask) { barrierCmd = DbgBarrierDrawMeshTasksIndirect; } else { barrierCmd = (isIndexed ? DbgBarrierDrawIndexed : DbgBarrierDrawNonIndexed) | DbgBarrierDrawIndirect; } DbgBarrierPreCmd(barrierCmd); m_perCmdBufDrawCallCounter++; // Increment per command buffer draw call counter ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants( m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif } else if (info.actionType == IndirectCommandsActionType::Dispatch) { barrierCmd = DbgBarrierDispatchIndirect; DbgBarrierPreCmd(barrierCmd); m_perCmdBufDispatchCallCounter++; // Increment per command buffer dispatch call counter if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindCompute) == false) { RebindPipeline(); } #if VKI_RAY_TRACING BindRayQueryConstants( m_allGpuState.pComputePipeline, Pal::PipelineBindPoint::Compute, 0, 0, 0, nullptr, 0, 0); #endif } #if VKI_RAY_TRACING else if (info.actionType == IndirectCommandsActionType::TraceRay) { barrierCmd = DbgTraceRays; DbgBarrierPreCmd(barrierCmd); } #endif else { VK_NEVER_CALLED(); } #if VKI_RAY_TRACING if (info.actionType == IndirectCommandsActionType::TraceRay) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); constexpr VkStridedDeviceAddressRegionKHR EmptyShaderBindingTable = {}; for (uint32_t cmdId = 0; cmdId < pInfo->maxSequenceCount; cmdId++) { TraceRaysIndirectPerDevice( deviceIdx, GpuRt::ExecuteIndirectArgType::DispatchDimenionsAndShaderTable, EmptyShaderBindingTable, EmptyShaderBindingTable, EmptyShaderBindingTable, EmptyShaderBindingTable, pInfo->indirectAddress + cmdId * info.strideInBytes, pLayout, GetUserMarkerContextValue()); } } while (deviceGroup.IterateNext()); } else #endif { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdExecuteIndirectCmds( *pLayout->PalIndirectCmdGenerator(deviceIdx), pInfo->indirectAddress, pInfo->maxSequenceCount, pInfo->sequenceCountAddress); } while (deviceGroup.IterateNext()); } DbgBarrierPostCmd(barrierCmd); } // ===================================================================================================================== // Performs a color clear (vkCmdClearColorImage) void CmdBuffer::ClearColorImage( VkImage image, VkImageLayout imageLayout, const VkClearColorValue* pColor, uint32_t rangeCount, const VkImageSubresourceRange* pRanges) { PalCmdSuspendPredication(true); const Image* pImage = Image::ObjectFromHandle(image); const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); VkFormat format = pImage->TreatAsSrgb() ? pImage->GetSrgbFormat() : pImage->GetFormat(); const Pal::SwizzledFormat palFormat = VkToPalFormat(format, settings); if (Pal::Formats::IsBlockCompressed(palFormat.format)) { return; } VirtualStackFrame virtStackFrame(m_pStackAllocator); const auto maxRanges = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::SubresRange)), MaxPalColorAspectsPerMask); auto rangeBatch = Util::Min(rangeCount * MaxPalColorAspectsPerMask, maxRanges); // Allocate space to store image subresource ranges Pal::SubresRange* pPalRanges = virtStackFrame.AllocArray(rangeBatch); if (pPalRanges != nullptr) { const Pal::ImageLayout layout = pImage->GetBarrierPolicy().GetTransferLayout( imageLayout, GetQueueFamilyIndex()); for (uint32_t rangeIdx = 0; rangeIdx < rangeCount;) { uint32_t palRangeCount = 0; while ((rangeIdx < rangeCount) && (palRangeCount <= (rangeBatch - MaxPalColorAspectsPerMask))) { // Only color aspect is allowed here VK_ASSERT(pRanges[rangeIdx].aspectMask == VK_IMAGE_ASPECT_COLOR_BIT); VkToPalSubresRange(pImage->GetFormat(), pRanges[rangeIdx], pImage->GetMipLevels(), pImage->GetArraySize(), pPalRanges, &palRangeCount, settings); ++rangeIdx; } PalCmdClearColorImage( *pImage, layout, VkToPalClearColor(*pColor, palFormat), palFormat, palRangeCount, pPalRanges, 0, nullptr, settings.addMissingBarrierForClearColorImage ? static_cast(Pal::ColorClearAutoSync) : 0); } virtStackFrame.FreeArray(pPalRanges); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } PalCmdSuspendPredication(false); } // ===================================================================================================================== // Performs a depth-stencil clear of an image (vkCmdClearDepthStencilImage) void CmdBuffer::ClearDepthStencilImage( VkImage image, VkImageLayout imageLayout, float depth, uint32_t stencil, uint32_t rangeCount, const VkImageSubresourceRange* pRanges) { PalCmdSuspendPredication(true); VirtualStackFrame virtStackFrame(m_pStackAllocator); const auto maxRanges = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::SubresRange)), MaxPalDepthAspectsPerMask); auto rangeBatch = Util::Min(rangeCount * MaxPalDepthAspectsPerMask, maxRanges); // Allocate space to store image subresource ranges (we need a separate region per PAL aspect) Pal::SubresRange* pPalRanges = virtStackFrame.AllocArray(rangeBatch); if (pPalRanges != nullptr) { const Image* pImage = Image::ObjectFromHandle(image); const Pal::ImageLayout layout = pImage->GetBarrierPolicy().GetTransferLayout( imageLayout, GetQueueFamilyIndex()); ValidateSamplePattern(pImage->GetImageSamples(), nullptr); for (uint32_t rangeIdx = 0; rangeIdx < rangeCount;) { uint32_t palRangeCount = 0; while ((rangeIdx < rangeCount) && (palRangeCount <= (rangeBatch - MaxPalDepthAspectsPerMask))) { // Only depth or stencil aspect is allowed here VK_ASSERT((pRanges[rangeIdx].aspectMask & ~(VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)) == 0); VkToPalSubresRange(pImage->GetFormat(), pRanges[rangeIdx], pImage->GetMipLevels(), pImage->GetArraySize(), pPalRanges, &palRangeCount, m_pDevice->GetRuntimeSettings()); ++rangeIdx; } PalCmdClearDepthStencil( *pImage, layout, layout, VkToPalClearDepth(depth), stencil, palRangeCount, pPalRanges, 0, nullptr, 0); } virtStackFrame.FreeArray(pPalRanges); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } PalCmdSuspendPredication(false); } // ===================================================================================================================== // Clears a set of attachments in the current subpass void CmdBuffer::ClearAttachments( uint32_t attachmentCount, const VkClearAttachment* pAttachments, uint32_t rectCount, const VkClearRect* pRects) { // if pRenderPass is null, than dynamic rendering is being used if (UsingDynamicRendering()) { if (m_flags.is2ndLvl == false) { ClearDynamicRenderingImages(attachmentCount, pAttachments, rectCount, pRects); } else { ClearDynamicRenderingBoundAttachments(attachmentCount, pAttachments, rectCount, pRects); } } else { if ((m_flags.is2ndLvl == false) && (m_allGpuState.pFramebuffer != nullptr)) { ClearImageAttachments(attachmentCount, pAttachments, rectCount, pRects); } else { ClearBoundAttachments(attachmentCount, pAttachments, rectCount, pRects); } } } // ===================================================================================================================== // Clears a set of attachments in the current dynamic rendering pass. void CmdBuffer::ClearDynamicRenderingImages( uint32_t attachmentCount, const VkClearAttachment* pAttachments, uint32_t rectCount, const VkClearRect* pRects) { // Note: Bound target clears are pipelined by the HW, so we do not have to insert any barriers VirtualStackFrame virtStackFrame(m_pStackAllocator); constexpr uint32 MinRects = 8; for (uint32_t idx = 0; idx < attachmentCount; ++idx) { const VkClearAttachment& clearInfo = pAttachments[idx]; // Detect if color clear or depth clear if ((clearInfo.aspectMask & VK_IMAGE_ASPECT_COLOR_BIT) != 0) { const DynamicRenderingAttachments& attachment = m_allGpuState.dynamicRenderingInstance.colorAttachments[clearInfo.colorAttachment]; // Clear only if the referenced attachment index is active if ((attachment.pImageView != nullptr) && (attachment.pImageView->GetImage() != nullptr)) { const Image* pImage = attachment.pImageView->GetImage(); const Pal::SwizzledFormat palFormat = VkToPalFormat(attachment.attachmentFormat, m_pDevice->GetRuntimeSettings()); Util::Vector clearBoxes { &virtStackFrame }; Util::Vector clearSubresRanges{ &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::Box) + sizeof(Pal::SubresRange)), MinRects); auto rectBatch = Util::Min(rectCount, maxRects); const auto palResult1 = clearBoxes.Reserve(rectBatch); const auto palResult2 = clearSubresRanges.Reserve(rectBatch); if ((palResult1 == Pal::Result::Success) && (palResult2 == Pal::Result::Success)) { for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { // Obtain the baseArrayLayer of the image view to apply it when clearing the image itself. const uint32_t zOffset = static_cast(attachment.pImageView->GetZRange().offset); rectBatch = Util::Min(rectCount - rectIdx, maxRects); CreateClearBoxes( rectCount, (pRects + rectIdx), m_allGpuState.dynamicRenderingInstance.viewMask, zOffset, pImage->GetImageType() == VK_IMAGE_TYPE_3D, &clearBoxes); CreateClearSubresRanges( attachment.pImageView, pImage->GetImageType() == VK_IMAGE_TYPE_3D, clearInfo, rectBatch, pRects + rectIdx, m_allGpuState.dynamicRenderingInstance.viewMask, &clearSubresRanges); PalCmdClearColorImage( *pImage, attachment.imageLayout, VkToPalClearColor(clearInfo.clearValue.color, palFormat), palFormat, clearSubresRanges.NumElements(), clearSubresRanges.Data(), clearBoxes.NumElements(), clearBoxes.Data(), Pal::ClearColorImageFlags::ColorClearAutoSync); } } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } } else { const DynamicRenderingAttachments& depthAttachment = m_allGpuState.dynamicRenderingInstance.depthAttachment; const DynamicRenderingAttachments& stencilAttachment = m_allGpuState.dynamicRenderingInstance.stencilAttachment; // Depth and Stencil Views are the same if both exist Pal::ImageLayout imageLayout = {}; const ImageView* pDepthStencilView = nullptr; if ((depthAttachment.pImageView != nullptr) && ((clearInfo.aspectMask & VK_IMAGE_ASPECT_DEPTH_BIT) != 0)) { pDepthStencilView = depthAttachment.pImageView; imageLayout = depthAttachment.imageLayout; } else if ((stencilAttachment.pImageView != nullptr) && ((clearInfo.aspectMask & VK_IMAGE_ASPECT_STENCIL_BIT) != 0)) { pDepthStencilView = stencilAttachment.pImageView; imageLayout = stencilAttachment.imageLayout; } // Clear only if the referenced attachment index is active if (pDepthStencilView != nullptr) { Util::Vector clearRects { &virtStackFrame }; Util::Vector clearSubresRanges{ &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::Rect) + sizeof(Pal::SubresRange)), MinRects); auto rectBatch = Util::Min((rectCount * MaxPalDepthAspectsPerMask), maxRects); const auto palResult1 = clearRects.Reserve(rectBatch); const auto palResult2 = clearSubresRanges.Reserve(rectBatch); if ((palResult1 == Pal::Result::Success) && (palResult2 == Pal::Result::Success)) { ValidateSamplePattern(pDepthStencilView->GetImage()->GetImageSamples(), nullptr); for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { // Obtain the baseArrayLayer of the image view to apply it when clearing the image itself. const uint32_t zOffset = static_cast(pDepthStencilView->GetZRange().offset); rectBatch = Util::Min(rectCount - rectIdx, maxRects); CreateClearRects( rectCount, (pRects + rectIdx), &clearRects); CreateClearSubresRanges( pDepthStencilView, false, // is3dImage clearInfo, rectBatch, pRects + rectIdx, m_allGpuState.dynamicRenderingInstance.viewMask, &clearSubresRanges); PalCmdClearDepthStencil( *pDepthStencilView->GetImage(), imageLayout, imageLayout, VkToPalClearDepth(clearInfo.clearValue.depthStencil.depth), clearInfo.clearValue.depthStencil.stencil, clearSubresRanges.NumElements(), clearSubresRanges.Data(), clearRects.NumElements(), clearRects.Data(), Pal::ClearDepthStencilFlags::DsClearAutoSync); } } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } } } } // ===================================================================================================================== // Clears a set of attachments in the current renderpass using PAL's CmdClearBound*Targets commands. void CmdBuffer::ClearDynamicRenderingBoundAttachments( uint32_t attachmentCount, const VkClearAttachment* pAttachments, uint32_t rectCount, const VkClearRect* pRects) { // Note: Bound target clears are pipelined by the HW, so we do not have to insert any barriers VirtualStackFrame virtStackFrame(m_pStackAllocator); constexpr uint32 MinRects = 8; Util::Vector clearRegions{ &virtStackFrame }; Util::Vector colorTargets{ &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::ClearBoundTargetRegion) + sizeof(Pal::BoundColorTarget)), MinRects); auto rectBatch = Util::Min(rectCount, maxRects); const auto palResult1 = clearRegions.Reserve(rectBatch); const auto palResult2 = colorTargets.Reserve(attachmentCount); m_recordingResult = ((palResult1 == Pal::Result::Success) && (palResult2 == Pal::Result::Success)) ? VK_SUCCESS : VK_ERROR_OUT_OF_HOST_MEMORY; if (m_recordingResult == VK_SUCCESS) { for (uint32_t idx = 0; idx < attachmentCount; ++idx) { const VkClearAttachment& clearInfo = pAttachments[idx]; // Detect if color clear or depth clear if ((clearInfo.aspectMask & VK_IMAGE_ASPECT_COLOR_BIT) != 0) { // Fill in bound target information for this target, but don't clear yet const uint32_t tgtIdx = clearInfo.colorAttachment; // Clear only if the attachment reference is active if (tgtIdx < m_allGpuState.dynamicRenderingInstance.colorAttachmentCount) { const auto& attachment = m_allGpuState.dynamicRenderingInstance.colorAttachments[tgtIdx]; if (attachment.attachmentFormat != VK_FORMAT_UNDEFINED) { const uint32_t remappedIdx = m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[tgtIdx]; Pal::BoundColorTarget target = {}; target.targetIndex = remappedIdx; target.swizzledFormat = VkToPalFormat(attachment.attachmentFormat, m_pDevice->GetRuntimeSettings()); target.samples = attachment.rasterizationSamples; target.fragments = attachment.rasterizationSamples; target.clearValue = VkToPalClearColor(clearInfo.clearValue.color, target.swizzledFormat); colorTargets.PushBack(target); } } } else // Depth-stencil clear { Pal::DepthStencilSelectFlags selectFlags = {}; selectFlags.depth = ((clearInfo.aspectMask & VK_IMAGE_ASPECT_DEPTH_BIT) != 0); selectFlags.stencil = ((clearInfo.aspectMask & VK_IMAGE_ASPECT_STENCIL_BIT) != 0); DbgBarrierPreCmd(DbgBarrierClearDepth); for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { rectBatch = Util::Min(rectCount - rectIdx, maxRects); uint32_t viewMask = m_allGpuState.dynamicRenderingInstance.viewMask; CreateClearRegions( rectBatch, pRects + rectIdx, viewMask, 0u, &clearRegions); // Clear the bound depth stencil target immediately PalCmdBuffer(DefaultDeviceIndex)->CmdClearBoundDepthStencilTargets( VkToPalClearDepth(clearInfo.clearValue.depthStencil.depth), clearInfo.clearValue.depthStencil.stencil, StencilWriteMaskFull, m_allGpuState.dynamicRenderingInstance.depthAttachment.rasterizationSamples, m_allGpuState.dynamicRenderingInstance.depthAttachment.rasterizationSamples, selectFlags, clearRegions.NumElements(), clearRegions.Data()); } DbgBarrierPostCmd(DbgBarrierClearDepth); } } if (colorTargets.NumElements() > 0) { DbgBarrierPreCmd(DbgBarrierClearColor); for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { rectBatch = Util::Min(rectCount - rectIdx, maxRects); uint32_t viewMask = m_allGpuState.dynamicRenderingInstance.viewMask; CreateClearRegions( rectBatch, pRects + rectIdx, viewMask, 0u, &clearRegions); // Clear the bound color targets PalCmdBuffer(DefaultDeviceIndex)->CmdClearBoundColorTargets( colorTargets.NumElements(), colorTargets.Data(), clearRegions.NumElements(), clearRegions.Data()); } DbgBarrierPostCmd(DbgBarrierClearColor); } } } // ===================================================================================================================== // Clears a set of attachments in the current subpass using PAL's CmdClearBound*Targets commands. void CmdBuffer::ClearBoundAttachments( uint32_t attachmentCount, const VkClearAttachment* pAttachments, uint32_t rectCount, const VkClearRect* pRects) { // Note: Bound target clears are pipelined by the HW, so we do not have to insert any barriers VirtualStackFrame virtStackFrame(m_pStackAllocator); // Get the current renderpass and subpass const RenderPass* pRenderPass = m_allGpuState.pRenderPass; const uint32_t subpass = m_renderPassInstance.subpass; constexpr uint32 MinRects = 8; Util::Vector clearRegions { &virtStackFrame }; Util::Vector colorTargets { &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::ClearBoundTargetRegion) + sizeof(Pal::BoundColorTarget)), MinRects); auto rectBatch = Util::Min(rectCount, maxRects); const auto palResult1 = clearRegions.Reserve(rectBatch); const auto palResult2 = colorTargets.Reserve(attachmentCount); m_recordingResult = ((palResult1 == Pal::Result::Success) && (palResult2 == Pal::Result::Success)) ? VK_SUCCESS : VK_ERROR_OUT_OF_HOST_MEMORY; if (m_recordingResult == VK_SUCCESS) { for (uint32_t idx = 0; idx < attachmentCount; ++idx) { const VkClearAttachment& clearInfo = pAttachments[idx]; // Detect if color clear or depth clear if ((clearInfo.aspectMask & VK_IMAGE_ASPECT_COLOR_BIT) != 0) { // Get the corresponding color reference in the current subpass const AttachmentReference& colorRef = pRenderPass->GetSubpassColorReference( subpass, clearInfo.colorAttachment); // Clear only if the attachment reference is active if (colorRef.attachment != VK_ATTACHMENT_UNUSED) { // Fill in bound target information for this target, but don't clear yet const uint32_t tgtIdx = clearInfo.colorAttachment; Pal::BoundColorTarget target = {}; target.targetIndex = tgtIdx; target.swizzledFormat = VkToPalFormat(pRenderPass->GetColorAttachmentFormat(subpass, tgtIdx), m_pDevice->GetRuntimeSettings()); target.samples = pRenderPass->GetColorAttachmentSamples(subpass, tgtIdx); target.fragments = pRenderPass->GetColorAttachmentSamples(subpass, tgtIdx); target.clearValue = VkToPalClearColor(clearInfo.clearValue.color, target.swizzledFormat); colorTargets.PushBack(target); } } else // Depth-stencil clear { // Get the corresponding color reference in the current subpass const AttachmentReference& depthStencilRef = pRenderPass->GetSubpassDepthStencilReference(subpass); // Clear only if the attachment reference is active if (depthStencilRef.attachment != VK_ATTACHMENT_UNUSED) { Pal::DepthStencilSelectFlags selectFlags = {}; selectFlags.depth = ((clearInfo.aspectMask & VK_IMAGE_ASPECT_DEPTH_BIT) != 0); selectFlags.stencil = ((clearInfo.aspectMask & VK_IMAGE_ASPECT_STENCIL_BIT) != 0); DbgBarrierPreCmd(DbgBarrierClearDepth); for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { rectBatch = Util::Min(rectCount - rectIdx, maxRects); uint32_t viewMask = pRenderPass->GetViewMask(subpass); CreateClearRegions( rectBatch, pRects + rectIdx, viewMask, 0u, &clearRegions); // Clear the bound depth stencil target immediately PalCmdBuffer(DefaultDeviceIndex)->CmdClearBoundDepthStencilTargets( VkToPalClearDepth(clearInfo.clearValue.depthStencil.depth), clearInfo.clearValue.depthStencil.stencil, StencilWriteMaskFull, pRenderPass->GetDepthStencilAttachmentSamples(subpass), pRenderPass->GetDepthStencilAttachmentSamples(subpass), selectFlags, clearRegions.NumElements(), clearRegions.Data()); } DbgBarrierPostCmd(DbgBarrierClearDepth); } } } if (colorTargets.NumElements() > 0) { DbgBarrierPreCmd(DbgBarrierClearColor); for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { rectBatch = Util::Min(rectCount - rectIdx, maxRects); uint32_t viewMask = pRenderPass->GetViewMask(subpass); CreateClearRegions( rectBatch, pRects + rectIdx, viewMask, 0u, &clearRegions); // Clear the bound color targets PalCmdBuffer(DefaultDeviceIndex)->CmdClearBoundColorTargets( colorTargets.NumElements(), colorTargets.Data(), clearRegions.NumElements(), clearRegions.Data()); } DbgBarrierPostCmd(DbgBarrierClearColor); } } } // ===================================================================================================================== void CmdBuffer::PalCmdClearColorImage( const Image& image, Pal::ImageLayout imageLayout, const Pal::ClearColor& color, const Pal::SwizzledFormat& clearFormat, uint32_t rangeCount, const Pal::SubresRange* pRanges, uint32_t boxCount, const Pal::Box* pBoxes, uint32_t flags) { DbgBarrierPreCmd(DbgBarrierClearColor); RegisterWriteToFlippableImage(&image); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdClearColorImage( *image.PalImage(deviceIdx), imageLayout, color, clearFormat, rangeCount, pRanges, boxCount, pBoxes, flags); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierClearColor); } // ===================================================================================================================== void CmdBuffer::PalCmdClearDepthStencil( const Image& image, Pal::ImageLayout depthLayout, Pal::ImageLayout stencilLayout, float depth, uint8_t stencil, uint32_t rangeCount, const Pal::SubresRange* pRanges, uint32_t rectCount, const Pal::Rect* pRects, uint32_t flags) { DbgBarrierPreCmd(DbgBarrierClearDepth); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdClearDepthStencil( *image.PalImage(deviceIdx), depthLayout, stencilLayout, depth, stencil, StencilWriteMaskFull, rangeCount, pRanges, rectCount, pRects, flags); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierClearDepth); } // ===================================================================================================================== void CmdBuffer::PalCmdResetEvent( Event* pEvent, uint32 stageMask) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdResetEvent(*pEvent->PalEvent(deviceIdx), stageMask); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdSetEvent( Event* pEvent, uint32 stageMask) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetEvent(*pEvent->PalEvent(deviceIdx), stageMask); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdResolveImage( const Image& srcImage, Pal::ImageLayout srcImageLayout, const Image& dstImage, Pal::ImageLayout dstImageLayout, Pal::ResolveMode resolveMode, uint32_t regionCount, const Pal::ImageResolveRegion* pRegions, uint32_t deviceMask) { DbgBarrierPreCmd(DbgBarrierResolve); RegisterWriteToFlippableImage(&dstImage); utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdResolveImage( *srcImage.PalImage(deviceIdx), srcImageLayout, *dstImage.PalImage(deviceIdx), dstImageLayout, resolveMode, regionCount, pRegions, 0); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierResolve); } // ===================================================================================================================== // Clears a set of attachments in the current subpass using PAL's CmdClear*Image() commands. void CmdBuffer::ClearImageAttachments( uint32_t attachmentCount, const VkClearAttachment* pAttachments, uint32_t rectCount, const VkClearRect* pRects) { VirtualStackFrame virtStackFrame(m_pStackAllocator); constexpr uint32 MinRects = 8; // Get the current renderpass and subpass const RenderPass* pRenderPass = m_allGpuState.pRenderPass; const uint32_t subpass = m_renderPassInstance.subpass; // Go through each of the clear attachment infos for (uint32_t idx = 0; idx < attachmentCount; ++idx) { const VkClearAttachment& clearInfo = pAttachments[idx]; // Detect if color clear or depth clear if ((clearInfo.aspectMask & VK_IMAGE_ASPECT_COLOR_BIT) != 0) { // Get the color target index (subpass color reference index) const uint32_t targetIdx = clearInfo.colorAttachment; // Get the corresponding color reference in the current subpass const AttachmentReference& colorRef = pRenderPass->GetSubpassColorReference(subpass, targetIdx); // Get the referenced attachment index in the framebuffer const uint32_t attachmentIdx = colorRef.attachment; // Clear only if the referenced attachment index is active if (attachmentIdx != VK_ATTACHMENT_UNUSED) { // Get the matching framebuffer attachment const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(attachmentIdx); // Get the layout that this color attachment is currently in within the render pass const Pal::ImageLayout targetLayout = RPGetAttachmentLayout(attachmentIdx, 0); Util::Vector clearBoxes { &virtStackFrame }; Util::Vector clearSubresRanges { &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::Box) + sizeof(Pal::SubresRange)), MinRects); auto rectBatch = Util::Min(rectCount, maxRects); const auto palResult1 = clearBoxes.Reserve(rectBatch); const auto palResult2 = clearSubresRanges.Reserve(rectBatch); if ((palResult1 == Pal::Result::Success) && (palResult2 == Pal::Result::Success)) { for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { // Obtain the baseArrayLayer of the image view to apply it when clearing the image itself. const uint32_t zOffset = static_cast(attachment.pView->GetZRange().offset); rectBatch = Util::Min(rectCount - rectIdx, maxRects); uint32_t viewMask = pRenderPass->GetViewMask(subpass); CreateClearBoxes( rectCount, pRects + rectIdx, viewMask, zOffset, attachment.pImage->GetImageType() == VK_IMAGE_TYPE_3D, &clearBoxes); CreateClearSubresRanges( attachment, attachment.pImage->GetImageType() == VK_IMAGE_TYPE_3D, clearInfo, rectBatch, pRects + rectIdx, *pRenderPass, subpass, &clearSubresRanges); PalCmdClearColorImage( *attachment.pImage, targetLayout, VkToPalClearColor(clearInfo.clearValue.color, attachment.viewFormat), attachment.viewFormat, clearSubresRanges.NumElements(), clearSubresRanges.Data(), clearBoxes.NumElements(), clearBoxes.Data(), Pal::ClearColorImageFlags::ColorClearAutoSync); } } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } } else // Depth-stencil clear { // Get the depth-stencil reference of the current subpass const AttachmentReference& depthStencilRef = pRenderPass->GetSubpassDepthStencilReference(subpass); // Get the referenced attachment index in the framebuffer const uint32_t attachmentIdx = depthStencilRef.attachment; // Clear only if the referenced attachment index is active if (attachmentIdx != VK_ATTACHMENT_UNUSED) { // Get the matching framebuffer attachment const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(attachmentIdx); // Get the layout(s) that this attachment is currently in within the render pass const Pal::ImageLayout depthLayout = RPGetAttachmentLayout(attachmentIdx, 0); const Pal::ImageLayout stencilLayout = RPGetAttachmentLayout(attachmentIdx, 1); Util::Vector clearRects { &virtStackFrame }; Util::Vector clearSubresRanges { &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::Rect) + sizeof(Pal::SubresRange)), MinRects); auto rectBatch = Util::Min(rectCount, maxRects); const auto palResult1 = clearRects.Reserve(rectBatch); const auto palResult2 = clearSubresRanges.Reserve(rectBatch); if ((palResult1 == Pal::Result::Success) && (palResult2 == Pal::Result::Success)) { ValidateSamplePattern(attachment.pImage->GetImageSamples(), nullptr); for (uint32_t rectIdx = 0; rectIdx < rectCount; rectIdx += rectBatch) { rectBatch = Util::Min(rectCount - rectIdx, maxRects); CreateClearRects( rectCount, pRects + rectIdx, &clearRects); CreateClearSubresRanges( attachment, false, // is3dImage clearInfo, rectBatch, pRects + rectIdx, *pRenderPass, subpass, &clearSubresRanges); PalCmdClearDepthStencil( *attachment.pImage, depthLayout, stencilLayout, VkToPalClearDepth(clearInfo.clearValue.depthStencil.depth), clearInfo.clearValue.depthStencil.stencil, clearSubresRanges.NumElements(), clearSubresRanges.Data(), clearRects.NumElements(), clearRects.Data(), Pal::ClearDepthStencilFlags::DsClearAutoSync); } } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } } } } // ===================================================================================================================== template void CmdBuffer::ResolveImage( VkImage srcImage, VkImageLayout srcImageLayout, VkImage destImage, VkImageLayout destImageLayout, uint32_t rectCount, const ImageResolveType* pRects) { PalCmdSuspendPredication(true); VirtualStackFrame virtStackFrame(m_pStackAllocator); const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::ImageResolveRegion)), MaxRangePerAttachment); auto rectBatch = Util::Min(rectCount * MaxRangePerAttachment, maxRects); // Allocate space to store image resolve regions (we need a separate region per PAL aspect) Pal::ImageResolveRegion* pPalRegions = virtStackFrame.AllocArray(rectBatch); if (pPalRegions != nullptr) { const Image* const pSrcImage = Image::ObjectFromHandle(srcImage); const Image* const pDstImage = Image::ObjectFromHandle(destImage); const Pal::SwizzledFormat srcFormat = VkToPalFormat(pSrcImage->GetFormat(), m_pDevice->GetRuntimeSettings()); const Pal::ImageLayout palSrcImageLayout = pSrcImage->GetBarrierPolicy().GetTransferLayout( srcImageLayout, GetQueueFamilyIndex()); const Pal::ImageLayout palDestImageLayout = pDstImage->GetBarrierPolicy().GetTransferLayout( destImageLayout, GetQueueFamilyIndex()); // If ever permitted by the spec, pQuadSamplePattern must be specified because the source image was created with // sampleLocsAlwaysKnown set. VK_ASSERT(pSrcImage->IsDepthStencilFormat() == false); for (uint32_t rectIdx = 0; rectIdx < rectCount;) { uint32_t palRegionCount = 0; while ((rectIdx < rectCount) && (palRegionCount <= (rectBatch - MaxPalAspectsPerMask))) { // We expect MSAA images to never have mipmaps VK_ASSERT(pRects[rectIdx].srcSubresource.mipLevel == 0); VkToPalImageResolveRegion( pRects[rectIdx], srcFormat, pSrcImage->GetArraySize(), pDstImage->TreatAsSrgb(), pPalRegions, &palRegionCount); ++rectIdx; } PalCmdResolveImage( *pSrcImage, palSrcImageLayout, *pDstImage, palDestImageLayout, Pal::ResolveMode::Average, palRegionCount, pPalRegions, m_curDeviceMask); } virtStackFrame.FreeArray(pPalRegions); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } PalCmdSuspendPredication(false); } // ===================================================================================================================== // Implementation of vkCmdSetEvent() void CmdBuffer::SetEvent( VkEvent event, PipelineStageFlags stageMask) { DbgBarrierPreCmd(DbgBarrierSetResetEvent); PalCmdSetEvent(Event::ObjectFromHandle(event), VkToPalPipelineStageFlags(stageMask, true)); DbgBarrierPostCmd(DbgBarrierSetResetEvent); } // ===================================================================================================================== // Implementation of vkCmdSetEvent2() void CmdBuffer::SetEvent2( VkEvent event, const VkDependencyInfoKHR* pDependencyInfo) { DbgBarrierPreCmd(DbgBarrierSetResetEvent); if (m_flags.useSplitReleaseAcquire) { ExecuteAcquireRelease2(1, &event, pDependencyInfo, Release, RgpBarrierExternalCmdWaitEvents); } else { PipelineStageFlags stageMask = 0; for (uint32_t i = 0; i < pDependencyInfo->memoryBarrierCount; i++) { stageMask |= pDependencyInfo->pMemoryBarriers[i].srcStageMask; } for (uint32_t i = 0; i < pDependencyInfo->bufferMemoryBarrierCount; i++) { stageMask |= pDependencyInfo->pBufferMemoryBarriers[i].srcStageMask; } for (uint32_t i = 0; i < pDependencyInfo->imageMemoryBarrierCount; i++) { stageMask |= pDependencyInfo->pImageMemoryBarriers[i].srcStageMask; } PalCmdSetEvent(Event::ObjectFromHandle(event), VkToPalPipelineStageFlags(stageMask, true)); } DbgBarrierPostCmd(DbgBarrierSetResetEvent); } // ===================================================================================================================== // Returns attachment's PAL subresource ranges defined by clearInfo for Dynamic Rendering LoadOp Clear. // When multiview is enabled, layer ranges are modified according active views during a renderpass. Util::Vector LoadOpClearSubresRanges( const uint32_t& viewMask, const Pal::SubresRange& subresRange) { // Note that no allocation will be performed, so Util::Vector allocator is nullptr. Util::Vector clearSubresRanges{ nullptr }; if (viewMask > 0) { const auto layerRanges = RangesOfOnesInBitMask(viewMask); for (auto layerRangeIt = layerRanges.Begin(); layerRangeIt.IsValid(); layerRangeIt.Next()) { clearSubresRanges.PushBack(subresRange); clearSubresRanges.Back().startSubres.arraySlice += layerRangeIt.Get().offset; clearSubresRanges.Back().numSlices = layerRangeIt.Get().extent; } } else { clearSubresRanges.PushBack(subresRange); } return clearSubresRanges; } // ===================================================================================================================== // Clear Color for VK_KHR_dynamic_rendering void CmdBuffer::LoadOpClearColor( const Pal::Rect* pDeviceGroupRenderArea, const VkRenderingInfo* pRenderingInfo) { if (m_pSqttState != nullptr) { m_pSqttState->BeginRenderPassColorClear(); } const ImageView* pImageViews[Pal::MaxColorTargets] = {}; Pal::ClearColor clearColors[Pal::MaxColorTargets] = {}; Pal::ImageLayout imageLayouts[Pal::MaxColorTargets] = {}; Pal::SubresRange ranges[Pal::MaxColorTargets] = {}; Pal::SwizzledFormat clearFormats[Pal::MaxColorTargets] = {}; uint32_t clearCount = 0; // Collect information on the number of clears to decide if we need to batch. for (uint32_t i = 0; i < pRenderingInfo->colorAttachmentCount; ++i) { const VkRenderingAttachmentInfo& attachmentInfo = pRenderingInfo->pColorAttachments[i]; if (attachmentInfo.loadOp == VK_ATTACHMENT_LOAD_OP_CLEAR) { // Get the image view from the attachment info const ImageView* const pImageView = ImageView::ObjectFromHandle(attachmentInfo.imageView); if (pImageView != nullptr) { pImageViews[clearCount] = pImageView; const Image* pImage = pImageView->GetImage(); // Convert the clear color to the format of the attachment view clearFormats[clearCount] = VkToPalFormat( pImageView->GetViewFormat(), m_pDevice->GetRuntimeSettings()); clearColors[clearCount] = VkToPalClearColor( attachmentInfo.clearValue.color, clearFormats[clearCount]); // Get subres range from the image view pImageView->GetFrameBufferAttachmentSubresRange(&ranges[clearCount]); // Override the number of slices with layerCount from pBeginRendering ranges[clearCount].numSlices = pRenderingInfo->layerCount; // Clear Layout imageLayouts[clearCount] = pImage->GetBarrierPolicy().GetAspectLayout( attachmentInfo.imageLayout, ranges[clearCount].startSubres.plane, GetQueueFamilyIndex(), pImage->GetFormat()); clearCount++; } } } if (clearCount > 1) { BatchedLoadOpClears(clearCount, pImageViews, clearColors, imageLayouts, ranges, clearFormats, pRenderingInfo->viewMask); } else if (clearCount == 1) { VK_ASSERT(pImageViews[0] != nullptr); const auto clearSubresRanges = LoadOpClearSubresRanges(pRenderingInfo->viewMask, ranges[0]); utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); // Clear Box Pal::Box clearBox = BuildClearBox(pDeviceGroupRenderArea[deviceIdx], *(pImageViews[0])); PalCmdBuffer(deviceIdx)->CmdClearColorImage( *(pImageViews[0]->GetImage()->PalImage(deviceIdx)), imageLayouts[0], clearColors[0], clearFormats[0], clearSubresRanges.NumElements(), clearSubresRanges.Data(), 1, &clearBox, Pal::ColorClearAutoSync); } while (deviceGroup.IterateNext()); } if (m_pSqttState != nullptr) { m_pSqttState->EndRenderPassColorClear(); } } // ===================================================================================================================== // Clear Depth Stencil for VK_KHR_dynamic_rendering void CmdBuffer::LoadOpClearDepthStencil( const Pal::Rect* pDeviceGroupRenderArea, const VkRenderingInfo* pRenderingInfo) { if (m_pSqttState != nullptr) { m_pSqttState->BeginRenderPassDepthStencilClear(); } // Note that no allocation will be performed, so Util::Vector allocator is nullptr. Util::Vector clearSubresRanges{ nullptr }; const Image* pDepthStencilImage = nullptr; Pal::SubresRange subresRange = {}; Pal::ImageLayout depthLayout = {}; Pal::ImageLayout stencilLayout = {}; float clearDepth = 0.0f; uint8 clearStencil = 0; const VkRenderingAttachmentInfo* pDepthAttachmentInfo = pRenderingInfo->pDepthAttachment; const VkRenderingAttachmentInfo* pStencilAttachmentInfo = pRenderingInfo->pStencilAttachment; if ((pStencilAttachmentInfo != nullptr) && (pStencilAttachmentInfo->imageView != VK_NULL_HANDLE)) { const ImageView* const pStencilImageView = ImageView::ObjectFromHandle(pStencilAttachmentInfo->imageView); if (pStencilImageView != VK_NULL_HANDLE) { pDepthStencilImage = pStencilImageView->GetImage(); GetImageLayout( pStencilAttachmentInfo->imageView, pStencilAttachmentInfo->imageLayout, VK_IMAGE_ASPECT_STENCIL_BIT, &subresRange, &stencilLayout); if (pStencilAttachmentInfo->loadOp == VK_ATTACHMENT_LOAD_OP_CLEAR) { clearSubresRanges.PushBack(subresRange); clearStencil = pStencilAttachmentInfo->clearValue.depthStencil.stencil; } } } if ((pDepthAttachmentInfo != nullptr) && (pDepthAttachmentInfo->imageView != VK_NULL_HANDLE)) { const ImageView* const pDepthImageView = ImageView::ObjectFromHandle(pDepthAttachmentInfo->imageView); if (pDepthImageView != VK_NULL_HANDLE) { pDepthStencilImage = pDepthImageView->GetImage(); GetImageLayout( pDepthAttachmentInfo->imageView, pDepthAttachmentInfo->imageLayout, VK_IMAGE_ASPECT_DEPTH_BIT, &subresRange, &depthLayout); if (pDepthAttachmentInfo->loadOp == VK_ATTACHMENT_LOAD_OP_CLEAR) { clearSubresRanges.PushBack(subresRange); clearDepth = pDepthAttachmentInfo->clearValue.depthStencil.depth; } } } if (pDepthStencilImage != nullptr) { ValidateSamplePattern(pDepthStencilImage->GetImageSamples(), nullptr); utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdClearDepthStencil( *pDepthStencilImage->PalImage(deviceIdx), depthLayout, stencilLayout, clearDepth, clearStencil, StencilWriteMaskFull, clearSubresRanges.NumElements(), clearSubresRanges.Data(), 1, &(pDeviceGroupRenderArea[deviceIdx]), Pal::DsClearAutoSync); } while (deviceGroup.IterateNext()); } if (m_pSqttState != nullptr) { m_pSqttState->EndRenderPassDepthStencilClear(); } } // ===================================================================================================================== // StoreAttachment for VK_KHR_dynamic_rendering void CmdBuffer::StoreAttachmentInfo( const VkRenderingAttachmentInfo& renderingAttachmentInfo, DynamicRenderingAttachments* pDynamicRenderingAttachement) { const ImageView* const pImageView = ImageView::ObjectFromHandle(renderingAttachmentInfo.imageView); if (pImageView != nullptr) { const Image* pColorImage = pImageView->GetImage(); Pal::ImageLayout colorImageLayout = pColorImage->GetAttachmentLayout( { renderingAttachmentInfo.imageLayout, 0 }, 0, this); pDynamicRenderingAttachement->attachmentFormat = pImageView->GetViewFormat(); pDynamicRenderingAttachement->resolveMode = renderingAttachmentInfo.resolveMode; pDynamicRenderingAttachement->pImageView = pImageView; pDynamicRenderingAttachement->imageLayout = colorImageLayout; pDynamicRenderingAttachement->pResolveImageView = ImageView::ObjectFromHandle( renderingAttachmentInfo.resolveImageView); if (pDynamicRenderingAttachement->pResolveImageView != nullptr) { const Image* pResolveImage = pDynamicRenderingAttachement->pResolveImageView->GetImage(); if (pResolveImage != nullptr) { const RPImageLayout resolveLayout = { renderingAttachmentInfo.resolveImageLayout, Pal::LayoutResolveDst }; pDynamicRenderingAttachement->resolveImageLayout = pResolveImage->GetAttachmentLayout(resolveLayout, 0, this); } } } else { *pDynamicRenderingAttachement = {}; } } // ===================================================================================================================== // vkCmdBeginRendering for VK_KHR_dynamic_rendering void CmdBuffer::BeginRendering( const VkRenderingInfo* pRenderingInfo) { VK_ASSERT(pRenderingInfo != nullptr); DbgBarrierPreCmd(DbgBarrierBeginRendering); bool isResuming = (pRenderingInfo->flags & VK_RENDERING_RESUMING_BIT); bool isSuspended = (pRenderingInfo->flags & VK_RENDERING_SUSPENDING_BIT); bool skipEverything = isResuming && m_flags.isRenderingSuspended; bool skipClears = isResuming && (m_flags.isRenderingSuspended == false); m_allGpuState.dynamicRenderingInstance.viewMask = pRenderingInfo->viewMask; m_allGpuState.dynamicRenderingInstance.colorAttachmentCount = pRenderingInfo->colorAttachmentCount; m_allGpuState.dynamicRenderingInstance.enableResolveTarget = false; m_allGpuState.dirtyGraphics.colorWriteMask = 1; for (uint32_t i = 0; i < pRenderingInfo->colorAttachmentCount; ++i) { const VkRenderingAttachmentInfo& colorAttachmentInfo = pRenderingInfo->pColorAttachments[i]; m_allGpuState.dynamicRenderingInstance.enableResolveTarget |= (colorAttachmentInfo.resolveImageView != VK_NULL_HANDLE); StoreAttachmentInfo( colorAttachmentInfo, &m_allGpuState.dynamicRenderingInstance.colorAttachments[i]); m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i] = i; } if (pRenderingInfo->pDepthAttachment != nullptr) { const VkRenderingAttachmentInfo& depthAttachmentInfo = *pRenderingInfo->pDepthAttachment; m_allGpuState.dynamicRenderingInstance.enableResolveTarget |= (depthAttachmentInfo.resolveImageView != VK_NULL_HANDLE); StoreAttachmentInfo( depthAttachmentInfo, &m_allGpuState.dynamicRenderingInstance.depthAttachment); } if (pRenderingInfo->pStencilAttachment != nullptr) { const VkRenderingAttachmentInfo& stencilAttachmentInfo = *pRenderingInfo->pStencilAttachment; m_allGpuState.dynamicRenderingInstance.enableResolveTarget |= (stencilAttachmentInfo.resolveImageView != VK_NULL_HANDLE); StoreAttachmentInfo( stencilAttachmentInfo, &m_allGpuState.dynamicRenderingInstance.stencilAttachment); } m_flags.isRenderingSuspended = isSuspended; if (!skipEverything) { EXTRACT_VK_STRUCTURES_2( RENDERING_INFO_KHR, RenderingInfoKHR, DeviceGroupRenderPassBeginInfo, RenderingFragmentShadingRateAttachmentInfoKHR, pRenderingInfo, RENDER_PASS_BEGIN_INFO, DEVICE_GROUP_RENDER_PASS_BEGIN_INFO, RENDERING_FRAGMENT_SHADING_RATE_ATTACHMENT_INFO_KHR) bool replicateRenderArea = true; if (pDeviceGroupRenderPassBeginInfo != nullptr) { SetDeviceMask(pDeviceGroupRenderPassBeginInfo->deviceMask); m_allGpuState.dynamicRenderingInstance.renderAreaCount = pDeviceGroupRenderPassBeginInfo->deviceRenderAreaCount; VK_ASSERT(m_allGpuState.dynamicRenderingInstance.renderAreaCount <= MaxPalDevices); VK_ASSERT(m_renderPassInstance.renderAreaCount <= MaxPalDevices); if (pDeviceGroupRenderPassBeginInfo->deviceRenderAreaCount > 0) { utils::IterateMask deviceGroup(pDeviceGroupRenderPassBeginInfo->deviceMask); VK_ASSERT(m_numPalDevices == pDeviceGroupRenderPassBeginInfo->deviceRenderAreaCount); do { const uint32_t deviceIdx = deviceGroup.Index(); const VkRect2D& srcRect = pDeviceGroupRenderPassBeginInfo->pDeviceRenderAreas[deviceIdx]; auto* pDstRect = &m_allGpuState.dynamicRenderingInstance.renderArea[deviceIdx]; *pDstRect = VkToPalRect(srcRect); } while (deviceGroup.IterateNext()); replicateRenderArea = false; } } if (replicateRenderArea) { m_allGpuState.dynamicRenderingInstance.renderAreaCount = m_numPalDevices; const auto& srcRect = pRenderingInfo->renderArea; for (uint32_t deviceIdx = 0; deviceIdx < m_numPalDevices; deviceIdx++) { auto* pDstRect = &m_allGpuState.dynamicRenderingInstance.renderArea[deviceIdx]; *pDstRect = VkToPalRect(srcRect); } } Pal::GlobalScissorParams scissorParams = {}; scissorParams.scissorRegion = VkToPalRect(pRenderingInfo->renderArea); utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetGlobalScissor(scissorParams); } while (deviceGroup.IterateNext()); if (skipClears == false) { PalCmdSuspendPredication(true); LoadOpClearColor( m_allGpuState.dynamicRenderingInstance.renderArea, pRenderingInfo); LoadOpClearDepthStencil( m_allGpuState.dynamicRenderingInstance.renderArea, pRenderingInfo); PalCmdSuspendPredication(false); } BindTargets(); if ((pRenderingFragmentShadingRateAttachmentInfoKHR != nullptr) && (pRenderingFragmentShadingRateAttachmentInfoKHR->imageView != VK_NULL_HANDLE)) { // Get the image view from the attachment info const ImageView* const pImageView = ImageView::ObjectFromHandle(pRenderingFragmentShadingRateAttachmentInfoKHR->imageView); // Get the attachment image const Image* pImage = pImageView->GetImage(); utils::IterateMask deviceIndices(GetDeviceMask()); do { const uint32_t deviceIdx = deviceIndices.Index(); PalCmdBuffer(deviceIdx)->CmdBindSampleRateImage(pImage->PalImage(deviceIdx)); } while (deviceIndices.IterateNext()); } uint32_t numMultiViews = Util::CountSetBits(pRenderingInfo->viewMask); uint32_t viewInstanceMask = (numMultiViews > 0) ? pRenderingInfo->viewMask : GetDeviceMask(); PalCmdBuffer(DefaultDeviceIndex)->CmdSetViewInstanceMask(viewInstanceMask); } DbgBarrierPostCmd(DbgBarrierBeginRendering); } // ===================================================================================================================== // Call resolve image for VK_KHR_dynamic_rendering void CmdBuffer::ResolveImage( VkImageAspectFlags aspectMask, const DynamicRenderingAttachments& dynamicRenderingAttachments) { if (m_pSqttState != nullptr) { m_pSqttState->BeginRenderPassResolve(); } Pal::ImageResolveRegion regions[MaxPalDevices] = {}; for (uint32_t idx = 0; idx < m_allGpuState.dynamicRenderingInstance.renderAreaCount; idx++) { const Pal::Rect& renderArea = m_allGpuState.dynamicRenderingInstance.renderArea[idx]; Pal::SubresRange subresRangeSrc = {}; Pal::SubresRange subresRangeDst = {}; dynamicRenderingAttachments.pResolveImageView->GetFrameBufferAttachmentSubresRange(&subresRangeDst); dynamicRenderingAttachments.pImageView->GetFrameBufferAttachmentSubresRange(&subresRangeSrc); const uint32_t sliceCount = Util::Min(subresRangeSrc.numSlices, subresRangeDst.numSlices); VkFormat viewFormat = dynamicRenderingAttachments.pImageView->GetViewFormat(); VkFormat resolveViewFormat = dynamicRenderingAttachments.pResolveImageView->GetViewFormat(); if ((viewFormat != dynamicRenderingAttachments.pImageView->GetImage()->GetFormat()) || (resolveViewFormat != dynamicRenderingAttachments.pResolveImageView->GetImage()->GetFormat())) { // VUID-VkRenderingAttachmentInfo-imageView-06865: // imageView and resolveImageView must have the same VkFormat VK_ASSERT(viewFormat == resolveViewFormat); regions[idx].swizzledFormat = VkToPalFormat(viewFormat, m_pDevice->GetRuntimeSettings()); } else { regions[idx].swizzledFormat = Pal::UndefinedSwizzledFormat; } regions[idx].extent.width = renderArea.extent.width; regions[idx].extent.height = renderArea.extent.height; regions[idx].extent.depth = 1; regions[idx].numSlices = 1; regions[idx].srcOffset.x = renderArea.offset.x; regions[idx].srcOffset.y = renderArea.offset.y; regions[idx].srcOffset.z = 0; regions[idx].dstOffset.x = renderArea.offset.x; regions[idx].dstOffset.y = renderArea.offset.y; regions[idx].dstOffset.z = 0; regions[idx].dstMipLevel = subresRangeDst.startSubres.mipLevel; regions[idx].dstSlice = subresRangeDst.startSubres.arraySlice; regions[idx].numSlices = sliceCount; if ((aspectMask == VK_IMAGE_ASPECT_STENCIL_BIT) && dynamicRenderingAttachments.pImageView->GetImage()->HasDepthAndStencil()) { regions[idx].srcPlane = 1; } if ((aspectMask == VK_IMAGE_ASPECT_STENCIL_BIT) && dynamicRenderingAttachments.pResolveImageView->GetImage()->HasDepthAndStencil()) { regions[idx].dstPlane = 1; } if (Formats::HasDepth(dynamicRenderingAttachments.pImageView->GetViewFormat())) { regions[idx].pQuadSamplePattern = Device::GetDefaultQuadSamplePattern( dynamicRenderingAttachments.pImageView->GetImage()->GetImageSamples()); } } PalCmdResolveImage( *dynamicRenderingAttachments.pImageView->GetImage(), dynamicRenderingAttachments.imageLayout, *dynamicRenderingAttachments.pResolveImageView->GetImage(), dynamicRenderingAttachments.resolveImageLayout, VkToPalResolveMode(dynamicRenderingAttachments.resolveMode), m_allGpuState.dynamicRenderingInstance.renderAreaCount, regions, m_curDeviceMask); if (m_pSqttState != nullptr) { m_pSqttState->EndRenderPassResolve(); } } // ===================================================================================================================== // For Dynamic Rendering we need to wait for draws to finish before we do resolves. void CmdBuffer::PostDrawPreResolveSync() { if (m_flags.useReleaseAcquire) { Pal::AcquireReleaseInfo barrierInfo = { .srcGlobalStageMask = Pal::PipelineStageColorTarget | Pal::PipelineStageDsTarget, .dstGlobalStageMask = Pal::PipelineStageBlt, .srcGlobalAccessMask = Pal::CoherColorTarget | Pal::CoherDepthStencilTarget, .dstGlobalAccessMask = Pal::CoherResolveSrc, .memoryBarrierCount = 0, .pMemoryBarriers = nullptr, .imageBarrierCount = 0, .pImageBarriers = nullptr, .reason = RgpBarrierExternalRenderPassSync }; PalCmdReleaseThenAcquire( &barrierInfo, nullptr, nullptr, nullptr, nullptr, m_curDeviceMask); } else { Pal::BarrierInfo barrierInfo = {}; barrierInfo.waitPoint = Pal::HwPipePreCs; const Pal::HwPipePoint pipePoint = Pal::HwPipePostPs; barrierInfo.pipePointWaitCount = 1; barrierInfo.pPipePoints = &pipePoint; Pal::BarrierTransition transition = {}; transition.srcCacheMask = Pal::CoherColorTarget | Pal::CoherDepthStencilTarget; transition.dstCacheMask = Pal::CoherShader; barrierInfo.transitionCount = 1; barrierInfo.pTransitions = &transition; PalCmdBarrier(barrierInfo, m_curDeviceMask); } } // ===================================================================================================================== // vkCmdEndRendering for VK_KHR_dynamic_rendering void CmdBuffer::EndRendering() { DbgBarrierPreCmd(DbgBarrierEndRenderPass); // Only do resolves if renderpass isn't suspended and // there are resolve targets if (m_allGpuState.dynamicRenderingInstance.enableResolveTarget && (m_flags.isRenderingSuspended == false)) { // Sync draws before resolves PostDrawPreResolveSync(); // Resolve Color Images for (uint32_t i = 0; i < m_allGpuState.dynamicRenderingInstance.colorAttachmentCount; ++i) { const DynamicRenderingAttachments& renderingAttachmentInfo = m_allGpuState.dynamicRenderingInstance.colorAttachments[i]; if ((renderingAttachmentInfo.resolveMode != VK_RESOLVE_MODE_NONE) && (renderingAttachmentInfo.pResolveImageView != nullptr)) { { ResolveImage( VK_IMAGE_ASPECT_COLOR_BIT, renderingAttachmentInfo); } } } // Resolve Depth Image if ((m_allGpuState.dynamicRenderingInstance.depthAttachment.resolveMode != VK_RESOLVE_MODE_NONE) && (m_allGpuState.dynamicRenderingInstance.depthAttachment.pResolveImageView != nullptr)) { ResolveImage( VK_IMAGE_ASPECT_DEPTH_BIT, m_allGpuState.dynamicRenderingInstance.depthAttachment); } // Resolve Stencil Image if ((m_allGpuState.dynamicRenderingInstance.stencilAttachment.resolveMode != VK_RESOLVE_MODE_NONE) && (m_allGpuState.dynamicRenderingInstance.stencilAttachment.pResolveImageView != nullptr)) { ResolveImage( VK_IMAGE_ASPECT_STENCIL_BIT, m_allGpuState.dynamicRenderingInstance.stencilAttachment); } } // Reset attachment counts at End of Rendering m_allGpuState.dynamicRenderingInstance.enableResolveTarget = false; m_allGpuState.dynamicRenderingInstance.colorAttachmentCount = 0; m_allGpuState.dynamicRenderingInstance.depthAttachment = {}; m_allGpuState.dynamicRenderingInstance.stencilAttachment = {}; DbgBarrierPostCmd(DbgBarrierEndRenderPass); } // ===================================================================================================================== void CmdBuffer::ResetEvent( VkEvent event, PipelineStageFlags stageMask) { DbgBarrierPreCmd(DbgBarrierSetResetEvent); Event* pEvent = Event::ObjectFromHandle(event); if (pEvent->IsUseToken()) { const Pal::ReleaseToken token = { {0xFFFFFF, 0xFF} }; pEvent->SetSyncToken(token); } else { PalCmdResetEvent(pEvent, VkToPalPipelineStageFlags(stageMask, true)); } DbgBarrierPostCmd(DbgBarrierSetResetEvent); } // ===================================================================================================================== // Helper function called from ExecuteBarriers void CmdBuffer::FlushBarriers( Pal::BarrierInfo* pBarrier, Pal::BarrierTransition* const pTransitions, const Image** pTransitionImages, uint32_t mainTransitionCount) { pBarrier->transitionCount = mainTransitionCount; pBarrier->pTransitions = pTransitions; PalCmdBarrier(pBarrier, pTransitions, pTransitionImages, m_curDeviceMask); // Remove any signaled events as we do not want to wait more than once. pBarrier->gpuEventWaitCount = 0; pBarrier->ppGpuEvents = nullptr; } // ===================================================================================================================== // ExecuteBarriers Called by vkCmdWaitEvents() and vkCmdPipelineBarrier(). void CmdBuffer::ExecuteBarriers( VirtualStackFrame* pVirtStackFrame, uint32_t memBarrierCount, const VkMemoryBarrier* pMemoryBarriers, uint32_t bufferMemoryBarrierCount, const VkBufferMemoryBarrier* pBufferMemoryBarriers, uint32_t imageMemoryBarrierCount, const VkImageMemoryBarrier* pImageMemoryBarriers, Pal::BarrierInfo* pBarrier) { // The sum of all memory barriers and execution barriers uint32_t barrierCount = memBarrierCount + bufferMemoryBarrierCount + imageMemoryBarrierCount + pBarrier->gpuEventWaitCount + pBarrier->pipePointWaitCount; if (barrierCount == 0) { return; } constexpr uint32_t MaxTransitionCount = 512; constexpr uint32_t MaxLocationCount = 128; pBarrier->globalSrcCacheMask = 0u; pBarrier->globalDstCacheMask = 0u; Pal::BarrierTransition* pTransitions = pVirtStackFrame->AllocArray(MaxTransitionCount); Pal::BarrierTransition* pNextMain = pTransitions; if (pTransitions == nullptr) { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; return; } const Image** pTransitionImages = (m_numPalDevices > 1) && (imageMemoryBarrierCount > 0) ? pVirtStackFrame->AllocArray(MaxTransitionCount) : nullptr; for (uint32_t i = 0; i < memBarrierCount; ++i) { *pNextMain = {}; m_pDevice->GetBarrierPolicy().ApplyBarrierCacheFlags( pMemoryBarriers[i].srcAccessMask, pMemoryBarriers[i].dstAccessMask, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, pNextMain); pNextMain->imageInfo.pImage = nullptr; VK_ASSERT(pMemoryBarriers[i].pNext == nullptr); ++pNextMain; const uint32_t mainTransitionCount = static_cast(pNextMain - pTransitions); if (MaxPalAspectsPerMask + mainTransitionCount > MaxTransitionCount) { FlushBarriers(pBarrier, pTransitions, nullptr, mainTransitionCount); pNextMain = pTransitions; } } for (uint32_t i = 0; i < bufferMemoryBarrierCount; ++i) { *pNextMain = {}; const Buffer* pBuffer = Buffer::ObjectFromHandle(pBufferMemoryBarriers[i].buffer); pBuffer->GetBarrierPolicy().ApplyBufferMemoryBarrier( GetQueueFamilyIndex(), pBufferMemoryBarriers[i], pNextMain); pNextMain->imageInfo.pImage = nullptr; VK_ASSERT(pBufferMemoryBarriers[i].pNext == nullptr); ++pNextMain; const uint32_t mainTransitionCount = static_cast(pNextMain - pTransitions); if (MaxPalAspectsPerMask + mainTransitionCount > MaxTransitionCount) { FlushBarriers(pBarrier, pTransitions, nullptr, mainTransitionCount); pNextMain = pTransitions; } } uint32_t locationIndex = 0; uint32_t locationCount = (imageMemoryBarrierCount > MaxLocationCount) ? MaxLocationCount : imageMemoryBarrierCount; Pal::MsaaQuadSamplePattern* pLocations = imageMemoryBarrierCount > 0 ? pVirtStackFrame->AllocArray(locationCount) : nullptr; for (uint32_t i = 0; i < imageMemoryBarrierCount; ++i) { const Image* pImage = Image::ObjectFromHandle(pImageMemoryBarriers[i].image); VkFormat format = pImage->GetFormat(); Pal::BarrierTransition barrierTransition = { 0 }; bool layoutChanging = false; Pal::ImageLayout oldLayouts[MaxPalAspectsPerMask]; Pal::ImageLayout newLayouts[MaxPalAspectsPerMask]; pImage->GetBarrierPolicy().ApplyImageMemoryBarrier( GetQueueFamilyIndex(), pImageMemoryBarriers[i], &barrierTransition, &layoutChanging, oldLayouts, newLayouts, true); uint32_t layoutIdx = 0; uint32_t palRangeIdx = 0; uint32_t palRangeCount = 0; Pal::SubresRange palRanges[MaxPalAspectsPerMask] = {}; VkToPalSubresRange( format, pImageMemoryBarriers[i].subresourceRange, pImage->GetMipLevels(), pImage->GetArraySize(), palRanges, &palRangeCount, m_pDevice->GetRuntimeSettings()); if (layoutChanging && Formats::HasStencil(format)) { if (palRangeCount == MaxPalDepthAspectsPerMask) { // Find the subset of an images subres ranges that need to be transitioned based changes between the // source and destination layouts. if ((oldLayouts[0].usages == newLayouts[0].usages) && (oldLayouts[0].engines == newLayouts[0].engines)) { // Skip the depth transition palRangeCount--; palRangeIdx++; layoutIdx++; } else if ((oldLayouts[1].usages == newLayouts[1].usages) && (oldLayouts[1].engines == newLayouts[1].engines)) { // Skip the stencil transition palRangeCount--; } } else if (pImageMemoryBarriers[i].subresourceRange.aspectMask & VK_IMAGE_ASPECT_STENCIL_BIT) { VK_ASSERT((pImageMemoryBarriers[i].subresourceRange.aspectMask & VK_IMAGE_ASPECT_DEPTH_BIT) == 0); // Always use the second layout for stencil transitions. It is the only valid one for combined depth // stencil layouts, and LayoutUsageHelper replicates stencil-only layouts to all aspects. layoutIdx++; } } VK_ASSERT(palRangeCount > 0 && palRangeCount <= MaxPalAspectsPerMask); const Image** pLocalImageTransition = pTransitionImages; Pal::BarrierTransition* const pDestTransition = pNextMain; pNextMain += palRangeCount; if (pTransitionImages != nullptr) { const size_t localOffset = (pDestTransition - pTransitions); for (uint32_t rangeIdx = 0; rangeIdx < palRangeCount; rangeIdx++) { pLocalImageTransition[localOffset + rangeIdx] = pImage; } } if (layoutChanging) { EXTRACT_VK_STRUCTURES_1( Barrier, ImageMemoryBarrier, SampleLocationsInfoEXT, &pImageMemoryBarriers[i], IMAGE_MEMORY_BARRIER, SAMPLE_LOCATIONS_INFO_EXT) const Pal::MsaaQuadSamplePattern* pQuadSamplePattern = nullptr; if ((pSampleLocationsInfoEXT != nullptr) && (pLocations != nullptr)) // Could be null due to an OOM error { VK_ASSERT(static_cast(pSampleLocationsInfoEXT->sType) == VK_STRUCTURE_TYPE_SAMPLE_LOCATIONS_INFO_EXT); VK_ASSERT(pImage->IsSampleLocationsCompatibleDepth()); ConvertToPalMsaaQuadSamplePattern(pSampleLocationsInfoEXT, &pLocations[locationIndex]); pQuadSamplePattern = &pLocations[locationIndex]; } for (uint32_t transitionIdx = 0; transitionIdx < palRangeCount; transitionIdx++) { pDestTransition[transitionIdx] = {}; pDestTransition[transitionIdx].srcCacheMask = barrierTransition.srcCacheMask; pDestTransition[transitionIdx].dstCacheMask = barrierTransition.dstCacheMask; pDestTransition[transitionIdx].imageInfo.pImage = pImage->PalImage(DefaultDeviceIndex); pDestTransition[transitionIdx].imageInfo.subresRange = palRanges[palRangeIdx]; pDestTransition[transitionIdx].imageInfo.oldLayout = oldLayouts[layoutIdx]; pDestTransition[transitionIdx].imageInfo.newLayout = newLayouts[layoutIdx]; pDestTransition[transitionIdx].imageInfo.pQuadSamplePattern = pQuadSamplePattern; layoutIdx++; palRangeIdx++; } if (pQuadSamplePattern != nullptr) { ++locationIndex; } } else { for (uint32_t transitionIdx = 0; transitionIdx < palRangeCount; transitionIdx++) { pDestTransition[transitionIdx] = {}; pDestTransition[transitionIdx].srcCacheMask = barrierTransition.srcCacheMask; pDestTransition[transitionIdx].dstCacheMask = barrierTransition.dstCacheMask; } } const uint32_t mainTransitionCount = static_cast(pNextMain - pTransitions); // Accounting for the maximum sub ranges, do we have enough space left for another image ? const bool full = ((MaxPalAspectsPerMask + mainTransitionCount) > MaxTransitionCount) || (locationIndex == locationCount); if (full) { FlushBarriers(pBarrier, pTransitions, pTransitionImages, mainTransitionCount); pNextMain = pTransitions; locationIndex = 0; } } const uint32_t mainTransitionCount = static_cast(pNextMain - pTransitions); FlushBarriers(pBarrier, pTransitions, pTransitionImages, mainTransitionCount); pVirtStackFrame->FreeArray(pLocations); if (pTransitionImages != nullptr) { pVirtStackFrame->FreeArray(pTransitionImages); } pVirtStackFrame->FreeArray(pTransitions); } // ===================================================================================================================== // Implementation of vkCmdWaitEvents() void CmdBuffer::WaitEvents( uint32_t eventCount, const VkEvent* pEvents, PipelineStageFlags srcStageMask, PipelineStageFlags dstStageMask, uint32_t memoryBarrierCount, const VkMemoryBarrier* pMemoryBarriers, uint32_t bufferMemoryBarrierCount, const VkBufferMemoryBarrier* pBufferMemoryBarriers, uint32_t imageMemoryBarrierCount, const VkImageMemoryBarrier* pImageMemoryBarriers) { DbgBarrierPreCmd(DbgBarrierPipelineBarrierWaitEvents); if (m_flags.useSplitReleaseAcquire) { uint32_t eventRangeCount = 0; for (uint32_t i = 0; i < eventCount; i += eventRangeCount) { eventRangeCount = 1; bool usesToken = Event::ObjectFromHandle(pEvents[i])->IsUseToken(); for (uint32_t j = i + 1; j < eventCount; j++) { if (Event::ObjectFromHandle(pEvents[j])->IsUseToken() == usesToken) { eventRangeCount++; } else { break; } } ExecuteAcquireRelease(eventRangeCount, pEvents + i, srcStageMask, dstStageMask, memoryBarrierCount, pMemoryBarriers, bufferMemoryBarrierCount, pBufferMemoryBarriers, imageMemoryBarrierCount, pImageMemoryBarriers, Acquire, RgpBarrierExternalCmdWaitEvents); } } else { VirtualStackFrame virtStackFrame(m_pStackAllocator); // Allocate space to store signaled event pointers (automatically rewound on unscope) const Pal::IGpuEvent** ppGpuEvents = virtStackFrame.AllocArray(NumDeviceEvents(eventCount)); if (ppGpuEvents != nullptr) { const uint32_t multiDeviceStride = eventCount; for (uint32_t i = 0; i < eventCount; ++i) { const Event* pEvent = Event::ObjectFromHandle(pEvents[i]); InsertDeviceEvents(ppGpuEvents, pEvent, i, multiDeviceStride); } Pal::BarrierInfo barrier = {}; // Tell PAL to wait at a specific point until the given set of GpuEvent objects is signaled. // We intentionally ignore the source stage flags (srcStagemask) as they are irrelevant in the // presence of event objects barrier.reason = RgpBarrierExternalCmdWaitEvents; barrier.waitPoint = VkToPalWaitPipePoint(dstStageMask); barrier.gpuEventWaitCount = eventCount; barrier.ppGpuEvents = ppGpuEvents; ExecuteBarriers(&virtStackFrame, memoryBarrierCount, pMemoryBarriers, bufferMemoryBarrierCount, pBufferMemoryBarriers, imageMemoryBarrierCount, pImageMemoryBarriers, &barrier); virtStackFrame.FreeArray(ppGpuEvents); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } DbgBarrierPostCmd(DbgBarrierPipelineBarrierWaitEvents); } // ===================================================================================================================== // Implementation of vkCmdWaitEvents2() void CmdBuffer::WaitEvents2( uint32_t eventCount, const VkEvent* pEvents, const VkDependencyInfoKHR* pDependencyInfos) { DbgBarrierPreCmd(DbgBarrierPipelineBarrierWaitEvents); // If the ASIC provides split CmdRelease()/CmdReleaseEvent() and CmdAcquire()/CmdAcquireEvent() to express barrier, // we will find range of gpu-only events and gpu events with cpu-access, we are assuming the case won't be to have // a mixture, it means we can find ranges in the event list that are sync token or not sync token, and then call // CmdAcquire() or CmdAcquireEvent() for each range. If the ASIC doesn't support it, we call // WaitEventsSync2ToSync1() for all events. if (m_flags.useSplitReleaseAcquire) { uint32_t eventRangeCount = 0; for (uint32_t i = 0; i < eventCount; i += eventRangeCount) { eventRangeCount = 1; bool usesToken = Event::ObjectFromHandle(pEvents[i])->IsUseToken(); for (uint32_t j = i + 1; j < eventCount; j++) { if (Event::ObjectFromHandle(pEvents[j])->IsUseToken() == usesToken) { eventRangeCount++; } else { break; } } ExecuteAcquireRelease2(eventRangeCount, pEvents + i, pDependencyInfos + i, Acquire, RgpBarrierExternalCmdWaitEvents); } } else { WaitEventsSync2ToSync1(eventCount, pEvents, eventCount, pDependencyInfos); } DbgBarrierPostCmd(DbgBarrierPipelineBarrierWaitEvents); } // ===================================================================================================================== // Implementation of WaitEvents2() void CmdBuffer::WaitEventsSync2ToSync1( uint32_t eventCount, const VkEvent* pEvents, uint32_t dependencyCount, const VkDependencyInfoKHR* pDependencyInfos) { VirtualStackFrame virtStackFrame(m_pStackAllocator); // Allocate space to store signaled event pointers (automatically rewound on unscope) const Pal::IGpuEvent** ppGpuEvents = virtStackFrame.AllocArray(NumDeviceEvents(eventCount)); if (ppGpuEvents != nullptr) { const uint32_t multiDeviceStride = eventCount; for (uint32_t i = 0; i < eventCount; ++i) { const Event* pEvent = Event::ObjectFromHandle(pEvents[i]); InsertDeviceEvents(ppGpuEvents, pEvent, i, multiDeviceStride); } for (uint32_t j = 0; j < dependencyCount; j++) { const VkDependencyInfoKHR* pThisDependencyInfo = &pDependencyInfos[j]; // convert structure VkDependencyInfoKHR to the formal parameters of WaitEvents PipelineStageFlags dstStageMask = 0; VkMemoryBarrier* pMemoryBarriers = pThisDependencyInfo->memoryBarrierCount > 0 ? virtStackFrame.AllocArray(pThisDependencyInfo->memoryBarrierCount) : nullptr; for (uint32_t i = 0; i < pThisDependencyInfo->memoryBarrierCount; i++) { dstStageMask |= pThisDependencyInfo->pMemoryBarriers[i].dstStageMask; pMemoryBarriers[i] = { VK_STRUCTURE_TYPE_MEMORY_BARRIER, pThisDependencyInfo->pMemoryBarriers[i].pNext, static_cast(pThisDependencyInfo->pMemoryBarriers[i].srcAccessMask), static_cast(pThisDependencyInfo->pMemoryBarriers[i].dstAccessMask) }; } VkBufferMemoryBarrier* pBufferMemoryBarriers = pThisDependencyInfo->bufferMemoryBarrierCount > 0 ? virtStackFrame.AllocArray(pThisDependencyInfo->bufferMemoryBarrierCount) : nullptr; for (uint32_t i = 0; i < pThisDependencyInfo->bufferMemoryBarrierCount; i++) { dstStageMask |= pThisDependencyInfo->pBufferMemoryBarriers[i].dstStageMask; pBufferMemoryBarriers[i] = { VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER, pThisDependencyInfo->pBufferMemoryBarriers[i].pNext, static_cast(pThisDependencyInfo->pBufferMemoryBarriers[i].srcAccessMask), static_cast(pThisDependencyInfo->pBufferMemoryBarriers[i].dstAccessMask), pThisDependencyInfo->pBufferMemoryBarriers[i].srcQueueFamilyIndex, pThisDependencyInfo->pBufferMemoryBarriers[i].dstQueueFamilyIndex, pThisDependencyInfo->pBufferMemoryBarriers[i].buffer, pThisDependencyInfo->pBufferMemoryBarriers[i].offset, pThisDependencyInfo->pBufferMemoryBarriers[i].size }; } VkImageMemoryBarrier* pImageMemoryBarriers = pThisDependencyInfo->imageMemoryBarrierCount > 0 ? virtStackFrame.AllocArray(pThisDependencyInfo->imageMemoryBarrierCount) : nullptr; for (uint32_t i = 0; i < pThisDependencyInfo->imageMemoryBarrierCount; i++) { dstStageMask |= pThisDependencyInfo->pImageMemoryBarriers[i].dstStageMask; pImageMemoryBarriers[i] = { VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, pThisDependencyInfo->pImageMemoryBarriers[i].pNext, static_cast(pThisDependencyInfo->pImageMemoryBarriers[i].srcAccessMask), static_cast(pThisDependencyInfo->pImageMemoryBarriers[i].dstAccessMask), pThisDependencyInfo->pImageMemoryBarriers[i].oldLayout, pThisDependencyInfo->pImageMemoryBarriers[i].newLayout, pThisDependencyInfo->pImageMemoryBarriers[i].srcQueueFamilyIndex, pThisDependencyInfo->pImageMemoryBarriers[i].dstQueueFamilyIndex, pThisDependencyInfo->pImageMemoryBarriers[i].image, pThisDependencyInfo->pImageMemoryBarriers[i].subresourceRange }; } Pal::BarrierInfo barrier = {}; barrier.reason = RgpBarrierExternalCmdWaitEvents; barrier.waitPoint = VkToPalWaitPipePoint(dstStageMask); barrier.gpuEventWaitCount = eventCount; barrier.ppGpuEvents = ppGpuEvents; ExecuteBarriers(&virtStackFrame, pThisDependencyInfo->memoryBarrierCount, pMemoryBarriers, pThisDependencyInfo->bufferMemoryBarrierCount, pBufferMemoryBarriers, pThisDependencyInfo->imageMemoryBarrierCount, pImageMemoryBarriers, &barrier); if (pMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pMemoryBarriers); } if (pBufferMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pBufferMemoryBarriers); } if (pImageMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pImageMemoryBarriers); } } virtStackFrame.FreeArray(ppGpuEvents); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } // ===================================================================================================================== // Helper function called from ExecuteAcquireRelease* to route barrier calls based on AcquireReleaseMode void CmdBuffer::FlushAcquireReleaseBarriers( Pal::AcquireReleaseInfo* pAcquireReleaseInfo, uint32_t eventCount, const VkEvent* pEvents, Pal::MemBarrier* const pBufferBarriers, const Buffer** const ppBuffers, Pal::ImgBarrier* const pImageBarriers, const Image** const ppImages, VirtualStackFrame* pVirtStackFrame, const AcquireReleaseMode acquireReleaseMode, uint32_t deviceMask) { if (acquireReleaseMode == Release) { pAcquireReleaseInfo->dstGlobalStageMask = 0; pAcquireReleaseInfo->dstGlobalAccessMask = 0; // If memoryBarrierCount is 0, set srcStageMask to Pal::PipelineStageTopOfPipe. if (pAcquireReleaseInfo->srcGlobalStageMask == 0) { pAcquireReleaseInfo->srcGlobalStageMask |= Pal::PipelineStageTopOfPipe; } for (uint32 i = 0; i < pAcquireReleaseInfo->memoryBarrierCount; i++) { pBufferBarriers[i].dstStageMask = 0; pBufferBarriers[i].dstAccessMask = 0; } for (uint32 i = 0; i < pAcquireReleaseInfo->imageBarrierCount; i++) { pImageBarriers[i].dstStageMask = 0; pImageBarriers[i].dstAccessMask = 0; } // The only possibility we are here would be as a result of vkCmdSetEvent2 in which case eventCount must be 1 VK_ASSERT(eventCount == 1); PalCmdRelease( pAcquireReleaseInfo, pEvents[0], pBufferBarriers, ppBuffers, pImageBarriers, ppImages, deviceMask); } else if (acquireReleaseMode == Acquire) { pAcquireReleaseInfo->srcGlobalStageMask = 0; pAcquireReleaseInfo->srcGlobalAccessMask = 0; for (uint32 i = 0; i < pAcquireReleaseInfo->memoryBarrierCount; i++) { pBufferBarriers[i].srcStageMask = 0; pBufferBarriers[i].srcAccessMask = 0; } for (uint32 i = 0; i < pAcquireReleaseInfo->imageBarrierCount; i++) { pImageBarriers[i].srcStageMask = 0; pImageBarriers[i].srcAccessMask = 0; } // The only possibility we are here would be as a result of vkCmdWaitEvents* in which case eventCount // must be non-zero VK_ASSERT(eventCount != 0); PalCmdAcquire( pAcquireReleaseInfo, eventCount, pEvents, pBufferBarriers, ppBuffers, pImageBarriers, ppImages, pVirtStackFrame, deviceMask); } else { PalCmdReleaseThenAcquire( pAcquireReleaseInfo, pBufferBarriers, ppBuffers, pImageBarriers, ppImages, deviceMask); } } // ===================================================================================================================== // Based on Dependency Info, execute Acquire or Release according to the mode. This funtion handles the // VK_KHR_synchronization2 barrier API calls void CmdBuffer::ExecuteAcquireRelease2( uint32_t dependencyCount, const VkEvent* pEvents, const VkDependencyInfoKHR* pDependencyInfos, const AcquireReleaseMode acquireReleaseMode, uint32_t rgpBarrierReasonType) { VK_ASSERT((acquireReleaseMode == ReleaseThenAcquire) || (pEvents != nullptr)); const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); uint32_t barrierCount = 0; uint32_t maxBufferMemoryBarriers = 0; uint32_t maxImageMemoryBarriers = 0; for (uint32_t i = 0; i < dependencyCount; i++) { barrierCount += pDependencyInfos[i].memoryBarrierCount + pDependencyInfos[i].bufferMemoryBarrierCount + pDependencyInfos[i].imageMemoryBarrierCount; // Determine the maximum number of buffer and image barriers among all the dependency infos passed in maxBufferMemoryBarriers = Util::Max(pDependencyInfos[i].bufferMemoryBarrierCount, maxBufferMemoryBarriers); maxImageMemoryBarriers = Util::Max(pDependencyInfos[i].imageMemoryBarrierCount, maxImageMemoryBarriers); } if ((pEvents != nullptr) || (barrierCount > 0)) { VirtualStackFrame virtStackFrame(m_pStackAllocator); constexpr uint32_t MaxTransitionCount = 512; constexpr uint32_t MaxSampleLocationCount = 128; // Keeps track of the number of barriers for which info has already been // stored in Pal::AcquireReleaseInfo uint32_t memoryBarrierIdx = 0; uint32_t bufferMemoryBarrierIdx = 0; uint32_t imageMemoryBarrierIdx = 0; uint32_t maxLocationCount = Util::Min(maxImageMemoryBarriers, MaxSampleLocationCount); uint32_t maxBufferBarrierCount = Util::Min(maxBufferMemoryBarriers, MaxTransitionCount); uint32_t maxImageBarrierCount = Util::Min((MaxPalAspectsPerMask * maxImageMemoryBarriers) + 1, MaxTransitionCount); Pal::MemBarrier* pPalBufferMemoryBarriers = (maxBufferMemoryBarriers > 0) ? virtStackFrame.AllocArray(maxBufferBarrierCount) : nullptr; const Buffer** ppBuffers = (maxBufferMemoryBarriers > 0) ? virtStackFrame.AllocArray(maxBufferBarrierCount) : nullptr; Pal::ImgBarrier* pPalImageBarriers = (maxImageMemoryBarriers > 0) ? virtStackFrame.AllocArray(maxImageBarrierCount) : nullptr; const Image** ppImages = (maxImageMemoryBarriers > 0) ? virtStackFrame.AllocArray(maxImageBarrierCount) : nullptr; Pal::MsaaQuadSamplePattern* pLocations = (maxImageMemoryBarriers > 0) ? virtStackFrame.AllocArray(maxLocationCount) : nullptr; const bool bufferAllocSuccess = (((maxBufferMemoryBarriers > 0) && (pPalBufferMemoryBarriers != nullptr) && (ppBuffers != nullptr)) || (maxBufferMemoryBarriers == 0)); const bool imageAllocSuccess = (((maxImageMemoryBarriers > 0) && (pPalImageBarriers != nullptr) && (ppImages != nullptr) && (pLocations != nullptr)) || (maxImageMemoryBarriers == 0)); if (bufferAllocSuccess && imageAllocSuccess) { for (uint32_t j = 0; j < dependencyCount; j++) { const VkDependencyInfoKHR* pThisDependencyInfo = &pDependencyInfos[j]; uint32_t memBarrierCount = pThisDependencyInfo->memoryBarrierCount; uint32_t bufferMemoryBarrierCount = pThisDependencyInfo->bufferMemoryBarrierCount; uint32_t imageMemoryBarrierCount = pThisDependencyInfo->imageMemoryBarrierCount; while ((memoryBarrierIdx < memBarrierCount) || (bufferMemoryBarrierIdx < bufferMemoryBarrierCount) || (imageMemoryBarrierIdx < imageMemoryBarrierCount)) { Pal::AcquireReleaseInfo acquireReleaseInfo = {}; acquireReleaseInfo.pMemoryBarriers = pPalBufferMemoryBarriers; acquireReleaseInfo.pImageBarriers = pPalImageBarriers; acquireReleaseInfo.reason = rgpBarrierReasonType; uint32_t locationIndex = 0; while (memoryBarrierIdx < memBarrierCount) { Pal::BarrierTransition tempTransition = {}; const VkMemoryBarrier2& memoryBarrier = pThisDependencyInfo->pMemoryBarriers[memoryBarrierIdx]; acquireReleaseInfo.srcGlobalStageMask |= VkToPalPipelineStageFlags(memoryBarrier.srcStageMask, true); acquireReleaseInfo.dstGlobalStageMask |= VkToPalPipelineStageFlags(memoryBarrier.dstStageMask, false); VkAccessFlags2KHR srcAccessMask = memoryBarrier.srcAccessMask; VkAccessFlags2KHR dstAccessMask = memoryBarrier.dstAccessMask; m_pDevice->GetBarrierPolicy().ApplyBarrierCacheFlags( srcAccessMask, dstAccessMask, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, &tempTransition); acquireReleaseInfo.srcGlobalAccessMask |= tempTransition.srcCacheMask; acquireReleaseInfo.dstGlobalAccessMask |= tempTransition.dstCacheMask; memoryBarrierIdx++; } while ((acquireReleaseInfo.memoryBarrierCount < maxBufferBarrierCount) && (bufferMemoryBarrierIdx < bufferMemoryBarrierCount)) { Pal::BarrierTransition tempTransition = {}; const VkBufferMemoryBarrier2& bufferMemoryBarrier = pThisDependencyInfo->pBufferMemoryBarriers[bufferMemoryBarrierIdx]; const Buffer* pBuffer = Buffer::ObjectFromHandle(bufferMemoryBarrier.buffer); pBuffer->GetBarrierPolicy().ApplyBufferMemoryBarrier( GetQueueFamilyIndex(), bufferMemoryBarrier, &tempTransition); Pal::MemBarrier* const pBarrier = &pPalBufferMemoryBarriers[acquireReleaseInfo.memoryBarrierCount]; *pBarrier = {}; pBarrier->srcStageMask = VkToPalPipelineStageFlags(bufferMemoryBarrier.srcStageMask, true); pBarrier->dstStageMask = VkToPalPipelineStageFlags(bufferMemoryBarrier.dstStageMask, false); pBarrier->srcAccessMask = tempTransition.srcCacheMask; pBarrier->dstAccessMask = tempTransition.dstCacheMask; ppBuffers[acquireReleaseInfo.memoryBarrierCount] = pBuffer; acquireReleaseInfo.memoryBarrierCount++; bufferMemoryBarrierIdx++; } // Accounting for the max sub ranges, if we do not have enough space left for another image, // break from this loop. The info for remaining barriers will be passed to PAL in subsequent calls. while (((MaxPalAspectsPerMask + acquireReleaseInfo.imageBarrierCount) < maxImageBarrierCount) && (locationIndex < maxLocationCount) && (imageMemoryBarrierIdx < imageMemoryBarrierCount)) { Pal::BarrierTransition tempTransition = {}; const VkImageMemoryBarrier2& imageMemoryBarrier = pThisDependencyInfo->pImageMemoryBarriers[imageMemoryBarrierIdx]; bool layoutChanging = false; Pal::ImageLayout oldLayouts[MaxPalAspectsPerMask]; Pal::ImageLayout newLayouts[MaxPalAspectsPerMask]; const Image* pImage = Image::ObjectFromHandle(imageMemoryBarrier.image); // Synchronization2 will use new PAL interfaces CmdAcquire(), CmdRelease() and // CmdReleaseThenAcquire() to execute barrier, Under these interfaces, vulkan driver does not // need to add an optimization for Image barrier with the same oldLayout & newLayout, like // VK_IMAGE_LAYOUT_GENERAL to VK_IMAGE_LAYOUT_GENERAL. PAL should not be doing any transition // logic and only flush/invalidate caches as apporiate. So we make use of the template flag // skipMatchingLayouts to skip this if-checking for the same layout change by setting the flag // skipMatchingLayouts to false. As for legacy synchronization, we should be careful of this // change, maybe will have some potential regressions, so currently we keep this optimization // unchanged by setting this flag to true. With the iterative update of vulkan driver, we should // also remove this optimization for legacy synchronization. pImage->GetBarrierPolicy().ApplyImageMemoryBarrier( GetQueueFamilyIndex(), imageMemoryBarrier, &tempTransition, &layoutChanging, oldLayouts, newLayouts, false); VkFormat format = pImage->GetFormat(); uint32_t layoutIdx = 0; uint32_t palRangeIdx = 0; uint32_t palRangeCount = 0; Pal::SubresRange palRanges[MaxPalAspectsPerMask]; VkToPalSubresRange( format, imageMemoryBarrier.subresourceRange, pImage->GetMipLevels(), pImage->GetArraySize(), palRanges, &palRangeCount, settings); if (Formats::HasStencil(format)) { const VkImageAspectFlags aspectMask = imageMemoryBarrier.subresourceRange.aspectMask; // Always use the second layout for stencil transitions. It is the only valid one for // combined depth stencil layouts, and LayoutUsageHelper replicates stencil-only layouts to // all aspects. if ((aspectMask & VK_IMAGE_ASPECT_STENCIL_BIT) && ((aspectMask & VK_IMAGE_ASPECT_DEPTH_BIT) == 0)) { layoutIdx++; } } EXTRACT_VK_STRUCTURES_1( Barrier, ImageMemoryBarrier2KHR, SampleLocationsInfoEXT, &imageMemoryBarrier, IMAGE_MEMORY_BARRIER_2_KHR, SAMPLE_LOCATIONS_INFO_EXT) const Pal::MsaaQuadSamplePattern* pQuadSamplePattern = nullptr; if (pSampleLocationsInfoEXT != nullptr) { VK_ASSERT(static_cast(pSampleLocationsInfoEXT->sType) == VK_STRUCTURE_TYPE_SAMPLE_LOCATIONS_INFO_EXT); VK_ASSERT(pImage->IsSampleLocationsCompatibleDepth()); ConvertToPalMsaaQuadSamplePattern(pSampleLocationsInfoEXT, &pLocations[locationIndex]); pQuadSamplePattern = &pLocations[locationIndex]; } for (uint32_t transitionIdx = 0; transitionIdx < palRangeCount; transitionIdx++) { Pal::ImgBarrier* const pBarrier = &pPalImageBarriers[acquireReleaseInfo.imageBarrierCount]; *pBarrier = {}; pBarrier->srcStageMask = VkToPalPipelineStageFlags(imageMemoryBarrier.srcStageMask, true); pBarrier->dstStageMask = VkToPalPipelineStageFlags(imageMemoryBarrier.dstStageMask, false); pBarrier->srcAccessMask = tempTransition.srcCacheMask; pBarrier->dstAccessMask = tempTransition.dstCacheMask; // We set the pImage to nullptr by default here. But, this will be computed correctly later // for each device including DefaultDeviceIndex based on the deviceId. pBarrier->pImage = nullptr; pBarrier->subresRange = palRanges[palRangeIdx]; pBarrier->oldLayout = oldLayouts[layoutIdx]; pBarrier->newLayout = newLayouts[layoutIdx]; pBarrier->pQuadSamplePattern = pQuadSamplePattern; ppImages[acquireReleaseInfo.imageBarrierCount] = pImage; acquireReleaseInfo.imageBarrierCount++; layoutIdx++; palRangeIdx++; } if (pQuadSamplePattern != nullptr) { ++locationIndex; } imageMemoryBarrierIdx++; } FlushAcquireReleaseBarriers( &acquireReleaseInfo, ((pEvents != nullptr) ? 1u : 0u), ((pEvents != nullptr) ? &pEvents[j] : nullptr), pPalBufferMemoryBarriers, ppBuffers, pPalImageBarriers, ppImages, &virtStackFrame, acquireReleaseMode, m_curDeviceMask); } } } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } if (pPalBufferMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pPalBufferMemoryBarriers); } if (ppBuffers != nullptr) { virtStackFrame.FreeArray(ppBuffers); } if (pPalImageBarriers != nullptr) { virtStackFrame.FreeArray(pPalImageBarriers); } if (ppImages != nullptr) { virtStackFrame.FreeArray(ppImages); } if (pLocations != nullptr) { virtStackFrame.FreeArray(pLocations); } } } // ===================================================================================================================== // Records acquire-release barriers into PAL structures and passes them to PAL. This funtion handles the // Synchronization_1 barrier API calls void CmdBuffer::ExecuteAcquireRelease( uint32_t eventCount, const VkEvent* pEvents, PipelineStageFlags srcStageMask, PipelineStageFlags dstStageMask, uint32_t memBarrierCount, const VkMemoryBarrier* pMemoryBarriers, uint32_t bufferMemoryBarrierCount, const VkBufferMemoryBarrier* pBufferMemoryBarriers, uint32_t imageMemoryBarrierCount, const VkImageMemoryBarrier* pImageMemoryBarriers, const AcquireReleaseMode acquireReleaseMode, uint32_t rgpBarrierReasonType) { VK_ASSERT((acquireReleaseMode == ReleaseThenAcquire) || (pEvents != nullptr)); if ((memBarrierCount + bufferMemoryBarrierCount + imageMemoryBarrierCount + eventCount) > 0) { VirtualStackFrame virtStackFrame(m_pStackAllocator); const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); constexpr uint32_t MaxTransitionCount = 512; constexpr uint32_t MaxSampleLocationCount = 128; // Keeps track of the number of barriers for which info has already been // stored in Pal::AcquireReleaseInfo uint32_t memoryBarrierIdx = 0; uint32_t bufferMemoryBarrierIdx = 0; uint32_t imageMemoryBarrierIdx = 0; uint32_t gpuEventCount = eventCount; uint32_t maxLocationCount = Util::Min(imageMemoryBarrierCount, MaxSampleLocationCount); uint32_t maxBufferBarrierCount = Util::Min(bufferMemoryBarrierCount, MaxTransitionCount); uint32_t maxImageBarrierCount = Util::Min((MaxPalAspectsPerMask * imageMemoryBarrierCount) + 1, MaxTransitionCount); Pal::MemBarrier* pPalBufferMemoryBarriers = (bufferMemoryBarrierCount > 0) ? virtStackFrame.AllocArray(maxBufferBarrierCount) : nullptr; const Buffer** ppBuffers = (bufferMemoryBarrierCount > 0) ? virtStackFrame.AllocArray(maxBufferBarrierCount) : nullptr; Pal::ImgBarrier* pPalImageBarriers = (imageMemoryBarrierCount > 0) ? virtStackFrame.AllocArray(maxImageBarrierCount) : nullptr; Pal::MsaaQuadSamplePattern* pLocations = (imageMemoryBarrierCount > 0) ? virtStackFrame.AllocArray(maxLocationCount) : nullptr; const Image** ppImages = (imageMemoryBarrierCount > 0) ? virtStackFrame.AllocArray(maxImageBarrierCount) : nullptr; const bool bufferAllocSuccess = (((bufferMemoryBarrierCount > 0) && (pPalBufferMemoryBarriers != nullptr) && (ppBuffers != nullptr)) || (bufferMemoryBarrierCount == 0)); const bool imageAllocSuccess = (((imageMemoryBarrierCount > 0) && (pPalImageBarriers != nullptr) && (ppImages != nullptr) && (pLocations != nullptr)) || (imageMemoryBarrierCount == 0)); if (bufferAllocSuccess && imageAllocSuccess) { while ((memoryBarrierIdx < memBarrierCount) || (bufferMemoryBarrierIdx < bufferMemoryBarrierCount) || (imageMemoryBarrierIdx < imageMemoryBarrierCount) || (gpuEventCount > 0)) { Pal::AcquireReleaseInfo acquireReleaseInfo = {}; acquireReleaseInfo.pMemoryBarriers = pPalBufferMemoryBarriers; acquireReleaseInfo.pImageBarriers = pPalImageBarriers; acquireReleaseInfo.reason = rgpBarrierReasonType; uint32_t palSrcStageMask = VkToPalPipelineStageFlags(srcStageMask, true); uint32_t palDstStageMask = VkToPalPipelineStageFlags(dstStageMask, false); uint32_t locationIndex = 0; while (memoryBarrierIdx < memBarrierCount) { Pal::BarrierTransition tempTransition = {}; VkAccessFlags srcAccessMask = pMemoryBarriers[memoryBarrierIdx].srcAccessMask; VkAccessFlags dstAccessMask = pMemoryBarriers[memoryBarrierIdx].dstAccessMask; m_pDevice->GetBarrierPolicy().ApplyBarrierCacheFlags( srcAccessMask, dstAccessMask, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, &tempTransition); acquireReleaseInfo.srcGlobalStageMask = palSrcStageMask; acquireReleaseInfo.dstGlobalStageMask = palDstStageMask; acquireReleaseInfo.srcGlobalAccessMask |= tempTransition.srcCacheMask; acquireReleaseInfo.dstGlobalAccessMask |= tempTransition.dstCacheMask; memoryBarrierIdx++; } while ((acquireReleaseInfo.memoryBarrierCount < maxBufferBarrierCount) && (bufferMemoryBarrierIdx < bufferMemoryBarrierCount)) { Pal::BarrierTransition tempTransition = {}; const Buffer* pBuffer = Buffer::ObjectFromHandle( pBufferMemoryBarriers[bufferMemoryBarrierIdx].buffer); pBuffer->GetBarrierPolicy().ApplyBufferMemoryBarrier( GetQueueFamilyIndex(), pBufferMemoryBarriers[bufferMemoryBarrierIdx], &tempTransition); Pal::MemBarrier* const pBarrier = &pPalBufferMemoryBarriers[acquireReleaseInfo.memoryBarrierCount]; *pBarrier = {}; pBarrier->srcStageMask = palSrcStageMask; pBarrier->dstStageMask = palDstStageMask; pBarrier->srcAccessMask = tempTransition.srcCacheMask; pBarrier->dstAccessMask = tempTransition.dstCacheMask; ppBuffers[acquireReleaseInfo.memoryBarrierCount] = pBuffer; acquireReleaseInfo.memoryBarrierCount++; bufferMemoryBarrierIdx++; } // Accounting for the max sub ranges, if we do not have enough space left for another image, // break from this loop. The info for remaining barriers will be passed to PAL in subsequent calls. while (((MaxPalAspectsPerMask + acquireReleaseInfo.imageBarrierCount) < maxImageBarrierCount) && (locationIndex < maxLocationCount) && (imageMemoryBarrierIdx < imageMemoryBarrierCount)) { Pal::BarrierTransition tempTransition = {}; bool layoutChanging = false; Pal::ImageLayout oldLayouts[MaxPalAspectsPerMask]; Pal::ImageLayout newLayouts[MaxPalAspectsPerMask]; const Image* pImage = Image::ObjectFromHandle(pImageMemoryBarriers[imageMemoryBarrierIdx].image); VkImageMemoryBarrier localBarrier = pImageMemoryBarriers[imageMemoryBarrierIdx]; if (m_flags.useBackupBuffer) { // If the backup cmd buffer is being used, the meaning of queue family indexes is reversed: // backupQueueFamilyIndex is the original (replaced) index, and queueFamilyIndex is the backup // index. We need to fix the barrier queues, or the ownership transfer won't make sense. if (localBarrier.srcQueueFamilyIndex == m_backupQueueFamilyIndex) { localBarrier.srcQueueFamilyIndex = m_queueFamilyIndex; } if (localBarrier.dstQueueFamilyIndex == m_backupQueueFamilyIndex) { localBarrier.dstQueueFamilyIndex = m_queueFamilyIndex; } } // When using CmdReleaseThenAcquire() to execute barriers, vulkan driver does not need to add an // optimization for Image barrier with the same oldLayout & newLayout,like VK_IMAGE_LAYOUT_GENERAL // to VK_IMAGE_LAYOUT_GENERAL. PAL should not be doing any transition logic and only flush or // invalidate caches as apporiate. so we make use of the template flag skipMatchingLayouts to skip // this if-checking for the same layout change by setting the flag skipMatchingLayouts to false. pImage->GetBarrierPolicy().ApplyImageMemoryBarrier( GetQueueFamilyIndex(), localBarrier, &tempTransition, &layoutChanging, oldLayouts, newLayouts, false); VkFormat format = pImage->GetFormat(); uint32_t layoutIdx = 0; uint32_t palRangeIdx = 0; uint32_t palRangeCount = 0; Pal::SubresRange palRanges[MaxPalAspectsPerMask]; VkToPalSubresRange( format, pImageMemoryBarriers[imageMemoryBarrierIdx].subresourceRange, pImage->GetMipLevels(), pImage->GetArraySize(), palRanges, &palRangeCount, settings); if (Formats::HasStencil(format)) { const VkImageAspectFlags aspectMask = pImageMemoryBarriers[imageMemoryBarrierIdx].subresourceRange.aspectMask; // Always use the second layout for stencil transitions. It is the only valid one for combined // depth stencil layouts, and LayoutUsageHelper replicates stencil-only layouts to all aspects. if ((aspectMask & VK_IMAGE_ASPECT_STENCIL_BIT) && ((aspectMask & VK_IMAGE_ASPECT_DEPTH_BIT) == 0)) { layoutIdx++; } } EXTRACT_VK_STRUCTURES_1( Barrier, ImageMemoryBarrier, SampleLocationsInfoEXT, &pImageMemoryBarriers[imageMemoryBarrierIdx], IMAGE_MEMORY_BARRIER, SAMPLE_LOCATIONS_INFO_EXT) const Pal::MsaaQuadSamplePattern* pQuadSamplePattern = nullptr; if (pSampleLocationsInfoEXT != nullptr) { VK_ASSERT(static_cast(pSampleLocationsInfoEXT->sType) == VK_STRUCTURE_TYPE_SAMPLE_LOCATIONS_INFO_EXT); VK_ASSERT(pImage->IsSampleLocationsCompatibleDepth()); ConvertToPalMsaaQuadSamplePattern(pSampleLocationsInfoEXT, &pLocations[locationIndex]); pQuadSamplePattern = &pLocations[locationIndex]; } for (uint32_t transitionIdx = 0; transitionIdx < palRangeCount; transitionIdx++) { Pal::ImgBarrier* const pBarrier = &pPalImageBarriers[acquireReleaseInfo.imageBarrierCount]; *pBarrier = {}; pBarrier->srcStageMask = palSrcStageMask; pBarrier->dstStageMask = palDstStageMask; pBarrier->srcAccessMask = tempTransition.srcCacheMask; pBarrier->dstAccessMask = tempTransition.dstCacheMask; // We set the pImage to nullptr by default here. But, this will be computed correctly later for // each device including DefaultDeviceIndex based on the deviceId. pBarrier->pImage = nullptr; pBarrier->subresRange = palRanges[palRangeIdx]; pBarrier->oldLayout = oldLayouts[layoutIdx]; pBarrier->newLayout = newLayouts[layoutIdx]; pBarrier->pQuadSamplePattern = pQuadSamplePattern; ppImages[acquireReleaseInfo.imageBarrierCount] = pImage; acquireReleaseInfo.imageBarrierCount++; layoutIdx++; palRangeIdx++; } if (pQuadSamplePattern != nullptr) { ++locationIndex; } imageMemoryBarrierIdx++; } FlushAcquireReleaseBarriers( &acquireReleaseInfo, gpuEventCount, pEvents, pPalBufferMemoryBarriers, ppBuffers, pPalImageBarriers, ppImages, &virtStackFrame, acquireReleaseMode, m_curDeviceMask); gpuEventCount = 0; } } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } if (pPalBufferMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pPalBufferMemoryBarriers); } if (ppBuffers != nullptr) { virtStackFrame.FreeArray(ppBuffers); } if (pPalImageBarriers != nullptr) { virtStackFrame.FreeArray(pPalImageBarriers); } if (ppImages != nullptr) { virtStackFrame.FreeArray(ppImages); } if (pLocations != nullptr) { virtStackFrame.FreeArray(pLocations); } } } // ===================================================================================================================== // Implements of vkCmdPipelineBarrier() void CmdBuffer::PipelineBarrier( PipelineStageFlags srcStageMask, PipelineStageFlags destStageMask, uint32_t memBarrierCount, const VkMemoryBarrier* pMemoryBarriers, uint32_t bufferMemoryBarrierCount, const VkBufferMemoryBarrier* pBufferMemoryBarriers, uint32_t imageMemoryBarrierCount, const VkImageMemoryBarrier* pImageMemoryBarriers) { DbgBarrierPreCmd(DbgBarrierPipelineBarrierWaitEvents); const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); if (settings.syncPreviousDrawForTransferStage && (srcStageMask == VK_PIPELINE_STAGE_TRANSFER_BIT) && (destStageMask == VK_PIPELINE_STAGE_TRANSFER_BIT)) { srcStageMask |= (VK_PIPELINE_STAGE_2_LATE_FRAGMENT_TESTS_BIT_KHR | VK_PIPELINE_STAGE_2_COLOR_ATTACHMENT_OUTPUT_BIT_KHR); } if (m_flags.useReleaseAcquire) { ExecuteAcquireRelease(0, nullptr, srcStageMask, destStageMask, memBarrierCount, pMemoryBarriers, bufferMemoryBarrierCount, pBufferMemoryBarriers, imageMemoryBarrierCount, pImageMemoryBarriers, ReleaseThenAcquire, RgpBarrierExternalCmdPipelineBarrier); } else { VirtualStackFrame virtStackFrame(m_pStackAllocator); Pal::BarrierInfo barrier = {}; // Tell PAL to wait at a specific point until the given set of pipeline events has been signaled (this version // does not use GpuEvent objects). barrier.reason = RgpBarrierExternalCmdPipelineBarrier; barrier.waitPoint = VkToPalWaitPipePoint(destStageMask); // Collect signal pipe points. Pal::HwPipePoint pipePoints[MaxHwPipePoints]; barrier.pipePointWaitCount = VkToPalSrcPipePoints(srcStageMask, pipePoints); barrier.pPipePoints = pipePoints; ExecuteBarriers(&virtStackFrame, memBarrierCount, pMemoryBarriers, bufferMemoryBarrierCount, pBufferMemoryBarriers, imageMemoryBarrierCount, pImageMemoryBarriers, &barrier); } DbgBarrierPostCmd(DbgBarrierPipelineBarrierWaitEvents); } // ===================================================================================================================== // Implements of vkCmdPipelineBarrier2() void CmdBuffer::PipelineBarrier2( const VkDependencyInfoKHR* pDependencyInfo) { DbgBarrierPreCmd(DbgBarrierPipelineBarrierWaitEvents); if (m_flags.useReleaseAcquire) { ExecuteAcquireRelease2(1, nullptr, pDependencyInfo, ReleaseThenAcquire, RgpBarrierExternalCmdPipelineBarrier); } else { PipelineBarrierSync2ToSync1(pDependencyInfo); } DbgBarrierPostCmd(DbgBarrierPipelineBarrierWaitEvents); } // ===================================================================================================================== // Implements of PipelineBarrier2 void CmdBuffer::PipelineBarrierSync2ToSync1( const VkDependencyInfoKHR* pDependencyInfo) { VirtualStackFrame virtStackFrame(m_pStackAllocator); // convert structure VkDependencyInfoKHR to the formal parameters of PipelineBarrier VK_ASSERT((pDependencyInfo->memoryBarrierCount + pDependencyInfo->bufferMemoryBarrierCount + pDependencyInfo->imageMemoryBarrierCount) != 0); PipelineStageFlags srcStageMask = 0; PipelineStageFlags dstStageMask = 0; VkMemoryBarrier* pMemoryBarriers = pDependencyInfo->memoryBarrierCount > 0 ? virtStackFrame.AllocArray(pDependencyInfo->memoryBarrierCount) : nullptr; for (uint32_t i = 0; i < pDependencyInfo->memoryBarrierCount; i++) { srcStageMask |= pDependencyInfo->pMemoryBarriers[i].srcStageMask; dstStageMask |= pDependencyInfo->pMemoryBarriers[i].dstStageMask; pMemoryBarriers[i] = { VK_STRUCTURE_TYPE_MEMORY_BARRIER, pDependencyInfo->pMemoryBarriers[i].pNext, static_cast(pDependencyInfo->pMemoryBarriers[i].srcAccessMask), static_cast(pDependencyInfo->pMemoryBarriers[i].dstAccessMask) }; } VkBufferMemoryBarrier* pBufferMemoryBarriers = pDependencyInfo->bufferMemoryBarrierCount > 0 ? virtStackFrame.AllocArray(pDependencyInfo->bufferMemoryBarrierCount) : nullptr; for (uint32_t i = 0; i < pDependencyInfo->bufferMemoryBarrierCount; i++) { srcStageMask |= pDependencyInfo->pBufferMemoryBarriers[i].srcStageMask; dstStageMask |= pDependencyInfo->pBufferMemoryBarriers[i].dstStageMask; pBufferMemoryBarriers[i] = { VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER, pDependencyInfo->pBufferMemoryBarriers[i].pNext, static_cast(pDependencyInfo->pBufferMemoryBarriers[i].srcAccessMask), static_cast(pDependencyInfo->pBufferMemoryBarriers[i].dstAccessMask), pDependencyInfo->pBufferMemoryBarriers[i].srcQueueFamilyIndex, pDependencyInfo->pBufferMemoryBarriers[i].dstQueueFamilyIndex, pDependencyInfo->pBufferMemoryBarriers[i].buffer, pDependencyInfo->pBufferMemoryBarriers[i].offset, pDependencyInfo->pBufferMemoryBarriers[i].size }; } VkImageMemoryBarrier* pImageMemoryBarriers = pDependencyInfo->imageMemoryBarrierCount > 0 ? virtStackFrame.AllocArray(pDependencyInfo->imageMemoryBarrierCount) : nullptr; for (uint32_t i = 0; i < pDependencyInfo->imageMemoryBarrierCount; i++) { srcStageMask |= pDependencyInfo->pImageMemoryBarriers[i].srcStageMask; dstStageMask |= pDependencyInfo->pImageMemoryBarriers[i].dstStageMask; pImageMemoryBarriers[i] = { VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, pDependencyInfo->pImageMemoryBarriers[i].pNext, static_cast(pDependencyInfo->pImageMemoryBarriers[i].srcAccessMask), static_cast(pDependencyInfo->pImageMemoryBarriers[i].dstAccessMask), pDependencyInfo->pImageMemoryBarriers[i].oldLayout, pDependencyInfo->pImageMemoryBarriers[i].newLayout, pDependencyInfo->pImageMemoryBarriers[i].srcQueueFamilyIndex, pDependencyInfo->pImageMemoryBarriers[i].dstQueueFamilyIndex, pDependencyInfo->pImageMemoryBarriers[i].image, pDependencyInfo->pImageMemoryBarriers[i].subresourceRange }; } Pal::BarrierInfo barrier = {}; // Tell PAL to wait at a specific point until the given set of pipeline events has been signaled (this version // does not use GpuEvent objects). barrier.reason = RgpBarrierExternalCmdPipelineBarrier; barrier.waitPoint = VkToPalWaitPipePoint(dstStageMask); // Collect signal pipe points. Pal::HwPipePoint pipePoints[MaxHwPipePoints]; barrier.pipePointWaitCount = VkToPalSrcPipePoints(srcStageMask, pipePoints); barrier.pPipePoints = pipePoints; ExecuteBarriers(&virtStackFrame, pDependencyInfo->memoryBarrierCount, pMemoryBarriers, pDependencyInfo->bufferMemoryBarrierCount, pBufferMemoryBarriers, pDependencyInfo->imageMemoryBarrierCount, pImageMemoryBarriers, &barrier); if (pMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pMemoryBarriers); } if (pBufferMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pBufferMemoryBarriers); } if (pImageMemoryBarriers != nullptr) { virtStackFrame.FreeArray(pImageMemoryBarriers); } } // ===================================================================================================================== void CmdBuffer::BeginQueryIndexed( VkQueryPool queryPool, uint32_t query, VkQueryControlFlags flags, uint32_t index) { DbgBarrierPreCmd(DbgBarrierQueryBeginEnd); const QueryPool* pBasePool = QueryPool::ObjectFromHandle(queryPool); { const auto palQueryControlFlags = VkToPalQueryControlFlags(pBasePool->GetQueryType(), flags); // NOTE: This function is illegal to call for TimestampQueryPools and AccelerationStructureQueryPools const PalQueryPool* pQueryPool = pBasePool->AsPalQueryPool(); Pal::QueryType queryType = pQueryPool->PalQueryType(); if (queryType == Pal::QueryType::StreamoutStats) { queryType = static_cast(static_cast(queryType) + index); } utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdBeginQuery(*pQueryPool->PalPool(deviceIdx), queryType, query, palQueryControlFlags); } while (deviceGroup.IterateNext()); const auto* const pRenderPass = m_allGpuState.pRenderPass; // If queries are used while executing a render pass instance that has multiview enabled, // the query uses N consecutive query indices in the query pool (starting at query) where // N is the number of bits set in the view mask in the subpass the query is used in. // // Implementations may write the total result to the first query and // write zero to the other queries. if (((UsingDynamicRendering() == false) && pRenderPass->IsMultiviewEnabled()) || (m_allGpuState.dynamicRenderingInstance.viewMask != 0)) { const auto viewMask = (pRenderPass != nullptr) ? pRenderPass->GetViewMask(m_renderPassInstance.subpass) : m_allGpuState.dynamicRenderingInstance.viewMask; const auto viewCount = Util::CountSetBits(viewMask); // Call Begin() and immediately call End() for all remaining queries, // to set value of each remaining query to 0 and to make them avaliable. for (uint32_t remainingQuery = 1; remainingQuery < viewCount; ++remainingQuery) { const auto remainingQueryIndex = query + remainingQuery; utils::IterateMask multiviewDeviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = multiviewDeviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdBeginQuery( *pQueryPool->PalPool(deviceIdx), pQueryPool->PalQueryType(), remainingQueryIndex, palQueryControlFlags); PalCmdBuffer(deviceIdx)->CmdEndQuery( *pQueryPool->PalPool(deviceIdx), pQueryPool->PalQueryType(), remainingQueryIndex); } while (multiviewDeviceGroup.IterateNext()); } } } DbgBarrierPostCmd(DbgBarrierQueryBeginEnd); } // ===================================================================================================================== void CmdBuffer::EndQueryIndexed( VkQueryPool queryPool, uint32_t query, uint32_t index) { DbgBarrierPreCmd(DbgBarrierQueryBeginEnd); const QueryPool* pBasePool = QueryPool::ObjectFromHandle(queryPool); { // NOTE: This function is illegal to call for TimestampQueryPools and AccelerationStructureQueryPools const PalQueryPool* pQueryPool = pBasePool->AsPalQueryPool(); Pal::QueryType queryType = pQueryPool->PalQueryType(); if (queryType == Pal::QueryType::StreamoutStats) { queryType = static_cast(static_cast(queryType) + index); } utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdEndQuery(*pQueryPool->PalPool(deviceIdx), queryType, query); } while (deviceGroup.IterateNext()); } DbgBarrierPostCmd(DbgBarrierQueryBeginEnd); } #if VKI_RAY_TRACING // ===================================================================================================================== void CmdBuffer::ResetAccelerationStructureQueryPool( const AccelerationStructureQueryPool& accelerationStructureQueryPool, const uint32_t firstQuery, const uint32_t queryCount) { // All the cache operations operating on the query pool's accelerationStructure memory // that may have occurred before/after this reset. static const uint32_t AccelerationStructureCoher = Pal::CoherShaderWrite | // vkWriteAccelerationStructuresProperties (CmdDispatch) Pal::CoherShaderRead | // vkCmdCopyQueryPoolResults Pal::CoherCopyDst; // vkCmdResetQueryPool (CmdFillMemory) // Wait for any accelerationStructure query pool events to complete prior to filling memory { if (m_pDevice->GetRuntimeSettings().useAcquireReleaseInterface) { Pal::AcquireReleaseInfo acqRelInfo = {}; Pal::MemBarrier memTransition = {}; memTransition.srcAccessMask = AccelerationStructureCoher; memTransition.dstAccessMask = Pal::CoherMemory; memTransition.srcStageMask = Pal::PipelineStageCs; memTransition.dstStageMask = Pal::PipelineStageBlt; acqRelInfo.pMemoryBarriers = &memTransition; acqRelInfo.memoryBarrierCount = 1; acqRelInfo.reason = RgpBarrierInternalPreResetQueryPoolSync; PalCmdReleaseThenAcquire(acqRelInfo, m_curDeviceMask); } else { static const Pal::HwPipePoint pipePoint = Pal::HwPipeBottom; static const Pal::BarrierTransition Transition = { AccelerationStructureCoher, // srcCacheMask Pal::CoherMemory, // dstCacheMask { } // imageInfo }; static const Pal::BarrierInfo Barrier = { Pal::HwPipeTop, // waitPoint 1, // pipePointWaitCount &pipePoint, // pPipePoints 0, // gpuEventCount nullptr, // ppGpuEvents 0, // rangeCheckedTargetWaitCount nullptr, // ppTargets 1, // transitionCount &Transition, // pTransitions 0, // globalSrcCacheMask 0, // globalDstCacheMask RgpBarrierInternalPreResetQueryPoolSync // reason }; PalCmdBarrier(Barrier, m_curDeviceMask); } } utils::IterateMask deviceGroup1(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup1.Index(); PalCmdBuffer(deviceIdx)->CmdFillMemory( accelerationStructureQueryPool.PalMemory(deviceIdx), accelerationStructureQueryPool.GetSlotOffset(firstQuery), accelerationStructureQueryPool.GetSlotSize() * queryCount, 0); } while (deviceGroup1.IterateNext()); // Wait for memory fill to complete { if (m_pDevice->GetRuntimeSettings().useAcquireReleaseInterface) { Pal::AcquireReleaseInfo acqRelInfo = {}; Pal::MemBarrier memTransition = {}; memTransition.srcAccessMask = Pal::CoherMemory | Pal::CoherCopyDst; memTransition.dstAccessMask = AccelerationStructureCoher; memTransition.srcStageMask = Pal::PipelineStageBlt; memTransition.dstStageMask = Pal::PipelineStageTopOfPipe; acqRelInfo.pMemoryBarriers = &memTransition; acqRelInfo.memoryBarrierCount = 1; acqRelInfo.reason = RgpBarrierInternalPreResetQueryPoolSync; PalCmdReleaseThenAcquire(acqRelInfo, m_curDeviceMask); } else { static const Pal::HwPipePoint pipePoint = Pal::HwPipePostBlt; static const Pal::BarrierTransition Transition = { Pal::CoherMemory | Pal::CoherCopyDst, // srcCacheMask AccelerationStructureCoher, // dstCacheMask { } // imageInfo }; static const Pal::BarrierInfo Barrier = { Pal::HwPipeTop, // waitPoint 1, // pipePointWaitCount &pipePoint, // pPipePoints 0, // gpuEventCount nullptr, // ppGpuEvents 0, // rangeCheckedTargetWaitCount nullptr, // ppTargets 1, // transitionCount &Transition, // pTransitions 0, // globalSrcCacheMask 0, // globalDstCacheMask RgpBarrierInternalPostResetQueryPoolSync // reason }; PalCmdBarrier(Barrier, m_curDeviceMask); } } } #endif // ===================================================================================================================== void CmdBuffer::FillTimestampQueryPool( const TimestampQueryPool& timestampQueryPool, const uint32_t firstQuery, const uint32_t queryCount, const uint32_t timestampChunk) { // All the cache operations operating on the query pool's timestamp memory // that may have occurred before/after this reset. static const uint32_t TimestampCoher = Pal::CoherShaderRead | // vkCmdCopyQueryPoolResults (CmdDispatch) Pal::CoherCopyDst | // vkCmdResetQueryPool (CmdFillMemory) Pal::CoherTimestamp; // vkCmdWriteTimestamp (CmdWriteTimestamp) // Wait for any timestamp query pool events to complete prior to filling memory { if (m_pDevice->GetRuntimeSettings().useAcquireReleaseInterface) { Pal::AcquireReleaseInfo acqRelInfo = {}; Pal::MemBarrier memTransition = {}; memTransition.srcAccessMask = TimestampCoher; memTransition.dstAccessMask = Pal::CoherMemory; memTransition.srcStageMask = Pal::PipelineStageBottomOfPipe; memTransition.dstStageMask = Pal::PipelineStageBlt; acqRelInfo.pMemoryBarriers = &memTransition; acqRelInfo.memoryBarrierCount = 1; acqRelInfo.reason = RgpBarrierInternalPreResetQueryPoolSync; PalCmdReleaseThenAcquire(acqRelInfo, m_curDeviceMask); } else { static const Pal::HwPipePoint pipePoint = Pal::HwPipeBottom; static const Pal::BarrierTransition Transition = { TimestampCoher, // srcCacheMask Pal::CoherMemory, // dstCacheMask { } // imageInfo }; static const Pal::BarrierInfo Barrier = { Pal::HwPipeTop, // waitPoint 1, // pipePointWaitCount &pipePoint, // pPipePoints 0, // gpuEventCount nullptr, // ppGpuEvents 0, // rangeCheckedTargetWaitCount nullptr, // ppTargets 1, // transitionCount &Transition, // pTransitions 0, // globalSrcCacheMask 0, // globalDstCacheMask RgpBarrierInternalPreResetQueryPoolSync // reason }; PalCmdBarrier(Barrier, m_curDeviceMask); } } // +----------------+----------------+ // | TimestampChunk | TimestampChunk | // |----------------+----------------| // | TimestampValue | // +---------------------------------+ // TimestampValue = (uint64_t(TimestampChunk) << 32) + TimestampChunk // // Write TimestampValue to all timestamps in TimestampQueryPool. // Note that each slot in TimestampQueryPool contains only timestamp value. // The availability info is generated on the fly from timestamp value. utils::IterateMask deviceGroup1(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup1.Index(); PalCmdBuffer(deviceIdx)->CmdFillMemory( timestampQueryPool.PalMemory(deviceIdx), timestampQueryPool.GetSlotOffset(firstQuery), timestampQueryPool.GetSlotSize() * queryCount, timestampChunk); } while (deviceGroup1.IterateNext()); // Wait for memory fill to complete { if (m_pDevice->GetRuntimeSettings().useAcquireReleaseInterface) { Pal::AcquireReleaseInfo acqRelInfo = {}; Pal::MemBarrier memTransition = {}; memTransition.srcAccessMask = Pal::CoherMemory | Pal::CoherCopyDst; memTransition.dstAccessMask = TimestampCoher; memTransition.srcStageMask = Pal::PipelineStageBlt; memTransition.dstStageMask = Pal::PipelineStageTopOfPipe; acqRelInfo.pMemoryBarriers = &memTransition; acqRelInfo.memoryBarrierCount = 1; acqRelInfo.reason = RgpBarrierInternalPreResetQueryPoolSync; PalCmdReleaseThenAcquire(acqRelInfo, m_curDeviceMask); } else { static const Pal::HwPipePoint pipePoint = Pal::HwPipePostBlt; static const Pal::BarrierTransition Transition = { Pal::CoherMemory | Pal::CoherCopyDst, // srcCacheMask TimestampCoher, // dstCacheMask { } // imageInfo }; static const Pal::BarrierInfo Barrier = { Pal::HwPipeTop, // waitPoint 1, // pipePointWaitCount &pipePoint, // pPipePoints 0, // gpuEventCount nullptr, // ppGpuEvents 0, // rangeCheckedTargetWaitCount nullptr, // ppTargets 1, // transitionCount &Transition, // pTransitions 0, // globalSrcCacheMask 0, // globalDstCacheMask RgpBarrierInternalPostResetQueryPoolSync // reason }; PalCmdBarrier(Barrier, m_curDeviceMask); } } } // ===================================================================================================================== void CmdBuffer::ResetQueryPool( VkQueryPool queryPool, uint32_t firstQuery, uint32_t queryCount) { DbgBarrierPreCmd(DbgBarrierQueryReset); PalCmdSuspendPredication(true); const QueryPool* pBasePool = QueryPool::ObjectFromHandle(queryPool); if (pBasePool->GetQueryType() == VK_QUERY_TYPE_TIMESTAMP) { const TimestampQueryPool* pQueryPool = pBasePool->AsTimestampQueryPool(); // Write TimestampNotReady to all timestamps in TimestampQueryPool. FillTimestampQueryPool( *pQueryPool, firstQuery, queryCount, TimestampQueryPool::TimestampNotReadyChunk); } #if VKI_RAY_TRACING else if (IsAccelerationStructureQueryType(pBasePool->GetQueryType())) { const AccelerationStructureQueryPool* pQueryPool = pBasePool->AsAccelerationStructureQueryPool(); ResetAccelerationStructureQueryPool( *pQueryPool, firstQuery, queryCount); } #endif else { const PalQueryPool* pQueryPool = pBasePool->AsPalQueryPool(); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdResetQueryPool( *pQueryPool->PalPool(deviceIdx), firstQuery, queryCount); } while (deviceGroup.IterateNext()); } PalCmdSuspendPredication(false); DbgBarrierPostCmd(DbgBarrierQueryReset); } // ===================================================================================================================== // This is the main hook for any CmdBarrier going into PAL. Always call this function instead of CmdBarrier directly. void CmdBuffer::PalCmdBarrier( const Pal::BarrierInfo& info, uint32_t deviceMask) { // If you trip this assert, you've forgotten to populate a value for this field. You should use one of the // RgpBarrierReason enum values from sqtt_rgp_annotations.h. Preferably you should add a new one as described // in the header, but temporarily you may use the generic "unknown" reason so as not to block your main code // change. VK_ASSERT(info.reason != 0); #if PAL_ENABLE_PRINTS_ASSERTS for (uint32_t i = 0; i < info.transitionCount; ++i) { // Detect if PAL may execute a barrier blt using this image VK_ASSERT(info.pTransitions[i].imageInfo.pImage == nullptr); // You need to use the other PalCmdBarrier method (below) which uses vk::Image ptrs to obtain the // corresponding Pal::IImage ptr for each image transition } #endif if (m_flags.useReleaseAcquire) { // Translate the Pal::BarrierInfo to an equivalent Pal::AcquireReleaseInfo struct and then call // Pal::CmdReleaseThenAcquire() instead of Pal::CmdBarrier() TranslateBarrierInfoToAcqRel(info, deviceMask); } else { utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdBarrier(info); } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== void CmdBuffer::PalCmdBarrier( Pal::BarrierInfo* pInfo, Pal::BarrierTransition* const pTransitions, const Image** const pTransitionImages, uint32_t deviceMask) { // If you trip this assert, you've forgot to populate a value for this field. You should use one of the // RgpBarrierReason enum values from sqtt_rgp_annotations.h. Preferably you should add a new one as described // in the header, but temporarily you may use the generic "unknown" reason so as not to block you. VK_ASSERT(pInfo->reason != 0); const Pal::IGpuEvent** ppOriginalGpuEvents = pInfo->ppGpuEvents; utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); // TODO: I have proposed a better solution with the Pal team. ie grouped images referenced from // Pal::BarrierTransition. Executebarriers already wrote the correct Pal::IImage for device 0, so this loop // needs to update the Pal::Image* after the first iteration. if (deviceIdx > 0) { for (uint32_t i = 0; i < pInfo->transitionCount; i++) { if (pTransitions[i].imageInfo.pImage != nullptr) { pTransitions[i].imageInfo.pImage = pTransitionImages[i]->PalImage(deviceIdx); } } pInfo->pTransitions = pTransitions; // Access the correct Gpu Events for this Pal device if (pInfo->ppGpuEvents != nullptr) { pInfo->ppGpuEvents = &ppOriginalGpuEvents[(pInfo->gpuEventWaitCount * deviceIdx)]; } } PalCmdBuffer(deviceIdx)->CmdBarrier(*pInfo); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== // Translates the Pal::BarrierInfo into equivalent Pal::AcquireReleaseInfo struct. This function does a 1-to-1 mapping // for struct members and hence should not be used in general. void CmdBuffer::TranslateBarrierInfoToAcqRel( const Pal::BarrierInfo& barrierInfo, uint32_t deviceMask) { VirtualStackFrame virtStackFrame(m_pStackAllocator); Pal::AcquireReleaseInfo info = {}; Pal::MemBarrier memoryBarriers = {}; uint32_t srcStageMask = 0; const uint32_t dstStageMask = ConvertWaitPointToPipeStage(barrierInfo.waitPoint); for (uint32_t i = 0; i < barrierInfo.pipePointWaitCount; i++) { srcStageMask |= ConvertPipePointToPipeStage(barrierInfo.pPipePoints[i]); } info.reason = barrierInfo.reason; // If the transition count is 0 then this barrier is used only for global if (barrierInfo.transitionCount == 0) { info.srcGlobalStageMask = srcStageMask; info.dstGlobalStageMask = dstStageMask; info.srcGlobalAccessMask = barrierInfo.globalSrcCacheMask; info.dstGlobalAccessMask = barrierInfo.globalDstCacheMask; } else { VK_ASSERT((barrierInfo.globalSrcCacheMask == 0) && (barrierInfo.globalDstCacheMask == 0)); for (uint32_t i = 0; i < barrierInfo.transitionCount; i++) { VK_ASSERT(barrierInfo.pTransitions[i].imageInfo.pImage == nullptr); // Pal::AcquireReleaseInfo::MemBarrier is valid only for buffers. For renderpasses we would need to // use image barriers but since we don't have any information about the relevant Pal::IImage object, the // best we can do is record the transition via global memory barrier. if (info.reason == RgpBarrierExternalRenderPassSync) { info.srcGlobalStageMask = srcStageMask; info.dstGlobalStageMask = dstStageMask; info.srcGlobalAccessMask |= barrierInfo.pTransitions[i].srcCacheMask; info.dstGlobalAccessMask |= barrierInfo.pTransitions[i].dstCacheMask; } else { memoryBarriers.srcStageMask = srcStageMask; memoryBarriers.dstStageMask = dstStageMask; memoryBarriers.srcAccessMask |= barrierInfo.pTransitions[i].srcCacheMask; memoryBarriers.dstAccessMask |= barrierInfo.pTransitions[i].dstCacheMask; // Just passing 1 memory barrier count and OR'ing the cache masks is enough for PAL. info.memoryBarrierCount = 1; } } } if (info.memoryBarrierCount > 0) { info.pMemoryBarriers = &memoryBarriers; } PalCmdReleaseThenAcquire(info, deviceMask); } // ===================================================================================================================== // This is the main hook for any CmdReleaseThenAcquire going into PAL. Always call this function instead of // CmdReleaseThenAcquire directly. void CmdBuffer::PalCmdReleaseThenAcquire( const Pal::AcquireReleaseInfo& info, uint32_t deviceMask) { // If you trip this assert, you've forgotten to populate a value for this field. You should use one of the // RgpBarrierReason enum values from sqtt_rgp_annotations.h. Preferably you should add a new one as described // in the header, but temporarily you may use the generic "unknown" reason so as not to block your main code change. VK_ASSERT(info.reason != 0); #if PAL_ENABLE_PRINTS_ASSERTS for (uint32_t i = 0; i < info.imageBarrierCount; ++i) { // Detect if PAL may execute a barrier blt using this image VK_ASSERT(info.pImageBarriers[i].pImage == nullptr); // You need to use the other PalCmdReleaseThenAcquire method (below) which uses vk::Image ptrs to obtain the // corresponding Pal::IImage ptr for each image transition } #endif utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdReleaseThenAcquire(info); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdReleaseThenAcquire( Pal::AcquireReleaseInfo* pAcquireReleaseInfo, Pal::MemBarrier* const pBufferBarriers, const Buffer** const ppBuffers, Pal::ImgBarrier* const pImageBarriers, const Image** const ppImages, uint32_t deviceMask) { // If you trip this assert, you've forgotten to populate a value for this field. You should use one of the // RgpBarrierReason enum values from sqtt_rgp_annotations.h. Preferably you should add a new one as described // in the header, but temporarily you may use the generic "unknown" reason so as not to block you. VK_ASSERT(pAcquireReleaseInfo->reason != 0); utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < pAcquireReleaseInfo->imageBarrierCount; i++) { if (ppImages != nullptr) { pImageBarriers[i].pImage = ppImages[i]->PalImage(deviceIdx); } } pAcquireReleaseInfo->pImageBarriers = pImageBarriers; pAcquireReleaseInfo->pMemoryBarriers = pBufferBarriers; PalCmdBuffer(deviceIdx)->CmdReleaseThenAcquire(*pAcquireReleaseInfo); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdAcquire( Pal::AcquireReleaseInfo* pAcquireReleaseInfo, uint32_t eventCount, const VkEvent* pEvents, Pal::MemBarrier* const pBufferBarriers, const Buffer** const ppBuffers, Pal::ImgBarrier* const pImageBarriers, const Image** const ppImages, VirtualStackFrame* pVirtStackFrame, uint32_t deviceMask) { // If you trip this assert, you've forgot to populate a value for this field. You should use one of the // RgpBarrierReason enum values from sqtt_rgp_annotations.h. Preferably you should add a new one as described // in the header, but temporarily you may use the generic "unknown" reason so as not to block you. VK_ASSERT(pAcquireReleaseInfo->reason != 0); Event* pEvent = Event::ObjectFromHandle(pEvents[0]); utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < pAcquireReleaseInfo->imageBarrierCount; i++) { if (ppImages != nullptr) { pImageBarriers[i].pImage = ppImages[i]->PalImage(deviceIdx); } } pAcquireReleaseInfo->pImageBarriers = pImageBarriers; pAcquireReleaseInfo->pMemoryBarriers = pBufferBarriers; // Whether syncToken is used or not is decided by the setting 'SyncTokenEnabled' if (pEvent->IsUseToken()) { // Allocate space to store sync token values (automatically rewound on unscope) Pal::ReleaseToken* pSyncTokens = (eventCount > 0) ? pVirtStackFrame->AllocArray(eventCount) : nullptr; if (pSyncTokens != nullptr) { for (uint32_t i = 0; i < eventCount; ++i) { pSyncTokens[i] = Event::ObjectFromHandle(pEvents[i])->GetSyncToken(); } PalCmdBuffer(deviceIdx)->CmdAcquire(*pAcquireReleaseInfo, eventCount, pSyncTokens); pVirtStackFrame->FreeArray(pSyncTokens); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } else { // Allocate space to store signaled event pointers (automatically rewound on unscope) const Pal::IGpuEvent** ppGpuEvents = (eventCount > 0) ? pVirtStackFrame->AllocArray(eventCount) : nullptr; if (ppGpuEvents != nullptr) { for (uint32_t i = 0; i < eventCount; ++i) { ppGpuEvents[i] = Event::ObjectFromHandle(pEvents[i])->PalEvent(deviceIdx); } PalCmdBuffer(deviceIdx)->CmdAcquireEvent(*pAcquireReleaseInfo, eventCount, ppGpuEvents); pVirtStackFrame->FreeArray(ppGpuEvents); } else { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } } } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdRelease( Pal::AcquireReleaseInfo* pAcquireReleaseInfo, const VkEvent event, Pal::MemBarrier* const pBufferBarriers, const Buffer** const ppBuffers, Pal::ImgBarrier* const pImageBarriers, const Image** const ppImages, uint32_t deviceMask) { // If you trip this assert, you've forgot to populate a value for this field. You should use one of the // RgpBarrierReason enum values from sqtt_rgp_annotations.h. Preferably you should add a new one as described // in the header, but temporarily you may use the generic "unknown" reason so as not to block you. VK_ASSERT(pAcquireReleaseInfo->reason != 0); Event* pEvent = Event::ObjectFromHandle(event); utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < pAcquireReleaseInfo->imageBarrierCount; i++) { if (ppImages != nullptr) { pImageBarriers[i].pImage = ppImages[i]->PalImage(deviceIdx); } } pAcquireReleaseInfo->pImageBarriers = pImageBarriers; pAcquireReleaseInfo->pMemoryBarriers = pBufferBarriers; if (pEvent->IsUseToken()) { pEvent->SetSyncToken(PalCmdBuffer(deviceIdx)->CmdRelease(*pAcquireReleaseInfo)); } else { PalCmdBuffer(deviceIdx)->CmdReleaseEvent(*pAcquireReleaseInfo, pEvent->PalEvent(deviceIdx)); } } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdBindMsaaStates( const Pal::IMsaaState* const * pStates) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBindMsaaState(PalCmdBuffer(deviceIdx), deviceIdx, (pStates != nullptr) ? pStates[deviceIdx] : nullptr); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdSetMsaaQuadSamplePattern( uint32_t numSamplesPerPixel, const Pal::MsaaQuadSamplePattern& quadSamplePattern) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetMsaaQuadSamplePattern(numSamplesPerPixel, quadSamplePattern); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::PalCmdSuspendPredication( bool suspend) { if (m_flags.hasConditionalRendering) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSuspendPredication(suspend); } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== // Command to write a timestamp value to a location in a Timestamp query pool void CmdBuffer::WriteTimestamp( PipelineStageFlags pipelineStage, const TimestampQueryPool* pQueryPool, uint32_t query) { DbgBarrierPreCmd(DbgBarrierWriteTimestamp); PalCmdSuspendPredication(true); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdWriteTimestamp( VkToPalSrcPipeStageFlagForTimestampWrite(pipelineStage), pQueryPool->PalMemory(deviceIdx), pQueryPool->GetSlotOffset(query)); const auto* const pRenderPass = m_allGpuState.pRenderPass; // If vkCmdWriteTimestamp is called while executing a render pass instance that has multiview enabled, // the timestamp uses N consecutive query indices in the query pool (starting at query) where // N is the number of bits set in the view mask of the subpass the command is executed in. // // The first query is a timestamp value and (if more than one bit is set in the view mask) // zero is written to the remaining queries. if (((UsingDynamicRendering() == false) && pRenderPass->IsMultiviewEnabled()) || (m_allGpuState.dynamicRenderingInstance.viewMask != 0)) { const auto viewMask = (pRenderPass != nullptr) ? pRenderPass->GetViewMask(m_renderPassInstance.subpass) : m_allGpuState.dynamicRenderingInstance.viewMask; const auto viewCount = Util::CountSetBits(viewMask); VK_ASSERT(viewCount > 0); const auto remainingQueryCount = viewCount - 1; if (remainingQueryCount > 0) { const auto firstRemainingQuery = query + 1; // Set value of each remaining query to 0 and to make them avaliable. // Note that values of remaining queries (to which 0 was written) are not considered timestamps. FillTimestampQueryPool( *pQueryPool, firstRemainingQuery, remainingQueryCount, 0u); } } } while (deviceGroup.IterateNext()); PalCmdSuspendPredication(false); DbgBarrierPostCmd(DbgBarrierWriteTimestamp); } // ===================================================================================================================== void CmdBuffer::SetSampleLocations( const VkSampleLocationsInfoEXT* pSampleLocationsInfo) { uint32_t sampleLocationsPerPixel = static_cast(pSampleLocationsInfo->sampleLocationsPerPixel); if (sampleLocationsPerPixel > 0) { ConvertToPalMsaaQuadSamplePattern(pSampleLocationsInfo, &m_allGpuState.samplePattern.locations); } m_allGpuState.samplePattern.sampleCount = sampleLocationsPerPixel; m_allGpuState.dirtyGraphics.samplePattern = 1; } // ===================================================================================================================== // Begins a render pass instance (vkCmdBeginRenderPass) void CmdBuffer::BeginRenderPass( const VkRenderPassBeginInfo* pRenderPassBegin, VkSubpassContents contents) { DbgBarrierPreCmd(DbgBarrierBeginRenderPass); m_allGpuState.pRenderPass = RenderPass::ObjectFromHandle(pRenderPassBegin->renderPass); m_allGpuState.pFramebuffer = Framebuffer::ObjectFromHandle(pRenderPassBegin->framebuffer); Pal::Result result = Pal::Result::Success; EXTRACT_VK_STRUCTURES_3( RP, RenderPassBeginInfo, DeviceGroupRenderPassBeginInfo, RenderPassSampleLocationsBeginInfoEXT, RenderPassAttachmentBeginInfo, pRenderPassBegin, RENDER_PASS_BEGIN_INFO, DEVICE_GROUP_RENDER_PASS_BEGIN_INFO, RENDER_PASS_SAMPLE_LOCATIONS_BEGIN_INFO_EXT, RENDER_PASS_ATTACHMENT_BEGIN_INFO) // Copy render areas (these may be per-device in a group) bool replicateRenderArea = true; // Set the render pass instance's device mask to the value the command buffer began with. SetRpDeviceMask(m_cbBeginDeviceMask); if (pDeviceGroupRenderPassBeginInfo != nullptr) { SetRpDeviceMask(pDeviceGroupRenderPassBeginInfo->deviceMask); SetDeviceMask(GetRpDeviceMask()); m_renderPassInstance.renderAreaCount = pDeviceGroupRenderPassBeginInfo->deviceRenderAreaCount; VK_ASSERT(m_renderPassInstance.renderAreaCount <= MaxPalDevices); if (pDeviceGroupRenderPassBeginInfo->deviceRenderAreaCount > 0) { utils::IterateMask deviceGroup(pDeviceGroupRenderPassBeginInfo->deviceMask); VK_ASSERT(m_numPalDevices == pDeviceGroupRenderPassBeginInfo->deviceRenderAreaCount); do { const uint32_t deviceIdx = deviceGroup.Index(); const VkRect2D& srcRect = pDeviceGroupRenderPassBeginInfo->pDeviceRenderAreas[deviceIdx]; auto* pDstRect = &m_renderPassInstance.renderArea[deviceIdx]; pDstRect->offset.x = srcRect.offset.x; pDstRect->offset.y = srcRect.offset.y; pDstRect->extent.width = srcRect.extent.width; pDstRect->extent.height = srcRect.extent.height; } while (deviceGroup.IterateNext()); replicateRenderArea = false; } } if (replicateRenderArea) { m_renderPassInstance.renderAreaCount = m_numPalDevices; const auto& srcRect = pRenderPassBeginInfo->renderArea; for (uint32_t deviceIdx = 0; deviceIdx < m_numPalDevices; deviceIdx++) { auto* pDstRect = &m_renderPassInstance.renderArea[deviceIdx]; pDstRect->offset.x = srcRect.offset.x; pDstRect->offset.y = srcRect.offset.y; pDstRect->extent.width = srcRect.extent.width; pDstRect->extent.height = srcRect.extent.height; } } const uint32_t attachmentCount = m_allGpuState.pRenderPass->GetAttachmentCount(); // Allocate enough per-attachment state space if (m_renderPassInstance.maxAttachmentCount < attachmentCount) { // Free old memory if (m_renderPassInstance.pAttachments != nullptr) { m_pDevice->VkInstance()->FreeMem(m_renderPassInstance.pAttachments); m_renderPassInstance.pAttachments = nullptr; m_renderPassInstance.maxAttachmentCount = 0; } // Allocate enough to cover new requirements const size_t maxAttachmentCount = Util::Max(attachmentCount, 8U); m_renderPassInstance.pAttachments = static_cast( m_pDevice->VkInstance()->AllocMem( sizeof(RenderPassInstanceState::AttachmentState) * maxAttachmentCount, VK_SYSTEM_ALLOCATION_SCOPE_OBJECT)); if (m_renderPassInstance.pAttachments != nullptr) { m_renderPassInstance.maxAttachmentCount = maxAttachmentCount; memset(m_renderPassInstance.pAttachments, 0, maxAttachmentCount * sizeof(RenderPassInstanceState::AttachmentState)); } else { result = Pal::Result::ErrorOutOfMemory; } } const uint32_t subpassCount = m_allGpuState.pRenderPass->GetSubpassCount(); // Allocate pSamplePatterns memory if (m_renderPassInstance.maxSubpassCount < subpassCount) { // Free old memory if (m_renderPassInstance.pSamplePatterns != nullptr) { m_pDevice->VkInstance()->FreeMem(m_renderPassInstance.pSamplePatterns); m_renderPassInstance.pSamplePatterns = nullptr; m_renderPassInstance.maxSubpassCount = 0; } // Allocate enough to cover new requirements m_renderPassInstance.pSamplePatterns = static_cast( m_pDevice->VkInstance()->AllocMem( sizeof(SamplePattern) * subpassCount, VK_SYSTEM_ALLOCATION_SCOPE_OBJECT)); if (m_renderPassInstance.pSamplePatterns != nullptr) { m_renderPassInstance.maxSubpassCount = subpassCount; memset(m_renderPassInstance.pSamplePatterns, 0, subpassCount * sizeof(SamplePattern)); } else { result = Pal::Result::ErrorOutOfMemory; } } if (pRenderPassAttachmentBeginInfo != nullptr) { if (m_allGpuState.pFramebuffer->Imageless() == false) { VK_ASSERT(pRenderPassAttachmentBeginInfo->attachmentCount == 0); } else { VK_ASSERT(pRenderPassAttachmentBeginInfo->attachmentCount == attachmentCount); VK_ASSERT(pRenderPassAttachmentBeginInfo->attachmentCount == m_allGpuState.pFramebuffer->GetAttachmentCount()); } m_allGpuState.pFramebuffer->SetImageViews(pRenderPassAttachmentBeginInfo); } if (result == Pal::Result::Success) { m_renderPassInstance.subpass = 0; // Copy clear values if (pRenderPassBeginInfo->pClearValues != nullptr) { const uint32_t clearValueCount = Util::Min(pRenderPassBeginInfo->clearValueCount, attachmentCount); for (uint32_t a = 0; a < clearValueCount; ++a) { m_renderPassInstance.pAttachments[a].clearValue = pRenderPassBeginInfo->pClearValues[a]; } } // Initialize current layout state based on attachment initial layout for (uint32_t a = 0; a < attachmentCount; ++a) { // Start current layouts to PAL version of initial layout for each attachment. constexpr Pal::ImageLayout NullLayout = {}; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(a); const uint32 firstPlane = attachment.subresRange[0].startSubres.plane; const RPImageLayout initialLayout = { m_allGpuState.pRenderPass->GetAttachmentDesc(a).initialLayout, 0 }; if (!attachment.pImage->IsDepthStencilFormat()) { RPSetAttachmentLayout( a, firstPlane, attachment.pImage->GetAttachmentLayout(initialLayout, firstPlane, this)); } else { // Note that we set both depth and stencil aspect layouts for depth/stencil formats to define // initial values for them. This avoids some (incorrect) PAL asserts when clearing depth- or // stencil-only surfaces. Here, the missing aspect will have a null usage but a non-null engine // component. VK_ASSERT((firstPlane == 0) || (firstPlane == 1)); const RPImageLayout initialStencilLayout = { m_allGpuState.pRenderPass->GetAttachmentDesc(a).stencilInitialLayout, 0 }; RPSetAttachmentLayout( a, 0, attachment.pImage->GetAttachmentLayout(initialLayout, 0, this)); RPSetAttachmentLayout( a, 1, attachment.pImage->GetAttachmentLayout(initialStencilLayout, 1, this)); } } if (pRenderPassSampleLocationsBeginInfoEXT != nullptr) { uint32_t attachmentInitialSampleLocationCount = pRenderPassSampleLocationsBeginInfoEXT->attachmentInitialSampleLocationsCount; for (uint32_t ai = 0; ai < attachmentInitialSampleLocationCount; ++ai) { const uint32_t attachmentIndex = pRenderPassSampleLocationsBeginInfoEXT->pAttachmentInitialSampleLocations[ai].attachmentIndex; VK_ASSERT(attachmentIndex < attachmentCount); const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(attachmentIndex); if (attachment.pImage->IsSampleLocationsCompatibleDepth()) { const VkSampleLocationsInfoEXT* pSampleLocationsInfo = &pRenderPassSampleLocationsBeginInfoEXT->pAttachmentInitialSampleLocations[ai].sampleLocationsInfo; m_renderPassInstance.pAttachments[attachmentIndex].initialSamplePattern.sampleCount = (uint32_t)pSampleLocationsInfo->sampleLocationsPerPixel; ConvertToPalMsaaQuadSamplePattern( pSampleLocationsInfo, &m_renderPassInstance.pAttachments[attachmentIndex].initialSamplePattern.locations); } } uint32_t postSubpassSampleLocationsCount = pRenderPassSampleLocationsBeginInfoEXT->postSubpassSampleLocationsCount; for (uint32_t ps = 0; ps < postSubpassSampleLocationsCount; ++ps) { const uint32_t psIndex = pRenderPassSampleLocationsBeginInfoEXT->pPostSubpassSampleLocations[ps].subpassIndex; const VkSampleLocationsInfoEXT* pSampleLocationsInfo = &pRenderPassSampleLocationsBeginInfoEXT->pPostSubpassSampleLocations[ps].sampleLocationsInfo; m_renderPassInstance.pSamplePatterns[psIndex].sampleCount = (uint32_t)pSampleLocationsInfo->sampleLocationsPerPixel; ConvertToPalMsaaQuadSamplePattern( pSampleLocationsInfo, &m_renderPassInstance.pSamplePatterns[psIndex].locations); } } // Begin the first subpass m_renderPassInstance.pExecuteInfo = m_allGpuState.pRenderPass->GetExecuteInfo(); utils::IterateMask deviceGroup(GetRpDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetGlobalScissor(m_allGpuState.pFramebuffer->GetGlobalScissorParams()); } while (deviceGroup.IterateNext()); RPBeginSubpass(); } else { // Set a dummy state such that other instance commands ignore the render pass instance. m_renderPassInstance.subpass = VK_SUBPASS_EXTERNAL; } DbgBarrierPostCmd(DbgBarrierBeginRenderPass); } // ===================================================================================================================== // Advances to the next sub-pass in the current render pass (vkCmdNextSubPass) void CmdBuffer::NextSubPass( VkSubpassContents contents) { DbgBarrierPreCmd(DbgBarrierNextSubpass); if (m_renderPassInstance.subpass != VK_SUBPASS_EXTERNAL) { // End the previous subpass RPEndSubpass(); // Advance the current subpass index m_renderPassInstance.subpass++; // Begin the next subpass RPBeginSubpass(); } DbgBarrierPostCmd(DbgBarrierNextSubpass); } // ===================================================================================================================== // Ends the current subpass during a render pass instance. void CmdBuffer::RPEndSubpass() { VK_ASSERT(m_renderPassInstance.subpass < m_allGpuState.pRenderPass->GetSubpassCount()); // Get current subpass execute state const auto& subpass = m_renderPassInstance.pExecuteInfo->pSubpasses[m_renderPassInstance.subpass]; VirtualStackFrame virtStack(m_pStackAllocator); // Synchronize preceding work before resolving if needed if (subpass.end.syncPreResolve.flags.active) { RPSyncPoint(subpass.end.syncPreResolve, &virtStack); } // Execute any multisample resolve attachment operations if (subpass.end.resolveCount > 0) { RPResolveAttachments(subpass.end.resolveCount, subpass.end.pResolves); } // Synchronize preceding work at the end of the subpass if (subpass.end.syncBottom.flags.active) { RPSyncPoint(subpass.end.syncBottom, &virtStack); } } // ===================================================================================================================== // Handles post-clear synchronization for load-op color clears when not auto-syncing. void CmdBuffer::RPSyncPostLoadOpColorClear( uint32_t colorClearCount, const RPLoadOpClearInfo* pClears) { if (m_flags.useReleaseAcquire) { VK_ASSERT(colorClearCount > 0); VirtualStackFrame virtStack(m_pStackAllocator); Pal::AcquireReleaseInfo barrierInfo = {}; barrierInfo.reason = RgpBarrierExternalRenderPassSync; Pal::ImgBarrier* pPalTransitions = (colorClearCount != 0) ? virtStack.AllocArray(colorClearCount) : nullptr; const Image** ppImages = (colorClearCount != 0) ? virtStack.AllocArray(colorClearCount) : nullptr; for (uint32_t i = 0; i < colorClearCount; ++i) { const RPLoadOpClearInfo& clear = pClears[i]; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(clear.attachment); VK_ASSERT(pPalTransitions != nullptr); VK_ASSERT(ppImages != nullptr); for (uint32_t sr = 0; sr < attachment.subresRangeCount; ++sr) { const uint32_t plane = attachment.subresRange[sr].startSubres.plane; const Pal::ImageLayout oldLayout = RPGetAttachmentLayout(clear.attachment, plane); const Pal::ImageLayout newLayout = { VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, 1 }; ppImages[barrierInfo.imageBarrierCount] = attachment.pImage; Pal::ImgBarrier* const pBarrier = &pPalTransitions[barrierInfo.imageBarrierCount]; *pBarrier = {}; pBarrier->srcStageMask = Pal::PipelineStageBlt; pBarrier->dstStageMask = Pal::PipelineStageEarlyDsTarget; pBarrier->srcAccessMask = Pal::CoherClear; pBarrier->dstAccessMask = Pal::CoherColorTarget; // We set the pImage to nullptr by default here. But, this will be computed correctly later for // each device including DefaultDeviceIndex based on the deviceId. pBarrier->pImage = nullptr; pBarrier->oldLayout = oldLayout; pBarrier->newLayout = newLayout; pBarrier->subresRange = attachment.subresRange[sr]; barrierInfo.imageBarrierCount++; } } barrierInfo.pImageBarriers = pPalTransitions; PalCmdReleaseThenAcquire( &barrierInfo, nullptr, nullptr, pPalTransitions, ppImages, GetRpDeviceMask()); if (pPalTransitions != nullptr) { virtStack.FreeArray(pPalTransitions); } if (ppImages != nullptr) { virtStack.FreeArray(ppImages); } } else { static const Pal::BarrierTransition transition = { Pal::CoherClear, Pal::CoherColorTarget, {} }; static const Pal::HwPipePoint PipePoint = Pal::HwPipePostBlt; static const Pal::BarrierInfo Barrier = { Pal::HwPipePreRasterization, // waitPoint 1, // pipePointWaitCount &PipePoint, // pPipePoints 0, // gpuEventWaitCount nullptr, // ppGpuEvents 0, // rangeCheckedTargetWaitCount nullptr, // ppTargets 1, // transitionCount &transition, // pTransitions 0, // globalSrcCacheMask 0, // globalDstCacheMask RgpBarrierExternalRenderPassSync // reason }; PalCmdBarrier(Barrier, GetRpDeviceMask()); } } // ===================================================================================================================== // Begins the current subpass during a render pass instance. void CmdBuffer::RPBeginSubpass() { VK_ASSERT(m_renderPassInstance.subpass < m_allGpuState.pRenderPass->GetSubpassCount()); // Get current subpass execute state const auto& subpass = m_renderPassInstance.pExecuteInfo->pSubpasses[m_renderPassInstance.subpass]; // Synchronize prior work (defined by subpass dependencies) prior to the top of this subpass, and handle any // layout transitions for this subpass's references. if (subpass.begin.syncTop.flags.active) { VirtualStackFrame virtStack(m_pStackAllocator); RPSyncPoint(subpass.begin.syncTop, &virtStack); } if (m_flags.subpassLoadOpClearsBoundAttachments) { // Bind targets RPBindTargets(subpass.begin.bindTargets); } if (m_allGpuState.pRenderPass->DoClearsUpfront()) { if (m_renderPassInstance.subpass == 0) { const auto& subpasses = m_renderPassInstance.pExecuteInfo->pSubpasses; const auto subpassCount = m_allGpuState.pRenderPass->GetSubpassCount(); PalCmdSuspendPredication(true); for (uint32_t i = 0; i < subpassCount; ++i) { if (subpasses[i].begin.loadOps.colorClearCount > 0) { RPLoadOpClearColor(subpasses[i].begin.loadOps.colorClearCount, subpasses[i].begin.loadOps.pColorClears); } } // If we are manually pre-syncing color clears, we must post-sync also if (subpasses[0].begin.syncTop.barrier.flags.preColorClearSync) { RPSyncPostLoadOpColorClear(subpasses[0].begin.loadOps.colorClearCount, subpasses[0].begin.loadOps.pColorClears); } for (uint32_t i = 0; i < subpassCount; ++i) { // Execute any depth-stencil clear load operations if (subpasses[i].begin.loadOps.dsClearCount > 0) { RPLoadOpClearDepthStencil(subpasses[i].begin.loadOps.dsClearCount, subpasses[i].begin.loadOps.pDsClears); } } PalCmdSuspendPredication(false); } } else { if ((subpass.begin.loadOps.colorClearCount > 0) || (subpass.begin.loadOps.dsClearCount > 0)) { PalCmdSuspendPredication(true); // Execute any color clear load operations if (subpass.begin.loadOps.colorClearCount > 0) { RPLoadOpClearColor(subpass.begin.loadOps.colorClearCount, subpass.begin.loadOps.pColorClears); } // If we are manually pre-syncing color clears, we must post-sync also if (subpass.begin.syncTop.barrier.flags.preColorClearSync) { RPSyncPostLoadOpColorClear(subpass.begin.loadOps.colorClearCount, subpass.begin.loadOps.pColorClears); } // Execute any depth-stencil clear load operations if (subpass.begin.loadOps.dsClearCount > 0) { RPLoadOpClearDepthStencil(subpass.begin.loadOps.dsClearCount, subpass.begin.loadOps.pDsClears); } PalCmdSuspendPredication(false); } } if (m_flags.subpassLoadOpClearsBoundAttachments == false) { // Bind targets RPBindTargets(subpass.begin.bindTargets); } // Set view instance mask, on devices in render pass instance's device mask SetViewInstanceMask(GetRpDeviceMask()); } // ===================================================================================================================== // Executes a "sync point" during a render pass instance using the legacy barriers. There are a number of these at // different stages between subpasses where we handle execution/memory dependencies from subpass dependencies as well as // trigger automatic layout transitions. void CmdBuffer::RPSyncPointLegacy( const RPSyncPointInfo& syncPoint, VirtualStackFrame* pVirtStack) { const auto& rpBarrier = syncPoint.barrier; Pal::BarrierInfo barrier = {}; barrier.reason = RgpBarrierExternalRenderPassSync; barrier.waitPoint = rpBarrier.waitPoint; barrier.pipePointWaitCount = rpBarrier.pipePointCount; barrier.pPipePoints = rpBarrier.pipePoints; const uint32_t maxTransitionCount = MaxPalAspectsPerMask * syncPoint.transitionCount; Pal::BarrierTransition* pPalTransitions = (maxTransitionCount != 0) ? pVirtStack->AllocArray(maxTransitionCount) : nullptr; const Image** ppImages = (maxTransitionCount != 0) ? pVirtStack->AllocArray(maxTransitionCount) : nullptr; // Construct global memory dependency to synchronize caches (subpass dependencies + implicit synchronization) if (rpBarrier.flags.needsGlobalTransition) { Pal::BarrierTransition globalTransition = { }; m_pDevice->GetBarrierPolicy().ApplyBarrierCacheFlags( rpBarrier.srcAccessMask, rpBarrier.dstAccessMask, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, &globalTransition); barrier.globalSrcCacheMask = globalTransition.srcCacheMask | rpBarrier.implicitSrcCacheMask; barrier.globalDstCacheMask = globalTransition.dstCacheMask | rpBarrier.implicitDstCacheMask; } if ((pPalTransitions != nullptr) && (ppImages != nullptr)) { // Construct attachment-specific layout transitions for (uint32_t t = 0; t < syncPoint.transitionCount; ++t) { const RPTransitionInfo& tr = syncPoint.pTransitions[t]; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(tr.attachment); for (uint32_t sr = 0; sr < attachment.subresRangeCount; ++sr) { const uint32_t plane = attachment.subresRange[sr].startSubres.plane; const RPImageLayout nextLayout = (attachment.pImage->IsDepthStencilFormat() && (plane == 1)) ? tr.nextStencilLayout : tr.nextLayout; const Pal::ImageLayout newLayout = attachment.pImage->GetAttachmentLayout( nextLayout, plane, this); const Pal::ImageLayout oldLayout = RPGetAttachmentLayout( tr.attachment, plane); if ((oldLayout.usages != newLayout.usages) || (oldLayout.engines != newLayout.engines)) { VK_ASSERT(barrier.transitionCount < maxTransitionCount); ppImages[barrier.transitionCount] = attachment.pImage; Pal::BarrierTransition* const pLayoutTransition = &pPalTransitions[barrier.transitionCount]; *pLayoutTransition = {}; pLayoutTransition->imageInfo.pImage = attachment.pImage->PalImage(DefaultDeviceIndex); pLayoutTransition->imageInfo.oldLayout = oldLayout; pLayoutTransition->imageInfo.newLayout = newLayout; pLayoutTransition->imageInfo.subresRange = attachment.subresRange[sr]; const Pal::MsaaQuadSamplePattern* pQuadSamplePattern = nullptr; if (attachment.pImage->IsSampleLocationsCompatibleDepth() && tr.flags.isInitialLayoutTransition) { VK_ASSERT(attachment.pImage->HasDepth()); // Use the provided sample locations for this attachment if this is its // initial layout transition pQuadSamplePattern = &m_renderPassInstance.pAttachments[tr.attachment].initialSamplePattern.locations; } else { // Otherwise, use the subpass' sample locations uint32_t subpass = m_renderPassInstance.subpass; pQuadSamplePattern = &m_renderPassInstance.pSamplePatterns[subpass].locations; } pLayoutTransition->imageInfo.pQuadSamplePattern = pQuadSamplePattern; RPSetAttachmentLayout(tr.attachment, plane, newLayout); barrier.transitionCount++; } } } barrier.pTransitions = pPalTransitions; } else if (maxTransitionCount != 0) { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } // If app specifies the src/dst access masks in the subpass dependencies without layout transition at the end of // renderpass, cache will not be flushed according to PAL barrier logic, which will cause dirty values in the memory. // To fix the above issue, we construct a dumb transition to match PAL's logic to sync cache. // Construct a dumb transition to sync cache const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); if (settings.enableDumbTransitionSync && (barrier.transitionCount == 0) && (rpBarrier.flags.needsGlobalTransition)) { if (pPalTransitions == nullptr) { pPalTransitions = pVirtStack->AllocArray(1); } if (pPalTransitions != nullptr) { Pal::BarrierTransition* const pDumbTransition = &pPalTransitions[0]; *pDumbTransition = {}; barrier.transitionCount = 1; barrier.pTransitions = pDumbTransition; } } // Execute the barrier if it actually did anything if ((barrier.waitPoint != Pal::HwPipeBottom) || (barrier.transitionCount > 0) || ((barrier.pipePointWaitCount > 1) || ((barrier.pipePointWaitCount == 1) && (barrier.pPipePoints[0] != Pal::HwPipeTop)))) { PalCmdBarrier(&barrier, pPalTransitions, ppImages, GetRpDeviceMask()); } if (pPalTransitions != nullptr) { pVirtStack->FreeArray(pPalTransitions); } if (ppImages != nullptr) { pVirtStack->FreeArray(ppImages); } } // ===================================================================================================================== // Executes a "sync point" during a render pass instance. There are a number of these at different stages between // subpasses where we handle execution/memory dependencies from subpass dependencies as well as trigger automatic // layout transitions. void CmdBuffer::RPSyncPoint( const RPSyncPointInfo& syncPoint, VirtualStackFrame* pVirtStack) { const auto& rpBarrier = syncPoint.barrier; const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); if (m_flags.useReleaseAcquire) { Pal::AcquireReleaseInfo acquireReleaseInfo = {}; acquireReleaseInfo.reason = RgpBarrierExternalRenderPassSync; const uint32_t srcStageMask = VkToPalPipelineStageFlags(rpBarrier.srcStageMask, true); const uint32_t dstStageMask = VkToPalPipelineStageFlags(rpBarrier.dstStageMask, false); const uint32_t maxTransitionCount = MaxPalAspectsPerMask * syncPoint.transitionCount; Pal::ImgBarrier* pPalTransitions = (maxTransitionCount != 0) ? pVirtStack->AllocArray(maxTransitionCount) : nullptr; const Image** ppImages = (maxTransitionCount != 0) ? pVirtStack->AllocArray(maxTransitionCount) : nullptr; if ((pPalTransitions != nullptr) && (ppImages != nullptr)) { // Construct attachment-specific layout transitions for (uint32_t t = 0; t < syncPoint.transitionCount; ++t) { const RPTransitionInfo& tr = syncPoint.pTransitions[t]; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(tr.attachment); Pal::BarrierTransition imageTransition = { }; // Remove depth stencil related stage/access masks for color attachment transitions and remove color // target related stage/access mask for depth stencils. const uint32_t excludeStageMask = attachment.pImage->IsColorFormat() ? (~Pal::PipelineStageDsTarget) : (attachment.pImage->IsDepthStencilFormat() ? (~Pal::PipelineStageColorTarget) : (Pal::PipelineStageAllStages)); const uint32_t excludeAccessMask = attachment.pImage->IsColorFormat() ? (~Pal::CoherDepthStencilTarget) : (attachment.pImage->IsDepthStencilFormat() ? (~Pal::CoherColorTarget) : (Pal::CoherAllUsages)); for (uint32_t sr = 0; sr < attachment.subresRangeCount; ++sr) { const uint32_t plane = attachment.subresRange[sr].startSubres.plane; const bool isStencil = attachment.pImage->IsDepthStencilFormat() && (plane == 1); const RPImageLayout nextLayout = isStencil ? tr.nextStencilLayout : tr.nextLayout; const Pal::ImageLayout newLayout = attachment.pImage->GetAttachmentLayout( nextLayout, plane, this); const Pal::ImageLayout oldLayout = RPGetAttachmentLayout( tr.attachment, plane); m_pDevice->GetBarrierPolicy().ApplyBarrierCacheFlags( rpBarrier.srcAccessMask, rpBarrier.dstAccessMask, (isStencil ? tr.prevStencilLayout.layout : tr.prevLayout.layout), (isStencil ? tr.nextStencilLayout.layout : tr.nextLayout.layout), &imageTransition); const bool needsLayoutTransition = ((oldLayout.usages != newLayout.usages) || (oldLayout.engines != newLayout.engines)); uint32_t srcAccessMask = imageTransition.srcCacheMask | rpBarrier.implicitSrcCacheMask; uint32_t dstAccessMask = imageTransition.dstCacheMask | rpBarrier.implicitDstCacheMask; VK_ASSERT(acquireReleaseInfo.imageBarrierCount < maxTransitionCount); ppImages[acquireReleaseInfo.imageBarrierCount] = attachment.pImage; Pal::ImgBarrier* const pBarrier = &pPalTransitions[acquireReleaseInfo.imageBarrierCount]; *pBarrier = {}; pBarrier->srcStageMask = srcStageMask & excludeStageMask; pBarrier->dstStageMask = dstStageMask & excludeStageMask; pBarrier->srcAccessMask = srcAccessMask & excludeAccessMask; pBarrier->dstAccessMask = dstAccessMask & excludeAccessMask; // Subpass dependencies span all attachments, and some combinations of masks/accesses may // not make sense. If we are working on an image for which the stage masks are not applicable, // we should drop that transition, assuming we don't need to change the layout. if ((needsLayoutTransition == false) && ((pBarrier->srcStageMask == 0) || (pBarrier->dstStageMask == 0))) { continue; } // We set the pImage to nullptr by default here. But, this will be computed correctly later for // each device including DefaultDeviceIndex based on the deviceId. pBarrier->pImage = nullptr; pBarrier->oldLayout = oldLayout; pBarrier->newLayout = newLayout; pBarrier->subresRange = attachment.subresRange[sr]; const Pal::MsaaQuadSamplePattern* pQuadSamplePattern = nullptr; if (attachment.pImage->IsSampleLocationsCompatibleDepth() && tr.flags.isInitialLayoutTransition) { VK_ASSERT(attachment.pImage->HasDepth()); // Use the provided sample locations for this attachment if this is its // initial layout transition pQuadSamplePattern = &m_renderPassInstance.pAttachments[tr.attachment].initialSamplePattern.locations; } else { // Otherwise, use the subpass' sample locations uint32_t subpass = m_renderPassInstance.subpass; pQuadSamplePattern = &m_renderPassInstance.pSamplePatterns[subpass].locations; } pBarrier->pQuadSamplePattern = pQuadSamplePattern; RPSetAttachmentLayout(tr.attachment, plane, newLayout); acquireReleaseInfo.imageBarrierCount++; } } acquireReleaseInfo.pImageBarriers = pPalTransitions; } else if (maxTransitionCount != 0) { m_recordingResult = VK_ERROR_OUT_OF_HOST_MEMORY; } const bool stageMasksNotEmpty = (((srcStageMask == 0) && (dstStageMask == 0)) == false); // We do not require a dumb transition here in acquire/release interface because unlike Legacy barriers, // PAL flushes caches even if only the global barriers are passed-in without any image/buffer memory barriers. // Execute the barrier if it actually did anything if (stageMasksNotEmpty) { PalCmdReleaseThenAcquire( &acquireReleaseInfo, nullptr, nullptr, pPalTransitions, ppImages, GetRpDeviceMask()); } if (pPalTransitions != nullptr) { pVirtStack->FreeArray(pPalTransitions); } if (ppImages != nullptr) { pVirtStack->FreeArray(ppImages); } } else { RPSyncPointLegacy(syncPoint, pVirtStack); } } // ===================================================================================================================== // Does one or more load-op color clears during a render pass instance. void CmdBuffer::RPLoadOpClearColor( uint32_t count, const RPLoadOpClearInfo* pClears) { if (m_pSqttState != nullptr) { m_pSqttState->BeginRenderPassColorClear(); } VirtualStackFrame virtStackFrame(m_pStackAllocator); constexpr uint32 MinRects = 8; Util::Vector clearRegions{ &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::ClearBoundTargetRegion)), MinRects); auto rectBatch = Util::Min(count, maxRects); const auto palResult = clearRegions.Reserve(rectBatch); VK_ASSERT(palResult == Pal::Result::Success); for (uint32_t i = 0; i < count; ++i) { const RPLoadOpClearInfo& clear = pClears[i]; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(clear.attachment); const VkClearColorValue zeroClear = {{ 0.0f, 0.0f, 0.0f, 1.0f }}; // Convert the clear color to the format of the attachment view Pal::ClearColor clearColor = VkToPalClearColor( (clear.isOptional == false) ? m_renderPassInstance.pAttachments[clear.attachment].clearValue.color : zeroClear, attachment.viewFormat); Pal::BoundColorTarget target = {}; if (m_flags.subpassLoadOpClearsBoundAttachments) { const RenderPass* pRenderPass = m_allGpuState.pRenderPass; const uint32_t subpass = m_renderPassInstance.subpass; uint32_t tgtIdx = VK_ATTACHMENT_UNUSED; // Find color target of current attachment for (uint32_t colorTgt = 0; colorTgt < pRenderPass->GetSubpassColorReferenceCount(subpass); ++colorTgt) { const AttachmentReference& colorRef = pRenderPass->GetSubpassColorReference(subpass, colorTgt); if (clear.attachment == colorRef.attachment) { tgtIdx = colorTgt; break; } } VK_ASSERT(tgtIdx != VK_ATTACHMENT_UNUSED); target.targetIndex = tgtIdx; target.swizzledFormat = attachment.viewFormat; target.samples = pRenderPass->GetColorAttachmentSamples(subpass, tgtIdx); target.fragments = pRenderPass->GetColorAttachmentSamples(subpass, tgtIdx); target.clearValue = clearColor; } Pal::SubresRange subresRange; attachment.pView->GetFrameBufferAttachmentSubresRange(&subresRange); const Pal::ImageLayout clearLayout = RPGetAttachmentLayout(clear.attachment, subresRange.startSubres.plane); VK_ASSERT(clearLayout.usages & Pal::LayoutColorTarget); const auto clearSubresRanges = LoadOpClearSubresRanges( attachment, clear, *m_allGpuState.pRenderPass); utils::IterateMask deviceGroup(GetRpDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); Pal::Box clearBox = BuildClearBox(m_renderPassInstance.renderArea[deviceIdx], attachment); if (m_flags.subpassLoadOpClearsBoundAttachments == false) { // Multi-RT clears are synchronized later in RPBeginSubpass() uint32 flags = 0; if (count == 1) { flags |= Pal::ColorClearAutoSync; } if (clear.isOptional) { flags |= Pal::ColorClearSkipIfSlow; } PalCmdBuffer(deviceIdx)->CmdClearColorImage( *attachment.pImage->PalImage(deviceIdx), clearLayout, clearColor, attachment.viewFormat, clearSubresRanges.NumElements(), clearSubresRanges.Data(), 1, &clearBox, flags); } else if (clear.isOptional == false) // Don't attempt optional bound clears yet { const RenderPass* pRenderPass = m_allGpuState.pRenderPass; const uint32_t subpass = m_renderPassInstance.subpass; uint32_t viewMask = pRenderPass->GetViewMask(subpass); const VkRect2D rect = { { clearBox.offset.x, clearBox.offset.y }, // VkOffset2D offset; { clearBox.extent.width, clearBox.extent.height}, // VkExtent2D extent; }; const VkClearRect clearRect = { rect, // VkRect2D rect; static_cast(clearBox.offset.z), // deUint32 baseArrayLayer; clearBox.extent.depth // deUint32 layerCount; }; CreateClearRegions( 1, &clearRect, viewMask, 0u, &clearRegions); // Clear the bound color targets // TODO: Batch color targets in one call PalCmdBuffer(deviceIdx)->CmdClearBoundColorTargets( 1, &target, clearRegions.NumElements(), clearRegions.Data()); } } while (deviceGroup.IterateNext()); } if (m_pSqttState != nullptr) { m_pSqttState->EndRenderPassColorClear(); } } // ===================================================================================================================== // Does one or more load-op depth-stencil clears during a render pass instance. void CmdBuffer::RPLoadOpClearDepthStencil( uint32_t count, const RPLoadOpClearInfo* pClears) { if (m_pSqttState != nullptr) { m_pSqttState->BeginRenderPassDepthStencilClear(); } VirtualStackFrame virtStackFrame(m_pStackAllocator); constexpr uint32 MinRects = 8; Util::Vector clearRegions{ &virtStackFrame }; const auto maxRects = Util::Max(EstimateMaxObjectsOnVirtualStack(sizeof(Pal::ClearBoundTargetRegion)), MinRects); auto rectBatch = Util::Min(count, maxRects); for (uint32_t i = 0; i < count; ++i) { const RPLoadOpClearInfo& clear = pClears[i]; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(clear.attachment); const Pal::ImageLayout depthLayout = RPGetAttachmentLayout(clear.attachment, 0); const Pal::ImageLayout stencilLayout = RPGetAttachmentLayout(clear.attachment, 1); // Convert the clear color to the format of the attachment view const VkClearValue& clearValue = m_renderPassInstance.pAttachments[clear.attachment].clearValue; float clearDepth = VkToPalClearDepth(clearValue.depthStencil.depth); Pal::uint8 clearStencil = clearValue.depthStencil.stencil; const auto clearSubresRanges = LoadOpClearSubresRanges( attachment, clear, *m_allGpuState.pRenderPass); utils::IterateMask deviceGroup(GetRpDeviceMask()); Pal::SubresRange subresRange; attachment.pView->GetFrameBufferAttachmentSubresRange(&subresRange); ValidateSamplePattern( attachment.pImage->GetImageSamples(), &m_renderPassInstance.pAttachments[clear.attachment].initialSamplePattern); do { const uint32_t deviceIdx = deviceGroup.Index(); const Pal::Rect& palClearRect = m_renderPassInstance.renderArea[deviceIdx]; if (m_flags.subpassLoadOpClearsBoundAttachments == false) { PalCmdBuffer(deviceIdx)->CmdClearDepthStencil( *attachment.pImage->PalImage(deviceIdx), depthLayout, stencilLayout, clearDepth, clearStencil, StencilWriteMaskFull, clearSubresRanges.NumElements(), clearSubresRanges.Data(), 1, &palClearRect, Pal::DsClearAutoSync); } else { const auto palResult = clearRegions.Reserve(rectBatch); VK_ASSERT(palResult == Pal::Result::Success); const RenderPass* pRenderPass = m_allGpuState.pRenderPass; const uint32_t subpass = m_renderPassInstance.subpass; uint32_t viewMask = pRenderPass->GetViewMask(subpass); // Get the corresponding color reference in the current subpass const AttachmentReference& depthStencilRef = pRenderPass->GetSubpassDepthStencilReference(subpass); VK_ASSERT(depthStencilRef.attachment != VK_ATTACHMENT_UNUSED); Pal::DepthStencilSelectFlags selectFlags = {}; selectFlags.depth = ((clear.aspect & VK_IMAGE_ASPECT_DEPTH_BIT ) != 0); selectFlags.stencil = ((clear.aspect & VK_IMAGE_ASPECT_STENCIL_BIT) != 0); const VkRect2D rect = { { palClearRect.offset.x, palClearRect.offset.y }, // VkOffset2D offset; { palClearRect.extent.width, palClearRect.extent.height}, // VkExtent2D extent; }; const VkClearRect clearRect = { rect, // VkRect2D rect; 0u, // deUint32 baseArrayLayer; subresRange.numSlices // deUint32 layerCount; }; CreateClearRegions( 1, &clearRect, viewMask, 0u, &clearRegions); // Clear the bound depth stencil target immediately PalCmdBuffer(DefaultDeviceIndex)->CmdClearBoundDepthStencilTargets( clearDepth, clearStencil, StencilWriteMaskFull, pRenderPass->GetDepthStencilAttachmentSamples(subpass), pRenderPass->GetDepthStencilAttachmentSamples(subpass), selectFlags, clearRegions.NumElements(), clearRegions.Data()); } } while (deviceGroup.IterateNext()); } if (m_pSqttState != nullptr) { m_pSqttState->EndRenderPassDepthStencilClear(); } } // ===================================================================================================================== // Launches one or more resolves during a render pass instance. void CmdBuffer::RPResolveAttachments( uint32_t count, const RPResolveInfo* pResolves) { for (uint32_t i = 0; i < count; ++i) { const RPResolveInfo& params = pResolves[i]; { RPResolveMsaa(params); } } } // ===================================================================================================================== // Launches one or more MSAA resolves during a render pass instance. void CmdBuffer::RPResolveMsaa( const RPResolveInfo& params) { if (m_pSqttState != nullptr) { m_pSqttState->BeginRenderPassResolve(); } const Framebuffer::Attachment& srcAttachment = m_allGpuState.pFramebuffer->GetAttachment(params.src.attachment); const Framebuffer::Attachment& dstAttachment = m_allGpuState.pFramebuffer->GetAttachment(params.dst.attachment); // Both color and depth-stencil resolves are allowed by resolve attachments // SubresRange shall be exactly same for src and dst. VK_ASSERT(srcAttachment.subresRangeCount == dstAttachment.subresRangeCount); VK_ASSERT(srcAttachment.subresRange[0].numMips == 1); const uint32_t sliceCount = Util::Min( srcAttachment.subresRange[0].numSlices, dstAttachment.subresRange[0].numSlices); // We expect MSAA images to never have mipmaps VK_ASSERT(srcAttachment.subresRange[0].startSubres.mipLevel == 0); uint32_t aspectRegionCount = 0; uint32_t srcResolvePlanes[MaxRangePerAttachment] = {}; uint32_t dstResolvePlanes[MaxRangePerAttachment] = {}; const VkFormat srcResolveFormat = srcAttachment.pView->GetViewFormat(); const VkFormat dstResolveFormat = dstAttachment.pView->GetViewFormat(); Pal::ResolveMode resolveModes[MaxRangePerAttachment] = {}; const Pal::MsaaQuadSamplePattern* pSampleLocations = nullptr; if (Formats::IsDepthStencilFormat(srcResolveFormat) == false) { resolveModes[0] = Pal::ResolveMode::Average; srcResolvePlanes[0] = 0; dstResolvePlanes[0] = 0; aspectRegionCount = 1; } else { const uint32_t subpass = m_renderPassInstance.subpass; const VkResolveModeFlagBits depthResolveMode = m_allGpuState.pRenderPass->GetDepthResolveMode(subpass); const VkResolveModeFlagBits stencilResolveMode = m_allGpuState.pRenderPass->GetStencilResolveMode(subpass); if (Formats::HasDepth(srcResolveFormat)) { // Must be specified because the source image was created with sampleLocsAlwaysKnown set pSampleLocations = &m_renderPassInstance.pSamplePatterns[subpass].locations; } if (depthResolveMode != VK_RESOLVE_MODE_NONE) { if (Formats::HasDepth(srcResolveFormat) && Formats::HasDepth(dstResolveFormat)) { resolveModes[aspectRegionCount] = VkToPalResolveMode(depthResolveMode); srcResolvePlanes[aspectRegionCount] = 0; dstResolvePlanes[aspectRegionCount] = 0; aspectRegionCount++; } } if (stencilResolveMode != VK_RESOLVE_MODE_NONE) { if (Formats::HasStencil(srcResolveFormat) && Formats::HasStencil(dstResolveFormat)) { resolveModes[aspectRegionCount] = VkToPalResolveMode(stencilResolveMode); srcResolvePlanes[aspectRegionCount] = Formats::HasDepth(srcResolveFormat) ? 1 : 0; dstResolvePlanes[aspectRegionCount] = Formats::HasDepth(dstResolveFormat) ? 1 : 0; aspectRegionCount++; } } } // Depth and stencil might have different resolve mode, so allowing resolve each aspect independently. for (uint32_t aspectRegionIndex = 0; aspectRegionIndex < aspectRegionCount; ++aspectRegionIndex) { // During split-frame-rendering, the image to resolve could be split across multiple devices. Pal::ImageResolveRegion regions[MaxPalDevices]; const Pal::ImageLayout srcLayout = RPGetAttachmentLayout(params.src.attachment, srcResolvePlanes[aspectRegionIndex]); const Pal::ImageLayout dstLayout = RPGetAttachmentLayout(params.dst.attachment, dstResolvePlanes[aspectRegionIndex]); for (uint32_t idx = 0; idx < m_renderPassInstance.renderAreaCount; idx++) { const Pal::Rect& renderArea = m_renderPassInstance.renderArea[idx]; regions[idx].srcPlane = srcResolvePlanes[aspectRegionIndex]; regions[idx].srcSlice = srcAttachment.subresRange[0].startSubres.arraySlice; regions[idx].srcOffset.x = renderArea.offset.x; regions[idx].srcOffset.y = renderArea.offset.y; regions[idx].srcOffset.z = 0; regions[idx].dstPlane = dstResolvePlanes[aspectRegionIndex]; regions[idx].dstMipLevel = dstAttachment.subresRange[0].startSubres.mipLevel; regions[idx].dstSlice = dstAttachment.subresRange[0].startSubres.arraySlice; regions[idx].dstOffset.x = renderArea.offset.x; regions[idx].dstOffset.y = renderArea.offset.y; regions[idx].dstOffset.z = 0; regions[idx].extent.width = renderArea.extent.width; regions[idx].extent.height = renderArea.extent.height; regions[idx].extent.depth = 1; regions[idx].numSlices = sliceCount; if ((srcResolveFormat != srcAttachment.pImage->GetFormat()) || (dstResolveFormat != dstAttachment.pImage->GetFormat())) { // VUID-VkSubpassDescription-pResolveAttachments-00850: // each resolve attachment must have the same VkFormat as its corresponding color attachment VK_ASSERT(srcResolveFormat == dstResolveFormat); regions[idx].swizzledFormat = VkToPalFormat(srcResolveFormat, m_pDevice->GetRuntimeSettings()); } else { regions[idx].swizzledFormat = Pal::UndefinedSwizzledFormat; } regions[idx].pQuadSamplePattern = pSampleLocations; } if (aspectRegionIndex >= 1) { Pal::ImgBarrier imageBarrier = {}; imageBarrier.pImage = nullptr; // This is set by PalCmdReleaseThenAcquire imageBarrier.subresRange = dstAttachment.subresRange[0]; imageBarrier.srcAccessMask = Pal::CoherResolve; imageBarrier.dstAccessMask = Pal::CoherResolve; imageBarrier.srcStageMask = Pal::PipelineStageBlt; imageBarrier.dstStageMask = Pal::PipelineStageBlt; imageBarrier.oldLayout = dstLayout; imageBarrier.newLayout = dstLayout; Pal::AcquireReleaseInfo barrierInfo = { .srcGlobalStageMask = 0, .dstGlobalStageMask = 0, .srcGlobalAccessMask = 0, .dstGlobalAccessMask = 0, .memoryBarrierCount = 0, .pMemoryBarriers = nullptr, .imageBarrierCount = 1, .pImageBarriers = &imageBarrier, .reason = Pal::Developer::BarrierReason::BarrierReasonResolveImage }; const vk::Image* pImage = dstAttachment.pImage; PalCmdReleaseThenAcquire( &barrierInfo, nullptr, nullptr, &imageBarrier, &pImage, m_curDeviceMask); } PalCmdResolveImage( *srcAttachment.pImage, srcLayout, *dstAttachment.pImage, dstLayout, resolveModes[aspectRegionIndex], m_renderPassInstance.renderAreaCount, regions, GetRpDeviceMask()); } if (m_pSqttState != nullptr) { m_pSqttState->EndRenderPassResolve(); } } // ===================================================================================================================== // Binds color/depth targets for a subpass during a render pass instance. void CmdBuffer::RPBindTargets( const RPBindTargetsInfo& targets) { Pal::BindTargetParams params = {}; params.colorTargetCount = targets.colorTargetCount; static constexpr Pal::ImageLayout NullLayout = {}; utils::IterateMask deviceGroup(GetRpDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < targets.colorTargetCount; ++i) { const RPAttachmentReference& reference = targets.colorTargets[i]; if (reference.attachment != VK_ATTACHMENT_UNUSED) { const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(reference.attachment); RegisterWriteToFlippableImage(attachment.pImage); params.colorTargets[i].pColorTargetView = attachment.pView->PalColorTargetView(deviceIdx); params.colorTargets[i].imageLayout = RPGetAttachmentLayout(reference.attachment, 0); } else { params.colorTargets[i].pColorTargetView = nullptr; params.colorTargets[i].imageLayout = NullLayout; } } if (targets.depthStencil.attachment != VK_ATTACHMENT_UNUSED) { uint32_t attachmentIdx = targets.depthStencil.attachment; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(attachmentIdx); params.depthTarget.pDepthStencilView = attachment.pView->PalDepthStencilView(deviceIdx); params.depthTarget.depthLayout = RPGetAttachmentLayout(attachmentIdx, 0); params.depthTarget.stencilLayout = RPGetAttachmentLayout(attachmentIdx, 1); } else { params.depthTarget.pDepthStencilView = nullptr; params.depthTarget.depthLayout = NullLayout; params.depthTarget.stencilLayout = NullLayout; } PalCmdBuffer(deviceIdx)->CmdBindTargets(params); if (targets.fragmentShadingRateTarget.attachment != VK_ATTACHMENT_UNUSED) { uint32_t attachmentIdx = targets.fragmentShadingRateTarget.attachment; const Framebuffer::Attachment& attachment = m_allGpuState.pFramebuffer->GetAttachment(attachmentIdx); PalCmdBuffer(deviceIdx)->CmdBindSampleRateImage(attachment.pImage->PalImage(deviceIdx)); } } while (deviceGroup.IterateNext()); } // ===================================================================================================================== // Get Pal Image aspect layout from imageView void CmdBuffer::GetImageLayout( VkImageView imageView, VkImageLayout imageLayout, VkImageAspectFlags aspectMask, Pal::SubresRange* palSubresRange, Pal::ImageLayout* palImageLayout) { // Get the image view from the attachment info const ImageView* const pImageView = ImageView::ObjectFromHandle(imageView); // Get the attachment image const Image* pImage = pImageView->GetImage(); // Get subres range from the image view pImageView->GetFrameBufferAttachmentSubresRange(palSubresRange); palSubresRange->startSubres.plane = VkToPalImagePlaneSingle( pImage->GetFormat(), aspectMask, m_pDevice->GetRuntimeSettings()); // Get the Depth Layout from the view image *palImageLayout = pImage->GetBarrierPolicy().GetAspectLayout( imageLayout, palSubresRange->startSubres.plane, GetQueueFamilyIndex(), pImage->GetFormat()); } // ===================================================================================================================== // Binds color/depth targets for VK_KHR_dynamic_rendering void CmdBuffer::BindTargets() { Pal::BindTargetParams params = {}; params.colorTargetCount = m_allGpuState.dynamicRenderingInstance.colorAttachmentCount; utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < params.colorTargetCount; ++i) { const uint32_t location = m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i]; if (location != VK_ATTACHMENT_UNUSED) { const DynamicRenderingAttachments& renderingAttachmentInfo = m_allGpuState.dynamicRenderingInstance.colorAttachments[i]; if (renderingAttachmentInfo.pImageView != VK_NULL_HANDLE) { RegisterWriteToFlippableImage(renderingAttachmentInfo.pImageView->GetImage()); params.colorTargets[location].pColorTargetView = renderingAttachmentInfo.pImageView->PalColorTargetView(deviceIdx); params.colorTargets[location].imageLayout = renderingAttachmentInfo.imageLayout; } } } const DynamicRenderingAttachments& stencilAttachmentInfo = m_allGpuState.dynamicRenderingInstance.stencilAttachment; if (stencilAttachmentInfo.pImageView != VK_NULL_HANDLE) { params.depthTarget.pDepthStencilView = stencilAttachmentInfo.pImageView->PalDepthStencilView(deviceIdx); params.depthTarget.stencilLayout = stencilAttachmentInfo.imageLayout; } const DynamicRenderingAttachments& depthAttachmentInfo = m_allGpuState.dynamicRenderingInstance.depthAttachment; if (depthAttachmentInfo.pImageView != VK_NULL_HANDLE) { params.depthTarget.pDepthStencilView = depthAttachmentInfo.pImageView->PalDepthStencilView(deviceIdx); params.depthTarget.depthLayout = depthAttachmentInfo.imageLayout; } PalCmdBuffer(deviceIdx)->CmdBindTargets(params); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== // Sets view instance mask for a subpass during a render pass instance (on devices within passed in device mask). void CmdBuffer::SetViewInstanceMask( uint32_t deviceMask) { uint32_t subpassViewMask = 0; if (UsingDynamicRendering() == false) { subpassViewMask = m_allGpuState.pRenderPass->GetViewMask(m_renderPassInstance.subpass); } else if (m_allGpuState.dynamicRenderingInstance.viewMask > 0) { subpassViewMask = m_allGpuState.dynamicRenderingInstance.viewMask; } utils::IterateMask deviceGroup(deviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); const uint32_t deviceViewMask = uint32_t { 0x1 } << deviceIdx; uint32_t viewMask = 0x0; if (m_allGpuState.viewIndexFromDeviceIndex && (Util::CountSetBits(deviceMask) > 1)) { // VK_KHR_multiview interaction with VK_KHR_device_group. // When GraphicsPipeline is created with flag // VK_PIPELINE_CREATE_VIEW_INDEX_FROM_DEVICE_INDEX_BIT // rendering to views is split across multiple devices. // Essentially this flag allows application to divide work // between devices when multiview rendering is enabled. // Basically each device renders one view. // Vulkan Spec: VK_PIPELINE_CREATE_VIEW_INDEX_FROM_DEVICE_INDEX_BIT // specifies that any shader input variables decorated as DeviceIndex // will be assigned values as if they were decorated as ViewIndex. // To satisfy above requirement DeviceMask and ViewMask has to match. VK_ASSERT(m_curDeviceMask == viewMask); // Currently Vulkan CTS lacks tests covering this functionality. VK_NOT_TESTED(); viewMask = deviceViewMask; } else { // In default mode work is duplicated on each device, // because the same viewMask is set for all devices. // Basically each device renders all views. viewMask = subpassViewMask; } PalCmdBuffer(deviceIdx)->CmdSetViewInstanceMask(viewMask); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== // Ends a render pass instance (vkCmdEndRenderPass) void CmdBuffer::EndRenderPass() { DbgBarrierPreCmd(DbgBarrierEndRenderPass); if (m_renderPassInstance.subpass != VK_SUBPASS_EXTERNAL) { // Close the previous subpass RPEndSubpass(); // Get the end state for this render pass instance const RPExecuteEndRenderPassInfo& end = m_allGpuState.pRenderPass->GetExecuteInfo()->end; // Synchronize any prior work before leaving the instance (external dependencies) and also handle final layout // transitions. if (end.syncEnd.flags.active) { VirtualStackFrame virtStack(m_pStackAllocator); RPSyncPoint(end.syncEnd, &virtStack); } } // Clean up instance state m_allGpuState.pRenderPass = nullptr; m_allGpuState.pFramebuffer = nullptr; m_renderPassInstance.pExecuteInfo = nullptr; DbgBarrierPostCmd(DbgBarrierEndRenderPass); } // ===================================================================================================================== void CmdBuffer::WritePushConstants( PipelineBindPoint apiBindPoint, Pal::PipelineBindPoint palBindPoint, const PipelineLayout* pLayout, uint32_t startInDwords, uint32_t lengthInDwords, const uint32_t* const pInputValues) { PipelineBindState* pBindState = &m_allGpuState.pipelineState[apiBindPoint]; Pal::uint32* pUserData = reinterpret_cast(&pBindState->pushConstData[0]); uint32_t* pUserDataPtr = pUserData + startInDwords; for (uint32_t i = 0; i < lengthInDwords; i++) { pUserDataPtr[i] = pInputValues[i]; } pBindState->pushedConstCount = Util::Max(pBindState->pushedConstCount, startInDwords + lengthInDwords); const UserDataLayout& userDataLayout = pLayout->GetInfo().userDataLayout; // Program the user data register only if the current user data layout base matches that of the given // layout. Otherwise, what's happening is that the application is pushing constants for a future // pipeline layout (e.g. at the top of the command buffer) and this register write will be redundant because // a future vkCmdBindPipeline will reprogram the user data registers during the rebase. if (PalPipelineBindingOwnedBy(palBindPoint, apiBindPoint) && (pBindState->userDataLayout.common.pushConstRegBase == userDataLayout.common.pushConstRegBase) && (pBindState->userDataLayout.common.pushConstRegCount >= (startInDwords + lengthInDwords))) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdSetUserData( palBindPoint, pBindState->userDataLayout.common.pushConstRegBase + startInDwords, lengthInDwords, pUserDataPtr); } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== // Set push constant values void CmdBuffer::PushConstants( VkPipelineLayout layout, VkShaderStageFlags stageFlags, uint32_t start, uint32_t length, const void* values) { DbgBarrierPreCmd(DbgBarrierBindSetsPushConstants); uint32_t startInDwords = start / sizeof(uint32_t); uint32_t lengthInDwords = PipelineLayout::GetPushConstantSizeInDword(length); const uint32_t* const pInputValues = reinterpret_cast(values); const PipelineLayout* pLayout = PipelineLayout::ObjectFromHandle(layout); stageFlags &= m_validShaderStageFlags; PushConstantsIssueWrites(pLayout, stageFlags, startInDwords, lengthInDwords, pInputValues); DbgBarrierPostCmd(DbgBarrierBindSetsPushConstants); } // ===================================================================================================================== void CmdBuffer::PushConstantsIssueWrites( const PipelineLayout* pLayout, VkShaderStageFlags stageFlags, uint32_t startInDwords, uint32_t lengthInDwords, const uint32_t* const pInputValues) { if ((stageFlags & VK_SHADER_STAGE_COMPUTE_BIT) != 0) { WritePushConstants(PipelineBindCompute, Pal::PipelineBindPoint::Compute, pLayout, startInDwords, lengthInDwords, pInputValues); } #if VKI_RAY_TRACING if ((stageFlags & RayTraceShaderStages) != 0) { WritePushConstants(PipelineBindRayTracing, Pal::PipelineBindPoint::Compute, pLayout, startInDwords, lengthInDwords, pInputValues); } #endif if ((stageFlags & ShaderStageAllGraphics) != 0) { WritePushConstants(PipelineBindGraphics, Pal::PipelineBindPoint::Graphics, pLayout, startInDwords, lengthInDwords, pInputValues); } } // ===================================================================================================================== // Creates or grows an internal descriptor set for the command buffer to push template VkDescriptorSet CmdBuffer::InitPushDescriptorSet( const DescriptorSetLayout* pDestSetLayout, const PipelineLayout::SetUserDataLayout& setLayoutInfo, const size_t descriptorSetSize, PipelineBindPoint bindPoint, const uint32_t alignmentInDwords) { // The descriptor writes must go to the command buffer's shadow to handle incremental updates. // Any used descriptors are required to be pushed before the pipeline is executed or else they are undefined, // which means the last push descriptor set's value or uninitialized memory because no special care is taken here. DescriptorSet* pSet = DescriptorSet::ObjectFromHandle( m_allGpuState.pipelineState[bindPoint].pushDescriptorSet); // Reuse the existing shadow buffer unless it wasn't created or needs to grow. if (descriptorSetSize > m_allGpuState.pipelineState[bindPoint].pushDescriptorSetMaxSize) { const size_t objSize = Util::Pow2Align(sizeof(DescriptorSet), VK_DEFAULT_MEM_ALIGN); // Note that descriptor sets don't require a destructor to be called m_pDevice->VkInstance()->FreeMem(m_allGpuState.pipelineState[bindPoint].pPushDescriptorSetMemory); void* pSetMem = m_pDevice->VkInstance()->AllocMem( (descriptorSetSize * numPalDevices) + objSize, (alignmentInDwords * sizeof(uint32_t)), VK_SYSTEM_ALLOCATION_SCOPE_OBJECT); if (pSetMem != nullptr) { pSet = VK_PLACEMENT_NEW (Util::VoidPtrInc(pSetMem, (descriptorSetSize * numPalDevices))) DescriptorSet(0); // Store the API handle to avoid templated parameters when using it. m_allGpuState.pipelineState[bindPoint].pushDescriptorSet = DescriptorSet::HandleFromObject(pSet); m_allGpuState.pipelineState[bindPoint].pPushDescriptorSetMemory = pSetMem; m_allGpuState.pipelineState[bindPoint].pushDescriptorSetMaxSize = descriptorSetSize; } else { PAL_ASSERT_ALWAYS(); pSet = nullptr; m_allGpuState.pipelineState[bindPoint].pushDescriptorSet = VK_NULL_HANDLE; m_allGpuState.pipelineState[bindPoint].pPushDescriptorSetMemory = nullptr; m_allGpuState.pipelineState[bindPoint].pushDescriptorSetMaxSize = 0; } } if (pSet != nullptr) { DescriptorAddr descriptorAddrs[numPalDevices] = {}; // If there is a set pointer, the shadow memory is that of the push descriptor set. Otherwise, the descriptor // set is written inline to the command buffer binding data set shadow memory. if (setLayoutInfo.setPtrRegOffset != PipelineLayout::InvalidReg) { for (uint32_t deviceIdx = 0; deviceIdx < numPalDevices; deviceIdx++) { descriptorAddrs[deviceIdx].staticCpuAddr = static_cast(Util::VoidPtrInc( m_allGpuState.pipelineState[bindPoint].pPushDescriptorSetMemory, descriptorSetSize * deviceIdx)); } } else { for (uint32_t deviceIdx = 0; deviceIdx < numPalDevices; deviceIdx++) { descriptorAddrs[deviceIdx].staticCpuAddr = &(PerGpuState(deviceIdx)->setBindingData[bindPoint][setLayoutInfo.firstRegOffset]); } } pSet->Reassign(pDestSetLayout, 0, descriptorAddrs, nullptr); } // Push descriptor sets don't use vkAllocateDescriptorSets, so if they must be written to the descriptor set, // every push of an immutable sampler must be honored instead of skipping as we do today. Write them all here // until it's known if not skipping them must be implemented. if (m_pDevice->MustWriteImmutableSamplers()) { VK_NOT_IMPLEMENTED; pSet->WriteImmutableSamplers(m_pDevice->GetProperties().descriptorSizes.imageView); } return DescriptorSet::HandleFromObject(pSet); } // ===================================================================================================================== template void CmdBuffer::PushDescriptorSet( VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t set, uint32_t descriptorWriteCount, const VkWriteDescriptorSet* pDescriptorWrites) { DbgBarrierPreCmd(DbgBarrierPushDescriptorSet); const PipelineLayout* pLayout = PipelineLayout::ObjectFromHandle(layout); const DescriptorSetLayout* pDestSetLayout = pLayout->GetSetLayouts(set); const PipelineLayout::SetUserDataLayout& setLayoutInfo = pLayout->GetSetUserData(set); const uint8 setPtrRegOffset = setLayoutInfo.setPtrRegOffset; Pal::PipelineBindPoint palBindPoint; PipelineBindPoint apiBindPoint; ConvertPipelineBindPoint(pipelineBindPoint, &palBindPoint, &apiBindPoint); const uint32_t descriptorSetSizeInDwords = pDestSetLayout->Info().sta.dwSize; const uint32_t alignmentInDwords = m_pDevice->GetProperties().descriptorSizes.alignmentInDwords; // An internal descriptor set is used to represent the shadow to be consistent with the // vkCmdPushDescriptorSetWithTemplate implementation only. WriteDescriptorSets would have to have // been modified to accept the destination set as a new parameter instead of using VkWriteDescriptorSet. VkDescriptorSet pushDescriptorSet = InitPushDescriptorSet( pDestSetLayout, setLayoutInfo, (descriptorSetSizeInDwords * sizeof(uint32_t)), apiBindPoint, alignmentInDwords); DescriptorSet* pDestSet = DescriptorSet::ObjectFromHandle(pushDescriptorSet); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); // Issue the descriptor writes using the destination address of the command buffer's shadow rather than the // descriptor set memory; the dstSet member of VkWriteDescriptorSet must be ignored for push descriptors. for (uint32_t i = 0; i < descriptorWriteCount; ++i) { const VkWriteDescriptorSet& params = pDescriptorWrites[i]; const DescriptorSetLayout::BindingInfo& destBinding = pDestSetLayout->Binding(params.dstBinding); uint32_t* pDestAddr = pDestSet->StaticCpuAddress(deviceIdx) + pDestSetLayout->GetDstStaOffset(destBinding, params.dstArrayElement); // Determine whether the binding has immutable sampler descriptors. const bool hasImmutableSampler = (destBinding.imm.dwSize != 0); switch (static_cast(params.descriptorType)) { case VK_DESCRIPTOR_TYPE_SAMPLER: if (hasImmutableSampler == false) { DescriptorUpdate::WriteSamplerDescriptors( params.pImageInfo, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); } break; case VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER: if (hasImmutableSampler) { if (destBinding.bindingFlags.ycbcrConversionUsage == 0) { // If the sampler part of the combined image sampler is immutable then we should only update // the image descriptors, but have to make sure to still use the appropriate stride. DescriptorUpdate::WriteImageDescriptors( params.pImageInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); } else { DescriptorUpdate::WriteImageDescriptorsYcbcr( params.pImageInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); } } else { DescriptorUpdate::WriteImageSamplerDescriptors( params.pImageInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); } break; case VK_DESCRIPTOR_TYPE_STORAGE_IMAGE: DescriptorUpdate::WriteImageDescriptors( params.pImageInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; case VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE: case VK_DESCRIPTOR_TYPE_INPUT_ATTACHMENT: DescriptorUpdate::WriteImageDescriptors( params.pImageInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; case VK_DESCRIPTOR_TYPE_UNIFORM_TEXEL_BUFFER: DescriptorUpdate::WriteBufferDescriptors( params.pTexelBufferView, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; case VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER: DescriptorUpdate::WriteBufferDescriptors( params.pTexelBufferView, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; case VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER: DescriptorUpdate::WriteBufferInfoDescriptors( m_pDevice, params.pBufferInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; case VK_DESCRIPTOR_TYPE_STORAGE_BUFFER: DescriptorUpdate::WriteBufferInfoDescriptors( m_pDevice, params.pBufferInfo, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; #if VKI_RAY_TRACING case VK_DESCRIPTOR_TYPE_ACCELERATION_STRUCTURE_KHR: { const auto* pWriteAccelStructKHR = reinterpret_cast( utils::GetExtensionStructure(reinterpret_cast(params.pNext), static_cast(VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET_ACCELERATION_STRUCTURE_KHR))); VK_ASSERT(pWriteAccelStructKHR != nullptr); VK_ASSERT(pWriteAccelStructKHR->accelerationStructureCount == params.descriptorCount); DescriptorUpdate::WriteAccelerationStructureDescriptors( m_pDevice, pWriteAccelStructKHR->pAccelerationStructures, deviceIdx, pDestAddr, params.descriptorCount, destBinding.sta.dwArrayStride); break; } #endif case VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC: case VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC: case VK_DESCRIPTOR_TYPE_INLINE_UNIFORM_BLOCK: default: VK_ASSERT(!"Unexpected descriptor type"); break; } } // If there is a set pointer, update the push descriptor set from the command buffer shadow set to an embedded // memory allocation. Otherwise, the shadow set contents will be directly written to user data instead of this // push descriptor set pointer. if (setPtrRegOffset != PipelineLayout::InvalidReg) { Pal::gpusize gpuAddr; uint32* pCpuAddr = PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData(descriptorSetSizeInDwords, alignmentInDwords, &gpuAddr); memcpy(pCpuAddr, pDestSet->StaticCpuAddress(deviceIdx), (descriptorSetSizeInDwords * sizeof(uint32_t))); // CmdAllocateEmbeddedData is allocated out of VaRange::DescriptorTable, so the upper half is // known by the shader as is the case for our descriptor pool allocations. PerGpuState(deviceIdx)->setBindingData[apiBindPoint][setPtrRegOffset] = static_cast(gpuAddr); } SetUserDataPipelineLayout(set, 1, pLayout, palBindPoint, apiBindPoint); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierPushDescriptorSet); } // ===================================================================================================================== template void CmdBuffer::PushDescriptorSetWithTemplate( VkDescriptorUpdateTemplate descriptorUpdateTemplate, VkPipelineLayout layout, uint32_t set, const void* pData) { DbgBarrierPreCmd(DbgBarrierPushDescriptorSet); const PipelineLayout* pLayout = PipelineLayout::ObjectFromHandle(layout); const DescriptorSetLayout* pDestSetLayout = pLayout->GetSetLayouts(set); DescriptorUpdateTemplate* pTemplate = DescriptorUpdateTemplate::ObjectFromHandle(descriptorUpdateTemplate); Pal::PipelineBindPoint palBindPoint; PipelineBindPoint apiBindPoint; ConvertPipelineBindPoint(pTemplate->GetPipelineBindPoint(), &palBindPoint, &apiBindPoint); const uint32_t descriptorSetSizeInDwords = pDestSetLayout->Info().sta.dwSize; const uint32_t alignmentInDwords = m_pDevice->GetProperties().descriptorSizes.alignmentInDwords; const PipelineLayout::SetUserDataLayout& setLayoutInfo = pLayout->GetSetUserData(set); // An internal descriptor set is used to represent the shadow to utilize normal descriptor write support // for updating the shadow. Push descriptors can be represented by only the static section of the descriptor set // layout because not all descriptor types are supported. VkDescriptorSet pushDescriptorSet = InitPushDescriptorSet( pDestSetLayout, setLayoutInfo, (descriptorSetSizeInDwords * sizeof(uint32_t)), apiBindPoint, alignmentInDwords); // Issue the descriptor template update using the internal descriptor set to use the destination address of the // command buffer's shadow rather than descriptor pool memory like regular descriptor sets. pTemplate->Update( m_pDevice, pushDescriptorSet, pData); const uint8 setPtrRegOffset = setLayoutInfo.setPtrRegOffset; utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); // If there is a set pointer, update the push descriptor set from the command buffer shadow set to an embedded // memory allocation. Otherwise, the shadow set contents will be directly written to user data instead of this // push descriptor set pointer. if (setPtrRegOffset != PipelineLayout::InvalidReg) { Pal::gpusize gpuAddr; uint32* pCpuAddr = PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData(descriptorSetSizeInDwords, alignmentInDwords, &gpuAddr); const DescriptorSet* pShadowSet = DescriptorSet::ObjectFromHandle(pushDescriptorSet); memcpy(pCpuAddr, pShadowSet->StaticCpuAddress(deviceIdx), (descriptorSetSizeInDwords * sizeof(uint32_t))); // CmdAllocateEmbeddedData is allocated out of VaRange::DescriptorTable, so the upper half is // known by the shader as is the case for our descriptor pool allocations. PerGpuState(deviceIdx)->setBindingData[apiBindPoint][setPtrRegOffset] = static_cast(gpuAddr); } SetUserDataPipelineLayout(set, 1, pLayout, palBindPoint, apiBindPoint); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgBarrierPushDescriptorSet); } // ===================================================================================================================== void CmdBuffer::SetViewport( uint32_t firstViewport, uint32_t viewportCount, const VkViewport* pViewports) { // If we hit this assert the application did not set the right number of viewports // in VkPipelineViewportStateCreateInfo.viewportCount. // VK_ASSERT((firstViewport + viewportCount) <= m_state.viewport.count); const bool khrMaintenance1 = ((m_pDevice->VkPhysicalDevice(DefaultDeviceIndex)->GetEnabledAPIVersion() >= VK_MAKE_API_VERSION( 0, 1, 1, 0)) || m_pDevice->IsExtensionEnabled(DeviceExtensions::KHR_MAINTENANCE1)); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIndex = deviceGroup.Index(); for (uint32_t i = 0; i < viewportCount; ++i) { VkToPalViewport(pViewports[i], firstViewport + i, khrMaintenance1, &PerGpuState(deviceIndex)->viewport); } } while (deviceGroup.IterateNext()); m_allGpuState.dirtyGraphics.viewport = 1; m_allGpuState.staticTokens.viewports = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetViewportWithCount( uint32_t viewportCount, const VkViewport* pViewports) { utils::IterateMask deviceGroup(m_curDeviceMask); do { PerGpuState(deviceGroup.Index())->viewport.count = viewportCount; } while (deviceGroup.IterateNext()); SetViewport(0, viewportCount, pViewports); } // ===================================================================================================================== void CmdBuffer::SetAllViewports( const Pal::ViewportParams& params, uint32_t staticToken) { VK_ASSERT(m_cbBeginDeviceMask == m_pDevice->GetPalDeviceMask()); utils::IterateMask deviceGroup(m_cbBeginDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < params.count; ++i) { PerGpuState(deviceIdx)->viewport.viewports[i] = params.viewports[i]; } PerGpuState(deviceIdx)->viewport.count = params.count; PerGpuState(deviceIdx)->viewport.depthRange = params.depthRange; } while (deviceGroup.IterateNext()); m_allGpuState.dirtyGraphics.viewport = 1; m_allGpuState.staticTokens.viewports = staticToken; } // ===================================================================================================================== void CmdBuffer::CmdSetDepthClampRangeEXT( VkDepthClampModeEXT depthClampMode, const VkDepthClampRangeEXT* pDepthClampRange) { switch (depthClampMode) { case VK_DEPTH_CLAMP_MODE_VIEWPORT_RANGE_EXT: m_allGpuState.depthClampOverride.minDepthClamp = 1.0f; m_allGpuState.depthClampOverride.maxDepthClamp = 0.0f; break; case VK_DEPTH_CLAMP_MODE_USER_DEFINED_RANGE_EXT: m_allGpuState.depthClampOverride = *pDepthClampRange; break; default: VK_ASSERT(!"Unexpected depthClampMode"); break; } } // ===================================================================================================================== void CmdBuffer::SetScissor( uint32_t firstScissor, uint32_t scissorCount, const VkRect2D* pScissors) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < scissorCount; ++i) { VkToPalScissorRect(pScissors[i], firstScissor + i, &PerGpuState(deviceIdx)->scissor); } } while (deviceGroup.IterateNext()); m_allGpuState.dirtyGraphics.scissor = 1; m_allGpuState.staticTokens.scissorRect = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetScissorWithCount( uint32_t scissorCount, const VkRect2D* pScissors) { utils::IterateMask deviceGroup(m_curDeviceMask); do { PerGpuState(deviceGroup.Index())->scissor.count = scissorCount; } while (deviceGroup.IterateNext()); SetScissor(0, scissorCount, pScissors); } // ===================================================================================================================== void CmdBuffer::SetAllScissors( const Pal::ScissorRectParams& params, uint32_t staticToken) { VK_ASSERT(m_cbBeginDeviceMask == m_pDevice->GetPalDeviceMask()); utils::IterateMask deviceGroup(m_cbBeginDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PerGpuState(deviceIdx)->scissor.count = params.count; for (uint32_t i = 0; i < params.count; ++i) { PerGpuState(deviceIdx)->scissor.scissors[i] = params.scissors[i]; } } while (deviceGroup.IterateNext()); m_allGpuState.dirtyGraphics.scissor = 1; m_allGpuState.staticTokens.scissorRect = staticToken; } // ===================================================================================================================== void CmdBuffer::SetLineWidth( float lineWidth) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); const VkPhysicalDeviceLimits& limits = m_pDevice->VkPhysicalDevice(DefaultDeviceIndex)->GetLimits(); const Pal::PointLineRasterStateParams params = { DefaultPointSize, Util::Clamp(lineWidth, limits.lineWidthRange[0], limits.lineWidthRange[1]), limits.pointSizeRange[0], limits.pointSizeRange[1] }; utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetPointLineRasterState(params); } while (deviceGroup.IterateNext()); m_allGpuState.staticTokens.pointLineRasterState = DynamicRenderStateToken; DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } // ===================================================================================================================== void CmdBuffer::SetDepthBias( float depthBias, float depthBiasClamp, float slopeScaledDepthBias) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); const Pal::DepthBiasParams params = {depthBias, depthBiasClamp, slopeScaledDepthBias}; utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetDepthBiasState(params); } while (deviceGroup.IterateNext()); m_allGpuState.staticTokens.depthBiasState = DynamicRenderStateToken; DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } // ===================================================================================================================== void CmdBuffer::SetBlendConstants( const float blendConst[4]) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); const Pal::BlendConstParams params = { blendConst[0], blendConst[1], blendConst[2], blendConst[3] }; utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetBlendConst(params); } while (deviceGroup.IterateNext()); m_allGpuState.staticTokens.blendConst = DynamicRenderStateToken; DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } // ===================================================================================================================== void CmdBuffer::SetDepthBounds( float minDepthBounds, float maxDepthBounds) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); const Pal::DepthBoundsParams params = { minDepthBounds, maxDepthBounds }; utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetDepthBounds(params); } while (deviceGroup.IterateNext()); m_allGpuState.staticTokens.depthBounds = DynamicRenderStateToken; DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } // ===================================================================================================================== void CmdBuffer::SetStencilCompareMask( VkStencilFaceFlags faceMask, uint32_t stencilCompareMask) { if (faceMask & VK_STENCIL_FACE_FRONT_BIT) { m_allGpuState.stencilRefMasks.frontReadMask = static_cast(stencilCompareMask); } if (faceMask & VK_STENCIL_FACE_BACK_BIT) { m_allGpuState.stencilRefMasks.backReadMask = static_cast(stencilCompareMask); } m_allGpuState.dirtyGraphics.stencilRef = 1; } // ===================================================================================================================== void CmdBuffer::SetStencilWriteMask( VkStencilFaceFlags faceMask, uint32_t stencilWriteMask) { if (faceMask & VK_STENCIL_FACE_FRONT_BIT) { m_allGpuState.stencilRefMasks.frontWriteMask = static_cast(stencilWriteMask); } if (faceMask & VK_STENCIL_FACE_BACK_BIT) { m_allGpuState.stencilRefMasks.backWriteMask = static_cast(stencilWriteMask); } m_allGpuState.dirtyGraphics.stencilRef = 1; } // ===================================================================================================================== void CmdBuffer::SetStencilReference( VkStencilFaceFlags faceMask, uint32_t stencilReference) { if (faceMask & VK_STENCIL_FACE_FRONT_BIT) { m_allGpuState.stencilRefMasks.frontRef = static_cast(stencilReference); } if (faceMask & VK_STENCIL_FACE_BACK_BIT) { m_allGpuState.stencilRefMasks.backRef = static_cast(stencilReference); } m_allGpuState.dirtyGraphics.stencilRef = 1; } // ===================================================================================================================== // Calculate the hash of dynamic vertex input info static uint64_t GetDynamicVertexInputHash( uint32_t vertexBindingDescriptionCount, const VkVertexInputBindingDescription2EXT* pVertexBindingDescriptions, uint32_t vertexAttributeDescriptionCount, const VkVertexInputAttributeDescription2EXT* pVertexAttributeDescriptions) { Util::MetroHash::Hash hash = {}; if (vertexBindingDescriptionCount > 0) { VK_ASSERT(vertexAttributeDescriptionCount > 0); Util::MetroHash64 hasher; hasher.Update(reinterpret_cast(pVertexBindingDescriptions), sizeof(VkVertexInputBindingDescription2EXT) * vertexBindingDescriptionCount); hasher.Update(reinterpret_cast(pVertexAttributeDescriptions), sizeof(VkVertexInputAttributeDescription2EXT) * vertexAttributeDescriptionCount); hasher.Finalize(hash.bytes); } return hash.qwords[0]; } // ===================================================================================================================== // Builds uber-fetch shader internal data according to dynamic vertex input info. DynamicVertexInputInternalData* CmdBuffer::BuildUberFetchShaderInternalData( uint32_t vertexBindingDescriptionCount, const VkVertexInputBindingDescription2EXT* pVertexBindingDescriptions, uint32_t vertexAttributeDescriptionCount, const VkVertexInputAttributeDescription2EXT* pVertexAttributeDescriptions) { uint64_t vertexInputHash = GetDynamicVertexInputHash( vertexBindingDescriptionCount, pVertexBindingDescriptions, vertexAttributeDescriptionCount, pVertexAttributeDescriptions); DynamicVertexInputInternalData* pVertexInputData = nullptr; bool existed = false; Util::Result result = m_uberFetchShaderInternalDataMap.FindAllocate(vertexInputHash, &existed, &pVertexInputData); if (result == Util::Result::Success) { if (existed == false) { if (m_pUberFetchShaderTempBuffer == nullptr) { m_pUberFetchShaderTempBuffer = m_pDevice->VkInstance()->AllocMem( PipelineCompiler::GetMaxUberFetchShaderInternalDataSize() * NumPalDevices(), VK_SYSTEM_ALLOCATION_SCOPE_COMMAND); } if (m_pUberFetchShaderTempBuffer != nullptr) { void* pUberFetchShaderInternalData = m_pUberFetchShaderTempBuffer; bool isDynamicStride = (m_pDevice->GetEnabledFeatures().deviceGeneratedCommands == true); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); uint32_t uberFetchShaderInternalDataSize = m_pDevice->GetCompiler(deviceIdx)->BuildUberFetchShaderInternalData( vertexBindingDescriptionCount, pVertexBindingDescriptions, vertexAttributeDescriptionCount, pVertexAttributeDescriptions, isDynamicStride, m_flags.offsetMode, pUberFetchShaderInternalData); Pal::gpusize gpuAddress = {}; if (uberFetchShaderInternalDataSize > 0) { void* pCpuAddr = PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData( uberFetchShaderInternalDataSize, 1, &gpuAddress); memcpy(pCpuAddr, pUberFetchShaderInternalData, uberFetchShaderInternalDataSize); } pVertexInputData->gpuAddress[deviceIdx] = gpuAddress; pUberFetchShaderInternalData = Util::VoidPtrInc(pUberFetchShaderInternalData, uberFetchShaderInternalDataSize); } while (deviceGroup.IterateNext()); // we needn't set any user data if internal size is 0. if (pVertexInputData->gpuAddress[0] == 0) { pVertexInputData = nullptr; } } else { // return nullptr for any fail case. VK_NEVER_CALLED(); pVertexInputData = nullptr; } } } else { VK_NEVER_CALLED(); pVertexInputData = nullptr; } return pVertexInputData; } // ===================================================================================================================== void CmdBuffer::SetVertexInput( uint32_t vertexBindingDescriptionCount, const VkVertexInputBindingDescription2EXT* pVertexBindingDescriptions, uint32_t vertexAttributeDescriptionCount, const VkVertexInputAttributeDescription2EXT* pVertexAttributeDescriptions) { PipelineBindState* pBindState = &m_allGpuState.pipelineState[PipelineBindGraphics]; const bool padVertexBuffers = m_flags.padVertexBuffers; pBindState->pVertexInputInternalData = BuildUberFetchShaderInternalData( vertexBindingDescriptionCount, pVertexBindingDescriptions, vertexAttributeDescriptionCount, pVertexAttributeDescriptions); if (pBindState->pVertexInputInternalData != nullptr) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); // Upload internal memory if (pBindState->hasDynamicVertexInput && (m_allGpuState.pGraphicsPipeline != nullptr)) { VK_ASSERT(GetUberFetchShaderUserData(&pBindState->userDataLayout) != PipelineLayout::InvalidReg); uint32_t internalBufferLow = pBindState->pVertexInputInternalData->gpuAddress[deviceIdx] & UINT32_MAX; PalCmdBuffer(deviceIdx)->CmdSetUserData( Pal::PipelineBindPoint::Graphics, GetUberFetchShaderUserData(&pBindState->userDataLayout), 1, &internalBufferLow); } // Update vertex buffer stride uint32_t firstChanged = UINT_MAX; uint32_t lastChanged = 0; uint32_t vertexBufferCount = 0; Pal::BufferViewInfo* pVbBindings = PerGpuState(deviceIdx)->vbBindings; for (uint32_t bindex = 0; bindex < vertexBindingDescriptionCount; ++bindex) { uint32_t byteStride = pVertexBindingDescriptions[bindex].stride; uint32_t binding = pVertexBindingDescriptions[bindex].binding; vertexBufferCount = Util::Max(binding + 1, vertexBufferCount); Pal::BufferViewInfo* pBinding = &pVbBindings[binding]; if (pBinding->stride != byteStride) { pBinding->stride = byteStride; if (pBinding->gpuAddr != 0) { firstChanged = Util::Min(firstChanged, binding); lastChanged = Util::Max(lastChanged, binding); } if (padVertexBuffers && (pBinding->stride != 0)) { pBinding->range = Util::RoundUpToMultiple(pBinding->range, pBinding->stride); } } } if (firstChanged <= lastChanged) { auto pBinding = &PerGpuState(deviceIdx)->vbBindings[firstChanged]; if (m_flags.offsetMode) { Pal::VertexBufferView vertexViews[Pal::MaxVertexBuffers] = {}; for (uint32_t idx = 0; idx < (lastChanged - firstChanged + 1); idx++) { vertexViews[idx].gpuva = pBinding[idx].gpuAddr; vertexViews[idx].sizeInBytes = pBinding[idx].range; vertexViews[idx].strideInBytes = pBinding[idx].stride; } const Pal::VertexBufferViews bufferViews = { .firstBuffer = firstChanged, .bufferCount = (lastChanged - firstChanged) + 1, .offsetMode = true, .pVertexBufferViews = vertexViews }; PalCmdBuffer(deviceIdx)->CmdSetVertexBuffers(bufferViews); } else { const Pal::VertexBufferViews bufferViews = { .firstBuffer = firstChanged, .bufferCount = (lastChanged - firstChanged) + 1, .offsetMode = false, .pBufferViewInfos = pBinding }; PalCmdBuffer(deviceIdx)->CmdSetVertexBuffers(bufferViews); } } if (vertexBufferCount != pBindState->dynamicBindInfo.gfxDynState.vertexBufferCount) { pBindState->dynamicBindInfo.gfxDynState.vertexBufferCount = vertexBufferCount; m_allGpuState.dirtyGraphics.pipeline = 1; } } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== void CmdBuffer::SetRenderingAttachmentLocations( const VkRenderingAttachmentLocationInfoKHR* pLocationInfo) { if ((pLocationInfo != nullptr) && (pLocationInfo->pColorAttachmentLocations != nullptr)) { for (uint32_t i = 0; i < pLocationInfo->colorAttachmentCount; ++i) { m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i] = pLocationInfo->pColorAttachmentLocations[i]; } BindTargets(); } } // ===================================================================================================================== void CmdBuffer::SetRenderingInputAttachmentIndices( const VkRenderingInputAttachmentIndexInfoKHR* pInputAttachmentIndexInfo) { } #if VKI_ENABLE_DEBUG_BARRIERS // ===================================================================================================================== // This function inserts a command before or after a particular Vulkan command if the given runtime settings are asking // for it. void CmdBuffer::DbgCmdBarrier(bool preCmd) { const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); static_assert(( (static_cast(Pal::PipelineStageFlag::PipelineStageTopOfPipe) == PipelineStageTopOfPipe) && (static_cast(Pal::PipelineStageFlag::PipelineStageFetchIndirectArgs) == PipelineStageFetchIndirectArgs) && (static_cast(Pal::PipelineStageFlag::PipelineStagePostPrefetch) == PipelineStagePostPrefetch) && (static_cast(Pal::PipelineStageFlag::PipelineStageFetchIndices) == PipelineStageFetchIndices) && (static_cast(Pal::PipelineStageFlag::PipelineStageStreamOut) == PipelineStageStreamOut) && (static_cast(Pal::PipelineStageFlag::PipelineStageVs) == PipelineStageVs) && (static_cast(Pal::PipelineStageFlag::PipelineStageHs) == PipelineStageHs) && (static_cast(Pal::PipelineStageFlag::PipelineStageDs) == PipelineStageDs) && (static_cast(Pal::PipelineStageFlag::PipelineStageGs) == PipelineStageGs) && (static_cast(Pal::PipelineStageFlag::PipelineStagePs) == PipelineStagePs) && (static_cast(Pal::PipelineStageFlag::PipelineStageSampleRate) == PipelineStageSampleRate) && (static_cast(Pal::PipelineStageFlag::PipelineStageEarlyDsTarget) == PipelineStageEarlyDsTarget)&& (static_cast(Pal::PipelineStageFlag::PipelineStageLateDsTarget) == PipelineStageLateDsTarget) && (static_cast(Pal::PipelineStageFlag::PipelineStageColorTarget) == PipelineStageColorTarget) && (static_cast(Pal::PipelineStageFlag::PipelineStageCs) == PipelineStageCs) && (static_cast(Pal::PipelineStageFlag::PipelineStageBlt) == PipelineStageBlt) && (static_cast(Pal::PipelineStageFlag::PipelineStageBottomOfPipe) == PipelineStageBottomOfPipe)), "The PAL::PipelineStageFlag enum has changed. Vulkan settings might need to be updated."); static_assert(( (static_cast(Pal::CacheCoherencyUsageFlags::CoherCpu) == CoherCpu) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherShaderRead) == CoherShaderRead) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherShaderWrite) == CoherShaderWrite) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherCopySrc) == CoherCopySrc) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherCopyDst) == CoherCopyDst) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherColorTarget) == CoherColorTarget) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherDepthStencilTarget) == CoherDepthStencilTarget) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherResolveSrc) == CoherResolveSrc) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherResolveDst) == CoherResolveDst) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherClear) == CoherClear) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherIndirectArgs) == CoherIndirectArgs) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherIndexData) == CoherIndexData) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherQueueAtomic) == CoherQueueAtomic) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherTimestamp) == CoherTimestamp) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherStreamOut) == CoherStreamOut) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherMemory) == CoherMemory) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherSampleRate) == CoherSampleRate) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherPresent) == CoherPresent) && (static_cast(Pal::CacheCoherencyUsageFlags::CoherCp) == CoherCp)), "The PAL::CacheCoherencyUsageFlags enum has changed. Vulkan settings might need to be updated."); uint32_t srcStageMask; uint32_t dstStageMask; uint32_t srcCacheMask; uint32_t dstCacheMask; if (preCmd) { dstStageMask = static_cast(settings.dbgBarrierPreWaitPipePoint); srcStageMask = static_cast(settings.dbgBarrierPreSignalPipePoint); srcCacheMask = settings.dbgBarrierPreCacheSrcMask; dstCacheMask = settings.dbgBarrierPreCacheDstMask; } else { dstStageMask = static_cast(settings.dbgBarrierPostWaitPipePoint); srcStageMask = static_cast(settings.dbgBarrierPostSignalPipePoint); srcCacheMask = settings.dbgBarrierPostCacheSrcMask; dstCacheMask = settings.dbgBarrierPostCacheDstMask; } Pal::AcquireReleaseInfo barrier = {}; barrier.reason = RgpBarrierUnknownReason; // This code is debug-only code. barrier.dstGlobalStageMask = dstStageMask; if ((dstStageMask != Pal::PipelineStageTopOfPipe) || (srcStageMask != Pal::PipelineStageTopOfPipe)) { barrier.srcGlobalStageMask = srcStageMask; } if (srcCacheMask != 0 || dstCacheMask != 0) { barrier.srcGlobalAccessMask = srcCacheMask; barrier.dstGlobalAccessMask = dstCacheMask; } PalCmdReleaseThenAcquire(barrier, m_curDeviceMask); } #endif // ===================================================================================================================== void CmdBuffer::WriteBufferMarker( PipelineStageFlags pipelineStage, VkBuffer dstBuffer, VkDeviceSize dstOffset, uint32_t marker) { const Buffer* pDestBuffer = Buffer::ObjectFromHandle(dstBuffer); const Pal::PipelineStageFlag pipePoint = VkToPalSrcPipeStageFlagForMarkers(pipelineStage, m_palEngineType); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PalCmdBuffer(deviceIdx)->CmdWriteImmediate( pipePoint, marker, Pal::ImmediateDataWidth::ImmediateData32Bit, pDestBuffer->GpuVirtAddr(deviceIdx) + dstOffset); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::BindTransformFeedbackBuffers( uint32_t firstBinding, uint32_t bindingCount, const VkBuffer* pBuffers, const VkDeviceSize* pOffsets, const VkDeviceSize* pSizes) { VK_ASSERT(firstBinding + bindingCount <= Pal::MaxStreamOutTargets); if (m_pTransformFeedbackState == nullptr) { void* pMemory = m_pDevice->VkInstance()->AllocMem(sizeof(TransformFeedbackState), VK_SYSTEM_ALLOCATION_SCOPE_OBJECT); if (pMemory != nullptr) { m_pTransformFeedbackState = reinterpret_cast(pMemory); memset(m_pTransformFeedbackState, 0, sizeof(TransformFeedbackState)); } else { VK_NEVER_CALLED(); } } if (m_pTransformFeedbackState != nullptr) { VK_ASSERT(m_pTransformFeedbackState->enabled == false); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); for (uint32_t i = 0; i < bindingCount; i++) { uint32_t slot = i + firstBinding; if (pBuffers[i] != VK_NULL_HANDLE) { Buffer* pFeedbackBuffer = Buffer::ObjectFromHandle(pBuffers[i]); VkDeviceSize curSize = 0; if ((pSizes == nullptr) || (pSizes[i] == VK_WHOLE_SIZE)) { curSize = pFeedbackBuffer->GetSize() - pOffsets[i]; } else { curSize = pSizes[i]; } m_pTransformFeedbackState->params.target[slot].gpuVirtAddr = pFeedbackBuffer->GpuVirtAddr(deviceIdx) + pOffsets[i]; m_pTransformFeedbackState->params.target[slot].size = curSize; m_pTransformFeedbackState->bindMask |= 1 << slot; } else { m_pTransformFeedbackState->params.target[slot].gpuVirtAddr = 0; m_pTransformFeedbackState->params.target[slot].size = 0; m_pTransformFeedbackState->bindMask &= ~(1 << slot); } } } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== void CmdBuffer::BeginTransformFeedback( uint32_t firstCounterBuffer, uint32_t counterBufferCount, const VkBuffer* pCounterBuffers, const VkDeviceSize* pCounterBufferOffsets) { utils::IterateMask deviceGroup(m_curDeviceMask); if (m_pTransformFeedbackState != nullptr) { do { uint64_t counterBufferAddr[Pal::MaxStreamOutTargets] = {}; const uint32_t deviceIdx = deviceGroup.Index(); if (pCounterBuffers != nullptr) { CalcCounterBufferAddrs(firstCounterBuffer, counterBufferCount, pCounterBuffers, pCounterBufferOffsets, counterBufferAddr, deviceIdx); } if (m_pTransformFeedbackState->bindMask != 0) { PalCmdBuffer(deviceIdx)->CmdBindStreamOutTargets(m_pTransformFeedbackState->params); PalCmdBuffer(deviceIdx)->CmdLoadBufferFilledSizes(counterBufferAddr); // If counter buffer is null, then stransform feedback will start capturing vertex data to byte offset zero. for (uint32_t i = 0; i < Pal::MaxStreamOutTargets; i++) { if ((m_pTransformFeedbackState->bindMask & (1 << i)) && (counterBufferAddr[i] == 0)) { PalCmdBuffer(deviceIdx)->CmdSetBufferFilledSize(i, 0); } } m_pTransformFeedbackState->enabled = true; } } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== void CmdBuffer::EndTransformFeedback( uint32_t firstCounterBuffer, uint32_t counterBufferCount, const VkBuffer* pCounterBuffers, const VkDeviceSize* pCounterBufferOffsets) { if ((m_pTransformFeedbackState != nullptr) && (m_pTransformFeedbackState->enabled)) { utils::IterateMask deviceGroup(m_curDeviceMask); do { uint64_t counterBufferAddr[Pal::MaxStreamOutTargets] = {}; const uint32_t deviceIdx = deviceGroup.Index(); if (pCounterBuffers != nullptr) { CalcCounterBufferAddrs(firstCounterBuffer, counterBufferCount, pCounterBuffers, pCounterBufferOffsets, counterBufferAddr, deviceIdx); } if (m_pTransformFeedbackState->bindMask != 0) { PalCmdBuffer(deviceIdx)->CmdSaveBufferFilledSizes(counterBufferAddr); // Disable transform feedback by setting bound buffer's size and stride to 0. const Pal::BindStreamOutTargetParams params = {}; PalCmdBuffer(deviceIdx)->CmdBindStreamOutTargets(params); m_pTransformFeedbackState->enabled = false; } } while (deviceGroup.IterateNext()); } } // ===================================================================================================================== void CmdBuffer::CalcCounterBufferAddrs( uint32_t firstCounterBuffer, uint32_t counterBufferCount, const VkBuffer* pCounterBuffers, const VkDeviceSize* pCounterBufferOffsets, uint64_t* counterBufferAddr, uint32_t deviceIdx) { for (uint32_t i = firstCounterBuffer; i < (firstCounterBuffer + counterBufferCount); i++) { if ((pCounterBuffers[i] != VK_NULL_HANDLE) && (m_pTransformFeedbackState->bindMask & (1 << i))) { Buffer* pCounterBuffer = Buffer::ObjectFromHandle(pCounterBuffers[i]); if (pCounterBufferOffsets != nullptr) { counterBufferAddr[i] = pCounterBuffer->GpuVirtAddr(deviceIdx) + pCounterBufferOffsets[i]; } else { counterBufferAddr[i] = pCounterBuffer->GpuVirtAddr(deviceIdx); } } } } // ===================================================================================================================== void CmdBuffer::DrawIndirectByteCount( uint32_t instanceCount, uint32_t firstInstance, VkBuffer counterBuffer, VkDeviceSize counterBufferOffset, uint32_t counterOffset, uint32_t vertexStride) { Buffer* pCounterBuffer = Buffer::ObjectFromHandle(counterBuffer); ValidateGraphicsStates(); #if VKI_RAY_TRACING BindRayQueryConstants(m_allGpuState.pGraphicsPipeline, Pal::PipelineBindPoint::Graphics, 0, 0, 0, nullptr, 0, 0); #endif utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); uint64_t counterBufferAddr = pCounterBuffer->GpuVirtAddr(deviceIdx) + counterBufferOffset; { PalCmdBuffer(deviceIdx)->CmdDrawOpaque( counterBufferAddr, counterOffset, vertexStride, firstInstance, instanceCount); } } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::SetLineStipple( uint32_t lineStippleFactor, uint16_t lineStipplePattern) { // The line stipple factor is adjusted by one (carried over from OpenGL) m_allGpuState.lineStipple.lineStippleScale = (lineStippleFactor - 1); // The bit field to describe the stipple pattern m_allGpuState.lineStipple.lineStippleValue = lineStipplePattern; utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetLineStippleState(m_allGpuState.lineStipple); } while (deviceGroup.IterateNext()); m_allGpuState.staticTokens.lineStippleState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::CmdSetPerDrawVrsRate( const VkExtent2D* pFragmentSize, const VkFragmentShadingRateCombinerOpKHR combinerOps[2]) { m_allGpuState.vrsRate.shadingRate = VkToPalShadingSize( VkClampShadingRate(*pFragmentSize, m_pDevice->GetMaxVrsShadingRate())); m_allGpuState.vrsRate.combinerState[static_cast(Pal::VrsCombinerStage::ProvokingVertex)] = VkToPalShadingRateCombinerOp(combinerOps[0]); m_allGpuState.vrsRate.combinerState[static_cast(Pal::VrsCombinerStage::Primitive)] = VkToPalShadingRateCombinerOp(combinerOps[0]); m_allGpuState.vrsRate.combinerState[static_cast(Pal::VrsCombinerStage::Image)] = VkToPalShadingRateCombinerOp(combinerOps[1]); m_allGpuState.vrsRate.combinerState[static_cast(Pal::VrsCombinerStage::PsIterSamples)] = Pal::VrsCombiner::Passthrough; // Don't call CmdSetPerDrawVrsRate here since we have to observe the // currently bound pipeline to see if we should clamp the rate. // Calling Pal->CmdSetPerDrawVrsRate will happen in ValidateGraphicsStates m_allGpuState.dirtyGraphics.vrs = 1; m_allGpuState.staticTokens.fragmentShadingRate = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::CmdBeginConditionalRendering( const VkConditionalRenderingBeginInfoEXT* pConditionalRenderingBegin) { // Make sure we have a properly aligned buffer offset. VK_ASSERT(Util::IsPow2Aligned(pConditionalRenderingBegin->offset, 4)); // Conditional rendering discards the commands if the 32-bit value is zero. // Our hardware works in the opposite way, so we have to reverse the polarity flag. // PM4CMDSETPREDICATION:predicationBoolean: // 0 = draw_if_not_visible_or_overflow // 1 = draw_if_visible_or_no_overflow const bool predPolarity = (pConditionalRenderingBegin->flags & VK_CONDITIONAL_RENDERING_INVERTED_BIT_EXT) == 0; const Buffer* pBuffer = Buffer::ObjectFromHandle(pConditionalRenderingBegin->buffer); utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetPredication( nullptr, 0, pBuffer->PalMemory(deviceGroup.Index()), pBuffer->MemOffset() + pConditionalRenderingBegin->offset, Pal::PredicateType::Boolean32, predPolarity, false, false); } while (deviceGroup.IterateNext()); m_flags.hasConditionalRendering = true; } // ===================================================================================================================== void CmdBuffer::CmdEndConditionalRendering() { utils::IterateMask deviceGroup(m_curDeviceMask); do { PalCmdBuffer(deviceGroup.Index())->CmdSetPredication( nullptr, 0, nullptr, 0, Pal::PredicateType::Boolean32, false, false, false); } while (deviceGroup.IterateNext()); m_flags.hasConditionalRendering = false; } // ===================================================================================================================== void CmdBuffer::CmdDebugMarkerBegin( const VkDebugMarkerMarkerInfoEXT* pMarkerInfo) { InsertDebugMarker(pMarkerInfo->pMarkerName, true); } // ===================================================================================================================== void CmdBuffer::CmdDebugMarkerEnd() { InsertDebugMarker(nullptr, false); } // ===================================================================================================================== void CmdBuffer::CmdBeginDebugUtilsLabel( const VkDebugUtilsLabelEXT* pLabelInfo) { InsertDebugMarker(pLabelInfo->pLabelName, true); } // ===================================================================================================================== void CmdBuffer::CmdEndDebugUtilsLabel() { InsertDebugMarker(nullptr, false); } // ===================================================================================================================== void CmdBuffer::BindAlternatingThreadGroupConstant() { uint32_t data = m_reverseThreadGroupState ? 1 : 0; const UserDataLayout* pUserDataLayout = m_allGpuState.pComputePipeline->GetUserDataLayout(); uint32_t userDataRegBase = pUserDataLayout->common.threadGroupReversalRegBase; if (userDataRegBase != PipelineLayout::InvalidReg) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); Pal::ICmdBuffer* pPalCmdBuffer = PalCmdBuffer(deviceIdx); Pal::gpusize constGpuAddr = 0; void* pConstData = pPalCmdBuffer->CmdAllocateEmbeddedData(1, 1, &constGpuAddr); memcpy(pConstData, &data, sizeof(data)); pPalCmdBuffer->CmdSetUserData( Pal::PipelineBindPoint::Compute, userDataRegBase, 2, reinterpret_cast(&constGpuAddr)); } while (deviceGroup.IterateNext()); } // Flip the reversal state m_reverseThreadGroupState = (m_reverseThreadGroupState == false); } #if VKI_RAY_TRACING // ===================================================================================================================== void CmdBuffer::BuildAccelerationStructures( uint32_t infoCount, const VkAccelerationStructureBuildGeometryInfoKHR* pInfos, const VkAccelerationStructureBuildRangeInfoKHR* const* ppBuildRangeInfos, const VkDeviceAddress* pIndirectDeviceAddresses, const uint32* pIndirectStrides, const uint32* const* ppMaxPrimitiveCounts) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); BuildAccelerationStructuresPerDevice( deviceIdx, infoCount, pInfos, ppBuildRangeInfos, pIndirectDeviceAddresses, pIndirectStrides, ppMaxPrimitiveCounts); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::BuildAccelerationStructuresPerDevice( const uint32_t deviceIndex, uint32_t infoCount, const VkAccelerationStructureBuildGeometryInfoKHR* pInfos, const VkAccelerationStructureBuildRangeInfoKHR* const* ppBuildRangeInfos, const VkDeviceAddress* pIndirectDeviceAddresses, const uint32* pIndirectStrides, const uint32* const* ppMaxPrimitiveCounts) { const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); Util::Vector m_gpurtInfos(VkInstance()->Allocator()); Util::Vector m_convHelpers(VkInstance()->Allocator()); for (uint32_t infoIdx = 0; infoIdx < infoCount; ++infoIdx) { const VkAccelerationStructureBuildGeometryInfoKHR* pInfo = &pInfos[infoIdx]; const VkAccelerationStructureBuildRangeInfoKHR* pBuildRangeInfos = (ppBuildRangeInfos != nullptr) ? ppBuildRangeInfos[infoIdx] : nullptr; AccelerationStructure* pDst = AccelerationStructure::ObjectFromHandle(pInfo->dstAccelerationStructure); const AccelerationStructure* pSrc = AccelerationStructure::ObjectFromHandle(pInfo->srcAccelerationStructure); // pDst must be a valid handle VK_ASSERT(pDst != nullptr); GpuRt::AccelStructBuildInfo info = {}; info.dstAccelStructGpuAddr = (pDst != nullptr) ? pDst->GetDeviceAddress(deviceIndex) : 0; info.srcAccelStructGpuAddr = ((pInfo->mode == VK_BUILD_ACCELERATION_STRUCTURE_MODE_UPDATE_KHR) && (pSrc != nullptr)) ? pSrc->GetDeviceAddress(deviceIndex) : 0; GeometryConvertHelper helper = {}; AccelerationStructure::ConvertBuildInputsKHR( false, VkDevice(), deviceIndex, *pInfo, pBuildRangeInfos, (ppMaxPrimitiveCounts != nullptr) ? ppMaxPrimitiveCounts[infoIdx] : nullptr, &helper, &info.inputs); const bool forceRebuildTopLevel = Util::TestAnyFlagSet(settings.forceRebuildForUpdates, ForceRebuildForUpdatesTopLevel); const bool forceRebuildBottomLevel = Util::TestAnyFlagSet(settings.forceRebuildForUpdates, ForceRebuildForUpdatesBottomLevel); // Skip all work depending on rtTossPoint setting and type of work. const uint32 rtTossPoint = settings.rtTossPoint; const bool isUpdate = Util::TestAnyFlagSet(info.inputs.flags, GpuRt::AccelStructBuildFlagPerformUpdate); const bool tossWork = (((info.inputs.type == GpuRt::AccelStructType::TopLevel) && (rtTossPoint >= RtTossPointTlas)) || ((info.inputs.type == GpuRt::AccelStructType::BottomLevel) && (rtTossPoint >= RtTossPointBlasBuild)) || ((info.inputs.type == GpuRt::AccelStructType::BottomLevel) && (rtTossPoint >= RtTossPointBlasUpdate) && isUpdate)); if (tossWork) { info.inputs.inputElemCount = 0; } if (((info.inputs.type == GpuRt::AccelStructType::TopLevel) && forceRebuildTopLevel) || ((info.inputs.type == GpuRt::AccelStructType::BottomLevel) && forceRebuildBottomLevel)) { info.inputs.flags &= ~(GpuRt::AccelStructBuildFlagAllowUpdate | GpuRt::AccelStructBuildFlagPerformUpdate); } info.scratchAddr.gpu = pInfo->scratchData.deviceAddress; // Set Indirect Values if (pIndirectDeviceAddresses != nullptr) { VK_ASSERT(pIndirectDeviceAddresses[infoIdx] > 0); info.indirect.indirectGpuAddr = pIndirectDeviceAddresses[infoIdx]; info.indirect.indirectStride = pIndirectStrides[infoIdx]; } if (settings.batchBvhBuilds == BatchBvhModeDisabled) { DbgBarrierPreCmd((pInfo->type == VK_ACCELERATION_STRUCTURE_TYPE_TOP_LEVEL_KHR) ? DbgBuildAccelerationStructureTLAS : DbgBuildAccelerationStructureBLAS); m_pDevice->RayTrace()->GpuRt(deviceIndex)->BuildAccelStruct( PalCmdBuffer(deviceIndex), info); DbgBarrierPostCmd((pInfo->type == VK_ACCELERATION_STRUCTURE_TYPE_TOP_LEVEL_KHR) ? DbgBuildAccelerationStructureTLAS : DbgBuildAccelerationStructureBLAS); } else { m_gpurtInfos.PushBack(info); m_convHelpers.PushBack(helper); } } if (m_gpurtInfos.IsEmpty() == false) { DbgBarrierPreCmd(DbgBuildAccelerationStructureTLAS | DbgBuildAccelerationStructureBLAS); VK_ASSERT(m_gpurtInfos.NumElements() == m_convHelpers.NumElements()); for (uint32 i = 0; i < m_gpurtInfos.NumElements(); ++i) { m_gpurtInfos[i].inputs.pClientData = &m_convHelpers[i]; } m_pDevice->RayTrace()->GpuRt(deviceIndex)->BuildAccelStructs( PalCmdBuffer(deviceIndex), m_gpurtInfos); DbgBarrierPostCmd(DbgBuildAccelerationStructureTLAS | DbgBuildAccelerationStructureBLAS); } } // ===================================================================================================================== void CmdBuffer::WriteAccelerationStructuresProperties( uint32_t accelerationStructureCount, const VkAccelerationStructureKHR* pAccelerationStructures, VkQueryType queryType, VkQueryPool queryPool, uint32_t firstQuery) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); WriteAccelerationStructuresPropertiesPerDevice( deviceIdx, accelerationStructureCount, pAccelerationStructures, queryType, queryPool, firstQuery); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::WriteAccelerationStructuresPropertiesPerDevice( const uint32_t deviceIndex, uint32_t accelerationStructureCount, const VkAccelerationStructureKHR* pAccelerationStructures, VkQueryType queryType, VkQueryPool queryPool, uint32_t firstQuery) { VK_ASSERT(IsAccelerationStructureQueryType(queryType)); GpuRt::AccelStructPostBuildInfo postBuildInfo = {}; postBuildInfo.srcAccelStructCount = 1; switch (static_cast(queryType)) { case VK_QUERY_TYPE_ACCELERATION_STRUCTURE_SIZE_KHR: postBuildInfo.desc.infoType = GpuRt::AccelStructPostBuildInfoType::CurrentSize; break; case VK_QUERY_TYPE_ACCELERATION_STRUCTURE_SERIALIZATION_BOTTOM_LEVEL_POINTERS_KHR: postBuildInfo.desc.infoType = GpuRt::AccelStructPostBuildInfoType::Serialization; break; case VK_QUERY_TYPE_ACCELERATION_STRUCTURE_COMPACTED_SIZE_KHR: postBuildInfo.desc.infoType = GpuRt::AccelStructPostBuildInfoType::CompactedSize; break; case VK_QUERY_TYPE_ACCELERATION_STRUCTURE_SERIALIZATION_SIZE_KHR: postBuildInfo.desc.infoType = GpuRt::AccelStructPostBuildInfoType::Serialization; break; default: VK_NEVER_CALLED(); break; } const AccelerationStructureQueryPool* pQueryPool = QueryPool::ObjectFromHandle(queryPool)->AsAccelerationStructureQueryPool(); const uint32_t emitSize = pQueryPool->GetSlotSize(); const Pal::gpusize basePoolAddr = pQueryPool->GpuVirtAddr(deviceIndex); GpuRt::IDevice* const pGpuRt = m_pDevice->RayTrace()->GpuRt(deviceIndex); for (uint32_t i = 0; i < accelerationStructureCount; i++) { const AccelerationStructure* pAccelStructure = AccelerationStructure::ObjectFromHandle(pAccelerationStructures[i]); Pal::gpusize gpuAddr = pAccelStructure->GetDeviceAddress(deviceIndex); postBuildInfo.desc.postBuildBufferAddr.gpu = basePoolAddr + ((firstQuery + i) * emitSize); postBuildInfo.pSrcAccelStructGpuAddrs = &gpuAddr; pGpuRt->EmitAccelStructPostBuildInfo(PalCmdBuffer(deviceIndex), postBuildInfo); } } // ===================================================================================================================== void CmdBuffer::TraceRays( const VkStridedDeviceAddressRegionKHR& raygenShaderBindingTable, const VkStridedDeviceAddressRegionKHR& missShaderBindingTable, const VkStridedDeviceAddressRegionKHR& hitShaderBindingTable, const VkStridedDeviceAddressRegionKHR& callableShaderBindingTable, uint32_t width, uint32_t height, uint32_t depth) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); TraceRaysPerDevice( deviceIdx, raygenShaderBindingTable, missShaderBindingTable, hitShaderBindingTable, callableShaderBindingTable, width, height, depth); } while (deviceGroup.IterateNext()); } // ===================================================================================================================== void CmdBuffer::GetRayTracingDispatchArgs( uint32_t deviceIdx, const RuntimeSettings& settings, CmdPool* pCmdPool, const RayTracingPipeline* pPipeline, uint32* pConstMem, Pal::gpusize constGpuAddr, uint32_t width, uint32_t height, uint32_t depth, const VkStridedDeviceAddressRegionKHR& raygenSbt, const VkStridedDeviceAddressRegionKHR& missSbt, const VkStridedDeviceAddressRegionKHR& hitSbt, const VkStridedDeviceAddressRegionKHR& callableSbt, GpuRt::DispatchRaysConstants* pConstants) { pConstants->descriptorTable.constData.rayGenerationTableAddressLo = Util::LowPart(raygenSbt.deviceAddress); pConstants->descriptorTable.constData.rayGenerationTableAddressHi = Util::HighPart(raygenSbt.deviceAddress); pConstants->descriptorTable.constData.rayDispatchWidth = width; pConstants->descriptorTable.constData.rayDispatchHeight = height; pConstants->descriptorTable.constData.rayDispatchDepth = depth; pConstants->descriptorTable.constData.rayDispatchMaxGroups = pPipeline->PersistentDispatchSize( width, height, depth); pConstants->descriptorTable.constData.missTableBaseAddressLo = Util::LowPart(missSbt.deviceAddress); pConstants->descriptorTable.constData.missTableBaseAddressHi = Util::HighPart(missSbt.deviceAddress); pConstants->descriptorTable.constData.missTableStrideInBytes = static_cast(missSbt.stride); pConstants->descriptorTable.constData.hitGroupTableBaseAddressLo = Util::LowPart(hitSbt.deviceAddress); pConstants->descriptorTable.constData.hitGroupTableBaseAddressHi = Util::HighPart(hitSbt.deviceAddress); pConstants->descriptorTable.constData.hitGroupTableStrideInBytes = static_cast(hitSbt.stride); pConstants->descriptorTable.constData.callableTableBaseAddressLo = Util::LowPart(callableSbt.deviceAddress); pConstants->descriptorTable.constData.callableTableBaseAddressHi = Util::HighPart(callableSbt.deviceAddress); pConstants->descriptorTable.constData.callableTableStrideInBytes = static_cast(callableSbt.stride); pConstants->descriptorTable.constData.traceRayGpuVaLo = Util::LowPart( pPipeline->GetTraceRayGpuVa(deviceIdx)); pConstants->descriptorTable.constData.traceRayGpuVaHi = Util::HighPart( pPipeline->GetTraceRayGpuVa(deviceIdx)); pConstants->descriptorTable.constData.profileMaxIterations = m_pDevice->RayTrace()->GetProfileMaxIterations(); pConstants->descriptorTable.constData.profileRayFlags = m_pDevice->RayTrace()->GetProfileRayFlags(); memcpy(pConstants->descriptorTable.accelStructTrackerSrd, m_pDevice->RayTrace()->GetAccelStructTrackerSrd(deviceIdx), m_pDevice->GetProperties().descriptorSizes.untypedBufferView); if (pPipeline->CheckIsCps()) { pConstants->descriptorTable.constData.cpsDispatchId = 0; pConstants->descriptorTable.constData.cpsDispatchIdAddressHi = Util::HighPart( constGpuAddr + DISPATCHRAYSCONSTANTDATA_STRUCT_OFFSET_DISPATCHID); pConstants->descriptorTable.constData.cpsDispatchIdAddressLo = Util::LowPart( constGpuAddr + DISPATCHRAYSCONSTANTDATA_STRUCT_OFFSET_DISPATCHID); Pal::CompilerStackSizes stackSizes = PerGpuState(deviceIdx)->maxPipelineStackSizes; if (PerGpuState(deviceIdx)->dynamicPipelineStackSize != 0) { stackSizes.frontendSize = PerGpuState(deviceIdx)->dynamicPipelineStackSize; } pConstants->descriptorTable.constData.cpsFrontendStackSize = stackSizes.frontendSize; pConstants->descriptorTable.constData.cpsBackendStackSize = stackSizes.backendSize; if ((settings.cpsFlags & CpsFlagStackInGlobalMem) != 0) { const uint32 numRays = width * height * depth; const gpusize cpsMemorySize = m_pDevice->RayTrace()->GpuRt(deviceIdx)->GetCpsMemoryBytes( stackSizes.frontendSize, numRays); m_cpsCmdBufferUtil.AddPatchCpsRequest( deviceIdx, reinterpret_cast(pConstMem), cpsMemorySize); } } static_assert(uint32_t(GpuRt::TraceRayCounterDisable) == uint32_t(TraceRayCounterDisable), "Wrong enum value, TraceRayCounterDisable != GpuRt::TraceRayCounterDisable"); static_assert(uint32_t(GpuRt::TraceRayCounterRayHistoryLight) == uint32_t(TraceRayCounterRayHistoryLight), "Wrong enum value, TraceRayCounterRayHistoryLight != GpuRt::TraceRayCounterRayHistoryLight"); static_assert(uint32_t(GpuRt::TraceRayCounterRayHistoryFull) == uint32_t(TraceRayCounterRayHistoryFull), "Wrong enum value, TraceRayCounterRayHistoryFull != GpuRt::TraceRayCounterRayHistoryFull"); static_assert(uint32_t(GpuRt::TraceRayCounterTraversal) == uint32_t(TraceRayCounterTraversal), "Wrong enum value, TraceRayCounterTraversal != GpuRt::TraceRayCounterTraversal"); static_assert(uint32_t(GpuRt::TraceRayCounterCustom) == uint32_t(TraceRayCounterCustom), "Wrong enum value, TraceRayCounterCustom != GpuRt::TraceRayCounterCustom"); static_assert(uint32_t(GpuRt::TraceRayCounterDispatch) == uint32_t(TraceRayCounterDispatch), "Wrong enum value, TraceRayCounterDispatch != GpuRt::TraceRayCounterDispatch"); if (width > 0) { m_pDevice->RayTrace()->TraceDispatch(deviceIdx, this, GpuRt::RtPipelineType::RayTracing, width, height, depth, pPipeline->GetShaderGroupCount() + 1, pPipeline->GetApiHash(), GetUserMarkerContextValue(), &raygenSbt, &missSbt, &hitSbt, pConstants); } } // ===================================================================================================================== void CmdBuffer::TraceRayPreSetup( const uint32_t deviceIdx, const VkStridedDeviceAddressRegionKHR& raygenShaderBindingTable, const VkStridedDeviceAddressRegionKHR& missShaderBindingTable, const VkStridedDeviceAddressRegionKHR& hitShaderBindingTable, const VkStridedDeviceAddressRegionKHR& callableShaderBindingTable, const uint32_t width, const uint32_t height, const uint32_t depth, Pal::gpusize* pConstGpuAddr) { const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); const RayTracingPipeline* pPipeline = m_allGpuState.pRayTracingPipeline; uint32* pConstMem = PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData(GpuRt::DispatchRaysConstantsDw, 1, pConstGpuAddr); GpuRt::DispatchRaysConstants constants = {}; GetRayTracingDispatchArgs(deviceIdx, settings, m_pCmdPool, pPipeline, pConstMem, *pConstGpuAddr, width, height, depth, raygenShaderBindingTable, missShaderBindingTable, hitShaderBindingTable, callableShaderBindingTable, &constants); memcpy(pConstMem, &constants, sizeof(constants)); } // ===================================================================================================================== void CmdBuffer::TraceRaysPerDevice( const uint32_t deviceIdx, const VkStridedDeviceAddressRegionKHR& raygenShaderBindingTable, const VkStridedDeviceAddressRegionKHR& missShaderBindingTable, const VkStridedDeviceAddressRegionKHR& hitShaderBindingTable, const VkStridedDeviceAddressRegionKHR& callableShaderBindingTable, uint32_t width, uint32_t height, uint32_t depth) { DbgBarrierPreCmd(DbgTraceRays); const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); const RayTracingPipeline* pPipeline = m_allGpuState.pRayTracingPipeline; if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindRayTracing) == false) { RebindPipeline(); } Pal::gpusize constGpuAddr; TraceRayPreSetup(deviceIdx, raygenShaderBindingTable, missShaderBindingTable, hitShaderBindingTable, callableShaderBindingTable, width, height, depth, &constGpuAddr); uint32_t dispatchRaysUserData = pPipeline->GetDispatchRaysUserDataOffset(); uint32_t constGpuAddrLow = Util::LowPart(constGpuAddr); PalCmdBuffer(deviceIdx)->CmdSetUserData(Pal::PipelineBindPoint::Compute, dispatchRaysUserData, 1, &constGpuAddrLow); m_pfnTraceRaysDispatchPerDevice(this, deviceIdx, width, height, depth); if (pPipeline->CheckIsCps() && ((settings.cpsFlags & CpsFlagStackInGlobalMem) != 0)) { // A memory barrier is needed here because the CPS global memory is reused by other CPS kernels. RayTracingDevice::SyncRtCommands(PalCmdBuffer(deviceIdx), RtBarrierMode::Dispatch); } DbgBarrierPostCmd(DbgTraceRays); } // ===================================================================================================================== void CmdBuffer::TraceRaysIndirect( GpuRt::ExecuteIndirectArgType indirectArgType, const VkStridedDeviceAddressRegionKHR& raygenShaderBindingTable, const VkStridedDeviceAddressRegionKHR& missShaderBindingTable, const VkStridedDeviceAddressRegionKHR& hitShaderBindingTable, const VkStridedDeviceAddressRegionKHR& callableShaderBindingTable, VkDeviceAddress indirectDeviceAddress) { DbgBarrierPreCmd(DbgTraceRays); utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); TraceRaysIndirectPerDevice( deviceIdx, indirectArgType, raygenShaderBindingTable, missShaderBindingTable, hitShaderBindingTable, callableShaderBindingTable, indirectDeviceAddress, nullptr, GetUserMarkerContextValue()); } while (deviceGroup.IterateNext()); DbgBarrierPostCmd(DbgTraceRays); } // ===================================================================================================================== // Sets a barrier from indirect_arg state to copy_source for rayquery copy arguments. void CmdBuffer::SyncIndirectCopy( Pal::ICmdBuffer* pCmdBuffer) { if (m_pDevice->GetRuntimeSettings().useAcquireReleaseInterface) { Pal::AcquireReleaseInfo acqRelInfo = {}; Pal::MemBarrier memTransition = {}; memTransition.srcAccessMask = Pal::CoherIndirectArgs; memTransition.dstAccessMask = Pal::CoherCopySrc | Pal::CoherIndirectArgs; memTransition.srcStageMask = Pal::PipelineStageCs; memTransition.dstStageMask = Pal::PipelineStageBlt; acqRelInfo.pMemoryBarriers = &memTransition; acqRelInfo.memoryBarrierCount = 1; acqRelInfo.reason = RgpBarrierInternalRayTracingSync; pCmdBuffer->CmdReleaseThenAcquire(acqRelInfo); } else { Pal::BarrierTransition transition = {}; transition.srcCacheMask = Pal::CoherIndirectArgs; transition.dstCacheMask = Pal::CoherCopySrc | Pal::CoherIndirectArgs; const Pal::HwPipePoint postBlt = Pal::HwPipePreBlt; Pal::BarrierInfo barrierInfo = {}; barrierInfo.pipePointWaitCount = 1; barrierInfo.pPipePoints = &postBlt; barrierInfo.waitPoint = Pal::HwPipeTop; barrierInfo.transitionCount = 1; barrierInfo.pTransitions = &transition; barrierInfo.reason = RgpBarrierInternalRayTracingSync; pCmdBuffer->CmdBarrier(barrierInfo); } } // ===================================================================================================================== void CmdBuffer::TraceRaysDispatchPerDevice( CmdBuffer* pCmdBuffer, uint32_t deviceIdx, uint32_t width, uint32_t height, uint32_t depth) { const RayTracingPipeline* pPipeline = pCmdBuffer->m_allGpuState.pRayTracingPipeline; const Pal::DispatchDims dispatchSize = pPipeline->GetDispatchSize({ .x = width, .y = height, .z = depth }); pCmdBuffer->PalCmdBuffer(deviceIdx)->CmdDispatch(dispatchSize, {}); } // ===================================================================================================================== void CmdBuffer::TraceRaysIndirectPerDevice( const uint32_t deviceIdx, GpuRt::ExecuteIndirectArgType indirectArgType, const VkStridedDeviceAddressRegionKHR& raygenShaderBindingTable, const VkStridedDeviceAddressRegionKHR& missShaderBindingTable, const VkStridedDeviceAddressRegionKHR& hitShaderBindingTable, const VkStridedDeviceAddressRegionKHR& callableShaderBindingTable, VkDeviceAddress indirectDeviceAddress, const IndirectCommandsLayout* pLayout, uint64_t userMarkerContext) { const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); const RayTracingPipeline* pPipeline = m_allGpuState.pRayTracingPipeline; Pal::gpusize constGpuAddr; TraceRayPreSetup(deviceIdx, raygenShaderBindingTable, missShaderBindingTable, hitShaderBindingTable, callableShaderBindingTable, 0, // Pre-pass will populate width x height x depth 0, 0, &constGpuAddr); // Pre-pass gpusize initConstantsVa = 0; // Allocate extra space in scratch buffer to accommodate pre-action arguments (if any) const gpusize scratchBufferSize = (pLayout != nullptr) ? sizeof(VkTraceRaysIndirectCommandKHR) + pLayout->GetIndirectCommandsInfo().preActionArgSizeInBytes : sizeof(VkTraceRaysIndirectCommandKHR); InternalMemory* pScratchMemory = nullptr; VkResult result = GetRayTracingIndirectMemory(scratchBufferSize, &pScratchMemory); VK_ASSERT(result == VK_SUCCESS); auto* pInitConstants = reinterpret_cast( PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData(GpuRt::InitExecuteIndirectConstantsDw, 2, &initConstantsVa)); memset(pInitConstants, 0, sizeof(GpuRt::InitExecuteIndirectConstants)); pInitConstants->maxIterations = m_pDevice->RayTrace()->GetProfileMaxIterations(); pInitConstants->profileRayFlags = m_pDevice->RayTrace()->GetProfileRayFlags(); pInitConstants->maxDispatchCount = 1; pInitConstants->pipelineCount = 1; pInitConstants->indirectMode = (indirectArgType == GpuRt::ExecuteIndirectArgType::DispatchDimensions) ? 0 : 1; // NOTE: For CPS, we only support flatten thread group so far. const uint32_t flattenThreadGroupSize = pPipeline->CheckIsCps() ? settings.dispatchRaysThreadGroupSize : settings.rtFlattenThreadGroupSize; if (flattenThreadGroupSize == 0) { pInitConstants->dispatchDimSwizzleMode = 0; pInitConstants->rtThreadGroupSizeX = settings.rtThreadGroupSizeX; pInitConstants->rtThreadGroupSizeY = settings.rtThreadGroupSizeY; pInitConstants->rtThreadGroupSizeZ = settings.rtThreadGroupSizeZ; } else { pInitConstants->dispatchDimSwizzleMode = 1; pInitConstants->rtThreadGroupSizeX = flattenThreadGroupSize; pInitConstants->rtThreadGroupSizeY = 1; pInitConstants->rtThreadGroupSizeZ = 1; } // Instruct InitExecuteIndirect shader to dump pre-action arguments (if any) to the scratch buffer pInitConstants->bindingArgsSize = (pLayout != nullptr) ? pLayout->GetIndirectCommandsInfo().preActionArgSizeInBytes : 0; pInitConstants->inputBytesPerDispatch = 0; pInitConstants->outputBytesPerDispatch = 0; GpuRt::InitExecuteIndirectUserData initUserData = {}; initUserData.constantsVa = initConstantsVa; initUserData.inputBufferVa = indirectDeviceAddress; initUserData.outputBufferVa = static_cast(pScratchMemory->GpuVirtAddr(deviceIdx)); initUserData.outputConstantsVa = constGpuAddr; initUserData.outputCounterMetaVa = 0uLL; m_pDevice->RayTrace()->TraceIndirectDispatch(deviceIdx, this, GpuRt::RtPipelineType::RayTracing, 0, 0, 0, pPipeline->GetShaderGroupCount() + 1, pPipeline->GetApiHash(), userMarkerContext, &raygenShaderBindingTable, &missShaderBindingTable, &hitShaderBindingTable, &initUserData.outputCounterMetaVa, pInitConstants); m_pDevice->RayTrace()->GpuRt(deviceIdx)->InitExecuteIndirect(PalCmdBuffer(deviceIdx), initUserData, 1, 1); // Wait for the argument buffer to be populated before continuing with TraceRaysIndirect RayTracingDevice::SyncRtCommands(PalCmdBuffer(deviceIdx), RtBarrierMode::IndirectArg); uint32_t dispatchRaysUserData = pPipeline->GetDispatchRaysUserDataOffset(); uint32_t constGpuAddrLow = uint32_t(constGpuAddr); // Switch to the raytracing pipeline if needed if (PalPipelineBindingOwnedBy(Pal::PipelineBindPoint::Compute, PipelineBindRayTracing) == false) { RebindPipeline(); } PalCmdBuffer(deviceIdx)->CmdSetUserData(Pal::PipelineBindPoint::Compute, dispatchRaysUserData, 1, &constGpuAddrLow); if (pLayout != nullptr) { PalCmdBuffer(deviceIdx)->CmdExecuteIndirectCmds( *pLayout->PalIndirectCmdGenerator(deviceIdx), pScratchMemory->GpuVirtAddr(deviceIdx), 1, // Single dispatch here 0); } else { PalCmdBuffer(deviceIdx)->CmdDispatchIndirect(pScratchMemory->GpuVirtAddr(deviceIdx)); } } // ===================================================================================================================== // Allocates gpu video memory. VkResult CmdBuffer::GetScratchVidMem( gpusize sizeInBytes, InternalSubAllocPool poolId, InternalMemory** ppInternalMemory) { VkResult result = VK_ERROR_OUT_OF_HOST_MEMORY; *ppInternalMemory = nullptr; // Allocate system memory for InternalMemory object InternalMemory* pInternalMemory = nullptr; void* pSystemMemory = m_pDevice->VkInstance()->AllocMem(sizeof(InternalMemory), VK_DEFAULT_MEM_ALIGN, VK_SYSTEM_ALLOCATION_SCOPE_DEVICE); if (pSystemMemory != nullptr) { pInternalMemory = VK_PLACEMENT_NEW(pSystemMemory) InternalMemory; } // Allocate GPU video memory if (pInternalMemory != nullptr) { InternalMemCreateInfo allocInfo = {}; allocInfo.pal.size = sizeInBytes; allocInfo.pal.alignment = 16; allocInfo.pal.priority = Pal::GpuMemPriority::Normal; m_pDevice->MemMgr()->GetCommonPool(poolId, &allocInfo); result = m_pDevice->MemMgr()->AllocGpuMem( allocInfo, pInternalMemory, m_pDevice->GetPalDeviceMask(), VK_OBJECT_TYPE_COMMAND_BUFFER, ApiCmdBuffer::IntValueFromHandle(ApiCmdBuffer::FromObject(this))); VK_ASSERT(result == VK_SUCCESS); if (result == VK_SUCCESS) { *ppInternalMemory = pInternalMemory; m_scratchVidMemList.PushBack(pInternalMemory); } } if (result != VK_SUCCESS) { // Clean up if fail if (pInternalMemory != nullptr) { Util::Destructor(pInternalMemory); } m_pDevice->VkInstance()->FreeMem(pSystemMemory); } return result; } // ===================================================================================================================== // Alloacates GPU video memory according for TraceRaysIndirect VkResult CmdBuffer::GetRayTracingIndirectMemory( gpusize size, InternalMemory** ppInternalMemory) { return GetScratchVidMem(size, InternalPoolGpuAccess, ppInternalMemory); } // ===================================================================================================================== // Free GPU video memory according for TraceRaysIndirect void CmdBuffer::FreeRayTracingScratchVidMemory() { // This data could be farily large and consumes framebuffer memory. // // This should always be done when vkResetCommandBuffer() is called to handle the case // where an app resets a command buffer but doesn't call vkBeginCommandBuffer right away. for (uint32_t i = 0; i < m_scratchVidMemList.NumElements(); ++i) { // Dump entry data InternalMemory* pVidMemory = m_scratchVidMemList.At(i); // Free memory m_pDevice->MemMgr()->FreeGpuMem(pVidMemory); Util::Destructor(pVidMemory); m_pDevice->VkInstance()->FreeMem(pVidMemory); } // Clear list m_scratchVidMemList.Clear(); } // ===================================================================================================================== // Set the dynamic stack size for a ray tracing pipeline void CmdBuffer::SetRayTracingPipelineStackSize( uint32_t pipelineStackSize) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); PerGpuState(deviceIdx)->dynamicPipelineStackSize = pipelineStackSize; } while (deviceGroup.IterateNext()); } // ===================================================================================================================== // Setup internal constants and descriptors required for shaders using RayQuery void CmdBuffer::BindRayQueryConstants( const Pipeline* pPipeline, Pal::PipelineBindPoint bindPoint, uint32_t width, uint32_t height, uint32_t depth, Buffer* pIndirectBuffer, VkDeviceSize indirectOffset, VkDeviceSize indirectBufferVirtAddr) { if ((pPipeline != nullptr) && pPipeline->HasRayTracing()) { utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); const bool asTrackingEnabled = VkDevice()->RayTrace()->AccelStructTrackerEnabled(deviceIdx); const bool rtCountersEnabled = (VkDevice()->RayTrace()->TraceRayCounterMode(deviceIdx) != GpuRt::TraceRayCounterMode::TraceRayCounterDisable); if (asTrackingEnabled || rtCountersEnabled) { GpuRt::DispatchRaysConstants constants = {}; Pal::ICmdBuffer* pPalCmdBuffer = PalCmdBuffer(deviceIdx); Pal::gpusize constGpuAddr = 0; void* pConstData = pPalCmdBuffer->CmdAllocateEmbeddedData(GpuRt::DispatchRaysConstantsDw, 1, &constGpuAddr); if (asTrackingEnabled) { memcpy(constants.descriptorTable.accelStructTrackerSrd, VkDevice()->RayTrace()->GetAccelStructTrackerSrd(deviceIdx), VkDevice()->GetProperties().descriptorSizes.untypedBufferView); } // Ray history dumps for Graphics pipelines are not yet supported if (VkDevice()->RayTrace()->RayHistoryTraceActive(deviceIdx) && (bindPoint == Pal::PipelineBindPoint::Compute)) { const uint32_t* pOrigThreadgroupDims = static_cast(pPipeline)->GetOrigThreadgroupDims(); constants.descriptorTable.constData.profileMaxIterations = m_pDevice->RayTrace()->GetProfileMaxIterations(); constants.descriptorTable.constData.profileRayFlags = m_pDevice->RayTrace()->GetProfileRayFlags(); gpusize indirectBufferVa = (pIndirectBuffer != nullptr) ? pIndirectBuffer->GpuVirtAddr(deviceIdx) + indirectOffset : indirectBufferVirtAddr; if (indirectBufferVa == 0) { constants.descriptorTable.constData.rayDispatchWidth = width * pOrigThreadgroupDims[0]; constants.descriptorTable.constData.rayDispatchHeight = height * pOrigThreadgroupDims[1]; constants.descriptorTable.constData.rayDispatchDepth = depth * pOrigThreadgroupDims[2]; m_pDevice->RayTrace()->TraceDispatch(deviceIdx, this, GpuRt::RtPipelineType::Compute, width * pOrigThreadgroupDims[0], height * pOrigThreadgroupDims[1], depth * pOrigThreadgroupDims[2], 1, pPipeline->GetApiHash(), GetUserMarkerContextValue(), nullptr, nullptr, nullptr, &constants); } else { uint64 counterMetadataGpuVa = 0uLL; m_pDevice->RayTrace()->TraceIndirectDispatch(deviceIdx, this, GpuRt::RtPipelineType::Compute, pOrigThreadgroupDims[0], pOrigThreadgroupDims[1], pOrigThreadgroupDims[2], 1, pPipeline->GetApiHash(), GetUserMarkerContextValue(), nullptr, nullptr, nullptr, &counterMetadataGpuVa, &constants); Pal::MemoryCopyRegion region = {}; region.srcOffset = 0; region.copySize = sizeof(GpuRt::IndirectCounterMetadata) - sizeof(uint64); SyncIndirectCopy(PalCmdBuffer(deviceIdx)); PalCmdBuffer(deviceIdx)->CmdCopyMemoryByGpuVa( indirectBufferVa, (counterMetadataGpuVa + offsetof(GpuRt::IndirectCounterMetadata, dispatchRayDimensionX)), 1, ®ion); } } memcpy(pConstData, &constants, sizeof(constants)); uint32_t dispatchRaysUserData = pPipeline->GetDispatchRaysUserDataOffset(); uint32_t constGpuAddrLow = Util::LowPart(constGpuAddr); pPalCmdBuffer->CmdSetUserData(bindPoint, dispatchRaysUserData, 1, &constGpuAddrLow); } } while (deviceGroup.IterateNext()); } } #endif // ===================================================================================================================== void CmdBuffer::InsertDebugMarker( const char* pLabelName, bool isBegin) { constexpr uint8 MarkerSourceApplication = 0; const IDevMode* pDevMode = m_pDevice->VkInstance()->GetDevModeMgr(); // Insert Crash Analysis markers if requested if ((pDevMode != nullptr) && (pDevMode->IsCrashAnalysisEnabled())) { PalCmdBuffer(DefaultDeviceIndex)->CmdInsertExecutionMarker(isBegin, MarkerSourceApplication, pLabelName, (pLabelName != nullptr) ? Util::StringLength(pLabelName) : 0); } } // ===================================================================================================================== uint32_t CmdBuffer::GetPipelineScratchSize( uint32_t deviceIdx) const { uint32_t scratchSize = 0; #if VKI_RAY_TRACING if (m_allGpuState.pRayTracingPipeline != nullptr) { const RuntimeSettings& settings = m_pDevice->GetRuntimeSettings(); auto stackSizes = PerGpuState(deviceIdx)->maxPipelineStackSizes; auto dynamicStackSize = PerGpuState(deviceIdx)->dynamicPipelineStackSize; if (m_allGpuState.pRayTracingPipeline->CheckIsCps()) { if ((settings.cpsFlags & CpsFlagStackInGlobalMem) == 0) { // Continuations with stack in scratch scratchSize = stackSizes.backendSize + Util::Max(stackSizes.frontendSize, dynamicStackSize); } else { // Continuations stack in global memory scratchSize = stackSizes.backendSize; } } else { // Non-continuations scratchSize = Util::Max(stackSizes.backendSize, dynamicStackSize); } } #endif return scratchSize; } // ===================================================================================================================== void CmdBuffer::BindDescriptorBuffers( uint32_t bufferCount, const VkDescriptorBufferBindingInfoEXT* pBindingInfos) { // Please check if EXT_DESCRIPTOR_BUFFER is enabled. VK_ASSERT(m_allGpuState.pDescBufBinding != nullptr); VK_ASSERT(bufferCount <= MaxDescriptorSets); for (uint32_t ndx = 0; ndx < bufferCount; ++ndx) { VK_ASSERT(pBindingInfos[ndx].sType == VK_STRUCTURE_TYPE_DESCRIPTOR_BUFFER_BINDING_INFO_EXT); m_allGpuState.pDescBufBinding->baseAddr[ndx] = pBindingInfos[ndx].address; } } // ===================================================================================================================== void CmdBuffer::SetDescriptorBufferOffsets( VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t firstSet, uint32_t setCount, const uint32_t* pBufferIndices, const VkDeviceSize* pOffsets) { // Please check if EXT_DESCRIPTOR_BUFFER is enabled. VK_ASSERT(m_allGpuState.pDescBufBinding != nullptr); DescriptorBuffers descBuffers[MaxDescriptorSets] = {}; for (uint32_t ndx = 0u; ndx < setCount; ++ndx) { const uint32_t descNdx = ndx + firstSet; descBuffers[descNdx].offset = pOffsets[ndx]; descBuffers[descNdx].baseAddrNdx = pBufferIndices[ndx]; // First baseAddr should be bound by BindDescriptorBuffers. VK_ASSERT(m_allGpuState.pDescBufBinding->baseAddr[pBufferIndices[ndx]] != 0); } BindDescriptorSetsBuffers(pipelineBindPoint, layout, firstSet, setCount, descBuffers); } // ===================================================================================================================== void CmdBuffer::BindDescriptorBufferEmbeddedSamplers( VkPipelineBindPoint pipelineBindPoint, VkPipelineLayout layout, uint32_t set) { const PipelineLayout* pLayout = PipelineLayout::ObjectFromHandle(layout); const PipelineLayout::SetUserDataLayout& setLayoutInfo = pLayout->GetSetUserData(set); VK_ASSERT(set <= pLayout->GetInfo().setCount); if (m_pDevice->MustWriteImmutableSamplers() && (setLayoutInfo.setPtrRegOffset != PipelineLayout::InvalidReg)) { Pal::PipelineBindPoint palBindPoint; PipelineBindPoint apiBindPoint; ConvertPipelineBindPoint(pipelineBindPoint, &palBindPoint, &apiBindPoint); const DescriptorSetLayout* pDestSetLayout = pLayout->GetSetLayouts(set); const DescriptorSetLayout::CreateInfo& destSetLayoutInfo = pDestSetLayout->Info(); const size_t descriptorSetSize = destSetLayoutInfo.sta.dwSize; const size_t alignmentInDwords = m_pDevice->GetProperties().descriptorSizes.alignmentInDwords; utils::IterateMask deviceGroup(m_curDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); Pal::gpusize gpuAddr; uint32* pCpuAddr = PalCmdBuffer(deviceIdx)->CmdAllocateEmbeddedData(descriptorSetSize, alignmentInDwords, &gpuAddr); for (uint32_t bindingIndex = 0; bindingIndex < destSetLayoutInfo.count; ++bindingIndex) { const DescriptorSetLayout::BindingInfo& bindingInfo = pDestSetLayout->Binding(bindingIndex); // Determine whether the binding has immutable sampler descriptors. if (bindingInfo.imm.dwSize != 0) { uint32_t* pSamplerDesc = destSetLayoutInfo.imm.pImmutableSamplerData + bindingInfo.imm.dwOffset; const size_t srcArrayStrideInDW = bindingInfo.imm.dwArrayStride; uint32_t numOfSamplers = bindingInfo.info.descriptorCount; for (uint32_t descriptorIdx = 0; descriptorIdx < numOfSamplers; ++descriptorIdx) { size_t destOffset = pDestSetLayout->GetDstStaOffset(bindingInfo, descriptorIdx); memcpy(pCpuAddr + destOffset, pSamplerDesc, (sizeof(uint32_t) * bindingInfo.imm.dwSize) / numOfSamplers); pSamplerDesc += srcArrayStrideInDW; } } } PerGpuState(deviceIdx)->setBindingData[apiBindPoint][setLayoutInfo.setPtrRegOffset] = static_cast(gpuAddr); } while (deviceGroup.IterateNext()); SetUserDataPipelineLayout(set, 1, pLayout, palBindPoint, apiBindPoint); } } // ===================================================================================================================== // Batch LoadOp clears on multiple color attachments instead of using PAL's ColorClearAutoSync, this will reduce the // amount of barriers from 2 per clear to 2 for the entire batch. This is currently only used for Dynamic Rendering // as the renderpass code has it's own version of this. void CmdBuffer::BatchedLoadOpClears( uint32_t clearCount, const ImageView** pImageViews, const Pal::ClearColor* pClearColors, const Pal::ImageLayout* pClearLayouts, const Pal::SubresRange* pRanges, const Pal::SwizzledFormat* pClearFormats, uint32_t viewMask) { VK_ASSERT_MSG(clearCount > 1, "Pal::ColorClearAutoSync is recommended for single clears"); Pal::ImgBarrier imageBarriers[Pal::MaxColorTargets] = {}; const Image* images[Pal::MaxColorTargets] = {}; for (uint32_t i = 0; i < clearCount; i++) { Pal::ImgBarrier* pPreSyncBarrier = &imageBarriers[i]; pPreSyncBarrier->srcStageMask = Pal::PipelineStageColorTarget; pPreSyncBarrier->dstStageMask = Pal::PipelineStageBlt; pPreSyncBarrier->srcAccessMask = Pal::CoherColorTarget; pPreSyncBarrier->dstAccessMask = Pal::CoherClear; pPreSyncBarrier->oldLayout = pClearLayouts[i]; pPreSyncBarrier->newLayout = pClearLayouts[i]; pImageViews[i]->GetFrameBufferAttachmentSubresRange(&pPreSyncBarrier->subresRange); // This is filled out later in PalCmdReleaseThenAcquire() pPreSyncBarrier->pImage = nullptr; images[i] = pImageViews[i]->GetImage(); } // Issue the pre sync barrier Pal::AcquireReleaseInfo acqRelInfo = {}; acqRelInfo.reason = Pal::Developer::BarrierReason::BarrierReasonPreSyncClear; acqRelInfo.imageBarrierCount = clearCount; acqRelInfo.pImageBarriers = imageBarriers; PalCmdReleaseThenAcquire(&acqRelInfo, nullptr, nullptr, imageBarriers, images, m_curDeviceMask); // Issue the actual clear for (uint32_t i = 0; i < clearCount; i++) { //Modify the barriers for postSync clear Pal::ImgBarrier* pPostSyncBarrier = &imageBarriers[i]; pPostSyncBarrier->srcStageMask = Pal::PipelineStageBlt; pPostSyncBarrier->dstStageMask = Pal::PipelineStageColorTarget; pPostSyncBarrier->srcAccessMask = Pal::CoherClear; pPostSyncBarrier->dstAccessMask = Pal::CoherColorTarget; const auto clearSubresRanges = LoadOpClearSubresRanges(viewMask, pRanges[i]); utils::IterateMask deviceGroup(GetDeviceMask()); do { const uint32_t deviceIdx = deviceGroup.Index(); // Clear Box Pal::Box clearBox = BuildClearBox( m_allGpuState.dynamicRenderingInstance.renderArea[deviceIdx], *(pImageViews[i])); PalCmdBuffer(deviceIdx)->CmdClearColorImage( *(images[i]->PalImage(deviceIdx)), pClearLayouts[i], pClearColors[i], pClearFormats[i], clearSubresRanges.NumElements(), clearSubresRanges.Data(), 1, &clearBox, 0); } while (deviceGroup.IterateNext()); } //Issue the post sync barrier acqRelInfo.reason = Pal::Developer::BarrierReason::BarrierReasonPostSyncClear; acqRelInfo.imageBarrierCount = clearCount; acqRelInfo.pImageBarriers = imageBarriers; PalCmdReleaseThenAcquire(&acqRelInfo, nullptr, nullptr, imageBarriers, images, m_curDeviceMask); } // ===================================================================================================================== bool CmdBuffer::IsFormatInvalidForLogicOp( uint32_t attachment) { Pal::ChNumFormat format = Pal::ChNumFormat::Undefined; if (UsingDynamicRendering()) { if (attachment < m_allGpuState.dynamicRenderingInstance.colorAttachmentCount) { const uint32_t loc = m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[attachment]; if (loc != VK_ATTACHMENT_UNUSED) { const VkFormat vkFormat = m_allGpuState.dynamicRenderingInstance.colorAttachments[loc].attachmentFormat; const Pal::SwizzledFormat palFormat = VkToPalFormat(vkFormat, m_pDevice->GetRuntimeSettings()); format = palFormat.format; } } } else if (m_allGpuState.pFramebuffer != nullptr) { if (attachment < m_allGpuState.pFramebuffer->GetAttachmentCount()) { format = m_allGpuState.pFramebuffer->GetAttachment(attachment).viewFormat.format; } } else if (m_allGpuState.pRenderPass != nullptr) { if (attachment < m_allGpuState.pRenderPass->GetAttachmentCount()) { const auto& att = m_allGpuState.pRenderPass->GetAttachmentDesc(attachment); const Pal::SwizzledFormat palFormat = VkToPalFormat(att.format, m_pDevice->GetRuntimeSettings()); format = palFormat.format; } } return Pal::Formats::IsFloat(format) || Pal::Formats::IsSrgb(format); } // ===================================================================================================================== void CmdBuffer::ValidateGraphicsStates() { if (m_allGpuState.dirtyGraphics.u32All != 0) { const DynamicDepthStencil* pDepthStencil = nullptr; const DynamicColorBlend* pColorBlend = nullptr; const DynamicMsaa* pMsaa = nullptr; const GraphicsPipeline* pGraphicsPipeline = m_allGpuState.pGraphicsPipeline; if (m_allGpuState.dirtyGraphics.msaa || m_allGpuState.dirtyGraphics.samplePattern) { uint32_t enable1xMsaaSampleLocations = (m_allGpuState.sampleLocationsEnable && (m_allGpuState.msaaCreateInfo.coverageSamples == 1)) ? 1 : 0; if (m_allGpuState.msaaCreateInfo.flags.enable1xMsaaSampleLocations != enable1xMsaaSampleLocations) { m_allGpuState.msaaCreateInfo.flags.enable1xMsaaSampleLocations = enable1xMsaaSampleLocations; m_allGpuState.dirtyGraphics.msaa = 1; } } if ((pGraphicsPipeline != nullptr) && (m_allGpuState.dirtyGraphics.msaa || m_allGpuState.dirtyGraphics.pipeline)) { const uint8 forceSampleRateShading = (m_allGpuState.msaaCreateInfo.pixelShaderSamples > 1) && (pGraphicsPipeline->GetPipelineFlags().sampleShadingEnable != 0); if (m_allGpuState.msaaCreateInfo.flags.forceSampleRateShading != forceSampleRateShading) { m_allGpuState.msaaCreateInfo.flags.forceSampleRateShading = forceSampleRateShading; m_allGpuState.dirtyGraphics.msaa = 1; } } if (m_allGpuState.dirtyGraphics.pipeline && m_allGpuState.logicOpEnable) { m_allGpuState.dirtyGraphics.colorBlend = 1; } auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; utils::IterateMask deviceGroup(m_cbBeginDeviceMask); do { const uint32_t deviceIdx = deviceGroup.Index(); if (m_allGpuState.dirtyGraphics.colorBlend) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); RenderStateCache* pRSCache = m_pDevice->GetRenderStateCache(); if (pColorBlend == nullptr) { DynamicColorBlend colorBlend = {}; Pal::ColorBlendStateCreateInfo colorBlendCreateInfo = m_allGpuState.colorBlendCreateInfo; for (uint32_t i = 0; i < m_allGpuState.dynamicRenderingInstance.colorAttachmentCount; ++i) { const uint32_t location = m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i]; if (location != VK_ATTACHMENT_UNUSED) { colorBlendCreateInfo.targets[location] = m_allGpuState.colorBlendCreateInfo.targets[i]; } } if (m_allGpuState.logicOpEnable) { const uint32_t cnt = UsingDynamicRendering() ? m_allGpuState.dynamicRenderingInstance.colorAttachmentCount : (m_allGpuState.pFramebuffer != nullptr) ? m_allGpuState.pFramebuffer->GetAttachmentCount() : (m_allGpuState.pRenderPass != nullptr) ? m_allGpuState.pRenderPass->GetAttachmentCount() : 0; for (uint32_t i = 0; i < cnt; ++i) { if (IsFormatInvalidForLogicOp(i)) { const auto location = UsingDynamicRendering() ? m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i] : i; colorBlendCreateInfo.targets[location].disableLogicOp = true; colorBlendCreateInfo.targets[location].blendEnable = false; } } } pRSCache->CreateColorBlendState(colorBlendCreateInfo, m_pDevice->VkInstance()->GetAllocCallbacks(), VK_SYSTEM_ALLOCATION_SCOPE_OBJECT, colorBlend.pPalColorBlend); // Check if pPalColorBlend is already in the m_palColorBlendState, destroy it and use the old one // if yes.The destroy is not expensive since it's just a refCount--. for (uint32_t i = 0; i < m_palColorBlendState.NumElements(); ++i) { const DynamicColorBlend& palColorBlendState = m_palColorBlendState.At(i); // Check device0 only should be sufficient if (palColorBlendState.pPalColorBlend[0] == colorBlend.pPalColorBlend[0]) { pRSCache->DestroyColorBlendState(colorBlend.pPalColorBlend, m_pDevice->VkInstance()->GetAllocCallbacks()); pColorBlend = &palColorBlendState; break; } } // Add it to the m_palColorBlendState if it doesn't exist if (pColorBlend == nullptr) { m_palColorBlendState.PushBack(colorBlend); pColorBlend = &m_palColorBlendState.Back(); } } VK_ASSERT(pColorBlend != nullptr); PalCmdBindColorBlendState( m_pPalCmdBuffers[deviceIdx], deviceIdx, pColorBlend->pPalColorBlend[deviceIdx]); bool dualSourceBlendEnable = m_pDevice->PalDevice(DefaultDeviceIndex)->CanEnableDualSourceBlend( m_allGpuState.colorBlendCreateInfo); if (dualSourceBlendEnable != pDynamicState->dualSourceBlendEnable) { pDynamicState->dualSourceBlendEnable = dualSourceBlendEnable; m_allGpuState.dirtyGraphics.pipeline = 1; } DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } // Reorder color mask if dynamic rendering index is used if (m_allGpuState.dirtyGraphics.colorWriteMask) { uint32 newColorWriteMask = 0; const uint32 orignalColorWriteMask = m_allGpuState.colorWriteMask & m_allGpuState.colorWriteEnable; if (UsingDynamicRendering()) { // See if the dynamic locations need remapped for (uint32_t i = 0; i < m_allGpuState.dynamicRenderingInstance.colorAttachmentCount; ++i) { const uint32_t remapLocation = m_allGpuState.dynamicRenderingInstance.colorAttachmentLocations[i]; if (remapLocation != VK_ATTACHMENT_UNUSED) { const uint32 idxMaskValue = ((orignalColorWriteMask >> (4 * i)) & 0xF); const uint32 remapMaskValue = (idxMaskValue << (4 * remapLocation)); newColorWriteMask |= (idxMaskValue > 0) ? remapMaskValue : 0; } } } else { newColorWriteMask = orignalColorWriteMask; } if (newColorWriteMask != pDynamicState->colorWriteMask) { pDynamicState->colorWriteMask = newColorWriteMask; m_allGpuState.dirtyGraphics.pipeline = 1; } } if ((pGraphicsPipeline != nullptr) && m_allGpuState.dirtyGraphics.pipeline) { Pal::PipelineBindParams params = {}; params.pipelineBindPoint = Pal::PipelineBindPoint::Graphics; params.pPipeline = pGraphicsPipeline->GetPalPipeline(deviceIdx); params.gfxDynState = m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; params.gfxShaderInfo = pGraphicsPipeline->GetBindInfo(); if (params.gfxDynState.enable.depthClampMode && (params.gfxDynState.enable.depthClipMode == false)) { bool clipEnable = params.gfxDynState.depthClampMode == Pal::DepthClampMode::_None; params.gfxDynState.enable.depthClipMode = true; params.gfxDynState.depthClipFarEnable = clipEnable; params.gfxDynState.depthClipNearEnable = clipEnable; } params.apiPsoHash = pGraphicsPipeline->GetApiHash(); PalCmdBuffer(deviceIdx)->CmdBindPipeline(params); } if (m_allGpuState.dirtyGraphics.viewport) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); const bool isPointSizeUsed = (pGraphicsPipeline != nullptr) && pGraphicsPipeline->IsPointSizeUsed(); Pal::ViewportParams viewportParams = PerGpuState(deviceIdx)->viewport; if (isPointSizeUsed) { // The default vaule is 1.0f which means the guardband is disabled. // Values more than 1.0f enable guardband. viewportParams.horzDiscardRatio = 10.0f; viewportParams.vertDiscardRatio = 10.0f; } if (m_allGpuState.depthClampOverride.minDepthClamp <= m_allGpuState.depthClampOverride.maxDepthClamp) { for (uint32_t i = 0; i < viewportParams.count; ++i) { auto* const pViewport = &viewportParams.viewports[i]; pViewport->minDepth = m_allGpuState.depthClampOverride.minDepthClamp; pViewport->maxDepth = m_allGpuState.depthClampOverride.maxDepthClamp; } } PalCmdBuffer(deviceIdx)->CmdSetViewports(viewportParams); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } if (m_allGpuState.dirtyGraphics.scissor) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); PalCmdBuffer(deviceIdx)->CmdSetScissorRects(PerGpuState(deviceIdx)->scissor); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } if (m_allGpuState.dirtyGraphics.rasterState) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); PalCmdBuffer(deviceIdx)->CmdSetTriangleRasterState(m_allGpuState.triangleRasterState); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } if (m_allGpuState.dirtyGraphics.stencilRef) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); PalCmdBuffer(deviceIdx)->CmdSetStencilRefMasks(m_allGpuState.stencilRefMasks); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } if (m_allGpuState.dirtyGraphics.inputAssembly) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); PalCmdBuffer(deviceIdx)->CmdSetInputAssemblyState(m_allGpuState.inputAssemblyState); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } if (m_allGpuState.dirtyGraphics.vrs) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); const bool force1x1 = (pGraphicsPipeline != nullptr) && (pGraphicsPipeline->Force1x1ShaderRateEnabled()); // CmdSetPerDrawVrsRate has been called for the dynamic state // Look at the currently bound pipeline and see if we need to force the values to 1x1 Pal::VrsRateParams vrsRate = m_allGpuState.vrsRate; if (force1x1 || (m_allGpuState.samplePattern.sampleCount == 8)) { Device::SetDefaultVrsRateParams(&vrsRate); } SetVrsCombinerStagePsIterSamples(m_allGpuState.msaaCreateInfo.pixelShaderSamples, m_allGpuState.vrsRate.flags.exposeVrsPixelsMask, vrsRate.combinerState); PalCmdBuffer(deviceIdx)->CmdSetPerDrawVrsRate(vrsRate); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } if (m_allGpuState.dirtyGraphics.depthStencil) { RenderStateCache* pRSCache = m_pDevice->GetRenderStateCache(); if (pDepthStencil == nullptr) { DynamicDepthStencil depthStencil = {}; pRSCache->CreateDepthStencilState(m_allGpuState.depthStencilCreateInfo, m_pDevice->VkInstance()->GetAllocCallbacks(), VK_SYSTEM_ALLOCATION_SCOPE_OBJECT, depthStencil.pPalDepthStencil); // Check if pPalDepthStencil is already in the m_allGpuState.palDepthStencilState, destroy it // and use the old one if yes. The destroy is not expensive since it's just a refCount--. for (uint32_t i = 0; i < m_palDepthStencilState.NumElements(); ++i) { const DynamicDepthStencil& palDepthStencilState = m_palDepthStencilState.At(i); // Check device0 only should be sufficient if (palDepthStencilState.pPalDepthStencil[0] == depthStencil.pPalDepthStencil[0]) { pRSCache->DestroyDepthStencilState(depthStencil.pPalDepthStencil, m_pDevice->VkInstance()->GetAllocCallbacks()); pDepthStencil = &palDepthStencilState; break; } } // Add it to the m_palDepthStencilState if it doesn't exist if (pDepthStencil == nullptr) { m_palDepthStencilState.PushBack(depthStencil); pDepthStencil = &m_palDepthStencilState.Back(); } } VK_ASSERT(pDepthStencil != nullptr); PalCmdBindDepthStencilState( m_pPalCmdBuffers[deviceIdx], deviceIdx, pDepthStencil->pPalDepthStencil[deviceIdx]); } if (m_allGpuState.dirtyGraphics.samplePattern) { if (m_allGpuState.samplePattern.sampleCount != 0) { PalCmdBuffer(deviceGroup.Index())->CmdSetMsaaQuadSamplePattern( m_allGpuState.samplePattern.sampleCount, m_allGpuState.sampleLocationsEnable ? m_allGpuState.samplePattern.locations : *Device::GetDefaultQuadSamplePattern(m_allGpuState.samplePattern.sampleCount)); } } if (m_allGpuState.dirtyGraphics.msaa) { DbgBarrierPreCmd(DbgBarrierSetDynamicPipelineState); RenderStateCache* pRSCache = m_pDevice->GetRenderStateCache(); if (pMsaa == nullptr) { DynamicMsaa msaa = {}; pRSCache->CreateMsaaState(m_allGpuState.msaaCreateInfo, m_pDevice->VkInstance()->GetAllocCallbacks(), VK_SYSTEM_ALLOCATION_SCOPE_OBJECT, msaa.pPalMsaa); // Check if pPalMsaa is already in the m_palMsaaState, destroy it and use the old one if yes. // The destroy is not expensive since it's just a refCount--. for (uint32_t i = 0; i < m_palMsaaState.NumElements(); ++i) { const DynamicMsaa& palMsaaState = m_palMsaaState.At(i); // Check device0 only should be sufficient if (palMsaaState.pPalMsaa[0] == msaa.pPalMsaa[0]) { pRSCache->DestroyMsaaState(msaa.pPalMsaa, m_pDevice->VkInstance()->GetAllocCallbacks()); pMsaa = &palMsaaState; break; } } // Add it to the m_palMsaaState if it doesn't exist if (pMsaa == nullptr) { m_palMsaaState.PushBack(msaa); pMsaa = &m_palMsaaState.Back(); } } VK_ASSERT(pMsaa != nullptr); PalCmdBindMsaaState( m_pPalCmdBuffers[deviceIdx], deviceIdx, pMsaa->pPalMsaa[deviceIdx]); DbgBarrierPostCmd(DbgBarrierSetDynamicPipelineState); } } while (deviceGroup.IterateNext()); // Clear the dirty bits m_allGpuState.dirtyGraphics.u32All = 0; } } // ===================================================================================================================== void CmdBuffer::ValidateSamplePattern( uint32_t sampleCount, SamplePattern* pSamplePattern) { if (m_palQueueType == Pal::QueueTypeUniversal) { // if the current sample count is different than the current state, // use the sample pattern passed in or the default one if (sampleCount != m_allGpuState.samplePattern.sampleCount) { const Pal::MsaaQuadSamplePattern* pLocations; if ((pSamplePattern != nullptr) && (pSamplePattern->sampleCount > 0)) { VK_ASSERT(sampleCount == pSamplePattern->sampleCount); PalCmdSetMsaaQuadSamplePattern(pSamplePattern->sampleCount, pSamplePattern->locations); pLocations = &pSamplePattern->locations; } else { pLocations = Device::GetDefaultQuadSamplePattern(sampleCount); PalCmdSetMsaaQuadSamplePattern(sampleCount, *pLocations); } // If the current state doesn't have a valid sample count/pattern, update to this and clear the dirty bit. // Otherwise, we have to assume that a draw may be issued next depending on the previous sample pattern. if (m_allGpuState.samplePattern.sampleCount == 0) { m_allGpuState.samplePattern.sampleCount = sampleCount; m_allGpuState.samplePattern.locations = *pLocations; m_allGpuState.dirtyGraphics.samplePattern = 0; } else { m_allGpuState.dirtyGraphics.samplePattern = 1; } } // set current sample pattern in the hardware if it hasn't been set yet else if (m_allGpuState.dirtyGraphics.samplePattern) { PalCmdSetMsaaQuadSamplePattern( m_allGpuState.samplePattern.sampleCount, m_allGpuState.sampleLocationsEnable ? m_allGpuState.samplePattern.locations : *Device::GetDefaultQuadSamplePattern(m_allGpuState.samplePattern.sampleCount)); m_allGpuState.dirtyGraphics.samplePattern = 0; } } } // ===================================================================================================================== void CmdBuffer::SetCullModeEXT( VkCullModeFlags cullMode) { Pal::CullMode palCullMode = VkToPalCullMode(cullMode); if (m_allGpuState.triangleRasterState.cullMode != palCullMode) { m_allGpuState.triangleRasterState.cullMode = palCullMode; m_allGpuState.dirtyGraphics.rasterState = 1; } m_allGpuState.staticTokens.triangleRasterState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetFrontFaceEXT( VkFrontFace frontFace) { Pal::FaceOrientation palFrontFace = VkToPalFaceOrientation(frontFace); if (m_allGpuState.triangleRasterState.frontFace != palFrontFace) { m_allGpuState.triangleRasterState.frontFace = palFrontFace; m_allGpuState.dirtyGraphics.rasterState = 1; } m_allGpuState.staticTokens.triangleRasterState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetPrimitiveTopologyEXT( VkPrimitiveTopology primitiveTopology) { Pal::PrimitiveTopology palTopology = VkToPalPrimitiveTopology(primitiveTopology); if (m_allGpuState.inputAssemblyState.topology != palTopology) { m_allGpuState.inputAssemblyState.topology = palTopology; m_allGpuState.dirtyGraphics.inputAssembly = 1; } m_allGpuState.staticTokens.inputAssemblyState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetDepthTestEnableEXT( VkBool32 depthTestEnable) { if (m_allGpuState.depthStencilCreateInfo.depthEnable != static_cast(depthTestEnable)) { m_allGpuState.depthStencilCreateInfo.depthEnable = depthTestEnable; m_allGpuState.dirtyGraphics.depthStencil = 1; } } // ===================================================================================================================== void CmdBuffer::SetDepthWriteEnableEXT( VkBool32 depthWriteEnable) { if (m_allGpuState.depthStencilCreateInfo.depthWriteEnable != static_cast(depthWriteEnable)) { m_allGpuState.depthStencilCreateInfo.depthWriteEnable = depthWriteEnable; m_allGpuState.dirtyGraphics.depthStencil = 1; } } // ===================================================================================================================== void CmdBuffer::SetDepthCompareOpEXT( VkCompareOp depthCompareOp) { Pal::CompareFunc compareOp = VkToPalCompareFunc(depthCompareOp); if (m_allGpuState.depthStencilCreateInfo.depthFunc != compareOp) { m_allGpuState.depthStencilCreateInfo.depthFunc = compareOp; m_allGpuState.dirtyGraphics.depthStencil = 1; } } // ===================================================================================================================== void CmdBuffer::SetDepthBoundsTestEnableEXT( VkBool32 depthBoundsTestEnable) { if (m_allGpuState.depthStencilCreateInfo.depthBoundsEnable != static_cast(depthBoundsTestEnable)) { m_allGpuState.depthStencilCreateInfo.depthBoundsEnable = depthBoundsTestEnable; m_allGpuState.dirtyGraphics.depthStencil = 1; } } // ===================================================================================================================== void CmdBuffer::SetStencilTestEnableEXT( VkBool32 stencilTestEnable) { if (m_allGpuState.depthStencilCreateInfo.stencilEnable != static_cast(stencilTestEnable)) { m_allGpuState.depthStencilCreateInfo.stencilEnable = stencilTestEnable; m_allGpuState.dirtyGraphics.depthStencil = 1; } } // ===================================================================================================================== void CmdBuffer::SetStencilOpEXT( VkStencilFaceFlags faceMask, VkStencilOp failOp, VkStencilOp passOp, VkStencilOp depthFailOp, VkCompareOp compareOp) { Pal::StencilOp palFailOp = VkToPalStencilOp(failOp); Pal::StencilOp palPassOp = VkToPalStencilOp(passOp); Pal::StencilOp palDepthFailOp = VkToPalStencilOp(depthFailOp); Pal::CompareFunc palCompareOp = VkToPalCompareFunc(compareOp); Pal::DepthStencilStateCreateInfo* pCreateInfo = &(m_allGpuState.depthStencilCreateInfo); if (faceMask & VK_STENCIL_FACE_FRONT_BIT) { if ((pCreateInfo->front.stencilFailOp != palFailOp) || (pCreateInfo->front.stencilPassOp != palPassOp) || (pCreateInfo->front.stencilDepthFailOp != palDepthFailOp) || (pCreateInfo->front.stencilFunc != palCompareOp)) { pCreateInfo->front.stencilFailOp = palFailOp; pCreateInfo->front.stencilPassOp = palPassOp; pCreateInfo->front.stencilDepthFailOp = palDepthFailOp; pCreateInfo->front.stencilFunc = palCompareOp; m_allGpuState.dirtyGraphics.depthStencil = 1; } } if (faceMask & VK_STENCIL_FACE_BACK_BIT) { if ((pCreateInfo->back.stencilFailOp != palFailOp) || (pCreateInfo->back.stencilPassOp != palPassOp) || (pCreateInfo->back.stencilDepthFailOp != palDepthFailOp) || (pCreateInfo->back.stencilFunc != palCompareOp)) { pCreateInfo->back.stencilFailOp = palFailOp; pCreateInfo->back.stencilPassOp = palPassOp; pCreateInfo->back.stencilDepthFailOp = palDepthFailOp; pCreateInfo->back.stencilFunc = palCompareOp; m_allGpuState.dirtyGraphics.depthStencil = 1; } } } // ===================================================================================================================== void CmdBuffer::SetColorWriteEnableEXT( uint32_t attachmentCount, const VkBool32* pColorWriteEnables) { if (pColorWriteEnables != nullptr) { attachmentCount = Util::Min(attachmentCount, Pal::MaxColorTargets); uint32_t colorWriteEnable = m_allGpuState.colorWriteEnable; for (uint32 i = 0; i < attachmentCount; ++i) { if (pColorWriteEnables[i]) { colorWriteEnable |= (0xF << (4 * i)); } else { colorWriteEnable &= ~(0xF << (4 * i)); } } if (colorWriteEnable != m_allGpuState.colorWriteEnable) { m_allGpuState.colorWriteEnable = colorWriteEnable; m_allGpuState.dirtyGraphics.colorWriteMask = 1; } } } // ===================================================================================================================== void CmdBuffer::SetRasterizerDiscardEnableEXT( VkBool32 rasterizerDiscardEnable) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; if (pDynamicState->rasterizerDiscardEnable != static_cast(rasterizerDiscardEnable)) { pDynamicState->rasterizerDiscardEnable = rasterizerDiscardEnable; if (pDynamicState->enable.rasterizerDiscardEnable) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetPrimitiveRestartEnableEXT( VkBool32 primitiveRestartEnable) { if (m_allGpuState.inputAssemblyState.primitiveRestartEnable != static_cast(primitiveRestartEnable)) { m_allGpuState.inputAssemblyState.primitiveRestartEnable = primitiveRestartEnable; m_allGpuState.dirtyGraphics.inputAssembly = 1; } m_allGpuState.staticTokens.inputAssemblyState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetDepthBiasEnableEXT( VkBool32 depthBiasEnable) { if ((m_allGpuState.triangleRasterState.flags.frontDepthBiasEnable != depthBiasEnable) || (m_allGpuState.triangleRasterState.flags.backDepthBiasEnable != depthBiasEnable)) { m_allGpuState.triangleRasterState.flags.frontDepthBiasEnable = depthBiasEnable; m_allGpuState.triangleRasterState.flags.backDepthBiasEnable = depthBiasEnable; m_allGpuState.dirtyGraphics.rasterState = 1; } m_allGpuState.staticTokens.triangleRasterState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetColorBlendEnable( uint32_t firstAttachment, uint32_t attachmentCount, const VkBool32* pColorBlendEnables) { uint32_t lastAttachment = Util::Min(firstAttachment + attachmentCount, Pal::MaxColorTargets); for (uint32_t i = firstAttachment; i < lastAttachment; i++) { if (m_allGpuState.colorBlendCreateInfo.targets[i].blendEnable != static_cast(pColorBlendEnables[i - firstAttachment])) { m_allGpuState.colorBlendCreateInfo.targets[i].blendEnable = static_cast(pColorBlendEnables[i - firstAttachment]); m_allGpuState.dirtyGraphics.colorBlend = 1; } } } // ===================================================================================================================== void CmdBuffer::SetColorBlendEquation( uint32_t firstAttachment, uint32_t attachmentCount, const VkColorBlendEquationEXT* pColorBlendEquations) { uint32_t lastAttachment = Util::Min(firstAttachment + attachmentCount, Pal::MaxColorTargets); for (uint32_t i = firstAttachment; i < lastAttachment; i++) { const VkColorBlendEquationEXT& colorBlendEquation = pColorBlendEquations[i - firstAttachment]; auto pTarget = &m_allGpuState.colorBlendCreateInfo.targets[i]; Pal::Blend srcBlendColor = VkToPalBlend(colorBlendEquation.srcColorBlendFactor); Pal::Blend dstBlendColor = VkToPalBlend(colorBlendEquation.dstColorBlendFactor); Pal::BlendFunc blendFuncColor = VkToPalBlendFunc(colorBlendEquation.colorBlendOp); Pal::Blend srcBlendAlpha = VkToPalBlend(colorBlendEquation.srcAlphaBlendFactor); Pal::Blend dstBlendAlpha = VkToPalBlend(colorBlendEquation.dstAlphaBlendFactor); Pal::BlendFunc blendFuncAlpha = VkToPalBlendFunc(colorBlendEquation.alphaBlendOp); if ((pTarget->srcBlendColor != srcBlendColor) || (pTarget->dstBlendColor != dstBlendColor) || (pTarget->blendFuncColor != blendFuncColor) || (pTarget->srcBlendAlpha != srcBlendAlpha) || (pTarget->dstBlendAlpha != dstBlendAlpha) || (pTarget->blendFuncAlpha != blendFuncAlpha)) { pTarget->srcBlendColor = srcBlendColor; pTarget->dstBlendColor = dstBlendColor; pTarget->blendFuncColor = blendFuncColor; pTarget->srcBlendAlpha = srcBlendAlpha; pTarget->dstBlendAlpha = dstBlendAlpha; pTarget->blendFuncAlpha = blendFuncAlpha; m_allGpuState.dirtyGraphics.colorBlend = 1; } } } // ===================================================================================================================== void CmdBuffer::SetRasterizationSamples( VkSampleCountFlagBits rasterizationSamples) { const uint8 rasterizationSampleCount = static_cast(rasterizationSamples); if (rasterizationSampleCount != m_allGpuState.msaaCreateInfo.coverageSamples) { m_allGpuState.msaaCreateInfo.coverageSamples = rasterizationSampleCount; m_allGpuState.msaaCreateInfo.exposedSamples = rasterizationSampleCount; m_allGpuState.msaaCreateInfo.sampleClusters = rasterizationSampleCount; if (m_allGpuState.minSampleShading > 0.0f) { m_allGpuState.msaaCreateInfo.pixelShaderSamples = Pow2Pad(static_cast(ceil(rasterizationSampleCount * m_allGpuState.minSampleShading))); } else { m_allGpuState.msaaCreateInfo.pixelShaderSamples = 1; } m_allGpuState.msaaCreateInfo.depthStencilSamples = rasterizationSampleCount; m_allGpuState.msaaCreateInfo.shaderExportMaskSamples = rasterizationSampleCount; m_allGpuState.msaaCreateInfo.alphaToCoverageSamples = rasterizationSampleCount; m_allGpuState.msaaCreateInfo.occlusionQuerySamples = rasterizationSampleCount; m_allGpuState.dirtyGraphics.msaa = 1; } ValidateSamplePattern(rasterizationSampleCount, nullptr); m_allGpuState.samplePattern.sampleCount = rasterizationSampleCount; } // ===================================================================================================================== void CmdBuffer::SetSampleMask( VkSampleCountFlagBits samples, const VkSampleMask* pSampleMask) { if (m_allGpuState.msaaCreateInfo.sampleMask != static_cast(*pSampleMask)) { m_allGpuState.msaaCreateInfo.sampleMask = static_cast(*pSampleMask); m_allGpuState.dirtyGraphics.msaa = 1; } } // ===================================================================================================================== void CmdBuffer::SetConservativeRasterizationMode( VkConservativeRasterizationModeEXT conservativeRasterizationMode) { VK_ASSERT(m_pDevice->IsExtensionEnabled(DeviceExtensions::EXT_CONSERVATIVE_RASTERIZATION)); bool enableConservativeRasterization = false; Pal::ConservativeRasterizationMode conservativeMode = Pal::ConservativeRasterizationMode::Overestimate; switch (conservativeRasterizationMode) { case VK_CONSERVATIVE_RASTERIZATION_MODE_DISABLED_EXT: { enableConservativeRasterization = false; } break; case VK_CONSERVATIVE_RASTERIZATION_MODE_OVERESTIMATE_EXT: { enableConservativeRasterization = true; conservativeMode = Pal::ConservativeRasterizationMode::Overestimate; } break; case VK_CONSERVATIVE_RASTERIZATION_MODE_UNDERESTIMATE_EXT: { enableConservativeRasterization = true; conservativeMode = Pal::ConservativeRasterizationMode::Underestimate; } break; default: break; } if ((m_allGpuState.msaaCreateInfo.flags.enableConservativeRasterization != enableConservativeRasterization) || (enableConservativeRasterization && (conservativeMode != m_allGpuState.msaaCreateInfo.conservativeRasterizationMode))) { m_allGpuState.msaaCreateInfo.flags.enableConservativeRasterization = enableConservativeRasterization; m_allGpuState.msaaCreateInfo.conservativeRasterizationMode = conservativeMode; m_allGpuState.dirtyGraphics.msaa = 1; } } // ===================================================================================================================== void CmdBuffer::SetExtraPrimitiveOverestimationSize( float extraPrimitiveOverestimationSize) { // Do nothing return; } // ===================================================================================================================== void CmdBuffer::SetLineStippleEnable( VkBool32 stippledLineEnable) { if (m_allGpuState.msaaCreateInfo.flags.enableLineStipple != stippledLineEnable) { m_allGpuState.msaaCreateInfo.flags.enableLineStipple = stippledLineEnable; m_allGpuState.dirtyGraphics.msaa = 1; } } // ===================================================================================================================== void CmdBuffer::SetPolygonMode( VkPolygonMode polygonMode) { Pal::FillMode fillMode = VkToPalFillMode(polygonMode); if (m_allGpuState.triangleRasterState.frontFillMode != fillMode) { m_allGpuState.triangleRasterState.frontFillMode = fillMode; m_allGpuState.triangleRasterState.backFillMode = fillMode; m_allGpuState.dirtyGraphics.rasterState = 1; } m_allGpuState.staticTokens.triangleRasterState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetProvokingVertexMode( VkProvokingVertexModeEXT provokingVertexMode) { Pal::ProvokingVertex provokingVertex = VkToPalProvokingVertex(provokingVertexMode); if (m_allGpuState.triangleRasterState.provokingVertex != provokingVertex) { m_allGpuState.triangleRasterState.provokingVertex = provokingVertex; m_allGpuState.dirtyGraphics.rasterState = 1; } m_allGpuState.staticTokens.triangleRasterState = DynamicRenderStateToken; } // ===================================================================================================================== void CmdBuffer::SetColorWriteMask( uint32_t firstAttachment, uint32_t attachmentCount, const VkColorComponentFlags* pColorWriteMasks) { uint32_t lastAttachment = Util::Min(firstAttachment + attachmentCount, Pal::MaxColorTargets); uint32_t colorWriteMask = m_allGpuState.colorWriteMask; for (uint32_t i = firstAttachment; i < lastAttachment; i++) { colorWriteMask &= ~(0xF << (4 * i)); colorWriteMask |= pColorWriteMasks[i - firstAttachment] << (4 * i); } if (colorWriteMask != m_allGpuState.colorWriteMask) { m_allGpuState.colorWriteMask = colorWriteMask; m_allGpuState.dirtyGraphics.colorWriteMask = 1; } } // ===================================================================================================================== void CmdBuffer::SetSampleLocationsEnable( VkBool32 sampleLocationsEnable) { if (m_allGpuState.sampleLocationsEnable != sampleLocationsEnable) { m_allGpuState.sampleLocationsEnable = sampleLocationsEnable; m_allGpuState.dirtyGraphics.samplePattern = 1; } } // ===================================================================================================================== void CmdBuffer::SetLineRasterizationMode( VkLineRasterizationModeEXT lineRasterizationMode) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; bool perpLineEndCapsEnable = lineRasterizationMode == VK_LINE_RASTERIZATION_MODE_RECTANGULAR_EXT; if (perpLineEndCapsEnable != pDynamicState->perpLineEndCapsEnable) { pDynamicState->perpLineEndCapsEnable = perpLineEndCapsEnable; if (pDynamicState->enable.perpLineEndCapsEnable) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetLogicOp( VkLogicOp logicOp) { if (m_allGpuState.logicOp != logicOp) { m_allGpuState.logicOp = logicOp; if (m_allGpuState.logicOpEnable) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; pDynamicState->logicOp = VkToPalLogicOp(logicOp); if (pDynamicState->enable.logicOp) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } } // ===================================================================================================================== void CmdBuffer::SetLogicOpEnable( VkBool32 logicOpEnable) { if (m_allGpuState.logicOpEnable != logicOpEnable) { m_allGpuState.logicOpEnable = logicOpEnable; auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; pDynamicState->logicOp = m_allGpuState.logicOpEnable ? VkToPalLogicOp(m_allGpuState.logicOp) : Pal::LogicOp::Copy; if (pDynamicState->enable.logicOp) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetTessellationDomainOrigin( VkTessellationDomainOrigin domainOrigin) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; bool switchWinding = domainOrigin == VK_TESSELLATION_DOMAIN_ORIGIN_LOWER_LEFT; if (switchWinding != pDynamicState->switchWinding) { pDynamicState->switchWinding = switchWinding; if (pDynamicState->enable.switchWinding) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetDepthClampEnable( VkBool32 depthClampEnable) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; Pal::DepthClampMode clampMode = depthClampEnable ? Pal::DepthClampMode::Viewport : Pal::DepthClampMode::_None; if (clampMode != pDynamicState->depthClampMode) { pDynamicState->depthClampMode = clampMode; if (pDynamicState->enable.depthClampMode) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetAlphaToCoverageEnable( VkBool32 alphaToCoverageEnable) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; if (static_cast(alphaToCoverageEnable) != pDynamicState->alphaToCoverageEnable) { pDynamicState->alphaToCoverageEnable = alphaToCoverageEnable; if (pDynamicState->enable.alphaToCoverageEnable) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetDepthClipEnable( VkBool32 depthClipEnable) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; if (static_cast(depthClipEnable) != pDynamicState->depthClipNearEnable) { pDynamicState->depthClipNearEnable = depthClipEnable; pDynamicState->depthClipFarEnable = depthClipEnable; if (pDynamicState->enable.depthClipMode) { m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== void CmdBuffer::SetDepthClipNegativeOneToOne( VkBool32 negativeOneToOne) { auto pDynamicState = &m_allGpuState.pipelineState[PipelineBindGraphics].dynamicBindInfo.gfxDynState; Pal::DepthRange depthRange = negativeOneToOne ? Pal::DepthRange::NegativeOneToOne : Pal::DepthRange::ZeroToOne; if (depthRange != pDynamicState->depthRange) { pDynamicState->depthRange = depthRange; if (pDynamicState->enable.depthRange) { utils::IterateMask deviceGroup(m_curDeviceMask); do { PerGpuState(deviceGroup.Index())->viewport.depthRange = depthRange; } while (deviceGroup.IterateNext()); m_allGpuState.dirtyGraphics.viewport = 1; m_allGpuState.staticTokens.viewports = DynamicRenderStateToken; m_allGpuState.dirtyGraphics.pipeline = 1; } } } // ===================================================================================================================== RenderPassInstanceState::RenderPassInstanceState( PalAllocator* pAllocator) : pExecuteInfo(nullptr), subpass(VK_SUBPASS_EXTERNAL), renderAreaCount(0), maxAttachmentCount(0), pAttachments(nullptr), maxSubpassCount(0), pSamplePatterns(nullptr) { memset(&renderArea[0], 0, sizeof(renderArea)); } // ===================================================================================================================== template VKAPI_ATTR void VKAPI_CALL CmdBuffer::CmdBindDescriptorSets2( VkCommandBuffer cmdBuffer, const VkBindDescriptorSetsInfoKHR* pBindDescriptorSetsInfo) { ApiCmdBuffer::ObjectFromHandle(cmdBuffer)->BindDescriptorSets2( pBindDescriptorSetsInfo); } // ===================================================================================================================== template void CmdBuffer::BindDescriptorSets2( const VkBindDescriptorSetsInfoKHR* pBindDescriptorSetsInfo) { if ((pBindDescriptorSetsInfo->stageFlags & ShaderStageAllGraphics) != 0) { BindDescriptorSets( VK_PIPELINE_BIND_POINT_GRAPHICS, pBindDescriptorSetsInfo->layout, pBindDescriptorSetsInfo->firstSet, pBindDescriptorSetsInfo->descriptorSetCount, pBindDescriptorSetsInfo->pDescriptorSets, pBindDescriptorSetsInfo->dynamicOffsetCount, pBindDescriptorSetsInfo->pDynamicOffsets); } if ((pBindDescriptorSetsInfo->stageFlags & VK_SHADER_STAGE_COMPUTE_BIT) != 0) { BindDescriptorSets( VK_PIPELINE_BIND_POINT_COMPUTE, pBindDescriptorSetsInfo->layout, pBindDescriptorSetsInfo->firstSet, pBindDescriptorSetsInfo->descriptorSetCount, pBindDescriptorSetsInfo->pDescriptorSets, pBindDescriptorSetsInfo->dynamicOffsetCount, pBindDescriptorSetsInfo->pDynamicOffsets); } #if VKI_RAY_TRACING if ((pBindDescriptorSetsInfo->stageFlags & ShaderStageAllRayTracing) != 0) { BindDescriptorSets( VK_PIPELINE_BIND_POINT_RAY_TRACING_KHR, pBindDescriptorSetsInfo->layout, pBindDescriptorSetsInfo->firstSet, pBindDescriptorSetsInfo->descriptorSetCount, pBindDescriptorSetsInfo->pDescriptorSets, pBindDescriptorSetsInfo->dynamicOffsetCount, pBindDescriptorSetsInfo->pDynamicOffsets); } #endif } // ===================================================================================================================== void CmdBuffer::PushConstants2( const VkPushConstantsInfoKHR* pPushConstantsInfo) { PushConstants(pPushConstantsInfo->layout, pPushConstantsInfo->stageFlags, pPushConstantsInfo->offset, pPushConstantsInfo->size, pPushConstantsInfo->pValues); } // ===================================================================================================================== template VKAPI_ATTR void VKAPI_CALL CmdBuffer::CmdPushDescriptorSet2( VkCommandBuffer commandBuffer, const VkPushDescriptorSetInfo* pPushDescriptorSetInfo) { CmdBuffer* pCmdBuffer = ApiCmdBuffer::ObjectFromHandle(commandBuffer); pCmdBuffer->PushDescriptorSet2 ( pPushDescriptorSetInfo); } // ===================================================================================================================== template void CmdBuffer::PushDescriptorSet2( const VkPushDescriptorSetInfo* pPushDescriptorSetInfo) { if ((pPushDescriptorSetInfo->stageFlags & ShaderStageAllGraphics) != 0) { PushDescriptorSet ( VK_PIPELINE_BIND_POINT_GRAPHICS, pPushDescriptorSetInfo->layout, pPushDescriptorSetInfo->set, pPushDescriptorSetInfo->descriptorWriteCount, pPushDescriptorSetInfo->pDescriptorWrites); } if ((pPushDescriptorSetInfo->stageFlags & VK_SHADER_STAGE_COMPUTE_BIT) != 0) { PushDescriptorSet ( VK_PIPELINE_BIND_POINT_COMPUTE, pPushDescriptorSetInfo->layout, pPushDescriptorSetInfo->set, pPushDescriptorSetInfo->descriptorWriteCount, pPushDescriptorSetInfo->pDescriptorWrites); } #if VKI_RAY_TRACING if ((pPushDescriptorSetInfo->stageFlags & ShaderStageAllRayTracing) != 0) { PushDescriptorSet ( VK_PIPELINE_BIND_POINT_RAY_TRACING_KHR, pPushDescriptorSetInfo->layout, pPushDescriptorSetInfo->set, pPushDescriptorSetInfo->descriptorWriteCount, pPushDescriptorSetInfo->pDescriptorWrites); } #endif } // ===================================================================================================================== template VKAPI_ATTR void VKAPI_CALL CmdBuffer::CmdPushDescriptorSetWithTemplate2( VkCommandBuffer commandBuffer, const VkPushDescriptorSetWithTemplateInfo* pPushDescriptorSetWithTemplateInfo) { CmdBuffer* pCmdBuffer = ApiCmdBuffer::ObjectFromHandle(commandBuffer); pCmdBuffer->PushDescriptorSetWithTemplate2( pPushDescriptorSetWithTemplateInfo); } // ===================================================================================================================== template void CmdBuffer::PushDescriptorSetWithTemplate2( const VkPushDescriptorSetWithTemplateInfo* pPushDescriptorSetWithTemplateInfo) { PushDescriptorSetWithTemplate( pPushDescriptorSetWithTemplateInfo->descriptorUpdateTemplate, pPushDescriptorSetWithTemplateInfo->layout, pPushDescriptorSetWithTemplateInfo->set, pPushDescriptorSetWithTemplateInfo->pData); } // ===================================================================================================================== void CmdBuffer::SetDescriptorBufferOffsets2EXT( const VkSetDescriptorBufferOffsetsInfoEXT* pSetDescriptorBufferOffsetsInfo) { if ((pSetDescriptorBufferOffsetsInfo->stageFlags & ShaderStageAllGraphics) != 0) { SetDescriptorBufferOffsets( VK_PIPELINE_BIND_POINT_GRAPHICS, pSetDescriptorBufferOffsetsInfo->layout, pSetDescriptorBufferOffsetsInfo->firstSet, pSetDescriptorBufferOffsetsInfo->setCount, pSetDescriptorBufferOffsetsInfo->pBufferIndices, pSetDescriptorBufferOffsetsInfo->pOffsets); } if ((pSetDescriptorBufferOffsetsInfo->stageFlags & VK_SHADER_STAGE_COMPUTE_BIT) != 0) { SetDescriptorBufferOffsets( VK_PIPELINE_BIND_POINT_COMPUTE, pSetDescriptorBufferOffsetsInfo->layout, pSetDescriptorBufferOffsetsInfo->firstSet, pSetDescriptorBufferOffsetsInfo->setCount, pSetDescriptorBufferOffsetsInfo->pBufferIndices, pSetDescriptorBufferOffsetsInfo->pOffsets); } #if VKI_RAY_TRACING if ((pSetDescriptorBufferOffsetsInfo->stageFlags & ShaderStageAllRayTracing) != 0) { SetDescriptorBufferOffsets( VK_PIPELINE_BIND_POINT_RAY_TRACING_KHR, pSetDescriptorBufferOffsetsInfo->layout, pSetDescriptorBufferOffsetsInfo->firstSet, pSetDescriptorBufferOffsetsInfo->setCount, pSetDescriptorBufferOffsetsInfo->pBufferIndices, pSetDescriptorBufferOffsetsInfo->pOffsets); } #endif } // ===================================================================================================================== void CmdBuffer::BindDescriptorBufferEmbeddedSamplers2EXT( const VkBindDescriptorBufferEmbeddedSamplersInfoEXT* pBindDescriptorBufferEmbeddedSamplersInfo) { if ((pBindDescriptorBufferEmbeddedSamplersInfo->stageFlags & ShaderStageAllGraphics) != 0) { BindDescriptorBufferEmbeddedSamplers( VK_PIPELINE_BIND_POINT_GRAPHICS, pBindDescriptorBufferEmbeddedSamplersInfo->layout, pBindDescriptorBufferEmbeddedSamplersInfo->set); } if ((pBindDescriptorBufferEmbeddedSamplersInfo->stageFlags & VK_SHADER_STAGE_COMPUTE_BIT) != 0) { BindDescriptorBufferEmbeddedSamplers( VK_PIPELINE_BIND_POINT_COMPUTE, pBindDescriptorBufferEmbeddedSamplersInfo->layout, pBindDescriptorBufferEmbeddedSamplersInfo->set); } #if VKI_RAY_TRACING if ((pBindDescriptorBufferEmbeddedSamplersInfo->stageFlags & ShaderStageAllRayTracing) != 0) { BindDescriptorBufferEmbeddedSamplers( VK_PIPELINE_BIND_POINT_RAY_TRACING_KHR, pBindDescriptorBufferEmbeddedSamplersInfo->layout, pBindDescriptorBufferEmbeddedSamplersInfo->set); } #endif } // ===================================================================================================================== PFN_vkCmdBindDescriptorSets2 CmdBuffer::GetCmdBindDescriptorSets2Func( const Device* pDevice) { PFN_vkCmdBindDescriptorSets2 pFunc = nullptr; switch (pDevice->NumPalDevices()) { case 1: pFunc = GetCmdBindDescriptorSets2Func<1>(pDevice); break; #if (VKI_BUILD_MAX_NUM_GPUS > 1) case 2: pFunc = GetCmdBindDescriptorSets2Func<2>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 2) case 3: pFunc = GetCmdBindDescriptorSets2Func<3>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 3) case 4: pFunc = GetCmdBindDescriptorSets2Func<4>(pDevice); break; #endif default: pFunc = nullptr; VK_NEVER_CALLED(); break; } return pFunc; } // ===================================================================================================================== template PFN_vkCmdBindDescriptorSets2 CmdBuffer::GetCmdBindDescriptorSets2Func( const Device* pDevice) { PFN_vkCmdBindDescriptorSets2 pFunc = nullptr; if (pDevice->UseCompactDynamicDescriptors()) { pFunc = CmdBindDescriptorSets2; } else { pFunc = CmdBindDescriptorSets2; } return pFunc; } // ===================================================================================================================== template PFN_vkCmdPushDescriptorSet2 CmdBuffer::GetCmdPushDescriptorSet2Func( const Device* pDevice) { const size_t imageDescSize = pDevice->GetProperties().descriptorSizes.imageView; const size_t samplerDescSize = pDevice->GetProperties().descriptorSizes.sampler; const size_t typedBufferDescSize = pDevice->GetProperties().descriptorSizes.typedBufferView; const size_t untypedBufferDescSize = pDevice->GetProperties().descriptorSizes.untypedBufferView; PFN_vkCmdPushDescriptorSet2 pFunc = nullptr; if ((imageDescSize == 32) && (samplerDescSize == 16) && (typedBufferDescSize == 16) && (untypedBufferDescSize == 16)) { pFunc = &CmdPushDescriptorSet2< 32, 16, 16, 16, numPalDevices>; } else if ((imageDescSize == 32) && (samplerDescSize == 16) && (typedBufferDescSize == 24) && (untypedBufferDescSize == 16)) { pFunc = &CmdPushDescriptorSet2< 32, 16, 24, 16, numPalDevices>; } else { VK_NEVER_CALLED(); } return pFunc; } // ===================================================================================================================== PFN_vkCmdPushDescriptorSet2 CmdBuffer::GetCmdPushDescriptorSet2Func( const Device* pDevice) { PFN_vkCmdPushDescriptorSet2 pFunc = nullptr; switch (pDevice->NumPalDevices()) { case 1: pFunc = GetCmdPushDescriptorSet2Func<1>(pDevice); break; #if (VKI_BUILD_MAX_NUM_GPUS > 1) case 2: pFunc = GetCmdPushDescriptorSet2Func<2>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 2) case 3: pFunc = GetCmdPushDescriptorSet2Func<3>(pDevice); break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 3) case 4: pFunc = GetCmdPushDescriptorSet2Func<4>(pDevice); break; #endif default: VK_NEVER_CALLED(); break; } return pFunc; } // ===================================================================================================================== PFN_vkCmdPushDescriptorSetWithTemplate2 CmdBuffer::GetCmdPushDescriptorSetWithTemplate2Func( const Device* pDevice) { PFN_vkCmdPushDescriptorSetWithTemplate2 pFunc = nullptr; switch (pDevice->NumPalDevices()) { case 1: pFunc = CmdPushDescriptorSetWithTemplate2<1>; break; #if (VKI_BUILD_MAX_NUM_GPUS > 1) case 2: pFunc = CmdPushDescriptorSetWithTemplate2<2>; break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 2) case 3: pFunc = CmdPushDescriptorSetWithTemplate2<3>; break; #endif #if (VKI_BUILD_MAX_NUM_GPUS > 3) case 4: pFunc = CmdPushDescriptorSetWithTemplate2<4>; break; #endif default: VK_NEVER_CALLED(); break; } return pFunc; } // ===================================================================================================================== uint64_t CmdBuffer::GetUserMarkerContextValue() const { return (m_pSqttState != nullptr) ? m_pSqttState->GetUserMarkerContextValue() : 0; } // ===================================================================================================================== // Template instantiation needed for references entry.cpp. template void CmdBuffer::DrawIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset); template void CmdBuffer::DrawIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset); template void CmdBuffer::DrawIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset); template void CmdBuffer::DrawIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset); template void CmdBuffer::DrawIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa); template void CmdBuffer::DrawIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa); template void CmdBuffer::DrawIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa); template void CmdBuffer::DrawIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa); template void CmdBuffer::DrawMeshTasksIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset); template void CmdBuffer::DrawMeshTasksIndirect( VkBuffer buffer, VkDeviceSize offset, uint32_t count, uint32_t stride, VkBuffer countBuffer, VkDeviceSize countOffset); template void CmdBuffer::DrawMeshTasksIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa); template void CmdBuffer::DrawMeshTasksIndirect( VkDeviceSize indirectBufferVa, VkDeviceSize indirectBufferSize, uint32_t count, uint32_t stride, VkDeviceSize countBufferVa); template void CmdBuffer::ResolveImage( VkImage srcImage, VkImageLayout srcImageLayout, VkImage destImage, VkImageLayout destImageLayout, uint32_t rectCount, const VkImageResolve* pRects); template void CmdBuffer::ResolveImage( VkImage srcImage, VkImageLayout srcImageLayout, VkImage destImage, VkImageLayout destImageLayout, uint32_t rectCount, const VkImageResolve2* pRects); } // namespace vk