mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-13 06:38:31 +09:00
Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
37111ae992 | ||
|
|
6b0c2a15ab | ||
|
|
e9ffd99313 | ||
|
|
7c01ddea0c | ||
|
|
2d4d6e9cfb | ||
|
|
76b8957b99 | ||
|
|
ec685b9fa7 | ||
|
|
a12068df52 | ||
|
|
0b344792cc | ||
|
|
9fa32bdad0 | ||
|
|
a4980f2b56 | ||
|
|
8ca20e28ca | ||
|
|
c353a2055f | ||
|
|
421c20984e |
@@ -16,18 +16,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_device = device;
|
||||
m_commandPool = commandPool;
|
||||
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount, VK_NULL_HANDLE);
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount * 2, VK_NULL_HANDLE);
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
allocInfo.commandPool = commandPool;
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = frameCount;
|
||||
allocInfo.commandBufferCount = frameCount * 2;
|
||||
VkResult result = vkAllocateCommandBuffers(device, &allocInfo, commandBuffers.data());
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
m_frames[i].commandBuffer = commandBuffers[i];
|
||||
m_frames[i].preCommandBuffer = commandBuffers[frameCount + i];
|
||||
}
|
||||
|
||||
VkSemaphoreCreateInfo semaphoreInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
|
||||
@@ -47,9 +48,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void FrameContext::Destroy(VkDevice device, VkCommandPool commandPool) {
|
||||
const Uint32 frameCount = static_cast<Uint32>(m_frames.size());
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount, VK_NULL_HANDLE);
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount * 2, VK_NULL_HANDLE);
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
commandBuffers[i] = m_frames[i].commandBuffer;
|
||||
commandBuffers[frameCount + i] = m_frames[i].preCommandBuffer;
|
||||
}
|
||||
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
@@ -60,7 +62,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (auto& frame : m_frames) {
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
}
|
||||
vkFreeCommandBuffers(device, commandPool, frameCount, commandBuffers.data());
|
||||
vkFreeCommandBuffers(device, commandPool, frameCount * 2, commandBuffers.data());
|
||||
}
|
||||
m_frames.clear();
|
||||
currentFrameIndex = 0;
|
||||
@@ -87,6 +89,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
currentFrameIndex = (currentFrameIndex + 1) % static_cast<Uint32>(m_frames.size());
|
||||
GetCurrent().isCommandRecording = false;
|
||||
GetCurrent().hasCommandBufferRecorded = false;
|
||||
GetCurrent().isPreCommandRecording = false;
|
||||
GetCurrent().hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
|
||||
VkCommandBuffer& FrameContext::BeginCommandRecording(VkCommandBufferUsageFlags flags,
|
||||
@@ -118,6 +122,41 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
}
|
||||
|
||||
VkCommandBuffer FrameContext::BeginPreCommandRecording() {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.isPreCommandRecording) {
|
||||
return frame.preCommandBuffer;
|
||||
}
|
||||
MOBILEGL_ASSERT(!frame.hasPreCommandBufferRecorded,
|
||||
"BeginPreCommandRecording: a recorded pre stream is still awaiting submission");
|
||||
VK_VERIFY(vkResetCommandBuffer(frame.preCommandBuffer, 0), "BeginPreCommandRecording, vkResetCommandBuffer");
|
||||
VkCommandBufferBeginInfo beginInfo{};
|
||||
beginInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO;
|
||||
VK_VERIFY(vkBeginCommandBuffer(frame.preCommandBuffer, &beginInfo),
|
||||
"BeginPreCommandRecording, vkBeginCommandBuffer");
|
||||
frame.isPreCommandRecording = true;
|
||||
return frame.preCommandBuffer;
|
||||
}
|
||||
|
||||
void FrameContext::EndPreCommandRecordingIfOpen() {
|
||||
auto& frame = GetCurrent();
|
||||
if (!frame.isPreCommandRecording) {
|
||||
return;
|
||||
}
|
||||
VK_VERIFY(vkEndCommandBuffer(frame.preCommandBuffer), "EndPreCommandRecordingIfOpen, vkEndCommandBuffer");
|
||||
frame.isPreCommandRecording = false;
|
||||
frame.hasPreCommandBufferRecorded = true;
|
||||
}
|
||||
|
||||
void FrameContext::AbandonPreCommandRecording() {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.isPreCommandRecording) {
|
||||
VK_VERIFY(vkEndCommandBuffer(frame.preCommandBuffer), "AbandonPreCommandRecording, vkEndCommandBuffer");
|
||||
}
|
||||
frame.isPreCommandRecording = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
|
||||
VkResult FrameContext::InitializeSwapchainSemaphores(VkDevice device, Uint32 swapchainImageCount) {
|
||||
DestroySwapchainSemaphores(device);
|
||||
if (swapchainImageCount == 0) {
|
||||
@@ -202,17 +241,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 swapchainImageIndex) const {
|
||||
const auto& frame = GetCurrent();
|
||||
MOBILEGL_ASSERT(!frame.isCommandRecording, "GetSubmitInfo called while command buffer recording is still active");
|
||||
MOBILEGL_ASSERT(!frame.isPreCommandRecording,
|
||||
"GetSubmitInfo called while the pre-pass stream is still recording");
|
||||
AssertValidSwapchainImageIndex(swapchainImageIndex);
|
||||
SubmitInfoPacket packet{};
|
||||
packet.waitSemaphore = frame.imageAvailableSemaphore;
|
||||
packet.signalSemaphore = m_swapchainImageRenderFinishedSemaphores[swapchainImageIndex];
|
||||
packet.commandBuffer = frame.commandBuffer;
|
||||
|
||||
Uint32 commandBufferCount = 0;
|
||||
// The pre-pass stream executes strictly before the frame's commands.
|
||||
if (frame.hasPreCommandBufferRecorded) {
|
||||
packet.commandBuffers[commandBufferCount++] = frame.preCommandBuffer;
|
||||
}
|
||||
if (shouldSubmitCommandBuffer) {
|
||||
packet.commandBuffers[commandBufferCount++] = frame.commandBuffer;
|
||||
}
|
||||
|
||||
packet.submitInfo.waitSemaphoreCount = frame.imageAvailableSemaphoreConsumed ? 0U : 1U;
|
||||
packet.submitInfo.pWaitSemaphores = frame.imageAvailableSemaphoreConsumed ? nullptr : &packet.waitSemaphore;
|
||||
packet.submitInfo.pWaitDstStageMask = frame.imageAvailableSemaphoreConsumed ? nullptr : &packet.waitDstStageMask;
|
||||
packet.submitInfo.commandBufferCount = shouldSubmitCommandBuffer ? 1U : 0U;
|
||||
packet.submitInfo.pCommandBuffers = shouldSubmitCommandBuffer ? &packet.commandBuffer : nullptr;
|
||||
packet.submitInfo.commandBufferCount = commandBufferCount;
|
||||
packet.submitInfo.pCommandBuffers = commandBufferCount > 0 ? packet.commandBuffers : nullptr;
|
||||
packet.submitInfo.signalSemaphoreCount = 1;
|
||||
packet.submitInfo.pSignalSemaphores = &packet.signalSemaphore;
|
||||
return packet;
|
||||
@@ -276,12 +325,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_recordingObserver = observer;
|
||||
}
|
||||
|
||||
VkResult FrameContext::RetireCurrentCommandBuffer() {
|
||||
VkResult FrameContext::RetireCurrentCommandBuffer(Bool retirePreCommandBuffer) {
|
||||
MOBILEGL_ASSERT(m_device != VK_NULL_HANDLE && m_commandPool != VK_NULL_HANDLE,
|
||||
"RetireCurrentCommandBuffer requires an initialized FrameContext");
|
||||
auto& frame = GetCurrent();
|
||||
MOBILEGL_ASSERT(!frame.isCommandRecording,
|
||||
"RetireCurrentCommandBuffer called while the command buffer is still recording");
|
||||
MOBILEGL_ASSERT(!frame.isPreCommandRecording,
|
||||
"RetireCurrentCommandBuffer called while the pre-pass stream is still recording");
|
||||
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
@@ -289,10 +340,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = 1;
|
||||
VkCommandBuffer replacement = VK_NULL_HANDLE;
|
||||
const VkResult result = vkAllocateCommandBuffers(m_device, &allocInfo, &replacement);
|
||||
VkResult result = vkAllocateCommandBuffers(m_device, &allocInfo, &replacement);
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
if (retirePreCommandBuffer) {
|
||||
VkCommandBuffer preReplacement = VK_NULL_HANDLE;
|
||||
result = vkAllocateCommandBuffers(m_device, &allocInfo, &preReplacement);
|
||||
if (result != VK_SUCCESS) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &replacement);
|
||||
return result;
|
||||
}
|
||||
frame.retiredCommandBuffers.push_back({frame.preCommandBuffer, frame.lastSubmitIndex});
|
||||
frame.preCommandBuffer = preReplacement;
|
||||
}
|
||||
// lastSubmitIndex was just written by the renderer for the submission
|
||||
// that carried this command buffer.
|
||||
frame.retiredCommandBuffers.push_back({frame.commandBuffer, frame.lastSubmitIndex});
|
||||
|
||||
@@ -29,7 +29,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPipelineStageFlags waitDstStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
VkSemaphore waitSemaphore = VK_NULL_HANDLE;
|
||||
VkSemaphore signalSemaphore = VK_NULL_HANDLE;
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
// [0] = pre-pass command buffer (when recorded), then the frame
|
||||
// command buffer; submitInfo.pCommandBuffers points here.
|
||||
VkCommandBuffer commandBuffers[2] = {VK_NULL_HANDLE, VK_NULL_HANDLE};
|
||||
VkSubmitInfo submitInfo{VK_STRUCTURE_TYPE_SUBMIT_INFO};
|
||||
};
|
||||
|
||||
@@ -52,10 +54,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
struct FrameData {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
// Pre-pass work stream: out-of-pass commands (deferred clear
|
||||
// materialization, sampled-layout transitions) for resources the
|
||||
// frame's recording has not touched yet. Submitted immediately
|
||||
// BEFORE commandBuffer in the same vkQueueSubmit, so recording
|
||||
// into it never has to split the frame's active render pass.
|
||||
VkCommandBuffer preCommandBuffer = VK_NULL_HANDLE;
|
||||
VkSemaphore imageAvailableSemaphore = VK_NULL_HANDLE;
|
||||
VkFence imageInFlightFence = VK_NULL_HANDLE;
|
||||
Bool isCommandRecording = false;
|
||||
Bool hasCommandBufferRecorded = false;
|
||||
Bool isPreCommandRecording = false;
|
||||
Bool hasPreCommandBufferRecorded = false;
|
||||
Bool imageAvailableSemaphoreConsumed = false;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands),
|
||||
// appended in submit order; freed once their submission is known
|
||||
@@ -77,6 +87,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkCommandBuffer& BeginCommandRecording(VkCommandBufferUsageFlags flags = 0,
|
||||
const VkCommandBufferInheritanceInfo* pInheritanceInfo = nullptr);
|
||||
void EndCommandRecording();
|
||||
// Lazily opens the pre-pass work stream (see FrameData::preCommandBuffer).
|
||||
VkCommandBuffer BeginPreCommandRecording();
|
||||
// Closes the pre stream if open, marking it for submission ahead of the
|
||||
// frame command buffer. Safe to call when it never opened.
|
||||
void EndPreCommandRecordingIfOpen();
|
||||
// Drops an in-progress or recorded-but-unsubmitted pre stream (dropped
|
||||
// frame recordings, swapchain recreation).
|
||||
void AbandonPreCommandRecording();
|
||||
VkResult InitializeSwapchainSemaphores(VkDevice device, Uint32 swapchainImageCount);
|
||||
void DestroySwapchainSemaphores(VkDevice device);
|
||||
Bool TransitionToPresent(VkImage image, VkImageLayout oldLayout,
|
||||
@@ -91,7 +109,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// can restart while the submitted buffer is still executing. Retired
|
||||
// buffers are freed after the slot's fence is next waited, or as soon
|
||||
// as their submission is observed complete.
|
||||
VkResult RetireCurrentCommandBuffer();
|
||||
VkResult RetireCurrentCommandBuffer(Bool retirePreCommandBuffer = false);
|
||||
|
||||
// Frees every retired command buffer whose tagged submission index is
|
||||
// known complete. Driven by the renderer's submit tracker on completion
|
||||
|
||||
@@ -262,6 +262,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_images.resize(imageCount, VK_NULL_HANDLE);
|
||||
VK_VERIFY(vkGetSwapchainImagesKHR(device, m_swapchain, &imageCount, m_images.data()));
|
||||
m_imageLayouts.assign(imageCount, VK_IMAGE_LAYOUT_UNDEFINED);
|
||||
// Fresh swapchain images hold garbage until a render pass stores into them.
|
||||
m_imageContentDefined.assign(imageCount, false);
|
||||
m_depthStencilContentDefined.assign(imageCount, false);
|
||||
|
||||
CreateImageViews(device);
|
||||
CreateDepthStencilResources(device, physicalDevice);
|
||||
@@ -433,9 +436,39 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_images.clear();
|
||||
m_imageLayouts.clear();
|
||||
m_imageContentDefined.clear();
|
||||
m_depthStencilContentDefined.clear();
|
||||
m_preTransform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
|
||||
}
|
||||
|
||||
Bool SwapchainObject::IsImageContentDefined(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_imageContentDefined.size(), "Swapchain image content index out of range");
|
||||
return m_imageContentDefined[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetImageContentDefined(Uint32 index, Bool defined) {
|
||||
MOBILEGL_ASSERT(index < m_imageContentDefined.size(), "Swapchain image content index out of range");
|
||||
m_imageContentDefined[index] = defined;
|
||||
}
|
||||
|
||||
Bool SwapchainObject::IsDepthStencilContentDefined(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilContentDefined.size(),
|
||||
"Swapchain depth/stencil content index out of range");
|
||||
return m_depthStencilContentDefined[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetDepthStencilContentDefined(Uint32 index, Bool defined) {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilContentDefined.size(),
|
||||
"Swapchain depth/stencil content index out of range");
|
||||
m_depthStencilContentDefined[index] = defined;
|
||||
}
|
||||
|
||||
void SwapchainObject::SetAllDepthStencilContentUndefined() {
|
||||
for (SizeT i = 0; i < m_depthStencilContentDefined.size(); ++i) {
|
||||
m_depthStencilContentDefined[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
VkImage SwapchainObject::GetImage(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_images.size(), "Swapchain image index out of range");
|
||||
return m_images[index];
|
||||
|
||||
@@ -52,6 +52,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void SetImageLayout(Uint32 index, VkImageLayout layout);
|
||||
SizeT GetImageCount() const { return m_images.size(); }
|
||||
|
||||
// EGL content-validity tracking for the default framebuffer. A color
|
||||
// buffer's content is undefined once its image has been presented
|
||||
// (EGL_BUFFER_DESTROYED swap behaviour, the implementation default),
|
||||
// and every ancillary (depth/stencil) buffer's content is undefined
|
||||
// after ANY swap regardless of swap behaviour (EGL 1.5 §3.10.1). The
|
||||
// render-pass manager turns an undefined attachment's tile load into
|
||||
// LOAD_OP_DONT_CARE. Flags start false (a fresh swapchain image holds
|
||||
// garbage) and a render pass storing into an attachment sets it back
|
||||
// to defined.
|
||||
Bool IsImageContentDefined(Uint32 index) const;
|
||||
void SetImageContentDefined(Uint32 index, Bool defined);
|
||||
Bool IsDepthStencilContentDefined(Uint32 index) const;
|
||||
void SetDepthStencilContentDefined(Uint32 index, Bool defined);
|
||||
void SetAllDepthStencilContentUndefined();
|
||||
|
||||
private:
|
||||
void CreateImageViews(VkDevice device);
|
||||
void CreateDepthStencilResources(VkDevice device, VkPhysicalDevice physicalDevice);
|
||||
@@ -77,5 +92,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<VkDeviceMemory> m_depthStencilImageMemories;
|
||||
Vector<VkImageView> m_depthStencilImageViews;
|
||||
Vector<VkImageLayout> m_depthStencilImageLayouts;
|
||||
Vector<Bool> m_imageContentDefined;
|
||||
Vector<Bool> m_depthStencilContentDefined;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include <Config.h>
|
||||
#include <algorithm>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
@@ -204,6 +205,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// The frame's descriptor sets are recycled above, so last frame's reuse target
|
||||
// is gone: start the per-draw descriptor-reuse cache fresh this frame.
|
||||
m_hasLastDescriptor = false;
|
||||
m_lastBindValid = false;
|
||||
// Re-fingerprint the bound sampler set fresh this frame so any GL object address
|
||||
// reuse cannot outlive a single frame (see SamplerResolveMemo).
|
||||
for (auto& memo : m_samplerResolveMemo) {
|
||||
@@ -1189,6 +1191,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bufferInfo.range = ubo.range;
|
||||
dynOffset = static_cast<Uint32>(ubo.dynamicOffset);
|
||||
} else {
|
||||
// Global-UBO slice reuse (see GlobalUboSliceMemo): unchanged
|
||||
// uniform bytes re-use the slice already uploaded this frame.
|
||||
const Bool isGlobalUbo =
|
||||
programObj.globalUboBinding == static_cast<Int>(binding) && element == 0;
|
||||
const Uint64 uboFrameSerial = m_bufferManager->GetFrameSerial();
|
||||
const Uint64 uboProgramLifetimeId = program.GetLifetimeId();
|
||||
const Uint32 uboContentVersion = program.GetUBOContentVersion();
|
||||
Bool reusedSlice = false;
|
||||
if (isGlobalUbo) {
|
||||
for (const auto& memo : m_globalUboMemo) {
|
||||
if (memo.buffer != VK_NULL_HANDLE &&
|
||||
memo.programLifetimeId == uboProgramLifetimeId &&
|
||||
memo.frameSerial == uboFrameSerial &&
|
||||
memo.uboContentVersion == uboContentVersion &&
|
||||
memo.range == static_cast<VkDeviceSize>(ubo.payloadSize)) {
|
||||
bufferInfo.buffer = memo.buffer;
|
||||
bufferInfo.range = memo.range;
|
||||
dynOffset = static_cast<Uint32>(memo.offset);
|
||||
reusedSlice = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!reusedSlice) {
|
||||
BufferSlice slice{};
|
||||
if (!m_bufferManager->UploadTransient(BufferKind::Uniform, frameIndex, ubo.payload,
|
||||
ubo.payloadSize, m_minDynamicOffsetAlignment, slice)) {
|
||||
@@ -1199,6 +1225,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bufferInfo.buffer = slice.buffer;
|
||||
bufferInfo.range = ubo.payloadSize;
|
||||
dynOffset = static_cast<Uint32>(slice.offset);
|
||||
if (isGlobalUbo) {
|
||||
m_globalUboMemo[m_globalUboMemoNext] = GlobalUboSliceMemo{
|
||||
uboProgramLifetimeId, uboFrameSerial, uboContentVersion,
|
||||
slice.buffer, slice.offset, static_cast<VkDeviceSize>(ubo.payloadSize)};
|
||||
m_globalUboMemoNext = (m_globalUboMemoNext + 1) % kGlobalUboMemoSize;
|
||||
}
|
||||
}
|
||||
}
|
||||
bufferInfos.push_back(bufferInfo);
|
||||
// Dynamic offsets are consumed in binding order, then array element order,
|
||||
@@ -1337,8 +1370,34 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_hasLastDescriptor = cacheable;
|
||||
}
|
||||
|
||||
// Skip the driver call when this exact binding is already live on the
|
||||
// command buffer (see the bind-dedup shadow in the header).
|
||||
const Uint32 offsetCount = static_cast<Uint32>(dynamicOffsets.size());
|
||||
Bool identicalBind = m_lastBindValid && m_lastBindSet == descriptorSet &&
|
||||
m_lastBindLayout == programObj.pipelineLayout && m_lastBindPoint == bindPoint &&
|
||||
m_lastBindOffsetCount == offsetCount && offsetCount <= kMaxShadowedDynamicOffsets;
|
||||
if (identicalBind) {
|
||||
for (Uint32 i = 0; i < offsetCount; ++i) {
|
||||
if (m_lastBindOffsets[i] != dynamicOffsets[i]) {
|
||||
identicalBind = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!identicalBind) {
|
||||
vkCmdBindDescriptorSets(commandBuffer, bindPoint, programObj.pipelineLayout, 0, 1,
|
||||
&descriptorSet, static_cast<Uint32>(dynamicOffsets.size()), dynamicOffsets.data());
|
||||
&descriptorSet, offsetCount, dynamicOffsets.data());
|
||||
if (offsetCount <= kMaxShadowedDynamicOffsets) {
|
||||
m_lastBindValid = true;
|
||||
m_lastBindSet = descriptorSet;
|
||||
m_lastBindLayout = programObj.pipelineLayout;
|
||||
m_lastBindPoint = bindPoint;
|
||||
m_lastBindOffsetCount = offsetCount;
|
||||
std::copy_n(dynamicOffsets.data(), offsetCount, m_lastBindOffsets);
|
||||
} else {
|
||||
m_lastBindValid = false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -39,6 +39,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void Shutdown();
|
||||
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// A command buffer (re)began recording: descriptor bindings recorded into
|
||||
// the previous buffer do not carry over, so drop the bind-dedup shadow.
|
||||
void OnCommandBufferBoundary() { m_lastBindValid = false; }
|
||||
// A ProgramFactory eviction just destroyed this layout: purge every frame
|
||||
// slot's cached descriptor sets for it, so a recycled handle value can never
|
||||
// stale-hit sets written for the dead layout's bindings. The sets are
|
||||
@@ -185,6 +188,35 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 m_lastDescriptorSignature = 0;
|
||||
Bool m_hasLastDescriptor = false;
|
||||
|
||||
// vkCmdBindDescriptorSets dedup: consecutive draws with a static uniform
|
||||
// block resolve to the same set AND the same dynamic offsets, so the
|
||||
// driver call can be skipped outright. Command-buffer-scope state; reset
|
||||
// via OnCommandBufferBoundary whenever a recording (re)begins. Keyed on
|
||||
// layout+bind point, so a pipeline-layout switch always rebinds.
|
||||
static constexpr Uint32 kMaxShadowedDynamicOffsets = 8;
|
||||
Bool m_lastBindValid = false;
|
||||
VkDescriptorSet m_lastBindSet = VK_NULL_HANDLE;
|
||||
VkPipelineLayout m_lastBindLayout = VK_NULL_HANDLE;
|
||||
VkPipelineBindPoint m_lastBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
|
||||
Uint32 m_lastBindOffsetCount = 0;
|
||||
Uint32 m_lastBindOffsets[kMaxShadowedDynamicOffsets] = {};
|
||||
|
||||
// Global-UBO transient-slice reuse: MC leaves the default uniform block
|
||||
// untouched across long GUI/terrain runs, so the per-draw re-upload of
|
||||
// the same bytes can reuse the slice uploaded earlier THIS frame (frame
|
||||
// serial guards arena recycling; the content version guards writes).
|
||||
struct GlobalUboSliceMemo {
|
||||
Uint64 programLifetimeId = 0;
|
||||
Uint64 frameSerial = 0;
|
||||
Uint32 uboContentVersion = 0;
|
||||
VkBuffer buffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize offset = 0;
|
||||
VkDeviceSize range = 0;
|
||||
};
|
||||
static constexpr Uint32 kGlobalUboMemoSize = 4;
|
||||
GlobalUboSliceMemo m_globalUboMemo[kGlobalUboMemoSize];
|
||||
Uint32 m_globalUboMemoNext = 0;
|
||||
|
||||
// Per-binding fast path over VkSamplerManager's content-hashed sampler cache, which
|
||||
// stays the source of truth: its key hashes all sampler+texture state, so two distinct
|
||||
// sampler objects with identical state still resolve to one VkSampler. This memo only
|
||||
|
||||
@@ -58,15 +58,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VertexInputStateFactory::BackendVertexInputState& VertexInputStateFactory::GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao) {
|
||||
return GetOrCreateVertexInputState(vao, GetOrComputeHash(vao));
|
||||
// Per-draw fast path: the VAO carries a pointer to its resolved entry,
|
||||
// valid while its config version and the cache's eviction epoch both
|
||||
// match - no re-hash, no map lookup.
|
||||
const void* memoState = nullptr;
|
||||
Uint64 memoEpoch = 0;
|
||||
if (vao.GetBackendStateMemo(memoState, memoEpoch) && memoEpoch == m_evictionEpoch) {
|
||||
const auto* entry = static_cast<const BackendVertexInputState*>(memoState);
|
||||
entry->lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return *entry;
|
||||
}
|
||||
const BackendVertexInputState& entry = GetOrCreateVertexInputState(vao, GetOrComputeHash(vao));
|
||||
vao.SetBackendStateMemo(&entry, m_evictionEpoch);
|
||||
return entry;
|
||||
}
|
||||
|
||||
const VertexInputStateFactory::BackendVertexInputState& VertexInputStateFactory::GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash) {
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return it->second;
|
||||
it->second->lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return *it->second;
|
||||
}
|
||||
|
||||
VertexInputStateBuilder builder;
|
||||
@@ -172,11 +184,37 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const auto& state = builder.Build();
|
||||
|
||||
auto& entry = m_cache[hash];
|
||||
auto& slot = m_cache[hash];
|
||||
if (!slot) {
|
||||
slot = MakeUnique<BackendVertexInputState>();
|
||||
}
|
||||
BackendVertexInputState& entry = *slot;
|
||||
entry.hash = hash;
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
entry.bindings = builder.GetBindings();
|
||||
entry.attributes = builder.GetAttributes();
|
||||
// See the layoutHash declaration: hash only the resolved layout, never
|
||||
// buffer identities, so identical layouts across VAOs/buffers agree.
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, 0));
|
||||
for (const auto& binding : entry.bindings) {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &binding.binding, sizeof(binding.binding)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &binding.stride, sizeof(binding.stride)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &binding.inputRate, sizeof(binding.inputRate)));
|
||||
}
|
||||
for (const auto& attribute : entry.attributes) {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.location, sizeof(attribute.location)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.binding, sizeof(attribute.binding)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.format, sizeof(attribute.format)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.offset, sizeof(attribute.offset)));
|
||||
}
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &unsupportedAttribMask, sizeof(unsupportedAttribMask)));
|
||||
entry.layoutHash = XXH64_digest(m_hashState);
|
||||
entry.attributeLocationMask = 0;
|
||||
for (const auto& attribute : entry.attributes) {
|
||||
if (attribute.location < 32u) {
|
||||
entry.attributeLocationMask |= (1u << attribute.location);
|
||||
}
|
||||
}
|
||||
entry.bindingBufferKeys = std::move(bindingBufferKeys);
|
||||
entry.bindingBaseOffsets = std::move(bindingBaseOffsets);
|
||||
entry.bindingAttributeLocations = std::move(bindingAttributeLocations);
|
||||
@@ -205,8 +243,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameBoundaryCounter - it->second.lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
if (m_frameBoundaryCounter - it->second->lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
it = m_cache.erase(it);
|
||||
// Invalidate every VAO's state-pointer memo: the erased node's
|
||||
// address may be reused by a future insert.
|
||||
++m_evictionEpoch;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
|
||||
@@ -27,9 +27,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
struct BackendVertexInputState {
|
||||
HashType hash = 0;
|
||||
// Hash of the resolved Vulkan vertex layout only (bindings, attributes,
|
||||
// unsupported mask) - NO buffer identities. `hash` mixes buffer heap
|
||||
// addresses so per-chunk VBOs mint a fresh identity per buffer; keying
|
||||
// pipelines on that minted one VkPipeline per chunk section for an
|
||||
// identical layout, defeating pipeline reuse and the per-draw memo.
|
||||
// Pipelines depend only on the layout, so they key on this instead.
|
||||
HashType layoutHash = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age are evicted (CPU heap only).
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
// Mutable: the VAO's state-pointer memo fast path stamps it through
|
||||
// a const entry reference.
|
||||
mutable Uint64 lastUsedFrameBoundary = 0;
|
||||
Vector<VkVertexInputBindingDescription> bindings;
|
||||
Vector<VkVertexInputAttributeDescription> attributes;
|
||||
Vector<SizeT> bindingBufferKeys;
|
||||
@@ -41,6 +50,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// absent from `attributes`, so without this mask the draw path cannot tell them apart from
|
||||
// a genuinely disabled array and would silently feed the shader the current attribute value.
|
||||
Uint32 unsupportedAttribMask = 0;
|
||||
// Bitmask of `attributes[i].location` - the draw path needs it up to
|
||||
// three times per draw, so it is baked once at build time.
|
||||
Uint32 attributeLocationMask = 0;
|
||||
VkPipelineVertexInputStateCreateInfo state{
|
||||
VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO
|
||||
};
|
||||
@@ -80,9 +92,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, BackendVertexInputState> m_cache;
|
||||
// Values are heap-allocated: FastSTL::unordered_map is open-addressing,
|
||||
// so INSERT invalidates references to stored values. The draw path (and
|
||||
// the VAOs' state-pointer memos) hold entry pointers across inserts;
|
||||
// only the unique_ptr cell moves, never the pointee.
|
||||
UnorderedMap<HashType, UniquePtr<BackendVertexInputState>> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
// Bumped whenever any cache entry is erased. VAOs memo a raw pointer to
|
||||
// their heap-allocated entry (stable across map insert/rehash by
|
||||
// construction); a memo is honored only while its recorded epoch
|
||||
// matches, so an evicted entry can never be dereferenced through a
|
||||
// stale memo.
|
||||
Uint64 m_evictionEpoch = 1;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -93,6 +93,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_pendingClears.clear();
|
||||
m_aliveObjects.clear();
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
TextureIdentity VkClearManager::MakeTextureIdentity(MG_State::GLState::ITextureObject* texture) {
|
||||
@@ -127,6 +128,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_pendingClears.erase(key);
|
||||
}
|
||||
m_aliveObjects.erase(identity);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
Bool VkClearManager::LockTextureIdentityLocked(const TextureIdentity& identity,
|
||||
@@ -221,6 +223,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_aliveObjects[MakeTextureIdentity(texture.get())] = texture;
|
||||
auto& pending = m_pendingClears[key];
|
||||
MergeClearPayload(pending, clearPayload);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
void VkClearManager::QueueClear(const ClearAttachmentPayload& clearPayload,
|
||||
@@ -238,6 +241,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_aliveObjects[MakeTextureIdentity(texture.get())] = texture;
|
||||
auto& pending = m_pendingClears[key];
|
||||
MergeClearPayload(pending, clearPayload);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
Bool VkClearManager::HasPendingClear(MG_State::GLState::ITextureObject* texture) {
|
||||
@@ -245,6 +249,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const Uint64 lifetimeId = texture->GetLifetimeId();
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
for (auto it = m_pendingClears.begin(); it != m_pendingClears.end(); ++it) {
|
||||
@@ -260,6 +268,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (key.texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
if (m_pendingClears.find(key) == m_pendingClears.end()) {
|
||||
@@ -287,6 +298,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (key.texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
if (!LockTextureLocked(key, outTexture)) {
|
||||
@@ -325,6 +339,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const Uint64 lifetimeId = texture->GetLifetimeId();
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
@@ -345,6 +362,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
const TextureIdentity identity = MakeTextureIdentity(texture);
|
||||
MGLOG_D("%s: Pop all pending clears for texture %d", __func__, texture->GetExternalIndex());
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
@@ -361,6 +381,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto it = m_pendingClears.find(key);
|
||||
if (it != m_pendingClears.end()) {
|
||||
m_pendingClears.erase(it);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
#include "MG_Util/Math/VectorTypes.h"
|
||||
|
||||
#include <Includes.h>
|
||||
#include <atomic>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
@@ -120,7 +121,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture);
|
||||
|
||||
Uint8 m_gcCounter = 0;
|
||||
public:
|
||||
// Lock-free probe for the consecutive-draw fast path: any pending clear
|
||||
// forces the full SetupDraw path (which materializes/consumes it).
|
||||
Bool HasAnyPendingClears() const { return m_pendingCount.load(std::memory_order_relaxed) != 0; }
|
||||
|
||||
private:
|
||||
mutable std::mutex m_mutex;
|
||||
// Lock-free mirror of m_pendingClears.size(), maintained under m_mutex
|
||||
// by every mutation. The per-draw probes (HasPendingClear/GetPending*)
|
||||
// read it before taking the lock: during draw batches the pending set
|
||||
// is almost always empty, so this turns several locked map probes per
|
||||
// draw into one relaxed load.
|
||||
std::atomic<Uint32> m_pendingCount{0};
|
||||
std::unordered_map<PendingClearKey, ClearAttachmentPayload, PendingClearKeyHash> m_pendingClears;
|
||||
std::unordered_map<TextureIdentity, WeakPtr<MG_State::GLState::ITextureObject>, TextureIdentityHash> m_aliveObjects;
|
||||
};
|
||||
|
||||
@@ -481,7 +481,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
VkRenderPassManager::HashType VkRenderPassManager::ComputeHash(
|
||||
const MG_State::GLState::FramebufferObject& fbo, Uint32 swapchainImageIndex, Bool includePendingClear) {
|
||||
const MG_State::GLState::FramebufferObject& fbo, Uint32 swapchainImageIndex, Bool includePendingClear,
|
||||
Bool includeDefaultFboDepthStencil) {
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, m_config.CacheVersion));
|
||||
const Bool isDefaultFbo = fbo.IsDefaultFramebuffer();
|
||||
if (isDefaultFbo) {
|
||||
@@ -560,9 +561,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
attachment <= FramebufferAttachmentType::BackRight);
|
||||
if (isDefaultColorAttachment) {
|
||||
currentLayout = m_swapchainObject.GetImageLayout(swapchainImageIndex);
|
||||
// Content validity feeds the attachment's loadOp (see the
|
||||
// creation path), so it must key the cache as well.
|
||||
if (!m_swapchainObject.IsImageContentDefined(swapchainImageIndex)) {
|
||||
currentLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
} else if (attachment == FramebufferAttachmentType::Depth ||
|
||||
attachment == FramebufferAttachmentType::Stencil) {
|
||||
currentLayout = m_swapchainObject.GetDepthStencilImageLayout(swapchainImageIndex);
|
||||
if (!m_swapchainObject.IsDepthStencilContentDefined(swapchainImageIndex)) {
|
||||
currentLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
auto* textureResource = m_textureManager.SyncTextureAndGetDescriptor(*texture);
|
||||
@@ -617,14 +626,49 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
combineFramebufferAttachmentObjHash(drawbuf);
|
||||
}
|
||||
|
||||
// The depth-less default-FBO flavor omits the depth/stencil attachment
|
||||
// entirely, so it must hash differently from the depth-full flavor.
|
||||
const Bool depthStencilIncluded = !isDefaultFbo || includeDefaultFboDepthStencil;
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &depthStencilIncluded, sizeof(depthStencilIncluded)));
|
||||
if (depthStencilIncluded) {
|
||||
combineFramebufferAttachmentObjHash(FramebufferAttachmentType::Depth);
|
||||
combineFramebufferAttachmentObjHash(FramebufferAttachmentType::Stencil);
|
||||
}
|
||||
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
RenderPassEntry& VkRenderPassManager::GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex) {
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil) {
|
||||
// Resolve the default-FBO depth flavor (see the header comment): keep the
|
||||
// depth attachment when the caller needs it, when a depth/stencil clear is
|
||||
// pending, or when the active pass already carries it (escalate-only, so
|
||||
// alternating depth-less draws never split an established depth pass).
|
||||
Bool includeDefaultFboDepthStencil = true;
|
||||
if (fbo.IsDefaultFramebuffer()) {
|
||||
Bool activeDefaultHasDepthStencil = false;
|
||||
if (const auto* active = GetActiveRenderPass()) {
|
||||
Bool activeIsSwapchainPass = false;
|
||||
Bool activeHasSwapchainDepthStencil = false;
|
||||
for (const auto& tracked : active->trackedAttachmentLayouts) {
|
||||
activeIsSwapchainPass |= tracked.target == TrackedAttachmentTarget::SwapchainColor;
|
||||
activeHasSwapchainDepthStencil |=
|
||||
tracked.target == TrackedAttachmentTarget::SwapchainDepthStencil;
|
||||
}
|
||||
activeDefaultHasDepthStencil = activeIsSwapchainPass && activeHasSwapchainDepthStencil;
|
||||
}
|
||||
const auto& defaultDepthAtt = fbo.GetAttachment(FramebufferAttachmentType::Depth);
|
||||
const auto& defaultStencilAtt = fbo.GetAttachment(FramebufferAttachmentType::Stencil);
|
||||
const Bool pendingDepthStencilClear =
|
||||
(defaultDepthAtt.IsTexture() && m_clearManager.HasPendingClear(defaultDepthAtt)) ||
|
||||
HasPendingRenderbufferClear(defaultDepthAtt) ||
|
||||
(defaultStencilAtt.IsTexture() && m_clearManager.HasPendingClear(defaultStencilAtt)) ||
|
||||
HasPendingRenderbufferClear(defaultStencilAtt);
|
||||
includeDefaultFboDepthStencil =
|
||||
drawUsesDepthStencil || activeDefaultHasDepthStencil || pendingDepthStencilClear;
|
||||
}
|
||||
|
||||
auto hasPendingClearOnFramebuffer = [&]() -> Bool {
|
||||
const auto& drawBuffers = fbo.GetDrawBuffers();
|
||||
for (auto attachment : drawBuffers) {
|
||||
@@ -674,6 +718,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastFboVersion == fbo.GetObjectVersion() && m_rpFastSwapchainIndex == swapchainImageIndex &&
|
||||
m_rpFastTexEpoch == m_textureManager.GetTextureImageEpoch() &&
|
||||
m_rpFastRbEpoch == m_renderbufferImageEpoch &&
|
||||
(!fbo.IsDefaultFramebuffer() || m_rpFastHadDepthStencil == includeDefaultFboDepthStencil) &&
|
||||
m_rpFastRenderPassHash == activeRenderPass->hash && !hasPendingClearOnFramebuffer()) {
|
||||
auto activeIt = m_renderPasses.find(activeRenderPass->hash);
|
||||
if (activeIt != m_renderPasses.end()) {
|
||||
@@ -682,7 +727,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
auto compatibilityHash = ComputeHash(fbo, swapchainImageIndex, false);
|
||||
auto compatibilityHash = ComputeHash(fbo, swapchainImageIndex, false, includeDefaultFboDepthStencil);
|
||||
if (activeRenderPass != nullptr &&
|
||||
activeRenderPass->CompatibleWith(compatibilityHash) &&
|
||||
!hasPendingClearOnFramebuffer()) {
|
||||
@@ -699,10 +744,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastTexEpoch = m_textureManager.GetTextureImageEpoch();
|
||||
m_rpFastRbEpoch = m_renderbufferImageEpoch;
|
||||
m_rpFastRenderPassHash = activeRenderPass->hash;
|
||||
m_rpFastHadDepthStencil = activeIt->second.hasDepthStencilAttachment;
|
||||
activeIt->second.lastUsedFrame = m_frameCounter;
|
||||
return activeIt->second;
|
||||
}
|
||||
auto hash = ComputeHash(fbo, swapchainImageIndex, true);
|
||||
auto hash = ComputeHash(fbo, swapchainImageIndex, true, includeDefaultFboDepthStencil);
|
||||
auto it = m_renderPasses.find(hash);
|
||||
if (it != m_renderPasses.end()) {
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
@@ -894,6 +940,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(swapchainImageIndex < swapchainViews.size(),
|
||||
"GetOrCreateRenderPass: swapchain image index out of range");
|
||||
trackedColorLayout = m_swapchainObject.GetImageLayout(swapchainImageIndex);
|
||||
// EGL: a presented color buffer's content is undefined when its
|
||||
// image comes back around (EGL_BUFFER_DESTROYED, the default
|
||||
// swap behaviour) - skip the tile load instead of reloading
|
||||
// stale pixels nobody may rely on.
|
||||
if (!hasClear && !m_swapchainObject.IsImageContentDefined(swapchainImageIndex)) {
|
||||
trackedColorLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
.target = TrackedAttachmentTarget::SwapchainColor,
|
||||
.swapchainImageIndex = swapchainImageIndex,
|
||||
@@ -913,6 +966,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
.target = TrackedAttachmentTarget::Texture,
|
||||
.texture = att.GetTexture(),
|
||||
.textureRaw = att.GetTexture().get(),
|
||||
.textureMipLevel = attachmentMipLevel,
|
||||
.finalLayout = desc.finalLayout,
|
||||
});
|
||||
@@ -976,6 +1030,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
const auto* selectedDepthStencilAttachment = isUsableDepthStencilAttachment(depthAtt) ? &depthAtt :
|
||||
(isUsableDepthStencilAttachment(stencilAtt) ? &stencilAtt : nullptr);
|
||||
// Depth-less default-FBO flavor: nothing in this pass touches depth/stencil
|
||||
// and their content is undefined anyway (EGL swap), so drop the attachment
|
||||
// and its whole tile load + store.
|
||||
if (isDefaultFbo && !includeDefaultFboDepthStencil) {
|
||||
selectedDepthStencilAttachment = nullptr;
|
||||
}
|
||||
const Bool hasDistinctDepthAndStencilAttachments =
|
||||
isUsableDepthStencilAttachment(depthAtt) && isUsableDepthStencilAttachment(stencilAtt) &&
|
||||
!sameDepthStencilAttachmentObject(depthAtt, stencilAtt);
|
||||
@@ -994,6 +1054,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkImageLayout trackedDepthLayout = isDefaultFbo ?
|
||||
m_swapchainObject.GetDepthStencilImageLayout(swapchainImageIndex) :
|
||||
VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
||||
// EGL 1.5 §3.10.1: every ancillary (depth/stencil) buffer's content is
|
||||
// undefined after a swap, so the first default-FBO pass of a frame can
|
||||
// skip the depth/stencil tile load outright.
|
||||
if (isDefaultFbo && !m_swapchainObject.IsDepthStencilContentDefined(swapchainImageIndex)) {
|
||||
trackedDepthLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
depthAttachmentDescription.flags = 0;
|
||||
VkSampleCountFlagBits depthAttachmentSampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
Int depthAttachmentId = 0;
|
||||
@@ -1077,6 +1143,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
.target = TrackedAttachmentTarget::Texture,
|
||||
.texture = selectedDepthStencilAttachment->GetTexture(),
|
||||
.textureRaw = selectedDepthStencilAttachment->GetTexture().get(),
|
||||
.textureMipLevel = attachmentMipLevel,
|
||||
.finalLayout = depthAttachmentDescription.finalLayout,
|
||||
});
|
||||
@@ -1122,6 +1189,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
const Bool hasDepthStencilAttachment = depthAttachmentRef.attachment != VK_ATTACHMENT_UNUSED;
|
||||
|
||||
// Declare only the used colour-reference span. The GL draw-buffer array
|
||||
// always spans 8 slots, so passes used to declare colorAttachmentCount=8
|
||||
// with trailing VK_ATTACHMENT_UNUSED holes - and Adreno configures its
|
||||
// per-pixel render-backend/export path from the DECLARED count, so every
|
||||
// fragment of every pass paid the 8-target export cost (measured on
|
||||
// Adreno 650 / MC 26.2: 11.9 -> 7.5 ms of GPU time per frame, with the
|
||||
// single-quad swapchain blit pass alone dropping 1.26 -> 0.40 ms).
|
||||
// Interior GL_NONE holes keep their slots so fragment-output locations
|
||||
// still line up; a fragment output at a location past the trimmed count
|
||||
// is discarded, which is exactly GL's semantic for writing to a draw
|
||||
// buffer set to GL_NONE.
|
||||
while (!colorAttachmentRefs.empty() &&
|
||||
colorAttachmentRefs.back().attachment == VK_ATTACHMENT_UNUSED) {
|
||||
colorAttachmentRefs.pop_back();
|
||||
}
|
||||
|
||||
// Subpass
|
||||
VkSubpassDescription subpassDesc;
|
||||
subpassDesc.flags = 0;
|
||||
@@ -1330,6 +1413,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
renderPassBeginInfo.pClearValues = clearValues.data();
|
||||
|
||||
vkCmdBeginRenderPass(commandBuffer, &renderPassBeginInfo, VK_SUBPASS_CONTENTS_INLINE);
|
||||
// Pre-pass stream bookkeeping: this pass's attachment images are now
|
||||
// referenced by the open frame recording.
|
||||
if (s_textureManager != nullptr) {
|
||||
for (const auto& tracked : renderPassEntry.trackedAttachmentLayouts) {
|
||||
if (tracked.target == TrackedAttachmentTarget::Texture) {
|
||||
if (const auto texture = tracked.texture.lock()) {
|
||||
s_textureManager->StampTextureRecordingUse(texture.get());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const auto& pending: renderPassEntry.pendingClearAttachments) {
|
||||
if (pending.hasInlinePayload) {
|
||||
if (s_renderPassManager != nullptr) {
|
||||
@@ -1382,11 +1476,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case TrackedAttachmentTarget::SwapchainColor:
|
||||
MOBILEGL_ASSERT(s_swapchainObject != nullptr, "EndRenderPass: swapchain object is null");
|
||||
s_swapchainObject->SetImageLayout(trackedAttachment.swapchainImageIndex, trackedAttachment.finalLayout);
|
||||
// The pass stored into the attachment: its content is defined
|
||||
// until the image is next presented.
|
||||
s_swapchainObject->SetImageContentDefined(trackedAttachment.swapchainImageIndex, true);
|
||||
break;
|
||||
case TrackedAttachmentTarget::SwapchainDepthStencil:
|
||||
MOBILEGL_ASSERT(s_swapchainObject != nullptr, "EndRenderPass: swapchain object is null");
|
||||
s_swapchainObject->SetDepthStencilImageLayout(trackedAttachment.swapchainImageIndex,
|
||||
trackedAttachment.finalLayout);
|
||||
s_swapchainObject->SetDepthStencilContentDefined(trackedAttachment.swapchainImageIndex, true);
|
||||
break;
|
||||
default:
|
||||
MOBILEGL_ASSERT(false, "EndRenderPass: unsupported tracked attachment target=%d",
|
||||
|
||||
@@ -42,6 +42,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
struct TrackedAttachmentLayoutInfo {
|
||||
TrackedAttachmentTarget target = TrackedAttachmentTarget::Texture;
|
||||
WeakPtr<MG_State::GLState::ITextureObject> texture;
|
||||
// Identity-compare shortcut for the per-draw "does the active pass use
|
||||
// this sampled texture" probe: comparing this against a LIVE texture's
|
||||
// address needs no weak_ptr::lock (two refcount atomics per probe).
|
||||
// May dangle once the texture dies - compare only, never dereference.
|
||||
MG_State::GLState::ITextureObject* textureRaw = nullptr;
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
Uint32 textureMipLevel = 0;
|
||||
Uint32 swapchainImageIndex = 0;
|
||||
@@ -188,8 +193,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
HashType ComputeHash(
|
||||
const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool includePendingClear = true);
|
||||
RenderPassEntry& GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo, Uint32 swapchainImageIndex);
|
||||
Bool includePendingClear = true,
|
||||
Bool includeDefaultFboDepthStencil = true);
|
||||
// drawUsesDepthStencil: whether the operation about to run inside the pass
|
||||
// reads or writes the depth/stencil buffer (depth test or stencil test
|
||||
// enabled, or a depth/stencil clear). Only consulted for the DEFAULT
|
||||
// framebuffer: EGL undefines its ancillary buffers at every swap, so a
|
||||
// default-FBO pass whose draws provably never touch depth/stencil is
|
||||
// created WITHOUT the depth attachment - on a tiler that skips the whole
|
||||
// depth tile load AND store. The flavor only escalates: once a pass with
|
||||
// depth is active, later depth-less draws keep using it, and a depth-using
|
||||
// draw against a depth-less active pass resolves to a new (incompatible)
|
||||
// entry, which the caller's compatibility check turns into a pass split;
|
||||
// the new pass's depth loads DONT_CARE (content was undefined all along).
|
||||
RenderPassEntry& GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil = true);
|
||||
void QueueRenderbufferClear(GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo);
|
||||
void QueueRenderbufferClear(const ClearAttachmentPayload& clearPayload,
|
||||
@@ -219,6 +238,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// image recreation.
|
||||
Uint64 m_renderbufferImageEpoch = 1;
|
||||
|
||||
public:
|
||||
// Bumped whenever a renderbuffer backing is (re)created; consecutive-draw
|
||||
// snapshots include it so an attachment respecify forces a re-resolve.
|
||||
Uint64 GetRenderbufferImageEpoch() const { return m_renderbufferImageEpoch; }
|
||||
|
||||
private:
|
||||
|
||||
// Per-draw fast-path memo for GetOrCreateRenderPass (dirty-flag state tracking): when the
|
||||
// framebuffer state is provably unchanged since the last resolution, the active render pass
|
||||
// is reused WITHOUT recomputing the expensive per-draw hash. Invalidated by FBO switch /
|
||||
@@ -231,6 +257,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 m_rpFastTexEpoch = 0;
|
||||
Uint64 m_rpFastRbEpoch = 0;
|
||||
Uint64 m_rpFastRenderPassHash = 0;
|
||||
// Whether the memoized entry carries a depth/stencil attachment; a
|
||||
// default-FBO resolution whose effective depth request differs must
|
||||
// miss the memo (the depth-less/depth-full flavors hash differently).
|
||||
Bool m_rpFastHadDepthStencil = false;
|
||||
|
||||
public:
|
||||
struct RenderbufferResource {
|
||||
|
||||
@@ -607,7 +607,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
void VkTextureManager::Shutdown() {
|
||||
if (m_device != VK_NULL_HANDLE) {
|
||||
ReclaimCompletedUploads(/*waitAll=*/true);
|
||||
}
|
||||
DestroyDeferredReleases();
|
||||
++m_resourceEraseEpoch; // every memoized resource pointer dies with the map
|
||||
m_textureResources.clear();
|
||||
m_aliveObjects.clear();
|
||||
m_storageImageTextures.clear();
|
||||
@@ -629,6 +633,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frameIndex, m_deferredViewReleases.size());
|
||||
m_currentFrameIndex = frameIndex;
|
||||
CollectDeferredReleases(frameIndex);
|
||||
ReclaimCompletedUploads();
|
||||
|
||||
// Frame-boundary GC: every 64 frame boundaries (~1 s at 60 fps) bounds the reclaim
|
||||
// latency for dead textures regardless of draw traffic — workloads that churn
|
||||
@@ -659,6 +664,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
m_aliveObjects.erase(identity);
|
||||
m_storageImageTextures.erase(identity);
|
||||
// Invalidate every cross-draw sampled-texture memo: the erased
|
||||
// resource's address may be reused by a future emplace.
|
||||
++m_resourceEraseEpoch;
|
||||
}
|
||||
|
||||
void VkTextureManager::PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture) {
|
||||
@@ -714,6 +722,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
// Cross-draw memo probe (see SyncedTextureMemoEntry): skips both map
|
||||
// lookups and the (re)registration path for repeat-bound textures.
|
||||
TextureResource* resourcePtr = nullptr;
|
||||
for (Uint32 i = 0; i < kSyncedTextureMemoSize; ++i) {
|
||||
const SyncedTextureMemoEntry& memo = m_syncedTextureMemo[i];
|
||||
if (memo.texture == &texture && memo.lifetimeId == identity.lifetimeId &&
|
||||
memo.eraseEpoch == m_resourceEraseEpoch) {
|
||||
resourcePtr = memo.resource;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (resourcePtr == nullptr) {
|
||||
auto aliveIt = m_aliveObjects.find(identity);
|
||||
if (aliveIt != m_aliveObjects.end() && aliveIt->second.expired()) {
|
||||
EraseTrackedTexture(aliveIt->first);
|
||||
@@ -751,8 +772,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto [insertIt, _] = m_textureResources.emplace(identity, Move(initial));
|
||||
it = insertIt;
|
||||
}
|
||||
resourcePtr = &(it->second);
|
||||
m_syncedTextureMemo[m_syncedTextureMemoNext] =
|
||||
SyncedTextureMemoEntry{&texture, identity.lifetimeId, m_resourceEraseEpoch, resourcePtr};
|
||||
m_syncedTextureMemoNext = (m_syncedTextureMemoNext + 1) % kSyncedTextureMemoSize;
|
||||
}
|
||||
|
||||
if (!SyncTexture(texture, it->second)) {
|
||||
if (!SyncTexture(texture, *resourcePtr)) {
|
||||
MGLOG_D("%s: Syncing texture %d failed", __func__, texture.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
@@ -766,11 +792,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
if (!recorded) {
|
||||
m_drawSyncedThisDraw.push_back({identity, &(it->second)});
|
||||
m_drawSyncedThisDraw.push_back({identity, resourcePtr});
|
||||
}
|
||||
}
|
||||
|
||||
return &(it->second);
|
||||
return resourcePtr;
|
||||
}
|
||||
|
||||
VkImageView VkTextureManager::GetOrCreateViewAtMipLevel(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel) {
|
||||
@@ -1049,6 +1075,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return view;
|
||||
}
|
||||
|
||||
void VkTextureManager::StampTextureRecordingUse(MG_State::GLState::ITextureObject* texture) {
|
||||
if (texture == nullptr) {
|
||||
return;
|
||||
}
|
||||
auto it = m_textureResources.find(MakeTextureIdentity(texture));
|
||||
if (it != m_textureResources.end()) {
|
||||
it->second.lastRecordingGeneration = m_recordingGeneration;
|
||||
}
|
||||
}
|
||||
|
||||
void VkTextureManager::UpdateTrackedImageLayout(MG_State::GLState::ITextureObject* texture, VkImageLayout newLayout) {
|
||||
MOBILEGL_ASSERT(texture != nullptr, "UpdateTrackedImageLayout: texture is null");
|
||||
auto it = m_textureResources.find(MakeTextureIdentity(texture));
|
||||
@@ -1076,6 +1112,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(writtenMipLevel < resource.mipLevels,
|
||||
"UpdateTrackedImageLayoutAfterAttachmentWrite: textureId=%d mipLevel=%u out of range %u",
|
||||
texture->GetExternalIndex(), writtenMipLevel, resource.mipLevels);
|
||||
// Pre-pass stream bookkeeping: the render pass that just ended wrote this image.
|
||||
StampResourceRecordingUse(resource);
|
||||
|
||||
if (resource.layout != newLayout && resource.mipLevels > 1) {
|
||||
VkPipelineStageFlags srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
@@ -1160,6 +1198,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels,
|
||||
resource->arrayLayers);
|
||||
MOBILEGL_ASSERT(ok, "TransitionTextureForSampling: transition failed for textureId=%d", texture.GetExternalIndex());
|
||||
// Pre-pass stream bookkeeping: a command referencing the image was recorded.
|
||||
StampResourceRecordingUse(*resource);
|
||||
return ok;
|
||||
}
|
||||
|
||||
@@ -1189,6 +1229,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource->aspect, 0, resource->mipLevels, resource->arrayLayers);
|
||||
MOBILEGL_ASSERT(ok, "TransitionTextureForStorageImage: transition failed for textureId=%d",
|
||||
texture.GetExternalIndex());
|
||||
// Pre-pass stream bookkeeping: a command referencing the image was recorded.
|
||||
StampResourceRecordingUse(*resource);
|
||||
return ok;
|
||||
}
|
||||
|
||||
@@ -1420,8 +1462,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
const Bool isMultisampleTexture = IsMultisampleTextureUploadTarget(uploadTarget);
|
||||
// A texture that has only ever defined level 0 gets a single-level backing
|
||||
// (ANGLE's model). Preallocating the full chain put every render target
|
||||
// onto Adreno's multi-mip image layout and grew each texture by a third
|
||||
// for levels most textures never define. Once a second level is defined
|
||||
// the backing is recreated ONE time with the full chain (the
|
||||
// preserve-copy path below carries the pixels over), so sequentially-
|
||||
// defined atlas mips do not recreate per level, and glGenerateMipmap -
|
||||
// which defines every level before syncing - works unchanged.
|
||||
const Uint32 backingMipLevels =
|
||||
isMultisampleTexture ? 1u : std::max(mipLevels, ComputeFullMipLevelCount(texelSize));
|
||||
isMultisampleTexture ? 1u
|
||||
: (mipLevels > 1 ? std::max(mipLevels, ComputeFullMipLevelCount(texelSize)) : 1u);
|
||||
TextureShapeInfo shapeInfo{};
|
||||
const Bool supportedShape = TryResolveTextureShapeInfo(texture, uploadTarget, texelSize, shapeInfo);
|
||||
MOBILEGL_ASSERT(supportedShape,
|
||||
@@ -1683,6 +1734,28 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_deferredViewReleases[frameIndex].clear();
|
||||
}
|
||||
|
||||
void VkTextureManager::ReclaimCompletedUploads(Bool waitAll) {
|
||||
if (m_pendingUploadReclaims.empty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
SizeT completed = 0;
|
||||
for (; completed < m_pendingUploadReclaims.size(); ++completed) {
|
||||
PendingUploadReclaim& entry = m_pendingUploadReclaims[completed];
|
||||
if (waitAll) {
|
||||
VK_VERIFY(vkWaitForFences(m_device, 1, &entry.fence, VK_TRUE, UINT64_MAX),
|
||||
"vkWaitForFences(texture upload reclaim)");
|
||||
} else if (vkGetFenceStatus(m_device, entry.fence) != VK_SUCCESS) {
|
||||
break;
|
||||
}
|
||||
vkDestroyFence(m_device, entry.fence, nullptr);
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &entry.commandBuffer);
|
||||
vmaDestroyBuffer(m_allocator, entry.stagingBuffer, entry.stagingAllocation);
|
||||
}
|
||||
m_pendingUploadReclaims.erase(m_pendingUploadReclaims.begin(),
|
||||
m_pendingUploadReclaims.begin() + static_cast<std::ptrdiff_t>(completed));
|
||||
}
|
||||
|
||||
void VkTextureManager::DestroyDeferredReleases() {
|
||||
for (auto& deferredReleases : m_deferredReleases) {
|
||||
deferredReleases.clear();
|
||||
@@ -1989,11 +2062,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VK_VERIFY(vkCreateFence(m_device, &fenceInfo, nullptr, &uploadFence), "vkCreateFence(texture upload)");
|
||||
|
||||
VK_VERIFY(vkQueueSubmit(m_graphicsQueue, 1, &submitInfo, uploadFence), "vkQueueSubmit(texture)");
|
||||
VK_VERIFY(vkWaitForFences(m_device, 1, &uploadFence, VK_TRUE, UINT64_MAX), "vkWaitForFences(texture upload)");
|
||||
vkDestroyFence(m_device, uploadFence, nullptr);
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &commandBuffer);
|
||||
|
||||
vmaDestroyBuffer(m_allocator, stagingBuffer, stagingAllocation);
|
||||
// Do NOT wait the fence here: this submit sits behind the previous
|
||||
// frame's rendering on the queue, so a synchronous wait stalls the CPU
|
||||
// until the GPU drains - a per-frame vkQueueWaitIdle for any workload
|
||||
// with animated textures. Ordering against the current frame's draws is
|
||||
// already guaranteed (its command buffer is submitted later, at
|
||||
// present), so only the transient objects need to survive execution;
|
||||
// park them until the fence signals.
|
||||
m_pendingUploadReclaims.push_back({uploadFence, commandBuffer, stagingBuffer, stagingAllocation});
|
||||
ReclaimCompletedUploads();
|
||||
// Backstop for pathological upload storms: bound in-flight staging
|
||||
// memory by blocking on the oldest upload only once the list is deep.
|
||||
constexpr SizeT kMaxPendingTextureUploads = 16;
|
||||
if (m_pendingUploadReclaims.size() > kMaxPendingTextureUploads) {
|
||||
VK_VERIFY(vkWaitForFences(m_device, 1, &m_pendingUploadReclaims.front().fence, VK_TRUE, UINT64_MAX),
|
||||
"vkWaitForFences(texture upload backstop)");
|
||||
ReclaimCompletedUploads();
|
||||
}
|
||||
|
||||
if (!ok) {
|
||||
MGLOG_D("%s: texture upload cmd failed", __func__);
|
||||
|
||||
@@ -28,6 +28,9 @@ public:
|
||||
// manager keys its per-draw fast path on this so an attachment's image recreation
|
||||
// invalidates the cached render pass (dirty-flag tracking; portable to Vulkan 1.1).
|
||||
Uint64 GetTextureImageEpoch() const { return m_textureImageEpoch; }
|
||||
// Bumped whenever any tracked texture resource is erased; cached
|
||||
// TextureResource pointers are valid only while this is unchanged.
|
||||
Uint64 GetResourceEraseEpoch() const { return m_resourceEraseEpoch; }
|
||||
|
||||
struct TextureIdentity {
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
@@ -172,6 +175,13 @@ public:
|
||||
// NeedsStorageImagePreparation cannot ask for a recreate that will never happen.
|
||||
Bool storageUsageResolved = false;
|
||||
Uint16 syncedTextureParamsVersion = 0;
|
||||
// Recording generation (VkTextureManager::GetRecordingGeneration) of the last
|
||||
// command referencing this image that was recorded into the CURRENT frame
|
||||
// command buffer. An image untouched by the open recording may have its
|
||||
// out-of-pass work (deferred clears, sampled-layout transitions) recorded
|
||||
// into the frame's PRE command buffer - which executes strictly before the
|
||||
// frame's commands - instead of splitting the active render pass.
|
||||
Uint64 lastRecordingGeneration = 0;
|
||||
// Snapshot of ITextureObject::GetContentVersion() at the last successful sync;
|
||||
// lets SyncTexture skip the whole re-check/re-upload when content is unchanged.
|
||||
Uint64 syncedContentVersion = 0;
|
||||
@@ -207,6 +217,7 @@ public:
|
||||
std::swap(this->usageFlags, that.usageFlags);
|
||||
std::swap(this->storageUsageResolved, that.storageUsageResolved);
|
||||
std::swap(this->syncedTextureParamsVersion, that.syncedTextureParamsVersion);
|
||||
std::swap(this->lastRecordingGeneration, that.lastRecordingGeneration);
|
||||
std::swap(this->syncedContentVersion, that.syncedContentVersion);
|
||||
std::swap(this->syncedMipLevelCount, that.syncedMipLevelCount);
|
||||
}
|
||||
@@ -307,6 +318,21 @@ public:
|
||||
VkImageLayout newLayout);
|
||||
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
|
||||
// Recording-generation bookkeeping for the pre-pass command stream. The
|
||||
// generation advances every time the frame command buffer (re)begins
|
||||
// recording; a resource whose stamp does not match was not referenced by
|
||||
// any command in the open recording, so its out-of-pass work may safely
|
||||
// execute ahead of the whole recording (in the pre command buffer).
|
||||
void AdvanceRecordingGeneration() { ++m_recordingGeneration; }
|
||||
void StampResourceRecordingUse(TextureResource& resource) const {
|
||||
resource.lastRecordingGeneration = m_recordingGeneration;
|
||||
}
|
||||
// Map-lookup variant for callers that only hold the GL texture object.
|
||||
void StampTextureRecordingUse(MG_State::GLState::ITextureObject* texture);
|
||||
Bool WasTouchedThisRecording(const TextureResource& resource) const {
|
||||
return resource.lastRecordingGeneration == m_recordingGeneration;
|
||||
}
|
||||
// Records that this texture is bound to a GL image unit, so its image must carry
|
||||
// VK_IMAGE_USAGE_STORAGE_BIT. Must be called before NeedsStorageImagePreparation, and
|
||||
// therefore before the render pass is committed: an image that has to be upgraded is
|
||||
@@ -364,6 +390,9 @@ public:
|
||||
private:
|
||||
// Bumped in SyncTextureResource right after vmaCreateImage(texture). See GetTextureImageEpoch().
|
||||
Uint64 m_textureImageEpoch = 1;
|
||||
// See AdvanceRecordingGeneration. Starts above every resource's default
|
||||
// stamp of 0 so a fresh resource counts as untouched.
|
||||
Uint64 m_recordingGeneration = 1;
|
||||
|
||||
Bool SyncTexture(MG_State::GLState::ITextureObject &texture,
|
||||
TextureResource &outResource);
|
||||
@@ -394,6 +423,11 @@ private:
|
||||
void DeferViewRelease(VkImageView view);
|
||||
void CollectDeferredReleases(Uint32 frameIndex);
|
||||
void DestroyDeferredReleases();
|
||||
// Frees the fence/command buffer/staging buffer of every in-flight texture
|
||||
// upload whose fence has signaled (submission order = completion order on
|
||||
// the single queue, so the scan stops at the first still-pending entry).
|
||||
// waitAll blocks on every entry - Shutdown's drain.
|
||||
void ReclaimCompletedUploads(Bool waitAll = false);
|
||||
static TextureIdentity MakeTextureIdentity(MG_State::GLState::ITextureObject* texture);
|
||||
void EraseTrackedTexture(const TextureIdentity& identity);
|
||||
void PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture);
|
||||
@@ -423,6 +457,23 @@ private:
|
||||
TextureResource* resource = nullptr;
|
||||
};
|
||||
Vector<DrawSyncedTexture> m_drawSyncedThisDraw;
|
||||
// Cross-draw sampled-texture memo: the same few textures (atlas, lightmap)
|
||||
// are resolved on every draw, so cache their resource pointers and skip the
|
||||
// alive/resource map lookups. Node-based std::unordered_map keeps the
|
||||
// pointees stable across inserts; erases bump m_resourceEraseEpoch, which
|
||||
// every memo entry must match. SyncTexture still runs on memo hits, so
|
||||
// content/param freshness is unaffected. A dead-then-reused texture address
|
||||
// cannot false-hit: the new object carries a new lifetime id.
|
||||
struct SyncedTextureMemoEntry {
|
||||
const MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
Uint64 lifetimeId = 0;
|
||||
Uint64 eraseEpoch = 0;
|
||||
TextureResource* resource = nullptr;
|
||||
};
|
||||
static constexpr Uint32 kSyncedTextureMemoSize = 8;
|
||||
SyncedTextureMemoEntry m_syncedTextureMemo[kSyncedTextureMemoSize];
|
||||
Uint32 m_syncedTextureMemoNext = 0;
|
||||
Uint64 m_resourceEraseEpoch = 1;
|
||||
// Formats whose mutable-image probe failed on this device; their images are created
|
||||
// without MUTABLE_FORMAT_BIT so repeat syncs neither re-probe nor flag-mismatch.
|
||||
std::unordered_set<VkFormat> m_mutableFormatUnsupported;
|
||||
@@ -432,5 +483,16 @@ private:
|
||||
std::unordered_set<TextureIdentity, TextureIdentityHash> m_storageImageTextures;
|
||||
Vector<Vector<TextureResource>> m_deferredReleases;
|
||||
Vector<Vector<VkImageView>> m_deferredViewReleases;
|
||||
// Texture uploads are submitted out-of-band but NOT waited on (waiting
|
||||
// behind the queue serialized the CPU against the previous frame's GPU
|
||||
// work every time an animated atlas re-uploaded). Their transient objects
|
||||
// are parked here and reclaimed once the upload fence signals.
|
||||
struct PendingUploadReclaim {
|
||||
VkFence fence = VK_NULL_HANDLE;
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
VkBuffer stagingBuffer = VK_NULL_HANDLE;
|
||||
VmaAllocation stagingAllocation = nullptr;
|
||||
};
|
||||
Vector<PendingUploadReclaim> m_pendingUploadReclaims;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -217,6 +217,73 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return static_cast<Int>((static_cast<Int64>(value) * toExtent + fromExtent / 2) / fromExtent);
|
||||
}
|
||||
|
||||
// Redundant dynamic-state elimination for the per-draw hot path: within one
|
||||
// command-buffer recording, a vkCmdSet* whose values already match what the
|
||||
// command buffer holds is skipped. Valid because every PipelineFactory
|
||||
// pipeline declares the same eight dynamic states, so the values persist
|
||||
// across those pipeline binds; the shadow resets whenever a recording
|
||||
// (re)begins, and whenever an auxiliary pipeline with a narrower dynamic
|
||||
// set (blit, depth-mipmap) binds - their static state makes the
|
||||
// corresponding dynamic values undefined per the spec.
|
||||
struct DynamicStateShadow {
|
||||
// Last graphics pipeline bound on the frame command buffer. Pipeline
|
||||
// binds are command-buffer state (they survive render-pass boundaries),
|
||||
// so the same reset points that invalidate dynamic state - recording
|
||||
// (re)begin and the aux blit pipelines' raw binds - are exactly the
|
||||
// points where this becomes unknown.
|
||||
Bool graphicsPipelineValid = false;
|
||||
VkPipeline graphicsPipeline = VK_NULL_HANDLE;
|
||||
// Index/vertex buffer binds are command-buffer state too. Terrain
|
||||
// sections and GUI quads share one sequential index buffer, and GUI
|
||||
// batches often reuse a vertex arena buffer, so skipping identical
|
||||
// rebinds removes a large share of per-draw driver calls.
|
||||
Bool indexBindValid = false;
|
||||
VkBuffer indexBuffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize indexOffset = 0;
|
||||
VkIndexType indexType = VK_INDEX_TYPE_MAX_ENUM;
|
||||
static constexpr Uint32 kMaxShadowedVertexBindings = 8;
|
||||
Bool vertexBindValid = false;
|
||||
Uint32 vertexBindingCount = 0;
|
||||
VkBuffer vertexBuffers[kMaxShadowedVertexBindings] = {};
|
||||
VkDeviceSize vertexOffsets[kMaxShadowedVertexBindings] = {};
|
||||
Bool viewportValid = false;
|
||||
VkViewport viewport{};
|
||||
Bool scissorValid = false;
|
||||
VkRect2D scissor{};
|
||||
Bool blendConstantsValid = false;
|
||||
Float blendConstants[4] = {0.0f, 0.0f, 0.0f, 0.0f};
|
||||
Bool depthBiasValid = false;
|
||||
Float depthBiasConstantFactor = 0.0f;
|
||||
Float depthBiasSlopeFactor = 0.0f;
|
||||
Bool lineWidthValid = false;
|
||||
Float lineWidth = 0.0f;
|
||||
Bool stencilValid = false;
|
||||
Uint32 stencilFrontCompareMask = 0;
|
||||
Uint32 stencilBackCompareMask = 0;
|
||||
Uint32 stencilFrontWriteMask = 0;
|
||||
Uint32 stencilBackWriteMask = 0;
|
||||
Uint32 stencilFrontReference = 0;
|
||||
Uint32 stencilBackReference = 0;
|
||||
};
|
||||
static DynamicStateShadow g_dynamicStateShadow;
|
||||
|
||||
static void ResetDynamicStateShadow() {
|
||||
g_dynamicStateShadow = {};
|
||||
}
|
||||
|
||||
static void ShadowedSetScissor(VkCommandBuffer commandBuffer, const VkRect2D& scissor) {
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.scissorValid && shadow.scissor.offset.x == scissor.offset.x &&
|
||||
shadow.scissor.offset.y == scissor.offset.y &&
|
||||
shadow.scissor.extent.width == scissor.extent.width &&
|
||||
shadow.scissor.extent.height == scissor.extent.height) {
|
||||
return;
|
||||
}
|
||||
shadow.scissorValid = true;
|
||||
shadow.scissor = scissor;
|
||||
vkCmdSetScissor(commandBuffer, 0, 1, &scissor);
|
||||
}
|
||||
|
||||
static void ApplyGLViewportState(VkCommandBuffer commandBuffer,
|
||||
const IntVec2& framebufferExtent,
|
||||
VkSurfaceTransformFlagBitsKHR preTransform,
|
||||
@@ -246,6 +313,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
viewport.height = static_cast<float>(viewportHeight);
|
||||
viewport.minDepth = depthRange.x();
|
||||
viewport.maxDepth = depthRange.y();
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.viewportValid && shadow.viewport.x == viewport.x && shadow.viewport.y == viewport.y &&
|
||||
shadow.viewport.width == viewport.width && shadow.viewport.height == viewport.height &&
|
||||
shadow.viewport.minDepth == viewport.minDepth && shadow.viewport.maxDepth == viewport.maxDepth) {
|
||||
return;
|
||||
}
|
||||
shadow.viewportValid = true;
|
||||
shadow.viewport = viewport;
|
||||
vkCmdSetViewport(commandBuffer, 0, 1, &viewport);
|
||||
}
|
||||
|
||||
@@ -257,6 +332,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
blendColor.z(),
|
||||
blendColor.w(),
|
||||
};
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.blendConstantsValid && shadow.blendConstants[0] == blendConstants[0] &&
|
||||
shadow.blendConstants[1] == blendConstants[1] && shadow.blendConstants[2] == blendConstants[2] &&
|
||||
shadow.blendConstants[3] == blendConstants[3]) {
|
||||
return;
|
||||
}
|
||||
shadow.blendConstantsValid = true;
|
||||
shadow.blendConstants[0] = blendConstants[0];
|
||||
shadow.blendConstants[1] = blendConstants[1];
|
||||
shadow.blendConstants[2] = blendConstants[2];
|
||||
shadow.blendConstants[3] = blendConstants[3];
|
||||
vkCmdSetBlendConstants(commandBuffer, blendConstants);
|
||||
}
|
||||
|
||||
@@ -272,8 +358,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
static void ApplyPolygonOffsetState(VkCommandBuffer commandBuffer) {
|
||||
vkCmdSetDepthBias(commandBuffer, MG_State::pGLContext->GetPolygonOffsetUnits(), 0.0f,
|
||||
MG_State::pGLContext->GetPolygonOffsetFactor());
|
||||
const Float constantFactor = MG_State::pGLContext->GetPolygonOffsetUnits();
|
||||
const Float slopeFactor = MG_State::pGLContext->GetPolygonOffsetFactor();
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.depthBiasValid && shadow.depthBiasConstantFactor == constantFactor &&
|
||||
shadow.depthBiasSlopeFactor == slopeFactor) {
|
||||
return;
|
||||
}
|
||||
shadow.depthBiasValid = true;
|
||||
shadow.depthBiasConstantFactor = constantFactor;
|
||||
shadow.depthBiasSlopeFactor = slopeFactor;
|
||||
vkCmdSetDepthBias(commandBuffer, constantFactor, 0.0f, slopeFactor);
|
||||
}
|
||||
|
||||
static void ApplyLineWidthState(VkCommandBuffer commandBuffer) {
|
||||
@@ -288,6 +383,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
lineWidth = maxLineWidth;
|
||||
}
|
||||
}
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.lineWidthValid && shadow.lineWidth == lineWidth) {
|
||||
return;
|
||||
}
|
||||
shadow.lineWidthValid = true;
|
||||
shadow.lineWidth = lineWidth;
|
||||
vkCmdSetLineWidth(commandBuffer, lineWidth);
|
||||
}
|
||||
|
||||
@@ -336,15 +437,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static void ApplyStencilState(VkCommandBuffer commandBuffer) {
|
||||
const StencilFaceState& frontStencil = MG_State::pGLContext->GetStencilState(StencilFace::Front);
|
||||
const StencilFaceState& backStencil = MG_State::pGLContext->GetStencilState(StencilFace::Back);
|
||||
const Uint32 frontReference = static_cast<Uint32>(std::max(frontStencil.Ref, 0));
|
||||
const Uint32 backReference = static_cast<Uint32>(std::max(backStencil.Ref, 0));
|
||||
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.stencilValid && shadow.stencilFrontCompareMask == frontStencil.ValueMask &&
|
||||
shadow.stencilBackCompareMask == backStencil.ValueMask &&
|
||||
shadow.stencilFrontWriteMask == frontStencil.WriteMask &&
|
||||
shadow.stencilBackWriteMask == backStencil.WriteMask &&
|
||||
shadow.stencilFrontReference == frontReference && shadow.stencilBackReference == backReference) {
|
||||
return;
|
||||
}
|
||||
shadow.stencilValid = true;
|
||||
shadow.stencilFrontCompareMask = frontStencil.ValueMask;
|
||||
shadow.stencilBackCompareMask = backStencil.ValueMask;
|
||||
shadow.stencilFrontWriteMask = frontStencil.WriteMask;
|
||||
shadow.stencilBackWriteMask = backStencil.WriteMask;
|
||||
shadow.stencilFrontReference = frontReference;
|
||||
shadow.stencilBackReference = backReference;
|
||||
|
||||
vkCmdSetStencilCompareMask(commandBuffer, VK_STENCIL_FACE_FRONT_BIT, frontStencil.ValueMask);
|
||||
vkCmdSetStencilCompareMask(commandBuffer, VK_STENCIL_FACE_BACK_BIT, backStencil.ValueMask);
|
||||
vkCmdSetStencilWriteMask(commandBuffer, VK_STENCIL_FACE_FRONT_BIT, frontStencil.WriteMask);
|
||||
vkCmdSetStencilWriteMask(commandBuffer, VK_STENCIL_FACE_BACK_BIT, backStencil.WriteMask);
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_FRONT_BIT,
|
||||
static_cast<Uint32>(std::max(frontStencil.Ref, 0)));
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_BACK_BIT,
|
||||
static_cast<Uint32>(std::max(backStencil.Ref, 0)));
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_FRONT_BIT, frontReference);
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_BACK_BIT, backReference);
|
||||
}
|
||||
|
||||
enum class NumericDomain {
|
||||
@@ -718,16 +835,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static_assert(kMaxVertexAttribs <= ProgramFactory::VkProgramObject::kMaxVertexInputLocations,
|
||||
"vertexInputTypes is indexed by vertex attribute location");
|
||||
|
||||
static Uint32 BuildVertexInputAttributeMask(const Vector<VkVertexInputAttributeDescription>& attributes) {
|
||||
Uint32 attributeMask = 0;
|
||||
for (const auto& attribute : attributes) {
|
||||
if (attribute.location < kMaxVertexAttribs) {
|
||||
attributeMask |= (1u << attribute.location);
|
||||
}
|
||||
}
|
||||
return attributeMask;
|
||||
}
|
||||
|
||||
static Bool TryGetCurrentVertexAttributeFormat(GLenum glType, VkFormat& outFormat) {
|
||||
switch (glType) {
|
||||
case GL_FLOAT:
|
||||
@@ -878,8 +985,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (trackedAttachment.target != TrackedAttachmentTarget::Texture) {
|
||||
continue;
|
||||
}
|
||||
const auto trackedTexture = trackedAttachment.texture.lock();
|
||||
if (trackedTexture && trackedTexture.get() == &texture) {
|
||||
// Raw identity compare (see textureRaw): the caller's texture is
|
||||
// live, so a dangling tracked pointer can never equal its address
|
||||
// unless the allocator reused it - and that false positive merely
|
||||
// ends the render pass early, never misses a genuine use.
|
||||
if (trackedAttachment.textureRaw == &texture) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -2814,7 +2924,7 @@ void main() {
|
||||
// the GetCurrentProgram + GetOrCreateProgram hash lookup every draw.
|
||||
auto& vertexInputState = m_vertexInputStateFactory->GetOrCreateVertexInputState(vao);
|
||||
const Uint32 activeAttribMask = programObj.activeVertexInputLocationMask;
|
||||
const Uint32 vertexInputAttribMask = BuildVertexInputAttributeMask(vertexInputState.attributes);
|
||||
const Uint32 vertexInputAttribMask = vertexInputState.attributeLocationMask;
|
||||
const Uint32 missingAttribMask = activeAttribMask & ~vertexInputAttribMask;
|
||||
|
||||
const auto bindingCount = vertexInputState.bindings.size() + static_cast<SizeT>(std::popcount(missingAttribMask));
|
||||
@@ -3079,8 +3189,29 @@ void main() {
|
||||
}
|
||||
|
||||
if (bindingCount > 0) {
|
||||
vkCmdBindVertexBuffers(commandBuffer, 0, static_cast<Uint32>(bindingCount), vkBuffers.data(),
|
||||
vkOffsets.data());
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
const Uint32 count = static_cast<Uint32>(bindingCount);
|
||||
Bool identical = shadow.vertexBindValid && shadow.vertexBindingCount == count &&
|
||||
count <= DynamicStateShadow::kMaxShadowedVertexBindings;
|
||||
if (identical) {
|
||||
for (Uint32 i = 0; i < count; ++i) {
|
||||
if (shadow.vertexBuffers[i] != vkBuffers[i] || shadow.vertexOffsets[i] != vkOffsets[i]) {
|
||||
identical = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!identical) {
|
||||
vkCmdBindVertexBuffers(commandBuffer, 0, count, vkBuffers.data(), vkOffsets.data());
|
||||
if (count <= DynamicStateShadow::kMaxShadowedVertexBindings) {
|
||||
shadow.vertexBindValid = true;
|
||||
shadow.vertexBindingCount = count;
|
||||
std::copy_n(vkBuffers.data(), count, shadow.vertexBuffers);
|
||||
std::copy_n(vkOffsets.data(), count, shadow.vertexOffsets);
|
||||
} else {
|
||||
shadow.vertexBindValid = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -3150,8 +3281,17 @@ void main() {
|
||||
MGLOG_E("DrawElements skipped: failed to sync resident index buffer");
|
||||
return false;
|
||||
}
|
||||
vkCmdBindIndexBuffer(frame.commandBuffer, slice.buffer,
|
||||
slice.offset + static_cast<VkDeviceSize>(pIndexBufferView->indexByteOffset), vkIndexType);
|
||||
const VkDeviceSize indexBindOffset =
|
||||
slice.offset + static_cast<VkDeviceSize>(pIndexBufferView->indexByteOffset);
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (!shadow.indexBindValid || shadow.indexBuffer != slice.buffer ||
|
||||
shadow.indexOffset != indexBindOffset || shadow.indexType != vkIndexType) {
|
||||
vkCmdBindIndexBuffer(frame.commandBuffer, slice.buffer, indexBindOffset, vkIndexType);
|
||||
shadow.indexBindValid = true;
|
||||
shadow.indexBuffer = slice.buffer;
|
||||
shadow.indexOffset = indexBindOffset;
|
||||
shadow.indexType = vkIndexType;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -3653,6 +3793,10 @@ void main() {
|
||||
vkCmdSetScissor(frame.commandBuffer, 0, 1, &scissor);
|
||||
|
||||
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
|
||||
// The depth-mipmap pipeline's narrower dynamic set (viewport/scissor
|
||||
// only) leaves the other dynamic states undefined; its raw scissor
|
||||
// and viewport writes also bypass the shadow.
|
||||
ResetDynamicStateShadow();
|
||||
|
||||
std::fill(depthProgramData,
|
||||
depthProgramData + m_depthMipmapResources.program->GetUBOSize(),
|
||||
@@ -3714,16 +3858,24 @@ void main() {
|
||||
// content hash (folds program identity + link version + transform flags + shader stages),
|
||||
// vertex-input hash (VAO layout), render-pass hash (render targets + the draw-buffer/format
|
||||
// driven blend & write-mask gating), and the render-state version (all fixed-function state).
|
||||
// Reset per-frame and on pipeline destruction so m_lastPipelineResult can never dangle.
|
||||
const Uint64 vertexInputHash = m_vertexInputStateFactory->GetOrComputeHash(vao);
|
||||
// Reset per-frame and on pipeline destruction so a memoized handle can never dangle.
|
||||
// The identity hash mixes buffer heap addresses (per-chunk VBOs mint a new
|
||||
// one per buffer); the memo and the pipeline payload key on the resolved
|
||||
// LAYOUT hash instead, so draws over identical layouts share one pipeline.
|
||||
// The one-arg fetch rides the VAO's state-pointer memo (no hash, no map).
|
||||
auto& vis = m_vertexInputStateFactory->GetOrCreateVertexInputState(vao);
|
||||
const Uint64 vertexLayoutHash = vis.layoutHash;
|
||||
const Uint64 renderPassHash = renderPassEntry.hash;
|
||||
const Uint renderStateVersion = MG_State::pGLContext->GetRenderStateParametersVersion();
|
||||
if (m_lastPipelineValid && m_lastPipelineResult != VK_NULL_HANDLE && m_lastPipelineMode == mode &&
|
||||
m_lastPipelineProgramHash == programObj.hash && m_lastPipelineVertexInputHash == vertexInputHash &&
|
||||
m_lastPipelineRenderPassHash == renderPassHash &&
|
||||
m_lastPipelineRenderStateVersion == renderStateVersion &&
|
||||
m_lastPipelineTransformFlags == transformFlags) {
|
||||
return m_lastPipelineResult;
|
||||
for (Uint32 i = 0; i < m_pipelineMemoCount; ++i) {
|
||||
const PipelineMemoEntry& entry = m_pipelineMemo[i];
|
||||
if (entry.pipeline != VK_NULL_HANDLE && entry.mode == mode &&
|
||||
entry.programHash == programObj.hash && entry.vertexInputHash == vertexLayoutHash &&
|
||||
entry.renderPassHash == renderPassHash &&
|
||||
entry.renderStateVersion == renderStateVersion &&
|
||||
entry.transformFlags == transformFlags) {
|
||||
return entry.pipeline;
|
||||
}
|
||||
}
|
||||
|
||||
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG
|
||||
@@ -3764,9 +3916,7 @@ void main() {
|
||||
}
|
||||
#endif
|
||||
|
||||
// vertexInputHash was computed above for the fast-path key; reuse it here.
|
||||
auto& vis = m_vertexInputStateFactory->GetOrCreateVertexInputState(vao, vertexInputHash);
|
||||
const Uint32 vertexInputAttribMask = BuildVertexInputAttributeMask(vis.attributes);
|
||||
const Uint32 vertexInputAttribMask = vis.attributeLocationMask;
|
||||
const Uint32 activeAttribMask = programObj.activeVertexInputLocationMask;
|
||||
const Uint32 missingAttribMask = activeAttribMask & ~vertexInputAttribMask;
|
||||
auto& patchedAttributes = m_patchedAttributesScratch;
|
||||
@@ -3876,7 +4026,7 @@ void main() {
|
||||
|
||||
PipelineFactory::PipelineCreatePayload payload {
|
||||
.programHash = programObj.hash,
|
||||
.vertexInputHash = vertexInputHash,
|
||||
.vertexInputHash = vertexLayoutHash,
|
||||
.pipelineLayout = programObj.pipelineLayout,
|
||||
.renderPass = renderPassEntry.renderPass,
|
||||
.colorAttachmentCount = renderPassEntry.colorAttachmentCount,
|
||||
@@ -3940,12 +4090,15 @@ void main() {
|
||||
payload.backStencilCompareOp = VK_COMPARE_OP_ALWAYS;
|
||||
}
|
||||
const Uint32 fragmentOutputMask = programObj.activeFragmentOutputLocationMask;
|
||||
MOBILEGL_ASSERT(
|
||||
(fragmentOutputMask >> payload.colorAttachmentCount) == 0,
|
||||
"GetOrCreatePipeline: fragmentOutputMask=0x%x exceeds colorAttachmentCount=%u for program=%u",
|
||||
fragmentOutputMask,
|
||||
payload.colorAttachmentCount,
|
||||
program.GetExternalIndex());
|
||||
// Outputs at locations past the render pass's trimmed colour span are
|
||||
// simply discarded - GL's semantic for a fragment output whose draw
|
||||
// buffer is GL_NONE (the trailing UNUSED slots no longer occupy
|
||||
// references, see GetOrCreateRenderPass).
|
||||
if ((fragmentOutputMask >> payload.colorAttachmentCount) != 0) {
|
||||
MGLOG_D("GetOrCreatePipeline: fragmentOutputMask=0x%x exceeds colorAttachmentCount=%u for program=%u; "
|
||||
"outputs past the span are discarded",
|
||||
fragmentOutputMask, payload.colorAttachmentCount, program.GetExternalIndex());
|
||||
}
|
||||
MOBILEGL_ASSERT(payload.colorAttachmentCount <= PipelineFactory::PipelineCreatePayload::kMaxColorAttachments,
|
||||
"GetOrCreatePipeline: colorAttachmentCount=%u exceeds payload capacity",
|
||||
payload.colorAttachmentCount);
|
||||
@@ -4180,14 +4333,16 @@ void main() {
|
||||
}
|
||||
VkPipeline pipeline = m_pipelineFactory->GetOrCreatePipeline(payload);
|
||||
if (pipeline != VK_NULL_HANDLE) {
|
||||
m_lastPipelineValid = true;
|
||||
m_lastPipelineMode = mode;
|
||||
m_lastPipelineProgramHash = programObj.hash;
|
||||
m_lastPipelineVertexInputHash = vertexInputHash;
|
||||
m_lastPipelineRenderPassHash = renderPassHash;
|
||||
m_lastPipelineRenderStateVersion = renderStateVersion;
|
||||
m_lastPipelineTransformFlags = transformFlags;
|
||||
m_lastPipelineResult = pipeline;
|
||||
PipelineMemoEntry& entry = m_pipelineMemo[m_pipelineMemoNext];
|
||||
entry.mode = mode;
|
||||
entry.programHash = programObj.hash;
|
||||
entry.vertexInputHash = vertexLayoutHash;
|
||||
entry.renderPassHash = renderPassHash;
|
||||
entry.renderStateVersion = renderStateVersion;
|
||||
entry.transformFlags = transformFlags;
|
||||
entry.pipeline = pipeline;
|
||||
m_pipelineMemoNext = (m_pipelineMemoNext + 1) % kPipelineMemoSize;
|
||||
m_pipelineMemoCount = std::min(m_pipelineMemoCount + 1, kPipelineMemoSize);
|
||||
}
|
||||
return pipeline;
|
||||
}
|
||||
@@ -4289,6 +4444,129 @@ void main() {
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
Bool VulkanRenderer::TrySetupDrawFastPath(FrameContext::FrameData& frame, GLenum mode,
|
||||
Flags<DrawSetupAspect> aspects, const DrawCmdParam& drawParams,
|
||||
const IndexBufferView* pIndexBufferView) {
|
||||
const SetupDrawSnapshot& snap = m_setupDrawSnapshot;
|
||||
if (!snap.valid || !frame.isCommandRecording) {
|
||||
return false;
|
||||
}
|
||||
if (snap.aspects != aspects.GetRaw() || snap.mode != mode) {
|
||||
return false;
|
||||
}
|
||||
if (m_clearManager->HasAnyPendingClears()) {
|
||||
return false;
|
||||
}
|
||||
const auto* activeRenderPass = VkRenderPassManager::GetActiveRenderPass();
|
||||
if (activeRenderPass == nullptr || activeRenderPass->hash != snap.renderPassHash ||
|
||||
snap.imageIndex != m_imageIndexAcquired) {
|
||||
return false;
|
||||
}
|
||||
const auto& program = *MG_State::pGLContext->GetCurrentProgram();
|
||||
if (program.GetLifetimeId() != snap.programLifetimeId ||
|
||||
program.GetBackendStateVersion() != snap.programVersion) {
|
||||
return false;
|
||||
}
|
||||
const auto& vao = *MG_State::pGLContext->GetBoundVertexArray();
|
||||
if (static_cast<const void*>(&vao) != snap.vao || vao.GetConfigVersion() != snap.vaoConfigVersion) {
|
||||
return false;
|
||||
}
|
||||
const auto& drawFbo =
|
||||
MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||
if (static_cast<const void*>(drawFbo.get()) != snap.drawFbo ||
|
||||
drawFbo->GetObjectVersion() != snap.fboVersion) {
|
||||
return false;
|
||||
}
|
||||
if (MG_State::pGLContext->GetRenderStateParametersVersion() != snap.renderStateVersion ||
|
||||
MG_State::pGLContext->GetTextureBindGeneration() != snap.bindGeneration) {
|
||||
return false;
|
||||
}
|
||||
if (GetShaderTransformFlags(m_swapchainObject.GetPreTransform()).GetRaw() != snap.baseTransformFlags) {
|
||||
return false;
|
||||
}
|
||||
if (m_textureManager->GetResourceEraseEpoch() != snap.textureEraseEpoch ||
|
||||
m_textureManager->GetTextureImageEpoch() != snap.textureImageEpoch ||
|
||||
m_renderPassManager->GetRenderbufferImageEpoch() != snap.renderbufferImageEpoch) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Same sampled set as the snapshotting draw (program/bind keys above);
|
||||
// verify content and params are untouched and every layout is still
|
||||
// sampleable, then stamp recording use exactly as the full path would.
|
||||
// A feedback case (sampled texture written by the active pass) fails the
|
||||
// layout check and falls back to the full path's end-pass handling.
|
||||
const auto& sampledTextures = m_sampledTexturesScratch;
|
||||
const auto& sampledResources = m_sampledResourcesScratch;
|
||||
if (sampledResources.size() != sampledTextures.size()) {
|
||||
return false;
|
||||
}
|
||||
Uint64 contentSum = 0;
|
||||
Uint64 paramsSum = 0;
|
||||
for (SizeT i = 0; i < sampledTextures.size(); ++i) {
|
||||
const auto* sampledTexture = sampledTextures[i];
|
||||
if (sampledTexture == nullptr) {
|
||||
continue;
|
||||
}
|
||||
const auto* resource = sampledResources[i];
|
||||
if (resource == nullptr || !IsValidSampledImageLayout(resource->layout)) {
|
||||
return false;
|
||||
}
|
||||
contentSum += sampledTexture->GetContentVersion();
|
||||
paramsSum += sampledTexture->GetTextureParamsVersion();
|
||||
}
|
||||
if (contentSum != snap.sampledContentSum || paramsSum != snap.sampledParamsSum) {
|
||||
return false;
|
||||
}
|
||||
for (SizeT i = 0; i < sampledTextures.size(); ++i) {
|
||||
if (sampledTextures[i] != nullptr && sampledResources[i] != nullptr) {
|
||||
m_textureManager->StampResourceRecordingUse(*sampledResources[i]);
|
||||
}
|
||||
}
|
||||
|
||||
// Everything the full path would re-resolve is provably unchanged; run
|
||||
// only the per-draw tail.
|
||||
if (!g_dynamicStateShadow.graphicsPipelineValid ||
|
||||
g_dynamicStateShadow.graphicsPipeline != snap.pipeline) {
|
||||
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, snap.pipeline);
|
||||
g_dynamicStateShadow.graphicsPipelineValid = true;
|
||||
g_dynamicStateShadow.graphicsPipeline = snap.pipeline;
|
||||
}
|
||||
const auto& programObj = m_programFactory->GetOrCreateProgram(
|
||||
program, ProgramFactory::CompileOptionFlags(snap.resolvedTransformFlags));
|
||||
if (!m_uniformManager->BindProgramUniformBuffers(frame.commandBuffer, program, programObj,
|
||||
m_frameContext.GetCurrentFrameIndex())) {
|
||||
return false;
|
||||
}
|
||||
if (!UploadAndBindVertexBuffers(frame.commandBuffer, vao, programObj, drawParams, pIndexBufferView)) {
|
||||
return false;
|
||||
}
|
||||
if (aspects & DrawSetupAspect::IndexBuffer) {
|
||||
const Bool idxUploadOk = UploadAndBindIndexBuffer(frame, vao, pIndexBufferView);
|
||||
MOBILEGL_ASSERT(idxUploadOk, "SetupDraw fast path: failed to upload index buffer");
|
||||
}
|
||||
ApplyGLViewportState(frame.commandBuffer, snap.renderPassExtent, m_swapchainObject.GetPreTransform(),
|
||||
snap.drawFboIsDefault);
|
||||
ApplyBlendConstants(frame.commandBuffer);
|
||||
ApplyPolygonOffsetState(frame.commandBuffer);
|
||||
ApplyLineWidthState(frame.commandBuffer);
|
||||
ApplyStencilState(frame.commandBuffer);
|
||||
const Bool scissorEnabled = MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::ScissorTest);
|
||||
VkRect2D scissor{};
|
||||
if (scissorEnabled) {
|
||||
const auto& scissorBox = MG_State::pGLContext->GetScissorBox();
|
||||
scissor = snap.drawFboIsDefault
|
||||
? MakeDefaultFramebufferScissorRect(scissorBox, snap.renderPassExtent,
|
||||
m_swapchainObject.GetPreTransform())
|
||||
: MakeClampedScissorRect(scissorBox, snap.renderPassExtent);
|
||||
} else {
|
||||
scissor.offset = {0, 0};
|
||||
scissor.extent = { (Uint)snap.renderPassExtent.x(), (Uint)snap.renderPassExtent.y() };
|
||||
}
|
||||
ShadowedSetScissor(frame.commandBuffer, scissor);
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VulkanRenderer::SetupDraw(FrameContext::FrameData& frame, GLenum mode, Flags<DrawSetupAspect> aspects,
|
||||
const DrawCmdParam& drawParams,
|
||||
const IndexBufferView* pIndexBufferView) {
|
||||
@@ -4297,6 +4575,12 @@ void main() {
|
||||
// otherwise each re-run the full SyncTexture path on the same textures.
|
||||
VkTextureManager::DrawSyncScope drawSyncScope(*m_textureManager);
|
||||
m_textureManager->CollectGarbage();
|
||||
if (TrySetupDrawFastPath(frame, mode, aspects, drawParams, pIndexBufferView)) {
|
||||
return true;
|
||||
}
|
||||
// The fast path declined: whatever it saw may be stale. The full path
|
||||
// below re-resolves everything and refreshes the snapshot on success.
|
||||
m_setupDrawSnapshot.valid = false;
|
||||
const auto& drawFbo =
|
||||
MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||
if (drawFbo != nullptr && IsUnsupportedFramebufferForDirectVulkan(*drawFbo)) {
|
||||
@@ -4306,16 +4590,52 @@ void main() {
|
||||
const auto& vao = *MG_State::pGLContext->GetBoundVertexArray();
|
||||
const auto& program = *MG_State::pGLContext->GetCurrentProgram();
|
||||
ProgramFactory::CompileOptionFlags transformFlags = GetShaderTransformFlags(m_swapchainObject.GetPreTransform());
|
||||
const auto* programObjPtr = &m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
// Sampling a colour render target through the driver's implicit-LOD path faults the GPU on
|
||||
// Adreno 650 (see ForceExplicitLod0SamplePass); ask for the explicit-LOD variant when doing
|
||||
// so cannot change a texel, i.e. when every sampler this program reads is pinned to a
|
||||
// single mip level.
|
||||
if (UniformManager::ProgramSamplesOnlySingleLevelTextures(program, *programObjPtr)) {
|
||||
transformFlags |= ProgramFactory::CompileOptionBit::ExplicitLod0Sampling;
|
||||
programObjPtr = &m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
// single mip level. The probe walks every sampler binding, so its verdict is memoized
|
||||
// under the sampled-set memo's key plus the sampled textures' params-version sum (level
|
||||
// range and filter changes live there); the previous draw's texture list is valid for the
|
||||
// sum exactly when that key matches (same program, same binds).
|
||||
{
|
||||
const Uint64 lodProgramLifetimeId = program.GetLifetimeId();
|
||||
const Uint32 lodProgramVersion = program.GetBackendStateVersion();
|
||||
const Uint64 lodBindGeneration = MG_State::pGLContext->GetTextureBindGeneration();
|
||||
Bool lodMemoHit = false;
|
||||
if (m_lastLodDecisionValid && m_lastSampledSetValid &&
|
||||
m_lastLodProgramLifetimeId == lodProgramLifetimeId &&
|
||||
m_lastLodProgramVersion == lodProgramVersion &&
|
||||
m_lastLodBindGeneration == lodBindGeneration && m_lastLodBaseFlags == transformFlags &&
|
||||
m_lastSampledSetProgramLifetimeId == lodProgramLifetimeId &&
|
||||
m_lastSampledSetProgramVersion == lodProgramVersion &&
|
||||
m_lastSampledSetBindGeneration == lodBindGeneration) {
|
||||
Uint64 paramsSum = 0;
|
||||
for (const auto* sampledTexture : m_sampledTexturesScratch) {
|
||||
if (sampledTexture != nullptr) {
|
||||
paramsSum += sampledTexture->GetTextureParamsVersion();
|
||||
}
|
||||
const auto& programObj = *programObjPtr;
|
||||
}
|
||||
if (paramsSum == m_lastLodParamsSum) {
|
||||
transformFlags = m_lastLodResultFlags;
|
||||
lodMemoHit = true;
|
||||
}
|
||||
}
|
||||
if (!lodMemoHit) {
|
||||
const ProgramFactory::CompileOptionFlags baseFlags = transformFlags;
|
||||
const auto& baseProgramObj = m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
if (UniformManager::ProgramSamplesOnlySingleLevelTextures(program, baseProgramObj)) {
|
||||
transformFlags |= ProgramFactory::CompileOptionBit::ExplicitLod0Sampling;
|
||||
}
|
||||
m_lastLodDecisionValid = true;
|
||||
m_lastLodProgramLifetimeId = lodProgramLifetimeId;
|
||||
m_lastLodProgramVersion = lodProgramVersion;
|
||||
m_lastLodBindGeneration = lodBindGeneration;
|
||||
m_lastLodBaseFlags = baseFlags;
|
||||
m_lastLodResultFlags = transformFlags;
|
||||
m_lastLodParamsSum = 0; // filled below once the sampled set is known
|
||||
}
|
||||
}
|
||||
const auto& programObj = m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
|
||||
// Begin command recording if not yet
|
||||
if (!frame.isCommandRecording) {
|
||||
@@ -4360,6 +4680,18 @@ void main() {
|
||||
m_lastSampledSetTransformFlags = transformFlags;
|
||||
m_lastSampledSetBindGeneration = bindGeneration;
|
||||
}
|
||||
// Complete a freshly-made LOD decision (see above): its params sum
|
||||
// can only be taken once the sampled set is known. A genuine
|
||||
// all-zero sum merely re-probes next draw.
|
||||
if (m_lastLodDecisionValid && m_lastLodParamsSum == 0) {
|
||||
Uint64 paramsSum = 0;
|
||||
for (const auto* sampledTexture : sampledTextures) {
|
||||
if (sampledTexture != nullptr) {
|
||||
paramsSum += sampledTexture->GetTextureParamsVersion();
|
||||
}
|
||||
}
|
||||
m_lastLodParamsSum = paramsSum;
|
||||
}
|
||||
}
|
||||
MGLOG_D("SetupDraw: program=%u drawFbo=%u sampledTextureCount=%zu activeRenderPass=%s",
|
||||
program.GetExternalIndex(), drawFbo ? drawFbo->GetExternalIndex() : 0u, sampledTextures.size(),
|
||||
@@ -4383,7 +4715,10 @@ void main() {
|
||||
activeRenderPass = nullptr;
|
||||
}
|
||||
Bool needSampledTextureTransitions = false;
|
||||
for (auto* sampledTexture : sampledTextures) {
|
||||
auto& sampledResources = m_sampledResourcesScratch;
|
||||
sampledResources.assign(sampledTextures.size(), nullptr);
|
||||
for (SizeT sampledIndex = 0; sampledIndex < sampledTextures.size(); ++sampledIndex) {
|
||||
auto* sampledTexture = sampledTextures[sampledIndex];
|
||||
if (!sampledTexture) {
|
||||
continue;
|
||||
}
|
||||
@@ -4392,13 +4727,36 @@ void main() {
|
||||
MOBILEGL_ASSERT(textureResource != nullptr,
|
||||
"%s: SyncTextureAndGetDescriptor failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
sampledResources[sampledIndex] = textureResource;
|
||||
MGLOG_D("SetupDraw: sampled textureId=%d layout(before)=%s(%d)",
|
||||
sampledTexture->GetExternalIndex(), VkImageLayoutToString(textureResource->layout),
|
||||
static_cast<Int>(textureResource->layout));
|
||||
if (m_clearManager->HasPendingClear(sampledTexture) ||
|
||||
!IsValidSampledImageLayout(textureResource->layout)) {
|
||||
// Out-of-pass work is needed (deferred clear materialization or
|
||||
// a sampled-layout transition). When the open frame recording
|
||||
// has not referenced this image yet, that work can execute
|
||||
// ahead of the WHOLE recording - record it into the pre-pass
|
||||
// stream instead of splitting the active render pass (ANGLE's
|
||||
// outside-render-pass command stream, restricted to the
|
||||
// provably reorderable case).
|
||||
if (activeRenderPass != nullptr &&
|
||||
!m_frameContext.GetCurrent().hasPreCommandBufferRecorded &&
|
||||
!m_textureManager->WasTouchedThisRecording(*textureResource)) {
|
||||
VkCommandBuffer preCommandBuffer = m_frameContext.BeginPreCommandRecording();
|
||||
const Bool preClearReady =
|
||||
MaterializePendingClearForTexture(preCommandBuffer, *sampledTexture);
|
||||
MOBILEGL_ASSERT(preClearReady,
|
||||
"%s: pre-pass MaterializePendingClearForTexture failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
const Bool preTransitionReady =
|
||||
m_textureManager->TransitionTextureForSampling(preCommandBuffer, *sampledTexture);
|
||||
MOBILEGL_ASSERT(preTransitionReady,
|
||||
"%s: pre-pass TransitionTextureForSampling failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
continue;
|
||||
}
|
||||
needSampledTextureTransitions = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4408,10 +4766,23 @@ void main() {
|
||||
activeRenderPass = nullptr;
|
||||
}
|
||||
|
||||
for (auto* sampledTexture : sampledTextures) {
|
||||
for (SizeT sampledIndex = 0; sampledIndex < sampledTextures.size(); ++sampledIndex) {
|
||||
auto* sampledTexture = sampledTextures[sampledIndex];
|
||||
if (!sampledTexture) {
|
||||
continue;
|
||||
}
|
||||
// Fast path: the first loop already resolved this texture, nothing
|
||||
// is pending against it, and its layout is still sampleable (the
|
||||
// layout re-check covers an EndRenderPass between the loops having
|
||||
// rewritten an attachment's layout). Skipping the materialize +
|
||||
// transition + re-resolve chain here is the difference between one
|
||||
// pointer read and three calls per sampled texture per draw.
|
||||
if (auto* fastResource = sampledResources[sampledIndex];
|
||||
fastResource != nullptr && !m_clearManager->HasPendingClear(sampledTexture) &&
|
||||
IsValidSampledImageLayout(fastResource->layout)) {
|
||||
m_textureManager->StampResourceRecordingUse(*fastResource);
|
||||
continue;
|
||||
}
|
||||
const Bool clearReady = MaterializePendingClearForTexture(frame.commandBuffer, *sampledTexture);
|
||||
MOBILEGL_ASSERT(clearReady, "%s: MaterializePendingClearForTexture failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
@@ -4422,16 +4793,28 @@ void main() {
|
||||
MOBILEGL_ASSERT(transitionedResource != nullptr,
|
||||
"%s: post-transition SyncTextureAndGetDescriptor failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
// Pre-pass stream bookkeeping: the draw about to be recorded reads
|
||||
// this image, so later out-of-pass work on it can no longer jump
|
||||
// ahead of the recording.
|
||||
m_textureManager->StampResourceRecordingUse(*transitionedResource);
|
||||
MGLOG_D("SetupDraw: sampled textureId=%d layout(after)=%s(%d)",
|
||||
sampledTexture->GetExternalIndex(), VkImageLayoutToString(transitionedResource->layout),
|
||||
static_cast<Int>(transitionedResource->layout));
|
||||
}
|
||||
|
||||
auto* renderPassEntry = &m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired);
|
||||
// Depth/stencil participation of THIS draw, for the default-FBO depth-less
|
||||
// pass flavor (GL: a disabled depth/stencil test neither reads nor writes
|
||||
// its buffer).
|
||||
const Bool drawUsesDepthStencil =
|
||||
MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::DepthTest) ||
|
||||
MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::StencilTest);
|
||||
auto* renderPassEntry =
|
||||
&m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired, drawUsesDepthStencil);
|
||||
if (activeRenderPass && !activeRenderPass->CompatibleWith(*renderPassEntry)) {
|
||||
VkRenderPassManager::EndRenderPass(frame.commandBuffer);
|
||||
activeRenderPass = nullptr;
|
||||
renderPassEntry = &m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired);
|
||||
renderPassEntry =
|
||||
&m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired, drawUsesDepthStencil);
|
||||
}
|
||||
if (renderPassEntry->attachmentCount == 0 || renderPassEntry->extent.x() <= 0 || renderPassEntry->extent.y() <= 0) {
|
||||
MGLOG_D("SetupDraw skipped: drawFbo=%u resolved to an empty render pass (attachmentCount=%u extent=%dx%d)",
|
||||
@@ -4463,7 +4846,7 @@ void main() {
|
||||
// Every genuinely disabled attribute the shader reads must have a current-value type we can
|
||||
// synthesize a binding for; otherwise the upload below would push a null payload.
|
||||
const Uint32 missingAttribMask =
|
||||
activeAttribMask & ~BuildVertexInputAttributeMask(vertexInputState.attributes);
|
||||
activeAttribMask & ~vertexInputState.attributeLocationMask;
|
||||
for (Uint32 location = 0; location < kMaxVertexAttribs; ++location) {
|
||||
if ((missingAttribMask & (1u << location)) == 0) continue;
|
||||
|
||||
@@ -4491,7 +4874,11 @@ void main() {
|
||||
MOBILEGL_ASSERT(ok, "%s: BeginRenderPass failed", __func__);
|
||||
}
|
||||
|
||||
if (!g_dynamicStateShadow.graphicsPipelineValid || g_dynamicStateShadow.graphicsPipeline != pipeline) {
|
||||
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
|
||||
g_dynamicStateShadow.graphicsPipelineValid = true;
|
||||
g_dynamicStateShadow.graphicsPipeline = pipeline;
|
||||
}
|
||||
|
||||
const Bool boundUniforms = m_uniformManager->BindProgramUniformBuffers(
|
||||
frame.commandBuffer, program, programObj, m_frameContext.GetCurrentFrameIndex());
|
||||
@@ -4531,7 +4918,49 @@ void main() {
|
||||
scissor.offset = {0, 0};
|
||||
scissor.extent = { (Uint)renderPassEntry->extent.x(), (Uint)renderPassEntry->extent.y() };
|
||||
}
|
||||
vkCmdSetScissor(frame.commandBuffer, 0, 1, &scissor);
|
||||
ShadowedSetScissor(frame.commandBuffer, scissor);
|
||||
|
||||
// Snapshot the fully resolved configuration for the consecutive-draw
|
||||
// fast path (see TrySetupDrawFastPath).
|
||||
{
|
||||
auto& snap = m_setupDrawSnapshot;
|
||||
const auto* nowActiveRenderPass = VkRenderPassManager::GetActiveRenderPass();
|
||||
if (nowActiveRenderPass != nullptr && !programObj.hasStorageImages) {
|
||||
snap.valid = true;
|
||||
snap.aspects = aspects.GetRaw();
|
||||
snap.mode = mode;
|
||||
snap.programLifetimeId = program.GetLifetimeId();
|
||||
snap.programVersion = program.GetBackendStateVersion();
|
||||
snap.vao = &vao;
|
||||
snap.vaoConfigVersion = vao.GetConfigVersion();
|
||||
snap.drawFbo = drawFbo.get();
|
||||
snap.fboVersion = drawFbo->GetObjectVersion();
|
||||
snap.drawFboIsDefault = drawFbo->IsDefaultFramebuffer();
|
||||
snap.renderStateVersion = MG_State::pGLContext->GetRenderStateParametersVersion();
|
||||
snap.bindGeneration = MG_State::pGLContext->GetTextureBindGeneration();
|
||||
snap.baseTransformFlags = GetShaderTransformFlags(m_swapchainObject.GetPreTransform()).GetRaw();
|
||||
snap.resolvedTransformFlags = transformFlags.GetRaw();
|
||||
snap.renderPassHash = nowActiveRenderPass->hash;
|
||||
snap.imageIndex = m_imageIndexAcquired;
|
||||
snap.textureEraseEpoch = m_textureManager->GetResourceEraseEpoch();
|
||||
snap.textureImageEpoch = m_textureManager->GetTextureImageEpoch();
|
||||
snap.renderbufferImageEpoch = m_renderPassManager->GetRenderbufferImageEpoch();
|
||||
snap.renderPassExtent = renderPassEntry->extent;
|
||||
snap.pipeline = pipeline;
|
||||
Uint64 snapContentSum = 0;
|
||||
Uint64 snapParamsSum = 0;
|
||||
for (const auto* sampledTexture : sampledTextures) {
|
||||
if (sampledTexture != nullptr) {
|
||||
snapContentSum += sampledTexture->GetContentVersion();
|
||||
snapParamsSum += sampledTexture->GetTextureParamsVersion();
|
||||
}
|
||||
}
|
||||
snap.sampledContentSum = snapContentSum;
|
||||
snap.sampledParamsSum = snapParamsSum;
|
||||
} else {
|
||||
snap.valid = false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -5222,8 +5651,12 @@ void main() {
|
||||
if (!m_clearManager->GetPendingClears(&texture, pendingClears)) {
|
||||
return true;
|
||||
}
|
||||
MOBILEGL_ASSERT(VkRenderPassManager::GetActiveRenderPass() == nullptr,
|
||||
"MaterializePendingClearForTexture requires no active render pass");
|
||||
// A pass may stay open on the FRAME command buffer while this clear is
|
||||
// recorded into the pre-pass stream (a different command buffer that
|
||||
// executes strictly before the frame's commands).
|
||||
MOBILEGL_ASSERT(VkRenderPassManager::GetActiveRenderPass() == nullptr ||
|
||||
commandBuffer != m_frameContext.GetCurrent().commandBuffer,
|
||||
"MaterializePendingClearForTexture requires no active render pass on the target buffer");
|
||||
|
||||
auto* resource = m_textureManager->SyncTextureAndGetDescriptor(texture);
|
||||
MOBILEGL_ASSERT(resource != nullptr,
|
||||
@@ -5453,7 +5886,10 @@ void main() {
|
||||
"TryBlitToDefaultFramebufferWithShader: failed to create sampled view for textureId=%d mip=%u",
|
||||
sourceTexture->GetExternalIndex(), srcBinding.mipLevel);
|
||||
|
||||
auto& renderPassEntry = m_renderPassManager->GetOrCreateRenderPass(drawFbo, m_imageIndexAcquired);
|
||||
// A color-only blit never touches depth/stencil: let the default-FBO pass
|
||||
// it opens skip the depth attachment (depth-less flavor).
|
||||
auto& renderPassEntry =
|
||||
m_renderPassManager->GetOrCreateRenderPass(drawFbo, m_imageIndexAcquired, /*drawUsesDepthStencil=*/false);
|
||||
const Bool ok = VkRenderPassManager::BeginRenderPass(frame.commandBuffer, renderPassEntry);
|
||||
MOBILEGL_ASSERT(ok, "%s: BeginRenderPass failed", __func__);
|
||||
|
||||
@@ -5469,6 +5905,10 @@ void main() {
|
||||
const VkPipeline pipeline = GetOrCreateBlitPipeline(renderPassEntry);
|
||||
MOBILEGL_ASSERT(pipeline != VK_NULL_HANDLE, "TryBlitToDefaultFramebufferWithShader: blit pipeline is null");
|
||||
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
|
||||
// The blit pipeline's narrower dynamic set (viewport/scissor only)
|
||||
// leaves the other dynamic states undefined; its raw viewport/scissor
|
||||
// writes also bypass the shadow.
|
||||
ResetDynamicStateShadow();
|
||||
|
||||
auto* blitProgramData = static_cast<Uint8*>(m_blitResources.program->MapUBO());
|
||||
MOBILEGL_ASSERT(blitProgramData != nullptr, "TryBlitToDefaultFramebufferWithShader: blit UBO is null");
|
||||
@@ -6263,9 +6703,13 @@ void main() {
|
||||
if (frame.isCommandRecording) {
|
||||
m_frameContext.EndCommandRecording();
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
m_lastPipelineValid = false; // command-buffer boundary: drop the pipeline memo
|
||||
InvalidatePipelineMemo(); // command-buffer boundary: drop the pipeline memo
|
||||
}
|
||||
if (!frame.hasCommandBufferRecorded) {
|
||||
// The pre-pass stream must never be submitted later than the recording
|
||||
// it was paired with (frame commands recorded after a pre-pass move
|
||||
// rely on the moved work having executed first).
|
||||
m_frameContext.EndPreCommandRecordingIfOpen();
|
||||
if (!frame.hasCommandBufferRecorded && !frame.hasPreCommandBufferRecorded) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -7346,8 +7790,7 @@ void main() {
|
||||
m_programFactory->OnFrameBoundary();
|
||||
}
|
||||
if (m_pipelineFactory && m_pipelineFactory->OnFrameBoundary() > 0) {
|
||||
m_lastPipelineValid = false;
|
||||
m_lastPipelineResult = VK_NULL_HANDLE;
|
||||
InvalidatePipelineMemo();
|
||||
}
|
||||
if (m_vertexInputStateFactory) {
|
||||
m_vertexInputStateFactory->OnFrameBoundary();
|
||||
@@ -7400,8 +7843,18 @@ void main() {
|
||||
submitInfo.pWaitSemaphores = &waitSemaphore;
|
||||
submitInfo.pWaitDstStageMask = &waitDstStageMask;
|
||||
}
|
||||
submitInfo.commandBufferCount = 1;
|
||||
submitInfo.pCommandBuffers = &frame.commandBuffer;
|
||||
// The pre-pass stream, when recorded, executes strictly before the
|
||||
// frame's commands within the same submission.
|
||||
VkCommandBuffer commandBuffers[2] = {VK_NULL_HANDLE, VK_NULL_HANDLE};
|
||||
Uint32 commandBufferCount = 0;
|
||||
if (frame.hasPreCommandBufferRecorded) {
|
||||
commandBuffers[commandBufferCount++] = frame.preCommandBuffer;
|
||||
}
|
||||
if (frame.hasCommandBufferRecorded) {
|
||||
commandBuffers[commandBufferCount++] = frame.commandBuffer;
|
||||
}
|
||||
submitInfo.commandBufferCount = commandBufferCount;
|
||||
submitInfo.pCommandBuffers = commandBuffers;
|
||||
const VkResult result = vkQueueSubmit(m_graphicsQueue, 1, &submitInfo, fence);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("SubmitPendingCommandBuffer: vkQueueSubmit returned %d", result);
|
||||
@@ -7409,6 +7862,7 @@ void main() {
|
||||
}
|
||||
frame.imageAvailableSemaphoreConsumed = true;
|
||||
frame.hasCommandBufferRecorded = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
RegisterSubmit(fence, pooledFence);
|
||||
frame.lastSubmitIndex = m_submitCounter;
|
||||
return true;
|
||||
@@ -7439,6 +7893,8 @@ void main() {
|
||||
}
|
||||
m_frameContext.EndCommandRecording();
|
||||
}
|
||||
m_frameContext.EndPreCommandRecordingIfOpen();
|
||||
const Bool submittingPreCommandBuffer = frame.hasPreCommandBufferRecorded;
|
||||
if (!SubmitPendingCommandBuffer(frame, fence, /*pooledFence=*/true)) {
|
||||
// Submit failure (device loss regime): the ended command buffer
|
||||
// stays marked recorded so Present can still try to submit it.
|
||||
@@ -7451,13 +7907,12 @@ void main() {
|
||||
// cache and the aging sweep could destroy it while the flushed submission
|
||||
// still references it. Mirrors the drops at the readback and Present
|
||||
// boundaries; costs one full pipeline lookup on the next draw.
|
||||
m_lastPipelineValid = false;
|
||||
m_lastPipelineResult = VK_NULL_HANDLE;
|
||||
InvalidatePipelineMemo();
|
||||
|
||||
// The submitted command buffer may still be executing; recording must
|
||||
// restart on a fresh one. If none can be allocated, fall back to
|
||||
// draining this submission so reusing the buffer stays legal.
|
||||
const VkResult retireResult = m_frameContext.RetireCurrentCommandBuffer();
|
||||
const VkResult retireResult = m_frameContext.RetireCurrentCommandBuffer(submittingPreCommandBuffer);
|
||||
if (retireResult != VK_SUCCESS) {
|
||||
MGLOG_E("FlushPendingCommands: RetireCurrentCommandBuffer returned %d; draining submission", retireResult);
|
||||
if (vkWaitForFences(m_device, 1, &fence, VK_TRUE, UINT64_MAX) == VK_SUCCESS) {
|
||||
@@ -7523,6 +7978,17 @@ void main() {
|
||||
}
|
||||
|
||||
void VulkanRenderer::OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer) {
|
||||
// Dynamic state does not survive a command-buffer boundary.
|
||||
ResetDynamicStateShadow();
|
||||
m_setupDrawSnapshot.valid = false;
|
||||
if (m_uniformManager) {
|
||||
m_uniformManager->OnCommandBufferBoundary();
|
||||
}
|
||||
// Pre-pass stream bookkeeping: a fresh frame recording references no
|
||||
// textures yet.
|
||||
if (m_textureManager) {
|
||||
m_textureManager->AdvanceRecordingGeneration();
|
||||
}
|
||||
if (m_timerQueryManager) {
|
||||
m_timerQueryManager->OnFrameCommandRecordingBegan(commandBuffer, m_frameContext.GetCurrentFrameIndex(),
|
||||
m_bufferManager.GetFrameSerial());
|
||||
@@ -7595,9 +8061,10 @@ void main() {
|
||||
if (suspendedFrame.isCommandRecording) {
|
||||
m_frameContext.EndCommandRecording();
|
||||
}
|
||||
m_frameContext.AbandonPreCommandRecording();
|
||||
suspendedFrame.isCommandRecording = false;
|
||||
suspendedFrame.hasCommandBufferRecorded = false;
|
||||
m_lastPipelineValid = false;
|
||||
InvalidatePipelineMemo();
|
||||
// The dropped recording is never submitted, so once the fence
|
||||
// poll shows the pre-suspension submissions complete the frame
|
||||
// transients (descriptor sets, transient arenas, deferred
|
||||
@@ -7633,8 +8100,11 @@ void main() {
|
||||
// performs a real, stamping lookup) and can never age out.
|
||||
m_programFactory->OnFrameBoundary();
|
||||
if (m_pipelineFactory->OnFrameBoundary() > 0) {
|
||||
m_lastPipelineValid = false; // an aged-out pipeline may still be memoized
|
||||
m_lastPipelineResult = VK_NULL_HANDLE;
|
||||
InvalidatePipelineMemo(); // an aged-out pipeline may still be memoized
|
||||
// A recreated pipeline could reuse a freed handle value and alias
|
||||
// the bind-dedup shadow; force the next draw to re-bind.
|
||||
g_dynamicStateShadow.graphicsPipelineValid = false;
|
||||
m_setupDrawSnapshot.valid = false;
|
||||
}
|
||||
m_vertexInputStateFactory->OnFrameBoundary();
|
||||
m_samplerManager->OnFrameBoundary();
|
||||
@@ -7657,18 +8127,21 @@ void main() {
|
||||
if (frame.isCommandRecording) {
|
||||
m_frameContext.EndCommandRecording();
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
m_lastPipelineValid = false; // command-buffer boundary: drop the pipeline memo
|
||||
InvalidatePipelineMemo(); // command-buffer boundary: drop the pipeline memo
|
||||
}
|
||||
m_frameContext.EndPreCommandRecordingIfOpen();
|
||||
|
||||
const Bool shouldSubmitCommandBuffer = frame.hasCommandBufferRecorded;
|
||||
|
||||
// 1) Submit current frame work.
|
||||
// 1) Submit current frame work (the pre-pass stream, when recorded,
|
||||
// rides the same submission strictly ahead of the frame commands).
|
||||
auto submitPacket = m_frameContext.GetSubmitInfo(shouldSubmitCommandBuffer, m_imageIndexAcquired);
|
||||
VK_VERIFY(vkQueueSubmit(m_graphicsQueue, 1, &submitPacket.submitInfo, frame.imageInFlightFence));
|
||||
RegisterSubmit(frame.imageInFlightFence, /*pooledFence=*/false);
|
||||
frame.lastSubmitIndex = m_submitCounter;
|
||||
frame.isCommandRecording = false;
|
||||
frame.hasCommandBufferRecorded = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
m_swapchainObject.SetImageLayout(m_imageIndexAcquired, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR);
|
||||
|
||||
// 2) Present current frame.
|
||||
@@ -7696,6 +8169,13 @@ void main() {
|
||||
result = VK_SUCCESS;
|
||||
}
|
||||
VK_VERIFY(result, "Present, vkQueuePresentKHR");
|
||||
// EGL swap semantics: the presented color buffer's content is undefined the
|
||||
// next time this image is acquired (EGL_BUFFER_DESTROYED, the default swap
|
||||
// behaviour), and EVERY ancillary depth/stencil buffer's content is
|
||||
// undefined after any swap. The render-pass manager turns the undefined
|
||||
// attachments' next tile loads into LOAD_OP_DONT_CARE.
|
||||
m_swapchainObject.SetImageContentDefined(m_imageIndexAcquired, false);
|
||||
m_swapchainObject.SetAllDepthStencilContentUndefined();
|
||||
// The authoritative check, done here - after the frame is presented, before the next
|
||||
// acquire. This is what makes a launcher-side resolution change take effect: shrinking
|
||||
// the window's buffer (SurfaceHolder.setFixedSize) moves currentExtent, the swapchain
|
||||
@@ -8650,11 +9130,17 @@ void main() {
|
||||
if (m_pipelineFactory) {
|
||||
m_pipelineFactory->DestroyAll();
|
||||
}
|
||||
m_lastPipelineValid = false; // pipelines freed -> the memoized handle would dangle
|
||||
InvalidatePipelineMemo(); // pipelines freed -> the memoized handle would dangle
|
||||
g_dynamicStateShadow.graphicsPipelineValid = false;
|
||||
m_setupDrawSnapshot.valid = false;
|
||||
DestroyComputePipelines();
|
||||
if (m_frameContext.GetFrameCount() > 0) {
|
||||
m_frameContext.GetCurrent().isCommandRecording = false;
|
||||
m_frameContext.GetCurrent().hasCommandBufferRecorded = false;
|
||||
// The pre-pass stream paired with the abandoned recording is
|
||||
// dropped with it (its next Begin resets the buffer).
|
||||
m_frameContext.GetCurrent().isPreCommandRecording = false;
|
||||
m_frameContext.GetCurrent().hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
const Bool okArena = m_bufferManager.RecreateTransientArenas(m_frameContext.GetFrameCount());
|
||||
MOBILEGL_ASSERT(okArena, "RecreateSwapchain: buffer manager transient arena initialization failed");
|
||||
@@ -8809,8 +9295,7 @@ void main() {
|
||||
// destroys them immediately. The memo must drop as well: it can hand out a
|
||||
// cached handle without touching the factory.
|
||||
if (m_pipelineFactory->EvictByRenderPasses(renderPasses) > 0) {
|
||||
m_lastPipelineValid = false;
|
||||
m_lastPipelineResult = VK_NULL_HANDLE;
|
||||
InvalidatePipelineMemo();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8829,8 +9314,7 @@ void main() {
|
||||
m_computePipelines.erase(computeIt);
|
||||
}
|
||||
if (m_pipelineFactory != nullptr && m_pipelineFactory->EvictByProgramHash(programHash) > 0) {
|
||||
m_lastPipelineValid = false;
|
||||
m_lastPipelineResult = VK_NULL_HANDLE;
|
||||
InvalidatePipelineMemo();
|
||||
}
|
||||
if (m_uniformManager != nullptr) {
|
||||
m_uniformManager->OnDescriptorSetLayoutDestroyed(descriptorSetLayout);
|
||||
|
||||
@@ -151,6 +151,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool SetupDraw(FrameContext::FrameData& frame, GLenum mode, Flags<DrawSetupAspect> aspects,
|
||||
const DrawCmdParam& drawParams,
|
||||
const IndexBufferView* pIndexBufferView = nullptr);
|
||||
// ANGLE-style consecutive-draw fast path: SetupDraw snapshots the fully
|
||||
// resolved draw configuration; the next draw whose cheap version/identity
|
||||
// checks all match skips the resolution half (LOD probe, sampled-set
|
||||
// walk, render-pass and pipeline resolution) and jumps straight to the
|
||||
// per-draw tail. Returns false (leaving no side effects that the full
|
||||
// path cannot redo idempotently) whenever anything might have changed.
|
||||
Bool TrySetupDrawFastPath(FrameContext::FrameData& frame, GLenum mode, Flags<DrawSetupAspect> aspects,
|
||||
const DrawCmdParam& drawParams, const IndexBufferView* pIndexBufferView);
|
||||
void ClearAttachmentsOnActiveRenderPass(VkCommandBuffer commandBuffer,
|
||||
const RenderPassEntry& compatibleRenderPassEntry);
|
||||
|
||||
@@ -455,14 +463,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// gather + synthetic vertex-input rebuild + payload hash + lookup) when the full pipeline
|
||||
// state is unchanged from the previous draw. The key provably covers every pipeline field.
|
||||
// Reset per-frame and on pipeline destruction so the cached handle can never dangle.
|
||||
Bool m_lastPipelineValid = false;
|
||||
GLenum m_lastPipelineMode = 0;
|
||||
Uint64 m_lastPipelineProgramHash = 0;
|
||||
Uint64 m_lastPipelineVertexInputHash = 0;
|
||||
Uint64 m_lastPipelineRenderPassHash = 0;
|
||||
Uint m_lastPipelineRenderStateVersion = 0;
|
||||
ProgramFactory::CompileOptionFlags m_lastPipelineTransformFlags = {};
|
||||
VkPipeline m_lastPipelineResult = VK_NULL_HANDLE;
|
||||
// Small N-way pipeline-resolution memo (round-robin replacement). A
|
||||
// single-entry memo thrashed on draw sequences that alternate a few
|
||||
// pipelines (GUI text/quad program ping-pong), paying the full
|
||||
// payload-hash lookup per draw; eight entries cover such working sets
|
||||
// while keeping the hit path a trivial linear scan.
|
||||
struct PipelineMemoEntry {
|
||||
GLenum mode = 0;
|
||||
Uint64 programHash = 0;
|
||||
Uint64 vertexInputHash = 0;
|
||||
Uint64 renderPassHash = 0;
|
||||
Uint renderStateVersion = 0;
|
||||
ProgramFactory::CompileOptionFlags transformFlags = {};
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
};
|
||||
static constexpr Uint32 kPipelineMemoSize = 8;
|
||||
PipelineMemoEntry m_pipelineMemo[kPipelineMemoSize];
|
||||
Uint32 m_pipelineMemoCount = 0;
|
||||
Uint32 m_pipelineMemoNext = 0;
|
||||
// Drops every memoized pipeline handle. Required at command-buffer
|
||||
// boundaries and whenever any pipeline may have been destroyed.
|
||||
void InvalidatePipelineMemo() {
|
||||
m_pipelineMemoCount = 0;
|
||||
m_pipelineMemoNext = 0;
|
||||
}
|
||||
UnorderedMap<ProgramFactory::HashType, VkPipeline> m_computePipelines;
|
||||
UniquePtr<ProgramFactory> m_programFactory;
|
||||
UniquePtr<UniformManager> m_uniformManager;
|
||||
@@ -491,9 +515,61 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
ProgramFactory::CompileOptionFlags m_lastSampledSetTransformFlags = {};
|
||||
Uint64 m_lastSampledSetBindGeneration = 0;
|
||||
|
||||
// Memo for the per-draw explicit-LOD-0 eligibility probe
|
||||
// (ProgramSamplesOnlySingleLevelTextures): same key family as the
|
||||
// sampled-set memo, plus the sampled textures' params-version sum so a
|
||||
// level-range or filter change re-probes. On a hit the resolved
|
||||
// transform flags are reused, which also collapses the two
|
||||
// GetOrCreateProgram lookups into one.
|
||||
Bool m_lastLodDecisionValid = false;
|
||||
Uint64 m_lastLodProgramLifetimeId = 0;
|
||||
Uint32 m_lastLodProgramVersion = 0;
|
||||
Uint64 m_lastLodBindGeneration = 0;
|
||||
Uint64 m_lastLodParamsSum = 0;
|
||||
ProgramFactory::CompileOptionFlags m_lastLodBaseFlags = {};
|
||||
ProgramFactory::CompileOptionFlags m_lastLodResultFlags = {};
|
||||
|
||||
// Snapshot behind TrySetupDrawFastPath. Values only: the program and
|
||||
// render-pass caches are open-addressing maps whose entries move on
|
||||
// insert, so no pointers into them are cached; the pipeline handle is
|
||||
// protected by the command-buffer-boundary reset plus the mid-frame
|
||||
// pipeline-destruction resets, and monotonic epochs guard everything
|
||||
// that can be destroyed or recreated between draws.
|
||||
struct SetupDrawSnapshot {
|
||||
Bool valid = false;
|
||||
Uint8 aspects = 0;
|
||||
GLenum mode = 0;
|
||||
Uint64 programLifetimeId = 0;
|
||||
Uint32 programVersion = 0;
|
||||
const void* vao = nullptr;
|
||||
Uint32 vaoConfigVersion = 0;
|
||||
const void* drawFbo = nullptr;
|
||||
Uint16 fboVersion = 0;
|
||||
Bool drawFboIsDefault = false;
|
||||
Uint renderStateVersion = 0;
|
||||
Uint64 bindGeneration = 0;
|
||||
Uint32 baseTransformFlags = 0;
|
||||
Uint32 resolvedTransformFlags = 0;
|
||||
Uint64 renderPassHash = 0;
|
||||
Uint32 imageIndex = 0;
|
||||
Uint64 textureEraseEpoch = 0;
|
||||
Uint64 textureImageEpoch = 0;
|
||||
Uint64 renderbufferImageEpoch = 0;
|
||||
Uint64 sampledContentSum = 0;
|
||||
Uint64 sampledParamsSum = 0;
|
||||
IntVec2 renderPassExtent = {0, 0};
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
};
|
||||
SetupDrawSnapshot m_setupDrawSnapshot;
|
||||
|
||||
// Per-draw scratch buffers (clear keeps capacity) — these paths run for every
|
||||
// draw call and must not allocate.
|
||||
Vector<MG_State::GLState::ITextureObject*> m_sampledTexturesScratch;
|
||||
// Parallel to m_sampledTexturesScratch, refilled by every SetupDraw's
|
||||
// first sampled-texture loop: the resolved backend resources, so the
|
||||
// post-transition loop can skip re-resolving textures whose layout is
|
||||
// already sampleable.
|
||||
Vector<VkTextureManager::TextureResource*> m_sampledResourcesScratch;
|
||||
Vector<MG_State::GLState::ITextureObject*> m_storageImageTexturesScratch;
|
||||
Vector<VkBuffer> m_vertexBuffersScratch;
|
||||
Vector<VkDeviceSize> m_vertexOffsetsScratch;
|
||||
|
||||
@@ -104,6 +104,23 @@ namespace MobileGL {
|
||||
m_backendHashMemoVersion = m_configVersion;
|
||||
}
|
||||
|
||||
// Backend-owned resolved-state memo: an opaque pointer into the
|
||||
// backend's vertex-input-state cache plus the cache's eviction
|
||||
// epoch, valid while the config version matches. Lets the
|
||||
// per-draw path skip the content hash AND the cache lookup; the
|
||||
// epoch guards against the cache evicting the pointee.
|
||||
Bool GetBackendStateMemo(const void*& outState, Uint64& outEpoch) const {
|
||||
if (m_backendStateMemoVersion != m_configVersion) return false;
|
||||
outState = m_backendStateMemo;
|
||||
outEpoch = m_backendStateMemoEpoch;
|
||||
return true;
|
||||
}
|
||||
void SetBackendStateMemo(const void* state, Uint64 epoch) const {
|
||||
m_backendStateMemo = state;
|
||||
m_backendStateMemoEpoch = epoch;
|
||||
m_backendStateMemoVersion = m_configVersion;
|
||||
}
|
||||
|
||||
private:
|
||||
void BumpAttributeFormatVersion(Uint index);
|
||||
void BumpAttributeBufferVersion(Uint index);
|
||||
@@ -137,6 +154,9 @@ namespace MobileGL {
|
||||
Uint32 m_configVersion = 0;
|
||||
mutable Uint64 m_backendHashMemo = 0;
|
||||
mutable Uint32 m_backendHashMemoVersion = ~0u;
|
||||
mutable const void* m_backendStateMemo = nullptr;
|
||||
mutable Uint64 m_backendStateMemoEpoch = 0;
|
||||
mutable Uint32 m_backendStateMemoVersion = ~0u;
|
||||
};
|
||||
} // namespace GLState
|
||||
} // namespace MG_State
|
||||
|
||||
Reference in New Issue
Block a user