mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-12 14:18:31 +09:00
Compare commits
4
Commits
562076f657
...
3181ed2c5a
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3181ed2c5a | ||
|
|
6aed3b08f3 | ||
|
|
281467a345 | ||
|
|
c7e36986e7 |
@@ -15,6 +15,7 @@
|
|||||||
#include <MG_Impl/GLImpl/Texture/ProxyTexture.h>
|
#include <MG_Impl/GLImpl/Texture/ProxyTexture.h>
|
||||||
#include <MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h>
|
#include <MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h>
|
||||||
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
||||||
|
#include <MG_Impl/GLImpl/Query/GL_Query.h>
|
||||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||||
|
|
||||||
@@ -51,6 +52,11 @@ namespace MobileGL {
|
|||||||
// before a re-initialized library could pair them with the wrong
|
// before a re-initialized library could pair them with the wrong
|
||||||
// backend's DeleteSync).
|
// backend's DeleteSync).
|
||||||
MG_Impl::GLImpl::DestroyAllSyncObjects();
|
MG_Impl::GLImpl::DestroyAllSyncObjects();
|
||||||
|
// Queries die with their contexts for the same reason, and their registry
|
||||||
|
// is the same shape of process-global map: drain it here too, while the
|
||||||
|
// function table can still pair each backend handle with the backend that
|
||||||
|
// minted it.
|
||||||
|
MG_Impl::GLImpl::DestroyAllQueryObjects();
|
||||||
MG_Backend::pActiveBackendObject.reset();
|
MG_Backend::pActiveBackendObject.reset();
|
||||||
MG_State::pGLContext.reset();
|
MG_State::pGLContext.reset();
|
||||||
MG_State::pEGLContext.reset();
|
MG_State::pEGLContext.reset();
|
||||||
|
|||||||
@@ -1431,6 +1431,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
g_fboSyncedSlotVersions[SizeT(target)] = slotVersion;
|
g_fboSyncedSlotVersions[SizeT(target)] = slotVersion;
|
||||||
g_fboSyncedObjectVersions[SizeT(target)] = objectVersion;
|
g_fboSyncedObjectVersions[SizeT(target)] = objectVersion;
|
||||||
g_fboSyncedObjects[SizeT(target)] = fbo;
|
g_fboSyncedObjects[SizeT(target)] = fbo;
|
||||||
|
g_fboSyncedBackendIdGenerations[SizeT(target)] = g_attachmentBackendIdGeneration;
|
||||||
}
|
}
|
||||||
|
|
||||||
void SyncCurrentFBO() {
|
void SyncCurrentFBO() {
|
||||||
@@ -1461,9 +1462,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
const Uint16 slotVersion = slot.GetVersion();
|
const Uint16 slotVersion = slot.GetVersion();
|
||||||
const Uint16 objectVersion = currentFBO ? currentFBO->GetObjectVersion() : 0;
|
const Uint16 objectVersion = currentFBO ? currentFBO->GetObjectVersion() : 0;
|
||||||
auto* currentPtr = currentFBO.get();
|
auto* currentPtr = currentFBO.get();
|
||||||
|
// The backend-id generation joins the triple: a backend texture re-mint
|
||||||
|
// (RecreateBackendTexture) moves no frontend version, so without it the
|
||||||
|
// early-out would keep the driver FBO on the deleted texture name.
|
||||||
if (slotVersion == g_fboSyncedSlotVersions[SizeT(target)] &&
|
if (slotVersion == g_fboSyncedSlotVersions[SizeT(target)] &&
|
||||||
objectVersion == g_fboSyncedObjectVersions[SizeT(target)] &&
|
objectVersion == g_fboSyncedObjectVersions[SizeT(target)] &&
|
||||||
currentPtr == g_fboSyncedObjects[SizeT(target)]) {
|
currentPtr == g_fboSyncedObjects[SizeT(target)] &&
|
||||||
|
g_fboSyncedBackendIdGenerations[SizeT(target)] == g_attachmentBackendIdGeneration) {
|
||||||
lastUpdatedFBO = currentPtr;
|
lastUpdatedFBO = currentPtr;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -2129,6 +2134,15 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
static Bool g_broadcastMemoValid = false;
|
static Bool g_broadcastMemoValid = false;
|
||||||
static Uint g_broadcastMemoCount = 1;
|
static Uint g_broadcastMemoCount = 1;
|
||||||
|
|
||||||
|
// The identity+version key above is only monotonic WITHIN one GLContext: a
|
||||||
|
// library teardown + re-init frees every FramebufferObject and restarts the
|
||||||
|
// draw slot's counter at zero, so a recycled FBO address with coinciding
|
||||||
|
// fresh versions would false-hit. Cleared at the same boundaries as the
|
||||||
|
// structurally identical SyncCurrentFBO trio (InvalidateFramebufferBindingCache).
|
||||||
|
void InvalidateBroadcastMemo() {
|
||||||
|
g_broadcastMemoValid = false;
|
||||||
|
}
|
||||||
|
|
||||||
void SyncCurrentProgram(const SharedPtr<MG_State::GLState::ProgramObject>& currentProgram) {
|
void SyncCurrentProgram(const SharedPtr<MG_State::GLState::ProgramObject>& currentProgram) {
|
||||||
#ifdef TRACY_ENABLE
|
#ifdef TRACY_ENABLE
|
||||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||||
@@ -2312,6 +2326,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
FramebufferImpl::g_fboSyncedSlotVersions[(SizeT)target] = slot.GetVersion();
|
FramebufferImpl::g_fboSyncedSlotVersions[(SizeT)target] = slot.GetVersion();
|
||||||
FramebufferImpl::g_fboSyncedObjectVersions[(SizeT)target] = fbo ? fbo->GetObjectVersion() : 0;
|
FramebufferImpl::g_fboSyncedObjectVersions[(SizeT)target] = fbo ? fbo->GetObjectVersion() : 0;
|
||||||
FramebufferImpl::g_fboSyncedObjects[(SizeT)target] = fbo.get();
|
FramebufferImpl::g_fboSyncedObjects[(SizeT)target] = fbo.get();
|
||||||
|
FramebufferImpl::g_fboSyncedBackendIdGenerations[(SizeT)target] =
|
||||||
|
FramebufferImpl::g_attachmentBackendIdGeneration;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void BindCurrentProgramWithResources(
|
static void BindCurrentProgramWithResources(
|
||||||
@@ -3950,8 +3966,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
Bool resolved = g_GLESFuncs.glCheckFramebufferStatus(GL_DRAW_FRAMEBUFFER) == GL_FRAMEBUFFER_COMPLETE;
|
Bool resolved = g_GLESFuncs.glCheckFramebufferStatus(GL_DRAW_FRAMEBUFFER) == GL_FRAMEBUFFER_COMPLETE;
|
||||||
if (resolved) {
|
if (resolved) {
|
||||||
DrainBlitErrors();
|
DrainBlitErrors();
|
||||||
|
// A blit is scissored like a draw (the replicate path's guard documents the
|
||||||
|
// same rule): the application's box would clip this resolve into the
|
||||||
|
// scratch, and the second blit would then copy never-written scratch texels
|
||||||
|
// into the destination - silently, since scissor clipping raises no GL
|
||||||
|
// error. Disable for the staging blit only; the caller-visible blit below
|
||||||
|
// keeps the blit's native scissor semantics. Tracked via the render-state
|
||||||
|
// shadow, exactly like ScopedScissorDisable.
|
||||||
|
const Bool scissorWasEnabled =
|
||||||
|
(RenderStateImpl::g_syncedRenderStateParameters.ScissorTestEnabledMask & 1u) != 0;
|
||||||
|
if (scissorWasEnabled) g_GLESFuncs.glDisable(GL_SCISSOR_TEST);
|
||||||
g_GLESFuncs.glBlitFramebuffer(left, bottom, right, top, 0, 0, width, height, GL_COLOR_BUFFER_BIT,
|
g_GLESFuncs.glBlitFramebuffer(left, bottom, right, top, 0, 0, width, height, GL_COLOR_BUFFER_BIT,
|
||||||
GL_NEAREST);
|
GL_NEAREST);
|
||||||
|
if (scissorWasEnabled) g_GLESFuncs.glEnable(GL_SCISSOR_TEST);
|
||||||
resolved = g_GLESFuncs.glGetError() == GL_NO_ERROR;
|
resolved = g_GLESFuncs.glGetError() == GL_NO_ERROR;
|
||||||
}
|
}
|
||||||
if (resolved) {
|
if (resolved) {
|
||||||
@@ -4097,13 +4124,25 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
// The per-draw-buffer colour masks are not covered by the non-indexed
|
// The per-draw-buffer colour masks are not covered by the non-indexed
|
||||||
// glColorMask above.
|
// glColorMask above. Restore what the SYNC actually pushed, not the raw
|
||||||
|
// application masks: a widened attachment's alpha write is forced off by
|
||||||
|
// SyncRenderState and memoized in g_syncedColorMaskAlphaWidenMask, and the
|
||||||
|
// next sync early-outs on an unchanged version - restoring the undoctored
|
||||||
|
// mask here would leave alpha writes enabled on the widened buffer with
|
||||||
|
// nothing left to repair it. Same three-way pointer fallback as
|
||||||
|
// SyncRenderState's push: gating on the core name alone left EXT/OES-only
|
||||||
|
// devices holding buffer 0's mask broadcast across every buffer.
|
||||||
|
const auto colorMaskiFn = g_GLESFuncs.glColorMaski ? g_GLESFuncs.glColorMaski
|
||||||
|
: g_GLESFuncs.glColorMaskiEXT ? g_GLESFuncs.glColorMaskiEXT
|
||||||
|
: g_GLESFuncs.glColorMaskiOES;
|
||||||
|
if (colorMaskiFn) {
|
||||||
for (Uint index = 0; index < MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS; ++index) {
|
for (Uint index = 0; index < MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS; ++index) {
|
||||||
const BoolVec4& colorMask = RenderStateImpl::g_syncedRenderStateParameters.ColorMasks[index];
|
BoolVec4 colorMask = RenderStateImpl::g_syncedRenderStateParameters.ColorMasks[index];
|
||||||
if (g_GLESFuncs.glColorMaski) {
|
if (index < 32 && (RenderStateImpl::g_syncedColorMaskAlphaWidenMask & (1u << index)) != 0) {
|
||||||
g_GLESFuncs.glColorMaski(index, colorMask.x() ? GL_TRUE : GL_FALSE,
|
colorMask.w() = false;
|
||||||
colorMask.y() ? GL_TRUE : GL_FALSE, colorMask.z() ? GL_TRUE : GL_FALSE,
|
}
|
||||||
colorMask.w() ? GL_TRUE : GL_FALSE);
|
colorMaskiFn(index, colorMask.x() ? GL_TRUE : GL_FALSE, colorMask.y() ? GL_TRUE : GL_FALSE,
|
||||||
|
colorMask.z() ? GL_TRUE : GL_FALSE, colorMask.w() ? GL_TRUE : GL_FALSE);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (m_pausedTransformFeedback && g_GLESFuncs.glResumeTransformFeedback) {
|
if (m_pausedTransformFeedback && g_GLESFuncs.glResumeTransformFeedback) {
|
||||||
@@ -8304,6 +8343,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// Conservatively drop the redundant-glUseProgram guard: re-issuing one bind
|
// Conservatively drop the redundant-glUseProgram guard: re-issuing one bind
|
||||||
// after a MakeCurrent is cheaper than trusting a possibly-reset context.
|
// after a MakeCurrent is cheaper than trusting a possibly-reset context.
|
||||||
PrgramImpl::g_lastUsedBackendProgramId = 0;
|
PrgramImpl::g_lastUsedBackendProgramId = 0;
|
||||||
|
// The GLContext becoming current may be a fresh one whose slot versions
|
||||||
|
// restarted at zero; the broadcast memo's key is only monotonic within one.
|
||||||
|
PrgramImpl::InvalidateBroadcastMemo();
|
||||||
BufferImpl::InvalidateIndexedBufferBindingCache();
|
BufferImpl::InvalidateIndexedBufferBindingCache();
|
||||||
BufferImpl::InvalidatePixelBufferBindingCaches();
|
BufferImpl::InvalidatePixelBufferBindingCaches();
|
||||||
FramebufferImpl::InvalidateFramebufferBindingCache();
|
FramebufferImpl::InvalidateFramebufferBindingCache();
|
||||||
@@ -8832,6 +8874,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
FramebufferImpl::InvalidateFramebufferBindingCache();
|
FramebufferImpl::InvalidateFramebufferBindingCache();
|
||||||
VertexArrayImpl::InvalidateVAOBindingCache();
|
VertexArrayImpl::InvalidateVAOBindingCache();
|
||||||
PixelStoreImpl::InvalidatePackStateCache();
|
PixelStoreImpl::InvalidatePackStateCache();
|
||||||
|
PrgramImpl::InvalidateBroadcastMemo();
|
||||||
// Texture ids belong to the dying context; wrappers destroyed later must
|
// Texture ids belong to the dying context; wrappers destroyed later must
|
||||||
// not glDeleteTextures a recycled name in a successor context.
|
// not glDeleteTextures a recycled name in a successor context.
|
||||||
++g_backendContextGeneration;
|
++g_backendContextGeneration;
|
||||||
|
|||||||
@@ -577,6 +577,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// immutable storage, and any prior mutable store is replaced anyway.
|
// immutable storage, and any prior mutable store is replaced anyway.
|
||||||
if (resource->id != 0) {
|
if (resource->id != 0) {
|
||||||
NoteBufferIdDeleted(resource->id);
|
NoteBufferIdDeleted(resource->id);
|
||||||
|
// Driver VAOs may have this id baked into attribute/element bindings
|
||||||
|
// keyed on frontend versions this re-mint does not move.
|
||||||
|
++g_bufferBackendIdGeneration;
|
||||||
g_GLESFuncs.glDeleteBuffers(1, &resource->id);
|
g_GLESFuncs.glDeleteBuffers(1, &resource->id);
|
||||||
resource->id = 0;
|
resource->id = 0;
|
||||||
resource->immutableStorage = false;
|
resource->immutableStorage = false;
|
||||||
@@ -833,6 +836,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
g_bufferMutationEpoch.fetch_add(1, std::memory_order_release);
|
g_bufferMutationEpoch.fetch_add(1, std::memory_order_release);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// See the declaration: re-mints of a live resource's driver id. Written only on
|
||||||
|
// the context thread (both re-mint sites run there), read only by the VAO sync.
|
||||||
|
Uint64 g_bufferBackendIdGeneration = 0;
|
||||||
|
|
||||||
void RegisterBufferBackendOps() {
|
void RegisterBufferBackendOps() {
|
||||||
MG_State::GLState::SetBufferBackendOps(&g_glesBufferBackendOps);
|
MG_State::GLState::SetBufferBackendOps(&g_glesBufferBackendOps);
|
||||||
// Frontend writes issued while ops were unregistered advanced change
|
// Frontend writes issued while ops were unregistered advanced change
|
||||||
@@ -960,6 +967,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// here, on the thread that can, and the id is re-minted below.
|
// here, on the thread that can, and the id is re-minted below.
|
||||||
if (resource->immutableStorage && !resource->persistentMapped && resource->id != 0) {
|
if (resource->immutableStorage && !resource->persistentMapped && resource->id != 0) {
|
||||||
NoteBufferIdDeleted(resource->id);
|
NoteBufferIdDeleted(resource->id);
|
||||||
|
// Same as the persistent-map re-mint: the dying id may be baked into
|
||||||
|
// driver VAO bindings whose frontend versions do not move for this.
|
||||||
|
++g_bufferBackendIdGeneration;
|
||||||
g_GLESFuncs.glDeleteBuffers(1, &resource->id);
|
g_GLESFuncs.glDeleteBuffers(1, &resource->id);
|
||||||
resource->id = 0;
|
resource->id = 0;
|
||||||
resource->immutableStorage = false;
|
resource->immutableStorage = false;
|
||||||
@@ -1640,8 +1650,24 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// PrepareForDraw's BindCurrentVAO establishes the draw binding regardless.
|
// PrepareForDraw's BindCurrentVAO establishes the draw binding regardless.
|
||||||
const Uint32 currentConfigVersion = stateVAOObject->GetConfigVersion();
|
const Uint32 currentConfigVersion = stateVAOObject->GetConfigVersion();
|
||||||
const Uint16 currentIndexBufferVersion = stateVAOObject->GetIndexBufferBindingSlot().GetVersion();
|
const Uint16 currentIndexBufferVersion = stateVAOObject->GetIndexBufferBindingSlot().GetVersion();
|
||||||
const Bool attributesDirty = !m_hasSyncedConfigVersion || m_syncedConfigVersion != currentConfigVersion;
|
// A live buffer's driver id was re-minted since this twin's last emit
|
||||||
const Bool indexBufferDirty = currentIndexBufferVersion != m_syncedIndexBufferVersion;
|
// (persistent-map adoption / immutable-store retire): every baked binding may
|
||||||
|
// hold the dead id while every frontend version still matches, so force a
|
||||||
|
// full re-emit. Read once; each buffer re-mints at most once per walk (its
|
||||||
|
// first EnsureBufferResource this draw), before its id is baked, so stamping
|
||||||
|
// the entry value at the end is exact - and a stale stamp only costs one
|
||||||
|
// extra full emit.
|
||||||
|
const Uint64 currentBufferIdGeneration = BufferImpl::g_bufferBackendIdGeneration;
|
||||||
|
const Bool bufferIdsRemitted = m_syncedBufferIdGeneration != currentBufferIdGeneration;
|
||||||
|
const Bool attributesDirty =
|
||||||
|
bufferIdsRemitted || !m_hasSyncedConfigVersion || m_syncedConfigVersion != currentConfigVersion;
|
||||||
|
// Identity joins the version compare: the slot version is a wrapping Uint16,
|
||||||
|
// so a wrapped-back count with a different buffer bound must still read dirty.
|
||||||
|
const MG_State::GLState::BufferObject* currentIndexBufferObject =
|
||||||
|
stateVAOObject->GetIndexBufferBindingSlot().GetBoundObject().get();
|
||||||
|
const Bool indexBufferDirty = bufferIdsRemitted ||
|
||||||
|
currentIndexBufferVersion != m_syncedIndexBufferVersion ||
|
||||||
|
currentIndexBufferObject != m_syncedIndexBufferObject;
|
||||||
|
|
||||||
// The baseInstance shift lives in the attribute offsets the driver already holds, so
|
// The baseInstance shift lives in the attribute offsets the driver already holds, so
|
||||||
// a change of baseInstance has to re-emit the divisor'd arrays even when the frontend
|
// a change of baseInstance has to re-emit the divisor'd arrays even when the frontend
|
||||||
@@ -1675,9 +1701,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Bool needsSyncFormat = allAttributeVersions[attribIndex].FormatVersion !=
|
Bool needsSyncFormat = bufferIdsRemitted || allAttributeVersions[attribIndex].FormatVersion !=
|
||||||
m_syncedAttributeVersions[attribIndex].FormatVersion;
|
m_syncedAttributeVersions[attribIndex].FormatVersion;
|
||||||
Bool needsSyncBuffer = allAttributeVersions[attribIndex].BufferVersion !=
|
Bool needsSyncBuffer = bufferIdsRemitted || allAttributeVersions[attribIndex].BufferVersion !=
|
||||||
m_syncedAttributeVersions[attribIndex].BufferVersion;
|
m_syncedAttributeVersions[attribIndex].BufferVersion;
|
||||||
if (!needsSyncFormat && !needsSyncBuffer && !needsSyncBaseInstance) continue;
|
if (!needsSyncFormat && !needsSyncBuffer && !needsSyncBaseInstance) continue;
|
||||||
|
|
||||||
@@ -1798,6 +1824,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
|
|
||||||
if (indexBufferSynced) {
|
if (indexBufferSynced) {
|
||||||
m_syncedIndexBufferVersion = currentIndexBufferVersion;
|
m_syncedIndexBufferVersion = currentIndexBufferVersion;
|
||||||
|
m_syncedIndexBufferObject = currentIndexBufferObject;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1809,6 +1836,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
if (emitAttributes) {
|
if (emitAttributes) {
|
||||||
m_syncedFetchBaseInstance = fetchBaseInstance;
|
m_syncedFetchBaseInstance = fetchBaseInstance;
|
||||||
}
|
}
|
||||||
|
m_syncedBufferIdGeneration = currentBufferIdGeneration;
|
||||||
}
|
}
|
||||||
|
|
||||||
void BackendVertexArrayObject::SyncClientSideAttributesForDrawArrays(
|
void BackendVertexArrayObject::SyncClientSideAttributesForDrawArrays(
|
||||||
@@ -1946,6 +1974,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
void BackendTextureObject::RecreateBackendTexture() {
|
void BackendTextureObject::RecreateBackendTexture() {
|
||||||
if (m_backendTextureId != 0) {
|
if (m_backendTextureId != 0) {
|
||||||
ScratchFBOImpl::NoteTextureIdDeleted(m_backendTextureId);
|
ScratchFBOImpl::NoteTextureIdDeleted(m_backendTextureId);
|
||||||
|
// Application FBO twins that attached the dying id memoize on FRONTEND
|
||||||
|
// attachment versions, which this backend-side re-mint does not move;
|
||||||
|
// without this bump their driver FBOs would keep the deleted name
|
||||||
|
// attached forever (see g_attachmentBackendIdGeneration).
|
||||||
|
++FramebufferImpl::g_attachmentBackendIdGeneration;
|
||||||
if (m_contextGeneration == g_backendContextGeneration) {
|
if (m_contextGeneration == g_backendContextGeneration) {
|
||||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||||
}
|
}
|
||||||
@@ -3517,6 +3550,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
m_backendReadBuffer = GL_NONE;
|
m_backendReadBuffer = GL_NONE;
|
||||||
std::fill(m_syncedFrontendAttachmentVersions.begin(), m_syncedFrontendAttachmentVersions.end(),
|
std::fill(m_syncedFrontendAttachmentVersions.begin(), m_syncedFrontendAttachmentVersions.end(),
|
||||||
static_cast<Uint16>(~0u));
|
static_cast<Uint16>(~0u));
|
||||||
|
// Every attachment version is invalidated above, so the next walk re-attaches
|
||||||
|
// everything regardless; stamp the generation so it does not re-arm twice.
|
||||||
|
m_syncedBackendIdGeneration = g_attachmentBackendIdGeneration;
|
||||||
}
|
}
|
||||||
|
|
||||||
static Bool SyncAttachmentObject(GLenum glFBOTarget,
|
static Bool SyncAttachmentObject(GLenum glFBOTarget,
|
||||||
@@ -4005,6 +4041,15 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// -------------------- Attach texture to backend FBO -----------------------
|
// -------------------- Attach texture to backend FBO -----------------------
|
||||||
|
// A backend texture id was re-minted since this twin's last walk
|
||||||
|
// (RecreateBackendTexture): any point here may still hold the dead id while
|
||||||
|
// its frontend attachment version is unchanged, so the memo below would skip
|
||||||
|
// exactly the attachment that needs repair. Re-arm every point first.
|
||||||
|
if (m_syncedBackendIdGeneration != g_attachmentBackendIdGeneration) {
|
||||||
|
std::fill(m_syncedFrontendAttachmentVersions.begin(), m_syncedFrontendAttachmentVersions.end(),
|
||||||
|
static_cast<Uint16>(~0u));
|
||||||
|
m_syncedBackendIdGeneration = g_attachmentBackendIdGeneration;
|
||||||
|
}
|
||||||
const auto& attachments = stateFBOObject->GetAllAttachmentObjects();
|
const auto& attachments = stateFBOObject->GetAllAttachmentObjects();
|
||||||
const auto& attachmentVersions = stateFBOObject->GetAllFramebufferAttachmentVersions();
|
const auto& attachmentVersions = stateFBOObject->GetAllFramebufferAttachmentVersions();
|
||||||
for (SizeT i = 0; i < attachments.size(); ++i) {
|
for (SizeT i = 0; i < attachments.size(); ++i) {
|
||||||
@@ -4093,6 +4138,18 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The walk itself can re-mint an id (SyncAttachmentObject ->
|
||||||
|
// SyncMipmapsToBackend -> RecreateBackendTexture), invalidating points this
|
||||||
|
// walk already attached or version-skipped - e.g. one texture attached at two
|
||||||
|
// points. Re-enter until the generation is quiescent: every pass syncs each
|
||||||
|
// dirty texture clean, so each repeat finds strictly fewer re-mints and the
|
||||||
|
// common case (no re-mint) never takes a second pass. The head's draw/read-
|
||||||
|
// buffer syncs are memoized against their own shadows, so a repeat re-walks
|
||||||
|
// only the attachments.
|
||||||
|
if (m_syncedBackendIdGeneration != g_attachmentBackendIdGeneration) {
|
||||||
|
SyncToBackend(stateFBOObject, asTarget);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
GLenum BackendFramebufferObject::GetBackendAttachmentType(FramebufferAttachmentType frontendAtt) const {
|
GLenum BackendFramebufferObject::GetBackendAttachmentType(FramebufferAttachmentType frontendAtt) const {
|
||||||
@@ -4119,6 +4176,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
Array<Uint16, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedObjectVersions = {0};
|
Array<Uint16, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedObjectVersions = {0};
|
||||||
Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
||||||
g_fboSyncedObjects = {};
|
g_fboSyncedObjects = {};
|
||||||
|
Uint64 g_attachmentBackendIdGeneration = 0;
|
||||||
|
Array<Uint64, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedBackendIdGenerations = {0};
|
||||||
} // namespace FramebufferImpl
|
} // namespace FramebufferImpl
|
||||||
|
|
||||||
namespace ScratchFBOImpl {
|
namespace ScratchFBOImpl {
|
||||||
|
|||||||
@@ -346,6 +346,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// client-attribute staging buffers): scrub every buffer-binding shadow that
|
// client-attribute staging buffers): scrub every buffer-binding shadow that
|
||||||
// could false-skip when the name is recycled.
|
// could false-skip when the name is recycled.
|
||||||
void NoteBufferIdDeleted(Uint id);
|
void NoteBufferIdDeleted(Uint id);
|
||||||
|
// Bumped whenever a live GLESBufferResource's driver id is retired and re-minted
|
||||||
|
// while its frontend buffer stays alive (persistent-map adoption, immutable-store
|
||||||
|
// retire). The VAO twins' baked glVertexAttribPointer / element-array bindings
|
||||||
|
// key on FRONTEND versions, which a backend-side re-mint does not move - without
|
||||||
|
// this generation the driver VAO would keep fetching through the deleted id (or
|
||||||
|
// its retained store) forever. Compared and stamped by
|
||||||
|
// BackendVertexArrayObject::SyncToBackend.
|
||||||
|
extern Uint64 g_bufferBackendIdGeneration;
|
||||||
// Redundant-bind cache for INDEXED buffer bindings (glBindBufferBase/Range on
|
// Redundant-bind cache for INDEXED buffer bindings (glBindBufferBase/Range on
|
||||||
// GL_UNIFORM_BUFFER / GL_SHADER_STORAGE_BUFFER): skips the GL call when the
|
// GL_UNIFORM_BUFFER / GL_SHADER_STORAGE_BUFFER): skips the GL call when the
|
||||||
// (id, range) already at that index matches, like the array-buffer/texture/
|
// (id, range) already at that index matches, like the array-buffer/texture/
|
||||||
@@ -465,6 +473,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
Array<Uint, MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS> m_clientAttributeBufferIds;
|
Array<Uint, MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS> m_clientAttributeBufferIds;
|
||||||
Bool m_isInitialized = false;
|
Bool m_isInitialized = false;
|
||||||
Uint16 m_syncedIndexBufferVersion = 0;
|
Uint16 m_syncedIndexBufferVersion = 0;
|
||||||
|
// Identity of the buffer the version above was stamped against. Raw and never
|
||||||
|
// dereferenced: the slot version is a wrapping Uint16 (see the ResolvedDrawBuffers
|
||||||
|
// IBO memo and the packed_pixels postmortem at BindCurrentFBO), so the version
|
||||||
|
// alone would read a wrapped-back count with a different buffer bound as clean.
|
||||||
|
const MG_State::GLState::BufferObject* m_syncedIndexBufferObject = nullptr;
|
||||||
// Aggregate gate over the per-attribute walk below: the frontend bumps its config
|
// Aggregate gate over the per-attribute walk below: the frontend bumps its config
|
||||||
// version on every per-attribute version bump (the three Bump*Version functions are
|
// version on every per-attribute version bump (the three Bump*Version functions are
|
||||||
// its only writers), so an unchanged config version proves every per-attribute
|
// its only writers), so an unchanged config version proves every per-attribute
|
||||||
@@ -480,6 +493,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// Kept here because it describes what was last EMITTED, which is what the next sync
|
// Kept here because it describes what was last EMITTED, which is what the next sync
|
||||||
// has to correct.
|
// has to correct.
|
||||||
Uint32 m_syncedFetchBaseInstance = 0;
|
Uint32 m_syncedFetchBaseInstance = 0;
|
||||||
|
// BufferImpl::g_bufferBackendIdGeneration as of this twin's last emit. A
|
||||||
|
// mismatch means some live buffer's driver id was re-minted since; the ids
|
||||||
|
// baked into the driver VAO's attribute/element bindings may be dead even
|
||||||
|
// though every frontend version matches, so the next sync re-emits them all.
|
||||||
|
Uint64 m_syncedBufferIdGeneration = 0;
|
||||||
};
|
};
|
||||||
|
|
||||||
extern StateBackendObjectRegistry<MG_State::GLState::VertexArrayObject, BackendVertexArrayObject>
|
extern StateBackendObjectRegistry<MG_State::GLState::VertexArrayObject, BackendVertexArrayObject>
|
||||||
@@ -799,6 +817,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
|
|
||||||
using FramebufferObject = MG_State::GLState::FramebufferObject;
|
using FramebufferObject = MG_State::GLState::FramebufferObject;
|
||||||
FramebufferObject::FramebufferAttachmentVersionArray m_syncedFrontendAttachmentVersions = {0};
|
FramebufferObject::FramebufferAttachmentVersionArray m_syncedFrontendAttachmentVersions = {0};
|
||||||
|
// g_attachmentBackendIdGeneration as of this twin's last attachment walk. A
|
||||||
|
// mismatch means some backend texture id was re-minted since, and any of this
|
||||||
|
// twin's attachment points may still hold the dead id even though the frontend
|
||||||
|
// attachment versions match - so the walk re-attaches everything first.
|
||||||
|
Uint64 m_syncedBackendIdGeneration = 0;
|
||||||
};
|
};
|
||||||
|
|
||||||
extern StateBackendObjectRegistry<MG_State::GLState::FramebufferObject, BackendFramebufferObject>
|
extern StateBackendObjectRegistry<MG_State::GLState::FramebufferObject, BackendFramebufferObject>
|
||||||
@@ -888,6 +911,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
extern Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
extern Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
||||||
g_fboSyncedObjects;
|
g_fboSyncedObjects;
|
||||||
|
|
||||||
|
// Bumped whenever a live backend texture's driver id is re-minted while its
|
||||||
|
// frontend texture may still be attached to application FBOs
|
||||||
|
// (BackendTextureObject::RecreateBackendTexture - e.g. a respecify of a texture
|
||||||
|
// whose backend storage went immutable). The FBO twins' attachment memos key on
|
||||||
|
// FRONTEND attachment versions, which a backend-side re-mint does not move, so
|
||||||
|
// the driver FBO would keep the deleted texture name attached forever. The
|
||||||
|
// SyncCurrentFBO gate compares this generation (below) to re-enter the sync,
|
||||||
|
// and each twin re-arms its per-attachment memo on a mismatch (SyncToBackend).
|
||||||
|
extern Uint64 g_attachmentBackendIdGeneration;
|
||||||
|
// What g_attachmentBackendIdGeneration was when SyncCurrentFBO last stamped each
|
||||||
|
// target; part of the synced tuple above.
|
||||||
|
extern Array<Uint64, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedBackendIdGenerations;
|
||||||
|
|
||||||
// Driver-level READ/DRAW framebuffer-binding shadow. Every backend
|
// Driver-level READ/DRAW framebuffer-binding shadow. Every backend
|
||||||
// glBindFramebuffer routes through BindFramebufferId so scoped helpers can
|
// glBindFramebuffer routes through BindFramebufferId so scoped helpers can
|
||||||
// save/restore the current binding without a glGetIntegerv round-trip (that
|
// save/restore the current binding without a glGetIntegerv round-trip (that
|
||||||
|
|||||||
@@ -69,6 +69,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
// slot's ownership unambiguous.
|
// slot's ownership unambiguous.
|
||||||
Uint64 programLifetimeId = 0;
|
Uint64 programLifetimeId = 0;
|
||||||
Uint32 backendStateVersion = 0;
|
Uint32 backendStateVersion = 0;
|
||||||
|
// glShaderStorageBlockBinding deliberately does NOT bump the backend state
|
||||||
|
// version, and the pipeline composite is unnamed so the in-place patch in
|
||||||
|
// DirectVulkan::ShaderStorageBlockBinding can never reach its slot - the
|
||||||
|
// mirror replay bumps only the program's block-binding version. Without this
|
||||||
|
// key the composite's slot kept serving the pre-rebind block.binding.
|
||||||
|
Uint32 blockBindingVersion = 0;
|
||||||
Vector<StorageBlockResource> storageBlocks;
|
Vector<StorageBlockResource> storageBlocks;
|
||||||
Vector<BufferVariableResource> bufferVariables;
|
Vector<BufferVariableResource> bufferVariables;
|
||||||
GLint computeWorkGroupSize[3] = {1, 1, 1};
|
GLint computeWorkGroupSize[3] = {1, 1, 1};
|
||||||
@@ -156,18 +162,33 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
auto& cache = g_programResourceCaches[program.GetExternalIndex()];
|
auto& cache = g_programResourceCaches[program.GetExternalIndex()];
|
||||||
const Uint64 programLifetimeId = program.GetLifetimeId();
|
const Uint64 programLifetimeId = program.GetLifetimeId();
|
||||||
const Uint32 backendStateVersion = program.GetBackendStateVersion();
|
const Uint32 backendStateVersion = program.GetBackendStateVersion();
|
||||||
|
const Uint32 blockBindingVersion = program.GetBlockBindingVersion();
|
||||||
// The lifetime id must match too: a new program that reuses a deleted
|
// The lifetime id must match too: a new program that reuses a deleted
|
||||||
// program's name and happens to land on the same backendStateVersion (both
|
// program's name and happens to land on the same backendStateVersion (both
|
||||||
// count from zero) would otherwise be served the dead program's reflection.
|
// count from zero) would otherwise be served the dead program's reflection.
|
||||||
if (cache.programLifetimeId == programLifetimeId &&
|
if (cache.programLifetimeId == programLifetimeId &&
|
||||||
cache.backendStateVersion == backendStateVersion &&
|
cache.backendStateVersion == backendStateVersion &&
|
||||||
(!cache.storageBlocks.empty() || !cache.bufferVariables.empty())) {
|
(!cache.storageBlocks.empty() || !cache.bufferVariables.empty())) {
|
||||||
|
if (cache.blockBindingVersion != blockBindingVersion) {
|
||||||
|
// Only the block bindings moved (glShaderStorageBlockBinding, or the
|
||||||
|
// pipeline composite's mirror replay - neither touches the backend
|
||||||
|
// state version): the reflection itself is unchanged, so re-apply the
|
||||||
|
// overrides by name instead of re-running spirv-reflect. Overrides
|
||||||
|
// only ever accumulate, so a block without one still holds its
|
||||||
|
// declared binding.
|
||||||
|
for (auto& block : cache.storageBlocks) {
|
||||||
|
const Int rebound = program.GetShaderStorageBlockBindingOverride(block.name);
|
||||||
|
if (rebound >= 0) block.binding = static_cast<Uint32>(rebound);
|
||||||
|
}
|
||||||
|
cache.blockBindingVersion = blockBindingVersion;
|
||||||
|
}
|
||||||
return cache;
|
return cache;
|
||||||
}
|
}
|
||||||
|
|
||||||
cache = {};
|
cache = {};
|
||||||
cache.programLifetimeId = programLifetimeId;
|
cache.programLifetimeId = programLifetimeId;
|
||||||
cache.backendStateVersion = backendStateVersion;
|
cache.backendStateVersion = backendStateVersion;
|
||||||
|
cache.blockBindingVersion = blockBindingVersion;
|
||||||
|
|
||||||
Vector<SpvReflectShaderModule> modules;
|
Vector<SpvReflectShaderModule> modules;
|
||||||
Vector<Bool> validModules;
|
Vector<Bool> validModules;
|
||||||
|
|||||||
@@ -3213,6 +3213,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
Vector<Uint> patchedSpirv;
|
Vector<Uint> patchedSpirv;
|
||||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(
|
if (MG_Util::ShaderTranspiler::ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(
|
||||||
moduleSpirvs[i], patchedSpirv, m_subgroupPolicy.nativeSubgroupSize,
|
moduleSpirvs[i], patchedSpirv, m_subgroupPolicy.nativeSubgroupSize,
|
||||||
|
m_subgroupPolicy.maxComputeSharedMemoryBytes,
|
||||||
enableSpirvValidation)) {
|
enableSpirvValidation)) {
|
||||||
moduleSpirvs[i] = std::move(patchedSpirv);
|
moduleSpirvs[i] = std::move(patchedSpirv);
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
@@ -287,8 +287,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
if (m_frameBoundaryCounter - it->second->lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
if (m_frameBoundaryCounter - it->second->lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||||
it = m_cache.erase(it);
|
it = m_cache.erase(it);
|
||||||
// Invalidate every VAO's state-pointer memo: the erased node's
|
// Invalidate every VAO's state-pointer memo: the erased node's
|
||||||
// address may be reused by a future insert.
|
// address may be reused by a future insert. Advance through the
|
||||||
++m_evictionEpoch;
|
// process-wide source so the value stays unique across factory
|
||||||
|
// instances (see the member comment).
|
||||||
|
m_evictionEpoch = ++s_evictionEpochSource;
|
||||||
} else {
|
} else {
|
||||||
++it;
|
++it;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -125,7 +125,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
// construction); a memo is honored only while its recorded epoch
|
// construction); a memo is honored only while its recorded epoch
|
||||||
// matches, so an evicted entry can never be dereferenced through a
|
// matches, so an evicted entry can never be dereferenced through a
|
||||||
// stale memo.
|
// stale memo.
|
||||||
Uint64 m_evictionEpoch = 1;
|
//
|
||||||
|
// Drawn from a process-wide source, never a per-instance counter: the VAO
|
||||||
|
// memos outlive this factory (they live on pGLContext's VAOs, the renderer
|
||||||
|
// is destroyed and recreated on EGL surface release/re-create), so a fresh
|
||||||
|
// factory restarting at a dead factory's epoch value would honor its
|
||||||
|
// dangling entry pointers. The constructor takes a value strictly greater
|
||||||
|
// than anything a predecessor ever stamped, so a dead factory's memo can
|
||||||
|
// never compare equal here - the same never-reused idiom as the lifetime ids.
|
||||||
|
// Single-threaded like the rest of the factory (renderer-thread only).
|
||||||
|
static inline Uint64 s_evictionEpochSource = 0;
|
||||||
|
Uint64 m_evictionEpoch = ++s_evictionEpochSource;
|
||||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||||
};
|
};
|
||||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||||
|
|||||||
@@ -166,7 +166,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
void VkClearManager::MergeClearPayload(ClearAttachmentPayload& dst, const ClearAttachmentPayload& src) {
|
void VkClearManager::MergeClearPayload(ClearAttachmentPayload& dst, const ClearAttachmentPayload& src) {
|
||||||
dst.mask |= src.mask;
|
dst.mask |= src.mask;
|
||||||
if ((src.mask & GL_COLOR_BUFFER_BIT) != 0) {
|
if ((src.mask & GL_COLOR_BUFFER_BIT) != 0) {
|
||||||
|
// The whole colour story travels together (same rule as
|
||||||
|
// VkRenderPassManager::QueueRenderbufferClear): a glClearBufferiv/uiv
|
||||||
|
// payload carries its value in colorInt/colorUint and its branch selector
|
||||||
|
// in colorEncoding - dropping them here would leave the pending clear
|
||||||
|
// reading as an all-zero float one.
|
||||||
dst.color = src.color;
|
dst.color = src.color;
|
||||||
|
dst.colorEncoding = src.colorEncoding;
|
||||||
|
dst.colorInt = src.colorInt;
|
||||||
|
dst.colorUint = src.colorUint;
|
||||||
}
|
}
|
||||||
if ((src.mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
if ((src.mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
||||||
dst.depth = src.depth;
|
dst.depth = src.depth;
|
||||||
|
|||||||
@@ -831,6 +831,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
// recreated since (texture + renderbuffer image epochs), and no pending clear (which alters
|
// recreated since (texture + renderbuffer image epochs), and no pending clear (which alters
|
||||||
// load ops). Any of these differing forces the full recompute below. Portable to VK 1.1.
|
// load ops). Any of these differing forces the full recompute below. Portable to VK 1.1.
|
||||||
if (activeRenderPass != nullptr && m_rpFastValid && m_rpFastFbo == &fbo &&
|
if (activeRenderPass != nullptr && m_rpFastValid && m_rpFastFbo == &fbo &&
|
||||||
|
m_rpFastFboLifetimeId == fbo.GetLifetimeId() &&
|
||||||
m_rpFastFboVersion == fbo.GetObjectVersion() && m_rpFastSwapchainIndex == swapchainImageIndex &&
|
m_rpFastFboVersion == fbo.GetObjectVersion() && m_rpFastSwapchainIndex == swapchainImageIndex &&
|
||||||
m_rpFastTexEpoch == m_textureManager.GetTextureImageEpoch() &&
|
m_rpFastTexEpoch == m_textureManager.GetTextureImageEpoch() &&
|
||||||
m_rpFastRbEpoch == m_renderbufferImageEpoch &&
|
m_rpFastRbEpoch == m_renderbufferImageEpoch &&
|
||||||
@@ -855,6 +856,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
// epochs AFTER ComputeHash: its attachment SyncTexture can create an image (bump the epoch).
|
// epochs AFTER ComputeHash: its attachment SyncTexture can create an image (bump the epoch).
|
||||||
m_rpFastValid = true;
|
m_rpFastValid = true;
|
||||||
m_rpFastFbo = &fbo;
|
m_rpFastFbo = &fbo;
|
||||||
|
m_rpFastFboLifetimeId = fbo.GetLifetimeId();
|
||||||
m_rpFastFboVersion = fbo.GetObjectVersion();
|
m_rpFastFboVersion = fbo.GetObjectVersion();
|
||||||
m_rpFastSwapchainIndex = swapchainImageIndex;
|
m_rpFastSwapchainIndex = swapchainImageIndex;
|
||||||
m_rpFastTexEpoch = m_textureManager.GetTextureImageEpoch();
|
m_rpFastTexEpoch = m_textureManager.GetTextureImageEpoch();
|
||||||
@@ -1507,7 +1509,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
ClearAttachmentPayload clearPayload{};
|
ClearAttachmentPayload clearPayload{};
|
||||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||||
if (pending.hasInlinePayload) {
|
if (pending.hasInlinePayload) {
|
||||||
|
// The inline payload was snapshotted when the entry was CREATED, but the
|
||||||
|
// clear VALUE is not part of the entry's hash - a cache hit with a newer
|
||||||
|
// glClear would replay the creation-time value and drop the new one (the
|
||||||
|
// texture path below is immune because it re-reads the live payload).
|
||||||
|
// Same defense as ClearAttachmentsOnActiveRenderPass: prefer the live
|
||||||
|
// pending clear, fall back to the snapshot only when none is queued.
|
||||||
|
if (s_renderPassManager != nullptr &&
|
||||||
|
s_renderPassManager->GetPendingRenderbufferClear(pending.renderbuffer, clearPayload)) {
|
||||||
|
if ((clearPayload.mask & GL_COLOR_BUFFER_BIT) != 0 && pending.renderbuffer != nullptr &&
|
||||||
|
MG_Util::GetBaseInternalFormatComponentCount(pending.renderbuffer->GetInternalFormat()) ==
|
||||||
|
3) {
|
||||||
|
// RGB renderbuffers are backed by an RGBA image; the missing alpha reads as 1.
|
||||||
|
ForceOpaqueClearAlpha(clearPayload);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
clearPayload = pending.inlinePayload;
|
clearPayload = pending.inlinePayload;
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
if (pending.key.texture == nullptr ||
|
if (pending.key.texture == nullptr ||
|
||||||
!s_clearManager->GetPendingClear(pending.key, clearPayload, liveTexture)) {
|
!s_clearManager->GetPendingClear(pending.key, clearPayload, liveTexture)) {
|
||||||
|
|||||||
@@ -289,6 +289,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
// or a pending clear. Portable to Vulkan 1.1 (no dynamic_rendering / imageless FB needed).
|
// or a pending clear. Portable to Vulkan 1.1 (no dynamic_rendering / imageless FB needed).
|
||||||
Bool m_rpFastValid = false;
|
Bool m_rpFastValid = false;
|
||||||
const MG_State::GLState::FramebufferObject* m_rpFastFbo = nullptr;
|
const MG_State::GLState::FramebufferObject* m_rpFastFbo = nullptr;
|
||||||
|
// The FBO's never-reused lifetime id joins the raw pointer + Uint16 version:
|
||||||
|
// a deleted FBO reallocated at the same address whose fresh setup performed
|
||||||
|
// the same number of version bumps would otherwise compare equal (both count
|
||||||
|
// from 0), serving the dead framebuffer's pass to the new object.
|
||||||
|
Uint64 m_rpFastFboLifetimeId = 0;
|
||||||
Uint16 m_rpFastFboVersion = 0;
|
Uint16 m_rpFastFboVersion = 0;
|
||||||
Uint32 m_rpFastSwapchainIndex = 0;
|
Uint32 m_rpFastSwapchainIndex = 0;
|
||||||
Uint64 m_rpFastTexEpoch = 0;
|
Uint64 m_rpFastTexEpoch = 0;
|
||||||
|
|||||||
@@ -1993,6 +1993,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
texture.GetExternalIndex(),
|
texture.GetExternalIndex(),
|
||||||
MG_Util::ConvertTextureUploadTargetToString(uploadTarget).c_str(),
|
MG_Util::ConvertTextureUploadTargetToString(uploadTarget).c_str(),
|
||||||
static_cast<Int>(format), static_cast<Uint32>(imageInfo.usage));
|
static_cast<Int>(format), static_cast<Uint32>(imageInfo.usage));
|
||||||
|
// The preserved image was written by GPU work that may still be in flight
|
||||||
|
// (preserve requires layout != UNDEFINED); park it on the deferred ring
|
||||||
|
// like every other destruction path instead of letting the unique_ptr
|
||||||
|
// destroy it synchronously under the GPU.
|
||||||
|
if (preservedResource) {
|
||||||
|
DeferResourceRelease(Move(*preservedResource));
|
||||||
|
}
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -2015,6 +2022,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
static_cast<Int>(imageInfo.samples), static_cast<Int>(imageInfo.format));
|
static_cast<Int>(imageInfo.samples), static_cast<Int>(imageInfo.format));
|
||||||
resource.image = VK_NULL_HANDLE;
|
resource.image = VK_NULL_HANDLE;
|
||||||
resource.allocation = nullptr;
|
resource.allocation = nullptr;
|
||||||
|
// Same as the probe failure above: the preserved live image must go through
|
||||||
|
// the deferred ring, never a synchronous destructor while frames that
|
||||||
|
// reference it are still in flight.
|
||||||
|
if (preservedResource) {
|
||||||
|
DeferResourceRelease(Move(*preservedResource));
|
||||||
|
}
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
++m_textureImageEpoch; // a new attachment image invalidates cached render passes
|
++m_textureImageEpoch; // a new attachment image invalidates cached render passes
|
||||||
|
|||||||
@@ -3323,6 +3323,11 @@ void main() {
|
|||||||
indexView.indexByteSize > bufferSize - indexView.indexByteOffset) {
|
indexView.indexByteSize > bufferSize - indexView.indexByteOffset) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
// Recorded-but-unexecuted GPU writes (XFB capture, SSBO, storage texel
|
||||||
|
// buffer) land in the coherent mapping this scan is about to read;
|
||||||
|
// submit-and-wait first, exactly like the restart-index rewrite does.
|
||||||
|
// A no-op unless the gpu-write flag is set.
|
||||||
|
indexBufferShared->SyncGpuWrites();
|
||||||
indexBufferShared->SyncPersistentMappedRange();
|
indexBufferShared->SyncPersistentMappedRange();
|
||||||
indexBytes = indexBufferShared->MappedData() + indexView.indexByteOffset;
|
indexBytes = indexBufferShared->MappedData() + indexView.indexByteOffset;
|
||||||
} else {
|
} else {
|
||||||
@@ -3560,6 +3565,16 @@ void main() {
|
|||||||
const Uint8* sourceData, SizeT sourceStride,
|
const Uint8* sourceData, SizeT sourceStride,
|
||||||
SizeT elementSize, SizeT elementCount,
|
SizeT elementSize, SizeT elementCount,
|
||||||
BufferSlice& outSlice) -> Bool {
|
BufferSlice& outSlice) -> Bool {
|
||||||
|
// A resolved stride of 0 is the binding model's "never advance" (see the
|
||||||
|
// factory's layout notes): exactly one element is converted and every vertex
|
||||||
|
// reads it. That single element is read at offset 0, so the stride is never
|
||||||
|
// actually used - but both converters reject 0 as a degenerate input, which
|
||||||
|
// made the documented single-element conversion unreachable and silently
|
||||||
|
// dropped every draw using such a binding. Substitute the element's own
|
||||||
|
// size; the caller's cache key still carries the distinct stride 0.
|
||||||
|
if (sourceStride == 0 && elementCount == 1) {
|
||||||
|
sourceStride = elementSize;
|
||||||
|
}
|
||||||
const void* uploadData = nullptr;
|
const void* uploadData = nullptr;
|
||||||
VkDeviceSize uploadSize = 0;
|
VkDeviceSize uploadSize = 0;
|
||||||
switch (conversion) {
|
switch (conversion) {
|
||||||
@@ -3693,6 +3708,12 @@ void main() {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A GPU-written source (XFB capture, SSBO, storage texel buffer) has its
|
||||||
|
// bytes produced by commands that are merely RECORDED at this point, and
|
||||||
|
// MappedData() aliases the coherent GPU memory they will write into -
|
||||||
|
// converting now would read pre-write garbage. Submit-and-wait first,
|
||||||
|
// mirroring the restart-index rewrite; a flag-test no-op otherwise.
|
||||||
|
sourceBufferShared->SyncGpuWrites();
|
||||||
sourceBufferShared->SyncPersistentMappedRange();
|
sourceBufferShared->SyncPersistentMappedRange();
|
||||||
const SizeT availableElementCount =
|
const SizeT availableElementCount =
|
||||||
sourceStride == 0 ? 1 : 1 + (sourceSize - baseOffset - elementSize) / sourceStride;
|
sourceStride == 0 ? 1 : 1 + (sourceSize - baseOffset - elementSize) / sourceStride;
|
||||||
@@ -3986,7 +4007,7 @@ void main() {
|
|||||||
// Skips the per-draw GetBackendResource chase into a cold resource object.
|
// Skips the per-draw GetBackendResource chase into a cold resource object.
|
||||||
Bool sliceStillValid = false;
|
Bool sliceStillValid = false;
|
||||||
const Uint64 frameSerial = m_bufferManager.GetFrameSerial();
|
const Uint64 frameSerial = m_bufferManager.GetFrameSerial();
|
||||||
if (indexMemo->indexFrameSerial == frameSerial &&
|
if (indexMemo->indexFrameSerial == frameSerial && !indexMemo->indexBufferMapped &&
|
||||||
indexMemo->indexSliceEpochCounter == m_bufferManager.GetSliceEpochCounter()) {
|
indexMemo->indexSliceEpochCounter == m_bufferManager.GetSliceEpochCounter()) {
|
||||||
sliceStillValid = true;
|
sliceStillValid = true;
|
||||||
}
|
}
|
||||||
@@ -4049,6 +4070,9 @@ void main() {
|
|||||||
indexMemo->indexVkBuffer = slice.buffer;
|
indexMemo->indexVkBuffer = slice.buffer;
|
||||||
indexMemo->indexSliceOffset = slice.offset;
|
indexMemo->indexSliceOffset = slice.offset;
|
||||||
indexMemo->indexFrameSerial = m_bufferManager.GetFrameSerial();
|
indexMemo->indexFrameSerial = m_bufferManager.GetFrameSerial();
|
||||||
|
// A host-mapped EBO can mutate its shadow with no epoch bump; the hit
|
||||||
|
// path declines on this flag (mirror of anyBufferMapped).
|
||||||
|
indexMemo->indexBufferMapped = indexBufferShared->IsMapped();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
const VkDeviceSize indexBindOffset =
|
const VkDeviceSize indexBindOffset =
|
||||||
@@ -5718,6 +5742,22 @@ void main() {
|
|||||||
if (program.GetBackendStateVersion() != snap.programVersion) {
|
if (program.GetBackendStateVersion() != snap.programVersion) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
// glBegin/EndTransformFeedback moves no key this fast path otherwise observes
|
||||||
|
// (the design makes capture a compile-option FLAG precisely because no version
|
||||||
|
// bumps, VulkanRenderer.h's pipeline-memo note) - but the snapshot bakes that
|
||||||
|
// flag into resolvedTransformFlags and the pipeline. Recompute the one dynamic
|
||||||
|
// bit (the full path's exact predicate) and decline on a mismatch, or the first
|
||||||
|
// captured draw after glBeginTransformFeedback would bind the undecorated
|
||||||
|
// variant and silently capture nothing while the CPU bookkeeping advances.
|
||||||
|
const Bool wantsXfbCapture = m_transformFeedbackFeatureEnabled &&
|
||||||
|
MG_State::pGLContext->IsTransformFeedbackActive() &&
|
||||||
|
program.GetTransformFeedbackVaryingCount() > 0;
|
||||||
|
const Bool snapHasXfbCapture =
|
||||||
|
static_cast<Bool>(ProgramFactory::CompileOptionFlags(snap.resolvedTransformFlags) &
|
||||||
|
ProgramFactory::CompileOptionBit::XfbCapture);
|
||||||
|
if (wantsXfbCapture != snapHasXfbCapture) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
// A changed VAO does NOT decline: the VAO only feeds the pipeline's vertex
|
// A changed VAO does NOT decline: the VAO only feeds the pipeline's vertex
|
||||||
// input state (re-resolved below through the layout-keyed memo, so N VAOs
|
// input state (re-resolved below through the layout-keyed memo, so N VAOs
|
||||||
// sharing one attribute layout share one pipeline) and the vertex/index
|
// sharing one attribute layout share one pipeline) and the vertex/index
|
||||||
@@ -5731,6 +5771,7 @@ void main() {
|
|||||||
const auto& drawFbo =
|
const auto& drawFbo =
|
||||||
MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||||
if (static_cast<const void*>(drawFbo.get()) != snap.drawFbo ||
|
if (static_cast<const void*>(drawFbo.get()) != snap.drawFbo ||
|
||||||
|
drawFbo->GetLifetimeId() != snap.drawFboLifetimeId ||
|
||||||
drawFbo->GetObjectVersion() != snap.fboVersion) {
|
drawFbo->GetObjectVersion() != snap.fboVersion) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -5914,8 +5955,14 @@ void main() {
|
|||||||
}
|
}
|
||||||
const Uint64 samplingResolutionGeneration = MG_State::pGLContext->GetSamplingResolutionGeneration();
|
const Uint64 samplingResolutionGeneration = MG_State::pGLContext->GetSamplingResolutionGeneration();
|
||||||
if (samplingResolutionGeneration != snap.samplingResolutionGeneration) {
|
if (samplingResolutionGeneration != snap.samplingResolutionGeneration) {
|
||||||
snap.samplingResolutionGeneration = samplingResolutionGeneration;
|
// Decline, not re-arm: snap.resolvedTransformFlags bakes the
|
||||||
samplerDescriptorsUnchanged = false;
|
// ExplicitLod0Sampling verdict, which reads the effective sampler's
|
||||||
|
// filters/aniso/LOD range - exactly the state this counter tracks.
|
||||||
|
// Re-arming the stamp here would rebuild the descriptors but keep the
|
||||||
|
// stale SPIR-V variant forever (every later draw compares equal again).
|
||||||
|
// Same shape as the erase-epoch declines above; costs one full-path draw
|
||||||
|
// per sampler/shape change, and the full path's LOD memo re-probes.
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Everything the full path would re-resolve is provably unchanged - or, for
|
// Everything the full path would re-resolve is provably unchanged - or, for
|
||||||
@@ -6095,11 +6142,18 @@ void main() {
|
|||||||
const Uint64 lodProgramLifetimeId = program.GetLifetimeId();
|
const Uint64 lodProgramLifetimeId = program.GetLifetimeId();
|
||||||
const Uint32 lodProgramVersion = program.GetBackendStateVersion();
|
const Uint32 lodProgramVersion = program.GetBackendStateVersion();
|
||||||
const Uint64 lodBindGeneration = MG_State::pGLContext->GetTextureBindGeneration();
|
const Uint64 lodBindGeneration = MG_State::pGLContext->GetTextureBindGeneration();
|
||||||
|
// The probe also reads the EFFECTIVE sampler's filters/aniso/LOD range
|
||||||
|
// (ProgramSamplesOnlySingleLevelTextures), and those setters bump ONLY the
|
||||||
|
// sampling-resolution generation - not the texture params version the sum
|
||||||
|
// below covers. Without this key a filter/aniso change would keep serving
|
||||||
|
// the stale verdict.
|
||||||
|
const Uint64 lodSamplingGeneration = MG_State::pGLContext->GetSamplingResolutionGeneration();
|
||||||
Bool lodMemoHit = false;
|
Bool lodMemoHit = false;
|
||||||
if (m_lastLodDecisionValid && m_lastSampledSetValid &&
|
if (m_lastLodDecisionValid && m_lastSampledSetValid &&
|
||||||
m_lastLodProgramLifetimeId == lodProgramLifetimeId &&
|
m_lastLodProgramLifetimeId == lodProgramLifetimeId &&
|
||||||
m_lastLodProgramVersion == lodProgramVersion &&
|
m_lastLodProgramVersion == lodProgramVersion &&
|
||||||
m_lastLodBindGeneration == lodBindGeneration && m_lastLodBaseFlags == transformFlags &&
|
m_lastLodBindGeneration == lodBindGeneration &&
|
||||||
|
m_lastLodSamplingGeneration == lodSamplingGeneration && m_lastLodBaseFlags == transformFlags &&
|
||||||
m_lastSampledSetProgramLifetimeId == lodProgramLifetimeId &&
|
m_lastSampledSetProgramLifetimeId == lodProgramLifetimeId &&
|
||||||
m_lastSampledSetProgramVersion == lodProgramVersion &&
|
m_lastSampledSetProgramVersion == lodProgramVersion &&
|
||||||
m_lastSampledSetBindGeneration == lodBindGeneration) {
|
m_lastSampledSetBindGeneration == lodBindGeneration) {
|
||||||
@@ -6124,6 +6178,7 @@ void main() {
|
|||||||
m_lastLodProgramLifetimeId = lodProgramLifetimeId;
|
m_lastLodProgramLifetimeId = lodProgramLifetimeId;
|
||||||
m_lastLodProgramVersion = lodProgramVersion;
|
m_lastLodProgramVersion = lodProgramVersion;
|
||||||
m_lastLodBindGeneration = lodBindGeneration;
|
m_lastLodBindGeneration = lodBindGeneration;
|
||||||
|
m_lastLodSamplingGeneration = lodSamplingGeneration;
|
||||||
m_lastLodBaseFlags = baseFlags;
|
m_lastLodBaseFlags = baseFlags;
|
||||||
m_lastLodResultFlags = transformFlags;
|
m_lastLodResultFlags = transformFlags;
|
||||||
m_lastLodParamsSum = 0; // filled below once the sampled set is known
|
m_lastLodParamsSum = 0; // filled below once the sampled set is known
|
||||||
@@ -6467,6 +6522,7 @@ void main() {
|
|||||||
snap.vaoLifetimeId = vao.GetLifetimeId();
|
snap.vaoLifetimeId = vao.GetLifetimeId();
|
||||||
snap.vaoConfigVersion = vao.GetConfigVersion();
|
snap.vaoConfigVersion = vao.GetConfigVersion();
|
||||||
snap.drawFbo = drawFbo.get();
|
snap.drawFbo = drawFbo.get();
|
||||||
|
snap.drawFboLifetimeId = drawFbo->GetLifetimeId();
|
||||||
snap.fboVersion = drawFbo->GetObjectVersion();
|
snap.fboVersion = drawFbo->GetObjectVersion();
|
||||||
snap.drawFboIsDefault = drawFboIsDefault;
|
snap.drawFboIsDefault = drawFboIsDefault;
|
||||||
snap.viewportCount = ResolveDrawViewportCount(programObj.writesViewportIndexBuiltin);
|
snap.viewportCount = ResolveDrawViewportCount(programObj.writesViewportIndexBuiltin);
|
||||||
@@ -13825,6 +13881,15 @@ void main() {
|
|||||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||||
VK_VERIFY(vkCreateComputePipelines(m_device, VK_NULL_HANDLE, 1, &pipelineInfo, nullptr, &pipeline),
|
VK_VERIFY(vkCreateComputePipelines(m_device, VK_NULL_HANDLE, 1, &pipelineInfo, nullptr, &pipeline),
|
||||||
"GetOrCreateComputePipeline, vkCreateComputePipelines");
|
"GetOrCreateComputePipeline, vkCreateComputePipelines");
|
||||||
|
// A failed creation must never be memoized - same contract as
|
||||||
|
// PipelineFactory::GetOrCreatePipeline: caching the null would serve it back
|
||||||
|
// for the rest of the process and every dispatch of this program would be
|
||||||
|
// silently skipped. Retrying costs one failed vkCreateComputePipelines per
|
||||||
|
// dispatch, which is the correct price.
|
||||||
|
if (pipeline == VK_NULL_HANDLE) {
|
||||||
|
MGLOG_E("GetOrCreateComputePipeline: vkCreateComputePipelines failed; not caching the failure");
|
||||||
|
return VK_NULL_HANDLE;
|
||||||
|
}
|
||||||
m_computePipelines.emplace(programObj.hash, pipeline);
|
m_computePipelines.emplace(programObj.hash, pipeline);
|
||||||
return pipeline;
|
return pipeline;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -807,6 +807,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
Uint32 m_lastLodProgramVersion = 0;
|
Uint32 m_lastLodProgramVersion = 0;
|
||||||
Uint64 m_lastLodBindGeneration = 0;
|
Uint64 m_lastLodBindGeneration = 0;
|
||||||
Uint64 m_lastLodParamsSum = 0;
|
Uint64 m_lastLodParamsSum = 0;
|
||||||
|
// Sampling-resolution generation at probe time. The probe reads the effective
|
||||||
|
// sampler's filters/aniso/LOD range, whose setters bump only this counter -
|
||||||
|
// the params-version sum above never moves for them.
|
||||||
|
Uint64 m_lastLodSamplingGeneration = 0;
|
||||||
ProgramFactory::CompileOptionFlags m_lastLodBaseFlags = {};
|
ProgramFactory::CompileOptionFlags m_lastLodBaseFlags = {};
|
||||||
ProgramFactory::CompileOptionFlags m_lastLodResultFlags = {};
|
ProgramFactory::CompileOptionFlags m_lastLodResultFlags = {};
|
||||||
|
|
||||||
@@ -844,6 +848,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
Uint64 vaoLifetimeId = 0;
|
Uint64 vaoLifetimeId = 0;
|
||||||
Uint32 vaoConfigVersion = 0;
|
Uint32 vaoConfigVersion = 0;
|
||||||
const void* drawFbo = nullptr;
|
const void* drawFbo = nullptr;
|
||||||
|
// Never-reused lifetime id beside the raw pointer + Uint16 version: a
|
||||||
|
// deleted FBO recycled at the same address with the same fresh version
|
||||||
|
// count would otherwise compare equal (same ABA as the render-pass
|
||||||
|
// manager's fast-path memo).
|
||||||
|
Uint64 drawFboLifetimeId = 0;
|
||||||
Uint16 fboVersion = 0;
|
Uint16 fboVersion = 0;
|
||||||
Bool drawFboIsDefault = false;
|
Bool drawFboIsDefault = false;
|
||||||
Uint renderStateVersion = 0;
|
Uint renderStateVersion = 0;
|
||||||
@@ -1067,6 +1076,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
|||||||
VkBuffer indexVkBuffer = VK_NULL_HANDLE;
|
VkBuffer indexVkBuffer = VK_NULL_HANDLE;
|
||||||
VkDeviceSize indexSliceOffset = 0;
|
VkDeviceSize indexSliceOffset = 0;
|
||||||
Uint64 indexFrameSerial = 0;
|
Uint64 indexFrameSerial = 0;
|
||||||
|
// The EBO carried a host map when the slice was recorded - the mirror of
|
||||||
|
// anyBufferMapped on the vertex half. A shadow-backed (non-adopted)
|
||||||
|
// persistent map mutates its shadow with no API call and no epoch bump, so
|
||||||
|
// the one-compare rescue must decline and re-run the acquire, whose
|
||||||
|
// SyncPersistentMappedRange is the push-down. A map taken AFTER the record
|
||||||
|
// is already covered: AcquirePersistentMap bumps the slice epoch for the
|
||||||
|
// request itself, adopted or declined.
|
||||||
|
Bool indexBufferMapped = false;
|
||||||
|
|
||||||
// Bound per draw (first bindingCount elements).
|
// Bound per draw (first bindingCount elements).
|
||||||
VkBuffer vkBuffers[kMaxBindings] = {};
|
VkBuffer vkBuffers[kMaxBindings] = {};
|
||||||
|
|||||||
@@ -1057,21 +1057,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
static Bool allowVSOnlyPrograms;
|
// Read fresh every link, never latched in a static: the capability is
|
||||||
static Bool initialized = false;
|
// per-backend, and a latch would freeze it across a backend teardown +
|
||||||
if (!initialized) {
|
// re-initialization (the previous function-static memo here never even set
|
||||||
|
// its own initialized flag, so it re-read every call anyway - this makes
|
||||||
|
// the always-fresh behavior the stated one). A struct-field read per
|
||||||
|
// glLinkProgram costs nothing.
|
||||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
||||||
if (!activeBackendObject) {
|
if (!activeBackendObject) {
|
||||||
MGLOG_E_ONCE("activeBackendObject is not initialized!");
|
MGLOG_E_ONCE("activeBackendObject is not initialized!");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const auto& rendererInfo = activeBackendObject->GetRendererInfo();
|
const Bool allowVSOnlyPrograms =
|
||||||
allowVSOnlyPrograms = (Int)rendererInfo.StaticBackendCapability.AllowVSOnlyPrograms;
|
activeBackendObject->GetRendererInfo().StaticBackendCapability.AllowVSOnlyPrograms;
|
||||||
}
|
|
||||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
|
||||||
if (activeBackendObject) {
|
|
||||||
programObject->SetMaxFragmentOutputColorNumber(activeBackendObject->GetDynamicParameters().MaxDrawBuffers);
|
programObject->SetMaxFragmentOutputColorNumber(activeBackendObject->GetDynamicParameters().MaxDrawBuffers);
|
||||||
}
|
|
||||||
programObject->Link(!allowVSOnlyPrograms);
|
programObject->Link(!allowVSOnlyPrograms);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -648,4 +648,39 @@ namespace MobileGL::MG_Impl::GLImpl {
|
|||||||
if (!ValidateQueryStreamIndex(__FUNCTION__, target, index)) return;
|
if (!ValidateQueryStreamIndex(__FUNCTION__, target, index)) return;
|
||||||
GetQueryiv(target, pname, params);
|
GetQueryiv(target, pname, params);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void DestroyAllQueryObjects() {
|
||||||
|
// Detach the registry under the lock, release outside it - same discipline
|
||||||
|
// (and the same accepted teardown race) as DestroyAllSyncObjects. Without
|
||||||
|
// this drain, every query the app left undeleted survived full library
|
||||||
|
// teardown in the process-global registry: the objects and their backend
|
||||||
|
// wrappers leaked across Destroy/Initialize cycles, stale ids kept
|
||||||
|
// answering IsQuery == GL_TRUE in the re-initialized library, and a later
|
||||||
|
// glDeleteQueries could hand the OLD backend's handle to a DIFFERENT
|
||||||
|
// backend's DeleteBackendQuery, which casts it to the wrong wrapper type.
|
||||||
|
UnorderedMap<GLuint, QueryObject*> orphans;
|
||||||
|
{
|
||||||
|
const std::lock_guard<std::mutex> lock(g_queryObjectsMutex);
|
||||||
|
orphans.swap(g_liveQueryObjects);
|
||||||
|
g_activeTimeElapsedQueryId = 0;
|
||||||
|
g_activePrimitivesWrittenQueryId = 0;
|
||||||
|
g_activePrimitivesGeneratedQueryId = 0;
|
||||||
|
g_activeSamplesPassedQueryId = 0;
|
||||||
|
}
|
||||||
|
if (orphans.empty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// Backend handles must be released by the backend that created them, so
|
||||||
|
// this runs while the function table is still populated. Both backends'
|
||||||
|
// DeleteBackendQuery are generation-guarded, so a handle whose renderer
|
||||||
|
// or ES context is already gone frees only the wrapper.
|
||||||
|
const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery;
|
||||||
|
for (const auto& [_, queryObject] : orphans) {
|
||||||
|
if (deleteBackendQuery && queryObject->backendHandle) {
|
||||||
|
deleteBackendQuery(queryObject->backendHandle);
|
||||||
|
}
|
||||||
|
delete queryObject;
|
||||||
|
}
|
||||||
|
MGLOG_D("DestroyAllQueryObjects: reclaimed %zu query object(s) the app left undeleted", orphans.size());
|
||||||
|
}
|
||||||
} // namespace MobileGL::MG_Impl::GLImpl
|
} // namespace MobileGL::MG_Impl::GLImpl
|
||||||
|
|||||||
@@ -29,4 +29,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
|||||||
void GetQueryBufferObjecti64v(GLuint id, GLuint buffer, GLenum pname, GLintptr offset);
|
void GetQueryBufferObjecti64v(GLuint id, GLuint buffer, GLenum pname, GLintptr offset);
|
||||||
void GetQueryBufferObjectui64v(GLuint id, GLuint buffer, GLenum pname, GLintptr offset);
|
void GetQueryBufferObjectui64v(GLuint id, GLuint buffer, GLenum pname, GLintptr offset);
|
||||||
void QueryCounter(GLuint id, GLenum target);
|
void QueryCounter(GLuint id, GLenum target);
|
||||||
|
// Destroys every still-registered query object exactly as DeleteQueries would.
|
||||||
|
// GL requires queries to die with their context; called only from full library
|
||||||
|
// teardown (DestroyImpl), where no context survives on any thread, so the
|
||||||
|
// process-global registry can be drained wholesale. Must run while the backend
|
||||||
|
// function table is still populated: each backend handle has to be released by
|
||||||
|
// the backend that created it, never by a later re-initialized one (whose
|
||||||
|
// DeleteBackendQuery would cast the wrapper to the wrong backend's type).
|
||||||
|
// Same contract as DestroyAllSyncObjects.
|
||||||
|
void DestroyAllQueryObjects();
|
||||||
} // namespace MobileGL::MG_Impl::GLImpl
|
} // namespace MobileGL::MG_Impl::GLImpl
|
||||||
|
|||||||
@@ -8,11 +8,13 @@
|
|||||||
//
|
//
|
||||||
// Scenario - THE FIXTURE-SHAPED SUBGROUP REDUCTION, ON WHATEVER WIDTH THE DEVICE HAS.
|
// Scenario - THE FIXTURE-SHAPED SUBGROUP REDUCTION, ON WHATEVER WIDTH THE DEVICE HAS.
|
||||||
//
|
//
|
||||||
// iterationRP's auto-exposure pass declares `shared vec2 prefixSumCache[32]` for a
|
// iterationRP hard-sizes the scratch its subgroup prefix scans write through
|
||||||
// 512-invocation workgroup and combines per-subgroup subtotals through
|
// prefixSumCache[gl_SubgroupID], and ships that idiom twice: the auto-exposure pass
|
||||||
// prefixSumCache[gl_SubgroupID]. The algorithm is width-agnostic; only the static 32
|
// declares `shared vec2 prefixSumCache[32]` for a 512-invocation workgroup, and the
|
||||||
// bakes in "at most 32 subgroups", which every desktop capture satisfies and an 8-lane
|
// RTW importance warp declares `shared float prefixSumCache[64]` for a 1024-invocation
|
||||||
// device (lavapipe: 64 subgroups) does not. DirectVulkan patches exactly that with
|
// one. Both algorithms are width-agnostic; only the static lengths bake in "at most 32
|
||||||
|
// (respectively 64) subgroups", which every desktop capture satisfies and an 8-lane
|
||||||
|
// device (lavapipe: 64 and 128 subgroups) does not. DirectVulkan patches exactly that with
|
||||||
// FixIterationRPSubgroupScratchPass, growing the array to ceil(invocations / native
|
// FixIterationRPSubgroupScratchPass, growing the array to ceil(invocations / native
|
||||||
// width) on the modules that match the pack's reduction fingerprint.
|
// width) on the modules that match the pack's reduction fingerprint.
|
||||||
//
|
//
|
||||||
@@ -47,6 +49,10 @@ namespace MGITest {
|
|||||||
constexpr std::uint32_t kInvocationCount = 512u;
|
constexpr std::uint32_t kInvocationCount = 512u;
|
||||||
// sum of 0..511, exactly representable and associativity-proof in fp32.
|
// sum of 0..511, exactly representable and associativity-proof in fp32.
|
||||||
constexpr float kExpectedTotal = 130816.0f;
|
constexpr float kExpectedTotal = 130816.0f;
|
||||||
|
// The RTW warp's shape: 1024 invocations into a 64-entry float scratch.
|
||||||
|
constexpr std::uint32_t kWideInvocationCount = 1024u;
|
||||||
|
// sum of 0..1023, likewise exact in fp32.
|
||||||
|
constexpr float kWideExpectedTotal = 523776.0f;
|
||||||
|
|
||||||
constexpr const char* kComputeSource = R"(#version 430 core
|
constexpr const char* kComputeSource = R"(#version 430 core
|
||||||
#extension GL_KHR_shader_subgroup_basic : require
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
@@ -87,6 +93,50 @@ void main() {
|
|||||||
}
|
}
|
||||||
atomicMax(outputData.maxSubgroupId, gl_SubgroupID);
|
atomicMax(outputData.maxSubgroupId, gl_SubgroupID);
|
||||||
}
|
}
|
||||||
|
)";
|
||||||
|
|
||||||
|
// The RTW importance warp's shape: a plain float scan over 1024 invocations
|
||||||
|
// into a 64-entry scratch. Same idiom, different dimensions - which is exactly
|
||||||
|
// what a fingerprint pinned to the exposure pass's shape walks past.
|
||||||
|
constexpr const char* kWideComputeSource = R"(#version 430 core
|
||||||
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
|
|
||||||
|
layout(local_size_x = 1024) in;
|
||||||
|
|
||||||
|
layout(std430, binding = 0) buffer Output {
|
||||||
|
float total;
|
||||||
|
uint numSubgroups;
|
||||||
|
uint maxSubgroupId;
|
||||||
|
} outputData;
|
||||||
|
|
||||||
|
shared float prefixSumCache[64];
|
||||||
|
|
||||||
|
void main() {
|
||||||
|
float importance = float(gl_LocalInvocationID.x);
|
||||||
|
float prefixSum = subgroupInclusiveAdd(importance);
|
||||||
|
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
|
||||||
|
prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||||
|
barrier();
|
||||||
|
|
||||||
|
uint loopLength = uint(findMSB(gl_NumSubgroups));
|
||||||
|
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
|
||||||
|
|
||||||
|
for (uint scanStage = 0u; scanStage < loopLength; ++scanStage) {
|
||||||
|
if ((gl_SubgroupID & (1u << scanStage)) > 0u) {
|
||||||
|
prefixSum += prefixSumCache[(gl_SubgroupID >> scanStage << scanStage) - 1u];
|
||||||
|
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
|
||||||
|
prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||||
|
}
|
||||||
|
barrier();
|
||||||
|
}
|
||||||
|
|
||||||
|
if (gl_LocalInvocationID.x == 1023u) {
|
||||||
|
outputData.total = prefixSum;
|
||||||
|
outputData.numSubgroups = gl_NumSubgroups;
|
||||||
|
}
|
||||||
|
atomicMax(outputData.maxSubgroupId, gl_SubgroupID);
|
||||||
|
}
|
||||||
)";
|
)";
|
||||||
|
|
||||||
struct OutputBlock {
|
struct OutputBlock {
|
||||||
@@ -130,8 +180,7 @@ void main() {
|
|||||||
"512-invocation workgroup";
|
"512-invocation workgroup";
|
||||||
}
|
}
|
||||||
|
|
||||||
m_program = CompileComputeProgram(kComputeSource);
|
m_maxInvocations = static_cast<std::uint32_t>(invocations);
|
||||||
ASSERT_NE(m_program, 0u) << m_buildLog;
|
|
||||||
|
|
||||||
glGenBuffers(1, &m_output);
|
glGenBuffers(1, &m_output);
|
||||||
glBindBuffer(GL_SHADER_STORAGE_BUFFER, m_output);
|
glBindBuffer(GL_SHADER_STORAGE_BUFFER, m_output);
|
||||||
@@ -181,7 +230,14 @@ void main() {
|
|||||||
return program;
|
return program;
|
||||||
}
|
}
|
||||||
|
|
||||||
OutputBlock Dispatch() {
|
// Re-poisons the block, compiles the shape under test and runs it once.
|
||||||
|
OutputBlock Dispatch(const char* source) {
|
||||||
|
const OutputBlock poison{-1.0f, 0xa5a5a5a5u, 0u};
|
||||||
|
glBindBuffer(GL_SHADER_STORAGE_BUFFER, m_output);
|
||||||
|
glBufferSubData(GL_SHADER_STORAGE_BUFFER, 0, sizeof(OutputBlock), &poison);
|
||||||
|
m_program = CompileComputeProgram(source);
|
||||||
|
EXPECT_NE(m_program, 0u) << m_buildLog;
|
||||||
|
if (m_program == 0u) return OutputBlock{};
|
||||||
glUseProgram(m_program);
|
glUseProgram(m_program);
|
||||||
glDispatchCompute(1, 1, 1);
|
glDispatchCompute(1, 1, 1);
|
||||||
glMemoryBarrier(GL_BUFFER_UPDATE_BARRIER_BIT);
|
glMemoryBarrier(GL_BUFFER_UPDATE_BARRIER_BIT);
|
||||||
@@ -193,12 +249,13 @@ void main() {
|
|||||||
|
|
||||||
GLuint m_program = 0;
|
GLuint m_program = 0;
|
||||||
GLuint m_output = 0;
|
GLuint m_output = 0;
|
||||||
|
std::uint32_t m_maxInvocations = 0;
|
||||||
std::string m_buildLog;
|
std::string m_buildLog;
|
||||||
};
|
};
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
TEST_F(IterationRPScratchFixScenario, FixtureShapedReductionSumsEveryInvocation) {
|
TEST_F(IterationRPScratchFixScenario, FixtureShapedReductionSumsEveryInvocation) {
|
||||||
const OutputBlock block = Dispatch();
|
const OutputBlock block = Dispatch(kComputeSource);
|
||||||
EXPECT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR));
|
EXPECT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR));
|
||||||
|
|
||||||
// The topology diagnostics catch the failure modes by name before the sum does:
|
// The topology diagnostics catch the failure modes by name before the sum does:
|
||||||
@@ -219,4 +276,27 @@ void main() {
|
|||||||
<< "workgroup reduction produced " << block.total << " with gl_NumSubgroups="
|
<< "workgroup reduction produced " << block.total << " with gl_NumSubgroups="
|
||||||
<< block.numSubgroups;
|
<< block.numSubgroups;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The pack's second instance of the same bug, and the one that kept the CI
|
||||||
|
// retrace red after the exposure pass alone was patched.
|
||||||
|
TEST_F(IterationRPScratchFixScenario, WideFixtureShapedReductionSumsEveryInvocation) {
|
||||||
|
if (m_maxInvocations < kWideInvocationCount) {
|
||||||
|
GTEST_SKIP() << "needs a " << kWideInvocationCount << "-invocation workgroup";
|
||||||
|
}
|
||||||
|
const OutputBlock block = Dispatch(kWideComputeSource);
|
||||||
|
EXPECT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR));
|
||||||
|
|
||||||
|
ASSERT_NE(block.numSubgroups, 0xa5a5a5a5u) << "invocation 1023 never reached its store";
|
||||||
|
EXPECT_GE(block.numSubgroups, 1u);
|
||||||
|
EXPECT_LE(block.numSubgroups, kWideInvocationCount);
|
||||||
|
EXPECT_LT(block.maxSubgroupId, block.numSubgroups)
|
||||||
|
<< "gl_SubgroupID exceeds gl_NumSubgroups - the inconsistency "
|
||||||
|
"DeriveNumSubgroupsPass exists to repair";
|
||||||
|
|
||||||
|
// Without the patch an 8-lane device writes prefixSumCache[64..127] out of
|
||||||
|
// bounds and this comparison fails.
|
||||||
|
EXPECT_EQ(block.total, kWideExpectedTotal)
|
||||||
|
<< "workgroup reduction produced " << block.total << " with gl_NumSubgroups="
|
||||||
|
<< block.numSubgroups;
|
||||||
|
}
|
||||||
} // namespace MGITest
|
} // namespace MGITest
|
||||||
|
|||||||
@@ -646,9 +646,17 @@ namespace MobileGL::MG_State {
|
|||||||
for (SizeT stage = 0; stage < ProgramPipelineObject::kGraphicsStageCount; ++stage) {
|
for (SizeT stage = 0; stage < ProgramPipelineObject::kGraphicsStageCount; ++stage) {
|
||||||
const auto& stageProgram = pipeline->GetStageProgram(static_cast<ShaderStage>(stage));
|
const auto& stageProgram = pipeline->GetStageProgram(static_cast<ShaderStage>(stage));
|
||||||
if (!stageProgram) continue;
|
if (!stageProgram) continue;
|
||||||
for (const auto& shader : stageProgram->GetAttachedShaders()) {
|
// The stage program contributes the shaders its LAST LINK consumed, never
|
||||||
if (!shader || static_cast<SizeT>(shader->GetShaderStage()) != stage) continue;
|
// its live attach list: per GL 4.6 7.3/7.4 a pipeline stage executes the
|
||||||
composite->AttachShader(shader);
|
// stage program as last linked - glAttachShader and glCompileShader take
|
||||||
|
// effect only at the program's next link - and neither of those moves the
|
||||||
|
// link version this cache keys on, so reading live state here would let a
|
||||||
|
// post-link attach or recompile leak into the composite while the signature
|
||||||
|
// still hits. The pinned (source, node) makes the composite's Link()
|
||||||
|
// consume the very inputs that link consumed.
|
||||||
|
for (const auto& ref : stageProgram->GetLinkedShaderSnapshot()) {
|
||||||
|
if (!ref.shader || static_cast<SizeT>(ref.shader->GetShaderStage()) != stage) continue;
|
||||||
|
composite->AttachShaderWithPinnedLinkInput(ref);
|
||||||
anyStage = true;
|
anyStage = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -9,7 +9,18 @@
|
|||||||
#include "FramebufferObject.h"
|
#include "FramebufferObject.h"
|
||||||
#include "MG_Util/Types.h"
|
#include "MG_Util/Types.h"
|
||||||
|
|
||||||
|
#include <atomic>
|
||||||
|
|
||||||
namespace MobileGL::MG_State::GLState {
|
namespace MobileGL::MG_State::GLState {
|
||||||
|
// Starts at 1 so a zero-initialized memo slot can never carry a live object's id.
|
||||||
|
// Atomic for the same reason as the VAO counter: it costs nothing, and a duplicate
|
||||||
|
// id would resurrect exactly the ABA this id exists to kill.
|
||||||
|
static std::atomic<Uint64> s_nextFramebufferLifetimeId{1};
|
||||||
|
|
||||||
|
Uint64 FramebufferObject::AllocateLifetimeId() {
|
||||||
|
return s_nextFramebufferLifetimeId.fetch_add(1, std::memory_order_relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
// FramebufferAttachmentObject
|
// FramebufferAttachmentObject
|
||||||
FramebufferAttachmentObject::FramebufferAttachmentObject(
|
FramebufferAttachmentObject::FramebufferAttachmentObject(
|
||||||
const SharedPtr<MG_State::GLState::ITextureObject>& texture, TextureUploadTarget textureUploadTarget, Int level,
|
const SharedPtr<MG_State::GLState::ITextureObject>& texture, TextureUploadTarget textureUploadTarget, Int level,
|
||||||
|
|||||||
@@ -148,13 +148,25 @@ namespace MobileGL {
|
|||||||
|
|
||||||
Uint16 GetObjectVersion() const { return m_objectVersion; }
|
Uint16 GetObjectVersion() const { return m_objectVersion; }
|
||||||
|
|
||||||
|
// Globally-unique, never-reused id for THIS object's lifetime - the same
|
||||||
|
// contract as VertexArrayObject::GetLifetimeId(), and needed for the same
|
||||||
|
// reason: neither the GL name nor the heap address can tell a
|
||||||
|
// deleted-and-recreated framebuffer from the original, and m_objectVersion
|
||||||
|
// starts at 0 for every new object, so a backend memo keyed on
|
||||||
|
// (pointer, version) alone would silently inherit the dead object's entry
|
||||||
|
// (see VkRenderPassManager's per-draw fast-path memo).
|
||||||
|
Uint64 GetLifetimeId() const { return m_lifetimeId; }
|
||||||
|
|
||||||
Uint GetExternalIndex() const;
|
Uint GetExternalIndex() const;
|
||||||
Bool IsDefaultFramebuffer() const { return m_externalIndex == 0; }
|
Bool IsDefaultFramebuffer() const { return m_externalIndex == 0; }
|
||||||
|
|
||||||
private:
|
private:
|
||||||
|
static Uint64 AllocateLifetimeId();
|
||||||
|
|
||||||
void BumpAttachmentVersion(FramebufferAttachmentType type);
|
void BumpAttachmentVersion(FramebufferAttachmentType type);
|
||||||
|
|
||||||
const Uint m_externalIndex = 0;
|
const Uint m_externalIndex = 0;
|
||||||
|
const Uint64 m_lifetimeId = AllocateLifetimeId();
|
||||||
FramebufferAttachmentObjectArray m_attachmentObjects;
|
FramebufferAttachmentObjectArray m_attachmentObjects;
|
||||||
FramebufferAttachmentVersionArray m_attachmentVersions;
|
FramebufferAttachmentVersionArray m_attachmentVersions;
|
||||||
|
|
||||||
|
|||||||
@@ -393,6 +393,14 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool ProgramObject::AttachShaderWithPinnedLinkInput(const LinkedShaderRef& ref) {
|
||||||
|
if (!AttachShader(ref.shader)) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
m_pinnedLinkInputs[ref.shader.get()] = ref;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
SizeT ProgramObject::DetachShader(const SharedPtr<ShaderObject>& shader) {
|
SizeT ProgramObject::DetachShader(const SharedPtr<ShaderObject>& shader) {
|
||||||
MGLOG_D("DetachShader called for shader %p from ProgramObject %u", shader.get(), m_externalIndex);
|
MGLOG_D("DetachShader called for shader %p from ProgramObject %u", shader.get(), m_externalIndex);
|
||||||
if (!ShaderIsAttached(shader)) {
|
if (!ShaderIsAttached(shader)) {
|
||||||
@@ -475,6 +483,8 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
AddDefaultFragmentShaderIfMissing();
|
AddDefaultFragmentShaderIfMissing();
|
||||||
}
|
}
|
||||||
if (m_shaders.empty()) {
|
if (m_shaders.empty()) {
|
||||||
|
// This IS the last link now, and it consumed nothing.
|
||||||
|
m_linkedShaderSnapshot.clear();
|
||||||
m_artifacts.infoLog = "No shader objects are attached to program.";
|
m_artifacts.infoLog = "No shader objects are attached to program.";
|
||||||
MGLOG_E("ProgramObject %u: Link failed - no shader objects attached.", m_externalIndex);
|
MGLOG_E("ProgramObject %u: Link failed - no shader objects attached.", m_externalIndex);
|
||||||
return;
|
return;
|
||||||
@@ -505,8 +515,19 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
Vector<SharedPtr<ShaderCompileTask>> deps;
|
Vector<SharedPtr<ShaderCompileTask>> deps;
|
||||||
deps.reserve(m_shaders.size());
|
deps.reserve(m_shaders.size());
|
||||||
task->in.shaders.reserve(m_shaders.size());
|
task->in.shaders.reserve(m_shaders.size());
|
||||||
|
m_linkedShaderSnapshot.clear();
|
||||||
|
m_linkedShaderSnapshot.reserve(m_shaders.size());
|
||||||
for (const auto& shader : m_shaders) {
|
for (const auto& shader : m_shaders) {
|
||||||
const SharedPtr<ShaderCompileTask>& node = shader->CompiledNodeForLink();
|
// A pipeline composite pins the (source, node) each stage program's LAST link
|
||||||
|
// consumed (AttachShaderWithPinnedLinkInput); an ordinary program takes the
|
||||||
|
// shader's current ones. Without the pin a post-link recompile would leak a
|
||||||
|
// shader the stage program never linked into the composite.
|
||||||
|
SharedPtr<const String> sourcePtr = shader->GetShaderSourcePtr();
|
||||||
|
SharedPtr<ShaderCompileTask> node = shader->CompiledNodeForLink();
|
||||||
|
if (const auto pinned = m_pinnedLinkInputs.find(shader.get()); pinned != m_pinnedLinkInputs.end()) {
|
||||||
|
sourcePtr = pinned->second.source;
|
||||||
|
node = pinned->second.node;
|
||||||
|
}
|
||||||
if (node) {
|
if (node) {
|
||||||
// This link is now an observer of that node's result, and the ShaderObject is
|
// This link is now an observer of that node's result, and the ShaderObject is
|
||||||
// no longer the only route to it: without the marker, the ordinary
|
// no longer the only route to it: without the marker, the ordinary
|
||||||
@@ -515,7 +536,10 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
node->MarkLinkReferenced();
|
node->MarkLinkReferenced();
|
||||||
if (!node->IsTerminal()) deps.push_back(node);
|
if (!node->IsTerminal()) deps.push_back(node);
|
||||||
}
|
}
|
||||||
task->in.shaders.push_back({shader->GetShaderStage(), shader->GetShaderSourcePtr(), node});
|
task->in.shaders.push_back({shader->GetShaderStage(), sourcePtr, node});
|
||||||
|
// What "as last linked" will mean for this program from now on - the pipeline
|
||||||
|
// composite cache rebuilds from exactly this set (GetProgramForDraw).
|
||||||
|
m_linkedShaderSnapshot.push_back({shader, sourcePtr, node});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Phase B of the same link: SPIR-V generation, spirv-opt and the global-UBO routing
|
// Phase B of the same link: SPIR-V generation, spirv-opt and the global-UBO routing
|
||||||
|
|||||||
@@ -60,6 +60,26 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
|
|
||||||
Vector<SharedPtr<ShaderObject>>& GetAttachedShaders();
|
Vector<SharedPtr<ShaderObject>>& GetAttachedShaders();
|
||||||
const Vector<SharedPtr<ShaderObject>>& GetAttachedShaders() const;
|
const Vector<SharedPtr<ShaderObject>>& GetAttachedShaders() const;
|
||||||
|
|
||||||
|
// One shader exactly as this program's last Link() consumed it: the object, the
|
||||||
|
// source snapshot, and the compile node taken at that link's enqueue. GL 4.6 7.3/7.4
|
||||||
|
// makes this triple - not the live attach list, not the shader's current compile -
|
||||||
|
// what a program pipeline stage executes ("as last linked"): glAttachShader and
|
||||||
|
// glCompileShader take effect only at the program's next link, yet neither moves
|
||||||
|
// m_linkVersion, so anything keyed on the link generation must consume this
|
||||||
|
// snapshot rather than re-read the live state.
|
||||||
|
struct LinkedShaderRef {
|
||||||
|
SharedPtr<ShaderObject> shader;
|
||||||
|
SharedPtr<const String> source;
|
||||||
|
SharedPtr<ShaderCompileTask> node;
|
||||||
|
};
|
||||||
|
// The last link's full input set; empty when this program has never linked (or its
|
||||||
|
// last link had no shaders attached). GL-thread-owned, rebuilt in Link()'s prologue.
|
||||||
|
const Vector<LinkedShaderRef>& GetLinkedShaderSnapshot() const { return m_linkedShaderSnapshot; }
|
||||||
|
// Pipeline-composite attach: AttachShader plus a pin that makes THIS program's
|
||||||
|
// Link() consume ref's (source, node) instead of the shader's current ones, so a
|
||||||
|
// post-link recompile of the stage program's shader cannot leak into the composite.
|
||||||
|
bool AttachShaderWithPinnedLinkInput(const LinkedShaderRef& ref);
|
||||||
const String& GetInfoLog() const { return Artifacts().infoLog; }
|
const String& GetInfoLog() const { return Artifacts().infoLog; }
|
||||||
// glCreateShaderProgramv folds the shader's compile log into the program's log, which
|
// glCreateShaderProgramv folds the shader's compile log into the program's log, which
|
||||||
// is the only place a caller can read it from once the shader name is gone.
|
// is the only place a caller can read it from once the shader name is gone.
|
||||||
@@ -786,6 +806,13 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
// order - and the name is the only coordinate all three agree on. Absent from the map
|
// order - and the name is the only coordinate all three agree on. Absent from the map
|
||||||
// means "never rebound", and the shader's declared binding still stands.
|
// means "never rebound", and the shader's declared binding still stands.
|
||||||
void SetShaderStorageBlockBinding(const String& blockName, Uint binding) {
|
void SetShaderStorageBlockBinding(const String& blockName, Uint binding) {
|
||||||
|
// Equality bail-out like SetUniformBlockBinding's: the pipeline composite
|
||||||
|
// mirror replays every override each draw, and without this every replay
|
||||||
|
// would churn m_blockBindingVersion and rebuild whatever keys on it.
|
||||||
|
const auto it = Artifacts().shaderStorageBlockBinding.find(blockName);
|
||||||
|
if (it != Artifacts().shaderStorageBlockBinding.end() && it->second == static_cast<Int>(binding)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
Artifacts().shaderStorageBlockBinding[blockName] = static_cast<Int>(binding);
|
Artifacts().shaderStorageBlockBinding[blockName] = static_cast<Int>(binding);
|
||||||
// Deliberately NOT m_backendStateVersion: Espryt's entry point never forces a
|
// Deliberately NOT m_backendStateVersion: Espryt's entry point never forces a
|
||||||
// program build off this, and bumping that version would start doing so. The
|
// program build off this, and bumping that version would start doing so. The
|
||||||
@@ -1223,6 +1250,13 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
// glGetAttachedShaders / GL_ATTACHED_SHADERS / the orphan-shader sweep need no join.
|
// glGetAttachedShaders / GL_ATTACHED_SHADERS / the orphan-shader sweep need no join.
|
||||||
Vector<SharedPtr<ShaderObject>> m_shaders;
|
Vector<SharedPtr<ShaderObject>> m_shaders;
|
||||||
Vector<SharedPtr<ShaderObject>> m_detachedShaders; // Store detached shaders and remove on next link
|
Vector<SharedPtr<ShaderObject>> m_detachedShaders; // Store detached shaders and remove on next link
|
||||||
|
// See GetLinkedShaderSnapshot. Holding the SharedPtrs here is deliberate: the
|
||||||
|
// "as last linked" set must survive detach-and-delete of its shaders (the
|
||||||
|
// glCreateShaderProgramv shape) until the next link replaces it.
|
||||||
|
Vector<LinkedShaderRef> m_linkedShaderSnapshot;
|
||||||
|
// See AttachShaderWithPinnedLinkInput. Populated only on pipeline composites,
|
||||||
|
// which never detach, so entries need no removal path. GL-thread-owned.
|
||||||
|
UnorderedMap<const ShaderObject*, LinkedShaderRef> m_pinnedLinkInputs;
|
||||||
|
|
||||||
// Link INPUTS (all "take effect at the next link" per GL): glBindAttribLocation,
|
// Link INPUTS (all "take effect at the next link" per GL): glBindAttribLocation,
|
||||||
// glBindFragDataLocation(Indexed), glTransformFeedbackVaryings, and the draw-buffer
|
// glBindFragDataLocation(Indexed), glTransformFeedbackVaryings, and the draw-buffer
|
||||||
|
|||||||
@@ -140,6 +140,13 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
}
|
}
|
||||||
|
|
||||||
void ShaderObject::Compile() {
|
void ShaderObject::Compile() {
|
||||||
|
// The compile-environment snapshot is taken HERE, on the GL thread, and handed to
|
||||||
|
// the job. Everything the pipeline needs to know about the device comes through it,
|
||||||
|
// never through pActiveBackendObject - that is what makes the body movable.
|
||||||
|
// Hoisted above the memo check because the memo must be env-disciplined too (below).
|
||||||
|
const SharedPtr<const MG_Util::ShaderTranspiler::CompileEnv> env =
|
||||||
|
MG_Util::ShaderTranspiler::GetCurrentCompileEnv();
|
||||||
|
|
||||||
// P0b layer 1, as a tri-state: the memo is "the node in m_compiled was built from
|
// P0b layer 1, as a tri-state: the memo is "the node in m_compiled was built from
|
||||||
// the string m_source still points at". SetShaderSource only swaps that pointer when
|
// the string m_source still points at". SetShaderSource only swaps that pointer when
|
||||||
// the text actually differs, so this is a pointer compare, and it covers Pending as
|
// the text actually differs, so this is a pointer compare, and it covers Pending as
|
||||||
@@ -152,7 +159,18 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
// ClaimParsedShader's on-demand re-parse needs - a real recompile would have handed
|
// ClaimParsedShader's on-demand re-parse needs - a real recompile would have handed
|
||||||
// the next link a fresh parse, the no-op hands it a fresh re-parse of the identical
|
// the next link a fresh parse, the no-op hands it a fresh re-parse of the identical
|
||||||
// source instead. Same result, one parse either way.
|
// source instead. Same result, one parse either way.
|
||||||
if (HasMemoizedCompile()) return;
|
//
|
||||||
|
// The environment joins the check (ShaderSourceKey.h's memo-hazard rule: a memo
|
||||||
|
// must never be handed back under an environment other than the one it was
|
||||||
|
// computed against). Layers 2 and 3 key on the fingerprint, but this memo sits
|
||||||
|
// ABOVE both, so without this compare a node computed against a dead environment
|
||||||
|
// - e.g. a compute shader rejected against the pre-capability fallback limits -
|
||||||
|
// would keep answering forever while a fresh object with byte-identical source
|
||||||
|
// compiles fine. The fingerprint is a content hash, so a republish of identical
|
||||||
|
// capabilities still hits.
|
||||||
|
if (HasMemoizedCompile() && m_compiled->env != nullptr && m_compiled->env->fingerprint == env->fingerprint) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
// Two reasons to stay on this thread, one rule. Without the async flag the whole
|
// Two reasons to stay on this thread, one rule. Without the async flag the whole
|
||||||
// path must be byte-identical to the synchronous implementation, and a cache-less
|
// path must be byte-identical to the synchronous implementation, and a cache-less
|
||||||
@@ -168,12 +186,6 @@ namespace MobileGL::MG_State::GLState {
|
|||||||
// glMaxShaderCompilerThreadsKHR(0) and a flag-off build both bypass sharing exactly
|
// glMaxShaderCompilerThreadsKHR(0) and a flag-off build both bypass sharing exactly
|
||||||
// as they bypass the pool, and their behaviour stays byte-identical to pre-stage-6.
|
// as they bypass the pool, and their behaviour stays byte-identical to pre-stage-6.
|
||||||
const Bool runOnPool = m_preprocessCache && MG_Util::Async::AsyncShaderCompileActive();
|
const Bool runOnPool = m_preprocessCache && MG_Util::Async::AsyncShaderCompileActive();
|
||||||
|
|
||||||
// The compile-environment snapshot is taken HERE, on the GL thread, and handed to
|
|
||||||
// the job. Everything the pipeline needs to know about the device comes through it,
|
|
||||||
// never through pActiveBackendObject - that is what makes the body movable.
|
|
||||||
const SharedPtr<const MG_Util::ShaderTranspiler::CompileEnv> env =
|
|
||||||
MG_Util::ShaderTranspiler::GetCurrentCompileEnv();
|
|
||||||
const Uint64 sourceHash = ShaderPreprocessCache::HashSource(*m_source);
|
const Uint64 sourceHash = ShaderPreprocessCache::HashSource(*m_source);
|
||||||
|
|
||||||
// ---- P1 stage 6: adopt an equivalent compile instead of enqueueing a duplicate ----
|
// ---- P1 stage 6: adopt an equivalent compile instead of enqueueing a duplicate ----
|
||||||
|
|||||||
@@ -109,10 +109,11 @@ namespace {
|
|||||||
return tools.Validate(spirv);
|
return tools.Validate(spirv);
|
||||||
}
|
}
|
||||||
|
|
||||||
// iterationRP's reduction fingerprint: 32x16x1, subgroupInclusiveAdd on a
|
// iterationRP's exposure reduction, as the pack ships it: 32x16 (512
|
||||||
// vec2, and the pack's own 32-entry gl_SubgroupID-indexed scratch. A second,
|
// invocations), subgroupInclusiveAdd on a vec2, and a 32-entry
|
||||||
// plainly indexed array rides along to prove the patch is surgical.
|
// gl_SubgroupID-indexed scratch. A second, plainly indexed array rides along
|
||||||
constexpr const char* kIterationRPShapedSource = R"(#version 450 core
|
// to prove the patch is surgical.
|
||||||
|
constexpr const char* kExposureShapedSource = R"(#version 450 core
|
||||||
#extension GL_KHR_shader_subgroup_basic : require
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
#extension GL_KHR_shader_subgroup_arithmetic : require
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
||||||
@@ -141,85 +142,195 @@ void main() {
|
|||||||
}
|
}
|
||||||
)";
|
)";
|
||||||
|
|
||||||
// Same scratch idiom, different workgroup shape - NOT iterationRP, so the
|
// The pack's OTHER instance of the same bug, which a fingerprint pinned to the
|
||||||
// fingerprint must refuse it even though it would break identically.
|
// exposure pass's dimensions walks straight past: the RTW importance warp
|
||||||
constexpr const char* kWrongWorkgroupShapeSource = R"(#version 450 core
|
// scans a plain float across 1024 invocations into a 64-entry scratch.
|
||||||
|
constexpr const char* kRtwWarpShapedSource = R"(#version 450 core
|
||||||
#extension GL_KHR_shader_subgroup_basic : require
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
#extension GL_KHR_shader_subgroup_arithmetic : require
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
layout(local_size_x = 64, local_size_y = 8, local_size_z = 1) in;
|
layout(local_size_x = 1024) in;
|
||||||
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
||||||
shared vec2 prefixSumCache[32];
|
shared float prefixSumCache[64];
|
||||||
void main() {
|
void main() {
|
||||||
vec2 v = subgroupInclusiveAdd(vec2(1.0, 0.0));
|
float importance = float(gl_LocalInvocationID.x) * 0.5;
|
||||||
|
float prefixSum = subgroupInclusiveAdd(importance);
|
||||||
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
|
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
|
||||||
prefixSumCache[gl_SubgroupID] = v;
|
prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||||
barrier();
|
barrier();
|
||||||
if (gl_LocalInvocationIndex == 0u)
|
uint loopLength = uint(findMSB(gl_NumSubgroups));
|
||||||
outputData.value = prefixSumCache[0].x;
|
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
|
||||||
|
for (uint scanStage = 0u; scanStage < loopLength; ++scanStage) {
|
||||||
|
if ((gl_SubgroupID & (1u << scanStage)) > 0u) {
|
||||||
|
prefixSum += prefixSumCache[(gl_SubgroupID >> scanStage << scanStage) - 1u];
|
||||||
|
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
|
||||||
|
prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||||
|
}
|
||||||
|
barrier();
|
||||||
|
}
|
||||||
|
if (gl_LocalInvocationID.x == 1023u) outputData.value = prefixSumCache[0];
|
||||||
}
|
}
|
||||||
)";
|
)";
|
||||||
|
|
||||||
// Right shape, but a float scan and a float[32] scratch - not the pack's
|
// A subgroup scan, but the scratch is indexed per invocation rather than per
|
||||||
// vec2 accumulator signature.
|
// subgroup: its size is not a subgroup-count assumption, so it is not ours.
|
||||||
constexpr const char* kWrongElementTypeSource = R"(#version 450 core
|
constexpr const char* kInvocationIndexedSource = R"(#version 450 core
|
||||||
#extension GL_KHR_shader_subgroup_basic : require
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
#extension GL_KHR_shader_subgroup_arithmetic : require
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
||||||
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
||||||
shared float cache[32];
|
shared vec2 perInvocation[32];
|
||||||
void main() {
|
void main() {
|
||||||
float v = subgroupInclusiveAdd(float(gl_LocalInvocationIndex));
|
vec2 v = subgroupInclusiveAdd(vec2(float(gl_LocalInvocationIndex), 0.0));
|
||||||
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
|
perInvocation[gl_LocalInvocationIndex & 31u] = v;
|
||||||
cache[gl_SubgroupID] = v;
|
|
||||||
barrier();
|
barrier();
|
||||||
if (gl_LocalInvocationIndex == 0u)
|
if (gl_LocalInvocationIndex == 0u) outputData.value = perInvocation[0].x;
|
||||||
outputData.value = cache[0];
|
}
|
||||||
|
)";
|
||||||
|
|
||||||
|
// gl_SubgroupID-indexed, but no subgroup scan feeds it and the element type is
|
||||||
|
// not the pack's float accumulator.
|
||||||
|
constexpr const char* kNonFloatScratchSource = R"(#version 450 core
|
||||||
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
|
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
||||||
|
layout(std430, binding = 0) buffer Output { uint value; } outputData;
|
||||||
|
shared uint tally[32];
|
||||||
|
void main() {
|
||||||
|
float scan = subgroupInclusiveAdd(float(gl_LocalInvocationIndex));
|
||||||
|
tally[gl_SubgroupID] = uint(scan);
|
||||||
|
barrier();
|
||||||
|
if (gl_LocalInvocationIndex == 0u) outputData.value = tally[0];
|
||||||
|
}
|
||||||
|
)";
|
||||||
|
|
||||||
|
// gl_SubgroupID-indexed, but masked into range: the declaration is bounded by
|
||||||
|
// construction, not a subgroup-count assumption, so it is not the pack's bug.
|
||||||
|
constexpr const char* kMaskedSubgroupIndexSource = R"(#version 450 core
|
||||||
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
|
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
||||||
|
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
||||||
|
shared vec2 bounded[8];
|
||||||
|
void main() {
|
||||||
|
vec2 v = subgroupInclusiveAdd(vec2(float(gl_LocalInvocationIndex), 0.0));
|
||||||
|
bounded[gl_SubgroupID & 7u] = v;
|
||||||
|
barrier();
|
||||||
|
if (gl_LocalInvocationIndex == 0u) outputData.value = bounded[0].x;
|
||||||
|
}
|
||||||
|
)";
|
||||||
|
|
||||||
|
// Neither of the pack's shapes: a small per-subgroup array in a 256-invocation
|
||||||
|
// workgroup, used to prove the width gate keeps EVERY module inert at >= 16 lanes.
|
||||||
|
constexpr const char* kForeignShapeSource = R"(#version 450 core
|
||||||
|
#extension GL_KHR_shader_subgroup_basic : require
|
||||||
|
#extension GL_KHR_shader_subgroup_arithmetic : require
|
||||||
|
layout(local_size_x = 256) in;
|
||||||
|
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
||||||
|
shared float partial[4];
|
||||||
|
void main() {
|
||||||
|
float v = subgroupInclusiveAdd(float(gl_LocalInvocationID.x));
|
||||||
|
if (gl_SubgroupID < 4u) partial[gl_SubgroupID] = v;
|
||||||
|
barrier();
|
||||||
|
if (gl_LocalInvocationID.x == 0u) outputData.value = partial[0];
|
||||||
|
}
|
||||||
|
)";
|
||||||
|
|
||||||
|
// No subgroup construct at all.
|
||||||
|
constexpr const char* kSubgroupFreeSource = R"(#version 450 core
|
||||||
|
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
|
||||||
|
layout(std430, binding = 0) buffer Output { float value; } outputData;
|
||||||
|
shared vec2 scratch[32];
|
||||||
|
void main() {
|
||||||
|
scratch[gl_LocalInvocationIndex & 31u] = vec2(float(gl_LocalInvocationIndex), 0.0);
|
||||||
|
barrier();
|
||||||
|
if (gl_LocalInvocationIndex == 0u) outputData.value = scratch[0].x;
|
||||||
}
|
}
|
||||||
)";
|
)";
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
TEST(FixIterationRPSubgroupScratchPass, GrowsThePacksScratchForNarrowSubgroups) {
|
TEST(FixIterationRPSubgroupScratchPass, GrowsTheExposureScratchForNarrowSubgroups) {
|
||||||
const Vector<Uint32> input = CompileCompute(kIterationRPShapedSource);
|
const Vector<Uint32> input = CompileCompute(kExposureShapedSource);
|
||||||
ASSERT_FALSE(input.empty());
|
ASSERT_FALSE(input.empty());
|
||||||
ASSERT_EQ(WorkgroupArrayLengths(input), (std::vector<Uint32>{4u, 32u}));
|
ASSERT_EQ(WorkgroupArrayLengths(input), (std::vector<Uint32>{4u, 32u}));
|
||||||
|
|
||||||
// lavapipe: 8-lane subgroups over 512 invocations need 64 entries; the
|
// lavapipe: 8-lane subgroups over 512 invocations need 64 entries; the
|
||||||
// plainly indexed neighbour must keep its 4.
|
// plainly indexed neighbour must keep its 4.
|
||||||
Vector<Uint32> output;
|
Vector<Uint32> output;
|
||||||
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, 8u, true));
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, 8u, 32768u, true));
|
||||||
EXPECT_EQ(WorkgroupArrayLengths(output), (std::vector<Uint32>{4u, 64u}));
|
EXPECT_EQ(WorkgroupArrayLengths(output), (std::vector<Uint32>{4u, 64u}));
|
||||||
EXPECT_TRUE(Validates(output));
|
EXPECT_TRUE(Validates(output));
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST(FixIterationRPSubgroupScratchPass, LeavesPackWidthAssumptionsAloneOnWideDevices) {
|
// The regression the CI retrace caught: patching only the exposure pass leaves
|
||||||
const Vector<Uint32> input = CompileCompute(kIterationRPShapedSource);
|
// this one writing 128 subgroups into 64 entries, and the frame stays wrong.
|
||||||
|
TEST(FixIterationRPSubgroupScratchPass, GrowsTheRtwWarpScratchForNarrowSubgroups) {
|
||||||
|
const Vector<Uint32> input = CompileCompute(kRtwWarpShapedSource);
|
||||||
ASSERT_FALSE(input.empty());
|
ASSERT_FALSE(input.empty());
|
||||||
|
ASSERT_EQ(WorkgroupArrayLengths(input), (std::vector<Uint32>{64u}));
|
||||||
|
|
||||||
// >= 16 lanes means at most 32 subgroups: the pack's declared size holds and
|
Vector<Uint32> output;
|
||||||
// the module must pass through byte-identical.
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, 8u, 32768u, true));
|
||||||
|
EXPECT_EQ(WorkgroupArrayLengths(output), (std::vector<Uint32>{128u}));
|
||||||
|
EXPECT_TRUE(Validates(output));
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(FixIterationRPSubgroupScratchPass, LeavesPackWidthAssumptionsAloneOnWideDevices) {
|
||||||
|
// Both shapes are sized for >= 16 lanes (512/16 = 32, 1024/16 = 64), so on
|
||||||
|
// every such device the modules must pass through byte-identical.
|
||||||
|
for (const char* source : {kExposureShapedSource, kRtwWarpShapedSource}) {
|
||||||
|
const Vector<Uint32> input = CompileCompute(source);
|
||||||
|
ASSERT_FALSE(input.empty());
|
||||||
for (const Uint32 nativeSize : {16u, 32u, 64u, 128u}) {
|
for (const Uint32 nativeSize : {16u, 32u, 64u, 128u}) {
|
||||||
Vector<Uint32> output;
|
Vector<Uint32> output;
|
||||||
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, nativeSize, true));
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(
|
||||||
|
input, output, nativeSize, 32768u, true));
|
||||||
EXPECT_EQ(output, input) << "native width " << nativeSize;
|
EXPECT_EQ(output, input) << "native width " << nativeSize;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST(FixIterationRPSubgroupScratchPass, RefusesAModuleOutsideTheFingerprint) {
|
TEST(FixIterationRPSubgroupScratchPass, RefusesAModuleOutsideTheIdiom) {
|
||||||
for (const char* source : {kWrongWorkgroupShapeSource, kWrongElementTypeSource}) {
|
for (const char* source : {kInvocationIndexedSource, kNonFloatScratchSource,
|
||||||
|
kSubgroupFreeSource, kMaskedSubgroupIndexSource}) {
|
||||||
const Vector<Uint32> input = CompileCompute(source);
|
const Vector<Uint32> input = CompileCompute(source);
|
||||||
ASSERT_FALSE(input.empty());
|
ASSERT_FALSE(input.empty());
|
||||||
Vector<Uint32> output;
|
Vector<Uint32> output;
|
||||||
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, 8u, true));
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, 8u, 32768u, true));
|
||||||
EXPECT_EQ(output, input);
|
EXPECT_EQ(output, input);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A grown array that would not fit the device's shared memory is left alone:
|
||||||
|
// a pipeline that cannot be created is worse than the pack's own overrun.
|
||||||
|
// The width gate is what keeps unrelated shaders untouched on the devices the pack
|
||||||
|
// was written for: at >= 16 lanes nothing is rewritten, whatever its shape.
|
||||||
|
TEST(FixIterationRPSubgroupScratchPass, LeavesEveryModuleAloneAtThePacksAssumedWidth) {
|
||||||
|
const Vector<Uint32> input = CompileCompute(kForeignShapeSource);
|
||||||
|
ASSERT_FALSE(input.empty());
|
||||||
|
for (const Uint32 nativeSize : {16u, 32u, 64u}) {
|
||||||
|
Vector<Uint32> output;
|
||||||
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(
|
||||||
|
input, output, nativeSize, 32768u, true));
|
||||||
|
EXPECT_EQ(output, input) << "native width " << nativeSize;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST(FixIterationRPSubgroupScratchPass, RefusesGrowthThatWouldNotFitSharedMemory) {
|
||||||
|
const Vector<Uint32> input = CompileCompute(kRtwWarpShapedSource);
|
||||||
|
ASSERT_FALSE(input.empty());
|
||||||
|
Vector<Uint32> output;
|
||||||
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, output, 8u, 256u, true));
|
||||||
|
EXPECT_EQ(output, input);
|
||||||
|
}
|
||||||
|
|
||||||
TEST(FixIterationRPSubgroupScratchPass, IsIdempotent) {
|
TEST(FixIterationRPSubgroupScratchPass, IsIdempotent) {
|
||||||
const Vector<Uint32> input = CompileCompute(kIterationRPShapedSource);
|
for (const char* source : {kExposureShapedSource, kRtwWarpShapedSource}) {
|
||||||
|
const Vector<Uint32> input = CompileCompute(source);
|
||||||
ASSERT_FALSE(input.empty());
|
ASSERT_FALSE(input.empty());
|
||||||
Vector<Uint32> once;
|
Vector<Uint32> once;
|
||||||
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, once, 8u, true));
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(input, once, 8u, 32768u, true));
|
||||||
Vector<Uint32> twice;
|
Vector<Uint32> twice;
|
||||||
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(once, twice, 8u, true));
|
ASSERT_TRUE(ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(once, twice, 8u, 32768u, true));
|
||||||
EXPECT_EQ(twice, once);
|
EXPECT_EQ(twice, once);
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -912,12 +912,13 @@ namespace MobileGL {
|
|||||||
|
|
||||||
bool ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(
|
bool ShaderCompiler::FixIterationRPSubgroupScratchForVulkan(
|
||||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
|
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
|
||||||
const Uint32 nativeSubgroupSize, const bool enableSpirvValidation) {
|
const Uint32 nativeSubgroupSize, const Uint32 maxWorkgroupScratchBytes,
|
||||||
|
const bool enableSpirvValidation) {
|
||||||
using namespace spvtools;
|
using namespace spvtools;
|
||||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||||
optimizer.RegisterPass(
|
optimizer.RegisterPass(
|
||||||
FixIterationRPSubgroupScratchPass::CreateFixIterationRPSubgroupScratchPass(
|
FixIterationRPSubgroupScratchPass::CreateFixIterationRPSubgroupScratchPass(
|
||||||
nativeSubgroupSize));
|
nativeSubgroupSize, maxWorkgroupScratchBytes));
|
||||||
|
|
||||||
return RunOptimizerChecked("FixIterationRPSubgroupScratchForVulkan", optimizer,
|
return RunOptimizerChecked("FixIterationRPSubgroupScratchForVulkan", optimizer,
|
||||||
inputBinary, outputBinary, true, enableSpirvValidation);
|
inputBinary, outputBinary, true, enableSpirvValidation);
|
||||||
|
|||||||
@@ -167,12 +167,17 @@ namespace MobileGL {
|
|||||||
Vector<uint32_t>& outputBinary,
|
Vector<uint32_t>& outputBinary,
|
||||||
Uint32 maxWorkgroupScratchBytes,
|
Uint32 maxWorkgroupScratchBytes,
|
||||||
bool enableSpirvValidation = false);
|
bool enableSpirvValidation = false);
|
||||||
// Patches iterationRP's under-declared prefixSumCache[32] on sub-16-lane
|
// Grows iterationRP's under-declared gl_SubgroupID-indexed scratch to the
|
||||||
// devices, fingerprint-gated to that pack's reduction; every other module
|
// subgroup count the device actually partitions into, fingerprint-gated to
|
||||||
// passes through byte-identical. See FixIterationRPSubgroupScratchPass.
|
// that pack's reduction idiom; every other module - and every device whose
|
||||||
|
// width the pack already assumed - passes through byte-identical.
|
||||||
|
// maxWorkgroupScratchBytes bounds the growth (pass the device's
|
||||||
|
// maxComputeSharedMemorySize; 0 falls back to the 16384-byte Vulkan
|
||||||
|
// minimum). See FixIterationRPSubgroupScratchPass.
|
||||||
static bool FixIterationRPSubgroupScratchForVulkan(const Vector<Uint32>& inputBinary,
|
static bool FixIterationRPSubgroupScratchForVulkan(const Vector<Uint32>& inputBinary,
|
||||||
Vector<uint32_t>& outputBinary,
|
Vector<uint32_t>& outputBinary,
|
||||||
Uint32 nativeSubgroupSize,
|
Uint32 nativeSubgroupSize,
|
||||||
|
Uint32 maxWorkgroupScratchBytes,
|
||||||
bool enableSpirvValidation = false);
|
bool enableSpirvValidation = false);
|
||||||
// Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair
|
// Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair
|
||||||
// (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no
|
// (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no
|
||||||
|
|||||||
@@ -91,9 +91,15 @@ namespace MobileGL {
|
|||||||
BlockRelayout(IRContext* irContext, Bool std140)
|
BlockRelayout(IRContext* irContext, Bool std140)
|
||||||
: m_irContext(irContext), m_std140(std140) {}
|
: m_irContext(irContext), m_std140(std140) {}
|
||||||
|
|
||||||
// Size and alignment of `typeId`, applying every stride decoration it implies
|
// Size and alignment of `typeId`, QUEUING every offset/stride decoration it
|
||||||
// on the way down. Zero size means "not a type this layout knows how to
|
// implies on the way down. Zero size means "not a type this layout knows how
|
||||||
// describe"; the caller then leaves the block alone rather than guessing.
|
// to describe"; the caller then leaves the block alone rather than guessing.
|
||||||
|
// The queue is what makes that fallback honest: measurement must be
|
||||||
|
// side-effect-free until it is known to succeed, or a mid-struct failure
|
||||||
|
// would leave the block half-relaid-out - members before the failing one at
|
||||||
|
// compacted 32-bit offsets, members after it at the original 64-bit ones, a
|
||||||
|
// layout matching neither convention. Commit() flushes the queue and is
|
||||||
|
// called only on a successful Measure of the whole block.
|
||||||
struct Extent {
|
struct Extent {
|
||||||
Uint32 size = 0;
|
Uint32 size = 0;
|
||||||
Uint32 alignment = 0;
|
Uint32 alignment = 0;
|
||||||
@@ -108,7 +114,29 @@ namespace MobileGL {
|
|||||||
return extent;
|
return extent;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Flushes the decoration writes a successful Measure queued. Call exactly
|
||||||
|
// once, only when Measure returned a non-zero size; a failed measurement's
|
||||||
|
// queue dies with this per-block instance, leaving the module untouched.
|
||||||
|
void Commit() {
|
||||||
|
for (const PendingDecoration& pending : m_pendingWrites) {
|
||||||
|
if (pending.member) {
|
||||||
|
ApplyMemberDecoration(pending.targetId, pending.memberIndex, pending.decoration,
|
||||||
|
pending.value);
|
||||||
|
} else {
|
||||||
|
ApplyTypeDecoration(pending.targetId, pending.decoration, pending.value);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
m_pendingWrites.clear();
|
||||||
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
|
struct PendingDecoration {
|
||||||
|
Bool member = false;
|
||||||
|
Uint32 targetId = 0;
|
||||||
|
Uint32 memberIndex = 0;
|
||||||
|
spv::Decoration decoration = spv::Decoration::Offset;
|
||||||
|
Uint32 value = 0;
|
||||||
|
};
|
||||||
Extent MeasureUncached(Uint32 typeId) {
|
Extent MeasureUncached(Uint32 typeId) {
|
||||||
const Instruction* type = m_irContext->get_def_use_mgr()->GetDef(typeId);
|
const Instruction* type = m_irContext->get_def_use_mgr()->GetDef(typeId);
|
||||||
if (type == nullptr) return {};
|
if (type == nullptr) return {};
|
||||||
@@ -204,7 +232,17 @@ namespace MobileGL {
|
|||||||
return length->GetSingleWordInOperand(0);
|
return length->GetSingleWordInOperand(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Queue-only during measurement; the module is mutated in Commit().
|
||||||
void SetTypeDecoration(Uint32 targetId, spv::Decoration decoration, Uint32 value) {
|
void SetTypeDecoration(Uint32 targetId, spv::Decoration decoration, Uint32 value) {
|
||||||
|
m_pendingWrites.push_back({false, targetId, 0, decoration, value});
|
||||||
|
}
|
||||||
|
|
||||||
|
void SetMemberDecoration(Uint32 structId, Uint32 member, spv::Decoration decoration,
|
||||||
|
Uint32 value) {
|
||||||
|
m_pendingWrites.push_back({true, structId, member, decoration, value});
|
||||||
|
}
|
||||||
|
|
||||||
|
void ApplyTypeDecoration(Uint32 targetId, spv::Decoration decoration, Uint32 value) {
|
||||||
for (Instruction& annotation : m_irContext->annotations()) {
|
for (Instruction& annotation : m_irContext->annotations()) {
|
||||||
if (annotation.opcode() != spv::Op::OpDecorate) continue;
|
if (annotation.opcode() != spv::Op::OpDecorate) continue;
|
||||||
if (annotation.GetSingleWordInOperand(0) != targetId) continue;
|
if (annotation.GetSingleWordInOperand(0) != targetId) continue;
|
||||||
@@ -216,7 +254,7 @@ namespace MobileGL {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void SetMemberDecoration(Uint32 structId, Uint32 member, spv::Decoration decoration,
|
void ApplyMemberDecoration(Uint32 structId, Uint32 member, spv::Decoration decoration,
|
||||||
Uint32 value) {
|
Uint32 value) {
|
||||||
for (Instruction& annotation : m_irContext->annotations()) {
|
for (Instruction& annotation : m_irContext->annotations()) {
|
||||||
if (annotation.opcode() != spv::Op::OpMemberDecorate) continue;
|
if (annotation.opcode() != spv::Op::OpMemberDecorate) continue;
|
||||||
@@ -233,6 +271,7 @@ namespace MobileGL {
|
|||||||
IRContext* m_irContext = nullptr;
|
IRContext* m_irContext = nullptr;
|
||||||
Bool m_std140 = true;
|
Bool m_std140 = true;
|
||||||
std::unordered_map<Uint32, Extent> m_extents;
|
std::unordered_map<Uint32, Extent> m_extents;
|
||||||
|
std::vector<PendingDecoration> m_pendingWrites;
|
||||||
};
|
};
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
@@ -529,9 +568,12 @@ namespace MobileGL {
|
|||||||
// A member shape the layout rules here do not describe. Leaving the block
|
// A member shape the layout rules here do not describe. Leaving the block
|
||||||
// at its 64-bit offsets keeps the module valid for Vulkan; SPIRV-Cross will
|
// at its 64-bit offsets keeps the module valid for Vulkan; SPIRV-Cross will
|
||||||
// decline it for ESSL, which is the same outcome as before the demotion.
|
// decline it for ESSL, which is the same outcome as before the demotion.
|
||||||
|
// Nothing was written: Measure only queues, and the queue dies here.
|
||||||
MGLOG_D("DemoteFloat64Pass: block %%%u contains a member this pass cannot lay "
|
MGLOG_D("DemoteFloat64Pass: block %%%u contains a member this pass cannot lay "
|
||||||
"out; its 64-bit offsets are left in place",
|
"out; its 64-bit offsets are left in place",
|
||||||
blockType->result_id());
|
blockType->result_id());
|
||||||
|
} else {
|
||||||
|
relayout.Commit();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+297
-91
@@ -26,13 +26,13 @@ namespace MobileGL {
|
|||||||
using spvtools::opt::Instruction;
|
using spvtools::opt::Instruction;
|
||||||
using spvtools::opt::IRContext;
|
using spvtools::opt::IRContext;
|
||||||
|
|
||||||
// iterationRP's reduction fingerprint, spelled out.
|
// The Vulkan minimum for maxComputeSharedMemorySize, used when the caller
|
||||||
constexpr uint32_t kIterationRPLocalSizeX = 32u;
|
// could not tell us the device's real limit.
|
||||||
constexpr uint32_t kIterationRPLocalSizeY = 16u;
|
constexpr uint32_t kMinimumSharedMemoryBytes = 16384u;
|
||||||
constexpr uint32_t kIterationRPLocalSizeZ = 1u;
|
|
||||||
constexpr uint32_t kIterationRPInvocations =
|
// The narrowest subgroup width iterationRP's declarations are sized for.
|
||||||
kIterationRPLocalSizeX * kIterationRPLocalSizeY * kIterationRPLocalSizeZ;
|
// At or above it both shipped shapes fit and nothing may be rewritten.
|
||||||
constexpr uint32_t kIterationRPScratchLength = 32u;
|
constexpr uint32_t kPackAssumedSubgroupWidth = 16u;
|
||||||
|
|
||||||
Instruction* FindBuiltinDefinition(IRContext* context, spv::BuiltIn builtin) {
|
Instruction* FindBuiltinDefinition(IRContext* context, spv::BuiltIn builtin) {
|
||||||
auto* defUseMgr = context->get_def_use_mgr();
|
auto* defUseMgr = context->get_def_use_mgr();
|
||||||
@@ -73,18 +73,121 @@ namespace MobileGL {
|
|||||||
return nullptr;
|
return nullptr;
|
||||||
}
|
}
|
||||||
|
|
||||||
// vec2 of 32-bit float - the type of iterationRP's luminance/exposure
|
// A 32-bit float scalar or vector - the shape of every accumulator the
|
||||||
// accumulator and of its prefixSumCache entries.
|
// pack runs through its scans (float, vec2 and vec4 all appear). Returns
|
||||||
bool IsVec2Float32(IRContext* context, uint32_t typeId) {
|
// the component count, or 0 for anything else.
|
||||||
const Instruction* type = context->get_def_use_mgr()->GetDef(typeId);
|
uint32_t Float32ComponentCount(IRContext* context, uint32_t typeId) {
|
||||||
if (type == nullptr || type->opcode() != spv::Op::OpTypeVector ||
|
auto* defUseMgr = context->get_def_use_mgr();
|
||||||
type->GetSingleWordInOperand(1) != 2u) {
|
const Instruction* type = defUseMgr->GetDef(typeId);
|
||||||
|
if (type == nullptr) return 0u;
|
||||||
|
uint32_t components = 1u;
|
||||||
|
if (type->opcode() == spv::Op::OpTypeVector) {
|
||||||
|
components = type->GetSingleWordInOperand(1);
|
||||||
|
if (components < 2u || components > 4u) return 0u;
|
||||||
|
type = defUseMgr->GetDef(type->GetSingleWordInOperand(0));
|
||||||
|
if (type == nullptr) return 0u;
|
||||||
|
}
|
||||||
|
if (type->opcode() != spv::Op::OpTypeFloat ||
|
||||||
|
type->GetSingleWordInOperand(0) != 32u) {
|
||||||
|
return 0u;
|
||||||
|
}
|
||||||
|
return components;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t RoundUp(uint32_t value, uint32_t alignment) {
|
||||||
|
return alignment == 0u ? value : ((value + alignment - 1u) / alignment) * alignment;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Size AND alignment of a workgroup-storage type. Drivers lay shared
|
||||||
|
// memory out at natural alignment and the limit
|
||||||
|
// (VUID-RuntimeSpirv-Workgroup-06530) counts the padding that produces,
|
||||||
|
// so a model that sums unpadded sizes would under-count exactly where the
|
||||||
|
// budget check matters. Returns false for anything not modelled here,
|
||||||
|
// which the caller answers by declining to grow at all rather than by
|
||||||
|
// certifying growth against a total it knows is an underestimate.
|
||||||
|
bool WorkgroupTypeLayout(IRContext* context, uint32_t typeId, uint32_t* size,
|
||||||
|
uint32_t* alignment, uint32_t depth = 0u) {
|
||||||
|
if (depth > 8u) return false;
|
||||||
|
auto* defUseMgr = context->get_def_use_mgr();
|
||||||
|
const Instruction* type = defUseMgr->GetDef(typeId);
|
||||||
|
if (type == nullptr) return false;
|
||||||
|
switch (type->opcode()) {
|
||||||
|
case spv::Op::OpTypeBool:
|
||||||
|
*size = 4u;
|
||||||
|
*alignment = 4u;
|
||||||
|
return true;
|
||||||
|
case spv::Op::OpTypeInt:
|
||||||
|
case spv::Op::OpTypeFloat: {
|
||||||
|
const uint32_t width = type->GetSingleWordInOperand(0) / 8u;
|
||||||
|
if (width == 0u) return false;
|
||||||
|
*size = width;
|
||||||
|
*alignment = width;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
case spv::Op::OpTypeVector: {
|
||||||
|
uint32_t componentSize = 0u;
|
||||||
|
uint32_t componentAlignment = 0u;
|
||||||
|
if (!WorkgroupTypeLayout(context, type->GetSingleWordInOperand(0),
|
||||||
|
&componentSize, &componentAlignment, depth + 1u)) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
const Instruction* component =
|
const uint32_t components = type->GetSingleWordInOperand(1);
|
||||||
context->get_def_use_mgr()->GetDef(type->GetSingleWordInOperand(0));
|
if (components < 2u || components > 4u) return false;
|
||||||
return component != nullptr && component->opcode() == spv::Op::OpTypeFloat &&
|
*size = componentSize * components;
|
||||||
component->GetSingleWordInOperand(0) == 32u;
|
// A three-component vector aligns like a four-component one.
|
||||||
|
*alignment = componentSize * (components == 3u ? 4u : components);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
case spv::Op::OpTypeMatrix:
|
||||||
|
case spv::Op::OpTypeArray: {
|
||||||
|
uint32_t elementSize = 0u;
|
||||||
|
uint32_t elementAlignment = 0u;
|
||||||
|
if (!WorkgroupTypeLayout(context, type->GetSingleWordInOperand(0), &elementSize,
|
||||||
|
&elementAlignment, depth + 1u)) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
uint32_t count = 0u;
|
||||||
|
if (type->opcode() == spv::Op::OpTypeMatrix) {
|
||||||
|
count = type->GetSingleWordInOperand(1);
|
||||||
|
} else {
|
||||||
|
const Instruction* length =
|
||||||
|
defUseMgr->GetDef(type->GetSingleWordInOperand(1));
|
||||||
|
if (length == nullptr || length->opcode() != spv::Op::OpConstant) {
|
||||||
|
return false; // spec-constant length: not sizeable here
|
||||||
|
}
|
||||||
|
count = length->GetSingleWordInOperand(0);
|
||||||
|
}
|
||||||
|
*size = RoundUp(elementSize, elementAlignment) * count;
|
||||||
|
*alignment = elementAlignment;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
case spv::Op::OpTypeStruct: {
|
||||||
|
uint32_t offset = 0u;
|
||||||
|
uint32_t structAlignment = 1u;
|
||||||
|
for (uint32_t i = 0; i < type->NumInOperands(); ++i) {
|
||||||
|
uint32_t memberSize = 0u;
|
||||||
|
uint32_t memberAlignment = 0u;
|
||||||
|
if (!WorkgroupTypeLayout(context, type->GetSingleWordInOperand(i),
|
||||||
|
&memberSize, &memberAlignment, depth + 1u)) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
offset = RoundUp(offset, memberAlignment) + memberSize;
|
||||||
|
if (memberAlignment > structAlignment) structAlignment = memberAlignment;
|
||||||
|
}
|
||||||
|
*size = RoundUp(offset, structAlignment);
|
||||||
|
*alignment = structAlignment;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The group operations the pack's prefix scans use.
|
||||||
|
bool IsScanOrReduce(spv::GroupOperation operation) {
|
||||||
|
return operation == spv::GroupOperation::Reduce ||
|
||||||
|
operation == spv::GroupOperation::InclusiveScan ||
|
||||||
|
operation == spv::GroupOperation::ExclusiveScan;
|
||||||
}
|
}
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
@@ -92,13 +195,15 @@ namespace MobileGL {
|
|||||||
auto* irContext = context();
|
auto* irContext = context();
|
||||||
auto* defUseMgr = irContext->get_def_use_mgr();
|
auto* defUseMgr = irContext->get_def_use_mgr();
|
||||||
|
|
||||||
// A device whose native width already satisfies the pack's assumption
|
// Without a known device width there is no topology to compare against;
|
||||||
// (>= 16 lanes -> at most 32 subgroups) needs no patch at all.
|
// and a width the pack already assumed needs no patch at all. Both of
|
||||||
if (m_nativeSubgroupSize == 0u || m_nativeSubgroupSize >= 16u) {
|
// iterationRP's shapes are sized for >= 16 lanes (512/16 = 32 entries,
|
||||||
|
// 1024/16 = 64), so every module on such a device - the pack's or anyone
|
||||||
|
// else's - must pass through byte-identical. The per-array length test
|
||||||
|
// further down is the second gate, not a replacement for this one.
|
||||||
|
if (m_nativeSubgroupSize == 0u || m_nativeSubgroupSize >= kPackAssumedSubgroupWidth) {
|
||||||
return Status::SuccessWithoutChange;
|
return Status::SuccessWithoutChange;
|
||||||
}
|
}
|
||||||
const uint32_t requiredLength =
|
|
||||||
(kIterationRPInvocations + m_nativeSubgroupSize - 1u) / m_nativeSubgroupSize;
|
|
||||||
|
|
||||||
for (const Instruction& entryPoint : irContext->module()->entry_points()) {
|
for (const Instruction& entryPoint : irContext->module()->entry_points()) {
|
||||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) !=
|
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) !=
|
||||||
@@ -107,7 +212,8 @@ namespace MobileGL {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Fingerprint 1: the pack's exposure-pass workgroup shape, 32x16x1.
|
// Fingerprint 1: a literal workgroup size, so the subgroup count the
|
||||||
|
// dispatch actually partitions into is known here.
|
||||||
const auto resolveUintConstant = [&](uint32_t id, uint32_t* value) {
|
const auto resolveUintConstant = [&](uint32_t id, uint32_t* value) {
|
||||||
const Instruction* def = defUseMgr->GetDef(id);
|
const Instruction* def = defUseMgr->GetDef(id);
|
||||||
if (def == nullptr || def->opcode() != spv::Op::OpConstant) return false;
|
if (def == nullptr || def->opcode() != spv::Op::OpConstant) return false;
|
||||||
@@ -139,26 +245,40 @@ namespace MobileGL {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!haveLocalSize || localSize[0] != kIterationRPLocalSizeX ||
|
if (!haveLocalSize || localSize[0] == 0u || localSize[1] == 0u || localSize[2] == 0u) {
|
||||||
localSize[1] != kIterationRPLocalSizeY || localSize[2] != kIterationRPLocalSizeZ) {
|
|
||||||
return Status::SuccessWithoutChange;
|
return Status::SuccessWithoutChange;
|
||||||
}
|
}
|
||||||
|
const uint64_t totalInvocations =
|
||||||
|
static_cast<uint64_t>(localSize[0]) * localSize[1] * localSize[2];
|
||||||
|
if (totalInvocations == 0u || totalInvocations > (1u << 20)) {
|
||||||
|
return Status::SuccessWithoutChange;
|
||||||
|
}
|
||||||
|
const uint32_t requiredLength = static_cast<uint32_t>(
|
||||||
|
(totalInvocations + m_nativeSubgroupSize - 1u) / m_nativeSubgroupSize);
|
||||||
|
|
||||||
// Fingerprint 2: the reduction's subgroupInclusiveAdd on a vec2.
|
// Fingerprint 2: a subgroup scan over a 32-bit float value - the pack's
|
||||||
bool sawVec2InclusiveAdd = false;
|
// prefix-sum reduction, and the reason its scratch is indexed per subgroup.
|
||||||
|
bool sawFloatSubgroupScan = false;
|
||||||
for (auto& function : *irContext->module()) {
|
for (auto& function : *irContext->module()) {
|
||||||
for (auto& block : function) {
|
for (auto& block : function) {
|
||||||
for (auto& inst : block) {
|
for (auto& inst : block) {
|
||||||
if (inst.opcode() == spv::Op::OpGroupNonUniformFAdd &&
|
if (inst.opcode() != spv::Op::OpGroupNonUniformFAdd &&
|
||||||
static_cast<spv::GroupOperation>(inst.GetSingleWordInOperand(1)) ==
|
inst.opcode() != spv::Op::OpGroupNonUniformFMin &&
|
||||||
spv::GroupOperation::InclusiveScan &&
|
inst.opcode() != spv::Op::OpGroupNonUniformFMax) {
|
||||||
IsVec2Float32(irContext, inst.type_id())) {
|
continue;
|
||||||
sawVec2InclusiveAdd = true;
|
}
|
||||||
|
if (inst.NumInOperands() < 2) continue;
|
||||||
|
if (!IsScanOrReduce(static_cast<spv::GroupOperation>(
|
||||||
|
inst.GetSingleWordInOperand(1)))) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (Float32ComponentCount(irContext, inst.type_id()) != 0u) {
|
||||||
|
sawFloatSubgroupScan = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!sawVec2InclusiveAdd) {
|
if (!sawFloatSubgroupScan) {
|
||||||
return Status::SuccessWithoutChange;
|
return Status::SuccessWithoutChange;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -171,60 +291,89 @@ namespace MobileGL {
|
|||||||
}
|
}
|
||||||
const uint32_t subgroupIdVariableId = subgroupIdVariable->result_id();
|
const uint32_t subgroupIdVariableId = subgroupIdVariable->result_id();
|
||||||
|
|
||||||
// Conservative taint walk over values, and through Function/Private
|
// The pack indexes its scratch with gl_SubgroupID ITSELF, so only values
|
||||||
// temporaries by variable (glslang routinely spills builtin loads into
|
// that ARE that id qualify - not everything computed from it. An index
|
||||||
// locals before they reach an index expression). Over-tainting is safe:
|
// that is masked or clamped (cache[gl_SubgroupID & 3u]) is bounded by
|
||||||
// the candidate filter below still demands the exact vec2[32] shape.
|
// construction and is none of this pass's business; accepting it would
|
||||||
std::unordered_map<uint32_t, bool> valueTainted; // result id -> tainted
|
// turn a targeted repair into a general array resizer. Identity survives
|
||||||
std::unordered_map<uint32_t, bool> variableTainted; // variable id -> tainted
|
// OpCopyObject, a signedness OpBitcast, and the Function/Private spill
|
||||||
bool changedTaint = true;
|
// glslang emits for a builtin load - and nothing else. A spill variable
|
||||||
while (changedTaint) {
|
// counts only when EVERY store into it is the id.
|
||||||
changedTaint = false;
|
std::unordered_map<uint32_t, bool> subgroupIdValues; // result id IS the id
|
||||||
|
std::unordered_map<uint32_t, bool> subgroupIdVariables; // spill holding only it
|
||||||
|
bool changedIdentity = true;
|
||||||
|
while (changedIdentity) {
|
||||||
|
changedIdentity = false;
|
||||||
|
|
||||||
|
std::unordered_map<uint32_t, uint32_t> totalStores;
|
||||||
|
std::unordered_map<uint32_t, uint32_t> idStores;
|
||||||
for (auto& function : *irContext->module()) {
|
for (auto& function : *irContext->module()) {
|
||||||
for (auto& block : function) {
|
for (auto& block : function) {
|
||||||
for (auto& inst : block) {
|
for (auto& inst : block) {
|
||||||
const spv::Op opcode = inst.opcode();
|
if (inst.opcode() != spv::Op::OpStore) continue;
|
||||||
if (opcode == spv::Op::OpStore) {
|
|
||||||
if (!valueTainted.count(inst.GetSingleWordInOperand(1))) continue;
|
|
||||||
const Instruction* root =
|
|
||||||
RootVariable(irContext, inst.GetSingleWordInOperand(0));
|
|
||||||
if (root == nullptr) continue;
|
|
||||||
if (!variableTainted.count(root->result_id())) {
|
|
||||||
variableTainted[root->result_id()] = true;
|
|
||||||
changedTaint = true;
|
|
||||||
}
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (inst.result_id() == 0 || valueTainted.count(inst.result_id())) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
bool tainted = false;
|
|
||||||
if (opcode == spv::Op::OpLoad) {
|
|
||||||
const uint32_t pointerId = inst.GetSingleWordInOperand(0);
|
const uint32_t pointerId = inst.GetSingleWordInOperand(0);
|
||||||
if (pointerId == subgroupIdVariableId) tainted = true;
|
const Instruction* target = defUseMgr->GetDef(pointerId);
|
||||||
const Instruction* root = RootVariable(irContext, pointerId);
|
if (target == nullptr || target->opcode() != spv::Op::OpVariable) {
|
||||||
if (root != nullptr && variableTainted.count(root->result_id())) {
|
continue;
|
||||||
tainted = true;
|
|
||||||
}
|
}
|
||||||
} else {
|
const auto storageClass = static_cast<spv::StorageClass>(
|
||||||
inst.ForEachInId([&](const uint32_t* operandId) {
|
target->GetSingleWordInOperand(0));
|
||||||
if (valueTainted.count(*operandId)) tainted = true;
|
if (storageClass != spv::StorageClass::Function &&
|
||||||
});
|
storageClass != spv::StorageClass::Private) {
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
if (tainted) {
|
totalStores[pointerId] += 1u;
|
||||||
valueTainted[inst.result_id()] = true;
|
if (subgroupIdValues.count(inst.GetSingleWordInOperand(1))) {
|
||||||
changedTaint = true;
|
idStores[pointerId] += 1u;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (const auto& entry : totalStores) {
|
||||||
|
if (entry.second != 0u && idStores[entry.first] == entry.second &&
|
||||||
|
!subgroupIdVariables.count(entry.first)) {
|
||||||
|
subgroupIdVariables[entry.first] = true;
|
||||||
|
changedIdentity = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (auto& function : *irContext->module()) {
|
||||||
|
for (auto& block : function) {
|
||||||
|
for (auto& inst : block) {
|
||||||
|
if (inst.result_id() == 0 ||
|
||||||
|
subgroupIdValues.count(inst.result_id())) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
bool isSubgroupId = false;
|
||||||
|
switch (inst.opcode()) {
|
||||||
|
case spv::Op::OpLoad: {
|
||||||
|
const uint32_t pointerId = inst.GetSingleWordInOperand(0);
|
||||||
|
isSubgroupId = pointerId == subgroupIdVariableId ||
|
||||||
|
subgroupIdVariables.count(pointerId) != 0u;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
case spv::Op::OpCopyObject:
|
||||||
|
case spv::Op::OpBitcast:
|
||||||
|
isSubgroupId =
|
||||||
|
subgroupIdValues.count(inst.GetSingleWordInOperand(0)) != 0u;
|
||||||
|
break;
|
||||||
|
default:
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (isSubgroupId) {
|
||||||
|
subgroupIdValues[inst.result_id()] = true;
|
||||||
|
changedIdentity = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (valueTainted.empty()) {
|
if (subgroupIdValues.empty()) {
|
||||||
return Status::SuccessWithoutChange;
|
return Status::SuccessWithoutChange;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Fingerprint 3: workgroup-shared vec2[32] arrays whose access-chain
|
// Fingerprint 3: workgroup-shared float arrays indexed by gl_SubgroupID
|
||||||
// index depends on gl_SubgroupID - the under-declared prefixSumCache.
|
// itself - the under-declared prefixSumCache.
|
||||||
std::map<uint32_t, Instruction*> candidates;
|
std::map<uint32_t, Instruction*> candidates;
|
||||||
for (auto& function : *irContext->module()) {
|
for (auto& function : *irContext->module()) {
|
||||||
for (auto& block : function) {
|
for (auto& block : function) {
|
||||||
@@ -234,7 +383,7 @@ namespace MobileGL {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (inst.NumInOperands() < 2) continue;
|
if (inst.NumInOperands() < 2) continue;
|
||||||
if (!valueTainted.count(inst.GetSingleWordInOperand(1))) continue;
|
if (!subgroupIdValues.count(inst.GetSingleWordInOperand(1))) continue;
|
||||||
Instruction* baseVariable =
|
Instruction* baseVariable =
|
||||||
defUseMgr->GetDef(inst.GetSingleWordInOperand(0));
|
defUseMgr->GetDef(inst.GetSingleWordInOperand(0));
|
||||||
if (baseVariable == nullptr ||
|
if (baseVariable == nullptr ||
|
||||||
@@ -252,7 +401,16 @@ namespace MobileGL {
|
|||||||
return Status::SuccessWithoutChange;
|
return Status::SuccessWithoutChange;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool changedModule = false;
|
// Everything that survives the filter, with the bytes each grown array
|
||||||
|
// will need. Nothing is mutated until the whole set fits the device's
|
||||||
|
// shared-memory budget, so a module is never left half-grown.
|
||||||
|
struct Growth {
|
||||||
|
Instruction* variable = nullptr;
|
||||||
|
uint32_t elementTypeId = 0;
|
||||||
|
uint32_t lengthTypeId = 0;
|
||||||
|
uint32_t addedBytes = 0;
|
||||||
|
};
|
||||||
|
std::vector<Growth> growths;
|
||||||
for (auto& entry : candidates) {
|
for (auto& entry : candidates) {
|
||||||
Instruction* variable = entry.second;
|
Instruction* variable = entry.second;
|
||||||
|
|
||||||
@@ -290,18 +448,70 @@ namespace MobileGL {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
const uint32_t elementTypeId = arrayType->GetSingleWordInOperand(0);
|
const uint32_t elementTypeId = arrayType->GetSingleWordInOperand(0);
|
||||||
if (!IsVec2Float32(irContext, elementTypeId)) continue;
|
const uint32_t components = Float32ComponentCount(irContext, elementTypeId);
|
||||||
|
if (components == 0u) continue;
|
||||||
const Instruction* lengthConstant =
|
const Instruction* lengthConstant =
|
||||||
defUseMgr->GetDef(arrayType->GetSingleWordInOperand(1));
|
defUseMgr->GetDef(arrayType->GetSingleWordInOperand(1));
|
||||||
uint32_t currentLength = 0;
|
if (lengthConstant == nullptr || lengthConstant->opcode() != spv::Op::OpConstant) {
|
||||||
if (lengthConstant == nullptr ||
|
|
||||||
lengthConstant->opcode() != spv::Op::OpConstant ||
|
|
||||||
!((currentLength = lengthConstant->GetSingleWordInOperand(0),
|
|
||||||
currentLength == kIterationRPScratchLength))) {
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
const uint32_t currentLength = lengthConstant->GetSingleWordInOperand(0);
|
||||||
|
|
||||||
|
// The pack's own assumption holds on this device: the declared array
|
||||||
|
// already covers every subgroup the workgroup partitions into. That is
|
||||||
|
// every >= 16-lane device for the shapes iterationRP ships, and those
|
||||||
|
// modules must pass through byte-identical.
|
||||||
if (currentLength >= requiredLength) continue;
|
if (currentLength >= requiredLength) continue;
|
||||||
|
|
||||||
|
// vec3 strides at its 16-byte alignment, so charge the padded stride.
|
||||||
|
const uint32_t elementStride = (components == 3u ? 4u : components) * 4u;
|
||||||
|
growths.push_back(Growth{variable, elementTypeId, lengthConstant->type_id(),
|
||||||
|
(requiredLength - currentLength) * elementStride});
|
||||||
|
}
|
||||||
|
if (growths.empty()) {
|
||||||
|
return Status::SuccessWithoutChange;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Growing must not push the module past what the device can launch: a
|
||||||
|
// pipeline that fails to create is worse than the pack's own overrun.
|
||||||
|
{
|
||||||
|
uint64_t declaredBytes = 0;
|
||||||
|
bool sawUnsizeable = false;
|
||||||
|
for (auto& global : irContext->module()->types_values()) {
|
||||||
|
if (global.opcode() != spv::Op::OpVariable ||
|
||||||
|
static_cast<spv::StorageClass>(global.GetSingleWordInOperand(0)) !=
|
||||||
|
spv::StorageClass::Workgroup) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const Instruction* pointerType = defUseMgr->GetDef(global.type_id());
|
||||||
|
uint32_t bytes = 0u;
|
||||||
|
uint32_t alignment = 0u;
|
||||||
|
if (pointerType == nullptr ||
|
||||||
|
pointerType->opcode() != spv::Op::OpTypePointer ||
|
||||||
|
!WorkgroupTypeLayout(irContext, pointerType->GetSingleWordInOperand(1),
|
||||||
|
&bytes, &alignment)) {
|
||||||
|
sawUnsizeable = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
declaredBytes = RoundUp(static_cast<uint32_t>(declaredBytes), alignment) + bytes;
|
||||||
|
}
|
||||||
|
// A declaration this pass cannot size leaves the total an
|
||||||
|
// underestimate, so the growth cannot be certified against the device
|
||||||
|
// limit at all - decline rather than guess.
|
||||||
|
if (sawUnsizeable) {
|
||||||
|
return Status::SuccessWithoutChange;
|
||||||
|
}
|
||||||
|
for (const Growth& growth : growths) declaredBytes += growth.addedBytes;
|
||||||
|
|
||||||
|
const uint32_t deviceBudget = m_maxWorkgroupScratchBytes != 0u
|
||||||
|
? m_maxWorkgroupScratchBytes
|
||||||
|
: kMinimumSharedMemoryBytes;
|
||||||
|
if (declaredBytes > deviceBudget) {
|
||||||
|
return Status::SuccessWithoutChange;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const Growth& growth : growths) {
|
||||||
// Build the grown array type. All three new instructions are inserted
|
// Build the grown array type. All three new instructions are inserted
|
||||||
// immediately BEFORE the variable so definition-before-use holds in the
|
// immediately BEFORE the variable so definition-before-use holds in the
|
||||||
// module's global section (manager-created instructions append to its
|
// module's global section (manager-created instructions append to its
|
||||||
@@ -310,17 +520,17 @@ namespace MobileGL {
|
|||||||
// scalar constant is legal SPIR-V); the fresh array type makes the
|
// scalar constant is legal SPIR-V); the fresh array type makes the
|
||||||
// pointer type unique by construction, so neither collides with an
|
// pointer type unique by construction, so neither collides with an
|
||||||
// existing declaration.
|
// existing declaration.
|
||||||
const uint32_t lengthTypeId = lengthConstant->type_id();
|
Instruction* variable = growth.variable;
|
||||||
const uint32_t newLengthId = irContext->TakeNextId();
|
const uint32_t newLengthId = irContext->TakeNextId();
|
||||||
variable->InsertBefore(spvtools::MakeUnique<Instruction>(
|
variable->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||||
irContext, spv::Op::OpConstant, lengthTypeId, newLengthId,
|
irContext, spv::Op::OpConstant, growth.lengthTypeId, newLengthId,
|
||||||
Instruction::OperandList{{SPV_OPERAND_TYPE_TYPED_LITERAL_NUMBER,
|
Instruction::OperandList{{SPV_OPERAND_TYPE_TYPED_LITERAL_NUMBER,
|
||||||
{requiredLength}}}));
|
{requiredLength}}}));
|
||||||
const uint32_t newArrayTypeId = irContext->TakeNextId();
|
const uint32_t newArrayTypeId = irContext->TakeNextId();
|
||||||
variable->InsertBefore(spvtools::MakeUnique<Instruction>(
|
variable->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||||
irContext, spv::Op::OpTypeArray, 0, newArrayTypeId,
|
irContext, spv::Op::OpTypeArray, 0, newArrayTypeId,
|
||||||
Instruction::OperandList{
|
Instruction::OperandList{
|
||||||
{SPV_OPERAND_TYPE_ID, {elementTypeId}},
|
{SPV_OPERAND_TYPE_ID, {growth.elementTypeId}},
|
||||||
{SPV_OPERAND_TYPE_ID, {newLengthId}}}));
|
{SPV_OPERAND_TYPE_ID, {newLengthId}}}));
|
||||||
const uint32_t newPointerTypeId = irContext->TakeNextId();
|
const uint32_t newPointerTypeId = irContext->TakeNextId();
|
||||||
variable->InsertBefore(spvtools::MakeUnique<Instruction>(
|
variable->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||||
@@ -331,21 +541,17 @@ namespace MobileGL {
|
|||||||
{SPV_OPERAND_TYPE_ID, {newArrayTypeId}}}));
|
{SPV_OPERAND_TYPE_ID, {newArrayTypeId}}}));
|
||||||
|
|
||||||
variable->SetResultType(newPointerTypeId);
|
variable->SetResultType(newPointerTypeId);
|
||||||
changedModule = true;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!changedModule) {
|
|
||||||
return Status::SuccessWithoutChange;
|
|
||||||
}
|
|
||||||
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||||
return Status::SuccessWithChange;
|
return Status::SuccessWithChange;
|
||||||
}
|
}
|
||||||
|
|
||||||
spvtools::Optimizer::PassToken
|
spvtools::Optimizer::PassToken
|
||||||
FixIterationRPSubgroupScratchPass::CreateFixIterationRPSubgroupScratchPass(
|
FixIterationRPSubgroupScratchPass::CreateFixIterationRPSubgroupScratchPass(
|
||||||
const Uint32 nativeSubgroupSize) {
|
const Uint32 nativeSubgroupSize, const Uint32 maxWorkgroupScratchBytes) {
|
||||||
return spvtools::Optimizer::PassToken(
|
return spvtools::Optimizer::PassToken(MakeUnique<FixIterationRPSubgroupScratchPass>(
|
||||||
MakeUnique<FixIterationRPSubgroupScratchPass>(nativeSubgroupSize));
|
nativeSubgroupSize, maxWorkgroupScratchBytes));
|
||||||
}
|
}
|
||||||
} // namespace ShaderTranspiler
|
} // namespace ShaderTranspiler
|
||||||
} // namespace MG_Util
|
} // namespace MG_Util
|
||||||
|
|||||||
@@ -16,48 +16,57 @@
|
|||||||
namespace MobileGL {
|
namespace MobileGL {
|
||||||
namespace MG_Util {
|
namespace MG_Util {
|
||||||
namespace ShaderTranspiler {
|
namespace ShaderTranspiler {
|
||||||
// Patches ONE known shader-pack defect: iterationRP's auto-exposure reduction
|
// Patches ONE known shader-pack defect: iterationRP hard-sizes the scratch
|
||||||
// declares `shared vec2 prefixSumCache[32]` for its 512-invocation workgroup
|
// its subgroup prefix scans write through prefixSumCache[gl_SubgroupID].
|
||||||
// and stores per-subgroup subtotals through prefixSumCache[gl_SubgroupID].
|
// The pack ships that idiom twice, sized for the >= 16-lane subgroups
|
||||||
// The pack hard-sized that scratch for the >=16-lane subgroups desktop GL
|
// desktop GL drivers give it:
|
||||||
// drivers ship; on a narrower Vulkan device (lavapipe's 8 lanes -> 64
|
// - the auto-exposure reduction: 32x16 (512 invocations), vec2[32];
|
||||||
// subgroups) every subgroup past entry 31 indexes shared memory out of
|
// - the RTW importance warp: 1024 invocations, float[64].
|
||||||
// bounds - on a CPU rasterizer that is literal heap corruption. The
|
// On a narrower Vulkan device (lavapipe's 8 lanes -> 64 and 128 subgroups)
|
||||||
// reduction ALGORITHM is width-agnostic (its combine loop is sized by
|
// every subgroup past the last declared entry indexes shared memory out of
|
||||||
// gl_NumSubgroups), so the faithful repair is to grow the one under-declared
|
// bounds - on a CPU rasterizer that is literal heap corruption. Both
|
||||||
// array to ceil(512 / native width) and change nothing else. This is the
|
// reduction ALGORITHMS are width-agnostic (their combine loops are sized by
|
||||||
// pack author's bug, not MobileGL's; the patch is therefore deliberately
|
// gl_NumSubgroups), so the faithful repair is to grow the under-declared
|
||||||
// NOT a general mechanism - it only rewrites modules that positively match
|
// arrays to ceil(invocations / native width) and change nothing else.
|
||||||
// iterationRP's reduction fingerprint:
|
|
||||||
// - GLCompute entry point with local size exactly 32x16x1;
|
|
||||||
// - a subgroupInclusiveAdd on a vec2 (OpGroupNonUniformFAdd InclusiveScan,
|
|
||||||
// the pack's luminance/exposure accumulator signature);
|
|
||||||
// - a workgroup-shared array of exactly vec2[32] whose access-chain index
|
|
||||||
// is data-dependent on gl_SubgroupID.
|
|
||||||
// Matching at the SPIR-V level keeps the recognition robust against
|
|
||||||
// whitespace/identifier-level drift that made the old source-text template
|
|
||||||
// rewrite (removed in 7769156) so brittle, while still refusing to touch
|
|
||||||
// anything that is not this pack's reduction. On devices whose native width
|
|
||||||
// already satisfies the pack's assumption (>= 16 lanes: desktop GL, Adreno),
|
|
||||||
// the grown length equals or undershoots the declared 32 and every module
|
|
||||||
// passes through byte-identical.
|
|
||||||
//
|
//
|
||||||
// The pass never fails a module: anything it cannot prove is this exact
|
// This is the pack author's bug, not MobileGL's, so the patch is
|
||||||
// pattern - or cannot grow safely (a whole-array use, a spec-constant
|
// deliberately NOT a general "resize shared arrays" mechanism. It rewrites
|
||||||
// length, an initializer) - is left exactly as it was.
|
// an array only when the module positively matches the pack's reduction
|
||||||
|
// idiom AND the device's own topology proves the declaration too small:
|
||||||
|
// - GLCompute entry point with a literal workgroup size;
|
||||||
|
// - a subgroup scan/reduce over a 32-bit float scalar or vector
|
||||||
|
// (OpGroupNonUniformF{Add,Min,Max}), the pack's accumulator signature;
|
||||||
|
// - a workgroup-shared array of 32-bit float scalars/vectors whose
|
||||||
|
// access-chain index is data-dependent on gl_SubgroupID;
|
||||||
|
// - a declared length strictly below ceil(invocations / native width).
|
||||||
|
// That last clause is what keeps the patch inert wherever the pack is
|
||||||
|
// correct: on any device whose width satisfies the pack's assumption
|
||||||
|
// (>= 16 lanes: desktop GL, Adreno) both shapes already fit and every
|
||||||
|
// module passes through byte-identical. Matching at the SPIR-V level keeps
|
||||||
|
// recognition robust against the whitespace/identifier drift that made the
|
||||||
|
// old source-text template rewrite (removed in 7769156) so brittle.
|
||||||
|
//
|
||||||
|
// The pass never fails a module: anything it cannot prove is this pattern -
|
||||||
|
// or cannot grow safely (a whole-array use, a spec-constant length, an
|
||||||
|
// initializer, or growth that would not fit maxWorkgroupScratchBytes) - is
|
||||||
|
// left exactly as it was. Pass the device's maxComputeSharedMemorySize as
|
||||||
|
// maxWorkgroupScratchBytes; 0 falls back to the 16384-byte Vulkan minimum.
|
||||||
class FixIterationRPSubgroupScratchPass : public spvtools::opt::Pass {
|
class FixIterationRPSubgroupScratchPass : public spvtools::opt::Pass {
|
||||||
public:
|
public:
|
||||||
explicit FixIterationRPSubgroupScratchPass(Uint32 nativeSubgroupSize)
|
FixIterationRPSubgroupScratchPass(Uint32 nativeSubgroupSize,
|
||||||
: m_nativeSubgroupSize(nativeSubgroupSize) {}
|
Uint32 maxWorkgroupScratchBytes)
|
||||||
|
: m_nativeSubgroupSize(nativeSubgroupSize),
|
||||||
|
m_maxWorkgroupScratchBytes(maxWorkgroupScratchBytes) {}
|
||||||
|
|
||||||
const char* name() const override { return "fix-iterationrp-subgroup-scratch"; }
|
const char* name() const override { return "fix-iterationrp-subgroup-scratch"; }
|
||||||
Status Process() override;
|
Status Process() override;
|
||||||
|
|
||||||
static spvtools::Optimizer::PassToken CreateFixIterationRPSubgroupScratchPass(
|
static spvtools::Optimizer::PassToken CreateFixIterationRPSubgroupScratchPass(
|
||||||
Uint32 nativeSubgroupSize);
|
Uint32 nativeSubgroupSize, Uint32 maxWorkgroupScratchBytes);
|
||||||
|
|
||||||
private:
|
private:
|
||||||
Uint32 m_nativeSubgroupSize;
|
Uint32 m_nativeSubgroupSize;
|
||||||
|
Uint32 m_maxWorkgroupScratchBytes;
|
||||||
};
|
};
|
||||||
} // namespace ShaderTranspiler
|
} // namespace ShaderTranspiler
|
||||||
} // namespace MG_Util
|
} // namespace MG_Util
|
||||||
|
|||||||
Reference in New Issue
Block a user