diff --git a/MobileGL/Config.h b/MobileGL/Config.h index 78626248..b67181d7 100644 --- a/MobileGL/Config.h +++ b/MobileGL/Config.h @@ -155,6 +155,11 @@ namespace MobileGL::MG_Config { // per-draw glBufferSubData path instead of the persistent-mapped ring allocator // (negative control / driver-bug escape hatch). Bool DisableUboRing = false; + // MOBILEGL_DISABLE_UNPACK_RING: force DirectGLES texture uploads back to + // glTexSubImage from the client pointer instead of staging them through the + // persistent-mapped unpack-PBO ring (negative control / driver-bug escape + // hatch). + Bool DisableUnpackRing = false; // MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES // depth/stencil reads and always go through the shader-sampling emulation. Core GL // ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does, diff --git a/MobileGL/ConfigLoader.cpp b/MobileGL/ConfigLoader.cpp index 23cb5f1d..7283791f 100644 --- a/MobileGL/ConfigLoader.cpp +++ b/MobileGL/ConfigLoader.cpp @@ -183,6 +183,7 @@ namespace MobileGL::MG_ConfigLoader { features.CoherentAsFlush = QueryEnvFlag("MOBILEGL_COHERENT_AS_FLUSH"); features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY"); features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING"); + features.DisableUnpackRing = QueryEnvFlag("MOBILEGL_DISABLE_UNPACK_RING"); features.EsprytForceDepthStencilReadbackEmulation = QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION"); features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS"); diff --git a/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp b/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp index a3b31cc1..68375387 100644 --- a/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp +++ b/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp @@ -9985,9 +9985,10 @@ namespace MobileGL::MG_Backend::DirectGLES { g_completedFrameSerial.store(completed, std::memory_order_relaxed); } - // After the watermark advanced: retire grown-away UBO-ring stores and record - // the frame's ring high-water mark for slot reclamation. + // After the watermark advanced: retire grown-away ring stores and record the + // frame's ring high-water marks for slot reclamation. BufferImpl::UboRingOnPresent(); + BufferImpl::UnpackRingOnPresent(); BufferImpl::TrimBufferPool(); } diff --git a/MobileGL/MG_Backend/DirectGLES/Managers.cpp b/MobileGL/MG_Backend/DirectGLES/Managers.cpp index 3ba144f7..df8f632c 100644 --- a/MobileGL/MG_Backend/DirectGLES/Managers.cpp +++ b/MobileGL/MG_Backend/DirectGLES/Managers.cpp @@ -630,11 +630,26 @@ namespace MobileGL::MG_Backend::DirectGLES { return 0; } - // --- Global-UBO ring (see Managers.h) ------------------------------------ + // --- Persistent-mapped bump rings (see Managers.h) ----------------------- + // Shared machinery behind BOTH the global-UBO ring and the texture + // unpack-PBO ring: one EXT_buffer_storage persistent|coherent map per ring, + // monotonic head/tail cursors, and reclamation riding the Present() + // frame-fence watermark. The two rings differ only in size cap, offset + // alignment and log label - the reclamation, context-loss and + // emergency-drain rules are the part that was hard to get right, so they + // are shared rather than copied. constexpr SizeT kUboRingInitialBytes = 4u * 1024u * 1024u; constexpr SizeT kUboRingMaxBytes = 64u * 1024u * 1024u; + constexpr SizeT kUnpackRingInitialBytes = 4u * 1024u * 1024u; + constexpr SizeT kUnpackRingMaxBytes = 64u * 1024u * 1024u; + // A PBO-sourced glTexSubImage constrains the offset only to the pixel + // TYPE's size (GL_INVALID_OPERATION otherwise), and no ES client type is + // wider than 4 bytes - unlike a UBO bind, which owes the driver + // GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT. 64 covers every type with room to + // spare and keeps consecutive staged blocks off each other's cache lines. + constexpr SizeT kUnpackRingAlignment = 64; - struct UboRingState { + struct PersistentRingStore { Uint id = 0; Uint8* mappedPtr = nullptr; SizeT size = 0; @@ -648,41 +663,64 @@ namespace MobileGL::MG_Backend::DirectGLES { Uint contextGeneration = 0; SizeT alignment = 256; // A hard storage-creation failure under this context; stop retrying - // per draw (cleared when the context generation moves on). + // per use (cleared when the context generation moves on). Bool creationFailed = false; }; - UboRingState g_uboRing; // Grown-away ring stores: deletable only once the GPU finished the last // frame that could reference them (same watermark as the buffer pool). - struct RetiredUboRing { + struct RetiredRingStore { Uint id = 0; Uint contextGeneration = 0; Uint64 retireSerial = 0; }; - Vector g_retiredUboRings; // Present()-time high-water marks: every byte below headAtPresent was // written during frames <= frameSerial, so once frameSerial completes, // tail may advance to headAtPresent. FIFO by construction. - struct UboRingFrameMark { + struct RingFrameMark { Uint64 frameSerial = 0; Uint64 headAtPresent = 0; }; - Vector g_uboRingFrameMarks; + + // One ring: its live store, its two reclamation lists, and the immutable + // knobs that tell it apart from the other one. + struct PersistentRing { + PersistentRingStore store; + Vector retired; + Vector frameMarks; + SizeT initialBytes = 0; + SizeT maxBytes = 0; + // 0: take the offset alignment from GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT + // at store-creation time (the UBO ring's binds require it). + SizeT fixedAlignment = 0; + const char* label = ""; + }; + + PersistentRing g_uboRing{{}, {}, {}, kUboRingInitialBytes, kUboRingMaxBytes, 0, "Global-UBO ring"}; + PersistentRing g_unpackRing{{}, + {}, + {}, + kUnpackRingInitialBytes, + kUnpackRingMaxBytes, + kUnpackRingAlignment, + "Texture unpack ring"}; // The ES context the ring's id/map belonged to is gone (or was never // seen): drop every handle without GL calls and re-arm creation. The // generation counter must survive the reset — frame serials also survive // context recreation, so a restarted counter could revalidate a stale // per-program slot cache against the new ring. - void ResetUboRingForNewContext() { - const Uint32 keptGeneration = g_uboRing.generation; - g_uboRing = {}; - g_uboRing.generation = keptGeneration; - g_uboRing.contextGeneration = g_bufferContextGeneration; - g_retiredUboRings.clear(); - g_uboRingFrameMarks.clear(); + void ResetRingForNewContext(PersistentRing& ring) { + const Uint32 keptGeneration = ring.store.generation; + ring.store = {}; + ring.store.generation = keptGeneration; + ring.store.contextGeneration = g_bufferContextGeneration; + // Keep the ring's own alignment across the wipe: the slow path rounds a + // request with it BEFORE the store that would set it exists. + if (ring.fixedAlignment != 0) ring.store.alignment = ring.fixedAlignment; + ring.retired.clear(); + ring.frameMarks.clear(); } GLESBufferResource* ResourceOf(BufferObject& bufferObject) { @@ -1097,9 +1135,10 @@ namespace MobileGL::MG_Backend::DirectGLES { InvalidateArrayBufferBindingCache(); InvalidateIndexedBufferBindingCache(); InvalidatePixelBufferBindingCaches(); - // The global-UBO ring's id and persistent map died with the context; - // drop the handles (no GL) and let the next draw recreate the ring. - ResetUboRingForNewContext(); + // The rings' ids and persistent maps died with the context; drop the + // handles (no GL) and let the next draw / texture upload recreate them. + ResetRingForNewContext(g_uboRing); + ResetRingForNewContext(g_unpackRing); } void ProcessDeferredBufferReleases() { @@ -1441,26 +1480,26 @@ namespace MobileGL::MG_Backend::DirectGLES { g_pooledBytes = 0; } - // --- Global-UBO ring (see Managers.h) ------------------------------------ + // --- Persistent-mapped bump rings (see Managers.h) ------------------------ namespace { // (Re)create the ring store with room for at least minBytes. Any live // store is retired (deleted once the GPU finished the last frame that // could reference its slots), never deleted in place. Returns false and // leaves the current store untouched when minBytes cannot fit under the // size cap; a GL failure loses the store and latches creationFailed so - // draws stop retrying under this context. - Bool CreateUboRingStorage(SizeT minBytes) { - SizeT newSize = kUboRingInitialBytes; + // later uses stop retrying under this context. + Bool CreateRingStorage(PersistentRing& ring, SizeT minBytes) { + SizeT newSize = ring.initialBytes; while (newSize < minBytes) newSize *= 2; - if (newSize > kUboRingMaxBytes) return false; + if (newSize > ring.maxBytes) return false; - if (g_uboRing.id != 0) { - g_retiredUboRings.push_back( - {g_uboRing.id, g_uboRing.contextGeneration, DirectGLES::CurrentFrameSerial() + 1}); + if (ring.store.id != 0) { + ring.retired.push_back( + {ring.store.id, ring.store.contextGeneration, DirectGLES::CurrentFrameSerial() + 1}); } - const Uint32 nextGeneration = g_uboRing.generation + 1; - g_uboRing.id = 0; - g_uboRing.mappedPtr = nullptr; + const Uint32 nextGeneration = ring.store.generation + 1; + ring.store.id = 0; + ring.store.mappedPtr = nullptr; Uint id = 0; g_GLESFuncs.glGenBuffers(1, &id); @@ -1477,213 +1516,249 @@ namespace MobileGL::MG_Backend::DirectGLES { g_GLESFuncs.glDeleteBuffers(1, &id); id = 0; } else { - g_uboRing.mappedPtr = static_cast(ptr); + ring.store.mappedPtr = static_cast(ptr); } } if (id == 0) { - MGLOG_E_ONCE("Global-UBO ring: persistent storage creation failed (%zu bytes); " - "falling back to glBufferSubData uploads.", - newSize); - g_uboRing.creationFailed = true; + MGLOG_E_ONCE("%s: persistent storage creation failed (%zu bytes); falling back to the " + "client-memory upload path.", + ring.label, newSize); + ring.store.creationFailed = true; return false; } - const GLint capsAlignment = g_GLESCapabilities.UniformBufferOffsetAlignment; - g_uboRing.id = id; - g_uboRing.size = newSize; - g_uboRing.head = 0; - g_uboRing.tail = 0; - g_uboRing.generation = nextGeneration; - g_uboRing.alignment = capsAlignment > 0 ? static_cast(capsAlignment) : 256; - g_uboRingFrameMarks.clear(); - MGLOG_D("Global-UBO ring: %zu MiB persistent store ready (id %u, gen %u, align %zu).", - newSize / (1024u * 1024u), id, nextGeneration, g_uboRing.alignment); + SizeT alignment = ring.fixedAlignment; + if (alignment == 0) { + const GLint capsAlignment = g_GLESCapabilities.UniformBufferOffsetAlignment; + alignment = capsAlignment > 0 ? static_cast(capsAlignment) : 256; + } + ring.store.id = id; + ring.store.size = newSize; + ring.store.head = 0; + ring.store.tail = 0; + ring.store.generation = nextGeneration; + ring.store.alignment = alignment; + ring.frameMarks.clear(); + MGLOG_D("%s: %zu MiB persistent store ready (id %u, gen %u, align %zu).", ring.label, + newSize / (1024u * 1024u), id, nextGeneration, alignment); return true; } - } // namespace - Bool UboRingAvailable() { - if (MG_Config::Features.DisableUboRing) return false; - // Reclamation rides the Present fence watermark; without working fences - // slots would never be provably GPU-idle (same rule as IsPoolable). - if (!g_GLESFuncs.glBufferStorageEXT || !g_GLESFuncs.glMapBufferRange || !g_GLESFuncs.glGenBuffers || - !g_GLESFuncs.glFenceSync || !g_GLESFuncs.glGetSynciv) { - return false; + // Shared half of the availability gate (the per-ring kill switch sits in + // the exported wrappers). Reclamation rides the Present fence watermark; + // without working fences slots would never be provably GPU-idle (same rule + // as IsPoolable). + Bool RingAvailable(PersistentRing& ring) { + if (!g_GLESFuncs.glBufferStorageEXT || !g_GLESFuncs.glMapBufferRange || !g_GLESFuncs.glGenBuffers || + !g_GLESFuncs.glFenceSync || !g_GLESFuncs.glGetSynciv) { + return false; + } + if (!CanTouchGLNow()) return false; + if (ring.store.contextGeneration != g_bufferContextGeneration) { + ResetRingForNewContext(ring); + } + return !ring.store.creationFailed; } - if (!CanTouchGLNow()) return false; - if (g_uboRing.contextGeneration != g_bufferContextGeneration) { - ResetUboRingForNewContext(); - } - return !g_uboRing.creationFailed; - } - namespace { // Division-based rounding fallback: the spec doesn't promise a power-of-two // alignment. Slot offsets stay multiples of the alignment because every // slot size is, and wrap padding restarts at ring offset 0. - inline SizeT UboRingAlignUp(SizeT size, SizeT alignment) { + inline SizeT RingAlignUp(SizeT size, SizeT alignment) { if ((alignment & (alignment - 1)) == 0) { return (size + alignment - 1) & ~(alignment - 1); } return (size + alignment - 1) / alignment * alignment; } - Bool UboRingAllocateSlow(SizeT size, SizeT& outOffset); - } // namespace + Bool RingAllocateSlow(PersistentRing& ring, SizeT size, SizeT& outOffset); - Bool UboRingAllocate(SizeT size, SizeT& outOffset) { - if (size == 0) return false; - // Fast path: a live ring under the current context with room before both + // Bump-allocate `size` bytes out of `ring`. + // + // Fast path: a live store under the current context with room before both // the wrap boundary and the in-flight tail. Touches no GL and probes no - // frame marks - the sole caller sits behind UboRingAvailable() in the - // draw preparation, so the context checks have already run this draw. - // `tail` may be stale here (marks are only retired on Present and on the - // slow path); staleness is conservative - the in-flight span reads too - // large, the check fails, and the slow path retires marks and re-tries. - auto& ring = g_uboRing; - if (ring.id != 0 && ring.contextGeneration == g_bufferContextGeneration) { - const SizeT alignedSize = UboRingAlignUp(size, ring.alignment); - // Ring sizes are kUboRingInitialBytes (a power of two) doubled some - // number of times, so the offset modulo reduces to a mask. - static_assert((kUboRingInitialBytes & (kUboRingInitialBytes - 1)) == 0, - "ring offset mask below requires power-of-two ring sizes"); - const SizeT offset = static_cast(ring.head & (ring.size - 1)); - if (offset + alignedSize <= ring.size && - ring.head + alignedSize - ring.tail <= ring.size) { - ring.head += alignedSize; - outOffset = offset; - return true; - } - } - return UboRingAllocateSlow(size, outOffset); - } - - namespace { - Bool UboRingAllocateSlow(SizeT size, SizeT& outOffset) { - if (!UboRingAvailable()) return false; - const SizeT alignedSize = UboRingAlignUp(size, g_uboRing.alignment); - if (g_uboRing.id == 0 && !CreateUboRingStorage(alignedSize)) { - return false; - } - - // Advance tail past every frame the GPU provably finished. - const Uint64 completed = DirectGLES::CompletedFrameSerial(); - SizeT retiredMarks = 0; - for (const auto& mark : g_uboRingFrameMarks) { - if (mark.frameSerial > completed) break; - if (mark.headAtPresent > g_uboRing.tail) g_uboRing.tail = mark.headAtPresent; - ++retiredMarks; - } - if (retiredMarks > 0) { - g_uboRingFrameMarks.erase(g_uboRingFrameMarks.begin(), - g_uboRingFrameMarks.begin() + static_cast(retiredMarks)); - } - - // A slot may not straddle the ring end; pad the cursor to the boundary. - SizeT offset = static_cast(g_uboRing.head % g_uboRing.size); - if (offset + alignedSize > g_uboRing.size) { - g_uboRing.head += g_uboRing.size - offset; - offset = 0; - } - - if (g_uboRing.head + alignedSize - g_uboRing.tail > g_uboRing.size) { - // In-flight span would overrun live slots: grow instead of overwrite. - if (CreateUboRingStorage(std::max(g_uboRing.size * 2, alignedSize))) { - offset = 0; - } else if (g_uboRing.creationFailed) { - return false; // store lost; callers fall back to glBufferSubData - } else { - // At the size cap (>kUboRingMaxBytes of uniforms in flight). First - // try to free room by waiting for the OLDEST in-flight frames to - // retire - a bounded wait that ends as soon as enough tail space - // exists, instead of draining the entire queue. - constexpr Uint64 kFrameWaitNs = 50ull * 1000 * 1000; // 50ms per frame - while (!g_uboRingFrameMarks.empty() && - g_uboRing.head + alignedSize - g_uboRing.tail > g_uboRing.size) { - const auto& oldest = g_uboRingFrameMarks.front(); - if (!DirectGLES::WaitForFrameSerialCompleted(oldest.frameSerial, kFrameWaitNs)) { - break; - } - if (oldest.headAtPresent > g_uboRing.tail) g_uboRing.tail = oldest.headAtPresent; - g_uboRingFrameMarks.erase(g_uboRingFrameMarks.begin()); - } - if (g_uboRing.head + alignedSize - g_uboRing.tail <= g_uboRing.size) { - offset = static_cast(g_uboRing.head % g_uboRing.size); - if (offset + alignedSize > g_uboRing.size) { - g_uboRing.head += g_uboRing.size - offset; - offset = 0; - } - g_uboRing.head += alignedSize; + // frame marks - callers sit behind the ring's availability gate, so the + // context checks have already run for this draw/upload. `tail` may be stale + // here (marks are only retired on Present and on the slow path); staleness + // is conservative - the in-flight span reads too large, the check fails, + // and the slow path retires marks and re-tries. + Bool RingAllocate(PersistentRing& ring, SizeT size, SizeT& outOffset) { + if (size == 0) return false; + auto& store = ring.store; + if (store.id != 0 && store.contextGeneration == g_bufferContextGeneration) { + const SizeT alignedSize = RingAlignUp(size, store.alignment); + // Ring sizes are the ring's initial size (a power of two) doubled + // some number of times, so the offset modulo reduces to a mask. + static_assert((kUboRingInitialBytes & (kUboRingInitialBytes - 1)) == 0, + "ring offset mask below requires power-of-two ring sizes"); + static_assert((kUnpackRingInitialBytes & (kUnpackRingInitialBytes - 1)) == 0, + "ring offset mask below requires power-of-two ring sizes"); + const SizeT offset = static_cast(store.head & (store.size - 1)); + if (offset + alignedSize <= store.size && store.head + alignedSize - store.tail <= store.size) { + store.head += alignedSize; outOffset = offset; return true; } - // No usable fence covers the oldest frames: drain once rather than - // corrupt live slots. - if (g_GLESFuncs.glFinish) g_GLESFuncs.glFinish(); - g_uboRing.tail = g_uboRing.head; - g_uboRingFrameMarks.clear(); - // Same-frame slots written before the drain may now be recycled by - // the very next allocations; a generation bump keeps later draws - // from rebinding those cached offsets. - ++g_uboRing.generation; - offset = static_cast(g_uboRing.head % g_uboRing.size); - if (offset + alignedSize > g_uboRing.size) { - g_uboRing.head += g_uboRing.size - offset; + } + return RingAllocateSlow(ring, size, outOffset); + } + + Bool RingAllocateSlow(PersistentRing& ring, SizeT size, SizeT& outOffset) { + if (!RingAvailable(ring)) return false; + auto& store = ring.store; + // Sizing the store first, so the request is rounded with the alignment + // the live store actually carries rather than the pre-creation default. + if (store.id == 0 && !CreateRingStorage(ring, RingAlignUp(size, store.alignment))) { + return false; + } + const SizeT alignedSize = RingAlignUp(size, store.alignment); + + // Advance tail past every frame the GPU provably finished. + const Uint64 completed = DirectGLES::CompletedFrameSerial(); + SizeT retiredMarks = 0; + for (const auto& mark : ring.frameMarks) { + if (mark.frameSerial > completed) break; + if (mark.headAtPresent > store.tail) store.tail = mark.headAtPresent; + ++retiredMarks; + } + if (retiredMarks > 0) { + ring.frameMarks.erase(ring.frameMarks.begin(), + ring.frameMarks.begin() + static_cast(retiredMarks)); + } + + // A slot may not straddle the ring end; pad the cursor to the boundary. + SizeT offset = static_cast(store.head % store.size); + if (offset + alignedSize > store.size) { + store.head += store.size - offset; + offset = 0; + } + + if (store.head + alignedSize - store.tail > store.size) { + // In-flight span would overrun live slots: grow instead of overwrite. + if (CreateRingStorage(ring, std::max(store.size * 2, alignedSize))) { offset = 0; + } else if (store.creationFailed) { + return false; // store lost; callers fall back to their legacy path + } else { + // At the size cap (>maxBytes in flight). First try to free room + // by waiting for the OLDEST in-flight frames to retire - a + // bounded wait that ends as soon as enough tail space exists, + // instead of draining the entire queue. + constexpr Uint64 kFrameWaitNs = 50ull * 1000 * 1000; // 50ms per frame + while (!ring.frameMarks.empty() && + store.head + alignedSize - store.tail > store.size) { + const auto& oldest = ring.frameMarks.front(); + if (!DirectGLES::WaitForFrameSerialCompleted(oldest.frameSerial, kFrameWaitNs)) { + break; + } + if (oldest.headAtPresent > store.tail) store.tail = oldest.headAtPresent; + ring.frameMarks.erase(ring.frameMarks.begin()); + } + if (store.head + alignedSize - store.tail <= store.size) { + offset = static_cast(store.head % store.size); + if (offset + alignedSize > store.size) { + store.head += store.size - offset; + offset = 0; + } + store.head += alignedSize; + outOffset = offset; + return true; + } + // No usable fence covers the oldest frames: drain once rather than + // corrupt live slots. + if (g_GLESFuncs.glFinish) g_GLESFuncs.glFinish(); + store.tail = store.head; + ring.frameMarks.clear(); + // Same-frame slots written before the drain may now be recycled by + // the very next allocations; a generation bump keeps later draws + // from rebinding those cached offsets. + ++store.generation; + offset = static_cast(store.head % store.size); + if (offset + alignedSize > store.size) { + store.head += store.size - offset; + offset = 0; + } } } + + store.head += alignedSize; + outOffset = offset; + return true; } - g_uboRing.head += alignedSize; - outOffset = offset; - return true; - } + // Present()-time upkeep shared by both rings. + void RingOnPresent(PersistentRing& ring) { + if (!CanTouchGLNow()) return; + + // Delete grown-away stores the GPU is provably done with. + const Uint64 completed = DirectGLES::CompletedFrameSerial(); + for (SizeT i = ring.retired.size(); i-- > 0;) { + RetiredRingStore& entry = ring.retired[i]; + const Bool staleContext = entry.contextGeneration != g_bufferContextGeneration; + if (!staleContext && entry.retireSerial > completed) continue; + if (!staleContext && entry.id != 0) { + ScrubBufferBindingShadowsForId(entry.id); + g_GLESFuncs.glDeleteBuffers(1, &entry.id); + } + ring.retired[i] = ring.retired.back(); + ring.retired.pop_back(); + } + + auto& store = ring.store; + if (store.id == 0 || store.contextGeneration != g_bufferContextGeneration) return; + // Retire completed marks here too — RingAllocate is the main consumer, + // but frames that used the ring for nothing would otherwise let the list + // grow one entry per Present, unboundedly. + SizeT retiredMarks = 0; + for (const auto& mark : ring.frameMarks) { + if (mark.frameSerial > completed) break; + if (mark.headAtPresent > store.tail) store.tail = mark.headAtPresent; + ++retiredMarks; + } + if (retiredMarks > 0) { + ring.frameMarks.erase(ring.frameMarks.begin(), + ring.frameMarks.begin() + static_cast(retiredMarks)); + } + // Record this frame's high-water mark (Present just fenced the serial now + // reported by CurrentFrameSerial()). A fence-less Present repeats the + // serial; fold into the existing mark. + const Uint64 serial = DirectGLES::CurrentFrameSerial(); + if (!ring.frameMarks.empty() && ring.frameMarks.back().frameSerial == serial) { + ring.frameMarks.back().headAtPresent = store.head; + } else { + ring.frameMarks.push_back({serial, store.head}); + } + } } // namespace - void* UboRingMappedPtr() { return g_uboRing.mappedPtr; } - Uint UboRingBufferId() { return g_uboRing.id; } - Uint32 UboRingGeneration() { return g_uboRing.generation; } - - void UboRingOnPresent() { - if (!CanTouchGLNow()) return; - - // Delete grown-away stores the GPU is provably done with. - const Uint64 completed = DirectGLES::CompletedFrameSerial(); - for (SizeT i = g_retiredUboRings.size(); i-- > 0;) { - RetiredUboRing& entry = g_retiredUboRings[i]; - const Bool staleContext = entry.contextGeneration != g_bufferContextGeneration; - if (!staleContext && entry.retireSerial > completed) continue; - if (!staleContext && entry.id != 0) { - ScrubBufferBindingShadowsForId(entry.id); - g_GLESFuncs.glDeleteBuffers(1, &entry.id); - } - g_retiredUboRings[i] = g_retiredUboRings.back(); - g_retiredUboRings.pop_back(); - } - - if (g_uboRing.id == 0 || g_uboRing.contextGeneration != g_bufferContextGeneration) return; - // Retire completed marks here too — UboRingAllocate is the main consumer, - // but frames with no global-UBO draws would otherwise let the list grow - // one entry per Present, unboundedly. - SizeT retiredMarks = 0; - for (const auto& mark : g_uboRingFrameMarks) { - if (mark.frameSerial > completed) break; - if (mark.headAtPresent > g_uboRing.tail) g_uboRing.tail = mark.headAtPresent; - ++retiredMarks; - } - if (retiredMarks > 0) { - g_uboRingFrameMarks.erase(g_uboRingFrameMarks.begin(), - g_uboRingFrameMarks.begin() + static_cast(retiredMarks)); - } - // Record this frame's high-water mark (Present just fenced the serial now - // reported by CurrentFrameSerial()). A fence-less Present repeats the - // serial; fold into the existing mark. - const Uint64 serial = DirectGLES::CurrentFrameSerial(); - if (!g_uboRingFrameMarks.empty() && g_uboRingFrameMarks.back().frameSerial == serial) { - g_uboRingFrameMarks.back().headAtPresent = g_uboRing.head; - } else { - g_uboRingFrameMarks.push_back({serial, g_uboRing.head}); - } + Bool UboRingAvailable() { + if (MG_Config::Features.DisableUboRing) return false; + return RingAvailable(g_uboRing); } + + Bool UboRingAllocate(SizeT size, SizeT& outOffset) { return RingAllocate(g_uboRing, size, outOffset); } + + void* UboRingMappedPtr() { return g_uboRing.store.mappedPtr; } + Uint UboRingBufferId() { return g_uboRing.store.id; } + Uint32 UboRingGeneration() { return g_uboRing.store.generation; } + + void UboRingOnPresent() { RingOnPresent(g_uboRing); } + + Bool UnpackRingAvailable() { + if (MG_Config::Features.DisableUnpackRing) return false; + return RingAvailable(g_unpackRing); + } + + Bool UnpackRingAllocate(SizeT size, SizeT& outOffset) { + // A request the ring could never satisfy even empty would otherwise walk + // the whole grow/drain ladder before failing. + if (size > kUnpackRingMaxBytes) return false; + return RingAllocate(g_unpackRing, size, outOffset); + } + + void* UnpackRingMappedPtr() { return g_unpackRing.store.mappedPtr; } + Uint UnpackRingBufferId() { return g_unpackRing.store.id; } + SizeT UnpackRingMaxBytes() { return kUnpackRingMaxBytes; } + + void UnpackRingOnPresent() { RingOnPresent(g_unpackRing); } } // namespace BufferImpl namespace VertexArrayImpl { @@ -2580,6 +2655,93 @@ namespace MobileGL::MG_Backend::DirectGLES { static inline GLint s_skipImages = 0; }; + // --- Unpack-ring staging (see BufferImpl::UnpackRingAvailable) --------------- + // One rectangular (or whole-level) region of a level shadow, repacked TIGHTLY + // into the persistently-mapped unpack PBO. The upload that follows passes the + // returned ring offset with UNPACK_ROW_LENGTH / UNPACK_IMAGE_HEIGHT left at the + // surrounding ScopedDefaultUnpackState's 0, so the driver reads exactly the + // bytes staged here - strictly fewer than the ROW_LENGTH-strided client-pointer + // upload it replaces, which made the driver walk the whole level's stride. + // + // `blocks` describes one or more source regions to stage back-to-back into a + // SINGLE ring allocation; the ring offset of block i lands in blocks[i].offset. + // Deliberately without default member initializers: call sites declare a + // kMaxDirtyRects-sized array of these per dirty level and fill only the entries + // they use, so the type stays trivially default-constructible and the + // declaration costs nothing on the levels that never reach the ring. + struct UnpackStagingBlock { + const Uint8* src; // top-left texel of the region in the level shadow + SizeT rowBytes; // bytes per region row (region width * bpp) + SizeT rows; // region height + SizeT slices; // region depth (1 for 2D) + SizeT srcRowStride; // level row pitch + SizeT srcSliceStride; // level slice pitch + SizeT offset; // out: byte offset into the ring store + }; + + // False (nothing staged, ring untouched) whenever the caller must keep the + // client-pointer path: ring unavailable, a degenerate block, or a total that + // overflows / exceeds the ring's cap. + static Bool StageBlocksIntoUnpackRing(UnpackStagingBlock* blocks, SizeT blockCount) { + if (blocks == nullptr || blockCount == 0) return false; + if (!BufferImpl::UnpackRingAvailable()) return false; + + const SizeT maxBytes = BufferImpl::UnpackRingMaxBytes(); + SizeT total = 0; + for (SizeT i = 0; i < blockCount; ++i) { + const UnpackStagingBlock& b = blocks[i]; + if (b.src == nullptr || b.rowBytes == 0 || b.rows == 0 || b.slices == 0) return false; + // Every factor is bounded by the level's own byte size, but the + // multiplications are on SizeT: check each against the cap instead of + // trusting a product that could have wrapped. + if (b.rowBytes > maxBytes || b.rows > maxBytes / b.rowBytes) return false; + const SizeT sliceBytes = b.rowBytes * b.rows; + if (b.slices > maxBytes / sliceBytes) return false; + const SizeT blockBytes = sliceBytes * b.slices; + if (blockBytes > maxBytes - total) return false; + total += blockBytes; + } + + SizeT base = 0; + if (!BufferImpl::UnpackRingAllocate(total, base)) return false; + // AFTER the allocation: growing the ring replaces the store and its map. + auto* ringBytes = static_cast(BufferImpl::UnpackRingMappedPtr()); + if (ringBytes == nullptr) return false; + + SizeT cursor = base; + for (SizeT i = 0; i < blockCount; ++i) { + UnpackStagingBlock& b = blocks[i]; + b.offset = cursor; + Uint8* dst = ringBytes + cursor; + // Each block's size is a whole number of texels, so consecutive block + // offsets stay multiples of the pixel type's size just as `base` is. + if (b.rowBytes == b.srcRowStride && b.rowBytes * b.rows == b.srcSliceStride) { + Memcpy(dst, b.src, b.rowBytes * b.rows * b.slices); // region IS the level + } else { + for (SizeT z = 0; z < b.slices; ++z) { + const Uint8* srcSlice = b.src + z * b.srcSliceStride; + if (b.rowBytes == b.srcRowStride) { + Memcpy(dst, srcSlice, b.rowBytes * b.rows); // full-width rows + dst += b.rowBytes * b.rows; + continue; + } + for (SizeT y = 0; y < b.rows; ++y) { + Memcpy(dst, srcSlice + y * b.srcRowStride, b.rowBytes); + dst += b.rowBytes; + } + } + } + cursor += b.rowBytes * b.rows * b.slices; + } + return true; + } + + // The `pixels` argument of a PBO-sourced glTexSubImage: a byte offset dressed + // up as a pointer. + static const void* UnpackRingPixelOffset(SizeT offset) { + return reinterpret_cast(static_cast(offset)); + } + static Uint GetNormFallbackComponentCount(TextureInternalFormat format) { switch (format) { case TextureInternalFormat::R8Snorm: @@ -3847,34 +4009,111 @@ namespace MobileGL::MG_Backend::DirectGLES { static_cast(rect.lo.y()) * levelRowBytes + static_cast(rect.lo.x()) * bpp; }; + // Unpack-ring staging plan, decided ONCE for whichever branch + // below runs: either every glTexSubImage of this level sources + // from the ring or none does, so the pixel-unpack binding is + // toggled exactly once per level. The plan's three shapes line + // up 1:1 with the switch's three branches. + // + // Regions are repacked TIGHTLY (row length = the region's own + // width), which is why the ring path issues no glPixelStorei at + // all: the surrounding ScopedDefaultUnpackState already holds + // ROW_LENGTH/IMAGE_HEIGHT at 0, which is exactly what a tight + // block wants. It also stages strictly fewer bytes than the + // client-pointer path makes the driver walk, which strides over + // the whole level width. + UnpackStagingBlock + stagingBlocks[MG_State::GLState::MipmapStorage::kMaxDirtyRects]; + SizeT stagingBlockCount = 0; + const Bool ringUsable = BufferImpl::UnpackRingAvailable(); + if (!ringUsable) { + // Nothing to plan: every branch below keeps the client + // pointer and its ROW_LENGTH striding, unchanged. + } else if (subRectEligible && dirtyRectCount >= 2) { + for (SizeT r = 0; r < dirtyRectCount; ++r) { + const auto& rect = dirtyRects[r]; + stagingBlocks[r] = { + rectShadowPtr(rect), + static_cast(rect.hi.x() - rect.lo.x()) * bpp, + static_cast(rect.hi.y() - rect.lo.y()), + static_cast(std::max(rect.hi.z() - rect.lo.z(), 1)), + levelRowBytes, + levelSliceBytes, + 0}; + } + stagingBlockCount = dirtyRectCount; + } else if (subRectEligible) { + stagingBlocks[0] = {regionPtr, + static_cast(regionSize.x()) * bpp, + static_cast(regionSize.y()), + static_cast(std::max(regionSize.z(), 1)), + levelRowBytes, + levelSliceBytes, + 0}; + stagingBlockCount = 1; + } else if (uploadData == mipData && texelCount > 0 && + byteSize % texelCount == 0) { + // Whole level, shadow bytes verbatim: `byteSize` is exactly + // what the driver would have read from the client pointer, + // and the shadow is already tightly packed. A CONVERTED + // level stays on the pointer path - its buffer's length is + // the conversion's, not the shadow's, so nothing here can + // size the driver's read of it. The whole-texels check is + // the same sanity the sub-rect path applies, and it is what + // makes "the shadow's bytes per texel == the transfer's" + // hold. (A 1D array's upload size permutes height and depth, + // but the texel count and the tight byte order are the same, + // so the flat copy still describes what the driver reads.) + stagingBlocks[0] = {static_cast(uploadData), + static_cast(byteSize), + 1, + 1, + static_cast(byteSize), + static_cast(byteSize), + 0}; + stagingBlockCount = 1; + } + const Bool ringStaged = + stagingBlockCount > 0 && + StageBlocksIntoUnpackRing(stagingBlocks, stagingBlockCount); + if (ringStaged) { + BufferImpl::BindPixelUnpackBufferId(BufferImpl::UnpackRingBufferId()); + } switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) { case TextureTarget::Texture2D: case TextureTarget::TextureCubeMap: if (subRectEligible && dirtyRectCount >= 2) { - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); + if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); for (SizeT r = 0; r < dirtyRectCount; ++r) { const auto& rect = dirtyRects[r]; g_GLESFuncs.glTexSubImage2D( glUploadTarget, static_cast(level), rect.lo.x(), rect.lo.y(), static_cast(rect.hi.x() - rect.lo.x()), static_cast(rect.hi.y() - rect.lo.y()), glFormat, - glType, rectShadowPtr(rect)); + glType, + ringStaged ? UnpackRingPixelOffset(stagingBlocks[r].offset) + : static_cast(rectShadowPtr(rect))); } // The surrounding ScopedDefaultUnpackState shadow says 0. - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); + if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); } else if (subRectEligible) { - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); + if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); g_GLESFuncs.glTexSubImage2D( glUploadTarget, static_cast(level), dirtyRegion.lo.x(), dirtyRegion.lo.y(), static_cast(regionSize.x()), - static_cast(regionSize.y()), glFormat, glType, regionPtr); + static_cast(regionSize.y()), glFormat, glType, + ringStaged ? UnpackRingPixelOffset(stagingBlocks[0].offset) + : static_cast(regionPtr)); // The surrounding ScopedDefaultUnpackState shadow says 0. - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); + if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); } else { g_GLESFuncs.glTexSubImage2D(glUploadTarget, static_cast(level), 0, 0, static_cast(uploadSize.x()), static_cast(uploadSize.y()), glFormat, - glType, uploadData); + glType, + ringStaged + ? UnpackRingPixelOffset(stagingBlocks[0].offset) + : uploadData); } break; case TextureTarget::Texture3D: @@ -3883,8 +4122,10 @@ namespace MobileGL::MG_Backend::DirectGLES { // like a 2D array whose depth is 6 * the cube count. case TextureTarget::TextureCubeMapArray: if (subRectEligible && dirtyRectCount >= 2) { - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); - g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y()); + if (!ringStaged) { + g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); + g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y()); + } for (SizeT r = 0; r < dirtyRectCount; ++r) { const auto& rect = dirtyRects[r]; g_GLESFuncs.glTexSubImage3D( @@ -3893,27 +4134,40 @@ namespace MobileGL::MG_Backend::DirectGLES { static_cast(rect.hi.x() - rect.lo.x()), static_cast(rect.hi.y() - rect.lo.y()), static_cast(rect.hi.z() - rect.lo.z()), glFormat, - glType, rectShadowPtr(rect)); + glType, + ringStaged ? UnpackRingPixelOffset(stagingBlocks[r].offset) + : static_cast(rectShadowPtr(rect))); + } + if (!ringStaged) { + g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); + g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0); } - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); - g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0); } else if (subRectEligible) { - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); - g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y()); + if (!ringStaged) { + g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x()); + g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y()); + } g_GLESFuncs.glTexSubImage3D( glUploadTarget, static_cast(level), dirtyRegion.lo.x(), dirtyRegion.lo.y(), dirtyRegion.lo.z(), static_cast(regionSize.x()), static_cast(regionSize.y()), - static_cast(regionSize.z()), glFormat, glType, regionPtr); - g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); - g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0); + static_cast(regionSize.z()), glFormat, glType, + ringStaged ? UnpackRingPixelOffset(stagingBlocks[0].offset) + : static_cast(regionPtr)); + if (!ringStaged) { + g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); + g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0); + } } else { g_GLESFuncs.glTexSubImage3D(glUploadTarget, static_cast(level), 0, 0, 0, static_cast(uploadSize.x()), static_cast(uploadSize.y()), static_cast(uploadSize.z()), glFormat, - glType, uploadData); + glType, + ringStaged + ? UnpackRingPixelOffset(stagingBlocks[0].offset) + : uploadData); } break; default: @@ -3921,6 +4175,12 @@ namespace MobileGL::MG_Backend::DirectGLES { MG_Util::ConvertTextureTargetToString(stateTextureObject->GetTarget()).c_str()); break; } + if (ringStaged) { + // Back to the resting unbound state every other upload site + // in this file assumes (their BindPixelUnpackBufferId(0) is + // meant to stay a shadow no-op). + BufferImpl::BindPixelUnpackBufferId(0); + } textureMipmapObject->MarkStorageDirty(uploadTarget, level, false); } } diff --git a/MobileGL/MG_Backend/DirectGLES/Managers.h b/MobileGL/MG_Backend/DirectGLES/Managers.h index 13227ab8..eb646143 100644 --- a/MobileGL/MG_Backend/DirectGLES/Managers.h +++ b/MobileGL/MG_Backend/DirectGLES/Managers.h @@ -510,6 +510,40 @@ namespace MobileGL::MG_Backend::DirectGLES { // Present()-time upkeep: records the frame's high-water mark for reclamation // and deletes grown-away ring stores once the GPU is done with them. void UboRingOnPresent(); + + // --- Texture unpack-PBO ring ---------------------------------------------- + // The same persistent-mapped bump allocator, staging TEXTURE UPLOADS. A + // glTexSubImage from client memory hands the driver a pointer it must read + // before the call returns, so the copy has to be ordered against whatever GPU + // work still reads the destination texture: Mali resolves that by BLOCKING the + // calling thread (osup_sync_object_wait) instead of ghosting, and Minecraft + // re-uploads animated atlas sprites and the lightmap every tick into textures + // the in-flight frame is still sampling. Staging the bytes into a + // GPU-visible unpack PBO and passing an OFFSET instead lets the driver queue + // the copy in the command stream with no CPU wait at all. + // + // Same reclamation contract as the UBO ring: no ring bytes are recycled before + // the frame that referenced them completed on the GPU, so a staged block stays + // intact for as long as the queued transfer can still be reading it. The store + // therefore settles at roughly (bytes staged per frame) x (frames in flight), + // which is what to watch if this ring ever shows up in an RSS regression: it + // grows on demand from 4 MiB and is capped, not unbounded. + // + // False when the feature is disabled (MOBILEGL_DISABLE_UNPACK_RING), + // EXT_buffer_storage / fences are missing, the ES context is not current, or + // ring creation already failed under this context. Callers then upload from + // the client pointer exactly as before. + Bool UnpackRingAvailable(); + // Bump-allocate `size` bytes aligned to 64 (a PBO-sourced glTexSubImage only + // owes the driver the pixel type's own alignment). Grows the ring when the + // in-flight span would be overrun; false when the request exceeds the ring's + // size cap or storage (re)creation fails. + Bool UnpackRingAllocate(SizeT size, SizeT& outOffset); + void* UnpackRingMappedPtr(); + Uint UnpackRingBufferId(); + // Largest single staging request the ring can ever satisfy. + SizeT UnpackRingMaxBytes(); + void UnpackRingOnPresent(); } // namespace BufferImpl namespace VertexArrayImpl {