diff --git a/MobileGL/Config.h b/MobileGL/Config.h index 014d4c56..900b9502 100644 --- a/MobileGL/Config.h +++ b/MobileGL/Config.h @@ -170,6 +170,12 @@ namespace MobileGL::MG_Config { // persistent-mapped unpack-PBO ring (negative control / driver-bug escape // hatch). Bool DisableUnpackRing = false; + // MOBILEGL_DISABLE_UPLOAD_RING: force DirectGLES app buffer updates + // (glBufferSubData / map flushes) back to the immediate driver upload instead + // of queueing them for the staged-copy flush through the persistent-mapped + // upload ring (negative control / driver-bug escape hatch; the immediate + // upload stalls on drivers that resolve the WAR hazard on the CPU, e.g. Mali). + Bool DisableUploadRing = false; // MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES // depth/stencil reads and always go through the shader-sampling emulation. Core GL // ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does, diff --git a/MobileGL/ConfigLoader.cpp b/MobileGL/ConfigLoader.cpp index 084f491a..6d2745c9 100644 --- a/MobileGL/ConfigLoader.cpp +++ b/MobileGL/ConfigLoader.cpp @@ -185,6 +185,7 @@ namespace MobileGL::MG_ConfigLoader { features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY"); features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING"); features.DisableUnpackRing = QueryEnvFlag("MOBILEGL_DISABLE_UNPACK_RING"); + features.DisableUploadRing = QueryEnvFlag("MOBILEGL_DISABLE_UPLOAD_RING"); features.EsprytForceDepthStencilReadbackEmulation = QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION"); features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS"); diff --git a/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp b/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp index 8df6f184..726cbf8d 100644 --- a/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp +++ b/MobileGL/MG_Backend/DirectGLES/DirectGLES.cpp @@ -10636,6 +10636,7 @@ namespace MobileGL::MG_Backend::DirectGLES { // frame's ring high-water marks for slot reclamation. BufferImpl::UboRingOnPresent(); BufferImpl::UnpackRingOnPresent(); + BufferImpl::UploadRingOnPresent(); BufferImpl::TrimBufferPool(); } diff --git a/MobileGL/MG_Backend/DirectGLES/Managers.cpp b/MobileGL/MG_Backend/DirectGLES/Managers.cpp index ffd79a3c..f03609e2 100644 --- a/MobileGL/MG_Backend/DirectGLES/Managers.cpp +++ b/MobileGL/MG_Backend/DirectGLES/Managers.cpp @@ -648,6 +648,11 @@ namespace MobileGL::MG_Backend::DirectGLES { // GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT. 64 covers every type with room to // spare and keeps consecutive staged blocks off each other's cache lines. constexpr SizeT kUnpackRingAlignment = 64; + // glCopyBufferSubData carries no offset-alignment requirement at all; 64 + // keeps staged blocks cache-line separated, same as the unpack ring. + constexpr SizeT kUploadRingInitialBytes = 4u * 1024u * 1024u; + constexpr SizeT kUploadRingMaxBytes = 64u * 1024u * 1024u; + constexpr SizeT kUploadRingAlignment = 64; struct PersistentRingStore { Uint id = 0; @@ -705,6 +710,21 @@ namespace MobileGL::MG_Backend::DirectGLES { kUnpackRingMaxBytes, kUnpackRingAlignment, "Texture unpack ring"}; + // Staging ring for app buffer updates whose destination store may still be + // referenced by in-flight GPU work. Mali's glBufferSubData resolves that WAR + // hazard by BLOCKING in the call (osup_sync_object_wait) until every + // referencing job retires - under Minecraft 26.3's per-frame UBO and + // chunk-mesh SubData streams that serialized whole frames (~1 fps while + // chunks stream in). Staging the bytes here and issuing a + // glCopyBufferSubData instead keeps the hazard on the GPU timeline where it + // is just job ordering, and the CPU never waits. + PersistentRing g_uploadRing{{}, + {}, + {}, + kUploadRingInitialBytes, + kUploadRingMaxBytes, + kUploadRingAlignment, + "Buffer upload ring"}; // The ES context the ring's id/map belonged to is gone (or was never // seen): drop every handle without GL calls and re-arm creation. The @@ -794,6 +814,67 @@ namespace MobileGL::MG_Backend::DirectGLES { bufferObject.MappedData() + start); } + // Ring machinery shared with the UBO/unpack rings; defined further down in + // this same unnamed namespace. + Bool RingAllocate(PersistentRing& ring, SizeT size, SizeT& outOffset); + Bool RingAvailable(PersistentRing& ring); + + // True when a pending-range flush can go through the staging ring right + // now: kill switch off, the ES copy entry point resolved, and the ring's + // own availability gate (EXT_buffer_storage + fences + live context) up. + Bool UploadRingUsableNow() { + if (MG_Config::Features.DisableUploadRing) return false; + if (!g_GLESFuncs.glCopyBufferSubData) return false; + return RingAvailable(g_uploadRing); + } + + // Push every queued range of `resource` from the shadow into the backend + // store. Ranges go through the staging ring + glCopyBufferSubData when the + // ring is usable - the copy resolves the WAR hazard against in-flight + // frames on the GPU timeline, where it is mere job ordering - and fall + // back to the direct (potentially stalling) glBufferSubData otherwise. + // The caller owns syncedChangeSerial; this only drains the queue. + void FlushPendingRangesNow(GLESBufferResource& resource, BufferObject& bufferObject) { +#ifdef TRACY_ENABLE + ZoneScopedC(TRACY_ZONECOLOR_BACKEND); +#endif + VecRange1D ranges; + { + const std::lock_guard lock(resource.pendingMutex); + if (resource.pendingRanges.empty()) return; + ranges = std::move(resource.pendingRanges); + resource.pendingRanges.clear(); + } + // Clamp against BOTH extents: the readback flush may run while the + // frontend size and the backend store disagree (a pending respecify + // resolves that later; bytes past either end have nowhere to land). + // + // The ranges are copied AS QUEUED (VecRange1D::Add already merges + // near-adjacent ones): this driver runs buffer copies as worker-thread + // memcpys, so BYTES are the cost axis - collapsing a scattered flush + // into its union re-copied nearly whole chunk-mesh arenas every frame + // and saturated the copy worker during camera pans. + const SizeT limit = std::min(bufferObject.GetSize(), resource.storageSize); + const Bool ringUsable = UploadRingUsableNow(); + for (const auto& range : ranges) { + const SizeT end = std::min(range.end, limit); + const SizeT start = std::min(range.start, end); + const SizeT size = end - start; + if (size == 0) continue; + SizeT ringOffset = 0; + if (ringUsable && size <= kUploadRingMaxBytes && + RingAllocate(g_uploadRing, size, ringOffset)) { + Memcpy(g_uploadRing.store.mappedPtr + ringOffset, bufferObject.MappedData() + start, size); + BindBufferId(GL_COPY_READ_BUFFER, g_uploadRing.store.id); + BindBufferId(GL_COPY_WRITE_BUFFER, resource.id); + g_GLESFuncs.glCopyBufferSubData(GL_COPY_READ_BUFFER, GL_COPY_WRITE_BUFFER, + (GLintptr)ringOffset, (GLintptr)start, (GLsizeiptr)size); + } else { + UploadRangeNow(resource, bufferObject, start, end); + } + } + } + // EXT_buffer_storage bit values (same numeric values as the desktop ARB // tokens); defined locally so this compiles regardless of which GLES headers // expose the EXT tokens. @@ -941,11 +1022,27 @@ namespace MobileGL::MG_Backend::DirectGLES { if (!CanTouchGLNow() || resource->id == 0 || resource->contextGeneration != g_bufferContextGeneration || !StorageMatches(*resource, bufferObject)) { + const std::lock_guard lock(resource->pendingMutex); resource->pendingRanges.Add({offset, offset + size}); return; } - UploadRangeNow(*resource, bufferObject, offset, offset + size); - resource->syncedChangeSerial = bufferObject.GetChangeSerial(); + // An immediate glBufferSubData resolves the WAR hazard against frames + // still referencing this store on the CPU on some drivers - Mali parks + // the thread in osup_sync_object_wait until every referencing job + // retires, which serialized Minecraft 26.3's per-frame UBO/chunk-mesh + // update streams into ~1 fps. Queue the range instead (the shadow + // already holds the bytes) and let draw-time sync push the merged + // ranges through the staging ring. The zero-copy persistent store + // keeps the legacy immediate upload: draw-time sync never flushes + // ranges for it, and its mapping publishes writes by itself. + if ((resource->persistentMapped && resource->persistentPtr) || + MG_Config::Features.DisableUploadRing) { + UploadRangeNow(*resource, bufferObject, offset, offset + size); + resource->syncedChangeSerial = bufferObject.GetChangeSerial(); + return; + } + const std::lock_guard lock(resource->pendingMutex); + resource->pendingRanges.Add({offset, offset + size}); } void Ops_FlushMappedRange(BufferObject& bufferObject, Range1D range, @@ -956,6 +1053,19 @@ namespace MobileGL::MG_Backend::DirectGLES { if (!CanTouchGLNow() || resource->id == 0 || resource->contextGeneration != g_bufferContextGeneration || !StorageMatches(*resource, bufferObject)) { + const std::lock_guard lock(resource->pendingMutex); + resource->pendingRanges.Add(range); + return; + } + + // Same WAR-hazard rule as Ops_SubData: an immediate upload (mapped or + // glBufferSubData) can park the thread on Mali until the frames still + // referencing this store retire. Queue the range for the staged-copy + // flush at draw-time sync; only the zero-copy persistent store and the + // negative-control kill switch keep the immediate paths below. + if (!(resource->persistentMapped && resource->persistentPtr) && + !MG_Config::Features.DisableUploadRing) { + const std::lock_guard lock(resource->pendingMutex); resource->pendingRanges.Add(range); return; } @@ -1007,6 +1117,10 @@ namespace MobileGL::MG_Backend::DirectGLES { const SizeT size = std::min(bufferObject.GetSize(), resource->storageSize); if (size == 0) return; + // Queued app writes must land in the backend store before it is read + // back, or the writeback below would revert them in the shadow. + FlushPendingRangesNow(*resource, bufferObject); + BindBufferId(TempBufferTarget, resource->id); void* mapped = g_GLESFuncs.glMapBufferRange(TempBufferTarget, 0, static_cast(size), GL_MAP_READ_BIT); @@ -1295,11 +1409,7 @@ namespace MobileGL::MG_Backend::DirectGLES { resource->storageSize != bufferObject->GetSize()) { RespecifyStorageNow(*resource, *bufferObject); } else if (!resource->pendingRanges.empty()) { - for (const auto& range : resource->pendingRanges) { - const SizeT end = std::min(range.end, bufferObject->GetSize()); - UploadRangeNow(*resource, *bufferObject, std::min(range.start, end), end); - } - resource->pendingRanges.clear(); + FlushPendingRangesNow(*resource, *bufferObject); resource->syncedChangeSerial = bufferObject->GetChangeSerial(); } else if (resource->syncedChangeSerial != bufferObject->GetChangeSerial()) { // Ops could not track some writes (e.g. the ops table was @@ -1621,6 +1731,8 @@ namespace MobileGL::MG_Backend::DirectGLES { "ring offset mask below requires power-of-two ring sizes"); static_assert((kUnpackRingInitialBytes & (kUnpackRingInitialBytes - 1)) == 0, "ring offset mask below requires power-of-two ring sizes"); + static_assert((kUploadRingInitialBytes & (kUploadRingInitialBytes - 1)) == 0, + "ring offset mask below requires power-of-two ring sizes"); const SizeT offset = static_cast(store.head & (store.size - 1)); if (offset + alignedSize <= store.size && store.head + alignedSize - store.tail <= store.size) { store.head += alignedSize; @@ -1789,6 +1901,8 @@ namespace MobileGL::MG_Backend::DirectGLES { SizeT UnpackRingMaxBytes() { return kUnpackRingMaxBytes; } void UnpackRingOnPresent() { RingOnPresent(g_unpackRing); } + + void UploadRingOnPresent() { RingOnPresent(g_uploadRing); } } // namespace BufferImpl namespace VertexArrayImpl { diff --git a/MobileGL/MG_Backend/DirectGLES/Managers.h b/MobileGL/MG_Backend/DirectGLES/Managers.h index 8e50d929..d0c46250 100644 --- a/MobileGL/MG_Backend/DirectGLES/Managers.h +++ b/MobileGL/MG_Backend/DirectGLES/Managers.h @@ -642,6 +642,23 @@ namespace MobileGL::MG_Backend::DirectGLES { // Largest single staging request the ring can ever satisfy. SizeT UnpackRingMaxBytes(); void UnpackRingOnPresent(); + + // --- Buffer upload ring --------------------------------------------------- + // The same persistent-mapped bump allocator, staging APP BUFFER UPDATES + // (glBufferSubData / non-persistent map flushes) whose destination store may + // still be referenced by in-flight GPU work. Mali resolves that WAR hazard by + // BLOCKING the calling glBufferSubData (osup_sync_object_wait) until every + // referencing job retires - Minecraft 26.3 rewrites its chunk-section and + // dynamic-transform UBOs and streams chunk meshes with per-frame SubData, and + // each such call serialized against the whole GPU queue (~1 fps while chunks + // stream in, and again on every camera pan). App SubData ranges are queued on + // the resource instead (the frontend shadow already holds the bytes) and + // draw-time sync drains them: bytes staged into this ring, then one + // glCopyBufferSubData per merged range - the copy is ordered on the GPU + // timeline, so the hazard costs no CPU wait. Reclamation contract identical + // to the other two rings. MOBILEGL_DISABLE_UPLOAD_RING restores the + // historical immediate-upload path (negative control / escape hatch). + void UploadRingOnPresent(); } // namespace BufferImpl namespace VertexArrayImpl {