[Fix] (DirectGLES): queue app buffer updates and flush them through a staged-copy upload ring instead of stalling in glBufferSubData

This commit is contained in:
2026-08-27 22:19:44 -04:00
parent 200c21336f
commit a28da07641
5 changed files with 146 additions and 7 deletions
+6
View File
@@ -170,6 +170,12 @@ namespace MobileGL::MG_Config {
// persistent-mapped unpack-PBO ring (negative control / driver-bug escape
// hatch).
Bool DisableUnpackRing = false;
// MOBILEGL_DISABLE_UPLOAD_RING: force DirectGLES app buffer updates
// (glBufferSubData / map flushes) back to the immediate driver upload instead
// of queueing them for the staged-copy flush through the persistent-mapped
// upload ring (negative control / driver-bug escape hatch; the immediate
// upload stalls on drivers that resolve the WAR hazard on the CPU, e.g. Mali).
Bool DisableUploadRing = false;
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
+1
View File
@@ -185,6 +185,7 @@ namespace MobileGL::MG_ConfigLoader {
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
features.DisableUnpackRing = QueryEnvFlag("MOBILEGL_DISABLE_UNPACK_RING");
features.DisableUploadRing = QueryEnvFlag("MOBILEGL_DISABLE_UPLOAD_RING");
features.EsprytForceDepthStencilReadbackEmulation =
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
@@ -10636,6 +10636,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
// frame's ring high-water marks for slot reclamation.
BufferImpl::UboRingOnPresent();
BufferImpl::UnpackRingOnPresent();
BufferImpl::UploadRingOnPresent();
BufferImpl::TrimBufferPool();
}
+119 -5
View File
@@ -648,6 +648,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
// GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT. 64 covers every type with room to
// spare and keeps consecutive staged blocks off each other's cache lines.
constexpr SizeT kUnpackRingAlignment = 64;
// glCopyBufferSubData carries no offset-alignment requirement at all; 64
// keeps staged blocks cache-line separated, same as the unpack ring.
constexpr SizeT kUploadRingInitialBytes = 4u * 1024u * 1024u;
constexpr SizeT kUploadRingMaxBytes = 64u * 1024u * 1024u;
constexpr SizeT kUploadRingAlignment = 64;
struct PersistentRingStore {
Uint id = 0;
@@ -705,6 +710,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
kUnpackRingMaxBytes,
kUnpackRingAlignment,
"Texture unpack ring"};
// Staging ring for app buffer updates whose destination store may still be
// referenced by in-flight GPU work. Mali's glBufferSubData resolves that WAR
// hazard by BLOCKING in the call (osup_sync_object_wait) until every
// referencing job retires - under Minecraft 26.3's per-frame UBO and
// chunk-mesh SubData streams that serialized whole frames (~1 fps while
// chunks stream in). Staging the bytes here and issuing a
// glCopyBufferSubData instead keeps the hazard on the GPU timeline where it
// is just job ordering, and the CPU never waits.
PersistentRing g_uploadRing{{},
{},
{},
kUploadRingInitialBytes,
kUploadRingMaxBytes,
kUploadRingAlignment,
"Buffer upload ring"};
// The ES context the ring's id/map belonged to is gone (or was never
// seen): drop every handle without GL calls and re-arm creation. The
@@ -794,6 +814,67 @@ namespace MobileGL::MG_Backend::DirectGLES {
bufferObject.MappedData() + start);
}
// Ring machinery shared with the UBO/unpack rings; defined further down in
// this same unnamed namespace.
Bool RingAllocate(PersistentRing& ring, SizeT size, SizeT& outOffset);
Bool RingAvailable(PersistentRing& ring);
// True when a pending-range flush can go through the staging ring right
// now: kill switch off, the ES copy entry point resolved, and the ring's
// own availability gate (EXT_buffer_storage + fences + live context) up.
Bool UploadRingUsableNow() {
if (MG_Config::Features.DisableUploadRing) return false;
if (!g_GLESFuncs.glCopyBufferSubData) return false;
return RingAvailable(g_uploadRing);
}
// Push every queued range of `resource` from the shadow into the backend
// store. Ranges go through the staging ring + glCopyBufferSubData when the
// ring is usable - the copy resolves the WAR hazard against in-flight
// frames on the GPU timeline, where it is mere job ordering - and fall
// back to the direct (potentially stalling) glBufferSubData otherwise.
// The caller owns syncedChangeSerial; this only drains the queue.
void FlushPendingRangesNow(GLESBufferResource& resource, BufferObject& bufferObject) {
#ifdef TRACY_ENABLE
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
#endif
VecRange1D ranges;
{
const std::lock_guard<std::mutex> lock(resource.pendingMutex);
if (resource.pendingRanges.empty()) return;
ranges = std::move(resource.pendingRanges);
resource.pendingRanges.clear();
}
// Clamp against BOTH extents: the readback flush may run while the
// frontend size and the backend store disagree (a pending respecify
// resolves that later; bytes past either end have nowhere to land).
//
// The ranges are copied AS QUEUED (VecRange1D::Add already merges
// near-adjacent ones): this driver runs buffer copies as worker-thread
// memcpys, so BYTES are the cost axis - collapsing a scattered flush
// into its union re-copied nearly whole chunk-mesh arenas every frame
// and saturated the copy worker during camera pans.
const SizeT limit = std::min(bufferObject.GetSize(), resource.storageSize);
const Bool ringUsable = UploadRingUsableNow();
for (const auto& range : ranges) {
const SizeT end = std::min(range.end, limit);
const SizeT start = std::min(range.start, end);
const SizeT size = end - start;
if (size == 0) continue;
SizeT ringOffset = 0;
if (ringUsable && size <= kUploadRingMaxBytes &&
RingAllocate(g_uploadRing, size, ringOffset)) {
Memcpy(g_uploadRing.store.mappedPtr + ringOffset, bufferObject.MappedData() + start, size);
BindBufferId(GL_COPY_READ_BUFFER, g_uploadRing.store.id);
BindBufferId(GL_COPY_WRITE_BUFFER, resource.id);
g_GLESFuncs.glCopyBufferSubData(GL_COPY_READ_BUFFER, GL_COPY_WRITE_BUFFER,
(GLintptr)ringOffset, (GLintptr)start, (GLsizeiptr)size);
} else {
UploadRangeNow(resource, bufferObject, start, end);
}
}
}
// EXT_buffer_storage bit values (same numeric values as the desktop ARB
// tokens); defined locally so this compiles regardless of which GLES headers
// expose the EXT tokens.
@@ -941,11 +1022,27 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (!CanTouchGLNow() || resource->id == 0 ||
resource->contextGeneration != g_bufferContextGeneration ||
!StorageMatches(*resource, bufferObject)) {
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
resource->pendingRanges.Add({offset, offset + size});
return;
}
// An immediate glBufferSubData resolves the WAR hazard against frames
// still referencing this store on the CPU on some drivers - Mali parks
// the thread in osup_sync_object_wait until every referencing job
// retires, which serialized Minecraft 26.3's per-frame UBO/chunk-mesh
// update streams into ~1 fps. Queue the range instead (the shadow
// already holds the bytes) and let draw-time sync push the merged
// ranges through the staging ring. The zero-copy persistent store
// keeps the legacy immediate upload: draw-time sync never flushes
// ranges for it, and its mapping publishes writes by itself.
if ((resource->persistentMapped && resource->persistentPtr) ||
MG_Config::Features.DisableUploadRing) {
UploadRangeNow(*resource, bufferObject, offset, offset + size);
resource->syncedChangeSerial = bufferObject.GetChangeSerial();
return;
}
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
resource->pendingRanges.Add({offset, offset + size});
}
void Ops_FlushMappedRange(BufferObject& bufferObject, Range1D range,
@@ -956,6 +1053,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (!CanTouchGLNow() || resource->id == 0 ||
resource->contextGeneration != g_bufferContextGeneration ||
!StorageMatches(*resource, bufferObject)) {
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
resource->pendingRanges.Add(range);
return;
}
// Same WAR-hazard rule as Ops_SubData: an immediate upload (mapped or
// glBufferSubData) can park the thread on Mali until the frames still
// referencing this store retire. Queue the range for the staged-copy
// flush at draw-time sync; only the zero-copy persistent store and the
// negative-control kill switch keep the immediate paths below.
if (!(resource->persistentMapped && resource->persistentPtr) &&
!MG_Config::Features.DisableUploadRing) {
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
resource->pendingRanges.Add(range);
return;
}
@@ -1007,6 +1117,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
const SizeT size = std::min<SizeT>(bufferObject.GetSize(), resource->storageSize);
if (size == 0) return;
// Queued app writes must land in the backend store before it is read
// back, or the writeback below would revert them in the shadow.
FlushPendingRangesNow(*resource, bufferObject);
BindBufferId(TempBufferTarget, resource->id);
void* mapped = g_GLESFuncs.glMapBufferRange(TempBufferTarget, 0, static_cast<GLsizeiptr>(size),
GL_MAP_READ_BIT);
@@ -1295,11 +1409,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
resource->storageSize != bufferObject->GetSize()) {
RespecifyStorageNow(*resource, *bufferObject);
} else if (!resource->pendingRanges.empty()) {
for (const auto& range : resource->pendingRanges) {
const SizeT end = std::min(range.end, bufferObject->GetSize());
UploadRangeNow(*resource, *bufferObject, std::min(range.start, end), end);
}
resource->pendingRanges.clear();
FlushPendingRangesNow(*resource, *bufferObject);
resource->syncedChangeSerial = bufferObject->GetChangeSerial();
} else if (resource->syncedChangeSerial != bufferObject->GetChangeSerial()) {
// Ops could not track some writes (e.g. the ops table was
@@ -1621,6 +1731,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
"ring offset mask below requires power-of-two ring sizes");
static_assert((kUnpackRingInitialBytes & (kUnpackRingInitialBytes - 1)) == 0,
"ring offset mask below requires power-of-two ring sizes");
static_assert((kUploadRingInitialBytes & (kUploadRingInitialBytes - 1)) == 0,
"ring offset mask below requires power-of-two ring sizes");
const SizeT offset = static_cast<SizeT>(store.head & (store.size - 1));
if (offset + alignedSize <= store.size && store.head + alignedSize - store.tail <= store.size) {
store.head += alignedSize;
@@ -1789,6 +1901,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
SizeT UnpackRingMaxBytes() { return kUnpackRingMaxBytes; }
void UnpackRingOnPresent() { RingOnPresent(g_unpackRing); }
void UploadRingOnPresent() { RingOnPresent(g_uploadRing); }
} // namespace BufferImpl
namespace VertexArrayImpl {
+17
View File
@@ -642,6 +642,23 @@ namespace MobileGL::MG_Backend::DirectGLES {
// Largest single staging request the ring can ever satisfy.
SizeT UnpackRingMaxBytes();
void UnpackRingOnPresent();
// --- Buffer upload ring ---------------------------------------------------
// The same persistent-mapped bump allocator, staging APP BUFFER UPDATES
// (glBufferSubData / non-persistent map flushes) whose destination store may
// still be referenced by in-flight GPU work. Mali resolves that WAR hazard by
// BLOCKING the calling glBufferSubData (osup_sync_object_wait) until every
// referencing job retires - Minecraft 26.3 rewrites its chunk-section and
// dynamic-transform UBOs and streams chunk meshes with per-frame SubData, and
// each such call serialized against the whole GPU queue (~1 fps while chunks
// stream in, and again on every camera pan). App SubData ranges are queued on
// the resource instead (the frontend shadow already holds the bytes) and
// draw-time sync drains them: bytes staged into this ring, then one
// glCopyBufferSubData per merged range - the copy is ordered on the GPU
// timeline, so the hazard costs no CPU wait. Reclamation contract identical
// to the other two rings. MOBILEGL_DISABLE_UPLOAD_RING restores the
// historical immediate-upload path (negative control / escape hatch).
void UploadRingOnPresent();
} // namespace BufferImpl
namespace VertexArrayImpl {