mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-07 19:58:32 +09:00
[Fix] (DirectGLES): queue app buffer updates and flush them through a staged-copy upload ring instead of stalling in glBufferSubData
This commit is contained in:
@@ -170,6 +170,12 @@ namespace MobileGL::MG_Config {
|
||||
// persistent-mapped unpack-PBO ring (negative control / driver-bug escape
|
||||
// hatch).
|
||||
Bool DisableUnpackRing = false;
|
||||
// MOBILEGL_DISABLE_UPLOAD_RING: force DirectGLES app buffer updates
|
||||
// (glBufferSubData / map flushes) back to the immediate driver upload instead
|
||||
// of queueing them for the staged-copy flush through the persistent-mapped
|
||||
// upload ring (negative control / driver-bug escape hatch; the immediate
|
||||
// upload stalls on drivers that resolve the WAR hazard on the CPU, e.g. Mali).
|
||||
Bool DisableUploadRing = false;
|
||||
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
|
||||
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
|
||||
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
|
||||
|
||||
@@ -185,6 +185,7 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
||||
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
||||
features.DisableUnpackRing = QueryEnvFlag("MOBILEGL_DISABLE_UNPACK_RING");
|
||||
features.DisableUploadRing = QueryEnvFlag("MOBILEGL_DISABLE_UPLOAD_RING");
|
||||
features.EsprytForceDepthStencilReadbackEmulation =
|
||||
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||
|
||||
@@ -10636,6 +10636,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// frame's ring high-water marks for slot reclamation.
|
||||
BufferImpl::UboRingOnPresent();
|
||||
BufferImpl::UnpackRingOnPresent();
|
||||
BufferImpl::UploadRingOnPresent();
|
||||
BufferImpl::TrimBufferPool();
|
||||
}
|
||||
|
||||
|
||||
@@ -648,6 +648,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT. 64 covers every type with room to
|
||||
// spare and keeps consecutive staged blocks off each other's cache lines.
|
||||
constexpr SizeT kUnpackRingAlignment = 64;
|
||||
// glCopyBufferSubData carries no offset-alignment requirement at all; 64
|
||||
// keeps staged blocks cache-line separated, same as the unpack ring.
|
||||
constexpr SizeT kUploadRingInitialBytes = 4u * 1024u * 1024u;
|
||||
constexpr SizeT kUploadRingMaxBytes = 64u * 1024u * 1024u;
|
||||
constexpr SizeT kUploadRingAlignment = 64;
|
||||
|
||||
struct PersistentRingStore {
|
||||
Uint id = 0;
|
||||
@@ -705,6 +710,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
kUnpackRingMaxBytes,
|
||||
kUnpackRingAlignment,
|
||||
"Texture unpack ring"};
|
||||
// Staging ring for app buffer updates whose destination store may still be
|
||||
// referenced by in-flight GPU work. Mali's glBufferSubData resolves that WAR
|
||||
// hazard by BLOCKING in the call (osup_sync_object_wait) until every
|
||||
// referencing job retires - under Minecraft 26.3's per-frame UBO and
|
||||
// chunk-mesh SubData streams that serialized whole frames (~1 fps while
|
||||
// chunks stream in). Staging the bytes here and issuing a
|
||||
// glCopyBufferSubData instead keeps the hazard on the GPU timeline where it
|
||||
// is just job ordering, and the CPU never waits.
|
||||
PersistentRing g_uploadRing{{},
|
||||
{},
|
||||
{},
|
||||
kUploadRingInitialBytes,
|
||||
kUploadRingMaxBytes,
|
||||
kUploadRingAlignment,
|
||||
"Buffer upload ring"};
|
||||
|
||||
// The ES context the ring's id/map belonged to is gone (or was never
|
||||
// seen): drop every handle without GL calls and re-arm creation. The
|
||||
@@ -794,6 +814,67 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
bufferObject.MappedData() + start);
|
||||
}
|
||||
|
||||
// Ring machinery shared with the UBO/unpack rings; defined further down in
|
||||
// this same unnamed namespace.
|
||||
Bool RingAllocate(PersistentRing& ring, SizeT size, SizeT& outOffset);
|
||||
Bool RingAvailable(PersistentRing& ring);
|
||||
|
||||
// True when a pending-range flush can go through the staging ring right
|
||||
// now: kill switch off, the ES copy entry point resolved, and the ring's
|
||||
// own availability gate (EXT_buffer_storage + fences + live context) up.
|
||||
Bool UploadRingUsableNow() {
|
||||
if (MG_Config::Features.DisableUploadRing) return false;
|
||||
if (!g_GLESFuncs.glCopyBufferSubData) return false;
|
||||
return RingAvailable(g_uploadRing);
|
||||
}
|
||||
|
||||
// Push every queued range of `resource` from the shadow into the backend
|
||||
// store. Ranges go through the staging ring + glCopyBufferSubData when the
|
||||
// ring is usable - the copy resolves the WAR hazard against in-flight
|
||||
// frames on the GPU timeline, where it is mere job ordering - and fall
|
||||
// back to the direct (potentially stalling) glBufferSubData otherwise.
|
||||
// The caller owns syncedChangeSerial; this only drains the queue.
|
||||
void FlushPendingRangesNow(GLESBufferResource& resource, BufferObject& bufferObject) {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
VecRange1D ranges;
|
||||
{
|
||||
const std::lock_guard<std::mutex> lock(resource.pendingMutex);
|
||||
if (resource.pendingRanges.empty()) return;
|
||||
ranges = std::move(resource.pendingRanges);
|
||||
resource.pendingRanges.clear();
|
||||
}
|
||||
// Clamp against BOTH extents: the readback flush may run while the
|
||||
// frontend size and the backend store disagree (a pending respecify
|
||||
// resolves that later; bytes past either end have nowhere to land).
|
||||
//
|
||||
// The ranges are copied AS QUEUED (VecRange1D::Add already merges
|
||||
// near-adjacent ones): this driver runs buffer copies as worker-thread
|
||||
// memcpys, so BYTES are the cost axis - collapsing a scattered flush
|
||||
// into its union re-copied nearly whole chunk-mesh arenas every frame
|
||||
// and saturated the copy worker during camera pans.
|
||||
const SizeT limit = std::min(bufferObject.GetSize(), resource.storageSize);
|
||||
const Bool ringUsable = UploadRingUsableNow();
|
||||
for (const auto& range : ranges) {
|
||||
const SizeT end = std::min(range.end, limit);
|
||||
const SizeT start = std::min(range.start, end);
|
||||
const SizeT size = end - start;
|
||||
if (size == 0) continue;
|
||||
SizeT ringOffset = 0;
|
||||
if (ringUsable && size <= kUploadRingMaxBytes &&
|
||||
RingAllocate(g_uploadRing, size, ringOffset)) {
|
||||
Memcpy(g_uploadRing.store.mappedPtr + ringOffset, bufferObject.MappedData() + start, size);
|
||||
BindBufferId(GL_COPY_READ_BUFFER, g_uploadRing.store.id);
|
||||
BindBufferId(GL_COPY_WRITE_BUFFER, resource.id);
|
||||
g_GLESFuncs.glCopyBufferSubData(GL_COPY_READ_BUFFER, GL_COPY_WRITE_BUFFER,
|
||||
(GLintptr)ringOffset, (GLintptr)start, (GLsizeiptr)size);
|
||||
} else {
|
||||
UploadRangeNow(resource, bufferObject, start, end);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// EXT_buffer_storage bit values (same numeric values as the desktop ARB
|
||||
// tokens); defined locally so this compiles regardless of which GLES headers
|
||||
// expose the EXT tokens.
|
||||
@@ -941,11 +1022,27 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (!CanTouchGLNow() || resource->id == 0 ||
|
||||
resource->contextGeneration != g_bufferContextGeneration ||
|
||||
!StorageMatches(*resource, bufferObject)) {
|
||||
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
|
||||
resource->pendingRanges.Add({offset, offset + size});
|
||||
return;
|
||||
}
|
||||
// An immediate glBufferSubData resolves the WAR hazard against frames
|
||||
// still referencing this store on the CPU on some drivers - Mali parks
|
||||
// the thread in osup_sync_object_wait until every referencing job
|
||||
// retires, which serialized Minecraft 26.3's per-frame UBO/chunk-mesh
|
||||
// update streams into ~1 fps. Queue the range instead (the shadow
|
||||
// already holds the bytes) and let draw-time sync push the merged
|
||||
// ranges through the staging ring. The zero-copy persistent store
|
||||
// keeps the legacy immediate upload: draw-time sync never flushes
|
||||
// ranges for it, and its mapping publishes writes by itself.
|
||||
if ((resource->persistentMapped && resource->persistentPtr) ||
|
||||
MG_Config::Features.DisableUploadRing) {
|
||||
UploadRangeNow(*resource, bufferObject, offset, offset + size);
|
||||
resource->syncedChangeSerial = bufferObject.GetChangeSerial();
|
||||
return;
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
|
||||
resource->pendingRanges.Add({offset, offset + size});
|
||||
}
|
||||
|
||||
void Ops_FlushMappedRange(BufferObject& bufferObject, Range1D range,
|
||||
@@ -956,6 +1053,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (!CanTouchGLNow() || resource->id == 0 ||
|
||||
resource->contextGeneration != g_bufferContextGeneration ||
|
||||
!StorageMatches(*resource, bufferObject)) {
|
||||
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
|
||||
resource->pendingRanges.Add(range);
|
||||
return;
|
||||
}
|
||||
|
||||
// Same WAR-hazard rule as Ops_SubData: an immediate upload (mapped or
|
||||
// glBufferSubData) can park the thread on Mali until the frames still
|
||||
// referencing this store retire. Queue the range for the staged-copy
|
||||
// flush at draw-time sync; only the zero-copy persistent store and the
|
||||
// negative-control kill switch keep the immediate paths below.
|
||||
if (!(resource->persistentMapped && resource->persistentPtr) &&
|
||||
!MG_Config::Features.DisableUploadRing) {
|
||||
const std::lock_guard<std::mutex> lock(resource->pendingMutex);
|
||||
resource->pendingRanges.Add(range);
|
||||
return;
|
||||
}
|
||||
@@ -1007,6 +1117,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const SizeT size = std::min<SizeT>(bufferObject.GetSize(), resource->storageSize);
|
||||
if (size == 0) return;
|
||||
|
||||
// Queued app writes must land in the backend store before it is read
|
||||
// back, or the writeback below would revert them in the shadow.
|
||||
FlushPendingRangesNow(*resource, bufferObject);
|
||||
|
||||
BindBufferId(TempBufferTarget, resource->id);
|
||||
void* mapped = g_GLESFuncs.glMapBufferRange(TempBufferTarget, 0, static_cast<GLsizeiptr>(size),
|
||||
GL_MAP_READ_BIT);
|
||||
@@ -1295,11 +1409,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
resource->storageSize != bufferObject->GetSize()) {
|
||||
RespecifyStorageNow(*resource, *bufferObject);
|
||||
} else if (!resource->pendingRanges.empty()) {
|
||||
for (const auto& range : resource->pendingRanges) {
|
||||
const SizeT end = std::min(range.end, bufferObject->GetSize());
|
||||
UploadRangeNow(*resource, *bufferObject, std::min(range.start, end), end);
|
||||
}
|
||||
resource->pendingRanges.clear();
|
||||
FlushPendingRangesNow(*resource, *bufferObject);
|
||||
resource->syncedChangeSerial = bufferObject->GetChangeSerial();
|
||||
} else if (resource->syncedChangeSerial != bufferObject->GetChangeSerial()) {
|
||||
// Ops could not track some writes (e.g. the ops table was
|
||||
@@ -1621,6 +1731,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
"ring offset mask below requires power-of-two ring sizes");
|
||||
static_assert((kUnpackRingInitialBytes & (kUnpackRingInitialBytes - 1)) == 0,
|
||||
"ring offset mask below requires power-of-two ring sizes");
|
||||
static_assert((kUploadRingInitialBytes & (kUploadRingInitialBytes - 1)) == 0,
|
||||
"ring offset mask below requires power-of-two ring sizes");
|
||||
const SizeT offset = static_cast<SizeT>(store.head & (store.size - 1));
|
||||
if (offset + alignedSize <= store.size && store.head + alignedSize - store.tail <= store.size) {
|
||||
store.head += alignedSize;
|
||||
@@ -1789,6 +1901,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
SizeT UnpackRingMaxBytes() { return kUnpackRingMaxBytes; }
|
||||
|
||||
void UnpackRingOnPresent() { RingOnPresent(g_unpackRing); }
|
||||
|
||||
void UploadRingOnPresent() { RingOnPresent(g_uploadRing); }
|
||||
} // namespace BufferImpl
|
||||
|
||||
namespace VertexArrayImpl {
|
||||
|
||||
@@ -642,6 +642,23 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Largest single staging request the ring can ever satisfy.
|
||||
SizeT UnpackRingMaxBytes();
|
||||
void UnpackRingOnPresent();
|
||||
|
||||
// --- Buffer upload ring ---------------------------------------------------
|
||||
// The same persistent-mapped bump allocator, staging APP BUFFER UPDATES
|
||||
// (glBufferSubData / non-persistent map flushes) whose destination store may
|
||||
// still be referenced by in-flight GPU work. Mali resolves that WAR hazard by
|
||||
// BLOCKING the calling glBufferSubData (osup_sync_object_wait) until every
|
||||
// referencing job retires - Minecraft 26.3 rewrites its chunk-section and
|
||||
// dynamic-transform UBOs and streams chunk meshes with per-frame SubData, and
|
||||
// each such call serialized against the whole GPU queue (~1 fps while chunks
|
||||
// stream in, and again on every camera pan). App SubData ranges are queued on
|
||||
// the resource instead (the frontend shadow already holds the bytes) and
|
||||
// draw-time sync drains them: bytes staged into this ring, then one
|
||||
// glCopyBufferSubData per merged range - the copy is ordered on the GPU
|
||||
// timeline, so the hazard costs no CPU wait. Reclamation contract identical
|
||||
// to the other two rings. MOBILEGL_DISABLE_UPLOAD_RING restores the
|
||||
// historical immediate-upload path (negative control / escape hatch).
|
||||
void UploadRingOnPresent();
|
||||
} // namespace BufferImpl
|
||||
|
||||
namespace VertexArrayImpl {
|
||||
|
||||
Reference in New Issue
Block a user