mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-07 19:58:32 +09:00
Compare commits
8
Commits
12df061e0b
...
0d2fceab0e
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0d2fceab0e | ||
|
|
82decded58 | ||
|
|
d04a3394de | ||
|
|
33eabfc2ff | ||
|
|
56e5d13dc9 | ||
|
|
34b7cc772f | ||
|
|
c1d6a3c908 | ||
|
|
386bd7e461 |
@@ -155,6 +155,11 @@ namespace MobileGL::MG_Config {
|
|||||||
// per-draw glBufferSubData path instead of the persistent-mapped ring allocator
|
// per-draw glBufferSubData path instead of the persistent-mapped ring allocator
|
||||||
// (negative control / driver-bug escape hatch).
|
// (negative control / driver-bug escape hatch).
|
||||||
Bool DisableUboRing = false;
|
Bool DisableUboRing = false;
|
||||||
|
// MOBILEGL_DISABLE_UNPACK_RING: force DirectGLES texture uploads back to
|
||||||
|
// glTexSubImage from the client pointer instead of staging them through the
|
||||||
|
// persistent-mapped unpack-PBO ring (negative control / driver-bug escape
|
||||||
|
// hatch).
|
||||||
|
Bool DisableUnpackRing = false;
|
||||||
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
|
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
|
||||||
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
|
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
|
||||||
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
|
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
|
||||||
|
|||||||
@@ -183,6 +183,7 @@ namespace MobileGL::MG_ConfigLoader {
|
|||||||
features.CoherentAsFlush = QueryEnvFlag("MOBILEGL_COHERENT_AS_FLUSH");
|
features.CoherentAsFlush = QueryEnvFlag("MOBILEGL_COHERENT_AS_FLUSH");
|
||||||
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
||||||
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
||||||
|
features.DisableUnpackRing = QueryEnvFlag("MOBILEGL_DISABLE_UNPACK_RING");
|
||||||
features.EsprytForceDepthStencilReadbackEmulation =
|
features.EsprytForceDepthStencilReadbackEmulation =
|
||||||
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
||||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||||
|
|||||||
@@ -9985,9 +9985,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
g_completedFrameSerial.store(completed, std::memory_order_relaxed);
|
g_completedFrameSerial.store(completed, std::memory_order_relaxed);
|
||||||
}
|
}
|
||||||
|
|
||||||
// After the watermark advanced: retire grown-away UBO-ring stores and record
|
// After the watermark advanced: retire grown-away ring stores and record the
|
||||||
// the frame's ring high-water mark for slot reclamation.
|
// frame's ring high-water marks for slot reclamation.
|
||||||
BufferImpl::UboRingOnPresent();
|
BufferImpl::UboRingOnPresent();
|
||||||
|
BufferImpl::UnpackRingOnPresent();
|
||||||
BufferImpl::TrimBufferPool();
|
BufferImpl::TrimBufferPool();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -630,11 +630,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- Global-UBO ring (see Managers.h) ------------------------------------
|
// --- Persistent-mapped bump rings (see Managers.h) -----------------------
|
||||||
|
// Shared machinery behind BOTH the global-UBO ring and the texture
|
||||||
|
// unpack-PBO ring: one EXT_buffer_storage persistent|coherent map per ring,
|
||||||
|
// monotonic head/tail cursors, and reclamation riding the Present()
|
||||||
|
// frame-fence watermark. The two rings differ only in size cap, offset
|
||||||
|
// alignment and log label - the reclamation, context-loss and
|
||||||
|
// emergency-drain rules are the part that was hard to get right, so they
|
||||||
|
// are shared rather than copied.
|
||||||
constexpr SizeT kUboRingInitialBytes = 4u * 1024u * 1024u;
|
constexpr SizeT kUboRingInitialBytes = 4u * 1024u * 1024u;
|
||||||
constexpr SizeT kUboRingMaxBytes = 64u * 1024u * 1024u;
|
constexpr SizeT kUboRingMaxBytes = 64u * 1024u * 1024u;
|
||||||
|
constexpr SizeT kUnpackRingInitialBytes = 4u * 1024u * 1024u;
|
||||||
|
constexpr SizeT kUnpackRingMaxBytes = 64u * 1024u * 1024u;
|
||||||
|
// A PBO-sourced glTexSubImage constrains the offset only to the pixel
|
||||||
|
// TYPE's size (GL_INVALID_OPERATION otherwise), and no ES client type is
|
||||||
|
// wider than 4 bytes - unlike a UBO bind, which owes the driver
|
||||||
|
// GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT. 64 covers every type with room to
|
||||||
|
// spare and keeps consecutive staged blocks off each other's cache lines.
|
||||||
|
constexpr SizeT kUnpackRingAlignment = 64;
|
||||||
|
|
||||||
struct UboRingState {
|
struct PersistentRingStore {
|
||||||
Uint id = 0;
|
Uint id = 0;
|
||||||
Uint8* mappedPtr = nullptr;
|
Uint8* mappedPtr = nullptr;
|
||||||
SizeT size = 0;
|
SizeT size = 0;
|
||||||
@@ -648,41 +663,64 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
Uint contextGeneration = 0;
|
Uint contextGeneration = 0;
|
||||||
SizeT alignment = 256;
|
SizeT alignment = 256;
|
||||||
// A hard storage-creation failure under this context; stop retrying
|
// A hard storage-creation failure under this context; stop retrying
|
||||||
// per draw (cleared when the context generation moves on).
|
// per use (cleared when the context generation moves on).
|
||||||
Bool creationFailed = false;
|
Bool creationFailed = false;
|
||||||
};
|
};
|
||||||
UboRingState g_uboRing;
|
|
||||||
|
|
||||||
// Grown-away ring stores: deletable only once the GPU finished the last
|
// Grown-away ring stores: deletable only once the GPU finished the last
|
||||||
// frame that could reference them (same watermark as the buffer pool).
|
// frame that could reference them (same watermark as the buffer pool).
|
||||||
struct RetiredUboRing {
|
struct RetiredRingStore {
|
||||||
Uint id = 0;
|
Uint id = 0;
|
||||||
Uint contextGeneration = 0;
|
Uint contextGeneration = 0;
|
||||||
Uint64 retireSerial = 0;
|
Uint64 retireSerial = 0;
|
||||||
};
|
};
|
||||||
Vector<RetiredUboRing> g_retiredUboRings;
|
|
||||||
|
|
||||||
// Present()-time high-water marks: every byte below headAtPresent was
|
// Present()-time high-water marks: every byte below headAtPresent was
|
||||||
// written during frames <= frameSerial, so once frameSerial completes,
|
// written during frames <= frameSerial, so once frameSerial completes,
|
||||||
// tail may advance to headAtPresent. FIFO by construction.
|
// tail may advance to headAtPresent. FIFO by construction.
|
||||||
struct UboRingFrameMark {
|
struct RingFrameMark {
|
||||||
Uint64 frameSerial = 0;
|
Uint64 frameSerial = 0;
|
||||||
Uint64 headAtPresent = 0;
|
Uint64 headAtPresent = 0;
|
||||||
};
|
};
|
||||||
Vector<UboRingFrameMark> g_uboRingFrameMarks;
|
|
||||||
|
// One ring: its live store, its two reclamation lists, and the immutable
|
||||||
|
// knobs that tell it apart from the other one.
|
||||||
|
struct PersistentRing {
|
||||||
|
PersistentRingStore store;
|
||||||
|
Vector<RetiredRingStore> retired;
|
||||||
|
Vector<RingFrameMark> frameMarks;
|
||||||
|
SizeT initialBytes = 0;
|
||||||
|
SizeT maxBytes = 0;
|
||||||
|
// 0: take the offset alignment from GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT
|
||||||
|
// at store-creation time (the UBO ring's binds require it).
|
||||||
|
SizeT fixedAlignment = 0;
|
||||||
|
const char* label = "";
|
||||||
|
};
|
||||||
|
|
||||||
|
PersistentRing g_uboRing{{}, {}, {}, kUboRingInitialBytes, kUboRingMaxBytes, 0, "Global-UBO ring"};
|
||||||
|
PersistentRing g_unpackRing{{},
|
||||||
|
{},
|
||||||
|
{},
|
||||||
|
kUnpackRingInitialBytes,
|
||||||
|
kUnpackRingMaxBytes,
|
||||||
|
kUnpackRingAlignment,
|
||||||
|
"Texture unpack ring"};
|
||||||
|
|
||||||
// The ES context the ring's id/map belonged to is gone (or was never
|
// The ES context the ring's id/map belonged to is gone (or was never
|
||||||
// seen): drop every handle without GL calls and re-arm creation. The
|
// seen): drop every handle without GL calls and re-arm creation. The
|
||||||
// generation counter must survive the reset — frame serials also survive
|
// generation counter must survive the reset — frame serials also survive
|
||||||
// context recreation, so a restarted counter could revalidate a stale
|
// context recreation, so a restarted counter could revalidate a stale
|
||||||
// per-program slot cache against the new ring.
|
// per-program slot cache against the new ring.
|
||||||
void ResetUboRingForNewContext() {
|
void ResetRingForNewContext(PersistentRing& ring) {
|
||||||
const Uint32 keptGeneration = g_uboRing.generation;
|
const Uint32 keptGeneration = ring.store.generation;
|
||||||
g_uboRing = {};
|
ring.store = {};
|
||||||
g_uboRing.generation = keptGeneration;
|
ring.store.generation = keptGeneration;
|
||||||
g_uboRing.contextGeneration = g_bufferContextGeneration;
|
ring.store.contextGeneration = g_bufferContextGeneration;
|
||||||
g_retiredUboRings.clear();
|
// Keep the ring's own alignment across the wipe: the slow path rounds a
|
||||||
g_uboRingFrameMarks.clear();
|
// request with it BEFORE the store that would set it exists.
|
||||||
|
if (ring.fixedAlignment != 0) ring.store.alignment = ring.fixedAlignment;
|
||||||
|
ring.retired.clear();
|
||||||
|
ring.frameMarks.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
GLESBufferResource* ResourceOf(BufferObject& bufferObject) {
|
GLESBufferResource* ResourceOf(BufferObject& bufferObject) {
|
||||||
@@ -1097,9 +1135,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
InvalidateArrayBufferBindingCache();
|
InvalidateArrayBufferBindingCache();
|
||||||
InvalidateIndexedBufferBindingCache();
|
InvalidateIndexedBufferBindingCache();
|
||||||
InvalidatePixelBufferBindingCaches();
|
InvalidatePixelBufferBindingCaches();
|
||||||
// The global-UBO ring's id and persistent map died with the context;
|
// The rings' ids and persistent maps died with the context; drop the
|
||||||
// drop the handles (no GL) and let the next draw recreate the ring.
|
// handles (no GL) and let the next draw / texture upload recreate them.
|
||||||
ResetUboRingForNewContext();
|
ResetRingForNewContext(g_uboRing);
|
||||||
|
ResetRingForNewContext(g_unpackRing);
|
||||||
}
|
}
|
||||||
|
|
||||||
void ProcessDeferredBufferReleases() {
|
void ProcessDeferredBufferReleases() {
|
||||||
@@ -1441,26 +1480,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
g_pooledBytes = 0;
|
g_pooledBytes = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- Global-UBO ring (see Managers.h) ------------------------------------
|
// --- Persistent-mapped bump rings (see Managers.h) ------------------------
|
||||||
namespace {
|
namespace {
|
||||||
// (Re)create the ring store with room for at least minBytes. Any live
|
// (Re)create the ring store with room for at least minBytes. Any live
|
||||||
// store is retired (deleted once the GPU finished the last frame that
|
// store is retired (deleted once the GPU finished the last frame that
|
||||||
// could reference its slots), never deleted in place. Returns false and
|
// could reference its slots), never deleted in place. Returns false and
|
||||||
// leaves the current store untouched when minBytes cannot fit under the
|
// leaves the current store untouched when minBytes cannot fit under the
|
||||||
// size cap; a GL failure loses the store and latches creationFailed so
|
// size cap; a GL failure loses the store and latches creationFailed so
|
||||||
// draws stop retrying under this context.
|
// later uses stop retrying under this context.
|
||||||
Bool CreateUboRingStorage(SizeT minBytes) {
|
Bool CreateRingStorage(PersistentRing& ring, SizeT minBytes) {
|
||||||
SizeT newSize = kUboRingInitialBytes;
|
SizeT newSize = ring.initialBytes;
|
||||||
while (newSize < minBytes) newSize *= 2;
|
while (newSize < minBytes) newSize *= 2;
|
||||||
if (newSize > kUboRingMaxBytes) return false;
|
if (newSize > ring.maxBytes) return false;
|
||||||
|
|
||||||
if (g_uboRing.id != 0) {
|
if (ring.store.id != 0) {
|
||||||
g_retiredUboRings.push_back(
|
ring.retired.push_back(
|
||||||
{g_uboRing.id, g_uboRing.contextGeneration, DirectGLES::CurrentFrameSerial() + 1});
|
{ring.store.id, ring.store.contextGeneration, DirectGLES::CurrentFrameSerial() + 1});
|
||||||
}
|
}
|
||||||
const Uint32 nextGeneration = g_uboRing.generation + 1;
|
const Uint32 nextGeneration = ring.store.generation + 1;
|
||||||
g_uboRing.id = 0;
|
ring.store.id = 0;
|
||||||
g_uboRing.mappedPtr = nullptr;
|
ring.store.mappedPtr = nullptr;
|
||||||
|
|
||||||
Uint id = 0;
|
Uint id = 0;
|
||||||
g_GLESFuncs.glGenBuffers(1, &id);
|
g_GLESFuncs.glGenBuffers(1, &id);
|
||||||
@@ -1477,213 +1516,249 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
g_GLESFuncs.glDeleteBuffers(1, &id);
|
g_GLESFuncs.glDeleteBuffers(1, &id);
|
||||||
id = 0;
|
id = 0;
|
||||||
} else {
|
} else {
|
||||||
g_uboRing.mappedPtr = static_cast<Uint8*>(ptr);
|
ring.store.mappedPtr = static_cast<Uint8*>(ptr);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (id == 0) {
|
if (id == 0) {
|
||||||
MGLOG_E_ONCE("Global-UBO ring: persistent storage creation failed (%zu bytes); "
|
MGLOG_E_ONCE("%s: persistent storage creation failed (%zu bytes); falling back to the "
|
||||||
"falling back to glBufferSubData uploads.",
|
"client-memory upload path.",
|
||||||
newSize);
|
ring.label, newSize);
|
||||||
g_uboRing.creationFailed = true;
|
ring.store.creationFailed = true;
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
const GLint capsAlignment = g_GLESCapabilities.UniformBufferOffsetAlignment;
|
SizeT alignment = ring.fixedAlignment;
|
||||||
g_uboRing.id = id;
|
if (alignment == 0) {
|
||||||
g_uboRing.size = newSize;
|
const GLint capsAlignment = g_GLESCapabilities.UniformBufferOffsetAlignment;
|
||||||
g_uboRing.head = 0;
|
alignment = capsAlignment > 0 ? static_cast<SizeT>(capsAlignment) : 256;
|
||||||
g_uboRing.tail = 0;
|
}
|
||||||
g_uboRing.generation = nextGeneration;
|
ring.store.id = id;
|
||||||
g_uboRing.alignment = capsAlignment > 0 ? static_cast<SizeT>(capsAlignment) : 256;
|
ring.store.size = newSize;
|
||||||
g_uboRingFrameMarks.clear();
|
ring.store.head = 0;
|
||||||
MGLOG_D("Global-UBO ring: %zu MiB persistent store ready (id %u, gen %u, align %zu).",
|
ring.store.tail = 0;
|
||||||
newSize / (1024u * 1024u), id, nextGeneration, g_uboRing.alignment);
|
ring.store.generation = nextGeneration;
|
||||||
|
ring.store.alignment = alignment;
|
||||||
|
ring.frameMarks.clear();
|
||||||
|
MGLOG_D("%s: %zu MiB persistent store ready (id %u, gen %u, align %zu).", ring.label,
|
||||||
|
newSize / (1024u * 1024u), id, nextGeneration, alignment);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
} // namespace
|
|
||||||
|
|
||||||
Bool UboRingAvailable() {
|
// Shared half of the availability gate (the per-ring kill switch sits in
|
||||||
if (MG_Config::Features.DisableUboRing) return false;
|
// the exported wrappers). Reclamation rides the Present fence watermark;
|
||||||
// Reclamation rides the Present fence watermark; without working fences
|
// without working fences slots would never be provably GPU-idle (same rule
|
||||||
// slots would never be provably GPU-idle (same rule as IsPoolable).
|
// as IsPoolable).
|
||||||
if (!g_GLESFuncs.glBufferStorageEXT || !g_GLESFuncs.glMapBufferRange || !g_GLESFuncs.glGenBuffers ||
|
Bool RingAvailable(PersistentRing& ring) {
|
||||||
!g_GLESFuncs.glFenceSync || !g_GLESFuncs.glGetSynciv) {
|
if (!g_GLESFuncs.glBufferStorageEXT || !g_GLESFuncs.glMapBufferRange || !g_GLESFuncs.glGenBuffers ||
|
||||||
return false;
|
!g_GLESFuncs.glFenceSync || !g_GLESFuncs.glGetSynciv) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (!CanTouchGLNow()) return false;
|
||||||
|
if (ring.store.contextGeneration != g_bufferContextGeneration) {
|
||||||
|
ResetRingForNewContext(ring);
|
||||||
|
}
|
||||||
|
return !ring.store.creationFailed;
|
||||||
}
|
}
|
||||||
if (!CanTouchGLNow()) return false;
|
|
||||||
if (g_uboRing.contextGeneration != g_bufferContextGeneration) {
|
|
||||||
ResetUboRingForNewContext();
|
|
||||||
}
|
|
||||||
return !g_uboRing.creationFailed;
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace {
|
|
||||||
// Division-based rounding fallback: the spec doesn't promise a power-of-two
|
// Division-based rounding fallback: the spec doesn't promise a power-of-two
|
||||||
// alignment. Slot offsets stay multiples of the alignment because every
|
// alignment. Slot offsets stay multiples of the alignment because every
|
||||||
// slot size is, and wrap padding restarts at ring offset 0.
|
// slot size is, and wrap padding restarts at ring offset 0.
|
||||||
inline SizeT UboRingAlignUp(SizeT size, SizeT alignment) {
|
inline SizeT RingAlignUp(SizeT size, SizeT alignment) {
|
||||||
if ((alignment & (alignment - 1)) == 0) {
|
if ((alignment & (alignment - 1)) == 0) {
|
||||||
return (size + alignment - 1) & ~(alignment - 1);
|
return (size + alignment - 1) & ~(alignment - 1);
|
||||||
}
|
}
|
||||||
return (size + alignment - 1) / alignment * alignment;
|
return (size + alignment - 1) / alignment * alignment;
|
||||||
}
|
}
|
||||||
Bool UboRingAllocateSlow(SizeT size, SizeT& outOffset);
|
Bool RingAllocateSlow(PersistentRing& ring, SizeT size, SizeT& outOffset);
|
||||||
} // namespace
|
|
||||||
|
|
||||||
Bool UboRingAllocate(SizeT size, SizeT& outOffset) {
|
// Bump-allocate `size` bytes out of `ring`.
|
||||||
if (size == 0) return false;
|
//
|
||||||
// Fast path: a live ring under the current context with room before both
|
// Fast path: a live store under the current context with room before both
|
||||||
// the wrap boundary and the in-flight tail. Touches no GL and probes no
|
// the wrap boundary and the in-flight tail. Touches no GL and probes no
|
||||||
// frame marks - the sole caller sits behind UboRingAvailable() in the
|
// frame marks - callers sit behind the ring's availability gate, so the
|
||||||
// draw preparation, so the context checks have already run this draw.
|
// context checks have already run for this draw/upload. `tail` may be stale
|
||||||
// `tail` may be stale here (marks are only retired on Present and on the
|
// here (marks are only retired on Present and on the slow path); staleness
|
||||||
// slow path); staleness is conservative - the in-flight span reads too
|
// is conservative - the in-flight span reads too large, the check fails,
|
||||||
// large, the check fails, and the slow path retires marks and re-tries.
|
// and the slow path retires marks and re-tries.
|
||||||
auto& ring = g_uboRing;
|
Bool RingAllocate(PersistentRing& ring, SizeT size, SizeT& outOffset) {
|
||||||
if (ring.id != 0 && ring.contextGeneration == g_bufferContextGeneration) {
|
if (size == 0) return false;
|
||||||
const SizeT alignedSize = UboRingAlignUp(size, ring.alignment);
|
auto& store = ring.store;
|
||||||
// Ring sizes are kUboRingInitialBytes (a power of two) doubled some
|
if (store.id != 0 && store.contextGeneration == g_bufferContextGeneration) {
|
||||||
// number of times, so the offset modulo reduces to a mask.
|
const SizeT alignedSize = RingAlignUp(size, store.alignment);
|
||||||
static_assert((kUboRingInitialBytes & (kUboRingInitialBytes - 1)) == 0,
|
// Ring sizes are the ring's initial size (a power of two) doubled
|
||||||
"ring offset mask below requires power-of-two ring sizes");
|
// some number of times, so the offset modulo reduces to a mask.
|
||||||
const SizeT offset = static_cast<SizeT>(ring.head & (ring.size - 1));
|
static_assert((kUboRingInitialBytes & (kUboRingInitialBytes - 1)) == 0,
|
||||||
if (offset + alignedSize <= ring.size &&
|
"ring offset mask below requires power-of-two ring sizes");
|
||||||
ring.head + alignedSize - ring.tail <= ring.size) {
|
static_assert((kUnpackRingInitialBytes & (kUnpackRingInitialBytes - 1)) == 0,
|
||||||
ring.head += alignedSize;
|
"ring offset mask below requires power-of-two ring sizes");
|
||||||
outOffset = offset;
|
const SizeT offset = static_cast<SizeT>(store.head & (store.size - 1));
|
||||||
return true;
|
if (offset + alignedSize <= store.size && store.head + alignedSize - store.tail <= store.size) {
|
||||||
}
|
store.head += alignedSize;
|
||||||
}
|
|
||||||
return UboRingAllocateSlow(size, outOffset);
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace {
|
|
||||||
Bool UboRingAllocateSlow(SizeT size, SizeT& outOffset) {
|
|
||||||
if (!UboRingAvailable()) return false;
|
|
||||||
const SizeT alignedSize = UboRingAlignUp(size, g_uboRing.alignment);
|
|
||||||
if (g_uboRing.id == 0 && !CreateUboRingStorage(alignedSize)) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Advance tail past every frame the GPU provably finished.
|
|
||||||
const Uint64 completed = DirectGLES::CompletedFrameSerial();
|
|
||||||
SizeT retiredMarks = 0;
|
|
||||||
for (const auto& mark : g_uboRingFrameMarks) {
|
|
||||||
if (mark.frameSerial > completed) break;
|
|
||||||
if (mark.headAtPresent > g_uboRing.tail) g_uboRing.tail = mark.headAtPresent;
|
|
||||||
++retiredMarks;
|
|
||||||
}
|
|
||||||
if (retiredMarks > 0) {
|
|
||||||
g_uboRingFrameMarks.erase(g_uboRingFrameMarks.begin(),
|
|
||||||
g_uboRingFrameMarks.begin() + static_cast<std::ptrdiff_t>(retiredMarks));
|
|
||||||
}
|
|
||||||
|
|
||||||
// A slot may not straddle the ring end; pad the cursor to the boundary.
|
|
||||||
SizeT offset = static_cast<SizeT>(g_uboRing.head % g_uboRing.size);
|
|
||||||
if (offset + alignedSize > g_uboRing.size) {
|
|
||||||
g_uboRing.head += g_uboRing.size - offset;
|
|
||||||
offset = 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (g_uboRing.head + alignedSize - g_uboRing.tail > g_uboRing.size) {
|
|
||||||
// In-flight span would overrun live slots: grow instead of overwrite.
|
|
||||||
if (CreateUboRingStorage(std::max(g_uboRing.size * 2, alignedSize))) {
|
|
||||||
offset = 0;
|
|
||||||
} else if (g_uboRing.creationFailed) {
|
|
||||||
return false; // store lost; callers fall back to glBufferSubData
|
|
||||||
} else {
|
|
||||||
// At the size cap (>kUboRingMaxBytes of uniforms in flight). First
|
|
||||||
// try to free room by waiting for the OLDEST in-flight frames to
|
|
||||||
// retire - a bounded wait that ends as soon as enough tail space
|
|
||||||
// exists, instead of draining the entire queue.
|
|
||||||
constexpr Uint64 kFrameWaitNs = 50ull * 1000 * 1000; // 50ms per frame
|
|
||||||
while (!g_uboRingFrameMarks.empty() &&
|
|
||||||
g_uboRing.head + alignedSize - g_uboRing.tail > g_uboRing.size) {
|
|
||||||
const auto& oldest = g_uboRingFrameMarks.front();
|
|
||||||
if (!DirectGLES::WaitForFrameSerialCompleted(oldest.frameSerial, kFrameWaitNs)) {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (oldest.headAtPresent > g_uboRing.tail) g_uboRing.tail = oldest.headAtPresent;
|
|
||||||
g_uboRingFrameMarks.erase(g_uboRingFrameMarks.begin());
|
|
||||||
}
|
|
||||||
if (g_uboRing.head + alignedSize - g_uboRing.tail <= g_uboRing.size) {
|
|
||||||
offset = static_cast<SizeT>(g_uboRing.head % g_uboRing.size);
|
|
||||||
if (offset + alignedSize > g_uboRing.size) {
|
|
||||||
g_uboRing.head += g_uboRing.size - offset;
|
|
||||||
offset = 0;
|
|
||||||
}
|
|
||||||
g_uboRing.head += alignedSize;
|
|
||||||
outOffset = offset;
|
outOffset = offset;
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
// No usable fence covers the oldest frames: drain once rather than
|
}
|
||||||
// corrupt live slots.
|
return RingAllocateSlow(ring, size, outOffset);
|
||||||
if (g_GLESFuncs.glFinish) g_GLESFuncs.glFinish();
|
}
|
||||||
g_uboRing.tail = g_uboRing.head;
|
|
||||||
g_uboRingFrameMarks.clear();
|
Bool RingAllocateSlow(PersistentRing& ring, SizeT size, SizeT& outOffset) {
|
||||||
// Same-frame slots written before the drain may now be recycled by
|
if (!RingAvailable(ring)) return false;
|
||||||
// the very next allocations; a generation bump keeps later draws
|
auto& store = ring.store;
|
||||||
// from rebinding those cached offsets.
|
// Sizing the store first, so the request is rounded with the alignment
|
||||||
++g_uboRing.generation;
|
// the live store actually carries rather than the pre-creation default.
|
||||||
offset = static_cast<SizeT>(g_uboRing.head % g_uboRing.size);
|
if (store.id == 0 && !CreateRingStorage(ring, RingAlignUp(size, store.alignment))) {
|
||||||
if (offset + alignedSize > g_uboRing.size) {
|
return false;
|
||||||
g_uboRing.head += g_uboRing.size - offset;
|
}
|
||||||
|
const SizeT alignedSize = RingAlignUp(size, store.alignment);
|
||||||
|
|
||||||
|
// Advance tail past every frame the GPU provably finished.
|
||||||
|
const Uint64 completed = DirectGLES::CompletedFrameSerial();
|
||||||
|
SizeT retiredMarks = 0;
|
||||||
|
for (const auto& mark : ring.frameMarks) {
|
||||||
|
if (mark.frameSerial > completed) break;
|
||||||
|
if (mark.headAtPresent > store.tail) store.tail = mark.headAtPresent;
|
||||||
|
++retiredMarks;
|
||||||
|
}
|
||||||
|
if (retiredMarks > 0) {
|
||||||
|
ring.frameMarks.erase(ring.frameMarks.begin(),
|
||||||
|
ring.frameMarks.begin() + static_cast<std::ptrdiff_t>(retiredMarks));
|
||||||
|
}
|
||||||
|
|
||||||
|
// A slot may not straddle the ring end; pad the cursor to the boundary.
|
||||||
|
SizeT offset = static_cast<SizeT>(store.head % store.size);
|
||||||
|
if (offset + alignedSize > store.size) {
|
||||||
|
store.head += store.size - offset;
|
||||||
|
offset = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (store.head + alignedSize - store.tail > store.size) {
|
||||||
|
// In-flight span would overrun live slots: grow instead of overwrite.
|
||||||
|
if (CreateRingStorage(ring, std::max(store.size * 2, alignedSize))) {
|
||||||
offset = 0;
|
offset = 0;
|
||||||
|
} else if (store.creationFailed) {
|
||||||
|
return false; // store lost; callers fall back to their legacy path
|
||||||
|
} else {
|
||||||
|
// At the size cap (>maxBytes in flight). First try to free room
|
||||||
|
// by waiting for the OLDEST in-flight frames to retire - a
|
||||||
|
// bounded wait that ends as soon as enough tail space exists,
|
||||||
|
// instead of draining the entire queue.
|
||||||
|
constexpr Uint64 kFrameWaitNs = 50ull * 1000 * 1000; // 50ms per frame
|
||||||
|
while (!ring.frameMarks.empty() &&
|
||||||
|
store.head + alignedSize - store.tail > store.size) {
|
||||||
|
const auto& oldest = ring.frameMarks.front();
|
||||||
|
if (!DirectGLES::WaitForFrameSerialCompleted(oldest.frameSerial, kFrameWaitNs)) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (oldest.headAtPresent > store.tail) store.tail = oldest.headAtPresent;
|
||||||
|
ring.frameMarks.erase(ring.frameMarks.begin());
|
||||||
|
}
|
||||||
|
if (store.head + alignedSize - store.tail <= store.size) {
|
||||||
|
offset = static_cast<SizeT>(store.head % store.size);
|
||||||
|
if (offset + alignedSize > store.size) {
|
||||||
|
store.head += store.size - offset;
|
||||||
|
offset = 0;
|
||||||
|
}
|
||||||
|
store.head += alignedSize;
|
||||||
|
outOffset = offset;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
// No usable fence covers the oldest frames: drain once rather than
|
||||||
|
// corrupt live slots.
|
||||||
|
if (g_GLESFuncs.glFinish) g_GLESFuncs.glFinish();
|
||||||
|
store.tail = store.head;
|
||||||
|
ring.frameMarks.clear();
|
||||||
|
// Same-frame slots written before the drain may now be recycled by
|
||||||
|
// the very next allocations; a generation bump keeps later draws
|
||||||
|
// from rebinding those cached offsets.
|
||||||
|
++store.generation;
|
||||||
|
offset = static_cast<SizeT>(store.head % store.size);
|
||||||
|
if (offset + alignedSize > store.size) {
|
||||||
|
store.head += store.size - offset;
|
||||||
|
offset = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
store.head += alignedSize;
|
||||||
|
outOffset = offset;
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
g_uboRing.head += alignedSize;
|
// Present()-time upkeep shared by both rings.
|
||||||
outOffset = offset;
|
void RingOnPresent(PersistentRing& ring) {
|
||||||
return true;
|
if (!CanTouchGLNow()) return;
|
||||||
}
|
|
||||||
|
// Delete grown-away stores the GPU is provably done with.
|
||||||
|
const Uint64 completed = DirectGLES::CompletedFrameSerial();
|
||||||
|
for (SizeT i = ring.retired.size(); i-- > 0;) {
|
||||||
|
RetiredRingStore& entry = ring.retired[i];
|
||||||
|
const Bool staleContext = entry.contextGeneration != g_bufferContextGeneration;
|
||||||
|
if (!staleContext && entry.retireSerial > completed) continue;
|
||||||
|
if (!staleContext && entry.id != 0) {
|
||||||
|
ScrubBufferBindingShadowsForId(entry.id);
|
||||||
|
g_GLESFuncs.glDeleteBuffers(1, &entry.id);
|
||||||
|
}
|
||||||
|
ring.retired[i] = ring.retired.back();
|
||||||
|
ring.retired.pop_back();
|
||||||
|
}
|
||||||
|
|
||||||
|
auto& store = ring.store;
|
||||||
|
if (store.id == 0 || store.contextGeneration != g_bufferContextGeneration) return;
|
||||||
|
// Retire completed marks here too — RingAllocate is the main consumer,
|
||||||
|
// but frames that used the ring for nothing would otherwise let the list
|
||||||
|
// grow one entry per Present, unboundedly.
|
||||||
|
SizeT retiredMarks = 0;
|
||||||
|
for (const auto& mark : ring.frameMarks) {
|
||||||
|
if (mark.frameSerial > completed) break;
|
||||||
|
if (mark.headAtPresent > store.tail) store.tail = mark.headAtPresent;
|
||||||
|
++retiredMarks;
|
||||||
|
}
|
||||||
|
if (retiredMarks > 0) {
|
||||||
|
ring.frameMarks.erase(ring.frameMarks.begin(),
|
||||||
|
ring.frameMarks.begin() + static_cast<std::ptrdiff_t>(retiredMarks));
|
||||||
|
}
|
||||||
|
// Record this frame's high-water mark (Present just fenced the serial now
|
||||||
|
// reported by CurrentFrameSerial()). A fence-less Present repeats the
|
||||||
|
// serial; fold into the existing mark.
|
||||||
|
const Uint64 serial = DirectGLES::CurrentFrameSerial();
|
||||||
|
if (!ring.frameMarks.empty() && ring.frameMarks.back().frameSerial == serial) {
|
||||||
|
ring.frameMarks.back().headAtPresent = store.head;
|
||||||
|
} else {
|
||||||
|
ring.frameMarks.push_back({serial, store.head});
|
||||||
|
}
|
||||||
|
}
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
void* UboRingMappedPtr() { return g_uboRing.mappedPtr; }
|
Bool UboRingAvailable() {
|
||||||
Uint UboRingBufferId() { return g_uboRing.id; }
|
if (MG_Config::Features.DisableUboRing) return false;
|
||||||
Uint32 UboRingGeneration() { return g_uboRing.generation; }
|
return RingAvailable(g_uboRing);
|
||||||
|
|
||||||
void UboRingOnPresent() {
|
|
||||||
if (!CanTouchGLNow()) return;
|
|
||||||
|
|
||||||
// Delete grown-away stores the GPU is provably done with.
|
|
||||||
const Uint64 completed = DirectGLES::CompletedFrameSerial();
|
|
||||||
for (SizeT i = g_retiredUboRings.size(); i-- > 0;) {
|
|
||||||
RetiredUboRing& entry = g_retiredUboRings[i];
|
|
||||||
const Bool staleContext = entry.contextGeneration != g_bufferContextGeneration;
|
|
||||||
if (!staleContext && entry.retireSerial > completed) continue;
|
|
||||||
if (!staleContext && entry.id != 0) {
|
|
||||||
ScrubBufferBindingShadowsForId(entry.id);
|
|
||||||
g_GLESFuncs.glDeleteBuffers(1, &entry.id);
|
|
||||||
}
|
|
||||||
g_retiredUboRings[i] = g_retiredUboRings.back();
|
|
||||||
g_retiredUboRings.pop_back();
|
|
||||||
}
|
|
||||||
|
|
||||||
if (g_uboRing.id == 0 || g_uboRing.contextGeneration != g_bufferContextGeneration) return;
|
|
||||||
// Retire completed marks here too — UboRingAllocate is the main consumer,
|
|
||||||
// but frames with no global-UBO draws would otherwise let the list grow
|
|
||||||
// one entry per Present, unboundedly.
|
|
||||||
SizeT retiredMarks = 0;
|
|
||||||
for (const auto& mark : g_uboRingFrameMarks) {
|
|
||||||
if (mark.frameSerial > completed) break;
|
|
||||||
if (mark.headAtPresent > g_uboRing.tail) g_uboRing.tail = mark.headAtPresent;
|
|
||||||
++retiredMarks;
|
|
||||||
}
|
|
||||||
if (retiredMarks > 0) {
|
|
||||||
g_uboRingFrameMarks.erase(g_uboRingFrameMarks.begin(),
|
|
||||||
g_uboRingFrameMarks.begin() + static_cast<std::ptrdiff_t>(retiredMarks));
|
|
||||||
}
|
|
||||||
// Record this frame's high-water mark (Present just fenced the serial now
|
|
||||||
// reported by CurrentFrameSerial()). A fence-less Present repeats the
|
|
||||||
// serial; fold into the existing mark.
|
|
||||||
const Uint64 serial = DirectGLES::CurrentFrameSerial();
|
|
||||||
if (!g_uboRingFrameMarks.empty() && g_uboRingFrameMarks.back().frameSerial == serial) {
|
|
||||||
g_uboRingFrameMarks.back().headAtPresent = g_uboRing.head;
|
|
||||||
} else {
|
|
||||||
g_uboRingFrameMarks.push_back({serial, g_uboRing.head});
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
Bool UboRingAllocate(SizeT size, SizeT& outOffset) { return RingAllocate(g_uboRing, size, outOffset); }
|
||||||
|
|
||||||
|
void* UboRingMappedPtr() { return g_uboRing.store.mappedPtr; }
|
||||||
|
Uint UboRingBufferId() { return g_uboRing.store.id; }
|
||||||
|
Uint32 UboRingGeneration() { return g_uboRing.store.generation; }
|
||||||
|
|
||||||
|
void UboRingOnPresent() { RingOnPresent(g_uboRing); }
|
||||||
|
|
||||||
|
Bool UnpackRingAvailable() {
|
||||||
|
if (MG_Config::Features.DisableUnpackRing) return false;
|
||||||
|
return RingAvailable(g_unpackRing);
|
||||||
|
}
|
||||||
|
|
||||||
|
Bool UnpackRingAllocate(SizeT size, SizeT& outOffset) {
|
||||||
|
// A request the ring could never satisfy even empty would otherwise walk
|
||||||
|
// the whole grow/drain ladder before failing.
|
||||||
|
if (size > kUnpackRingMaxBytes) return false;
|
||||||
|
return RingAllocate(g_unpackRing, size, outOffset);
|
||||||
|
}
|
||||||
|
|
||||||
|
void* UnpackRingMappedPtr() { return g_unpackRing.store.mappedPtr; }
|
||||||
|
Uint UnpackRingBufferId() { return g_unpackRing.store.id; }
|
||||||
|
SizeT UnpackRingMaxBytes() { return kUnpackRingMaxBytes; }
|
||||||
|
|
||||||
|
void UnpackRingOnPresent() { RingOnPresent(g_unpackRing); }
|
||||||
} // namespace BufferImpl
|
} // namespace BufferImpl
|
||||||
|
|
||||||
namespace VertexArrayImpl {
|
namespace VertexArrayImpl {
|
||||||
@@ -2580,6 +2655,93 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
static inline GLint s_skipImages = 0;
|
static inline GLint s_skipImages = 0;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// --- Unpack-ring staging (see BufferImpl::UnpackRingAvailable) ---------------
|
||||||
|
// One rectangular (or whole-level) region of a level shadow, repacked TIGHTLY
|
||||||
|
// into the persistently-mapped unpack PBO. The upload that follows passes the
|
||||||
|
// returned ring offset with UNPACK_ROW_LENGTH / UNPACK_IMAGE_HEIGHT left at the
|
||||||
|
// surrounding ScopedDefaultUnpackState's 0, so the driver reads exactly the
|
||||||
|
// bytes staged here - strictly fewer than the ROW_LENGTH-strided client-pointer
|
||||||
|
// upload it replaces, which made the driver walk the whole level's stride.
|
||||||
|
//
|
||||||
|
// `blocks` describes one or more source regions to stage back-to-back into a
|
||||||
|
// SINGLE ring allocation; the ring offset of block i lands in blocks[i].offset.
|
||||||
|
// Deliberately without default member initializers: call sites declare a
|
||||||
|
// kMaxDirtyRects-sized array of these per dirty level and fill only the entries
|
||||||
|
// they use, so the type stays trivially default-constructible and the
|
||||||
|
// declaration costs nothing on the levels that never reach the ring.
|
||||||
|
struct UnpackStagingBlock {
|
||||||
|
const Uint8* src; // top-left texel of the region in the level shadow
|
||||||
|
SizeT rowBytes; // bytes per region row (region width * bpp)
|
||||||
|
SizeT rows; // region height
|
||||||
|
SizeT slices; // region depth (1 for 2D)
|
||||||
|
SizeT srcRowStride; // level row pitch
|
||||||
|
SizeT srcSliceStride; // level slice pitch
|
||||||
|
SizeT offset; // out: byte offset into the ring store
|
||||||
|
};
|
||||||
|
|
||||||
|
// False (nothing staged, ring untouched) whenever the caller must keep the
|
||||||
|
// client-pointer path: ring unavailable, a degenerate block, or a total that
|
||||||
|
// overflows / exceeds the ring's cap.
|
||||||
|
static Bool StageBlocksIntoUnpackRing(UnpackStagingBlock* blocks, SizeT blockCount) {
|
||||||
|
if (blocks == nullptr || blockCount == 0) return false;
|
||||||
|
if (!BufferImpl::UnpackRingAvailable()) return false;
|
||||||
|
|
||||||
|
const SizeT maxBytes = BufferImpl::UnpackRingMaxBytes();
|
||||||
|
SizeT total = 0;
|
||||||
|
for (SizeT i = 0; i < blockCount; ++i) {
|
||||||
|
const UnpackStagingBlock& b = blocks[i];
|
||||||
|
if (b.src == nullptr || b.rowBytes == 0 || b.rows == 0 || b.slices == 0) return false;
|
||||||
|
// Every factor is bounded by the level's own byte size, but the
|
||||||
|
// multiplications are on SizeT: check each against the cap instead of
|
||||||
|
// trusting a product that could have wrapped.
|
||||||
|
if (b.rowBytes > maxBytes || b.rows > maxBytes / b.rowBytes) return false;
|
||||||
|
const SizeT sliceBytes = b.rowBytes * b.rows;
|
||||||
|
if (b.slices > maxBytes / sliceBytes) return false;
|
||||||
|
const SizeT blockBytes = sliceBytes * b.slices;
|
||||||
|
if (blockBytes > maxBytes - total) return false;
|
||||||
|
total += blockBytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
SizeT base = 0;
|
||||||
|
if (!BufferImpl::UnpackRingAllocate(total, base)) return false;
|
||||||
|
// AFTER the allocation: growing the ring replaces the store and its map.
|
||||||
|
auto* ringBytes = static_cast<Uint8*>(BufferImpl::UnpackRingMappedPtr());
|
||||||
|
if (ringBytes == nullptr) return false;
|
||||||
|
|
||||||
|
SizeT cursor = base;
|
||||||
|
for (SizeT i = 0; i < blockCount; ++i) {
|
||||||
|
UnpackStagingBlock& b = blocks[i];
|
||||||
|
b.offset = cursor;
|
||||||
|
Uint8* dst = ringBytes + cursor;
|
||||||
|
// Each block's size is a whole number of texels, so consecutive block
|
||||||
|
// offsets stay multiples of the pixel type's size just as `base` is.
|
||||||
|
if (b.rowBytes == b.srcRowStride && b.rowBytes * b.rows == b.srcSliceStride) {
|
||||||
|
Memcpy(dst, b.src, b.rowBytes * b.rows * b.slices); // region IS the level
|
||||||
|
} else {
|
||||||
|
for (SizeT z = 0; z < b.slices; ++z) {
|
||||||
|
const Uint8* srcSlice = b.src + z * b.srcSliceStride;
|
||||||
|
if (b.rowBytes == b.srcRowStride) {
|
||||||
|
Memcpy(dst, srcSlice, b.rowBytes * b.rows); // full-width rows
|
||||||
|
dst += b.rowBytes * b.rows;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
for (SizeT y = 0; y < b.rows; ++y) {
|
||||||
|
Memcpy(dst, srcSlice + y * b.srcRowStride, b.rowBytes);
|
||||||
|
dst += b.rowBytes;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
cursor += b.rowBytes * b.rows * b.slices;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The `pixels` argument of a PBO-sourced glTexSubImage: a byte offset dressed
|
||||||
|
// up as a pointer.
|
||||||
|
static const void* UnpackRingPixelOffset(SizeT offset) {
|
||||||
|
return reinterpret_cast<const void*>(static_cast<std::uintptr_t>(offset));
|
||||||
|
}
|
||||||
|
|
||||||
static Uint GetNormFallbackComponentCount(TextureInternalFormat format) {
|
static Uint GetNormFallbackComponentCount(TextureInternalFormat format) {
|
||||||
switch (format) {
|
switch (format) {
|
||||||
case TextureInternalFormat::R8Snorm:
|
case TextureInternalFormat::R8Snorm:
|
||||||
@@ -3840,6 +4002,15 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
dirtyRectCount = textureMipmapObject->GetStorageDirtyRects(
|
dirtyRectCount = textureMipmapObject->GetStorageDirtyRects(
|
||||||
uploadTarget, level, dirtyRects,
|
uploadTarget, level, dirtyRects,
|
||||||
MG_State::GLState::MipmapStorage::kMaxDirtyRects);
|
MG_State::GLState::MipmapStorage::kMaxDirtyRects);
|
||||||
|
// The scatter refinement pays only on the client-pointer path,
|
||||||
|
// where fewer bytes mean less driver-side copying. Through the
|
||||||
|
// unpack ring every glTexSubImage is a GPU copy job (Mali), so
|
||||||
|
// ~100 sprite rects become ~100 jobs whose fixed cost dwarfs
|
||||||
|
// the union box's extra bytes - measured +6 ms/frame of GPU
|
||||||
|
// time in MC's animated-atlas ticks. One box, one job.
|
||||||
|
if (BufferImpl::UnpackRingAvailable()) {
|
||||||
|
dirtyRectCount = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
const auto rectShadowPtr = [&](const MG_State::GLState::MipmapDirtyRegion& rect) {
|
const auto rectShadowPtr = [&](const MG_State::GLState::MipmapDirtyRegion& rect) {
|
||||||
return static_cast<const Uint8*>(uploadData) +
|
return static_cast<const Uint8*>(uploadData) +
|
||||||
@@ -3847,34 +4018,111 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
static_cast<SizeT>(rect.lo.y()) * levelRowBytes +
|
static_cast<SizeT>(rect.lo.y()) * levelRowBytes +
|
||||||
static_cast<SizeT>(rect.lo.x()) * bpp;
|
static_cast<SizeT>(rect.lo.x()) * bpp;
|
||||||
};
|
};
|
||||||
|
// Unpack-ring staging plan, decided ONCE for whichever branch
|
||||||
|
// below runs: either every glTexSubImage of this level sources
|
||||||
|
// from the ring or none does, so the pixel-unpack binding is
|
||||||
|
// toggled exactly once per level. The plan's three shapes line
|
||||||
|
// up 1:1 with the switch's three branches.
|
||||||
|
//
|
||||||
|
// Regions are repacked TIGHTLY (row length = the region's own
|
||||||
|
// width), which is why the ring path issues no glPixelStorei at
|
||||||
|
// all: the surrounding ScopedDefaultUnpackState already holds
|
||||||
|
// ROW_LENGTH/IMAGE_HEIGHT at 0, which is exactly what a tight
|
||||||
|
// block wants. It also stages strictly fewer bytes than the
|
||||||
|
// client-pointer path makes the driver walk, which strides over
|
||||||
|
// the whole level width.
|
||||||
|
UnpackStagingBlock
|
||||||
|
stagingBlocks[MG_State::GLState::MipmapStorage::kMaxDirtyRects];
|
||||||
|
SizeT stagingBlockCount = 0;
|
||||||
|
const Bool ringUsable = BufferImpl::UnpackRingAvailable();
|
||||||
|
if (!ringUsable) {
|
||||||
|
// Nothing to plan: every branch below keeps the client
|
||||||
|
// pointer and its ROW_LENGTH striding, unchanged.
|
||||||
|
} else if (subRectEligible && dirtyRectCount >= 2) {
|
||||||
|
for (SizeT r = 0; r < dirtyRectCount; ++r) {
|
||||||
|
const auto& rect = dirtyRects[r];
|
||||||
|
stagingBlocks[r] = {
|
||||||
|
rectShadowPtr(rect),
|
||||||
|
static_cast<SizeT>(rect.hi.x() - rect.lo.x()) * bpp,
|
||||||
|
static_cast<SizeT>(rect.hi.y() - rect.lo.y()),
|
||||||
|
static_cast<SizeT>(std::max(rect.hi.z() - rect.lo.z(), 1)),
|
||||||
|
levelRowBytes,
|
||||||
|
levelSliceBytes,
|
||||||
|
0};
|
||||||
|
}
|
||||||
|
stagingBlockCount = dirtyRectCount;
|
||||||
|
} else if (subRectEligible) {
|
||||||
|
stagingBlocks[0] = {regionPtr,
|
||||||
|
static_cast<SizeT>(regionSize.x()) * bpp,
|
||||||
|
static_cast<SizeT>(regionSize.y()),
|
||||||
|
static_cast<SizeT>(std::max(regionSize.z(), 1)),
|
||||||
|
levelRowBytes,
|
||||||
|
levelSliceBytes,
|
||||||
|
0};
|
||||||
|
stagingBlockCount = 1;
|
||||||
|
} else if (uploadData == mipData && texelCount > 0 &&
|
||||||
|
byteSize % texelCount == 0) {
|
||||||
|
// Whole level, shadow bytes verbatim: `byteSize` is exactly
|
||||||
|
// what the driver would have read from the client pointer,
|
||||||
|
// and the shadow is already tightly packed. A CONVERTED
|
||||||
|
// level stays on the pointer path - its buffer's length is
|
||||||
|
// the conversion's, not the shadow's, so nothing here can
|
||||||
|
// size the driver's read of it. The whole-texels check is
|
||||||
|
// the same sanity the sub-rect path applies, and it is what
|
||||||
|
// makes "the shadow's bytes per texel == the transfer's"
|
||||||
|
// hold. (A 1D array's upload size permutes height and depth,
|
||||||
|
// but the texel count and the tight byte order are the same,
|
||||||
|
// so the flat copy still describes what the driver reads.)
|
||||||
|
stagingBlocks[0] = {static_cast<const Uint8*>(uploadData),
|
||||||
|
static_cast<SizeT>(byteSize),
|
||||||
|
1,
|
||||||
|
1,
|
||||||
|
static_cast<SizeT>(byteSize),
|
||||||
|
static_cast<SizeT>(byteSize),
|
||||||
|
0};
|
||||||
|
stagingBlockCount = 1;
|
||||||
|
}
|
||||||
|
const Bool ringStaged =
|
||||||
|
stagingBlockCount > 0 &&
|
||||||
|
StageBlocksIntoUnpackRing(stagingBlocks, stagingBlockCount);
|
||||||
|
if (ringStaged) {
|
||||||
|
BufferImpl::BindPixelUnpackBufferId(BufferImpl::UnpackRingBufferId());
|
||||||
|
}
|
||||||
switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) {
|
switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) {
|
||||||
case TextureTarget::Texture2D:
|
case TextureTarget::Texture2D:
|
||||||
case TextureTarget::TextureCubeMap:
|
case TextureTarget::TextureCubeMap:
|
||||||
if (subRectEligible && dirtyRectCount >= 2) {
|
if (subRectEligible && dirtyRectCount >= 2) {
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
||||||
for (SizeT r = 0; r < dirtyRectCount; ++r) {
|
for (SizeT r = 0; r < dirtyRectCount; ++r) {
|
||||||
const auto& rect = dirtyRects[r];
|
const auto& rect = dirtyRects[r];
|
||||||
g_GLESFuncs.glTexSubImage2D(
|
g_GLESFuncs.glTexSubImage2D(
|
||||||
glUploadTarget, static_cast<GLint>(level), rect.lo.x(),
|
glUploadTarget, static_cast<GLint>(level), rect.lo.x(),
|
||||||
rect.lo.y(), static_cast<GLsizei>(rect.hi.x() - rect.lo.x()),
|
rect.lo.y(), static_cast<GLsizei>(rect.hi.x() - rect.lo.x()),
|
||||||
static_cast<GLsizei>(rect.hi.y() - rect.lo.y()), glFormat,
|
static_cast<GLsizei>(rect.hi.y() - rect.lo.y()), glFormat,
|
||||||
glType, rectShadowPtr(rect));
|
glType,
|
||||||
|
ringStaged ? UnpackRingPixelOffset(stagingBlocks[r].offset)
|
||||||
|
: static_cast<const void*>(rectShadowPtr(rect)));
|
||||||
}
|
}
|
||||||
// The surrounding ScopedDefaultUnpackState shadow says 0.
|
// The surrounding ScopedDefaultUnpackState shadow says 0.
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
||||||
} else if (subRectEligible) {
|
} else if (subRectEligible) {
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
||||||
g_GLESFuncs.glTexSubImage2D(
|
g_GLESFuncs.glTexSubImage2D(
|
||||||
glUploadTarget, static_cast<GLint>(level), dirtyRegion.lo.x(),
|
glUploadTarget, static_cast<GLint>(level), dirtyRegion.lo.x(),
|
||||||
dirtyRegion.lo.y(), static_cast<GLsizei>(regionSize.x()),
|
dirtyRegion.lo.y(), static_cast<GLsizei>(regionSize.x()),
|
||||||
static_cast<GLsizei>(regionSize.y()), glFormat, glType, regionPtr);
|
static_cast<GLsizei>(regionSize.y()), glFormat, glType,
|
||||||
|
ringStaged ? UnpackRingPixelOffset(stagingBlocks[0].offset)
|
||||||
|
: static_cast<const void*>(regionPtr));
|
||||||
// The surrounding ScopedDefaultUnpackState shadow says 0.
|
// The surrounding ScopedDefaultUnpackState shadow says 0.
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
if (!ringStaged) g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
||||||
} else {
|
} else {
|
||||||
g_GLESFuncs.glTexSubImage2D(glUploadTarget, static_cast<GLint>(level), 0, 0,
|
g_GLESFuncs.glTexSubImage2D(glUploadTarget, static_cast<GLint>(level), 0, 0,
|
||||||
static_cast<GLsizei>(uploadSize.x()),
|
static_cast<GLsizei>(uploadSize.x()),
|
||||||
static_cast<GLsizei>(uploadSize.y()), glFormat,
|
static_cast<GLsizei>(uploadSize.y()), glFormat,
|
||||||
glType, uploadData);
|
glType,
|
||||||
|
ringStaged
|
||||||
|
? UnpackRingPixelOffset(stagingBlocks[0].offset)
|
||||||
|
: uploadData);
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
case TextureTarget::Texture3D:
|
case TextureTarget::Texture3D:
|
||||||
@@ -3883,8 +4131,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// like a 2D array whose depth is 6 * the cube count.
|
// like a 2D array whose depth is 6 * the cube count.
|
||||||
case TextureTarget::TextureCubeMapArray:
|
case TextureTarget::TextureCubeMapArray:
|
||||||
if (subRectEligible && dirtyRectCount >= 2) {
|
if (subRectEligible && dirtyRectCount >= 2) {
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
if (!ringStaged) {
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y());
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
||||||
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y());
|
||||||
|
}
|
||||||
for (SizeT r = 0; r < dirtyRectCount; ++r) {
|
for (SizeT r = 0; r < dirtyRectCount; ++r) {
|
||||||
const auto& rect = dirtyRects[r];
|
const auto& rect = dirtyRects[r];
|
||||||
g_GLESFuncs.glTexSubImage3D(
|
g_GLESFuncs.glTexSubImage3D(
|
||||||
@@ -3893,27 +4143,40 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
static_cast<GLsizei>(rect.hi.x() - rect.lo.x()),
|
static_cast<GLsizei>(rect.hi.x() - rect.lo.x()),
|
||||||
static_cast<GLsizei>(rect.hi.y() - rect.lo.y()),
|
static_cast<GLsizei>(rect.hi.y() - rect.lo.y()),
|
||||||
static_cast<GLsizei>(rect.hi.z() - rect.lo.z()), glFormat,
|
static_cast<GLsizei>(rect.hi.z() - rect.lo.z()), glFormat,
|
||||||
glType, rectShadowPtr(rect));
|
glType,
|
||||||
|
ringStaged ? UnpackRingPixelOffset(stagingBlocks[r].offset)
|
||||||
|
: static_cast<const void*>(rectShadowPtr(rect)));
|
||||||
|
}
|
||||||
|
if (!ringStaged) {
|
||||||
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
||||||
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0);
|
||||||
}
|
}
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0);
|
|
||||||
} else if (subRectEligible) {
|
} else if (subRectEligible) {
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
if (!ringStaged) {
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y());
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, texelSize.x());
|
||||||
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, texelSize.y());
|
||||||
|
}
|
||||||
g_GLESFuncs.glTexSubImage3D(
|
g_GLESFuncs.glTexSubImage3D(
|
||||||
glUploadTarget, static_cast<GLint>(level), dirtyRegion.lo.x(),
|
glUploadTarget, static_cast<GLint>(level), dirtyRegion.lo.x(),
|
||||||
dirtyRegion.lo.y(), dirtyRegion.lo.z(),
|
dirtyRegion.lo.y(), dirtyRegion.lo.z(),
|
||||||
static_cast<GLsizei>(regionSize.x()),
|
static_cast<GLsizei>(regionSize.x()),
|
||||||
static_cast<GLsizei>(regionSize.y()),
|
static_cast<GLsizei>(regionSize.y()),
|
||||||
static_cast<GLsizei>(regionSize.z()), glFormat, glType, regionPtr);
|
static_cast<GLsizei>(regionSize.z()), glFormat, glType,
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
ringStaged ? UnpackRingPixelOffset(stagingBlocks[0].offset)
|
||||||
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0);
|
: static_cast<const void*>(regionPtr));
|
||||||
|
if (!ringStaged) {
|
||||||
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
||||||
|
g_GLESFuncs.glPixelStorei(GL_UNPACK_IMAGE_HEIGHT, 0);
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
g_GLESFuncs.glTexSubImage3D(glUploadTarget, static_cast<GLint>(level), 0, 0, 0,
|
g_GLESFuncs.glTexSubImage3D(glUploadTarget, static_cast<GLint>(level), 0, 0, 0,
|
||||||
static_cast<GLsizei>(uploadSize.x()),
|
static_cast<GLsizei>(uploadSize.x()),
|
||||||
static_cast<GLsizei>(uploadSize.y()),
|
static_cast<GLsizei>(uploadSize.y()),
|
||||||
static_cast<GLsizei>(uploadSize.z()), glFormat,
|
static_cast<GLsizei>(uploadSize.z()), glFormat,
|
||||||
glType, uploadData);
|
glType,
|
||||||
|
ringStaged
|
||||||
|
? UnpackRingPixelOffset(stagingBlocks[0].offset)
|
||||||
|
: uploadData);
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
default:
|
default:
|
||||||
@@ -3921,6 +4184,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
MG_Util::ConvertTextureTargetToString(stateTextureObject->GetTarget()).c_str());
|
MG_Util::ConvertTextureTargetToString(stateTextureObject->GetTarget()).c_str());
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
if (ringStaged) {
|
||||||
|
// Back to the resting unbound state every other upload site
|
||||||
|
// in this file assumes (their BindPixelUnpackBufferId(0) is
|
||||||
|
// meant to stay a shadow no-op).
|
||||||
|
BufferImpl::BindPixelUnpackBufferId(0);
|
||||||
|
}
|
||||||
textureMipmapObject->MarkStorageDirty(uploadTarget, level, false);
|
textureMipmapObject->MarkStorageDirty(uploadTarget, level, false);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -510,6 +510,40 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
|||||||
// Present()-time upkeep: records the frame's high-water mark for reclamation
|
// Present()-time upkeep: records the frame's high-water mark for reclamation
|
||||||
// and deletes grown-away ring stores once the GPU is done with them.
|
// and deletes grown-away ring stores once the GPU is done with them.
|
||||||
void UboRingOnPresent();
|
void UboRingOnPresent();
|
||||||
|
|
||||||
|
// --- Texture unpack-PBO ring ----------------------------------------------
|
||||||
|
// The same persistent-mapped bump allocator, staging TEXTURE UPLOADS. A
|
||||||
|
// glTexSubImage from client memory hands the driver a pointer it must read
|
||||||
|
// before the call returns, so the copy has to be ordered against whatever GPU
|
||||||
|
// work still reads the destination texture: Mali resolves that by BLOCKING the
|
||||||
|
// calling thread (osup_sync_object_wait) instead of ghosting, and Minecraft
|
||||||
|
// re-uploads animated atlas sprites and the lightmap every tick into textures
|
||||||
|
// the in-flight frame is still sampling. Staging the bytes into a
|
||||||
|
// GPU-visible unpack PBO and passing an OFFSET instead lets the driver queue
|
||||||
|
// the copy in the command stream with no CPU wait at all.
|
||||||
|
//
|
||||||
|
// Same reclamation contract as the UBO ring: no ring bytes are recycled before
|
||||||
|
// the frame that referenced them completed on the GPU, so a staged block stays
|
||||||
|
// intact for as long as the queued transfer can still be reading it. The store
|
||||||
|
// therefore settles at roughly (bytes staged per frame) x (frames in flight),
|
||||||
|
// which is what to watch if this ring ever shows up in an RSS regression: it
|
||||||
|
// grows on demand from 4 MiB and is capped, not unbounded.
|
||||||
|
//
|
||||||
|
// False when the feature is disabled (MOBILEGL_DISABLE_UNPACK_RING),
|
||||||
|
// EXT_buffer_storage / fences are missing, the ES context is not current, or
|
||||||
|
// ring creation already failed under this context. Callers then upload from
|
||||||
|
// the client pointer exactly as before.
|
||||||
|
Bool UnpackRingAvailable();
|
||||||
|
// Bump-allocate `size` bytes aligned to 64 (a PBO-sourced glTexSubImage only
|
||||||
|
// owes the driver the pixel type's own alignment). Grows the ring when the
|
||||||
|
// in-flight span would be overrun; false when the request exceeds the ring's
|
||||||
|
// size cap or storage (re)creation fails.
|
||||||
|
Bool UnpackRingAllocate(SizeT size, SizeT& outOffset);
|
||||||
|
void* UnpackRingMappedPtr();
|
||||||
|
Uint UnpackRingBufferId();
|
||||||
|
// Largest single staging request the ring can ever satisfy.
|
||||||
|
SizeT UnpackRingMaxBytes();
|
||||||
|
void UnpackRingOnPresent();
|
||||||
} // namespace BufferImpl
|
} // namespace BufferImpl
|
||||||
|
|
||||||
namespace VertexArrayImpl {
|
namespace VertexArrayImpl {
|
||||||
|
|||||||
@@ -218,7 +218,8 @@ add_library(glretrace_common STATIC
|
|||||||
"${APITRACE_ROOT}/retrace/metric_backend_opengl.cpp"
|
"${APITRACE_ROOT}/retrace/metric_backend_opengl.cpp"
|
||||||
"${APITRACE_ROOT}/retrace/metric_helper.cpp"
|
"${APITRACE_ROOT}/retrace/metric_helper.cpp"
|
||||||
"${APITRACE_ROOT}/retrace/metric_writer.cpp"
|
"${APITRACE_ROOT}/retrace/metric_writer.cpp"
|
||||||
apitrace_glws_android.cpp)
|
apitrace_glws_android.cpp
|
||||||
|
trace_benchmark.cpp)
|
||||||
target_include_directories(glretrace_common PUBLIC
|
target_include_directories(glretrace_common PUBLIC
|
||||||
"${APITRACE_GENERATED_DIR}"
|
"${APITRACE_GENERATED_DIR}"
|
||||||
"${APITRACE_ROOT}/dispatch"
|
"${APITRACE_ROOT}/dispatch"
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
#include "apitrace_fbo_dump.hpp"
|
#include "apitrace_fbo_dump.hpp"
|
||||||
#include "glws.hpp"
|
#include "glws.hpp"
|
||||||
#include "retrace.hpp"
|
#include "retrace.hpp"
|
||||||
|
#include "trace_benchmark.hpp"
|
||||||
|
|
||||||
#include <android/native_window.h>
|
#include <android/native_window.h>
|
||||||
#include <EGL/egl.h>
|
#include <EGL/egl.h>
|
||||||
@@ -186,6 +187,9 @@ public:
|
|||||||
char callNo[32];
|
char callNo[32];
|
||||||
snprintf(callNo, sizeof(callNo), "%u", retrace::callNo);
|
snprintf(callNo, sizeof(callNo), "%u", retrace::callNo);
|
||||||
gEgl.swapBuffers(gDisplay, surface);
|
gEgl.swapBuffers(gDisplay, surface);
|
||||||
|
// Frame boundary: retrace_eglSwapBuffers() has already run frame_complete() and
|
||||||
|
// handed the frame to us. No-op unless benchmark mode armed the timer.
|
||||||
|
mobilegl_trace::benchmark::OnFrameBoundary();
|
||||||
HoldAfterTargetPresent(callNo);
|
HoldAfterTargetPresent(callNo);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,94 @@
|
|||||||
|
#include "trace_benchmark.hpp"
|
||||||
|
|
||||||
|
#include <chrono>
|
||||||
|
#include <cstdlib>
|
||||||
|
#include <utility>
|
||||||
|
#include <dlfcn.h>
|
||||||
|
|
||||||
|
namespace mobilegl_trace {
|
||||||
|
namespace benchmark {
|
||||||
|
namespace {
|
||||||
|
|
||||||
|
using Clock = std::chrono::steady_clock;
|
||||||
|
using GlFinishFn = void (*)();
|
||||||
|
|
||||||
|
constexpr std::size_t kFrameReserve = 4096;
|
||||||
|
|
||||||
|
bool gEnabled = false;
|
||||||
|
bool gFinishEachFrame = false;
|
||||||
|
bool gResolvedGlFinish = false;
|
||||||
|
GlFinishFn gGlFinish = nullptr;
|
||||||
|
Clock::time_point gStart;
|
||||||
|
Clock::time_point gLastBoundary;
|
||||||
|
std::vector<double> gFrameMs;
|
||||||
|
|
||||||
|
// Same resolution order the glws layers use for MobileGL's entry points: the replay driver
|
||||||
|
// already dlopen()ed the library with RTLD_GLOBAL before retrace started, so RTLD_NOLOAD
|
||||||
|
// finds that handle instead of loading a second copy, and RTLD_DEFAULT is the fallback for
|
||||||
|
// the case where it was linked in rather than dlopen()ed.
|
||||||
|
GlFinishFn ResolveGlFinish() {
|
||||||
|
void *handle = nullptr;
|
||||||
|
const char *library = std::getenv("MOBILEGL_TRACE_LIBRARY");
|
||||||
|
if (library != nullptr && library[0] != '\0') {
|
||||||
|
handle = dlopen(library, RTLD_NOW | RTLD_GLOBAL | RTLD_NOLOAD);
|
||||||
|
}
|
||||||
|
if (handle == nullptr) {
|
||||||
|
handle = dlopen("libMobileGL.so", RTLD_NOW | RTLD_GLOBAL | RTLD_NOLOAD);
|
||||||
|
}
|
||||||
|
if (handle != nullptr) {
|
||||||
|
void *symbol = dlsym(handle, "glFinish");
|
||||||
|
if (symbol != nullptr) {
|
||||||
|
return reinterpret_cast<GlFinishFn>(symbol);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return reinterpret_cast<GlFinishFn>(dlsym(RTLD_DEFAULT, "glFinish"));
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
void Begin(bool finishEachFrame) {
|
||||||
|
gFrameMs.clear();
|
||||||
|
gFrameMs.reserve(kFrameReserve);
|
||||||
|
gFinishEachFrame = finishEachFrame;
|
||||||
|
gResolvedGlFinish = false;
|
||||||
|
gGlFinish = nullptr;
|
||||||
|
gStart = Clock::now();
|
||||||
|
gLastBoundary = gStart;
|
||||||
|
gEnabled = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
void OnFrameBoundary() {
|
||||||
|
if (!gEnabled) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (gFinishEachFrame) {
|
||||||
|
// Resolved on the first boundary rather than in Begin(): a context only exists once
|
||||||
|
// the trace has created one, and glFinish before that would be pointless anyway.
|
||||||
|
if (!gResolvedGlFinish) {
|
||||||
|
gGlFinish = ResolveGlFinish();
|
||||||
|
gResolvedGlFinish = true;
|
||||||
|
}
|
||||||
|
if (gGlFinish != nullptr) {
|
||||||
|
gGlFinish();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const Clock::time_point now = Clock::now();
|
||||||
|
gFrameMs.push_back(std::chrono::duration<double, std::milli>(now - gLastBoundary).count());
|
||||||
|
gLastBoundary = now;
|
||||||
|
}
|
||||||
|
|
||||||
|
Report End() {
|
||||||
|
Report report;
|
||||||
|
if (!gEnabled) {
|
||||||
|
return report;
|
||||||
|
}
|
||||||
|
gEnabled = false;
|
||||||
|
gFinishEachFrame = false;
|
||||||
|
report.totalSeconds = std::chrono::duration<double>(Clock::now() - gStart).count();
|
||||||
|
report.frameMs = std::move(gFrameMs);
|
||||||
|
gFrameMs.clear();
|
||||||
|
return report;
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace benchmark
|
||||||
|
} // namespace mobilegl_trace
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
namespace mobilegl_trace {
|
||||||
|
namespace benchmark {
|
||||||
|
|
||||||
|
// Per-frame wall-clock timing for the retrace loop, shared by the Android replay runner
|
||||||
|
// and the desktop CLI. Disarmed unless Begin() armed it, and the frame-boundary hook is a
|
||||||
|
// single bool test in that case, so the correctness harness pays nothing for it.
|
||||||
|
//
|
||||||
|
// Retrace runs --singlethread, so all of this is deliberately plain globals: Begin(),
|
||||||
|
// OnFrameBoundary() and End() are only ever reached from the one retrace thread.
|
||||||
|
|
||||||
|
// Arms timing for the retrace that is about to run.
|
||||||
|
//
|
||||||
|
// finishEachFrame issues a full glFinish through the replayed context at every frame
|
||||||
|
// boundary, so a recorded frame time covers GPU completion and not just CPU submission.
|
||||||
|
// That matters on tiled mobile GPUs, where a swap without a sync returns long before the
|
||||||
|
// tiler is done and the numbers degenerate into "how fast can we feed the driver". The
|
||||||
|
// price is that finishing every frame serializes CPU/GPU overlap, so the absolute frame
|
||||||
|
// times are pessimistic against a real running game - they are deterministic and
|
||||||
|
// comparable between backends and revisions, which is what a benchmark fixture is for.
|
||||||
|
// With finishEachFrame off the run measures CPU-side submission only.
|
||||||
|
void Begin(bool finishEachFrame);
|
||||||
|
|
||||||
|
// Frame-boundary hook. Called from the platform glws swapBuffers override, which is where
|
||||||
|
// apitrace's replay loop advances the frame: retrace_eglSwapBuffers() calls
|
||||||
|
// frame_complete() and then Drawable::swapBuffers().
|
||||||
|
void OnFrameBoundary();
|
||||||
|
|
||||||
|
struct Report {
|
||||||
|
// Wall time of every completed frame, in milliseconds.
|
||||||
|
std::vector<double> frameMs;
|
||||||
|
// Begin() to End(), in seconds. Covers trace parsing and the leading partial frame too,
|
||||||
|
// which is why it is reported next to the per-frame statistics rather than derived from
|
||||||
|
// them.
|
||||||
|
double totalSeconds = 0.0;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Disarms timing and hands back what was recorded.
|
||||||
|
Report End();
|
||||||
|
|
||||||
|
} // namespace benchmark
|
||||||
|
} // namespace mobilegl_trace
|
||||||
@@ -3,10 +3,13 @@
|
|||||||
#include <dlfcn.h>
|
#include <dlfcn.h>
|
||||||
#include "apitrace_exit.hpp"
|
#include "apitrace_exit.hpp"
|
||||||
#include "png.h"
|
#include "png.h"
|
||||||
|
#include "trace_benchmark.hpp"
|
||||||
|
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <cerrno>
|
#include <cerrno>
|
||||||
#include <chrono>
|
#include <chrono>
|
||||||
|
#include <cmath>
|
||||||
|
#include <cstddef>
|
||||||
#include <cstdint>
|
#include <cstdint>
|
||||||
#include <cstdio>
|
#include <cstdio>
|
||||||
#include <cstdlib>
|
#include <cstdlib>
|
||||||
@@ -421,32 +424,30 @@ std::string SnapshotCallSet(const Request& request) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
int RunRetraceMain(const Request& request) {
|
int RunRetraceMain(const Request& request) {
|
||||||
std::string prefix = request.outputDir + "/actual.";
|
std::vector<std::string> args;
|
||||||
std::string callSet = SnapshotCallSet(request);
|
args.emplace_back("mobilegl-glretrace");
|
||||||
|
args.emplace_back("-b");
|
||||||
|
args.emplace_back("--singlethread");
|
||||||
|
args.emplace_back("--no-context-check");
|
||||||
|
if (!request.benchmark) {
|
||||||
|
// The snapshot callset is also what stops the replay: apitrace exits once it has
|
||||||
|
// dumped the last call in -S. Benchmark mode wants the whole trace, so the -s/-S
|
||||||
|
// pair is left off entirely, which drops the readback and the PNG encode with it.
|
||||||
|
args.emplace_back("--snapshot-alpha");
|
||||||
|
args.emplace_back("-s");
|
||||||
|
args.emplace_back(request.outputDir + "/actual.");
|
||||||
|
args.emplace_back("-S");
|
||||||
|
args.emplace_back(SnapshotCallSet(request));
|
||||||
|
}
|
||||||
|
args.emplace_back(request.tracePath);
|
||||||
|
|
||||||
std::string arg0 = "mobilegl-glretrace";
|
std::vector<char*> argv;
|
||||||
std::string argBenchmark = "-b";
|
argv.reserve(args.size() + 1);
|
||||||
std::string argSingleThread = "--singlethread";
|
for (std::string& arg : args) {
|
||||||
std::string argNoContextCheck = "--no-context-check";
|
argv.push_back(arg.data());
|
||||||
std::string argSnapshotAlpha = "--snapshot-alpha";
|
}
|
||||||
std::string argSnapshotPrefix = "-s";
|
argv.push_back(nullptr);
|
||||||
std::string argSnapshotCall = "-S";
|
return MOBILEGL_APITRACE_RETRACE_MAIN(static_cast<int>(args.size()), argv.data());
|
||||||
std::string tracePath = request.tracePath;
|
|
||||||
|
|
||||||
char* argv[] = {
|
|
||||||
arg0.data(),
|
|
||||||
argBenchmark.data(),
|
|
||||||
argSingleThread.data(),
|
|
||||||
argNoContextCheck.data(),
|
|
||||||
argSnapshotAlpha.data(),
|
|
||||||
argSnapshotPrefix.data(),
|
|
||||||
prefix.data(),
|
|
||||||
argSnapshotCall.data(),
|
|
||||||
callSet.data(),
|
|
||||||
tracePath.data(),
|
|
||||||
nullptr,
|
|
||||||
};
|
|
||||||
return MOBILEGL_APITRACE_RETRACE_MAIN(10, argv);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
bool RunRetrace(const Request& request, Result& result) {
|
bool RunRetrace(const Request& request, Result& result) {
|
||||||
@@ -475,6 +476,12 @@ bool RunRetrace(const Request& request, Result& result) {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (request.benchmark) {
|
||||||
|
// Nothing was snapshotted, so there is nothing to collect or compare here; the
|
||||||
|
// caller turns the recorded frame times into the result instead.
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
std::string snapshotPath = SnapshotPathForCall(request);
|
std::string snapshotPath = SnapshotPathForCall(request);
|
||||||
if (!Exists(snapshotPath)) {
|
if (!Exists(snapshotPath)) {
|
||||||
result.statusCode = STATUS_RETRACE_FAILED;
|
result.statusCode = STATUS_RETRACE_FAILED;
|
||||||
@@ -786,6 +793,82 @@ bool CompareWithGolden(const Request& request, Result& result) {
|
|||||||
return result.passed;
|
return result.passed;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::string BenchmarkResultPath(const Request& request) {
|
||||||
|
return request.benchmarkResultPath.empty() ? request.outputDir + "/benchmark.json"
|
||||||
|
: request.benchmarkResultPath;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Folds the recorded frame times into the headline numbers. Everything but totalSeconds and
|
||||||
|
// the frame count is computed over the trailing benchmarkTailFrames frames only.
|
||||||
|
void SummarizeBenchmark(const Request& request, const benchmark::Report& report, Result& result) {
|
||||||
|
result.benchmarkFrames = static_cast<long long>(report.frameMs.size());
|
||||||
|
result.benchmarkTotalSeconds = report.totalSeconds;
|
||||||
|
result.benchmarkTailFrames = 0;
|
||||||
|
if (report.frameMs.empty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const int requestedTail =
|
||||||
|
request.benchmarkTailFrames > 0 ? request.benchmarkTailFrames : kDefaultBenchmarkTailFrames;
|
||||||
|
const std::size_t tail =
|
||||||
|
std::min(static_cast<std::size_t>(requestedTail), report.frameMs.size());
|
||||||
|
result.benchmarkTailFrames = static_cast<int>(tail);
|
||||||
|
|
||||||
|
std::vector<double> window(report.frameMs.end() - static_cast<std::ptrdiff_t>(tail),
|
||||||
|
report.frameMs.end());
|
||||||
|
double sum = 0.0;
|
||||||
|
for (double frameMs : window) {
|
||||||
|
sum += frameMs;
|
||||||
|
}
|
||||||
|
result.benchmarkMeanMs = sum / static_cast<double>(tail);
|
||||||
|
|
||||||
|
std::sort(window.begin(), window.end());
|
||||||
|
result.benchmarkMedianMs = (tail % 2 == 1)
|
||||||
|
? window[tail / 2]
|
||||||
|
: 0.5 * (window[tail / 2 - 1] + window[tail / 2]);
|
||||||
|
// Nearest-rank p95, so the reported value is always an observed frame time.
|
||||||
|
std::size_t rank = static_cast<std::size_t>(std::ceil(0.95 * static_cast<double>(tail)));
|
||||||
|
if (rank == 0) {
|
||||||
|
rank = 1;
|
||||||
|
}
|
||||||
|
result.benchmarkP95Ms = window[rank - 1];
|
||||||
|
result.benchmarkFps = result.benchmarkMeanMs > 0.0 ? 1000.0 / result.benchmarkMeanMs : -1.0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool WriteBenchmarkJson(const Request& request,
|
||||||
|
const Result& result,
|
||||||
|
const benchmark::Report& report) {
|
||||||
|
std::ofstream file(result.benchmarkResultPath, std::ios::out | std::ios::trunc);
|
||||||
|
if (!file) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
file << "{\n";
|
||||||
|
file << " \"tracePath\": \"" << JsonEscape(request.tracePath) << "\",\n";
|
||||||
|
file << " \"backend\": \"" << JsonEscape(request.backend) << "\",\n";
|
||||||
|
file << " \"benchmarkFinish\": " << (request.benchmarkFinish ? "true" : "false") << ",\n";
|
||||||
|
file << " \"width\": " << request.width << ",\n";
|
||||||
|
file << " \"height\": " << request.height << ",\n";
|
||||||
|
file << " \"totalFrames\": " << result.benchmarkFrames << ",\n";
|
||||||
|
file << " \"tailFrames\": " << result.benchmarkTailFrames << ",\n";
|
||||||
|
file << std::fixed << std::setprecision(6);
|
||||||
|
file << " \"totalSeconds\": " << result.benchmarkTotalSeconds << ",\n";
|
||||||
|
file << std::setprecision(3);
|
||||||
|
file << " \"meanFrameMs\": " << result.benchmarkMeanMs << ",\n";
|
||||||
|
file << " \"medianFrameMs\": " << result.benchmarkMedianMs << ",\n";
|
||||||
|
file << " \"p95FrameMs\": " << result.benchmarkP95Ms << ",\n";
|
||||||
|
file << " \"fps\": " << result.benchmarkFps << ",\n";
|
||||||
|
file << " \"frameTimesMs\": [";
|
||||||
|
for (std::size_t i = 0; i < report.frameMs.size(); ++i) {
|
||||||
|
if (i > 0) {
|
||||||
|
file << ", ";
|
||||||
|
}
|
||||||
|
file << report.frameMs[i];
|
||||||
|
}
|
||||||
|
file << "]\n";
|
||||||
|
file << "}\n";
|
||||||
|
return static_cast<bool>(file);
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
extern "C" [[noreturn]] void mobilegl_apitrace_exit(int status) {
|
extern "C" [[noreturn]] void mobilegl_apitrace_exit(int status) {
|
||||||
@@ -838,7 +921,25 @@ bool WriteResultJson(const Request& request, const Result& result) {
|
|||||||
file << " \"deriveNumSubgroups\": " << (request.deriveNumSubgroups ? "true" : "false") << ",\n";
|
file << " \"deriveNumSubgroups\": " << (request.deriveNumSubgroups ? "true" : "false") << ",\n";
|
||||||
file << " \"iterationRPFixBarrier\": " << (request.iterationRPFixBarrier ? "true" : "false") << ",\n";
|
file << " \"iterationRPFixBarrier\": " << (request.iterationRPFixBarrier ? "true" : "false") << ",\n";
|
||||||
file << " \"holdMs\": " << request.holdMs << ",\n";
|
file << " \"holdMs\": " << request.holdMs << ",\n";
|
||||||
file << " \"mismatchPixels\": " << result.mismatchPixels << "\n";
|
file << " \"mismatchPixels\": " << result.mismatchPixels;
|
||||||
|
if (request.benchmark) {
|
||||||
|
// Headline numbers only; the per-frame array lives in benchmarkResultPath.
|
||||||
|
file << ",\n";
|
||||||
|
file << " \"benchmark\": true,\n";
|
||||||
|
file << " \"benchmarkResultPath\": \"" << JsonEscape(result.benchmarkResultPath) << "\",\n";
|
||||||
|
file << " \"benchmarkFinish\": " << (request.benchmarkFinish ? "true" : "false") << ",\n";
|
||||||
|
file << " \"benchmarkFrames\": " << result.benchmarkFrames << ",\n";
|
||||||
|
file << " \"benchmarkTailFrames\": " << result.benchmarkTailFrames << ",\n";
|
||||||
|
file << std::setprecision(6);
|
||||||
|
file << " \"benchmarkTotalSeconds\": " << result.benchmarkTotalSeconds << ",\n";
|
||||||
|
file << std::setprecision(3);
|
||||||
|
file << " \"benchmarkMeanFrameMs\": " << result.benchmarkMeanMs << ",\n";
|
||||||
|
file << " \"benchmarkMedianFrameMs\": " << result.benchmarkMedianMs << ",\n";
|
||||||
|
file << " \"benchmarkP95FrameMs\": " << result.benchmarkP95Ms << ",\n";
|
||||||
|
file << " \"benchmarkFps\": " << result.benchmarkFps << "\n";
|
||||||
|
} else {
|
||||||
|
file << "\n";
|
||||||
|
}
|
||||||
file << "}\n";
|
file << "}\n";
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -848,6 +949,9 @@ Result RunTraceReplay(const Request& request) {
|
|||||||
result.resultPath = request.outputDir + "/result.json";
|
result.resultPath = request.outputDir + "/result.json";
|
||||||
result.actualPath = request.outputDir + "/actual.png";
|
result.actualPath = request.outputDir + "/actual.png";
|
||||||
result.diffPath = request.diffPath;
|
result.diffPath = request.diffPath;
|
||||||
|
if (request.benchmark) {
|
||||||
|
result.benchmarkResultPath = BenchmarkResultPath(request);
|
||||||
|
}
|
||||||
const std::string mobileGlLogPath = request.outputDir + "/mobilegl.log";
|
const std::string mobileGlLogPath = request.outputDir + "/mobilegl.log";
|
||||||
|
|
||||||
if (!EnsureDirectory(request.outputDir)) {
|
if (!EnsureDirectory(request.outputDir)) {
|
||||||
@@ -868,7 +972,8 @@ Result RunTraceReplay(const Request& request) {
|
|||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (request.targetCall < 0) {
|
// Benchmark mode never snapshots, so it has no target call to stop at.
|
||||||
|
if (!request.benchmark && request.targetCall < 0) {
|
||||||
result.statusCode = STATUS_INVALID_ARGUMENT;
|
result.statusCode = STATUS_INVALID_ARGUMENT;
|
||||||
result.message = "target_call must be set for dump-images style replay";
|
result.message = "target_call must be set for dump-images style replay";
|
||||||
return result;
|
return result;
|
||||||
@@ -883,6 +988,40 @@ Result RunTraceReplay(const Request& request) {
|
|||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (request.benchmark) {
|
||||||
|
benchmark::Begin(request.benchmarkFinish);
|
||||||
|
const bool retraced = RunRetrace(request, result);
|
||||||
|
const benchmark::Report report = benchmark::End();
|
||||||
|
SummarizeBenchmark(request, report, result);
|
||||||
|
// Written even when the retrace failed: a partial timing series says where the
|
||||||
|
// replay got to, which is exactly what is wanted when triaging one.
|
||||||
|
const bool wroteJson = WriteBenchmarkJson(request, result, report);
|
||||||
|
HoldAfterRetrace(request);
|
||||||
|
if (!retraced) {
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
if (!wroteJson) {
|
||||||
|
result.statusCode = STATUS_IO_ERROR;
|
||||||
|
result.message = "benchmark completed but failed to write " + result.benchmarkResultPath;
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
// "Passed" in benchmark mode means the replay ran the trace to the end without
|
||||||
|
// error; there is no golden to be right or wrong about.
|
||||||
|
result.passed = true;
|
||||||
|
result.statusCode = STATUS_OK;
|
||||||
|
std::ostringstream message;
|
||||||
|
message << std::fixed << std::setprecision(3)
|
||||||
|
<< "benchmark completed; frames=" << result.benchmarkFrames
|
||||||
|
<< ", tailFrames=" << result.benchmarkTailFrames
|
||||||
|
<< ", meanMs=" << result.benchmarkMeanMs
|
||||||
|
<< ", medianMs=" << result.benchmarkMedianMs
|
||||||
|
<< ", p95Ms=" << result.benchmarkP95Ms
|
||||||
|
<< ", fps=" << result.benchmarkFps
|
||||||
|
<< ", benchmarkResultPath=" << result.benchmarkResultPath;
|
||||||
|
result.message = message.str();
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
if (!RunRetrace(request, result)) {
|
if (!RunRetrace(request, result)) {
|
||||||
HoldAfterRetrace(request);
|
HoldAfterRetrace(request);
|
||||||
return result;
|
return result;
|
||||||
|
|||||||
@@ -15,6 +15,8 @@ enum StatusCode {
|
|||||||
STATUS_COMPARE_FAILED = 6,
|
STATUS_COMPARE_FAILED = 6,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
constexpr int kDefaultBenchmarkTailFrames = 200;
|
||||||
|
|
||||||
struct Request {
|
struct Request {
|
||||||
std::string tracePath;
|
std::string tracePath;
|
||||||
std::string goldenPath;
|
std::string goldenPath;
|
||||||
@@ -30,6 +32,19 @@ struct Request {
|
|||||||
// Named GL_TEXTURE_2D dump points, each `CALL,TEXTURE,LEVEL,DIR`. Debug-only; the replay
|
// Named GL_TEXTURE_2D dump points, each `CALL,TEXTURE,LEVEL,DIR`. Debug-only; the replay
|
||||||
// behaves exactly as before when this is empty.
|
// behaves exactly as before when this is empty.
|
||||||
std::vector<std::string> texture2dDumps;
|
std::vector<std::string> texture2dDumps;
|
||||||
|
// Benchmark (frame-timing) mode. Off by default. When on, the replay runs the whole
|
||||||
|
// trace from start to finish and records a wall-clock timestamp at every frame
|
||||||
|
// boundary; no snapshot is taken and no golden comparison runs.
|
||||||
|
bool benchmark = false;
|
||||||
|
// Number of trailing frames the summary statistics are computed over. Clamped to the
|
||||||
|
// number of frames actually recorded. The tail is what is comparable between runs: the
|
||||||
|
// head of a trace is dominated by shader compiles and first-use uploads.
|
||||||
|
int benchmarkTailFrames = kDefaultBenchmarkTailFrames;
|
||||||
|
// glFinish through the replayed context at every frame boundary, so a frame time
|
||||||
|
// includes GPU completion instead of only CPU submission. See trace_benchmark.hpp.
|
||||||
|
bool benchmarkFinish = true;
|
||||||
|
// Where the timing JSON goes. Defaults to <outputDir>/benchmark.json.
|
||||||
|
std::string benchmarkResultPath;
|
||||||
int targetFrame = -1;
|
int targetFrame = -1;
|
||||||
long long targetCall = -1;
|
long long targetCall = -1;
|
||||||
int width = 0;
|
int width = 0;
|
||||||
@@ -60,6 +75,16 @@ struct Result {
|
|||||||
std::string matchedGoldenPath;
|
std::string matchedGoldenPath;
|
||||||
double ssim = -1.0;
|
double ssim = -1.0;
|
||||||
long long mismatchPixels = -1;
|
long long mismatchPixels = -1;
|
||||||
|
// Benchmark headline numbers. Left at the defaults below unless the request asked for
|
||||||
|
// benchmark mode; benchmarkResultPath then names the JSON with the per-frame array.
|
||||||
|
std::string benchmarkResultPath;
|
||||||
|
long long benchmarkFrames = -1;
|
||||||
|
int benchmarkTailFrames = 0;
|
||||||
|
double benchmarkTotalSeconds = -1.0;
|
||||||
|
double benchmarkMeanMs = -1.0;
|
||||||
|
double benchmarkMedianMs = -1.0;
|
||||||
|
double benchmarkP95Ms = -1.0;
|
||||||
|
double benchmarkFps = -1.0;
|
||||||
};
|
};
|
||||||
|
|
||||||
Result RunTraceReplay(const Request& request);
|
Result RunTraceReplay(const Request& request);
|
||||||
|
|||||||
@@ -125,7 +125,11 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
|
|||||||
jboolean fixIterationRPSubgroupScratch,
|
jboolean fixIterationRPSubgroupScratch,
|
||||||
jboolean deriveNumSubgroups,
|
jboolean deriveNumSubgroups,
|
||||||
jboolean iterationRPFixBarrier,
|
jboolean iterationRPFixBarrier,
|
||||||
jstring texture2dDumps) {
|
jstring texture2dDumps,
|
||||||
|
jboolean benchmarkMode,
|
||||||
|
jint benchmarkTailFrames,
|
||||||
|
jboolean benchmarkFinish,
|
||||||
|
jstring benchmarkResultPath) {
|
||||||
mobilegl_trace::Request request;
|
mobilegl_trace::Request request;
|
||||||
request.tracePath = ToString(env, tracePath);
|
request.tracePath = ToString(env, tracePath);
|
||||||
request.goldenPath = ToString(env, goldenPath);
|
request.goldenPath = ToString(env, goldenPath);
|
||||||
@@ -156,6 +160,12 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
|
|||||||
request.fixIterationRPSubgroupScratch = fixIterationRPSubgroupScratch == JNI_TRUE;
|
request.fixIterationRPSubgroupScratch = fixIterationRPSubgroupScratch == JNI_TRUE;
|
||||||
request.deriveNumSubgroups = deriveNumSubgroups == JNI_TRUE;
|
request.deriveNumSubgroups = deriveNumSubgroups == JNI_TRUE;
|
||||||
request.iterationRPFixBarrier = iterationRPFixBarrier == JNI_TRUE;
|
request.iterationRPFixBarrier = iterationRPFixBarrier == JNI_TRUE;
|
||||||
|
request.benchmark = benchmarkMode == JNI_TRUE;
|
||||||
|
request.benchmarkTailFrames = benchmarkTailFrames > 0
|
||||||
|
? benchmarkTailFrames
|
||||||
|
: mobilegl_trace::kDefaultBenchmarkTailFrames;
|
||||||
|
request.benchmarkFinish = benchmarkFinish == JNI_TRUE;
|
||||||
|
request.benchmarkResultPath = ToString(env, benchmarkResultPath);
|
||||||
|
|
||||||
ScopedTraceReplayState replayState;
|
ScopedTraceReplayState replayState;
|
||||||
mobilegl_trace_set_requested_size(request.width, request.height);
|
mobilegl_trace_set_requested_size(request.width, request.height);
|
||||||
|
|||||||
+33
-4
@@ -119,7 +119,11 @@ public final class TraceReplayActivity extends Activity {
|
|||||||
request.fixIterationRPSubgroupScratch,
|
request.fixIterationRPSubgroupScratch,
|
||||||
request.deriveNumSubgroups,
|
request.deriveNumSubgroups,
|
||||||
request.iterationRPFixBarrier,
|
request.iterationRPFixBarrier,
|
||||||
request.texture2dDumps
|
request.texture2dDumps,
|
||||||
|
request.benchmark,
|
||||||
|
request.benchmarkTailFrames,
|
||||||
|
request.benchmarkFinish,
|
||||||
|
request.benchmarkResultPath
|
||||||
);
|
);
|
||||||
Log.i(TAG, result.toString());
|
Log.i(TAG, result.toString());
|
||||||
TraceReplayResult finalResult = result;
|
TraceReplayResult finalResult = result;
|
||||||
@@ -155,7 +159,11 @@ public final class TraceReplayActivity extends Activity {
|
|||||||
boolean fixIterationRPSubgroupScratch,
|
boolean fixIterationRPSubgroupScratch,
|
||||||
boolean deriveNumSubgroups,
|
boolean deriveNumSubgroups,
|
||||||
boolean iterationRPFixBarrier,
|
boolean iterationRPFixBarrier,
|
||||||
String texture2dDumps
|
String texture2dDumps,
|
||||||
|
boolean benchmark,
|
||||||
|
int benchmarkTailFrames,
|
||||||
|
boolean benchmarkFinish,
|
||||||
|
String benchmarkResultPath
|
||||||
);
|
);
|
||||||
|
|
||||||
private static final class TraceReplayRequest {
|
private static final class TraceReplayRequest {
|
||||||
@@ -184,6 +192,13 @@ public final class TraceReplayActivity extends Activity {
|
|||||||
final boolean deriveNumSubgroups;
|
final boolean deriveNumSubgroups;
|
||||||
final boolean iterationRPFixBarrier;
|
final boolean iterationRPFixBarrier;
|
||||||
final String texture2dDumps;
|
final String texture2dDumps;
|
||||||
|
// Benchmark (frame-timing) mode. Replays the whole trace, times every frame
|
||||||
|
// boundary, and skips the snapshot and the SSIM comparison; "passed" then only
|
||||||
|
// means the replay ran to the end without error.
|
||||||
|
final boolean benchmark;
|
||||||
|
final int benchmarkTailFrames;
|
||||||
|
final boolean benchmarkFinish;
|
||||||
|
final String benchmarkResultPath;
|
||||||
|
|
||||||
private TraceReplayRequest(
|
private TraceReplayRequest(
|
||||||
String tracePath,
|
String tracePath,
|
||||||
@@ -210,7 +225,11 @@ public final class TraceReplayActivity extends Activity {
|
|||||||
boolean fixIterationRPSubgroupScratch,
|
boolean fixIterationRPSubgroupScratch,
|
||||||
boolean deriveNumSubgroups,
|
boolean deriveNumSubgroups,
|
||||||
boolean iterationRPFixBarrier,
|
boolean iterationRPFixBarrier,
|
||||||
String texture2dDumps
|
String texture2dDumps,
|
||||||
|
boolean benchmark,
|
||||||
|
int benchmarkTailFrames,
|
||||||
|
boolean benchmarkFinish,
|
||||||
|
String benchmarkResultPath
|
||||||
) {
|
) {
|
||||||
this.tracePath = tracePath;
|
this.tracePath = tracePath;
|
||||||
this.goldenPath = goldenPath;
|
this.goldenPath = goldenPath;
|
||||||
@@ -237,11 +256,17 @@ public final class TraceReplayActivity extends Activity {
|
|||||||
this.deriveNumSubgroups = deriveNumSubgroups;
|
this.deriveNumSubgroups = deriveNumSubgroups;
|
||||||
this.iterationRPFixBarrier = iterationRPFixBarrier;
|
this.iterationRPFixBarrier = iterationRPFixBarrier;
|
||||||
this.texture2dDumps = texture2dDumps;
|
this.texture2dDumps = texture2dDumps;
|
||||||
|
this.benchmark = benchmark;
|
||||||
|
this.benchmarkTailFrames = benchmarkTailFrames;
|
||||||
|
this.benchmarkFinish = benchmarkFinish;
|
||||||
|
this.benchmarkResultPath = benchmarkResultPath;
|
||||||
}
|
}
|
||||||
|
|
||||||
static TraceReplayRequest from(Intent intent, File filesDir, String defaultBackend) {
|
static TraceReplayRequest from(Intent intent, File filesDir, String defaultBackend) {
|
||||||
String outputDir = readString(intent, "output_dir", new File(filesDir, "trace-replay").getAbsolutePath());
|
String outputDir = readString(intent, "output_dir", new File(filesDir, "trace-replay").getAbsolutePath());
|
||||||
String diffPath = readString(intent, "diff_path", "");
|
String diffPath = readString(intent, "diff_path", "");
|
||||||
|
String benchmarkResultPath =
|
||||||
|
readString(intent, "benchmark_result_path", outputDir + "/benchmark.json");
|
||||||
return new TraceReplayRequest(
|
return new TraceReplayRequest(
|
||||||
readString(intent, "trace_path", ""),
|
readString(intent, "trace_path", ""),
|
||||||
readString(intent, "golden_path", ""),
|
readString(intent, "golden_path", ""),
|
||||||
@@ -267,7 +292,11 @@ public final class TraceReplayActivity extends Activity {
|
|||||||
intent.getBooleanExtra("fix_iterationrp_subgroup_scratch", false),
|
intent.getBooleanExtra("fix_iterationrp_subgroup_scratch", false),
|
||||||
intent.getBooleanExtra("derive_num_subgroups", false),
|
intent.getBooleanExtra("derive_num_subgroups", false),
|
||||||
intent.getBooleanExtra("iterationrp_fix_barrier", false),
|
intent.getBooleanExtra("iterationrp_fix_barrier", false),
|
||||||
readString(intent, "texture_2d_dumps", "")
|
readString(intent, "texture_2d_dumps", ""),
|
||||||
|
intent.getBooleanExtra("benchmark", false),
|
||||||
|
intent.getIntExtra("benchmark_tail_frames", 200),
|
||||||
|
intent.getBooleanExtra("benchmark_finish", true),
|
||||||
|
benchmarkResultPath
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -32,6 +32,10 @@ Usage:
|
|||||||
[--avoid-angle-llvmpipe-explicit-lod-bias] \
|
[--avoid-angle-llvmpipe-explicit-lod-bias] \
|
||||||
[--coherent-as-flush] \
|
[--coherent-as-flush] \
|
||||||
[--dump-texture-2d CALL,TEXTURE,LEVEL,DIR] \
|
[--dump-texture-2d CALL,TEXTURE,LEVEL,DIR] \
|
||||||
|
[--benchmark] \
|
||||||
|
[--benchmark-tail-frames N] \
|
||||||
|
[--benchmark-finish 0|1] \
|
||||||
|
[--reuse-fixture] \
|
||||||
--timeout-seconds N
|
--timeout-seconds N
|
||||||
|
|
||||||
Set MOBILEGL_USE_ANGLE=1 to run DirectGLES replay with packaged ANGLE
|
Set MOBILEGL_USE_ANGLE=1 to run DirectGLES replay with packaged ANGLE
|
||||||
@@ -50,6 +54,12 @@ sample with an explicit LOD that ANGLE llvmpipe cannot take a LOD bias on
|
|||||||
(MOBILEGL_AVOID_EXPLICIT_LOD_BIAS=1).
|
(MOBILEGL_AVOID_EXPLICIT_LOD_BIAS=1).
|
||||||
Pass --coherent-as-flush for traces whose engine writes persistent
|
Pass --coherent-as-flush for traces whose engine writes persistent
|
||||||
GL_MAP_FLUSH_EXPLICIT_BIT maps it never flushes (MOBILEGL_COHERENT_AS_FLUSH=1).
|
GL_MAP_FLUSH_EXPLICIT_BIT maps it never flushes (MOBILEGL_COHERENT_AS_FLUSH=1).
|
||||||
|
Pass --benchmark to replay the whole trace as a frame-timing benchmark instead of
|
||||||
|
snapshotting one frame and comparing it against the golden. The run then also
|
||||||
|
copies benchmark.json (per-frame times plus mean/median/p95) out of the app, and
|
||||||
|
"passed" only means the replay reached the end of the trace without an error.
|
||||||
|
Pass --reuse-fixture to skip re-extracting and re-pushing the trace, for repeat
|
||||||
|
runs of a case whose fixture is already in /data/local/tmp.
|
||||||
EOF
|
EOF
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -109,6 +119,10 @@ avoid_angle_llvmpipe_sampler_mipmap_min_filter=0
|
|||||||
avoid_angle_llvmpipe_explicit_lod_bias=0
|
avoid_angle_llvmpipe_explicit_lod_bias=0
|
||||||
coherent_as_flush=0
|
coherent_as_flush=0
|
||||||
texture_2d_dumps=""
|
texture_2d_dumps=""
|
||||||
|
benchmark=0
|
||||||
|
benchmark_tail_frames=200
|
||||||
|
benchmark_finish=1
|
||||||
|
reuse_fixture=0
|
||||||
timeout_seconds=""
|
timeout_seconds=""
|
||||||
|
|
||||||
while [ "$#" -gt 0 ]; do
|
while [ "$#" -gt 0 ]; do
|
||||||
@@ -150,6 +164,10 @@ while [ "$#" -gt 0 ]; do
|
|||||||
;;
|
;;
|
||||||
--coherent-as-flush) coherent_as_flush=1; shift 1 ;;
|
--coherent-as-flush) coherent_as_flush=1; shift 1 ;;
|
||||||
--dump-texture-2d) texture_2d_dumps="$(next_arg "$@")"; shift 2 ;;
|
--dump-texture-2d) texture_2d_dumps="$(next_arg "$@")"; shift 2 ;;
|
||||||
|
--benchmark) benchmark=1; shift 1 ;;
|
||||||
|
--benchmark-tail-frames) benchmark_tail_frames="$(next_arg "$@")"; shift 2 ;;
|
||||||
|
--benchmark-finish) benchmark_finish="$(next_arg "$@")"; shift 2 ;;
|
||||||
|
--reuse-fixture) reuse_fixture=1; shift 1 ;;
|
||||||
--timeout-seconds) timeout_seconds="$(next_arg "$@")"; shift 2 ;;
|
--timeout-seconds) timeout_seconds="$(next_arg "$@")"; shift 2 ;;
|
||||||
-h|--help) usage; exit 0 ;;
|
-h|--help) usage; exit 0 ;;
|
||||||
*) die "unknown argument: $1" ;;
|
*) die "unknown argument: $1" ;;
|
||||||
@@ -339,7 +357,11 @@ run_retrace() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
mkdir -p "${result_dir}"
|
mkdir -p "${result_dir}"
|
||||||
"${ADB}" install -r "$(host_path_for_adb "${apk_file}")"
|
# A repeat run of a case the previous invocation already installed and pushed only needs
|
||||||
|
# the on-device copy back into a fresh app output directory.
|
||||||
|
if [ "${reuse_fixture}" -eq 0 ]; then
|
||||||
|
"${ADB}" install -r "$(host_path_for_adb "${apk_file}")"
|
||||||
|
fi
|
||||||
copy_fixture_to_app
|
copy_fixture_to_app
|
||||||
adb_device_path shell am force-stop "${package_name}"
|
adb_device_path shell am force-stop "${package_name}"
|
||||||
"${ADB}" logcat -c
|
"${ADB}" logcat -c
|
||||||
@@ -378,6 +400,16 @@ run_retrace() {
|
|||||||
if [ -n "${texture_2d_dumps}" ]; then
|
if [ -n "${texture_2d_dumps}" ]; then
|
||||||
set -- "$@" --es texture_2d_dumps "${texture_2d_dumps}"
|
set -- "$@" --es texture_2d_dumps "${texture_2d_dumps}"
|
||||||
fi
|
fi
|
||||||
|
if [ "${benchmark}" -eq 1 ]; then
|
||||||
|
set -- "$@" --ez benchmark true
|
||||||
|
set -- "$@" --ei benchmark_tail_frames "${benchmark_tail_frames}"
|
||||||
|
set -- "$@" --es benchmark_result_path "${app_dir}/output/benchmark.json"
|
||||||
|
if [ "${benchmark_finish}" = "0" ]; then
|
||||||
|
set -- "$@" --ez benchmark_finish false
|
||||||
|
else
|
||||||
|
set -- "$@" --ez benchmark_finish true
|
||||||
|
fi
|
||||||
|
fi
|
||||||
set -- "$@" \
|
set -- "$@" \
|
||||||
--es output_dir "${app_dir}/output" \
|
--es output_dir "${app_dir}/output" \
|
||||||
--es diff_path "${app_dir}/output/${safe_case}-diff.png" \
|
--es diff_path "${app_dir}/output/${safe_case}-diff.png" \
|
||||||
@@ -434,10 +466,16 @@ run_retrace() {
|
|||||||
fi
|
fi
|
||||||
adb_device_path exec-out run-as "${package_name}" cat "${app_dir}/output/result.json" > "${result_dir}/result.json"
|
adb_device_path exec-out run-as "${package_name}" cat "${app_dir}/output/result.json" > "${result_dir}/result.json"
|
||||||
cat "${result_dir}/result.json"
|
cat "${result_dir}/result.json"
|
||||||
copy_app_artifact "${app_dir}/output/actual.png" "${result_dir}/${safe_case}-${backend}-actual.png"
|
# A benchmark run takes no snapshot, so there is no actual/diff pair to copy.
|
||||||
copy_app_artifact "${app_dir}/output/${safe_case}-diff.png" "${result_dir}/${safe_case}-${backend}-diff.png"
|
if [ "${benchmark}" -eq 0 ]; then
|
||||||
|
copy_app_artifact "${app_dir}/output/actual.png" "${result_dir}/${safe_case}-${backend}-actual.png"
|
||||||
|
copy_app_artifact "${app_dir}/output/${safe_case}-diff.png" "${result_dir}/${safe_case}-${backend}-diff.png"
|
||||||
|
fi
|
||||||
copy_app_artifact "${app_dir}/output/retrace.log" "${result_dir}/retrace.log"
|
copy_app_artifact "${app_dir}/output/retrace.log" "${result_dir}/retrace.log"
|
||||||
copy_app_artifact "${app_dir}/output/mobilegl.log" "${result_dir}/mobilegl.log"
|
copy_app_artifact "${app_dir}/output/mobilegl.log" "${result_dir}/mobilegl.log"
|
||||||
|
if [ "${benchmark}" -eq 1 ]; then
|
||||||
|
copy_app_artifact "${app_dir}/output/benchmark.json" "${result_dir}/benchmark.json"
|
||||||
|
fi
|
||||||
copy_texture_2d_dumps
|
copy_texture_2d_dumps
|
||||||
|
|
||||||
# A replay that wrote result.json but did not pass used to print nothing but
|
# A replay that wrote result.json but did not pass used to print nothing but
|
||||||
@@ -468,6 +506,8 @@ run_retrace() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
mkdir -p "${fixture_root}" "${result_root}"
|
mkdir -p "${fixture_root}" "${result_root}"
|
||||||
prepare_fixture
|
if [ "${reuse_fixture}" -eq 0 ]; then
|
||||||
|
prepare_fixture
|
||||||
|
fi
|
||||||
|
|
||||||
run_retrace
|
run_retrace
|
||||||
|
|||||||
@@ -31,6 +31,15 @@ android {
|
|||||||
}
|
}
|
||||||
fordebug {
|
fordebug {
|
||||||
debuggable true
|
debuggable true
|
||||||
|
// Keep the APK debuggable (simpleperf --app, JDWP) but build the
|
||||||
|
// native library like a release: AGP maps debuggable build types to
|
||||||
|
// CMAKE_BUILD_TYPE=Debug, which compiles MobileGL at -O0 and makes
|
||||||
|
// every performance measurement on this flavor meaningless.
|
||||||
|
externalNativeBuild {
|
||||||
|
cmake {
|
||||||
|
arguments "-DCMAKE_BUILD_TYPE=RelWithDebInfo"
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (standalonePluginBuild) {
|
if (standalonePluginBuild) {
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
results/
|
||||||
|
leveldat_*
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
# device_bench — in-game FPS benchmark harness
|
||||||
|
|
||||||
|
Scripted, repeatable in-game FPS measurement for MobileGL's two Android backends
|
||||||
|
(Espryt/DirectGLES and Magma/DirectVulkan) plus a MobileGlues reference run,
|
||||||
|
driven through the FCL fordebug flavor. Intended for A/B performance work and
|
||||||
|
release regression gates on real devices.
|
||||||
|
|
||||||
|
## How it measures
|
||||||
|
|
||||||
|
FCL's in-game FPS overlay counts `eglSwapBuffers` calls natively (renderer-
|
||||||
|
agnostic, not vsync-capped when the game runs with vsync off). When the overlay
|
||||||
|
is enabled, FCL's FPS thread logs one `FCLFPS: <n>` logcat line per second;
|
||||||
|
`bench.sh` collects those lines during the measurement window and reports
|
||||||
|
mean / median / min / max / stdev, alongside GPU busy%, SoC temperature, and
|
||||||
|
frequency-pin integrity.
|
||||||
|
|
||||||
|
## One-time setup (per device / world)
|
||||||
|
|
||||||
|
1. Install the FCL **fordebug** flavor (`com.tungsten.fcl.mgdebug.debug`). Its
|
||||||
|
splash auto-launches the selected profile into the prepared world after a 5 s
|
||||||
|
countdown.
|
||||||
|
2. In-game menu: enable **show FPS** (persists in `files/menu_setting.json`).
|
||||||
|
3. Prepare the benchmark world: fixed camera position, gamerules
|
||||||
|
`doMobSpawning/doDaylightCycle/doWeatherCycle=false`, then save & quit once.
|
||||||
|
`bench.sh` always `am force-stop`s the game (never saves), so every run
|
||||||
|
replays the same state.
|
||||||
|
4. `options.txt`: desired `renderDistance`, `enableVsync:false`, high `maxFps`,
|
||||||
|
`inactivityFpsLimit:"minimized"` (the "afk" default locks 30 fps after 60 s
|
||||||
|
without input and ruins the window).
|
||||||
|
5. Root required (frequency pinning, GPU busy sampling).
|
||||||
|
6. Write a device profile under `devices/` (see `devices/odinlite.env`).
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
```
|
||||||
|
./bench.sh --device devices/odinlite.env --backend magma # 30 samples, 180 s warmup
|
||||||
|
./bench.sh --device devices/odinlite.env --backend espryt --label after-fix-X
|
||||||
|
./bench.sh --device devices/odinlite.env --backend mobileglues # reference
|
||||||
|
```
|
||||||
|
|
||||||
|
Results append to `results/results.jsonl`; per-run screenshots (`pre.png`,
|
||||||
|
`post.png`) land in `results/<timestamp>-<backend>[-label]/` — always eyeball
|
||||||
|
them: the pre/post pair must show the same scene, or the run is invalid.
|
||||||
|
|
||||||
|
## Protocol discipline (hard-won, do not skip)
|
||||||
|
|
||||||
|
- **Thermal gate**: the script waits for the profile's start-temperature
|
||||||
|
threshold. Runs started hot are not comparable to runs started cool.
|
||||||
|
- **Warmup 180 s**: ART JIT takes ~3 min to plateau (62→67→84 fps ramp was
|
||||||
|
measured); short warmups underestimate by 10-20%.
|
||||||
|
- **Pins can be overridden by the thermal engine.** The result JSON records
|
||||||
|
`big_cur/little_cur/gpu_cur_khz` sampled at window end — discard the run if
|
||||||
|
they do not match the profile pins.
|
||||||
|
- **Paired runs**: absolute FPS drifts across sessions (camera angle, world
|
||||||
|
state). A/B comparisons must be back-to-back runs in the same session.
|
||||||
|
- **F3 off** for standard numbers (the F3 debug overlay multiplies per-draw
|
||||||
|
overhead and skews backends differently).
|
||||||
|
- The FPS overlay itself must be ON (it is what produces the FCLFPS lines).
|
||||||
Executable
+215
@@ -0,0 +1,215 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# In-game FPS benchmark for MobileGL on a real device, driven through FCL.
|
||||||
|
#
|
||||||
|
# Prerequisites (one-time, manual):
|
||||||
|
# - FCL fordebug flavor installed (com.tungsten.fcl.mgdebug.debug); its splash
|
||||||
|
# screen auto-launches the selected profile/version into the prepared world.
|
||||||
|
# - FCL in-game menu "show FPS" toggle enabled (menu_setting.json showFps=true):
|
||||||
|
# the FPS overlay thread logs one "FCLFPS: <n>" logcat line per second.
|
||||||
|
# - The benchmark world saved with a deterministic state (mob spawning /
|
||||||
|
# daylight cycle / weather gamerules off) and the desired options.txt
|
||||||
|
# (renderDistance, vsync off, maxFps high).
|
||||||
|
# - Rooted device (frequency pinning + GPU utilization sampling).
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# bench.sh --device devices/odinlite.env --backend magma [--samples 30]
|
||||||
|
# [--warmup 180] [--label mylabel] [--no-pin]
|
||||||
|
# backend: magma | espryt | mobileglues (reference)
|
||||||
|
#
|
||||||
|
# Output: one JSON line on stdout (also appended to results/results.jsonl) with
|
||||||
|
# mean/median/min/max FPS, GPU busy%, temperatures, and pin-integrity flags.
|
||||||
|
# Screenshots (pre/post measurement) land in results/<timestamp>-<label>/.
|
||||||
|
|
||||||
|
set -u -o pipefail
|
||||||
|
cd "$(dirname "$0")"
|
||||||
|
# Git Bash: stop MSYS from rewriting /sys/... arguments into C:/Program Files/...
|
||||||
|
export MSYS_NO_PATHCONV=1 MSYS2_ARG_CONV_EXCL='*'
|
||||||
|
|
||||||
|
PKG=com.tungsten.fcl.mgdebug.debug
|
||||||
|
ACTIVITY=$PKG/com.tungsten.fcl.activity.SplashActivity
|
||||||
|
RENDERER_ESPRYT=5e273ee2-baca-4c81-8e48-b63feefb9ba8
|
||||||
|
RENDERER_MAGMA=2be0dc10-1eef-4ce2-b512-b266dd33fd9e
|
||||||
|
RENDERER_MOBILEGLUES=com.fcl.plugin.mobileglues
|
||||||
|
|
||||||
|
DEVICE_ENV=""
|
||||||
|
BACKEND=""
|
||||||
|
SAMPLES=30
|
||||||
|
WARMUP=180
|
||||||
|
LABEL=""
|
||||||
|
DO_PIN=1
|
||||||
|
WORLD_LOAD_TIMEOUT=420
|
||||||
|
|
||||||
|
while [ $# -gt 0 ]; do
|
||||||
|
case "$1" in
|
||||||
|
--device) DEVICE_ENV=$2; shift 2 ;;
|
||||||
|
--backend) BACKEND=$2; shift 2 ;;
|
||||||
|
--samples) SAMPLES=$2; shift 2 ;;
|
||||||
|
--warmup) WARMUP=$2; shift 2 ;;
|
||||||
|
--label) LABEL=$2; shift 2 ;;
|
||||||
|
--no-pin) DO_PIN=0; shift ;;
|
||||||
|
*) echo "unknown arg: $1" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
[ -n "$DEVICE_ENV" ] && [ -n "$BACKEND" ] || { echo "need --device and --backend" >&2; exit 2; }
|
||||||
|
# shellcheck disable=SC1090
|
||||||
|
. "$DEVICE_ENV"
|
||||||
|
|
||||||
|
case "$BACKEND" in
|
||||||
|
espryt) RENDERER=$RENDERER_ESPRYT ;;
|
||||||
|
magma) RENDERER=$RENDERER_MAGMA ;;
|
||||||
|
mobileglues) RENDERER=$RENDERER_MOBILEGLUES ;;
|
||||||
|
*) echo "unknown backend: $BACKEND" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
ADB="adb -s $DEVICE_SERIAL"
|
||||||
|
STAMP=$(date +%Y%m%d-%H%M%S)
|
||||||
|
RUNLABEL="${STAMP}-${BACKEND}${LABEL:+-$LABEL}"
|
||||||
|
OUTDIR="results/$RUNLABEL"
|
||||||
|
mkdir -p "$OUTDIR"
|
||||||
|
|
||||||
|
log() { echo "[bench] $*" >&2; }
|
||||||
|
# Quote the whole su invocation for the DEVICE shell, or redirects run unprivileged.
|
||||||
|
sushell() { $ADB shell "su -c '$*'"; }
|
||||||
|
|
||||||
|
read_temp() {
|
||||||
|
$ADB shell "for tz in /sys/class/thermal/thermal_zone*; do
|
||||||
|
if [ \"\$(cat \$tz/type)\" = \"$THERMAL_ZONE_TYPE\" ]; then cat \$tz/temp; break; fi; done" | tr -d '\r'
|
||||||
|
}
|
||||||
|
|
||||||
|
# MTK: plain cpufreq sysfs writes are reverted by the vendor boost/PowerHAL within
|
||||||
|
# seconds — pin through ppm hard_userlimit instead (cluster indices: 0=little, 1=big).
|
||||||
|
pin_freqs() {
|
||||||
|
log "pinning CPU big=$CPU_BIG_FREQ little=$CPU_LITTLE_FREQ gpu=${GPU_PIN_KHZ}kHz (ppm)"
|
||||||
|
sushell "echo 1 $CPU_BIG_FREQ > /proc/ppm/policy/hard_userlimit_max_cpu_freq;
|
||||||
|
echo 1 $CPU_BIG_FREQ > /proc/ppm/policy/hard_userlimit_min_cpu_freq;
|
||||||
|
echo 0 $CPU_LITTLE_FREQ > /proc/ppm/policy/hard_userlimit_max_cpu_freq;
|
||||||
|
echo 0 $CPU_LITTLE_FREQ > /proc/ppm/policy/hard_userlimit_min_cpu_freq" >/dev/null
|
||||||
|
sushell "echo $GPU_PIN_KHZ > /proc/gpufreq/gpufreq_opp_freq" >/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
unpin_freqs() {
|
||||||
|
log "unpinning frequencies (restore DVFS)"
|
||||||
|
sushell "echo 1 -1 > /proc/ppm/policy/hard_userlimit_max_cpu_freq;
|
||||||
|
echo 1 -1 > /proc/ppm/policy/hard_userlimit_min_cpu_freq;
|
||||||
|
echo 0 -1 > /proc/ppm/policy/hard_userlimit_max_cpu_freq;
|
||||||
|
echo 0 -1 > /proc/ppm/policy/hard_userlimit_min_cpu_freq" >/dev/null
|
||||||
|
sushell "echo 0 > /proc/gpufreq/gpufreq_opp_freq" >/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
$ADB shell am force-stop $PKG >/dev/null 2>&1
|
||||||
|
[ "$DO_PIN" = 1 ] && unpin_freqs
|
||||||
|
$ADB shell svc power stayon false >/dev/null 2>&1
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
|
||||||
|
# --- 1. Wake, unlock, keep screen on, fan to sport ---------------------------
|
||||||
|
$ADB shell input keyevent KEYCODE_WAKEUP >/dev/null
|
||||||
|
$ADB shell input keyevent 82 >/dev/null
|
||||||
|
$ADB shell svc power stayon true >/dev/null
|
||||||
|
$ADB shell settings put global fan_mode 3 2>/dev/null
|
||||||
|
wakefulness=$($ADB shell dumpsys power | grep -o 'mWakefulness=[A-Za-z]*' | head -1)
|
||||||
|
log "screen: $wakefulness"
|
||||||
|
|
||||||
|
# --- 2. Thermal gate ----------------------------------------------------------
|
||||||
|
log "thermal gate: waiting for $THERMAL_ZONE_TYPE <= $THERMAL_START_MAX_MC"
|
||||||
|
for i in $(seq 1 60); do
|
||||||
|
T=$(read_temp)
|
||||||
|
[ "$T" -le "$THERMAL_START_MAX_MC" ] && break
|
||||||
|
log " temp=$T, cooling... ($i)"
|
||||||
|
sleep 10
|
||||||
|
done
|
||||||
|
TEMP_START=$(read_temp)
|
||||||
|
log "start temp: $TEMP_START"
|
||||||
|
|
||||||
|
# --- 3. Select renderer (device-side sed, proven under run-as) ---------------
|
||||||
|
for known in $RENDERER_ESPRYT $RENDERER_MAGMA $RENDERER_MOBILEGLUES; do
|
||||||
|
[ "$known" = "$RENDERER" ] && continue
|
||||||
|
$ADB shell run-as $PKG sed -i "s/$known/$RENDERER/g" files/config.json
|
||||||
|
done
|
||||||
|
log "renderer now: $($ADB shell run-as $PKG grep renderer files/config.json | tr -d '\r' | tr -s ' ' | sort -u | tr '\n' ' ')"
|
||||||
|
|
||||||
|
# --- 4. Pin frequencies -------------------------------------------------------
|
||||||
|
[ "$DO_PIN" = 1 ] && pin_freqs
|
||||||
|
|
||||||
|
# --- 5. Launch, wait for world (retry: the JVM occasionally dies with
|
||||||
|
# "exited due to signal 34" right after JLI_Launch on this device) -------------
|
||||||
|
MCLOG="/storage/emulated/0/FCL/.minecraft/versions/*/logs/latest.log"
|
||||||
|
WORLD_UP=0
|
||||||
|
for attempt in 1 2 3; do
|
||||||
|
$ADB shell am force-stop $PKG
|
||||||
|
sleep 5
|
||||||
|
$ADB shell "rm -f $MCLOG /sdcard/MG/latest.log" 2>/dev/null
|
||||||
|
$ADB logcat -c 2>/dev/null
|
||||||
|
log "launching $ACTIVITY (attempt $attempt)"
|
||||||
|
$ADB shell am start -n $ACTIVITY >/dev/null
|
||||||
|
for i in $(seq 1 $((WORLD_LOAD_TIMEOUT / 5))); do
|
||||||
|
sleep 5
|
||||||
|
if $ADB shell "grep -l -e 'logged in with entity id' -e 'Preparing spawn area: 100' $MCLOG" >/dev/null 2>&1; then
|
||||||
|
WORLD_UP=1; break
|
||||||
|
fi
|
||||||
|
if ! $ADB shell pidof $PKG >/dev/null; then
|
||||||
|
sleep 3
|
||||||
|
if ! $ADB shell pidof $PKG >/dev/null; then
|
||||||
|
log "attempt $attempt: game process died; retrying"
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
[ "$WORLD_UP" = 1 ] && break
|
||||||
|
done
|
||||||
|
if [ "$WORLD_UP" != 1 ]; then
|
||||||
|
log "world did not load within ${WORLD_LOAD_TIMEOUT}s"
|
||||||
|
$ADB exec-out screencap -p > "$OUTDIR/failed-load.png" 2>/dev/null
|
||||||
|
echo "{\"label\":\"$RUNLABEL\",\"error\":\"world-load-timeout\"}"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
log "world is up; warmup ${WARMUP}s"
|
||||||
|
sleep "$WARMUP"
|
||||||
|
|
||||||
|
# --- 6. Measure ---------------------------------------------------------------
|
||||||
|
$ADB exec-out screencap -p > "$OUTDIR/pre.png" 2>/dev/null
|
||||||
|
TEMP_MID=$(read_temp)
|
||||||
|
GPU_BUSY_SAMPLES=""
|
||||||
|
FPS_FILE="$OUTDIR/fps.txt"
|
||||||
|
: > "$FPS_FILE"
|
||||||
|
log "sampling $SAMPLES FPS values (1/s) + GPU busy"
|
||||||
|
$ADB logcat -v raw -s FCLFPS:I > "$OUTDIR/fclfps.log" &
|
||||||
|
LOGCAT_PID=$!
|
||||||
|
for i in $(seq 1 "$SAMPLES"); do
|
||||||
|
sleep 1
|
||||||
|
B=$(sushell "cat $GPU_UTIL_NODE" | tr -d '\r' | awk '{print $1}')
|
||||||
|
GPU_BUSY_SAMPLES="$GPU_BUSY_SAMPLES $B"
|
||||||
|
done
|
||||||
|
kill $LOGCAT_PID 2>/dev/null
|
||||||
|
wait $LOGCAT_PID 2>/dev/null
|
||||||
|
grep -E '^[0-9]+$' "$OUTDIR/fclfps.log" | tail -n "$SAMPLES" > "$FPS_FILE"
|
||||||
|
$ADB exec-out screencap -p > "$OUTDIR/post.png" 2>/dev/null
|
||||||
|
TEMP_END=$(read_temp)
|
||||||
|
|
||||||
|
# Pin integrity: sample live freqs right at the end of the window (game still hot).
|
||||||
|
BIG_CUR=$($ADB shell cat /sys/devices/system/cpu/cpufreq/$CPU_BIG_POLICY/scaling_cur_freq | tr -d '\r')
|
||||||
|
LITTLE_CUR=$($ADB shell cat /sys/devices/system/cpu/cpufreq/$CPU_LITTLE_POLICY/scaling_cur_freq | tr -d '\r')
|
||||||
|
GPU_CUR=$(sushell "cat $GPU_CURFREQ_NODE" | tr -d '\r' | awk '{print $NF}')
|
||||||
|
|
||||||
|
# --- 7. Stats -----------------------------------------------------------------
|
||||||
|
STATS=$(sort -n "$FPS_FILE" | awk '
|
||||||
|
{ v[NR]=$1; s+=$1 }
|
||||||
|
END {
|
||||||
|
if (NR==0) { print "0 0 0 0 0 0"; exit }
|
||||||
|
mean=s/NR; med=v[int((NR+1)/2)];
|
||||||
|
for(i=1;i<=NR;i++) ss+=(v[i]-mean)^2;
|
||||||
|
sd=(NR>1)?sqrt(ss/(NR-1)):0;
|
||||||
|
printf "%d %.1f %d %d %d %.1f", NR, mean, med, v[1], v[NR], sd
|
||||||
|
}')
|
||||||
|
set -- $STATS
|
||||||
|
N=$1 MEAN=$2 MED=$3 MIN=$4 MAX=$5 SD=$6
|
||||||
|
GPU_BUSY_MEAN=$(echo "$GPU_BUSY_SAMPLES" | tr ' ' '\n' | grep -E '^[0-9]+$' | awk '{s+=$1;n++} END{if(n) printf "%.0f", s/n; else print 0}')
|
||||||
|
|
||||||
|
RESULT=$(printf '{"label":"%s","backend":"%s","samples":%s,"fps_mean":%s,"fps_median":%s,"fps_min":%s,"fps_max":%s,"fps_sd":%s,"gpu_busy_mean":%s,"temp_start_mc":%s,"temp_mid_mc":%s,"temp_end_mc":%s,"big_cur":%s,"little_cur":%s,"gpu_cur_khz":%s,"pinned":%s,"warmup_s":%s}' \
|
||||||
|
"$RUNLABEL" "$BACKEND" "$N" "$MEAN" "$MED" "$MIN" "$MAX" "$SD" "$GPU_BUSY_MEAN" \
|
||||||
|
"$TEMP_START" "$TEMP_MID" "$TEMP_END" "$BIG_CUR" "$LITTLE_CUR" "${GPU_CUR:-0}" "$DO_PIN" "$WARMUP")
|
||||||
|
mkdir -p results
|
||||||
|
echo "$RESULT" >> results/results.jsonl
|
||||||
|
echo "$RESULT"
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
# Device profile: AYN Odin Lite (MT6877 Dimensity 900, Mali-G68 MC4, Android 11)
|
||||||
|
# 2x A78 (policy6, max 2400000) + 6x A55 (policy0, max 2000000), GPU top OPP 902MHz.
|
||||||
|
# Panel: 1080x1920 @ 60Hz (presented FPS caps at 60 - render-side FPS comes from the
|
||||||
|
# FCLFPS logcat tag, which counts eglSwapBuffers; it is NOT vsync-capped when the
|
||||||
|
# game runs with vsync off).
|
||||||
|
DEVICE_SERIAL=MTK0002207301023500
|
||||||
|
|
||||||
|
# Frequency pins, enforced via /proc/ppm/policy/hard_userlimit_* (plain cpufreq
|
||||||
|
# sysfs writes are reverted by the vendor game boost within seconds). The stock
|
||||||
|
# performance_mode=2 boost pins big=2400000/little=2000000 anyway, so we pin at
|
||||||
|
# the same values; bench.sh records live freqs at window end to catch thermal
|
||||||
|
# clamping. Fan must be in sport mode (settings put global fan_mode 3) or the
|
||||||
|
# SoC runs away past 80C.
|
||||||
|
CPU_BIG_POLICY=policy6
|
||||||
|
CPU_BIG_FREQ=2400000
|
||||||
|
CPU_LITTLE_POLICY=policy0
|
||||||
|
CPU_LITTLE_FREQ=2000000
|
||||||
|
|
||||||
|
# GPU pin via legacy MTK gpufreq: echo <khz> > /proc/gpufreq/gpufreq_opp_freq
|
||||||
|
# (0 restores DVFS). Top OPP on this device is 902000.
|
||||||
|
GPU_PIN_KHZ=902000
|
||||||
|
|
||||||
|
# Thermal gate: wait until this zone is at or below the threshold before starting.
|
||||||
|
THERMAL_ZONE_TYPE=mtktscpu
|
||||||
|
THERMAL_START_MAX_MC=50000
|
||||||
|
|
||||||
|
# MTK ged GPU utilization node; first field of the triple is busy %.
|
||||||
|
GPU_UTIL_NODE=/sys/kernel/ged/hal/gpu_utilization
|
||||||
|
GPU_CURFREQ_NODE=/sys/kernel/ged/hal/current_freqency
|
||||||
Executable
+70
@@ -0,0 +1,70 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# CPU-profile the running game with simpleperf (DWARF call graphs) and produce
|
||||||
|
# a symbolized report on the host. Run while the game is in-world (e.g. during
|
||||||
|
# a bench.sh warmup, or standalone after launching the game manually).
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# profile.sh --device devices/odinlite.env [--duration 30] [--label hot1]
|
||||||
|
# [--freq 800]
|
||||||
|
#
|
||||||
|
# Requires: debuggable app (fordebug flavor), host NDK simpleperf, and the
|
||||||
|
# unstripped libMobileGL.so from the same build as the installed APK
|
||||||
|
# (MobileGL/build/intermediates/merged_native_libs/fordebug/mergeFordebugNativeLibs/out/lib/arm64-v8a).
|
||||||
|
#
|
||||||
|
# Notes carried over from earlier campaigns:
|
||||||
|
# - DWARF unwinding (-g): frame-pointer call graphs are broken on these builds.
|
||||||
|
# - The MC render thread is a JVM thread with a generic name (Thread-NN, varies
|
||||||
|
# per run) — find it in the report with --sort comm first.
|
||||||
|
# - Use Git Bash, not PowerShell (binary pull corruption via > redirect).
|
||||||
|
|
||||||
|
set -u -o pipefail
|
||||||
|
cd "$(dirname "$0")"
|
||||||
|
# Git Bash: stop MSYS from rewriting /data/... arguments into C:/Program Files/...
|
||||||
|
export MSYS_NO_PATHCONV=1 MSYS2_ARG_CONV_EXCL='*'
|
||||||
|
|
||||||
|
PKG=com.tungsten.fcl.mgdebug.debug
|
||||||
|
DEVICE_ENV=""
|
||||||
|
DURATION=30
|
||||||
|
FREQ=800
|
||||||
|
LABEL=prof
|
||||||
|
|
||||||
|
while [ $# -gt 0 ]; do
|
||||||
|
case "$1" in
|
||||||
|
--device) DEVICE_ENV=$2; shift 2 ;;
|
||||||
|
--duration) DURATION=$2; shift 2 ;;
|
||||||
|
--freq) FREQ=$2; shift 2 ;;
|
||||||
|
--label) LABEL=$2; shift 2 ;;
|
||||||
|
*) echo "unknown arg: $1" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
[ -n "$DEVICE_ENV" ] || { echo "need --device" >&2; exit 2; }
|
||||||
|
# shellcheck disable=SC1090
|
||||||
|
. "$DEVICE_ENV"
|
||||||
|
|
||||||
|
ADB="adb -s $DEVICE_SERIAL"
|
||||||
|
STAMP=$(date +%Y%m%d-%H%M%S)
|
||||||
|
OUTDIR="results/${STAMP}-${LABEL}"
|
||||||
|
mkdir -p "$OUTDIR"
|
||||||
|
|
||||||
|
$ADB shell pidof $PKG >/dev/null || { echo "game not running" >&2; exit 1; }
|
||||||
|
|
||||||
|
echo "[profile] recording ${DURATION}s @ ${FREQ}Hz (DWARF)..." >&2
|
||||||
|
$ADB shell simpleperf record --app $PKG -e cpu-clock -f "$FREQ" -g \
|
||||||
|
--duration "$DURATION" -o /data/local/tmp/mgprof.data || exit 1
|
||||||
|
(cd "$OUTDIR" && MSYS_NO_PATHCONV=1 adb -s "$DEVICE_SERIAL" pull /data/local/tmp/mgprof.data perf.data >/dev/null)
|
||||||
|
echo "[profile] pulled to $OUTDIR/perf.data" >&2
|
||||||
|
|
||||||
|
# Host-side symbolization if the NDK simpleperf scripts are available.
|
||||||
|
SIMPLEPERF_DIR="${SIMPLEPERF_DIR:-$LOCALAPPDATA/Android/Sdk/ndk/28.2.13676358/simpleperf}"
|
||||||
|
SYMDIR="../../build/intermediates/merged_native_libs/fordebug/mergeFordebugNativeLibs/out/lib/arm64-v8a"
|
||||||
|
if [ -d "$SIMPLEPERF_DIR" ] && [ -d "$SYMDIR" ]; then
|
||||||
|
echo "[profile] building binary cache (symbolized)..." >&2
|
||||||
|
(cd "$OUTDIR" && python "$SIMPLEPERF_DIR/binary_cache_builder.py" -i perf.data -lib "../../$SYMDIR" >/dev/null 2>&1)
|
||||||
|
echo "[profile] per-thread summary:" >&2
|
||||||
|
"$SIMPLEPERF_DIR/bin/windows/x86_64/simpleperf.exe" report -i "$OUTDIR/perf.data" \
|
||||||
|
--symfs "$OUTDIR/binary_cache" --sort comm -n 2>/dev/null | head -25
|
||||||
|
echo "[profile] done; drill down with:" >&2
|
||||||
|
echo " $SIMPLEPERF_DIR/bin/windows/x86_64/simpleperf.exe report -i $OUTDIR/perf.data --symfs $OUTDIR/binary_cache --comms <renderthread> --sort symbol -n | head -40" >&2
|
||||||
|
else
|
||||||
|
echo "[profile] NDK simpleperf or symbol dir missing; raw perf.data kept at $OUTDIR" >&2
|
||||||
|
fi
|
||||||
Executable
+92
@@ -0,0 +1,92 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Launch the game into the benchmark world and leave it running (for profiling
|
||||||
|
# or ad-hoc probing). Handles the flaky-launch failure mode seen on Odin Lite:
|
||||||
|
# the JVM occasionally dies with "exited due to signal 34" right after
|
||||||
|
# JLI_Launch (~40% of launches, cause unknown) — so launching retries up to
|
||||||
|
# --retries times before giving up.
|
||||||
|
#
|
||||||
|
# Usage: session.sh --device devices/odinlite.env [--backend magma|espryt|mobileglues]
|
||||||
|
# [--settle 150] [--retries 3] [--no-pin]
|
||||||
|
# Exits 0 with the game in-world (after settle seconds), 1 otherwise.
|
||||||
|
# NOTE: leaves the game running AND the frequency pins active (that is the
|
||||||
|
# point of a session). When done: am force-stop the game and unpin via
|
||||||
|
# `echo <cluster> -1 > /proc/ppm/policy/hard_userlimit_{min,max}_cpu_freq`
|
||||||
|
# and `echo 0 > /proc/gpufreq/gpufreq_opp_freq` (or run a bench.sh, whose
|
||||||
|
# cleanup unpins).
|
||||||
|
|
||||||
|
set -u -o pipefail
|
||||||
|
cd "$(dirname "$0")"
|
||||||
|
export MSYS_NO_PATHCONV=1 MSYS2_ARG_CONV_EXCL='*'
|
||||||
|
|
||||||
|
PKG=com.tungsten.fcl.mgdebug.debug
|
||||||
|
ACTIVITY=$PKG/com.tungsten.fcl.activity.SplashActivity
|
||||||
|
RENDERER_ESPRYT=5e273ee2-baca-4c81-8e48-b63feefb9ba8
|
||||||
|
RENDERER_MAGMA=2be0dc10-1eef-4ce2-b512-b266dd33fd9e
|
||||||
|
RENDERER_MOBILEGLUES=com.fcl.plugin.mobileglues
|
||||||
|
|
||||||
|
DEVICE_ENV="" BACKEND="" SETTLE=150 RETRIES=3 DO_PIN=1
|
||||||
|
while [ $# -gt 0 ]; do
|
||||||
|
case "$1" in
|
||||||
|
--device) DEVICE_ENV=$2; shift 2 ;;
|
||||||
|
--backend) BACKEND=$2; shift 2 ;;
|
||||||
|
--settle) SETTLE=$2; shift 2 ;;
|
||||||
|
--retries) RETRIES=$2; shift 2 ;;
|
||||||
|
--no-pin) DO_PIN=0; shift ;;
|
||||||
|
*) echo "unknown arg: $1" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
[ -n "$DEVICE_ENV" ] || { echo "need --device" >&2; exit 2; }
|
||||||
|
# shellcheck disable=SC1090
|
||||||
|
. "$DEVICE_ENV"
|
||||||
|
ADB="adb -s $DEVICE_SERIAL"
|
||||||
|
log() { echo "[session] $*" >&2; }
|
||||||
|
|
||||||
|
if [ -n "$BACKEND" ]; then
|
||||||
|
case "$BACKEND" in
|
||||||
|
espryt) RENDERER=$RENDERER_ESPRYT ;;
|
||||||
|
magma) RENDERER=$RENDERER_MAGMA ;;
|
||||||
|
mobileglues) RENDERER=$RENDERER_MOBILEGLUES ;;
|
||||||
|
*) echo "unknown backend: $BACKEND" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
for known in $RENDERER_ESPRYT $RENDERER_MAGMA $RENDERER_MOBILEGLUES; do
|
||||||
|
[ "$known" = "$RENDERER" ] && continue
|
||||||
|
$ADB shell run-as $PKG sed -i "s/$known/$RENDERER/g" files/config.json
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
|
||||||
|
$ADB shell "input keyevent KEYCODE_WAKEUP; sleep 1; input keyevent 82; svc power stayon true; settings put global fan_mode 3"
|
||||||
|
if [ "$DO_PIN" = 1 ]; then
|
||||||
|
$ADB shell "su -c 'echo 1 $CPU_BIG_FREQ > /proc/ppm/policy/hard_userlimit_max_cpu_freq;
|
||||||
|
echo 1 $CPU_BIG_FREQ > /proc/ppm/policy/hard_userlimit_min_cpu_freq;
|
||||||
|
echo 0 $CPU_LITTLE_FREQ > /proc/ppm/policy/hard_userlimit_max_cpu_freq;
|
||||||
|
echo 0 $CPU_LITTLE_FREQ > /proc/ppm/policy/hard_userlimit_min_cpu_freq;
|
||||||
|
echo $GPU_PIN_KHZ > /proc/gpufreq/gpufreq_opp_freq'"
|
||||||
|
fi
|
||||||
|
|
||||||
|
MCLOG=/storage/emulated/0/FCL/.minecraft/versions/1.21.4/logs/latest.log
|
||||||
|
for attempt in $(seq 1 "$RETRIES"); do
|
||||||
|
$ADB shell am force-stop $PKG
|
||||||
|
sleep 5
|
||||||
|
$ADB shell "rm -f $MCLOG"
|
||||||
|
$ADB shell am start -n $ACTIVITY >/dev/null
|
||||||
|
log "attempt $attempt: launched"
|
||||||
|
DEADLINE=$((SECONDS + 300))
|
||||||
|
while [ $SECONDS -lt $DEADLINE ]; do
|
||||||
|
sleep 5
|
||||||
|
if $ADB shell "grep -q 'joined the game' $MCLOG" 2>/dev/null; then
|
||||||
|
log "world up (attempt $attempt); settling ${SETTLE}s"
|
||||||
|
sleep "$SETTLE"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
if ! $ADB shell pidof $PKG >/dev/null; then
|
||||||
|
# process may still be between splash and JVM start briefly; confirm twice
|
||||||
|
sleep 3
|
||||||
|
if ! $ADB shell pidof $PKG >/dev/null; then
|
||||||
|
log "attempt $attempt: process died (signal-34 flake?); retrying"
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
done
|
||||||
|
log "no world after $RETRIES attempts"
|
||||||
|
exit 1
|
||||||
@@ -225,7 +225,8 @@ add_library(mobilegl_trace_glretrace_common STATIC
|
|||||||
"${APITRACE_ROOT}/retrace/metric_helper.cpp"
|
"${APITRACE_ROOT}/retrace/metric_helper.cpp"
|
||||||
"${APITRACE_ROOT}/retrace/metric_writer.cpp"
|
"${APITRACE_ROOT}/retrace/metric_writer.cpp"
|
||||||
"${MOBILEGL_TRACE_ROOT}/apitrace_fbo_dump.cpp"
|
"${MOBILEGL_TRACE_ROOT}/apitrace_fbo_dump.cpp"
|
||||||
"${MOBILEGL_TRACE_ROOT}/apitrace_glws_egl.cpp")
|
"${MOBILEGL_TRACE_ROOT}/apitrace_glws_egl.cpp"
|
||||||
|
"${MOBILEGL_TRACE_SHARED_CPP_DIR}/trace_benchmark.cpp")
|
||||||
if(APPLE)
|
if(APPLE)
|
||||||
set(MOBILEGL_TRACE_APPLE_FRAMEWORKS
|
set(MOBILEGL_TRACE_APPLE_FRAMEWORKS
|
||||||
"-framework Cocoa"
|
"-framework Cocoa"
|
||||||
@@ -239,7 +240,8 @@ target_include_directories(mobilegl_trace_glretrace_common PUBLIC
|
|||||||
"${APITRACE_GENERATED_DIR}"
|
"${APITRACE_GENERATED_DIR}"
|
||||||
"${APITRACE_ROOT}/dispatch"
|
"${APITRACE_ROOT}/dispatch"
|
||||||
"${APITRACE_ROOT}/helpers"
|
"${APITRACE_ROOT}/helpers"
|
||||||
"${APITRACE_ROOT}/retrace")
|
"${APITRACE_ROOT}/retrace"
|
||||||
|
"${MOBILEGL_TRACE_SHARED_CPP_DIR}")
|
||||||
target_compile_definitions(mobilegl_trace_glretrace_common PRIVATE
|
target_compile_definitions(mobilegl_trace_glretrace_common PRIVATE
|
||||||
exit=mobilegl_apitrace_exit
|
exit=mobilegl_apitrace_exit
|
||||||
main=mobilegl_apitrace_main)
|
main=mobilegl_apitrace_main)
|
||||||
|
|||||||
@@ -17,6 +17,12 @@ The bundled fixtures cover:
|
|||||||

|

|
||||||
- minecraft-1.21.4-in-world: captured from Minecraft 1.21.4 after entering a singleplayer world.
|
- minecraft-1.21.4-in-world: captured from Minecraft 1.21.4 after entering a singleplayer world.
|
||||||

|

|
||||||
|
- minecraft-1.21.4-rd12-odinlite-in-world: captured on an Android device (FCL MobileGL Magma capture, 854x480) from
|
||||||
|
vanilla Minecraft 1.21.4 at render distance 12, drifting down a river valley in a boat. Unlike the other in-world
|
||||||
|
fixtures this one is a 251-frame window (gltrim `-f 1094-1343`) rather than a single frame, so it doubles as the
|
||||||
|
campaign's benchmark scene: benchmark mode replays the whole window and the tail frames measure steady-state
|
||||||
|
in-world frame time. The golden is still the final frame, so it works as an ordinary correctness case too.
|
||||||
|

|
||||||
- minecraft-1.21.4-fabric-sodium-in-world: captured from Minecraft 1.21.4 Fabric with Sodium after entering a
|
- minecraft-1.21.4-fabric-sodium-in-world: captured from Minecraft 1.21.4 Fabric with Sodium after entering a
|
||||||
singleplayer world with Fancy graphics.
|
singleplayer world with Fancy graphics.
|
||||||

|

|
||||||
@@ -280,6 +286,72 @@ unflushed persistent maps, e.g. the Create fixtures), pass
|
|||||||
sundial-lite fixture), pass `--ez avoid_angle_llvmpipe_explicit_lod_bias true` so
|
sundial-lite fixture), pass `--ez avoid_angle_llvmpipe_explicit_lod_bias true` so
|
||||||
the replay runs with `MOBILEGL_AVOID_EXPLICIT_LOD_BIAS=1`.
|
the replay runs with `MOBILEGL_AVOID_EXPLICIT_LOD_BIAS=1`.
|
||||||
|
|
||||||
|
## Benchmark mode (frame timing)
|
||||||
|
|
||||||
|
Benchmark mode reuses the same fixtures as a performance harness instead of a
|
||||||
|
correctness one: it replays the trace from the first call to the last, takes a
|
||||||
|
wall-clock timestamp at every frame boundary, and takes no snapshot and runs no
|
||||||
|
SSIM comparison. `passed` then only means the replay reached the end of the trace
|
||||||
|
without an error.
|
||||||
|
|
||||||
|
Frame times include GPU completion by default, because a swap boundary on a tiled
|
||||||
|
mobile GPU returns long before the tiler is done and would otherwise time CPU
|
||||||
|
submission alone. That is what `benchmark_finish` / `--benchmark-finish` controls:
|
||||||
|
on (the default) it issues a `glFinish` through the replayed context at every
|
||||||
|
frame boundary, which serializes CPU/GPU overlap - pessimistic against a real
|
||||||
|
running game, but deterministic and comparable between backends and revisions.
|
||||||
|
Turn it off to measure CPU-side submission only.
|
||||||
|
|
||||||
|
The mean/median/p95 are computed over the last `benchmark_tail_frames` frames
|
||||||
|
(default 200, clamped to the frames actually recorded); the head of a trace is
|
||||||
|
dominated by shader compiles and first-use uploads. The full per-frame array is in
|
||||||
|
the timing JSON, next to the headline numbers, which also appear in `result.json`.
|
||||||
|
|
||||||
|
Whole device runs, one line per run plus the best of the repeats:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
python tools/trace_replay/run_android_retrace_local.py --benchmark \
|
||||||
|
--case minecraft-1.21.4-fabric-iris-photon-in-world --backend DirectVulkan \
|
||||||
|
--benchmark-repeats 3
|
||||||
|
```
|
||||||
|
|
||||||
|
`--benchmark-tail-frames N`, `--benchmark-no-finish` and
|
||||||
|
`--benchmark-timeout-seconds N` are available; only the first repeat installs the
|
||||||
|
APK and pushes the trace. Each run's `benchmark.json` is kept next to the case
|
||||||
|
result as `benchmark-run<N>.json`.
|
||||||
|
|
||||||
|
The Activity takes the same settings directly:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
adb shell am start -a top.mobilegl.plugin.TRACE_REPLAY \
|
||||||
|
-n $PKG/top.mobilegl.plugin.trace.TraceReplayActivity \
|
||||||
|
--es trace_path $APP_DIR/input/openra.trace \
|
||||||
|
--es output_dir $APP_DIR/output \
|
||||||
|
--es backend DirectGLES \
|
||||||
|
--ei width 640 --ei height 480 \
|
||||||
|
--ez benchmark true \
|
||||||
|
--ei benchmark_tail_frames 200 \
|
||||||
|
--ez benchmark_finish true \
|
||||||
|
--es benchmark_result_path $APP_DIR/output/benchmark.json
|
||||||
|
adb exec-out run-as $PKG cat files/trace-replay/output/benchmark.json > benchmark.json
|
||||||
|
```
|
||||||
|
|
||||||
|
`golden_path` and `target_call` are not needed in benchmark mode. The Linux CLI
|
||||||
|
takes the same options:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
./mobilegl_trace_replay --trace openra.trace --output out --backend DirectGLES \
|
||||||
|
--benchmark --benchmark-tail-frames=200 --benchmark-finish=1 \
|
||||||
|
--benchmark-result=out/benchmark.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Two caveats when reading the numbers. Frame times are taken at `eglSwapBuffers`,
|
||||||
|
so a trace that ends frames with `glFrameTerminatorGREMEDY` instead is not timed.
|
||||||
|
And on a window surface the swap can block on the compositor, which pins frame
|
||||||
|
times to the display refresh; replay against the pbuffer surface
|
||||||
|
(`--ez use_pbuffer true`) to measure the renderer rather than the presentation
|
||||||
|
path.
|
||||||
|
|
||||||
## Reproducing the Android DirectGLES lane on Linux (ANGLE on lavapipe)
|
## Reproducing the Android DirectGLES lane on Linux (ANGLE on lavapipe)
|
||||||
|
|
||||||
The APK workflow's DirectGLES lane is not the same stack as the Linux one, which
|
The APK workflow's DirectGLES lane is not the same stack as the Linux one, which
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
#include "retrace.hpp"
|
#include "retrace.hpp"
|
||||||
|
|
||||||
#include "apitrace_fbo_dump.hpp"
|
#include "apitrace_fbo_dump.hpp"
|
||||||
|
#include "trace_benchmark.hpp"
|
||||||
|
|
||||||
#include <EGL/egl.h>
|
#include <EGL/egl.h>
|
||||||
#include <EGL/eglext.h>
|
#include <EGL/eglext.h>
|
||||||
@@ -414,6 +415,9 @@ public:
|
|||||||
char callNo[32];
|
char callNo[32];
|
||||||
snprintf(callNo, sizeof(callNo), "%u", retrace::callNo);
|
snprintf(callNo, sizeof(callNo), "%u", retrace::callNo);
|
||||||
gEgl.swapBuffers(gDisplay, surface);
|
gEgl.swapBuffers(gDisplay, surface);
|
||||||
|
// Frame boundary: retrace_eglSwapBuffers() has already run frame_complete() and
|
||||||
|
// handed the frame to us. No-op unless benchmark mode armed the timer.
|
||||||
|
mobilegl_trace::benchmark::OnFrameBoundary();
|
||||||
#if defined(__APPLE__)
|
#if defined(__APPLE__)
|
||||||
PumpMacOSEvents();
|
PumpMacOSEvents();
|
||||||
if (window && !windowShown) {
|
if (window && !windowShown) {
|
||||||
|
|||||||
BIN
Binary file not shown.
Binary file not shown.
@@ -119,7 +119,7 @@ def render_summary():
|
|||||||
shutil.copyfile(SUMMARY_DIR / SUMMARY_HTML, SUMMARY_DIR / "index.html")
|
shutil.copyfile(SUMMARY_DIR / SUMMARY_HTML, SUMMARY_DIR / "index.html")
|
||||||
|
|
||||||
|
|
||||||
def run_case(case, backend):
|
def run_case(case, backend, extra_args=None, timeout_seconds=None):
|
||||||
backend_info = BACKENDS[backend]
|
backend_info = BACKENDS[backend]
|
||||||
apk = find_trace_apk()
|
apk = find_trace_apk()
|
||||||
trace_archive = FIXTURES / case["trace_archive"]
|
trace_archive = FIXTURES / case["trace_archive"]
|
||||||
@@ -176,8 +176,9 @@ def run_case(case, backend):
|
|||||||
"--crop-height",
|
"--crop-height",
|
||||||
str(case["crop_height"]),
|
str(case["crop_height"]),
|
||||||
"--timeout-seconds",
|
"--timeout-seconds",
|
||||||
str(case["timeout_seconds"]),
|
str(timeout_seconds if timeout_seconds is not None else case["timeout_seconds"]),
|
||||||
]
|
]
|
||||||
|
command.extend(extra_args or [])
|
||||||
if alternate is not None:
|
if alternate is not None:
|
||||||
command[command.index("--target-call"):command.index("--target-call")] = ["--alternate-golden", bash_path(alternate)]
|
command[command.index("--target-call"):command.index("--target-call")] = ["--alternate-golden", bash_path(alternate)]
|
||||||
if backend_info["use_pbuffer"]:
|
if backend_info["use_pbuffer"]:
|
||||||
@@ -204,12 +205,118 @@ def run_case(case, backend):
|
|||||||
return result.returncode
|
return result.returncode
|
||||||
|
|
||||||
|
|
||||||
|
def read_benchmark(case, backend, run_index):
|
||||||
|
"""Reads the benchmark.json the run just pulled and files it under the run number."""
|
||||||
|
result_dir = RESULT_ROOT / f"{safe_case(case['name'])}-{backend}"
|
||||||
|
source = result_dir / "benchmark.json"
|
||||||
|
if not source.exists():
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
report = json.loads(source.read_text(encoding="utf-8"))
|
||||||
|
except (OSError, ValueError) as error:
|
||||||
|
print(f"failed to read {source}: {error}", file=sys.stderr)
|
||||||
|
return None
|
||||||
|
shutil.copyfile(source, result_dir / f"benchmark-run{run_index}.json")
|
||||||
|
return report
|
||||||
|
|
||||||
|
|
||||||
|
def format_benchmark(report):
|
||||||
|
return (
|
||||||
|
f"frames={report.get('totalFrames', -1)}"
|
||||||
|
f" total={report.get('totalSeconds', -1):.1f}s"
|
||||||
|
f" tail={report.get('tailFrames', -1)}"
|
||||||
|
f" mean={report.get('meanFrameMs', -1):.3f}ms"
|
||||||
|
f" median={report.get('medianFrameMs', -1):.3f}ms"
|
||||||
|
f" p95={report.get('p95FrameMs', -1):.3f}ms"
|
||||||
|
f" fps={report.get('fps', -1):.1f}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def run_benchmark_case(case, backend, args):
|
||||||
|
"""Runs the case as a frame-timing benchmark `--benchmark-repeats` times.
|
||||||
|
|
||||||
|
Only the first run installs the APK and pushes the trace; the repeats reuse what is
|
||||||
|
already on the device, so the numbers are not paying for an adb push each time.
|
||||||
|
"""
|
||||||
|
label = f"{case['name']} / {backend}"
|
||||||
|
reports = []
|
||||||
|
failures = 0
|
||||||
|
for run_index in range(1, args.benchmark_repeats + 1):
|
||||||
|
extra_args = [
|
||||||
|
"--benchmark",
|
||||||
|
"--benchmark-tail-frames",
|
||||||
|
str(args.benchmark_tail_frames),
|
||||||
|
"--benchmark-finish",
|
||||||
|
"0" if args.benchmark_no_finish else "1",
|
||||||
|
]
|
||||||
|
if run_index > 1:
|
||||||
|
extra_args.append("--reuse-fixture")
|
||||||
|
# The previous repeat's file would otherwise be read back as this run's result.
|
||||||
|
stale = RESULT_ROOT / f"{safe_case(case['name'])}-{backend}" / "benchmark.json"
|
||||||
|
if stale.exists():
|
||||||
|
stale.unlink()
|
||||||
|
rc = run_case(case, backend, extra_args=extra_args, timeout_seconds=args.benchmark_timeout_seconds)
|
||||||
|
report = read_benchmark(case, backend, run_index)
|
||||||
|
if rc != 0 or report is None:
|
||||||
|
print(f"{label} run {run_index}/{args.benchmark_repeats}: FAILED (exit {rc})", flush=True)
|
||||||
|
failures += 1
|
||||||
|
continue
|
||||||
|
reports.append((run_index, report))
|
||||||
|
print(
|
||||||
|
f"{label} run {run_index}/{args.benchmark_repeats}: {format_benchmark(report)}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
def mean_frame_ms(entry):
|
||||||
|
# A run that recorded no frames reports -1; it must not win "best" by being smallest.
|
||||||
|
mean = entry[1].get("meanFrameMs", -1)
|
||||||
|
return mean if mean > 0 else float("inf")
|
||||||
|
|
||||||
|
if reports:
|
||||||
|
best_index, best = min(reports, key=mean_frame_ms)
|
||||||
|
print(
|
||||||
|
f"{label} best of {args.benchmark_repeats} (run {best_index}): {format_benchmark(best)}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
return 1 if failures else 0
|
||||||
|
|
||||||
|
|
||||||
def parse_args():
|
def parse_args():
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument("--case", action="append", dest="cases", help="Case name to run; may be repeated.")
|
parser.add_argument("--case", action="append", dest="cases", help="Case name to run; may be repeated.")
|
||||||
parser.add_argument("--backend", action="append", choices=sorted(BACKENDS), help="Backend to run; may be repeated.")
|
parser.add_argument("--backend", action="append", choices=sorted(BACKENDS), help="Backend to run; may be repeated.")
|
||||||
parser.add_argument("--all", action="store_true", help="Run every case in the APK workflow matrix.")
|
parser.add_argument("--all", action="store_true", help="Run every case in the APK workflow matrix.")
|
||||||
parser.add_argument("--keep-results", action="store_true", help="Do not clear the previous result root.")
|
parser.add_argument("--keep-results", action="store_true", help="Do not clear the previous result root.")
|
||||||
|
parser.add_argument(
|
||||||
|
"--benchmark",
|
||||||
|
action="store_true",
|
||||||
|
help="Replay each case end to end as a frame-timing benchmark instead of comparing "
|
||||||
|
"one frame against its golden.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--benchmark-repeats",
|
||||||
|
type=int,
|
||||||
|
default=3,
|
||||||
|
help="Benchmark runs per case/backend; the best (lowest mean frame time) is reported.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--benchmark-tail-frames",
|
||||||
|
type=int,
|
||||||
|
default=200,
|
||||||
|
help="Frames at the end of the run the statistics are computed over.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--benchmark-no-finish",
|
||||||
|
action="store_true",
|
||||||
|
help="Do not glFinish at every frame boundary, so frame times measure CPU submission "
|
||||||
|
"only instead of GPU completion.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--benchmark-timeout-seconds",
|
||||||
|
type=int,
|
||||||
|
default=900,
|
||||||
|
help="Per-run timeout; a benchmark replays the whole trace, not just up to target_call.",
|
||||||
|
)
|
||||||
return parser.parse_args()
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
@@ -221,12 +328,20 @@ def main():
|
|||||||
if not selected_cases:
|
if not selected_cases:
|
||||||
print("No cases selected. Use --all or --case NAME.", file=sys.stderr)
|
print("No cases selected. Use --all or --case NAME.", file=sys.stderr)
|
||||||
return 2
|
return 2
|
||||||
|
if args.benchmark and args.benchmark_repeats < 1:
|
||||||
|
print("--benchmark-repeats must be at least 1.", file=sys.stderr)
|
||||||
|
return 2
|
||||||
if not args.keep_results and RESULT_ROOT.exists():
|
if not args.keep_results and RESULT_ROOT.exists():
|
||||||
shutil.rmtree(RESULT_ROOT)
|
shutil.rmtree(RESULT_ROOT)
|
||||||
RESULT_ROOT.mkdir(parents=True, exist_ok=True)
|
RESULT_ROOT.mkdir(parents=True, exist_ok=True)
|
||||||
failures = 0
|
failures = 0
|
||||||
for case in selected_cases:
|
for case in selected_cases:
|
||||||
for backend in selected_backends:
|
for backend in selected_backends:
|
||||||
|
if args.benchmark:
|
||||||
|
print(f"=== Android benchmark: {case['name']} / {backend} ===", flush=True)
|
||||||
|
# No SSIM verdicts to render here; the summary page is for the correctness lane.
|
||||||
|
failures += run_benchmark_case(case, backend, args)
|
||||||
|
continue
|
||||||
print(f"=== Android retrace: {case['name']} / {backend} ===", flush=True)
|
print(f"=== Android retrace: {case['name']} / {backend} ===", flush=True)
|
||||||
rc = run_case(case, backend)
|
rc = run_case(case, backend)
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -60,6 +60,13 @@
|
|||||||
"golden": "minecraft-1.21.4-in-world.0000280000.png",
|
"golden": "minecraft-1.21.4-in-world.0000280000.png",
|
||||||
"target_call": 280000
|
"target_call": 280000
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"name": "minecraft-1.21.4-rd12-odinlite-in-world",
|
||||||
|
"trace_archive": "minecraft-1.21.4-rd12-odinlite-in-world.tgz",
|
||||||
|
"golden": "minecraft-1.21.4-rd12-odinlite-in-world.0004660351.png",
|
||||||
|
"target_call": 4660351,
|
||||||
|
"timeout_seconds": 1800
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"name": "minecraft-1.21.4-fabric-sodium-in-world",
|
"name": "minecraft-1.21.4-fabric-sodium-in-world",
|
||||||
"trace_archive": "minecraft-1.21.4-fabric-sodium-in-world.tgz",
|
"trace_archive": "minecraft-1.21.4-fabric-sodium-in-world.tgz",
|
||||||
|
|||||||
@@ -29,6 +29,15 @@ void PrintUsage(const char *argv0) {
|
|||||||
<< " --crop-width N Compare crop width\n"
|
<< " --crop-width N Compare crop width\n"
|
||||||
<< " --crop-height N Compare crop height\n"
|
<< " --crop-height N Compare crop height\n"
|
||||||
<< " --coherent-as-flush Set MOBILEGL_COHERENT_AS_FLUSH=1 for the replay\n"
|
<< " --coherent-as-flush Set MOBILEGL_COHERENT_AS_FLUSH=1 for the replay\n"
|
||||||
|
<< " --benchmark Frame-timing mode: replay the whole trace, record a\n"
|
||||||
|
<< " wall-clock timestamp at every frame boundary, and skip\n"
|
||||||
|
<< " the snapshot and the golden comparison. --target-call is\n"
|
||||||
|
<< " not required in this mode.\n"
|
||||||
|
<< " --benchmark-tail-frames=N Frames at the end of the run the mean/median/p95 are\n"
|
||||||
|
<< " computed over (default: 200, clamped to the frame count)\n"
|
||||||
|
<< " --benchmark-finish=0/1 glFinish at every frame boundary so a frame time covers\n"
|
||||||
|
<< " GPU completion and not just CPU submission (default: 1)\n"
|
||||||
|
<< " --benchmark-result=PATH Timing JSON output (default: OUTPUT/benchmark.json)\n"
|
||||||
<< " --dump-fbo-attachments CALL:DIR[:FBO,FBO,...]\n"
|
<< " --dump-fbo-attachments CALL:DIR[:FBO,FBO,...]\n"
|
||||||
<< " At CALL, write every colour attachment and the depth\n"
|
<< " At CALL, write every colour attachment and the depth\n"
|
||||||
<< " attachment of every live framebuffer object into DIR as\n"
|
<< " attachment of every live framebuffer object into DIR as\n"
|
||||||
@@ -77,6 +86,23 @@ bool ReadEnvFlag(const char *name) {
|
|||||||
return value != nullptr && std::string(value) == "1";
|
return value != nullptr && std::string(value) == "1";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The benchmark options are documented as --name=VALUE. The space-separated spelling every
|
||||||
|
// other option here uses is accepted too, so --benchmark-tail-frames=300 and
|
||||||
|
// --benchmark-tail-frames 300 both work; hasInlineValue says which one was given.
|
||||||
|
bool MatchOption(const std::string &arg, const char *name, std::string &inlineValue, bool &hasInlineValue) {
|
||||||
|
if (arg == name) {
|
||||||
|
hasInlineValue = false;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
const std::string prefix = std::string(name) + "=";
|
||||||
|
if (arg.compare(0, prefix.size(), prefix) == 0) {
|
||||||
|
inlineValue = arg.substr(prefix.size());
|
||||||
|
hasInlineValue = true;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
bool ParseArgs(int argc, char **argv, mobilegl_trace::Request &request) {
|
bool ParseArgs(int argc, char **argv, mobilegl_trace::Request &request) {
|
||||||
request.backend = "DirectGLES";
|
request.backend = "DirectGLES";
|
||||||
request.fixIterationRPSubgroupScratch = ReadEnvFlag("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
|
request.fixIterationRPSubgroupScratch = ReadEnvFlag("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
|
||||||
@@ -85,6 +111,8 @@ bool ParseArgs(int argc, char **argv, mobilegl_trace::Request &request) {
|
|||||||
|
|
||||||
for (int i = 1; i < argc; ++i) {
|
for (int i = 1; i < argc; ++i) {
|
||||||
const std::string arg = argv[i];
|
const std::string arg = argv[i];
|
||||||
|
std::string optionValue;
|
||||||
|
bool optionHasValue = false;
|
||||||
if (arg == "--trace") {
|
if (arg == "--trace") {
|
||||||
if (!ReadValue(argc, argv, i, request.tracePath)) return false;
|
if (!ReadValue(argc, argv, i, request.tracePath)) return false;
|
||||||
} else if (arg == "--golden") {
|
} else if (arg == "--golden") {
|
||||||
@@ -129,6 +157,17 @@ bool ParseArgs(int argc, char **argv, mobilegl_trace::Request &request) {
|
|||||||
if (!ReadInt(argc, argv, i, request.cropHeight)) return false;
|
if (!ReadInt(argc, argv, i, request.cropHeight)) return false;
|
||||||
} else if (arg == "--coherent-as-flush") {
|
} else if (arg == "--coherent-as-flush") {
|
||||||
request.coherentAsFlush = true;
|
request.coherentAsFlush = true;
|
||||||
|
} else if (arg == "--benchmark") {
|
||||||
|
request.benchmark = true;
|
||||||
|
} else if (MatchOption(arg, "--benchmark-tail-frames", optionValue, optionHasValue)) {
|
||||||
|
if (!optionHasValue && !ReadValue(argc, argv, i, optionValue)) return false;
|
||||||
|
request.benchmarkTailFrames = std::atoi(optionValue.c_str());
|
||||||
|
} else if (MatchOption(arg, "--benchmark-finish", optionValue, optionHasValue)) {
|
||||||
|
if (!optionHasValue && !ReadValue(argc, argv, i, optionValue)) return false;
|
||||||
|
request.benchmarkFinish = optionValue != "0";
|
||||||
|
} else if (MatchOption(arg, "--benchmark-result", optionValue, optionHasValue)) {
|
||||||
|
if (!optionHasValue && !ReadValue(argc, argv, i, optionValue)) return false;
|
||||||
|
request.benchmarkResultPath = optionValue;
|
||||||
} else if (arg == "--dump-fbo-attachments") {
|
} else if (arg == "--dump-fbo-attachments") {
|
||||||
std::string dumpPoint;
|
std::string dumpPoint;
|
||||||
if (!ReadValue(argc, argv, i, dumpPoint)) return false;
|
if (!ReadValue(argc, argv, i, dumpPoint)) return false;
|
||||||
@@ -153,7 +192,7 @@ bool ParseArgs(int argc, char **argv, mobilegl_trace::Request &request) {
|
|||||||
std::cerr << "--output is required\n";
|
std::cerr << "--output is required\n";
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (request.targetCall < 0) {
|
if (request.targetCall < 0 && !request.benchmark) {
|
||||||
std::cerr << "--target-call is required\n";
|
std::cerr << "--target-call is required\n";
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -161,6 +200,10 @@ bool ParseArgs(int argc, char **argv, mobilegl_trace::Request &request) {
|
|||||||
std::cerr << "--hold-ms must be non-negative\n";
|
std::cerr << "--hold-ms must be non-negative\n";
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
if (request.benchmarkTailFrames <= 0) {
|
||||||
|
std::cerr << "--benchmark-tail-frames must be positive\n";
|
||||||
|
return false;
|
||||||
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -181,9 +224,15 @@ int main(int argc, char **argv) {
|
|||||||
mobilegl_trace::WriteResultJson(request, result);
|
mobilegl_trace::WriteResultJson(request, result);
|
||||||
}
|
}
|
||||||
|
|
||||||
std::cout << result.message << "\n"
|
if (request.benchmark) {
|
||||||
<< "result: " << result.resultPath << "\n"
|
std::cout << result.message << "\n"
|
||||||
<< "actual: " << result.actualPath << "\n"
|
<< "result: " << result.resultPath << "\n"
|
||||||
<< "diff: " << result.diffPath << "\n";
|
<< "benchmark: " << result.benchmarkResultPath << "\n";
|
||||||
|
} else {
|
||||||
|
std::cout << result.message << "\n"
|
||||||
|
<< "result: " << result.resultPath << "\n"
|
||||||
|
<< "actual: " << result.actualPath << "\n"
|
||||||
|
<< "diff: " << result.diffPath << "\n";
|
||||||
|
}
|
||||||
return result.passed ? 0 : result.statusCode;
|
return result.passed ? 0 : result.statusCode;
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user