diff --git a/CMakeLists.txt b/CMakeLists.txt index 01abae32..32614803 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -270,6 +270,7 @@ set(SOURCE_FILES MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp MobileGL/MG_Util/ShaderTranspiler/SpvcSession.cpp MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp + MobileGL/MG_Util/ShaderTranspiler/TranslationCache.cpp MobileGL/MG_Util/ShaderTranspiler/glslang/TMglGlslIoResolver.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenInterfaceStructPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EliminateFloatEqualsZeroPass.cpp diff --git a/MobileGL/Config.h b/MobileGL/Config.h index 1ccfa8bb..3feac0eb 100644 --- a/MobileGL/Config.h +++ b/MobileGL/Config.h @@ -203,6 +203,14 @@ namespace MobileGL::MG_Config { // immediately stay serial by their own construction). Off by default; never // advertise it. QuirkOverride AsyncOptimisticShaderStatus = QuirkOverride::Auto; + // MOBILEGL_SHADER_CACHE: the two-level, in-memory shader translation memo + // (MG_Util/ShaderTranspiler/TranslationCache.h) - L1 memoizes a linked + // program's sanitized SPIR-V, L2 memoizes DirectGLES's emitted ESSL. Auto + // is ON; ForceOff turns BOTH levels off and makes every translation run + // from scratch. The escape hatch exists because a wrong cache hit is a + // silently miscompiled shader: if a device ever renders differently with + // the cache on, one run with this falsy says so. + QuirkOverride ShaderTranslationCache = QuirkOverride::Auto; }; extern FeaturesTable Features; } // namespace MobileGL::MG_Config diff --git a/MobileGL/ConfigLoader.cpp b/MobileGL/ConfigLoader.cpp index cb1fe00f..cb4f467b 100644 --- a/MobileGL/ConfigLoader.cpp +++ b/MobileGL/ConfigLoader.cpp @@ -194,6 +194,7 @@ namespace MobileGL::MG_ConfigLoader { features.AsyncShaderCompileThreads = QueryEnvUint32("MOBILEGL_ASYNC_SHADER_COMPILE_THREADS", 0, 0, 64); features.AsyncOptimisticShaderStatus = QueryEnvQuirkOverride("MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS"); + features.ShaderTranslationCache = QueryEnvQuirkOverride("MOBILEGL_SHADER_CACHE"); } inline void InitBackendType() { diff --git a/MobileGL/Init.cpp b/MobileGL/Init.cpp index 59554a1f..2e4497d3 100644 --- a/MobileGL/Init.cpp +++ b/MobileGL/Init.cpp @@ -18,6 +18,7 @@ #include #include #include +#include #include #include @@ -72,6 +73,12 @@ namespace MobileGL { // built-in symbol tables the prewarm latch stands for, so leaving it set would // make the next Initialize() skip a prewarm it genuinely needs. MG_Util::ShaderTranspiler::ShaderCompiler::ResetPrewarmLatch(); + // The two-level translation memo. Nothing in it references a glslang object - + // both levels hold plain bytes - so this is RSS hygiene rather than a lifetime + // requirement, and it is safe either side of FinalizeProcess. Stats first: an + // fordebug build gets one line per level saying how the run went. + MG_Util::ShaderTranspiler::LogShaderTranslationCacheStats(); + MG_Util::ShaderTranspiler::ClearShaderTranslationCaches(); MG_Backend::gBackendFunctionsTable = {}; g_isInitialized = false; if (logLifecycle) { diff --git a/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.cpp b/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.cpp index 9194d3ea..59096bd7 100644 --- a/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.cpp +++ b/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.cpp @@ -497,11 +497,53 @@ namespace MobileGL::MG_State::GLState { spirvHandoff.reflection.uniformIndexInTProgram = artifacts.uniformIndexInTProgram; spirvHandoff.reflection.tProgramUniformIndexToGl = artifacts.tProgramUniformIndexToGl; spirvHandoff.reflection.maxUniformLocation = artifacts.maxUniformLocation; + spirvHandoff.spirvCacheKey = BuildSpirvCacheKey(env); spirvHandoff.ready = true; MGLOG_D("ProgramObject %u: phase A done, %zu module(s) handed to the SPIR-V job", in.externalIndex, spirvHandoff.shaderTypes.size()); } + // The L1 key. Every input below is one that can change the SPIR-V this program + // generates; see the key inventory on SpirvTranslationKeyInputs. + // + // Deliberately NOT keyed on: the transform-feedback request + // (ResolveTransformFeedbackVaryings only READS the linked intermediates - it sets no + // XFB qualifier, and the ESSL capture rename happens in the backend, behind L2's own + // key), the fragment-output count limit (a link-failure gate, never an emission input), + // and reflection (verified non-mutating on this glslang pin; see the ordering note in + // RunBody). + MG_Util::ShaderTranspiler::TranslationCacheKey ProgramLinkTask::BuildSpirvCacheKey( + const MG_Util::ShaderTranspiler::CompileEnv& env) const { + using namespace MG_Util::ShaderTranspiler; + if (!ShaderTranslationCacheEnabled()) return {}; + + SpirvTranslationKeyInputs keyInputs; + keyInputs.envFingerprint = env.fingerprint; + // Always 0 on both production parse paths (ShaderCompileTask::RunCompilePipeline and + // ClaimParsedShader's re-parse). In the key regardless, so that a future non-zero + // value cannot alias a module parsed without it. + keyInputs.shaderCompileFlags = 0; + keyInputs.enableSpirvValidation = in.enableSpirvValidation; + keyInputs.stages.reserve(in.shaders.size()); + for (const LinkShaderInput& shader : in.shaders) { + const ShaderCompileArtifacts& compiled = CompiledArtifacts(shader.compiled); + if (compiled.preprocessedSource.empty()) { + // No text to key on - an internal shader object, or an artifact this build + // did not populate. Refuse to key rather than key on nothing. + return {}; + } + keyInputs.stages.push_back(SpirvTranslationKeyInputs::Stage{ + .type = MG_Util::ConvertShaderStageToGLEnum(shader.stage), + .preprocessedSource = StringView(compiled.preprocessedSource)}); + } + if (keyInputs.stages.empty()) return {}; + keyInputs.explicitVertexInLocations = &in.explicitAttribLocations; + keyInputs.explicitFragmentOutLocations = &in.explicitFragDataLocation; + keyInputs.explicitFragmentOutIndices = &in.explicitFragDataIndex; + keyInputs.explicitOpaqueUniformBindings = &artifacts.explicitOpaqueUniformBindings; + return BuildSpirvTranslationKey(keyInputs); + } + Bool ProgramLinkTask::ConsumeShaders(Vector>& outShaders) { outShaders.assign(in.shaders.size(), nullptr); diff --git a/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.h b/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.h index 1486ab4b..d60ae273 100644 --- a/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.h +++ b/MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.h @@ -12,6 +12,7 @@ #include #include #include +#include namespace MobileGL::MG_State::GLState { // One attached shader, as the link sees it: never the ShaderObject, always a snapshot. @@ -116,6 +117,19 @@ namespace MobileGL::MG_State::GLState { // for phase B after the join has moved `artifacts` away. ProgramObject::LinkArtifacts reflection; + // L1 shader-translation memo key for this program's SPIR-V (see + // MG_Util/ShaderTranspiler/TranslationCache.h). Built HERE, at the tail of phase + // A, and not by phase B - two reasons, both structural: + // * the key covers the four link-time request maps and the merged opaque + // bindings, and one of those (explicitOpaqueUniformBindings) lives in + // `artifacts`, which phase B is forbidden to read because the GL-thread join + // moves it out from under phase B; + // * built once, it serves both the lookup and the insert, so the program's + // sources are copied into the blob exactly once per link. + // Invalid (null blob) when the cache is disabled, or when a stage arrived + // without preprocessed source - in which case phase B simply translates. + MG_Util::ShaderTranspiler::TranslationCacheKey spirvCacheKey; + // The one flag phase B tests before doing anything: false means this link never // reached the tail of RunBody (it failed, or was cancelled mid-body). Bool ready = false; @@ -143,6 +157,13 @@ namespace MobileGL::MG_State::GLState { // Each returns false to abort the link with `artifacts.infoLog` already set, which is // GL's definition of a failed link: LINK_STATUS false plus a log, never a GL error. Bool ConsumeShaders(Vector>& outShaders); + + // The L1 memo key for the SPIR-V this program is about to generate, or an invalid + // key when the cache is off or a stage has no preprocessed source to key on. + // Called at the tail of RunBody, where every input it needs is still owned by this + // node and `artifacts` has not yet been published. + MG_Util::ShaderTranspiler::TranslationCacheKey BuildSpirvCacheKey( + const MG_Util::ShaderTranspiler::CompileEnv& env) const; Bool DoReflection(const MG_Util::ShaderTranspiler::CompileEnv& env); Bool ValidateFragmentOutputLocations(); Bool ResolveTransformFeedbackVaryings(); diff --git a/MobileGL/MG_State/GLState/ProgramState/ProgramSpirvTask.cpp b/MobileGL/MG_State/GLState/ProgramState/ProgramSpirvTask.cpp index d12cd49a..a0ff57a2 100644 --- a/MobileGL/MG_State/GLState/ProgramState/ProgramSpirvTask.cpp +++ b/MobileGL/MG_State/GLState/ProgramState/ProgramSpirvTask.cpp @@ -12,6 +12,7 @@ #include #include #include +#include #include #include @@ -152,6 +153,28 @@ namespace MobileGL::MG_State::GLState { using namespace MG_Util::ShaderTranspiler; MGLOG_D("ProgramObject %u: GenerateSpirv - start", externalIndex); + // L1 of the shader translation memo. The segment this short-circuits is the whole + // of GlslangToSpv plus the 11-pass SanitizeAndOptimizeBinary chain, for every stage + // of the program at once - ~136 us per stage on the RelWithDebInfo host measurement. + // The key was built at the tail of phase A (ProgramLinkTask::BuildSpirvCacheKey) and + // covers every input that can move these bytes; see TranslationCache.h. + // + // Note what a HIT does NOT skip: the glslang parse and link, which already happened + // in phase A because the frontend's whole GL query surface is built out of the + // TProgram they produce. + auto& spirvCache = GetSpirvTranslationCache(); + const TranslationCacheKey& cacheKey = handoff.spirvCacheKey; + if (cacheKey.Valid()) { + if (const SpirvTranslationResultPtr hit = spirvCache.Find(cacheKey); + hit && hit->modules.size() == handoff.shaderTypes.size()) { + artifacts.generatedSpirv = hit->modules; + artifacts.spirvStatus = true; + MGLOG_D("ProgramObject %u: GenerateSpirv - L1 cache hit, %zu module(s) reused", + externalIndex, artifacts.generatedSpirv.size()); + return; + } + } + // The shaders were parsed once, in the link-compatible (relaxed Vulkan-rules) // configuration, and the handoff's program linked those parses - so it IS the program // the backends consume. Generate SPIR-V straight from its intermediates, which the @@ -195,6 +218,16 @@ namespace MobileGL::MG_State::GLState { } } artifacts.spirvStatus = allOptimized; + + // Only a clean run is memoized. A failed optimizer run leaves `spv` as whatever the + // chain got to before it gave up, and that is exactly the binary no other program + // should ever be handed. + if (allOptimized && cacheKey.Valid()) { + auto payload = MakeShared(); + payload->modules = artifacts.generatedSpirv; + const SizeT payloadBytes = SpirvTranslationResultBytes(*payload); + spirvCache.Insert(cacheKey, SpirvTranslationResultPtr(Move(payload)), payloadBytes); + } } void ProgramSpirvTask::BuildGlobalUboRouting(const ProgramLinkTask::SpirvHandoff& handoff, diff --git a/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.cpp b/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.cpp new file mode 100644 index 00000000..24f8f992 --- /dev/null +++ b/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.cpp @@ -0,0 +1,178 @@ +// MobileGL - MobileGL/MG_Util/ShaderTranspiler/TranslationCache.cpp +// Copyright (c) 2025-2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#include "TranslationCache.h" + +#include + +namespace MobileGL::MG_Util::ShaderTranspiler { + namespace { + // Tags keep two different key builders from ever producing the same blob, + // even if their inputs happened to serialize identically. + constexpr Uint32 kSpirvKeyTag = 0x4d474c31u; // "MGL1" + constexpr Uint32 kEsslKeyTag = 0x4d474c32u; // "MGL2" + + // Bumped whenever the SHAPE of a key changes (a field added, a field's + // meaning changed). It is in every blob, so a stale in-memory entry from a + // previous shape cannot be honoured - and a future disk tier gets the same + // protection for free. + constexpr Uint32 kKeyLayoutVersion = 1u; + + // The repo's existing cache epoch (MG_Config::CacheVersion, the seed + // ProgramFactory::ComputeHash uses). Strictly redundant for an in-memory + // cache - one process cannot hold two of them - but it is the knob a disk + // tier would have to turn, and putting it in now means the blob format does + // not have to change when that tier arrives. + void AppendCommonKeyPrefix(TranslationKeyBuilder& builder, const Uint32 tag) { + builder.Value(tag); + builder.Value(kKeyLayoutVersion); + builder.Value(MG_Config::CacheVersion); + } + + // ---- L1 caps ------------------------------------------------------- + // 64 entries / 12 MiB. + // + // The win this cache exists for is REPETITION, not coverage: a CTS smoke + // case compiles a handful of distinct sources 2592 times, and a handful of + // entries serves it completely. The opposite workload - an Iris shaderpack + // load - is ~300-600 MOSTLY DISTINCT programs, which would never hit no + // matter how large the cache is, so a large cap there buys nothing and + // costs resident memory on a phone. 64 entries is comfortably above the + // distinct-source count of every repetition workload measured, and the + // 12 MiB ceiling bounds the pathological case (a pack whose ~100 KB stages + // ARE re-linked) at the same order as the existing 8 MiB + // ShaderPreprocessCache budget. + constexpr SizeT kSpirvCacheMaxEntries = 64; + constexpr SizeT kSpirvCacheMaxBytes = 12u * 1024u * 1024u; + + // ---- L2 caps ------------------------------------------------------- + // 128 entries / 12 MiB. Same reasoning, twice the entry count: L2 is keyed + // per STAGE rather than per program, so the same program population needs + // roughly twice the slots. The byte budget stays put - an L2 entry (SPIR-V + // in, ESSL text out) is smaller than an L1 one (all stages' source in, all + // stages' SPIR-V out). + constexpr SizeT kEsslCacheMaxEntries = 128; + constexpr SizeT kEsslCacheMaxBytes = 12u * 1024u * 1024u; + } // namespace + + Bool ShaderTranslationCacheEnabled() { + // Read live rather than latched into a function-local static. MG_Config::Features + // is a plain global of scalars written once by MG_ConfigLoader::Init() - a load + // costs nothing, no worker ever touches the environment through it, and the unit + // tests (which flip the field directly, as AsyncCompileTest and QueryTest already + // do) need the switch to actually take effect when they flip it. + return MG_Config::Features.ShaderTranslationCache != MG_Config::QuirkOverride::ForceOff; + } + + void TranslationKeyBuilder::Bytes(const void* data, const SizeT length) { + if (length == 0) return; + m_blob.append(static_cast(data), length); + } + + void TranslationKeyBuilder::Text(const StringView text) { + Value(static_cast(text.size())); + Bytes(text.data(), text.size()); + } + + void TranslationKeyBuilder::Words(const Vector& words) { + Value(static_cast(words.size())); + Bytes(words.data(), words.size() * sizeof(Uint32)); + } + + void TranslationKeyBuilder::NameSet(const std::set& names) { + Value(static_cast(names.size())); + for (const String& name : names) Text(name); + } + + TranslationCacheKey MakeTranslationCacheKey(String blob) { + TranslationCacheKey key; + key.hash = static_cast(XXH64(blob.data(), blob.size(), 0)); + key.blob = MakeShared(Move(blob)); + return key; + } + + TranslationCacheKey BuildSpirvTranslationKey(const SpirvTranslationKeyInputs& inputs) { + TranslationKeyBuilder builder; + AppendCommonKeyPrefix(builder, kSpirvKeyTag); + builder.Value(inputs.envFingerprint); + builder.Value(inputs.shaderCompileFlags); + builder.Value(static_cast(inputs.enableSpirvValidation)); + builder.Value(static_cast(inputs.stages.size())); + for (const auto& stage : inputs.stages) { + builder.Value(static_cast(stage.type)); + builder.Text(stage.preprocessedSource); + } + static const UnorderedMap kEmpty; + builder.NameMap(inputs.explicitVertexInLocations ? *inputs.explicitVertexInLocations : kEmpty); + builder.NameMap(inputs.explicitFragmentOutLocations ? *inputs.explicitFragmentOutLocations : kEmpty); + builder.NameMap(inputs.explicitFragmentOutIndices ? *inputs.explicitFragmentOutIndices : kEmpty); + builder.NameMap(inputs.explicitOpaqueUniformBindings ? *inputs.explicitOpaqueUniformBindings : kEmpty); + return MakeTranslationCacheKey(builder); + } + + SizeT SpirvTranslationResultBytes(const SpirvTranslationResult& result) { + SizeT bytes = 0; + for (const auto& module : result.modules) bytes += module.size() * sizeof(Uint32); + return bytes; + } + + TranslationCacheKey BuildEsslTranslationKey(const EsslTranslationKeyInputs& inputs) { + TranslationKeyBuilder builder; + AppendCommonKeyPrefix(builder, kEsslKeyTag); + builder.Value(static_cast(inputs.shaderType)); + builder.Value(static_cast(inputs.supportsViewportArray)); + builder.Value(static_cast(inputs.supportsNoperspectiveInterpolation)); + builder.Value(inputs.maxColorTextureSamples); + builder.Value(inputs.maxIntegerSamples); + builder.Value(inputs.maxDepthTextureSamples); + builder.Value(inputs.advertisedMaxSamples); + builder.Value(static_cast(inputs.esslVersion)); + builder.Value(static_cast(inputs.enableSpirvValidation)); + static const std::set kEmptySet; + builder.NameSet(inputs.xfbCaptureBlockNames ? *inputs.xfbCaptureBlockNames : kEmptySet); + static const UnorderedMap kEmptyFormats; + builder.NameMap(inputs.glFormatByUniformName ? *inputs.glFormatByUniformName : kEmptyFormats); + static const UnorderedMap kEmptyBindings; + builder.NameMap(inputs.storageBlockBindingOverrides ? *inputs.storageBlockBindingOverrides + : kEmptyBindings); + static const Vector kEmptyWords; + builder.Words(inputs.spirv ? *inputs.spirv : kEmptyWords); + return MakeTranslationCacheKey(builder); + } + + SizeT EsslTranslationResultBytes(const EsslTranslationResult& result) { + SizeT bytes = result.essl.size(); + for (const String& name : result.flattenedXfbBlockNames) bytes += name.size(); + return bytes; + } + + BoundedTranslationCache& GetSpirvTranslationCache() { + // Function-local static: the caches must not be constructed before + // MG_Config is loaded, and they must survive every context teardown (the + // key carries the CompileEnv fingerprint, so surviving is safe). + static BoundedTranslationCache kCache( + "ShaderTranslationCache L1 (GLSL->SPIR-V)", kSpirvCacheMaxEntries, kSpirvCacheMaxBytes); + return kCache; + } + + BoundedTranslationCache& GetEsslTranslationCache() { + static BoundedTranslationCache kCache( + "ShaderTranslationCache L2 (SPIR-V->ESSL)", kEsslCacheMaxEntries, kEsslCacheMaxBytes); + return kCache; + } + + void ClearShaderTranslationCaches() { + GetSpirvTranslationCache().Clear(); + GetEsslTranslationCache().Clear(); + } + + void LogShaderTranslationCacheStats() { + GetSpirvTranslationCache().LogStats(); + GetEsslTranslationCache().LogStats(); + } +} // namespace MobileGL::MG_Util::ShaderTranspiler diff --git a/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h b/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h new file mode 100644 index 00000000..a3d242c6 --- /dev/null +++ b/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h @@ -0,0 +1,464 @@ +// MobileGL - MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h +// Copyright (c) 2025-2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#pragma once +#include + +#include +#include +#include + +namespace MobileGL::MG_Util::ShaderTranspiler { + // =========================================================================== + // The two-level shader translation memo. + // + // MOTIVATION (measured). KHR-GL33.texture_swizzle.smoke_* builds 2592 programs + // per case out of a handful of DISTINCT sources - the CTS template substitutes + // BASIC_TYPE and little else within one case - and the process is CPU-bound at + // 93% cpu/wall with the device driver's own compiler at 0.15%. Every one of + // those 2592 programs walks the whole translation chain again: + // + // GLSL --[glslang parse]--> AST --[link + mapIO]--> TProgram + // --[GlslangToSpv]--> SPIR-V --[SanitizeAndOptimizeBinary]--> SPIR-V' + // --[backend SPIR-V pass chain]--> SPIR-V'' --[SPIRV-Cross]--> ESSL + // + // L1 memoizes the segment from the parsed program to SPIR-V'; L2 memoizes the + // segment from SPIR-V' to the emitted backend payload. The two are kept apart + // on purpose: L1 is backend-agnostic (the same module feeds DirectGLES and + // DirectVulkan), while L2's key is made almost entirely of BACKEND capability + // bits, and folding them into one key would make every DirectGLES capability a + // reason to miss on the frontend half as well. + // + // WHAT IS DELIBERATELY NOT MEMOIZED: the glslang parse and the glslang link. + // Both produce a TShader/TProgram, and the frontend's whole GL query surface + // (ProgramObject::LinkArtifacts, BuildGlobalUboRouting) is built by asking that + // TProgram questions - so skipping them means caching a live glslang object + // graph and sharing it between ProgramObjects, which is a different change with + // its own aliasing and consume-once hazards. See the report in the branch + // history; the parse is ~50% of the per-stage cost and is the next campaign. + // + // CORRECTNESS RULE, non-negotiable. A wrong hit is a silently miscompiled + // shader - far worse than a slow one. So: + // * the key blob carries the FULL bytes of every input, never a digest, and + // every candidate hit is confirmed by comparing those bytes. The 64-bit + // hash is a bucket selector only; a collision degrades to a miss. + // * every input that can change the output is in the blob. Adding an input + // to a translation step MEANS adding it to that level's key builder. + // * MOBILEGL_SHADER_CACHE=0 turns both levels off, so a field miscompile can + // be bisected against the cache in one run. + // + // NO DISK TIER IN THIS CHANGE. Persistence needs its own invalidation story + // (driver/vendor string, MobileGL build id, glslang and SPIRV-Cross revisions) + // and its own answer to "what if the file is hostile", and neither belongs in + // a performance change. Where it WOULD attach: BoundedTranslationCache::Find, + // on the miss path, would consult a disk tier keyed by the same blob before + // returning null, and Insert would write through to it. Nothing in the design + // below forecloses that - the key is already a self-contained byte string and + // the payloads are already plain data. + // =========================================================================== + + // The process-wide master switch, mirroring MOBILEGL_SHADER_CACHE. + // QuirkOverride semantics: unset (Auto) is ON, an explicitly falsy value is + // OFF. Read once from MG_Config::Features, so a worker never touches the + // environment. + Bool ShaderTranslationCacheEnabled(); + + // Serializes the exact bytes of a cache key. Every appender is + // length-prefixed or fixed-width, so no two different input tuples can + // serialize to the same byte string by running into each other. + class TranslationKeyBuilder { + public: + void Bytes(const void* data, SizeT length); + + template + void Value(const T& value) { + static_assert(std::is_trivially_copyable_v, + "TranslationKeyBuilder::Value hashes the object representation"); + Bytes(&value, sizeof(T)); + } + + // Length-prefixed, so "ab"+"c" and "a"+"bc" cannot collide. + void Text(StringView text); + void Words(const Vector& words); + + // Hash maps and sets are serialized in SORTED order, never in iteration + // order: ska::flat_hash_map's iteration order depends on insertion history + // and capacity, so two logically identical maps could otherwise serialize + // differently and cause spurious misses. Sorting makes the blob canonical. + // These maps are all tiny (explicit locations, image formats, storage-block + // rebindings), so the sort is free. + template + void NameMap(const UnorderedMap& map) { + static_assert(std::is_trivially_copyable_v); + Vector> sorted; + sorted.reserve(map.size()); + for (const auto& [name, value] : map) sorted.emplace_back(StringView(name), value); + std::sort(sorted.begin(), sorted.end(), + [](const auto& a, const auto& b) { return a.first < b.first; }); + Value(static_cast(sorted.size())); + for (const auto& [name, value] : sorted) { + Text(name); + Value(value); + } + } + + // std::set is already ordered, but it gets the same length prefix. + void NameSet(const std::set& names); + + const String& Blob() const { return m_blob; } + String Take() { return Move(m_blob); } + + private: + String m_blob; + }; + + // A cache key: the full bytes, plus the hash that selects a bucket for them. + // The blob is shared rather than copied so that indexing an entry by its key + // does not double the memory a 100 KB shaderpack stage costs. + struct TranslationCacheKey { + Uint64 hash = 0; + SharedPtr blob; + + Bool Valid() const { return blob != nullptr; } + SizeT Bytes() const { return blob ? blob->size() : 0u; } + + // FULL comparison, always. This is what makes a hash collision a miss + // rather than a miscompiled shader. + Bool operator==(const TranslationCacheKey& other) const { + if (hash != other.hash) return false; + if (blob == other.blob) return true; // the same buffer + if (!blob || !other.blob) return false; + return *blob == *other.blob; + } + }; + + struct TranslationCacheKeyHasher { + SizeT operator()(const TranslationCacheKey& key) const { return static_cast(key.hash); } + }; + + // Seals a builder's bytes into a key. + TranslationCacheKey MakeTranslationCacheKey(String blob); + inline TranslationCacheKey MakeTranslationCacheKey(TranslationKeyBuilder& builder) { + return MakeTranslationCacheKey(builder.Take()); + } + + struct TranslationCacheStats { + Uint64 hits = 0; + Uint64 misses = 0; + Uint64 inserts = 0; + Uint64 evictions = 0; + // Entries whose own key+payload already exceed the whole byte budget. + // Caching one would evict everything else and then itself. + Uint64 rejectedOversize = 0; + // Two workers missed on the same key and both computed it. Harmless (the + // key covers every input, so both results are equal), but worth counting: + // a large number would mean the redundancy is no longer a startup artifact. + Uint64 duplicateInserts = 0; + }; + + // A bounded, thread-safe, process-lifetime memo. + // + // EVICTION is FIFO, bounded by BOTH an entry count and a stored-byte budget, + // whichever binds first - the same policy (and the same reasoning) as + // ShaderPreprocessCache. Translation workloads are bursts of mostly-distinct + // inputs whose reuse clusters around insertion time, and FIFO keeps Find() a + // read-only operation: with N pool workers hammering the same cache, an LRU + // splice on every hit would turn the shared hit path into a writer. + // + // THREAD SAFETY. The mutex guards the containers only; the expensive + // translation always runs OUTSIDE it, between the Find and the Insert. Two + // workers that miss on the same key therefore both compute it, and the second + // Insert is dropped. That is deliberate: the alternative - one worker waits + // for the other's result - would block a pool worker inside a job body, which + // is precisely the invariant (JobNode I4) that keeps ShaderCompilePool from + // deadlocking when the waiting job holds the only worker the awaited job needs. + // The waste is bounded by the worker count and only happens on the first burst. + // + // LIFETIME. Hits hand out shared ownership of the payload, never a pointer into + // the entry list, so a reader keeps its payload alive across any concurrent + // eviction - and across Clear() and the cache's own destruction. + template + class BoundedTranslationCache { + public: + using PayloadPtr = SharedPtr; + + BoundedTranslationCache(const char* name, SizeT maxEntries, SizeT maxBytes) + : m_name(name), m_maxEntries(maxEntries), m_maxBytes(maxBytes) {} + + PayloadPtr Find(const TranslationCacheKey& key) const { + if (!key.Valid()) return nullptr; + const std::lock_guard lock(m_mutex); + const auto it = m_index.find(key); + if (it == m_index.end()) { + ++m_stats.misses; + return nullptr; + } + ++m_stats.hits; + return it->second->payload; + } + + void Insert(TranslationCacheKey key, PayloadPtr payload, SizeT payloadBytes) { + if (!key.Valid() || !payload) return; + const SizeT entryBytes = key.Bytes() + payloadBytes; + const std::lock_guard lock(m_mutex); + if (entryBytes > m_maxBytes) { + ++m_stats.rejectedOversize; + return; + } + if (m_index.find(key) != m_index.end()) { + // A concurrent miss on the same key computed it too. The incumbent + // is kept: the key covers every input, so the two payloads are + // equal, and replacing would only move a demonstrably-wanted entry + // to the back of the FIFO. + ++m_stats.duplicateInserts; + return; + } + m_entries.push_back(Entry{key, Move(payload), entryBytes}); + m_index.emplace(Move(key), std::prev(m_entries.end())); + m_storedBytes += entryBytes; + ++m_stats.inserts; + EvictUntilWithinBudgetLocked(); + } + + void Clear() { + const std::lock_guard lock(m_mutex); + m_index.clear(); + m_entries.clear(); + m_storedBytes = 0; + } + + TranslationCacheStats Stats() const { + const std::lock_guard lock(m_mutex); + return m_stats; + } + + SizeT EntryCount() const { + const std::lock_guard lock(m_mutex); + return m_entries.size(); + } + + SizeT StoredBytes() const { + const std::lock_guard lock(m_mutex); + return m_storedBytes; + } + + // MGLOG_D, so an INFO build compiles this out entirely. + void LogStats() const { + const TranslationCacheStats stats = Stats(); + const Uint64 lookups = stats.hits + stats.misses; + MGLOG_D("%s: %llu/%llu hits (%.1f%%), %llu inserts, %llu evictions, %llu oversize, " + "%llu duplicate, %zu entries / %zu KiB", + m_name, static_cast(stats.hits), + static_cast(lookups), + lookups ? 100.0 * static_cast(stats.hits) / static_cast(lookups) : 0.0, + static_cast(stats.inserts), + static_cast(stats.evictions), + static_cast(stats.rejectedOversize), + static_cast(stats.duplicateInserts), EntryCount(), + StoredBytes() / 1024u); + } + + // Tests only: makes the caps small enough to exercise eviction without + // building megabytes of shaders. Clears the cache, because shrinking the + // caps under live entries would otherwise leave it over budget. + void SetCapsForTesting(SizeT maxEntries, SizeT maxBytes) { + const std::lock_guard lock(m_mutex); + m_maxEntries = maxEntries; + m_maxBytes = maxBytes; + m_index.clear(); + m_entries.clear(); + m_storedBytes = 0; + m_stats = {}; + } + + private: + struct Entry { + TranslationCacheKey key; + PayloadPtr payload; + SizeT bytes = 0; + }; + using EntryList = std::list; + + void EvictUntilWithinBudgetLocked() { + while (!m_entries.empty() && + (m_entries.size() > m_maxEntries || m_storedBytes > m_maxBytes)) { + const auto victim = m_entries.begin(); + m_storedBytes -= victim->bytes; + m_index.erase(victim->key); + m_entries.erase(victim); + ++m_stats.evictions; + } + } + + const char* m_name = ""; + SizeT m_maxEntries = 0; + SizeT m_maxBytes = 0; + + mutable std::mutex m_mutex; + mutable TranslationCacheStats m_stats; + EntryList m_entries; // front = oldest = FIFO victim + UnorderedMap m_index; + SizeT m_storedBytes = 0; + }; + + // ======================================================================= + // L1 - the FRONT END: parsed GLSL program -> sanitized SPIR-V modules. + // ======================================================================= + // + // The cached artifact is the module AFTER SanitizeAndOptimizeBinary, not the + // raw GlslangToSpv output. That is a deliberate choice and it is safe: + // SanitizeAndOptimizeBinary is a fixed 11-pass spirv-opt chain with no + // arguments but the module, and its two remaining parameters (`validateOutput`, + // `enableSpirvValidation`) only decide whether the OUTPUT is handed to the + // validator and logged - RunOptimizerChecked runs the optimizer first and + // identically either way. Nothing between GlslangToSpv and Sanitize reads + // backend state. So caching after Sanitize saves the 96 us/stage the chain + // costs on top of the 40 us GlslangToSpv, and gives the backends exactly the + // bytes they would have got. + // + // WHAT IS IN THE KEY (each one is an input that can change the modules): + // * the CompileEnv fingerprint - covers the glslang resource limits + // (BuildTBuiltInResource reads env->params), the backend identity, the + // advertised extension set and the compute limits; + // * per stage, in link order: the GL stage enum and the FULL preprocessed + // source, which is literally the text ParseShaderSource was given; + // * the four link-time request maps mapIO resolves against + // (glBindAttribLocation / glBindFragDataLocation / + // glBindFragDataLocationIndexed, and the merged layout(binding=) opaque + // units) - these steer TMglGlslIoResolver and therefore the Locations and + // Bindings baked into every module; + // * the ShaderCompileBits the parse ran under (always 0 in production; in + // the key so a future non-zero value cannot alias); + // * the SPIR-V validation switch (byte-identical output either way, but it + // costs one byte to be sure). + // + // The key is a PROGRAM-level key, not a per-stage one, and that is forced: + // glslang's mapIO resolves a fragment stage's input Locations against the + // vertex stage's outputs, so a stage's SPIR-V is NOT a function of that + // stage's source alone. A per-stage key here would be exactly the silent + // miscompile this cache must never produce. + struct SpirvTranslationResult { + // One module per stage, in the same order as ProgramLinkTask's + // spirvHandoff.shaderTypes. + Vector> modules; + }; + using SpirvTranslationResultPtr = SharedPtr; + + struct SpirvTranslationKeyInputs { + struct Stage { + GLenum type = 0; + StringView preprocessedSource; + }; + + Uint64 envFingerprint = 0; + Vector stages; + const UnorderedMap* explicitVertexInLocations = nullptr; + const UnorderedMap* explicitFragmentOutLocations = nullptr; + const UnorderedMap* explicitFragmentOutIndices = nullptr; + const UnorderedMap* explicitOpaqueUniformBindings = nullptr; + Uint32 shaderCompileFlags = 0; + Bool enableSpirvValidation = false; + }; + + TranslationCacheKey BuildSpirvTranslationKey(const SpirvTranslationKeyInputs& inputs); + SizeT SpirvTranslationResultBytes(const SpirvTranslationResult& result); + + // Process-global, and safe to be: the CompileEnv fingerprint is in the key, so + // a module computed under one context's limits can never be handed to another + // context with different ones. Global rather than per-context because the + // producer (ProgramSpirvTask) runs on a pool worker and must not reach + // MG_State::pGLContext. + BoundedTranslationCache& GetSpirvTranslationCache(); + + // ======================================================================= + // L2 - the BACK END: sanitized SPIR-V -> DirectGLES ESSL payload. + // ======================================================================= + // + // DIRECTGLES ONLY. DirectVulkan runs a different pass chain, steered by Vulkan + // device features, and gets no L2 in this change; giving it one means giving it + // its OWN instance with its OWN key, never this one. + // + // The memoized segment is BackendProgramObjectImpl::SyncToBackend's per-stage + // block from the draw-parameter lowering down to (and including) the + // SPIRV-Cross Compile() that produces the ESSL text. The text-level passes that + // follow it are deliberately outside: they are cheap string work, and they read + // a long tail of live per-program state (RebindImageUniformsToFrontendUnits + // walks the ProgramObject's uniform reflection, the norm-clamp masks and the + // fragColor broadcast count are live globals) whose inclusion would make the + // key both huge and fragile for no measurable saving. + // + // WHAT IS IN THE KEY: + // * the FULL SPIR-V module (the input); + // * the GL stage enum - three passes are stage-gated (draw parameters and + // array vertex inputs on vertex, fragment-output index legalization on + // fragment); + // * SupportsViewportArray - arms LowerViewportIndexForEssl; + // * the four sample ceilings (color / integer / depth / advertised) - both + // ARM ClampMultisampleFetchesForEssl and PARAMETERIZE it; + // * SupportsNoperspectiveInterpolation - arms EmulateNoPerspectiveForEssl; + // * the transform-feedback capture block names - the argument to + // FlattenXfbInterfaceBlocksForEssl, and the reason the payload has to carry + // the names it actually flattened; + // * the image-format bake map (uniform name -> GL internal format), which is + // derived from LIVE glBindImageTexture state and is the one genuinely + // per-draw-state input in here; + // * the storage-block binding overrides handed to SPIRV-Cross; + // * the ESSL version SPIRV-Cross targets (ResolveBackendEsslVersion, i.e. the + // driver's GLES version) - the remaining two SPIRV-Cross options are + // compile-time constants (GLSL_ES true, VULKAN_SEMANTICS false); + // * the SPIR-V validation switch, as in L1. + // + // Unconditional passes (StripUboMemberRelaxedPrecision, LowerRectImages, + // Lower1DArrayImages) take no input but the module and so need no key material. + struct EsslTranslationResult { + String essl; + // Which interface blocks FlattenXfbInterfaceBlocksForEssl actually rewrote + // in THIS stage. The caller unions these across stages and the transform- + // feedback capture list follows them, so a payload that dropped them would + // silently un-rename every capture on a cache hit. + std::set flattenedXfbBlockNames; + }; + using EsslTranslationResultPtr = SharedPtr; + + struct EsslTranslationKeyInputs { + const Vector* spirv = nullptr; + GLenum shaderType = 0; + + // --- driver capability bits that arm or steer a pass --- + Bool supportsViewportArray = false; + Bool supportsNoperspectiveInterpolation = false; + Int32 maxColorTextureSamples = 0; + Int32 maxIntegerSamples = 0; + Int32 maxDepthTextureSamples = 0; + Int32 advertisedMaxSamples = 0; + + // --- per-program / per-context inputs --- + const std::set* xfbCaptureBlockNames = nullptr; + const UnorderedMap* glFormatByUniformName = nullptr; + const UnorderedMap* storageBlockBindingOverrides = nullptr; + + // --- SPIRV-Cross options --- + Uint esslVersion = 300; + + Bool enableSpirvValidation = false; + }; + + TranslationCacheKey BuildEsslTranslationKey(const EsslTranslationKeyInputs& inputs); + SizeT EsslTranslationResultBytes(const EsslTranslationResult& result); + + BoundedTranslationCache& GetEsslTranslationCache(); + + // Drops both levels. Called from the same teardown that resets the glslang + // prewarm latch: nothing here holds a glslang object, so this is RSS hygiene + // rather than a correctness requirement. + void ClearShaderTranslationCaches(); + + // One MGLOG_D line per level. Called at teardown and cheap enough to call from + // a test. + void LogShaderTranslationCacheStats(); +} // namespace MobileGL::MG_Util::ShaderTranspiler