diff --git a/.github/workflows/apk.yml b/.github/workflows/apk.yml index d8cb80e8..aeb42efe 100644 --- a/.github/workflows/apk.yml +++ b/.github/workflows/apk.yml @@ -420,6 +420,9 @@ jobs: MOBILEGL_USE_ANGLE: ${{ matrix.backend.name == 'DirectGLES' && '1' || '0' }} MOBILEGL_TRACE_ANGLE_VARIANT: ${{ matrix.case.name == 'minecraft-1.21.4-fabric-iris-bliss-in-world' && '90a62123d794' || 'ec889e6ea831' }} MOBILEGL_MAGMA_R11G11B10F_FALLBACK: ${{ matrix.backend.name == 'DirectVulkan' && '1' || '0' }} + MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }} + MOBILEGL_DERIVE_NUM_SUBGROUPS: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }} + MOBILEGL_ITERATIONRP_FIX_BARRIER: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }} run: | apk_file="android-retrace-apks/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk" test -f "${apk_file}" diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index c325393c..c977505f 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -639,6 +639,12 @@ jobs: if [ '${{ matrix.backend }}' = 'DirectVulkan' ]; then export MOBILEGL_MAGMA_R11G11B10F_FALLBACK=1 fi + if [ '${{ matrix.backend }}' = 'DirectVulkan' ] \ + && [ '${{ matrix.case }}' = 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' ]; then + export MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH=1 + export MOBILEGL_DERIVE_NUM_SUBGROUPS=1 + export MOBILEGL_ITERATIONRP_FIX_BARRIER=1 + fi # The blended depth-write quirk auto-enables only on Qualcomm, which no CI # runner has, so force it on for the OIT case it exists to fix. ForceOn # bypasses only the vendor gate, so this exercises the real strip on diff --git a/CMakeLists.txt b/CMakeLists.txt index d9873406..93efd897 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -285,6 +285,7 @@ set(SOURCE_FILES MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DeriveNumSubgroupsPass.cpp + MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPSubgroupScratchPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateSubgroupsPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp diff --git a/MobileGL/Config.h b/MobileGL/Config.h index ff659bd0..1ccfa8bb 100644 --- a/MobileGL/Config.h +++ b/MobileGL/Config.h @@ -100,6 +100,10 @@ namespace MobileGL::MG_Config { // itself on >= 16-lane devices. Auto is ON; ForceOff replays the pack's bug // verbatim. QuirkOverride FixIterationRPSubgroupScratch = QuirkOverride::Auto; + // MOBILEGL_ITERATIONRP_FIX_BARRIER: repair Program 203's missing workgroup + // rendezvous between its two reductions over prefixSumCache. Off by default and + // fingerprint-gated by FixIterationRPBarrierPass when enabled. + Bool IterationRPFixBarrier = false; // MOBILEGL_DERIVE_NUM_SUBGROUPS: replace compute gl_NumSubgroups loads with // ceil(workgroup invocations / gl_SubgroupSize) on the NATIVE subgroup path // (ShaderTranspiler::DeriveNumSubgroupsPass). Auto is ON: GL requires diff --git a/MobileGL/ConfigLoader.cpp b/MobileGL/ConfigLoader.cpp index f23a43d5..cb1fe00f 100644 --- a/MobileGL/ConfigLoader.cpp +++ b/MobileGL/ConfigLoader.cpp @@ -171,6 +171,7 @@ namespace MobileGL::MG_ConfigLoader { features.MagmaEmulateSubgroup = QueryEnvFlag("MOBILEGL_MAGMA_EMULATE_SUBGROUP"); features.FixIterationRPSubgroupScratch = QueryEnvQuirkOverride("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH"); + features.IterationRPFixBarrier = QueryEnvFlag("MOBILEGL_ITERATIONRP_FIX_BARRIER"); features.DeriveNumSubgroups = QueryEnvQuirkOverride("MOBILEGL_DERIVE_NUM_SUBGROUPS"); features.AdvertiseFp64 = QueryEnvFlag("MOBILEGL_ADVERTISE_FP64"); features.MagmaR11G11B10FFallback = QueryEnvFlag("MOBILEGL_MAGMA_R11G11B10F_FALLBACK"); diff --git a/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.cpp b/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.cpp index 38291867..de5214fc 100644 --- a/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.cpp +++ b/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.cpp @@ -3190,10 +3190,25 @@ namespace MobileGL::MG_Backend::DirectVulkan { } // GL_KHR_shader_subgroup handling (SubgroupSupportPolicy.h). Native subgroup - // operations execute natively; two module repairs keep the GL contract intact + // operations execute natively; module repairs keep the GL contract intact // around them. The opt-in emulation path replaces them only on devices with no // subgroup support at all (MOBILEGL_MAGMA_EMULATE_SUBGROUP). if (shaders[i] && shaders[i]->GetShaderStage() == ShaderStage::Compute) { + // Program 203 broadcasts the first reduction through + // prefixSumCache[0], then lets the second reduction overwrite that + // scratch without first rendezvousing all readers. Patch that exact + // fingerprint before either native or emulated subgroup lowering. + if (m_subgroupPolicy.fixIterationRPBarrier) { + Vector patchedSpirv; + if (MG_Util::ShaderTranspiler::ShaderCompiler::FixIterationRPBarrierForVulkan( + moduleSpirvs[i], patchedSpirv, enableSpirvValidation)) { + moduleSpirvs[i] = std::move(patchedSpirv); + } else { + MGLOG_E("ProgramFactory: iterationRP barrier patch failed for program %u; " + "Program 203 keeps its shared-scratch race", + program.GetExternalIndex()); + } + } if (m_subgroupPolicy.emulateSubgroups) { Vector emulatedSpirv; if (MG_Util::ShaderTranspiler::ShaderCompiler::EmulateSubgroupsForVulkan( diff --git a/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.h b/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.h index 560aec89..2fe6c62f 100644 --- a/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.h +++ b/MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.h @@ -381,6 +381,7 @@ namespace MobileGL::MG_Backend::DirectVulkan { struct SubgroupLoweringPolicy { Bool emulateSubgroups = false; // MOBILEGL_MAGMA_EMULATE_SUBGROUP, no-native-support devices Bool fixIterationRPSubgroupScratch = false; // patch iterationRP's under-declared scratch + Bool fixIterationRPBarrier = false; // repair Program 203's shared-scratch race Bool deriveNumSubgroups = false; // repair the NumSubgroups builtin Bool requireFullSubgroups = false; // computeFullSubgroups enabled on the device Uint32 nativeSubgroupSize = 0; diff --git a/MobileGL/MG_Backend/DirectVulkan/Renderer/VulkanRenderer.cpp b/MobileGL/MG_Backend/DirectVulkan/Renderer/VulkanRenderer.cpp index f6f24e6f..4fd23b38 100644 --- a/MobileGL/MG_Backend/DirectVulkan/Renderer/VulkanRenderer.cpp +++ b/MobileGL/MG_Backend/DirectVulkan/Renderer/VulkanRenderer.cpp @@ -3063,6 +3063,7 @@ void main() { subgroupPolicy.emulateSubgroups = ShouldEmulateSubgroups(m_nativeSubgroupSupported); subgroupPolicy.fixIterationRPSubgroupScratch = m_nativeSubgroupSupported && ShouldFixIterationRPSubgroupScratch(); + subgroupPolicy.fixIterationRPBarrier = ShouldFixIterationRPBarrier(); subgroupPolicy.deriveNumSubgroups = m_nativeSubgroupSupported && ShouldDeriveNumSubgroups(); subgroupPolicy.requireFullSubgroups = m_computeFullSubgroupsFeatureEnabled; diff --git a/MobileGL/MG_Backend/DirectVulkan/SubgroupSupportPolicy.h b/MobileGL/MG_Backend/DirectVulkan/SubgroupSupportPolicy.h index 0f06e7bc..f5f48e74 100644 --- a/MobileGL/MG_Backend/DirectVulkan/SubgroupSupportPolicy.h +++ b/MobileGL/MG_Backend/DirectVulkan/SubgroupSupportPolicy.h @@ -18,9 +18,11 @@ namespace MobileGL::MG_Backend::DirectVulkan { // // Native subgroups are the implementation whenever the device has them, whatever // their width - subgroup operations execute on the hardware paths they were made - // for. Two module-level repairs keep the GL contract intact around them: + // for. Module-level repairs keep the GL contract intact around them: // - FixIterationRPSubgroupScratchPass patches the one known pack bug: iterationRP's // prefixSumCache[32], under-declared for sub-16-lane devices (8-lane lavapipe); + // - FixIterationRPBarrierPass repairs Program 203's race between two reductions + // reusing that scratch, when explicitly enabled; // - DeriveNumSubgroupsPass replaces the one builtin drivers get wrong // (gl_NumSubgroups) with the value the rest of the topology implies. // The 32-lane shared-memory emulation (EmulateSubgroupsPass) is a LAST RESORT for @@ -47,6 +49,10 @@ namespace MobileGL::MG_Backend::DirectVulkan { MG_Config::QuirkOverride::ForceOff; } + inline Bool ShouldFixIterationRPBarrier() { + return MG_Config::Features.IterationRPFixBarrier; + } + inline Bool ShouldDeriveNumSubgroups() { // Auto is ON: gl_NumSubgroups must agree with the gl_SubgroupID range for the GL // contract to hold, and the derived ceil() value is the one the renderer can pin diff --git a/MobileGL/MG_Test/ShaderTranspiler/CMakeLists.txt b/MobileGL/MG_Test/ShaderTranspiler/CMakeLists.txt index 9f16aa6f..fd389fb4 100644 --- a/MobileGL/MG_Test/ShaderTranspiler/CMakeLists.txt +++ b/MobileGL/MG_Test/ShaderTranspiler/CMakeLists.txt @@ -4,6 +4,7 @@ add_executable( SpirvPassTest SpirvPassTest.cpp DeriveNumSubgroupsTest.cpp + FixIterationRPBarrierTest.cpp FixIterationRPSubgroupScratchTest.cpp EmulateSubgroupsTest.cpp DemoteFloat64Test.cpp diff --git a/MobileGL/MG_Test/ShaderTranspiler/FixIterationRPBarrierTest.cpp b/MobileGL/MG_Test/ShaderTranspiler/FixIterationRPBarrierTest.cpp new file mode 100644 index 00000000..bb6004b0 --- /dev/null +++ b/MobileGL/MG_Test/ShaderTranspiler/FixIterationRPBarrierTest.cpp @@ -0,0 +1,206 @@ +// MobileGL - MobileGL/MG_Test/ShaderTranspiler/FixIterationRPBarrierTest.cpp +// Copyright (c) 2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#include + +#define SPV_ENABLE_UTILITY_CODE +#include "glslang/SPIRV/spirv.hpp11" +#undef SPV_ENABLE_UTILITY_CODE + +#include "Includes.h" +#include +#include + +#include + +#include +#include + +using namespace MobileGL; +using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler; + +namespace { + constexpr SizeT kSpirvHeaderWordCount = 5u; + + template + void ForEachInstruction(const Vector& spirv, Visitor&& visit) { + for (SizeT offset = kSpirvHeaderWordCount; offset < spirv.size();) { + const Uint32 wordCount = spirv[offset] >> 16u; + if (wordCount == 0u || offset + wordCount > spirv.size()) break; + visit(static_cast(spirv[offset] & 0xffffu), &spirv[offset], wordCount); + offset += wordCount; + } + } + + Vector CompileCompute(const String& source) { + using namespace MobileGL::MG_Util::ShaderTranspiler; + ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source}; + auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib); + EXPECT_TRUE(shaderResult) << (shaderResult ? String{} : shaderResult.error().log); + if (!shaderResult) return {}; + + ProgramAttrib programAttrib{.shaders = {shaderResult.value()}}; + auto programResult = ShaderCompiler::LinkProgram(programAttrib); + EXPECT_TRUE(programResult) << (programResult ? String{} : programResult.error().log); + if (!programResult) return {}; + + ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()}; + auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib); + EXPECT_TRUE(binaryResult) << (binaryResult ? String{} : binaryResult.error().log); + if (!binaryResult || binaryResult->empty()) return {}; + return binaryResult->front(); + } + + bool Validates(const Vector& spirv) { + spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1); + tools.SetMessageConsumer( + [](spv_message_level_t, const char*, const spv_position_t& position, const char* message) { + ADD_FAILURE() << "spirv-val at word " << position.index << ": " << message; + }); + return tools.Validate(spirv); + } + + Uint32 CountOpcode(const Vector& spirv, spv::Op wanted) { + Uint32 count = 0u; + ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32*, Uint32) { + if (opcode == wanted) ++count; + }); + return count; + } + + bool HasWorkgroupBarrierImmediatelyBeforeSecondScan(const Vector& spirv) { + std::map uintConstants; + spv::Op previous = spv::Op::OpNop; + Uint32 scanCount = 0u; + bool found = false; + const Uint32* previousWords = nullptr; + Uint32 previousWordCount = 0u; + ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32* words, Uint32 wordCount) { + if (opcode == spv::Op::OpConstant && wordCount >= 4u) { + uintConstants[words[2]] = words[3]; + } + if (opcode == spv::Op::OpGroupNonUniformFAdd && wordCount >= 6u && + static_cast(words[4]) == spv::GroupOperation::InclusiveScan && ++scanCount == 2u && + previous == spv::Op::OpControlBarrier && previousWordCount == 4u) { + found = + uintConstants[previousWords[1]] == static_cast(spv::Scope::Workgroup) && + uintConstants[previousWords[2]] == static_cast(spv::Scope::Workgroup) && + uintConstants[previousWords[3]] == (static_cast(spv::MemorySemanticsMask::AcquireRelease) | + static_cast(spv::MemorySemanticsMask::WorkgroupMemory)); + } + previous = opcode; + previousWords = words; + previousWordCount = wordCount; + }); + return found; + } + + constexpr const char* kProgram203RaceShape = R"(#version 450 core +#extension GL_KHR_shader_subgroup_basic : require +#extension GL_KHR_shader_subgroup_arithmetic : require +layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in; +layout(std430, binding = 0) buffer Output { vec2 value; } outputData; +shared vec2 prefixSumCache[32]; +void main() { + vec2 sampleLuminance = subgroupInclusiveAdd( + vec2(float(gl_LocalInvocationIndex), 1.0)); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) + prefixSumCache[gl_SubgroupID] = sampleLuminance; + barrier(); + if (gl_LocalInvocationIndex == 511u) + prefixSumCache[0] = sampleLuminance / 512.0; + barrier(); + + float avg = prefixSumCache[0].x; + float weight = avg > 0.0 ? float(gl_LocalInvocationIndex + 1u) / avg : 0.0; + vec2 sampleExposure = subgroupInclusiveAdd(vec2(weight, 1.0)); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) + prefixSumCache[gl_SubgroupID] = sampleExposure; + barrier(); + if (gl_LocalInvocationIndex == 511u) + outputData.value = sampleExposure; +} +)"; + + constexpr const char* kAlreadySynchronizedShape = R"(#version 450 core +#extension GL_KHR_shader_subgroup_basic : require +#extension GL_KHR_shader_subgroup_arithmetic : require +layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in; +layout(std430, binding = 0) buffer Output { vec2 value; } outputData; +shared vec2 prefixSumCache[32]; +void main() { + vec2 first = subgroupInclusiveAdd(vec2(float(gl_LocalInvocationIndex), 1.0)); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) + prefixSumCache[gl_SubgroupID] = first; + barrier(); + if (gl_LocalInvocationIndex == 511u) prefixSumCache[0] = first / 512.0; + barrier(); + float avg = prefixSumCache[0].x; + barrier(); + vec2 second = subgroupInclusiveAdd(vec2(avg, 1.0)); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) + prefixSumCache[gl_SubgroupID] = second; + barrier(); + if (gl_LocalInvocationIndex == 511u) outputData.value = second; +} +)"; + + constexpr const char* kForeignSingleScanShape = R"(#version 450 core +#extension GL_KHR_shader_subgroup_basic : require +#extension GL_KHR_shader_subgroup_arithmetic : require +layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in; +layout(std430, binding = 0) buffer Output { vec2 value; } outputData; +shared vec2 prefixSumCache[32]; +void main() { + vec2 value = subgroupInclusiveAdd(vec2(float(gl_LocalInvocationIndex), 1.0)); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) + prefixSumCache[gl_SubgroupID] = value; + barrier(); + if (gl_LocalInvocationIndex == 0u) outputData.value = prefixSumCache[0]; +} +)"; +} // namespace + +TEST(FixIterationRPBarrierPass, InsertsWorkgroupBarrierBeforeSecondReduction) { + const Vector input = CompileCompute(kProgram203RaceShape); + ASSERT_FALSE(input.empty()); + const Uint32 inputBarrierCount = CountOpcode(input, spv::Op::OpControlBarrier); + EXPECT_FALSE(HasWorkgroupBarrierImmediatelyBeforeSecondScan(input)); + + Vector output; + ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, output, true)); + EXPECT_EQ(CountOpcode(output, spv::Op::OpControlBarrier), inputBarrierCount + 1u); + EXPECT_TRUE(HasWorkgroupBarrierImmediatelyBeforeSecondScan(output)); + EXPECT_TRUE(Validates(output)); +} + +TEST(FixIterationRPBarrierPass, LeavesOtherShapesByteIdentical) { + const Vector input = CompileCompute(kForeignSingleScanShape); + ASSERT_FALSE(input.empty()); + Vector output; + ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, output, true)); + EXPECT_EQ(output, input); +} + +TEST(FixIterationRPBarrierPass, LeavesAnAlreadySynchronizedShaderByteIdentical) { + const Vector input = CompileCompute(kAlreadySynchronizedShape); + ASSERT_FALSE(input.empty()); + Vector output; + ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, output, true)); + EXPECT_EQ(output, input); +} + +TEST(FixIterationRPBarrierPass, IsIdempotent) { + const Vector input = CompileCompute(kProgram203RaceShape); + ASSERT_FALSE(input.empty()); + Vector once; + ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, once, true)); + Vector twice; + ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(once, twice, true)); + EXPECT_EQ(twice, once); +} diff --git a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp index 73d4f025..49e10680 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp +++ b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp @@ -27,6 +27,7 @@ #include "SpirvPasses/ZeroBaseVertexPass.h" #include "SpirvPasses/DeriveNumSubgroupsPass.h" #include "SpirvPasses/EmulateSubgroupsPass.h" +#include "SpirvPasses/FixIterationRPBarrierPass.h" #include "SpirvPasses/FixIterationRPSubgroupScratchPass.h" #include "SpirvPasses/NormalizeRectCoordinatesPass.h" #include "SpirvPasses/Lower1DArrayImagesPass.h" @@ -924,6 +925,17 @@ namespace MobileGL { inputBinary, outputBinary, true, enableSpirvValidation); } + bool ShaderCompiler::FixIterationRPBarrierForVulkan( + const Vector& inputBinary, Vector& outputBinary, + const bool enableSpirvValidation) { + using namespace spvtools; + Optimizer optimizer(SPV_ENV_VULKAN_1_1); + optimizer.RegisterPass(FixIterationRPBarrierPass::CreateFixIterationRPBarrierPass()); + + return RunOptimizerChecked("FixIterationRPBarrierForVulkan", optimizer, + inputBinary, outputBinary, true, enableSpirvValidation); + } + bool ShaderCompiler::DecoratePositionInvariantForVulkan(const Vector& inputBinary, Vector& outputBinary, const bool enableSpirvValidation) { using namespace spvtools; diff --git a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h index d3578124..6a371603 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h +++ b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h @@ -179,6 +179,12 @@ namespace MobileGL { Uint32 nativeSubgroupSize, Uint32 maxWorkgroupScratchBytes, bool enableSpirvValidation = false); + // Inserts the missing workgroup rendezvous between Program 203's two + // prefixSumCache reductions. Fingerprint-gated to the iterationRP shape; + // unrelated and already-repaired modules pass through byte-identical. + static bool FixIterationRPBarrierForVulkan(const Vector& inputBinary, + Vector& outputBinary, + bool enableSpirvValidation = false); // Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair // (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no // VK_FORMAT_R64*_SFLOAT is needed - lavapipe advertises none of them for vertex diff --git a/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp new file mode 100644 index 00000000..bedc89c3 --- /dev/null +++ b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp @@ -0,0 +1,232 @@ +// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp +// Copyright (c) 2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#include "FixIterationRPBarrierPass.h" + +#include "spirv.hpp" +#include "source/opt/constants.h" +#include "source/opt/def_use_manager.h" +#include "source/opt/instruction.h" +#include "source/opt/ir_context.h" +#include "source/opt/module.h" +#include "source/util/make_unique.h" + +#include + +namespace MobileGL::MG_Util::ShaderTranspiler { + namespace { + using spvtools::opt::Instruction; + using spvtools::opt::IRContext; + using spvtools::opt::Operand; + + const Instruction* RootVariable(IRContext* context, uint32_t pointerId) { + const Instruction* def = context->get_def_use_mgr()->GetDef(pointerId); + while (def != nullptr) { + switch (def->opcode()) { + case spv::Op::OpVariable: + return def; + case spv::Op::OpAccessChain: + case spv::Op::OpInBoundsAccessChain: + case spv::Op::OpCopyObject: + def = context->get_def_use_mgr()->GetDef(def->GetSingleWordInOperand(0)); + break; + default: + return nullptr; + } + } + return nullptr; + } + + bool IsUintConstant(IRContext* context, uint32_t id, uint32_t wanted) { + const Instruction* def = context->get_def_use_mgr()->GetDef(id); + return def != nullptr && def->opcode() == spv::Op::OpConstant && def->NumInOperands() == 1u && + def->GetSingleWordInOperand(0) == wanted; + } + + bool IsZeroElementPointer(IRContext* context, uint32_t pointerId, const Instruction** root) { + const Instruction* pointer = context->get_def_use_mgr()->GetDef(pointerId); + if (pointer == nullptr || + (pointer->opcode() != spv::Op::OpAccessChain && pointer->opcode() != spv::Op::OpInBoundsAccessChain) || + pointer->NumInOperands() < 2u) { + return false; + } + for (uint32_t i = 1u; i < pointer->NumInOperands(); ++i) { + if (!IsUintConstant(context, pointer->GetSingleWordInOperand(i), 0u)) return false; + } + *root = RootVariable(context, pointerId); + return *root != nullptr; + } + + bool IsWorkgroupVec2Array(IRContext* context, const Instruction* variable) { + if (variable == nullptr || variable->opcode() != spv::Op::OpVariable || variable->NumInOperands() < 1u || + static_cast(variable->GetSingleWordInOperand(0)) != spv::StorageClass::Workgroup) { + return false; + } + auto* defUseMgr = context->get_def_use_mgr(); + const Instruction* pointerType = defUseMgr->GetDef(variable->type_id()); + if (pointerType == nullptr || pointerType->opcode() != spv::Op::OpTypePointer || + pointerType->NumInOperands() < 2u) { + return false; + } + const Instruction* arrayType = defUseMgr->GetDef(pointerType->GetSingleWordInOperand(1)); + if (arrayType == nullptr || arrayType->opcode() != spv::Op::OpTypeArray || + arrayType->NumInOperands() < 2u) { + return false; + } + const Instruction* length = defUseMgr->GetDef(arrayType->GetSingleWordInOperand(1)); + if (length == nullptr || length->opcode() != spv::Op::OpConstant || length->NumInOperands() != 1u) { + return false; + } + const uint32_t arrayLength = length->GetSingleWordInOperand(0); + if (arrayLength < 32u || arrayLength > 512u) return false; + + const Instruction* vectorType = defUseMgr->GetDef(arrayType->GetSingleWordInOperand(0)); + if (vectorType == nullptr || vectorType->opcode() != spv::Op::OpTypeVector || + vectorType->NumInOperands() < 2u || vectorType->GetSingleWordInOperand(1) != 2u) { + return false; + } + const Instruction* scalarType = defUseMgr->GetDef(vectorType->GetSingleWordInOperand(0)); + return scalarType != nullptr && scalarType->opcode() == spv::Op::OpTypeFloat && + scalarType->NumInOperands() == 1u && scalarType->GetSingleWordInOperand(0) == 32u; + } + + bool IsVec2FloatInclusiveAdd(IRContext* context, const Instruction* inst) { + if (inst->opcode() != spv::Op::OpGroupNonUniformFAdd || inst->NumInOperands() < 3u || + static_cast(inst->GetSingleWordInOperand(1)) != + spv::GroupOperation::InclusiveScan) { + return false; + } + const Instruction* vectorType = context->get_def_use_mgr()->GetDef(inst->type_id()); + if (vectorType == nullptr || vectorType->opcode() != spv::Op::OpTypeVector || + vectorType->NumInOperands() < 2u || vectorType->GetSingleWordInOperand(1) != 2u) { + return false; + } + const Instruction* scalarType = context->get_def_use_mgr()->GetDef(vectorType->GetSingleWordInOperand(0)); + return scalarType != nullptr && scalarType->opcode() == spv::Op::OpTypeFloat && + scalarType->NumInOperands() == 1u && scalarType->GetSingleWordInOperand(0) == 32u; + } + + bool HasProgram203LocalSize(IRContext* context) { + for (const Instruction& entryPoint : context->module()->entry_points()) { + if (static_cast(entryPoint.GetSingleWordInOperand(0)) != + spv::ExecutionModel::GLCompute) { + return false; + } + } + for (const Instruction& mode : context->module()->execution_modes()) { + if (mode.opcode() == spv::Op::OpExecutionMode && mode.NumInOperands() >= 5u && + static_cast(mode.GetSingleWordInOperand(1)) == spv::ExecutionMode::LocalSize) { + return mode.GetSingleWordInOperand(2) == 32u && mode.GetSingleWordInOperand(3) == 16u && + mode.GetSingleWordInOperand(4) == 1u; + } + } + return false; + } + + bool IsStoreToRoot(IRContext* context, const Instruction* inst, const Instruction* root) { + return inst->opcode() == spv::Op::OpStore && inst->NumInOperands() >= 2u && + RootVariable(context, inst->GetSingleWordInOperand(0)) == root; + } + } // namespace + + spvtools::opt::Pass::Status FixIterationRPBarrierPass::Process() { + auto* irContext = context(); + if (!HasProgram203LocalSize(irContext)) return Status::SuccessWithoutChange; + + for (auto& function : *irContext->module()) { + std::vector instructions; + std::vector scans; + for (auto& block : function) { + for (auto& inst : block) { + if (IsVec2FloatInclusiveAdd(irContext, &inst)) scans.push_back(instructions.size()); + instructions.push_back(&inst); + } + } + // Program 203 has exactly two vec2 inclusive adds: the luminance reduction + // and the weighted-exposure reduction. More or fewer is not our fingerprint. + if (scans.size() != 2u) continue; + + const size_t firstScan = scans[0]; + const size_t secondScan = scans[1]; + const Instruction* scratch = nullptr; + size_t averageLoad = instructions.size(); + + for (size_t i = firstScan + 1u; i < secondScan; ++i) { + Instruction* inst = instructions[i]; + if (inst->opcode() != spv::Op::OpLoad || inst->NumInOperands() < 1u) continue; + const Instruction* root = nullptr; + if (!IsZeroElementPointer(irContext, inst->GetSingleWordInOperand(0), &root) || + !IsWorkgroupVec2Array(irContext, root)) { + continue; + } + // The broadcast is read as prefixSumCache[0].x, hence a scalar load. + const Instruction* type = irContext->get_def_use_mgr()->GetDef(inst->type_id()); + if (type == nullptr || type->opcode() != spv::Op::OpTypeFloat || type->NumInOperands() != 1u || + type->GetSingleWordInOperand(0) != 32u) { + continue; + } + scratch = root; + averageLoad = i; + break; + } + if (scratch == nullptr) continue; + + bool sawZeroBroadcastStore = false; + bool sawPublishBarrier = false; + for (size_t i = firstScan + 1u; i < averageLoad; ++i) { + const Instruction* root = nullptr; + if (instructions[i]->opcode() == spv::Op::OpStore && + IsZeroElementPointer(irContext, instructions[i]->GetSingleWordInOperand(0), &root) && + root == scratch) { + sawZeroBroadcastStore = true; + } else if (sawZeroBroadcastStore && instructions[i]->opcode() == spv::Op::OpControlBarrier) { + sawPublishBarrier = true; + } + } + if (!sawZeroBroadcastStore || !sawPublishBarrier) continue; + + bool alreadySynchronized = false; + for (size_t i = averageLoad + 1u; i < secondScan; ++i) { + if (instructions[i]->opcode() == spv::Op::OpControlBarrier) { + alreadySynchronized = true; + break; + } + } + if (alreadySynchronized) return Status::SuccessWithoutChange; + + bool secondPhaseReusesScratch = false; + for (size_t i = secondScan + 1u; i < instructions.size(); ++i) { + if (IsStoreToRoot(irContext, instructions[i], scratch)) { + secondPhaseReusesScratch = true; + break; + } + } + if (!secondPhaseReusesScratch) continue; + + auto* constantMgr = irContext->get_constant_mgr(); + const uint32_t scopeId = constantMgr->GetUIntConstId(static_cast(spv::Scope::Workgroup)); + const uint32_t semanticsId = + constantMgr->GetUIntConstId(static_cast(spv::MemorySemanticsMask::AcquireRelease) | + static_cast(spv::MemorySemanticsMask::WorkgroupMemory)); + if (scopeId == 0u || semanticsId == 0u) return Status::Failure; + + instructions[secondScan]->InsertBefore(spvtools::MakeUnique( + irContext, spv::Op::OpControlBarrier, 0u, 0u, + Instruction::OperandList{Operand{SPV_OPERAND_TYPE_ID, {scopeId}}, + Operand{SPV_OPERAND_TYPE_ID, {scopeId}}, + Operand{SPV_OPERAND_TYPE_ID, {semanticsId}}})); + irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone); + return Status::SuccessWithChange; + } + return Status::SuccessWithoutChange; + } + + spvtools::Optimizer::PassToken FixIterationRPBarrierPass::CreateFixIterationRPBarrierPass() { + return spvtools::Optimizer::PassToken(spvtools::MakeUnique()); + } +} // namespace MobileGL::MG_Util::ShaderTranspiler diff --git a/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.h b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.h new file mode 100644 index 00000000..0c85ef7c --- /dev/null +++ b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.h @@ -0,0 +1,28 @@ +// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.h +// Copyright (c) 2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#pragma once + +#include "source/opt/pass.h" +#include "spirv-tools/optimizer.hpp" + +namespace MobileGL::MG_Util::ShaderTranspiler { + // Repairs iterationRP Program 203's missing workgroup rendezvous between two + // reductions that reuse prefixSumCache. The first phase broadcasts its result + // through prefixSumCache[0], but the second phase may overwrite that element before + // every invocation has read it. The pass fingerprints that exact two-scan, + // 512-invocation shape and inserts one Workgroup control barrier immediately before + // the second scan. Unrelated modules and already-repaired modules are byte-identical. + class FixIterationRPBarrierPass : public spvtools::opt::Pass { + public: + const char* name() const override { return "fix-iterationrp-barrier"; } + Status Process() override; + + static spvtools::Optimizer::PassToken CreateFixIterationRPBarrierPass(); + }; +} // namespace MobileGL::MG_Util::ShaderTranspiler diff --git a/android-plugin/app/src/trace/cpp/trace_replay_core.cpp b/android-plugin/app/src/trace/cpp/trace_replay_core.cpp index 151fa98c..8bd0a8b7 100644 --- a/android-plugin/app/src/trace/cpp/trace_replay_core.cpp +++ b/android-plugin/app/src/trace/cpp/trace_replay_core.cpp @@ -164,6 +164,21 @@ bool LoadMobileGL(const Request& request, std::string& error) { } else { unsetenv("MOBILEGL_COHERENT_AS_FLUSH"); } + if (request.fixIterationRPSubgroupScratch) { + setenv("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH", "1", 1); + } else { + unsetenv("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH"); + } + if (request.deriveNumSubgroups) { + setenv("MOBILEGL_DERIVE_NUM_SUBGROUPS", "1", 1); + } else { + unsetenv("MOBILEGL_DERIVE_NUM_SUBGROUPS"); + } + if (request.iterationRPFixBarrier) { + setenv("MOBILEGL_ITERATIONRP_FIX_BARRIER", "1", 1); + } else { + unsetenv("MOBILEGL_ITERATIONRP_FIX_BARRIER"); + } if (request.fboAttachmentDumps.empty()) { unsetenv("MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS"); } else { @@ -818,6 +833,10 @@ bool WriteResultJson(const Request& request, const Result& result) { << (request.avoidAngleLlvmpipeSamplerMipmapMinFilter ? "true" : "false") << ",\n"; file << " \"avoidAngleLlvmpipeExplicitLodBias\": " << (request.avoidAngleLlvmpipeExplicitLodBias ? "true" : "false") << ",\n"; + file << " \"fixIterationRPSubgroupScratch\": " << (request.fixIterationRPSubgroupScratch ? "true" : "false") + << ",\n"; + file << " \"deriveNumSubgroups\": " << (request.deriveNumSubgroups ? "true" : "false") << ",\n"; + file << " \"iterationRPFixBarrier\": " << (request.iterationRPFixBarrier ? "true" : "false") << ",\n"; file << " \"holdMs\": " << request.holdMs << ",\n"; file << " \"mismatchPixels\": " << result.mismatchPixels << "\n"; file << "}\n"; diff --git a/android-plugin/app/src/trace/cpp/trace_replay_core.hpp b/android-plugin/app/src/trace/cpp/trace_replay_core.hpp index cc550378..01267c93 100644 --- a/android-plugin/app/src/trace/cpp/trace_replay_core.hpp +++ b/android-plugin/app/src/trace/cpp/trace_replay_core.hpp @@ -44,6 +44,9 @@ struct Request { bool avoidAngleLlvmpipeSamplerMipmapMinFilter = false; bool avoidAngleLlvmpipeExplicitLodBias = false; bool coherentAsFlush = false; + bool fixIterationRPSubgroupScratch = false; + bool deriveNumSubgroups = false; + bool iterationRPFixBarrier = false; int holdMs = 0; }; diff --git a/android-plugin/app/src/trace/cpp/trace_replay_jni.cpp b/android-plugin/app/src/trace/cpp/trace_replay_jni.cpp index 8c6e1a6e..9b136039 100644 --- a/android-plugin/app/src/trace/cpp/trace_replay_jni.cpp +++ b/android-plugin/app/src/trace/cpp/trace_replay_jni.cpp @@ -122,6 +122,9 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv* jboolean avoidAngleLlvmpipeSamplerMipmapMinFilter, jboolean avoidAngleLlvmpipeExplicitLodBias, jboolean coherentAsFlush, + jboolean fixIterationRPSubgroupScratch, + jboolean deriveNumSubgroups, + jboolean iterationRPFixBarrier, jstring texture2dDumps) { mobilegl_trace::Request request; request.tracePath = ToString(env, tracePath); @@ -150,6 +153,9 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv* avoidAngleLlvmpipeSamplerMipmapMinFilter == JNI_TRUE; request.avoidAngleLlvmpipeExplicitLodBias = avoidAngleLlvmpipeExplicitLodBias == JNI_TRUE; request.coherentAsFlush = coherentAsFlush == JNI_TRUE; + request.fixIterationRPSubgroupScratch = fixIterationRPSubgroupScratch == JNI_TRUE; + request.deriveNumSubgroups = deriveNumSubgroups == JNI_TRUE; + request.iterationRPFixBarrier = iterationRPFixBarrier == JNI_TRUE; ScopedTraceReplayState replayState; mobilegl_trace_set_requested_size(request.width, request.height); diff --git a/android-plugin/app/src/trace/java/top/mobilegl/plugin/trace/TraceReplayActivity.java b/android-plugin/app/src/trace/java/top/mobilegl/plugin/trace/TraceReplayActivity.java index e3c91aa1..eecdacb4 100644 --- a/android-plugin/app/src/trace/java/top/mobilegl/plugin/trace/TraceReplayActivity.java +++ b/android-plugin/app/src/trace/java/top/mobilegl/plugin/trace/TraceReplayActivity.java @@ -116,6 +116,9 @@ public final class TraceReplayActivity extends Activity { request.avoidAngleLlvmpipeSamplerMipmapMinFilter, request.avoidAngleLlvmpipeExplicitLodBias, request.coherentAsFlush, + request.fixIterationRPSubgroupScratch, + request.deriveNumSubgroups, + request.iterationRPFixBarrier, request.texture2dDumps ); Log.i(TAG, result.toString()); @@ -149,6 +152,9 @@ public final class TraceReplayActivity extends Activity { boolean avoidAngleLlvmpipeSamplerMipmapMinFilter, boolean avoidAngleLlvmpipeExplicitLodBias, boolean coherentAsFlush, + boolean fixIterationRPSubgroupScratch, + boolean deriveNumSubgroups, + boolean iterationRPFixBarrier, String texture2dDumps ); @@ -174,6 +180,9 @@ public final class TraceReplayActivity extends Activity { final boolean avoidAngleLlvmpipeSamplerMipmapMinFilter; final boolean avoidAngleLlvmpipeExplicitLodBias; final boolean coherentAsFlush; + final boolean fixIterationRPSubgroupScratch; + final boolean deriveNumSubgroups; + final boolean iterationRPFixBarrier; final String texture2dDumps; private TraceReplayRequest( @@ -198,6 +207,9 @@ public final class TraceReplayActivity extends Activity { boolean avoidAngleLlvmpipeSamplerMipmapMinFilter, boolean avoidAngleLlvmpipeExplicitLodBias, boolean coherentAsFlush, + boolean fixIterationRPSubgroupScratch, + boolean deriveNumSubgroups, + boolean iterationRPFixBarrier, String texture2dDumps ) { this.tracePath = tracePath; @@ -221,6 +233,9 @@ public final class TraceReplayActivity extends Activity { this.avoidAngleLlvmpipeSamplerMipmapMinFilter = avoidAngleLlvmpipeSamplerMipmapMinFilter; this.avoidAngleLlvmpipeExplicitLodBias = avoidAngleLlvmpipeExplicitLodBias; this.coherentAsFlush = coherentAsFlush; + this.fixIterationRPSubgroupScratch = fixIterationRPSubgroupScratch; + this.deriveNumSubgroups = deriveNumSubgroups; + this.iterationRPFixBarrier = iterationRPFixBarrier; this.texture2dDumps = texture2dDumps; } @@ -249,6 +264,9 @@ public final class TraceReplayActivity extends Activity { intent.getBooleanExtra("avoid_angle_llvmpipe_sampler_mipmap_min_filter", false), intent.getBooleanExtra("avoid_angle_llvmpipe_explicit_lod_bias", false), intent.getBooleanExtra("coherent_as_flush", false), + intent.getBooleanExtra("fix_iterationrp_subgroup_scratch", false), + intent.getBooleanExtra("derive_num_subgroups", false), + intent.getBooleanExtra("iterationrp_fix_barrier", false), readString(intent, "texture_2d_dumps", "") ); } diff --git a/android-plugin/trace-replay-ci.sh b/android-plugin/trace-replay-ci.sh index f80a79e2..08c0639f 100644 --- a/android-plugin/trace-replay-ci.sh +++ b/android-plugin/trace-replay-ci.sh @@ -40,6 +40,9 @@ Set MOBILEGL_TRACE_ANGLE_VARIANT to the packaged ANGLE short hash used by DirectGLES replay. Set MOBILEGL_RETRACE_USE_PBUFFER=1 or pass --use-pbuffer to run DirectGLES against an offscreen EGL pbuffer instead of the Activity surface. +Set MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH=1, +MOBILEGL_DERIVE_NUM_SUBGROUPS=1, and MOBILEGL_ITERATIONRP_FIX_BARRIER=1 to +forward the corresponding iterationRP SPIR-V repairs into the APK process. Pass --avoid-angle-llvmpipe-sampler-mipmap-min-filter for DirectGLES traces that need ANGLE llvmpipe sampler mipmap filters downgraded to avoid driver stalls. Pass --avoid-angle-llvmpipe-explicit-lod-bias for DirectGLES traces whose shaders @@ -363,6 +366,15 @@ run_retrace() { if [ "${coherent_as_flush}" -eq 1 ]; then set -- "$@" --ez coherent_as_flush true fi + if [ "${MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH:-}" = "1" ]; then + set -- "$@" --ez fix_iterationrp_subgroup_scratch true + fi + if [ "${MOBILEGL_DERIVE_NUM_SUBGROUPS:-}" = "1" ]; then + set -- "$@" --ez derive_num_subgroups true + fi + if [ "${MOBILEGL_ITERATIONRP_FIX_BARRIER:-}" = "1" ]; then + set -- "$@" --ez iterationrp_fix_barrier true + fi if [ -n "${texture_2d_dumps}" ]; then set -- "$@" --es texture_2d_dumps "${texture_2d_dumps}" fi