[Fix, Test] (DirectVulkan, ShaderTranspiler, TraceReplay): repair iterationRP's missing reduction barrier

Program 203 reuses prefixSumCache for a second subgroup reduction before every workgroup invocation has consumed the first result. Add a fingerprint-gated SPIR-V pass that inserts the missing Workgroup acquire-release barrier while preserving native subgroup operations.

Keep the repair opt-in behind MOBILEGL_ITERATIONRP_FIX_BARRIER, cover insertion, pass-through, and idempotence, and enable it together with the existing iterationRP subgroup repairs for the matching Linux and Android CI retraces.
This commit is contained in:
2026-08-19 23:42:17 -04:00
parent 5bd8ef01e5
commit c09045fe59
20 changed files with 583 additions and 2 deletions
+3
View File
@@ -420,6 +420,9 @@ jobs:
MOBILEGL_USE_ANGLE: ${{ matrix.backend.name == 'DirectGLES' && '1' || '0' }}
MOBILEGL_TRACE_ANGLE_VARIANT: ${{ matrix.case.name == 'minecraft-1.21.4-fabric-iris-bliss-in-world' && '90a62123d794' || 'ec889e6ea831' }}
MOBILEGL_MAGMA_R11G11B10F_FALLBACK: ${{ matrix.backend.name == 'DirectVulkan' && '1' || '0' }}
MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
MOBILEGL_DERIVE_NUM_SUBGROUPS: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
MOBILEGL_ITERATIONRP_FIX_BARRIER: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
run: |
apk_file="android-retrace-apks/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk"
test -f "${apk_file}"
+6
View File
@@ -639,6 +639,12 @@ jobs:
if [ '${{ matrix.backend }}' = 'DirectVulkan' ]; then
export MOBILEGL_MAGMA_R11G11B10F_FALLBACK=1
fi
if [ '${{ matrix.backend }}' = 'DirectVulkan' ] \
&& [ '${{ matrix.case }}' = 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' ]; then
export MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH=1
export MOBILEGL_DERIVE_NUM_SUBGROUPS=1
export MOBILEGL_ITERATIONRP_FIX_BARRIER=1
fi
# The blended depth-write quirk auto-enables only on Qualcomm, which no CI
# runner has, so force it on for the OIT case it exists to fix. ForceOn
# bypasses only the vendor gate, so this exercises the real strip on
+1
View File
@@ -285,6 +285,7 @@ set(SOURCE_FILES
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DeriveNumSubgroupsPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPSubgroupScratchPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateSubgroupsPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp
+4
View File
@@ -100,6 +100,10 @@ namespace MobileGL::MG_Config {
// itself on >= 16-lane devices. Auto is ON; ForceOff replays the pack's bug
// verbatim.
QuirkOverride FixIterationRPSubgroupScratch = QuirkOverride::Auto;
// MOBILEGL_ITERATIONRP_FIX_BARRIER: repair Program 203's missing workgroup
// rendezvous between its two reductions over prefixSumCache. Off by default and
// fingerprint-gated by FixIterationRPBarrierPass when enabled.
Bool IterationRPFixBarrier = false;
// MOBILEGL_DERIVE_NUM_SUBGROUPS: replace compute gl_NumSubgroups loads with
// ceil(workgroup invocations / gl_SubgroupSize) on the NATIVE subgroup path
// (ShaderTranspiler::DeriveNumSubgroupsPass). Auto is ON: GL requires
+1
View File
@@ -171,6 +171,7 @@ namespace MobileGL::MG_ConfigLoader {
features.MagmaEmulateSubgroup = QueryEnvFlag("MOBILEGL_MAGMA_EMULATE_SUBGROUP");
features.FixIterationRPSubgroupScratch =
QueryEnvQuirkOverride("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
features.IterationRPFixBarrier = QueryEnvFlag("MOBILEGL_ITERATIONRP_FIX_BARRIER");
features.DeriveNumSubgroups = QueryEnvQuirkOverride("MOBILEGL_DERIVE_NUM_SUBGROUPS");
features.AdvertiseFp64 = QueryEnvFlag("MOBILEGL_ADVERTISE_FP64");
features.MagmaR11G11B10FFallback = QueryEnvFlag("MOBILEGL_MAGMA_R11G11B10F_FALLBACK");
@@ -3190,10 +3190,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
}
// GL_KHR_shader_subgroup handling (SubgroupSupportPolicy.h). Native subgroup
// operations execute natively; two module repairs keep the GL contract intact
// operations execute natively; module repairs keep the GL contract intact
// around them. The opt-in emulation path replaces them only on devices with no
// subgroup support at all (MOBILEGL_MAGMA_EMULATE_SUBGROUP).
if (shaders[i] && shaders[i]->GetShaderStage() == ShaderStage::Compute) {
// Program 203 broadcasts the first reduction through
// prefixSumCache[0], then lets the second reduction overwrite that
// scratch without first rendezvousing all readers. Patch that exact
// fingerprint before either native or emulated subgroup lowering.
if (m_subgroupPolicy.fixIterationRPBarrier) {
Vector<Uint> patchedSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::FixIterationRPBarrierForVulkan(
moduleSpirvs[i], patchedSpirv, enableSpirvValidation)) {
moduleSpirvs[i] = std::move(patchedSpirv);
} else {
MGLOG_E("ProgramFactory: iterationRP barrier patch failed for program %u; "
"Program 203 keeps its shared-scratch race",
program.GetExternalIndex());
}
}
if (m_subgroupPolicy.emulateSubgroups) {
Vector<Uint> emulatedSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::EmulateSubgroupsForVulkan(
@@ -381,6 +381,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
struct SubgroupLoweringPolicy {
Bool emulateSubgroups = false; // MOBILEGL_MAGMA_EMULATE_SUBGROUP, no-native-support devices
Bool fixIterationRPSubgroupScratch = false; // patch iterationRP's under-declared scratch
Bool fixIterationRPBarrier = false; // repair Program 203's shared-scratch race
Bool deriveNumSubgroups = false; // repair the NumSubgroups builtin
Bool requireFullSubgroups = false; // computeFullSubgroups enabled on the device
Uint32 nativeSubgroupSize = 0;
@@ -3063,6 +3063,7 @@ void main() {
subgroupPolicy.emulateSubgroups = ShouldEmulateSubgroups(m_nativeSubgroupSupported);
subgroupPolicy.fixIterationRPSubgroupScratch =
m_nativeSubgroupSupported && ShouldFixIterationRPSubgroupScratch();
subgroupPolicy.fixIterationRPBarrier = ShouldFixIterationRPBarrier();
subgroupPolicy.deriveNumSubgroups =
m_nativeSubgroupSupported && ShouldDeriveNumSubgroups();
subgroupPolicy.requireFullSubgroups = m_computeFullSubgroupsFeatureEnabled;
@@ -18,9 +18,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
//
// Native subgroups are the implementation whenever the device has them, whatever
// their width - subgroup operations execute on the hardware paths they were made
// for. Two module-level repairs keep the GL contract intact around them:
// for. Module-level repairs keep the GL contract intact around them:
// - FixIterationRPSubgroupScratchPass patches the one known pack bug: iterationRP's
// prefixSumCache[32], under-declared for sub-16-lane devices (8-lane lavapipe);
// - FixIterationRPBarrierPass repairs Program 203's race between two reductions
// reusing that scratch, when explicitly enabled;
// - DeriveNumSubgroupsPass replaces the one builtin drivers get wrong
// (gl_NumSubgroups) with the value the rest of the topology implies.
// The 32-lane shared-memory emulation (EmulateSubgroupsPass) is a LAST RESORT for
@@ -47,6 +49,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
MG_Config::QuirkOverride::ForceOff;
}
inline Bool ShouldFixIterationRPBarrier() {
return MG_Config::Features.IterationRPFixBarrier;
}
inline Bool ShouldDeriveNumSubgroups() {
// Auto is ON: gl_NumSubgroups must agree with the gl_SubgroupID range for the GL
// contract to hold, and the derived ceil() value is the one the renderer can pin
@@ -4,6 +4,7 @@ add_executable(
SpirvPassTest
SpirvPassTest.cpp
DeriveNumSubgroupsTest.cpp
FixIterationRPBarrierTest.cpp
FixIterationRPSubgroupScratchTest.cpp
EmulateSubgroupsTest.cpp
DemoteFloat64Test.cpp
@@ -0,0 +1,206 @@
// MobileGL - MobileGL/MG_Test/ShaderTranspiler/FixIterationRPBarrierTest.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include <gtest/gtest.h>
#define SPV_ENABLE_UTILITY_CODE
#include "glslang/SPIRV/spirv.hpp11"
#undef SPV_ENABLE_UTILITY_CODE
#include "Includes.h"
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/Types.h>
#include <spirv-tools/libspirv.hpp>
#include <map>
#include <vector>
using namespace MobileGL;
using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler;
namespace {
constexpr SizeT kSpirvHeaderWordCount = 5u;
template <typename Visitor>
void ForEachInstruction(const Vector<Uint32>& spirv, Visitor&& visit) {
for (SizeT offset = kSpirvHeaderWordCount; offset < spirv.size();) {
const Uint32 wordCount = spirv[offset] >> 16u;
if (wordCount == 0u || offset + wordCount > spirv.size()) break;
visit(static_cast<spv::Op>(spirv[offset] & 0xffffu), &spirv[offset], wordCount);
offset += wordCount;
}
}
Vector<Uint32> CompileCompute(const String& source) {
using namespace MobileGL::MG_Util::ShaderTranspiler;
ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source};
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
EXPECT_TRUE(shaderResult) << (shaderResult ? String{} : shaderResult.error().log);
if (!shaderResult) return {};
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
EXPECT_TRUE(programResult) << (programResult ? String{} : programResult.error().log);
if (!programResult) return {};
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()};
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
EXPECT_TRUE(binaryResult) << (binaryResult ? String{} : binaryResult.error().log);
if (!binaryResult || binaryResult->empty()) return {};
return binaryResult->front();
}
bool Validates(const Vector<Uint32>& spirv) {
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
tools.SetMessageConsumer(
[](spv_message_level_t, const char*, const spv_position_t& position, const char* message) {
ADD_FAILURE() << "spirv-val at word " << position.index << ": " << message;
});
return tools.Validate(spirv);
}
Uint32 CountOpcode(const Vector<Uint32>& spirv, spv::Op wanted) {
Uint32 count = 0u;
ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32*, Uint32) {
if (opcode == wanted) ++count;
});
return count;
}
bool HasWorkgroupBarrierImmediatelyBeforeSecondScan(const Vector<Uint32>& spirv) {
std::map<Uint32, Uint32> uintConstants;
spv::Op previous = spv::Op::OpNop;
Uint32 scanCount = 0u;
bool found = false;
const Uint32* previousWords = nullptr;
Uint32 previousWordCount = 0u;
ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32* words, Uint32 wordCount) {
if (opcode == spv::Op::OpConstant && wordCount >= 4u) {
uintConstants[words[2]] = words[3];
}
if (opcode == spv::Op::OpGroupNonUniformFAdd && wordCount >= 6u &&
static_cast<spv::GroupOperation>(words[4]) == spv::GroupOperation::InclusiveScan && ++scanCount == 2u &&
previous == spv::Op::OpControlBarrier && previousWordCount == 4u) {
found =
uintConstants[previousWords[1]] == static_cast<Uint32>(spv::Scope::Workgroup) &&
uintConstants[previousWords[2]] == static_cast<Uint32>(spv::Scope::Workgroup) &&
uintConstants[previousWords[3]] == (static_cast<Uint32>(spv::MemorySemanticsMask::AcquireRelease) |
static_cast<Uint32>(spv::MemorySemanticsMask::WorkgroupMemory));
}
previous = opcode;
previousWords = words;
previousWordCount = wordCount;
});
return found;
}
constexpr const char* kProgram203RaceShape = R"(#version 450 core
#extension GL_KHR_shader_subgroup_basic : require
#extension GL_KHR_shader_subgroup_arithmetic : require
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
layout(std430, binding = 0) buffer Output { vec2 value; } outputData;
shared vec2 prefixSumCache[32];
void main() {
vec2 sampleLuminance = subgroupInclusiveAdd(
vec2(float(gl_LocalInvocationIndex), 1.0));
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = sampleLuminance;
barrier();
if (gl_LocalInvocationIndex == 511u)
prefixSumCache[0] = sampleLuminance / 512.0;
barrier();
float avg = prefixSumCache[0].x;
float weight = avg > 0.0 ? float(gl_LocalInvocationIndex + 1u) / avg : 0.0;
vec2 sampleExposure = subgroupInclusiveAdd(vec2(weight, 1.0));
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = sampleExposure;
barrier();
if (gl_LocalInvocationIndex == 511u)
outputData.value = sampleExposure;
}
)";
constexpr const char* kAlreadySynchronizedShape = R"(#version 450 core
#extension GL_KHR_shader_subgroup_basic : require
#extension GL_KHR_shader_subgroup_arithmetic : require
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
layout(std430, binding = 0) buffer Output { vec2 value; } outputData;
shared vec2 prefixSumCache[32];
void main() {
vec2 first = subgroupInclusiveAdd(vec2(float(gl_LocalInvocationIndex), 1.0));
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = first;
barrier();
if (gl_LocalInvocationIndex == 511u) prefixSumCache[0] = first / 512.0;
barrier();
float avg = prefixSumCache[0].x;
barrier();
vec2 second = subgroupInclusiveAdd(vec2(avg, 1.0));
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = second;
barrier();
if (gl_LocalInvocationIndex == 511u) outputData.value = second;
}
)";
constexpr const char* kForeignSingleScanShape = R"(#version 450 core
#extension GL_KHR_shader_subgroup_basic : require
#extension GL_KHR_shader_subgroup_arithmetic : require
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
layout(std430, binding = 0) buffer Output { vec2 value; } outputData;
shared vec2 prefixSumCache[32];
void main() {
vec2 value = subgroupInclusiveAdd(vec2(float(gl_LocalInvocationIndex), 1.0));
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = value;
barrier();
if (gl_LocalInvocationIndex == 0u) outputData.value = prefixSumCache[0];
}
)";
} // namespace
TEST(FixIterationRPBarrierPass, InsertsWorkgroupBarrierBeforeSecondReduction) {
const Vector<Uint32> input = CompileCompute(kProgram203RaceShape);
ASSERT_FALSE(input.empty());
const Uint32 inputBarrierCount = CountOpcode(input, spv::Op::OpControlBarrier);
EXPECT_FALSE(HasWorkgroupBarrierImmediatelyBeforeSecondScan(input));
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, output, true));
EXPECT_EQ(CountOpcode(output, spv::Op::OpControlBarrier), inputBarrierCount + 1u);
EXPECT_TRUE(HasWorkgroupBarrierImmediatelyBeforeSecondScan(output));
EXPECT_TRUE(Validates(output));
}
TEST(FixIterationRPBarrierPass, LeavesOtherShapesByteIdentical) {
const Vector<Uint32> input = CompileCompute(kForeignSingleScanShape);
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, output, true));
EXPECT_EQ(output, input);
}
TEST(FixIterationRPBarrierPass, LeavesAnAlreadySynchronizedShaderByteIdentical) {
const Vector<Uint32> input = CompileCompute(kAlreadySynchronizedShape);
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, output, true));
EXPECT_EQ(output, input);
}
TEST(FixIterationRPBarrierPass, IsIdempotent) {
const Vector<Uint32> input = CompileCompute(kProgram203RaceShape);
ASSERT_FALSE(input.empty());
Vector<Uint32> once;
ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(input, once, true));
Vector<Uint32> twice;
ASSERT_TRUE(ShaderCompiler::FixIterationRPBarrierForVulkan(once, twice, true));
EXPECT_EQ(twice, once);
}
@@ -27,6 +27,7 @@
#include "SpirvPasses/ZeroBaseVertexPass.h"
#include "SpirvPasses/DeriveNumSubgroupsPass.h"
#include "SpirvPasses/EmulateSubgroupsPass.h"
#include "SpirvPasses/FixIterationRPBarrierPass.h"
#include "SpirvPasses/FixIterationRPSubgroupScratchPass.h"
#include "SpirvPasses/NormalizeRectCoordinatesPass.h"
#include "SpirvPasses/Lower1DArrayImagesPass.h"
@@ -924,6 +925,17 @@ namespace MobileGL {
inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::FixIterationRPBarrierForVulkan(
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(FixIterationRPBarrierPass::CreateFixIterationRPBarrierPass());
return RunOptimizerChecked("FixIterationRPBarrierForVulkan", optimizer,
inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
using namespace spvtools;
@@ -179,6 +179,12 @@ namespace MobileGL {
Uint32 nativeSubgroupSize,
Uint32 maxWorkgroupScratchBytes,
bool enableSpirvValidation = false);
// Inserts the missing workgroup rendezvous between Program 203's two
// prefixSumCache reductions. Fingerprint-gated to the iterationRP shape;
// unrelated and already-repaired modules pass through byte-identical.
static bool FixIterationRPBarrierForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair
// (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no
// VK_FORMAT_R64*_SFLOAT is needed - lavapipe advertises none of them for vertex
@@ -0,0 +1,232 @@
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include "FixIterationRPBarrierPass.h"
#include "spirv.hpp"
#include "source/opt/constants.h"
#include "source/opt/def_use_manager.h"
#include "source/opt/instruction.h"
#include "source/opt/ir_context.h"
#include "source/opt/module.h"
#include "source/util/make_unique.h"
#include <vector>
namespace MobileGL::MG_Util::ShaderTranspiler {
namespace {
using spvtools::opt::Instruction;
using spvtools::opt::IRContext;
using spvtools::opt::Operand;
const Instruction* RootVariable(IRContext* context, uint32_t pointerId) {
const Instruction* def = context->get_def_use_mgr()->GetDef(pointerId);
while (def != nullptr) {
switch (def->opcode()) {
case spv::Op::OpVariable:
return def;
case spv::Op::OpAccessChain:
case spv::Op::OpInBoundsAccessChain:
case spv::Op::OpCopyObject:
def = context->get_def_use_mgr()->GetDef(def->GetSingleWordInOperand(0));
break;
default:
return nullptr;
}
}
return nullptr;
}
bool IsUintConstant(IRContext* context, uint32_t id, uint32_t wanted) {
const Instruction* def = context->get_def_use_mgr()->GetDef(id);
return def != nullptr && def->opcode() == spv::Op::OpConstant && def->NumInOperands() == 1u &&
def->GetSingleWordInOperand(0) == wanted;
}
bool IsZeroElementPointer(IRContext* context, uint32_t pointerId, const Instruction** root) {
const Instruction* pointer = context->get_def_use_mgr()->GetDef(pointerId);
if (pointer == nullptr ||
(pointer->opcode() != spv::Op::OpAccessChain && pointer->opcode() != spv::Op::OpInBoundsAccessChain) ||
pointer->NumInOperands() < 2u) {
return false;
}
for (uint32_t i = 1u; i < pointer->NumInOperands(); ++i) {
if (!IsUintConstant(context, pointer->GetSingleWordInOperand(i), 0u)) return false;
}
*root = RootVariable(context, pointerId);
return *root != nullptr;
}
bool IsWorkgroupVec2Array(IRContext* context, const Instruction* variable) {
if (variable == nullptr || variable->opcode() != spv::Op::OpVariable || variable->NumInOperands() < 1u ||
static_cast<spv::StorageClass>(variable->GetSingleWordInOperand(0)) != spv::StorageClass::Workgroup) {
return false;
}
auto* defUseMgr = context->get_def_use_mgr();
const Instruction* pointerType = defUseMgr->GetDef(variable->type_id());
if (pointerType == nullptr || pointerType->opcode() != spv::Op::OpTypePointer ||
pointerType->NumInOperands() < 2u) {
return false;
}
const Instruction* arrayType = defUseMgr->GetDef(pointerType->GetSingleWordInOperand(1));
if (arrayType == nullptr || arrayType->opcode() != spv::Op::OpTypeArray ||
arrayType->NumInOperands() < 2u) {
return false;
}
const Instruction* length = defUseMgr->GetDef(arrayType->GetSingleWordInOperand(1));
if (length == nullptr || length->opcode() != spv::Op::OpConstant || length->NumInOperands() != 1u) {
return false;
}
const uint32_t arrayLength = length->GetSingleWordInOperand(0);
if (arrayLength < 32u || arrayLength > 512u) return false;
const Instruction* vectorType = defUseMgr->GetDef(arrayType->GetSingleWordInOperand(0));
if (vectorType == nullptr || vectorType->opcode() != spv::Op::OpTypeVector ||
vectorType->NumInOperands() < 2u || vectorType->GetSingleWordInOperand(1) != 2u) {
return false;
}
const Instruction* scalarType = defUseMgr->GetDef(vectorType->GetSingleWordInOperand(0));
return scalarType != nullptr && scalarType->opcode() == spv::Op::OpTypeFloat &&
scalarType->NumInOperands() == 1u && scalarType->GetSingleWordInOperand(0) == 32u;
}
bool IsVec2FloatInclusiveAdd(IRContext* context, const Instruction* inst) {
if (inst->opcode() != spv::Op::OpGroupNonUniformFAdd || inst->NumInOperands() < 3u ||
static_cast<spv::GroupOperation>(inst->GetSingleWordInOperand(1)) !=
spv::GroupOperation::InclusiveScan) {
return false;
}
const Instruction* vectorType = context->get_def_use_mgr()->GetDef(inst->type_id());
if (vectorType == nullptr || vectorType->opcode() != spv::Op::OpTypeVector ||
vectorType->NumInOperands() < 2u || vectorType->GetSingleWordInOperand(1) != 2u) {
return false;
}
const Instruction* scalarType = context->get_def_use_mgr()->GetDef(vectorType->GetSingleWordInOperand(0));
return scalarType != nullptr && scalarType->opcode() == spv::Op::OpTypeFloat &&
scalarType->NumInOperands() == 1u && scalarType->GetSingleWordInOperand(0) == 32u;
}
bool HasProgram203LocalSize(IRContext* context) {
for (const Instruction& entryPoint : context->module()->entry_points()) {
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) !=
spv::ExecutionModel::GLCompute) {
return false;
}
}
for (const Instruction& mode : context->module()->execution_modes()) {
if (mode.opcode() == spv::Op::OpExecutionMode && mode.NumInOperands() >= 5u &&
static_cast<spv::ExecutionMode>(mode.GetSingleWordInOperand(1)) == spv::ExecutionMode::LocalSize) {
return mode.GetSingleWordInOperand(2) == 32u && mode.GetSingleWordInOperand(3) == 16u &&
mode.GetSingleWordInOperand(4) == 1u;
}
}
return false;
}
bool IsStoreToRoot(IRContext* context, const Instruction* inst, const Instruction* root) {
return inst->opcode() == spv::Op::OpStore && inst->NumInOperands() >= 2u &&
RootVariable(context, inst->GetSingleWordInOperand(0)) == root;
}
} // namespace
spvtools::opt::Pass::Status FixIterationRPBarrierPass::Process() {
auto* irContext = context();
if (!HasProgram203LocalSize(irContext)) return Status::SuccessWithoutChange;
for (auto& function : *irContext->module()) {
std::vector<Instruction*> instructions;
std::vector<size_t> scans;
for (auto& block : function) {
for (auto& inst : block) {
if (IsVec2FloatInclusiveAdd(irContext, &inst)) scans.push_back(instructions.size());
instructions.push_back(&inst);
}
}
// Program 203 has exactly two vec2 inclusive adds: the luminance reduction
// and the weighted-exposure reduction. More or fewer is not our fingerprint.
if (scans.size() != 2u) continue;
const size_t firstScan = scans[0];
const size_t secondScan = scans[1];
const Instruction* scratch = nullptr;
size_t averageLoad = instructions.size();
for (size_t i = firstScan + 1u; i < secondScan; ++i) {
Instruction* inst = instructions[i];
if (inst->opcode() != spv::Op::OpLoad || inst->NumInOperands() < 1u) continue;
const Instruction* root = nullptr;
if (!IsZeroElementPointer(irContext, inst->GetSingleWordInOperand(0), &root) ||
!IsWorkgroupVec2Array(irContext, root)) {
continue;
}
// The broadcast is read as prefixSumCache[0].x, hence a scalar load.
const Instruction* type = irContext->get_def_use_mgr()->GetDef(inst->type_id());
if (type == nullptr || type->opcode() != spv::Op::OpTypeFloat || type->NumInOperands() != 1u ||
type->GetSingleWordInOperand(0) != 32u) {
continue;
}
scratch = root;
averageLoad = i;
break;
}
if (scratch == nullptr) continue;
bool sawZeroBroadcastStore = false;
bool sawPublishBarrier = false;
for (size_t i = firstScan + 1u; i < averageLoad; ++i) {
const Instruction* root = nullptr;
if (instructions[i]->opcode() == spv::Op::OpStore &&
IsZeroElementPointer(irContext, instructions[i]->GetSingleWordInOperand(0), &root) &&
root == scratch) {
sawZeroBroadcastStore = true;
} else if (sawZeroBroadcastStore && instructions[i]->opcode() == spv::Op::OpControlBarrier) {
sawPublishBarrier = true;
}
}
if (!sawZeroBroadcastStore || !sawPublishBarrier) continue;
bool alreadySynchronized = false;
for (size_t i = averageLoad + 1u; i < secondScan; ++i) {
if (instructions[i]->opcode() == spv::Op::OpControlBarrier) {
alreadySynchronized = true;
break;
}
}
if (alreadySynchronized) return Status::SuccessWithoutChange;
bool secondPhaseReusesScratch = false;
for (size_t i = secondScan + 1u; i < instructions.size(); ++i) {
if (IsStoreToRoot(irContext, instructions[i], scratch)) {
secondPhaseReusesScratch = true;
break;
}
}
if (!secondPhaseReusesScratch) continue;
auto* constantMgr = irContext->get_constant_mgr();
const uint32_t scopeId = constantMgr->GetUIntConstId(static_cast<uint32_t>(spv::Scope::Workgroup));
const uint32_t semanticsId =
constantMgr->GetUIntConstId(static_cast<uint32_t>(spv::MemorySemanticsMask::AcquireRelease) |
static_cast<uint32_t>(spv::MemorySemanticsMask::WorkgroupMemory));
if (scopeId == 0u || semanticsId == 0u) return Status::Failure;
instructions[secondScan]->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpControlBarrier, 0u, 0u,
Instruction::OperandList{Operand{SPV_OPERAND_TYPE_ID, {scopeId}},
Operand{SPV_OPERAND_TYPE_ID, {scopeId}},
Operand{SPV_OPERAND_TYPE_ID, {semanticsId}}}));
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
return Status::SuccessWithChange;
}
return Status::SuccessWithoutChange;
}
spvtools::Optimizer::PassToken FixIterationRPBarrierPass::CreateFixIterationRPBarrierPass() {
return spvtools::Optimizer::PassToken(spvtools::MakeUnique<FixIterationRPBarrierPass>());
}
} // namespace MobileGL::MG_Util::ShaderTranspiler
@@ -0,0 +1,28 @@
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.h
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#pragma once
#include "source/opt/pass.h"
#include "spirv-tools/optimizer.hpp"
namespace MobileGL::MG_Util::ShaderTranspiler {
// Repairs iterationRP Program 203's missing workgroup rendezvous between two
// reductions that reuse prefixSumCache. The first phase broadcasts its result
// through prefixSumCache[0], but the second phase may overwrite that element before
// every invocation has read it. The pass fingerprints that exact two-scan,
// 512-invocation shape and inserts one Workgroup control barrier immediately before
// the second scan. Unrelated modules and already-repaired modules are byte-identical.
class FixIterationRPBarrierPass : public spvtools::opt::Pass {
public:
const char* name() const override { return "fix-iterationrp-barrier"; }
Status Process() override;
static spvtools::Optimizer::PassToken CreateFixIterationRPBarrierPass();
};
} // namespace MobileGL::MG_Util::ShaderTranspiler
@@ -164,6 +164,21 @@ bool LoadMobileGL(const Request& request, std::string& error) {
} else {
unsetenv("MOBILEGL_COHERENT_AS_FLUSH");
}
if (request.fixIterationRPSubgroupScratch) {
setenv("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH", "1", 1);
} else {
unsetenv("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
}
if (request.deriveNumSubgroups) {
setenv("MOBILEGL_DERIVE_NUM_SUBGROUPS", "1", 1);
} else {
unsetenv("MOBILEGL_DERIVE_NUM_SUBGROUPS");
}
if (request.iterationRPFixBarrier) {
setenv("MOBILEGL_ITERATIONRP_FIX_BARRIER", "1", 1);
} else {
unsetenv("MOBILEGL_ITERATIONRP_FIX_BARRIER");
}
if (request.fboAttachmentDumps.empty()) {
unsetenv("MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS");
} else {
@@ -818,6 +833,10 @@ bool WriteResultJson(const Request& request, const Result& result) {
<< (request.avoidAngleLlvmpipeSamplerMipmapMinFilter ? "true" : "false") << ",\n";
file << " \"avoidAngleLlvmpipeExplicitLodBias\": "
<< (request.avoidAngleLlvmpipeExplicitLodBias ? "true" : "false") << ",\n";
file << " \"fixIterationRPSubgroupScratch\": " << (request.fixIterationRPSubgroupScratch ? "true" : "false")
<< ",\n";
file << " \"deriveNumSubgroups\": " << (request.deriveNumSubgroups ? "true" : "false") << ",\n";
file << " \"iterationRPFixBarrier\": " << (request.iterationRPFixBarrier ? "true" : "false") << ",\n";
file << " \"holdMs\": " << request.holdMs << ",\n";
file << " \"mismatchPixels\": " << result.mismatchPixels << "\n";
file << "}\n";
@@ -44,6 +44,9 @@ struct Request {
bool avoidAngleLlvmpipeSamplerMipmapMinFilter = false;
bool avoidAngleLlvmpipeExplicitLodBias = false;
bool coherentAsFlush = false;
bool fixIterationRPSubgroupScratch = false;
bool deriveNumSubgroups = false;
bool iterationRPFixBarrier = false;
int holdMs = 0;
};
@@ -122,6 +122,9 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
jboolean avoidAngleLlvmpipeSamplerMipmapMinFilter,
jboolean avoidAngleLlvmpipeExplicitLodBias,
jboolean coherentAsFlush,
jboolean fixIterationRPSubgroupScratch,
jboolean deriveNumSubgroups,
jboolean iterationRPFixBarrier,
jstring texture2dDumps) {
mobilegl_trace::Request request;
request.tracePath = ToString(env, tracePath);
@@ -150,6 +153,9 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
avoidAngleLlvmpipeSamplerMipmapMinFilter == JNI_TRUE;
request.avoidAngleLlvmpipeExplicitLodBias = avoidAngleLlvmpipeExplicitLodBias == JNI_TRUE;
request.coherentAsFlush = coherentAsFlush == JNI_TRUE;
request.fixIterationRPSubgroupScratch = fixIterationRPSubgroupScratch == JNI_TRUE;
request.deriveNumSubgroups = deriveNumSubgroups == JNI_TRUE;
request.iterationRPFixBarrier = iterationRPFixBarrier == JNI_TRUE;
ScopedTraceReplayState replayState;
mobilegl_trace_set_requested_size(request.width, request.height);
@@ -116,6 +116,9 @@ public final class TraceReplayActivity extends Activity {
request.avoidAngleLlvmpipeSamplerMipmapMinFilter,
request.avoidAngleLlvmpipeExplicitLodBias,
request.coherentAsFlush,
request.fixIterationRPSubgroupScratch,
request.deriveNumSubgroups,
request.iterationRPFixBarrier,
request.texture2dDumps
);
Log.i(TAG, result.toString());
@@ -149,6 +152,9 @@ public final class TraceReplayActivity extends Activity {
boolean avoidAngleLlvmpipeSamplerMipmapMinFilter,
boolean avoidAngleLlvmpipeExplicitLodBias,
boolean coherentAsFlush,
boolean fixIterationRPSubgroupScratch,
boolean deriveNumSubgroups,
boolean iterationRPFixBarrier,
String texture2dDumps
);
@@ -174,6 +180,9 @@ public final class TraceReplayActivity extends Activity {
final boolean avoidAngleLlvmpipeSamplerMipmapMinFilter;
final boolean avoidAngleLlvmpipeExplicitLodBias;
final boolean coherentAsFlush;
final boolean fixIterationRPSubgroupScratch;
final boolean deriveNumSubgroups;
final boolean iterationRPFixBarrier;
final String texture2dDumps;
private TraceReplayRequest(
@@ -198,6 +207,9 @@ public final class TraceReplayActivity extends Activity {
boolean avoidAngleLlvmpipeSamplerMipmapMinFilter,
boolean avoidAngleLlvmpipeExplicitLodBias,
boolean coherentAsFlush,
boolean fixIterationRPSubgroupScratch,
boolean deriveNumSubgroups,
boolean iterationRPFixBarrier,
String texture2dDumps
) {
this.tracePath = tracePath;
@@ -221,6 +233,9 @@ public final class TraceReplayActivity extends Activity {
this.avoidAngleLlvmpipeSamplerMipmapMinFilter = avoidAngleLlvmpipeSamplerMipmapMinFilter;
this.avoidAngleLlvmpipeExplicitLodBias = avoidAngleLlvmpipeExplicitLodBias;
this.coherentAsFlush = coherentAsFlush;
this.fixIterationRPSubgroupScratch = fixIterationRPSubgroupScratch;
this.deriveNumSubgroups = deriveNumSubgroups;
this.iterationRPFixBarrier = iterationRPFixBarrier;
this.texture2dDumps = texture2dDumps;
}
@@ -249,6 +264,9 @@ public final class TraceReplayActivity extends Activity {
intent.getBooleanExtra("avoid_angle_llvmpipe_sampler_mipmap_min_filter", false),
intent.getBooleanExtra("avoid_angle_llvmpipe_explicit_lod_bias", false),
intent.getBooleanExtra("coherent_as_flush", false),
intent.getBooleanExtra("fix_iterationrp_subgroup_scratch", false),
intent.getBooleanExtra("derive_num_subgroups", false),
intent.getBooleanExtra("iterationrp_fix_barrier", false),
readString(intent, "texture_2d_dumps", "")
);
}
+12
View File
@@ -40,6 +40,9 @@ Set MOBILEGL_TRACE_ANGLE_VARIANT to the packaged ANGLE short hash used by
DirectGLES replay.
Set MOBILEGL_RETRACE_USE_PBUFFER=1 or pass --use-pbuffer to run DirectGLES
against an offscreen EGL pbuffer instead of the Activity surface.
Set MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH=1,
MOBILEGL_DERIVE_NUM_SUBGROUPS=1, and MOBILEGL_ITERATIONRP_FIX_BARRIER=1 to
forward the corresponding iterationRP SPIR-V repairs into the APK process.
Pass --avoid-angle-llvmpipe-sampler-mipmap-min-filter for DirectGLES traces that
need ANGLE llvmpipe sampler mipmap filters downgraded to avoid driver stalls.
Pass --avoid-angle-llvmpipe-explicit-lod-bias for DirectGLES traces whose shaders
@@ -363,6 +366,15 @@ run_retrace() {
if [ "${coherent_as_flush}" -eq 1 ]; then
set -- "$@" --ez coherent_as_flush true
fi
if [ "${MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH:-}" = "1" ]; then
set -- "$@" --ez fix_iterationrp_subgroup_scratch true
fi
if [ "${MOBILEGL_DERIVE_NUM_SUBGROUPS:-}" = "1" ]; then
set -- "$@" --ez derive_num_subgroups true
fi
if [ "${MOBILEGL_ITERATIONRP_FIX_BARRIER:-}" = "1" ]; then
set -- "$@" --ez iterationrp_fix_barrier true
fi
if [ -n "${texture_2d_dumps}" ]; then
set -- "$@" --es texture_2d_dumps "${texture_2d_dumps}"
fi