diff --git a/MobileGL/Config.h b/MobileGL/Config.h index 97b928da..9d5e522d 100644 --- a/MobileGL/Config.h +++ b/MobileGL/Config.h @@ -128,9 +128,10 @@ namespace MobileGL::MG_Config { // explicitly request a core profile via EGL_CONTEXT_OPENGL_PROFILE_MASK / a >=3.1 // version request. Bool RelaxedSemantics = false; - // MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that - // rewrites the recognized workgroup prefix-scan template on Qualcomm devices with - // subgroups wider than 32 lanes (see ShaderSourceProcessor's quirk registry). + // MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that rewrites + // recognized subgroup reductions when narrow Vulkan subgroups overflow fixed scratch + // arrays, or when Qualcomm subgroups are wider than the captured 32-lane model (see + // ShaderSourceProcessor's quirk registry). QuirkOverride SubgroupPrefixScanQuirk = QuirkOverride::Auto; // MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE: overrides the DirectVulkan quirk that // strips depth writes from accumulation-blended pipelines (MIN/MAX or additive diff --git a/MobileGL/MG_Backend/DirectVulkan/BackendObject_DirectVulkan.cpp b/MobileGL/MG_Backend/DirectVulkan/BackendObject_DirectVulkan.cpp index 0ec1bf03..37cc313c 100644 --- a/MobileGL/MG_Backend/DirectVulkan/BackendObject_DirectVulkan.cpp +++ b/MobileGL/MG_Backend/DirectVulkan/BackendObject_DirectVulkan.cpp @@ -10,6 +10,7 @@ #include "MG_Backend/BackendObject.h" #include "DirectVulkan.h" #include "MG_State/GLState/FramebufferState/FramebufferObject.h" +#include "MG_State/GLState/Core.h" #include "MG_State/GLState/TextureState/TextureState.h" #include "MG_Util/Classifiers/TextureEnumClassifier.h" #include "MG_Util/Converters/MGToGL/TextureEnumConverter.h" @@ -383,6 +384,9 @@ namespace MobileGL::MG_Backend::DirectVulkan { } UpdateDynamicBackendParameters(); UpdateAdvertisedExtensions(); + if (MG_State::pGLContext) { + MG_State::pGLContext->InvalidateCompileEnv(); + } PopulateFormatCapabilities(physicalDevice.handle, vkGetPhysicalDeviceFormatProperties, m_vulkanCaps, MutableFormatCapabilities()); PrintFormatCapabilities(GetFormatCapabilities()); @@ -687,6 +691,9 @@ namespace MobileGL::MG_Backend::DirectVulkan { m_vulkanCaps = capabilities; UpdateDynamicBackendParameters(); UpdateAdvertisedExtensions(); + if (MG_State::pGLContext) { + MG_State::pGLContext->InvalidateCompileEnv(); + } MutableFormatCapabilities().Clear(); } diff --git a/MobileGL/MG_State/GLState/Core.cpp b/MobileGL/MG_State/GLState/Core.cpp index 03c5c165..32b9dd2d 100644 --- a/MobileGL/MG_State/GLState/Core.cpp +++ b/MobileGL/MG_State/GLState/Core.cpp @@ -39,6 +39,11 @@ namespace MobileGL::MG_State { return m_compileEnv; } + void GLContext::InvalidateCompileEnv() { + m_compileEnv.reset(); + m_compileEnvBackend = nullptr; + } + // Error void GLContext::RecordError(ErrorCode code, UniquePtr info) { // Invariant I1, mechanically enforced: the GL error state is GL-thread-owned. diff --git a/MobileGL/MG_State/GLState/Core.h b/MobileGL/MG_State/GLState/Core.h index 73b28fcf..4dfb1078 100644 --- a/MobileGL/MG_State/GLState/Core.h +++ b/MobileGL/MG_State/GLState/Core.h @@ -413,9 +413,12 @@ namespace MobileGL { // cannot be captured in MG_State::Init() - that runs BEFORE MG_Backend::Init(), // so there is no backend to query yet. Re-captured whenever the active backend // object changes, which also rolls the fingerprint and therefore invalidates - // every P0b preprocess memo keyed against the old one. + // every P0b preprocess memo keyed against the old one. A backend whose dynamic + // capabilities become available without changing object identity must call + // InvalidateCompileEnv() after publishing them. // GL thread only. const SharedPtr& GetCompileEnv(); + void InvalidateCompileEnv(); private: // State Components diff --git a/MobileGL/MG_Test/Program/ProgramUtilTest.cpp b/MobileGL/MG_Test/Program/ProgramUtilTest.cpp index ab162764..21dc3a52 100644 --- a/MobileGL/MG_Test/Program/ProgramUtilTest.cpp +++ b/MobileGL/MG_Test/Program/ProgramUtilTest.cpp @@ -2104,15 +2104,92 @@ void main() { float warp = (prefixSum - importance) / sum - float(gl_LocalInvocationID.x + 1u) / float(1024); outputValues[gl_GlobalInvocationID.x] = warp; } +)"; + } + + String MakeWeightedExposureSubgroupReductionShader() { + return R"(#version 460 core +#extension GL_KHR_shader_subgroup_arithmetic : enable +uniform int frameCounter; +uniform float frameTime; +uniform float aspectRatio; +uniform vec2 pixelSize; +uniform sampler2D colortex2; +uniform sampler2D pixelData2D; +layout(local_size_x = 32, local_size_y = 16) in; +layout(rg16f) uniform image2D img_pixelData2D; +shared vec2 prefixSumCache[32]; + +float remapSaturate(float value, float edge0, float edge1) { + return clamp((value - edge0) / (edge1 - edge0), 0.0f, 1.0f); +} + +float GetExposureValue(float luminance) { + return max(luminance, 1.0E-5f); +} + +void main() { + vec2 texCoord = (vec2(gl_GlobalInvocationID.xy) + 0.5f) * vec2(1.0f / 32.0f, 1.0f / 16.0f); + vec2 sampleCoord = texCoord * (1.0f / 64.0f); + sampleCoord.x += (15.0f / 32.0f) + pixelSize.x * 12.0f; + float tileExposure = dot(textureLod(colortex2, sampleCoord, 0.0f).rgb, vec3(0.2125f, 0.7154f, 0.0721f)); + vec2 sampleLuminance = vec2(tileExposure, 0.0f); + sampleLuminance = subgroupInclusiveAdd(sampleLuminance); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleLuminance; + barrier(); + uint loopLength = uint(findMSB(gl_NumSubgroups)); + loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u); + for (uint i = 0; i < loopLength; i++) { + if ((gl_SubgroupID & (1u << i)) > 0u) { + sampleLuminance += prefixSumCache[(gl_SubgroupID >> i << i) - 1u]; + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleLuminance; + } + barrier(); + } + if (gl_LocalInvocationIndex == 511u) prefixSumCache[0] = sampleLuminance / 512.0f; + ; + barrier(); + float avg = prefixSumCache[0].x; + vec2 tileDistance = texCoord * 2.0f - 1.0f; + tileDistance.y /= aspectRatio; + float centerDistance = length(tileDistance); + float tileWeight = remapSaturate(centerDistance, 0.6f, 0.4f); + tileExposure = max(7.0E-7f, tileExposure); + float lumaWeight = avg / tileExposure; + lumaWeight = pow(lumaWeight, remapSaturate(avg, 0.02f, 0.001f) * 0.4f + 0.2f); + tileWeight *= lumaWeight; + vec2 sampleExposure = vec2(tileExposure * tileWeight, tileWeight); + sampleExposure = subgroupInclusiveAdd(sampleExposure); + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleExposure; + barrier(); + for (uint i = 0; i < loopLength; i++) { + if ((gl_SubgroupID & (1u << i)) > 0u) { + sampleExposure += prefixSumCache[(gl_SubgroupID >> i << i) - 1u]; + if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleExposure; + } + barrier(); + } + if (gl_LocalInvocationIndex == 511u) { + float avgExposure = max(sampleExposure.x / sampleExposure.y * 29.3f, 1.0E-10f); + avgExposure = log2(avgExposure); + float prevAvgExposure = log2(texelFetch(pixelData2D, ivec2(0, 0), 0).x); + float frameTimeFixed = frameTime + step(frameCounter, 20) * 100.0f; + float exposureTime = clamp(frameTimeFixed * (2.0f / 1.0f), 0.0f, 1.0f); + avgExposure = mix(prevAvgExposure, avgExposure, exposureTime); + avgExposure = max(exp2(avgExposure), 1.0E-5f); + float exposure = GetExposureValue(avgExposure); + imageStore(img_pixelData2D, ivec2(0, 0), vec4(avgExposure, exposure, 0.0f, 0.0f)); + } +} )"; } } // namespace -TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanUsesSharedMemoryAndProducesValidSpirv) { +TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanForNarrowSubgroupsProducesValidSpirv) { using namespace MG_Util::ShaderTranspiler; String source = MakeLinearSubgroupPrefixScanShader(); - ASSERT_TRUE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source)); + ASSERT_TRUE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 8, source)); EXPECT_NE(source.find("shared float prefixSumCache[1024]"), String::npos) << source; EXPECT_NE(source.find("mglVirtualSubgroupInvocation"), String::npos) << source; @@ -2121,7 +2198,7 @@ TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanUsesSharedMemoryAndProduc EXPECT_EQ(source.find("gl_Subgroup"), String::npos) << source; const String onceRewritten = source; - EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source)); + EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 8, source)); EXPECT_EQ(source, onceRewritten); ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source}; @@ -2150,12 +2227,21 @@ TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanUsesSharedMemoryAndProduc EXPECT_EQ(spirvText.find("OpGroupNonUniform"), String::npos) << spirvText; } +TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanStillSupportsWideQualcommSubgroups) { + using namespace MG_Util::ShaderTranspiler; + + String source = MakeLinearSubgroupPrefixScanShader(); + EXPECT_TRUE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source)); + EXPECT_EQ(source.find("subgroupInclusiveAdd"), String::npos) << source; +} + TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsOtherStagesAndSubgroupWidths) { using namespace MG_Util::ShaderTranspiler; const String original = MakeLinearSubgroupPrefixScanShader(); for (const auto& [stage, subgroupSize] : - {std::pair{ShaderStage::Compute, Uint32{32}}, std::pair{ShaderStage::Fragment, Uint32{64}}, + {std::pair{ShaderStage::Compute, Uint32{0}}, std::pair{ShaderStage::Compute, Uint32{16}}, + std::pair{ShaderStage::Compute, Uint32{32}}, std::pair{ShaderStage::Fragment, Uint32{64}}, std::pair{ShaderStage::Compute, Uint32{96}}}) { String source = original; EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(stage, subgroupSize, source)); @@ -2220,6 +2306,92 @@ TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsPartialOrUnsafeTem expectUnchanged(std::move(nvShuffleCall)); } +TEST_F(ProgramUtilTest, RewriteWeightedExposureReductionForEightLaneSubgroupsProducesValidSpirv) { + using namespace MG_Util::ShaderTranspiler; + + String source = MakeWeightedExposureSubgroupReductionShader(); + ASSERT_TRUE(RewriteWeightedExposureSubgroupReductionForVulkan(ShaderStage::Compute, 8, source)); + + EXPECT_NE(source.find("mglExposureWeightedSum"), String::npos) << source; + EXPECT_EQ(source.find("subgroupInclusiveAdd"), String::npos) << source; + EXPECT_EQ(source.find("gl_Subgroup"), String::npos) << source; + EXPECT_EQ(source.find("barrier()"), String::npos) << source; + + const String onceRewritten = source; + EXPECT_FALSE(RewriteWeightedExposureSubgroupReductionForVulkan(ShaderStage::Compute, 8, source)); + EXPECT_EQ(source, onceRewritten); + + ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source}; + auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib); + ASSERT_TRUE(shaderResult) << shaderResult.error().log << "\nsource:\n" << source; + + ProgramAttrib programAttrib{.shaders = {shaderResult.value()}}; + auto programResult = ShaderCompiler::LinkProgram(programAttrib); + ASSERT_TRUE(programResult) << programResult.error().log; + + ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()}; + auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib); + ASSERT_TRUE(binaryResult) << binaryResult.error().log; + ASSERT_EQ(binaryResult->size(), 1u); + + String validationDiagnostics; + spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1); + tools.SetMessageConsumer([&](spv_message_level_t, const char*, const spv_position_t&, const char* message) { + validationDiagnostics += message; + validationDiagnostics += '\n'; + }); + EXPECT_TRUE(tools.Validate(binaryResult->front())) << validationDiagnostics; + + String spirvText; + ASSERT_TRUE(tools.Disassemble(binaryResult->front(), &spirvText)); + EXPECT_EQ(spirvText.find("OpGroupNonUniform"), String::npos) << spirvText; +} + +TEST_F(ProgramUtilTest, RewriteWeightedExposureReductionRejectsOtherWidthsAndUnsafeTemplates) { + using namespace MG_Util::ShaderTranspiler; + + const String original = MakeWeightedExposureSubgroupReductionShader(); + const auto expectUnchanged = [&](ShaderStage stage, Uint32 subgroupSize, String source) { + const String before = source; + EXPECT_FALSE(RewriteWeightedExposureSubgroupReductionForVulkan(stage, subgroupSize, source)); + EXPECT_EQ(source, before); + }; + + expectUnchanged(ShaderStage::Compute, 0, original); + expectUnchanged(ShaderStage::Compute, 16, original); + expectUnchanged(ShaderStage::Compute, 32, original); + expectUnchanged(ShaderStage::Fragment, 8, original); + + String wrongScratchSize = original; + wrongScratchSize.replace(wrongScratchSize.find("prefixSumCache[32]"), std::strlen("prefixSumCache[32]"), + "prefixSumCache[64]"); + expectUnchanged(ShaderStage::Compute, 8, std::move(wrongScratchSize)); + + String changedWeighting = original; + changedWeighting.replace(changedWeighting.find("29.3f"), std::strlen("29.3f"), "30.0f"); + expectUnchanged(ShaderStage::Compute, 8, std::move(changedWeighting)); + + String extraSubgroupUse = original; + extraSubgroupUse.insert(extraSubgroupUse.find("void main()"), + "float extraSubgroupUse(float value) { return subgroupAdd(value); }\n"); + expectUnchanged(ShaderStage::Compute, 8, std::move(extraSubgroupUse)); +} + +TEST_F(ProgramUtilTest, NarrowSubgroupQuirkRunsForDirectVulkanWithoutVendorSpoofing) { + using namespace MG_Util::ShaderTranspiler; + + CompileEnv env; + env.backend = BackendType::DirectVulkan; + env.params.SubgroupSize = 8; + env.params.GpuVendor = MG_Backend::GpuVendorKind::Unknown; + env.subgroupPrefixScanQuirk = MG_Config::QuirkOverride::Auto; + + String source = MakeWeightedExposureSubgroupReductionShader(); + PreprocessShaderSource(ShaderStage::Compute, source, env); + EXPECT_NE(source.find("mglExposureWeightedSum"), String::npos) << source; + EXPECT_EQ(source.find("subgroupInclusiveAdd"), String::npos) << source; +} + // The LEXICAL half must fire at the source level (before the parse) for the // preempt-list names - the end-to-end ESSL tests cannot tell which half did the // rename, and for these names the parse would fail without the source rewrite. diff --git a/MobileGL/MG_Test/SanityTest.cpp b/MobileGL/MG_Test/SanityTest.cpp index f0fccae9..dd9bee54 100644 --- a/MobileGL/MG_Test/SanityTest.cpp +++ b/MobileGL/MG_Test/SanityTest.cpp @@ -31,6 +31,7 @@ #include #include #include +#include #include #include #include @@ -710,6 +711,37 @@ TEST(DirectVulkanSanity, AdvertisesSubgroupOnlyWhenVulkanReportsUsableSupport) { EXPECT_TRUE(backend.GetDynamicParameters().SubgroupQuadOperationsInAllStages); } +TEST(DirectVulkanSanity, CapabilityRefreshInvalidatesTheCachedCompileEnvironment) { + using namespace MobileGL; + + auto previousContext = Move(MG_State::pGLContext); + auto previousBackend = Move(MG_Backend::pActiveBackendObject); + MG_State::pGLContext = MakeUnique(); + + auto backend = MakeUnique(); + auto* backendPtr = backend.get(); + MG_Backend::pActiveBackendObject = Move(backend); + + const auto before = MG_State::pGLContext->GetCompileEnv(); + EXPECT_EQ(before->params.SubgroupSize, 0u); + + MG_External::VulkanCapabilities caps; + caps.SupportsShaderSubgroup = true; + caps.SubgroupSize = 8; + caps.SubgroupSupportedStages = VK_SHADER_STAGE_COMPUTE_BIT; + caps.SubgroupSupportedOperations = VK_SUBGROUP_FEATURE_BASIC_BIT | VK_SUBGROUP_FEATURE_ARITHMETIC_BIT; + backendPtr->ApplyVulkanCapabilitiesForTesting(caps); + + const auto after = MG_State::pGLContext->GetCompileEnv(); + EXPECT_NE(after.get(), before.get()); + EXPECT_NE(after->fingerprint, before->fingerprint); + EXPECT_EQ(after->backend, BackendType::DirectVulkan); + EXPECT_EQ(after->params.SubgroupSize, 8u); + + MG_Backend::pActiveBackendObject = Move(previousBackend); + MG_State::pGLContext = Move(previousContext); +} + TEST(DirectVulkanSanity, KeepsOptionalGpuShaderInt64BranchForVoxyQuadDecode) { using namespace MobileGL; diff --git a/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp b/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp index 213d2de9..0414dba0 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp +++ b/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp @@ -23,6 +23,7 @@ namespace { using MobileGL::SizeT; using MobileGL::String; + using MobileGL::Uint32; using MobileGL::Vector; bool IsIdentifierChar(char ch) { @@ -254,6 +255,46 @@ namespace { return true; } + SizeT CountTokenSequence(const Vector& tokens, + std::initializer_list expected) { + SizeT count = 0; + for (SizeT position = 0; position < tokens.size(); ++position) { + if (MatchTokenSequence(tokens, position, expected)) { + ++count; + } + } + return count; + } + + bool FindUniqueTokenSequence(const Vector& tokens, const Vector& expected, + SizeT& sourceBegin, SizeT& sourceEnd) { + if (expected.empty() || expected.size() > tokens.size()) { + return false; + } + + SizeT matchCount = 0; + for (SizeT position = 0; position + expected.size() <= tokens.size(); ++position) { + bool matches = true; + for (SizeT expectedIndex = 0; expectedIndex < expected.size(); ++expectedIndex) { + if (tokens[position + expectedIndex].text != expected[expectedIndex].text) { + matches = false; + break; + } + } + if (!matches) { + continue; + } + ++matchCount; + sourceBegin = tokens[position].begin; + sourceEnd = tokens[position + expected.size() - 1].end; + } + return matchCount == 1; + } + + bool IsPowerOfTwo(Uint32 value) { + return value != 0u && (value & (value - 1u)) == 0u; + } + struct LinearPrefixScanMatch { SizeT sharedArraySizeBegin = 0; SizeT sharedArraySizeEnd = 0; @@ -506,6 +547,164 @@ namespace { return replacement; } + struct WeightedExposureReductionMatch { + SizeT mainBegin = 0; + SizeT mainEnd = 0; + }; + + bool ParseWeightedExposureReductionTemplate(const Vector& tokens, + WeightedExposureReductionMatch& match) { + // IterationRP's exposure pass is a complete, stable shader-pack template. Match the + // whole main body before replacing it: a partial match would be unsafe because the + // replacement deliberately replays the 32x16 sample grid from one invocation. + static const Vector expectedMain = TokenizeCode(R"glsl( +void main() { +vec2 texCoord = (vec2(gl_GlobalInvocationID.xy) + 0.5f) * vec2(1.0f / 32.0f, 1.0f / 16.0f); +vec2 sampleCoord = texCoord * (1.0f / 64.0f); +sampleCoord.x += (15.0f / 32.0f) + pixelSize.x * 12.0f; +float tileExposure = dot(textureLod(colortex2, sampleCoord, 0.0f).rgb, vec3(0.2125f, 0.7154f, 0.0721f)); +vec2 sampleLuminance = vec2(tileExposure, 0.0f); +sampleLuminance = subgroupInclusiveAdd(sampleLuminance); +if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleLuminance; +barrier(); +uint loopLength = uint(findMSB(gl_NumSubgroups)); +loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u); +for (uint i = 0; i < loopLength; i++) { +if ((gl_SubgroupID & (1u << i)) > 0u) { +sampleLuminance += prefixSumCache[(gl_SubgroupID >> i << i) - 1u]; +if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleLuminance; +} +barrier(); +} +if (gl_LocalInvocationIndex == 511u) prefixSumCache[0] = sampleLuminance / 512.0f; +; +barrier(); +float avg = prefixSumCache[0].x; +vec2 tileDistance = texCoord * 2.0f - 1.0f; +tileDistance.y /= aspectRatio; +float centerDistance = length(tileDistance); +float tileWeight = remapSaturate(centerDistance, 0.6f, 0.4f); +tileExposure = max(7.0E-7f, tileExposure); +float lumaWeight = avg / tileExposure; +lumaWeight = pow(lumaWeight, remapSaturate(avg, 0.02f, 0.001f) * 0.4f + 0.2f); +tileWeight *= lumaWeight; +vec2 sampleExposure = vec2(tileExposure * tileWeight, tileWeight); +sampleExposure = subgroupInclusiveAdd(sampleExposure); +if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleExposure; +barrier(); +for (uint i = 0; i < loopLength; i++) { +if ((gl_SubgroupID & (1u << i)) > 0u) { +sampleExposure += prefixSumCache[(gl_SubgroupID >> i << i) - 1u]; +if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = sampleExposure; +} +barrier(); +} +if (gl_LocalInvocationIndex == 511u) { +float avgExposure = max(sampleExposure.x / sampleExposure.y * 29.3f, 1.0E-10f); +avgExposure = log2(avgExposure); +float prevAvgExposure = log2(texelFetch(pixelData2D, ivec2(0, 0), 0).x); +float frameTimeFixed = frameTime + step(frameCounter, 20) * 100.0f; +float exposureTime = clamp(frameTimeFixed * (2.0f / 1.0f), 0.0f, 1.0f); +avgExposure = mix(prevAvgExposure, avgExposure, exposureTime); +avgExposure = max(exp2(avgExposure), 1.0E-5f); +float exposure = GetExposureValue(avgExposure); +imageStore(img_pixelData2D, ivec2(0, 0), vec4(avgExposure, exposure, 0.0f, 0.0f)); +} +} +)glsl"); + + if (!FindUniqueTokenSequence(tokens, expectedMain, match.mainBegin, match.mainEnd) || + CountTokenSequence(tokens, + {"layout", "(", "local_size_x", "=", "32", ",", "local_size_y", "=", "16", + ")", "in", ";"}) != 1 || + CountTokenSequence(tokens, {"shared", "vec2", "prefixSumCache", "[", "32", "]", ";"}) != 1 || + CountTokenSequence(tokens, + {"float", "GetExposureValue", "(", "float", "luminance", ")", "{"}) != 1 || + CountTokenSequence(tokens, {"uniform", "int", "frameCounter", ";"}) != 1 || + CountTokenSequence(tokens, {"uniform", "float", "frameTime", ";"}) != 1 || + CountTokenSequence(tokens, {"uniform", "float", "aspectRatio", ";"}) != 1 || + CountTokenSequence(tokens, {"uniform", "vec2", "pixelSize", ";"}) != 1 || + CountTokenSequence(tokens, {"uniform", "sampler2D", "colortex2", ";"}) != 1 || + CountTokenSequence(tokens, {"uniform", "sampler2D", "pixelData2D", ";"}) != 1 || + CountTokenSequence(tokens, + {"layout", "(", "rg16f", ")", "uniform", "image2D", "img_pixelData2D", ";"}) != + 1) { + return false; + } + + // No second user of the scratch array or lane-width-sensitive builtin may survive the + // rewrite. These counts describe the fully matched main body plus its one declaration. + if (CountToken(tokens, "prefixSumCache") != 9 || CountToken(tokens, "GetExposureValue") != 2 || + CountToken(tokens, "subgroupInclusiveAdd") != 2 || + CountToken(tokens, "gl_SubgroupInvocationID") != 4 || CountToken(tokens, "gl_SubgroupSize") != 4 || + CountToken(tokens, "gl_SubgroupID") != 8 || CountToken(tokens, "gl_NumSubgroups") != 2 || + CountToken(tokens, "gl_LocalInvocationIndex") != 2 || CountToken(tokens, "barrier") != 5 || + CountToken(tokens, "findMSB") != 1 || + HasIdentifierWithPrefixOutsideAllowed(tokens, "subgroup", {"subgroupInclusiveAdd"}) || + HasIdentifierWithPrefixOutsideAllowed( + tokens, "gl_Subgroup", + {"gl_SubgroupInvocationID", "gl_SubgroupSize", "gl_SubgroupID", "gl_NumSubgroups"}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SubGroup", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Warp", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Thread", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SMID", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "ballot", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "shuffle", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "readInvocation", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "readFirstInvocation", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "anyInvocation", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "allInvocations", {}) || + HasIdentifierWithPrefixOutsideAllowed(tokens, "mglExposure", {})) { + return false; + } + return true; + } + + String BuildWeightedExposureReductionReplacement() { + return R"glsl(void main() { +if (gl_LocalInvocationIndex != 0u) return; +float mglExposureAverage = 0.0f; +for (uint mglExposureY = 0u; mglExposureY < 16u; ++mglExposureY) { +for (uint mglExposureX = 0u; mglExposureX < 32u; ++mglExposureX) { +vec2 mglExposureTexCoord = (vec2(mglExposureX, mglExposureY) + 0.5f) * vec2(1.0f / 32.0f, 1.0f / 16.0f); +vec2 mglExposureSampleCoord = mglExposureTexCoord * (1.0f / 64.0f); +mglExposureSampleCoord.x += (15.0f / 32.0f) + pixelSize.x * 12.0f; +mglExposureAverage += dot(textureLod(colortex2, mglExposureSampleCoord, 0.0f).rgb, + vec3(0.2125f, 0.7154f, 0.0721f)); +} +} +mglExposureAverage /= 512.0f; +vec2 mglExposureWeightedSum = vec2(0.0f); +for (uint mglExposureY = 0u; mglExposureY < 16u; ++mglExposureY) { +for (uint mglExposureX = 0u; mglExposureX < 32u; ++mglExposureX) { +vec2 mglExposureTexCoord = (vec2(mglExposureX, mglExposureY) + 0.5f) * vec2(1.0f / 32.0f, 1.0f / 16.0f); +vec2 mglExposureSampleCoord = mglExposureTexCoord * (1.0f / 64.0f); +mglExposureSampleCoord.x += (15.0f / 32.0f) + pixelSize.x * 12.0f; +float mglExposureTile = dot(textureLod(colortex2, mglExposureSampleCoord, 0.0f).rgb, + vec3(0.2125f, 0.7154f, 0.0721f)); +vec2 mglExposureDistance = mglExposureTexCoord * 2.0f - 1.0f; +mglExposureDistance.y /= aspectRatio; +float mglExposureWeight = remapSaturate(length(mglExposureDistance), 0.6f, 0.4f); +mglExposureTile = max(7.0E-7f, mglExposureTile); +float mglExposureLumaWeight = mglExposureAverage / mglExposureTile; +mglExposureLumaWeight = pow(mglExposureLumaWeight, + remapSaturate(mglExposureAverage, 0.02f, 0.001f) * 0.4f + 0.2f); +mglExposureWeight *= mglExposureLumaWeight; +mglExposureWeightedSum += vec2(mglExposureTile * mglExposureWeight, mglExposureWeight); +} +} +float avgExposure = max(mglExposureWeightedSum.x / mglExposureWeightedSum.y * 29.3f, 1.0E-10f); +avgExposure = log2(avgExposure); +float prevAvgExposure = log2(texelFetch(pixelData2D, ivec2(0, 0), 0).x); +float frameTimeFixed = frameTime + step(frameCounter, 20) * 100.0f; +float exposureTime = clamp(frameTimeFixed * (2.0f / 1.0f), 0.0f, 1.0f); +avgExposure = mix(prevAvgExposure, avgExposure, exposureTime); +avgExposure = max(exp2(avgExposure), 1.0E-5f); +float exposure = GetExposureValue(avgExposure); +imageStore(img_pixelData2D, ivec2(0, 0), vec4(avgExposure, exposure, 0.0f, 0.0f)); +})glsl"; + } + void SkipDirectiveWhitespace(const MobileGL::String& source, SizeT& pos, SizeT lineEnd) { while (pos < lineEnd && std::isspace(static_cast(source[pos]))) { pos++; @@ -1256,15 +1455,12 @@ namespace MobileGL { Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize, String& source) { constexpr Uint32 capturedSubgroupSize = 32; - if (stage != ShaderStage::Compute || nativeSubgroupSize <= capturedSubgroupSize || - nativeSubgroupSize % capturedSubgroupSize != 0) { - return false; - } - - // Vulkan subgroup widths are powers of two. Keep the workaround restricted to - // wider widths which are a power-of-two multiple of the captured 32-lane model. - const Uint32 subgroupScale = nativeSubgroupSize / capturedSubgroupSize; - if ((subgroupScale & (subgroupScale - 1u)) != 0u) { + const Bool narrowSubgroup = nativeSubgroupSize != 0u && nativeSubgroupSize < 16u && + capturedSubgroupSize % nativeSubgroupSize == 0u; + const Bool wideSubgroup = nativeSubgroupSize > capturedSubgroupSize && + nativeSubgroupSize % capturedSubgroupSize == 0u; + if (stage != ShaderStage::Compute || !IsPowerOfTwo(nativeSubgroupSize) || + (!narrowSubgroup && !wideSubgroup)) { return false; } @@ -1276,8 +1472,8 @@ namespace MobileGL { // silently falls back to the driver's miscompiled path. Make that visible. if (CountToken(tokens, "subgroupInclusiveAdd") > 0) { MGLOG_W_ONCE("%s: subgroupInclusiveAdd present but the linear prefix-scan template " - "did not match; the wide-subgroup rewrite was NOT applied", - __func__); + "did not match; the subgroup-compatibility rewrite was NOT applied", + __func__); } return false; } @@ -1291,6 +1487,30 @@ namespace MobileGL { return true; } + Bool RewriteWeightedExposureSubgroupReductionForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize, + String& source) { + if (stage != ShaderStage::Compute || !IsPowerOfTwo(nativeSubgroupSize) || + nativeSubgroupSize >= 16u) { + return false; + } + + const Vector tokens = TokenizeCode(source); + WeightedExposureReductionMatch match; + if (!ParseWeightedExposureReductionTemplate(tokens, match)) { + if (CountToken(tokens, "subgroupInclusiveAdd") == 2 && + CountToken(tokens, "GetExposureValue") > 0) { + MGLOG_W_ONCE("%s: weighted exposure subgroup reductions were present but the complete " + "template did not match; the narrow-subgroup rewrite was NOT applied", + __func__); + } + return false; + } + + source.replace(match.mainBegin, match.mainEnd - match.mainBegin, + BuildWeightedExposureReductionReplacement()); + return true; + } + namespace { struct ShaderSourceQuirkContext { ShaderStage stage = ShaderStage::Unknown; @@ -1320,16 +1540,20 @@ namespace MobileGL { "subgroup-prefix-scan-rewrite", [](const CompileEnv& env) { return env.subgroupPrefixScanQuirk; }, [](const ShaderSourceQuirkContext& ctx) { - // Qualcomm's Vulkan driver miscompiles the recognized float - // InclusiveScan pattern for native subgroups wider than the - // captured 32 lanes; other vendors compile it correctly and - // should keep their native scan. + // Narrow subgroups overflow the pack's fixed subgroup-result scratch + // arrays. Qualcomm also miscompiles the recognized float InclusiveScan + // pattern when its native subgroup is wider than the captured 32 lanes. return ctx.backend == BackendType::DirectVulkan && - ctx.vendor == MG_Backend::GpuVendorKind::Qualcomm; + ((ctx.subgroupSize != 0u && ctx.subgroupSize < 16u) || + (ctx.vendor == MG_Backend::GpuVendorKind::Qualcomm && + ctx.subgroupSize > 32u)); }, [](const ShaderSourceQuirkContext& ctx, String& source) { - return RewriteLinearSubgroupPrefixScanForVulkan(ctx.stage, ctx.subgroupSize, - source); + const Bool exposureRewritten = RewriteWeightedExposureSubgroupReductionForVulkan( + ctx.stage, ctx.subgroupSize, source); + const Bool prefixScanRewritten = RewriteLinearSubgroupPrefixScanForVulkan( + ctx.stage, ctx.subgroupSize, source); + return exposureRewritten || prefixScanRewritten; }, }, }; diff --git a/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.h b/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.h index d2027256..5589c711 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.h +++ b/MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.h @@ -30,17 +30,16 @@ namespace MobileGL { // tests and diagnostics that drive the preprocessor standalone. void PreprocessShaderSource(ShaderStage stage, String& source); - // Some desktop-captured compute shaders build a workgroup-wide linear prefix scan - // from subgroupInclusiveAdd plus a shared array of subgroup totals. Qualcomm's - // Vulkan driver miscompiles that exact float InclusiveScan path for native subgroups - // wider than the capture's 32 lanes. For the narrowly recognized, uniform-control- - // flow template, replace the subgroup-local scan with a shared-memory, strict - // left-fold over virtual 32-lane segments. Returns true only when the complete safe - // template was recognized and rewritten. PreprocessShaderSource reaches this through - // its device-quirk registry: by default only on detected Qualcomm Vulkan devices, - // overridable either way with MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN=1/0. The explicit - // entry point exists for deterministic tests. + // Some desktop-captured compute shaders size shared scratch for a 32-lane subgroup + // model. Narrow Vulkan subgroups can produce more subgroup totals than that storage + // holds; Qualcomm also miscompiles one recognized scan when its subgroup is wider. + // These entry points replace only complete, known-safe templates with lane-independent + // algorithms. PreprocessShaderSource reaches them through its device-quirk registry; + // MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN=1/0 overrides the automatic device gate. The + // explicit entry points exist for deterministic tests. Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize, String& source); + Bool RewriteWeightedExposureSubgroupReductionForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize, + String& source); // Rewrites a "#version 330 core" directive that PreprocessShaderSource normalized down // from a legacy desktop version back up to "#version 460 core". Returns false (leaving