diff --git a/CMakeLists.txt b/CMakeLists.txt index 5bb93da1..4c357c43 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -293,6 +293,7 @@ set(SOURCE_FILES MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateSubgroupsPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DArrayImagesPass.cpp + MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/BakeImageFormatsPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/WidenImageFormatsPass.cpp MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ClampMultisampleFetchPass.cpp diff --git a/MobileGL/MG_Backend/DirectGLES/Managers.cpp b/MobileGL/MG_Backend/DirectGLES/Managers.cpp index c5abeea7..9220ac87 100644 --- a/MobileGL/MG_Backend/DirectGLES/Managers.cpp +++ b/MobileGL/MG_Backend/DirectGLES/Managers.cpp @@ -5665,6 +5665,27 @@ namespace MobileGL::MG_Backend::DirectGLES { effectiveSpirv = &arrayImageSpirv; } + // The SAMPLER half of the same 1D story, and a defect one layer deeper than the one + // above. SPIRV-Cross DOES widen a 1D sampler's coordinate for ES - it just prints the + // OFFSET and the two GRADIENT operands with the arity the desktop shader spelled, so + // textureLodOffset(sampler1DArray, vec2, float, int) is emitted against a + // sampler2DArray and the driver answers "no matching overloaded function found", + // losing the stage and silently no-oping every dispatch that used it. Widening the + // operands alone would be an INVALID module (the validator derives the required arity + // from the image's own Dim), so the pass moves the type to 2D and widens coordinate, + // offset and gradients together. + // + // NO KEY MATERIAL, by the same test LegalizeStorageBlockArrayIndexingForEssl passes: + // it takes the module and nothing else, no capability bit arms it, and it self-gates + // on the module's own content (BinaryHasOffsetOrGrad1DSampledImage). The module is + // already the largest thing in the L2 key, so it is covered completely. + Vector sampled1DSpirv; + if (MG_Util::ShaderTranspiler::ShaderCompiler::Lower1DSampledImagesForEssl( + *effectiveSpirv, sampled1DSpirv, enableSpirvValidation) && + !sampled1DSpirv.empty()) { + effectiveSpirv = &sampled1DSpirv; + } + // GLSL ES has no format-less image: `writeonly uniform uimage2D` is legal desktop // GLSL 4.2 and an Adreno ES compile error ("all images have to define layout // format"), which loses the whole program. Give each such image the format the diff --git a/MobileGL/MG_Test/Program/ProgramUtilTest.cpp b/MobileGL/MG_Test/Program/ProgramUtilTest.cpp index 4a680bd0..c3c08fe7 100644 --- a/MobileGL/MG_Test/Program/ProgramUtilTest.cpp +++ b/MobileGL/MG_Test/Program/ProgramUtilTest.cpp @@ -22,6 +22,7 @@ #include #include #include +#include #include #include #include @@ -3651,6 +3652,290 @@ void main() { ssb.sum = uint(imageSize(i0).x) + imageLoad(i0, ivec2(0, 0)).r; } << "declining means the 1D-array type is still there for the driver to reject"; } +// --- 1D SAMPLED images (Lower1DSampledImagesPass) ---------------------------------------------- +// +// The other half of the 1D story. SPIRV-Cross DOES widen a 1D sampler's coordinate for ES - the +// test above pins that - but it prints the OFFSET and the two GRADIENT operands with the arity the +// desktop shader spelled, against a sampler it has just declared 2D. The result has no ESSL +// overload, the driver says "no matching overloaded function found", and the stage is lost. + +namespace { + // Same word walk as the storage-image counters, for Sampled == 1. + SizeT Count1DSampledImageTypes(const Vector& spirv) { + constexpr unsigned kOpTypeImage = 25, kDim1D = 0; + SizeT count = 0; + for (SizeT i = 5; i < spirv.size();) { + const unsigned wordCount = spirv[i] >> 16; + const unsigned opcode = spirv[i] & 0xFFFFu; + if (wordCount == 0 || i + wordCount > spirv.size()) break; + if (opcode == kOpTypeImage && wordCount >= 8 && spirv[i + 3] == kDim1D && + spirv[i + 7] == 1u) { + ++count; + } + i += wordCount; + } + return count; + } + + // KHR-GL43.compute_shader.resource-texture's own sampler1DArray lookup, minus the other eight + // samplers: a textureLodOffset whose offset is the scalar GL gives a 1D array. + const char* k1DArraySamplerOffsetCompute = R"(#version 440 core +layout (local_size_x = 1) in; +uniform sampler1DArray g_sampler4; +layout (std430, binding = 0) buffer SSB { vec4 data; } ssb; +void main() { ssb.data = textureLodOffset(g_sampler4, vec2(0.5, 1.0), 0.0, 0); } +)"; +} // namespace + +// The negative control, and the whole reason the pass exists: SPIRV-Cross emits the sampler as 2D +// and widens the coordinate, then hands the scalar offset straight through. Pinning the upstream +// behaviour here means that if a future SPIRV-Cross bump fixes it, this test fails and says so, +// rather than the pass quietly becoming dead weight. +TEST_F(ProgramUtilTest, SpirvCrossEmitsAScalarOffsetFor1DSamplers) { + using namespace MG_Util::ShaderTranspiler; + + const Vector spirv = BuildSpirvForStage(k1DArraySamplerOffsetCompute, GL_COMPUTE_SHADER); + ASSERT_FALSE(spirv.empty()); + ASSERT_EQ(Count1DSampledImageTypes(spirv), 1u) + << "glslang no longer emits a Dim1D/Sampled=1 image for sampler1DArray"; + + const String essl = DecompileToEssl(spirv); + ASSERT_FALSE(essl.empty()); + EXPECT_NE(essl.find("sampler2DArray"), String::npos) + << "SPIRV-Cross declares the 1D array sampler as 2D on ES; that half it does do:\n" << essl; + EXPECT_EQ(essl.find("ivec2"), String::npos) + << "SPIRV-Cross is expected to pass the SCALAR offset straight through, so nothing in this " + "fixture builds an ivec2 - its absence IS the defect, because ESSL has no " + "textureLodOffset(sampler2DArray, vec3, float, int). If this no longer happens, " + "Lower1DSampledImagesForEssl may no longer be needed:\n" + << essl; +} + +// The fix: the type becomes a 2D array and the offset becomes two components, so the call +// type-checks against the declaration SPIRV-Cross was already emitting. +TEST_F(ProgramUtilTest, Lower1DSampledImagesWidensTheOffsetOfA1DArrayLookup) { + using namespace MG_Util::ShaderTranspiler; + + const Vector raw = BuildSpirvForStage(k1DArraySamplerOffsetCompute, GL_COMPUTE_SHADER); + ASSERT_FALSE(raw.empty()); + + // Through the shared chain first, exactly as the DirectGLES transpile path does - the same + // reason the storage-image tests above do it: the pass runs on sanitized bytes, and validating + // raw glslang output would latch pre-existing properties against this pass. + Vector spirv; + ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv)); + ASSERT_TRUE(Lower1DSampledImagesPass::BinaryHasOffsetOrGrad1DSampledImage(spirv)) + << "the fixture must reproduce the defect before the fix is asked to remove it:\n" + << DisassembleSpirv(spirv); + + const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount(); + + Vector lowered; + ASSERT_TRUE(ShaderCompiler::Lower1DSampledImagesForEssl(spirv, lowered, true)); + ASSERT_FALSE(lowered.empty()); + + EXPECT_EQ(Count1DSampledImageTypes(lowered), 0u) + << "no 1D sampled image type may survive the pass:\n" + << DisassembleSpirv(lowered); + // The point of moving the TYPE rather than only the operand: an ivec2 offset against a type + // still declared Dim1D is an invalid module, and the validator would say so. + EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore) + << "the lowered module must stay validator-clean:\n" + << DisassembleSpirv(lowered); + + const String essl = DecompileToEssl(lowered); + ASSERT_FALSE(essl.empty()); + EXPECT_NE(essl.find("sampler2DArray"), String::npos) + << "the sampler must still be declared as the 2D array the texture is stored as:\n" << essl; + EXPECT_NE(essl.find("ivec2"), String::npos) + << "the offset must now be the two-component one ESSL's sampler2DArray overload takes:\n" + << essl; +} + +// The gradients take the identical repair, and through a different SPIRV-Cross branch - the offset +// is emitted at `if (args.offset)` and the gradients at `if (args.grad_x || args.grad_y)`, so one +// fixture cannot cover both. +TEST_F(ProgramUtilTest, Lower1DSampledImagesWidensTheGradientsOfA1DLookup) { + using namespace MG_Util::ShaderTranspiler; + + const Vector raw = BuildSpirvForStage(R"(#version 440 core +layout (local_size_x = 1) in; +uniform sampler1D g_sampler0; +layout (std430, binding = 0) buffer SSB { vec4 data; } ssb; +void main() { ssb.data = textureGrad(g_sampler0, 0.5, 0.25, 0.125); } +)", + GL_COMPUTE_SHADER); + ASSERT_FALSE(raw.empty()); + + Vector spirv; + ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv)); + ASSERT_TRUE(Lower1DSampledImagesPass::BinaryHasOffsetOrGrad1DSampledImage(spirv)) + << DisassembleSpirv(spirv); + + const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount(); + + Vector lowered; + ASSERT_TRUE(ShaderCompiler::Lower1DSampledImagesForEssl(spirv, lowered, true)); + ASSERT_FALSE(lowered.empty()); + + EXPECT_EQ(Count1DSampledImageTypes(lowered), 0u) << DisassembleSpirv(lowered); + EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore) + << "the lowered module must stay validator-clean:\n" + << DisassembleSpirv(lowered); + + const String essl = DecompileToEssl(lowered); + ASSERT_FALSE(essl.empty()); + EXPECT_NE(essl.find("textureGrad"), String::npos) << essl; + // Both derivatives have to be widened, not just the first: ESSL's overload takes two vec2s. + EXPECT_NE(essl.find("vec2(0.25, 0.0)"), String::npos) + << "dPdx must be widened to two components:\n" << essl; + EXPECT_NE(essl.find("vec2(0.125, 0.0)"), String::npos) + << "dPdy must be widened too:\n" << essl; +} + +// Scope: a 1D sampler that is only SAMPLED or FETCHED is emitted correctly by the very same +// SPIRV-Cross code, so the pass must not touch it. Replacing working emission with our own buys +// nothing and risks everything - the same rule the storage-image sibling applies to a 1D image +// with no atomic on it. resource-texture's own sampler1D is exactly this shape (it only calls +// texelFetch), so this is not a hypothetical. +TEST_F(ProgramUtilTest, Lower1DSampledImagesLeavesPlainLookupsToSpirvCross) { + using namespace MG_Util::ShaderTranspiler; + + const Vector spirv = BuildSpirvForStage(R"(#version 440 core +layout (local_size_x = 1) in; +uniform sampler1D g_sampler0; +uniform sampler1DArray g_sampler4; +layout (std430, binding = 0) buffer SSB { vec4 data; } ssb; +void main() { + ssb.data = texelFetch(g_sampler0, 2, 0) + texture(g_sampler4, vec2(0.5, 1.0)); +} +)", + GL_COMPUTE_SHADER); + ASSERT_FALSE(spirv.empty()); + ASSERT_EQ(Count1DSampledImageTypes(spirv), 2u); + EXPECT_FALSE(Lower1DSampledImagesPass::BinaryHasOffsetOrGrad1DSampledImage(spirv)) + << "no offset and no gradient here, so the probe must say there is nothing to do"; + + Vector lowered; + ASSERT_TRUE(ShaderCompiler::Lower1DSampledImagesForEssl(spirv, lowered, true)); + EXPECT_EQ(lowered, spirv) << "a 1D sampler with no offset or gradient must pass through byte " + "for byte"; +} + +// The gate is per arrayed-ness, matching the two distinct OpTypeImage declarations glslang emits: +// the sampler1DArray carries the offset and is rewritten, while the sampler1D in the same module +// is left to SPIRV-Cross. This is resource-texture's own shape. +TEST_F(ProgramUtilTest, Lower1DSampledImagesRewritesOnlyTheArrayednessThatCarriesTheOffset) { + using namespace MG_Util::ShaderTranspiler; + + const Vector raw = BuildSpirvForStage(R"(#version 440 core +layout (local_size_x = 1) in; +uniform sampler1D g_sampler0; +uniform sampler1DArray g_sampler4; +layout (std430, binding = 0) buffer SSB { vec4 data; } ssb; +void main() { + ssb.data = texelFetch(g_sampler0, 2, 0) + + textureLodOffset(g_sampler4, vec2(0.5, 1.0), 0.0, 0); +} +)", + GL_COMPUTE_SHADER); + ASSERT_FALSE(raw.empty()); + + Vector spirv; + ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv)); + ASSERT_EQ(Count1DSampledImageTypes(spirv), 2u); + + const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount(); + + Vector lowered; + ASSERT_TRUE(ShaderCompiler::Lower1DSampledImagesForEssl(spirv, lowered, true)); + ASSERT_FALSE(lowered.empty()); + + EXPECT_EQ(Count1DSampledImageTypes(lowered), 1u) + << "the arrayed sampler must be rewritten and the non-arrayed one left alone:\n" + << DisassembleSpirv(lowered); + EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore) + << "the lowered module must stay validator-clean:\n" + << DisassembleSpirv(lowered); + + // Both spellings coincide on ES, which is why a partial rewrite is safe here and is NOT safe + // for the storage-image sibling: SPIRV-Cross prints Dim1D as "2D" already, so the stage that + // was rewritten and the stage that was not declare the same ESSL type. + const String essl = DecompileToEssl(lowered); + ASSERT_FALSE(essl.empty()); + EXPECT_EQ(essl.find("sampler1D"), String::npos) + << "nothing may reach the driver still spelled 1D:\n" << essl; +} + +// The shape that would emit INVALID SPIR-V without the deduplication, and the shape the +// conformance case actually has: a 1D sampler and a real 2D sampler of the same sampled type in +// one module. Rewriting the first one's Dim in place makes the two OpTypeImage declarations +// structurally identical, and SPIR-V forbids duplicate non-aggregate types. +TEST_F(ProgramUtilTest, Lower1DSampledImagesDeduplicatesAgainstAnExisting2DSampler) { + using namespace MG_Util::ShaderTranspiler; + + const Vector raw = BuildSpirvForStage(R"(#version 440 core +layout (local_size_x = 1) in; +uniform sampler1D g_sampler0; +uniform sampler2D g_sampler1; +layout (std430, binding = 0) buffer SSB { vec4 data; } ssb; +void main() { + ssb.data = textureLodOffset(g_sampler0, 0.5, 0.0, 1) + + textureLod(g_sampler1, vec2(0.5), 0.0); +} +)", + GL_COMPUTE_SHADER); + ASSERT_FALSE(raw.empty()); + + Vector spirv; + ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv)); + ASSERT_EQ(Count1DSampledImageTypes(spirv), 1u); + + const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount(); + + Vector lowered; + ASSERT_TRUE(ShaderCompiler::Lower1DSampledImagesForEssl(spirv, lowered, true)); + ASSERT_FALSE(lowered.empty()); + + EXPECT_EQ(Count1DSampledImageTypes(lowered), 0u) << DisassembleSpirv(lowered); + EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore) + << "the rewritten 1D sampler collided with the module's own 2D sampler and left a " + "duplicate type declaration behind:\n" + << DisassembleSpirv(lowered); +} + +// The declined shape, for the sibling's reason: textureSize(sampler1D) yields an int and +// textureSize(sampler2D) an ivec2, so rewriting the type while leaving the query would hand the +// shader a value of the wrong shape. The module is returned untouched rather than half-translated. +TEST_F(ProgramUtilTest, Lower1DSampledImagesDeclinesAModuleThatQueriesTheTextureSize) { + using namespace MG_Util::ShaderTranspiler; + + const Vector raw = BuildSpirvForStage(R"(#version 440 core +layout (local_size_x = 1) in; +uniform sampler1D g_sampler0; +layout (std430, binding = 0) buffer SSB { vec4 data; } ssb; +void main() { + ssb.data = textureLodOffset(g_sampler0, 0.5, 0.0, 1) + float(textureSize(g_sampler0, 0)); +} +)", + GL_COMPUTE_SHADER); + ASSERT_FALSE(raw.empty()); + + Vector spirv; + ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv)); + ASSERT_TRUE(Lower1DSampledImagesPass::BinaryHasOffsetOrGrad1DSampledImage(spirv)) + << "the fixture must still carry the offset that arms the pass, so that the decline is " + "what leaves the module alone rather than the gate:\n" + << DisassembleSpirv(spirv); + + Vector lowered; + ASSERT_TRUE(ShaderCompiler::Lower1DSampledImagesForEssl(spirv, lowered, true)); + EXPECT_EQ(lowered, spirv) + << "a declined module must be handed back untouched, not partly rewritten"; + EXPECT_EQ(Count1DSampledImageTypes(lowered), 1u) + << "declining means the 1D type is still there for the driver to reject"; +} + // --- image format qualifier bake (BakeImageFormatsPass) --------------------------------------- // // Desktop GLSL 4.2 lets a writeonly image declaration omit its format layout qualifier; GLSL ES diff --git a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp index bd15b5aa..1e5c5b70 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp +++ b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp @@ -33,6 +33,7 @@ #include "SpirvPasses/FixIterationRPSubgroupScratchPass.h" #include "SpirvPasses/NormalizeRectCoordinatesPass.h" #include "SpirvPasses/Lower1DArrayImagesPass.h" +#include "SpirvPasses/Lower1DSampledImagesPass.h" #include "SpirvPasses/BakeImageFormatsPass.h" #include "SpirvPasses/WidenImageFormatsPass.h" #include "SpirvPasses/ClampMultisampleFetchPass.h" @@ -1128,6 +1129,39 @@ namespace MobileGL { return RunOptimizerChecked("Lower1DArrayImagesForEssl", optimizer, inputBinary, outputBinary, true, enableSpirvValidation); } + bool ShaderCompiler::Lower1DSampledImagesForEssl(const Vector& inputBinary, + Vector& outputBinary, + const bool enableSpirvValidation) { + using namespace spvtools; + + // The overwhelmingly common answer, and the reason the probe exists: no 1D sampler + // is reached by an offset or a gradient, so the module is handed back byte for + // byte without an Optimizer ever being built. Every ESSL shader in the process + // passes through here, so the cost of the case with nothing to do is the cost of + // this pass. Note the probe is deliberately NARROWER than "declares a 1D sampler": + // SPIRV-Cross emits the plain sample and fetch forms correctly, and taking those + // over would be a regression looking for somewhere to happen. + if (!Lower1DSampledImagesPass::BinaryHasOffsetOrGrad1DSampledImage(inputBinary)) { + outputBinary = inputBinary; + return true; + } + + Optimizer optimizer(SPV_ENV_VULKAN_1_1); + optimizer.RegisterPass(Lower1DSampledImagesPass::CreateLower1DSampledImagesPass()); + // Mandatory, not tidying - the same collision Lower1DArrayImagesForEssl documents + // one screen up. Rewriting a 1D sampled image type to the 2D one makes it + // structurally IDENTICAL to any real 2D sampled image of the same sampled type the + // module already declared, and SPIR-V forbids duplicate non-aggregate type + // declarations. That is not exotic here: it is the exact shape of the headline + // case, whose compute shader declares sampler1D and sampler2D side by side. The + // same applies to the OpTypeSampledImage and OpTypePointer instructions above + // them, and to the Sampled1D capability the rewrite turns into a second Shader. + optimizer.RegisterPass(CreateRemoveDuplicatesPass()); + + return RunOptimizerChecked("Lower1DSampledImagesForEssl", optimizer, inputBinary, + outputBinary, true, enableSpirvValidation); + } + bool ShaderCompiler::RebaseInstanceIndexForVulkan(const Vector& inputBinary, Vector& outputBinary, const bool enableSpirvValidation) { using namespace spvtools; diff --git a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h index 5f844d6f..bf92aafd 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h +++ b/MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.h @@ -209,6 +209,19 @@ namespace MobileGL { static bool Lower1DArrayImagesForEssl(const Vector& inputBinary, Vector& outputBinary, bool enableSpirvValidation = false); + // The SAMPLED-image counterpart. SPIRV-Cross widens a 1D sampler's COORDINATE for + // ES and prints the OFFSET and GRADIENT operands with their original 1D arity, so + // textureOffset / textureLodOffset / texelFetchOffset / textureGrad on a + // sampler1D(Array) come out with no ESSL overload ("no matching overloaded + // function found") and the stage is lost. Rewrites the type to 2D and widens + // coordinate, offset and gradients together. DirectGLES transpile path only - + // Vulkan has 1D images natively. Copies the input through untouched unless the + // module actually carries such an operand on a 1D sampler, so a shader that only + // samples or fetches keeps SPIRV-Cross's own correct emission. See + // Lower1DSampledImagesPass for what it declines and why. + static bool Lower1DSampledImagesForEssl(const Vector& inputBinary, + Vector& outputBinary, + bool enableSpirvValidation = false); // Gives each format-less storage image the format bound to its image unit, so // the emitted ESSL can carry the format layout qualifier GLSL ES requires of // every image and desktop GLSL lets a writeonly declaration omit. `glFormatByName` diff --git a/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.cpp b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.cpp new file mode 100644 index 00000000..7c8e406c --- /dev/null +++ b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.cpp @@ -0,0 +1,652 @@ +// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.cpp +// Copyright (c) 2025-2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#include "Lower1DSampledImagesPass.h" + +#include "spirv.hpp" +#include "source/opt/build_module.h" +#include "source/opt/constants.h" +#include "source/opt/def_use_manager.h" +#include "source/opt/instruction.h" +#include "source/opt/ir_builder.h" +#include "source/opt/ir_context.h" +#include "source/opt/module.h" +#include "source/opt/type_manager.h" +#include "source/opt/types.h" +#include "source/util/make_unique.h" + +#include +#include + +namespace MobileGL { + namespace MG_Util { + namespace ShaderTranspiler { + namespace { + using spvtools::opt::Instruction; + using spvtools::opt::InstructionBuilder; + using spvtools::opt::IRContext; + namespace analysis = spvtools::opt::analysis; + + // OpTypeImage in-operands: 0 sampled type, 1 Dim, 2 Depth, 3 Arrayed, 4 MS, + // 5 Sampled, 6 Format. + constexpr uint32_t kDimOperand = 1; + constexpr uint32_t kArrayedOperand = 3; + constexpr uint32_t kSampledOperand = 5; + + // Sampled == 1 is SPIR-V's "used WITH a sampler", i.e. exactly the sampler + // uniforms this pass exists for. Sampled == 2 is the storage image + // Lower1DArrayImagesPass owns, and Sampled == 0 ("either") is a shape glslang + // never emits from GLSL - left out so an unexpected module is declined rather + // than rewritten on a guess. + bool Is1DSampledImageType(const Instruction* imageType) { + return imageType != nullptr && imageType->opcode() == spv::Op::OpTypeImage && + imageType->NumInOperands() > kSampledOperand && + static_cast(imageType->GetSingleWordInOperand(kDimOperand)) == + spv::Dim::Dim1D && + imageType->GetSingleWordInOperand(kSampledOperand) == 1u; + } + + bool Is1DSampledImageTypeOfArrayedness(const Instruction* imageType, bool arrayed) { + return Is1DSampledImageType(imageType) && + (imageType->GetSingleWordInOperand(kArrayedOperand) == 1u) == arrayed; + } + + // Any Dim1D image still declared with Sampled == 1. Used only to decide whether + // the Sampled1D capability is still needed after the rewrite. + bool AnyDim1DSampledTypeLeft(IRContext* context) { + for (const Instruction& type : context->module()->types_values()) { + if (Is1DSampledImageType(&type)) return true; + } + return false; + } + + // The OpTypeImage behind whatever an image operation was handed - a bare image, a + // sampled image, or a pointer/array of either. Same unwrapping as + // Lower1DArrayImagesPass, which needs the identical walk. + Instruction* ResolveImageType(IRContext* context, uint32_t objectId) { + auto* defUseMgr = context->get_def_use_mgr(); + Instruction* object = defUseMgr->GetDef(objectId); + if (object == nullptr) return nullptr; + Instruction* type = defUseMgr->GetDef(object->type_id()); + while (type != nullptr) { + switch (type->opcode()) { + case spv::Op::OpTypeImage: + return type; + case spv::Op::OpTypeSampledImage: + case spv::Op::OpTypePointer: + case spv::Op::OpTypeArray: + case spv::Op::OpTypeRuntimeArray: + // Each names its element type in its last in-operand, except arrays, + // whose element type is the FIRST. Both are reached here because a + // sampler uniform may be declared as an array of samplers. + type = defUseMgr->GetDef( + type->opcode() == spv::Op::OpTypeArray || + type->opcode() == spv::Op::OpTypeRuntimeArray + ? type->GetSingleWordInOperand(0) + : type->GetSingleWordInOperand(type->NumInOperands() - 1)); + continue; + default: + return nullptr; + } + } + return nullptr; + } + + // How this pass classifies an opcode that can touch one of these images. + enum class OpKind { + // Not an image operation at all: it may CARRY the image or sampled-image + // value (OpLoad, OpSampledImage, OpCopyObject, ...) but it names no + // coordinate, so the rewrite does not reach it. + NotImageOp, + // Addresses texels: has a coordinate at in-operand 1 and, from + // `imageOperandsIndex`, an optional image-operands mask. + Texel, + // Reads a property whose result does not depend on Dim. Safe to leave. + DimIndependentQuery, + // Recognised, and refused: rewriting the type would change the shape of what + // the shader consumes, or the operation is one this pass has no translation + // for. + Decline, + }; + + struct OpClassification { + OpKind kind = OpKind::NotImageOp; + uint32_t coordinateOperand = 1; + // In-operand index of the ImageOperands mask, when the opcode has one. The + // mask itself is OPTIONAL for the implicit-Lod, fetch and gather forms, so + // this is an index to test against NumInOperands(), not a promise. + uint32_t imageOperandsIndex = 0; + }; + + OpClassification ClassifyOpcode(spv::Op opcode) { + switch (opcode) { + // (image, coordinate, [operands]) - the mask, when present, is in-operand 2. + case spv::Op::OpImageSampleImplicitLod: + case spv::Op::OpImageSampleExplicitLod: + case spv::Op::OpImageSampleProjImplicitLod: + case spv::Op::OpImageSampleProjExplicitLod: + case spv::Op::OpImageFetch: + case spv::Op::OpImageSparseSampleImplicitLod: + case spv::Op::OpImageSparseSampleExplicitLod: + case spv::Op::OpImageSparseSampleProjImplicitLod: + case spv::Op::OpImageSparseSampleProjExplicitLod: + case spv::Op::OpImageSparseFetch: + return {OpKind::Texel, 1u, 2u}; + + // (image, coordinate, D_ref, [operands]) - one operand more before the mask. + case spv::Op::OpImageSampleDrefImplicitLod: + case spv::Op::OpImageSampleDrefExplicitLod: + case spv::Op::OpImageSampleProjDrefImplicitLod: + case spv::Op::OpImageSampleProjDrefExplicitLod: + case spv::Op::OpImageSparseSampleDrefImplicitLod: + case spv::Op::OpImageSparseSampleDrefExplicitLod: + case spv::Op::OpImageSparseSampleProjDrefImplicitLod: + case spv::Op::OpImageSparseSampleProjDrefExplicitLod: + return {OpKind::Texel, 1u, 3u}; + + // OpImageQueryLod names a coordinate and no mask. Its coordinate is the PLANE + // components only (no array layer), which the insert-at-1 rule widens just as + // correctly as a sampling coordinate. + case spv::Op::OpImageQueryLod: + return {OpKind::Texel, 1u, /*no mask*/ 0xFFFFFFFFu}; + + // Scalar result, identical for Dim1D and Dim2D. + case spv::Op::OpImageQueryLevels: + return {OpKind::DimIndependentQuery, 0u, 0u}; + + // textureSize: int for a sampler1D, ivec2 for the sampler2D it would become. + // There is no correct narrower answer to substitute, so the module is left + // alone - the sibling pass refuses the same shape for the same reason. + case spv::Op::OpImageQuerySize: + case spv::Op::OpImageQuerySizeLod: + // Gather is not available for 1D samplers in GLSL, so reaching one here means + // an input this pass did not anticipate; and its ConstOffsets operand is an + // ARRAY of offsets whose widening this pass does not implement. + case spv::Op::OpImageGather: + case spv::Op::OpImageDrefGather: + case spv::Op::OpImageSparseGather: + case spv::Op::OpImageSparseDrefGather: + // Storage-image traffic has no business reaching a Sampled == 1 image; if it + // does, the module is not the shape this pass reasoned about. + case spv::Op::OpImageRead: + case spv::Op::OpImageWrite: + case spv::Op::OpImageSparseRead: + case spv::Op::OpImageTexelPointer: + case spv::Op::OpImageQuerySamples: + return {OpKind::Decline, 0u, 0u}; + + default: + return {OpKind::NotImageOp, 0u, 0u}; + } + } + + // How many ids each ImageOperands bit contributes, in the bit order SPIR-V lays + // them out in. Only the bits that carry ids need an entry; the rest contribute + // nothing and are skipped by having a count of zero. + struct ImageOperandBit { + spv::ImageOperandsMask bit; + uint32_t idCount; + }; + constexpr ImageOperandBit kImageOperandBits[] = { + {spv::ImageOperandsMask::Bias, 1u}, + {spv::ImageOperandsMask::Lod, 1u}, + {spv::ImageOperandsMask::Grad, 2u}, + {spv::ImageOperandsMask::ConstOffset, 1u}, + {spv::ImageOperandsMask::Offset, 1u}, + {spv::ImageOperandsMask::ConstOffsets, 1u}, + {spv::ImageOperandsMask::Sample, 1u}, + {spv::ImageOperandsMask::MinLod, 1u}, + {spv::ImageOperandsMask::MakeTexelAvailable, 1u}, + {spv::ImageOperandsMask::MakeTexelVisible, 1u}, + {spv::ImageOperandsMask::NonPrivateTexel, 0u}, + {spv::ImageOperandsMask::VolatileTexel, 0u}, + {spv::ImageOperandsMask::SignExtend, 0u}, + {spv::ImageOperandsMask::ZeroExtend, 0u}, + {spv::ImageOperandsMask::Nontemporal, 0u}, + {spv::ImageOperandsMask::Offsets, 1u}, + }; + + // Where each of the operands this pass rewrites sits, for one instruction. An + // index of 0 means "not present" - in-operand 0 is always the image, so it can + // never be a real position for one of these. + struct OperandPositions { + uint32_t gradX = 0; + uint32_t gradY = 0; + uint32_t constOffset = 0; + uint32_t offset = 0; + // A bit this pass does not know how to widen appeared on a covered image. + bool unsupported = false; + + bool Any() const { return gradX != 0 || constOffset != 0 || offset != 0; } + }; + + OperandPositions LocateOperands(const Instruction& instruction, + uint32_t imageOperandsIndex) { + OperandPositions positions; + if (imageOperandsIndex == 0xFFFFFFFFu || + instruction.NumInOperands() <= imageOperandsIndex) { + return positions; + } + const uint32_t mask = instruction.GetSingleWordInOperand(imageOperandsIndex); + uint32_t next = imageOperandsIndex + 1u; + for (const ImageOperandBit& entry : kImageOperandBits) { + if ((mask & static_cast(entry.bit)) == 0u) continue; + switch (entry.bit) { + case spv::ImageOperandsMask::Grad: + positions.gradX = next; + positions.gradY = next + 1u; + break; + case spv::ImageOperandsMask::ConstOffset: + positions.constOffset = next; + break; + case spv::ImageOperandsMask::Offset: + positions.offset = next; + break; + case spv::ImageOperandsMask::ConstOffsets: + case spv::ImageOperandsMask::Offsets: + // An array of offsets, only meaningful for gather - which is declined + // above. Refuse rather than translate half of it. + positions.unsupported = true; + break; + default: + break; + } + next += entry.idCount; + } + // Every id the mask claimed has to actually be there; a truncated operand + // list means the instruction is not the shape this walk assumed. + if (next > instruction.NumInOperands()) { + positions.unsupported = true; + } + return positions; + } + + // Whether this instruction so much as mentions a value whose type resolves to a + // covered image. Used to make sure nothing reaches these images through an opcode + // this pass never considered: the answer decides between rewriting and declining, + // never between two different rewrites. + template + bool MentionsCoveredImage(IRContext* context, const Instruction& instruction, + const CoveredFn& covered) { + bool mentions = false; + instruction.ForEachInId([&](const uint32_t* id) { + if (mentions || id == nullptr) return; + if (covered(ResolveImageType(context, *id))) mentions = true; + }); + return mentions; + } + + // The component type of a value, and how many of them it has. A scalar reports a + // count of 1; anything that is neither an int/float scalar nor a vector of one + // reports 0, which every caller treats as "not a shape this pass translates". + struct ValueShape { + const analysis::Type* componentType = nullptr; + uint32_t componentCount = 0; + bool IsScalar() const { return componentCount == 1u; } + }; + + ValueShape DescribeValue(IRContext* context, uint32_t valueId) { + ValueShape shape; + Instruction* def = context->get_def_use_mgr()->GetDef(valueId); + if (def == nullptr) return shape; + const analysis::Type* type = context->get_type_mgr()->GetType(def->type_id()); + if (type == nullptr) return shape; + const analysis::Vector* asVector = type->AsVector(); + const analysis::Type* component = + asVector != nullptr ? asVector->element_type() : type; + if (component == nullptr) return shape; + if (component->AsInteger() == nullptr && component->AsFloat() == nullptr) { + return shape; + } + shape.componentType = component; + shape.componentCount = asVector != nullptr ? asVector->element_count() : 1u; + return shape; + } + + // Which 1D sampled images this module is to be rewritten for, decided per + // arrayed-ness because that is the granularity of the OpTypeImage declarations + // glslang emits. A category is in scope only when the module actually performs a + // lookup on it carrying an Offset, ConstOffset or Grad - the operands SPIRV-Cross + // prints with the wrong arity - so a shader that only samples and fetches keeps + // SPIRV-Cross's own correct emission untouched. + struct LoweringScope { + bool arrayed = false; + bool nonArrayed = false; + + bool Any() const { return arrayed || nonArrayed; } + bool Covers(const Instruction* imageType) const { + return (arrayed && Is1DSampledImageTypeOfArrayedness(imageType, true)) || + (nonArrayed && Is1DSampledImageTypeOfArrayedness(imageType, false)); + } + }; + + LoweringScope ResolveLoweringScope(IRContext* context) { + LoweringScope scope; + // The type table settles the common case, and it is nearly every shader: no + // 1D sampled image declared at all, so the code is never walked. + bool declared = false; + for (const Instruction& type : context->module()->types_values()) { + if (Is1DSampledImageType(&type)) { + declared = true; + break; + } + } + if (!declared) return scope; + + for (auto& function : *context->module()) { + for (auto& block : function) { + for (auto& instruction : block) { + const OpClassification classification = + ClassifyOpcode(instruction.opcode()); + if (classification.kind != OpKind::Texel || + instruction.NumInOperands() <= classification.coordinateOperand) { + continue; + } + const Instruction* imageType = + ResolveImageType(context, instruction.GetSingleWordInOperand(0)); + if (!Is1DSampledImageType(imageType)) continue; + const OperandPositions positions = + LocateOperands(instruction, classification.imageOperandsIndex); + if (!positions.Any()) continue; + if (imageType->GetSingleWordInOperand(kArrayedOperand) == 1u) { + scope.arrayed = true; + } else { + scope.nonArrayed = true; + } + } + } + } + return scope; + } + + // Everything this pass will touch, collected before a single word is changed. + // Planning first is what lets every refusal be a clean "leave the module alone": + // there is no point at which the module is half converted and the pass then + // discovers it cannot finish. + struct RewritePlan { + struct Site { + Instruction* instruction = nullptr; + uint32_t coordinateOperand = 0; + OperandPositions operands; + }; + std::vector sites; + bool declined = false; + }; + + RewritePlan PlanRewrite(IRContext* context, const LoweringScope& scope) { + RewritePlan plan; + const auto covered = [&scope](const Instruction* type) { + return scope.Covers(type); + }; + + for (auto& function : *context->module()) { + for (auto& block : function) { + for (auto& instruction : block) { + const OpClassification classification = + ClassifyOpcode(instruction.opcode()); + + if (classification.kind == OpKind::NotImageOp || + classification.kind == OpKind::DimIndependentQuery) { + // These name no coordinate, so they need no rewrite - but an + // opcode this pass has never classified must not reach one of + // these images unnoticed. NotImageOp is the catch-all, so the + // check is on it. + if (classification.kind == OpKind::NotImageOp && + instruction.opcode() != spv::Op::OpLoad && + instruction.opcode() != spv::Op::OpStore && + instruction.opcode() != spv::Op::OpCopyObject && + instruction.opcode() != spv::Op::OpSampledImage && + instruction.opcode() != spv::Op::OpImage && + instruction.opcode() != spv::Op::OpAccessChain && + instruction.opcode() != spv::Op::OpInBoundsAccessChain && + instruction.opcode() != spv::Op::OpPhi && + instruction.opcode() != spv::Op::OpSelect && + instruction.opcode() != spv::Op::OpFunctionCall && + MentionsCoveredImage(context, instruction, covered)) { + plan.declined = true; + return plan; + } + continue; + } + + if (instruction.NumInOperands() < 1) continue; + const Instruction* imageType = + ResolveImageType(context, instruction.GetSingleWordInOperand(0)); + if (!scope.Covers(imageType)) continue; + + if (classification.kind == OpKind::Decline) { + plan.declined = true; + return plan; + } + if (instruction.NumInOperands() <= classification.coordinateOperand) { + plan.declined = true; + return plan; + } + + const OperandPositions positions = + LocateOperands(instruction, classification.imageOperandsIndex); + if (positions.unsupported) { + plan.declined = true; + return plan; + } + + // Confirm here, before anything is written, that every operand + // about to be widened has the shape the widening assumes. The + // coordinate may be a scalar or a short vector; the offset and + // the two gradients must be SCALARS, which for a Dim1D image is + // not an assumption but the validator's own rule + // (GetPlaneCoordSize(1D) == 1). Checking it up front is what + // keeps the apply phase total. + const ValueShape coordinate = DescribeValue( + context, instruction.GetSingleWordInOperand( + classification.coordinateOperand)); + if (coordinate.componentCount == 0u || coordinate.componentCount > 3u) { + plan.declined = true; + return plan; + } + const uint32_t scalarOperands[] = {positions.gradX, positions.gradY, + positions.offset, + positions.constOffset}; + for (const uint32_t position : scalarOperands) { + if (position == 0u) continue; + if (!DescribeValue(context, + instruction.GetSingleWordInOperand(position)) + .IsScalar()) { + plan.declined = true; + return plan; + } + } + // ConstOffset has to stay a constant expression, so its widened + // form is built as a module-scope constant - which is only + // possible if the operand really is one. + if (positions.constOffset != 0u && + context->get_constant_mgr()->FindDeclaredConstant( + instruction.GetSingleWordInOperand(positions.constOffset)) == + nullptr) { + plan.declined = true; + return plan; + } + + plan.sites.push_back( + {&instruction, classification.coordinateOperand, positions}); + } + } + } + return plan; + } + } // namespace + + bool Lower1DSampledImagesPass::BinaryHasOffsetOrGrad1DSampledImage( + const Vector& binary) { + if (binary.empty()) { + return false; + } + std::unique_ptr context = spvtools::BuildModule( + SPV_ENV_VULKAN_1_1, + [](spv_message_level_t, const char*, const spv_position_t&, const char*) {}, + binary.data(), binary.size()); + if (!context) { + return false; + } + return ResolveLoweringScope(context.get()).Any(); + } + + spvtools::opt::Pass::Status Lower1DSampledImagesPass::Process() { + auto* irContext = context(); + auto* typeMgr = irContext->get_type_mgr(); + auto* constantMgr = irContext->get_constant_mgr(); + + const LoweringScope scope = ResolveLoweringScope(irContext); + if (!scope.Any()) { + return Status::SuccessWithoutChange; + } + + RewritePlan plan = PlanRewrite(irContext, scope); + if (plan.declined) { + return Status::SuccessWithoutChange; + } + + // A zero of a given 32-bit scalar type. The literal word is the VALUE's bit + // pattern, which for a float zero is 0 as well - so one helper serves the integer + // coordinate of a fetch, the float coordinate of a sample and the float gradients + // alike, without a second spelling to keep in step. + const auto zeroOf = [&](const analysis::Type* componentType, + uint32_t componentTypeId) -> uint32_t { + const analysis::Constant* constant = + constantMgr->GetConstant(componentType, {0u}); + if (constant == nullptr) return 0u; + const Instruction* defining = + constantMgr->GetDefiningInstruction(constant, componentTypeId); + return defining != nullptr ? defining->result_id() : 0u; + }; + + // The whole of the arity repair, in one place: insert a zero at component 1. + // Scalar u becomes (u, 0); (u, layer) becomes (u, 0, layer); (u, q) becomes + // (u, 0, q). See the header for why one rule covers every shape. + const auto widen = [&](uint32_t valueId, Instruction* before, + bool mustBeConstant) -> uint32_t { + const ValueShape shape = DescribeValue(irContext, valueId); + if (shape.componentCount == 0u) return 0u; + + const uint32_t componentTypeId = typeMgr->GetTypeInstruction(shape.componentType); + if (componentTypeId == 0u) return 0u; + analysis::Vector widenedCandidate(shape.componentType, shape.componentCount + 1u); + const uint32_t widenedTypeId = typeMgr->GetTypeInstruction(&widenedCandidate); + const uint32_t zeroId = zeroOf(shape.componentType, componentTypeId); + if (widenedTypeId == 0u || zeroId == 0u) return 0u; + + // ConstOffset must remain a constant expression - the validator says so + // outright ("Expected Image Operand ConstOffset to be a const object") - so + // for it the widened value is built as a module-scope OpConstantComposite + // rather than as an instruction in the block. Only the scalar shape is + // reachable: the plan phase refuses anything else, because a Dim1D image's + // offset has exactly one component by the validator's own arity rule. + if (mustBeConstant) { + if (!shape.IsScalar()) return 0u; + const analysis::Type* widenedType = typeMgr->GetType(widenedTypeId); + const analysis::Constant* widenedConstant = + widenedType != nullptr + ? constantMgr->GetConstant(widenedType, {valueId, zeroId}) + : nullptr; + if (widenedConstant == nullptr) return 0u; + const Instruction* defining = + constantMgr->GetDefiningInstruction(widenedConstant, widenedTypeId); + return defining != nullptr ? defining->result_id() : 0u; + } + + InstructionBuilder builder( + irContext, before, + IRContext::kAnalysisDefUse | IRContext::kAnalysisInstrToBlockMapping); + std::vector componentIds; + componentIds.reserve(shape.componentCount + 1u); + if (shape.IsScalar()) { + componentIds.push_back(valueId); + componentIds.push_back(zeroId); + } else { + for (uint32_t i = 0; i < shape.componentCount; ++i) { + Instruction* extracted = + builder.AddCompositeExtract(componentTypeId, valueId, {i}); + if (extracted == nullptr) return 0u; + componentIds.push_back(extracted->result_id()); + if (i == 0u) componentIds.push_back(zeroId); + } + } + Instruction* widened = + builder.AddCompositeConstruct(widenedTypeId, componentIds); + return widened != nullptr ? widened->result_id() : 0u; + }; + + for (RewritePlan::Site& site : plan.sites) { + Instruction* instruction = site.instruction; + + struct Target { + uint32_t position; + bool mustBeConstant; + }; + const Target targets[] = { + {site.coordinateOperand, false}, + {site.operands.gradX, false}, + {site.operands.gradY, false}, + {site.operands.offset, false}, + {site.operands.constOffset, true}, + }; + for (const Target& target : targets) { + // Position 0 is the image operand, so it is this plan's "absent" marker + // for everything except the coordinate, which is never 0. + if (target.position == 0u) continue; + const uint32_t widenedId = + widen(instruction->GetSingleWordInOperand(target.position), instruction, + target.mustBeConstant); + if (widenedId == 0u) { + // Reachable only if the module's shapes disagree with what the plan + // recorded. Failing here makes the caller keep the input binary, + // which is the same outcome as a decline. + return Status::Failure; + } + instruction->SetInOperand(target.position, {widenedId}); + } + irContext->UpdateDefUse(instruction); + } + + // Only now, with no lookup still spelling a 1D coordinate, does the type become + // the 2D one - which is what ES stores a GL_TEXTURE_1D(_ARRAY) as anyway + // (MapToBackendTextureTarget), and what SPIRV-Cross was already PRINTING for it. + for (Instruction& type : irContext->types_values()) { + if (scope.Covers(&type)) { + type.SetInOperand(kDimOperand, {static_cast(spv::Dim::Dim2D)}); + } + } + + // Sampled1D describes the types just rewritten. Drop it only if no 1D SAMPLED + // image is left at all - a module may still hold one this pass left alone (a + // category with no offset or gradient on it), and that one still needs the + // capability. Image1D is deliberately untouched: it belongs to the storage images + // Lower1DArrayImagesPass owns, and they may still be Dim1D here. Shader is + // declared by any module reaching this point, so restating it keeps the + // instruction valid and RemoveDuplicates collapses the pair. + if (!AnyDim1DSampledTypeLeft(irContext)) { + for (Instruction& capability : irContext->capabilities()) { + const auto value = + static_cast(capability.GetSingleWordInOperand(0)); + if (value == spv::Capability::Sampled1D) { + capability.SetInOperand(0, {static_cast(spv::Capability::Shader)}); + } + } + } + + irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone); + return Status::SuccessWithChange; + } + + spvtools::Optimizer::PassToken Lower1DSampledImagesPass::CreateLower1DSampledImagesPass() { + return spvtools::Optimizer::PassToken( + spvtools::MakeUnique()); + } + } // namespace ShaderTranspiler + } // namespace MG_Util +} // namespace MobileGL diff --git a/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.h b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.h new file mode 100644 index 00000000..6b16a85a --- /dev/null +++ b/MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.h @@ -0,0 +1,116 @@ +// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.h +// Copyright (c) 2025-2026 MobileGL-Dev +// Licensed under the GNU Lesser General Public License v3.0: +// https://www.gnu.org/licenses/gpl-3.0.txt +// https://www.gnu.org/licenses/lgpl-3.0.txt +// SPDX-License-Identifier: LGPL-3.0-only +// End of Source File Header + +#pragma once + +#include "spirv-tools/optimizer.hpp" +#include "source/opt/pass.h" + +#include + +namespace MobileGL { + namespace MG_Util { + namespace ShaderTranspiler { + // The SAMPLED-image half of the 1D story. Lower1DArrayImagesPass owns the storage + // half and says there, correctly for what it needed, that SPIRV-Cross's SAMPLER path + // "already handles the 1D-array shape correctly and must be left to it". That is true + // of the COORDINATE and false of everything else the lookup carries. + // + // ES has no 1D texture, so SPIRV-Cross emits a 1D sampler as a 2D one - `case Dim1D: + // res += options.es ? "2D" : "1D"` - and fakes the missing coordinate component at + // each call site (spirv_glsl.cpp, the `imgtype.image.dim == Dim1D && options.es` + // branches: `vec2(coord, 0.0)` non-arrayed, `vec3(coord.x, 0.0, coord.y)` arrayed, + // which is the same (u, 0, layer) the 2D-array texture actually stores). But the + // OFFSET operand and the two GRADIENT operands are printed straight through with + // their original 1D arity: + // + // if (args.offset) { ...; farg_str += bitcast_expression(SPIRType::Int, args.offset); } + // if (args.grad_x || args.grad_y) { ...; farg_str += to_expression(args.grad_x); ... } + // + // So a `textureLodOffset(sampler1DArray, vec2, float, int)` comes out as + // `textureLodOffset(sampler2DArray, vec3, float, int)`, for which ESSL has no + // overload, and the driver answers "'textureLodOffset' : no matching overloaded + // function found". That loses the stage, and with it the program - which is how ONE + // sampler1DArray lookup took down the nine-sampler compute shader of + // KHR-GL43.compute_shader.resource-texture, whose dispatch then silently did nothing + // and left the SSBO reading back the zeros the test uploaded. + // + // Observed failing on an Adreno 830 by isolating each form: textureOffset, + // textureLodOffset and texelFetchOffset on both sampler1D and sampler1DArray, and + // textureGrad on sampler1DArray. The same shaders with a 2D sampler compile, so the + // functions exist - only the argument arity is wrong. + // + // WHY NOT PATCH SPIRV-CROSS. 3rdparty/SPIRV-Cross is a submodule pinned to KhronosGroup + // upstream, not to a MobileGL fork (contrast 3rdparty/glslang), so an in-tree edit + // would live outside this repository's history. + // + // WHY NOT WIDEN JUST THE OPERANDS. Emitting an ivec2 offset against a type still + // declared Dim1D is an INVALID module, not a clever shortcut: the validator computes + // the required arity from the image's own Dim (validate_image.cpp, GetPlaneCoordSize + // -> "Expected Image Operand Offset to have 1 component") and would latch a failure on + // every validating lane. So the type has to move too, and once it does the coordinate + // has to move with it - which is what this pass does, in the module, before + // SPIRV-Cross ever applies its own emulation. + // + // The rewrite is exactly SPIRV-Cross's own, restated on the SPIR-V side so that + // coordinate, offset and gradient are all widened by one piece of code: a zero is + // INSERTED AT COMPONENT 1 of each. That single rule is right for every shape, because + // a 1D coordinate lays out as [u][array layer][proj q] and the plane occupies index 0 + // alone - so (u) -> (u, 0), (u, layer) -> (u, 0, layer) and (u, q) -> (u, 0, q) all + // fall out of it, and so do the scalar offset -> ivec2 and the scalar gradients -> + // vec2. The Dref value is a separate SPIR-V operand rather than a coordinate + // component, so the shadow forms need nothing extra. + // + // NO CROSS-STAGE HAZARD, and this is the one place this pass is on firmer ground than + // its storage-image sibling, whose header records the opposite as a known limitation. + // That pass can rewrite uimage1DArray to uimage2DArray in one stage and decline in + // another, and the two then spell the SAME uniform `uimage2D` and `uimage2DArray` and + // the ES link fails on a type mismatch. Here the two spellings COINCIDE: SPIRV-Cross + // prints Dim1D as "2D" on ES already, so a stage this pass rewrote and a stage it left + // alone both declare `sampler2D` / `sampler2DArray`. Partial application across a + // program's stages is therefore invisible at the interface. + // + // Deliberately narrow, on three axes - the sibling's reasoning, applied to this + // resource: + // + // * SAMPLED images only (Sampled == 1). Storage images are the sibling's. + // * Only when the module actually carries an Offset, ConstOffset or Grad operand on + // a 1D sampled image, i.e. only where SPIRV-Cross's emission is ALREADY broken. + // A shader that only calls texture()/textureLod()/texelFetch() on a sampler1D + // keeps taking SPIRV-Cross's own (correct) output byte for byte, so this pass has + // no way to regress it. The gate is decided per arrayed-ness, matching the two + // distinct OpTypeImage declarations glslang emits. + // * ESSL only. Vulkan has VK_IMAGE_VIEW_TYPE_1D natively and the offset and gradient + // arities are the ones the module already spells, so DirectVulkan must see the + // module unchanged. + // + // A size query on a covered image is DECLINED rather than half-translated, for the + // sibling's reason: textureSize(sampler1D) yields an int and textureSize(sampler2D) an + // ivec2, so rewriting the type while leaving the query would hand the shader a value of + // the wrong shape. Refusing leaves the module byte for byte and is no worse than today. + // + // Every decline is decided BEFORE anything is rewritten - the pass plans the whole + // edit, and only then applies it - so there is no state in which it has half-converted + // a module and then given up. Anything it does not recognise reaching one of these + // images (a gather, an unexpected image opcode) is a decline, not a guess. + class Lower1DSampledImagesPass final : public spvtools::opt::Pass { + public: + const char* name() const override { return "mobilegl-lower-1d-sampled-images"; } + Status Process() override; + + // Whether a module carries the shape this pass exists for: a 1D SAMPLED image + // reached by a lookup with an Offset, ConstOffset or Grad operand. One parse + // answers it, and the answer is no for very nearly every shader - the common path + // must not build an Optimizer at all. + static bool BinaryHasOffsetOrGrad1DSampledImage(const Vector& binary); + + static spvtools::Optimizer::PassToken CreateLower1DSampledImagesPass(); + }; + } // namespace ShaderTranspiler + } // namespace MG_Util +} // namespace MobileGL diff --git a/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h b/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h index b371e8f0..45f89a22 100644 --- a/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h +++ b/MobileGL/MG_Util/ShaderTranspiler/TranslationCache.h @@ -566,9 +566,9 @@ namespace MobileGL::MG_Util::ShaderTranspiler { // // Unconditional passes take no input but the module and so need no key material: // StripUboMemberRelaxedPrecision, LowerRectImages, Lower1DArrayImages, - // LegalizeStorageBlockArrayIndexing and FlattenAtomicCounterBlockOffsets. Each self-gates - // on the module's own content and is armed by nothing, so the SPIR-V already in this key - // covers them completely. + // Lower1DSampledImages, LegalizeStorageBlockArrayIndexing and + // FlattenAtomicCounterBlockOffsets. Each self-gates on the module's own content and is + // armed by nothing, so the SPIR-V already in this key covers them completely. // // THE TEST FOR THAT CLAIM IS NOT THE SIGNATURE. LowerViewportIndexForEssl is equally // module-only to look at, yet SupportsViewportArray is in this key because that bit ARMS