mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-08 20:28:32 +09:00
[Merge] (ShaderTranspiler, DirectGLES): land the fp64 block layout, colour-index and readonly-writeonly repairs
This commit is contained in:
@@ -279,6 +279,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecomposeWorkgroupVec3Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecoratePositionInvariantPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DemoteFloat64Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenFloat64StorageBlockPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerDrawParametersPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerViewportIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PackDoubleVertexInputsPass.cpp
|
||||
|
||||
@@ -6106,6 +6106,22 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
spvcSession.SetAtomicCounterBlockBindings(atomicCounterEsslBindingTop,
|
||||
outAtomicCounterGlBindings);
|
||||
|
||||
// `layout(index = 0)` is the GL default spelled out loud, and GLSL ES has no such
|
||||
// qualifier in core - a stage that prints it is refused with "index layout
|
||||
// qualifier requires EXT_blend_func_extended" and the whole program then draws
|
||||
// nothing. Drop the decoration when it carries the default; a REAL dual-source
|
||||
// index (1) is left alone, because that one genuinely needs the extension and the
|
||||
// driver has to see it. Fragment stage only: no other stage can carry it.
|
||||
if (glShaderType == GL_FRAGMENT_SHADER) {
|
||||
spvcSession.DropDefaultFragmentOutputColorIndex();
|
||||
}
|
||||
|
||||
// `readonly writeonly` together says the buffer variable can only be asked its
|
||||
// .length(), which the frontend has already enforced - so the pair is inert, and
|
||||
// printing it is not. Mesa's ES compiler refuses a block spelled that way and the
|
||||
// stage never reaches the program.
|
||||
spvcSession.RelaxReadWriteExclusiveStorageBuffers();
|
||||
|
||||
const char* result = nullptr;
|
||||
spvcSession.Compile(&result);
|
||||
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
// which is the whole point.
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
@@ -153,6 +154,151 @@ void main() {
|
||||
std::string m_buildLog;
|
||||
};
|
||||
|
||||
// A SHADER STORAGE BLOCK that holds doubles is the one place the narrowing is NOT free:
|
||||
// demoting `double` to `float` also repacks the block, and the bytes the application
|
||||
// wrote into the buffer do not move with it. Every member past the first double then
|
||||
// reads and writes at the wrong offset, and the block is simply shorter than the one
|
||||
// that was bound - the tail of it is never touched at all
|
||||
// (KHR-GL43.shader_storage_buffer_object.basic-stdLayout-case3, whose output matched its
|
||||
// input up to the first double's slot and was zero from there on).
|
||||
//
|
||||
// The block layout is fixed by GL 4.6 core 7.6.2.2 and is asserted here as literal byte
|
||||
// offsets rather than queried, so this says what the SPEC requires and not what MobileGL
|
||||
// happens to report. Both packings are covered because they differ in exactly the places
|
||||
// that matter: std140 rounds an array's stride and a matrix's column stride up to 16,
|
||||
// std430 does not, and only std430 packs the scalars tightly.
|
||||
//
|
||||
// Every value is exactly representable in binary32, so a correct implementation copies
|
||||
// the block BYTE FOR BYTE even though it narrows each double on the way through.
|
||||
constexpr const char* kBlockCopySource = R"(#version 430 core
|
||||
layout(local_size_x = 1) in;
|
||||
layout(std140, binding = 0) buffer In140 {
|
||||
int data0;
|
||||
float data1[3];
|
||||
mat3x2 data2;
|
||||
double data3;
|
||||
double data4[2];
|
||||
int data5;
|
||||
dvec3 data6;
|
||||
} g_in140;
|
||||
layout(std430, binding = 1) buffer In430 {
|
||||
int data0;
|
||||
float data1[3];
|
||||
mat3x2 data2;
|
||||
double data3;
|
||||
double data4[2];
|
||||
int data5;
|
||||
dvec3 data6;
|
||||
} g_in430;
|
||||
layout(std140, binding = 2) buffer Out140 {
|
||||
int data0;
|
||||
float data1[3];
|
||||
mat3x2 data2;
|
||||
double data3;
|
||||
double data4[2];
|
||||
int data5;
|
||||
dvec3 data6;
|
||||
} g_out140;
|
||||
layout(std430, binding = 3) buffer Out430 {
|
||||
int data0;
|
||||
float data1[3];
|
||||
mat3x2 data2;
|
||||
double data3;
|
||||
double data4[2];
|
||||
int data5;
|
||||
dvec3 data6;
|
||||
} g_out430;
|
||||
void main() {
|
||||
g_out140.data0 = g_in140.data0;
|
||||
for (int i = 0; i < 3; ++i) g_out140.data1[i] = g_in140.data1[i];
|
||||
g_out140.data2 = g_in140.data2;
|
||||
g_out140.data3 = g_in140.data3;
|
||||
for (int i = 0; i < 2; ++i) g_out140.data4[i] = g_in140.data4[i];
|
||||
g_out140.data5 = g_in140.data5;
|
||||
g_out140.data6 = g_in140.data6;
|
||||
|
||||
g_out430.data0 = g_in430.data0;
|
||||
for (int i = 0; i < 3; ++i) g_out430.data1[i] = g_in430.data1[i];
|
||||
g_out430.data2 = g_in430.data2;
|
||||
g_out430.data3 = g_in430.data3;
|
||||
for (int i = 0; i < 2; ++i) g_out430.data4[i] = g_in430.data4[i];
|
||||
g_out430.data5 = g_in430.data5;
|
||||
g_out430.data6 = g_in430.data6;
|
||||
}
|
||||
)";
|
||||
|
||||
// GL 4.6 core 7.6.2.2 rule by rule, for the block above.
|
||||
// std140: an array's element stride and a matrix's column stride round up to 16, a
|
||||
// double aligns to 8 and a dvec3 to 32.
|
||||
// std430: the same without the rounding - so the scalars pack tightly and only the
|
||||
// dvec3's 32-byte alignment leaves a hole.
|
||||
struct BlockLayout {
|
||||
int data0;
|
||||
int data1;
|
||||
int data1Stride;
|
||||
int data2;
|
||||
int data2ColumnStride;
|
||||
int data3;
|
||||
int data4;
|
||||
int data4Stride;
|
||||
int data5;
|
||||
int data6;
|
||||
int size;
|
||||
};
|
||||
constexpr BlockLayout kStd140{0, 16, 16, 64, 16, 112, 128, 16, 160, 192, 216};
|
||||
constexpr BlockLayout kStd430{0, 4, 4, 16, 8, 40, 48, 8, 64, 96, 120};
|
||||
|
||||
void PokeInt(std::vector<unsigned char>& bytes, int offset, int value) {
|
||||
std::memcpy(&bytes[static_cast<std::size_t>(offset)], &value, sizeof(value));
|
||||
}
|
||||
void PokeFloat(std::vector<unsigned char>& bytes, int offset, float value) {
|
||||
std::memcpy(&bytes[static_cast<std::size_t>(offset)], &value, sizeof(value));
|
||||
}
|
||||
void PokeDouble(std::vector<unsigned char>& bytes, int offset, double value) {
|
||||
std::memcpy(&bytes[static_cast<std::size_t>(offset)], &value, sizeof(value));
|
||||
}
|
||||
|
||||
// The block's contents, at the offsets the standard puts them. Padding stays zero, which
|
||||
// is what makes a byte-for-byte comparison against the (zero-initialised) output buffer
|
||||
// catch a member that landed somewhere it should not have.
|
||||
std::vector<unsigned char> MakeBlockContents(const BlockLayout& layout) {
|
||||
std::vector<unsigned char> bytes(static_cast<std::size_t>(layout.size), 0);
|
||||
PokeInt(bytes, layout.data0, 1);
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
PokeFloat(bytes, layout.data1 + i * layout.data1Stride, 2.0f + static_cast<float>(i));
|
||||
}
|
||||
// Column-major, two rows per column.
|
||||
for (int column = 0; column < 3; ++column) {
|
||||
for (int row = 0; row < 2; ++row) {
|
||||
PokeFloat(bytes, layout.data2 + column * layout.data2ColumnStride + row * 4,
|
||||
5.0f + static_cast<float>(column * 2 + row));
|
||||
}
|
||||
}
|
||||
PokeDouble(bytes, layout.data3, 11.0);
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
PokeDouble(bytes, layout.data4 + i * layout.data4Stride, 12.0 + static_cast<double>(i));
|
||||
}
|
||||
PokeInt(bytes, layout.data5, 14);
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
PokeDouble(bytes, layout.data6 + i * 8, 15.0 + static_cast<double>(i));
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
|
||||
// Names the first byte that differs, and which member owns it, so a failure is a
|
||||
// diagnosis rather than "the buffer is wrong".
|
||||
std::string DescribeOffset(const BlockLayout& layout, int offset) {
|
||||
const std::pair<int, const char*> members[] = {
|
||||
{layout.data0, "data0"}, {layout.data1, "data1"}, {layout.data2, "data2"},
|
||||
{layout.data3, "data3"}, {layout.data4, "data4"}, {layout.data5, "data5"},
|
||||
{layout.data6, "data6"}};
|
||||
const char* owner = "(padding before data0)";
|
||||
for (const auto& [start, name] : members) {
|
||||
if (offset >= start) owner = name;
|
||||
}
|
||||
return std::string(owner);
|
||||
}
|
||||
|
||||
// Every double-typed uniform shape GLSL has, all thirteen of them, in one program - the
|
||||
// shape of KHR-GL43.compute_shader.fp64-case2. The scalar and the square matrices are
|
||||
// covered by the cases above; what only a set like this reaches is the NON-SQUARE
|
||||
@@ -819,5 +965,66 @@ void main() { o_color = vec4(0.0, 1.0, 0.0, 1.0); }
|
||||
EXPECT_EQ(FirstGLError(), 0u);
|
||||
}
|
||||
|
||||
TEST_F(DoublePrecisionScenario, AStorageBlockWithDoublesKeepsTheLayoutItWasBoundWith) {
|
||||
if (!Ready()) return;
|
||||
|
||||
GLint blocks = 0;
|
||||
glGetIntegerv(GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS, &blocks);
|
||||
if (blocks < 4) {
|
||||
GTEST_SKIP() << "GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS is " << blocks << "; this needs 4";
|
||||
}
|
||||
|
||||
const unsigned int program = CompileComputeProgram(kBlockCopySource);
|
||||
ASSERT_NE(program, 0u) << m_buildLog;
|
||||
|
||||
const std::vector<unsigned char> in140 = MakeBlockContents(kStd140);
|
||||
const std::vector<unsigned char> in430 = MakeBlockContents(kStd430);
|
||||
const std::vector<unsigned char> zero140(in140.size(), 0);
|
||||
const std::vector<unsigned char> zero430(in430.size(), 0);
|
||||
|
||||
GLuint buffers[4] = {};
|
||||
glGenBuffers(4, buffers);
|
||||
const std::vector<unsigned char>* contents[4] = {&in140, &in430, &zero140, &zero430};
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, static_cast<GLuint>(i), buffers[i]);
|
||||
glBufferData(GL_SHADER_STORAGE_BUFFER, static_cast<GLsizeiptr>(contents[i]->size()),
|
||||
contents[i]->data(), GL_DYNAMIC_COPY);
|
||||
}
|
||||
ASSERT_EQ(FirstGLError(), 0u);
|
||||
|
||||
glUseProgram(program);
|
||||
glDispatchCompute(1, 1, 1);
|
||||
glMemoryBarrier(GL_BUFFER_UPDATE_BARRIER_BIT);
|
||||
EXPECT_EQ(FirstGLError(), 0u);
|
||||
|
||||
for (int pass = 0; pass < 2; ++pass) {
|
||||
const BlockLayout& layout = pass == 0 ? kStd140 : kStd430;
|
||||
const std::vector<unsigned char>& expected = pass == 0 ? in140 : in430;
|
||||
const char* packing = pass == 0 ? "std140" : "std430";
|
||||
std::vector<unsigned char> observed(expected.size(), 0xEE);
|
||||
glBindBuffer(GL_SHADER_STORAGE_BUFFER, buffers[2 + pass]);
|
||||
glGetBufferSubData(GL_SHADER_STORAGE_BUFFER, 0,
|
||||
static_cast<GLsizeiptr>(observed.size()), observed.data());
|
||||
int mismatches = 0;
|
||||
int firstMismatch = -1;
|
||||
for (std::size_t i = 0; i < expected.size(); ++i) {
|
||||
if (expected[i] == observed[i]) continue;
|
||||
++mismatches;
|
||||
if (firstMismatch < 0) firstMismatch = static_cast<int>(i);
|
||||
}
|
||||
EXPECT_EQ(mismatches, 0)
|
||||
<< packing << " block: " << mismatches << " of " << expected.size()
|
||||
<< " bytes differ, first at byte " << firstMismatch << " (in "
|
||||
<< DescribeOffset(layout, firstMismatch < 0 ? 0 : firstMismatch)
|
||||
<< "); a block that was repacked around its doubles reads and writes every "
|
||||
"member after the first one at the wrong offset";
|
||||
}
|
||||
|
||||
glUseProgram(0);
|
||||
glDeleteProgram(program);
|
||||
glDeleteBuffers(4, buffers);
|
||||
EXPECT_EQ(FirstGLError(), 0u);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace MGITest
|
||||
|
||||
@@ -179,6 +179,13 @@ void main()
|
||||
in flat uint v_index;
|
||||
out vec4 o_color;
|
||||
void main() { o_color = vec4(0.0, 1.0, 0.0, 1.0); }
|
||||
)";
|
||||
|
||||
// The colour index spelled out at its default value. Says nothing that
|
||||
// `layout(location = 0)` alone does not, and must therefore cost nothing.
|
||||
constexpr const char* kExplicitColorIndexFS = R"(#version 420 core
|
||||
layout(location = 0, index = 0) out vec4 o_color;
|
||||
void main() { o_color = vec4(0.0, 1.0, 0.0, 1.0); }
|
||||
)";
|
||||
|
||||
class Glsl420DeclarationScenario : public ScenarioTest {
|
||||
@@ -473,4 +480,24 @@ void main() { o_color = vec4(0.0, 1.0, 0.0, 1.0); }
|
||||
EXPECT_EQ(centre.g, 255) << "the atomic-counter shader linked but painted nothing";
|
||||
}
|
||||
|
||||
// `layout(location = 0, index = 0)` is the GL default written out loud, and an application
|
||||
// is entitled to write it - KHR-GL43.shader_atomic_counters.basic-program-query does. It has
|
||||
// to reach the driver as an ORDINARY single-source output: GLSL ES has no `index` qualifier
|
||||
// in core, so a transpiler that prints the decoration back gets "index layout qualifier
|
||||
// requires EXT_blend_func_extended", the stage never compiles, the program runs with a stage
|
||||
// missing and the draw paints nothing at all. Black, not red - which is why the conformance
|
||||
// case looked like the atomic counters had stopped counting.
|
||||
TEST_F(Glsl420DeclarationScenario, AnExplicitDefaultColorIndexStillDraws) {
|
||||
if (!Ready()) return;
|
||||
|
||||
const GLuint program = Build(kQuadVS, kExplicitColorIndexFS);
|
||||
if (program == 0) return;
|
||||
|
||||
const Rgba8 centre = DrawAndRead(program);
|
||||
EXPECT_EQ(FirstGLError(), 0u);
|
||||
EXPECT_EQ(centre.g, 255) << "a fragment output declared layout(location = 0, index = 0) painted "
|
||||
"nothing; its stage was almost certainly refused by the driver";
|
||||
EXPECT_EQ(centre.r, 0u);
|
||||
}
|
||||
|
||||
} // namespace MGITest
|
||||
|
||||
@@ -64,6 +64,25 @@ void main() {
|
||||
g_length[2] = g_input23[0].data.length();
|
||||
g_length[3] = g_input23[1].data.length();
|
||||
}
|
||||
)";
|
||||
|
||||
// GL 4.6 core 4.10 lets a buffer variable be declared readonly AND writeonly at once:
|
||||
// it can then be neither read nor written, and `.length()` is the only thing left that
|
||||
// may be asked of it. The pair is inert - and printing it into ESSL is not, because
|
||||
// SPIRV-Cross hoists the qualifiers every member shares onto the BLOCK and Mesa's ES
|
||||
// compiler refuses that spelling ("Interface block sets both readonly and writeonly").
|
||||
// Lifted from KHR-GL43.shader_storage_buffer_object.basic-readonly-writeonly.
|
||||
constexpr const char* kReadonlyWriteonlyComputeSource = R"(#version 430 core
|
||||
layout(local_size_x = 1) in;
|
||||
layout(std430, binding = 0) buffer Input {
|
||||
readonly writeonly int g_in[];
|
||||
};
|
||||
layout(std430, binding = 4) buffer Output {
|
||||
int g_length[];
|
||||
};
|
||||
void main() {
|
||||
g_length[0] = g_in.length();
|
||||
}
|
||||
)";
|
||||
|
||||
constexpr int kElementBytes = 16; // ivec4, std430
|
||||
@@ -212,4 +231,33 @@ void main() {
|
||||
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 0, input0);
|
||||
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 3, input3);
|
||||
}
|
||||
|
||||
// A buffer variable qualified readonly AND writeonly can only be asked its length, and that
|
||||
// question still has to be answered. A stage the driver refused answers 0 - and refuses
|
||||
// silently, because the program links without it and the dispatch is then a no-op.
|
||||
TEST_F(SsboArrayLengthScenario, AReadonlyWriteonlyArrayStillReportsItsLength) {
|
||||
if (!Ready() || IsSkipped()) return;
|
||||
|
||||
const GLuint program = CompileComputeProgram(kReadonlyWriteonlyComputeSource);
|
||||
ASSERT_NE(program, 0u) << m_buildLog;
|
||||
|
||||
const GLuint input = MakeStorageBuffer(6); // 6 ivec4 = 24 ints
|
||||
const GLuint output = MakeStorageBuffer(1);
|
||||
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 0, input);
|
||||
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 4, output);
|
||||
ASSERT_EQ(FirstGLError(), 0u);
|
||||
|
||||
glUseProgram(program);
|
||||
glDispatchCompute(1, 1, 1);
|
||||
glMemoryBarrier(GL_BUFFER_UPDATE_BARRIER_BIT);
|
||||
int length = -1;
|
||||
glBindBuffer(GL_SHADER_STORAGE_BUFFER, output);
|
||||
glGetBufferSubData(GL_SHADER_STORAGE_BUFFER, 0, sizeof(length), &length);
|
||||
EXPECT_EQ(FirstGLError(), 0u);
|
||||
EXPECT_EQ(length, 24) << "a readonly+writeonly runtime array reported length " << length
|
||||
<< "; 0 means the stage never reached the program";
|
||||
|
||||
glUseProgram(m_program);
|
||||
glDeleteProgram(program);
|
||||
}
|
||||
} // namespace MGITest
|
||||
|
||||
@@ -8,6 +8,7 @@ add_executable(
|
||||
FixIterationRPSubgroupScratchTest.cpp
|
||||
EmulateSubgroupsTest.cpp
|
||||
DemoteFloat64Test.cpp
|
||||
FlattenFloat64StorageBlockTest.cpp
|
||||
FlattenXfbInterfaceBlocksTest.cpp
|
||||
UniquifyIoBlockNamesTest.cpp
|
||||
LowerViewportIndexTest.cpp
|
||||
|
||||
@@ -0,0 +1,305 @@
|
||||
// MobileGL - MobileGL/MG_Test/ShaderTranspiler/FlattenFloat64StorageBlockTest.cpp
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// FlattenFloat64StorageBlockPass, over the module the production chain actually hands it:
|
||||
// ShaderCompiler::SanitizeAndOptimizeBinary, where the pass sits immediately before the fp64
|
||||
// demotion. The behavioural half - that a block copied through the flattened words comes back
|
||||
// byte for byte - is DoublePrecisionScenario's; what only a module walk can say is WHICH blocks
|
||||
// were flattened, how wide, and that the ones this pass must not touch came through unchanged.
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <cstring>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "Includes.h"
|
||||
#include "Init.h"
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
|
||||
#include <spirv-tools/libspirv.hpp>
|
||||
|
||||
using namespace MobileGL;
|
||||
using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler;
|
||||
|
||||
namespace {
|
||||
// A test-side reference walker, deliberately independent of the production code: a bug in
|
||||
// the pass must not be able to hide behind the same helper.
|
||||
constexpr Uint32 kSpirvHeaderWordCount = 5;
|
||||
constexpr Uint32 kOpName = 5;
|
||||
constexpr Uint32 kOpDecorate = 71;
|
||||
constexpr Uint32 kOpMemberDecorate = 72;
|
||||
constexpr Uint32 kOpTypeInt = 21;
|
||||
constexpr Uint32 kOpTypeFloat = 22;
|
||||
constexpr Uint32 kOpTypeArray = 28;
|
||||
constexpr Uint32 kOpTypeStruct = 30;
|
||||
constexpr Uint32 kOpConstant = 43;
|
||||
constexpr Uint32 kDecorationArrayStride = 6;
|
||||
constexpr Uint32 kDecorationOffset = 35;
|
||||
|
||||
template <typename Visitor>
|
||||
void ForEachInstruction(const Vector<Uint32>& spirv, Visitor&& visit) {
|
||||
for (SizeT i = kSpirvHeaderWordCount; i < spirv.size();) {
|
||||
const Uint32 wordCount = spirv[i] >> 16;
|
||||
const Uint32 opcode = spirv[i] & 0xFFFFu;
|
||||
if (wordCount == 0 || i + wordCount > spirv.size()) break;
|
||||
visit(opcode, &spirv[i], wordCount);
|
||||
i += wordCount;
|
||||
}
|
||||
}
|
||||
|
||||
Uint32 StructIdNamed(const Vector<Uint32>& spirv, const String& name) {
|
||||
Uint32 structId = 0;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpName || wordCount < 3 || structId != 0) return;
|
||||
const char* text = reinterpret_cast<const char*>(&words[2]);
|
||||
const SizeT available = static_cast<SizeT>(wordCount - 2) * sizeof(Uint32);
|
||||
// The whole name, not a prefix of it: "Wide" must not match "WideOther".
|
||||
if (available <= name.size() || text[name.size()] != 0) return;
|
||||
if (std::strncmp(text, name.c_str(), name.size()) == 0) structId = words[1];
|
||||
});
|
||||
return structId;
|
||||
}
|
||||
|
||||
// The operands of OpTypeStruct <structId>, i.e. one type id per member.
|
||||
Vector<Uint32> MemberTypesOf(const Vector<Uint32>& spirv, Uint32 structId) {
|
||||
Vector<Uint32> members;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpTypeStruct || wordCount < 2 || words[1] != structId) return;
|
||||
for (Uint32 i = 2; i < wordCount; ++i) members.push_back(words[i]);
|
||||
});
|
||||
return members;
|
||||
}
|
||||
|
||||
Vector<Uint32> MemberOffsetsOf(const Vector<Uint32>& spirv, Uint32 structId) {
|
||||
std::map<Uint32, Uint32> byMember;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpMemberDecorate || wordCount < 5 || words[1] != structId) return;
|
||||
if (words[3] != kDecorationOffset) return;
|
||||
byMember[words[2]] = words[4];
|
||||
});
|
||||
Vector<Uint32> offsets;
|
||||
for (const auto& [member, offset] : byMember) offsets.push_back(offset);
|
||||
return offsets;
|
||||
}
|
||||
|
||||
Uint32 DecorationValueOf(const Vector<Uint32>& spirv, Uint32 id, Uint32 decoration) {
|
||||
Uint32 value = 0xFFFFFFFFu;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpDecorate || wordCount < 4 || words[1] != id || words[2] != decoration) return;
|
||||
value = words[3];
|
||||
});
|
||||
return value;
|
||||
}
|
||||
|
||||
// (element type id, declared length) of OpTypeArray <arrayId>, or (0, 0).
|
||||
std::pair<Uint32, Uint32> ArrayShapeOf(const Vector<Uint32>& spirv, Uint32 arrayId) {
|
||||
Uint32 elementTypeId = 0;
|
||||
Uint32 lengthConstantId = 0;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpTypeArray || wordCount < 4 || words[1] != arrayId) return;
|
||||
elementTypeId = words[2];
|
||||
lengthConstantId = words[3];
|
||||
});
|
||||
if (elementTypeId == 0) return {0, 0};
|
||||
Uint32 length = 0;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpConstant || wordCount < 4 || words[2] != lengthConstantId) return;
|
||||
length = words[3];
|
||||
});
|
||||
return {elementTypeId, length};
|
||||
}
|
||||
|
||||
Bool IsUint32Type(const Vector<Uint32>& spirv, Uint32 typeId) {
|
||||
Bool isUint = false;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode != kOpTypeInt || wordCount < 4 || words[1] != typeId) return;
|
||||
isUint = words[2] == 32u && words[3] == 0u;
|
||||
});
|
||||
return isUint;
|
||||
}
|
||||
|
||||
Uint32 CountFloatTypesOfWidth(const Vector<Uint32>& spirv, Uint32 width) {
|
||||
Uint32 count = 0;
|
||||
ForEachInstruction(spirv, [&](Uint32 opcode, const Uint32* words, Uint32 wordCount) {
|
||||
if (opcode == kOpTypeFloat && wordCount >= 3 && words[2] == width) ++count;
|
||||
});
|
||||
return count;
|
||||
}
|
||||
|
||||
String Disassemble(const Vector<Uint32>& spirv) {
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String text;
|
||||
tools.Disassemble(spirv, &text);
|
||||
return text;
|
||||
}
|
||||
|
||||
Vector<Uint32> CompileToSpirv(GLenum stage, const String& source) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
ShaderAttrib shaderAttrib{.shaderType = stage, .sourceStr = source};
|
||||
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
|
||||
EXPECT_TRUE(shaderResult) << (shaderResult ? String{} : shaderResult.error().log);
|
||||
if (!shaderResult) return {};
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
|
||||
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
|
||||
EXPECT_TRUE(programResult) << (programResult ? String{} : programResult.error().log);
|
||||
if (!programResult) return {};
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {stage}, .program = *programResult.value()};
|
||||
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
EXPECT_TRUE(binaryResult) << (binaryResult ? String{} : binaryResult.error().log);
|
||||
if (!binaryResult || binaryResult->empty()) return {};
|
||||
return binaryResult->front();
|
||||
}
|
||||
|
||||
// The whole shared chain, exactly as the frontend runs it at link.
|
||||
Vector<Uint32> Sanitize(const Vector<Uint32>& input) {
|
||||
Vector<Uint32> output;
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
|
||||
return output;
|
||||
}
|
||||
|
||||
// The block std140 lays out as data0@0, data1[3]@16 stride 16, data2@64 column stride 16,
|
||||
// data3@112, data4[2]@128 stride 16, data5@160, data6@192 - 216 bytes, i.e. 54 words.
|
||||
constexpr const char* kStd140BlockSource = R"(#version 430 core
|
||||
layout(local_size_x = 1) in;
|
||||
layout(std140, binding = 0) buffer Wide {
|
||||
int data0;
|
||||
float data1[3];
|
||||
mat3x2 data2;
|
||||
double data3;
|
||||
double data4[2];
|
||||
int data5;
|
||||
dvec3 data6;
|
||||
} g_wide;
|
||||
void main() {
|
||||
g_wide.data0 = 1;
|
||||
for (int i = 0; i < 3; ++i) g_wide.data1[i] = float(i);
|
||||
g_wide.data2 = mat3x2(1.0);
|
||||
g_wide.data3 = 2.0lf;
|
||||
for (int i = 0; i < 2; ++i) g_wide.data4[i] = double(i);
|
||||
g_wide.data5 = 3;
|
||||
g_wide.data6 = dvec3(4.0lf);
|
||||
}
|
||||
)";
|
||||
} // namespace
|
||||
|
||||
class FlattenFloat64StorageBlockTest : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
MobileGL::Initialize();
|
||||
m_validationFailuresAtStart = ShaderCompiler::SpirvValidationFailureCount();
|
||||
}
|
||||
|
||||
void TearDown() override {
|
||||
// The wrapper validates its output on every run, so this covers every rewrite the test
|
||||
// performed without any of them having to say so.
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), m_validationFailuresAtStart)
|
||||
<< "the flattened module did not survive spirv-val";
|
||||
}
|
||||
|
||||
Uint64 m_validationFailuresAtStart = 0;
|
||||
};
|
||||
|
||||
TEST_F(FlattenFloat64StorageBlockTest, AStorageBlockWithDoublesBecomesOneWordArray) {
|
||||
const Vector<Uint32> input = CompileToSpirv(GL_COMPUTE_SHADER, kStd140BlockSource);
|
||||
ASSERT_FALSE(input.empty());
|
||||
// Before: seven members, at the std140 offsets the standard requires WITH the doubles.
|
||||
const Uint32 inputStructId = StructIdNamed(input, "Wide");
|
||||
ASSERT_NE(inputStructId, 0u) << Disassemble(input);
|
||||
EXPECT_EQ(MemberOffsetsOf(input, inputStructId),
|
||||
(Vector<Uint32>{0, 16, 64, 112, 128, 160, 192}))
|
||||
<< Disassemble(input);
|
||||
|
||||
const Vector<Uint32> output = Sanitize(input);
|
||||
ASSERT_FALSE(output.empty());
|
||||
|
||||
const Uint32 structId = StructIdNamed(output, "Wide");
|
||||
ASSERT_NE(structId, 0u) << Disassemble(output);
|
||||
const Vector<Uint32> members = MemberTypesOf(output, structId);
|
||||
ASSERT_EQ(members.size(), 1u) << "the block should have collapsed to one member\n"
|
||||
<< Disassemble(output);
|
||||
EXPECT_EQ(MemberOffsetsOf(output, structId), (Vector<Uint32>{0}));
|
||||
|
||||
const auto [elementTypeId, length] = ArrayShapeOf(output, members[0]);
|
||||
ASSERT_NE(elementTypeId, 0u) << "member 0 is not an array\n" << Disassemble(output);
|
||||
EXPECT_TRUE(IsUint32Type(output, elementTypeId)) << Disassemble(output);
|
||||
// 216 bytes is where the standard puts the end of this block; 216 / 4 = 54 words.
|
||||
EXPECT_EQ(length, 54u) << Disassemble(output);
|
||||
EXPECT_EQ(DecorationValueOf(output, members[0], kDecorationArrayStride), 4u);
|
||||
|
||||
// And the demotion that runs straight afterwards still has nothing 64-bit left to find.
|
||||
EXPECT_EQ(CountFloatTypesOfWidth(output, 64), 0u) << Disassemble(output);
|
||||
}
|
||||
|
||||
// The gate, from the other side: a storage block with no 64-bit member keeps every member and
|
||||
// every offset it was compiled with. This is what makes the pass free for every shader that does
|
||||
// not use doubles - which is all of them but a handful.
|
||||
TEST_F(FlattenFloat64StorageBlockTest, AStorageBlockWithoutDoublesIsLeftAlone) {
|
||||
const String source = R"(#version 430 core
|
||||
layout(local_size_x = 1) in;
|
||||
layout(std140, binding = 0) buffer Plain {
|
||||
int data0;
|
||||
float data1[3];
|
||||
mat3x2 data2;
|
||||
int data3;
|
||||
} g_plain;
|
||||
void main() {
|
||||
g_plain.data0 = 1;
|
||||
for (int i = 0; i < 3; ++i) g_plain.data1[i] = float(i);
|
||||
g_plain.data2 = mat3x2(1.0);
|
||||
g_plain.data3 = 2;
|
||||
}
|
||||
)";
|
||||
const Vector<Uint32> input = CompileToSpirv(GL_COMPUTE_SHADER, source);
|
||||
ASSERT_FALSE(input.empty());
|
||||
const Vector<Uint32> output = Sanitize(input);
|
||||
ASSERT_FALSE(output.empty());
|
||||
|
||||
const Uint32 structId = StructIdNamed(output, "Plain");
|
||||
ASSERT_NE(structId, 0u) << Disassemble(output);
|
||||
EXPECT_EQ(MemberTypesOf(output, structId).size(), 4u) << Disassemble(output);
|
||||
EXPECT_EQ(MemberOffsetsOf(output, structId), (Vector<Uint32>{0, 16, 64, 112}))
|
||||
<< Disassemble(output);
|
||||
}
|
||||
|
||||
// A plain UNIFORM block is deliberately NOT flattened, however many doubles it holds: the
|
||||
// frontend's glUniform*d routing is built by reflecting the DEMOTED module
|
||||
// (ProgramSpirvTask::BuildGlobalUboRouting), so a representation change there would have to move
|
||||
// with it. It keeps its members and takes the demotion's repacking, exactly as before.
|
||||
TEST_F(FlattenFloat64StorageBlockTest, AUniformBlockWithDoublesIsLeftToTheDemotion) {
|
||||
const String source = R"(#version 430 core
|
||||
layout(local_size_x = 1) in;
|
||||
layout(std140, binding = 0) uniform Params {
|
||||
int data0;
|
||||
double data1;
|
||||
int data2;
|
||||
} g_params;
|
||||
layout(std430, binding = 0) buffer Sink {
|
||||
float g_out[];
|
||||
};
|
||||
void main() {
|
||||
g_out[0] = float(g_params.data0) + float(g_params.data1) + float(g_params.data2);
|
||||
}
|
||||
)";
|
||||
const Vector<Uint32> input = CompileToSpirv(GL_COMPUTE_SHADER, source);
|
||||
ASSERT_FALSE(input.empty());
|
||||
const Vector<Uint32> output = Sanitize(input);
|
||||
ASSERT_FALSE(output.empty());
|
||||
|
||||
const Uint32 structId = StructIdNamed(output, "Params");
|
||||
ASSERT_NE(structId, 0u) << Disassemble(output);
|
||||
EXPECT_EQ(MemberTypesOf(output, structId).size(), 3u)
|
||||
<< "a uniform block must not be flattened\n"
|
||||
<< Disassemble(output);
|
||||
// The demotion's re-derived std140 layout for `int, float, int`, which is what the frontend
|
||||
// reflects and what glUniform*d then writes into.
|
||||
EXPECT_EQ(MemberOffsetsOf(output, structId), (Vector<Uint32>{0, 4, 8})) << Disassemble(output);
|
||||
}
|
||||
@@ -19,6 +19,7 @@
|
||||
#include "SpirvPasses/DecomposeWorkgroupVec3Pass.h"
|
||||
#include "SpirvPasses/DecoratePositionInvariantPass.h"
|
||||
#include "SpirvPasses/DemoteFloat64Pass.h"
|
||||
#include "SpirvPasses/FlattenFloat64StorageBlockPass.h"
|
||||
#include "SpirvPasses/LowerDrawParametersPass.h"
|
||||
#include "SpirvPasses/LowerViewportIndexPass.h"
|
||||
#include "SpirvPasses/PackDoubleVertexInputsPass.h"
|
||||
@@ -797,6 +798,17 @@ namespace MobileGL {
|
||||
// in particular it runs before the backends' PackDoubleVertexInputsPass, whose
|
||||
// OpBitcast this one would otherwise decline on. Costs one types_values() walk on
|
||||
// the overwhelming majority of modules, which declare no 64-bit float at all.
|
||||
// ...but demoting a double that lives in a SHADER STORAGE BLOCK also repacks that
|
||||
// block, and the bytes an application put in the buffer do not move with it. This
|
||||
// runs first and takes those blocks out of the demotion's hands: each becomes a
|
||||
// flat `uint` array whose index arithmetic carries the std140/std430 offsets
|
||||
// glslang computed WITH the doubles in place, so the layout survives byte for byte
|
||||
// and only the VALUES narrow. Gated on a block actually holding a 64-bit float, so
|
||||
// every other module pays one types_values() walk and nothing else, and it declines
|
||||
// (leaving the block for the demotion to handle the old way) on any shape it cannot
|
||||
// re-address exactly. See FlattenFloat64StorageBlockPass.h.
|
||||
optimizer.RegisterPass(
|
||||
FlattenFloat64StorageBlockPass::CreateFlattenFloat64StorageBlockPass());
|
||||
optimizer.RegisterPass(DemoteFloat64Pass::CreateDemoteFloat64Pass());
|
||||
|
||||
return RunOptimizerChecked("SanitizeAndOptimizeBinary", optimizer, inputBinary,
|
||||
|
||||
@@ -50,12 +50,10 @@ namespace MobileGL {
|
||||
// for the same reason - writes exactly where the demoted shader reads. Blocks with no
|
||||
// 64-bit member anywhere are never touched.
|
||||
//
|
||||
// THE MEASURED COST, so the next wave does not re-diagnose it. Four GL 4.3 conformance
|
||||
// cases fail on BOTH backends and on both an Adreno 830 and a Mali G925 - i.e. on every
|
||||
// device, because no device has shaderFloat64 and the demotion therefore always runs:
|
||||
// WHAT RE-DERIVING STILL COSTS, so the next wave does not re-diagnose it. Two GL 4.3
|
||||
// conformance cases fail on BOTH backends and on every device, because no device has
|
||||
// shaderFloat64 and the demotion therefore always runs:
|
||||
//
|
||||
// KHR-GL43.shader_storage_buffer_object.basic-stdLayout-case3-cs
|
||||
// KHR-GL43.shader_storage_buffer_object.basic-stdLayout-case3-vs
|
||||
// KHR-GL43.compute_shader.fp64-case1
|
||||
// KHR-GL43.compute_shader.fp64-case3
|
||||
//
|
||||
@@ -63,44 +61,28 @@ namespace MobileGL {
|
||||
// against it: it is blocked on GLSL subroutines ("FP64 support - subroutines"), which
|
||||
// glslang deletes when targeting SPIR-V, and is out of scope by standing instruction.
|
||||
//
|
||||
// The other three fail in the two ways this comment predicts and in no other.
|
||||
// stdLayout-case3 copies a block byte for byte: the output matches the input for
|
||||
// bytes [0, 76) and is zero from there on, which is exactly the block's size once
|
||||
// every double became a float and the layout repacked tightly. Re-derived byte-exactly
|
||||
// in 2026-08: the block is `int data0; float data1[5]; mat3x2 data2; double data3;
|
||||
// double data4[2]; int data5; dvec3 data6`, and demoting every double to float and
|
||||
// repacking std430 gives data0@0, data1@4..23, data2@24..47, data3@48, data4@52..59,
|
||||
// data5@60, data6@64..75 - 76 bytes. EVERY mismatching byte the QPA reports is >= 76
|
||||
// and every expected-non-zero byte below 76 matched, on both the std140 output and the
|
||||
// std430 one.
|
||||
//
|
||||
// ONE TRAP FOR THE NEXT READER, because it reads as evidence AGAINST demotion and is
|
||||
// not: in the std430 output the doubles below the boundary appear to have round-tripped
|
||||
// BIT-EXACTLY, which looks like fp64 surviving. It is an artifact. The shader reads and
|
||||
// writes through the SAME demoted offset, so those four bytes are copied verbatim
|
||||
// whatever they are interpreted as - the copy proves nothing about the width.
|
||||
//
|
||||
// fp64-case1 reports ceil(2.2) as 2: the uniform's double 2.0 is 0x4000000000000000,
|
||||
// the demoted read takes its low 32 bits (0.0), ceil(0.0 + 0.2) = 1.0f = 0x3F800000
|
||||
// lands in the low half of the 8-byte output slot and the whole thing prints as 2.
|
||||
// Index 0 of the same case PASSES by accident, for the same reason - writing 0.0f into
|
||||
// the low half of 1.0 leaves it unchanged - so a partial pass here is not progress.
|
||||
// Fixing it means carrying a double in the DEFAULT UNIFORM block without re-deriving
|
||||
// its layout, and that block's routing is built by reflecting the module this pass
|
||||
// produces, so the representation change ripples into every glUniform*d. Deliberately
|
||||
// not attempted. compute_shader.fp64-case2 passes today and any attempt has to keep it
|
||||
// green.
|
||||
//
|
||||
// Both backends produce a CHARACTER-FOR-CHARACTER identical QPA byte list, which is
|
||||
// the cheapest available proof that the defect is in this shared pass and in neither
|
||||
// backend. A future wave that wants to re-open this should start by re-checking that
|
||||
// identity rather than by re-deriving the layout.
|
||||
//
|
||||
// Fixing them means NOT demoting a double that lives in a buffer block, and carrying
|
||||
// it as a uvec2 word pair instead - preserving the application's byte layout exactly,
|
||||
// unpacking to fp32 for arithmetic and repacking on store. That is a large pass with
|
||||
// the same dmat problem the paragraph above describes (a uvec2 representation cannot
|
||||
// express a matrix stride either, so it would have to decline dmat types), and the
|
||||
// default-uniform routing above reflects the demoted module, so a representation
|
||||
// change there ripples into every glUniform*d. THREE actionable cases of 16085 (the
|
||||
// fourth, fp64-case3, is subroutine-blocked and unreachable from here); deliberately
|
||||
// not attempted, and re-confirmed as not worth attempting in the 2026-08 wave.
|
||||
// compute_shader.fp64-case2 passes today and any attempt has to keep it green.
|
||||
// SHADER STORAGE BLOCKS ARE NO LONGER IN THAT LIST, and the two cases that used to be
|
||||
// (shader_storage_buffer_object.basic-stdLayout-case3-cs and -vs, which copy a block
|
||||
// byte for byte and used to come back zero from the first double's slot onwards) pass
|
||||
// on both backends. FlattenFloat64StorageBlockPass runs immediately before this one
|
||||
// and takes every storage block holding a 64-bit float out of its hands, rewriting the
|
||||
// block into a flat `uint` array whose index arithmetic carries the offsets glslang
|
||||
// computed WITH the doubles in place. A flat array has no layout for SPIRV-Cross to
|
||||
// re-derive, which is what makes it expressible where a padded struct is not, and an
|
||||
// offset in an address computation has none of the dmat trouble the paragraph above
|
||||
// describes. See that pass's header. Everything below still describes what happens to
|
||||
// every OTHER block, and to the doubles in the function bodies of all of them.
|
||||
//
|
||||
// Declines (leaves the module byte-identical, so the caller's existing "this module
|
||||
// still declares Float64" failure path reports it) when the module contains an
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,91 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenFloat64StorageBlockPass.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include "source/opt/pass.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
// Rewrites a SHADER STORAGE BLOCK that contains a 64-bit float into a flat
|
||||
// `uint` word array, and turns every access to it into address arithmetic over
|
||||
// that array. The application's byte layout survives exactly; the VALUES are
|
||||
// still narrowed to 32-bit floats, because that is all any target here has.
|
||||
//
|
||||
// WHY THIS EXISTS. DemoteFloat64Pass rewrites `double` to `float` in place and
|
||||
// lets SPIRV-Cross re-derive the block's packing from the declared types, because
|
||||
// GLSL ES has no member `layout(offset=)` and SPIRV-Cross refuses any block whose
|
||||
// stated offsets it cannot express as std140 or std430. That re-derivation moves
|
||||
// every member past the first double: the block a shader reads and writes stops
|
||||
// being the block the application filled. Byte-for-byte, on the shape
|
||||
// KHR-GL43.shader_storage_buffer_object.basic-stdLayout-case3 uses, the output
|
||||
// matched the input up to the first double's slot and was zero from there on -
|
||||
// the demoted block is simply shorter than the one that was bound.
|
||||
//
|
||||
// A flat `uint[]` has no layout to re-derive: one member, offset 0, ArrayStride 4,
|
||||
// which IS std430, so SPIRV-Cross prints it unconditionally and the driver lays it
|
||||
// out the only way it can. Every member's real byte offset - the std140 or std430
|
||||
// one glslang computed WITH the doubles in place - then lives in the index
|
||||
// arithmetic this pass emits, not in the declaration. The two ways the earlier
|
||||
// attempt at this was blocked both disappear with it:
|
||||
//
|
||||
// * dmat: a `uvec2`-per-double representation cannot express a MatrixStride, so
|
||||
// it would have had to decline matrices of doubles. Here a stride is a number
|
||||
// in an address computation and nothing else, so dmat needs no special case.
|
||||
// * the default-uniform block: its routing is built by reflecting the DEMOTED
|
||||
// module (ProgramSpirvTask::BuildGlobalUboRouting), so changing how a double
|
||||
// is carried there would ripple into every glUniform*d. This pass touches
|
||||
// StorageBuffer blocks only and never that one.
|
||||
//
|
||||
// WHAT GL SEES IS UNCHANGED, and becomes CORRECT rather than merely unchanged:
|
||||
// glGetProgramResourceiv answers from glslang's reflection of the pre-demotion
|
||||
// module (ProgramInterface.cpp reads TObjectReflection::offset), i.e. the true
|
||||
// fp64 offsets. Before this pass those offsets described a layout no shader used;
|
||||
// now they describe the one it does.
|
||||
//
|
||||
// PRECISION, stated plainly. A double still becomes a float: the load narrows the
|
||||
// stored binary64 to binary32 and the store widens it back, so a value that does
|
||||
// not survive a round trip through 32 bits does not survive this either. The
|
||||
// narrowing truncates the discarded mantissa bits rather than rounding to nearest,
|
||||
// and flushes what binary32 can only hold as a subnormal to a signed zero; NaN
|
||||
// stays NaN and an out-of-range magnitude becomes an infinity. That is the same
|
||||
// fp32 promise DemoteFloat64Pass already makes - what changes is only that the
|
||||
// BYTES around the value stay where the application put them.
|
||||
//
|
||||
// DECLINES, leaving the block exactly as it was for DemoteFloat64Pass to handle the
|
||||
// old way, whenever it meets something it cannot rewrite exactly:
|
||||
// - a block whose variable is used as anything but an access-chain base (loaded
|
||||
// whole, handed to a function, asked its OpArrayLength);
|
||||
// - an access chain that is not rooted at the variable, or whose result feeds
|
||||
// anything but a plain OpLoad / OpStore (an atomic, OpCopyMemory, a further
|
||||
// chain);
|
||||
// - a non-constant index into a struct, a runtime array anywhere in the block, a
|
||||
// RowMajor matrix (its columns are not contiguous, so a whole-column access is
|
||||
// not one range), a member width other than 32 or 64 bits, or an offset or
|
||||
// stride that is not a multiple of 4;
|
||||
// - a load or store whose type decomposes into more scalars than the cap below,
|
||||
// so legalizing a block can never explode the module.
|
||||
//
|
||||
// ORDERING: must run BEFORE DemoteFloat64Pass, which is what turns the doubles this
|
||||
// pass leaves in the function body into floats - the OpFConvert pairs emitted here
|
||||
// are width-preserving by then and collapse to their operands. It emits only 32-bit
|
||||
// OpBitcasts, so it never trips that pass's "bitcast across the 64-bit boundary"
|
||||
// decline.
|
||||
class FlattenFloat64StorageBlockPass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "mobilegl-flatten-float64-storage-block"; }
|
||||
Status Process() override;
|
||||
|
||||
static spvtools::Optimizer::PassToken CreateFlattenFloat64StorageBlockPass();
|
||||
};
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
@@ -442,6 +442,60 @@ namespace MobileGL {
|
||||
SPVC_CHK_RETURN
|
||||
}
|
||||
|
||||
spvc_result SpvcSession::DropDefaultFragmentOutputColorIndex() {
|
||||
if (!(usage & SessionUsageBit::Transpile)) return SPVC_ERROR_INVALID_ARGUMENT;
|
||||
|
||||
SPVC_CHK_INIT
|
||||
const spvc_reflected_resource* list = nullptr;
|
||||
size_t count = 0;
|
||||
SPVC_CHK_RESULT(spvc_resources_get_resource_list_for_type(
|
||||
resources, SPVC_RESOURCE_TYPE_STAGE_OUTPUT, &list, &count));
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
const spvc_reflected_resource& resource = list[i];
|
||||
if (!spvc_compiler_has_decoration(compiler, resource.id, SpvDecorationIndex)) continue;
|
||||
if (spvc_compiler_get_decoration(compiler, resource.id, SpvDecorationIndex) != 0u) continue;
|
||||
spvc_compiler_unset_decoration(compiler, resource.id, SpvDecorationIndex);
|
||||
}
|
||||
SPVC_CHK_RETURN
|
||||
}
|
||||
|
||||
spvc_result SpvcSession::RelaxReadWriteExclusiveStorageBuffers() {
|
||||
if (!(usage & SessionUsageBit::Transpile)) return SPVC_ERROR_INVALID_ARGUMENT;
|
||||
|
||||
SPVC_CHK_INIT
|
||||
const spvc_reflected_resource* list = nullptr;
|
||||
size_t count = 0;
|
||||
SPVC_CHK_RESULT(spvc_resources_get_resource_list_for_type(
|
||||
resources, SPVC_RESOURCE_TYPE_STORAGE_BUFFER, &list, &count));
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
const spvc_reflected_resource& resource = list[i];
|
||||
// The variable itself, for a block the application qualified as a whole.
|
||||
if (spvc_compiler_has_decoration(compiler, resource.id, SpvDecorationNonReadable) &&
|
||||
spvc_compiler_has_decoration(compiler, resource.id, SpvDecorationNonWritable)) {
|
||||
spvc_compiler_unset_decoration(compiler, resource.id, SpvDecorationNonReadable);
|
||||
spvc_compiler_unset_decoration(compiler, resource.id, SpvDecorationNonWritable);
|
||||
}
|
||||
// ...and each member, which is where the qualifiers usually sit and where
|
||||
// SPIRV-Cross reads them from before hoisting the ones every member shares.
|
||||
const spvc_type blockType = spvc_compiler_get_type_handle(compiler, resource.base_type_id);
|
||||
if (blockType == nullptr) continue;
|
||||
const unsigned memberCount = spvc_type_get_num_member_types(blockType);
|
||||
for (unsigned member = 0; member < memberCount; ++member) {
|
||||
if (!spvc_compiler_has_member_decoration(compiler, resource.base_type_id, member,
|
||||
SpvDecorationNonReadable) ||
|
||||
!spvc_compiler_has_member_decoration(compiler, resource.base_type_id, member,
|
||||
SpvDecorationNonWritable)) {
|
||||
continue;
|
||||
}
|
||||
spvc_compiler_unset_member_decoration(compiler, resource.base_type_id, member,
|
||||
SpvDecorationNonReadable);
|
||||
spvc_compiler_unset_member_decoration(compiler, resource.base_type_id, member,
|
||||
SpvDecorationNonWritable);
|
||||
}
|
||||
}
|
||||
SPVC_CHK_RETURN
|
||||
}
|
||||
|
||||
spvc_result SpvcSession::Compile(const char** result) {
|
||||
if (!(usage & SessionUsageBit::Transpile)) return SPVC_ERROR_INVALID_ARGUMENT;
|
||||
SPVC_CHK_INIT
|
||||
|
||||
@@ -120,6 +120,46 @@ namespace MobileGL {
|
||||
// `outGlBindings` is appended to, so one vector can collect a whole program's
|
||||
// stages; it may repeat a binding declared by several of them.
|
||||
spvc_result SetAtomicCounterBlockBindings(Int topBinding, Vector<Int>& outGlBindings);
|
||||
// Drops the Index decoration from every fragment output that carries the DEFAULT
|
||||
// colour index 0, so the emitted ESSL does not print `index = 0`.
|
||||
//
|
||||
// Index 0 is what every single-source fragment output already is, in GL and in
|
||||
// ESSL alike, and SPIR-V carries the decoration only because the application
|
||||
// spelled the qualifier out - `layout(location = 0, index = 0) out vec4 c;` is
|
||||
// legal desktop GLSL and says nothing. Printing it back into ESSL is NOT
|
||||
// harmless: GLSL ES has no `index` layout qualifier in core, so the driver
|
||||
// answers "index layout qualifier requires EXT_blend_func_extended" and refuses
|
||||
// the stage. The program then links nothing and every draw with it renders
|
||||
// NOTHING - verified on Mesa 26.1.4 llvmpipe with no MobileGL in the process,
|
||||
// and it is why KHR-GL43.shader_atomic_counters.basic-program-query read back a
|
||||
// black render target.
|
||||
//
|
||||
// A NON-zero index is left exactly as it is: that one really does select the
|
||||
// second dual-source input and cannot be expressed without the extension, so it
|
||||
// must keep reaching the driver (the frontend's own glBindFragDataLocationIndexed
|
||||
// path already emits only non-zero indices for the same reason).
|
||||
spvc_result DropDefaultFragmentOutputColorIndex();
|
||||
// Drops `readonly` and `writeonly` from every shader storage block - and every
|
||||
// block member - that carries BOTH of them.
|
||||
//
|
||||
// GL 4.6 core 4.10 lets a buffer variable be declared readonly AND writeonly at
|
||||
// once: it then cannot be read or written at all, and the only thing left that
|
||||
// it can be used for is `.length()`. The pair is therefore inert by
|
||||
// construction - the frontend has already rejected any access to it - so
|
||||
// dropping it cannot change what the shader does.
|
||||
//
|
||||
// Emitting it does change whether the shader EXISTS. SPIRV-Cross hoists the
|
||||
// qualifiers every member shares onto the block, and Mesa's ES compiler rejects
|
||||
// that spelling outright ("Interface block sets both readonly and writeonly",
|
||||
// verified on Mesa 26.1.4 llvmpipe with no MobileGL in the process, against the
|
||||
// exact source this transpiler emitted). The stage then never compiles, the
|
||||
// program links without it, and every dispatch or draw is a silent no-op -
|
||||
// which is how KHR-GL43.shader_storage_buffer_object.basic-readonly-writeonly
|
||||
// read back 0 instead of the array length.
|
||||
//
|
||||
// A block carrying only ONE of the two is left exactly as it is: those really do
|
||||
// constrain the accesses the shader makes, and the driver is entitled to know.
|
||||
spvc_result RelaxReadWriteExclusiveStorageBuffers();
|
||||
spvc_result Compile(const char** result);
|
||||
const SpvcMetadata& GetMetadata() const;
|
||||
const char* GetLastErrorString() const;
|
||||
|
||||
Reference in New Issue
Block a user