[Merge] (DirectGLES, ShaderTranspiler): land GL43 wave4 with the interface-block rename inside the L2 boundary

This commit is contained in:
2026-08-20 21:13:55 -04:00
42 changed files with 3280 additions and 96 deletions
+1
View File
@@ -283,6 +283,7 @@ set(SOURCE_FILES
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerViewportIndexPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PackDoubleVertexInputsPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenXfbInterfaceBlocksPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/UniquifyIoBlockNamesPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/SplitArrayVertexInputsPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
+24
View File
@@ -372,8 +372,32 @@ namespace MobileGL {
Int MaxComputeImageUniforms = 8;
Int MaxDrawBuffers = 8;
Int MaxColorAttachments = 8;
// GL_MAX_CLIP_DISTANCES. Zero is a legal answer here, not a placeholder, and a
// backend that cannot host a clip distance MUST report it: advertising eight the
// backend will refuse does not make gl_ClipDistance work, it only moves the failure
// from an honest "unsupported" at query time to a backend shader-compile error the
// frontend never surfaces, after which every draw with that program silently renders
// nothing. DirectGLES fills it from GL_EXT_clip_cull_distance, DirectVulkan from the
// shaderClipDistance device feature. The DEFAULT stays at the GL 4.3 core minimum
// because it describes the no-backend case (standalone shader compiles, unit tests),
// where there is no device to be honest about and BuildTBuiltInResource still has to
// hand glslang a workable gl_MaxClipDistances.
Int MaxClipDistances = 8;
Int MaxViewports = 16;
// GL_LAYER_PROVOKING_VERTEX / GL_VIEWPORT_INDEX_PROVOKING_VERTEX: which vertex of a
// primitive supplies gl_Layer and gl_ViewportIndex. GL 4.6 table 23.65 makes
// GL_UNDEFINED_VERTEX a legal answer for both, and it is the honest default - naming
// a convention is a statement about behaviour, so a backend that does not pin one
// must not claim it does. DirectGLES fills the layer one from the ES 3.2 query and
// the viewport one from GL_OES_viewport_array, and leaves UNDEFINED where the
// capability is absent: without the viewport array extension only viewport 0 is ever
// rasterized, so no convention selects anything. DirectVulkan keeps UNDEFINED for
// both - which vertex provokes is decided per pipeline by
// VulkanRenderer::SelectProvokingVertexMode out of VK_EXT_provoking_vertex,
// provokingVertexModePerPipeline and the topology, so no single convention is true
// of the backend.
GLenum LayerProvokingVertex = GL_UNDEFINED_VERTEX;
GLenum ViewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
Int MaxViewportWidth = 16384;
Int MaxViewportHeight = 16384;
Float ViewportBoundsRangeMin = 0.0f;
@@ -213,7 +213,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (options & PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget) {
reasons.push_back("no colour-renderable three-channel format on OpenGL ES");
}
if (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) {
// A format is either 8- or 16-bit signed normalized, so at most one of the two ever
// survives GetApplicablePixelFormatNormalizeOptions and the reason is not duplicated.
if ((options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
reasons.push_back("EXT_render_snorm not supported");
}
@@ -1336,6 +1339,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
m_dynamicParameters.MaxColorAttachments = m_GLESCapabilities.MaxColorAttachments;
m_dynamicParameters.MaxClipDistances = m_GLESCapabilities.MaxClipDistances;
m_dynamicParameters.MaxViewports = m_GLESCapabilities.MaxViewports;
// Whatever the driver said about which vertex supplies gl_Layer, and GL_UNDEFINED_VERTEX
// for gl_ViewportIndex on every driver without GL_OES_viewport_array - which is both test
// devices. That is not a shortfall being hidden: without the extension only viewport 0 is
// ever rasterized, so no vertex "selects" a viewport index and naming a convention would
// describe behaviour this backend does not implement.
m_dynamicParameters.LayerProvokingVertex = m_GLESCapabilities.LayerProvokingVertex;
m_dynamicParameters.ViewportIndexProvokingVertex = m_GLESCapabilities.ViewportIndexProvokingVertex;
m_dynamicParameters.MaxViewportWidth = m_GLESCapabilities.MaxViewportWidth;
m_dynamicParameters.MaxViewportHeight = m_GLESCapabilities.MaxViewportHeight;
m_dynamicParameters.ViewportBoundsRangeMin = m_GLESCapabilities.ViewportBoundsRangeMin;
+188 -9
View File
@@ -7684,6 +7684,149 @@ namespace MobileGL::MG_Backend::DirectGLES {
return true;
}
// ---- Bit-exact readback of a 32-bit packed colour level ---------------------------------------
//
// glGetTexImage of a packed format read with its OWN client type owes the application the words
// the image HOLDS, and neither of the two routes above can promise that once anything other than
// a glTexImage has written the level:
//
// * the colour-attachment route reads GL_RGBA/GL_FLOAT and re-encodes, which canonicalizes an
// RGB9_E5 shared exponent (0xf8fc0000 -> 0xe7e00000, same value, different bits) and
// collapses an R11F_G11F_B10F NaN to the canonical payload 1
// (MG_Util::EncodeFloatToUnsignedSmallFloat) - and a copy-image from RGB9_E5 lands exactly
// such a NaN in the 10-bit blue field every time, because the source's shared-exponent
// field is all ones;
// * the CPU shadow only ever holds what was UPLOADED, so for a level glCopyImageSubData wrote
// it answers with the PRE-COPY contents. MirrorCopyImageIntoDestinationShadow patches that
// up for the shapes it can address texel-exactly and declines for the rest - a renderbuffer
// source (which has no shadow to mirror from at all), a cube or 1D-array endpoint, a
// self-copy - and the decline is silent, so the stale words are served as truth.
//
// glCopyImageSubData is a raw texel-block move and EXT_copy_image puts every 32-bit colour
// format in one compatibility class, so copying the level into a scratch GL_R32UI image and
// reading THAT back as unsigned integers hands over the stored words themselves, whoever wrote
// them. This is what lets the shadow stop being the authority for these formats: it is tried
// first, and every step reports rather than guesses, so a driver that turns any of it down
// simply leaves the old shadow/attachment fallbacks to run.
static GLuint g_packedWordScratchTextureId = 0;
static GLsizei g_packedWordScratchWidth = 0;
static GLsizei g_packedWordScratchHeight = 0;
// Grow-only, so a readback sweep over a mip chain allocates once. Zero when the driver refused
// the storage, which is a decline and not an error.
static GLuint EnsurePackedWordScratchTexture(GLsizei width, GLsizei height) {
if (g_packedWordScratchTextureId != 0 && g_packedWordScratchWidth >= width &&
g_packedWordScratchHeight >= height) {
return g_packedWordScratchTextureId;
}
const GLsizei newWidth = std::max(width, g_packedWordScratchWidth);
const GLsizei newHeight = std::max(height, g_packedWordScratchHeight);
if (g_packedWordScratchTextureId != 0) {
// A scratch FBO may still name the old id, and the driver is free to hand the same
// number back for the replacement - which would false-skip the re-attach.
ScratchFBOImpl::NoteTextureIdDeleted(g_packedWordScratchTextureId);
g_GLESFuncs.glDeleteTextures(1, &g_packedWordScratchTextureId);
g_packedWordScratchTextureId = 0;
g_packedWordScratchWidth = 0;
g_packedWordScratchHeight = 0;
}
GLuint texture = 0;
g_GLESFuncs.glGenTextures(1, &texture);
if (texture == 0) return 0;
ClearGLErrors();
TextureImpl::ActivateTextureUnit(TextureImpl::TempTextureUnit);
g_GLESFuncs.glBindTexture(GL_TEXTURE_2D, texture);
// Immutable single-level storage: glCopyImageSubData wants a complete image, and
// glTexStorage clamps TEXTURE_MAX_LEVEL, which is what makes a one-level texture complete
// under the default mipmapping filter.
g_GLESFuncs.glTexStorage2D(GL_TEXTURE_2D, 1, GL_R32UI, newWidth, newHeight);
const GLenum storageError = g_GLESFuncs.glGetError();
// Re-bind whatever the binding cache says lives on the temp unit, so the cache stays
// truthful without a driver query (same discipline as CopyR32FTexture2D).
auto* cachedBound = TextureImpl::g_boundTexturesCache[TextureImpl::TempTextureUnit]
[static_cast<SizeT>(TextureTarget::Texture2D)];
g_GLESFuncs.glBindTexture(GL_TEXTURE_2D, cachedBound ? cachedBound->GetBackendTextureId() : 0);
if (storageError != GL_NO_ERROR) {
g_GLESFuncs.glDeleteTextures(1, &texture);
MGLOG_D("GetTexImage: no %dx%d GL_R32UI scratch image (%s); the verbatim word readback is unavailable",
newWidth, newHeight, MG_Util::ConvertGLEnumToString(storageError).c_str());
return 0;
}
g_packedWordScratchTextureId = texture;
g_packedWordScratchWidth = newWidth;
g_packedWordScratchHeight = newHeight;
return texture;
}
static void ReleasePackedWordScratchTexture() {
// The ES context (and the name with it) is gone; deleting here would target a recycled
// name in the successor context.
g_packedWordScratchTextureId = 0;
g_packedWordScratchWidth = 0;
g_packedWordScratchHeight = 0;
}
// One slice of `backendTarget`'s level, as width*height stored 32-bit words in `outWords`.
static Bool ReadPackedLevelWordsViaScratch(GLuint texture, GLenum backendTarget, GLint level, GLint slice,
GLsizei width, GLsizei height, Uint32* outWords) {
if (texture == 0 || outWords == nullptr || width <= 0 || height <= 0 || level < 0 || slice < 0) return false;
if (!g_GLESFuncs.glCopyImageSubData) return false;
// Horizontal bands, so neither the scratch image nor the staging buffer scales with the
// level. The scratch is grow-only on purpose - a sweep down a mip chain must not
// reallocate per level - which without a band cap would leave a 4096x4096 readback's
// 64 MiB image parked for the rest of the process. The cap is 1 MiB of GL_R32UI, with
// 4 MiB of staging behind it because the read lands four words per texel.
constexpr SizeT kMaxScratchTexels = SizeT{1} << 18;
const GLsizei bandRows = std::max<GLsizei>(
1, static_cast<GLsizei>(std::min<SizeT>(kMaxScratchTexels / static_cast<SizeT>(width),
static_cast<SizeT>(height))));
const GLuint scratch = EnsurePackedWordScratchTexture(width, bandRows);
if (scratch == 0) return false;
ScopedFramebufferBinding readBinding(/*saveRead=*/true, /*saveDraw=*/false);
auto& scratchFB = ScratchFBOImpl::BlitReadFramebuffer();
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, ScratchFBOImpl::EnsureId(scratchFB));
ScratchFBOImpl::EnsureColorAttachment2D(scratchFB, GL_READ_FRAMEBUFFER, scratch, GL_TEXTURE_2D, 0);
ScratchFBOImpl::EnsureReadBuffer(scratchFB, GL_COLOR_ATTACHMENT0);
if (g_GLESFuncs.glCheckFramebufferStatus(GL_READ_FRAMEBUFFER) != GL_FRAMEBUFFER_COMPLETE) {
MGLOG_D("GetTexImage: the GL_R32UI scratch attachment is incomplete; falling back");
return false;
}
// GL_RGBA_INTEGER/GL_UNSIGNED_INT is the one combination ES guarantees for an integer
// colour buffer, so the read lands four words per texel and the red one is compacted out
// here. The PACK scope is the tight default rather than the application's, so a row comes
// back packed at exactly `width * 4` words. One glGetError covers the whole loop: it
// accumulates, and a failure anywhere means the caller falls back rather than trusting a
// partial result.
const SizeT wordsPerRow = static_cast<SizeT>(width) * 4;
Vector<Uint32> staging(static_cast<SizeT>(bandRows) * wordsPerRow);
ScopedPixelPackBuffer packBuffer(0);
ScopedPackState packState(PixelStoreImpl::PackState{4, 0, 0, 0});
ClearGLErrors();
for (GLsizei y = 0; y < height; y += bandRows) {
const GLsizei rows = std::min(bandRows, height - y);
g_GLESFuncs.glCopyImageSubData(texture, backendTarget, level, 0, y, slice, scratch, GL_TEXTURE_2D, 0, 0,
0, 0, width, rows, 1);
g_GLESFuncs.glReadPixels(0, 0, width, rows, GL_RGBA_INTEGER, GL_UNSIGNED_INT, staging.data());
for (GLsizei row = 0; row < rows; ++row) {
const Uint32* srcRow = staging.data() + static_cast<SizeT>(row) * wordsPerRow;
Uint32* dstRow = outWords + static_cast<SizeT>(y + row) * static_cast<SizeT>(width);
for (GLsizei x = 0; x < width; ++x) dstRow[x] = srcRow[static_cast<SizeT>(x) * 4];
}
}
const GLenum error = g_GLESFuncs.glGetError();
if (error != GL_NO_ERROR) {
MGLOG_D("GetTexImage: the GL_R32UI word readback of %s was refused (%s); falling back",
MG_Util::ConvertGLEnumToString(backendTarget).c_str(),
MG_Util::ConvertGLEnumToString(error).c_str());
return false;
}
return true;
}
static Bool IsLegacyNativeReadPixelsFormat(GLenum format) {
return format == GL_RGBA || format == GL_RGBA_INTEGER || format == GL_RED || format == GL_RED_INTEGER ||
format == GL_DEPTH_COMPONENT || format == GL_STENCIL_INDEX || format == GL_DEPTH_STENCIL;
@@ -8058,17 +8201,52 @@ namespace MobileGL::MG_Backend::DirectGLES {
// value 8064, different words), and the conformance suite compares the words
// ("CopyImageSubData modified contents of source image"). The scratch FBO does NOT
// decide this for us: Adreno reports an RGB9_E5 colour attachment complete, so the
// shadow branch further down was unreachable. Serve the verbatim-word pairs from the
// shadow first and keep the GPU attempts as the fallback for a level the shadow never
// received. Every other format still prefers the GPU, so a rendered-into texture is
// unaffected; RGB9_E5 is not colour-renderable, so its shadow stays authoritative -
// and the one path that GPU-writes it, CopyImageSubData, mirrors itself into the
// shadow for exactly this reason.
// shadow branch further down was unreachable. Every other format still prefers the
// GPU, so a rendered-into texture is unaffected.
const Bool rawPackedWordRead = MG_Util::PixelStoreProcessor::IsRawPackedPixelTransfer(
textureObject->GetFormat(), MG_Util::ConvertGLEnumToTextureInputFormat(format),
MG_Util::ConvertGLEnumToTexturePixelDataType(type));
// ...and the GPU CAN answer with the stored words after all, for any 32-bit packed
// format and whoever wrote the level, by going through a scratch GL_R32UI image (see
// ReadPackedLevelWordsViaScratch). Preferred over both routes below because it is the
// only one that is right for a level glCopyImageSubData wrote: the shadow may never
// have seen that write, and re-encoding the attachment cannot reproduce an RGB9_E5
// shared exponent or an R11F_G11F_B10F NaN payload. A multisample image is excluded
// because copy-image requires matching sample counts.
if (rawPackedWordRead && textureObject->GetSamples() == 0) {
// Copy-image addresses a cube map as ONE image with the face on z, where
// glGetTexImage names the face in its target.
const auto readUploadTarget = MG_Util::ConvertGLEnumToTextureUploadTarget(target);
const GLint copyBaseSlice =
(readUploadTarget >= TextureUploadTarget::CubeMapPositiveX &&
readUploadTarget <= TextureUploadTarget::CubeMapNegativeZ)
? static_cast<GLint>(readUploadTarget) -
static_cast<GLint>(TextureUploadTarget::CubeMapPositiveX)
: 0;
const GLenum copyTarget =
TextureImpl::ConvertTextureTargetToBackendGLEnum(textureObject->GetTarget());
const SizeT sliceWords = static_cast<SizeT>(size.x()) * static_cast<SizeT>(size.y());
Vector<Uint32> words(sliceWords * static_cast<SizeT>(sliceCount));
Bool allSlicesRead = true;
for (GLsizei slice = 0; slice < sliceCount && allSlicesRead; ++slice) {
allSlicesRead = ReadPackedLevelWordsViaScratch(backendTexId, copyTarget, level,
copyBaseSlice + slice, size.x(), size.y(),
words.data() + sliceWords * static_cast<SizeT>(slice));
}
if (allSlicesRead &&
ReadbackImpl::StorePackedWordsToClient(reinterpret_cast<const Uint8*>(words.data()), size.x(),
size.y(), sliceCount, type, pixels,
applyPackImageParams)) {
MGLOG_D("GetTexImage: finished %d slice(s) via the bit-exact GL_R32UI word readback", sliceCount);
return;
}
}
// The last resort for the one format the attachment route can never answer for: the
// shadow is only right while nothing but a glTexImage has written the level, which is
// why CopyImageSubData mirrors itself into it where it can.
const Bool verbatimPackedShadowRead =
MG_Util::PixelStoreProcessor::HasRedundantPackedEncoding(textureObject->GetFormat()) &&
MG_Util::PixelStoreProcessor::IsRawPackedPixelTransfer(
textureObject->GetFormat(), MG_Util::ConvertGLEnumToTextureInputFormat(format),
MG_Util::ConvertGLEnumToTexturePixelDataType(type));
rawPackedWordRead;
if (verbatimPackedShadowRead &&
GetTexImageViaShadowConversion(textureMipmapObject,
MG_Util::ConvertGLEnumToTextureUploadTarget(target), level, size.x(),
@@ -9262,6 +9440,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
XfbImpl::OnBackendContextDestroyed();
MultiDrawImpl::OnBackendContextDestroyed();
ScratchFBOImpl::OnBackendContextDestroyed();
ReleasePackedWordScratchTexture();
FramebufferImpl::InvalidateFramebufferBindingCache();
VertexArrayImpl::InvalidateVAOBindingCache();
PixelStoreImpl::InvalidatePackStateCache();
+168
View File
@@ -30,6 +30,7 @@
#include <algorithm>
#include <cctype>
#include <cstdlib>
#include <map>
#include <mutex>
#include <cstring>
#include <regex>
@@ -4725,6 +4726,34 @@ namespace MobileGL::MG_Backend::DirectGLES {
return signature + entry;
}
// Pipeline position of a shader stage. Names the PRODUCER of an inter-stage
// interface block: a block one stage consumes was written by the stage before it.
// ShaderStage is declared in pipeline order, so the enum value IS the position;
// compute has no inter-stage interface at all and is reported as -1.
Int InterStagePipelineIndex(ShaderStage stage) {
switch (stage) {
case ShaderStage::Vertex:
case ShaderStage::TessControl:
case ShaderStage::TessEval:
case ShaderStage::Geometry:
case ShaderStage::Fragment:
return static_cast<Int>(stage);
default:
return -1;
}
}
// Whether a stage can declare interface blocks in BOTH directions at once, i.e.
// whether one block name can name two different blocks inside it. A vertex INPUT
// and a fragment OUTPUT cannot be blocks and compute has neither, so only these
// three can. This is what keeps the module probe off every program without
// tessellation or geometry - which is every program Minecraft and its shader packs
// build.
Bool CanDeclareBlocksInBothDirections(ShaderStage stage) {
return stage == ShaderStage::TessControl || stage == ShaderStage::TessEval ||
stage == ShaderStage::Geometry;
}
// Reflection names an array uniform after its first element ("g_image[0]") at every
// location it spans; SPIR-V names the variable once, without the subscript. This is
// the name both sides agree on.
@@ -4929,6 +4958,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
const Vector<unsigned int>& spirvCode, const GLenum glShaderType,
const std::set<String>& xfbCaptureBlockNames, const ImageFormatBakeInputs& imageFormatBake,
const UnorderedMap<String, Int>& storageBlockBindingOverrides,
const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames,
const Int atomicCounterEsslBindingTop, const Bool enableSpirvValidation, String& outSource,
std::set<String>& outFlattenedXfbBlockNames, Vector<Int>& outAtomicCounterGlBindings,
String& outError) const {
@@ -5051,6 +5082,41 @@ namespace MobileGL::MG_Backend::DirectGLES {
}
}
// The producer-keyed interface-block rename, planned program-wide by the caller and
// applied to this stage: the blocks it CONSUMES are spelled after the previous stage
// present in the program and the ones it PRODUCES after itself, so a tessellation
// evaluation stage's two TCSOutputBlocks stop being one name and every other stage
// still agrees with it. See UniquifyIoBlockNamesPass for why Mali needs it.
//
// INSIDE THE MEMOIZED SEGMENT, and at exactly the position it was written in - after
// the XFB flatten, before the UBO precision strip. Both halves of that matter:
// * INSIDE, because it rewrites the MODULE and the emitted ESSL carries the result.
// Left outside, a second program sharing this stage's key would be served ESSL
// with the blocks un-renamed and the repair would silently stop working - the
// same trap SetAtomicCounterBlockBindings sets one screen down.
// * AT THIS POSITION, because moving a SPIR-V pass in a chain is a behavioural
// change, and this one arrived device-verified on Mali. Hoisting it above the
// cache probe instead would have needed no key material at all (the rename would
// already be in the module bytes the key hashes) and was rejected for that
// reason: it reorders the chain, and it would re-serialise the module on every
// build including the ones the memo is there to make free.
// Its two rename maps are therefore KEY MATERIAL - see the caller.
Vector<unsigned int> uniquifiedIoBlockSpirv;
if (!inputBlockRenames.empty() || !outputBlockRenames.empty()) {
std::set<String> stageRenamedIoBlockNames;
if (MG_Util::ShaderTranspiler::ShaderCompiler::UniquifyIoBlockNamesForEssl(
*effectiveSpirv, inputBlockRenames, outputBlockRenames,
stageRenamedIoBlockNames, uniquifiedIoBlockSpirv, enableSpirvValidation) &&
!uniquifiedIoBlockSpirv.empty() && !stageRenamedIoBlockNames.empty()) {
effectiveSpirv = &uniquifiedIoBlockSpirv;
MGLOG_D("Program %u stage %s: %zu inter-stage interface block(s) renamed per "
"producing stage, because some stage of this program declares the same "
"block name in both directions and the ES driver may alias the two.",
m_backendProgramId, MG_Util::ConvertGLEnumToString(glShaderType).c_str(),
stageRenamedIoBlockNames.size());
}
}
// ESSL stage-matches uniform blocks by member precision, but SPIRV-Cross prints
// a RelaxedPrecision member as explicit "mediump" in the vertex stage and as
// UNQUALIFIED (mediump-by-default) in the fragment stage; after
@@ -5306,6 +5372,61 @@ namespace MobileGL::MG_Backend::DirectGLES {
}
std::set<String> flattenedXfbBlockNames;
// Desktop GLSL keeps SEPARATE name namespaces for input and output interface
// blocks, so ONE stage may legally declare `in FOO {...}` and `out FOO {...}` at
// the same time - which the tessellation evaluation stage of both interface-block
// tests in KHR-GL42/43.shading_language_420pack does ("in TCSOutputBlock ... out
// TCSOutputBlock"). SPIRV-Cross keeps the same split (block_input_names vs
// block_output_names) and re-emits BOTH under the name FOO, so the generated ESSL
// declares two different blocks called FOO in one shader. Adreno's ES compiler
// keeps them apart; Mali's does not - the stage compiles, the program links, and
// the output block's payload never reaches the next stage. All 22 of that group's
// Mali failures are exactly the two tests that write this shape, and every one of
// them passes on Adreno and on DirectVulkan.
//
// The repair is a rename keyed on the PRODUCING stage, planned here and applied
// per stage below so a producer and its consumer keep naming the same block.
// Gated twice over, because a re-serialised module is not free (it cost the
// create-indirect retrace 0.15 SSIM the first time the array-input split missed
// its gate): only a tessellation or geometry stage can declare blocks in both
// directions at all, and even then the probe has to FIND a collision before any
// stage is rewritten.
std::set<String> collidingIoBlockNames;
std::set<String> declaredIoBlockNames;
Vector<Int> stagePipelineIndices(attachedShaders.size(), -1);
Bool anyStageCanDeclareBlocksInBothDirections = false;
for (SizeT index = 0; index < attachedShaders.size(); ++index) {
const ShaderStage stage = attachedShaders[index]->GetShaderStage();
stagePipelineIndices[index] = InterStagePipelineIndex(stage);
if (CanDeclareBlocksInBothDirections(stage)) anyStageCanDeclareBlocksInBothDirections = true;
}
if (anyStageCanDeclareBlocksInBothDirections) {
for (SizeT index = 0; index < attachedShaders.size() && index < shaderSpirvs.size(); ++index) {
MG_Util::ShaderTranspiler::ShaderCompiler::ProbeIoBlockNamesForEssl(
shaderSpirvs[index], collidingIoBlockNames, declaredIoBlockNames);
}
// A block a capture request names is resolved BY NAME at
// glTransformFeedbackVaryings time - and flattened away entirely by the pass
// below - so renaming one would ask the driver for a block the request does
// not spell.
for (const auto& xfbCaptureBlockName : xfbCaptureBlockNames) {
collidingIoBlockNames.erase(xfbCaptureBlockName);
}
}
// The one spelling every stage of THIS program agrees on for `blockName` as written
// by pipeline stage `producerPipelineIndex`. "__" is reserved in GLSL, so a name
// already ending in '_' does not get another one, and the digit-suffix loop steps
// off any name the program already spells.
const auto uniqueIoBlockName = [&declaredIoBlockNames](const String& blockName,
Int producerPipelineIndex) {
const char* separator = (!blockName.empty() && blockName.back() == '_') ? "" : "_";
String candidate = blockName + separator + "mgio" + std::to_string(producerPipelineIndex);
while (declaredIoBlockNames.find(candidate) != declaredIoBlockNames.end()) {
candidate += "0";
}
return candidate;
};
for (int index = 0; index < attachedShaders.size(); ++index) {
auto& shader = attachedShaders[index];
GLenum glShaderType = MG_Util::ConvertShaderStageToGLEnum(shader->GetShaderStage());
@@ -5359,6 +5480,52 @@ namespace MobileGL::MG_Backend::DirectGLES {
esslKeyInputs.storageBlockBindingOverrides = &storageBlockBindingOverrides;
esslKeyInputs.esslVersion = ResolveBackendEsslVersion();
esslKeyInputs.atomicCounterEsslBindingTop = m_atomicCounterEsslBindingTop;
// THIS STAGE's share of the program-wide interface-block rename plan built above
// the loop. Resolved here, outside the memoized segment, because it is planning
// and not translation - exactly like the image-format bake map - and because that
// makes the two maps a plain function argument the L2 key can carry.
//
// KEYING ON THE RESOLVED MAPS rather than on what they were derived from
// (collidingIoBlockNames, declaredIoBlockNames, stagePipelineIndices, this
// stage's index) is deliberate: the maps ARE the pass's arguments, so they are
// exactly as fine as the pass's behaviour and no finer. Two programs whose
// collision plans differ but whose maps for THIS stage come out identical really
// do produce the same ESSL and should share the entry.
//
// A block whose other end is NOT in this program is deliberately left out of the
// plan: in a separate-shader-objects pipeline the interface it matches across
// lives in another program that never saw this plan, and renaming one side of
// THAT would break a program pipeline to repair a driver quirk. That is what the
// producer/consumer presence tests below are for - in a monolithic program both
// are trivially satisfied for every interface the collision can touch.
std::map<String, String> inputBlockRenames;
std::map<String, String> outputBlockRenames;
if (!collidingIoBlockNames.empty() && stagePipelineIndices[index] >= 0) {
const Int myPipelineIndex = stagePipelineIndices[index];
Int producerPipelineIndex = -1;
Bool hasConsumerStage = false;
for (const Int otherPipelineIndex : stagePipelineIndices) {
if (otherPipelineIndex < 0) continue;
if (otherPipelineIndex < myPipelineIndex &&
otherPipelineIndex > producerPipelineIndex) {
producerPipelineIndex = otherPipelineIndex;
}
if (otherPipelineIndex > myPipelineIndex) hasConsumerStage = true;
}
for (const auto& collidingBlockName : collidingIoBlockNames) {
if (producerPipelineIndex >= 0) {
inputBlockRenames[collidingBlockName] =
uniqueIoBlockName(collidingBlockName, producerPipelineIndex);
}
if (hasConsumerStage) {
outputBlockRenames[collidingBlockName] =
uniqueIoBlockName(collidingBlockName, myPipelineIndex);
}
}
}
esslKeyInputs.inputBlockRenames = &inputBlockRenames;
esslKeyInputs.outputBlockRenames = &outputBlockRenames;
esslKeyInputs.enableSpirvValidation = enableSpirvValidation;
auto& esslCache = MG_Util::ShaderTranspiler::GetEsslTranslationCache();
@@ -5384,6 +5551,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
String transpileError;
if (!TranspileSpirvToEssl(spirvCode, glShaderType, xfbCaptureBlockNames,
imageFormatBake, storageBlockBindingOverrides,
inputBlockRenames, outputBlockRenames,
m_atomicCounterEsslBindingTop,
enableSpirvValidation, source,
stageFlattenedXfbBlockNames,
@@ -1246,6 +1246,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
const std::set<String>& xfbCaptureBlockNames,
const ImageFormatBakeInputs& imageFormatBake,
const UnorderedMap<String, Int>& storageBlockBindingOverrides,
const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames,
Int atomicCounterEsslBindingTop, Bool enableSpirvValidation,
String& outSource,
std::set<String>& outFlattenedXfbBlockNames,
+6
View File
@@ -171,6 +171,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (!capabilities.SupportsRenderSnorm || !capabilities.SupportsNorm16Texture) {
options |= PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget;
}
// 8-bit signed-normalized storage is core ES, so only the rendering half is in
// question here; the 16-bit bit above additionally needs EXT_texture_norm16 for the
// encoding to exist at all.
if (!capabilities.SupportsRenderSnorm) {
options |= PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget;
}
return options;
}
@@ -905,8 +905,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
const Int maxSupportedDrawBuffers = static_cast<Int>(MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS);
m_dynamicParameters.MaxDrawBuffers = std::min(m_vulkanCaps.MaxDrawBuffers, maxSupportedDrawBuffers);
m_dynamicParameters.MaxColorAttachments = std::min(m_vulkanCaps.MaxColorAttachments, maxSupportedDrawBuffers);
m_dynamicParameters.MaxClipDistances = m_vulkanCaps.MaxClipDistances;
// Same shape as the image-uniform limits three lines above: maxClipDistances is reported
// by every device, but declaring ClipDistance in a module needs the shaderClipDistance
// FEATURE, which VulkanRenderer enables exactly where the physical device has it. Without
// it the limit describes a capacity no shader may use, so report none.
m_dynamicParameters.MaxClipDistances =
m_vulkanCaps.SupportsShaderClipDistance ? std::max(m_vulkanCaps.MaxClipDistances, 0) : 0;
m_dynamicParameters.MaxViewports = m_vulkanCaps.MaxViewports;
// Assigned explicitly rather than left to the struct's defaults, like every other
// parameter here, so a second fill cannot inherit a stale value. GL_UNDEFINED_VERTEX is
// the truthful answer for DirectVulkan and a legal one (GL 4.6 table 23.65): which vertex
// provokes is chosen per pipeline by VulkanRenderer::SelectProvokingVertexMode out of
// VK_EXT_provoking_vertex, provokingVertexModePerPipeline and the topology, so there is no
// one convention to name. Vulkan's own default is FIRST, which is the opposite of the
// GL_LAST_VERTEX_CONVENTION this used to claim unconditionally.
m_dynamicParameters.LayerProvokingVertex = GL_UNDEFINED_VERTEX;
m_dynamicParameters.ViewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
m_dynamicParameters.MaxViewportWidth = m_vulkanCaps.MaxViewportWidth;
m_dynamicParameters.MaxViewportHeight = m_vulkanCaps.MaxViewportHeight;
m_dynamicParameters.ViewportBoundsRangeMin = m_vulkanCaps.ViewportBoundsRangeMin;
+19 -7
View File
@@ -1572,9 +1572,6 @@ namespace MobileGL::MG_Impl::GLImpl {
case GL_LINE_WIDTH:
*params = static_cast<GLint>(MG_State::pGLContext->GetLineWidth());
return;
case GL_LAYER_PROVOKING_VERTEX:
*params = GL_LAST_VERTEX_CONVENTION;
return;
case GL_LOGIC_OP_MODE:
*params = static_cast<GLint>(MG_Util::ConvertLogicOperationToGLEnum(MG_State::pGLContext->GetLogicOp()));
return;
@@ -1664,7 +1661,11 @@ namespace MobileGL::MG_Impl::GLImpl {
*params = MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::Multisample) ? GL_TRUE : GL_FALSE;
return;
case GL_MIN_MAP_BUFFER_ALIGNMENT:
*params = 64; // TODO
// The same constant the map paths align to (MG_State/GLState/BufferState/
// PipeResource.h), never a literal: this number is a PROMISE about the pointers
// glMapBuffer and glMapBufferRange return, and the two used to be unrelated - the
// query said 64 while the pointers came out of a std::vector aligned to 16.
*params = static_cast<GLint>(MG_State::GLState::MIN_MAP_BUFFER_ALIGNMENT);
return;
case GL_MAX_LABEL_LENGTH:
*params = 256; // TODO
@@ -2113,9 +2114,6 @@ namespace MobileGL::MG_Impl::GLImpl {
params[3] = vp.w();
return;
}
case GL_VIEWPORT_INDEX_PROVOKING_VERTEX:
*params = GL_LAST_VERTEX_CONVENTION;
return;
case GL_MAX_ELEMENT_INDEX:
*params = 1024 * 1024; // TODO
return;
@@ -2203,6 +2201,20 @@ namespace MobileGL::MG_Impl::GLImpl {
case GL_MAX_CLIP_DISTANCES:
*params = dynamicParameters.MaxClipDistances;
break;
// Both were a hard-coded GL_LAST_VERTEX_CONVENTION, derived from nothing. GL 4.6 table
// 23.65 permits GL_UNDEFINED_VERTEX for either, and that is what the backends report
// wherever they do not actually pin a convention - claiming one is a statement about
// which vertex of a primitive supplies gl_Layer / gl_ViewportIndex, and DirectGLES
// rasterizes only viewport 0 on a driver without GL_OES_viewport_array while
// DirectVulkan picks its provoking mode per pipeline. KHR-GLxx.viewport_array.query
// accepts all four values, and .provoking_vertex - which failed on both devices, in
// OPPOSITE directions - stops verifying as soon as either answer is undefined.
case GL_LAYER_PROVOKING_VERTEX:
*params = static_cast<GLint>(dynamicParameters.LayerProvokingVertex);
break;
case GL_VIEWPORT_INDEX_PROVOKING_VERTEX:
*params = static_cast<GLint>(dynamicParameters.ViewportIndexProvokingVertex);
break;
case GL_MAX_COLOR_TEXTURE_SAMPLES:
*params = std::max(dynamicParameters.MaxColorTextureSamples, GetAdvertisedMaxSamples());
break;
+57 -4
View File
@@ -3749,12 +3749,57 @@ namespace MobileGL::MG_Impl::GLImpl {
}
return GetCopyImageLevelSize(endpoint.Texture, uploadTarget, level);
}
// How far the region's z axis may reach. It does not mean the same thing on every target
// GL 4.6 core 18.3.2 accepts: on a CUBE MAP it selects among the six faces, which this
// frontend keeps as six separate one-slice upload targets - so the level's own extent
// says 1 and the real bound is 6. A cube-map ARRAY is one upload target whose depth
// already counts layer-faces, and a 1D array carries its layers on y (which is where GL
// puts them for this entry point too), so both are answered by the level extent.
Int GetCopyImageEndpointLayerCount(const MG_Backend::CopyImageEndpoint& endpoint,
const IntVec3& levelSize) {
if (!endpoint.IsRenderbuffer() && endpoint.Texture &&
endpoint.Texture->GetTarget() == TextureTarget::TextureCubeMap) {
return 6;
}
return std::max(levelSize.z(), 1);
}
// GL 4.6 core 18.3.2 requires INVALID_VALUE when the region exceeds either image's
// boundaries. The only bounds-shaped call this validator used to make was
// ValidateCopyImageBlockAlignment, whose first line returns true for every UNCOMPRESSED
// format - so no uncompressed copy was bounded at all, and the z extent could not be
// bounded even in principle because srcZ/dstZ never reached the validator. Texture
// endpoints were covered only by accident, through the ES driver's own error, which the
// DirectGLES backend logs and swallows rather than reporting; a GL_RENDERBUFFER endpoint
// got neither (KHR-GL43.copy_image.exceeding_boundaries).
Bool ValidateCopyImageRegionBounds(const MG_Backend::CopyImageEndpoint& endpoint, const IntVec3& levelSize,
GLint x, GLint y, GLint z, GLsizei width, GLsizei height, GLsizei depth,
const char* endpointName) {
// An extent this frontend does not know cannot bound anything, and guessing would
// reject a copy GL allows. Every caller has already established that the level
// exists and that the image is complete, so this is a belt-and-braces guard.
if (levelSize.x() <= 0 || levelSize.y() <= 0) return true;
const Int layers = GetCopyImageEndpointLayerCount(endpoint, levelSize);
if (x >= 0 && y >= 0 && z >= 0 && static_cast<Int64>(x) + width <= levelSize.x() &&
static_cast<Int64>(y) + height <= levelSize.y() && static_cast<Int64>(z) + depth <= layers) {
return true;
}
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>(
"MG_Impl/GLImpl", "ValidateCopyImageSubData_State",
std::format("The {} region [{}, {}, {}] + [{} x {} x {}] does not fit inside the {} x {} x {} "
"image.",
endpointName, x, y, z, width, height, depth, levelSize.x(), levelSize.y(), layers)));
return false;
}
} // namespace
Bool ValidateCopyImageSubData_State(const MG_Backend::CopyImageEndpoint& src,
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY,
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
const MG_Backend::CopyImageEndpoint& dst,
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY,
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth) {
if (!ValidateCopyImageObjectExists(src, "source") ||
!ValidateCopyImageObjectExists(dst, "destination")) {
@@ -3849,6 +3894,14 @@ namespace MobileGL::MG_Impl::GLImpl {
dstLevelSize.x(), dstLevelSize.y(), "destination")) {
return false;
}
// One region extent, measured against both images: GL 4.6 core 18.3.2 gives the copy a
// single width/height/depth and requires it to fit in the source AND the destination.
if (!ValidateCopyImageRegionBounds(src, srcLevelSize, srcX, srcY, srcZ, srcWidth, srcHeight, srcDepth,
"source") ||
!ValidateCopyImageRegionBounds(dst, dstLevelSize, dstX, dstY, dstZ, srcWidth, srcHeight, srcDepth,
"destination")) {
return false;
}
return true;
}
@@ -6082,8 +6135,8 @@ namespace MobileGL::MG_Impl::GLImpl {
};
const MG_Backend::CopyImageEndpoint src = resolveEndpoint(srcName, srcTarget);
const MG_Backend::CopyImageEndpoint dst = resolveEndpoint(dstName, dstTarget);
if (!ValidateCopyImageSubData_State(src, srcTarget, srcLevel, srcX, srcY, dst, dstTarget,
dstLevel, dstX, dstY, srcWidth, srcHeight, srcDepth)) {
if (!ValidateCopyImageSubData_State(src, srcTarget, srcLevel, srcX, srcY, srcZ, dst, dstTarget,
dstLevel, dstX, dstY, dstZ, srcWidth, srcHeight, srcDepth)) {
return;
}
CopyImageSubData_Backend(src, srcTarget, srcLevel, srcX, srcY, srcZ, dst, dstTarget, dstLevel,
@@ -59,6 +59,7 @@ add_executable(MobileGLIntegrationTest
Scenarios/AsyncCompileScenario.cpp
Scenarios/XfbAfterClipDistanceScenario.cpp
Scenarios/ThreeChannelAttachmentScenario.cpp
Scenarios/SnormAttachmentScenario.cpp
Scenarios/PipelineFailureScenario.cpp
Scenarios/AdvertisedLimitsScenario.cpp
Scenarios/PixelStoreSweepScenario.cpp
@@ -83,6 +84,7 @@ add_executable(MobileGLIntegrationTest
Scenarios/ImageSizeAfterRespecScenario.cpp
Scenarios/SsboDeclarationFormScenario.cpp
Scenarios/Glsl420DeclarationScenario.cpp
Scenarios/IoBlockNameCollisionScenario.cpp
Scenarios/FragmentOutputArrayIndexScenario.cpp
Scenarios/BufferTextureScenario.cpp
Scenarios/VertexAttribBindingScenario.cpp
@@ -91,6 +93,7 @@ add_executable(MobileGLIntegrationTest
Scenarios/VertexArrayEnableDisableScenario.cpp
Scenarios/CopyImageLevelRangeScenario.cpp
Scenarios/CopyImageLayeredScenario.cpp
Scenarios/PackedWordReadbackScenario.cpp
Scenarios/LayeredAttachmentBarrierScenario.cpp
Scenarios/LayeredTextureReadbackScenario.cpp
Scenarios/AtomicCounterScenario.cpp
@@ -151,6 +151,18 @@ void main() { fragColor = vec4(0.0, 1.0, 0.0, 1.0); }
glReadPixels(x, y, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, out);
}
// GL_MAX_CLIP_DISTANCES is a real backend answer, not a constant: DirectGLES reports
// 0 on a driver without GL_EXT_clip_cull_distance, and DirectVulkan reports 0 without
// the shaderClipDistance device feature. On such a stack the shader above cannot
// compile - and MUST not, because declaring a clip distance the backend cannot host
// is exactly what used to link cleanly and then render nothing. Skip rather than
// fail: there is no clipping to assert about.
static bool BackendHostsTwoClipDistances() {
GLint maxClipDistances = 0;
glGetIntegerv(GL_MAX_CLIP_DISTANCES, &maxClipDistances);
return maxClipDistances >= 2;
}
// Never assume the eight start disabled - see the header note about
// XfbAfterClipDistanceScenario leaving one on for the rest of the process.
static void DisableEveryClipDistance() {
@@ -229,6 +241,9 @@ void main() { fragColor = vec4(0.0, 1.0, 0.0, 1.0); }
// The claim: an enabled clip distance removes the fragments where it is negative.
TEST_F(ClipDistanceScenario, AnEnabledClipDistanceRemovesTheNegativeHalf) {
if (!Ready()) return;
if (!BackendHostsTwoClipDistances()) {
GTEST_SKIP() << "this backend advertises no clip distances, so there is nothing to clip with";
}
HeadlessGL& gl = Gl();
const int width = gl.Width();
const int height = gl.Height();
@@ -280,6 +295,9 @@ void main() { fragColor = vec4(0.0, 1.0, 0.0, 1.0); }
// draw simply failed - would pass the case above.
TEST_F(ClipDistanceScenario, ADisabledClipDistanceRemovesNothing) {
if (!Ready()) return;
if (!BackendHostsTwoClipDistances()) {
GTEST_SKIP() << "this backend advertises no clip distances, so there is nothing to clip with";
}
HeadlessGL& gl = Gl();
const int width = gl.Width();
const int height = gl.Height();
@@ -329,6 +347,9 @@ void main() { fragColor = vec4(0.0, 1.0, 0.0, 1.0); }
// passes both cases above and fails this one.
TEST_F(ClipDistanceScenario, TheEnablesAreIndependentPerDistance) {
if (!Ready()) return;
if (!BackendHostsTwoClipDistances()) {
GTEST_SKIP() << "this backend advertises no clip distances, so there is nothing to clip with";
}
HeadlessGL& gl = Gl();
const int width = gl.Width();
const int height = gl.Height();
@@ -0,0 +1,389 @@
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/IoBlockNameCollisionScenario.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Scenario - ONE BLOCK NAME USED IN BOTH DIRECTIONS BY ONE STAGE STILL CARRIES ITS PAYLOAD.
//
// Desktop GLSL keeps SEPARATE name namespaces for input and output interface blocks, so a
// single stage may legally write
//
// in TcsData { ... } tes_in[];
// out TcsData { ... } tes_out;
//
// The tessellation evaluation stage of both interface-block tests in
// KHR-GL42/43.shading_language_420pack does exactly that, and MobileGL's backend used to
// hand the shape straight through: SPIRV-Cross splits the namespace the same way glslang
// does (block_input_names vs block_output_names) and re-emits BOTH blocks under the name
// TcsData, so the generated ESSL declares two different blocks of one name in one shader.
// Adreno's ES compiler keeps them apart. Mali's does not - the stage compiles, the program
// links, and the evaluation stage's writes never reach the geometry stage, which is all 22
// of that group's Mali failures and none of Adreno's or DirectVulkan's.
//
// Both cases below drive the SAME five-stage pipeline (vertex -> tessellation control ->
// tessellation evaluation -> geometry -> fragment) and differ only in whether the
// evaluation stage reuses one name. The distinct-name case is the negative control: it is
// what says a red pixel in the colliding case is about the name and not about this machine's
// tessellation, its geometry stage, or the block mechanism in general.
//
// Colour code, so a failure names its own cause:
// green - the payload crossed all four stage boundaries, which is the pass.
// blue - the clear colour: nothing was drawn at all (the program did not link, or the
// backend program was rejected and every draw became a no-op).
// red - the pipeline ran but the plain (non-block) varying did not arrive, i.e. the
// failure is not about interface blocks.
// black - the pipeline ran, the plain varying arrived, and the BLOCK payload came back
// zeroed or garbage. That is the defect this scenario exists for.
//
// llvmpipe and lavapipe run this faithfully but do NOT reproduce the original defect - the
// aliasing is a Mali ES compiler behaviour. Read a green run here as "the rename did not
// break the ordinary path"; the claim it pins on the device is the CTS group above.
#include <string>
#include <vector>
#include "../Harness/HeadlessGL.h"
#include "../Harness/ScenarioFixture.h"
#ifdef GLAPI
#undef GLAPI
#endif
#define GL_GLEXT_PROTOTYPES
#include <GL/gl.h>
#include <GL/glcorearb.h>
#undef GL_GLEXT_PROTOTYPES
namespace MGITest {
namespace {
// The payload starts here and is copied, unmodified, through every block below.
const char* const kVertexSource = R"(#version 420 core
out VsData {
vec4 payload;
} vs_out;
void main()
{
vs_out.payload = vec4(0.0, 1.0, 0.0, 1.0);
gl_Position = vec4(0.0, 0.0, 0.0, 1.0);
}
)";
const char* const kTessControlSource = R"(#version 420 core
layout(vertices = 1) out;
in VsData {
vec4 payload;
} tcs_in[];
out TcsData {
vec4 payload;
} tcs_out[];
void main()
{
tcs_out[gl_InvocationID].payload = tcs_in[gl_InvocationID].payload;
gl_TessLevelOuter[0] = 1.0;
gl_TessLevelOuter[1] = 1.0;
gl_TessLevelOuter[2] = 1.0;
gl_TessLevelOuter[3] = 1.0;
gl_TessLevelInner[0] = 1.0;
gl_TessLevelInner[1] = 1.0;
}
)";
// THE CASE UNDER TEST: one name, both directions, in one stage.
const char* const kCollidingTessEvalSource = R"(#version 420 core
layout(isolines, point_mode) in;
in TcsData {
vec4 payload;
} tes_in[];
out TcsData {
vec4 payload;
} tes_out;
out float tes_gs_alive;
void main()
{
tes_out.payload = tes_in[0].payload;
tes_gs_alive = 1.0;
}
)";
// The negative control: byte-identical but for the output block's name.
const char* const kDistinctTessEvalSource = R"(#version 420 core
layout(isolines, point_mode) in;
in TcsData {
vec4 payload;
} tes_in[];
out TesData {
vec4 payload;
} tes_out;
out float tes_gs_alive;
void main()
{
tes_out.payload = tes_in[0].payload;
tes_gs_alive = 1.0;
}
)";
// One geometry source per evaluation stage, because the block it consumes is named
// after the block the evaluation stage produced.
const char* const kCollidingGeometrySource = R"(#version 420 core
layout(points) in;
layout(triangle_strip, max_vertices = 4) out;
in TcsData {
vec4 payload;
} gs_in[];
in float tes_gs_alive[];
out GsData {
vec4 payload;
} gs_out;
out float gs_fs_alive;
void EmitCorner(vec2 corner)
{
gs_out.payload = gs_in[0].payload;
gs_fs_alive = tes_gs_alive[0];
gl_Position = vec4(corner, 0.0, 1.0);
EmitVertex();
}
void main()
{
EmitCorner(vec2(-1.0, -1.0));
EmitCorner(vec2(-1.0, 1.0));
EmitCorner(vec2( 1.0, -1.0));
EmitCorner(vec2( 1.0, 1.0));
}
)";
const char* const kDistinctGeometrySource = R"(#version 420 core
layout(points) in;
layout(triangle_strip, max_vertices = 4) out;
in TesData {
vec4 payload;
} gs_in[];
in float tes_gs_alive[];
out GsData {
vec4 payload;
} gs_out;
out float gs_fs_alive;
void EmitCorner(vec2 corner)
{
gs_out.payload = gs_in[0].payload;
gs_fs_alive = tes_gs_alive[0];
gl_Position = vec4(corner, 0.0, 1.0);
EmitVertex();
}
void main()
{
EmitCorner(vec2(-1.0, -1.0));
EmitCorner(vec2(-1.0, 1.0));
EmitCorner(vec2( 1.0, -1.0));
EmitCorner(vec2( 1.0, 1.0));
}
)";
// Red when the PLAIN varying did not arrive, so "the pipeline is broken" and "the
// block payload is broken" cannot be confused for one another.
const char* const kFragmentSource = R"(#version 420 core
in GsData {
vec4 payload;
} fs_in;
in float gs_fs_alive;
out vec4 fragColor;
void main()
{
fragColor = gs_fs_alive > 0.5 ? fs_in.payload : vec4(1.0, 0.0, 0.0, 1.0);
}
)";
class IoBlockNameCollisionScenario : public ScenarioTest {
protected:
void SetUp() override {
ScenarioTest::SetUp();
if (!Ready()) return;
glGenVertexArrays(1, &m_vao);
glBindVertexArray(m_vao);
if (!BackendHostsTessellationAndGeometry()) {
GTEST_SKIP() << "no tessellation/geometry stages on " << Gl().BackendName() << " ("
<< Gl().RendererString() << "); there is no five-stage pipeline to "
<< "carry a block through";
}
}
void TearDown() override {
if (!Ready()) return;
glUseProgram(0);
for (const GLuint program : m_programs) {
glDeleteProgram(program);
}
m_programs.clear();
glBindVertexArray(0);
if (m_vao != 0) glDeleteVertexArrays(1, &m_vao);
m_vao = 0;
}
// GL_MAX_TESS_GEN_LEVEL is a real backend answer, not a frontend constant: it
// reads 0 on a DirectGLES driver without GL_EXT_tessellation_shader and on a
// DirectVulkan device without the tessellationShader feature. There is no
// five-stage pipeline to assert about on such a stack.
static bool BackendHostsTessellationAndGeometry() {
GLint maxTessGenLevel = 0;
glGetIntegerv(GL_MAX_TESS_GEN_LEVEL, &maxTessGenLevel);
GLint maxGeometryOutputVertices = 0;
glGetIntegerv(GL_MAX_GEOMETRY_OUTPUT_VERTICES, &maxGeometryOutputVertices);
while (glGetError() != GL_NO_ERROR) {
}
return maxTessGenLevel >= 1 && maxGeometryOutputVertices >= 4;
}
GLuint BuildPipeline(const char* tessEvalSource, const char* geometrySource) {
const GLenum stages[] = {GL_VERTEX_SHADER, GL_TESS_CONTROL_SHADER,
GL_TESS_EVALUATION_SHADER, GL_GEOMETRY_SHADER,
GL_FRAGMENT_SHADER};
const char* const sources[] = {kVertexSource, kTessControlSource, tessEvalSource,
geometrySource, kFragmentSource};
GLuint shaders[5] = {0, 0, 0, 0, 0};
bool ok = true;
for (int i = 0; i < 5; ++i) {
shaders[i] = glCreateShader(stages[i]);
glShaderSource(shaders[i], 1, &sources[i], nullptr);
glCompileShader(shaders[i]);
GLint compiled = 0;
glGetShaderiv(shaders[i], GL_COMPILE_STATUS, &compiled);
if (!compiled) {
m_buildLog = InfoLog(shaders[i], true);
ok = false;
break;
}
}
if (!ok) {
for (const GLuint shader : shaders) {
if (shader != 0) glDeleteShader(shader);
}
return 0;
}
const GLuint program = glCreateProgram();
for (const GLuint shader : shaders) {
glAttachShader(program, shader);
}
glLinkProgram(program);
GLint linked = 0;
glGetProgramiv(program, GL_LINK_STATUS, &linked);
for (const GLuint shader : shaders) {
glDeleteShader(shader);
}
if (!linked) {
m_buildLog = InfoLog(program, false);
glDeleteProgram(program);
return 0;
}
m_programs.push_back(program);
return program;
}
// Clears to BLUE, so "the draw painted nothing" is a colour of its own rather
// than something that could be mistaken for a zeroed payload.
Rgba8 DrawAndReadCentre(GLuint program) const {
glViewport(0, 0, Gl().Width(), Gl().Height());
glClearColor(0.0f, 0.0f, 1.0f, 1.0f);
glClear(GL_COLOR_BUFFER_BIT);
glUseProgram(program);
glPatchParameteri(GL_PATCH_VERTICES, 1);
glDrawArrays(GL_PATCHES, 0, 1);
Rgba8 pixel{};
glReadPixels(Gl().Width() / 2, Gl().Height() / 2, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, &pixel);
return pixel;
}
static bool IsGreen(const Rgba8& pixel) {
return pixel.r < 64 && pixel.g > 192 && pixel.b < 64;
}
const std::string& BuildLog() const { return m_buildLog; }
static GLenum FirstGLError() {
const GLenum first = glGetError();
while (glGetError() != GL_NO_ERROR) {
}
return first;
}
private:
static std::string InfoLog(GLuint object, bool isShader) {
GLint length = 0;
if (isShader) {
glGetShaderiv(object, GL_INFO_LOG_LENGTH, &length);
} else {
glGetProgramiv(object, GL_INFO_LOG_LENGTH, &length);
}
std::vector<char> log(static_cast<std::size_t>(length > 1 ? length : 1), '\0');
if (isShader) {
glGetShaderInfoLog(object, static_cast<GLsizei>(log.size()), nullptr, log.data());
} else {
glGetProgramInfoLog(object, static_cast<GLsizei>(log.size()), nullptr, log.data());
}
return std::string(log.data());
}
GLuint m_vao = 0;
std::vector<GLuint> m_programs;
std::string m_buildLog;
};
// The negative control, and it runs first on purpose: if this one is not green there
// is nothing to conclude from the case below it.
//
// It is also the CALIBRATION. GL_MAX_TESS_GEN_LEVEL answers for the tessellation
// stages honestly, but nothing MobileGL reports answers for the geometry stage the
// same way (GL_MAX_GEOMETRY_* are frontend constants and an ES driver may legitimately
// report zero geometry storage blocks while having geometry shaders), so a stack that
// cannot build a five-stage program at all is recognised here, by trying.
TEST_F(IoBlockNameCollisionScenario, DistinctlyNamedBlocksCarryThePayloadThroughFiveStages) {
if (!Ready()) return;
const GLuint program = BuildPipeline(kDistinctTessEvalSource, kDistinctGeometrySource);
if (program == 0) {
GTEST_SKIP() << "this stack cannot build a five-stage tessellation+geometry program on "
<< Gl().BackendName() << ", so there is no block to carry through: "
<< BuildLog();
}
const Rgba8 centre = DrawAndReadCentre(program);
EXPECT_EQ(FirstGLError(), 0u);
EXPECT_TRUE(IsGreen(centre)) << "the control pipeline did not deliver its payload: " << centre;
}
TEST_F(IoBlockNameCollisionScenario, OneBlockNameInBothDirectionsStillCarriesThePayload) {
if (!Ready()) return;
// Same calibration as the case above, and for the same reason: a five-stage program
// this stack cannot build at all is not evidence about block names. Only once the
// DISTINCT-name build succeeds does a failure of the colliding one mean something.
if (BuildPipeline(kDistinctTessEvalSource, kDistinctGeometrySource) == 0) {
GTEST_SKIP() << "this stack cannot build a five-stage tessellation+geometry program on "
<< Gl().BackendName() << ", so there is no block to carry through: "
<< BuildLog();
}
// Legal desktop GLSL: input and output block names live in separate namespaces, so
// the evaluation stage below declares TcsData twice and must still compile. The
// control above having built is what makes this assertion about the NAME.
const GLuint program = BuildPipeline(kCollidingTessEvalSource, kCollidingGeometrySource);
ASSERT_NE(program, 0u)
<< "an interface block name reused across the two directions of one stage is legal "
"desktop GLSL, but the program did not build: "
<< BuildLog();
const Rgba8 centre = DrawAndReadCentre(program);
EXPECT_EQ(FirstGLError(), 0u);
EXPECT_TRUE(IsGreen(centre))
<< "the payload did not survive the stage that names its input and output block "
"the same: "
<< centre << " (blue: nothing drew; red: the plain varying was lost too; black: "
"the block arrived empty)";
}
} // namespace
} // namespace MGITest
@@ -0,0 +1,220 @@
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/PackedWordReadbackScenario.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// glGetTexImage of a 32-bit packed format read with its OWN client type owes the application the
// words the image HOLDS, and KHR-GL43.copy_image compares exactly those words. Two routes used to
// answer, and both are wrong for a level glCopyImageSubData wrote:
//
// * the colour-attachment route reads GL_RGBA/GL_FLOAT and re-encodes, which canonicalizes an
// RGB9_E5 shared exponent and collapses an R11F_G11F_B10F NaN payload to 1;
// * the CPU shadow only holds what was UPLOADED, and the mirror that replays a copy into it
// declines - silently - for a renderbuffer source, which has no shadow to mirror from.
//
// Both are pinned here with words the CTS itself uses, because both failures are invisible to a
// value comparison: every assertion below is on BITS that decode to the very value the wrong
// answer also decodes to.
//
// The fix is a raw-word route (DirectGLES::ReadPackedLevelWordsViaScratch: copy the level into a
// scratch GL_R32UI image, read that back as unsigned integers), and DirectVulkan reaches the same
// place through PackReadbackToClientOrPbo's raw-word branch over the staging bytes - so these
// scenarios are backend-agnostic on purpose.
#include <cstddef>
#include <ios>
#include <vector>
#include "../Harness/HeadlessGL.h"
#include "../Harness/ScenarioFixture.h"
#ifdef GLAPI
#undef GLAPI
#endif
#define GL_GLEXT_PROTOTYPES
#include <GL/gl.h>
#include <GL/glcorearb.h>
#undef GL_GLEXT_PROTOTYPES
namespace MGITest {
namespace {
constexpr GLsizei kExtent = 4;
// The non-canonical RGB9_E5 word KHR-GL43.copy_image writes: R=0, G=0, B mantissa 63,
// shared exponent 31, i.e. the value 8064, which the spec's own encoder would emit as
// 0xe7e00000 instead. Anything that decodes and re-encodes hands back the canonical word.
//
// Reinterpreted in the destination of an RGB9_E5 -> R11F_G11F_B10F copy it is R=0,
// G=1920, B=995 - and B's 5-bit exponent is all ones with a nonzero mantissa, i.e. a NaN
// whose payload 3 does not survive a float32 round trip (it comes back as the canonical
// payload 1, B=993, word 0xf87c0000). The two defects therefore land on the same word.
constexpr GLuint kRgb9E5Word = 0xf8fc0000u;
// The R11F_G11F_B10F word the same test pairs with it: R=0, G=0, B = exponent 12,
// mantissa 0 = 0.125. As an RGB9_E5 word it is all-zero channels with a shared exponent of
// 12, which the canonical encoder would write as 0x00000000 - so a decode/re-encode of THIS
// one loses every bit that distinguishes it.
constexpr GLuint kR11fG11fB10fWord = 0x60000000u;
class PackedWordReadbackScenario : public ScenarioTest {
protected:
void SetUp() override {
ScenarioTest::SetUp();
if (!Ready()) return;
DrainErrors();
}
void TearDown() override {
if (!Ready()) return;
DeleteObjects();
DrainErrors();
ScenarioTest::TearDown();
}
static void DrainErrors() {
for (int i = 0; i < 16 && glGetError() != GL_NO_ERROR; ++i) {
}
}
void DeleteObjects() {
if (m_src != 0) glDeleteTextures(1, &m_src);
if (m_dst != 0) glDeleteTextures(1, &m_dst);
if (m_rbo != 0) glDeleteRenderbuffers(1, &m_rbo);
m_src = 0;
m_dst = 0;
m_rbo = 0;
}
// A complete single-level texture whose every texel holds `word`, uploaded through the
// packed client type so the stored bits are the client's bits and nothing has had a
// chance to re-encode them.
GLuint MakePackedTexture(GLenum internalFormat, GLenum type, GLuint word) {
const std::vector<GLuint> words(static_cast<std::size_t>(kExtent) * kExtent, word);
GLuint texture = 0;
glGenTextures(1, &texture);
glBindTexture(GL_TEXTURE_2D, texture);
glTexImage2D(GL_TEXTURE_2D, 0, static_cast<GLint>(internalFormat), kExtent, kExtent, 0, GL_RGB, type,
words.data());
// What Utils::makeTextureComplete does in the conformance cases, and what
// glCopyImageSubData requires of both endpoints.
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_BASE_LEVEL, 0);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAX_LEVEL, 0);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
glBindTexture(GL_TEXTURE_2D, 0);
return texture;
}
// Every texel of level 0, as raw client words.
std::vector<GLuint> ReadPackedWords(GLuint texture, GLenum type) {
std::vector<GLuint> words(static_cast<std::size_t>(kExtent) * kExtent, 0xDEADBEEFu);
glBindTexture(GL_TEXTURE_2D, texture);
glGetTexImage(GL_TEXTURE_2D, 0, GL_RGB, type, words.data());
glBindTexture(GL_TEXTURE_2D, 0);
return words;
}
// The copy under test. Returns the error it raised so a driver that cannot perform the
// move at all can skip rather than fail: the point of these cases is which BITS come
// back, and there are none to compare if the copy never happened.
GLenum CopyWholeImage(GLuint srcName, GLenum srcTarget, GLuint dstName, GLenum dstTarget) {
DrainErrors();
glCopyImageSubData(srcName, srcTarget, 0, 0, 0, 0, dstName, dstTarget, 0, 0, 0, 0, kExtent, kExtent,
1);
const GLenum error = glGetError();
EXPECT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR)) << "the copy recorded more than one error";
return error;
}
static void ExpectEveryTexel(const std::vector<GLuint>& words, GLuint expected, const char* what) {
for (std::size_t i = 0; i < words.size(); ++i) {
ASSERT_EQ(words[i], expected)
<< what << ": texel " << i << " read 0x" << std::hex << words[i] << ", expected 0x"
<< expected;
}
}
GLuint m_src = 0;
GLuint m_dst = 0;
GLuint m_rbo = 0;
};
// The control that has to hold before either regression means anything: a packed word
// uploaded and read straight back must be the SAME word, not merely the same colour.
TEST_F(PackedWordReadbackScenario, AnUploadedPackedWordReadsBackVerbatim) {
if (!Ready()) GTEST_SKIP();
m_src = MakePackedTexture(GL_RGB9_E5, GL_UNSIGNED_INT_5_9_9_9_REV, kRgb9E5Word);
ASSERT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR)) << "RGB9_E5 upload";
ExpectEveryTexel(ReadPackedWords(m_src, GL_UNSIGNED_INT_5_9_9_9_REV), kRgb9E5Word, "RGB9_E5 round trip");
m_dst = MakePackedTexture(GL_R11F_G11F_B10F, GL_UNSIGNED_INT_10F_11F_11F_REV, kR11fG11fB10fWord);
ASSERT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR)) << "R11F_G11F_B10F upload";
ExpectEveryTexel(ReadPackedWords(m_dst, GL_UNSIGNED_INT_10F_11F_11F_REV), kR11fG11fB10fWord,
"R11F_G11F_B10F round trip");
}
// KHR-GL43.copy_image.functional rgb9_e5 -> r11f_g11f_b10f, all nine target combinations of
// which failed on both GPUs. glCopyImageSubData is a raw block move, so the destination
// physically holds the source's word - but the readback decoded it to float and re-encoded,
// and the destination's blue field is a NaN whose payload float32 does not carry. Every
// texel came back 0xf87c0000 (payload 1) instead of 0xf8fc0000 (payload 3): the same
// "colour", two bits apart.
TEST_F(PackedWordReadbackScenario, ACopiedRgb9E5WordSurvivesInAnR11fG11fB10fDestination) {
if (!Ready()) GTEST_SKIP();
m_src = MakePackedTexture(GL_RGB9_E5, GL_UNSIGNED_INT_5_9_9_9_REV, kRgb9E5Word);
m_dst = MakePackedTexture(GL_R11F_G11F_B10F, GL_UNSIGNED_INT_10F_11F_11F_REV, 0u);
ASSERT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR)) << "texture setup";
const GLenum copyError = CopyWholeImage(m_src, GL_TEXTURE_2D, m_dst, GL_TEXTURE_2D);
if (copyError != static_cast<GLenum>(GL_NO_ERROR)) {
GTEST_SKIP() << "this driver declined the RGB9_E5 -> R11F_G11F_B10F copy (" << copyError << ")";
}
ExpectEveryTexel(ReadPackedWords(m_dst, GL_UNSIGNED_INT_10F_11F_11F_REV), kRgb9E5Word,
"copied word in the R11F_G11F_B10F destination");
// ...and the source is still the source. This is verify()'s FIRST check in the
// conformance case, and the half that a canonicalizing readback fails on its own.
ExpectEveryTexel(ReadPackedWords(m_src, GL_UNSIGNED_INT_5_9_9_9_REV), kRgb9E5Word,
"the RGB9_E5 source after the copy");
}
// KHR-GL43.copy_image.functional *->rgb9_e5 with a GL_RENDERBUFFER source: exactly the three
// renderbuffer combinations of each such family failed, and no texture one did. The
// destination's CPU shadow is what the readback answered from, the mirror that replays a
// copy into it declines when an endpoint is a renderbuffer (there is no shadow to mirror
// FROM), and the decline is silent - so glGetTexImage handed back the destination's
// pre-copy contents. The word chosen here makes that unmissable: it decodes to the same
// all-zero channels the canonical encoder would write as 0x00000000.
TEST_F(PackedWordReadbackScenario, ACopyThroughARenderbufferReachesAnRgb9E5Destination) {
if (!Ready()) GTEST_SKIP();
m_src = MakePackedTexture(GL_R11F_G11F_B10F, GL_UNSIGNED_INT_10F_11F_11F_REV, kR11fG11fB10fWord);
m_dst = MakePackedTexture(GL_RGB9_E5, GL_UNSIGNED_INT_5_9_9_9_REV, 0xFFFFFFFFu);
glGenRenderbuffers(1, &m_rbo);
glBindRenderbuffer(GL_RENDERBUFFER, m_rbo);
glRenderbufferStorage(GL_RENDERBUFFER, GL_R11F_G11F_B10F, kExtent, kExtent);
glBindRenderbuffer(GL_RENDERBUFFER, 0);
ASSERT_EQ(glGetError(), static_cast<GLenum>(GL_NO_ERROR)) << "renderbuffer setup";
// The conformance case's own shape: texture -> renderbuffer -> texture.
const GLenum toRenderbuffer = CopyWholeImage(m_src, GL_TEXTURE_2D, m_rbo, GL_RENDERBUFFER);
if (toRenderbuffer != static_cast<GLenum>(GL_NO_ERROR)) {
GTEST_SKIP() << "this driver declined a renderbuffer copy destination (" << toRenderbuffer << ")";
}
const GLenum fromRenderbuffer = CopyWholeImage(m_rbo, GL_RENDERBUFFER, m_dst, GL_TEXTURE_2D);
if (fromRenderbuffer != static_cast<GLenum>(GL_NO_ERROR)) {
GTEST_SKIP() << "this driver declined a renderbuffer copy source (" << fromRenderbuffer << ")";
}
ExpectEveryTexel(ReadPackedWords(m_dst, GL_UNSIGNED_INT_5_9_9_9_REV), kR11fG11fB10fWord,
"copied word in the RGB9_E5 destination");
}
} // namespace
} // namespace MGITest
@@ -0,0 +1,245 @@
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/SnormAttachmentScenario.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Scenario - SIGNED-NORMALIZED COLOUR ATTACHMENTS, on a live driver.
//
// The bug: a GLES driver without GL_EXT_render_snorm treats every signed-normalized format as
// texture-only. DirectGLES had a colour-renderable substitute for exactly one of the eight
// (GL_RGB16_SNORM, through the three-channel widening), so an R8_SNORM or R16_SNORM attachment got
// no storage the driver would render into: the ES framebuffer was incomplete, the draw landed
// nowhere, and glGetTexImage fell through to the CPU shadow - all zeroes for a texture created with
// no data. KHR-GL4x.texture_swizzle renders into a SINGLE-CHANNEL SNORM output for every one of its
// SNORM source formats, which is why all 46 of its GL43 SNORM cases failed on Mali.
//
// THE OTHER HALF, and the reason this scenario asserts VALUES rather than only completeness: the
// substitute has to be exact. A half float's 11-bit mantissa cannot represent a 16-bit SNORM
// channel - 23451/32767 quantizes about six SNORM steps away, against a conformance window of one -
// so the 16-bit formats must land on a 32-bit float even though the 8-bit ones are fine in a half.
// Trading 46 visible failures for silent precision loss in Iris' SNORM normal buffers would be the
// worse outcome, so the round trip below is pinned tightly enough to fail on a half-float substitute
// (tolerance two SNORM steps, half-float error six).
//
// WHAT THIS GATE CAN AND CANNOT SEE. Both CI drivers (Mesa llvmpipe) and Adreno expose
// GL_EXT_render_snorm, so they take the NATIVE path here and the substitution stays dead. That is
// precisely why the assertions are written as invariants of the format rather than of the fallback:
// "a signed-normalized colour attachment is complete and round-trips its channel values" has to
// hold whichever path answers it, so the scenario fails if anyone ever routes these formats to a
// lossy storage on a driver where it IS live. The substitution itself can only be observed on a
// device without EXT_render_snorm (Mali Immortalis-G925).
//
// DirectGLES only, like the three-channel scenario next door: DirectVulkan resolves SNORM formats
// on its own terms and asserting Espryt's answers there would pin a coincidence.
#include <string>
#include <vector>
#include "../Harness/HeadlessGL.h"
#include "../Harness/ScenarioFixture.h"
#ifdef GLAPI
#undef GLAPI
#endif
#define GL_GLEXT_PROTOTYPES
#include <GL/gl.h>
#include <GL/glcorearb.h>
#undef GL_GLEXT_PROTOTYPES
namespace MGITest {
namespace {
constexpr const char* kVS = R"(#version 330 core
in vec2 aPos;
void main() {
gl_Position = vec4(aPos, 0.0, 1.0);
}
)";
// A uniform rather than a literal so nothing can constant-fold the value into a different
// precision than the one the attachment stores.
constexpr const char* kFS = R"(#version 330 core
out vec4 oColor;
uniform float uValue;
void main() { oColor = vec4(uValue, 0.0, 0.0, 1.0); }
)";
constexpr int kSize = 8;
// The two channel values the round trip is pinned on. Both are positive on purpose:
// glReadPixels applies GL_CLAMP_READ_COLOR (GL_FIXED_ONLY by default) to a fixed-point
// colour buffer, so the negative half of a SNORM attachment reads back as 0 and would
// measure the clamp instead of the storage.
constexpr int kSnorm8Value = 99;
constexpr int kSnorm16Value = 23451;
class SnormAttachmentScenario : public ScenarioTest {
protected:
void SetUp() override {
ScenarioTest::SetUp();
if (!Ready()) return;
if (Gl().BackendName() != "DirectGLES") {
GTEST_SKIP() << "the signed-normalized substitution is a DirectGLES fallback; backend is "
<< Gl().BackendName();
}
}
// A single-level 2D texture in `internalFormat`, or 0 when the driver rejects the
// storage outright (which is a different failure from rejecting the ATTACHMENT).
static GLuint MakeTexture(GLenum internalFormat) {
GLuint texture = 0;
glGenTextures(1, &texture);
glBindTexture(GL_TEXTURE_2D, texture);
glTexStorage2D(GL_TEXTURE_2D, 1, internalFormat, kSize, kSize);
if (glGetError() != GL_NO_ERROR) {
glDeleteTextures(1, &texture);
return 0;
}
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
glBindTexture(GL_TEXTURE_2D, 0);
return texture;
}
static GLenum SingleAttachmentStatus(GLenum internalFormat) {
const GLuint texture = MakeTexture(internalFormat);
if (texture == 0) return GL_NONE;
GLuint fbo = 0;
glGenFramebuffers(1, &fbo);
glBindFramebuffer(GL_DRAW_FRAMEBUFFER, fbo);
glFramebufferTexture2D(GL_DRAW_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, texture, 0);
const GLenum status = glCheckFramebufferStatus(GL_DRAW_FRAMEBUFFER);
glBindFramebuffer(GL_DRAW_FRAMEBUFFER, 0);
glDeleteFramebuffers(1, &fbo);
glDeleteTextures(1, &texture);
return status;
}
// Renders `value` into the red channel of a fresh `internalFormat` attachment and hands
// back what glReadPixels sees. Returns false when the framebuffer never came up, which
// is the failure mode this scenario exists for - a draw into an incomplete framebuffer
// is dropped by the driver and leaves the caller reading the cleared texture.
bool RenderAndReadRed(GLenum internalFormat, float value, float* outRed) {
std::string error;
const GLuint program = CompileProgram(kVS, kFS, &error);
EXPECT_NE(program, 0u) << error;
if (program == 0) return false;
const GLint valueLocation = glGetUniformLocation(program, "uValue");
EXPECT_GE(valueLocation, 0);
const GLuint texture = MakeTexture(internalFormat);
EXPECT_NE(texture, 0u) << "the driver refused the texture storage itself";
if (texture == 0) {
glDeleteProgram(program);
return false;
}
GLuint fbo = 0;
glGenFramebuffers(1, &fbo);
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
glFramebufferTexture2D(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, texture, 0);
const bool complete = glCheckFramebufferStatus(GL_FRAMEBUFFER) == GL_FRAMEBUFFER_COMPLETE;
if (complete) {
const float quad[] = {-1.0f, -1.0f, 1.0f, -1.0f, -1.0f, 1.0f, 1.0f, 1.0f};
GLuint vao = 0;
GLuint vbo = 0;
glGenVertexArrays(1, &vao);
glBindVertexArray(vao);
glGenBuffers(1, &vbo);
glBindBuffer(GL_ARRAY_BUFFER, vbo);
glBufferData(GL_ARRAY_BUFFER, sizeof(quad), quad, GL_STATIC_DRAW);
glEnableVertexAttribArray(0);
glVertexAttribPointer(0, 2, GL_FLOAT, GL_FALSE, 2 * sizeof(float), nullptr);
glUseProgram(program);
glUniform1f(valueLocation, value);
glViewport(0, 0, kSize, kSize);
// Cleared to zero so a dropped draw cannot be mistaken for a correct one.
glClearColor(0.0f, 0.0f, 0.0f, 0.0f);
glClear(GL_COLOR_BUFFER_BIT);
glDrawArrays(GL_TRIANGLE_STRIP, 0, 4);
std::vector<float> pixels(static_cast<std::size_t>(kSize) * kSize * 4, -1.0f);
glReadBuffer(GL_COLOR_ATTACHMENT0);
glReadPixels(0, 0, kSize, kSize, GL_RGBA, GL_FLOAT, pixels.data());
if (outRed) *outRed = pixels[0];
glDeleteBuffers(1, &vbo);
glDeleteVertexArrays(1, &vao);
}
glBindFramebuffer(GL_FRAMEBUFFER, 0);
glDeleteFramebuffers(1, &fbo);
glDeleteTextures(1, &texture);
glDeleteProgram(program);
return complete;
}
};
// THE regression gate for the frontend's answer. Every one of these used to be
// GL_FRAMEBUFFER_UNSUPPORTED on a driver without EXT_render_snorm, and nothing in the CTS
// (or in Iris) checks the status before drawing, so the failure was silent all the way to a
// readback of zeroes.
TEST_F(SnormAttachmentScenario, SignedNormalizedColorAttachmentsReportComplete) {
if (!Ready() || IsSkipped()) return;
// GL_R8 is the control: colour-renderable in ES core, so it must pass with or without
// any substitution. If it ever fails, nothing below means anything.
EXPECT_EQ(SingleAttachmentStatus(GL_R8), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE))
<< "GL_R8 is ES-core colour-renderable";
// The single-channel pair KHR-GL4x.texture_swizzle renders into for every SNORM source
// format - the whole 46-case failure.
EXPECT_EQ(SingleAttachmentStatus(GL_R8_SNORM), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE));
EXPECT_EQ(SingleAttachmentStatus(GL_R16_SNORM), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE));
// ...and the two- and four-channel siblings, which are what a shaderpack actually
// declares (Iris colortex buffers in RGBA16_SNORM).
EXPECT_EQ(SingleAttachmentStatus(GL_RG8_SNORM), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE));
EXPECT_EQ(SingleAttachmentStatus(GL_RG16_SNORM), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE));
EXPECT_EQ(SingleAttachmentStatus(GL_RGBA8_SNORM), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE));
EXPECT_EQ(SingleAttachmentStatus(GL_RGBA16_SNORM), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE));
EXPECT_EQ(FirstGLError(), 0u) << GLErrorName(FirstGLError());
}
// The other half: whatever storage answers for the attachment has to hold the channel value
// to the format's own precision. This is the assertion that fails if the 16-bit formats are
// ever routed to a half float - the substitute an implementer naturally reaches for, because
// it is what the 8-bit ones correctly use.
TEST_F(SnormAttachmentScenario, SignedNormalizedAttachmentsRoundTripTheirChannelValues) {
if (!Ready() || IsSkipped()) return;
const float snorm8Expected = static_cast<float>(kSnorm8Value) / 127.0f;
float red8 = -1.0f;
ASSERT_TRUE(RenderAndReadRed(GL_R8_SNORM, snorm8Expected, &red8))
<< "an R8_SNORM colour attachment must be complete before any value can be asserted";
// Two 8-bit SNORM steps. A half float is exact here (worst case 0.03 of a step), so this
// only has to catch a storage that quantizes harder than the format itself.
EXPECT_NEAR(red8, snorm8Expected, 2.0f / 127.0f)
<< "R8_SNORM attachment lost its channel value";
EXPECT_GT(red8, 0.5f) << "the draw never landed - this is the cleared texture, not the rendered one";
const float snorm16Expected = static_cast<float>(kSnorm16Value) / 32767.0f;
float red16 = -1.0f;
ASSERT_TRUE(RenderAndReadRed(GL_R16_SNORM, snorm16Expected, &red16))
<< "an R16_SNORM colour attachment must be complete before any value can be asserted";
// Two 16-bit SNORM steps (6.1e-5). A half float would land 1.9e-4 away - three times
// this window - which is exactly the failure this bound exists to catch.
EXPECT_NEAR(red16, snorm16Expected, 2.0f / 32767.0f)
<< "R16_SNORM attachment was stored in something that cannot hold 16 signed bits";
EXPECT_GT(red16, 0.5f) << "the draw never landed - this is the cleared texture, not the rendered one";
float red16x4 = -1.0f;
ASSERT_TRUE(RenderAndReadRed(GL_RGBA16_SNORM, snorm16Expected, &red16x4))
<< "an RGBA16_SNORM colour attachment must be complete before any value can be asserted";
EXPECT_NEAR(red16x4, snorm16Expected, 2.0f / 32767.0f)
<< "RGBA16_SNORM attachment was stored in something that cannot hold 16 signed bits";
EXPECT_EQ(FirstGLError(), 0u) << GLErrorName(FirstGLError());
}
} // namespace
} // namespace MGITest
@@ -163,7 +163,7 @@ namespace MobileGL::MG_State::GLState {
if (!m_resource.IsGpuResident() &&
!(m_mappingAccess & BufferMappingAccessBit::FlushExplicit)) { // if we didn't flush explicitly
if (!(m_mappingAccess & BufferMappingAccessBit::Persistent)) {
Memcpy(m_resource.Bytes() + m_mappedRange.start, m_stagingData.data(),
Memcpy(m_resource.Bytes() + m_mappedRange.start, m_stagingData.data() + m_stagingBias,
m_mappedRange.end - m_mappedRange.start);
}
NotifyFlushMappedRange(m_mappedRange, m_mappingAccess);
@@ -175,6 +175,7 @@ namespace MobileGL::MG_State::GLState {
m_isMapped = false;
m_mappingAccess = BufferMappingAccessBit::Null;
m_mappedRange = {0, 0};
m_stagingBias = 0;
m_ownsStagingData = false;
}
@@ -193,7 +194,7 @@ namespace MobileGL::MG_State::GLState {
// FLUSH_EXPLICIT maps are never GPU-resident (only coherent maps are adopted), so
// the staged bytes must be copied into the shadow before the backend reads them.
if (!(m_mappingAccess & BufferMappingAccessBit::Persistent)) {
Memcpy(m_resource.Bytes() + start, m_stagingData.data() + offset, length);
Memcpy(m_resource.Bytes() + start, m_stagingData.data() + m_stagingBias + offset, length);
}
NotifyFlushMappedRange({start, end}, m_mappingAccess);
}
@@ -311,6 +312,9 @@ namespace MobileGL::MG_State::GLState {
m_mappedRange = {0, m_size};
if (m_mappingAccess & BufferMappingAccessBit::Write) {
// glMapBuffer maps from offset 0, so no bias: the allocation's own
// GL_MIN_MAP_BUFFER_ALIGNMENT-aligned base is what the application must get.
m_stagingBias = 0;
m_stagingData.resize(m_size);
m_ownsStagingData = true;
@@ -372,14 +376,21 @@ namespace MobileGL::MG_State::GLState {
}
if (access & BufferMappingAccessBit::Write) {
m_stagingData.resize(range.end - range.start);
// ARB_map_buffer_alignment constrains (returned pointer - offset), not the pointer:
// a map at offset 63 must hand back a pointer 63 bytes past the alignment grid, which
// is exactly what the read path below gets for free from shadowBase + offset. The
// staging store has to be biased by the same phase to match, so it over-allocates by
// it and the mapped bytes start at data() + m_stagingBias.
m_stagingBias = range.start % MIN_MAP_BUFFER_ALIGNMENT;
const SizeT mappedLength = range.end - range.start;
m_stagingData.resize(m_stagingBias + mappedLength);
m_ownsStagingData = true;
if (!(access & (BufferMappingAccessBit::InvalidateRange | BufferMappingAccessBit::InvalidateBuffer))) {
Memcpy(m_stagingData.data(), m_resource.Bytes() + range.start, m_stagingData.size());
Memcpy(m_stagingData.data() + m_stagingBias, m_resource.Bytes() + range.start, mappedLength);
}
return m_stagingData.data();
return m_stagingData.data() + m_stagingBias;
} else {
m_ownsStagingData = false;
return m_resource.Bytes() + range.start;
@@ -438,7 +449,7 @@ namespace MobileGL::MG_State::GLState {
return const_cast<Uint8*>(m_resource.Bytes()) + m_mappedRange.start;
}
if (m_ownsStagingData) {
return const_cast<Uint8*>(m_stagingData.data());
return const_cast<Uint8*>(m_stagingData.data()) + m_stagingBias;
}
return const_cast<Uint8*>(m_resource.Bytes()) + m_mappedRange.start;
}
@@ -239,7 +239,14 @@ namespace MobileGL {
// Set by MarkGpuWritten, cleared by SyncGpuWrites once the shadow is refreshed.
Bool m_gpuWritePending = false;
Range1D m_mappedRange;
Vector<Uint8> m_stagingData;
// The write-map staging store. MapAlignedData because the application is handed a
// pointer into it, and biased by m_stagingBias because ARB_map_buffer_alignment
// requires (returned pointer - offset) to be aligned, not the pointer itself: a range
// map at offset 63 must hand back a pointer sitting 63 bytes past the alignment grid.
// The bias is the offset's phase, so the mapped bytes still start at
// m_stagingData.data() + m_stagingBias and the allocation is that much longer.
MapAlignedData m_stagingData;
SizeT m_stagingBias = 0;
Bool m_ownsStagingData;
};
} // namespace MG_State::GLState
@@ -10,8 +10,56 @@
#include <Includes.h>
#include <MG_Util/Types.h>
#include <bit>
#include <new>
#include <vector>
namespace MobileGL::MG_State::GLState {
// GL_MIN_MAP_BUFFER_ALIGNMENT. GL 4.2 / ARB_map_buffer_alignment fix the minimum at 64 and
// MobileGL advertises exactly that (MG_Impl/GLImpl/Getter/GL_Getter.cpp reads this constant),
// so under-reporting is not available - the implementation has to be brought up to the number
// instead. The promise is about POINTERS, not just the query: glMapBuffer must return a
// 64-byte-aligned pointer, and glMapBufferRange must return one whose base - the returned
// pointer minus the offset the caller asked for - is. Every pointer the frontend hands out
// comes from the shadow below or from BufferObject's staging buffer, and std::vector only
// promises alignof(std::max_align_t) (16 on aarch64), so both allocations carry the alignment
// themselves. One constant for the getter and the allocator, because the two may never
// disagree - the same reason the atomic-counter limits are shared through
// MG_Util/ShaderTranspiler/Types.h.
inline constexpr SizeT MIN_MAP_BUFFER_ALIGNMENT = 64;
// Allocator that gives every allocation MIN_MAP_BUFFER_ALIGNMENT. Deliberately minimal: the
// vectors it backs hold raw bytes and are only ever sized, so allocate/deallocate plus the
// rebinding and equality boilerplate std::vector requires is the whole interface.
template <typename T>
struct MapAlignedAllocator {
using value_type = T;
MapAlignedAllocator() noexcept = default;
template <typename U>
MapAlignedAllocator(const MapAlignedAllocator<U>&) noexcept {}
T* allocate(SizeT count) {
if (count == 0) return nullptr;
return static_cast<T*>(
::operator new(count * sizeof(T), std::align_val_t{MIN_MAP_BUFFER_ALIGNMENT}));
}
void deallocate(T* pointer, SizeT) noexcept {
::operator delete(pointer, std::align_val_t{MIN_MAP_BUFFER_ALIGNMENT});
}
template <typename U>
Bool operator==(const MapAlignedAllocator<U>&) const noexcept {
return true;
}
template <typename U>
Bool operator!=(const MapAlignedAllocator<U>&) const noexcept {
return false;
}
};
// Byte store for anything the application may end up holding a mapped pointer into.
using MapAlignedData = std::vector<Uint8, MapAlignedAllocator<Uint8>>;
// Opaque, refcounted handle to the backend's GPU storage for one buffer
// (the driver-side resource). The active backend derives from it and attaches
// its own payload (VkBufferResource / GLESBufferResource). Held by PipeResource.
@@ -57,8 +105,8 @@ namespace MobileGL::MG_State::GLState {
}
// Direct shadow access, used only by the backend's upload-from-shadow path,
// which never runs for a GPU-resident (persistent) buffer.
Data& Shadow() { return *m_shadow; }
const Data& Shadow() const { return *m_shadow; }
MapAlignedData& Shadow() { return *m_shadow; }
const MapAlignedData& Shadow() const { return *m_shadow; }
// Transition to persistent GPU residency: adopt the backend's coherent
// mapped base as the source of truth and drop the CPU shadow. The caller
@@ -85,7 +133,10 @@ namespace MobileGL::MG_State::GLState {
SharedPtr<BackendBufferResource> ReleaseBackend() { return std::move(m_backend); }
private:
SharedPtr<Data> m_shadow = MakeShared<Data>();
// MapAlignedData, not Data: a read-only glMapBuffer hands the application this very
// pointer, and a range map hands it base + offset, so the base has to be on the
// GL_MIN_MAP_BUFFER_ALIGNMENT grid for either to satisfy ARB_map_buffer_alignment.
SharedPtr<MapAlignedData> m_shadow = MakeShared<MapAlignedData>();
void* m_gpuMapped = nullptr;
SharedPtr<BackendBufferResource> m_backend;
};
@@ -45,6 +45,31 @@ namespace {
GLint maxFragmentSsboBlocks = 9;
bool tessAndGeometrySsboBlocksQueried = false;
bool perStageSsboBlockQueryRaisesError = false;
// GL_MAX_CLIP_DISTANCES. Not ES core in any version - it exists only as
// GL_MAX_CLIP_DISTANCES_EXT under GL_EXT_clip_cull_distance - so asking a driver without
// the extension raises GL_INVALID_ENUM and leaves the out-param untouched. The "queried"
// flag is what pins the gating; the "raises error" knob is what pins the drain.
GLint maxClipDistances = 8;
bool maxClipDistancesQueried = false;
bool clipDistanceQueryRaisesError = false;
// GL_MAX_VIEWPORTS / GL_VIEWPORT_SUBPIXEL_BITS / GL_VIEWPORT_BOUNDS_RANGE are
// GL_OES_viewport_array state and, like the clip-distance pname, exist nowhere in ES core.
GLint maxViewports = 32;
GLint viewportSubpixelBits = 8;
bool viewportArrayLimitsQueried = false;
// GL_LAYER_PROVOKING_VERTEX is ES 3.2 core; GL_VIEWPORT_INDEX_PROVOKING_VERTEX comes with
// GL_OES_viewport_array. Both must go unasked where they do not exist, and a driver answer
// outside the four legal conventions must not be forwarded as one.
GLint layerProvokingVertex = GL_FIRST_VERTEX_CONVENTION;
GLint viewportIndexProvokingVertex = GL_LAST_VERTEX_CONVENTION;
bool layerProvokingVertexQueried = false;
// A driver rejecting one of the UNCONDITIONAL probes. GL_SMOOTH_LINE_WIDTH_RANGE is the
// realistic one - it is desktop-only state that every GLES driver refuses - and it stands
// in for the whole run: whatever it leaves behind must not reach the application.
bool smoothLineWidthQueryRaisesError = false;
// What the driver answers for the four multisample ceilings. Zero is the value that has
// to be floored away: the frontend would otherwise advertise a sample count it rejects.
GLint multisampleCeiling = 4;
GLfloat minFragmentInterpolationOffset = -0.75f;
GLfloat maxFragmentInterpolationOffset = 0.625f;
GLint fragmentInterpolationOffsetBits = 6;
@@ -160,6 +185,37 @@ namespace {
case GL_MAX_COMPUTE_IMAGE_UNIFORMS:
*data = g_fake.maxComputeImageUniforms;
break;
case GL_MAX_CLIP_DISTANCES:
g_fake.maxClipDistancesQueried = true;
if (g_fake.clipDistanceQueryRaisesError) {
g_fake.pendingError = GL_INVALID_ENUM;
} else {
*data = g_fake.maxClipDistances;
}
break;
case GL_MAX_VIEWPORTS:
g_fake.viewportArrayLimitsQueried = true;
*data = g_fake.maxViewports;
break;
case GL_VIEWPORT_SUBPIXEL_BITS:
g_fake.viewportArrayLimitsQueried = true;
*data = g_fake.viewportSubpixelBits;
break;
case GL_VIEWPORT_INDEX_PROVOKING_VERTEX:
g_fake.viewportArrayLimitsQueried = true;
*data = g_fake.viewportIndexProvokingVertex;
break;
case GL_LAYER_PROVOKING_VERTEX:
g_fake.layerProvokingVertexQueried = true;
*data = g_fake.layerProvokingVertex;
break;
case GL_MAX_COLOR_TEXTURE_SAMPLES:
case GL_MAX_DEPTH_TEXTURE_SAMPLES:
case GL_MAX_FRAMEBUFFER_SAMPLES:
case GL_MAX_INTEGER_SAMPLES:
case GL_MAX_SAMPLES:
*data = g_fake.multisampleCeiling;
break;
case GL_FRAGMENT_INTERPOLATION_OFFSET_BITS:
g_fake.fragmentInterpolationLimitsQueried = true;
if (g_fake.fragmentInterpolationQueryRaisesError) {
@@ -239,11 +295,22 @@ namespace {
data[0] = g_fake.maxFragmentInterpolationOffset;
}
break;
case GL_SMOOTH_LINE_WIDTH_RANGE:
if (g_fake.smoothLineWidthQueryRaisesError) {
g_fake.pendingError = GL_INVALID_ENUM;
} else {
data[0] = 0.0f;
data[1] = 0.0f;
}
break;
case GL_VIEWPORT_BOUNDS_RANGE:
g_fake.viewportArrayLimitsQueried = true;
data[0] = 0.0f;
data[1] = 0.0f;
break;
// Two-component range queries.
case GL_ALIASED_LINE_WIDTH_RANGE:
case GL_SMOOTH_LINE_WIDTH_RANGE:
case GL_ALIASED_POINT_SIZE_RANGE:
case GL_VIEWPORT_BOUNDS_RANGE:
data[0] = 0.0f;
data[1] = 0.0f;
break;
@@ -642,6 +709,186 @@ TEST(PerStageStorageBlockCapabilities, ARejectedQueryIsDrainedAndFallsBackToTheS
EXPECT_EQ(g_fake.pendingError, static_cast<GLenum>(GL_NO_ERROR));
}
// GL_MAX_CLIP_DISTANCES is the same defect as the per-stage storage blocks above, one pname
// over: the query does not exist without GL_EXT_clip_cull_distance, so an unguarded probe left
// an optimistic 8 behind on every ARM driver. Advertising eight clip planes a driver cannot host
// does not make gl_ClipDistance work - SPIRV-Cross emits it behind an `#extension ... : require`
// the ESSL compiler rejects, DirectGLES has nowhere to put the per-distance enables, and the
// draw renders nothing while LINK_STATUS says everything is fine.
TEST(ClipDistanceCapabilities, NoExtensionMeansNoClipDistancesAndNoQuery) {
const auto funcs = MakeFakeGLESFunctions();
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
MobileGL::MG_External::GLESCapabilities caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(caps, funcs));
EXPECT_FALSE(caps.SupportsClipDistance);
EXPECT_EQ(caps.MaxClipDistances, 0);
EXPECT_FALSE(g_fake.maxClipDistancesQueried)
<< "GL_MAX_CLIP_DISTANCES is not ES core; asking for it without the extension only leaks "
"a GL_INVALID_ENUM";
}
// The other half of the same claim, and the one that keeps this from being a blanket zero: a
// driver that HAS the extension must have its real limit come through untouched. Adreno does,
// and it passes the clip-distance conformance cases on the strength of it.
TEST(ClipDistanceCapabilities, TheExtensionIsQueriedAndItsLimitIsReportedVerbatim) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.extensions.emplace_back("GL_EXT_clip_cull_distance");
g_fake.maxClipDistances = 6;
const auto funcs = MakeFakeGLESFunctions();
MobileGL::MG_External::GLESCapabilities caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(caps, funcs));
EXPECT_TRUE(caps.SupportsClipDistance);
EXPECT_TRUE(g_fake.maxClipDistancesQueried);
EXPECT_EQ(caps.MaxClipDistances, 6);
}
// A driver that advertises the extension and then refuses the query is a driver fault, not a
// missing feature - but the answer has to be the honest zero either way, and the error must not
// be left for the application's first glGetError to find.
TEST(ClipDistanceCapabilities, ARejectedQueryIsDrainedAndReportsZero) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.extensions.emplace_back("GL_EXT_clip_cull_distance");
g_fake.clipDistanceQueryRaisesError = true;
const auto funcs = MakeFakeGLESFunctions();
MobileGL::MG_External::GLESCapabilities caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(caps, funcs));
EXPECT_TRUE(g_fake.maxClipDistancesQueried);
EXPECT_EQ(caps.MaxClipDistances, 0);
EXPECT_EQ(funcs.glGetError(), GL_NO_ERROR) << "the failed query must not leave an error behind";
}
// The same defect one more time, for the three GL_OES_viewport_array pnames. Their advertised
// values do not come from the driver (GL_Getter answers GL_MAX_VIEWPORTS from the frontend state
// width and floors GL_SUBPIXEL_BITS at its own constant), so what this pins is the other half of
// the class defect: a pname that does not exist must not be asked for, because the GL_INVALID_ENUM
// it raises is then attributed to whatever the application calls next.
TEST(ViewportArrayCapabilities, TheLimitsAreOnlyAskedForWhenTheExtensionIsPresent) {
const auto funcs = MakeFakeGLESFunctions();
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
MobileGL::MG_External::GLESCapabilities withoutCaps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(withoutCaps, funcs));
EXPECT_FALSE(withoutCaps.SupportsViewportArray);
EXPECT_FALSE(g_fake.viewportArrayLimitsQueried);
EXPECT_EQ(withoutCaps.MaxViewports, 16) << "the OpenGL core minimum, not a driver answer";
EXPECT_FLOAT_EQ(withoutCaps.ViewportBoundsRangeMin, -32768.0f);
EXPECT_FLOAT_EQ(withoutCaps.ViewportBoundsRangeMax, 32767.0f);
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.extensions.emplace_back("GL_OES_viewport_array");
MobileGL::MG_External::GLESCapabilities withCaps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(withCaps, funcs));
EXPECT_TRUE(withCaps.SupportsViewportArray);
EXPECT_TRUE(g_fake.viewportArrayLimitsQueried);
EXPECT_EQ(withCaps.MaxViewports, g_fake.maxViewports);
EXPECT_EQ(withCaps.ViewportSubpixelBits, g_fake.viewportSubpixelBits);
}
// GL_LAYER_PROVOKING_VERTEX and GL_VIEWPORT_INDEX_PROVOKING_VERTEX name which vertex of a
// primitive supplies gl_Layer and gl_ViewportIndex. MobileGL used to answer a hard-coded
// GL_LAST_VERTEX_CONVENTION for both, derived from nothing, and got it wrong on both test devices
// in OPPOSITE directions. GL_UNDEFINED_VERTEX is a legal answer (GL 4.6 table 23.65) and it is
// the honest one wherever the capability that would give the convention meaning is absent.
TEST(ProvokingVertexConventions, AreTakenFromTheDriverOnlyWhereThePnameExists) {
const auto funcs = MakeFakeGLESFunctions();
// ES 3.1, no viewport array: neither pname exists, so neither is asked for.
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
MobileGL::MG_External::GLESCapabilities es31Caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(es31Caps, funcs));
EXPECT_FALSE(g_fake.layerProvokingVertexQueried);
EXPECT_EQ(es31Caps.LayerProvokingVertex, static_cast<GLenum>(GL_UNDEFINED_VERTEX));
EXPECT_EQ(es31Caps.ViewportIndexProvokingVertex, static_cast<GLenum>(GL_UNDEFINED_VERTEX));
// ES 3.2 with the viewport array: both exist and both driver answers come through verbatim.
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.glesMinorVersion = 2;
g_fake.extensions.emplace_back("GL_OES_viewport_array");
MobileGL::MG_External::GLESCapabilities es32Caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(es32Caps, funcs));
EXPECT_TRUE(g_fake.layerProvokingVertexQueried);
EXPECT_EQ(es32Caps.LayerProvokingVertex, static_cast<GLenum>(GL_FIRST_VERTEX_CONVENTION));
EXPECT_EQ(es32Caps.ViewportIndexProvokingVertex, static_cast<GLenum>(GL_LAST_VERTEX_CONVENTION));
// ES 3.2 WITHOUT the viewport array - the shape of both test devices. The layer convention is
// real and comes from the driver; the viewport-index one describes a selection that never
// happens, because only viewport 0 is ever rasterized, and stays undefined.
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.glesMinorVersion = 2;
MobileGL::MG_External::GLESCapabilities deviceLikeCaps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(deviceLikeCaps, funcs));
EXPECT_EQ(deviceLikeCaps.LayerProvokingVertex, static_cast<GLenum>(GL_FIRST_VERTEX_CONVENTION));
EXPECT_EQ(deviceLikeCaps.ViewportIndexProvokingVertex, static_cast<GLenum>(GL_UNDEFINED_VERTEX));
}
// A driver answering something that is not one of the four legal conventions must not have it
// forwarded as one: GL_UNDEFINED_VERTEX describes "MobileGL cannot tell you" exactly.
TEST(ProvokingVertexConventions, AnIllegalDriverAnswerBecomesUndefined) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.glesMinorVersion = 2;
g_fake.layerProvokingVertex = 0x1234;
const auto funcs = MakeFakeGLESFunctions();
MobileGL::MG_External::GLESCapabilities caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(caps, funcs));
EXPECT_TRUE(g_fake.layerProvokingVertexQueried);
EXPECT_EQ(caps.LayerProvokingVertex, static_cast<GLenum>(GL_UNDEFINED_VERTEX));
}
// The multisample ceilings are ES 3.1 state; a driver that answers zero - or an older context
// that answers nothing - must not have that reach GL_Getter, which would then reject the sample
// count it just advertised.
TEST(MultisampleCapabilities, TheAdvertisedSampleCountsNeverFallBelowOne) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.multisampleCeiling = 0;
const auto funcs = MakeFakeGLESFunctions();
MobileGL::MG_External::GLESCapabilities caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(caps, funcs));
EXPECT_EQ(caps.MaxColorTextureSamples, 1);
EXPECT_EQ(caps.MaxDepthTextureSamples, 1);
EXPECT_EQ(caps.MaxFramebufferSamples, 1);
EXPECT_EQ(caps.MaxIntegerSamples, 1);
EXPECT_EQ(caps.MaxSamples, 1);
EXPECT_EQ(caps.MaxSampleMaskWords, 1);
}
// The whole point of the drain, stated once at the level that matters: capability init is the
// first thing that ever touches the driver, so an error it leaves behind surfaces at the
// APPLICATION's first glGetError and is blamed on an unrelated call. GL_SMOOTH_LINE_WIDTH_RANGE
// is the stand-in because it is desktop-only state that every real GLES driver refuses.
TEST(CapabilityProbeHygiene, ARejectedUnconditionalProbeLeavesNoErrorBehind) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.smoothLineWidthQueryRaisesError = true;
const auto funcs = MakeFakeGLESFunctions();
MobileGL::MG_External::GLESCapabilities caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(caps, funcs));
EXPECT_EQ(funcs.glGetError(), GL_NO_ERROR)
<< "capability init must not hand the application an error it never caused";
}
TEST(FragmentInterpolationCapabilities, QueriesOnlyWhenSupportedAndPreservesDriverLimits) {
const auto funcs = MakeFakeGLESFunctions();
+111
View File
@@ -8,6 +8,7 @@
#include <gtest/gtest.h>
#include <cstdint>
#include <limits>
#include "Includes.h"
@@ -267,6 +268,116 @@ TEST_F(BufferTest, AcquireMemoryRangeWithExplicit) {
ASSERT_EQ(actual, expected);
}
// GL_MIN_MAP_BUFFER_ALIGNMENT is a promise about POINTERS, and MobileGL used to keep only the
// query half of it: glGetIntegerv answered 64 while every mapped pointer came out of a plain
// std::vector, aligned to alignof(std::max_align_t) - 16 on aarch64. GL 4.2 /
// ARB_map_buffer_alignment fix the minimum at 64, so under-reporting is not available and the
// implementation has to be brought up to the number instead. Note the two different constraints:
// glMapBuffer's pointer must be aligned outright, while glMapBufferRange's must be aligned AFTER
// subtracting the offset the caller asked for - i.e. it sits at the offset's own alignment phase.
// KHR-GLxx.map_buffer_alignment.functional asserts exactly these two, at offset 63, for 24
// storage-flag combinations across 14 targets, and failed identically on both test devices.
TEST_F(BufferTest, MappedPointersHonourTheAdvertisedMapBufferAlignment) {
GLint advertisedAlignment = 0;
MobileGL::MG_Impl::GLImpl::GetIntegerv(GL_MIN_MAP_BUFFER_ALIGNMENT, &advertisedAlignment);
ASSERT_EQ(advertisedAlignment, static_cast<GLint>(MobileGL::MG_State::GLState::MIN_MAP_BUFFER_ALIGNMENT))
<< "the query and the allocator must read the same constant";
ASSERT_GE(advertisedAlignment, 64) << "GL 4.2 fixes the minimum at 64";
const SizeT alignment = static_cast<SizeT>(advertisedAlignment);
auto& slot = MobileGL::MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Uniform);
Vector<Uint> bufferNames;
MobileGL::MG_State::pGLContext->GenBufferNames(1, bufferNames);
auto bufObj = MobileGL::MG_State::pGLContext->CreateBufferObject(bufferNames[0]);
slot.Bind(bufObj);
// The conformance test's own shape: a buffer two alignments long, mapped from the last byte
// inside the first alignment - the offset most likely to expose a base-aligned-only fix.
const SizeT bufferSize = 2 * alignment;
const SizeT offset = alignment - 1;
bufObj->Resize(bufferSize);
Vector<Uint8> initData(bufferSize);
for (SizeT i = 0; i < bufferSize; ++i) initData[i] = static_cast<Uint8>(i);
bufObj->UploadData(DataPtr{.data = initData.data(), .size = bufferSize}, 0);
const auto addressOf = [](const void* pointer) { return reinterpret_cast<std::uintptr_t>(pointer); };
// glMapBuffer, read-only: the shadow base itself is handed out.
void* readMapped = bufObj->AcquireMemory(true, true, false);
ASSERT_NE(readMapped, nullptr);
EXPECT_EQ(addressOf(readMapped) % alignment, 0u) << "glMapBuffer(GL_READ_ONLY) returned an unaligned pointer";
bufObj->ReleaseMemory();
// glMapBuffer, write: the staging store is handed out instead.
void* writeMapped = bufObj->AcquireMemory(true, false, true);
ASSERT_NE(writeMapped, nullptr);
EXPECT_EQ(addressOf(writeMapped) % alignment, 0u) << "glMapBuffer(GL_WRITE_ONLY) returned an unaligned pointer";
EXPECT_EQ(bufObj->GetMappedPointer(), writeMapped)
<< "GL_BUFFER_MAP_POINTER must report the pointer the map returned";
bufObj->ReleaseMemory();
// glMapBufferRange, read-only: shadow base + offset, so the phase falls out for free.
const Range1D mapRange{.start = offset, .end = bufferSize};
void* rangeRead = bufObj->AcquireMemoryRange(mapRange, BufferMappingAccessBit::Read);
ASSERT_NE(rangeRead, nullptr);
EXPECT_EQ((addressOf(rangeRead) - offset) % alignment, 0u)
<< "glMapBufferRange(READ) returned a pointer whose base is unaligned";
bufObj->ReleaseMemory();
// glMapBufferRange, write: the staging store has to be biased to the same phase, and the
// write-back has to follow the bias or the bytes land at the wrong place in the shadow.
Uint8* rangeWrite = static_cast<Uint8*>(bufObj->AcquireMemoryRange(mapRange, BufferMappingAccessBit::Write));
ASSERT_NE(rangeWrite, nullptr);
EXPECT_EQ((addressOf(rangeWrite) - offset) % alignment, 0u)
<< "glMapBufferRange(WRITE) returned a pointer whose base is unaligned";
EXPECT_EQ(bufObj->GetMappedPointer(), rangeWrite)
<< "GL_BUFFER_MAP_POINTER must report the pointer the map returned";
// Seeded from the shadow, so the mapped view starts at the offset's byte.
EXPECT_EQ(rangeWrite[0], static_cast<Uint8>(offset));
rangeWrite[0] = 0xAB;
rangeWrite[bufferSize - offset - 1] = 0xCD;
bufObj->ReleaseMemory();
Vector<Uint8> readBack(bufferSize);
bufObj->DownloadSubData(readBack.data(), 0, bufferSize);
EXPECT_EQ(readBack[offset], 0xAB) << "the biased staging write-back landed at the wrong offset";
EXPECT_EQ(readBack[bufferSize - 1], 0xCD) << "the biased staging write-back landed at the wrong offset";
EXPECT_EQ(readBack[offset - 1], static_cast<Uint8>(offset - 1)) << "the write-back overran the mapped range";
}
// The explicit-flush path reads through the same bias, one flush offset further in: a flush of
// [offset + 4, offset + 8) must copy the bytes the application wrote at rangeWrite[4..8), not the
// ones sitting four bytes into the raw allocation.
TEST_F(BufferTest, ExplicitFlushOfARangeMapFollowsTheAlignmentBias) {
auto& slot = MobileGL::MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Uniform);
Vector<Uint> bufferNames;
MobileGL::MG_State::pGLContext->GenBufferNames(1, bufferNames);
auto bufObj = MobileGL::MG_State::pGLContext->CreateBufferObject(bufferNames[0]);
slot.Bind(bufObj);
const SizeT alignment = MobileGL::MG_State::GLState::MIN_MAP_BUFFER_ALIGNMENT;
const SizeT bufferSize = 2 * alignment;
const SizeT offset = alignment - 1;
bufObj->Resize(bufferSize);
Vector<Uint8> initData(bufferSize, 0);
bufObj->UploadData(DataPtr{.data = initData.data(), .size = bufferSize}, 0);
const Range1D mapRange{.start = offset, .end = bufferSize};
Uint8* mapped = static_cast<Uint8*>(bufObj->AcquireMemoryRange(
mapRange, BufferMappingAccessBit::Write | BufferMappingAccessBit::FlushExplicit));
ASSERT_NE(mapped, nullptr);
mapped[4] = 0x5A;
mapped[5] = 0x5B;
bufObj->FlushMemoryRange(4, 2);
bufObj->ReleaseMemory();
Vector<Uint8> readBack(bufferSize);
bufObj->DownloadSubData(readBack.data(), 0, bufferSize);
EXPECT_EQ(readBack[offset + 4], 0x5A);
EXPECT_EQ(readBack[offset + 5], 0x5B);
EXPECT_EQ(readBack[offset + 3], 0x00) << "the explicit flush copied bytes outside the flushed range";
}
TEST_F(BufferTest, CopyBufferSubData) {
auto& srcSlot = MobileGL::MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::CopyRead);
auto& dstSlot = MobileGL::MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::CopyWrite);
+49
View File
@@ -643,6 +643,55 @@ TEST(DirectGLESSanity, PreservesHostPerStageImageUniformLimits) {
EXPECT_EQ(params.MaxComputeImageUniforms, 5);
}
// maxClipDistances is a LIMIT every Vulkan device reports; declaring ClipDistance in a module
// needs the shaderClipDistance FEATURE, which is separate and which VulkanRenderer enables only
// where the physical device has it. Forwarding the limit without the feature advertises eight
// clip planes no shader may use - the same shape as the image-uniform limits above, and the same
// shape as the GL_EXT_clip_cull_distance lie on DirectGLES. Not a blanket zero: a device WITH the
// feature keeps its real number.
TEST(DirectVulkanSanity, GatesClipDistancesOnTheShaderClipDistanceFeature) {
using namespace MobileGL;
MG_Backend::DirectVulkan::BackendObject_DirectVulkan backend;
MG_External::VulkanCapabilities caps;
caps.MaxClipDistances = 8;
caps.SupportsShaderClipDistance = false;
backend.ApplyVulkanCapabilitiesForTesting(caps);
EXPECT_EQ(backend.GetDynamicParameters().MaxClipDistances, 0);
caps.SupportsShaderClipDistance = true;
backend.ApplyVulkanCapabilitiesForTesting(caps);
EXPECT_EQ(backend.GetDynamicParameters().MaxClipDistances, 8);
}
// GL_LAYER_PROVOKING_VERTEX / GL_VIEWPORT_INDEX_PROVOKING_VERTEX were a hard-coded
// GL_LAST_VERTEX_CONVENTION for both backends, derived from nothing, and wrong on both test
// devices in opposite directions. DirectGLES now forwards what its loader resolved; DirectVulkan
// reports GL_UNDEFINED_VERTEX, which GL 4.6 table 23.65 permits and which is what the backend
// honestly implements - the provoking mode is chosen per pipeline out of VK_EXT_provoking_vertex,
// provokingVertexModePerPipeline and the topology.
TEST(ProvokingVertexConventions, EachBackendReportsWhatItActuallyPins) {
using namespace MobileGL;
MG_Backend::DirectGLES::BackendObject_DirectGLES glesBackend;
MG_External::GLESCapabilities glesCaps;
glesCaps.LayerProvokingVertex = GL_FIRST_VERTEX_CONVENTION;
glesCaps.ViewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
glesBackend.ApplyGLESCapabilitiesForTesting(glesCaps);
EXPECT_EQ(glesBackend.GetDynamicParameters().LayerProvokingVertex,
static_cast<GLenum>(GL_FIRST_VERTEX_CONVENTION));
EXPECT_EQ(glesBackend.GetDynamicParameters().ViewportIndexProvokingVertex,
static_cast<GLenum>(GL_UNDEFINED_VERTEX));
MG_Backend::DirectVulkan::BackendObject_DirectVulkan vkBackend;
MG_External::VulkanCapabilities vkCaps;
vkBackend.ApplyVulkanCapabilitiesForTesting(vkCaps);
EXPECT_EQ(vkBackend.GetDynamicParameters().LayerProvokingVertex, static_cast<GLenum>(GL_UNDEFINED_VERTEX));
EXPECT_EQ(vkBackend.GetDynamicParameters().ViewportIndexProvokingVertex,
static_cast<GLenum>(GL_UNDEFINED_VERTEX));
}
TEST(FragmentInterpolationCapabilities, PlumbsGLESAndBothVulkanPropertyPaths) {
using namespace MobileGL;
@@ -9,6 +9,7 @@ add_executable(
EmulateSubgroupsTest.cpp
DemoteFloat64Test.cpp
FlattenXfbInterfaceBlocksTest.cpp
UniquifyIoBlockNamesTest.cpp
LowerViewportIndexTest.cpp
ClampMultisampleFetchTest.cpp
)
@@ -627,6 +627,9 @@ TEST_F(TranslationCacheTest, TheFrontendFingerprintMovesWithEveryFrontendLimit)
// real drivers produce (z = 64 on ES against 1024 elsewhere).
{"params.MaxComputeTextureImageUnits",
[](CompileEnv& e) { e.params.MaxComputeTextureImageUnits += 1; }},
// wave4's 4fc3531d: glslang rejects gl_ClipDistance[i] past this at parse AND expands
// gl_MaxClipDistances from it, so it is both a compile gate and a baked constant.
{"params.MaxClipDistances", [](CompileEnv& e) { e.params.MaxClipDistances += 1; }},
{"maxComputeWorkGroupSize[0]", [](CompileEnv& e) { e.maxComputeWorkGroupSize[0] += 1; }},
{"maxComputeWorkGroupSize[1]", [](CompileEnv& e) { e.maxComputeWorkGroupSize[1] += 1; }},
{"maxComputeWorkGroupSize[2]", [](CompileEnv& e) { e.maxComputeWorkGroupSize[2] += 1; }},
@@ -830,6 +833,8 @@ TEST_F(TranslationCacheTest, L2KeyMovesWithEveryGateThatSteersTheEsslChain) {
const std::set<String> xfbBlocks{"StageData"};
const UnorderedMap<String, Uint> imageFormats{{"gImage", 0x8236u /*GL_R32UI*/}};
const UnorderedMap<String, Int> storageBindings{{"Data", 3}};
const std::map<String, String> ioBlockRenames{{"TCSOutputBlock", "TCSOutputBlock_mgio1"}};
const std::map<String, String> otherIoBlockRenames{{"TCSOutputBlock", "TCSOutputBlock_mgio2"}};
const EsslTranslationKeyInputs base = BaselineEsslInputs(spirv);
const TranslationCacheKey baseKey = BuildEsslTranslationKey(base);
@@ -898,6 +903,31 @@ TEST_F(TranslationCacheTest, L2KeyMovesWithEveryGateThatSteersTheEsslChain) {
v.esslVersion = 300;
variants.emplace_back("esslVersion", BuildEsslTranslationKey(v));
}
{ // SpvcSession::SetAtomicCounterBlockBindings - printed into the layout(binding=)
// qualifier of every synthesized counter block, so it changes the emitted text.
EsslTranslationKeyInputs v = base;
v.atomicCounterEsslBindingTop = 6;
variants.emplace_back("atomicCounterEsslBindingTop", BuildEsslTranslationKey(v));
}
{ // the two arguments to UniquifyIoBlockNamesForEssl. Separate cases, because a stage
// that CONSUMES a block renames it after the previous stage while one that PRODUCES
// it renames after itself - so the same block name legitimately maps to different
// spellings in the two maps, and a key that folded them together would let a
// consumer's plan be served to a producer.
EsslTranslationKeyInputs v = base;
v.inputBlockRenames = &ioBlockRenames;
variants.emplace_back("inputBlockRenames", BuildEsslTranslationKey(v));
}
{
EsslTranslationKeyInputs v = base;
v.outputBlockRenames = &ioBlockRenames;
variants.emplace_back("outputBlockRenames", BuildEsslTranslationKey(v));
}
{ // ...and a DIFFERENT target spelling for the same block name must not share either.
EsslTranslationKeyInputs v = base;
v.outputBlockRenames = &otherIoBlockRenames;
variants.emplace_back("outputBlockRenames(other target)", BuildEsslTranslationKey(v));
}
{
EsslTranslationKeyInputs v = base;
v.enableSpirvValidation = true;
@@ -0,0 +1,262 @@
// MobileGL - MobileGL/MG_Test/ShaderTranspiler/UniquifyIoBlockNamesTest.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include <gtest/gtest.h>
#include <map>
#include <set>
#include <string>
#include <vector>
#include "Includes.h"
#include "Init.h"
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/SpvcSession.h>
#include <MG_Util/ShaderTranspiler/Types.h>
#include <spirv-tools/libspirv.hpp>
using namespace MobileGL;
using MobileGL::MG_Util::ShaderTranspiler::SessionUsageBit;
using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler;
using MobileGL::MG_Util::ShaderTranspiler::SpvcSession;
namespace {
Vector<Uint32> CompileToSpirv(GLenum stage, const String& source) {
using namespace MG_Util::ShaderTranspiler;
ShaderAttrib shaderAttrib{.shaderType = stage, .sourceStr = source};
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
EXPECT_TRUE(shaderResult) << (shaderResult ? String{} : shaderResult.error().log);
if (!shaderResult) return {};
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
EXPECT_TRUE(programResult) << (programResult ? String{} : programResult.error().log);
if (!programResult) return {};
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {stage}, .program = *programResult.value()};
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
EXPECT_TRUE(binaryResult) << (binaryResult ? String{} : binaryResult.error().log);
if (!binaryResult || binaryResult->empty()) return {};
return binaryResult->front();
}
String Disassemble(const Vector<Uint32>& spirv) {
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
String text;
tools.Disassemble(spirv, &text);
return text;
}
String Transpile(const Vector<Uint32>& spirv) {
SpvcSession session(spirv, SessionUsageBit::Transpile);
auto essl = ShaderCompiler::DecompileShader(session);
EXPECT_TRUE(essl) << (essl ? String{} : essl.error().log);
return essl ? essl.value() : String{};
}
// The tessellation evaluation stage of
// KHR-GL42/43.shading_language_420pack.length_of_vector_and_matrix_* and
// .qualifier_order_block_*, reduced to the shape that matters: ONE block name used for
// both the block this stage consumes and the block it produces. Legal desktop GLSL - the
// input and output block namespaces are separate - and something SPIRV-Cross re-emits
// verbatim, so the ESSL it produces declares two different blocks called TCSOutputBlock.
const char* kCollidingTessEvalSource = R"(#version 420 core
layout(isolines, point_mode) in;
in vec4 tcs_tes_result[];
out vec4 tes_gs_result;
in TCSOutputBlock {
vec4 tcs_tes_variable;
} input_block[];
out TCSOutputBlock {
vec4 tes_gs_variable;
} output_block;
void main()
{
tes_gs_result = tcs_tes_result[0];
output_block.tes_gs_variable = input_block[0].tcs_tes_variable;
}
)";
// The same stage with the two blocks already named apart, which is the overwhelmingly
// common shape and the one that must go through untouched.
const char* kDistinctTessEvalSource = R"(#version 420 core
layout(isolines, point_mode) in;
in vec4 tcs_tes_result[];
out vec4 tes_gs_result;
in TCSOutputBlock {
vec4 tcs_tes_variable;
} input_block[];
out TESOutputBlock {
vec4 tes_gs_variable;
} output_block;
void main()
{
tes_gs_result = tcs_tes_result[0];
output_block.tes_gs_variable = input_block[0].tcs_tes_variable;
}
)";
// gl_PerVertex is an Input block AND an Output block of one name in every tessellation
// and geometry stage. It is the language's block, not the shader's, so it must never be
// reported and never be renamed.
const char* kBuiltinBlockOnlyTessEvalSource = R"(#version 420 core
layout(isolines, point_mode) in;
void main()
{
gl_Position = gl_in[0].gl_Position;
}
)";
} // namespace
class UniquifyIoBlockNamesTest : public ::testing::Test {
protected:
void SetUp() override {
MobileGL::Initialize();
m_validationFailuresAtStart = ShaderCompiler::SpirvValidationFailureCount();
}
void TearDown() override {
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), m_validationFailuresAtStart)
<< "the renamed module did not survive spirv-val";
}
Uint64 m_validationFailuresAtStart = 0;
};
TEST_F(UniquifyIoBlockNamesTest, ProbeReportsABlockNameUsedInBothDirections) {
const Vector<Uint32> input = CompileToSpirv(GL_TESS_EVALUATION_SHADER, kCollidingTessEvalSource);
ASSERT_FALSE(input.empty());
std::set<String> colliding;
std::set<String> declared;
ShaderCompiler::ProbeIoBlockNamesForEssl(input, colliding, declared);
EXPECT_EQ(colliding, (std::set<String>{"TCSOutputBlock"}));
// The name set the caller picks a replacement out of has to contain what the module
// already spells, or the replacement could land on top of an existing declaration.
EXPECT_NE(declared.find("TCSOutputBlock"), declared.end());
EXPECT_NE(declared.find("input_block"), declared.end());
EXPECT_NE(declared.find("output_block"), declared.end());
}
TEST_F(UniquifyIoBlockNamesTest, ProbeIgnoresAStageWhoseBlocksAlreadyHaveDistinctNames) {
const Vector<Uint32> input = CompileToSpirv(GL_TESS_EVALUATION_SHADER, kDistinctTessEvalSource);
ASSERT_FALSE(input.empty());
std::set<String> colliding;
std::set<String> declared;
ShaderCompiler::ProbeIoBlockNamesForEssl(input, colliding, declared);
EXPECT_TRUE(colliding.empty());
EXPECT_NE(declared.find("TCSOutputBlock"), declared.end());
}
TEST_F(UniquifyIoBlockNamesTest, ProbeNeverReportsTheBuiltinBlock) {
const Vector<Uint32> input =
CompileToSpirv(GL_TESS_EVALUATION_SHADER, kBuiltinBlockOnlyTessEvalSource);
ASSERT_FALSE(input.empty());
std::set<String> colliding;
std::set<String> declared;
ShaderCompiler::ProbeIoBlockNamesForEssl(input, colliding, declared);
// gl_PerVertex is read through gl_in and written through gl_Position, i.e. it is exactly
// the in-and-out-under-one-name shape - and renaming it would invent a block no driver
// knows.
EXPECT_TRUE(colliding.empty()) << "gl_PerVertex must never enter the rename plan";
}
TEST_F(UniquifyIoBlockNamesTest, RenamesTheTwoBlocksApartInTheEmittedEssl) {
const Vector<Uint32> input = CompileToSpirv(GL_TESS_EVALUATION_SHADER, kCollidingTessEvalSource);
ASSERT_FALSE(input.empty());
// The generated ESSL really does declare the block twice under one name before the fix -
// pinning the defect, not just the repair.
const String before = Transpile(input);
EXPECT_NE(before.find("in TCSOutputBlock"), String::npos) << before;
EXPECT_NE(before.find("out TCSOutputBlock"), String::npos) << before;
// The plan the DirectGLES program build makes for a five-stage program: what this stage
// consumes is spelled after the tessellation control stage (pipeline index 1) and what it
// produces after itself (pipeline index 2).
const std::map<String, String> inputRenames{{"TCSOutputBlock", "TCSOutputBlock_mgio1"}};
const std::map<String, String> outputRenames{{"TCSOutputBlock", "TCSOutputBlock_mgio2"}};
std::set<String> renamed;
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::UniquifyIoBlockNamesForEssl(input, inputRenames, outputRenames, renamed,
output, true));
ASSERT_FALSE(output.empty());
EXPECT_EQ(renamed, (std::set<String>{"TCSOutputBlock"}));
const String dis = Disassemble(output);
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
ASSERT_TRUE(tools.Validate(output)) << dis;
EXPECT_EQ(dis.find("\"TCSOutputBlock\""), String::npos)
<< "the colliding name is still on a block struct:\n"
<< dis;
EXPECT_NE(dis.find("\"TCSOutputBlock_mgio1\""), String::npos) << dis;
EXPECT_NE(dis.find("\"TCSOutputBlock_mgio2\""), String::npos) << dis;
const String after = Transpile(output);
EXPECT_NE(after.find("TCSOutputBlock_mgio1"), String::npos) << after;
EXPECT_NE(after.find("TCSOutputBlock_mgio2"), String::npos) << after;
// Only the block TYPE name moves: the instance names are what the body reads and writes
// through, and the member names are half of what ES matches the interface by.
EXPECT_NE(after.find("input_block"), String::npos) << after;
EXPECT_NE(after.find("output_block"), String::npos) << after;
EXPECT_NE(after.find("tcs_tes_variable"), String::npos) << after;
EXPECT_NE(after.find("tes_gs_variable"), String::npos) << after;
}
TEST_F(UniquifyIoBlockNamesTest, RenamesOnlyTheDirectionTheCallerPlanned) {
const Vector<Uint32> input = CompileToSpirv(GL_TESS_EVALUATION_SHADER, kCollidingTessEvalSource);
ASSERT_FALSE(input.empty());
// A separate-shader-objects program that ends at this stage plans no output rename,
// because the block's consumer lives in another program that never saw the plan.
const std::map<String, String> inputRenames{{"TCSOutputBlock", "TCSOutputBlock_mgio1"}};
std::set<String> renamed;
Vector<Uint32> output;
ASSERT_TRUE(
ShaderCompiler::UniquifyIoBlockNamesForEssl(input, inputRenames, {}, renamed, output, true));
ASSERT_FALSE(output.empty());
EXPECT_EQ(renamed, (std::set<String>{"TCSOutputBlock"}));
const String dis = Disassemble(output);
EXPECT_NE(dis.find("\"TCSOutputBlock_mgio1\""), String::npos) << dis;
// The output block keeps the name the other program still spells.
EXPECT_NE(dis.find("\"TCSOutputBlock\""), String::npos) << dis;
EXPECT_EQ(dis.find("\"TCSOutputBlock_mgio2\""), String::npos) << dis;
}
TEST_F(UniquifyIoBlockNamesTest, ReportsNothingWhenThePlanNamesNoBlockThisStageDeclares) {
const Vector<Uint32> input = CompileToSpirv(GL_TESS_EVALUATION_SHADER, kDistinctTessEvalSource);
ASSERT_FALSE(input.empty());
const std::map<String, String> renames{{"SomeOtherBlock", "SomeOtherBlock_mgio2"}};
std::set<String> renamed;
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::UniquifyIoBlockNamesForEssl(input, renames, renames, renamed, output, true));
// Empty is what tells the DirectGLES program build to keep the module it already had
// instead of adopting the optimizer's re-serialised copy.
EXPECT_TRUE(renamed.empty());
const String dis = Disassemble(output);
EXPECT_NE(dis.find("\"TCSOutputBlock\""), String::npos) << dis;
EXPECT_NE(dis.find("\"TESOutputBlock\""), String::npos) << dis;
}
+343 -7
View File
@@ -3197,11 +3197,19 @@ TEST_F(TextureTest, NormalizeLegacySizedFormatsMapToCanonicalShadowLayouts) {
GLenum type;
};
const Case cases[] = {
// Legacy <=8-bit-per-channel formats store as UNorm8 component arrays.
{GL_R3_G3_B2, GL_RGB565, GL_RGB, GL_UNSIGNED_BYTE},
{GL_RGB4, GL_RGB565, GL_RGB, GL_UNSIGNED_BYTE},
{GL_RGB5, GL_RGB565, GL_RGB, GL_UNSIGNED_BYTE},
{GL_RGBA2, GL_RGBA4, GL_RGBA, GL_UNSIGNED_BYTE},
// Legacy <=8-bit-per-channel DESKTOP-ONLY formats store as UNorm8 component arrays, in the
// 8-bit-per-channel ES format that layout already is. Storing them in the narrower
// GL_RGB565/GL_RGBA4 they nominally fit in made the driver requantize the shadow bytes on
// every upload, which is not lossless: 5-bit 2 -> UNorm8 16 -> 16/255*31 = 1.945, which a
// truncating driver reads back as 1 (KHR-GL43.copy_image rgb4->rgb4, 12/12 failing on Mali).
{GL_R3_G3_B2, GL_RGB8, GL_RGB, GL_UNSIGNED_BYTE},
{GL_RGB4, GL_RGB8, GL_RGB, GL_UNSIGNED_BYTE},
{GL_RGB5, GL_RGB8, GL_RGB, GL_UNSIGNED_BYTE},
{GL_RGBA2, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE},
// The two that are ES formats in their own right keep their native storage: an application
// that asks for GL_RGBA4 or GL_RGB5_A1 is asking for the smaller image, and the same
// normalization also picks the storage for glRenderbufferStorage, where those two are
// ordinary ES render targets rather than a desktop-compatibility shim.
{GL_RGBA4, GL_RGBA4, GL_RGBA, GL_UNSIGNED_BYTE},
{GL_RGB5_A1, GL_RGB5_A1, GL_RGBA, GL_UNSIGNED_BYTE},
// 10/12-bit channels store as UNorm16 component arrays.
@@ -3860,6 +3868,131 @@ TEST_F(TextureTest, ColorAttachableTargetsRequestTheThreeChannelWidening) {
PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget);
EXPECT_FALSE(GetRenderTargetNormalizeOptions(capabilities, texture2DIndex) &
PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget);
// ...and neither can an 8-bit one. That half of the answer used to be missing entirely, which
// is why an R8_SNORM / RG8_SNORM colour attachment got no substitute at all on a driver
// without EXT_render_snorm.
EXPECT_TRUE(GetRenderTargetNormalizeOptions(noSnormCapabilities, texture2DIndex) &
PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget);
EXPECT_FALSE(GetRenderTargetNormalizeOptions(capabilities, texture2DIndex) &
PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget);
EXPECT_FALSE(GetRenderTargetNormalizeOptions(noSnormCapabilities, bufferIndex) &
PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget);
// 8-bit signed-normalized storage is core ES, so only EXT_render_snorm gates the 8-bit bit;
// the 16-bit one also needs EXT_texture_norm16 for the encoding to exist at all.
MG_External::GLESCapabilities noNorm16Capabilities{};
noNorm16Capabilities.SupportsRenderSnorm = true;
noNorm16Capabilities.SupportsNorm16Texture = false;
EXPECT_TRUE(GetRenderTargetNormalizeOptions(noNorm16Capabilities, texture2DIndex) &
PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget);
EXPECT_FALSE(GetRenderTargetNormalizeOptions(noNorm16Capabilities, texture2DIndex) &
PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget);
}
// ---- Signed-normalized colour-renderable substitution (KHR-GL4x.texture_swizzle on Mali) -------
//
// A driver without GL_EXT_render_snorm treats every signed-normalized format as texture-only, so a
// colour attachment in one of them leaves the ES framebuffer incomplete: the draw lands nowhere and
// the readback falls through to the CPU shadow, which for a glTexImage2D(..., nullptr) output
// texture is all zeroes. The render-target bits used to reach GL_RGB16_SNORM alone, so five of the
// eight SNORM formats - and in particular the single-channel GL_R8_SNORM / GL_R16_SNORM that
// KHR-GL4x.texture_swizzle renders into for EVERY SNORM source format - had no fallback at all.
TEST_F(TextureTest, SnormRenderTargetOptionsApplyToEverySignedNormalizedFormat) {
using MG_Util::TextureFormatProcessor::GetApplicablePixelFormatNormalizeOptions;
const Flags<PixelFormatNormalizeOptionBit> requested =
PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget | PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget;
for (const GLenum internalFormat : {GL_R8_SNORM, GL_RG8_SNORM, GL_RGB8_SNORM, GL_RGBA8_SNORM}) {
const auto applicable = GetApplicablePixelFormatNormalizeOptions(internalFormat, requested);
EXPECT_TRUE(applicable & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)
<< "internalformat 0x" << std::hex << internalFormat;
// The two bits are per precision class, so the 16-bit one never reaches an 8-bit format -
// that is what keeps the fallback reason from naming both.
EXPECT_FALSE(applicable & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget)
<< "internalformat 0x" << std::hex << internalFormat;
}
for (const GLenum internalFormat : {GL_R16_SNORM, GL_RG16_SNORM, GL_RGB16_SNORM, GL_RGBA16_SNORM}) {
const auto applicable = GetApplicablePixelFormatNormalizeOptions(internalFormat, requested);
EXPECT_TRUE(applicable & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget)
<< "internalformat 0x" << std::hex << internalFormat;
EXPECT_FALSE(applicable & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)
<< "internalformat 0x" << std::hex << internalFormat;
}
// GL_RGB16_SNORM used to be granted the 16-bit bit only when the three-channel widening was
// requested alongside it, which made the answer depend on the order the caller assembled its
// option set in. The capability probe and the runtime storage choice assemble different sets.
EXPECT_TRUE(GetApplicablePixelFormatNormalizeOptions(GL_RGB16_SNORM,
PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) &
PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget);
// Nothing else responds to either bit; an unsigned-normalized or float format keeps its storage.
for (const GLenum internalFormat : {GL_R8, GL_R16, GL_RGBA8, GL_RGBA16, GL_RGB16F, GL_RGBA32F, GL_RGB9_E5}) {
EXPECT_FALSE(GetApplicablePixelFormatNormalizeOptions(internalFormat, requested))
<< "internalformat 0x" << std::hex << internalFormat;
}
}
TEST_F(TextureTest, SnormRenderTargetSubstitutesKeepEveryChannelValueExactly) {
using MG_Util::TextureFormatProcessor::NormalizePixelFormat;
struct Case {
GLenum requested;
Flags<PixelFormatNormalizeOptionBit> options;
GLenum internalFormat;
GLenum format;
GLenum type;
};
const Flags<PixelFormatNormalizeOptionBit> snorm8RT = PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget;
const Flags<PixelFormatNormalizeOptionBit> snorm16RT = PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget;
const Case cases[] = {
// 8-bit: a half float represents every v/127 exactly (the worst case, -123/127, quantizes
// 0.03 of a SNORM step away), so it is the same storage GL_RGBA8_SNORM already always got.
{GL_R8_SNORM, snorm8RT, GL_R16F, GL_RED, GL_FLOAT},
{GL_RG8_SNORM, snorm8RT, GL_RG16F, GL_RG, GL_FLOAT},
{GL_RGBA8_SNORM, snorm8RT, GL_RGBA16F, GL_RGBA, GL_FLOAT},
// 16-bit: NOT a half float. Its spacing just below 1.0 is some 16 SNORM steps, so it hands
// -23451/32767 back as -23457 against a conformance window of one step; a 32-bit float
// round-trips all 65535 channel values.
{GL_R16_SNORM, snorm16RT, GL_R32F, GL_RED, GL_FLOAT},
{GL_RG16_SNORM, snorm16RT, GL_RG32F, GL_RG, GL_FLOAT},
{GL_RGBA16_SNORM, snorm16RT, GL_RGBA32F, GL_RGBA, GL_FLOAT},
// The render-target bit outranks the narrower fallbacks, whichever way the caller's option
// set was assembled: the capability probe folds the driver options in, the runtime storage
// choice can see the render-target bit alone, and the two have to pick the same storage.
{GL_R16_SNORM, snorm16RT | PixelFormatNormalizeOptionBit::NoNorm16, GL_R32F, GL_RED, GL_FLOAT},
{GL_RG16_SNORM, snorm16RT | PixelFormatNormalizeOptionBit::NoSnorm16, GL_RG32F, GL_RG, GL_FLOAT},
{GL_RGBA16_SNORM,
snorm16RT | PixelFormatNormalizeOptionBit::NoNorm16 | PixelFormatNormalizeOptionBit::NoSnorm16,
GL_RGBA32F, GL_RGBA, GL_FLOAT},
// The three-channel formats go on through the widening, which outranks everything.
{GL_RGB8_SNORM, snorm8RT | PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget, GL_RGBA16F, GL_RGBA,
GL_FLOAT},
{GL_RGB16_SNORM, snorm16RT | PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget, GL_RGBA32F, GL_RGBA,
GL_FLOAT},
// Control: with EXT_render_snorm neither bit is ever set, so the driver that renders to the
// signed-normalized encoding keeps storing it byte for byte. This is the shape Adreno and
// llvmpipe take, which is why the substitution is invisible on every gate the project runs.
{GL_R8_SNORM, PixelFormatNormalizeOptionBit::None, GL_R8_SNORM, GL_RED, GL_BYTE},
{GL_RG8_SNORM, PixelFormatNormalizeOptionBit::None, GL_RG8_SNORM, GL_RG, GL_BYTE},
{GL_R16_SNORM, PixelFormatNormalizeOptionBit::None, GL_R16_SNORM, GL_RED, GL_SHORT},
{GL_RG16_SNORM, PixelFormatNormalizeOptionBit::None, GL_RG16_SNORM, GL_RG, GL_SHORT},
{GL_RGBA16_SNORM, PixelFormatNormalizeOptionBit::None, GL_RGBA16_SNORM, GL_RGBA, GL_SHORT},
// ...and the bit for the other precision class does nothing on its own.
{GL_R8_SNORM, snorm16RT, GL_R8_SNORM, GL_RED, GL_BYTE},
{GL_R16_SNORM, snorm8RT, GL_R16_SNORM, GL_RED, GL_SHORT},
};
for (const auto& testCase : cases) {
GLenum internalFormat = 0;
GLenum format = 0;
GLenum type = 0;
NormalizePixelFormat(testCase.requested, testCase.options, &internalFormat, &format, &type);
EXPECT_EQ(internalFormat, testCase.internalFormat) << "requested 0x" << std::hex << testCase.requested;
EXPECT_EQ(format, testCase.format) << "requested 0x" << std::hex << testCase.requested;
EXPECT_EQ(type, testCase.type) << "requested 0x" << std::hex << testCase.requested;
}
}
TEST_F(TextureTest, ThreeChannelRenderTargetOptionAppliesToEveryDeniedThreeChannelFormat) {
@@ -3910,9 +4043,10 @@ TEST_F(TextureTest, ThreeChannelWideningRetargetsInternalFormatAndTransferPairTo
{GL_RGB16F, widen, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT},
{GL_RGB32F, widen, GL_RGBA32F, GL_RGBA, GL_FLOAT},
// 16-bit SNORM keeps its encoding where EXT_render_snorm can render to it; a half float's
// 11-bit mantissa cannot represent a 16-bit SNORM channel exactly.
// 11-bit mantissa cannot represent a 16-bit SNORM channel exactly, so the driver that
// cannot render to the encoding gets the 32-bit float rather than the half.
{GL_RGB16_SNORM, widen, GL_RGBA16_SNORM, GL_RGBA, GL_SHORT},
{GL_RGB16_SNORM, widenNoSnorm16, GL_RGBA16F, GL_RGBA, GL_FLOAT},
{GL_RGB16_SNORM, widenNoSnorm16, GL_RGBA32F, GL_RGBA, GL_FLOAT},
// 16-bit UNORM and the legacy 10/12-bit formats stored as RGB16.
{GL_RGB16, widen, GL_RGBA32F, GL_RGBA, GL_FLOAT},
{GL_RGB10, widen, GL_RGBA32F, GL_RGBA, GL_FLOAT},
@@ -4682,6 +4816,208 @@ TEST_F(TextureTest, CopyImageSubDataChecksARenderbufferLevelAndStorage) {
ExpectSingleGlError(GL_INVALID_OPERATION);
}
// GL 4.6 core 18.3.2 requires INVALID_VALUE when the region exceeds either image's boundaries, and
// this validator had no bounds check whatsoever: the one call shaped like one,
// ValidateCopyImageBlockAlignment, returns true on its first line for every UNCOMPRESSED format.
// Texture endpoints only looked covered because the ES driver raised its own error - which
// DirectGLES logs and swallows, so the application saw GL_NO_ERROR and a destination that never
// changed (KHR-GL43.copy_image.exceeding_boundaries).
TEST_F(TextureTest, CopyImageSubDataRejectsARegionThatLeavesTheImage) {
const ScopedTextureBackendFunctionsOverride backendGuard;
MG_Backend::gBackendFunctionsTable.GL.CopyImageSubData = RecordCopyImageSubData;
g_copyImageSubDataCall = {};
GLuint srcTexture = 0;
GLuint dstTexture = 0;
MakeCopyImagePair(GL_RGBA8, GL_RGBA8, srcTexture, dstTexture);
ASSERT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
// The region that exactly reaches the far edge is the boundary this must NOT reject - a
// validator that answered INVALID_VALUE to every non-origin region would satisfy the negatives
// below and break every legal partial copy.
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D, 0, 4, 4, 0, dstTexture, GL_TEXTURE_2D, 0, 4, 4, 0,
4, 4, 1);
EXPECT_TRUE(g_copyImageSubDataCall.Called);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
// One texel past it on x, on y, and on the destination side.
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D, 0, 5, 4, 0, dstTexture, GL_TEXTURE_2D, 0, 0, 0, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D, 0, 4, 5, 0, dstTexture, GL_TEXTURE_2D, 0, 0, 0, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D, 0, 0, 0, 0, dstTexture, GL_TEXTURE_2D, 0, 5, 5, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
// A negative origin is out of bounds on the other side of the same rule.
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D, 0, -1, 0, 0, dstTexture, GL_TEXTURE_2D, 0, 0, 0, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
}
// The endpoint the missing bounds check actually cost: a renderbuffer never reaches the ES
// driver's texture-shaped checks either, so a 4x4 region at y = 14 of a 16x16 renderbuffer - the
// exact sub-case KHR-GL43.copy_image.exceeding_boundaries starts with, GL_RENDERBUFFER being first
// in its target list - was accepted outright.
TEST_F(TextureTest, CopyImageSubDataBoundsARenderbufferRegion) {
const ScopedTextureBackendFunctionsOverride backendGuard;
MG_Backend::gBackendFunctionsTable.GL.CopyImageSubData = RecordCopyImageSubData;
g_copyImageSubDataCall = {};
GLuint texture = 0;
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_2D, 1, &texture);
MG_Impl::GLImpl::TextureStorage2D(texture, 1, GL_RGBA8, 16, 16);
GLuint renderbuffer = 0;
MG_Impl::GLImpl::CreateRenderbuffers(1, &renderbuffer);
MG_Impl::GLImpl::NamedRenderbufferStorage(renderbuffer, GL_RGBA8, 16, 16);
ASSERT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
MG_Impl::GLImpl::CopyImageSubData(renderbuffer, GL_RENDERBUFFER, 0, 0, 12, 0, texture, GL_TEXTURE_2D, 0, 0, 0, 0,
4, 4, 1);
EXPECT_TRUE(g_copyImageSubDataCall.Called);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(renderbuffer, GL_RENDERBUFFER, 0, 0, 14, 0, texture, GL_TEXTURE_2D, 0, 0, 0, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
// ...and as the destination, where the same renderbuffer has the same one image.
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(texture, GL_TEXTURE_2D, 0, 0, 0, 0, renderbuffer, GL_RENDERBUFFER, 0, 14, 0, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
// A renderbuffer has exactly one slice, so any z at all is out of range.
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(renderbuffer, GL_RENDERBUFFER, 0, 0, 0, 1, texture, GL_TEXTURE_2D, 0, 0, 0, 0,
4, 4, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
}
// The z axis was structurally unbounded - srcZ/dstZ did not even reach the validator - so a layer
// range running off the end of an array reached the backend as an out-of-range image subresource.
TEST_F(TextureTest, CopyImageSubDataBoundsTheLayerRangeOfAnArray) {
const ScopedTextureBackendFunctionsOverride backendGuard;
MG_Backend::gBackendFunctionsTable.GL.CopyImageSubData = RecordCopyImageSubData;
g_copyImageSubDataCall = {};
GLuint srcTexture = 0;
GLuint dstTexture = 0;
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_2D_ARRAY, 1, &srcTexture);
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_2D_ARRAY, 1, &dstTexture);
MG_Impl::GLImpl::TextureStorage3D(srcTexture, 1, GL_RGBA8, 8, 8, 12);
MG_Impl::GLImpl::TextureStorage3D(dstTexture, 1, GL_RGBA8, 8, 8, 12);
ASSERT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
// Layers 5..11 of a 12-layer array: the last one the range may reach.
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D_ARRAY, 0, 0, 0, 5, dstTexture, GL_TEXTURE_2D_ARRAY,
0, 0, 0, 5, 4, 4, 7);
EXPECT_TRUE(g_copyImageSubDataCall.Called);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D_ARRAY, 0, 0, 0, 6, dstTexture, GL_TEXTURE_2D_ARRAY,
0, 0, 0, 0, 4, 4, 7);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_2D_ARRAY, 0, 0, 0, 0, dstTexture, GL_TEXTURE_2D_ARRAY,
0, 0, 0, 6, 4, 4, 7);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
}
// The convention the bounds check has to get right, and the one that would silently reject legal
// copies if it did not: on a CUBE MAP the z axis selects among the six faces, which this frontend
// keeps as six separate one-slice upload targets - so the level's own extent reports depth 1 and a
// bound taken from it would refuse every whole-cube copy.
TEST_F(TextureTest, CopyImageSubDataCountsCubeMapFacesOnTheZAxis) {
const ScopedTextureBackendFunctionsOverride backendGuard;
MG_Backend::gBackendFunctionsTable.GL.CopyImageSubData = RecordCopyImageSubData;
g_copyImageSubDataCall = {};
GLuint srcTexture = 0;
GLuint dstTexture = 0;
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_CUBE_MAP, 1, &srcTexture);
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_CUBE_MAP, 1, &dstTexture);
MG_Impl::GLImpl::TextureStorage2D(srcTexture, 1, GL_RGBA8, 8, 8);
MG_Impl::GLImpl::TextureStorage2D(dstTexture, 1, GL_RGBA8, 8, 8);
DrainPendingGlErrors();
const auto srcObject = MG_State::pGLContext->GetTextureObject(srcTexture);
const auto dstObject = MG_State::pGLContext->GetTextureObject(dstTexture);
ASSERT_NE(srcObject, nullptr);
ASSERT_NE(dstObject, nullptr);
if (!srcObject->IsComplete() || !dstObject->IsComplete()) {
GTEST_SKIP() << "this context could not give the cube maps storage";
}
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_CUBE_MAP, 0, 0, 0, 0, dstTexture, GL_TEXTURE_CUBE_MAP,
0, 0, 0, 0, 8, 8, 6);
EXPECT_TRUE(g_copyImageSubDataCall.Called);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
// A seventh face does not exist.
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_CUBE_MAP, 0, 0, 0, 1, dstTexture, GL_TEXTURE_CUBE_MAP,
0, 0, 0, 0, 8, 8, 6);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
}
// The other axis convention: GL puts a 1D ARRAY's layers on y for this entry point (srcY is the
// first layer, srcHeight the layer count), which is also where this frontend keeps them - so the
// level extent answers directly and z stays a single slice.
TEST_F(TextureTest, CopyImageSubDataBoundsA1DArraysLayersOnTheYAxis) {
const ScopedTextureBackendFunctionsOverride backendGuard;
MG_Backend::gBackendFunctionsTable.GL.CopyImageSubData = RecordCopyImageSubData;
g_copyImageSubDataCall = {};
GLuint srcTexture = 0;
GLuint dstTexture = 0;
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_1D_ARRAY, 1, &srcTexture);
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_1D_ARRAY, 1, &dstTexture);
MG_Impl::GLImpl::TextureStorage2D(srcTexture, 1, GL_RGBA8, 16, 8);
MG_Impl::GLImpl::TextureStorage2D(dstTexture, 1, GL_RGBA8, 16, 8);
DrainPendingGlErrors();
const auto srcObject = MG_State::pGLContext->GetTextureObject(srcTexture);
const auto dstObject = MG_State::pGLContext->GetTextureObject(dstTexture);
ASSERT_NE(srcObject, nullptr);
ASSERT_NE(dstObject, nullptr);
if (!srcObject->IsComplete() || !dstObject->IsComplete()) {
GTEST_SKIP() << "this context could not give the 1D arrays storage";
}
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_1D_ARRAY, 0, 0, 3, 0, dstTexture, GL_TEXTURE_1D_ARRAY,
0, 0, 3, 0, 4, 5, 1);
EXPECT_TRUE(g_copyImageSubDataCall.Called);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
g_copyImageSubDataCall = {};
MG_Impl::GLImpl::CopyImageSubData(srcTexture, GL_TEXTURE_1D_ARRAY, 0, 0, 4, 0, dstTexture, GL_TEXTURE_1D_ARRAY,
0, 0, 0, 0, 4, 5, 1);
EXPECT_FALSE(g_copyImageSubDataCall.Called);
ExpectSingleGlError(GL_INVALID_VALUE);
}
// A 16-byte RGTC2 block and a 16-byte RGBA32UI texel are in the same size class, so GL 4.6 core
// 18.3.2 requires this copy to succeed. It did not for an ARRAY source: glTexImage3D recorded no
// specific-compressed-format tag, so the level was measured as the 2-byte RG8 storage RGTC2
+177 -23
View File
@@ -831,6 +831,35 @@ namespace MobileGL::MG_Util::BackendLoader {
return includesBase;
}
// GL 4.6 table 23.65 admits exactly four answers for GL_LAYER_PROVOKING_VERTEX and
// GL_VIEWPORT_INDEX_PROVOKING_VERTEX. Anything else means the driver wrote something MobileGL
// cannot forward as a convention, and GL_UNDEFINED_VERTEX - a legal answer, not a placeholder
// - is the accurate thing to say about it.
static GLenum NormalizeProvokingVertexConvention(GLint driverValue) {
switch (static_cast<GLenum>(driverValue)) {
case GL_FIRST_VERTEX_CONVENTION:
case GL_LAST_VERTEX_CONVENTION:
case GL_PROVOKING_VERTEX:
case GL_UNDEFINED_VERTEX:
return static_cast<GLenum>(driverValue);
default:
return GL_UNDEFINED_VERTEX;
}
}
static const char* ProvokingVertexConventionName(GLenum convention) {
switch (convention) {
case GL_FIRST_VERTEX_CONVENTION:
return "GL_FIRST_VERTEX_CONVENTION";
case GL_LAST_VERTEX_CONVENTION:
return "GL_LAST_VERTEX_CONVENTION";
case GL_PROVOKING_VERTEX:
return "GL_PROVOKING_VERTEX";
default:
return "GL_UNDEFINED_VERTEX";
}
}
Bool FillInGLESCapabilities(MG_External::GLESCapabilities& caps, const MG_External::GLESFunctionsTable& glesFuncs) {
if (!glesFuncs.glGetString || !glesFuncs.glGetIntegerv) {
MGLOG_E("Required GLES functions are not loaded, cannot query capabilities");
@@ -1063,8 +1092,21 @@ namespace MobileGL::MG_Util::BackendLoader {
GLint maxComputeImageUniforms = 8;
GLint maxDrawBuffers = 8;
GLint maxColorAttachments = 8;
GLint maxClipDistances = 8;
// Zero is a legal answer, not a placeholder. GL_MAX_CLIP_DISTANCES exists in ES only as
// GL_MAX_CLIP_DISTANCES_EXT under GL_EXT_clip_cull_distance, so on a driver without that
// extension there is nowhere to put a clip distance at all: SPIRV-Cross emits
// gl_ClipDistance behind an `#extension ... : require` the ESSL compiler rejects, and
// DirectGLES has no state to forward the per-distance enables into (see the gate in
// DirectGLES::SyncRenderState). Starting at 8 meant a probe that could never run left an
// optimistic 8 behind, so the frontend promised eight clip planes and every draw with a
// clipping program silently rendered nothing. The guarded probe below only ever widens it.
GLint maxClipDistances = 0;
GLint maxViewports = 16;
// GL_UNDEFINED_VERTEX is what stands when the probes below cannot run, and it is a legal
// answer rather than a placeholder: with neither geometry shaders nor a viewport array
// there is no layered or multi-viewport draw for a convention to describe.
GLenum layerProvokingVertex = GL_UNDEFINED_VERTEX;
GLenum viewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
GLfloat minFragmentInterpolationOffset = -0.5f;
GLfloat maxFragmentInterpolationOffset = 0.4375f;
GLint fragmentInterpolationOffsetBits = 4;
@@ -1074,11 +1116,39 @@ namespace MobileGL::MG_Util::BackendLoader {
GLint maxProgramTextureGatherOffset = 7;
GLint maxPatchVertices = 32;
GLint maxTessGenLevel = 64;
// Function-scope, and used by every probe group below rather than redeclared inside each
// one. Returns whether anything was drained, which is what lets a group tell "the driver
// answered" from "the driver rejected the pname and left my local alone".
const auto drainErrors = [&glesFuncs]() {
Bool hadError = false;
if (glesFuncs.glGetError) {
while (glesFuncs.glGetError() != GL_NO_ERROR) hadError = true;
}
return hadError;
};
// THE GENERATOR OF THIS WHOLE BUG FAMILY, closed here. A bare glGetIntegerv/glGetFloatv
// of a pname the driver does not have does two damaging things at once: it leaves the
// local at whatever the declaration initialised it to - an optimistic number the frontend
// then advertises as a capability - and it leaves a GL_INVALID_ENUM in the queue where
// the next unrelated probe's caller, or the application's first glGetError, gets blamed
// for it. The per-stage storage block, fragment interpolation and buffer texture probes
// below already drain and fall back; this unconditional run did neither, which is how
// GL_MAX_CLIP_DISTANCES came to be advertised as 8 on a driver with no clip distances at
// all. Every pname here that is not ES core is now either gated on the capability that
// makes it exist or floored at the value a rejected probe would have left, and the whole
// run is bracketed by a drain.
drainErrors();
glesFuncs.glGetFloatv(GL_ALIASED_LINE_WIDTH_RANGE, aliasedLineWidthRange);
// GL_SMOOTH_LINE_WIDTH_RANGE / GL_SMOOTH_LINE_WIDTH_GRANULARITY (0x0B22 / 0x0B23) are
// desktop-only - ES has never had an antialiased line width query - so on a real GLES
// driver these two raise GL_INVALID_ENUM. Kept as probes rather than dropped because the
// ANGLE and desktop-GL hosts MobileGL also runs on do answer them; the initialisers are
// the GL 4.6 table 23.55 minimum of [1, 1], which is both the honest answer for a driver
// that cannot say and what an untouched out-param already holds.
glesFuncs.glGetFloatv(GL_SMOOTH_LINE_WIDTH_RANGE, smoothLineWidthRange);
glesFuncs.glGetFloatv(GL_SMOOTH_LINE_WIDTH_GRANULARITY, &smoothLineWidthGranularity);
glesFuncs.glGetFloatv(GL_ALIASED_POINT_SIZE_RANGE, aliasedPointSizeRange);
glesFuncs.glGetFloatv(GL_VIEWPORT_BOUNDS_RANGE, viewportBoundsRange);
glesFuncs.glGetIntegerv(GL_MAX_3D_TEXTURE_SIZE, &max3DTextureSize);
glesFuncs.glGetIntegerv(GL_MAX_ARRAY_TEXTURE_LAYERS, &maxArrayTextureLayers);
glesFuncs.glGetIntegerv(GL_MAX_CUBE_MAP_TEXTURE_SIZE, &maxCubeMapTextureSize);
@@ -1102,8 +1172,25 @@ namespace MobileGL::MG_Util::BackendLoader {
// single test case. 1 is a spec-legal value (the minimum required), so cap
// to what is actually implemented instead of forwarding the raw driver limit.
maxSampleMaskWords = std::min(maxSampleMaskWords, 1);
// The multisample ceilings above are ES 3.1 state apart from GL_MAX_SAMPLES, which is ES
// 3.0, so a 3.0 context rejects five of the six and leaves whatever the out-param held.
// One sample is what a rejected probe leaves behind and is also the smallest legal
// answer, so clamp rather than trust: a zero reaching GL_Getter would have the frontend
// reject the very sample count it just advertised (see GetAdvertisedMaxSamples).
maxColorTextureSamples = std::max(maxColorTextureSamples, 1);
maxDepthTextureSamples = std::max(maxDepthTextureSamples, 1);
maxFramebufferSamples = std::max(maxFramebufferSamples, 1);
maxIntegerSamples = std::max(maxIntegerSamples, 1);
maxSamples = std::max(maxSamples, 1);
maxSampleMaskWords = std::max(maxSampleMaskWords, 1);
// ES 3.2 core, or EXT_tessellation_shader on 3.1. Probed rather than version-gated so a
// 3.1 driver that HAS the extension still gets to answer; the clamp below is what makes a
// rejected query safe, since GL 4.6 table 23.66 and ES 3.2 table 21.45 set the same
// minimums the initialisers carry and neither API permits less.
glesFuncs.glGetIntegerv(GL_MAX_PATCH_VERTICES, &maxPatchVertices);
glesFuncs.glGetIntegerv(GL_MAX_TESS_GEN_LEVEL, &maxTessGenLevel);
maxPatchVertices = std::max(maxPatchVertices, 32);
maxTessGenLevel = std::max(maxTessGenLevel, 64);
glesFuncs.glGetIntegerv(GL_MIN_PROGRAM_TEXTURE_GATHER_OFFSET, &minProgramTextureGatherOffset);
glesFuncs.glGetIntegerv(GL_MAX_PROGRAM_TEXTURE_GATHER_OFFSET, &maxProgramTextureGatherOffset);
// A driver that leaves the probe untouched (pre-ES 3.1, or an ignored enum) must not
@@ -1140,6 +1227,13 @@ namespace MobileGL::MG_Util::BackendLoader {
(caps.GLESVersion.Major == 3 && caps.GLESVersion.Minor >= 2)) {
glesFuncs.glGetIntegerv(GL_MAX_GEOMETRY_IMAGE_UNIFORMS, &maxGeometryImageUniforms);
}
// Closes the bracket opened before the run: every local above now holds either the
// driver's answer or a floor, and nothing this function asked for is left in the error
// queue for a later probe - or the application - to be blamed for.
if (drainErrors()) {
MGLOG_W("One or more capability queries were rejected by this driver; the affected "
"limits keep MobileGL's spec-minimum floors");
}
// Per-stage storage-block counts. Deliberately NOT batched with the unconditional probes
// above, for the reason GL_MAX_TEXTURE_BUFFER_SIZE is not: the vertex and fragment pnames
// are ES 3.1, but the tessellation and geometry ones only exist from ES 3.2 on (or under
@@ -1152,14 +1246,6 @@ namespace MobileGL::MG_Util::BackendLoader {
// stages is 0. That is the honest answer: DirectGLES emits ESSL 3.10 on an ES 3.1 context,
// where those stages do not exist at all.
{
const auto drainErrors = [&glesFuncs]() {
Bool hadError = false;
if (glesFuncs.glGetError) {
while (glesFuncs.glGetError() != GL_NO_ERROR) hadError = true;
}
return hadError;
};
// Isolate from errors raised by the preceding probes so the drain below reports on
// these queries only.
drainErrors();
@@ -1195,19 +1281,67 @@ namespace MobileGL::MG_Util::BackendLoader {
}
glesFuncs.glGetIntegerv(GL_MAX_DRAW_BUFFERS, &maxDrawBuffers);
glesFuncs.glGetIntegerv(GL_MAX_COLOR_ATTACHMENTS, &maxColorAttachments);
glesFuncs.glGetIntegerv(GL_MAX_CLIP_DISTANCES, &maxClipDistances);
glesFuncs.glGetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
// GL_MAX_CLIP_DISTANCES is 0x0D32, which ES only ever spells GL_MAX_CLIP_DISTANCES_EXT and
// only ever has under GL_EXT_clip_cull_distance. The extension was already resolved into
// caps.SupportsClipDistance a few hundred lines above and is the same flag DirectGLES
// gates the CLIP_DISTANCEi enable forwarding on, so ask the driver only where the pname
// exists; everywhere else the honest 0 stands and no GL_INVALID_ENUM is left behind for an
// unrelated query - or the application's first glGetError - to trip over.
if (caps.SupportsClipDistance) {
drainErrors();
glesFuncs.glGetIntegerv(GL_MAX_CLIP_DISTANCES, &maxClipDistances);
if (drainErrors()) {
MGLOG_W("GL_EXT_clip_cull_distance is advertised but GL_MAX_CLIP_DISTANCES was "
"rejected; reporting no clip distances");
maxClipDistances = 0;
}
}
glesFuncs.glGetIntegerv(GL_MAX_VIEWPORT_DIMS, maxViewportDims);
glesFuncs.glGetIntegerv(GL_VIEWPORT_SUBPIXEL_BITS, &viewportSubpixelBits);
// GL_LAYER_PROVOKING_VERTEX is ES 3.2 core (it arrives with geometry shaders, which is
// what gl_Layer needs). Ask the driver where the pname exists rather than asserting a
// convention: it is a statement about which vertex of a primitive supplies gl_Layer, and
// MobileGL forwards the geometry stage to the driver rather than implementing the
// selection itself, so the driver's answer IS MobileGL's answer. Below ES 3.2 there are
// no layered draws to have a convention for and GL_UNDEFINED_VERTEX stands, which GL 4.6
// table 23.65 explicitly permits.
if (esAtLeast32) {
GLint driverLayerConvention = static_cast<GLint>(GL_UNDEFINED_VERTEX);
drainErrors();
glesFuncs.glGetIntegerv(GL_LAYER_PROVOKING_VERTEX, &driverLayerConvention);
if (!drainErrors()) {
layerProvokingVertex = NormalizeProvokingVertexConvention(driverLayerConvention);
}
}
// GL_MAX_VIEWPORTS (0x825B), GL_VIEWPORT_SUBPIXEL_BITS (0x825C) and GL_VIEWPORT_BOUNDS_RANGE
// (0x825D) all arrive with GL_OES_viewport_array and exist nowhere in ES core, so on the
// drivers DirectGLES actually runs on all three raise GL_INVALID_ENUM. The values MobileGL
// advertises do not change by asking: GL_Getter answers GL_MAX_VIEWPORTS from the frontend
// state width (indexed viewport entry points validate against RenderStateParameters::
// MAX_VIEWPORTS, so a device answer of 1 would reject indices the state can legitimately
// hold), floors GL_SUBPIXEL_BITS at its own 4, and the bounds range is clamped to the core
// minimum below. What changes is that the errors stop being manufactured.
if (caps.SupportsViewportArray) {
GLint driverViewportIndexConvention = static_cast<GLint>(GL_UNDEFINED_VERTEX);
drainErrors();
glesFuncs.glGetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
glesFuncs.glGetIntegerv(GL_VIEWPORT_SUBPIXEL_BITS, &viewportSubpixelBits);
glesFuncs.glGetIntegerv(GL_VIEWPORT_INDEX_PROVOKING_VERTEX, &driverViewportIndexConvention);
if (glesFuncs.glGetFloatv) {
glesFuncs.glGetFloatv(GL_VIEWPORT_BOUNDS_RANGE, viewportBoundsRange);
}
if (drainErrors()) {
MGLOG_W("GL_OES_viewport_array is advertised but its viewport limit queries were "
"rejected; keeping the OpenGL core minimums");
maxViewports = 16;
viewportSubpixelBits = 0;
viewportBoundsRange[0] = -32768.0f;
viewportBoundsRange[1] = 32767.0f;
} else {
viewportIndexProvokingVertex =
NormalizeProvokingVertexConvention(driverViewportIndexConvention);
}
}
if (caps.SupportsShaderMultisampleInterpolation && glesFuncs.glGetFloatv) {
const auto drainErrors = [&glesFuncs]() {
Bool hadError = false;
if (glesFuncs.glGetError) {
while (glesFuncs.glGetError() != GL_NO_ERROR) hadError = true;
}
return hadError;
};
// Isolate these optional queries from errors raised by preceding capability
// probes, then consume any query error so initialization never leaks it into
// the application's first glGetError call.
@@ -1368,8 +1502,12 @@ namespace MobileGL::MG_Util::BackendLoader {
caps.MaxComputeImageUniforms = maxComputeImageUniforms;
caps.MaxDrawBuffers = maxDrawBuffers;
caps.MaxColorAttachments = maxColorAttachments;
caps.MaxClipDistances = maxClipDistances;
// A driver is free to write nonsense into an out-param it then rejects, and without the
// extension the probe above never ran at all - so the flag, not the local, decides.
caps.MaxClipDistances = caps.SupportsClipDistance ? std::max(maxClipDistances, 0) : 0;
caps.MaxViewports = maxViewports;
caps.LayerProvokingVertex = layerProvokingVertex;
caps.ViewportIndexProvokingVertex = viewportIndexProvokingVertex;
caps.MaxViewportWidth = maxViewportDims[0];
caps.MaxViewportHeight = maxViewportDims[1];
// Only ever WIDER than the core minimum: a driver that answered the query is allowed to
@@ -1450,12 +1588,20 @@ namespace MobileGL::MG_Util::BackendLoader {
MGLOG_I(" GL_MAX_COMPUTE_IMAGE_UNIFORMS: %d", caps.MaxComputeImageUniforms);
MGLOG_I(" GL_MAX_DRAW_BUFFERS: %d", caps.MaxDrawBuffers);
MGLOG_I(" GL_MAX_COLOR_ATTACHMENTS: %d", caps.MaxColorAttachments);
MGLOG_I(" GL_MAX_CLIP_DISTANCES: %d", caps.MaxClipDistances);
// Worth spelling the reason out for the same reason the per-stage storage block counts
// are: a zero here is what stops an application's gl_ClipDistance from ever clipping, and
// reading it back from an artifact is the difference between "MobileGL dropped my draw"
// and "this driver has no clip distances".
MGLOG_I(" GL_MAX_CLIP_DISTANCES: %d%s", caps.MaxClipDistances,
caps.SupportsClipDistance ? "" : " (no GL_EXT_clip_cull_distance on this driver)");
MGLOG_I(" GL_MAX_VIEWPORTS: %d", caps.MaxViewports);
MGLOG_I(" GL_MAX_VIEWPORT_DIMS: [%d, %d]", caps.MaxViewportWidth, caps.MaxViewportHeight);
MGLOG_I(" GL_VIEWPORT_BOUNDS_RANGE: [%.3f, %.3f]", caps.ViewportBoundsRangeMin,
caps.ViewportBoundsRangeMax);
MGLOG_I(" GL_VIEWPORT_SUBPIXEL_BITS: %d", caps.ViewportSubpixelBits);
MGLOG_I(" GL_LAYER_PROVOKING_VERTEX: %s", ProvokingVertexConventionName(caps.LayerProvokingVertex));
MGLOG_I(" GL_VIEWPORT_INDEX_PROVOKING_VERTEX: %s",
ProvokingVertexConventionName(caps.ViewportIndexProvokingVertex));
caps.IndirectDrawInstanceIdIncludesBaseInstance =
ProbeIndirectInstanceIdIncludesBaseInstance(caps, glesFuncs);
@@ -1479,6 +1625,14 @@ namespace MobileGL::MG_Util::BackendLoader {
caps.AvoidSamplerMipmapMinFilter ? "true" : "false");
MGLOG_I(" Avoid explicit LOD bias: %s", caps.AvoidExplicitLodBias ? "true" : "false");
// Last line of defence. Capability init is the very first thing that touches the driver,
// so anything it leaves in the error queue surfaces at the APPLICATION's first
// glGetError and gets attributed to whatever call the app happened to make. Every group
// above drains its own, but a probe added later must not be able to reintroduce the leak.
if (drainErrors()) {
MGLOG_W("Capability initialization left a GL error behind; it has been consumed so it "
"cannot surface at the application's first glGetError");
}
return true;
}
} // namespace MobileGL::MG_Util::BackendLoader
@@ -1275,8 +1275,16 @@ namespace MobileGL {
Int MaxComputeImageUniforms = 8;
Int MaxDrawBuffers = 8;
Int MaxColorAttachments = 8;
Int MaxClipDistances = 8;
// Zero is a legal answer, not a placeholder: ES reaches clip distances only through
// GL_EXT_clip_cull_distance, so a driver without it has none. See the guarded probe
// in FillInGLESCapabilities.
Int MaxClipDistances = 0;
Int MaxViewports = 16;
// GL_LAYER_PROVOKING_VERTEX (ES 3.2 core) and GL_VIEWPORT_INDEX_PROVOKING_VERTEX
// (GL_OES_viewport_array). GL_UNDEFINED_VERTEX is a legal answer for both and is what
// a driver that has neither is honestly saying.
GLenum LayerProvokingVertex = GL_UNDEFINED_VERTEX;
GLenum ViewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
Int MaxViewportWidth = 16384;
Int MaxViewportHeight = 16384;
Float ViewportBoundsRangeMin = 0.0f;
@@ -237,6 +237,7 @@ namespace MobileGL::MG_Util::BackendLoader {
supportedFeatures.vertexPipelineStoresAndAtomics == VK_TRUE;
caps.SupportsFragmentStoresAndAtomics = supportedFeatures.fragmentStoresAndAtomics == VK_TRUE;
caps.SupportsGeometryShader = supportedFeatures.geometryShader == VK_TRUE;
caps.SupportsShaderClipDistance = supportedFeatures.shaderClipDistance == VK_TRUE;
caps.MaxShaderStorageBlockSize = static_cast<SizeT>(p.limits.maxStorageBufferRange);
const Bool supportsShaderSubgroup = vk.vkGetPhysicalDeviceProperties2 &&
HasUsableShaderSubgroupSupport(subgroupProps);
@@ -331,6 +332,7 @@ namespace MobileGL::MG_Util::BackendLoader {
caps.SupportsVertexPipelineStoresAndAtomics = false;
caps.SupportsFragmentStoresAndAtomics = false;
caps.SupportsGeometryShader = false;
caps.SupportsShaderClipDistance = false;
caps.MaxShaderStorageBlockSize = static_cast<SizeT>(properties.limits.maxStorageBufferRange);
caps.SupportsShaderSubgroup = false;
caps.SubgroupSize = 0;
@@ -96,6 +96,12 @@ namespace MobileGL {
Bool SupportsVertexPipelineStoresAndAtomics = false;
Bool SupportsFragmentStoresAndAtomics = false;
Bool SupportsGeometryShader = false;
// VkPhysicalDeviceFeatures::shaderClipDistance. maxClipDistances is a LIMIT and is
// reported whatever the feature says, so the limit alone does not mean a module may
// declare ClipDistance - VulkanRenderer enables the feature only where the physical
// device has it, and without it a shader writing gl_ClipDistance is invalid. Very
// widely supported, hence read from the device features and never assumed false.
Bool SupportsShaderClipDistance = false;
SizeT MaxShaderStorageBlockSize = 128 * 1024 * 1024;
Bool SupportsShaderSubgroup = false;
Uint32 SubgroupSize = 0;
+3 -2
View File
@@ -558,8 +558,9 @@ namespace MobileGL::MG_Util::SelfTest {
} else {
builder.Warn("GL_EXT_render_snorm",
"not supported; signed-normalized formats are texture-only, so every SNORM "
"render target is stored as a float (GL_RGBA8_SNORM/GL_RGB8_SNORM -> "
"GL_RGBA16F) and its fragment outputs are clamped to [-1,1] in software");
"render target is stored as a float (8-bit -> *16F, 16-bit -> *32F, which "
"is the narrowest float that still holds a 16-bit SNORM channel exactly) "
"and its fragment outputs are clamped to [-1,1] in software");
}
// FAIL, not WARN: ES 3.x core makes every float format texture-only, and every Iris
// shaderpack renders into at least GL_R11F_G11F_B10F (Complementary's colortex0, BSL's
@@ -62,6 +62,11 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
// gl_MaxComputeTextureImageUnits built-in constant, so a compute module that reads
// that constant generates DIFFERENT SPIR-V under two backends that disagree on it.
HashValue(state, env.params.MaxComputeTextureImageUnits);
// Added when wave4 (4fc3531d) made this one env-derived, and the same class again:
// glslang REJECTS gl_ClipDistance[i] for i >= maxClipDistances at parse (ParseHelper)
// and expands gl_MaxClipDistances from the same number, so it decides both whether a
// shader compiles at all and what a module that reads the constant generates.
HashValue(state, env.params.MaxClipDistances);
// The compute work-group limits, likewise added by wave3 (cb155c5b). They used to be
// hardcoded maxima in BuildTBuiltInResource, and the L1 key comment said in so many
// words that the day they became backend-derived they would have to move in here -
@@ -84,9 +84,11 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
// * the DynamicBackendParameters fields BuildTBuiltInResource copies into
// TBuiltInResource - MaxImageUnits, MaxDrawBuffers, MaxVertexImageUniforms,
// MaxGeometryImageUniforms, MaxFragmentImageUniforms, MaxComputeImageUniforms,
// MaxCombinedImageUniforms, MaxComputeTextureImageUnits. glslang enforces those at
// parse, so they decide whether a shader compiles at all and can change the link
// result.
// MaxCombinedImageUniforms, MaxComputeTextureImageUnits, MaxClipDistances. glslang
// enforces those at parse, so they decide whether a shader compiles at all and can
// change the link result. MaxClipDistances moved in at the wave4 merge (4fc3531d),
// the third time in three waves that a hardcoded TBuiltInResource field became
// env-derived - assume the next wave does it again and re-audit.
// * maxComputeWorkGroupSize and maxComputeWorkGroupCount, all three components each.
// These moved IN at the dev merge that brought wave3's cb155c5b, which made
// BuildTBuiltInResource read them from the env instead of hardcoding a permissive
@@ -23,6 +23,7 @@
#include "SpirvPasses/LowerViewportIndexPass.h"
#include "SpirvPasses/PackDoubleVertexInputsPass.h"
#include "SpirvPasses/FlattenXfbInterfaceBlocksPass.h"
#include "SpirvPasses/UniquifyIoBlockNamesPass.h"
#include "SpirvPasses/SplitArrayVertexInputsPass.h"
#include "SpirvPasses/RebaseInstanceIndexPass.h"
#include "SpirvPasses/ZeroBaseVertexPass.h"
@@ -82,7 +83,6 @@ namespace MobileGL {
Resources.maxFragmentInputVectors = 15;
Resources.minProgramTexelOffset = -8;
Resources.maxProgramTexelOffset = 7;
Resources.maxClipDistances = 8;
Resources.maxComputeUniformComponents = MAX_COMPUTE_UNIFORM_COMPONENTS;
Resources.maxComputeTextureImageUnits = 16;
Resources.maxComputeImageUniforms = 8;
@@ -181,6 +181,14 @@ namespace MobileGL {
Resources.maxComputeImageUniforms = dynamicParameters.MaxComputeImageUniforms;
Resources.maxCombinedImageUniforms = dynamicParameters.MaxCombinedImageUniforms;
Resources.maxComputeTextureImageUnits = dynamicParameters.MaxComputeTextureImageUnits;
// Load-bearing, not cosmetic. glslang rejects gl_ClipDistance[i] for
// i >= maxClipDistances (ParseHelper.cpp) and expands gl_MaxClipDistances from the
// same number, so tracking the backend limit is what turns "the program links,
// the backend's shader compile fails somewhere the frontend never surfaces, and
// the draw renders nothing" into an honest glCompileShader error with a log. It is
// also what makes glGetIntegerv(GL_MAX_CLIP_DISTANCES) and gl_MaxClipDistances
// agree, which KHR-GLxx.clip_distance.coverage compares directly.
Resources.maxClipDistances = dynamicParameters.MaxClipDistances;
// The compute work-group limits are the env's, not the backend parameters': they
// are the only ones that come from a REAL indexed driver query, which
@@ -782,6 +790,42 @@ namespace MobileGL {
outName);
}
void ShaderCompiler::ProbeIoBlockNamesForEssl(const Vector<Uint32>& binary,
std::set<String>& collidingBlockNames,
std::set<String>& declaredNames) {
if (binary.empty()) {
// Same reasoning as ModuleDeclaresBufferTextureSampler: a stage that produced
// no SPIR-V has no block names to report, and parsing it would push a
// spurious diagnostic through the message consumer.
return;
}
std::unique_ptr<spvtools::opt::IRContext> context = spvtools::BuildModule(
SPV_ENV_VULKAN_1_1, MakeSpirvMessageConsumer("ProbeIoBlockNamesForEssl"), binary.data(),
binary.size());
if (!context) {
// Unparseable here means unusable downstream too; let the ordinary transpile
// path produce the error rather than inventing a rename plan from it.
return;
}
UniquifyIoBlockNamesPass::ProbeIoBlockNames(context.get(), collidingBlockNames, declaredNames);
}
bool ShaderCompiler::UniquifyIoBlockNamesForEssl(const Vector<Uint32>& inputBinary,
const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames,
std::set<String>& renamedBlockNames,
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
if (inputBlockRenames.empty() && outputBlockRenames.empty()) return false;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(UniquifyIoBlockNamesPass::CreateUniquifyIoBlockNamesPass(
inputBlockRenames, outputBlockRenames, &renamedBlockNames));
return RunOptimizerChecked("UniquifyIoBlockNamesForEssl", optimizer, inputBinary,
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::PackDoubleVertexInputsForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
@@ -12,6 +12,7 @@
#include "glslang/TVarEntryInfo.h"
#include "glslang/TMglGlslIoResolver.h"
#include <map>
#include <set>
namespace MobileGL {
@@ -101,6 +102,29 @@ namespace MobileGL {
static bool RewriteXfbCaptureNameForFlattenedBlock(const String& captureName,
const std::set<String>& flattenedBlockNames,
String& outName);
// Adds to `collidingBlockNames` every inter-stage interface block this stage
// declares in BOTH directions at once (`in FOO {...}; out FOO {...}`, which
// desktop GLSL allows because its input and output block namespaces are
// separate), and to `declaredNames` every name the module spells. The gate for
// UniquifyIoBlockNamesForEssl below, and the source of the name set a
// replacement has to avoid. Reads the module; never rewrites it.
static void ProbeIoBlockNamesForEssl(const Vector<Uint32>& binary,
std::set<String>& collidingBlockNames,
std::set<String>& declaredNames);
// Renames inter-stage interface BLOCK types so the collision the probe above
// found gets one spelling per producing stage. `inputBlockRenames` applies to
// blocks this stage consumes and `outputBlockRenames` to blocks it produces,
// both planned program-wide by the caller so a producer and its consumer keep
// matching; `renamedBlockNames` reports the original names this stage actually
// rewrote. SPIRV-Cross re-emits two same-named blocks verbatim and the Mali ES
// driver then loses the output block's payload. Only for the DirectGLES
// transpile path. See UniquifyIoBlockNamesPass.
static bool UniquifyIoBlockNamesForEssl(const Vector<Uint32>& inputBinary,
const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames,
std::set<String>& renamedBlockNames,
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Drops RelaxedPrecision member decorations from uniform-block structs so
// SPIRV-Cross prints the same (highp) member precision in every stage; ES
// drivers reject cross-stage uniform blocks whose member precisions differ.
@@ -0,0 +1,231 @@
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/UniquifyIoBlockNamesPass.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include "UniquifyIoBlockNamesPass.h"
#include "spirv.hpp"
#include "source/opt/def_use_manager.h"
#include "source/opt/instruction.h"
#include "source/opt/ir_context.h"
#include "source/opt/module.h"
#include "source/util/make_unique.h"
#include "source/util/string_utils.h"
#include <unordered_map>
#include <unordered_set>
namespace MobileGL {
namespace MG_Util {
namespace ShaderTranspiler {
namespace {
using spvtools::opt::Instruction;
using spvtools::opt::IRContext;
// Which storage classes a block struct is reachable from. A struct seen in both
// directions inside ONE module cannot be renamed per direction (there is only
// one name to change), so it is skipped rather than guessed at.
constexpr Uint32 kSeenAsInput = 1u;
constexpr Uint32 kSeenAsOutput = 2u;
// Every struct type carrying the Block decoration, minus the ones with a builtin
// member (gl_PerVertex): those are named by the language, not by the shader, and
// renaming one would invent a block no driver knows.
std::unordered_set<uint32_t> CollectUserBlockStructIds(IRContext* irContext) {
std::unordered_set<uint32_t> blockStructIds;
std::unordered_set<uint32_t> builtinStructIds;
for (Instruction& annotation : irContext->module()->annotations()) {
if (annotation.opcode() == spv::Op::OpDecorate) {
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(1)) ==
spv::Decoration::Block) {
blockStructIds.insert(annotation.GetSingleWordInOperand(0));
}
} else if (annotation.opcode() == spv::Op::OpMemberDecorate) {
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(2)) ==
spv::Decoration::BuiltIn) {
builtinStructIds.insert(annotation.GetSingleWordInOperand(0));
}
}
}
for (uint32_t builtinStructId : builtinStructIds) {
blockStructIds.erase(builtinStructId);
}
return blockStructIds;
}
// The block struct an Input/Output variable declares, or 0 when the variable is
// not an interface block of the kind this pass renames. Tessellation and geometry
// interfaces are arrays of the block struct, so one array level is unwrapped -
// the same shape StripUboMemberRelaxedPrecisionPass unwraps for instance-arrayed
// uniform blocks.
uint32_t GetInterfaceBlockStructId(IRContext* irContext, Instruction& variable,
const std::unordered_set<uint32_t>& blockStructIds,
spv::StorageClass& outStorageClass) {
if (variable.opcode() != spv::Op::OpVariable) return 0;
const auto storageClass =
static_cast<spv::StorageClass>(variable.GetSingleWordInOperand(0));
if (storageClass != spv::StorageClass::Input &&
storageClass != spv::StorageClass::Output) {
return 0;
}
auto* defUseMgr = irContext->get_def_use_mgr();
Instruction* pointerType = defUseMgr->GetDef(variable.type_id());
if (pointerType == nullptr || pointerType->opcode() != spv::Op::OpTypePointer) return 0;
uint32_t pointeeId = pointerType->GetSingleWordInOperand(1);
Instruction* pointee = defUseMgr->GetDef(pointeeId);
while (pointee != nullptr && (pointee->opcode() == spv::Op::OpTypeArray ||
pointee->opcode() == spv::Op::OpTypeRuntimeArray)) {
pointeeId = pointee->GetSingleWordInOperand(0);
pointee = defUseMgr->GetDef(pointeeId);
}
if (pointee == nullptr || pointee->opcode() != spv::Op::OpTypeStruct) return 0;
if (blockStructIds.find(pointeeId) == blockStructIds.end()) return 0;
outStorageClass = storageClass;
return pointeeId;
}
String FindName(IRContext* irContext, uint32_t id) {
for (Instruction& debugInst : irContext->debugs2()) {
if (debugInst.opcode() != spv::Op::OpName) continue;
if (debugInst.GetSingleWordInOperand(0) != id) continue;
return debugInst.GetInOperand(1).AsString();
}
return String();
}
// Replaces an EXISTING OpName only. A block struct with no name of its own is
// one SPIRV-Cross would spell from a fallback, which the consuming stage would
// not agree with anyway - leave it alone rather than invent a name for it.
Bool ReplaceExistingName(IRContext* irContext, uint32_t id, const String& newName) {
for (Instruction& debugInst : irContext->debugs2()) {
if (debugInst.opcode() != spv::Op::OpName) continue;
if (debugInst.GetSingleWordInOperand(0) != id) continue;
debugInst.SetInOperand(
1, spvtools::utils::MakeVector<spvtools::opt::Operand::OperandData>(newName));
return true;
}
return false;
}
// Not a real id: "this name reached two different struct types in the same
// direction", which is already an illegal shader (glslang refuses to reuse a
// block name inside one interface) and which no rename could repair - two
// structs would come out with one new name. Both the probe and the rewrite
// decline it.
constexpr uint32_t kAmbiguousStructId = 0xffffffffu;
// The module's interface blocks indexed the way both halves of this pass need
// them: by name within each direction, plus which directions each struct type
// is reached from.
struct IoBlockIndex {
std::map<String, uint32_t> inputStructByName;
std::map<String, uint32_t> outputStructByName;
std::unordered_map<uint32_t, Uint32> storageMaskByStructId;
};
IoBlockIndex IndexIoBlocks(IRContext* irContext,
const std::unordered_set<uint32_t>& blockStructIds) {
IoBlockIndex index;
for (Instruction& variable : irContext->module()->types_values()) {
spv::StorageClass storageClass = spv::StorageClass::Input;
const uint32_t structId =
GetInterfaceBlockStructId(irContext, variable, blockStructIds, storageClass);
if (structId == 0) continue;
const Bool isInput = storageClass == spv::StorageClass::Input;
index.storageMaskByStructId[structId] |= isInput ? kSeenAsInput : kSeenAsOutput;
const String blockName = FindName(irContext, structId);
if (blockName.empty()) continue;
std::map<String, uint32_t>& byName =
isInput ? index.inputStructByName : index.outputStructByName;
const auto inserted = byName.emplace(blockName, structId);
if (!inserted.second && inserted.first->second != structId) {
inserted.first->second = kAmbiguousStructId;
}
}
return index;
}
} // namespace
void UniquifyIoBlockNamesPass::ProbeIoBlockNames(spvtools::opt::IRContext* irContext,
std::set<String>& outCollidingBlockNames,
std::set<String>& outDeclaredNames) {
if (irContext == nullptr) return;
for (Instruction& debugInst : irContext->debugs2()) {
if (debugInst.opcode() != spv::Op::OpName) continue;
outDeclaredNames.insert(debugInst.GetInOperand(1).AsString());
}
const std::unordered_set<uint32_t> blockStructIds = CollectUserBlockStructIds(irContext);
if (blockStructIds.empty()) return;
const IoBlockIndex index = IndexIoBlocks(irContext, blockStructIds);
for (const auto& input : index.inputStructByName) {
const auto output = index.outputStructByName.find(input.first);
if (output == index.outputStructByName.end()) continue;
if (input.second == kAmbiguousStructId || output->second == kAmbiguousStructId) continue;
// Same struct type on both sides: there is one name to rename and two
// directions wanting different ones, so the collision cannot be repaired.
if (input.second == output->second) continue;
outCollidingBlockNames.insert(input.first);
}
}
spvtools::opt::Pass::Status UniquifyIoBlockNamesPass::Process() {
if (m_inputBlockRenames.empty() && m_outputBlockRenames.empty()) {
return Status::SuccessWithoutChange;
}
auto* irContext = context();
const std::unordered_set<uint32_t> blockStructIds = CollectUserBlockStructIds(irContext);
if (blockStructIds.empty()) return Status::SuccessWithoutChange;
// Indexed BEFORE anything is renamed, so every decline below is decided against
// the names the module arrived with rather than against a half-renamed one.
const IoBlockIndex index = IndexIoBlocks(irContext, blockStructIds);
Bool modified = false;
for (int direction = 0; direction < 2; ++direction) {
const Bool isInput = direction == 0;
const std::map<String, uint32_t>& byName =
isInput ? index.inputStructByName : index.outputStructByName;
const std::map<String, String>& renames =
isInput ? m_inputBlockRenames : m_outputBlockRenames;
const Uint32 wantedMask = isInput ? kSeenAsInput : kSeenAsOutput;
for (const auto& block : byName) {
if (block.second == kAmbiguousStructId) continue;
const auto rename = renames.find(block.first);
if (rename == renames.end()) continue;
if (rename->second.empty() || rename->second == block.first) continue;
// A struct type reached from BOTH directions carries one name for two
// interfaces, so renaming it for this direction would rename it for the
// other one too. Leave the module as it was.
const auto mask = index.storageMaskByStructId.find(block.second);
if (mask == index.storageMaskByStructId.end() || mask->second != wantedMask) continue;
if (!ReplaceExistingName(irContext, block.second, rename->second)) continue;
if (m_renamedBlockNames != nullptr) m_renamedBlockNames->insert(block.first);
modified = true;
}
}
return modified ? Status::SuccessWithChange : Status::SuccessWithoutChange;
}
spvtools::Optimizer::PassToken UniquifyIoBlockNamesPass::CreateUniquifyIoBlockNamesPass(
const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames, std::set<String>* renamedBlockNames) {
return spvtools::Optimizer::PassToken(MakeUnique<UniquifyIoBlockNamesPass>(
inputBlockRenames, outputBlockRenames, renamedBlockNames));
}
} // namespace ShaderTranspiler
} // namespace MG_Util
} // namespace MobileGL
@@ -0,0 +1,90 @@
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/UniquifyIoBlockNamesPass.h
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#pragma once
#include "source/opt/pass.h"
#include "spirv-tools/optimizer.hpp"
#include <Includes.h>
#include <map>
#include <set>
namespace MobileGL {
namespace MG_Util {
namespace ShaderTranspiler {
// Renames the STRUCT of an inter-stage interface block, so a block name a stage
// declares in both directions at once gets one spelling per producing stage.
//
// WHY. Desktop GLSL keeps SEPARATE name namespaces for input and output interface
// blocks, so a single stage may legally write
//
// in TCSOutputBlock { ... } input_block[];
// out TCSOutputBlock { ... } output_block;
//
// which is exactly what the tessellation evaluation stage of
// KHR-GL42/43.shading_language_420pack.length_of_vector_and_matrix_* and
// .qualifier_order_block_* does. glslang accepts it deliberately (ParseHelper
// errors only when the two share a storage qualifier) and SPIRV-Cross re-emits
// BOTH under the name TCSOutputBlock, because it too splits the namespace
// (block_input_names vs block_output_names). The generated ESSL 3.20 then declares
// two different blocks called TCSOutputBlock in one shader. Adreno's ES compiler
// keeps them apart; Mali's does not - the stage compiles, the program links, and
// the output block's payload never reaches the next stage, which is all 22 of
// that group's Mali failures and none of Adreno's or DirectVulkan's.
//
// WHAT. The rename is planned program-wide by the CALLER and keyed on the
// PRODUCING stage, so a producer and its consumer keep naming the same block:
// the tessellation control stage's `out TCSOutputBlock` and the evaluation
// stage's `in TCSOutputBlock` both become <name>_mgio<TCS>, while the evaluation
// stage's own `out TCSOutputBlock` and the geometry stage's `in TCSOutputBlock`
// both become <name>_mgio<TES>. Only the block TYPE name changes; instance names,
// member names, locations and every decoration are left exactly as they were, and
// ES matches inter-stage blocks by block name plus member sequence.
//
// DirectGLES only: DirectVulkan hands the module to the driver as SPIR-V, where
// the two blocks are distinct type ids and the debug names carry no meaning.
class UniquifyIoBlockNamesPass : public spvtools::opt::Pass {
public:
// `inputBlockRenames` applies to blocks this stage CONSUMES and
// `outputBlockRenames` to blocks it PRODUCES, both keyed by the block's
// current name. `renamedBlockNames` receives the ORIGINAL names this stage
// actually rewrote, so the caller can adopt the re-serialised module only
// when there was something to rewrite.
UniquifyIoBlockNamesPass(const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames,
std::set<String>* renamedBlockNames)
: m_inputBlockRenames(inputBlockRenames), m_outputBlockRenames(outputBlockRenames),
m_renamedBlockNames(renamedBlockNames) {}
const char* name() const override { return "mobilegl-uniquify-io-block-names"; }
Status Process() override;
// Reads a module WITHOUT rewriting it, for the caller's gate. Adds to
// `outCollidingBlockNames` every block name this module declares in BOTH Input
// and Output storage under two DIFFERENT struct types - the only shape the
// rename above can repair - and to `outDeclaredNames` every name the module
// spells, so the caller can pick a replacement that collides with none of them.
// Builtin blocks (gl_PerVertex and friends) are never reported.
static void ProbeIoBlockNames(spvtools::opt::IRContext* irContext,
std::set<String>& outCollidingBlockNames,
std::set<String>& outDeclaredNames);
static spvtools::Optimizer::PassToken CreateUniquifyIoBlockNamesPass(
const std::map<String, String>& inputBlockRenames,
const std::map<String, String>& outputBlockRenames,
std::set<String>* renamedBlockNames);
private:
std::map<String, String> m_inputBlockRenames;
std::map<String, String> m_outputBlockRenames;
std::set<String>* m_renamedBlockNames = nullptr;
};
} // namespace ShaderTranspiler
} // namespace MG_Util
} // namespace MobileGL
@@ -25,7 +25,8 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
//
// 2: L2 gained atomicCounterEsslBindingTop (wave3's atomic-counter block rebinding
// prints it into the emitted ESSL), and L1c was added.
constexpr Uint32 kKeyLayoutVersion = 2u;
// 3: L2 gained the two interface-block rename maps (wave4's UniquifyIoBlockNames).
constexpr Uint32 kKeyLayoutVersion = 3u;
// The repo's existing cache epoch (MG_Config::CacheVersion, the seed
// ProgramFactory::ComputeHash uses). Strictly redundant for an in-memory
@@ -92,6 +93,14 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
for (const String& value : values) Text(value);
}
void TranslationKeyBuilder::StringMap(const std::map<String, String>& map) {
Value(static_cast<Uint64>(map.size()));
for (const auto& [name, value] : map) {
Text(name);
Text(value);
}
}
void TranslationKeyBuilder::NameSet(const std::set<String>& names) {
Value(static_cast<Uint64>(names.size()));
for (const String& name : names) Text(name);
@@ -167,6 +176,9 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
static const UnorderedMap<String, Int> kEmptyBindings;
builder.NameMap(inputs.storageBlockBindingOverrides ? *inputs.storageBlockBindingOverrides
: kEmptyBindings);
static const std::map<String, String> kEmptyRenames;
builder.StringMap(inputs.inputBlockRenames ? *inputs.inputBlockRenames : kEmptyRenames);
builder.StringMap(inputs.outputBlockRenames ? *inputs.outputBlockRenames : kEmptyRenames);
static const Vector<Uint32> kEmptyWords;
builder.Words(inputs.spirv ? *inputs.spirv : kEmptyWords);
return MakeTranslationCacheKey(builder);
@@ -10,6 +10,7 @@
#include <Includes.h>
#include <list>
#include <map>
#include <mutex>
#include <set>
@@ -130,6 +131,9 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
// std::set is already ordered, but it gets the same length prefix.
void NameSet(const std::set<String>& names);
// std::map is ordered too, so it needs no sort - but both halves are TEXT, so each
// gets its own length prefix and the pair cannot run into the next one.
void StringMap(const std::map<String, String>& map);
// ORDER-SENSITIVE, unlike NameMap: a transform-feedback capture list is a sequence,
// and gl_NextBuffer / gl_SkipComponentsN make its order load-bearing.
void TextList(const Vector<String>& values);
@@ -553,6 +557,8 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
// * the storage-block binding overrides handed to SPIRV-Cross;
// * the atomic-counter binding top, which SetAtomicCounterBlockBindings turns into the
// layout(binding=) qualifier every synthesized counter block is printed with;
// * this stage's two interface-block rename maps, which UniquifyIoBlockNamesForEssl
// turns into the block type names the emitted ESSL spells;
// * the ESSL version SPIRV-Cross targets (ResolveBackendEsslVersion, i.e. the
// driver's GLES version) - the remaining two SPIRV-Cross options are
// compile-time constants (GLSL_ES true, VULKAN_SEMANTICS false);
@@ -593,6 +599,14 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
const std::set<String>* xfbCaptureBlockNames = nullptr;
const UnorderedMap<String, Uint>* glFormatByUniformName = nullptr;
const UnorderedMap<String, Int>* storageBlockBindingOverrides = nullptr;
// THIS STAGE's share of the program-wide interface-block rename plan - the two
// arguments UniquifyIoBlockNamesForEssl is called with, which decide which block type
// names the emitted ESSL spells. Empty for every program without a tessellation or
// geometry stage that declares one block name in both directions, i.e. for all but a
// handful. The maps rather than what they were derived from: they ARE the pass's
// arguments, so they are exactly as fine as its behaviour and no finer.
const std::map<String, String>* inputBlockRenames = nullptr;
const std::map<String, String>* outputBlockRenames = nullptr;
// The top of the reserved storage-block window atomic-counter blocks are moved into
// (`top - N` for GL binding N). Derived from the driver's
+11 -5
View File
@@ -47,12 +47,18 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
// True when a packed internal format has REDUNDANT encodings, so decoding a texel and
// re-encoding it keeps the VALUE but not the BITS. Only RGB9_E5 does: its shared exponent can
// be lowered with the mantissas shifted up to match, and the spec's encoder always emits the
// canonical form. RGB10_A2, RGB10_A2UI and R11F_G11F_B10F round-trip through float32
// bit-exactly, so a GPU readback can answer for them.
// canonical form, so no readback that goes through a decode cycle can return the stored words.
//
// This is what decides whether the CPU shadow has to stay authoritative for a format: a
// readback of an RGB9_E5 level through a colour attachment cannot return the stored words, no
// matter how well behaved the driver is.
// Read this as "a FINITE value re-encodes to different bits", and nothing wider. This comment
// used to assert that RGB10_A2, RGB10_A2UI and R11F_G11F_B10F "round-trip through float32
// bit-exactly, so a GPU readback can answer for them", and that is false for
// R11F_G11F_B10F: a field whose 5-bit exponent is all ones is an Inf or a NaN, and a NaN's
// payload does not survive the trip (EncodeFloatToUnsignedSmallFloat re-encodes every NaN as
// the canonical payload 1). glCopyImageSubData from an RGB9_E5 source produces exactly such a
// word in the blue field on every texel, because the source's shared-exponent field is all
// ones. The bit-exact answer for all four formats is the raw-word route,
// DirectGLES::ReadPackedLevelWordsViaScratch; this predicate only picks which of the older
// fallbacks to prefer when that route is unavailable.
Bool HasRedundantPackedEncoding(TextureInternalFormat internalFormat);
// Decodes the canonical shadow-mip storage of `internalFormat` into wide RGBA texels for CPU
@@ -31,32 +31,41 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoRgb16;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget;
break;
// The two render-target bits reach EVERY signed-normalized format, one-, two- and
// four-channel included. They used to be granted to GL_RGB16_SNORM alone, which left the
// other seven with no colour-renderable fallback at all on a driver without
// EXT_render_snorm: an R8_SNORM or R16_SNORM attachment (what KHR-GL4x.texture_swizzle
// renders into for every SNORM source format) got no substitute, so the ES framebuffer was
// incomplete, the draw landed nowhere and the readback fell through to the never-written
// CPU shadow.
case GL_RGB16_SNORM:
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoRGB16Snorm;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoNorm16;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm16;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget;
if (options & PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget) {
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget;
}
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget;
break;
case GL_RGBA16_SNORM:
case GL_RG16_SNORM:
case GL_R16_SNORM:
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoNorm16;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm16;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget;
break;
case GL_RGBA8_SNORM:
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm8;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoRGBA8Snorm;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget;
break;
case GL_RGB8_SNORM:
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm8;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget;
break;
case GL_RG8_SNORM:
case GL_R8_SNORM:
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm8;
applicableOptions |= options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget;
break;
// The rest of the three-channel formats no real ES driver renders to. They have no
// other fallback: none of the driver/forced option bits names them, so before the
@@ -113,9 +122,12 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
return {GL_RGBA16F, GL_RGBA, GL_FLOAT};
case GL_RGB16_SNORM:
// A half float loses the low bits of a 16-bit SNORM channel, so keep the
// signed-normalized encoding whenever the driver can render to it.
// signed-normalized encoding whenever the driver can render to it - and when it
// cannot, widen to the 32-bit float, which is the only renderable storage that
// still holds all 65535 channel values exactly. GL_RGBA16F here handed -23451/32767
// back as -23457, six times the +/-1-step window KHR-GL4x.texture_swizzle allows.
return (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget)
? ThreeChannelWidening{GL_RGBA16F, GL_RGBA, GL_FLOAT}
? ThreeChannelWidening{GL_RGBA32F, GL_RGBA, GL_FLOAT}
: ThreeChannelWidening{GL_RGBA16_SNORM, GL_RGBA, GL_SHORT};
// Unsigned-normalized 16-bit (and the legacy 10/12-bit formats stored as RGB16):
// GL_RGB32F is a legal ES texture format but is not colour-renderable either.
@@ -203,7 +215,17 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
}
*outInternalFormat = internalFormat;
break;
// NoSnorm16RenderTarget outranks the other two 16-bit fallbacks on purpose: it is the
// only one whose substitute has to be EXACT, so it picks the 32-bit float rather than
// the half the driver/ANGLE fallbacks settle for. The capability probe folds the
// driver options and the render-target options into one set while the runtime storage
// choice can see the render-target bit alone (GetRuntimeFallbackNormalizeOptions), so
// the two would disagree on the storage format without a fixed precedence.
case GL_RGBA16_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) {
*outInternalFormat = GL_RGBA32F;
break;
}
if ((options & PixelFormatNormalizeOptionBit::NoNorm16) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm16)) {
*outInternalFormat = GL_RGBA16F;
@@ -212,6 +234,12 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
*outInternalFormat = internalFormat;
break;
case GL_RGB16_SNORM:
// The three-channel widening below replaces this whenever the target has to stay
// renderable; GL_RGB32F keeps the precision for the targets that do not.
if (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) {
*outInternalFormat = GL_RGB32F;
break;
}
if ((options & PixelFormatNormalizeOptionBit::NoNorm16) ||
(options & PixelFormatNormalizeOptionBit::NoRGB16Snorm) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm16)) {
@@ -221,6 +249,10 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
*outInternalFormat = internalFormat;
break;
case GL_RG16_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) {
*outInternalFormat = GL_RG32F;
break;
}
if ((options & PixelFormatNormalizeOptionBit::NoNorm16) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm16)) {
*outInternalFormat = GL_RG16F;
@@ -229,6 +261,10 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
*outInternalFormat = internalFormat;
break;
case GL_R16_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) {
*outInternalFormat = GL_R32F;
break;
}
if ((options & PixelFormatNormalizeOptionBit::NoNorm16) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm16)) {
*outInternalFormat = GL_R16F;
@@ -236,30 +272,36 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
}
*outInternalFormat = internalFormat;
break;
// 8-bit SNORM: the half float already IS exact here, so the render-target bit lands on
// the same storage the other two 8-bit fallbacks pick.
case GL_RGBA8_SNORM:
if ((options & PixelFormatNormalizeOptionBit::NoSnorm8) ||
(options & PixelFormatNormalizeOptionBit::NoRGBA8Snorm)) {
(options & PixelFormatNormalizeOptionBit::NoRGBA8Snorm) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
*outInternalFormat = GL_RGBA16F;
break;
}
*outInternalFormat = internalFormat;
break;
case GL_RGB8_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm8) {
if ((options & PixelFormatNormalizeOptionBit::NoSnorm8) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
*outInternalFormat = GL_RGB16F;
break;
}
*outInternalFormat = internalFormat;
break;
case GL_RG8_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm8) {
if ((options & PixelFormatNormalizeOptionBit::NoSnorm8) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
*outInternalFormat = GL_RG16F;
break;
}
*outInternalFormat = internalFormat;
break;
case GL_R8_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm8) {
if ((options & PixelFormatNormalizeOptionBit::NoSnorm8) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
*outInternalFormat = GL_R16F;
break;
}
@@ -270,10 +312,30 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
// per-channel precision (extra precision stays inside the CTS comparison epsilon, which is
// derived from the requested format's bit widths). The upload (format, type) below matches
// the canonical shadow layout in PixelStoreProcessor (UNorm8 / UNorm16 component arrays).
//
// The <=8-bit ones land on the 8-bit-per-channel storage that layout ALREADY is, rather
// than on the narrower GL_RGB565/GL_RGBA4 they nominally fit in. Storing them narrower
// made the driver requantize the UNorm8 shadow bytes on every upload, and that step is
// exact only by luck: 5-bit value 2 encodes as UNorm8 16, and 16/255*31 = 1.945 sits
// astride the 5-bit boundary, so a driver that truncates hands back 1 (all twelve
// KHR-GL43.copy_image.functional rgb4->rgb4 cases fail on Mali, at verify()'s FIRST
// check - a plain glTexImage/glGetTexImage round trip with no copy involved). The
// 8-bit store removes the requantization entirely; the client word round-trips
// exactly, because encoding an n-bit field to UNorm8 with rounding and back is the
// identity for every n <= 8. It is also what DirectVulkan has always done with them
// (VkTextureManager::ResolveTextureFormatInfo resolves all six legacy low-bit formats
// to R8G8B8A8_UNORM), so the two backends now agree here.
//
// Only the DESKTOP-ONLY formats move. GL_RGBA4 and GL_RGB5_A1 are ES formats an
// application can legitimately ask for - the same normalization picks the storage for
// glRenderbufferStorage - so widening them would be a memory decision, not a
// correctness one. Nothing about the REPORTED precision moves either way:
// GL_TEXTURE_*_SIZE and glGetInternalformativ answer from TextureMetrics, keyed on the
// requested format, not on the ES storage.
case GL_R3_G3_B2:
case GL_RGB4:
case GL_RGB5:
*outInternalFormat = GL_RGB565;
*outInternalFormat = GL_RGB8;
break;
case GL_RGB10:
case GL_RGB12:
@@ -283,7 +345,7 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
: GL_RGB16;
break;
case GL_RGBA2:
*outInternalFormat = GL_RGBA4;
*outInternalFormat = GL_RGBA8;
break;
case GL_RGBA12:
*outInternalFormat =
@@ -513,7 +575,8 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
if ((options & PixelFormatNormalizeOptionBit::NoNorm16) ||
(internalFormat == GL_RGB16_SNORM &&
(options & PixelFormatNormalizeOptionBit::NoRGB16Snorm)) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm16)) {
(options & PixelFormatNormalizeOptionBit::NoSnorm16) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget)) {
*outType = GL_FLOAT;
break;
} else {
@@ -523,7 +586,8 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
case GL_RGB8_SNORM:
case GL_RG8_SNORM:
case GL_R8_SNORM:
if (options & PixelFormatNormalizeOptionBit::NoSnorm8) {
if ((options & PixelFormatNormalizeOptionBit::NoSnorm8) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
*outType = GL_FLOAT;
break;
}
@@ -531,7 +595,8 @@ namespace MobileGL::MG_Util::TextureFormatProcessor {
break;
case GL_RGBA8_SNORM:
if ((options & PixelFormatNormalizeOptionBit::NoSnorm8) ||
(options & PixelFormatNormalizeOptionBit::NoRGBA8Snorm)) {
(options & PixelFormatNormalizeOptionBit::NoRGBA8Snorm) ||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
*outType = GL_FLOAT;
break;
}
@@ -29,11 +29,21 @@ namespace MobileGL {
// three-channel client data with an alpha of 1.0, and sampling/readback has to hide
// the added alpha again (BackendTextureFormatAddsAlpha).
NoThreeChannelRenderTarget = 1 << 7,
// Pairs with the bit above: the widened four-channel format has to stay renderable AND
// keep 16-bit signed-normalized precision, which needs both EXT_texture_norm16 and
// EXT_render_snorm. Without them the only renderable widening left is a half float, whose
// 11-bit mantissa cannot represent a 16-bit SNORM channel exactly.
// A 16-bit signed-normalized image has to back a colour attachment, and the driver cannot
// render to the signed-normalized encoding itself: that needs both EXT_texture_norm16 and
// EXT_render_snorm, and without either one an R16_SNORM / RG16_SNORM / RGB16_SNORM /
// RGBA16_SNORM attachment is texture-only, so the framebuffer is never complete and the
// draw silently lands nowhere. The substitute is a 32-bit float, NOT the half float the
// other SNORM fallbacks use: a half's 11-bit mantissa cannot represent a 16-bit SNORM
// channel exactly - its spacing just below 1.0 is 2^-11, some 16 SNORM steps, so
// -23451/32767 comes back as -23457 - while a 32-bit float round-trips every one of the
// 65535 channel values bit for bit.
NoSnorm16RenderTarget = 1 << 8,
// The 8-bit twin of the bit above: without EXT_render_snorm an R8_SNORM / RG8_SNORM /
// RGB8_SNORM / RGBA8_SNORM colour attachment is not renderable either. Here a half float
// IS exact - every value in [-127, 127] divided by 127 round-trips through a half - so the
// substitute matches what the always-on GL_RGBA8_SNORM fallback already picks.
NoSnorm8RenderTarget = 1 << 9,
None = 0,
};
namespace MG_Util::TextureFormatProcessor {
+59 -1
View File
@@ -57,6 +57,39 @@ def mem_available_kb(serial):
return int(m.group(1)) if m else None
def core_max_frequencies(serial):
r = adb(serial, "shell",
"for d in /sys/devices/system/cpu/cpu*/cpufreq; do "
"cat $d/cpuinfo_max_freq 2>/dev/null || echo 0; done", timeout=30)
freqs = [int(x) for x in re.findall(r"\d+", r.stdout or "")]
return freqs if freqs and max(freqs) > 0 else []
def derive_cpu_mask(serial, mode):
"""taskset mask for `mode`: 'prime' (fastest core only) or 'fast' (fast cluster).
Measured on the Mali G925, one texture_swizzle smoke case, two rounds in opposite
orders: unpinned 13.50/13.74 s, fast cluster 10.22/9.89 s, prime core 6.82/4.38 s.
Peak thread count was 11 in every configuration, so pinning does NOT cost MobileGL
any of its compile-pool parallelism - the unpinned run is simply losing to the
scheduler parking a CPU-bound load on the little cluster.
"""
freqs = core_max_frequencies(serial)
if not freqs:
return None
top = max(freqs)
if mode == "prime":
# The single fastest core. Fastest of the three in measurement, though with the
# widest spread, which is why it is opt-in rather than the default.
return f"{1 << freqs.index(top):x}"
cutoff = top * 0.7
mask = 0
for cpu, freq in enumerate(freqs):
if freq >= cutoff:
mask |= 1 << cpu
return f"{mask:x}" if mask else None
def completed_cases(qpa_path):
"""Return (finished_case_names, last_started_case_or_None).
@@ -99,6 +132,20 @@ def main():
# the contradiction instead of papering over it.
ap.add_argument("--gl-config-name", default="rgba8888d24s8",
help="--deqp-gl-config-name value (empty string to leave it unset)")
# A CTS run is CPU-bound (measured: cpu/wall = 93% on a texture_swizzle smoke case,
# which spends its time in glslang and spirv-tools, not in the driver), and Android's
# scheduler parks that load on the little cluster. Measured on the Mali G925 device,
# one smoke case: unpinned 16 s, cores 4-7 6 s, core 7 alone 5 s (unpinned re-run 16 s,
# so this is not drift). Pinning is worth 2.7-3.2x, and a NARROWER mask was faster,
# not slower - the compile pool's parallelism does not pay for the cross-core migration
# once the translation cache absorbs most of the compiles. "auto" keeps every core
# within 70% of the fastest, which drops the little cluster; that leaves room to run
# shards on separate cores, which is worth more than the last 20%.
ap.add_argument("--cpu-mask", default="fast",
help="CPU affinity for glcts: 'fast' (every core within 70%% of the "
"fastest, i.e. the big cluster), 'prime' (the single fastest core, "
"quickest measured but with the widest spread), 'none' (leave "
"affinity alone), or an explicit taskset hex mask")
ap.add_argument("--max-rounds", type=int, default=4000)
ap.add_argument("--max-empty-streak", type=int, default=64,
help="abort after this many consecutive chunks that produce no log at all")
@@ -128,6 +175,16 @@ def main():
total = len(remaining)
print(f"[run_cts] {args.backend} on {args.serial}: {total} cases")
cpu_mask = None
if args.cpu_mask in ("fast", "prime", "auto"): # auto kept as an alias for fast
cpu_mask = derive_cpu_mask(args.serial, "prime" if args.cpu_mask == "prime" else "fast")
if cpu_mask is None:
print("[run_cts] could not read cpufreq; leaving affinity alone", file=sys.stderr)
elif args.cpu_mask != "none":
cpu_mask = args.cpu_mask
if cpu_mask:
print(f"[run_cts] pinning glcts to CPU mask 0x{cpu_mask} (--cpu-mask {args.cpu_mask})")
crashed = []
hung = []
done = set()
@@ -167,12 +224,13 @@ def main():
config_flag = (
f"--deqp-gl-config-name={args.gl_config_name} " if args.gl_config_name else ""
)
taskset_prefix = f"taskset {cpu_mask} " if cpu_mask else ""
# The trailing sync makes the qpa durable: a hard GPU hang reboots the
# device, and f2fs rolls back unsynced writes, silently eating the log.
cmd = (
f"cd {args.device_dir} && "
f"MOBILEGL_BACKEND_TYPE={args.backend} LD_LIBRARY_PATH=. {extra_env}"
f"./glcts --deqp-caselist-file={dev_list} "
f"{taskset_prefix}./glcts --deqp-caselist-file={dev_list} "
f"--deqp-surface-type={args.surface} "
f"--deqp-surface-width={args.surface_size} "
f"--deqp-surface-height={args.surface_size} "