Compare commits

...
Author SHA1 Message Date
swung0x48 2b6c2b561c [Fix, Test] (DirectVulkan, ShaderTranspiler, TraceReplay): derive NumSubgroups behind opt-in quirk 2026-08-18 22:31:57 -04:00
swung0x48andClaude Fable 5 12c94111b5 [Fix, Test] (DirectVulkan, SelfTest, MG_IntegrationTest, TraceReplay): snapshot sampler/image feedback and add Program 203 diagnostics
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-18 03:04:09 -04:00
swung0x48 7769156cfc [Fix, Test] (ShaderTranspiler, WGL, TraceReplay): remove subgroup pack quirks and tune iterationRP 2026-08-17 01:55:17 -04:00
swung0x48 0ecfdff4e7 [Fix, Test] (MG_State, MG_Util, DirectVulkan, MG_Test): replay narrow-subgroup reductions correctly 2026-08-16 13:33:01 -04:00
swung0x48 6df5a6137f [CI] (trace-replay): run iterationRP only on DirectVulkan 2026-08-16 11:09:56 -04:00
swung0x48 b3794f4e6a [Feat] (MG_Impl, MG_State, MG_Util, DirectGLES, DirectVulkan, MG_Test): implement ARB_clear_buffer_object correctly 2026-08-16 01:12:28 -04:00
swung0x48 14d3901d30 [Chore, Test] (MG_State, MG_Util, DirectGLES, DirectVulkan): make SPIR-V validation task-local 2026-08-15 22:58:28 -04:00
swung0x48 d4766513e4 [Fix, Test] (DirectVulkan): replay Photon descriptor pressure correctly 2026-08-15 22:35:39 -04:00
swung0x48 72dc7aa6aa [Chore] (MG_Config, MG_State, MG_Util): gate SPIR-V validation at startup 2026-08-15 22:35:39 -04:00
swung0x48 9d1b280375 [Fix, Test] (MG_Util, MG_Test): rename Photon-conflicting MSL identifiers 2026-08-15 09:40:13 -04:00
swung0x48 8acd885594 [Fix, Test] (DirectVulkan): preserve viewport-index program metadata 2026-08-15 09:40:13 -04:00
swung0x48 10ff5e2b18 [Fix, Test] (DirectGLES, DirectVulkan): advertise indirect draw capabilities accurately
- advertise GL_ARB_draw_indirect when supported
- gate GL_ARB_base_instance on complete non-zero firstInstance semantics
- synchronize Driver POST reporting
- add capability and extension-advertisement regression tests
2026-08-15 06:30:44 -04:00
swung0x48 a6e52476f3 [Chore] (CMake): skip embedded SPIRV-Tools executables 2026-08-15 05:48:41 -04:00
swung0x48 0deff52a1b [Fix, Test] (DirectVulkan, trace-replay): replay quarter-turn surfaces correctly 2026-08-13 06:45:28 -04:00
swung0x48 50fefca959 Merge branch "feat/cts-viewport-array" into dev 2026-08-13 04:57:40 -04:00
swung0x48 42ad62b54c [Fix, Test] (MG_Util, MG_IntegrationTest): advertise the GL 4.3 VIEWPORT_BOUNDS_RANGE floor on a GLES driver that has no such query, instead of a range admitting no origin 2026-08-13 04:44:23 -04:00
swung0x48 f41403e227 [Feat, Test] (MG_Backend/DirectVulkan, MG_Test, MG_IntegrationTest): rasterize the viewport gl_ViewportIndex selects, instead of collapsing all sixteen onto viewport 0 2026-08-13 04:44:23 -04:00
swung0x48 5fbb17f6b9 [Feat, Fix, Test] (MG_State, MG_Impl, MG_Backend, MG_Test): give ARB_viewport_array real 16-element indexed state instead of eight stubs and a viewport-0 echo 2026-08-13 04:44:23 -04:00
swung0x48 92d8f7269b [Fix] (MG_IntegrationTest): drop the executable bit a copied-in scenario file carried 2026-08-13 04:29:23 -04:00
swung0x48 822e405c77 Merge branch "feat/vk-barrier-layer-ranges" into dev 2026-08-13 04:28:57 -04:00
swung0x48 b8233f9c4e [Fix, Test] (MG_Backend/DirectVulkan, MG_IntegrationTest): the barrier before every attachment transfer only ever moved layer 0, so a copy off a non-zero layer read a layout nothing had transitioned 2026-08-13 04:01:20 -04:00
Swung0x48 91475a7b6f Merge branch feat/cts-copyimage-frontend into dev 2026-08-13 03:59:16 -04:00
97 changed files with 7566 additions and 1275 deletions
+5 -8
View File
@@ -209,7 +209,7 @@ jobs:
- name: Load trace cases
id: trace-cases
run: |
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk)" >> "$GITHUB_OUTPUT"
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk-matrix)" >> "$GITHUB_OUTPUT"
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
trace-fixtures:
@@ -337,13 +337,7 @@ jobs:
strategy:
fail-fast: false
max-parallel: 4
matrix:
backend:
- name: DirectGLES
gpu: software
- name: DirectVulkan
gpu: lavapipe
case: ${{ fromJSON(needs.trace-cases.outputs.android) }}
matrix: ${{ fromJSON(needs.trace-cases.outputs.android) }}
steps:
- name: Set Swap Space
uses: pierotofy/set-swap-space@v1.0
@@ -441,6 +435,9 @@ jobs:
if [ "${{ matrix.case.coherent_as_flush || false }}" = "true" ]; then
extra_retrace_args+=(--coherent-as-flush)
fi
if [ "${{ matrix.case.num_subgroups_quirk || false }}" = "true" ]; then
extra_retrace_args+=(--num-subgroups-quirk)
fi
run_retrace() {
timeout "$(( ${{ matrix.case.timeout_seconds }} + 300 ))" sh android-plugin/trace-replay-ci.sh \
+5 -6
View File
@@ -491,6 +491,7 @@ jobs:
- benchmark
- integration
outputs:
matrix: ${{ steps.trace-cases.outputs.matrix }}
names: ${{ steps.trace-cases.outputs.names }}
steps:
- name: Checkout repo
@@ -498,7 +499,9 @@ jobs:
- name: Load trace cases
id: trace-cases
run: echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
run: |
echo "matrix=$(python3 tools/trace_replay/trace_cases.py --ci --format github-test-matrix)" >> "$GITHUB_OUTPUT"
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
trace-fixtures:
name: trace fixture (${{ matrix.case }})
@@ -577,11 +580,7 @@ jobs:
strategy:
fail-fast: false
max-parallel: 4
matrix:
backend:
- DirectGLES
- DirectVulkan
case: ${{ fromJSON(needs.trace-cases.outputs.names) }}
matrix: ${{ fromJSON(needs.trace-cases.outputs.matrix) }}
steps:
- name: Set Swap Space
+3
View File
@@ -182,6 +182,7 @@ set(ENABLE_SPVREMAPPER OFF CACHE BOOL "Enable SPVRemapper" FORCE)
set(ENABLE_OPT ON CACHE BOOL "Enable SPIRV-Tools opt usage in glslang" FORCE)
set(BUILD_EXTERNAL ON CACHE BOOL "Build external deps in External/" FORCE)
set(ENABLE_GLSLANG_INSTALL OFF CACHE BOOL "Install glslang targets" FORCE)
set(SPIRV_SKIP_EXECUTABLES ON CACHE BOOL "Skip building SPIRV-Tools executables" FORCE)
set(SPIRV_CROSS_C_API ON CACHE BOOL "Enable C API" FORCE)
set(SPIRV_CROSS_ENABLE_GLSL ON CACHE BOOL "Enable GLSL backend" FORCE)
@@ -283,6 +284,7 @@ set(SOURCE_FILES
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/SplitArrayVertexInputsPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DeriveNumSubgroupsPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DArrayImagesPass.cpp
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/BakeImageFormatsPass.cpp
@@ -297,6 +299,7 @@ set(SOURCE_FILES
MobileGL/MG_Util/BackendLoaders/Vulkan/Loader.cpp
MobileGL/MG_Util/SelfTest/DriverPost.cpp
MobileGL/MG_Util/SelfTest/DriverPostProgram203Witness.cpp
MobileGL/MG_Util/Texture/PixelStoreProcessor.cpp
MobileGL/MG_Util/Texture/TextureFormatProcessor.cpp
+8 -9
View File
@@ -66,14 +66,12 @@ namespace MobileGL::MG_Config {
// - DISPLAY: X11 session variable, not MobileGL configuration.
// - MOBILEGL_LOG_FILE_PATH: log-file init runs before MG_ConfigLoader::Init
// (see MG_Util/Debug/Log.cpp).
// - MOBILEGL_VALIDATE_SPIRV: test suites like SpirvPassTest exercise
// ShaderCompiler without ever running MobileGL::Initialize(), and every
// Initialize() re-runs MG_ConfigLoader::Init, which would clobber a
// programmatic override stored here (see ShaderCompiler.cpp,
// SpirvValidationEnabled).
struct FeaturesTable {
// MOBILEGL_DISABLE_TIMERQUERY: do not advertise or use GPU timer queries.
Bool DisableTimerQuery = false;
// MOBILEGL_ENABLE_SPIRV_VALIDATION: validate generated and transformed SPIR-V.
// Disabled by default because validation is a diagnostics-only cost.
Bool EnableSpirvValidation = false;
// MOBILEGL_USE_ANGLE: load ANGLE EGL/GLES libraries.
Bool UseAngle = false;
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
@@ -82,6 +80,11 @@ namespace MobileGL::MG_Config {
#endif
// MOBILEGL_DISABLE_SUBGROUP: force-disable Vulkan shader subgroup support.
Bool DisableSubgroup = false;
// MOBILEGL_NUM_SUBGROUPS_QUIRK: derive compute gl_NumSubgroups from the local
// workgroup dimensions and gl_SubgroupSize instead of reading Vulkan's
// NumSubgroups builtin. Off by default; enable only for drivers whose builtin
// disagrees with the SubgroupId topology emitted by the same dispatch.
Bool NumSubgroupsQuirk = false;
// MOBILEGL_ADVERTISE_FP64: add GL_ARB_gpu_shader_fp64 to the advertised extension
// string. `double` in a shader always WORKS - it is narrowed to 32 bits before any
// module reaches a backend (ShaderTranspiler::DemoteFloat64Pass) - but the extension
@@ -130,10 +133,6 @@ namespace MobileGL::MG_Config {
// explicitly request a core profile via EGL_CONTEXT_OPENGL_PROFILE_MASK / a >=3.1
// version request.
Bool RelaxedSemantics = false;
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that
// rewrites the recognized workgroup prefix-scan template on Qualcomm devices with
// subgroups wider than 32 lanes (see ShaderSourceProcessor's quirk registry).
QuirkOverride SubgroupPrefixScanQuirk = QuirkOverride::Auto;
// MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE: overrides the DirectVulkan quirk that
// strips depth writes from accumulation-blended pipelines (MIN/MAX or additive
// ONE+ONE - the multi-pass depth-equality signature) on drivers without
+2 -1
View File
@@ -162,11 +162,13 @@ namespace MobileGL::MG_ConfigLoader {
inline void InitFeatures() {
auto& features = MG_Config::Features;
features.DisableTimerQuery = QueryEnvFlag("MOBILEGL_DISABLE_TIMERQUERY");
features.EnableSpirvValidation = QueryEnvFlag("MOBILEGL_ENABLE_SPIRV_VALIDATION");
features.UseAngle = QueryEnvFlag("MOBILEGL_USE_ANGLE");
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
QueryEnvVariable("MOBILEGL_TRACE_ANGLE_VARIANT", features.TraceAngleVariant, "");
#endif
features.DisableSubgroup = QueryEnvFlag("MOBILEGL_DISABLE_SUBGROUP");
features.NumSubgroupsQuirk = QueryEnvFlag("MOBILEGL_NUM_SUBGROUPS_QUIRK");
features.AdvertiseFp64 = QueryEnvFlag("MOBILEGL_ADVERTISE_FP64");
features.MagmaR11G11B10FFallback = QueryEnvFlag("MOBILEGL_MAGMA_R11G11B10F_FALLBACK");
features.MagmaFramesInFlight = QueryEnvUint32("MOBILEGL_MAGMA_FRAMESINFLIGHT", 3, 1, 64);
@@ -179,7 +181,6 @@ namespace MobileGL::MG_ConfigLoader {
features.EsprytForceDepthStencilReadbackEmulation =
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
features.SubgroupPrefixScanQuirk = QueryEnvQuirkOverride("MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN");
features.MagmaDisableBlendedDepthWriteQuirk =
QueryEnvQuirkOverride("MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE");
features.DisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS");
@@ -712,9 +712,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
{
.TargetGLVersion = {4, 0, 0}, // GL target version
.TargetGLSLVersion = {4, 6, 0}, // Target Shading Language Version
// Baseline advertisement (no timer queries / anisotropy yet); reconciled
// once the ES capabilities exist, see UpdateAdvertisedCapabilityExtensions.
.Extensions = BuildAdvertisedExtensions(false, false),
// Baseline advertisement (no runtime capabilities yet); reconciled once
// the ES capabilities exist, see UpdateAdvertisedCapabilityExtensions.
.Extensions = BuildAdvertisedExtensions(false, false, false, false),
.IsCompatibilityProfile = false // Is Compatibility Profile
},
.StaticBackendCapability = {.AllowVSOnlyPrograms = false} // Backend Capability
@@ -734,9 +734,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
// thread can only observe the extension string after the
// advertisement for its context has settled; rebuilding the whole
// list keeps the re-run after a context recreation idempotent.
void UpdateAdvertisedCapabilityExtensions(Bool anisotropicFilteringSupported) {
MutableRendererInfo().RendererGLInfo.Extensions =
BuildAdvertisedExtensions(AreTimerQueriesSupported(), anisotropicFilteringSupported);
void UpdateAdvertisedCapabilityExtensions(const MG_External::GLESCapabilities& capabilities) {
MutableRendererInfo().RendererGLInfo.Extensions = BuildAdvertisedExtensions(
AreTimerQueriesSupported(), capabilities.SupportsTextureFilterAnisotropy,
capabilities.SupportsDrawIndirect,
capabilities.SupportsDrawIndirect && capabilities.SupportsBaseInstance);
}
} // namespace
@@ -779,11 +781,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
return false;
}
DirectGLES::SetGLESCapabilities(m_GLESCapabilities);
// Now that g_GLESCapabilities knows about GL_EXT_disjoint_timer_query and
// GL_EXT_texture_filter_anisotropic, reconcile the advertisement (see the comment on
// UpdateAdvertisedCapabilityExtensions for why it cannot happen when the extension
// list is first built).
UpdateAdvertisedCapabilityExtensions(m_GLESCapabilities.SupportsTextureFilterAnisotropy);
// Now that g_GLESCapabilities knows the host extensions, entry points, and ES version,
// reconcile every runtime-gated advertisement (see the comment on
// UpdateAdvertisedCapabilityExtensions for why this cannot happen when the list is first
// built).
UpdateAdvertisedCapabilityExtensions(m_GLESCapabilities);
UpdateDynamicBackendParameters();
PopulateFormatCapabilities(m_GLESFunctions, m_GLESCapabilities, MutableFormatCapabilities());
PrintFormatCapabilities(GetFormatCapabilities());
@@ -924,11 +926,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
return MutableRendererInfo();
}
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported) {
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported,
Bool drawIndirectSupported,
Bool nonZeroIndirectBaseInstanceSupported) {
Vector<GLExtension> extensions = {
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage, E_GL_ARB_texture_storage,
E_GL_ARB_texture_storage_multisample, E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access,
E_GL_ARB_multi_draw_indirect, E_GL_ARB_indirect_parameters, E_GL_ARB_shader_draw_parameters,
@@ -955,6 +959,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
// extension explicitly permits. It is also the only thing that
// exposes glProgramParameteri before GL 4.1.
E_GL_ARB_get_program_binary};
// Minecraft 26.3 checks this prerequisite before it even considers
// GL_ARB_multi_draw_indirect. ES 3.1 supplies both single-draw entry points; the loader
// folds the version and pointer checks into SupportsDrawIndirect.
if (drawIndirectSupported) {
extensions.push_back(E_GL_ARB_draw_indirect);
}
// ARB_base_instance also defines the last word of an indirect command. Direct calls are
// emulated on every Espryt device, but without host GL_EXT_base_instance a native indirect
// draw cannot shift divisor attributes by a GPU-authored non-zero value, so do not promise
// that incomplete case.
if (drawIndirectSupported && nonZeroIndirectBaseInstanceSupported) {
extensions.push_back(E_GL_ARB_base_instance);
}
// GL_KHR_parallel_shader_compile is MobileGL's own capability, not the host ES
// driver's: the compiler threads are MobileGL's, and glCompileShader/glLinkProgram
// are serviced entirely inside the frontend. Whether the device driver advertises
@@ -67,9 +67,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
const RendererInfo& GetRendererIdentity();
// The full OpenGL extension list Espryt advertises (glGetString(GL_EXTENSIONS))
// for a device whose timer queries / anisotropic filtering are (or are not) usable.
// for a device whose timer queries / anisotropic filtering / native indirect draws /
// non-zero indirect baseInstance semantics are (or are not) usable.
// The MOBILEGL_DISABLE_TIMERQUERY escape hatch is applied inside.
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported);
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported,
Bool drawIndirectSupported,
Bool nonZeroIndirectBaseInstanceSupported);
// Format: <OpenGL ES Renderer>, OpenGL ES <Major>.<Minor> — the exact string an
// initialized backend returns from GetBackendAPIVersionString (and that ends up
+25 -11
View File
@@ -1600,7 +1600,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
!g_hasSyncedRenderState || std::memcmp(currentBytes + kBlendSpanEnd, syncedBytes + kBlendSpanEnd,
sizeof(RenderStateParameters) - kBlendSpanEnd) != 0;
IntVec4 backendViewport = parameters.Viewport;
IntVec4 backendViewport = MG_State::pGLContext->GetViewport();
if (backendViewport.z() <= 0 || backendViewport.w() <= 0) {
Int surfaceWidth = 0;
Int surfaceHeight = 0;
@@ -1614,7 +1614,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
g_syncedBackendViewport = backendViewport;
}
// All 12 capability bools live after LogicOp in the struct, i.e. in the tail span.
// Every capability bool (and the scissor-test mask below) lives after LogicOp in the
// struct, i.e. in the tail span.
if (tailSpanDirty) {
#define SYNC_CAPABILITY(cap_mg, cap_gl) \
if (forceFullPush || parameters.cap_mg##Enabled != g_syncedRenderStateParameters.cap_mg##Enabled) { \
@@ -1633,11 +1634,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
SYNC_CAPABILITY(SampleMask, GL_SAMPLE_MASK);
SYNC_CAPABILITY(PolygonOffsetFill, GL_POLYGON_OFFSET_FILL);
SYNC_CAPABILITY(RasterizerDiscard, GL_RASTERIZER_DISCARD);
SYNC_CAPABILITY(ScissorTest, GL_SCISSOR_TEST);
SYNC_CAPABILITY(StencilTest, GL_STENCIL_TEST);
SYNC_CAPABILITY(CullFace, GL_CULL_FACE);
#undef SYNC_CAPABILITY
// GL_SCISSOR_TEST is per-viewport enable state (ARB_viewport_array), so it is a
// 16-bit mask and not a "<Name>Enabled" bool the macro above could key off. ES
// has exactly one scissor rectangle and one scissor enable, so only bit 0 - the
// index every ES draw rasterizes against - can be forwarded; a program that
// enables the test for viewport 3 alone gets viewport 0's answer here. That is
// the same limitation as the unemulated gl_ViewportIndex on this backend and is
// why the multi-viewport half of KHR-GL43.viewport_array stays red on Espryt.
{
const Bool scissorTest = (parameters.ScissorTestEnabledMask & 1u) != 0;
const Bool syncedScissorTest =
(g_syncedRenderStateParameters.ScissorTestEnabledMask & 1u) != 0;
if (forceFullPush || scissorTest != syncedScissorTest) {
scissorTest ? g_GLESFuncs.glEnable(GL_SCISSOR_TEST) : g_GLESFuncs.glDisable(GL_SCISSOR_TEST);
}
}
}
if (tailSpanDirty && g_GLESCapabilities.SupportsClipDistance) {
@@ -1864,8 +1880,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (forceFullPush || parameters.DepthMask != g_syncedRenderStateParameters.DepthMask) {
g_GLESFuncs.glDepthMask(parameters.DepthMask ? GL_TRUE : GL_FALSE);
}
if (forceFullPush || parameters.DepthRange != g_syncedRenderStateParameters.DepthRange) {
g_GLESFuncs.glDepthRangef(parameters.DepthRange.x(), parameters.DepthRange.y());
if (forceFullPush || parameters.DepthRanges[0] != g_syncedRenderStateParameters.DepthRanges[0]) {
g_GLESFuncs.glDepthRangef(parameters.DepthRanges[0].x(), parameters.DepthRanges[0].y());
}
}
@@ -2003,7 +2019,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
// everything drawn with GL_SCISSOR_TEST enabled before the app's first glScissor
// is clipped away - Minecraft 26.2 keeps only its unscissored sky and hand and
// loses the terrain and the whole GUI.
IntVec4 backendScissorBox = parameters.ScissorBox;
IntVec4 backendScissorBox = parameters.ScissorBoxes[0];
if (backendScissorBox.z() <= 0 || backendScissorBox.w() <= 0) {
Int surfaceWidth = 0;
Int surfaceHeight = 0;
@@ -3045,10 +3061,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
}
static Bool SupportsNativeIndirectDraws() {
const auto& version = g_GLESCapabilities.GLESVersion;
const Bool esVersionOk = version.Major > 3 || (version.Major == 3 && version.Minor >= 1);
return esVersionOk && g_GLESFuncs.glDrawElementsIndirect != nullptr &&
g_GLESFuncs.glDrawArraysIndirect != nullptr;
return g_GLESCapabilities.SupportsDrawIndirect;
}
// Runs an (indexed) indirect multi-draw. When a GL_DRAW_INDIRECT_BUFFER is bound the draws
@@ -4762,7 +4775,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
// restores the app state on exit, tracked via the render-state shadow.
class ScopedScissorDisable {
public:
ScopedScissorDisable() : m_wasEnabled(RenderStateImpl::g_syncedRenderStateParameters.ScissorTestEnabled) {
ScopedScissorDisable()
: m_wasEnabled((RenderStateImpl::g_syncedRenderStateParameters.ScissorTestEnabledMask & 1u) != 0) {
if (m_wasEnabled) g_GLESFuncs.glDisable(GL_SCISSOR_TEST);
}
~ScopedScissorDisable() {
+17 -10
View File
@@ -729,8 +729,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
void Ops_ReadbackFromGpu(BufferObject& bufferObject) {
auto* resource = ResourceOf(bufferObject);
if (!resource || resource->id == 0 || !resource->storageInitialized) return;
if (resource->persistentMapped) return; // shadow already IS the GPU storage
if (!CanTouchGLNow() || resource->contextGeneration != g_bufferContextGeneration) return;
if (resource->persistentMapped) {
// Host writes to a persistent map must not race shader writes already queued
// on this context. There is no backend copy to read back in this case.
if (g_GLESFuncs.glFinish) g_GLESFuncs.glFinish();
return;
}
if (!g_GLESFuncs.glMapBufferRange || !g_GLESFuncs.glUnmapBuffer) return;
const SizeT size = std::min<SizeT>(bufferObject.GetSize(), resource->storageSize);
if (size == 0) return;
@@ -4741,6 +4746,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
MGLOG_D("%s:", src.empty() ? "" : src.c_str());
}
auto& shaderSpirvs = stateProgramObject->GetGeneratedSpirv();
const Bool enableSpirvValidation = stateProgramObject->GetSpirvValidationEnabled();
// Blocks a transform-feedback capture request names a member of ("StageData" of
// "StageData.attrib[0]"). The Adreno ES driver accepts such a request, links, and
@@ -4794,7 +4800,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
Vector<unsigned int> loweredSpirv;
const Vector<unsigned int>* effectiveSpirv = &spirvCode;
if (glShaderType == GL_VERTEX_SHADER &&
MG_Util::ShaderTranspiler::ShaderCompiler::LowerDrawParametersForEssl(spirvCode, loweredSpirv) &&
MG_Util::ShaderTranspiler::ShaderCompiler::LowerDrawParametersForEssl(spirvCode, loweredSpirv, enableSpirvValidation) &&
!loweredSpirv.empty()) {
effectiveSpirv = &loweredSpirv;
}
@@ -4804,7 +4810,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
Vector<unsigned int> splitArrayInputSpirv;
if (glShaderType == GL_VERTEX_SHADER &&
MG_Util::ShaderTranspiler::ShaderCompiler::SplitArrayVertexInputsForEssl(
*effectiveSpirv, splitArrayInputSpirv) &&
*effectiveSpirv, splitArrayInputSpirv, enableSpirvValidation) &&
!splitArrayInputSpirv.empty() && splitArrayInputSpirv != *effectiveSpirv) {
// Only when the pass ACTUALLY split something. The optimizer hands back a
// re-serialised copy either way, and adopting that copy for every vertex
@@ -4826,7 +4832,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (!xfbCaptureBlockNames.empty() &&
MG_Util::ShaderTranspiler::ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(
*effectiveSpirv, xfbCaptureBlockNames, stageFlattenedXfbBlockNames,
flattenedXfbSpirv) &&
flattenedXfbSpirv, enableSpirvValidation) &&
!flattenedXfbSpirv.empty() && !stageFlattenedXfbBlockNames.empty()) {
effectiveSpirv = &flattenedXfbSpirv;
flattenedXfbBlockNames.insert(stageFlattenedXfbBlockNames.begin(),
@@ -4842,7 +4848,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
// declare the member highp; nothing else about emission changes.
Vector<unsigned int> uboPrecisionSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::StripUboMemberRelaxedPrecisionForEssl(
*effectiveSpirv, uboPrecisionSpirv) &&
*effectiveSpirv, uboPrecisionSpirv, enableSpirvValidation) &&
!uboPrecisionSpirv.empty()) {
effectiveSpirv = &uboPrecisionSpirv;
}
@@ -4857,7 +4863,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
Vector<unsigned int> noperspectiveSpirv;
if (!g_GLESCapabilities.SupportsNoperspectiveInterpolation &&
MG_Util::ShaderTranspiler::ShaderCompiler::EmulateNoPerspectiveForEssl(
*effectiveSpirv, noperspectiveSpirv) &&
*effectiveSpirv, noperspectiveSpirv, enableSpirvValidation) &&
!noperspectiveSpirv.empty()) {
effectiveSpirv = &noperspectiveSpirv;
}
@@ -4867,7 +4873,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
// divides the coordinate of every normalized-coordinate lookup by the texture
// size, which is the whole of the difference between the two.
Vector<unsigned int> rectLoweredSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(*effectiveSpirv, rectLoweredSpirv) &&
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(*effectiveSpirv, rectLoweredSpirv, enableSpirvValidation) &&
!rectLoweredSpirv.empty()) {
effectiveSpirv = &rectLoweredSpirv;
}
@@ -4881,7 +4887,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
// coordinate to (u, 0, layer) - before SPIRV-Cross can apply its own.
Vector<unsigned int> arrayImageSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::Lower1DArrayImagesForEssl(*effectiveSpirv,
arrayImageSpirv) &&
arrayImageSpirv, enableSpirvValidation) &&
!arrayImageSpirv.empty()) {
effectiveSpirv = &arrayImageSpirv;
}
@@ -4900,7 +4906,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
if (!imageFormatBake.glFormatByUniformName.empty() &&
MG_Util::ShaderTranspiler::ShaderCompiler::DeclaresFormatlessStorageImage(*effectiveSpirv) &&
MG_Util::ShaderTranspiler::ShaderCompiler::BakeImageFormatsForEssl(
*effectiveSpirv, imageFormatBake.glFormatByUniformName, imageFormatSpirv) &&
*effectiveSpirv, imageFormatBake.glFormatByUniformName, imageFormatSpirv,
enableSpirvValidation) &&
!imageFormatSpirv.empty()) {
effectiveSpirv = &imageFormatSpirv;
}
@@ -4916,7 +4923,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
Vector<unsigned int> outputIndexSpirv;
if (glShaderType == GL_FRAGMENT_SHADER &&
MG_Util::ShaderTranspiler::ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(
*effectiveSpirv, outputIndexSpirv) &&
*effectiveSpirv, outputIndexSpirv, enableSpirvValidation) &&
!outputIndexSpirv.empty()) {
effectiveSpirv = &outputIndexSpirv;
}
@@ -10,6 +10,7 @@
#include "MG_Backend/BackendObject.h"
#include "DirectVulkan.h"
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
#include "MG_State/GLState/Core.h"
#include "MG_State/GLState/TextureState/TextureState.h"
#include "MG_Util/Classifiers/TextureEnumClassifier.h"
#include "MG_Util/Converters/MGToGL/TextureEnumConverter.h"
@@ -383,6 +384,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
}
UpdateDynamicBackendParameters();
UpdateAdvertisedExtensions();
if (MG_State::pGLContext) {
MG_State::pGLContext->InvalidateCompileEnv();
}
PopulateFormatCapabilities(physicalDevice.handle, vkGetPhysicalDeviceFormatProperties, m_vulkanCaps,
MutableFormatCapabilities());
PrintFormatCapabilities(GetFormatCapabilities());
@@ -497,20 +501,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
.ExtraVendor = Nullopt,
.RendererGLInfo = {.TargetGLVersion = {4, 0, 0},
.TargetGLSLVersion = {4, 6, 0},
// Baseline advertisement (no shader subgroup, no timer queries); a
// live backend reconciles its copy in UpdateAdvertisedExtensions.
.Extensions = BuildAdvertisedExtensions(false, false, false),
// Baseline advertisement (no runtime-gated capabilities); a live
// backend reconciles its copy in UpdateAdvertisedExtensions.
.Extensions = BuildAdvertisedExtensions(false, false, false, false),
.IsCompatibilityProfile = false},
.StaticBackendCapability = {.AllowVSOnlyPrograms = false}};
return rendererInfo;
}
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
Bool anisotropicFilteringSupported) {
Bool anisotropicFilteringSupported,
Bool nonZeroIndirectBaseInstanceSupported) {
Vector<GLExtension> extensions = {
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_multi_draw_indirect,
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_draw_indirect,
E_GL_ARB_multi_draw_indirect,
E_GL_ARB_indirect_parameters, E_GL_EXT_framebuffer_object, E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage,
E_GL_ARB_texture_storage, E_GL_ARB_texture_storage_multisample, E_GL_ARB_texture_multisample,
E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access, E_GL_ARB_shader_draw_parameters,
@@ -530,6 +536,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// extension explicitly permits. It is also the only thing that
// exposes glProgramParameteri before GL 4.1.
E_GL_ARB_get_program_binary};
// Vulkan's drawIndirectFirstInstance feature is optional. Direct base-instance calls work
// without it, but ARB_base_instance also promises non-zero firstInstance in GPU indirect
// commands; the renderer supplies true only when that word is legal and gl_InstanceID can
// be rebased to OpenGL's zero-based semantics.
if (nonZeroIndirectBaseInstanceSupported) {
extensions.push_back(E_GL_ARB_base_instance);
}
if (shaderSubgroupSupported && !MG_Config::Features.DisableSubgroup) {
extensions.push_back(E_GL_KHR_shader_subgroup);
}
@@ -678,6 +691,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
m_vulkanCaps = capabilities;
UpdateDynamicBackendParameters();
UpdateAdvertisedExtensions();
if (MG_State::pGLContext) {
MG_State::pGLContext->InvalidateCompileEnv();
}
MutableFormatCapabilities().Clear();
}
@@ -690,7 +706,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// the whole list keeps re-runs idempotent.
m_rendererInfo.RendererGLInfo.Extensions = BuildAdvertisedExtensions(
m_vulkanCaps.SupportsShaderSubgroup, pVulkanRenderer && pVulkanRenderer->IsTimerQuerySupported(),
pVulkanRenderer && pVulkanRenderer->IsSamplerAnisotropySupported());
pVulkanRenderer && pVulkanRenderer->IsSamplerAnisotropySupported(),
pVulkanRenderer && pVulkanRenderer->IsNonZeroIndirectBaseInstanceSupported());
}
void BackendObject_DirectVulkan::UpdateDynamicBackendParameters() {
@@ -62,8 +62,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// POST screen shows.
// Static identity of the Magma renderer (renderer/backend names, target GL/GLSL
// versions, ExtraVendor) with the baseline extension advertisement (no shader
// subgroup, no timer queries). A live backend copies this in its constructor and
// versions, ExtraVendor) with the baseline extension advertisement (no runtime-gated
// capabilities). A live backend copies this in its constructor and
// reconciles the Extensions in UpdateAdvertisedExtensions once real capabilities
// exist; callers that need the advertised list for a known capability set must
// use BuildAdvertisedExtensions instead.
@@ -74,7 +74,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// MOBILEGL_DISABLE_TIMERQUERY escape hatches are applied inside, so callers pass
// the detected device support (passing an already-gated value is harmless).
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
Bool anisotropicFilteringSupported);
Bool anisotropicFilteringSupported,
Bool nonZeroIndirectBaseInstanceSupported);
// Format: <GPU Name>, Vulkan <Vulkan Version>, Driver <Driver Version> — the exact
// string an initialized backend returns from GetBackendAPIVersionString (and that
@@ -206,6 +206,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
XXHASH_VERIFY(
XXH64_update(m_hashState, &payload.primitiveRestartEnable, sizeof(payload.primitiveRestartEnable)));
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.patchControlPoints, sizeof(payload.patchControlPoints)));
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.viewportCount, sizeof(payload.viewportCount)));
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.polygonMode, sizeof(payload.polygonMode)));
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.cullMode, sizeof(payload.cullMode)));
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.frontFace, sizeof(payload.frontFace)));
@@ -406,8 +407,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
tessellation.patchControlPoints = payload.patchControlPoints;
VkPipelineViewportStateCreateInfo vpci{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO};
vpci.viewportCount = 1;
vpci.scissorCount = 1;
// Both counts move together: GL has one scissor rectangle per viewport, and Vulkan
// requires viewportCount == scissorCount whenever both are dynamic
// (VUID-VkPipelineViewportStateCreateInfo-scissorCount-04136). The caller has already
// clamped this to the device's multiViewport capability.
vpci.viewportCount = std::max<Uint32>(payload.viewportCount, 1u);
vpci.scissorCount = vpci.viewportCount;
VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO};
raster.polygonMode = payload.polygonMode;
@@ -42,6 +42,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
Bool primitiveRestartEnable = false;
// GL_PATCH_VERTICES; only read for a PATCH_LIST topology.
Uint32 patchControlPoints = 3;
// How many of ARB_viewport_array's viewports this pipeline rasterizes into. 1 for
// every program that never assigns gl_ViewportIndex, which is all of them outside the
// conformance suite - the wide shape costs a longer vkCmdSetViewport/Scissor per state
// change and can cost hardware fast paths, so it is opt-in per program. Baked into the
// pipeline (viewportCount is not dynamic without VK_EXT_extended_dynamic_state) and
// therefore hashed; the DYNAMIC viewport/scissor arrays the draw pushes must have
// exactly this many elements (VUID-vkCmdDraw-viewportCount-03417/-03418).
Uint32 viewportCount = 1;
VkPolygonMode polygonMode = VK_POLYGON_MODE_FILL;
VkCullModeFlags cullMode = VK_CULL_MODE_BACK_BIT;
VkFrontFace frontFace = VK_FRONT_FACE_CLOCKWISE;
@@ -8,6 +8,7 @@
#include "ProgramFactory.h"
#include "Config.h"
#include "MG_Backend/DirectVulkan/DirectVulkanResourceState.h"
#include "MG_Util/ShaderTranspiler/ShaderCompiler.h"
#include "MG_Util/ShaderTranspiler/SpvcSession.h"
@@ -1997,6 +1998,29 @@ namespace MobileGL::MG_Backend::DirectVulkan {
return ReflectedDeclaresInputBuiltin(reflectModule, SpvBuiltInBaseVertex);
}
// gl_ViewportIndex on the last pre-rasterization stage. glslang emits it natively for Vulkan
// (BuiltIn ViewportIndex plus OpCapability MultiViewport), and nothing in the SpirvPasses
// chain touches it, so a plain reflection of the declared output builtins is the whole test.
Bool ProgramFactory::ReflectedWritesViewportIndexBuiltin(const SpvReflectShaderModule& reflectModule) {
return ReflectedDeclaresOutputBuiltin(reflectModule, SpvBuiltInViewportIndex);
}
Bool ProgramFactory::ReflectedDeclaresOutputBuiltin(const SpvReflectShaderModule& reflectModule,
SpvBuiltIn builtin) {
for (Uint32 entryIndex = 0; entryIndex < reflectModule.entry_point_count; ++entryIndex) {
const SpvReflectEntryPoint& entryPoint = reflectModule.entry_points[entryIndex];
for (Uint32 variableIndex = 0; variableIndex < entryPoint.output_variable_count; ++variableIndex) {
const SpvReflectInterfaceVariable* variable = entryPoint.output_variables[variableIndex];
if (variable != nullptr &&
(variable->decoration_flags & SPV_REFLECT_DECORATION_BUILT_IN) != 0 &&
variable->built_in == builtin) {
return true;
}
}
}
return false;
}
Bool ProgramFactory::ReflectedDeclaresInputBuiltin(const SpvReflectShaderModule& reflectModule,
SpvBuiltIn builtin) {
for (Uint32 entryIndex = 0; entryIndex < reflectModule.entry_point_count; ++entryIndex) {
@@ -2339,6 +2363,46 @@ namespace MobileGL::MG_Backend::DirectVulkan {
}
}
// Which pre-rasterization stage assigns gl_ViewportIndex is not fixed: GL 4.1 allows only the
// geometry stage, ARB_shader_viewport_layer_array/GL 4.6 also the vertex and tessellation
// evaluation stages. Rather than guess which one is last, every non-fragment, non-compute
// module is asked - one writer anywhere means this program's draws need a multi-viewport
// pipeline, and a false positive costs only a wider viewportCount.
void ProgramFactory::ReflectViewportIndexUsage(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
const Vector<Vector<Uint>>& spirv,
VkProgramObject& entry) const {
entry.writesViewportIndexBuiltin = false;
for (SizeT moduleIndex = 0; moduleIndex < shaders.size() && moduleIndex < spirv.size(); ++moduleIndex) {
if (!shaders[moduleIndex]) continue;
const ShaderStage stage = shaders[moduleIndex]->GetShaderStage();
if (stage == ShaderStage::Fragment || stage == ShaderStage::Compute) continue;
const auto& module = spirv[moduleIndex];
if (module.empty()) continue;
SpvReflectShaderModule reflectModule{};
const SpvReflectResult createResult =
spvReflectCreateShaderModule(module.size() * sizeof(Uint), module.data(), &reflectModule);
if (createResult != SPV_REFLECT_RESULT_SUCCESS) {
// Fail toward the wide pipeline. Missing a real gl_ViewportIndex writer would
// silently collapse every viewport onto 0 (the exact bug this reflection exists
// to fix); over-declaring costs one extra viewport slot on a program that never
// uses it.
MGLOG_E_ONCE("ProgramFactory::ReflectViewportIndexUsage: reflection failed (result=%d); assuming the "
"program writes gl_ViewportIndex",
static_cast<Int>(createResult));
entry.writesViewportIndexBuiltin = true;
continue;
}
if (ReflectedWritesViewportIndexBuiltin(reflectModule)) {
entry.writesViewportIndexBuiltin = true;
}
spvReflectDestroyShaderModule(&reflectModule);
}
}
void ProgramFactory::ReflectFragmentOutputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
const Vector<Vector<Uint>>& spirv,
VkProgramObject& entry) const {
@@ -2910,8 +2974,73 @@ namespace MobileGL::MG_Backend::DirectVulkan {
bindings.push_back(layoutBinding);
}
// UPDATE_AFTER_BIND is strictly an optional per-layout acceleration. The GL
// descriptor model still resolves every sampler uniform element independently
// (including its texture-unit sampler-object override); selecting this path
// changes neither that resolution nor the set versioning in UniformManager.
// A conservative count keeps a layout on ordinary descriptors whenever any
// relevant update-after-bind limit is not large enough, rather than asking a
// driver to reject it during vkCreateDescriptorSetLayout.
Uint32 updateAfterBindSamplers = 0;
Uint32 updateAfterBindUniformBuffers = 0;
Uint32 updateAfterBindStorageBuffers = 0;
Uint32 updateAfterBindSampledImages = 0;
Uint32 updateAfterBindStorageImages = 0;
for (Uint32 binding = 0; binding < m_maxBindings; ++binding) {
const Uint32 count = entry.bindingDescriptorCounts[binding];
switch (entry.bindingKinds[binding]) {
case DescriptorBindingKind::UniformBufferDynamic:
updateAfterBindUniformBuffers += count;
break;
case DescriptorBindingKind::CombinedImageSampler:
updateAfterBindSamplers += count;
updateAfterBindSampledImages += count;
break;
case DescriptorBindingKind::UniformTexelBuffer:
updateAfterBindSampledImages += count;
break;
case DescriptorBindingKind::StorageBuffer:
case DescriptorBindingKind::StorageTexelBuffer:
updateAfterBindStorageBuffers += count;
break;
case DescriptorBindingKind::StorageImage:
updateAfterBindStorageImages += count;
break;
case DescriptorBindingKind::None:
break;
}
}
const Uint32 updateAfterBindResources = updateAfterBindUniformBuffers + updateAfterBindStorageBuffers +
updateAfterBindSampledImages + updateAfterBindStorageImages;
const auto& uab = m_updateAfterBindLimits;
entry.usesUpdateAfterBind =
uab.enabled && updateAfterBindSamplers <= uab.maxPerStageSamplers &&
updateAfterBindUniformBuffers <= uab.maxPerStageUniformBuffers &&
updateAfterBindStorageBuffers <= uab.maxPerStageStorageBuffers &&
updateAfterBindSampledImages <= uab.maxPerStageSampledImages &&
updateAfterBindStorageImages <= uab.maxPerStageStorageImages &&
updateAfterBindResources <= uab.maxPerStageResources &&
updateAfterBindSamplers <= uab.maxSetSamplers &&
updateAfterBindUniformBuffers <= uab.maxSetUniformBuffers &&
updateAfterBindUniformBuffers <= uab.maxSetUniformBuffersDynamic &&
updateAfterBindStorageBuffers <= uab.maxSetStorageBuffers &&
updateAfterBindStorageBuffers <= uab.maxSetStorageBuffersDynamic &&
updateAfterBindSampledImages <= uab.maxSetSampledImages &&
updateAfterBindStorageImages <= uab.maxSetStorageImages;
Vector<VkDescriptorBindingFlags> bindingFlags;
VkDescriptorSetLayoutBindingFlagsCreateInfo bindingFlagsInfo{};
if (entry.usesUpdateAfterBind) {
bindingFlags.assign(bindings.size(), VK_DESCRIPTOR_BINDING_UPDATE_AFTER_BIND_BIT);
bindingFlagsInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_BINDING_FLAGS_CREATE_INFO;
bindingFlagsInfo.bindingCount = static_cast<Uint32>(bindingFlags.size());
bindingFlagsInfo.pBindingFlags = bindingFlags.data();
}
VkDescriptorSetLayoutCreateInfo setLayoutInfo{};
setLayoutInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO;
setLayoutInfo.flags = entry.usesUpdateAfterBind ? VK_DESCRIPTOR_SET_LAYOUT_CREATE_UPDATE_AFTER_BIND_POOL_BIT : 0;
setLayoutInfo.pNext = entry.usesUpdateAfterBind ? &bindingFlagsInfo : nullptr;
setLayoutInfo.bindingCount = static_cast<Uint32>(bindings.size());
setLayoutInfo.pBindings = bindings.data();
VK_VERIFY(vkCreateDescriptorSetLayout(m_device, &setLayoutInfo, nullptr, &entry.descriptorSetLayout),
@@ -2991,6 +3120,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
auto& shaders = program.GetAttachedShaders();
auto& spirv = program.GetGeneratedSpirv();
Vector<Vector<Uint>> moduleSpirvs(spirv.size());
const Bool enableSpirvValidation = program.GetSpirvValidationEnabled();
if (enableSpirvValidation) {
MG_Util::ShaderTranspiler::ShaderCompiler::PrepareSpirvValidation();
}
const ShaderStage fixupStage = PickClipFixupStage(shaders);
@@ -3031,12 +3164,29 @@ namespace MobileGL::MG_Backend::DirectVulkan {
}
}
// NumSubgroups is defined by the local workgroup dimensions and SubgroupSize. Derive
// it in SPIR-V instead of trusting a driver builtin that can disagree with the
// SubgroupId topology produced by the same compute dispatch (Adreno reports 1 while
// emitting IDs 0..7 for a 512-invocation, 64-wide workgroup).
if (MG_Config::Features.NumSubgroupsQuirk && shaders[i] &&
shaders[i]->GetShaderStage() == ShaderStage::Compute) {
Vector<Uint> derivedNumSubgroupsSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::DeriveNumSubgroupsForVulkan(
moduleSpirvs[i], derivedNumSubgroupsSpirv, enableSpirvValidation)) {
moduleSpirvs[i] = std::move(derivedNumSubgroupsSpirv);
} else {
MGLOG_E("ProgramFactory: failed to derive gl_NumSubgroups for program %u; "
"compute shaders may observe a driver-inconsistent subgroup count",
program.GetExternalIndex());
}
}
// Vulkan's SPIR-V environment has no rectangle image dimension, so a
// GL_TEXTURE_RECTANGLE lookup has to become the 2D one the texture is really
// stored as - which addresses [0,1] where the application addressed texels.
{
Vector<Uint> rectLoweredSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(moduleSpirvs[i], rectLoweredSpirv) &&
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(moduleSpirvs[i], rectLoweredSpirv, enableSpirvValidation) &&
!rectLoweredSpirv.empty()) {
moduleSpirvs[i] = Move(rectLoweredSpirv);
}
@@ -3049,7 +3199,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
{
Vector<Uint> invariantSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::DecoratePositionInvariantForVulkan(
moduleSpirvs[i], invariantSpirv)) {
moduleSpirvs[i], invariantSpirv, enableSpirvValidation)) {
moduleSpirvs[i] = std::move(invariantSpirv);
} else {
// The pass round-trips through SPIRV-Tools IR, so an unparseable module
@@ -3073,7 +3223,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
m_shaderDrawParametersEnabled) {
Vector<Uint> rebasedSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::RebaseInstanceIndexForVulkan(moduleSpirvs[i],
rebasedSpirv)) {
rebasedSpirv, enableSpirvValidation)) {
moduleSpirvs[i] = std::move(rebasedSpirv);
} else {
MGLOG_E("ProgramFactory: failed to rebase gl_InstanceID for program %u; "
@@ -3091,7 +3241,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
(flags & CompileOptionBit::ZeroBaseVertex)) {
Vector<Uint> zeroedSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::ZeroBaseVertexForVulkan(moduleSpirvs[i],
zeroedSpirv)) {
zeroedSpirv, enableSpirvValidation)) {
moduleSpirvs[i] = std::move(zeroedSpirv);
} else {
// Failing open keeps the native builtin, which is the pre-fix behavior:
@@ -3114,7 +3264,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
if (shaders[i] && shaders[i]->GetShaderStage() == ShaderStage::Vertex) {
Vector<Uint> packedSpirv;
const Bool packOk = MG_Util::ShaderTranspiler::ShaderCompiler::PackDoubleVertexInputsForVulkan(
moduleSpirvs[i], packedSpirv);
moduleSpirvs[i], packedSpirv, enableSpirvValidation);
MOBILEGL_ASSERT(packOk,
"ProgramFactory: 64-bit vertex input packing failed for program %u; the "
"vertex-input format and the shader input type now disagree",
@@ -3138,7 +3288,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
if (m_unformattedFloatStorageImagesEnabled) {
Vector<Uint> unformattedSpirv;
if (MG_Util::ShaderTranspiler::ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(
moduleSpirvs[i], unformattedSpirv)) {
moduleSpirvs[i], unformattedSpirv, enableSpirvValidation)) {
moduleSpirvs[i] = std::move(unformattedSpirv);
} else {
MGLOG_E("ProgramFactory: failed to make float storage images unformatted for program %u",
@@ -3159,7 +3309,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
#else
// Final module the driver receives; also checked in the INFO-level CI/test
// lanes, where the DEBUG gate above is compiled out.
if (MG_Util::ShaderTranspiler::ShaderCompiler::SpirvValidationEnabled()) {
if (enableSpirvValidation) {
ValidateTransformedSpirv(moduleSpv, shaders[i]->GetShaderStage(), program.GetExternalIndex());
}
#endif
@@ -3189,6 +3339,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
ValidateRasterizationStageInterface(shaders, moduleSpirvs, entry, program.GetExternalIndex());
#endif
ReflectVertexInputs(shaders, moduleSpirvs, entry);
ReflectViewportIndexUsage(shaders, moduleSpirvs, entry);
ReflectFragmentOutputs(shaders, moduleSpirvs, entry);
ReflectPassthroughTessControlNeed(shaders, moduleSpirvs, entry);
ReflectLayout(program, moduleSpirvs, entry);
@@ -3235,14 +3386,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
const VkDescriptorSetLayout descriptorSetLayout = it->second.descriptorSetLayout;
MGLOG_D("ProgramFactory::OnFrameBoundary: evicting idle program entry hash=0x%llx",
static_cast<unsigned long long>(hash));
// erase runs ~VkProgramObject (modules/layouts destroyed); notify after
// so an observer never observes a half-destroyed entry through a lookup.
// Observers only need the handle values to purge their keyed caches.
++m_cacheStructureEpoch; // erase moves/kills entries: memoised pointers die
it = m_cache.erase(it);
// The observer destroys dependent pipelines and frees descriptor sets while
// this entry still owns its layout. Vulkan requires every descriptor set to be
// freed before its VkDescriptorSetLayout is destroyed.
if (m_evictionObserver != nullptr) {
m_evictionObserver->OnProgramEvicted(hash, descriptorSetLayout);
}
++m_cacheStructureEpoch; // erase moves/kills entries: memoised pointers die
it = m_cache.erase(it);
} else {
++it;
}
@@ -3376,7 +3527,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG
ValidateTransformedSpirv(spirv, ShaderStage::TessControl, 0);
#else
if (MG_Util::ShaderTranspiler::ShaderCompiler::SpirvValidationEnabled()) {
if (m_enableSpirvValidation) {
MG_Util::ShaderTranspiler::ShaderCompiler::PrepareSpirvValidation();
ValidateTransformedSpirv(spirv, ShaderStage::TessControl, 0);
}
#endif
@@ -76,6 +76,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
using CompileOptionFlags = Flags<CompileOptionBit>;
using HashType = Uint64;
struct UpdateAfterBindLimits {
Bool enabled = false;
Uint32 maxPerStageSamplers = 0;
Uint32 maxPerStageUniformBuffers = 0;
Uint32 maxPerStageStorageBuffers = 0;
Uint32 maxPerStageSampledImages = 0;
Uint32 maxPerStageStorageImages = 0;
Uint32 maxPerStageResources = 0;
Uint32 maxSetSamplers = 0;
Uint32 maxSetUniformBuffers = 0;
Uint32 maxSetUniformBuffersDynamic = 0;
Uint32 maxSetStorageBuffers = 0;
Uint32 maxSetStorageBuffersDynamic = 0;
Uint32 maxSetSampledImages = 0;
Uint32 maxSetStorageImages = 0;
};
struct VkProgramObject {
static constexpr Uint32 kMaxVertexInputLocations = 32;
@@ -88,6 +105,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// Layout data (previously in separate VkProgramLayout)
VkDescriptorSetLayout descriptorSetLayout = VK_NULL_HANDLE;
// True only when this layout passed every descriptor-indexing feature and
// update-after-bind limit gate at reflection time. It controls both the
// layout/binding flags and the pool class used by UniformManager.
Bool usesUpdateAfterBind = false;
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
Vector<DescriptorBindingKind> bindingKinds;
// The bindings this program actually declares, ascending. bindingKinds is sized to the
@@ -151,6 +172,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// PROGRAM rather than of the variant: the zeroed variant leaves the variable
// declared, so both variants answer the same and the draw path can ask either.
Bool readsBaseVertexBuiltin = false;
// Some pre-rasterization stage assigns gl_ViewportIndex. Its pipeline declares
// viewportCount = the renderer's rasterizable viewport count instead of 1, and its
// draws push the whole viewport/scissor array; every other program keeps the
// single-viewport fast path untouched. Part of the program's identity (folded into
// the pipeline hash through programHash), so no memo can serve the wrong shape.
Bool writesViewportIndexBuiltin = false;
// This program has a tessellation EVALUATION stage and no tessellation CONTROL
// stage. GL allows that (4.6 core 11.2.2: with no control shader the input patch
// is passed through unmodified, the output patch size is PATCH_VERTICES, and the
@@ -190,6 +217,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// a pipeline failure would be reported against the wrong SPIR-V.
stageSpirvDigests = std::move(other.stageSpirvDigests);
descriptorSetLayout = other.descriptorSetLayout;
usesUpdateAfterBind = other.usesUpdateAfterBind;
pipelineLayout = other.pipelineLayout;
bindingKinds = std::move(other.bindingKinds);
activeBindings = std::move(other.activeBindings);
@@ -218,11 +246,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
fragmentInputComponentCount = other.fragmentInputComponentCount;
fragmentReplacesDepth = other.fragmentReplacesDepth;
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
writesViewportIndexBuiltin = other.writesViewportIndexBuiltin;
needsPassthroughTessControl = other.needsPassthroughTessControl;
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
lastUsedFrame = other.lastUsedFrame;
other.hash = 0;
other.descriptorSetLayout = VK_NULL_HANDLE;
other.usesUpdateAfterBind = false;
other.pipelineLayout = VK_NULL_HANDLE;
other.hasStorageImages = false;
other.declinedDescriptors = false;
@@ -234,6 +264,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
other.fragmentInputComponentCount = 0;
other.fragmentReplacesDepth = false;
other.readsBaseVertexBuiltin = false;
other.writesViewportIndexBuiltin = false;
other.needsPassthroughTessControl = false;
other.passthroughTessControlEmulatable = false;
other.lastUsedFrame = 0;
@@ -248,6 +279,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
modules = std::move(other.modules);
stageSpirvDigests = std::move(other.stageSpirvDigests); // travels with `modules` - see the move ctor
descriptorSetLayout = other.descriptorSetLayout;
usesUpdateAfterBind = other.usesUpdateAfterBind;
pipelineLayout = other.pipelineLayout;
bindingKinds = std::move(other.bindingKinds);
activeBindings = std::move(other.activeBindings);
@@ -276,11 +308,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
fragmentInputComponentCount = other.fragmentInputComponentCount;
fragmentReplacesDepth = other.fragmentReplacesDepth;
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
writesViewportIndexBuiltin = other.writesViewportIndexBuiltin;
needsPassthroughTessControl = other.needsPassthroughTessControl;
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
lastUsedFrame = other.lastUsedFrame;
other.hash = 0;
other.descriptorSetLayout = VK_NULL_HANDLE;
other.usesUpdateAfterBind = false;
other.pipelineLayout = VK_NULL_HANDLE;
other.hasStorageImages = false;
other.declinedDescriptors = false;
@@ -292,6 +326,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
other.fragmentInputComponentCount = 0;
other.fragmentReplacesDepth = false;
other.readsBaseVertexBuiltin = false;
other.writesViewportIndexBuiltin = false;
other.needsPassthroughTessControl = false;
other.passthroughTessControlEmulatable = false;
other.lastUsedFrame = 0;
@@ -337,12 +372,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
virtual void OnProgramEvicted(HashType programHash, VkDescriptorSetLayout descriptorSetLayout) = 0;
};
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings = 16,
Bool shaderDrawParametersEnabled = false,
Bool unformattedFloatStorageImagesEnabled = false)
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings,
Bool shaderDrawParametersEnabled,
Bool unformattedFloatStorageImagesEnabled,
Bool enableSpirvValidation,
UpdateAfterBindLimits updateAfterBindLimits)
: m_device(device), m_maxBindings(maxBindings), m_config(config),
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled),
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled) {
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled),
m_enableSpirvValidation(enableSpirvValidation),
m_updateAfterBindLimits(updateAfterBindLimits) {
VkProgramObject::s_device = device;
}
// Destroys the pass-through tessellation control modules. Runs while the device is
@@ -400,6 +439,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// Shared by the two above: does any entry point list an input variable decorated with
// this builtin?
static Bool ReflectedDeclaresInputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
// True when an entry point writes the ViewportIndex builtin (gl_ViewportIndex), i.e. when
// the program can route primitives to a viewport other than 0 and its pipeline therefore
// has to declare more than one. Asks about OUTPUT variables because that is the direction
// a pre-rasterization stage declares it in.
static Bool ReflectedWritesViewportIndexBuiltin(const SpvReflectShaderModule& reflectModule);
static Bool ReflectedDeclaresOutputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
// The pass-through tessellation control stage GL 4.6 core 11.2.2 describes for a
// program that has an evaluation stage and no control stage, for an input patch of
@@ -434,6 +479,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
void ReflectVertexInputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
const Vector<Vector<Uint>>& spirv,
VkProgramObject& entry) const;
void ReflectViewportIndexUsage(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
const Vector<Vector<Uint>>& spirv,
VkProgramObject& entry) const;
void ReflectFragmentOutputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
const Vector<Vector<Uint>>& spirv,
VkProgramObject& entry) const;
@@ -456,6 +504,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// True only when the logical device enabled both
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
Bool m_unformattedFloatStorageImagesEnabled = false;
// Startup snapshot used only by internally synthesized shader modules, which do not
// originate from a ProgramLinkTask.
Bool m_enableSpirvValidation = false;
// Device feature and limit gate resolved before vkCreateDevice. Keeping it in
// the factory lets each reflected layout choose ordinary descriptors when its
// own counts would exceed the update-after-bind budget.
UpdateAfterBindLimits m_updateAfterBindLimits{};
// See SetDefaultFramebufferHeight. 0 means "not known yet"; the FragCoordYFlip bit is
// never set before the swapchain exists, so no variant can be compiled against it.
Uint32 m_defaultFramebufferHeight = 0;
@@ -156,13 +156,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
frame.descriptorPools.clear();
VkDescriptorPool initialPool = VK_NULL_HANDLE;
if (!CreateDescriptorPool(m_setsPerFrame, initialPool)) {
if (!CreateDescriptorPool(m_setsPerFrame, false, initialPool)) {
MGLOG_E_ONCE("UniformDescriptorBinder::Initialize failed: cannot create frame descriptor pool %u",
frameIndex);
Shutdown();
return false;
}
frame.descriptorPools.push_back({initialPool, m_setsPerFrame, 0});
frame.descriptorPools.push_back({initialPool, m_setsPerFrame, 0, false});
MGLOG_D("UniformDescriptorBinder: frame %u descriptor pool created (maxSets=%u)", frameIndex,
m_setsPerFrame);
}
@@ -305,7 +305,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// texture/sampler resolution, completeness probe, sync, layout handling, sampler
// and view lookups - would recompute the identical descriptor.
if (trustUnchangedHint && descriptorMemoUsable && binding < m_samplerResolveMemo.size() &&
m_samplerResolveMemo[binding].infoValid) {
m_samplerResolveMemo[binding].infoValid &&
m_samplerResolveMemo[binding].infoProgramLifetimeId == program.GetLifetimeId()) {
outImageInfo = m_samplerResolveMemo[binding].info;
return true;
}
@@ -504,6 +505,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
if (binding < m_samplerResolveMemo.size()) {
if (descriptorMemoUsable) {
m_samplerResolveMemo[binding].info = outImageInfo;
m_samplerResolveMemo[binding].infoProgramLifetimeId = program.GetLifetimeId();
m_samplerResolveMemo[binding].infoValid = true;
} else {
// An arrayed binding publishes nothing here, and clears what a previous program
@@ -540,11 +542,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
outImageInfo = {
.sampler = m_samplerManager->GetOrCreateSampler(*samplerBindingOverride.sampler,
*samplerBindingOverride.texture),
*samplerBindingOverride.texture,
samplerBindingOverride.forceNearestFiltering,
resource->sampledLevelCount),
.imageView = samplerBindingOverride.imageView != VK_NULL_HANDLE ?
samplerBindingOverride.imageView :
(resource->sampledView != VK_NULL_HANDLE ? resource->sampledView : resource->fullView),
.imageLayout = resource->layout,
.imageLayout = samplerBindingOverride.imageLayout != VK_IMAGE_LAYOUT_UNDEFINED ?
samplerBindingOverride.imageLayout : resource->layout,
};
return outImageInfo.sampler != VK_NULL_HANDLE;
}
@@ -1267,6 +1272,87 @@ namespace MobileGL::MG_Backend::DirectVulkan {
return true;
}
Bool UniformManager::SamplerOverlapsWritableImageSubresource(Int samplerBaseLevel, Int samplerMaxLevel,
GLint imageLevel, GLenum imageAccess) {
return imageAccess != GL_READ_ONLY && imageLevel >= samplerBaseLevel && imageLevel <= samplerMaxLevel;
}
Bool UniformManager::CollectSamplerImageFeedback(
const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj,
Vector<SamplerImageFeedbackBinding>& outBindings) const {
outBindings.clear();
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr,
"CollectSamplerImageFeedback: GL context is null");
if (programObj.declinedDescriptors) return true;
for (const Uint32 samplerBinding : programObj.activeBindings) {
if (samplerBinding >= m_maxBindings ||
programObj.bindingKinds[samplerBinding] != ProgramFactory::DescriptorBindingKind::CombinedImageSampler) {
continue;
}
const Uint32 samplerCount = BindingDescriptorCount(programObj, samplerBinding);
for (Uint32 samplerElement = 0; samplerElement < samplerCount; ++samplerElement) {
MG_State::GLState::ITextureObject* sampledTexture = nullptr;
const MG_State::GLState::SamplerObject* sampledSampler = nullptr;
if (!ResolveSampledBinding(program, programObj, samplerBinding, samplerElement,
sampledTexture, sampledSampler) ||
sampledTexture == nullptr || sampledSampler == nullptr ||
MG_State::GLState::SamplesAsIncompleteTexture(sampledTexture, sampledSampler)) {
// ResolveSamplerDescriptor uses a fallback in these cases, which cannot
// alias the image-unit binding of the original texture.
continue;
}
// Multisample source images intentionally omit TRANSFER_SRC usage. Keep their existing
// direct binding instead of turning otherwise valid sampler2DMS/image2DMS dispatches
// into failed dispatches; a correct snapshot for them needs a same-sample-count path.
const TextureTarget sampledTarget = sampledTexture->GetTarget();
if (sampledTarget == TextureTarget::Texture2DMultisample ||
sampledTarget == TextureTarget::Texture2DMultisampleArray) {
continue;
}
const auto& levelRange = sampledTexture->GetLevelRange();
Bool aliasesWritableImage = false;
for (const Uint32 imageBinding : programObj.activeBindings) {
if (imageBinding >= m_maxBindings ||
programObj.bindingKinds[imageBinding] != ProgramFactory::DescriptorBindingKind::StorageImage) {
continue;
}
if (imageBinding >= programObj.samplerUniformLocationByBinding.size()) return false;
const Int baseLocation = programObj.samplerUniformLocationByBinding[imageBinding];
if (baseLocation < 0) return false;
const Uint32 imageCount = BindingDescriptorCount(programObj, imageBinding);
for (Uint32 imageElement = 0; imageElement < imageCount; ++imageElement) {
const Int location = ResolveDescriptorElementLocation(program, baseLocation, imageElement);
if (location < 0) return false;
const Int imageUnit = program.GetUniformSamplerOrImageUnitIndex(static_cast<Uint>(location));
if (imageUnit < 0 || imageUnit >= MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS) {
return false;
}
const auto& image = MG_State::pGLContext->GetImageTextureBinding(imageUnit);
// A sampler view exposes all layers of its target; equal texture plus an
// overlapping mip therefore aliases the writable image subresource.
if (image.Texture.get() == sampledTexture &&
SamplerOverlapsWritableImageSubresource(levelRange.x(), levelRange.y(),
image.Level, image.Access)) {
aliasesWritableImage = true;
break;
}
}
if (aliasesWritableImage) break;
}
if (aliasesWritableImage) {
outBindings.push_back({.samplerBinding = samplerBinding,
.samplerElement = samplerElement,
.texture = sampledTexture,
.sampler = sampledSampler,
.numericDomain = programObj.samplerNumericDomainByBinding[samplerBinding]});
}
}
}
return true;
}
Bool UniformManager::ResolveUniformBufferPayload(const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
Uint32 arrayElement, UboBindResult& out) const {
@@ -1388,7 +1474,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
return true;
}
Bool UniformManager::CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const {
Bool UniformManager::CreateDescriptorPool(Uint32 maxSets, Bool updateAfterBind, VkDescriptorPool& outPool) const {
outPool = VK_NULL_HANDLE;
if (m_device == VK_NULL_HANDLE || maxSets == 0 || m_maxBindings == 0) {
return false;
@@ -1431,7 +1517,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// (OnDescriptorSetLayoutDestroyed) so program churn recycles pool capacity.
// The cost is on set allocation only, which happens when a layout's per-frame
// cache grows - never on the per-draw reuse path.
poolInfo.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT;
poolInfo.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT |
(updateAfterBind ? VK_DESCRIPTOR_POOL_CREATE_UPDATE_AFTER_BIND_BIT : 0);
poolInfo.maxSets = maxSets;
poolInfo.poolSizeCount = static_cast<Uint32>(std::size(poolSizes));
poolInfo.pPoolSizes = poolSizes;
@@ -1445,24 +1532,28 @@ namespace MobileGL::MG_Backend::DirectVulkan {
return true;
}
Bool UniformManager::GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex) {
Bool UniformManager::GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex, Bool updateAfterBind) {
if (frame.descriptorPools.empty()) {
return false;
}
const auto& currentBucket = frame.descriptorPools[frame.activeDescriptorPoolIndex];
const Uint32 currentMaxSets = std::max<Uint32>(1, currentBucket.maxSets);
const auto matchingBucket = std::find_if(
frame.descriptorPools.begin(), frame.descriptorPools.end(),
[updateAfterBind](const DescriptorPoolBucket& candidate) { return candidate.updateAfterBind == updateAfterBind; });
const Uint32 currentMaxSets = matchingBucket != frame.descriptorPools.end()
? std::max<Uint32>(1, matchingBucket->maxSets)
: m_setsPerFrame;
const Uint32 grownMaxSets = currentMaxSets <= (std::numeric_limits<Uint32>::max() / 2) ? (currentMaxSets * 2)
: currentMaxSets;
VkDescriptorPool grownPool = VK_NULL_HANDLE;
if (!CreateDescriptorPool(grownMaxSets, grownPool)) {
if (!CreateDescriptorPool(grownMaxSets, updateAfterBind, grownPool)) {
MGLOG_E_ONCE("UniformDescriptorBinder::GrowFrameDescriptorPool failed: cannot create grown pool (%u -> %u sets)",
currentMaxSets, grownMaxSets);
return false;
}
frame.descriptorPools.push_back({grownPool, grownMaxSets, 0});
frame.descriptorPools.push_back({grownPool, grownMaxSets, 0, updateAfterBind});
frame.activeDescriptorPoolIndex = static_cast<Uint32>(frame.descriptorPools.size() - 1);
MGLOG_D(
"UniformDescriptorBinder: frame %u descriptor pool exhausted, grew pool (%u -> %u sets), poolCount=%zu",
@@ -1472,14 +1563,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
VkResult UniformManager::AllocateDescriptorSetsFromActivePool(Uint32 frameIndex, const ProgramFactory::VkProgramObject& programObj, VkDescriptorSet& outDescriptorSet) {
auto& frame = m_frames[frameIndex];
if (frame.activeDescriptorPoolIndex >= frame.descriptorPools.size()) {
frame.activeDescriptorPoolIndex = 0;
}
if (frame.descriptorPools[frame.activeDescriptorPoolIndex].allocatedSets >=
frame.descriptorPools[frame.activeDescriptorPoolIndex].maxSets) {
const Bool updateAfterBind = programObj.usesUpdateAfterBind;
if (frame.activeDescriptorPoolIndex >= frame.descriptorPools.size() ||
frame.descriptorPools[frame.activeDescriptorPoolIndex].updateAfterBind != updateAfterBind ||
frame.descriptorPools[frame.activeDescriptorPoolIndex].allocatedSets >=
frame.descriptorPools[frame.activeDescriptorPoolIndex].maxSets) {
const auto availableBucket = std::find_if(
frame.descriptorPools.begin(), frame.descriptorPools.end(),
[](const DescriptorPoolBucket& candidate) { return candidate.allocatedSets < candidate.maxSets; });
[updateAfterBind](const DescriptorPoolBucket& candidate) {
return candidate.updateAfterBind == updateAfterBind && candidate.allocatedSets < candidate.maxSets;
});
if (availableBucket == frame.descriptorPools.end()) {
outDescriptorSet = VK_NULL_HANDLE;
return VK_ERROR_OUT_OF_POOL_MEMORY;
@@ -1515,7 +1608,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
} else {
VkResult allocResult = AllocateDescriptorSetsFromActivePool(frameIndex, programObj, outDescriptorSet);
if (allocResult == VK_ERROR_OUT_OF_POOL_MEMORY || allocResult == VK_ERROR_FRAGMENTED_POOL) {
if (!GrowFrameDescriptorPool(frame, frameIndex)) {
if (!GrowFrameDescriptorPool(frame, frameIndex, programObj.usesUpdateAfterBind)) {
MGLOG_E_ONCE("UniformDescriptorBinder::AcquireDescriptorSet failed: descriptor pool growth failed");
return allocResult;
}
@@ -1633,7 +1726,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
Uint32 frameIndex,
VkPipelineBindPoint bindPoint,
const SamplerBindingOverride* samplerBindingOverride,
Bool samplerDescriptorsUnchangedHint) {
Bool samplerDescriptorsUnchangedHint,
const Vector<SamplerBindingOverride>* samplerBindingOverrides) {
// This program has a descriptor MobileGL could not resolve (see
// VkProgramObject::declinedDescriptors). Refusing here is the whole of the decline: the
// binding is still declared in the layout, so the pipeline is consistent with the shader
@@ -1660,7 +1754,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// sampler binding, and an unchanged (buffer, range) for the single
// dynamic UBO covers the rest - except the dynamic offset, which rebinding
// the SAME set delivers without any descriptor write.
const Bool cacheable = (samplerBindingOverride == nullptr);
const Bool cacheable = samplerBindingOverride == nullptr &&
(samplerBindingOverrides == nullptr || samplerBindingOverrides->empty());
if (cacheable && samplerDescriptorsUnchangedHint && m_fastRebindMemo.valid &&
m_fastRebindMemo.frameIndex == frameIndex &&
m_fastRebindMemo.programLifetimeId == program.GetLifetimeId() &&
@@ -1884,13 +1979,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
const SizeT firstImageInfoIndex = imageInfos.size();
for (Uint32 element = 0; element < descriptorCount; ++element) {
VkDescriptorImageInfo imageInfo{};
Bool hasImage = false;
if (overrideThisBinding && element == 0) {
hasImage = ResolveSamplerDescriptorOverride(*samplerBindingOverride, imageInfo);
} else {
hasImage = ResolveSamplerDescriptor(commandBuffer, program, programObj, binding, element,
imageInfo, samplerDescriptorsUnchangedHint);
const SamplerBindingOverride* overrideForElement =
overrideThisBinding && element == 0 ? samplerBindingOverride : nullptr;
if (overrideForElement == nullptr && samplerBindingOverrides != nullptr) {
const auto overrideIt = std::find_if(
samplerBindingOverrides->begin(), samplerBindingOverrides->end(),
[binding, element](const SamplerBindingOverride& candidate) {
return candidate.binding == binding && candidate.element == element;
});
if (overrideIt != samplerBindingOverrides->end()) {
overrideForElement = &*overrideIt;
}
}
const Bool hasImage = overrideForElement != nullptr
? ResolveSamplerDescriptorOverride(*overrideForElement, imageInfo)
: ResolveSamplerDescriptor(commandBuffer, program, programObj, binding,
element, imageInfo,
samplerDescriptorsUnchangedHint);
if (!hasImage) {
MGLOG_E_ONCE(
"UniformDescriptorBinder::BindProgramUniformBuffers failed: sampler binding %u element %u "
@@ -26,9 +26,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
public:
struct SamplerBindingOverride {
Uint32 binding = 0;
Uint32 element = 0;
MG_State::GLState::ITextureObject* texture = nullptr;
const MG_State::GLState::SamplerObject* sampler = nullptr;
VkImageView imageView = VK_NULL_HANDLE;
VkImageLayout imageLayout = VK_IMAGE_LAYOUT_UNDEFINED;
Bool forceNearestFiltering = false;
};
struct SamplerImageFeedbackBinding {
Uint32 samplerBinding = 0;
Uint32 samplerElement = 0;
MG_State::GLState::ITextureObject* texture = nullptr;
const MG_State::GLState::SamplerObject* sampler = nullptr;
SamplerNumericDomain numericDomain = SamplerNumericDomain::Unknown;
};
Bool Initialize(VkDevice device, VkBufferManager* bufferManager,
@@ -79,6 +90,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
Bool CollectStorageImageTextures(const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj,
Vector<MG_State::GLState::ITextureObject*>& outTextures) const;
Bool CollectSamplerImageFeedback(
const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj,
Vector<SamplerImageFeedbackBinding>& outBindings) const;
static Bool SamplerOverlapsWritableImageSubresource(Int samplerBaseLevel, Int samplerMaxLevel,
GLint imageLevel, GLenum imageAccess);
// samplerDescriptorsUnchangedHint: the caller (SetupDraw fast path) proved that
// every input of every combined-image-sampler resolution is unchanged since the
// previous draw's resolve - same (texture, sampler) per binding, texture params
@@ -91,7 +108,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
Uint32 frameIndex,
VkPipelineBindPoint bindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS,
const SamplerBindingOverride* samplerBindingOverride = nullptr,
Bool samplerDescriptorsUnchangedHint = false);
Bool samplerDescriptorsUnchangedHint = false,
const Vector<SamplerBindingOverride>* samplerBindingOverrides = nullptr);
// Pure format-policy helper kept public for host regression tests. Formatted storage
// images use their shader qualifier; transformed float images use glBindImageTexture's
@@ -114,6 +132,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
VkDescriptorPool handle = VK_NULL_HANDLE;
Uint32 maxSets = 0;
Uint32 allocatedSets = 0;
Bool updateAfterBind = false;
};
// A cached descriptor set together with the pool it was allocated from, so a
@@ -223,8 +242,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
void BindDescriptorSetDeduped(VkCommandBuffer commandBuffer, VkPipelineBindPoint bindPoint,
VkPipelineLayout pipelineLayout, VkDescriptorSet descriptorSet,
const Vector<Uint32>& dynamicOffsets);
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
Bool CreateDescriptorPool(Uint32 maxSets, Bool updateAfterBind, VkDescriptorPool& outPool) const;
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex, Bool updateAfterBind);
VkResult AllocateDescriptorSetsFromActivePool(
Uint32 frameIndex, const ProgramFactory::VkProgramObject& programObj, VkDescriptorSet& outDescriptorSet);
VkResult AcquireDescriptorSet(Uint32 frameIndex,
@@ -341,8 +360,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// lifetime id, so a freed-and-reallocated sampler or texture at the same heap address
// always gets a fresh id and misses (a raw pointer would false-hit that ABA) - so a
// stale guess can only miss and fall through to the hash, never resolve wrong. Still
// reset each frame alongside the descriptor-set cache. Indexed by binding.
// reset each frame alongside the descriptor-set cache. Indexed by binding, but the
// whole-descriptor entry is additionally keyed by program lifetime: Vulkan binding
// numbers are layout-local and unrelated programs routinely reuse binding 0/1.
struct SamplerResolveMemo {
Uint64 infoProgramLifetimeId = 0;
Uint64 samplerLifetimeId = 0;
Uint64 textureLifetimeId = 0;
VkSampler sampler = VK_NULL_HANDLE;
@@ -300,8 +300,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
Bool ok = VkTextureManager::TransitionImageLayout(
commandBuffer, newResource.image, newResource.layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT,
0, VK_ACCESS_TRANSFER_WRITE_BIT, newResource.aspect, 0, newResource.mipLevels,
newResource.arrayLayers);
0, VK_ACCESS_TRANSFER_WRITE_BIT, newResource.aspect, 0, newResource.mipLevels);
MOBILEGL_ASSERT(ok, "PreserveTextureContentsOnRecreate: failed to prepare destination image");
VkImageLayout srcTrackedLayout = oldResource.layout;
@@ -311,8 +310,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
ok = VkTextureManager::TransitionImageLayout(
commandBuffer, oldResource.image, srcTrackedLayout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
srcStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT,
srcAccessMask, VK_ACCESS_TRANSFER_READ_BIT, oldResource.aspect, 0, preservedMipLevels,
oldResource.arrayLayers);
srcAccessMask, VK_ACCESS_TRANSFER_READ_BIT, oldResource.aspect, 0, preservedMipLevels);
MOBILEGL_ASSERT(ok, "PreserveTextureContentsOnRecreate: failed to prepare source image");
Vector<VkImageCopy> copyRegions;
@@ -344,8 +342,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
ok = VkTextureManager::TransitionImageLayout(
commandBuffer, newResource.image, newResource.layout, oldResource.layout,
VK_PIPELINE_STAGE_TRANSFER_BIT, dstStageMask,
VK_ACCESS_TRANSFER_WRITE_BIT, dstAccessMask, newResource.aspect, 0, newResource.mipLevels,
newResource.arrayLayers);
VK_ACCESS_TRANSFER_WRITE_BIT, dstAccessMask, newResource.aspect, 0, newResource.mipLevels);
MOBILEGL_ASSERT(ok, "PreserveTextureContentsOnRecreate: failed to restore destination layout");
VK_VERIFY(vkEndCommandBuffer(commandBuffer), "vkEndCommandBuffer(texture preserve)");
@@ -1191,7 +1188,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
const Bool lowerTransitioned = TransitionImageLayout(
commandBuffer, resource.image, lowerMipLayout, newLayout,
srcStageMask, dstStageMask, srcAccessMask, dstAccessMask,
resource.aspect, 0, writtenMipLevel, resource.arrayLayers);
resource.aspect, 0, writtenMipLevel);
MOBILEGL_ASSERT(lowerTransitioned,
"UpdateTrackedImageLayoutAfterAttachmentWrite: failed to transition lower mip levels for textureId=%d",
texture->GetExternalIndex());
@@ -1203,8 +1200,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
const Bool upperTransitioned = TransitionImageLayout(
commandBuffer, resource.image, upperMipLayout, newLayout,
srcStageMask, dstStageMask, srcAccessMask, dstAccessMask,
resource.aspect, upperBaseMipLevel, resource.mipLevels - upperBaseMipLevel,
resource.arrayLayers);
resource.aspect, upperBaseMipLevel, resource.mipLevels - upperBaseMipLevel);
MOBILEGL_ASSERT(upperTransitioned,
"UpdateTrackedImageLayoutAfterAttachmentWrite: failed to transition upper mip levels for textureId=%d",
texture->GetExternalIndex());
@@ -1257,8 +1253,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
const Bool ok = TransitionImageLayout(commandBuffer, resource->image, resource->layout, targetLayout, srcStageMask,
s_sampledReadStages, srcAccessMask,
VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels,
resource->arrayLayers);
VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels);
MOBILEGL_ASSERT(ok, "TransitionTextureForSampling: transition failed for textureId=%d", texture.GetExternalIndex());
// Pre-pass stream bookkeeping: a command referencing the image was recorded.
StampResourceRecordingUse(*resource);
@@ -1288,7 +1283,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
VK_IMAGE_LAYOUT_GENERAL, srcStageMask,
VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, srcAccessMask,
VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT,
resource->aspect, 0, resource->mipLevels, resource->arrayLayers);
resource->aspect, 0, resource->mipLevels);
MOBILEGL_ASSERT(ok, "TransitionTextureForStorageImage: transition failed for textureId=%d",
texture.GetExternalIndex());
// Pre-pass stream bookkeeping: a command referencing the image was recorded.
@@ -1296,6 +1291,158 @@ namespace MobileGL::MG_Backend::DirectVulkan {
return ok;
}
Bool VkTextureManager::SnapshotTextureForSampling(VkCommandBuffer commandBuffer,
MG_State::GLState::ITextureObject& texture,
SamplerNumericDomain numericDomain,
VkPipelineStageFlags consumerShaderStageMask,
SampledTextureSnapshot& outSnapshot) {
outSnapshot = {};
TextureResource* source = SyncTextureAndGetDescriptor(texture);
if (source == nullptr || source->image == VK_NULL_HANDLE || source->sampleCount != VK_SAMPLE_COUNT_1_BIT ||
source->sampledLevelCount == 0) {
return false;
}
const VkFormat sampledFormat = ResolveSampledImageViewFormat(source->format, numericDomain);
if (sampledFormat == VK_FORMAT_UNDEFINED ||
!AreSampledImageViewFormatsCompatible(source->format, sampledFormat)) {
MGLOG_E_ONCE("SnapshotTextureForSampling: textureId=%d cannot create sampled view format=%d from image format=%d",
texture.GetExternalIndex(), static_cast<Int>(sampledFormat), static_cast<Int>(source->format));
return false;
}
if (sampledFormat != source->format &&
(source->imageCreateFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) == 0) {
MGLOG_E_ONCE("SnapshotTextureForSampling: textureId=%d needs unavailable mutable image format=%d for sampled view=%d",
texture.GetExternalIndex(), static_cast<Int>(source->format), static_cast<Int>(sampledFormat));
return false;
}
VkImageType imageType = VK_IMAGE_TYPE_2D;
switch (source->viewType) {
case VK_IMAGE_VIEW_TYPE_1D:
case VK_IMAGE_VIEW_TYPE_1D_ARRAY:
imageType = VK_IMAGE_TYPE_1D;
break;
case VK_IMAGE_VIEW_TYPE_3D:
imageType = VK_IMAGE_TYPE_3D;
break;
default:
break;
}
TextureResource snapshot{};
VkImageCreateInfo imageInfo{};
imageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
imageInfo.flags = source->imageCreateFlags;
imageInfo.imageType = imageType;
imageInfo.extent = {source->extent.width, source->extent.height, source->depth};
imageInfo.mipLevels = source->mipLevels;
imageInfo.arrayLayers = source->arrayLayers;
imageInfo.format = source->format;
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
imageInfo.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT;
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT;
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
// Keep the temporary's view-format list just as narrow as the source's sampler use. This
// has no storage-image usage, so unlike an app image binding the exact list is knowable.
Vector<VkFormat> viewFormats;
VkImageFormatListCreateInfo formatListInfo{};
if (m_imageFormatListSupported && (imageInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) {
viewFormats.push_back(source->format);
if (sampledFormat != source->format) {
viewFormats.push_back(sampledFormat);
}
formatListInfo.sType = VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO;
formatListInfo.viewFormatCount = static_cast<Uint32>(viewFormats.size());
formatListInfo.pViewFormats = viewFormats.data();
imageInfo.pNext = &formatListInfo;
}
VmaAllocationCreateInfo allocationInfo{};
allocationInfo.usage = VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE;
allocationInfo.requiredFlags = VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
const VkResult createResult =
vmaCreateImage(m_allocator, &imageInfo, &allocationInfo, &snapshot.image, &snapshot.allocation, nullptr);
if (createResult != VK_SUCCESS) {
MGLOG_E_ONCE("SnapshotTextureForSampling: vmaCreateImage failed result=%d textureId=%d", createResult,
texture.GetExternalIndex());
return false;
}
snapshot.extent = source->extent;
snapshot.depth = source->depth;
snapshot.arrayLayers = source->arrayLayers;
snapshot.mipLevels = source->mipLevels;
snapshot.sampledBaseMipLevel = source->sampledBaseMipLevel;
snapshot.sampledLevelCount = source->sampledLevelCount;
snapshot.format = source->format;
snapshot.aspect = source->aspect;
snapshot.viewType = source->viewType;
snapshot.sampleCount = VK_SAMPLE_COUNT_1_BIT;
snapshot.imageCreateFlags = imageInfo.flags;
snapshot.usageFlags = imageInfo.usage;
const TextureFormatInfo formatInfo = ResolveTextureFormatInfo(texture.GetFormat());
const VkComponentMapping sampledComponents = ResolveSampledViewComponents(texture, formatInfo);
const VkImageAspectFlags sampledAspect =
ResolveSampledImageViewAspectMask(snapshot.aspect, texture.GetDepthStencilTextureMode());
snapshot.sampledView = CreateImageView(snapshot.image, sampledFormat, sampledAspect, snapshot.viewType,
snapshot.sampledBaseMipLevel, snapshot.sampledLevelCount, 0,
snapshot.arrayLayers, &sampledComponents);
if (snapshot.sampledView == VK_NULL_HANDLE) {
MGLOG_E_ONCE("SnapshotTextureForSampling: failed to create sampled view textureId=%d", texture.GetExternalIndex());
return false;
}
VkPipelineStageFlags sourceStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
VkAccessFlags sourceAccessMask = 0;
const VkImageLayout sourceLayout = source->layout;
GetImageTransitionSourceState(sourceLayout, sourceStageMask, sourceAccessMask);
if (!TransitionImageLayout(commandBuffer, source->image, source->layout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
sourceStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT, sourceAccessMask,
VK_ACCESS_TRANSFER_READ_BIT, source->aspect, 0, source->mipLevels) ||
!TransitionImageLayout(commandBuffer, snapshot.image, snapshot.layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0,
VK_ACCESS_TRANSFER_WRITE_BIT, snapshot.aspect, snapshot.sampledBaseMipLevel,
snapshot.sampledLevelCount)) {
return false;
}
Vector<VkImageCopy> copyRegions;
copyRegions.reserve(snapshot.sampledLevelCount);
for (Uint32 level = snapshot.sampledBaseMipLevel;
level < snapshot.sampledBaseMipLevel + snapshot.sampledLevelCount; ++level) {
VkImageCopy copy{};
copy.srcSubresource = {source->aspect, level, 0, source->arrayLayers};
copy.dstSubresource = {snapshot.aspect, level, 0, snapshot.arrayLayers};
copy.extent = {std::max(source->extent.width >> level, 1u),
std::max(source->extent.height >> level, 1u),
std::max(source->depth >> level, 1u)};
copyRegions.push_back(copy);
}
vkCmdCopyImage(commandBuffer, source->image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, snapshot.image,
VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, static_cast<Uint32>(copyRegions.size()), copyRegions.data());
if (!TransitionImageLayout(commandBuffer, snapshot.image, snapshot.layout,
ResolveSampledReadOnlyLayout(snapshot.aspect), VK_PIPELINE_STAGE_TRANSFER_BIT,
consumerShaderStageMask, VK_ACCESS_TRANSFER_WRITE_BIT,
VK_ACCESS_SHADER_READ_BIT, snapshot.aspect, snapshot.sampledBaseMipLevel,
snapshot.sampledLevelCount) ||
!TransitionImageLayout(commandBuffer, source->image, source->layout, sourceLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, consumerShaderStageMask,
VK_ACCESS_TRANSFER_READ_BIT, VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT,
source->aspect, 0, source->mipLevels)) {
return false;
}
StampResourceRecordingUse(*source);
outSnapshot = {.imageView = snapshot.sampledView, .layout = snapshot.layout};
DeferResourceRelease(Move(snapshot));
return true;
}
void VkTextureManager::MarkStorageImageTexture(MG_State::GLState::ITextureObject& texture) {
m_storageImageTextures.insert(MakeTextureIdentity(&texture));
}
@@ -1355,8 +1502,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
VkImageLayout& trackedLayout, VkImageLayout newLayout,
VkPipelineStageFlags srcStageMask, VkPipelineStageFlags dstStageMask,
VkAccessFlags srcAccessMask, VkAccessFlags dstAccessMask,
VkImageAspectFlags aspectMask, Uint32 baseMipLevel, Uint32 levelCount,
Uint32 layerCount) {
VkImageAspectFlags aspectMask, Uint32 baseMipLevel,
Uint32 levelCount) {
MOBILEGL_ASSERT(image != VK_NULL_HANDLE, "TransitionImageLayout: m_image == VK_NULL_HANDLE");
MOBILEGL_ASSERT(!((dstAccessMask & VK_ACCESS_TRANSFER_READ_BIT) != 0 &&
(dstStageMask & VK_PIPELINE_STAGE_TRANSFER_BIT) == 0),
@@ -1381,7 +1528,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
barrier.subresourceRange.baseMipLevel = baseMipLevel;
barrier.subresourceRange.levelCount = levelCount;
barrier.subresourceRange.baseArrayLayer = 0;
barrier.subresourceRange.layerCount = layerCount;
// Every layer, always - see the declaration for why layout tracking leaves no other
// correct answer. VK_REMAINING_ARRAY_LAYERS rather than the image's own `arrayLayers`
// because those are not the same number for a 3D image: MobileGL creates 3D images
// 2D_ARRAY_COMPATIBLE and their arrayLayers is 1, which today Vulkan reads as "all depth
// slices" but will read as "depth slice 0" once VK_KHR_maintenance9 is enabled. The
// validation layer warns about that literal 1 by name.
barrier.subresourceRange.layerCount = VK_REMAINING_ARRAY_LAYERS;
vkCmdPipelineBarrier(commandBuffer, srcStageMask, dstStageMask, 0, 0, nullptr, 0, nullptr, 1, &barrier);
trackedLayout = newLayout;
@@ -2605,7 +2758,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
VK_PIPELINE_STAGE_TRANSFER_BIT,
uploadSrcAccessMask,
VK_ACCESS_TRANSFER_WRITE_BIT,
aspectMask, 0, outResource.mipLevels, outResource.arrayLayers);
aspectMask, 0, outResource.mipLevels);
MOBILEGL_ASSERT(ok, "TransitionImageLayout to VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL failed");
// Array textures keep their GL "depth" in VkImage array layers, so the
@@ -2709,7 +2862,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
s_sampledReadStages,
VK_ACCESS_TRANSFER_WRITE_BIT,
VK_ACCESS_SHADER_READ_BIT,
aspectMask, 0, outResource.mipLevels, outResource.arrayLayers);
aspectMask, 0, outResource.mipLevels);
MOBILEGL_ASSERT(ok, "TransitionImageLayout to sampled read-only layout failed");
outResource.layout = finalLayout;
@@ -310,6 +310,11 @@ public:
static inline VmaAllocator s_allocator = VK_NULL_HANDLE;
};
struct SampledTextureSnapshot {
VkImageView imageView = VK_NULL_HANDLE;
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
};
Bool Initialize(const InitInfo& initInfo);
void Shutdown();
void BeginFrame(Uint32 frameIndex);
@@ -343,6 +348,13 @@ public:
VkImageLayout newLayout);
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
// Copies the complete sampler-visible mip range into a transient sampled image. The source is
// restored to its prior layout, so image-store descriptors continue to name the original image.
// The transient ownership is tied to the current frame slot and is safe through its submission.
Bool SnapshotTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture,
SamplerNumericDomain numericDomain,
VkPipelineStageFlags consumerShaderStageMask,
SampledTextureSnapshot& outSnapshot);
// Recording-generation bookkeeping for the pre-pass command stream. The
// generation advances every time the frame command buffer (re)begins
@@ -388,12 +400,24 @@ public:
static Bool AreSampledImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
static Bool AreStorageImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
// Moves `image` to `newLayout` and writes the new layout back through `trackedLayout`.
//
// The barrier covers EVERY array layer of the image, and there is deliberately no layer
// parameter to say otherwise: layout here is tracked per IMAGE (one `TextureResource::layout`,
// or one caller-owned variable), so a barrier narrower than the image would leave the layers it
// skipped in the old layout while the tracker claims they moved. Every transfer against a
// framebuffer attachment above layer 0 - glReadPixels, glBlitFramebuffer, glCopyTexSubImage,
// glCopyImageSubData - then ran its copy on a layer no barrier had transitioned.
//
// The mip range IS a parameter, because mip levels really are transitioned piecewise (see
// UpdateTrackedImageLayoutAfterAttachmentWrite and the mipmap generation loops): those callers
// move the complement of the level they wrote so the whole image converges on one layout again.
// Nothing does, or can, do that per layer.
static Bool TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image, VkImageLayout& trackedLayout,
VkImageLayout newLayout, VkPipelineStageFlags srcStageMask,
VkPipelineStageFlags dstStageMask, VkAccessFlags srcAccessMask,
VkAccessFlags dstAccessMask, VkImageAspectFlags aspectMask,
Uint32 baseMipLevel = 0, Uint32 levelCount = 1,
Uint32 layerCount = 1);
Uint32 baseMipLevel = 0, Uint32 levelCount = 1);
SizeT CollectGarbage();
@@ -238,11 +238,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// nothing else across all of gl33.
//
// The mapping below is derived from - and at full extent exactly reproduces - the pixel
// mapping RemapDefaultFboReadbackToGLOrientation has always used:
// mapping VulkanRenderer::RemapDefaultFramebufferReadback uses:
// identity : image(x, H-1-y) -> flip Y
// 180 : image(W-1-x, y) -> mirror X (the rotation already flips the rows)
// Quarter turns swap the axes; nothing in this renderer models that (the readback declines to
// remap them and the viewport path only rescales), so they are left exactly as they were.
// Quarter turns swap the axes and are handled by MapDefaultFramebufferReadbackRect rather than
// this same-axis helper.
struct DefaultFramebufferRectMapping {
Bool flipY = false;
Bool mirrorX = false;
@@ -335,19 +335,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// Complete input inventory of ApplyDynamicDrawStateTail, one line per reader
// (each accessor it replaces is a verified plain field read of the same
// RenderStateParameters field - RenderState.cpp):
// ApplyGLViewportState : Viewport, DepthRange, + extent/isDefaultFbo/preTransform
// ApplyGLViewportState : Viewports[0], DepthRanges[0], + extent/isDefaultFbo/preTransform
// ApplyBlendConstants : BlendColor
// ApplyPolygonOffsetState : PolygonOffsetUnits, PolygonOffsetFactor
// ApplyLineWidthState : LineWidth (see the caveat below)
// ApplyStencilState : StencilStates[0..1].{ValueMask, WriteMask, Ref}
// scissor rect : ScissorTestEnabled, ScissorBox,
// scissor rect : ScissorTestEnabledMask bit 0, ScissorBoxes[0],
// + extent/isDefaultFbo/preTransform
// Caveat, unchanged from the version-only gate: ApplyLineWidthState also clamps
// to the ACTIVE BACKEND OBJECT's aliased line-width range. Those are device
// limits queried once at backend init and constant for the renderer's lifetime,
// so they are not part of the key (the version gate never covered them either).
struct DynamicTailKey {
Int viewport[4] = {0, 0, 0, 0};
Float viewport[4] = {0.0f, 0.0f, 0.0f, 0.0f};
Float depthRange[2] = {0.0f, 0.0f};
Float blendColor[4] = {0.0f, 0.0f, 0.0f, 0.0f};
Float polygonOffsetFactor = 0.0f;
@@ -437,12 +437,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
vkCmdSetScissor(commandBuffer, 0, 1, &scissor);
}
static void ApplyGLViewportState(VkCommandBuffer commandBuffer,
const IntVec2& framebufferExtent,
VkSurfaceTransformFlagBitsKHR preTransform,
Bool isDefaultFramebuffer) {
const IntVec4& viewportState = MG_State::pGLContext->GetViewport();
const FloatVec2& depthRange = MG_State::pGLContext->GetDepthRange();
// One viewport of the ARB_viewport_array state, mapped into Vulkan's frame. Split out of
// ApplyGLViewportState so the multi-viewport path derives index i through EXACTLY the same
// arithmetic as index 0 - the default-framebuffer Y-flip and pre-transform rotation
// especially, which is the classic way a multi-viewport port comes out upside down for every
// index but the one that was tested.
static VkViewport ComputeGLViewport(Uint32 index,
const IntVec2& framebufferExtent,
VkSurfaceTransformFlagBitsKHR preTransform,
Bool isDefaultFramebuffer) {
// Snapped to integers. The viewport is float STATE (glViewportIndexedf may set a
// fractional origin, and GetFloati_v hands it back verbatim), but what rasterizes here is
// the rounded rectangle - a deliberate, documented infidelity rather than a spec claim:
// MobileGL passes the driver's VIEWPORT_SUBPIXEL_BITS through, so it does advertise
// subpixel viewport precision it does not deliver. Nothing in KHR-GL43.viewport_array or
// in Minecraft sets a fractional viewport (the conformance checks are all on the state
// round trip), which is why the honest-but-lossy path was kept over widening every
// default-framebuffer Y-flip/pre-transform helper to floats. See the KNOWN INFIDELITY
// note in MG_IntegrationTest/Scenarios/AdvertisedLimitsScenario.cpp.
const FloatVec4& stored = MG_State::pGLContext->GetViewportIndexed(index);
const IntVec4 viewportState(static_cast<Int>(std::lround(stored.x())),
static_cast<Int>(std::lround(stored.y())),
static_cast<Int>(std::lround(stored.z())),
static_cast<Int>(std::lround(stored.w())));
const FloatVec2& depthRange = MG_State::pGLContext->GetDepthRangeIndexed(index);
const IntVec2 logicalExtent = isDefaultFramebuffer
? ResolveDefaultFramebufferLogicalExtent(preTransform, framebufferExtent)
: framebufferExtent;
@@ -477,6 +495,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
viewport.height = static_cast<float>(viewportHeight);
viewport.minDepth = depthRange.x();
viewport.maxDepth = depthRange.y();
return viewport;
}
static void ApplyGLViewportState(VkCommandBuffer commandBuffer,
const IntVec2& framebufferExtent,
VkSurfaceTransformFlagBitsKHR preTransform,
Bool isDefaultFramebuffer) {
const VkViewport viewport = ComputeGLViewport(0, framebufferExtent, preTransform, isDefaultFramebuffer);
auto& shadow = g_dynamicStateShadow;
if (shadow.viewportValid && shadow.viewport.x == viewport.x && shadow.viewport.y == viewport.y &&
shadow.viewport.width == viewport.width && shadow.viewport.height == viewport.height &&
@@ -1454,7 +1480,8 @@ void main() {
const char* label = nullptr;
};
static Uint32 ComputeMaxProgramBindings(const VkPhysicalDeviceProperties& properties) {
static Uint32 ComputeMaxProgramBindings(const VkPhysicalDeviceProperties& properties,
const ProgramFactory::UpdateAfterBindLimits& updateAfterBindLimits) {
const auto& limits = properties.limits;
static constexpr Uint32 kMinProgramBindings = 16;
static constexpr Uint32 kMaxProgramBindingsCap = 256;
@@ -1469,6 +1496,21 @@ void main() {
maxBindings = std::min(maxBindings, maxCombinedImageSamplers);
maxBindings = std::min(maxBindings, maxSampledImages + maxDynamicUniformBuffers);
if (updateAfterBindLimits.enabled) {
const Uint32 updateAfterBindSamplers = std::min(updateAfterBindLimits.maxPerStageSamplers,
updateAfterBindLimits.maxSetSamplers);
const Uint32 updateAfterBindSampledImages = std::min(updateAfterBindLimits.maxPerStageSampledImages,
updateAfterBindLimits.maxSetSampledImages);
const Uint32 updateAfterBindDynamicUniformBuffers =
std::min(updateAfterBindLimits.maxPerStageUniformBuffers,
updateAfterBindLimits.maxSetUniformBuffersDynamic);
Uint32 updateAfterBindBindings = updateAfterBindLimits.maxPerStageResources;
updateAfterBindBindings = std::min(updateAfterBindBindings, updateAfterBindSamplers);
updateAfterBindBindings =
std::min(updateAfterBindBindings, updateAfterBindSampledImages + updateAfterBindDynamicUniformBuffers);
maxBindings = std::max(maxBindings, updateAfterBindBindings);
}
maxBindings = std::max(kMinProgramBindings, maxBindings);
maxBindings = std::min(kMaxProgramBindingsCap, maxBindings);
return maxBindings;
@@ -2125,47 +2167,6 @@ void main() {
return static_cast<Uint8>(value * 255.0f + 0.5f);
}
// Re-order the copied BLOCK - not the whole image - from the default framebuffer's stored
// orientation into GL's. The caller has already aimed the copy at the right place with
// MapDefaultFramebufferRectAxis, so what arrives here is exactly the requested
// rectWidth x rectHeight rect, and all that is left is the order of rows (identity) or of
// columns (180) WITHIN it.
//
// This used to iterate the full swapchain extent and index both sides with that stride,
// which is why its caller could only use it on an exact full-extent read - and why every
// partial glReadPixels of the default framebuffer came back in Vulkan row order. Only
// identity/180 share the swapchain extent with the default framebuffer; 90/270 swap
// extents and are still declined.
static Bool RemapDefaultFboReadbackToGLOrientation(const Uint8* rawPixels,
Uint32 rectWidth,
Uint32 rectHeight,
VkSurfaceTransformFlagBitsKHR preTransform,
SizeT texelSize,
Uint8* outPixels) {
if (IsQuarterTurnPreTransform(preTransform)) {
return false;
}
if (rectWidth == 0 || rectHeight == 0 || texelSize == 0) {
return false;
}
const DefaultFramebufferRectMapping mapping = GetDefaultFramebufferRectMapping(preTransform);
const SizeT rowBytes = static_cast<SizeT>(rectWidth) * texelSize;
for (Uint32 outY = 0; outY < rectHeight; ++outY) {
const Uint32 srcY = mapping.flipY ? (rectHeight - 1 - outY) : outY;
const Uint8* srcRow = rawPixels + static_cast<SizeT>(srcY) * rowBytes;
Uint8* dstRow = outPixels + static_cast<SizeT>(outY) * rowBytes;
if (!mapping.mirrorX) {
Memcpy(dstRow, srcRow, rowBytes);
continue;
}
for (Uint32 outX = 0; outX < rectWidth; ++outX) {
Memcpy(dstRow + static_cast<SizeT>(outX) * texelSize,
srcRow + static_cast<SizeT>(rectWidth - 1 - outX) * texelSize, texelSize);
}
}
return true;
}
static SizeT AlignPixelRow(SizeT rowBytes, Int alignment) {
const SizeT resolvedAlignment = static_cast<SizeT>(std::max(alignment, 1));
return (rowBytes + resolvedAlignment - 1) & ~(resolvedAlignment - 1);
@@ -2712,6 +2713,95 @@ void main() {
return formatInfo.texel_block_size;
}
Bool VulkanRenderer::MapDefaultFramebufferReadbackRect(
GLint x, GLint y, GLsizei width, GLsizei height, VkExtent2D imageExtent,
VkSurfaceTransformFlagBitsKHR preTransform, VkOffset2D* imageOffset,
VkExtent2D* imageCopyExtent) {
if (width <= 0 || height <= 0 || imageOffset == nullptr || imageCopyExtent == nullptr) {
return false;
}
const Int imageWidth = static_cast<Int>(imageExtent.width);
const Int imageHeight = static_cast<Int>(imageExtent.height);
Int mappedX = x;
Int mappedY = y;
Uint32 mappedWidth = static_cast<Uint32>(width);
Uint32 mappedHeight = static_cast<Uint32>(height);
// InsertPositionFixup first flips GL Y and then applies the surface transform. In pixel
// coordinates that gives these half-open rectangle mappings into the stored image:
// identity: (x, H-y-h), 90: (y, x), 180: (W-x-w, y), 270: (H-y-h, W-x-w).
// Quarter turns also transpose the copied block's extent.
switch (preTransform) {
case VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR:
mappedX = y;
mappedY = x;
mappedWidth = static_cast<Uint32>(height);
mappedHeight = static_cast<Uint32>(width);
break;
case VK_SURFACE_TRANSFORM_ROTATE_180_BIT_KHR:
mappedX = imageWidth - x - width;
mappedY = y;
break;
case VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR:
mappedX = imageWidth - y - height;
mappedY = imageHeight - x - width;
mappedWidth = static_cast<Uint32>(height);
mappedHeight = static_cast<Uint32>(width);
break;
default:
mappedY = imageHeight - y - height;
break;
}
if (mappedX < 0 || mappedY < 0 || mappedWidth > imageExtent.width ||
mappedHeight > imageExtent.height ||
static_cast<Uint64>(mappedX) + mappedWidth > imageExtent.width ||
static_cast<Uint64>(mappedY) + mappedHeight > imageExtent.height) {
return false;
}
*imageOffset = {mappedX, mappedY};
*imageCopyExtent = {mappedWidth, mappedHeight};
return true;
}
Bool VulkanRenderer::RemapDefaultFramebufferReadback(
const Uint8* rawPixels, Uint32 logicalWidth, Uint32 logicalHeight,
VkSurfaceTransformFlagBitsKHR preTransform, SizeT texelSize, Uint8* outPixels) {
if (rawPixels == nullptr || outPixels == nullptr || logicalWidth == 0 || logicalHeight == 0 ||
texelSize == 0) {
return false;
}
const Uint32 rawWidth = IsQuarterTurnPreTransform(preTransform) ? logicalHeight : logicalWidth;
for (Uint32 outY = 0; outY < logicalHeight; ++outY) {
for (Uint32 outX = 0; outX < logicalWidth; ++outX) {
Uint32 srcX = outX;
Uint32 srcY = outY;
switch (preTransform) {
case VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR:
srcX = outY;
srcY = outX;
break;
case VK_SURFACE_TRANSFORM_ROTATE_180_BIT_KHR:
srcX = logicalWidth - 1 - outX;
break;
case VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR:
srcX = logicalHeight - 1 - outY;
srcY = logicalWidth - 1 - outX;
break;
default:
srcY = logicalHeight - 1 - outY;
break;
}
Memcpy(outPixels + (static_cast<SizeT>(outY) * logicalWidth + outX) * texelSize,
rawPixels + (static_cast<SizeT>(srcY) * rawWidth + srcX) * texelSize,
texelSize);
}
}
return true;
}
Bool VulkanRenderer::ConvertReadbackPixels(const Uint8* sourcePixels, VkFormat sourceFormat,
GLsizei width, GLsizei height, GLenum destinationFormat,
GLenum destinationType, SizeT destinationRowStride,
@@ -2936,7 +3026,7 @@ void main() {
succeeded = m_renderPassManager->Initialize();
MOBILEGL_ASSERT(succeeded, "VkRenderPassManager initialization failed.");
const Uint32 maxProgramBindings = ComputeMaxProgramBindings(m_physicalDevice.properties);
const Uint32 maxProgramBindings = ComputeMaxProgramBindings(m_physicalDevice.properties, m_updateAfterBindLimits);
MGLOG_I("DirectVulkan: using %u program descriptor bindings", maxProgramBindings);
if (IsPowerVRDevice(m_physicalDevice.properties)) {
m_config.DisablePipelineCache = true;
@@ -2970,7 +3060,9 @@ void main() {
}
m_programFactory = MakeUnique<ProgramFactory>(m_device, m_config, maxProgramBindings,
m_shaderDrawParametersFeatureEnabled,
m_unformattedFloatStorageImagesEnabled);
m_unformattedFloatStorageImagesEnabled,
MG_Config::Features.EnableSpirvValidation,
m_updateAfterBindLimits);
MOBILEGL_ASSERT(m_programFactory != nullptr, "ProgramFactory creation failed.");
// The swapchain already exists at this point (Initialize creates it first), so seed the
// height the factory could not be told about from CreateSwapchain.
@@ -4879,6 +4971,7 @@ void main() {
.topology = vkTopology,
.primitiveRestartEnable = primitiveRestartEnabled,
.patchControlPoints = static_cast<Uint32>(MG_State::pGLContext->GetPatchVertices()),
.viewportCount = ResolveDrawViewportCount(programObj.writesViewportIndexBuiltin),
.polygonMode = effectivePolygonMode,
.cullMode = cullFaceEnabled
? MG_Util::ConvertCullFaceModeToVkEnum(MG_State::pGLContext->GetCullFaceMode(), invertClockwise)
@@ -5319,11 +5412,147 @@ void main() {
}
return true;
}
Bool VulkanRenderer::PrepareSamplerImageFeedbackSnapshots(
FrameContext::FrameData& frame,
const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj,
VkPipelineStageFlags consumerShaderStageMask) {
auto& feedbackBindings = m_samplerImageFeedbackScratch;
auto& overrides = m_samplerImageBindingOverridesScratch;
overrides.clear();
if (!programObj.hasStorageImages) {
feedbackBindings.clear();
return true;
}
if (!m_uniformManager->CollectSamplerImageFeedback(program, programObj, feedbackBindings)) {
MGLOG_E_ONCE("%s: failed to collect sampler/image feedback for program=%u", __func__,
program.GetExternalIndex());
return false;
}
if (feedbackBindings.empty()) {
return true;
}
// Copy and layout barriers cannot be recorded inside a render pass. A graphics draw only
// gets here after an actual sampled/writable-image mip overlap was found, so ordinary
// graphics draws retain the active pass.
if (VkRenderPassManager::GetActiveRenderPass() != nullptr) {
VkRenderPassManager::EndRenderPass(frame.commandBuffer);
}
struct SnapshotCacheEntry {
MG_State::GLState::ITextureObject* texture = nullptr;
SamplerNumericDomain numericDomain = SamplerNumericDomain::Unknown;
VkTextureManager::SampledTextureSnapshot snapshot{};
};
Vector<SnapshotCacheEntry> snapshotCache;
snapshotCache.reserve(feedbackBindings.size());
overrides.reserve(feedbackBindings.size());
for (const auto& feedback : feedbackBindings) {
VkTextureManager::SampledTextureSnapshot snapshot{};
const auto existing = std::find_if(
snapshotCache.begin(), snapshotCache.end(), [&feedback](const SnapshotCacheEntry& candidate) {
return candidate.texture == feedback.texture && candidate.numericDomain == feedback.numericDomain;
});
if (existing != snapshotCache.end()) {
snapshot = existing->snapshot;
} else {
if (!m_textureManager->SnapshotTextureForSampling(frame.commandBuffer, *feedback.texture,
feedback.numericDomain, consumerShaderStageMask,
snapshot) ||
snapshot.imageView == VK_NULL_HANDLE) {
MGLOG_E_ONCE("%s: failed to snapshot textureId=%d for sampler binding=%u element=%u", __func__,
feedback.texture != nullptr ? feedback.texture->GetExternalIndex() : 0,
feedback.samplerBinding, feedback.samplerElement);
return false;
}
snapshotCache.push_back({.texture = feedback.texture,
.numericDomain = feedback.numericDomain,
.snapshot = snapshot});
}
overrides.push_back({
.binding = feedback.samplerBinding,
.element = feedback.samplerElement,
.texture = feedback.texture,
.sampler = feedback.sampler,
.imageView = snapshot.imageView,
.imageLayout = snapshot.layout,
.forceNearestFiltering = feedback.numericDomain == SamplerNumericDomain::SignedInteger ||
feedback.numericDomain == SamplerNumericDomain::UnsignedInteger,
});
if (program.GetExternalIndex() == 194 && feedback.texture->GetExternalIndex() == 75) {
MGLOG_D_ONCE("sampler/image feedback snapshot: program=194 texture=75 binding=%u element=%u view=%p",
feedback.samplerBinding, feedback.samplerElement, snapshot.imageView);
}
}
return true;
}
// The scissor rectangle Vulkan needs for ARB_viewport_array index `index`. Vulkan has no
// per-viewport scissor-test TOGGLE - a scissor rectangle always applies - so an index whose
// GL scissor test is disabled gets the whole framebuffer, which is exactly "the test always
// passes" (GL 4.6 core 17.3.2).
VkRect2D VulkanRenderer::ComputeGLScissorRect(Uint32 index, const IntVec2& extent,
VkSurfaceTransformFlagBitsKHR preTransform,
Bool isDefaultFbo) const {
const auto& parameters = MG_State::pGLContext->GetRenderStateParameters();
if ((parameters.ScissorTestEnabledMask & (1u << index)) == 0) {
VkRect2D full{};
full.offset = {0, 0};
full.extent = {static_cast<Uint32>(extent.x()), static_cast<Uint32>(extent.y())};
return full;
}
const IntVec4& scissorBox = parameters.ScissorBoxes[index];
return isDefaultFbo ? MakeDefaultFramebufferScissorRect(scissorBox, extent, preTransform)
: MakeClampedScissorRect(scissorBox, extent);
}
// The wide half of ApplyDynamicDrawStateTail: a pipeline built for a gl_ViewportIndex-writing
// program declares viewportCount > 1, and Vulkan then requires that many viewports AND that
// many scissors to have been set before the draw
// (VUID-vkCmdDraw-viewportCount-03417/-03418). Deliberately unmemoized: only conformance
// shaders reach it, the single-element dynamic-state shadow cannot describe an array, and
// leaving that shadow invalidated is what makes the next ordinary draw re-push its own
// single viewport instead of believing the array's element 0 is already bound.
void VulkanRenderer::ApplyMultiViewportDynamicState(VkCommandBuffer commandBuffer, Uint32 viewportCount,
const IntVec2& extent,
VkSurfaceTransformFlagBitsKHR preTransform,
Bool isDefaultFbo) {
MOBILEGL_ASSERT(viewportCount <= RenderStateParameters::MAX_VIEWPORTS,
"ApplyMultiViewportDynamicState: viewportCount=%u exceeds the indexed state width",
viewportCount);
const Uint32 count = std::min<Uint32>(viewportCount, RenderStateParameters::MAX_VIEWPORTS);
Array<VkViewport, RenderStateParameters::MAX_VIEWPORTS> viewports{};
Array<VkRect2D, RenderStateParameters::MAX_VIEWPORTS> scissors{};
for (Uint32 i = 0; i < count; ++i) {
viewports[i] = ComputeGLViewport(i, extent, preTransform, isDefaultFbo);
scissors[i] = ComputeGLScissorRect(i, extent, preTransform, isDefaultFbo);
}
vkCmdSetViewport(commandBuffer, 0, count, viewports.data());
vkCmdSetScissor(commandBuffer, 0, count, scissors.data());
auto& shadow = g_dynamicStateShadow;
shadow.viewportValid = false;
shadow.scissorValid = false;
shadow.dynamicTailValid = false;
}
void VulkanRenderer::ApplyDynamicDrawStateTail(FrameContext::FrameData& frame, const IntVec2& extent,
Bool isDefaultFbo) {
Bool isDefaultFbo, Uint32 viewportCount) {
auto& shadow = g_dynamicStateShadow;
if (viewportCount > 1) {
// The other five Apply* still run: blend constants, depth bias, line width and the
// stencil masks are not per-viewport and a multi-viewport draw needs them just as
// much. Only the viewport/scissor pair takes the array shape.
ApplyBlendConstants(frame.commandBuffer);
ApplyPolygonOffsetState(frame.commandBuffer);
ApplyLineWidthState(frame.commandBuffer);
ApplyStencilState(frame.commandBuffer);
ApplyMultiViewportDynamicState(frame.commandBuffer, viewportCount, extent,
m_swapchainObject.GetPreTransform(), isDefaultFbo);
return;
}
// One compare for the whole tail: see the gate's declaration in
// DynamicStateShadow for why (version, extent, default-FBO flag) pins every
// input the six Apply* below read.
@@ -5342,12 +5571,14 @@ void main() {
DynamicStateShadow::DynamicTailKey key;
{
const RenderStateParameters& p = MG_State::pGLContext->GetRenderStateParameters();
key.viewport[0] = p.Viewport.x();
key.viewport[1] = p.Viewport.y();
key.viewport[2] = p.Viewport.z();
key.viewport[3] = p.Viewport.w();
key.depthRange[0] = p.DepthRange.x();
key.depthRange[1] = p.DepthRange.y();
// Viewport 0 and its depth range: ApplyGLViewportState reads exactly those two
// (per-index state for indices > 0 is keyed separately, see multiViewportKey below).
key.viewport[0] = p.Viewports[0].x();
key.viewport[1] = p.Viewports[0].y();
key.viewport[2] = p.Viewports[0].z();
key.viewport[3] = p.Viewports[0].w();
key.depthRange[0] = p.DepthRanges[0].x();
key.depthRange[1] = p.DepthRanges[0].y();
key.blendColor[0] = p.BlendColor.x();
key.blendColor[1] = p.BlendColor.y();
key.blendColor[2] = p.BlendColor.z();
@@ -5362,11 +5593,11 @@ void main() {
key.stencilWriteMask[face] = p.StencilStates[face].WriteMask;
key.stencilRef[face] = p.StencilStates[face].Ref;
}
key.scissorEnabled = p.ScissorTestEnabled;
key.scissorBox[0] = p.ScissorBox.x();
key.scissorBox[1] = p.ScissorBox.y();
key.scissorBox[2] = p.ScissorBox.z();
key.scissorBox[3] = p.ScissorBox.w();
key.scissorEnabled = (p.ScissorTestEnabledMask & 1u) != 0;
key.scissorBox[0] = p.ScissorBoxes[0].x();
key.scissorBox[1] = p.ScissorBoxes[0].y();
key.scissorBox[2] = p.ScissorBoxes[0].z();
key.scissorBox[3] = p.ScissorBoxes[0].w();
key.extentX = extent.x();
key.extentY = extent.y();
key.preTransform = static_cast<Uint32>(preTransform);
@@ -5759,7 +5990,7 @@ void main() {
const Bool idxUploadOk = UploadAndBindIndexBuffer(frame, vao, pIndexBufferView);
MOBILEGL_ASSERT(idxUploadOk, "SetupDraw fast path: failed to upload index buffer");
}
ApplyDynamicDrawStateTail(frame, snap.renderPassExtent, snap.drawFboIsDefault);
ApplyDynamicDrawStateTail(frame, snap.renderPassExtent, snap.drawFboIsDefault, snap.viewportCount);
return true;
}
@@ -5937,6 +6168,11 @@ void main() {
MGLOG_E_ONCE("SetupDraw skipped: storage image preparation failed");
return false;
}
if (!PrepareSamplerImageFeedbackSnapshots(frame, program, programObj,
VK_PIPELINE_STAGE_ALL_GRAPHICS_BIT)) {
MGLOG_E_ONCE("SetupDraw skipped: sampler/image feedback snapshot failed");
return false;
}
auto* activeRenderPass = VkRenderPassManager::GetActiveRenderPass();
@@ -6181,7 +6417,9 @@ void main() {
}
const Bool boundUniforms = m_uniformManager->BindProgramUniformBuffers(
frame.commandBuffer, program, programObj, m_frameContext.GetCurrentFrameIndex());
frame.commandBuffer, program, programObj, m_frameContext.GetCurrentFrameIndex(),
VK_PIPELINE_BIND_POINT_GRAPHICS, nullptr, false,
m_samplerImageBindingOverridesScratch.empty() ? nullptr : &m_samplerImageBindingOverridesScratch);
if (!boundUniforms) {
MGLOG_E_ONCE("SetupDraw skipped: BindProgramUniformBuffers failed");
return false;
@@ -6199,7 +6437,8 @@ void main() {
MOBILEGL_ASSERT(idxUploadOk, "SetupDraw skipped: failed to upload index buffer");
}
ApplyDynamicDrawStateTail(frame, renderPassEntry->extent, drawFbo->IsDefaultFramebuffer());
ApplyDynamicDrawStateTail(frame, renderPassEntry->extent, drawFbo->IsDefaultFramebuffer(),
ResolveDrawViewportCount(programObj.writesViewportIndexBuiltin));
// Snapshot the fully resolved configuration for the consecutive-draw
// fast path (see TrySetupDrawFastPath).
@@ -6218,6 +6457,7 @@ void main() {
snap.drawFbo = drawFbo.get();
snap.fboVersion = drawFbo->GetObjectVersion();
snap.drawFboIsDefault = drawFboIsDefault;
snap.viewportCount = ResolveDrawViewportCount(programObj.writesViewportIndexBuiltin);
snap.renderStateVersion = MG_State::pGLContext->GetPipelineStateVersion();
snap.bindGeneration = MG_State::pGLContext->GetTextureBindGeneration();
snap.baseTransformFlags = GetBaseTransformFlagsRaw(drawFboIsDefault);
@@ -6302,6 +6542,11 @@ void main() {
MGLOG_E_ONCE("DispatchCompute skipped: storage image preparation failed");
return;
}
if (!PrepareSamplerImageFeedbackSnapshots(frame, program, programObj,
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT)) {
MGLOG_E_ONCE("DispatchCompute skipped: sampler/image feedback snapshot failed");
return;
}
const VkPipeline pipeline = GetOrCreateComputePipeline(programObj);
if (pipeline == VK_NULL_HANDLE) {
@@ -6313,7 +6558,8 @@ void main() {
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_COMPUTE, pipeline);
const Bool boundUniforms = m_uniformManager->BindProgramUniformBuffers(
frame.commandBuffer, program, programObj, m_frameContext.GetCurrentFrameIndex(),
VK_PIPELINE_BIND_POINT_COMPUTE);
VK_PIPELINE_BIND_POINT_COMPUTE, nullptr, false,
m_samplerImageBindingOverridesScratch.empty() ? nullptr : &m_samplerImageBindingOverridesScratch);
if (!boundUniforms) {
MGLOG_E_ONCE("DispatchCompute skipped: BindProgramUniformBuffers failed");
return;
@@ -6348,6 +6594,11 @@ void main() {
MGLOG_E_ONCE("DispatchComputeIndirect skipped: storage image preparation failed");
return;
}
if (!PrepareSamplerImageFeedbackSnapshots(frame, program, programObj,
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT)) {
MGLOG_E_ONCE("DispatchComputeIndirect skipped: sampler/image feedback snapshot failed");
return;
}
const VkPipeline pipeline = GetOrCreateComputePipeline(programObj);
if (pipeline == VK_NULL_HANDLE) {
@@ -6359,7 +6610,8 @@ void main() {
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_COMPUTE, pipeline);
const Bool boundUniforms = m_uniformManager->BindProgramUniformBuffers(
frame.commandBuffer, program, programObj, m_frameContext.GetCurrentFrameIndex(),
VK_PIPELINE_BIND_POINT_COMPUTE);
VK_PIPELINE_BIND_POINT_COMPUTE, nullptr, false,
m_samplerImageBindingOverridesScratch.empty() ? nullptr : &m_samplerImageBindingOverridesScratch);
if (!boundUniforms) {
MGLOG_E_ONCE("DispatchComputeIndirect skipped: BindProgramUniformBuffers failed");
return;
@@ -7113,7 +7365,7 @@ void main() {
Bool ok = VkTextureManager::TransitionImageLayout(
commandBuffer, resource->image, resource->layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
srcStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT, srcAccessMask, VK_ACCESS_TRANSFER_WRITE_BIT,
resource->aspect, 0, resource->mipLevels, resource->arrayLayers);
resource->aspect, 0, resource->mipLevels);
MOBILEGL_ASSERT(ok,
"MaterializePendingClearForTexture: failed to transition textureId=%d to TRANSFER_DST",
texture.GetExternalIndex());
@@ -7231,8 +7483,7 @@ void main() {
ok = VkTextureManager::TransitionImageLayout(
commandBuffer, resource->image, clearLayout, sampledLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_ALL_GRAPHICS_BIT,
VK_ACCESS_TRANSFER_WRITE_BIT, VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels,
resource->arrayLayers);
VK_ACCESS_TRANSFER_WRITE_BIT, VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels);
MOBILEGL_ASSERT(ok,
"MaterializePendingClearForTexture: failed to transition textureId=%d to sampled layout",
texture.GetExternalIndex());
@@ -7270,7 +7521,7 @@ void main() {
Bool ok = VkTextureManager::TransitionImageLayout(
commandBuffer, resource->image, resource->layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
srcStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT, srcAccessMask, VK_ACCESS_TRANSFER_WRITE_BIT,
resource->aspect, 0, 1, 1);
resource->aspect, 0, 1);
MOBILEGL_ASSERT(ok,
"MaterializePendingClearForRenderbuffer: failed to transition renderbuffer %u to TRANSFER_DST",
renderbuffer->GetExternalIndex());
@@ -7321,7 +7572,7 @@ void main() {
VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT |
VK_ACCESS_TRANSFER_READ_BIT,
resource->aspect, 0, 1, 1);
resource->aspect, 0, 1);
MOBILEGL_ASSERT(ok,
"MaterializePendingClearForRenderbuffer: failed to transition renderbuffer %u to steady layout",
renderbuffer->GetExternalIndex());
@@ -7928,6 +8179,9 @@ void main() {
VkPipelineStageFlags srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
VkAccessFlags srcAccessMask = 0;
GetImageTransitionSourceState(srcOriginalLayout, srcStageMask, srcAccessMask);
// Both blit regions below name `baseArrayLayer` from their binding, and a layered depth
// attachment puts that above 0. These barriers carry a mip range only - their layer
// range is every layer (see VkTextureManager::TransitionImageLayout).
if (readIsDefaultFbo) {
VkImageLayout srcTrackedLayout = srcOriginalLayout;
Bool ok = VkTextureManager::TransitionImageLayout(
@@ -8755,30 +9009,22 @@ void main() {
VkAccessFlags srcAccessMask = 0;
GetImageTransitionSourceState(srcOriginalLayout, srcStageMask, srcAccessMask);
VkImageLayout srcCopyLayout = srcOriginalLayout;
// The barrier has to name every layer the copy touches, not just layer 0 - otherwise the
// slice fix above lands the copy on layers the barrier never transitioned, which is the
// same defect one level down. TransitionImageLayout always starts its range at
// baseArrayLayer 0, so VK_REMAINING_ARRAY_LAYERS is the whole range and a superset of
// [baseSlice, baseSlice + depth).
//
// Not `arrayLayers`, which is 1 for a 3D image: MobileGL creates 3D images
// 2D_ARRAY_COMPATIBLE, and a literal 1 on one of those means "every depth slice" today but
// "depth slice 0" once VK_KHR_maintenance9 is enabled - i.e. it would silently become a
// single-slice barrier again on a newer driver. The validation layer says so by name.
static constexpr Uint32 kAllLayers = VK_REMAINING_ARRAY_LAYERS;
// The barriers below name a MIP range only. Their layer range is not a parameter:
// TransitionImageLayout always covers every layer of the image, which is a superset of the
// [baseSlice, baseSlice + depth) the slice mapping above hands the copy.
if (srcOriginalLayout == VK_IMAGE_LAYOUT_UNDEFINED) {
Bool srcReady = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, srcResource->image, srcResource->layout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
srcStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT,
srcAccessMask, VK_ACCESS_TRANSFER_READ_BIT,
srcResource->aspect, 0, srcResource->mipLevels, kAllLayers);
srcResource->aspect, 0, srcResource->mipLevels);
MOBILEGL_ASSERT(srcReady, "%s: failed to transition undefined source image", __func__);
srcCopyLayout = srcResource->layout;
} else {
Bool srcReady = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, srcResource->image, srcCopyLayout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
srcStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT,
srcAccessMask, VK_ACCESS_TRANSFER_READ_BIT, copyAspectMask, srcMipLevel, 1, kAllLayers);
srcAccessMask, VK_ACCESS_TRANSFER_READ_BIT, copyAspectMask, srcMipLevel, 1);
MOBILEGL_ASSERT(srcReady, "%s: failed to transition source image", __func__);
}
@@ -8791,14 +9037,14 @@ void main() {
frame.commandBuffer, dstResource->image, dstResource->layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
dstStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT,
dstAccessMask, VK_ACCESS_TRANSFER_WRITE_BIT,
dstResource->aspect, 0, dstResource->mipLevels, kAllLayers);
dstResource->aspect, 0, dstResource->mipLevels);
MOBILEGL_ASSERT(dstReady, "%s: failed to transition undefined destination image", __func__);
dstCopyLayout = dstResource->layout;
} else {
Bool dstReady = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, dstResource->image, dstCopyLayout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
dstStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT,
dstAccessMask, VK_ACCESS_TRANSFER_WRITE_BIT, copyAspectMask, dstMipLevel, 1, kAllLayers);
dstAccessMask, VK_ACCESS_TRANSFER_WRITE_BIT, copyAspectMask, dstMipLevel, 1);
MOBILEGL_ASSERT(dstReady, "%s: failed to transition destination image", __func__);
}
@@ -8840,13 +9086,13 @@ void main() {
frame.commandBuffer, srcResource->image, srcResource->layout, srcRestoreLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, srcRestoreStageMask,
VK_ACCESS_TRANSFER_READ_BIT, srcRestoreAccessMask,
srcResource->aspect, 0, srcResource->mipLevels, kAllLayers);
srcResource->aspect, 0, srcResource->mipLevels);
MOBILEGL_ASSERT(srcRestored, "%s: failed to restore undefined source image layout", __func__);
} else {
Bool srcRestored = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, srcResource->image, srcCopyLayout, srcRestoreLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, srcRestoreStageMask,
VK_ACCESS_TRANSFER_READ_BIT, srcRestoreAccessMask, copyAspectMask, srcMipLevel, 1, kAllLayers);
VK_ACCESS_TRANSFER_READ_BIT, srcRestoreAccessMask, copyAspectMask, srcMipLevel, 1);
MOBILEGL_ASSERT(srcRestored, "%s: failed to restore source image layout", __func__);
}
@@ -8858,13 +9104,13 @@ void main() {
frame.commandBuffer, dstResource->image, dstResource->layout, dstRestoreLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, dstRestoreStageMask,
VK_ACCESS_TRANSFER_WRITE_BIT, dstRestoreAccessMask,
dstResource->aspect, 0, dstResource->mipLevels, kAllLayers);
dstResource->aspect, 0, dstResource->mipLevels);
MOBILEGL_ASSERT(dstRestored, "%s: failed to restore undefined destination image layout", __func__);
} else {
Bool dstRestored = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, dstResource->image, dstCopyLayout, dstRestoreLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, dstRestoreStageMask,
VK_ACCESS_TRANSFER_WRITE_BIT, dstRestoreAccessMask, copyAspectMask, dstMipLevel, 1, kAllLayers);
VK_ACCESS_TRANSFER_WRITE_BIT, dstRestoreAccessMask, copyAspectMask, dstMipLevel, 1);
MOBILEGL_ASSERT(dstRestored, "%s: failed to restore destination image layout", __func__);
}
@@ -9023,6 +9269,9 @@ void main() {
VkPipelineStageFlags srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
VkAccessFlags srcAccessMask = 0;
GetImageTransitionSourceState(srcOriginalLayout, srcStageMask, srcAccessMask);
// The copy below reads `srcBinding.baseArrayLayer`, which for a glFramebufferTextureLayer
// attachment is any layer of the array - the barrier covers all of them (see
// VkTextureManager::TransitionImageLayout), so the layer being read is one it moved.
if (readIsDefaultFbo) {
VkImageLayout trackedLayout = srcOriginalLayout;
Bool ok = VkTextureManager::TransitionImageLayout(
@@ -9048,19 +9297,18 @@ void main() {
// The GL rect, aimed at the default framebuffer's stored orientation. Using the GL y
// verbatim copied rows [y, y+h) counted from the TOP of the image, i.e. the wrong band for
// every read that was not full-height.
Int32 copyOffsetX = x;
Int32 copyOffsetY = y;
VkOffset2D copyOffset{x, y};
VkExtent2D copyExtent{static_cast<Uint32>(width), static_cast<Uint32>(height)};
if (readIsDefaultFbo) {
const VkExtent2D defaultFboExtent = m_swapchainObject.GetExtent();
const DefaultFramebufferRectMapping mapping =
GetDefaultFramebufferRectMapping(m_swapchainObject.GetPreTransform());
copyOffsetX = MapDefaultFramebufferRectAxis(x, width, static_cast<Int>(defaultFboExtent.width),
mapping.mirrorX);
copyOffsetY = MapDefaultFramebufferRectAxis(y, height, static_cast<Int>(defaultFboExtent.height),
mapping.flipY);
const Bool mapped = MapDefaultFramebufferReadbackRect(
x, y, width, height, defaultFboExtent, m_swapchainObject.GetPreTransform(), &copyOffset,
&copyExtent);
MOBILEGL_ASSERT(mapped, "ReadPixels: default framebuffer read rectangle is out of bounds");
if (!mapped) return;
}
copyRegion.imageOffset = {copyOffsetX, copyOffsetY, static_cast<Int32>(srcBinding.depthOffset)};
copyRegion.imageExtent = {static_cast<Uint32>(width), static_cast<Uint32>(height), 1};
copyRegion.imageOffset = {copyOffset.x, copyOffset.y, static_cast<Int32>(srcBinding.depthOffset)};
copyRegion.imageExtent = {copyExtent.width, copyExtent.height, 1};
vkCmdCopyImageToBuffer(frame.commandBuffer, srcBinding.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
readback.GetHandle(), 1, &copyRegion);
@@ -9102,16 +9350,14 @@ void main() {
// already aimed with the same mapping. The gate is exactly what made every partial
// read of the default framebuffer come back in Vulkan row order.
Vector<Uint8> remapped(static_cast<SizeT>(width) * static_cast<SizeT>(height) * sourceTexelSize);
if (RemapDefaultFboReadbackToGLOrientation(mapped, static_cast<Uint32>(width),
static_cast<Uint32>(height), preTransform, sourceTexelSize,
remapped.data())) {
if (RemapDefaultFramebufferReadback(mapped, static_cast<Uint32>(width),
static_cast<Uint32>(height), preTransform, sourceTexelSize,
remapped.data())) {
PackReadbackToClientOrPbo(remapped.data(), srcFormat, width, height, 1, format, type, pixels,
/*applyPackImageParams=*/false, /*applyReadColorClamp=*/true);
return;
}
// Only a quarter-turn pre-transform reaches this, and nothing in this renderer models
// one. MGLOG_I because the INFO builds are the ones that run conformance.
MGLOG_D("DirectVulkan::ReadPixels: default-FBO remap declined (w=%d h=%d preTransform=%d); falling back "
MGLOG_D("DirectVulkan::ReadPixels: default-FBO remap failed (w=%d h=%d preTransform=%d); falling back "
"to raw readback",
width, height, static_cast<Int>(preTransform));
}
@@ -9492,16 +9738,15 @@ void main() {
// The swapchain's depth/stencil image is stored display-side-up like its colour twin, so
// the GL rect has to be mapped into that space before the copy and the copied rows
// re-oriented afterwards - the same two halves the colour ReadPixels path applies.
Int32 copyOffsetX = x;
Int32 copyOffsetY = y;
VkOffset2D copyOffset{x, y};
VkExtent2D copyExtent{static_cast<Uint32>(width), static_cast<Uint32>(height)};
if (defaultFramebufferOrientation) {
const VkExtent2D defaultFboExtent = m_swapchainObject.GetExtent();
const DefaultFramebufferRectMapping mapping =
GetDefaultFramebufferRectMapping(m_swapchainObject.GetPreTransform());
copyOffsetX = MapDefaultFramebufferRectAxis(x, width, static_cast<Int>(defaultFboExtent.width),
mapping.mirrorX);
copyOffsetY = MapDefaultFramebufferRectAxis(y, height, static_cast<Int>(defaultFboExtent.height),
mapping.flipY);
const Bool mapped = MapDefaultFramebufferReadbackRect(
x, y, width, height, defaultFboExtent, m_swapchainObject.GetPreTransform(), &copyOffset,
&copyExtent);
MOBILEGL_ASSERT(mapped, "ReadDepthStencilPixels: default framebuffer read rectangle is out of bounds");
if (!mapped) return;
}
VkBufferImageCopy regions[2]{};
@@ -9513,8 +9758,8 @@ void main() {
region.imageSubresource.mipLevel = mipLevel;
region.imageSubresource.baseArrayLayer = baseArrayLayer;
region.imageSubresource.layerCount = 1;
region.imageOffset = {copyOffsetX, copyOffsetY, 0};
region.imageExtent = {static_cast<Uint32>(width), static_cast<Uint32>(height), 1};
region.imageOffset = {copyOffset.x, copyOffset.y, 0};
region.imageExtent = {copyExtent.width, copyExtent.height, 1};
}
if (wantStencil) {
auto& region = regions[regionCount++];
@@ -9523,8 +9768,8 @@ void main() {
region.imageSubresource.mipLevel = mipLevel;
region.imageSubresource.baseArrayLayer = baseArrayLayer;
region.imageSubresource.layerCount = 1;
region.imageOffset = {copyOffsetX, copyOffsetY, 0};
region.imageExtent = {static_cast<Uint32>(width), static_cast<Uint32>(height), 1};
region.imageOffset = {copyOffset.x, copyOffset.y, 0};
region.imageExtent = {copyExtent.width, copyExtent.height, 1};
}
vkCmdCopyImageToBuffer(frame.commandBuffer, image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback.GetHandle(),
regionCount, regions);
@@ -9558,23 +9803,21 @@ void main() {
Bool remapped = true;
if (wantDepth && depthCopyBytes > 0) {
remappedDepth.resize(pixelCount * depthCopyBytes);
remapped = RemapDefaultFboReadbackToGLOrientation(depthSrc, static_cast<Uint32>(width),
static_cast<Uint32>(height), preTransform,
depthCopyBytes, remappedDepth.data());
remapped = RemapDefaultFramebufferReadback(depthSrc, static_cast<Uint32>(width),
static_cast<Uint32>(height), preTransform,
depthCopyBytes, remappedDepth.data());
}
if (remapped && wantStencil) {
remappedStencil.resize(pixelCount);
remapped = RemapDefaultFboReadbackToGLOrientation(stencilSrc, static_cast<Uint32>(width),
static_cast<Uint32>(height), preTransform, 1,
remappedStencil.data());
remapped = RemapDefaultFramebufferReadback(stencilSrc, static_cast<Uint32>(width),
static_cast<Uint32>(height), preTransform, 1,
remappedStencil.data());
}
if (remapped) {
if (!remappedDepth.empty()) depthSrc = remappedDepth.data();
if (!remappedStencil.empty()) stencilSrc = remappedStencil.data();
} else {
// Only a quarter-turn pre-transform reaches this, and nothing in this renderer
// models one. MGLOG_I because the INFO builds are the ones that run conformance.
MGLOG_D("DirectVulkan::ReadDepthStencilPixels: default-FBO remap declined (w=%d h=%d "
MGLOG_D("DirectVulkan::ReadDepthStencilPixels: default-FBO remap failed (w=%d h=%d "
"preTransform=%d); falling back to raw readback",
width, height, static_cast<Int>(preTransform));
}
@@ -9825,14 +10068,13 @@ void main() {
VkPipelineStageFlags srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
VkAccessFlags srcAccessMask = 0;
GetImageTransitionSourceState(originalLayout, srcStageMask, srcAccessMask);
// The copy below reads EVERY layer of the level, so the barrier has to name every layer
// too; a layerCount of 1 left an array texture's layers 1.. in whatever layout they were
// last left in while the transfer read them.
// The copy below reads EVERY layer of the level, which is exactly the range
// TransitionImageLayout barriers cover.
Bool ok = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, resource->image, resource->layout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
srcStageMask, VK_PIPELINE_STAGE_TRANSFER_BIT,
srcAccessMask, VK_ACCESS_TRANSFER_READ_BIT, resource->aspect,
static_cast<Uint32>(level), 1, VK_REMAINING_ARRAY_LAYERS);
static_cast<Uint32>(level), 1);
MOBILEGL_ASSERT(ok, "%s: failed to transition texture image", __func__);
VkBufferImageCopy copyRegion{};
@@ -9852,7 +10094,7 @@ void main() {
frame.commandBuffer, resource->image, resource->layout, originalLayout,
VK_PIPELINE_STAGE_TRANSFER_BIT, restoreStageMask,
VK_ACCESS_TRANSFER_READ_BIT, restoreAccessMask, resource->aspect,
static_cast<Uint32>(level), 1, VK_REMAINING_ARRAY_LAYERS);
static_cast<Uint32>(level), 1);
MOBILEGL_ASSERT(ok, "%s: failed to restore texture image layout", __func__);
if (!SubmitReadbackCommandsAndWait(frame)) {
@@ -9960,7 +10202,7 @@ void main() {
Bool transitioned = VkTextureManager::TransitionImageLayout(
frame.commandBuffer, resource->image, resource->layout, finalLayout,
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT,
0, VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels, resource->arrayLayers);
0, VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels);
MOBILEGL_ASSERT(transitioned, "GenerateMipmap: failed to transition uninitialized mip chain");
return;
}
@@ -11768,6 +12010,17 @@ void main() {
}
void VulkanRenderer::CreateInstance() {
#if defined(VK_USE_PLATFORM_METAL_EXT)
// MoltenVK snapshots its configuration when the loader first discovers the ICD. Set
// this before instance-extension enumeration, while preserving an explicit user value.
if (std::getenv("MVK_CONFIG_USE_METAL_ARGUMENT_BUFFERS") == nullptr) {
if (::setenv("MVK_CONFIG_USE_METAL_ARGUMENT_BUFFERS", "1", 0) == 0) {
MGLOG_I("MoltenVK: enabling Metal argument buffers");
} else {
MGLOG_W("MoltenVK: could not enable Metal argument buffers before ICD discovery");
}
}
#endif
m_extensions = EnumerateInstanceExtensions();
MGLOG_I("Got %d Vulkan instance extensions: ", m_extensions.size());
for (auto& extension : m_extensions) {
@@ -11909,17 +12162,19 @@ void main() {
auto debugMessengerCreateInfo = PopulateDebugMessengerCreateInfo();
// Layers
const void* instanceCreatePNext = nullptr;
if (m_validationLayersEnabled) {
MGLOG_I("Enabling validation layer...");
instanceInfo.enabledLayerCount = static_cast<uint32_t>(std::size(s_validationLayerNames));
instanceInfo.ppEnabledLayerNames = s_validationLayerNames;
// Chaining the messenger create-info is only legal with the extension on.
instanceInfo.pNext = debugUtilsAvailable ? &debugMessengerCreateInfo : nullptr;
instanceCreatePNext = debugUtilsAvailable ? &debugMessengerCreateInfo : nullptr;
} else {
instanceInfo.enabledLayerCount = 0;
instanceInfo.pNext = nullptr;
}
instanceInfo.pNext = instanceCreatePNext;
VK_VERIFY(vkCreateInstance(&instanceInfo, nullptr, &m_instance), "vkCreateInstance failed");
if (debugUtilsAvailable) {
@@ -12218,6 +12473,28 @@ void main() {
m_fillModeNonSolidFeatureEnabled = deviceFeatures.fillModeNonSolid == VK_TRUE;
deviceFeatures.dualSrcBlend = supportedDeviceFeatures.dualSrcBlend;
m_dualSrcBlendFeatureEnabled = deviceFeatures.dualSrcBlend == VK_TRUE;
// ARB_viewport_array rasterization. Without multiViewport a pipeline may declare exactly
// one viewport (VUID-VkPipelineViewportStateCreateInfo-viewportCount-01216), so a shader's
// gl_ViewportIndex can only ever select viewport 0 and the other fifteen rectangles are
// state with nowhere to go. The GL state stays 16 wide either way - GL 4.3 core requires
// MAX_VIEWPORTS >= 16 and that is a frontend promise, not a device one; this gate decides
// only whether a DRAW can rasterize into more than one of them.
deviceFeatures.multiViewport = supportedDeviceFeatures.multiViewport;
m_multiViewportFeatureEnabled = deviceFeatures.multiViewport == VK_TRUE;
m_maxRasterizableViewports =
m_multiViewportFeatureEnabled
? std::min<Uint32>(RenderStateParameters::MAX_VIEWPORTS,
std::max<Uint32>(m_physicalDevice.properties.limits.maxViewports, 1u))
: 1u;
MGLOG_I("Vulkan: multiViewport %s; rasterizable viewports=%u (device limit %u, GL state width %u)",
m_multiViewportFeatureEnabled ? "enabled" : "UNAVAILABLE", m_maxRasterizableViewports,
m_physicalDevice.properties.limits.maxViewports,
static_cast<Uint32>(RenderStateParameters::MAX_VIEWPORTS));
if (!m_multiViewportFeatureEnabled) {
MGLOG_W("Vulkan: the device does not support the multiViewport feature; gl_ViewportIndex will always "
"select viewport 0 and per-viewport scissor/depth-range state past index 0 cannot be "
"rasterized (the state itself is still stored and queryable)");
}
deviceFeatures.logicOp = supportedDeviceFeatures.logicOp;
deviceFeatures.shaderClipDistance = supportedDeviceFeatures.shaderClipDistance;
deviceFeatures.shaderCullDistance = supportedDeviceFeatures.shaderCullDistance;
@@ -12323,6 +12600,75 @@ void main() {
vkGetInstanceProcAddr(m_instance, "vkGetPhysicalDeviceFeatures2KHR"));
}
m_updateAfterBindLimits = {};
VkPhysicalDeviceDescriptorIndexingFeatures descriptorIndexingFeatures{};
descriptorIndexingFeatures.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DESCRIPTOR_INDEXING_FEATURES;
VkPhysicalDeviceDescriptorIndexingProperties descriptorIndexingProperties{};
descriptorIndexingProperties.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DESCRIPTOR_INDEXING_PROPERTIES;
const Bool descriptorIndexingCore = m_physicalDevice.properties.apiVersion >= VK_API_VERSION_1_2;
const Bool descriptorIndexingExtension =
IsExtensionSupported(availableExtensions, VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME);
auto getPhysicalDeviceProperties2 = reinterpret_cast<PFN_vkGetPhysicalDeviceProperties2>(
vkGetInstanceProcAddr(m_instance, "vkGetPhysicalDeviceProperties2"));
if (getPhysicalDeviceProperties2 == nullptr) {
getPhysicalDeviceProperties2 = reinterpret_cast<PFN_vkGetPhysicalDeviceProperties2>(
vkGetInstanceProcAddr(m_instance, "vkGetPhysicalDeviceProperties2KHR"));
}
if ((descriptorIndexingCore || descriptorIndexingExtension) && getPhysicalDeviceFeatures2 != nullptr &&
getPhysicalDeviceProperties2 != nullptr) {
VkPhysicalDeviceFeatures2 featureQuery{};
featureQuery.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2;
featureQuery.pNext = &descriptorIndexingFeatures;
getPhysicalDeviceFeatures2(m_physicalDevice.handle, &featureQuery);
VkPhysicalDeviceProperties2 propertyQuery{};
propertyQuery.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2;
propertyQuery.pNext = &descriptorIndexingProperties;
getPhysicalDeviceProperties2(m_physicalDevice.handle, &propertyQuery);
// This renderer emits every descriptor category listed below, including
// dynamic UBOs and combined image samplers. Do not enable a partial
// descriptor-indexing contract: it would make a later reflected program
// fail in the driver instead of choosing its ordinary descriptor layout.
const Bool allUpdateAfterBindFeatures =
descriptorIndexingFeatures.descriptorBindingUniformBufferUpdateAfterBind == VK_TRUE &&
descriptorIndexingFeatures.descriptorBindingSampledImageUpdateAfterBind == VK_TRUE &&
descriptorIndexingFeatures.descriptorBindingStorageImageUpdateAfterBind == VK_TRUE &&
descriptorIndexingFeatures.descriptorBindingStorageBufferUpdateAfterBind == VK_TRUE &&
descriptorIndexingFeatures.descriptorBindingUniformTexelBufferUpdateAfterBind == VK_TRUE &&
descriptorIndexingFeatures.descriptorBindingStorageTexelBufferUpdateAfterBind == VK_TRUE &&
(!deviceFeatures.robustBufferAccess || descriptorIndexingProperties.robustBufferAccessUpdateAfterBind);
if (allUpdateAfterBindFeatures) {
if (!descriptorIndexingCore && !IsExtensionAlreadyEnabled(
enabledDeviceExtensions,
VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME)) {
enabledDeviceExtensions.push_back(VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME);
}
descriptorIndexingFeatures.pNext = const_cast<void*>(deviceCreateInfo.pNext);
deviceCreateInfo.pNext = &descriptorIndexingFeatures;
m_updateAfterBindLimits = {
true,
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindSamplers,
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindUniformBuffers,
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindStorageBuffers,
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindSampledImages,
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindStorageImages,
descriptorIndexingProperties.maxPerStageUpdateAfterBindResources,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindSamplers,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindUniformBuffers,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindUniformBuffersDynamic,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindStorageBuffers,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindStorageBuffersDynamic,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindSampledImages,
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindStorageImages};
MGLOG_I("Vulkan: update-after-bind descriptor layouts enabled");
} else {
MGLOG_I("Vulkan: descriptor indexing is present but lacks the complete update-after-bind feature set; "
"using ordinary descriptor layouts");
}
} else {
MGLOG_I("Vulkan: descriptor indexing unavailable; using ordinary descriptor layouts");
}
VkPhysicalDeviceIndexTypeUint8Features indexTypeUint8Features{};
indexTypeUint8Features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_INDEX_TYPE_UINT8_FEATURES;
if (indexTypeUint8ExtensionName != nullptr) {
@@ -229,6 +229,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
GLint dstY, GLint width, GLint height, VkImageLayout srcRestoreLayout,
VkImageLayout dstRestoreLayout, Bool stencilAspect);
static SizeT GetReadbackTexelSize(VkFormat sourceFormat);
// Map a GL bottom-left-origin rectangle into the display-oriented swapchain image.
// Quarter-turn surface transforms swap the copy extent's axes.
static Bool MapDefaultFramebufferReadbackRect(GLint x, GLint y, GLsizei width, GLsizei height,
VkExtent2D imageExtent,
VkSurfaceTransformFlagBitsKHR preTransform,
VkOffset2D* imageOffset, VkExtent2D* imageCopyExtent);
// Reorder a tightly packed block copied with MapDefaultFramebufferReadbackRect back into
// GL row order. The input block has swapped dimensions for 90/270 degree transforms.
static Bool RemapDefaultFramebufferReadback(const Uint8* rawPixels, Uint32 logicalWidth,
Uint32 logicalHeight,
VkSurfaceTransformFlagBitsKHR preTransform,
SizeT texelSize, Uint8* outPixels);
static Bool ConvertReadbackPixels(const Uint8* sourcePixels, VkFormat sourceFormat,
GLsizei width, GLsizei height, GLenum destinationFormat,
GLenum destinationType, SizeT destinationRowStride,
@@ -298,6 +310,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// The samplerAnisotropy device feature was granted, so GL_TEXTURE_MAX_ANISOTROPY_EXT is
// honored rather than accepted-and-ignored.
Bool IsSamplerAnisotropySupported() const { return m_samplerAnisotropyFeatureEnabled; }
// ARB_base_instance extends indirect command records with a non-zero firstInstance and
// requires gl_InstanceID to remain zero-based. Vulkan needs both features to honor that
// complete contract: one legalizes the command word, the other enables the shader rebase.
Bool IsNonZeroIndirectBaseInstanceSupported() const {
return m_drawIndirectFirstInstanceFeatureEnabled && m_shaderDrawParametersFeatureEnabled;
}
// Ensures the frame command buffer is recording (same lazy pattern as
// SetupDraw) and writes a bottom-of-pipe timestamp into the current
// frame's pool. Null when unsupported or the pool is exhausted.
@@ -537,6 +555,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
Bool m_shaderDrawParametersExtensionEnabled = false;
Bool m_shaderDrawParametersFeatureEnabled = false;
Bool m_unformattedFloatStorageImagesEnabled = false;
// Set only after descriptor-indexing feature AND property queries prove that
// update-after-bind is legal for every descriptor category this renderer emits.
ProgramFactory::UpdateAfterBindLimits m_updateAfterBindLimits{};
// fillModeNonSolid gates VK_POLYGON_MODE_LINE/_POINT (glPolygonMode); independentBlend gates
// per-draw-buffer color write masks (glColorMaski). Both are cached at device creation and
// drive a runtime fallback when the device lacks them.
@@ -547,6 +568,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// needs no feature). Both cached at device creation and drive a hard-fail-at-draw when absent.
Bool m_dualSrcBlendFeatureEnabled = false;
Bool m_primitiveTopologyListRestartFeatureEnabled = false;
// multiViewport gates rasterizing into more than one of ARB_viewport_array's 16 viewports
// (gl_ViewportIndex). m_maxRasterizableViewports is min(MAX_VIEWPORTS, device limit), or 1
// when the feature is off, and is the viewportCount a gl_ViewportIndex-writing pipeline
// declares - it is NOT what GL_MAX_VIEWPORTS reports, which is the frontend state width.
Bool m_multiViewportFeatureEnabled = false;
Uint32 m_maxRasterizableViewports = 1;
// Union of shader stages sampled-read barriers may name; built at device creation
// because geometry/tessellation stage bits are invalid in a barrier when their
// feature is off (VUID-vkCmdPipelineBarrier-srcStageMask-04090/-04091), and
@@ -830,6 +857,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// re-resolve just the pipeline against the active pass; a change that
// flips it must fall back to the full path's pass selection.
Bool drawUsesDepthStencil = false;
// The snapshotting draw's pipeline viewportCount. A pure function of the PROGRAM
// (writesViewportIndexBuiltin) and of a device feature fixed at renderer init, both
// of which the programLifetimeId/programVersion guards above already pin - carried
// here so the fast path does not re-fetch the program object to re-derive it.
Uint32 viewportCount = 1;
IntVec2 renderPassExtent = {0, 0};
// colorAttachmentCount of the snapshotting draw's render pass: the
// pipeline-state hash input, so the fast path can refresh that hash and
@@ -897,6 +929,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
// already sampleable.
Vector<VkTextureManager::TextureResource*> m_sampledResourcesScratch;
Vector<MG_State::GLState::ITextureObject*> m_storageImageTexturesScratch;
Vector<UniformManager::SamplerImageFeedbackBinding> m_samplerImageFeedbackScratch;
Vector<UniformManager::SamplerBindingOverride> m_samplerImageBindingOverridesScratch;
Vector<VkBuffer> m_vertexBuffersScratch;
Vector<VkDeviceSize> m_vertexOffsetsScratch;
Vector<VkVertexInputAttributeDescription> m_patchedAttributesScratch;
@@ -1116,11 +1150,34 @@ namespace MobileGL::MG_Backend::DirectVulkan {
FrameContext::FrameData& frame,
const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj);
// Vulkan forbids a sampled descriptor and writable storage descriptor from naming the
// same image subresource in one shader operation. Snapshot only the sampler side; the
// storage descriptor continues to name the application texture.
Bool PrepareSamplerImageFeedbackSnapshots(
FrameContext::FrameData& frame,
const MG_State::GLState::ProgramObject& program,
const ProgramFactory::VkProgramObject& programObj,
VkPipelineStageFlags consumerShaderStageMask);
// The per-draw dynamic-state tail (viewport, scissor, blend constants, depth
// bias, line width, stencil), gated behind one render-state-parameters-version
// compare per command buffer - see the gate fields in DynamicStateShadow.
void ApplyDynamicDrawStateTail(FrameContext::FrameData& frame, const IntVec2& extent, Bool isDefaultFbo);
// viewportCount is the bound pipeline's declared viewport count: 1 for every program that
// does not write gl_ViewportIndex (the memoized fast path), otherwise the renderer's
// rasterizable viewport count, which takes the unmemoized array path.
void ApplyDynamicDrawStateTail(FrameContext::FrameData& frame, const IntVec2& extent, Bool isDefaultFbo,
Uint32 viewportCount = 1);
void ApplyMultiViewportDynamicState(VkCommandBuffer commandBuffer, Uint32 viewportCount, const IntVec2& extent,
VkSurfaceTransformFlagBitsKHR preTransform, Bool isDefaultFbo);
VkRect2D ComputeGLScissorRect(Uint32 index, const IntVec2& extent,
VkSurfaceTransformFlagBitsKHR preTransform, Bool isDefaultFbo) const;
// How many viewports a draw with this program rasterizes into: 1 unless the program
// assigns gl_ViewportIndex AND the device enabled multiViewport. Both the pipeline's
// baked viewportCount and the dynamic arrays come from this one answer, so they cannot
// disagree.
Uint32 ResolveDrawViewportCount(Bool programWritesViewportIndex) const {
return programWritesViewportIndex && m_multiViewportFeatureEnabled ? m_maxRasterizableViewports : 1u;
}
Bool UploadAndBindVertexBuffers(VkCommandBuffer commandBuffer, const MG_State::GLState::VertexArrayObject& vao,
const ProgramFactory::VkProgramObject& programObj,
+85 -30
View File
@@ -18,6 +18,7 @@
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
#include <MG_Util/Converters/GLToMG/BufferEnumConverter.h>
#include <MG_Util/Converters/MGToGL/BufferEnumConverter.h>
#include <MG_Util/Texture/PixelStoreProcessor.h>
namespace MobileGL::MG_Impl::GLImpl {
namespace {
@@ -31,6 +32,8 @@ namespace MobileGL::MG_Impl::GLImpl {
NamedBufferData,
NamedBufferSubData,
CopyNamedBufferSubData,
ClearBufferData,
ClearBufferSubData,
ClearNamedBufferData,
ClearNamedBufferSubData,
MapBufferRange,
@@ -65,6 +68,10 @@ namespace MobileGL::MG_Impl::GLImpl {
return "NamedBufferSubData";
case BufferOp::CopyNamedBufferSubData:
return "CopyNamedBufferSubData";
case BufferOp::ClearBufferData:
return "ClearBufferData";
case BufferOp::ClearBufferSubData:
return "ClearBufferSubData";
case BufferOp::ClearNamedBufferData:
return "ClearNamedBufferData";
case BufferOp::ClearNamedBufferSubData:
@@ -143,16 +150,6 @@ namespace MobileGL::MG_Impl::GLImpl {
return 0;
}
// The pattern is replicated verbatim, which is only the whole story while the client
// layout already matches the internal format - the case every entry point in practice
// uses, and the only one the conversion machinery here can express. Say so rather than
// quietly writing a differently-sized pattern.
const SizeT sourceSize = MG_Util::GetInputBytesPerPixel(inputFormat, pixelType);
if (sourceSize != elementSize) {
MGLOG_W_ONCE("%s: clear pattern is %zu bytes but internalformat 0x%X stores %zu; "
"converting between them is not implemented",
GetBufferOpName(op), sourceSize, internalformat, elementSize);
}
return elementSize;
}
@@ -194,27 +191,59 @@ namespace MobileGL::MG_Impl::GLImpl {
return true;
}
void ClearNamedBufferRange_State(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size,
GLenum format, GLenum type, const void* data, BufferOp op) {
Bool BuildClearPattern(GLenum internalformat, GLenum format, GLenum type, const void* data,
SizeT patternSize, BufferOp op, Vector<Uint8>& pattern) {
const TextureInternalFormat internal = MG_Util::ConvertGLEnumToTextureInternalFormat(internalformat);
const TextureInputFormat inputFormat = MG_Util::ConvertGLEnumToTextureInputFormat(format);
const TexturePixelDataType inputType = MG_Util::ConvertGLEnumToTexturePixelDataType(type);
Vector<Uint8> zeroInput;
const void* inputPixel = data;
if (inputPixel == nullptr) {
const SizeT inputSize = MG_Util::GetInputBytesPerPixel(inputFormat, inputType);
if (inputSize == 0) {
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", GetBufferOpName(op),
"format and type do not describe a source pixel."));
return false;
}
zeroInput.resize(inputSize);
inputPixel = zeroInput.data();
}
if (!MG_Util::PixelStoreProcessor::ConvertOnePixelToInternal(
internal, inputFormat, inputType, inputPixel, pattern)) {
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>(
"MG_Impl/GLImpl", GetBufferOpName(op),
std::format("Cannot convert one ({}, {}) pixel into internalformat 0x{:X}.",
MG_Util::ConvertGLEnumToString(format), MG_Util::ConvertGLEnumToString(type),
internalformat)));
return false;
}
if (data == nullptr) {
// GL defines a null clear value as all zero bits in the destination store, while
// retaining the format/type validation above.
pattern.assign(patternSize, 0);
}
return true;
}
void ClearBufferRange_State(const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
GLenum internalformat, GLintptr offset, GLsizeiptr size,
GLenum format, GLenum type, const void* data, BufferOp op) {
const SizeT patternSize = GetClearPatternSize(internalformat, format, type, op);
if (patternSize == 0) return;
auto bufferObject = GetNamedBufferObject(buffer, op);
if (!bufferObject) return;
if (!ValidateBufferClearRange(bufferObject, offset, size, patternSize, op)) return;
if (size == 0) return;
Vector<Uint8> clearData(static_cast<SizeT>(size));
if (data) {
const auto* pattern = static_cast<const Uint8*>(data);
for (SizeT at = 0; at < clearData.size(); at += patternSize) {
Memcpy(clearData.data() + at, pattern, patternSize);
}
} else {
Memset(clearData.data(), 0, clearData.size());
}
bufferObject->UploadSubData({clearData.data(), clearData.size()}, static_cast<SizeT>(offset));
Vector<Uint8> pattern;
if (!BuildClearPattern(internalformat, format, type, data, patternSize, op, pattern)) return;
bufferObject->FillSubData({pattern.data(), pattern.size()}, static_cast<SizeT>(offset),
static_cast<SizeT>(size));
}
auto& GetBufferBindingSlot(BufferTarget target) {
@@ -1197,17 +1226,34 @@ namespace MobileGL::MG_Impl::GLImpl {
static_cast<SizeT>(writeOffset), static_cast<SizeT>(size));
}
void ClearBufferData_State(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) {
auto bufferObject = GetBoundBufferObject(target, BufferOp::ClearBufferData);
if (!bufferObject) return;
ClearBufferRange_State(bufferObject, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
type, data, BufferOp::ClearBufferData);
}
void ClearBufferSubData_State(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size,
GLenum format, GLenum type, const void* data) {
auto bufferObject = GetBoundBufferObject(target, BufferOp::ClearBufferSubData);
if (!bufferObject) return;
ClearBufferRange_State(bufferObject, internalformat, offset, size, format, type, data,
BufferOp::ClearBufferSubData);
}
void ClearNamedBufferData_State(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data) {
auto bufferObject = GetNamedBufferObject(buffer, BufferOp::ClearNamedBufferData);
if (!bufferObject) return;
ClearNamedBufferRange_State(buffer, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
type, data, BufferOp::ClearNamedBufferData);
ClearBufferRange_State(bufferObject, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
type, data, BufferOp::ClearNamedBufferData);
}
void ClearNamedBufferSubData_State(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size,
GLenum format, GLenum type, const void* data) {
ClearNamedBufferRange_State(buffer, internalformat, offset, size, format, type, data,
BufferOp::ClearNamedBufferSubData);
auto bufferObject = GetNamedBufferObject(buffer, BufferOp::ClearNamedBufferSubData);
if (!bufferObject) return;
ClearBufferRange_State(bufferObject, internalformat, offset, size, format, type, data,
BufferOp::ClearNamedBufferSubData);
}
void* MapNamedBuffer_State(GLuint buffer, GLenum access) {
@@ -1662,6 +1708,15 @@ namespace MobileGL::MG_Impl::GLImpl {
CopyNamedBufferSubData_State(readBuffer, writeBuffer, readOffset, writeOffset, size);
}
void ClearBufferData(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) {
ClearBufferData_State(target, internalformat, format, type, data);
}
void ClearBufferSubData(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
GLenum type, const void* data) {
ClearBufferSubData_State(target, internalformat, offset, size, format, type, data);
}
void ClearNamedBufferData(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data) {
ClearNamedBufferData_State(buffer, internalformat, format, type, data);
}
@@ -27,6 +27,9 @@ namespace MobileGL::MG_Impl::GLImpl {
void NamedBufferSubData(GLuint buffer, GLintptr offset, GLsizeiptr size, const void* data);
void CopyNamedBufferSubData(GLuint readBuffer, GLuint writeBuffer, GLintptr readOffset, GLintptr writeOffset,
GLsizeiptr size);
void ClearBufferData(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data);
void ClearBufferSubData(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
GLenum type, const void* data);
void ClearNamedBufferData(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data);
void ClearNamedBufferSubData(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
GLenum type, const void* data);
@@ -969,14 +969,14 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribL3dv, GLuint index, const GLdoub
DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribL4dv, GLuint index, const GLdouble* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, VertexAttribL4dv, index, v)
DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribLPointer, GLuint index, GLint size, GLenum type, GLsizei stride, const void* pointer) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, VertexAttribLPointer, index, size, type, stride, pointer)
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetVertexAttribLdv, GLuint index, GLenum pname, GLdouble* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetVertexAttribLdv, index, pname, params)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ViewportArrayv, GLuint first, GLsizei count, const GLfloat* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ViewportArrayv, first, count, v)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ViewportIndexedf, GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ViewportIndexedf, index, x, y, w, h)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ViewportIndexedfv, GLuint index, const GLfloat* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ViewportIndexedfv, index, v)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ScissorArrayv, GLuint first, GLsizei count, const GLint* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ScissorArrayv, first, count, v)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ScissorIndexed, GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ScissorIndexed, index, left, bottom, width, height)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ScissorIndexedv, GLuint index, const GLint* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ScissorIndexedv, index, v)
DECLARE_GL_FUNCTION_STUB_HEAD(void, DepthRangeArrayv, GLuint first, GLsizei count, const GLdouble* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DepthRangeArrayv, first, count, v)
DECLARE_GL_FUNCTION_STUB_HEAD(void, DepthRangeIndexed, GLuint index, GLdouble n, GLdouble f) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DepthRangeIndexed, index, n, f)
DECLARE_GL_FUNCTION_HEAD(void, ViewportArrayv, GLuint first, GLsizei count, const GLfloat* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ViewportArrayv, first, count, v)
DECLARE_GL_FUNCTION_HEAD(void, ViewportIndexedf, GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ViewportIndexedf, index, x, y, w, h)
DECLARE_GL_FUNCTION_HEAD(void, ViewportIndexedfv, GLuint index, const GLfloat* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ViewportIndexedfv, index, v)
DECLARE_GL_FUNCTION_HEAD(void, ScissorArrayv, GLuint first, GLsizei count, const GLint* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ScissorArrayv, first, count, v)
DECLARE_GL_FUNCTION_HEAD(void, ScissorIndexed, GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ScissorIndexed, index, left, bottom, width, height)
DECLARE_GL_FUNCTION_HEAD(void, ScissorIndexedv, GLuint index, const GLint* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ScissorIndexedv, index, v)
DECLARE_GL_FUNCTION_HEAD(void, DepthRangeArrayv, GLuint first, GLsizei count, const GLdouble* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DepthRangeArrayv, first, count, v)
DECLARE_GL_FUNCTION_HEAD(void, DepthRangeIndexed, GLuint index, GLdouble n, GLdouble f) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DepthRangeIndexed, index, n, f)
DECLARE_GL_FUNCTION_HEAD(void, GetFloati_v, GLenum target, GLuint index, GLfloat* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, GetFloati_v, target, index, data)
DECLARE_GL_FUNCTION_HEAD(void, GetDoublei_v, GLenum target, GLuint index, GLdouble* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, GetDoublei_v, target, index, data)
DECLARE_GL_FUNCTION_HEAD(void, DrawArraysInstancedBaseInstance, GLenum mode, GLint first, GLsizei count, GLsizei instancecount, GLuint baseinstance) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawArraysInstancedBaseInstance, mode, first, count, instancecount, baseinstance)
@@ -985,8 +985,8 @@ DECLARE_GL_FUNCTION_HEAD(void, DrawElementsInstancedBaseVertexBaseInstance, GLen
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetActiveAtomicCounterBufferiv, GLuint program, GLuint bufferIndex, GLenum pname, GLint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetActiveAtomicCounterBufferiv, program, bufferIndex, pname, params)
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackInstanced, GLenum mode, GLuint id, GLsizei instancecount) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackInstanced, mode, id, instancecount)
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackStreamInstanced, GLenum mode, GLuint id, GLuint stream, GLsizei instancecount) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackStreamInstanced, mode, id, stream, instancecount)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearBufferData, GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearBufferData, target, internalformat, format, type, data)
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearBufferSubData, GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearBufferSubData, target, internalformat, offset, size, format, type, data)
DECLARE_GL_FUNCTION_HEAD(void, ClearBufferData, GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearBufferData, target, internalformat, format, type, data)
DECLARE_GL_FUNCTION_HEAD(void, ClearBufferSubData, GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearBufferSubData, target, internalformat, offset, size, format, type, data)
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetInternalformati64v, GLenum target, GLenum internalformat, GLenum pname, GLsizei count, GLint64* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetInternalformati64v, target, internalformat, pname, count, params)
DECLARE_GL_FUNCTION_STUB_HEAD(void, InvalidateTexSubImage, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, InvalidateTexSubImage, texture, level, xoffset, yoffset, zoffset, width, height, depth)
DECLARE_GL_FUNCTION_STUB_HEAD(void, InvalidateTexImage, GLuint texture, GLint level) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, InvalidateTexImage, texture, level)
+115 -16
View File
@@ -16,6 +16,7 @@
#include <MG_State/GLState/ErrorState/ErrorInfo.h>
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
#include <MG_Util/Converters/GLToMG/BufferEnumConverter.h>
#include <MG_Util/Converters/GLToMG/RenderStateEnumConverter.h>
#include <MG_Util/Converters/MGToGL/FramebufferEnumConverter.h>
#include <MG_Util/Converters/MGToGL/ErrorCodeConverter.h>
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
@@ -27,6 +28,11 @@
#include <MG_Backend/BackendObjects.h>
namespace MobileGL::MG_Impl::GLImpl {
// Declared rather than #included from GL_RenderState.h on purpose: that header also declares
// a free function named BlendEquation, which would hide the ::MobileGL::BlendEquation enum
// this file's blend-state queries name unqualified.
GLboolean IsEnabledi(GLenum target, GLuint index);
namespace {
enum class IndexedBufferQueryKind {
Binding,
@@ -339,26 +345,70 @@ namespace MobileGL::MG_Impl::GLImpl {
return sampler ? static_cast<GLint>(sampler->GetExternalIndex()) : 0;
}
// The ARB_viewport_array indexed rectangles. MobileGL keeps exactly one viewport, one
// scissor box and one depth range, so every in-range index answers with that single
// value - but it has to come from the frontend state the non-indexed getters read.
// The generic path at the bottom of GetIntegeri_v is a raw backend passthrough that
// has no case for these, so routing them through it returned zeros.
// The ARB_viewport_array indexed rectangles. Each of these is genuinely per-viewport
// frontend state (RenderStateParameters::Viewports / ScissorBoxes / DepthRanges), so the
// indexed getters must read the indexed storage - the generic path at the bottom of
// GetIntegeri_v is a raw backend passthrough that has no case for them and returned
// zeros, and routing them to the NON-indexed getter (what this used to do) answered every
// index with viewport 0's value, which is what
// KHR-GL43.viewport_array.{viewport,scissor,depth_range}_api caught.
Bool IsIndexedViewportQuery(GLenum target) {
return target == GL_VIEWPORT || target == GL_SCISSOR_BOX || target == GL_DEPTH_RANGE;
}
// ARB_viewport_array: `index` selects a viewport and MAX_VIEWPORTS bounds it.
// Component count of an indexed viewport-array query, so every width of getter writes the
// caller's whole buffer instead of just element 0 (GL 4.6 core 22.1).
GLsizei IndexedViewportQueryComponents(GLenum target) {
return target == GL_DEPTH_RANGE ? 2 : 4;
}
// ARB_viewport_array: `index` selects a viewport and MAX_VIEWPORTS bounds it. The bound is
// the frontend's own state width, which is also exactly what GL_MAX_VIEWPORTS reports -
// taking it from the backend caps instead would let a device limit of 1 (a Vulkan device
// without the multiViewport feature) make index 1 illegal even though the state exists.
Bool ValidateViewportQueryIndex(GLuint index, const char* caller) {
GLint maxViewports = 0;
GetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
if (index < static_cast<GLuint>(std::max(maxViewports, 1))) return true;
if (index < RenderStateParameters::MAX_VIEWPORTS) return true;
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller, "Viewport index is out of range."));
return false;
}
// The indexed viewport/scissor/depth-range state as floats, which is the widest lossless
// shape MobileGL stores (the viewport really is float state; the scissor box is integral
// and well inside float's exact range, and every depth range is in [0, 1]). Every indexed
// getter width funnels through this so they can never disagree with each other.
void ReadIndexedViewportStateFloat(GLenum target, GLuint index, GLfloat* out) {
switch (target) {
case GL_VIEWPORT: {
const FloatVec4& viewport = MG_State::pGLContext->GetViewportIndexed(index);
out[0] = viewport.x();
out[1] = viewport.y();
out[2] = viewport.z();
out[3] = viewport.w();
return;
}
case GL_SCISSOR_BOX: {
const IntVec4& box = MG_State::pGLContext->GetScissorBoxIndexed(index);
out[0] = static_cast<GLfloat>(box.x());
out[1] = static_cast<GLfloat>(box.y());
out[2] = static_cast<GLfloat>(box.z());
out[3] = static_cast<GLfloat>(box.w());
return;
}
case GL_DEPTH_RANGE: {
const FloatVec2& range = MG_State::pGLContext->GetDepthRangeIndexed(index);
out[0] = range.x();
out[1] = range.y();
return;
}
default:
MOBILEGL_ASSERT(false, "ReadIndexedViewportStateFloat: unexpected target 0x%x",
static_cast<Uint32>(target));
return;
}
}
void CopyIntsToBooleans(const GLint* src, SizeT count, GLboolean* dst) {
for (SizeT i = 0; i < count; ++i) {
dst[i] = src[i] ? GL_TRUE : GL_FALSE;
@@ -629,6 +679,17 @@ namespace MobileGL::MG_Impl::GLImpl {
params[1] = dynamicParameters.ViewportBoundsRangeMax;
return;
}
// Viewport 0's rectangle, verbatim. Falling through to the integer width below would
// round the fractional rectangle a glViewportIndexedf(0, ...) is allowed to set, and
// glGetFloatv(GL_VIEWPORT) is a lossless query of float state.
case GL_VIEWPORT: {
const FloatVec4& viewport = MG_State::pGLContext->GetViewportIndexed(0);
params[0] = viewport.x();
params[1] = viewport.y();
params[2] = viewport.z();
params[3] = viewport.w();
return;
}
case GL_MIN_FRAGMENT_INTERPOLATION_OFFSET:
case GL_MAX_FRAGMENT_INTERPOLATION_OFFSET:
case GL_FRAGMENT_INTERPOLATION_OFFSET_BITS: {
@@ -792,15 +853,32 @@ namespace MobileGL::MG_Impl::GLImpl {
return;
}
// GL 4.6 core 22.1: an indexed query answers EVERY indexed state, and GL_SCISSOR_TEST is
// indexed by viewport just like GL_BLEND is by draw buffer. Without this the integer
// width fell through to the backend passthrough and answered GL_INVALID_ENUM, which is
// the sticky error KHR-GL43.viewport_array.queries trips over at its next error check.
if (MG_Util::ConvertGLEnumToCapabilityInput(target) != CapabilityInput::Unknown) {
*data = IsEnabledi(target, index);
return;
}
switch (target) {
// ARB_viewport_array queries the indexed rectangles through glGetIntegeri_v as well
// (gl4cMultiBindTests and the viewport_array group both do). The frontend keeps one
// viewport and one scissor box, so every in-range index reports that one.
// (gl4cMultiBindTests and the viewport_array group both do).
case GL_VIEWPORT:
case GL_SCISSOR_BOX:
case GL_DEPTH_RANGE: {
if (!ValidateViewportQueryIndex(index, __func__)) return;
GetIntegerv(target, data);
GLfloat values[4] = {};
ReadIndexedViewportStateFloat(target, index, values);
const GLsizei components = IndexedViewportQueryComponents(target);
for (GLsizei i = 0; i < components; ++i) {
// Round, not truncate: glGetIntegerv on floating-point state rounds to nearest
// (GL 4.6 core 22.2), so a 255.875-wide viewport reads back as 256 and not 255.
data[i] = static_cast<GLint>(std::lround(values[i]));
}
return;
}
// The vertex buffer binding points of the vertex array object that is bound. Indexed by
// binding point, not by attribute (GL 4.6 core 10.3.1).
case GL_VERTEX_BINDING_BUFFER:
@@ -927,7 +1005,10 @@ namespace MobileGL::MG_Impl::GLImpl {
}
if (IsIndexedViewportQuery(target)) {
if (!ValidateViewportQueryIndex(index, __func__)) return;
GetFloatv(target, data);
// Verbatim, NOT via the integer width: the viewport is float state and
// KHR-GL43.viewport_array.viewport_api compares the read-back with ==, so a
// glViewportIndexedf(i, 0.125f, ...) has to come back as 0.125f exactly.
ReadIndexedViewportStateFloat(target, index, data);
return;
}
GLint ints[4] = {};
@@ -944,7 +1025,12 @@ namespace MobileGL::MG_Impl::GLImpl {
}
if (IsIndexedViewportQuery(target)) {
if (!ValidateViewportQueryIndex(index, __func__)) return;
GetDoublev(target, data);
GLfloat values[4] = {};
ReadIndexedViewportStateFloat(target, index, values);
const GLsizei components = IndexedViewportQueryComponents(target);
for (GLsizei i = 0; i < components; ++i) {
data[i] = static_cast<GLdouble>(values[i]);
}
return;
}
GLint ints[4] = {};
@@ -1020,7 +1106,12 @@ namespace MobileGL::MG_Impl::GLImpl {
// frontend-only value simply is not in the driver's table.
GLint values[4] = {};
GetIntegeri_v(target, index, values);
*data = static_cast<GLint64>(values[0]);
// The viewport-array rectangles are the only multi-component indexed state here; every
// other pname is scalar, so widening element 0 alone would silently truncate them.
const GLsizei components = IsIndexedViewportQuery(target) ? IndexedViewportQueryComponents(target) : 1;
for (GLsizei i = 0; i < components; ++i) {
data[i] = static_cast<GLint64>(values[i]);
}
}
void GetInteger64v(GLenum pname, GLint64* params) {
@@ -2192,7 +2283,15 @@ namespace MobileGL::MG_Impl::GLImpl {
params[1] = dynamicParameters.MaxViewportHeight;
break;
case GL_MAX_VIEWPORTS:
*params = dynamicParameters.MaxViewports;
// The frontend's own state width, not the backend's device limit. GL 4.3 core
// requires MAX_VIEWPORTS >= 16 and every indexed viewport entry point validates
// against RenderStateParameters::MAX_VIEWPORTS, so reporting anything else would
// either advertise viewports the state cannot hold or reject indices it can. A
// Vulkan device without the multiViewport feature reports maxViewports == 1, which
// limits what can be RASTERIZED to more than one rectangle (see the multiViewport
// gate in VulkanRenderer), not what the GL state can hold; caps.MaxViewports keeps
// carrying that device number for exactly that decision.
*params = static_cast<GLint>(RenderStateParameters::MAX_VIEWPORTS);
break;
case GL_MINOR_VERSION:
*params = rendererInfo.RendererGLInfo.TargetGLVersion.Minor;
@@ -20,28 +20,118 @@ namespace MobileGL::MG_Impl::GLImpl {
return std::clamp(static_cast<Float>(value), 0.0f, 1.0f);
}
static Bool ValidateIndexedBlendCapability(GLenum target, GLuint index, const char* functionName) {
if (target != GL_BLEND) {
// GL 4.6 core 17.3.2 and 22.1 give exactly two indexed capabilities: GL_BLEND, indexed by
// draw buffer, and GL_SCISSOR_TEST, indexed by viewport. They have DIFFERENT bounds
// (MAX_DRAW_BUFFERS vs MAX_VIEWPORTS), so the limit is picked per target rather than shared.
static Bool ValidateIndexedCapability(GLenum target, GLuint index, const char* functionName) {
GLuint limit = 0;
const char* indexName = nullptr;
switch (target) {
case GL_BLEND:
limit = MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS;
indexName = "Buffer";
break;
case GL_SCISSOR_TEST:
limit = RenderStateParameters::MAX_VIEWPORTS;
indexName = "Viewport";
break;
default:
MG_State::pGLContext->RecordError(
ErrorCode::InvalidEnum,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
"Only GL_BLEND is supported for indexed capability state."));
"Only GL_BLEND and GL_SCISSOR_TEST are supported for indexed "
"capability state."));
return false;
}
if (index >= MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS) {
if (index >= limit) {
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>(
"MG_Impl/GLImpl", functionName,
"Buffer index " + std::to_string(index) + " is out of range. Max supported is " +
std::to_string(MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS - 1) + "."));
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
String(indexName) + " index " + std::to_string(index) +
" is out of range. Max supported is " + std::to_string(limit - 1) +
"."));
return false;
}
return true;
}
// ------------------ ARB_viewport_array parameter validation ------------------
// All three families share the same two shapes, so they share the two checkers. GL 4.6 core
// 13.6.1/17.3.2: an out-of-range index is GL_INVALID_VALUE, and so is a negative width or
// height. `first + count == MAX_VIEWPORTS` is LEGAL - only strictly greater is an error,
// which KHR-GL43.viewport_array.api_errors checks explicitly in both directions.
static Bool ValidateViewportIndex(GLuint index, const char* functionName) {
if (index < RenderStateParameters::MAX_VIEWPORTS) return true;
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
"Viewport index " + std::to_string(index) +
" is out of range. Max supported is " +
std::to_string(RenderStateParameters::MAX_VIEWPORTS - 1) + "."));
return false;
}
static Bool ValidateViewportRange(GLuint first, GLsizei count, const char* functionName) {
if (count < 0) {
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "count must not be negative."));
return false;
}
// Widened before adding: first is a GLuint and count a GLsizei, so `first + count` in
// 32 bits can wrap past MAX_VIEWPORTS and let an out-of-range range through.
const Uint64 last = static_cast<Uint64>(first) + static_cast<Uint64>(count);
if (last > RenderStateParameters::MAX_VIEWPORTS) {
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
"first (" + std::to_string(first) + ") + count (" +
std::to_string(count) + ") exceeds GL_MAX_VIEWPORTS (" +
std::to_string(RenderStateParameters::MAX_VIEWPORTS) + ")."));
return false;
}
return true;
}
template <typename T>
static Bool ValidateNonNegativeExtent(T width, T height, const char* functionName) {
if (width >= T(0) && height >= T(0)) return true;
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "Width and height must be non-negative."));
return false;
}
// The array forms are all-or-nothing: one bad element rejects the whole call with a SINGLE
// GL_INVALID_VALUE and leaves every rectangle untouched. api_errors relies on both halves -
// it passes a full 16-element array with exactly one negative extent and then asserts the
// error queue holds exactly one entry.
template <typename T>
static Bool ValidateArrayExtents(GLsizei count, const T* v, const char* functionName) {
for (GLsizei i = 0; i < count; ++i) {
if (v[i * 4 + 2] >= T(0) && v[i * 4 + 3] >= T(0)) continue;
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
"Width and height must be non-negative (element " + std::to_string(i) +
")."));
return false;
}
return true;
}
static Bool ValidateNonNullArray(const void* v, const char* functionName) {
if (v != nullptr) return true;
MG_State::pGLContext->RecordError(
ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "value pointer cannot be null."));
return false;
}
static Bool TryConvertBlendEquation(GLenum mode, const char* functionName,
::MobileGL::BlendEquation& outEquation) {
outEquation = MG_Util::ConvertGLEnumToBlendEquation(mode);
@@ -93,16 +183,70 @@ namespace MobileGL::MG_Impl::GLImpl {
}
void Viewport_State(GLint x, GLint y, GLsizei width, GLsizei height) {
if (width < 0 || height < 0) {
MG_State::pGLContext->RecordError(ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "Viewport_State",
"Width abd height must be non-negative."));
return;
}
if (!ValidateNonNegativeExtent(width, height, "Viewport_State")) return;
MG_State::pGLContext->SetViewport(IntVec4(x, y, width, height));
}
// ------------------ ARB_viewport_array setters ------------------
void ViewportArrayv_State(GLuint first, GLsizei count, const GLfloat* v) {
if (!ValidateViewportRange(first, count, "ViewportArrayv_State")) return;
if (count == 0) return;
if (!ValidateNonNullArray(v, "ViewportArrayv_State")) return;
if (!ValidateArrayExtents(count, v, "ViewportArrayv_State")) return;
for (GLsizei i = 0; i < count; ++i) {
MG_State::pGLContext->SetViewportIndexed(first + static_cast<GLuint>(i),
FloatVec4(v[i * 4 + 0], v[i * 4 + 1], v[i * 4 + 2], v[i * 4 + 3]));
}
}
void ViewportIndexedf_State(GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) {
if (!ValidateViewportIndex(index, "ViewportIndexedf_State")) return;
if (!ValidateNonNegativeExtent(w, h, "ViewportIndexedf_State")) return;
MG_State::pGLContext->SetViewportIndexed(index, FloatVec4(x, y, w, h));
}
void ScissorArrayv_State(GLuint first, GLsizei count, const GLint* v) {
if (!ValidateViewportRange(first, count, "ScissorArrayv_State")) return;
if (count == 0) return;
if (!ValidateNonNullArray(v, "ScissorArrayv_State")) return;
if (!ValidateArrayExtents(count, v, "ScissorArrayv_State")) return;
for (GLsizei i = 0; i < count; ++i) {
MG_State::pGLContext->SetScissorBoxIndexed(first + static_cast<GLuint>(i),
IntVec4(v[i * 4 + 0], v[i * 4 + 1], v[i * 4 + 2], v[i * 4 + 3]));
}
}
void ScissorIndexed_State(GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) {
if (!ValidateViewportIndex(index, "ScissorIndexed_State")) return;
if (!ValidateNonNegativeExtent(width, height, "ScissorIndexed_State")) return;
MG_State::pGLContext->SetScissorBoxIndexed(index, IntVec4(left, bottom, width, height));
}
void DepthRangeArrayv_State(GLuint first, GLsizei count, const GLdouble* v) {
if (!ValidateViewportRange(first, count, "DepthRangeArrayv_State")) return;
if (count == 0) return;
if (!ValidateNonNullArray(v, "DepthRangeArrayv_State")) return;
for (GLsizei i = 0; i < count; ++i) {
MG_State::pGLContext->SetDepthRangeIndexed(
first + static_cast<GLuint>(i),
FloatVec2(ClampUnitFloat(static_cast<GLfloat>(v[i * 2 + 0])),
ClampUnitFloat(static_cast<GLfloat>(v[i * 2 + 1]))));
}
}
void DepthRangeIndexed_State(GLuint index, GLdouble n, GLdouble f) {
if (!ValidateViewportIndex(index, "DepthRangeIndexed_State")) return;
MG_State::pGLContext->SetDepthRangeIndexed(
index, FloatVec2(ClampUnitFloat(static_cast<GLfloat>(n)), ClampUnitFloat(static_cast<GLfloat>(f))));
}
void StencilOpSeparate_State(GLenum face, GLenum sfail, GLenum dpfail, GLenum dppass) {
Bool applyFront = false;
Bool applyBack = false;
@@ -175,12 +319,7 @@ namespace MobileGL::MG_Impl::GLImpl {
}
void Scissor_State(GLint x, GLint y, GLsizei width, GLsizei height) {
if (width < 0 || height < 0) {
MG_State::pGLContext->RecordError(ErrorCode::InvalidValue,
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "Scissor_State",
"Width abd height must be non-negative."));
return;
}
if (!ValidateNonNegativeExtent(width, height, "Scissor_State")) return;
MG_State::pGLContext->SetScissorBox(IntVec4(x, y, width, height));
}
@@ -336,7 +475,7 @@ namespace MobileGL::MG_Impl::GLImpl {
}
GLboolean IsEnabledi_State(GLenum target, GLuint index) {
if (!ValidateIndexedBlendCapability(target, index, "IsEnabledi_State")) {
if (!ValidateIndexedCapability(target, index, "IsEnabledi_State")) {
return GL_FALSE;
}
@@ -392,7 +531,14 @@ namespace MobileGL::MG_Impl::GLImpl {
}
GLint values[4] = {};
GetIntegeri_v(target, index, values);
*data = values[0] != 0 ? GL_TRUE : GL_FALSE;
// The ARB_viewport_array rectangles are the only multi-component indexed state that
// reaches here; writing element 0 alone would leave the caller's other three untouched.
const GLsizei components = target == GL_VIEWPORT || target == GL_SCISSOR_BOX
? 4
: (target == GL_DEPTH_RANGE ? 2 : 1);
for (GLsizei i = 0; i < components; ++i) {
data[i] = values[i] != 0 ? GL_TRUE : GL_FALSE;
}
}
GLboolean IsEnabled_State(GLenum cap) {
@@ -725,7 +871,7 @@ namespace MobileGL::MG_Impl::GLImpl {
}
void Disablei_State(GLenum target, GLuint index) {
if (!ValidateIndexedBlendCapability(target, index, "Disablei_State")) {
if (!ValidateIndexedCapability(target, index, "Disablei_State")) {
return;
}
@@ -743,7 +889,7 @@ namespace MobileGL::MG_Impl::GLImpl {
}
void Enablei_State(GLenum target, GLuint index) {
if (!ValidateIndexedBlendCapability(target, index, "Enablei_State")) {
if (!ValidateIndexedCapability(target, index, "Enablei_State")) {
return;
}
@@ -797,6 +943,44 @@ namespace MobileGL::MG_Impl::GLImpl {
Viewport_State(x, y, width, height);
}
void ViewportArrayv(GLuint first, GLsizei count, const GLfloat* v) {
ViewportArrayv_State(first, count, v);
}
void ViewportIndexedf(GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) {
ViewportIndexedf_State(index, x, y, w, h);
}
void ViewportIndexedfv(GLuint index, const GLfloat* v) {
// The index is validated before the pointer is touched: glViewportIndexedfv(MAX, nullptr)
// must be one GL_INVALID_VALUE, not a null dereference.
if (!ValidateViewportIndex(index, "ViewportIndexedfv")) return;
if (!ValidateNonNullArray(v, "ViewportIndexedfv")) return;
ViewportIndexedf_State(index, v[0], v[1], v[2], v[3]);
}
void ScissorArrayv(GLuint first, GLsizei count, const GLint* v) {
ScissorArrayv_State(first, count, v);
}
void ScissorIndexed(GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) {
ScissorIndexed_State(index, left, bottom, width, height);
}
void ScissorIndexedv(GLuint index, const GLint* v) {
if (!ValidateViewportIndex(index, "ScissorIndexedv")) return;
if (!ValidateNonNullArray(v, "ScissorIndexedv")) return;
ScissorIndexed_State(index, v[0], v[1], v[2], v[3]);
}
void DepthRangeArrayv(GLuint first, GLsizei count, const GLdouble* v) {
DepthRangeArrayv_State(first, count, v);
}
void DepthRangeIndexed(GLuint index, GLdouble n, GLdouble f) {
DepthRangeIndexed_State(index, n, f);
}
void StencilOpSeparate(GLenum face, GLenum sfail, GLenum dpfail, GLenum dppass) {
StencilOpSeparate_State(face, sfail, dpfail, dppass);
}
@@ -20,6 +20,16 @@ namespace MobileGL::MG_Impl::GLImpl {
void Enablei(GLenum target, GLuint index);
void BlendFunc(GLenum sfactor, GLenum dfactor);
void Viewport(GLint x, GLint y, GLsizei width, GLsizei height);
// ARB_viewport_array (core since GL 4.1). Every one of these addresses the same 16-element
// indexed state the classic glViewport/glScissor/glDepthRange trio broadcasts to.
void ViewportArrayv(GLuint first, GLsizei count, const GLfloat* v);
void ViewportIndexedf(GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h);
void ViewportIndexedfv(GLuint index, const GLfloat* v);
void ScissorArrayv(GLuint first, GLsizei count, const GLint* v);
void ScissorIndexed(GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height);
void ScissorIndexedv(GLuint index, const GLint* v);
void DepthRangeArrayv(GLuint first, GLsizei count, const GLdouble* v);
void DepthRangeIndexed(GLuint index, GLdouble n, GLdouble f);
void StencilOpSeparate(GLenum face, GLenum sfail, GLenum dpfail, GLenum dppass);
void StencilOp(GLenum fail, GLenum zfail, GLenum zpass);
void StencilMaskSeparate(GLenum face, GLuint mask);
@@ -63,10 +63,12 @@ add_executable(MobileGLIntegrationTest
Scenarios/DepthStencilReadbackMatrixScenario.cpp
Scenarios/DepthStencilReadbackAttachmentShapeScenario.cpp
Scenarios/ClipDistanceScenario.cpp
Scenarios/ViewportArrayScenario.cpp
Scenarios/SsboArrayLengthScenario.cpp
Scenarios/DoublePrecisionScenario.cpp
Scenarios/UniformInitializerScenario.cpp
Scenarios/SwizzleAccessRoutineScenario.cpp
Scenarios/Program203FirstReductionScenario.cpp
Scenarios/ProgramPipelineScenario.cpp
Scenarios/ImageLoadStoreSsoScenario.cpp
Scenarios/ImageTargetKindScenario.cpp
@@ -80,6 +82,7 @@ add_executable(MobileGLIntegrationTest
Scenarios/VertexArrayEnableDisableScenario.cpp
Scenarios/CopyImageLevelRangeScenario.cpp
Scenarios/CopyImageLayeredScenario.cpp
Scenarios/LayeredAttachmentBarrierScenario.cpp
)
target_include_directories(MobileGLIntegrationTest PRIVATE
@@ -15,6 +15,11 @@
#include <ostream>
#include <sstream>
#if defined(_WIN32)
#define WIN32_LEAN_AND_MEAN
#include <windows.h>
#endif
// MobileGL's own headers, in the order MobileGL/Includes.h uses them: GL/gl.h
// first, then glcorearb.h for the 3.x+ entry points. This binary links
// MobileGL_s, so every gl*/egl* below binds to MobileGL's implementation, not
@@ -53,6 +58,37 @@ namespace MGITest {
constexpr int kSurfaceWidth = 128;
constexpr int kSurfaceHeight = 96;
#if defined(_WIN32)
HWND g_testWindow = nullptr;
HWND CreateTestWindow() {
static const wchar_t* const kClassName = L"MobileGLIntegrationTestWindow";
static bool registered = false;
if (!registered) {
WNDCLASSW windowClass{};
windowClass.lpfnWndProc = DefWindowProcW;
windowClass.hInstance = GetModuleHandleW(nullptr);
windowClass.lpszClassName = kClassName;
if (RegisterClassW(&windowClass) == 0 && GetLastError() != ERROR_CLASS_ALREADY_EXISTS) {
return nullptr;
}
registered = true;
}
return CreateWindowExW(0, kClassName, L"MobileGL Integration Test", WS_OVERLAPPEDWINDOW,
CW_USEDEFAULT, CW_USEDEFAULT, kSurfaceWidth, kSurfaceHeight, nullptr, nullptr,
GetModuleHandleW(nullptr), nullptr);
}
#endif
bool UseWindowSurface() {
#if defined(_WIN32)
const char* value = std::getenv("MOBILEGL_ITEST_WINDOW_SURFACE");
return value != nullptr && value[0] != '\0' && std::strcmp(value, "0") != 0;
#else
return false;
#endif
}
std::string EnvOr(const char* name, const char* fallback) {
const char* value = std::getenv(name);
return (value != nullptr && value[0] != '\0') ? std::string(value) : std::string(fallback);
@@ -134,8 +170,9 @@ namespace MGITest {
return 3;
}
const bool useWindowSurface = UseWindowSurface();
const EGLint configAttribs[] = {EGL_SURFACE_TYPE,
EGL_PBUFFER_BIT,
useWindowSurface ? EGL_WINDOW_BIT : EGL_PBUFFER_BIT,
EGL_RED_SIZE,
8,
EGL_GREEN_SIZE,
@@ -152,7 +189,9 @@ namespace MGITest {
EGLConfig config = nullptr;
EGLint configCount = 0;
if (eglChooseConfig(display, configAttribs, &config, 1, &configCount) != EGL_TRUE || configCount < 1) {
outReason = WithEglError("eglChooseConfig found no pbuffer-capable RGBA8/D24 config");
outReason = WithEglError(useWindowSurface
? "eglChooseConfig found no window-capable RGBA8/D24 config"
: "eglChooseConfig found no pbuffer-capable RGBA8/D24 config");
return 4;
}
@@ -166,10 +205,23 @@ namespace MGITest {
return 5;
}
const EGLint pbufferAttribs[] = {EGL_WIDTH, kSurfaceWidth, EGL_HEIGHT, kSurfaceHeight, EGL_NONE};
EGLSurface surface = eglCreatePbufferSurface(display, config, pbufferAttribs);
EGLSurface surface = EGL_NO_SURFACE;
if (useWindowSurface) {
#if defined(_WIN32)
if (g_testWindow == nullptr) g_testWindow = CreateTestWindow();
if (g_testWindow == nullptr) {
outReason = "failed to create the Windows integration-test window";
return 6;
}
surface = eglCreateWindowSurface(display, config, g_testWindow, nullptr);
#endif
} else {
const EGLint pbufferAttribs[] = {EGL_WIDTH, kSurfaceWidth, EGL_HEIGHT, kSurfaceHeight, EGL_NONE};
surface = eglCreatePbufferSurface(display, config, pbufferAttribs);
}
if (surface == EGL_NO_SURFACE) {
outReason = WithEglError("eglCreatePbufferSurface failed");
outReason = WithEglError(useWindowSurface ? "eglCreateWindowSurface failed"
: "eglCreatePbufferSurface failed");
return 6;
}
// The step that brings the whole backend up (DirectVulkan creates its
@@ -491,6 +543,12 @@ namespace MGITest {
if (m_context != nullptr) eglDestroyContext(display, static_cast<EGLContext>(m_context));
if (m_surface != nullptr) eglDestroySurface(display, static_cast<EGLSurface>(m_surface));
eglTerminate(display);
#if defined(_WIN32)
if (g_testWindow != nullptr) {
DestroyWindow(g_testWindow);
g_testWindow = nullptr;
}
#endif
m_context = nullptr;
m_surface = nullptr;
m_display = nullptr;
@@ -199,5 +199,61 @@ namespace MGITest {
"derived component limits are computed in";
}
// ARB_viewport_array's own limits. They are advertised from three different places -
// GL_MAX_VIEWPORTS from the frontend's indexed state width, the bounds range and the
// subpixel bits from the backend caps table - and each backend fills that table from a
// different source, so all three are checked on both lanes.
//
// GL_VIEWPORT_BOUNDS_RANGE is the one that shipped wrong: GLES has no such query, the
// DirectGLES loader's glGetFloatv(GL_VIEWPORT_BOUNDS_RANGE) therefore raised
// GL_INVALID_ENUM and left the probe's zero-initialized array in place, and MobileGL
// advertised [0, 0] - a range that admits no viewport origin at all, and the check that
// kept KHR-GL43.viewport_array.queries red on Espryt after the indexed-state work.
TEST_F(AdvertisedLimitsScenario, ViewportArrayLimitsMeetTheirGL43Floors) {
GLint maxViewports = -1;
glGetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
EXPECT_GE(maxViewports, 16) << "GL 4.3 core table 23.53 sets the MAX_VIEWPORTS minimum at 16";
EXPECT_LE(maxViewports, 256) << "one viewport rectangle of indexed state is allocated per advertised "
"viewport, and the CTS sizes its arrays off this number";
GLfloat boundsRange[2] = {1.0f, -1.0f};
glGetFloatv(GL_VIEWPORT_BOUNDS_RANGE, boundsRange);
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
EXPECT_LE(boundsRange[0], -32768.0f)
<< "GL 4.6 core table 23.60 sets the VIEWPORT_BOUNDS_RANGE minimum at [-32768, 32767]; got ["
<< boundsRange[0] << ", " << boundsRange[1] << "]";
EXPECT_GE(boundsRange[1], 32767.0f)
<< "GL 4.6 core table 23.60 sets the VIEWPORT_BOUNDS_RANGE minimum at [-32768, 32767]; got ["
<< boundsRange[0] << ", " << boundsRange[1] << "]";
// KNOWN INFIDELITY, pinned here rather than hidden. MobileGL reports the driver's own
// VIEWPORT_SUBPIXEL_BITS (4 on llvmpipe, i.e. 1/16-pixel viewport precision), but the
// float viewport rectangle glViewportIndexedf stores is snapped to integers on its
// way to both backends (ComputeGLViewport, DirectGLES SyncRenderState). The STATE
// round trip is exact - which is all KHR-GL43.viewport_array.viewport_api checks, and
// all this cluster set out to fix - so the gap is in rasterization only: a fractional
// viewport origin rasterizes as if it had been rounded. Nothing in the suite or in
// Minecraft sets one. Only the spec floor is asserted; tightening this to EQ(0) would
// mean advertising no subpixel precision at all, which is a separate decision about a
// limit MobileGL currently passes through from the driver.
GLint subpixelBits = -1;
glGetIntegerv(GL_VIEWPORT_SUBPIXEL_BITS, &subpixelBits);
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
EXPECT_GE(subpixelBits, 0) << "GL 4.6 core table 23.60: VIEWPORT_SUBPIXEL_BITS has a minimum of 0, and "
"a negative value is what a sign-flipped uint32 looks like";
GLint viewportDims[2] = {-1, -1};
glGetIntegerv(GL_MAX_VIEWPORT_DIMS, viewportDims);
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
GLint maxRenderbufferSize = -1;
glGetIntegerv(GL_MAX_RENDERBUFFER_SIZE, &maxRenderbufferSize);
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
// GL 4.6 core 13.6.1: MAX_VIEWPORT_DIMS must be at least as large as the largest
// renderable surface, or a full-size framebuffer could not be fully viewported.
EXPECT_GE(viewportDims[0], maxRenderbufferSize);
EXPECT_GE(viewportDims[1], maxRenderbufferSize);
}
} // namespace
} // namespace MGITest
@@ -0,0 +1,378 @@
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/LayeredAttachmentBarrierScenario.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Scenario - A TRANSFER OFF A NON-ZERO ATTACHMENT LAYER READS THE LAYER THE BARRIER MOVED.
//
// Every transfer DirectVulkan performs against a framebuffer attachment is three commands: a
// barrier that puts the image in TRANSFER_SRC/DST, the copy or blit itself, and a barrier that
// puts it back. The copy names the attachment's layer - glFramebufferTextureLayer(.., layer) ends
// up in `srcSubresource.baseArrayLayer` - but TransitionImageLayout used to emit `layerCount = 1`
// from `baseArrayLayer 0`, so for every attachment on a layer above zero the barrier moved layer 0
// and the copy read layer N. The layer the transfer touched was never transitioned: it sat in
// COLOR_ATTACHMENT_OPTIMAL (or DEPTH_STENCIL_ATTACHMENT_OPTIMAL) while being read as TRANSFER_SRC.
//
// That is undefined behaviour, not a guaranteed wrong pixel: a layout is a compression/tiling
// promise, so a driver that stores both layouts identically returns the right bytes anyway. The
// software lanes (lavapipe) are exactly such a driver, which is why this scenario is paired with a
// validation-layer run - the layer names the mismatch outright
// (VUID-vkCmdCopyImageToBuffer-srcImageLayout-00189, "srcImageLayout ... doesn't match the actual
// current layout") where the pixels here cannot. On a tiler that really does re-tile per layout,
// these are the reads that come back as garbage.
//
// The four cases below are the four transfer paths that take an attachment layer from GL:
//
// glReadPixels (colour) -> VulkanRenderer::ReadPixels
// glBlitFramebuffer (colour) -> VulkanRenderer::BlitNamedFramebuffer
// glReadPixels (GL_DEPTH_COMPONENT) -> VulkanRenderer::ReadDepthStencilImageToClient
// glBlitFramebuffer (GL_DEPTH_BUFFER_BIT) -> VulkanRenderer::BlitNamedFramebuffer, depth leg
//
// Each one renders or clears INTO the non-zero layer first, so the image is genuinely sitting in
// its attachment layout when the transfer starts - a scenario that only uploaded texels would
// leave it in a transfer layout already and the mismatched barrier would be a no-op.
//
// Every case also asserts the layers it did not name still hold their own fill, so a backend that
// "fixed" the miss by transferring the whole image passes neither half.
//
// DirectGLES is the control: it hands the same calls to the driver, so a failure on both backends
// means the scenario is wrong and a failure on DirectVulkan alone means Magma is.
#include <cmath>
#include <string>
#include <vector>
#include "../Harness/HeadlessGL.h"
#include "../Harness/ScenarioFixture.h"
#ifdef GLAPI
#undef GLAPI
#endif
#define GL_GLEXT_PROTOTYPES
#include <GL/gl.h>
#include <GL/glcorearb.h>
#undef GL_GLEXT_PROTOTYPES
namespace MGITest {
namespace {
constexpr int kWidth = 8;
constexpr int kHeight = 8;
// Four layers with the subject at index 2: layers on both sides of it stay untouched, so
// "moved the whole image" and "moved layer 0" are both distinguishable from correct.
constexpr int kLayers = 4;
constexpr int kSubjectLayer = 2;
// A value no correct read can produce, so "the backend wrote nothing" fails loudly.
constexpr float kDepthPoison = 0.2f;
std::string Describe(const Rgba8& color) {
return "(" + std::to_string(color.r) + ", " + std::to_string(color.g) + ", " + std::to_string(color.b) +
", " + std::to_string(color.a) + ")";
}
// Per-layer fill, uniform within a layer: the defect is about WHICH layer is addressed, and
// a value that also varied inside the layer would make the assertions depend on row order.
Rgba8 LayerFill(int layer) {
return {static_cast<GLubyte>(17 + layer * 30), static_cast<GLubyte>(200 - layer * 25),
static_cast<GLubyte>(60 + layer * 40), 255};
}
// What the draw paints - matches kFS below, and is deliberately none of the LayerFill
// values so "the draw never landed" cannot read as a pass.
constexpr Rgba8 kPaintedColor{26, 51, 204, 255};
constexpr const char* kVS = R"(#version 330 core
in vec2 aPos;
void main() { gl_Position = vec4(aPos, 0.0, 1.0); }
)";
constexpr const char* kFS = R"(#version 330 core
out vec4 o_color;
void main() { o_color = vec4(0.1, 0.2, 0.8, 1.0); }
)";
void DrawFullViewportQuad(unsigned int program) {
static const float kQuad[] = {-1.0f, -1.0f, 1.0f, -1.0f, -1.0f, 1.0f, 1.0f, 1.0f};
GLuint vao = 0, vbo = 0;
glGenVertexArrays(1, &vao);
glBindVertexArray(vao);
glGenBuffers(1, &vbo);
glBindBuffer(GL_ARRAY_BUFFER, vbo);
glBufferData(GL_ARRAY_BUFFER, sizeof(kQuad), kQuad, GL_STATIC_DRAW);
glEnableVertexAttribArray(0);
glVertexAttribPointer(0, 2, GL_FLOAT, GL_FALSE, 2 * sizeof(float), nullptr);
glUseProgram(program);
glDrawArrays(GL_TRIANGLE_STRIP, 0, 4);
glBindVertexArray(0);
glDeleteBuffers(1, &vbo);
glDeleteVertexArrays(1, &vao);
}
class LayeredAttachmentBarrierScenario : public ScenarioTest {
protected:
void SetUp() override {
ScenarioTest::SetUp();
if (!Ready()) return;
std::string error;
m_program = CompileProgram(kVS, kFS, &error);
ASSERT_NE(m_program, 0u) << error;
}
void TearDown() override {
if (!Ready()) return;
glBindFramebuffer(GL_FRAMEBUFFER, 0);
for (const GLuint fbo : m_fbos) {
glDeleteFramebuffers(1, &fbo);
}
m_fbos.clear();
for (const GLuint texture : m_textures) {
glDeleteTextures(1, &texture);
}
m_textures.clear();
if (m_program != 0) {
glUseProgram(0);
glDeleteProgram(m_program);
m_program = 0;
}
}
// An RGBA8 2D array with a different uniform colour per layer.
GLuint MakeColorArray() {
GLuint texture = 0;
glGenTextures(1, &texture);
m_textures.push_back(texture);
glBindTexture(GL_TEXTURE_2D_ARRAY, texture);
glTexStorage3D(GL_TEXTURE_2D_ARRAY, 1, GL_RGBA8, kWidth, kHeight, kLayers);
glTexParameteri(GL_TEXTURE_2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
glTexParameteri(GL_TEXTURE_2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
for (int layer = 0; layer < kLayers; ++layer) {
const std::vector<Rgba8> texels(static_cast<std::size_t>(kWidth) * kHeight, LayerFill(layer));
glTexSubImage3D(GL_TEXTURE_2D_ARRAY, 0, 0, 0, layer, kWidth, kHeight, 1, GL_RGBA,
GL_UNSIGNED_BYTE, texels.data());
}
glBindTexture(GL_TEXTURE_2D_ARRAY, 0);
return texture;
}
// A depth 2D array. No initial upload: depth arrays are filled by clearing through an
// attachment, which is also the state the transfer paths have to cope with.
GLuint MakeDepthArray() {
GLuint texture = 0;
glGenTextures(1, &texture);
m_textures.push_back(texture);
glBindTexture(GL_TEXTURE_2D_ARRAY, texture);
glTexStorage3D(GL_TEXTURE_2D_ARRAY, 1, GL_DEPTH_COMPONENT24, kWidth, kHeight, kLayers);
glTexParameteri(GL_TEXTURE_2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
glTexParameteri(GL_TEXTURE_2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
glBindTexture(GL_TEXTURE_2D_ARRAY, 0);
return texture;
}
// One FBO naming `layer` of the given arrays. Depth is optional (0 = colour only).
GLuint MakeLayerFbo(GLuint colorArray, GLuint depthArray, int layer) {
GLuint fbo = 0;
glGenFramebuffers(1, &fbo);
m_fbos.push_back(fbo);
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
glFramebufferTextureLayer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, colorArray, 0, layer);
if (depthArray != 0) {
glFramebufferTextureLayer(GL_FRAMEBUFFER, GL_DEPTH_ATTACHMENT, depthArray, 0, layer);
}
EXPECT_EQ(glCheckFramebufferStatus(GL_FRAMEBUFFER), static_cast<GLenum>(GL_FRAMEBUFFER_COMPLETE))
<< "layer " << layer << " is not attachable";
return fbo;
}
// glReadPixels of one whole layer, through an FBO that names it.
Rgba8 ReadLayer(GLuint colorArray, int layer) {
const GLuint fbo = MakeLayerFbo(colorArray, 0, layer);
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
glReadBuffer(GL_COLOR_ATTACHMENT0);
glPixelStorei(GL_PACK_ALIGNMENT, 1);
std::vector<Rgba8> pixels(static_cast<std::size_t>(kWidth) * kHeight, Rgba8{});
glReadPixels(0, 0, kWidth, kHeight, GL_RGBA, GL_UNSIGNED_BYTE, pixels.data());
glBindFramebuffer(GL_FRAMEBUFFER, 0);
// The fill is uniform within a layer, so any disagreement between texels is itself
// a failure - reported here rather than silently reduced to pixels[0].
for (std::size_t i = 1; i < pixels.size(); ++i) {
EXPECT_TRUE(pixels[i] == pixels[0])
<< "layer " << layer << " is not uniform: texel 0 is " << Describe(pixels[0]) << ", texel "
<< i << " is " << Describe(pixels[i]);
}
return pixels[0];
}
// Every layer but `changed` still holds its own fill.
void ExpectOtherLayersUntouched(GLuint colorArray, int changed, const char* what) {
for (int layer = 0; layer < kLayers; ++layer) {
if (layer == changed) continue;
const Rgba8 actual = ReadLayer(colorArray, layer);
EXPECT_TRUE(actual == LayerFill(layer))
<< what << ": layer " << layer << " should still hold its fill but is " << Describe(actual)
<< ", expected " << Describe(LayerFill(layer));
}
}
float ReadDepthAt(int x, int y) const {
float depth = kDepthPoison;
glReadPixels(x, y, 1, 1, GL_DEPTH_COMPONENT, GL_FLOAT, &depth);
return depth;
}
std::vector<GLuint> m_textures;
std::vector<GLuint> m_fbos;
unsigned int m_program = 0;
};
// glReadPixels straight off a layer that was just rendered to. The image is in
// COLOR_ATTACHMENT_OPTIMAL when the readback barrier runs, so the barrier and the copy
// disagreeing about the layer is a live layout mismatch, not a bookkeeping detail.
TEST_F(LayeredAttachmentBarrierScenario, ReadPixelsOffRenderedNonZeroLayer) {
if (!Ready()) return;
const GLuint colorArray = MakeColorArray();
ASSERT_EQ(FirstGLError(), 0u) << "texture setup failed";
const GLuint fbo = MakeLayerFbo(colorArray, 0, kSubjectLayer);
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
glViewport(0, 0, kWidth, kHeight);
glDisable(GL_SCISSOR_TEST);
glDisable(GL_DEPTH_TEST);
glDrawBuffer(GL_COLOR_ATTACHMENT0);
DrawFullViewportQuad(m_program);
glReadBuffer(GL_COLOR_ATTACHMENT0);
glPixelStorei(GL_PACK_ALIGNMENT, 1);
std::vector<Rgba8> pixels(static_cast<std::size_t>(kWidth) * kHeight, Rgba8{});
glReadPixels(0, 0, kWidth, kHeight, GL_RGBA, GL_UNSIGNED_BYTE, pixels.data());
glBindFramebuffer(GL_FRAMEBUFFER, 0);
EXPECT_EQ(FirstGLError(), 0u);
for (std::size_t i = 0; i < pixels.size(); ++i) {
ASSERT_NEAR(pixels[i].r, kPaintedColor.r, 2)
<< "texel " << i << " of the rendered layer is " << Describe(pixels[i]);
ASSERT_NEAR(pixels[i].g, kPaintedColor.g, 2) << "texel " << i;
ASSERT_NEAR(pixels[i].b, kPaintedColor.b, 2) << "texel " << i;
}
ExpectOtherLayersUntouched(colorArray, kSubjectLayer, "readback off a rendered layer");
}
// glBlitFramebuffer between two non-zero layers of two different arrays. Both endpoints are
// above layer 0, so the source and destination barriers are each wrong on their own side.
TEST_F(LayeredAttachmentBarrierScenario, BlitBetweenNonZeroColorLayers) {
if (!Ready()) return;
const GLuint sourceArray = MakeColorArray();
const GLuint destinationArray = MakeColorArray();
ASSERT_EQ(FirstGLError(), 0u) << "texture setup failed";
constexpr int kSourceLayer = 3;
constexpr int kDestinationLayer = 1;
const GLuint sourceFbo = MakeLayerFbo(sourceArray, 0, kSourceLayer);
glBindFramebuffer(GL_FRAMEBUFFER, sourceFbo);
glViewport(0, 0, kWidth, kHeight);
glDisable(GL_SCISSOR_TEST);
glDisable(GL_DEPTH_TEST);
glDrawBuffer(GL_COLOR_ATTACHMENT0);
DrawFullViewportQuad(m_program);
const GLuint destinationFbo = MakeLayerFbo(destinationArray, 0, kDestinationLayer);
glBindFramebuffer(GL_READ_FRAMEBUFFER, sourceFbo);
glReadBuffer(GL_COLOR_ATTACHMENT0);
glBindFramebuffer(GL_DRAW_FRAMEBUFFER, destinationFbo);
glDrawBuffer(GL_COLOR_ATTACHMENT0);
glBlitFramebuffer(0, 0, kWidth, kHeight, 0, 0, kWidth, kHeight, GL_COLOR_BUFFER_BIT, GL_NEAREST);
glBindFramebuffer(GL_FRAMEBUFFER, 0);
EXPECT_EQ(FirstGLError(), 0u);
const Rgba8 blitted = ReadLayer(destinationArray, kDestinationLayer);
EXPECT_NEAR(blitted.r, kPaintedColor.r, 2) << "blit destination layer is " << Describe(blitted);
EXPECT_NEAR(blitted.g, kPaintedColor.g, 2);
EXPECT_NEAR(blitted.b, kPaintedColor.b, 2);
ExpectOtherLayersUntouched(destinationArray, kDestinationLayer, "colour blit destination");
// The source layer was rendered, not blitted into, so it is checked separately.
const Rgba8 source = ReadLayer(sourceArray, kSourceLayer);
EXPECT_NEAR(source.r, kPaintedColor.r, 2) << "blit source layer is " << Describe(source);
ExpectOtherLayersUntouched(sourceArray, kSourceLayer, "colour blit source");
}
// The depth aspect of the same readback path: the depth image sits in
// DEPTH_STENCIL_ATTACHMENT_OPTIMAL after the clear, and the copy names the attached layer.
TEST_F(LayeredAttachmentBarrierScenario, ReadDepthOffClearedNonZeroLayer) {
if (!Ready()) return;
const GLuint colorArray = MakeColorArray();
const GLuint depthArray = MakeDepthArray();
ASSERT_EQ(FirstGLError(), 0u) << "texture setup failed";
const GLuint fbo = MakeLayerFbo(colorArray, depthArray, kSubjectLayer);
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
glViewport(0, 0, kWidth, kHeight);
glDisable(GL_SCISSOR_TEST);
glDepthMask(GL_TRUE);
glClearDepth(0.375);
glClear(GL_DEPTH_BUFFER_BIT);
const float centre = ReadDepthAt(kWidth / 2, kHeight / 2);
glBindFramebuffer(GL_FRAMEBUFFER, 0);
EXPECT_EQ(FirstGLError(), 0u);
EXPECT_NEAR(centre, 0.375f, 1.0f / 4096.0f)
<< "glReadPixels(GL_DEPTH_COMPONENT) off layer " << kSubjectLayer << " returned " << centre
<< (std::fabs(centre - kDepthPoison) < 1e-6f ? " - the destination was never written at all" : "");
}
// The depth leg of the blit path, both endpoints above layer 0. Verified by reading the
// destination's depth back, which is the same readback the case above pins - so a failure
// here with that one passing is the blit, not the readback.
TEST_F(LayeredAttachmentBarrierScenario, BlitDepthBetweenNonZeroLayers) {
if (!Ready()) return;
const GLuint sourceColor = MakeColorArray();
const GLuint sourceDepth = MakeDepthArray();
const GLuint destinationColor = MakeColorArray();
const GLuint destinationDepth = MakeDepthArray();
ASSERT_EQ(FirstGLError(), 0u) << "texture setup failed";
constexpr int kSourceLayer = 3;
constexpr int kDestinationLayer = 1;
const GLuint sourceFbo = MakeLayerFbo(sourceColor, sourceDepth, kSourceLayer);
glBindFramebuffer(GL_FRAMEBUFFER, sourceFbo);
glViewport(0, 0, kWidth, kHeight);
glDisable(GL_SCISSOR_TEST);
glDepthMask(GL_TRUE);
glClearDepth(0.625);
glClear(GL_DEPTH_BUFFER_BIT);
// A destination pre-cleared to something the blit must overwrite, so "the blit did
// nothing" and "the blit landed" are different answers.
const GLuint destinationFbo = MakeLayerFbo(destinationColor, destinationDepth, kDestinationLayer);
glBindFramebuffer(GL_FRAMEBUFFER, destinationFbo);
glViewport(0, 0, kWidth, kHeight);
glDepthMask(GL_TRUE);
glClearDepth(0.125);
glClear(GL_DEPTH_BUFFER_BIT);
glBindFramebuffer(GL_READ_FRAMEBUFFER, sourceFbo);
glBindFramebuffer(GL_DRAW_FRAMEBUFFER, destinationFbo);
glBlitFramebuffer(0, 0, kWidth, kHeight, 0, 0, kWidth, kHeight, GL_DEPTH_BUFFER_BIT, GL_NEAREST);
EXPECT_EQ(FirstGLError(), 0u);
glBindFramebuffer(GL_FRAMEBUFFER, destinationFbo);
const float blitted = ReadDepthAt(kWidth / 2, kHeight / 2);
glBindFramebuffer(GL_FRAMEBUFFER, 0);
EXPECT_EQ(FirstGLError(), 0u);
EXPECT_NEAR(blitted, 0.625f, 1.0f / 4096.0f)
<< "depth blitted onto layer " << kDestinationLayer << " reads back as " << blitted
<< (std::fabs(blitted - 0.125f) < 1e-3f ? " - the destination kept its own clear" : "");
}
} // namespace
} // namespace MGITest
@@ -0,0 +1,878 @@
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/Program203FirstReductionScenario.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Scenario - PROGRAM 203'S FIRST SUBGROUP REDUCTION.
//
// Program 203 reduces a 32 x 16 exposure tile with a vector subgroup inclusive add,
// then a shared-memory scan of subgroup totals. The source assumes that every
// subgroup has a last lane, that there are 2..32 subgroups, and that local index
// 511 belongs to the last subgroup and its last lane. Those are source assumptions,
// not API contracts. This probe intentionally does not repair them: it records the
// observed topology and makes each handoff independently observable.
#include <algorithm>
#include <array>
#include <bit>
#include <cstddef>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <iomanip>
#include <iostream>
#include <limits>
#include <sstream>
#include <string>
#include <type_traits>
#include <utility>
#include <vector>
#include "../Harness/HeadlessGL.h"
#include "../Harness/ScenarioFixture.h"
#ifdef GLAPI
#undef GLAPI
#endif
#define GL_GLEXT_PROTOTYPES
#include <GL/gl.h>
#include <GL/glcorearb.h>
#undef GL_GLEXT_PROTOTYPES
namespace MGITest {
namespace {
constexpr std::size_t kInvocationCount = 512;
constexpr std::size_t kScanStageCount = 6;
constexpr std::uint32_t kQuietNanBits = 0x7fc00000u;
constexpr std::size_t kNoSlot = std::numeric_limits<std::size_t>::max();
struct UVec4 {
std::uint32_t x;
std::uint32_t y;
std::uint32_t z;
std::uint32_t w;
};
struct Vec4 {
float x;
float y;
float z;
float w;
};
// Matches the std430 block exactly. uvec4/vec4 arrays have a 16-byte
// stride, floats are a dense scalar array, and the outer scan array is
// stage-major in both GLSL and C++.
struct ProbeOutput {
std::array<UVec4, kInvocationCount> invocation;
std::array<UVec4, kInvocationCount> subgroup;
std::array<Vec4, kInvocationCount> reduction;
std::array<float, kInvocationCount> finalAverage;
std::array<std::array<float, kInvocationCount>, kScanStageCount> scanAfter;
};
static_assert(sizeof(UVec4) == 16);
static_assert(sizeof(Vec4) == 16);
static_assert(std::is_standard_layout_v<ProbeOutput>);
static_assert(offsetof(ProbeOutput, invocation) == 0);
static_assert(offsetof(ProbeOutput, subgroup) == 8192);
static_assert(offsetof(ProbeOutput, reduction) == 16384);
static_assert(offsetof(ProbeOutput, finalAverage) == 24576);
static_assert(offsetof(ProbeOutput, scanAfter) == 26624);
static_assert(sizeof(ProbeOutput) == 38912);
enum class InputMode {
SampledRgba32f,
IndexedSsbo,
};
const char* InputModeName(InputMode mode) {
return mode == InputMode::SampledRgba32f ? "sampled RGBA32F" : "indexed SSBO";
}
std::uint32_t FloatBits(float value) {
return std::bit_cast<std::uint32_t>(value);
}
bool SameBits(float lhs, float rhs) {
return FloatBits(lhs) == FloatBits(rhs);
}
bool IsQuietNanSentinel(float value) {
return FloatBits(value) == kQuietNanBits;
}
bool DrainGlErrors() {
bool hadError = false;
while (glGetError() != GL_NO_ERROR) hadError = true;
return hadError;
}
bool HasExtension(const char* wanted) {
GLint extensionCount = 0;
glGetIntegerv(GL_NUM_EXTENSIONS, &extensionCount);
for (GLint i = 0; i < extensionCount; ++i) {
const auto* extension = reinterpret_cast<const char*>(glGetStringi(GL_EXTENSIONS, static_cast<GLuint>(i)));
if (extension != nullptr && std::string(extension) == wanted) return true;
}
return false;
}
struct CapabilityInfo {
bool subgroupExtension = false;
GLint subgroupSize = 0;
GLint supportedStages = 0;
GLint supportedFeatures = 0;
GLint maxComputeStorageBlocks = 0;
GLint maxStorageBindings = 0;
GLint maxWorkGroupInvocations = 0;
std::array<GLint, 3> maxWorkGroupSize{};
bool queryHadError = false;
bool SupportsProbe() const {
const auto stages = static_cast<GLbitfield>(supportedStages);
const auto features = static_cast<GLbitfield>(supportedFeatures);
return !queryHadError && subgroupExtension &&
(stages & GL_COMPUTE_SHADER_BIT) != 0 &&
(features & (GL_SUBGROUP_FEATURE_BASIC_BIT_KHR | GL_SUBGROUP_FEATURE_ARITHMETIC_BIT_KHR)) ==
(GL_SUBGROUP_FEATURE_BASIC_BIT_KHR | GL_SUBGROUP_FEATURE_ARITHMETIC_BIT_KHR) &&
maxComputeStorageBlocks >= 2 && maxStorageBindings >= 2 &&
maxWorkGroupInvocations >= static_cast<GLint>(kInvocationCount) && maxWorkGroupSize[0] >= 32 &&
maxWorkGroupSize[1] >= 16 && maxWorkGroupSize[2] >= 1;
}
std::string MissingRequirements() const {
std::vector<std::string> missing;
const auto stages = static_cast<GLbitfield>(supportedStages);
const auto features = static_cast<GLbitfield>(supportedFeatures);
if (queryHadError) missing.emplace_back("a subgroup/compute capability query generated GL error");
if (!subgroupExtension) missing.emplace_back("GL_KHR_shader_subgroup");
if ((stages & GL_COMPUTE_SHADER_BIT) == 0) {
missing.emplace_back("GL_COMPUTE_SHADER_BIT in GL_SUBGROUP_SUPPORTED_STAGES_KHR");
}
const auto requiredFeatures =
GL_SUBGROUP_FEATURE_BASIC_BIT_KHR | GL_SUBGROUP_FEATURE_ARITHMETIC_BIT_KHR;
if ((features & requiredFeatures) != requiredFeatures) {
missing.emplace_back("basic|arithmetic in GL_SUBGROUP_SUPPORTED_FEATURES_KHR");
}
if (maxComputeStorageBlocks < 2 || maxStorageBindings < 2) {
missing.emplace_back("two compute SSBO bindings");
}
if (maxWorkGroupInvocations < static_cast<GLint>(kInvocationCount) || maxWorkGroupSize[0] < 32 ||
maxWorkGroupSize[1] < 16 || maxWorkGroupSize[2] < 1) {
missing.emplace_back("a 32x16x1 / 512-invocation compute workgroup");
}
std::ostringstream message;
for (std::size_t i = 0; i < missing.size(); ++i) {
if (i != 0) message << ", ";
message << missing[i];
}
return message.str();
}
};
CapabilityInfo QueryCapabilities() {
CapabilityInfo info;
DrainGlErrors();
info.subgroupExtension = HasExtension("GL_KHR_shader_subgroup");
glGetIntegerv(GL_SUBGROUP_SIZE_KHR, &info.subgroupSize);
glGetIntegerv(GL_SUBGROUP_SUPPORTED_STAGES_KHR, &info.supportedStages);
glGetIntegerv(GL_SUBGROUP_SUPPORTED_FEATURES_KHR, &info.supportedFeatures);
glGetIntegerv(GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS, &info.maxComputeStorageBlocks);
glGetIntegerv(GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS, &info.maxStorageBindings);
glGetIntegerv(GL_MAX_COMPUTE_WORK_GROUP_INVOCATIONS, &info.maxWorkGroupInvocations);
for (GLuint axis = 0; axis < info.maxWorkGroupSize.size(); ++axis) {
glGetIntegeri_v(GL_MAX_COMPUTE_WORK_GROUP_SIZE, axis, &info.maxWorkGroupSize[axis]);
}
info.queryHadError = DrainGlErrors();
return info;
}
void PrintMetadata(const CapabilityInfo& info, std::ostream& output) {
output << "Program203FirstReductionScenario metadata: "
<< "GL_SUBGROUP_SIZE_KHR=" << info.subgroupSize
<< ", GL_SUBGROUP_SUPPORTED_STAGES_KHR=0x" << std::hex
<< static_cast<GLbitfield>(info.supportedStages)
<< ", GL_SUBGROUP_SUPPORTED_FEATURES_KHR=0x"
<< static_cast<GLbitfield>(info.supportedFeatures) << std::dec
<< ", subgroupExtension=" << info.subgroupExtension
<< ", GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS=" << info.maxComputeStorageBlocks
<< ", GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS=" << info.maxStorageBindings
<< ", GL_MAX_COMPUTE_WORK_GROUP_INVOCATIONS=" << info.maxWorkGroupInvocations
<< ", GL_MAX_COMPUTE_WORK_GROUP_SIZE=" << info.maxWorkGroupSize[0] << 'x'
<< info.maxWorkGroupSize[1] << 'x' << info.maxWorkGroupSize[2]
<< ", queryHadError=" << info.queryHadError << '\n';
}
bool DumpRequested() {
const char* value = std::getenv("MOBILEGL_ITEST_SUBGROUP_PROBE_DUMP");
return value != nullptr && std::string(value) == "1";
}
constexpr const char* kShaderPreamble = R"(#version 430 core
#extension GL_KHR_shader_subgroup_basic : require
#extension GL_KHR_shader_subgroup_arithmetic : require
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
layout(std430, binding = 1) buffer SubgroupProbeOutput {
uvec4 invocation[512];
uvec4 subgroup[512];
vec4 reduction[512];
float finalAverage[512];
float scanAfter[6][512];
} outProbe;
shared vec2 prefixSumCache[32];
)";
constexpr const char* kSampledInput = R"(
uniform sampler2D colortex2;
uniform vec2 pixelSize;
)";
constexpr const char* kIndexedInput = R"(
layout(std430, binding = 0) readonly buffer Input {
float value[512];
} inputData;
)";
// Only the expression producing tileExposure differs between the two
// tests. The remainder is the program-203 first reduction, with stores
// placed after its existing barriers to expose each handoff.
constexpr const char* kSampledTileExposure = R"(
vec2 texCoord = (vec2(gl_GlobalInvocationID.xy) + 0.5) *
vec2(1.0 / 32.0, 1.0 / 16.0);
vec2 sampleCoord = texCoord * (1.0 / 64.0);
sampleCoord.x += (15.0 / 32.0) + pixelSize.x * 12.0;
float tileExposure = dot(
textureLod(colortex2, sampleCoord, 0.0).rgb,
vec3(0.2125, 0.7154, 0.0721));
)";
constexpr const char* kIndexedTileExposure = R"(
float tileExposure = inputData.value[gl_LocalInvocationIndex];
)";
constexpr const char* kReductionBody = R"(
vec2 sampleLuminance = vec2(tileExposure, 0.0);
sampleLuminance = subgroupInclusiveAdd(sampleLuminance);
float nativeInclusive = sampleLuminance.x;
// This is a uniform, safety-only branch: it leaves an invalid source
// contract visible without indexing past the 32-entry cache or underflowing
// loopLength - 1. It is deliberately a failure on the CPU, not a skip.
bool sourceDomain = gl_NumSubgroups >= 2u && gl_NumSubgroups <= 32u;
if (!sourceDomain) {
float qNaN = uintBitsToFloat(0x7fc00000u);
uint localIndex = gl_LocalInvocationIndex;
outProbe.invocation[localIndex] = uvec4(localIndex, gl_LocalInvocationID);
outProbe.subgroup[localIndex] = uvec4(gl_SubgroupSize, gl_NumSubgroups, gl_SubgroupID,
gl_SubgroupInvocationID);
outProbe.reduction[localIndex] = vec4(tileExposure, nativeInclusive, qNaN, qNaN);
outProbe.finalAverage[localIndex] = qNaN;
for (uint stage = 0u; stage < 6u; ++stage)
outProbe.scanAfter[stage][localIndex] = qNaN;
return;
}
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = sampleLuminance;
barrier();
float sourceRawSubtotal = prefixSumCache[gl_SubgroupID].x;
uint loopLength = uint(findMSB(gl_NumSubgroups));
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
for (uint scanStage = 0u; scanStage < loopLength; ++scanStage) {
if ((gl_SubgroupID & (1u << scanStage)) > 0u) {
sampleLuminance += prefixSumCache[(gl_SubgroupID >> scanStage << scanStage) - 1u];
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = sampleLuminance;
}
barrier();
outProbe.scanAfter[scanStage][gl_LocalInvocationIndex] = sampleLuminance.x;
}
float sourceMergedPrefix = sampleLuminance.x;
if (gl_LocalInvocationIndex == 511u)
prefixSumCache[0] = sampleLuminance / 512.0;
barrier();
float avg = prefixSumCache[0].x;
uint localIndex = gl_LocalInvocationIndex;
outProbe.invocation[localIndex] = uvec4(localIndex, gl_LocalInvocationID);
outProbe.subgroup[localIndex] = uvec4(gl_SubgroupSize, gl_NumSubgroups, gl_SubgroupID,
gl_SubgroupInvocationID);
outProbe.reduction[localIndex] = vec4(tileExposure, nativeInclusive, sourceRawSubtotal, sourceMergedPrefix);
outProbe.finalAverage[localIndex] = avg;
}
)";
std::string BuildProbeShader(InputMode mode) {
std::string source = kShaderPreamble;
source += mode == InputMode::SampledRgba32f ? kSampledInput : kIndexedInput;
source += "\nvoid main() {\n";
source += mode == InputMode::SampledRgba32f ? kSampledTileExposure : kIndexedTileExposure;
source += kReductionBody;
return source;
}
std::string FormatFloat(float value) {
std::ostringstream text;
text << std::hexfloat << value;
return text.str();
}
struct ValidationResult {
bool ok = true;
std::string phase;
std::string message;
bool scanStageMismatch = false;
int scanStage = -1;
bool ownerEvaluated = false;
bool index511IsSourceLastLaneWriter = false;
bool index511IsHighestSubgroupMember = false;
std::uint32_t highestObservedSubgroup = 0;
};
ValidationResult Failure(std::string phase, std::string message) {
ValidationResult result;
result.ok = false;
result.phase = std::move(phase);
result.message = std::move(message);
return result;
}
constexpr float kSampledLuminance = 0.2125f + 0.7154f + 0.0721f;
float ExpectedInput(InputMode mode, std::uint32_t localIndex) {
return mode == InputMode::SampledRgba32f ? kSampledLuminance : static_cast<float>(localIndex + 1u);
}
ValidationResult ValidateProbe(const ProbeOutput& output, InputMode mode) {
std::array<std::size_t, kInvocationCount> slotForLocal{};
slotForLocal.fill(kNoSlot);
// 1. Record identity. Slots are only used to locate each reported
// local index; all subgroup behavior below groups recorded IDs/lanes.
for (std::size_t slot = 0; slot < kInvocationCount; ++slot) {
const std::uint32_t localIndex = output.invocation[slot].x;
if (localIndex >= kInvocationCount) {
std::ostringstream message;
message << "output slot " << slot << " reports localIndex " << localIndex << " outside [0, 511]";
return Failure("record identity", message.str());
}
if (slotForLocal[localIndex] != kNoSlot) {
std::ostringstream message;
message << "localIndex " << localIndex << " appears in output slots " << slotForLocal[localIndex]
<< " and " << slot;
return Failure("record identity", message.str());
}
slotForLocal[localIndex] = slot;
}
for (std::size_t localIndex = 0; localIndex < kInvocationCount; ++localIndex) {
if (slotForLocal[localIndex] == kNoSlot) {
std::ostringstream message;
message << "localIndex " << localIndex << " is missing from all 512 records";
return Failure("record identity", message.str());
}
}
for (std::size_t localIndex = 0; localIndex < kInvocationCount; ++localIndex) {
const std::size_t slot = slotForLocal[localIndex];
const UVec4& invocation = output.invocation[slot];
const std::uint32_t expectedX = static_cast<std::uint32_t>(localIndex % 32u);
const std::uint32_t expectedY = static_cast<std::uint32_t>(localIndex / 32u);
if (invocation.y != expectedX || invocation.z != expectedY || invocation.w != 0u) {
std::ostringstream message;
message << "localIndex " << localIndex << " reports local invocation (" << invocation.y << ','
<< invocation.z << ',' << invocation.w << "), expected (" << expectedX << ',' << expectedY
<< ",0)";
return Failure("record identity", message.str());
}
const float expectedInput = ExpectedInput(mode, static_cast<std::uint32_t>(localIndex));
const float actualInput = output.reduction[slot].x;
if (!SameBits(actualInput, expectedInput)) {
std::ostringstream message;
message << "localIndex " << localIndex << " input was " << FormatFloat(actualInput) << ", expected "
<< FormatFloat(expectedInput);
return Failure("input", message.str());
}
}
// 2. Observed topology. Do not derive lanes or subgroup membership
// from local invocation indices: only the values the shader recorded
// participate in grouping.
const std::uint32_t reportedNumSubgroups = output.subgroup[slotForLocal[0]].y;
if (reportedNumSubgroups == 0u) {
return Failure("observed topology", "localIndex 0 reported gl_NumSubgroups == 0");
}
if (reportedNumSubgroups > kInvocationCount) {
std::ostringstream message;
message << "reported gl_NumSubgroups=" << reportedNumSubgroups
<< " exceeds the 512 recorded invocations, so at least one subgroup ID is missing";
return Failure("observed topology", message.str());
}
std::vector<std::vector<std::size_t>> subgroupSlots(reportedNumSubgroups);
for (std::size_t localIndex = 0; localIndex < kInvocationCount; ++localIndex) {
const std::size_t slot = slotForLocal[localIndex];
const UVec4& subgroup = output.subgroup[slot];
if (subgroup.x == 0u || subgroup.y == 0u) {
std::ostringstream message;
message << "localIndex " << localIndex << " reported subgroupSize=" << subgroup.x
<< ", numSubgroups=" << subgroup.y;
return Failure("observed topology", message.str());
}
if (subgroup.y != reportedNumSubgroups) {
std::ostringstream message;
message << "localIndex " << localIndex << " reported numSubgroups=" << subgroup.y
<< ", while localIndex 0 reported " << reportedNumSubgroups;
return Failure("observed topology", message.str());
}
if (subgroup.z >= reportedNumSubgroups) {
std::ostringstream message;
message << "localIndex " << localIndex << " reported subgroupID=" << subgroup.z
<< " outside [0, " << (reportedNumSubgroups - 1u) << ']';
return Failure("observed topology", message.str());
}
if (subgroup.w >= subgroup.x) {
std::ostringstream message;
message << "localIndex " << localIndex << " reported laneID=" << subgroup.w
<< " outside its subgroupSize=" << subgroup.x;
return Failure("observed topology", message.str());
}
subgroupSlots[subgroup.z].push_back(slot);
}
for (std::uint32_t subgroupID = 0; subgroupID < reportedNumSubgroups; ++subgroupID) {
if (subgroupSlots[subgroupID].empty()) {
std::ostringstream message;
message << "reported gl_NumSubgroups=" << reportedNumSubgroups
<< " but subgroupID " << subgroupID << " has no recorded members";
return Failure("observed topology", message.str());
}
auto& members = subgroupSlots[subgroupID];
std::sort(members.begin(), members.end(), [&output](std::size_t lhs, std::size_t rhs) {
return output.subgroup[lhs].w < output.subgroup[rhs].w;
});
for (std::size_t i = 1; i < members.size(); ++i) {
if (output.subgroup[members[i - 1]].w == output.subgroup[members[i]].w) {
std::ostringstream message;
message << "subgroupID " << subgroupID << " contains duplicate laneID "
<< output.subgroup[members[i]].w;
return Failure("observed topology", message.str());
}
}
}
// 3. Native subgroup arithmetic, in the actual lane ordering emitted
// by the driver. The fixture values and all partial sums are exactly
// representable binary32 values, so compare representation, not epsilon.
std::array<float, kInvocationCount> nativePrefix{};
std::vector<float> nativeSubtotal(reportedNumSubgroups, 0.0f);
for (std::uint32_t subgroupID = 0; subgroupID < reportedNumSubgroups; ++subgroupID) {
float inclusive = 0.0f;
for (const std::size_t slot : subgroupSlots[subgroupID]) {
const std::uint32_t localIndex = output.invocation[slot].x;
inclusive += ExpectedInput(mode, localIndex);
nativePrefix[slot] = inclusive;
const float actualNative = output.reduction[slot].y;
if (!SameBits(actualNative, inclusive)) {
std::ostringstream message;
message << "subgroupID " << subgroupID << ", laneID " << output.subgroup[slot].w
<< ", localIndex " << localIndex << " nativeInclusive was " << FormatFloat(actualNative)
<< ", expected " << FormatFloat(inclusive);
return Failure("native subgroup arithmetic", message.str());
}
}
nativeSubtotal[subgroupID] = inclusive;
}
// sourceDomain is the narrow source-side safety branch. It is checked
// after native arithmetic so an unsupported source topology still
// reports native subgroup behavior before failing explicitly.
if (reportedNumSubgroups < 2u || reportedNumSubgroups > 32u) {
for (std::size_t localIndex = 0; localIndex < kInvocationCount; ++localIndex) {
const std::size_t slot = slotForLocal[localIndex];
const Vec4& reduction = output.reduction[slot];
if (!IsQuietNanSentinel(reduction.z) || !IsQuietNanSentinel(reduction.w) ||
!IsQuietNanSentinel(output.finalAverage[slot])) {
std::ostringstream message;
message << "program 203 source reduction has no valid contract for gl_NumSubgroups="
<< reportedNumSubgroups << "; localIndex " << localIndex
<< " did not preserve its qNaN source-reduction sentinel";
return Failure("source domain", message.str());
}
for (std::size_t stage = 0; stage < kScanStageCount; ++stage) {
if (!IsQuietNanSentinel(output.scanAfter[stage][slot])) {
std::ostringstream message;
message << "program 203 source reduction has no valid contract for gl_NumSubgroups="
<< reportedNumSubgroups << "; localIndex " << localIndex << ", scan stage " << stage
<< " did not preserve its qNaN source-reduction sentinel";
return Failure("source domain", message.str());
}
}
}
std::ostringstream message;
message << "program 203 source reduction has no valid contract for observed gl_NumSubgroups="
<< reportedNumSubgroups << " (requires 2..32); native subgroup results were recorded";
return Failure("source domain", message.str());
}
// 4. Program-203 source writer and first shared-memory handoff.
std::vector<std::size_t> sourceWriter(reportedNumSubgroups, kNoSlot);
for (std::uint32_t subgroupID = 0; subgroupID < reportedNumSubgroups; ++subgroupID) {
std::size_t writerCount = 0;
for (const std::size_t slot : subgroupSlots[subgroupID]) {
const UVec4& subgroup = output.subgroup[slot];
if (subgroup.w == subgroup.x - 1u) {
sourceWriter[subgroupID] = slot;
++writerCount;
}
}
if (writerCount != 1u) {
std::ostringstream message;
message << "subgroupID " << subgroupID << " has " << writerCount
<< " recorded lane(s) where laneID == subgroupSize - 1; program 203 leaves that "
"shared-cache entry unwritten";
return Failure("source writer", message.str());
}
for (const std::size_t slot : subgroupSlots[subgroupID]) {
const float actualRawSubtotal = output.reduction[slot].z;
if (!SameBits(actualRawSubtotal, nativeSubtotal[subgroupID])) {
std::ostringstream message;
message << "subgroupID " << subgroupID << ", localIndex " << output.invocation[slot].x
<< " sourceRawSubtotal was " << FormatFloat(actualRawSubtotal) << ", expected "
<< FormatFloat(nativeSubtotal[subgroupID]);
return Failure("source raw subtotal", message.str());
}
}
}
// 5. Reproduce the source loop exactly, including the redundant final
// scan iteration on power-of-two subgroup counts. Reads and writes in
// one iteration target disjoint cache entries, so update the cache at
// the CPU equivalent of the source barrier.
std::array<float, kInvocationCount> mergedPrefix = nativePrefix;
std::vector<float> cache = nativeSubtotal;
std::uint32_t loopLength = std::bit_width(reportedNumSubgroups) - 1u;
loopLength +=
static_cast<std::uint32_t>(reportedNumSubgroups - (1u << (loopLength - 1u)) > 0u);
for (std::uint32_t scanStage = 0u; scanStage < loopLength; ++scanStage) {
std::vector<float> cacheAfterStage = cache;
for (std::uint32_t subgroupID = 0; subgroupID < reportedNumSubgroups; ++subgroupID) {
if ((subgroupID & (1u << scanStage)) == 0u) continue;
const std::uint32_t sourceCacheIndex = (subgroupID >> scanStage << scanStage) - 1u;
const float sourcePrefix = cache[sourceCacheIndex];
for (const std::size_t slot : subgroupSlots[subgroupID]) {
mergedPrefix[slot] += sourcePrefix;
}
cacheAfterStage[subgroupID] = mergedPrefix[sourceWriter[subgroupID]];
}
cache.swap(cacheAfterStage);
for (std::size_t localIndex = 0; localIndex < kInvocationCount; ++localIndex) {
const std::size_t slot = slotForLocal[localIndex];
const float actualAfterStage = output.scanAfter[scanStage][slot];
if (!SameBits(actualAfterStage, mergedPrefix[slot])) {
std::ostringstream message;
message << "scanStage " << scanStage << ", subgroupID " << output.subgroup[slot].z
<< ", laneID " << output.subgroup[slot].w << ", localIndex " << localIndex
<< " scanAfter was " << FormatFloat(actualAfterStage) << ", expected "
<< FormatFloat(mergedPrefix[slot]);
ValidationResult result = Failure("source scan", message.str());
result.scanStageMismatch = true;
result.scanStage = static_cast<int>(scanStage);
return result;
}
}
}
for (std::size_t localIndex = 0; localIndex < kInvocationCount; ++localIndex) {
const std::size_t slot = slotForLocal[localIndex];
const float actualMergedPrefix = output.reduction[slot].w;
if (!SameBits(actualMergedPrefix, mergedPrefix[slot])) {
std::ostringstream message;
message << "localIndex " << localIndex << " sourceMergedPrefix was "
<< FormatFloat(actualMergedPrefix) << ", expected " << FormatFloat(mergedPrefix[slot]);
return Failure("source scan", message.str());
}
}
// 6. Final owner and average. The uniformity check is intentionally
// separate from the source's topology contract at local index 511.
const float firstAverage = output.finalAverage[slotForLocal[0]];
for (std::size_t localIndex = 1; localIndex < kInvocationCount; ++localIndex) {
const float actualAverage = output.finalAverage[slotForLocal[localIndex]];
if (!SameBits(actualAverage, firstAverage)) {
std::ostringstream message;
message << "finalAverage differs: localIndex 0 has " << FormatFloat(firstAverage)
<< ", localIndex " << localIndex << " has " << FormatFloat(actualAverage);
return Failure("final average", message.str());
}
}
ValidationResult ownerResult;
ownerResult.ownerEvaluated = true;
for (std::uint32_t subgroupID = 0; subgroupID < reportedNumSubgroups; ++subgroupID) {
if (!subgroupSlots[subgroupID].empty()) {
ownerResult.highestObservedSubgroup = std::max(ownerResult.highestObservedSubgroup, subgroupID);
}
}
const std::size_t index511Slot = slotForLocal[kInvocationCount - 1u];
const UVec4& index511Subgroup = output.subgroup[index511Slot];
ownerResult.index511IsSourceLastLaneWriter =
index511Subgroup.w == index511Subgroup.x - 1u;
ownerResult.index511IsHighestSubgroupMember =
index511Subgroup.z == ownerResult.highestObservedSubgroup;
if (!ownerResult.index511IsSourceLastLaneWriter || !ownerResult.index511IsHighestSubgroupMember) {
std::ostringstream message;
message << "program 203 topology incompatibility: localIndex 511 is sourceLastLaneWriter="
<< ownerResult.index511IsSourceLastLaneWriter << ", highestSubgroupMember="
<< ownerResult.index511IsHighestSubgroupMember << " (subgroupID=" << index511Subgroup.z
<< ", highest observed subgroupID=" << ownerResult.highestObservedSubgroup << ')';
ownerResult.ok = false;
ownerResult.phase = "final average";
ownerResult.message = message.str();
return ownerResult;
}
float total = 0.0f;
for (const float subtotal : nativeSubtotal) total += subtotal;
float sampledExpectedTotal = 0.0f;
for (std::size_t i = 0; i < kInvocationCount; ++i) sampledExpectedTotal += kSampledLuminance;
const float expectedTotal = mode == InputMode::IndexedSsbo ? 131328.0f : sampledExpectedTotal;
if (!SameBits(total, expectedTotal) || !SameBits(mergedPrefix[index511Slot], expectedTotal)) {
std::ostringstream message;
message << "program 203 source total was " << FormatFloat(mergedPrefix[index511Slot])
<< " (native total " << FormatFloat(total) << "), expected " << FormatFloat(expectedTotal);
ownerResult.ok = false;
ownerResult.phase = "final average";
ownerResult.message = message.str();
return ownerResult;
}
const float expectedAverage = mode == InputMode::IndexedSsbo ? 256.5f : sampledExpectedTotal / 512.0f;
if (!SameBits(firstAverage, expectedAverage)) {
std::ostringstream message;
message << "finalAverage was " << FormatFloat(firstAverage) << ", expected "
<< FormatFloat(expectedAverage);
ownerResult.ok = false;
ownerResult.phase = "final average";
ownerResult.message = message.str();
return ownerResult;
}
return ownerResult;
}
void DumpProbe(const ProbeOutput& output, const CapabilityInfo& capabilities, const ValidationResult& validation,
bool includeScanStages) {
PrintMetadata(capabilities, std::cout);
if (validation.ok) {
std::cout << "Program203FirstReductionScenario firstFailure=none\n";
} else {
std::cout << "Program203FirstReductionScenario firstFailure=" << validation.phase << ": "
<< validation.message << '\n';
}
std::cout << "localIndex,localX,localY,localZ,subgroupSize,numSubgroups,subgroupID,laneID,input,"
"nativeInclusive,subgroupSubtotal,mergedPrefix,finalAverage\n";
for (std::size_t slot = 0; slot < kInvocationCount; ++slot) {
const UVec4& invocation = output.invocation[slot];
const UVec4& subgroup = output.subgroup[slot];
const Vec4& reduction = output.reduction[slot];
std::cout << invocation.x << ',' << invocation.y << ',' << invocation.z << ',' << invocation.w << ','
<< subgroup.x << ',' << subgroup.y << ',' << subgroup.z << ',' << subgroup.w << ','
<< std::hexfloat << reduction.x << ',' << reduction.y << ',' << reduction.z << ','
<< reduction.w << ',' << output.finalAverage[slot] << std::defaultfloat << '\n';
}
if (includeScanStages) {
std::cout << "scanStage,localIndex,scanAfter\n";
for (std::size_t scanStage = 0; scanStage < kScanStageCount; ++scanStage) {
for (std::size_t slot = 0; slot < kInvocationCount; ++slot) {
std::cout << scanStage << ',' << output.invocation[slot].x << ',' << std::hexfloat
<< output.scanAfter[scanStage][slot] << std::defaultfloat << '\n';
}
}
}
}
class Program203FirstReductionScenario : public ScenarioTest {
protected:
void SetUp() override {
ScenarioTest::SetUp();
if (!Ready()) return;
m_capabilities = QueryCapabilities();
// GL_SUBGROUP_SIZE_KHR is diagnostic only. It is deliberately
// never used to infer lane placement or an expected group count.
PrintMetadata(m_capabilities, std::cout);
RecordProperty("program203_gl_subgroup_size_khr", std::to_string(m_capabilities.subgroupSize));
if (!m_capabilities.SupportsProbe()) {
GTEST_SKIP() << "subgroup probe requires " << m_capabilities.MissingRequirements();
}
}
void TearDown() override {
if (!Ready()) return;
glUseProgram(0);
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 0, 0);
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 1, 0);
glBindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
glActiveTexture(GL_TEXTURE3);
glBindTexture(GL_TEXTURE_2D, 0);
glActiveTexture(GL_TEXTURE0);
if (m_texture != 0) glDeleteTextures(1, &m_texture);
if (m_inputBuffer != 0) glDeleteBuffers(1, &m_inputBuffer);
if (m_outputBuffer != 0) glDeleteBuffers(1, &m_outputBuffer);
if (m_program != 0) glDeleteProgram(m_program);
m_texture = 0;
m_inputBuffer = 0;
m_outputBuffer = 0;
m_program = 0;
}
GLuint CompileComputeProgram(const std::string& source, std::string* outError) {
const char* text = source.c_str();
const GLuint shader = glCreateShader(GL_COMPUTE_SHADER);
if (shader == 0) {
*outError = "glCreateShader(GL_COMPUTE_SHADER) returned 0";
return 0;
}
glShaderSource(shader, 1, &text, nullptr);
glCompileShader(shader);
GLint compiled = GL_FALSE;
glGetShaderiv(shader, GL_COMPILE_STATUS, &compiled);
if (compiled == GL_FALSE) {
char log[8192] = {};
glGetShaderInfoLog(shader, sizeof(log) - 1, nullptr, log);
*outError = std::string("the subgroup probe compute shader did not compile: ") + log;
glDeleteShader(shader);
return 0;
}
const GLuint program = glCreateProgram();
glAttachShader(program, shader);
glLinkProgram(program);
glDeleteShader(shader);
GLint linked = GL_FALSE;
glGetProgramiv(program, GL_LINK_STATUS, &linked);
if (linked == GL_FALSE) {
char log[8192] = {};
glGetProgramInfoLog(program, sizeof(log) - 1, nullptr, log);
*outError = std::string("the subgroup probe compute program did not link: ") + log;
glDeleteProgram(program);
return 0;
}
return program;
}
bool RunProbe(InputMode mode, ProbeOutput* output, std::string* outError) {
m_program = CompileComputeProgram(BuildProbeShader(mode), outError);
if (m_program == 0) return false;
ProbeOutput poison{};
std::memset(&poison, 0xa5, sizeof(poison));
glGenBuffers(1, &m_outputBuffer);
glBindBuffer(GL_SHADER_STORAGE_BUFFER, m_outputBuffer);
glBufferData(GL_SHADER_STORAGE_BUFFER, sizeof(ProbeOutput), &poison, GL_DYNAMIC_COPY);
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 1, m_outputBuffer);
if (mode == InputMode::IndexedSsbo) {
std::array<float, kInvocationCount> values{};
for (std::size_t i = 0; i < values.size(); ++i) values[i] = static_cast<float>(i + 1u);
glGenBuffers(1, &m_inputBuffer);
glBindBuffer(GL_SHADER_STORAGE_BUFFER, m_inputBuffer);
glBufferData(GL_SHADER_STORAGE_BUFFER, sizeof(values), values.data(), GL_STATIC_DRAW);
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 0, m_inputBuffer);
} else {
constexpr std::array<float, 4> kOneTexel = {1.0f, 1.0f, 1.0f, 1.0f};
glGenTextures(1, &m_texture);
glActiveTexture(GL_TEXTURE3);
glBindTexture(GL_TEXTURE_2D, m_texture);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_EDGE);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE);
glTexImage2D(GL_TEXTURE_2D, 0, GL_RGBA32F, 1, 1, 0, GL_RGBA, GL_FLOAT, kOneTexel.data());
}
if (const GLenum error = FirstGLError(); error != GL_NO_ERROR) {
std::ostringstream message;
message << "subgroup probe resource setup left " << GLErrorName(error);
*outError = message.str();
return false;
}
glUseProgram(m_program);
if (mode == InputMode::SampledRgba32f) {
const GLint sampler = glGetUniformLocation(m_program, "colortex2");
const GLint pixelSize = glGetUniformLocation(m_program, "pixelSize");
if (sampler == -1 || pixelSize == -1) {
*outError = "the sampled probe uniforms were optimized away or not reflected";
return false;
}
glUniform1i(sampler, 3);
glUniform2f(pixelSize, 1.0f / 854.0f, 1.0f / 480.0f);
}
glDispatchCompute(1, 1, 1);
glMemoryBarrier(GL_ALL_BARRIER_BITS);
glBindBuffer(GL_SHADER_STORAGE_BUFFER, m_outputBuffer);
glGetBufferSubData(GL_SHADER_STORAGE_BUFFER, 0, sizeof(ProbeOutput), output);
if (const GLenum error = FirstGLError(); error != GL_NO_ERROR) {
std::ostringstream message;
message << "subgroup probe dispatch/readback left " << GLErrorName(error);
*outError = message.str();
return false;
}
return true;
}
void RunAndValidate(InputMode mode) {
ProbeOutput output{};
std::string error;
ASSERT_TRUE(RunProbe(mode, &output, &error)) << InputModeName(mode) << ": " << error;
const ValidationResult validation = ValidateProbe(output, mode);
if (validation.ownerEvaluated) {
RecordProperty("program203_index511_source_last_lane_writer",
validation.index511IsSourceLastLaneWriter ? "true" : "false");
RecordProperty("program203_index511_highest_subgroup_member",
validation.index511IsHighestSubgroupMember ? "true" : "false");
RecordProperty("program203_highest_observed_subgroup",
std::to_string(validation.highestObservedSubgroup));
std::cout << "Program203FirstReductionScenario owner: localIndex511 sourceLastLaneWriter="
<< validation.index511IsSourceLastLaneWriter << ", highestSubgroupMember="
<< validation.index511IsHighestSubgroupMember << ", highestObservedSubgroup="
<< validation.highestObservedSubgroup << '\n';
}
if (!validation.ok || DumpRequested()) {
DumpProbe(output, m_capabilities, validation, validation.scanStageMismatch || DumpRequested());
}
EXPECT_TRUE(validation.ok) << validation.phase << ": " << validation.message;
}
CapabilityInfo m_capabilities;
GLuint m_program = 0;
GLuint m_inputBuffer = 0;
GLuint m_outputBuffer = 0;
GLuint m_texture = 0;
};
} // namespace
TEST_F(Program203FirstReductionScenario, SampledRgba32fFirstAverage) {
if (!Ready() || IsSkipped()) return;
RunAndValidate(InputMode::SampledRgba32f);
}
TEST_F(Program203FirstReductionScenario, IndexedInputTopologyAndReduction) {
if (!Ready() || IsSkipped()) return;
RunAndValidate(InputMode::IndexedSsbo);
}
} // namespace MGITest
@@ -0,0 +1,524 @@
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/ViewportArrayScenario.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Scenario - gl_ViewportIndex ACTUALLY ROUTES, AND THE PER-INDEX STATE IT SELECTS IS REAL.
//
// The state half of ARB_viewport_array is asserted in MG_Test/State/RenderStateTest.cpp, which
// is a pure set/get exercise and would pass just as green against a backend that stores all 16
// rectangles and rasterizes only the first. This file is the other half: every case here routes
// primitives to a viewport OTHER than 0 and then looks at where the pixels landed.
//
// Three claims, one per case:
// 1. gl_ViewportIndex selects the viewport RECTANGLE - a 4x4 grid of 32x32 viewports, one
// geometry-shader invocation per cell, and every cell must hold its own index.
// 2. gl_ViewportIndex selects the DEPTH RANGE - 16 one-pixel-wide viewports whose ranges are
// (i/16, 1 - i/16), a quad at each end of clip space, and gl_FragCoord.z read back.
// This is the claim that fails loudest against a single-viewport backend, because the
// geometry is still in the right place while every depth comes back as viewport 0's.
// 3. The per-index SCISSOR TEST ENABLE is honoured. Vulkan has no per-viewport scissor-test
// toggle, so a disabled index has to be given the whole framebuffer as its rectangle; the
// case draws the same primitive into the same index twice, once with the test off and once
// with it on, and requires the two results to differ in the documented direction.
//
// Case 1 runs a second time against the DEFAULT framebuffer. MobileGL Y-flips (and pre-transform
// rotates) the default framebuffer's rectangles and does not touch an FBO's, so a port that
// applies the flip to viewport 0 and forgets the other fifteen renders a correct-looking FBO and
// an upside-down window - the classic multi-viewport bug, and invisible to every FBO-only case.
//
// HONEST LIMIT OF THIS FILE. DirectGLES SKIPS every case: GLES has one viewport, one scissor
// rectangle and no gl_ViewportIndex, so routing to index > 0 is an emulation feature that has
// not been built (the Espryt half of KHR-GL43.viewport_array's rendering group is deliberately
// still red). The skip is explicit rather than silent so a future emulation lands here as a
// failing test and not as a test that was quietly never running. DirectVulkan additionally
// skips when the device lacks the multiViewport feature - Vulkan then forbids a pipeline from
// declaring more than one viewport at all, which is a device limit and not a MobileGL bug;
// lavapipe (every CI lane) and both Mali/Adreno devices support it, so the cases do run where
// it matters.
#include <cmath>
#include <string>
#include <vector>
#include "../Harness/HeadlessGL.h"
#include "../Harness/ScenarioFixture.h"
#ifdef GLAPI
#undef GLAPI
#endif
#define GL_GLEXT_PROTOTYPES
#include <GL/gl.h>
#include <GL/glcorearb.h>
#undef GL_GLEXT_PROTOTYPES
namespace MGITest {
namespace {
constexpr int kViewportCount = 16;
constexpr int kGridSide = 4; // 4x4 grid of viewports
constexpr int kCellSize = 32; // ... each 32x32
constexpr int kSurfaceSide = kGridSide * kCellSize;
constexpr GLint kUnwritten = -1;
// A geometry shader is the only stage GL 4.1 lets write gl_ViewportIndex, and
// `invocations` runs it once per viewport off a single input point - the same shape
// KHR-GL43.viewport_array.draw_to_single_layer_with_multiple_viewports uses.
const char* const kVertexSource = R"(#version 410 core
void main() { gl_Position = vec4(0.0, 0.0, 0.0, 1.0); }
)";
const char* const kGridGeometrySource = R"(#version 410 core
layout(points, invocations = 16) in;
layout(triangle_strip, max_vertices = 4) out;
flat out int gsIndex;
void main() {
gsIndex = gl_InvocationID;
gl_ViewportIndex = gl_InvocationID;
gl_Position = vec4(-1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 1.0, 0.0, 1.0); EmitVertex();
EndPrimitive();
}
)";
// One invocation, viewport chosen by a uniform: lets a case draw the SAME primitive into
// the SAME index twice under two different scissor-enable states.
const char* const kSingleGeometrySource = R"(#version 410 core
layout(points, invocations = 1) in;
layout(triangle_strip, max_vertices = 4) out;
uniform int uViewport;
flat out int gsIndex;
void main() {
gsIndex = uViewport;
gl_ViewportIndex = uViewport;
gl_Position = vec4(-1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 1.0, 0.0, 1.0); EmitVertex();
EndPrimitive();
}
)";
const char* const kIntFragmentSource = R"(#version 410 core
flat in int gsIndex;
layout(location = 0) out int fragColor;
void main() { fragColor = gsIndex; }
)";
// Two quads, one at each end of clip space, so the fragment stage can report the depth
// the viewport's range mapped them to. gl_FragCoord.z IS the post-range window depth, so
// it reads back the per-viewport minDepth/maxDepth directly.
const char* const kDepthGeometrySource = R"(#version 410 core
layout(points, invocations = 16) in;
layout(triangle_strip, max_vertices = 8) out;
void main() {
gl_ViewportIndex = gl_InvocationID;
gl_Position = vec4(-1.0, -1.0, -1.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, -1.0, -1.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 0.0, -1.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 0.0, -1.0, 1.0); EmitVertex();
EndPrimitive();
gl_Position = vec4(-1.0, 0.0, 1.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 0.0, 1.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 1.0, 1.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 1.0, 1.0, 1.0); EmitVertex();
EndPrimitive();
}
)";
const char* const kDepthFragmentSource = R"(#version 410 core
layout(location = 0) out float fragColor;
void main() { fragColor = gl_FragCoord.z; }
)";
class ViewportArrayScenario : public ScenarioTest {
protected:
void SetUp() override {
ScenarioTest::SetUp();
if (!Ready()) return;
if (Gl().BackendName() == "DirectGLES") {
GTEST_SKIP() << "gl_ViewportIndex routing is not emulated on DirectGLES: GLES has one viewport "
"and one scissor rectangle, so every index rasterizes as index 0. The indexed "
"STATE is still asserted (MG_Test RenderStateTest); this is the deferred "
"rendering half of KHR-GL43.viewport_array.";
}
GLint maxViewports = 0;
glGetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
ASSERT_GE(maxViewports, kViewportCount) << "GL 4.3 core requires GL_MAX_VIEWPORTS >= 16";
m_program = BuildProgram(kGridGeometrySource, kIntFragmentSource);
ASSERT_NE(m_program, 0u) << "grid program failed to build: " << m_buildLog;
glGenVertexArrays(1, &m_vao);
glBindVertexArray(m_vao);
ResetViewportArrayState();
ASSERT_EQ(glGetError(), GL_NO_ERROR) << "setup left a GL error behind";
}
void TearDown() override {
if (!Ready() || IsSkipped()) return;
ResetViewportArrayState();
if (m_vao != 0) glDeleteVertexArrays(1, &m_vao);
if (m_program != 0) glDeleteProgram(m_program);
glBindFramebuffer(GL_FRAMEBUFFER, 0);
while (glGetError() != GL_NO_ERROR) {
}
}
// Every case starts from the same slate: this fixture shares its context with every
// other scenario in the process, and a leftover per-index scissor enable is exactly
// the kind of state that would make a later case pass or fail for the wrong reason.
static void ResetViewportArrayState() {
for (int i = 0; i < kViewportCount; ++i) {
glDisablei(GL_SCISSOR_TEST, static_cast<GLuint>(i));
}
glDisable(GL_SCISSOR_TEST);
glViewport(0, 0, kSurfaceSide, kSurfaceSide);
glScissor(0, 0, kSurfaceSide, kSurfaceSide);
glDepthRange(0.0, 1.0);
glDisable(GL_DEPTH_TEST);
}
// The 4x4 grid: viewport y*4+x covers the cell whose lower-left corner is
// (x*cellW, y*cellH), in GL's bottom-left-origin window coordinates. Parameterized on
// the cell size because the default framebuffer this scenario also renders into is
// deliberately non-square (HeadlessGL is 128x96, so a transposing bug cannot hide).
static void SetupGridViewports(int cellW, int cellH) {
std::vector<GLfloat> data(static_cast<size_t>(kViewportCount) * 4);
for (int y = 0; y < kGridSide; ++y) {
for (int x = 0; x < kGridSide; ++x) {
const size_t base = static_cast<size_t>(y * kGridSide + x) * 4;
data[base + 0] = static_cast<GLfloat>(x * cellW);
data[base + 1] = static_cast<GLfloat>(y * cellH);
data[base + 2] = static_cast<GLfloat>(cellW);
data[base + 3] = static_cast<GLfloat>(cellH);
}
}
glViewportArrayv(0, kViewportCount, data.data());
}
GLuint BuildProgram(const char* geometrySource, const char* fragmentSource) {
const GLuint vs = CompileStage(GL_VERTEX_SHADER, kVertexSource);
if (vs == 0) return 0;
const GLuint gs = CompileStage(GL_GEOMETRY_SHADER, geometrySource);
if (gs == 0) {
glDeleteShader(vs);
return 0;
}
const GLuint fs = CompileStage(GL_FRAGMENT_SHADER, fragmentSource);
if (fs == 0) {
glDeleteShader(vs);
glDeleteShader(gs);
return 0;
}
const GLuint program = glCreateProgram();
glAttachShader(program, vs);
glAttachShader(program, gs);
glAttachShader(program, fs);
glLinkProgram(program);
GLint linked = 0;
glGetProgramiv(program, GL_LINK_STATUS, &linked);
glDeleteShader(vs);
glDeleteShader(gs);
glDeleteShader(fs);
if (!linked) {
GLint length = 0;
glGetProgramiv(program, GL_INFO_LOG_LENGTH, &length);
std::vector<char> log(static_cast<size_t>(length > 1 ? length : 1), '\0');
glGetProgramInfoLog(program, static_cast<GLsizei>(log.size()), nullptr, log.data());
m_buildLog = log.data();
glDeleteProgram(program);
return 0;
}
return program;
}
GLuint CompileStage(GLenum stage, const char* source) {
const GLuint shader = glCreateShader(stage);
glShaderSource(shader, 1, &source, nullptr);
glCompileShader(shader);
GLint compiled = 0;
glGetShaderiv(shader, GL_COMPILE_STATUS, &compiled);
if (compiled) return shader;
GLint length = 0;
glGetShaderiv(shader, GL_INFO_LOG_LENGTH, &length);
std::vector<char> log(static_cast<size_t>(length > 1 ? length : 1), '\0');
glGetShaderInfoLog(shader, static_cast<GLsizei>(log.size()), nullptr, log.data());
m_buildLog = log.data();
glDeleteShader(shader);
return 0;
}
// An R32I colour target, pre-filled with kUnwritten so "nothing was drawn here" is
// distinguishable from "index 0 was drawn here".
struct IntTarget {
GLuint fbo = 0;
GLuint texture = 0;
};
// The "nothing drawn here" value is UPLOADED, not cleared: the CTS fills its R32I
// targets the same way (fillTexture), and an upload cannot be confused with a clear
// that a backend defers, reorders or drops - which is exactly the ambiguity a case
// asserting "this cell must be untouched" cannot afford.
static void FillIntTarget(const IntTarget& target, int width, int height) {
const std::vector<GLint> unwritten(static_cast<size_t>(width) * height, kUnwritten);
glBindTexture(GL_TEXTURE_2D, target.texture);
glTexSubImage2D(GL_TEXTURE_2D, 0, 0, 0, width, height, GL_RED_INTEGER, GL_INT, unwritten.data());
}
static IntTarget MakeIntTarget(int width, int height) {
IntTarget target;
glGenTextures(1, &target.texture);
glBindTexture(GL_TEXTURE_2D, target.texture);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
glTexImage2D(GL_TEXTURE_2D, 0, GL_R32I, width, height, 0, GL_RED_INTEGER, GL_INT, nullptr);
glGenFramebuffers(1, &target.fbo);
glBindFramebuffer(GL_FRAMEBUFFER, target.fbo);
glFramebufferTexture2D(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, target.texture, 0);
FillIntTarget(target, width, height);
return target;
}
static void DestroyIntTarget(IntTarget& target) {
glBindFramebuffer(GL_FRAMEBUFFER, 0);
if (target.fbo != 0) glDeleteFramebuffers(1, &target.fbo);
if (target.texture != 0) glDeleteTextures(1, &target.texture);
}
static std::vector<GLint> ReadInts(int width, int height) {
std::vector<GLint> pixels(static_cast<size_t>(width) * height, 0);
glReadPixels(0, 0, width, height, GL_RED_INTEGER, GL_INT, pixels.data());
return pixels;
}
// The centre of grid cell (x, y), in the bottom-left-origin coordinates glReadPixels
// returns. Sampling the centre rather than a corner keeps the assertion about WHICH
// viewport was selected rather than about edge rounding.
static GLint CellCentre(const std::vector<GLint>& pixels, int stride, int x, int y) {
const int px = x * kCellSize + kCellSize / 2;
const int py = y * kCellSize + kCellSize / 2;
return pixels[static_cast<size_t>(py) * stride + px];
}
std::string m_buildLog;
GLuint m_program = 0;
GLuint m_vao = 0;
};
// --- 1. the viewport rectangle -------------------------------------------------------
TEST_F(ViewportArrayScenario, EachViewportIndexRasterizesIntoItsOwnRectangle) {
IntTarget target = MakeIntTarget(kSurfaceSide, kSurfaceSide);
SetupGridViewports(kCellSize, kCellSize);
glUseProgram(m_program);
glBindVertexArray(m_vao);
glDrawArrays(GL_POINTS, 0, 1);
ASSERT_EQ(glGetError(), GL_NO_ERROR);
const std::vector<GLint> pixels = ReadInts(kSurfaceSide, kSurfaceSide);
for (int y = 0; y < kGridSide; ++y) {
for (int x = 0; x < kGridSide; ++x) {
const GLint expected = y * kGridSide + x;
EXPECT_EQ(CellCentre(pixels, kSurfaceSide, x, y), expected)
<< "cell (" << x << ", " << y << ") should hold viewport index " << expected
<< "; a single-viewport backend paints the whole image with 15 (the last invocation)";
}
}
DestroyIntTarget(target);
}
// The same claim against the DEFAULT framebuffer, where MobileGL applies its Y-flip and
// pre-transform rotation. Index 0 alone getting the mapping is the classic bug.
TEST_F(ViewportArrayScenario, TheDefaultFramebufferAppliesTheSameFlipToEveryViewport) {
const int surfaceW = Gl().Width();
const int surfaceH = Gl().Height();
ASSERT_GE(surfaceW, kGridSide);
ASSERT_GE(surfaceH, kGridSide);
const int cellW = surfaceW / kGridSide;
const int cellH = surfaceH / kGridSide;
glBindFramebuffer(GL_FRAMEBUFFER, 0);
// Paint a value no viewport index can produce, so an unwritten cell is obvious.
glClearColor(0.0f, 0.0f, 0.0f, 1.0f);
glClear(GL_COLOR_BUFFER_BIT);
// The default framebuffer is 8-bit RGBA, so the index travels as a colour: cell i is
// painted with red = i * 16, which is exact in 8 bits for i in [0, 16).
const char* const kColorFragmentSource = R"(#version 410 core
flat in int gsIndex;
layout(location = 0) out vec4 fragColor;
void main() { fragColor = vec4(float(gsIndex) * 16.0 / 255.0, 0.0, 0.0, 1.0); }
)";
const GLuint colorProgram = BuildProgram(kGridGeometrySource, kColorFragmentSource);
ASSERT_NE(colorProgram, 0u) << "colour program failed to build: " << m_buildLog;
SetupGridViewports(cellW, cellH);
glUseProgram(colorProgram);
glBindVertexArray(m_vao);
glDrawArrays(GL_POINTS, 0, 1);
ASSERT_EQ(glGetError(), GL_NO_ERROR);
std::vector<unsigned char> pixels(static_cast<size_t>(surfaceW) * surfaceH * 4, 0);
glReadPixels(0, 0, surfaceW, surfaceH, GL_RGBA, GL_UNSIGNED_BYTE, pixels.data());
for (int y = 0; y < kGridSide; ++y) {
for (int x = 0; x < kGridSide; ++x) {
const int px = x * cellW + cellW / 2;
const int py = y * cellH + cellH / 2;
const int red = pixels[(static_cast<size_t>(py) * surfaceW + px) * 4];
const int expected = (y * kGridSide + x) * 16;
// One LSB of slack for an 8-bit round trip; the values are 16 apart, so this
// cannot confuse two neighbouring indices.
EXPECT_LE(std::abs(red - expected), 1)
<< "default-framebuffer cell (" << x << ", " << y << ") holds red=" << red << ", expected "
<< expected << ". A vertically mirrored grid means the Y-flip was applied to viewport 0 "
<< "only";
}
}
glDeleteProgram(colorProgram);
}
// --- 2. the depth range --------------------------------------------------------------
TEST_F(ViewportArrayScenario, EachViewportIndexUsesItsOwnDepthRange) {
// 16 columns one pixel wide and two rows tall: row 0 gets the near-plane quad, row 1
// the far-plane one, so both ends of viewport i's range land in the same column.
constexpr int kWidth = kViewportCount;
constexpr int kHeight = 2;
GLuint texture = 0;
GLuint fbo = 0;
glGenTextures(1, &texture);
glBindTexture(GL_TEXTURE_2D, texture);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
glTexImage2D(GL_TEXTURE_2D, 0, GL_R32F, kWidth, kHeight, 0, GL_RED, GL_FLOAT, nullptr);
glGenFramebuffers(1, &fbo);
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
glFramebufferTexture2D(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, texture, 0);
const GLfloat clearValue[4] = {-1.0f, 0.0f, 0.0f, 0.0f};
glClearBufferfv(GL_COLOR, 0, clearValue);
std::vector<GLfloat> viewports(static_cast<size_t>(kViewportCount) * 4);
std::vector<GLdouble> ranges(static_cast<size_t>(kViewportCount) * 2);
for (int i = 0; i < kViewportCount; ++i) {
viewports[static_cast<size_t>(i) * 4 + 0] = static_cast<GLfloat>(i);
viewports[static_cast<size_t>(i) * 4 + 1] = 0.0f;
viewports[static_cast<size_t>(i) * 4 + 2] = 1.0f;
viewports[static_cast<size_t>(i) * 4 + 3] = 2.0f;
ranges[static_cast<size_t>(i) * 2 + 0] = static_cast<GLdouble>(i) / 16.0;
ranges[static_cast<size_t>(i) * 2 + 1] = 1.0 - static_cast<GLdouble>(i) / 16.0;
}
glViewportArrayv(0, kViewportCount, viewports.data());
glDepthRangeArrayv(0, kViewportCount, ranges.data());
const GLuint depthProgram = BuildProgram(kDepthGeometrySource, kDepthFragmentSource);
ASSERT_NE(depthProgram, 0u) << "depth program failed to build: " << m_buildLog;
glUseProgram(depthProgram);
glBindVertexArray(m_vao);
glDrawArrays(GL_POINTS, 0, 1);
ASSERT_EQ(glGetError(), GL_NO_ERROR);
std::vector<GLfloat> pixels(static_cast<size_t>(kWidth) * kHeight, 0.0f);
glReadPixels(0, 0, kWidth, kHeight, GL_RED, GL_FLOAT, pixels.data());
for (int i = 0; i < kViewportCount; ++i) {
const float nearDepth = static_cast<float>(i) / 16.0f;
const float farDepth = 1.0f - static_cast<float>(i) / 16.0f;
// The tolerance covers depth-buffer-free rasterization of gl_FragCoord.z on a
// software rasterizer; the per-index values are 1/16 apart, so it cannot let a
// neighbouring viewport's range through, and viewport 0's range (0, 1) differs
// from every other index by at least 1/16.
EXPECT_NEAR(pixels[i], nearDepth, 1.0e-3f)
<< "viewport " << i << " near-plane depth; got viewport 0's range if this is 0";
EXPECT_NEAR(pixels[static_cast<size_t>(kWidth) + i], farDepth, 1.0e-3f)
<< "viewport " << i << " far-plane depth; got viewport 0's range if this is 1";
}
glDeleteProgram(depthProgram);
glBindFramebuffer(GL_FRAMEBUFFER, 0);
glDeleteFramebuffers(1, &fbo);
glDeleteTextures(1, &texture);
}
// --- 3. the per-index scissor-test enable --------------------------------------------
TEST_F(ViewportArrayScenario, AnIndexedScissorEnableClipsOnlyThatIndex) {
IntTarget target = MakeIntTarget(kSurfaceSide, kSurfaceSide);
// One full-size viewport per index so the scissor rectangle is the ONLY thing that
// can shrink the quad - the same separation KHR-GL43.viewport_array.scissor uses.
glViewport(0, 0, kSurfaceSide, kSurfaceSide);
std::vector<GLint> boxes(static_cast<size_t>(kViewportCount) * 4);
for (int y = 0; y < kGridSide; ++y) {
for (int x = 0; x < kGridSide; ++x) {
const size_t base = static_cast<size_t>(y * kGridSide + x) * 4;
boxes[base + 0] = x * kCellSize;
boxes[base + 1] = y * kCellSize;
boxes[base + 2] = kCellSize;
boxes[base + 3] = kCellSize;
}
}
glScissorArrayv(0, kViewportCount, boxes.data());
const GLuint singleProgram = BuildProgram(kSingleGeometrySource, kIntFragmentSource);
ASSERT_NE(singleProgram, 0u) << "single-viewport program failed to build: " << m_buildLog;
glUseProgram(singleProgram);
glBindVertexArray(m_vao);
const GLint uViewport = glGetUniformLocation(singleProgram, "uViewport");
ASSERT_NE(uViewport, -1);
constexpr GLint kProbeIndex = 6; // grid cell (2, 1)
constexpr int kProbeX = kProbeIndex % kGridSide;
constexpr int kProbeY = kProbeIndex / kGridSide;
// (a) scissor test ENABLED for this index: the quad is clipped to its 32x32 box.
glUniform1i(uViewport, kProbeIndex);
glEnablei(GL_SCISSOR_TEST, kProbeIndex);
glDrawArrays(GL_POINTS, 0, 1);
ASSERT_EQ(glGetError(), GL_NO_ERROR);
{
const std::vector<GLint> pixels = ReadInts(kSurfaceSide, kSurfaceSide);
EXPECT_EQ(CellCentre(pixels, kSurfaceSide, kProbeX, kProbeY), kProbeIndex)
<< "the scissored index must still paint inside its own box";
for (int y = 0; y < kGridSide; ++y) {
for (int x = 0; x < kGridSide; ++x) {
if (x == kProbeX && y == kProbeY) continue;
EXPECT_EQ(CellCentre(pixels, kSurfaceSide, x, y), kUnwritten)
<< "cell (" << x << ", " << y << ") is outside scissor rectangle " << kProbeIndex
<< " and must be untouched";
}
}
}
// (b) scissor test DISABLED for the same index, everything else identical: with no
// per-viewport toggle in Vulkan this is the case that needs the disabled index to be
// given the full framebuffer rectangle, and it is exactly where "leave the last
// rectangle bound" would show up as a still-clipped quad.
FillIntTarget(target, kSurfaceSide, kSurfaceSide);
glBindFramebuffer(GL_FRAMEBUFFER, target.fbo);
glDisablei(GL_SCISSOR_TEST, kProbeIndex);
glDrawArrays(GL_POINTS, 0, 1);
ASSERT_EQ(glGetError(), GL_NO_ERROR);
{
const std::vector<GLint> pixels = ReadInts(kSurfaceSide, kSurfaceSide);
for (int y = 0; y < kGridSide; ++y) {
for (int x = 0; x < kGridSide; ++x) {
EXPECT_EQ(CellCentre(pixels, kSurfaceSide, x, y), kProbeIndex)
<< "with the scissor test off for index " << kProbeIndex
<< ", its full-viewport quad must cover cell (" << x << ", " << y << ")";
}
}
}
glDeleteProgram(singleProgram);
DestroyIntTarget(target);
}
} // namespace
} // namespace MGITest
@@ -250,6 +250,34 @@ namespace MobileGL::MG_State::GLState {
NotifyContentWrite(atOffset, data.size);
}
void BufferObject::FillSubData(DataPtr pattern, SizeT atOffset, SizeT size) {
MOBILEGL_ASSERT(pattern.data != nullptr && pattern.size > 0,
"FillSubData requires a non-empty pattern.");
MOBILEGL_ASSERT(size % pattern.size == 0,
"FillSubData size (%zu) must be a multiple of pattern size (%zu).", size, pattern.size);
MOBILEGL_ASSERT(atOffset <= m_size && size <= m_size - atOffset,
"FillSubData out of bounds: atOffset (%zu) + size (%zu) > m_size (%zu)", atOffset, size,
m_size);
MOBILEGL_ASSERT(!m_isMapped || (m_mappingAccess & BufferMappingAccessBit::Persistent),
"Cannot fill data while buffer is non-persistently mapped.");
if (size == 0) return;
// A clear is ordered after all earlier GPU writes. Partial clears additionally need the
// retained shadow bytes; whole-store clears need the same synchronization before writing
// an adopted persistent mapping that the GPU may still be accessing.
SyncGpuWrites();
Uint8* dst = m_resource.Bytes() + atOffset;
if (pattern.size == 1) {
Memset(dst, *static_cast<const Uint8*>(pattern.data), size);
} else {
for (SizeT at = 0; at < size; at += pattern.size) {
Memcpy(dst + at, pattern.data, pattern.size);
}
}
NotifyContentWrite(atOffset, size);
}
void BufferObject::DownloadSubData(void* dst, SizeT atOffset, SizeT size) const {
MOBILEGL_ASSERT(atOffset + size <= m_size,
"DownloadSubData out of bounds: atOffset (%zu) + size (%zu) > m_size (%zu)", atOffset, size,
@@ -132,6 +132,9 @@ namespace MobileGL {
void UploadData(DataPtr data, SizeT atOffset);
void UploadSubData(DataPtr data, SizeT atOffset);
// Repeats one already-converted element through [atOffset, atOffset + size) and
// publishes the range as one content mutation.
void FillSubData(DataPtr pattern, SizeT atOffset, SizeT size);
// Reads `size` bytes from the CPU shadow at `atOffset` into `dst` (glGetBufferSubData).
// The shadow reflects CPU writes (BufferData/SubData/maps) and backend write-backs, but not
// arbitrary GPU-side writes.
+30 -1
View File
@@ -39,6 +39,11 @@ namespace MobileGL::MG_State {
return m_compileEnv;
}
void GLContext::InvalidateCompileEnv() {
m_compileEnv.reset();
m_compileEnvBackend = nullptr;
}
// Error
void GLContext::RecordError(ErrorCode code, UniquePtr<ErrorInfo> info) {
// Invariant I1, mechanically enforced: the GL error state is GL-thread-owned.
@@ -712,10 +717,18 @@ namespace MobileGL::MG_State {
m_renderState.SetViewport(viewport);
}
const IntVec4& GLContext::GetViewport() const {
IntVec4 GLContext::GetViewport() const {
return m_renderState.GetViewport();
}
void GLContext::SetViewportIndexed(Uint index, FloatVec4 viewport) {
m_renderState.SetViewportIndexed(index, viewport);
}
const FloatVec4& GLContext::GetViewportIndexed(Uint index) const {
return m_renderState.GetViewportIndexed(index);
}
void GLContext::SetLineWidth(Float width) {
m_renderState.SetLineWidth(width);
}
@@ -953,6 +966,14 @@ namespace MobileGL::MG_State {
return m_renderState.GetDepthRange();
}
void GLContext::SetDepthRangeIndexed(Uint index, FloatVec2 range) {
m_renderState.SetDepthRangeIndexed(index, range);
}
const FloatVec2& GLContext::GetDepthRangeIndexed(Uint index) const {
return m_renderState.GetDepthRangeIndexed(index);
}
void GLContext::SetSampleCoverage(Float value, Bool invert) {
m_renderState.SetSampleCoverage(value, invert);
}
@@ -1017,6 +1038,14 @@ namespace MobileGL::MG_State {
return m_renderState.GetScissorBox();
}
void GLContext::SetScissorBoxIndexed(Uint index, IntVec4 box) {
m_renderState.SetScissorBoxIndexed(index, box);
}
const IntVec4& GLContext::GetScissorBoxIndexed(Uint index) const {
return m_renderState.GetScissorBoxIndexed(index);
}
// Framebuffer
void GLContext::GenFramebufferNames(Uint number, Vector<Uint>& framebuffers) {
m_framebufferState.GenerateNames(number, framebuffers);
+15 -6
View File
@@ -198,8 +198,10 @@ namespace MobileGL {
// Only the pipeline-relevant subset - see RenderState::m_pipelineStateVersion.
Uint GetPipelineStateVersion() const;
const RenderStateParameters& GetRenderStateParameters() const;
void SetViewport(IntVec4 viewport); // x, y, width, height
const IntVec4& GetViewport() const; // x, y, width, height
void SetViewport(IntVec4 viewport); // x, y, width, height; writes ALL viewports
IntVec4 GetViewport() const; // x, y, width, height; viewport 0, rounded
void SetViewportIndexed(Uint index, FloatVec4 viewport);
const FloatVec4& GetViewportIndexed(Uint index) const;
void SetLineWidth(Float width);
Float GetLineWidth() const;
void SetPointSize(Float size);
@@ -260,8 +262,10 @@ namespace MobileGL {
Uint32 GetClearStencil() const;
void SetBlendColor(FloatVec4 color);
const FloatVec4& GetBlendColor() const;
void SetDepthRange(FloatVec2 range);
void SetDepthRange(FloatVec2 range); // writes ALL viewports' depth ranges
const FloatVec2& GetDepthRange() const;
void SetDepthRangeIndexed(Uint index, FloatVec2 range);
const FloatVec2& GetDepthRangeIndexed(Uint index) const;
void SetSampleCoverage(Float value, Bool invert);
Float GetSampleCoverageValue() const;
Bool GetSampleCoverageInvert() const;
@@ -276,8 +280,10 @@ namespace MobileGL {
FrontFaceMode GetFrontFaceMode() const;
void SetProvokingVertexMode(ProvokingVertexMode mode);
ProvokingVertexMode GetProvokingVertexMode() const;
void SetScissorBox(IntVec4 box); // x, y, width, height
const IntVec4& GetScissorBox() const; // x, y, width, height
void SetScissorBox(IntVec4 box); // x, y, width, height; writes ALL rectangles
const IntVec4& GetScissorBox() const; // x, y, width, height; rectangle 0
void SetScissorBoxIndexed(Uint index, IntVec4 box);
const IntVec4& GetScissorBoxIndexed(Uint index) const;
// Transform feedback. The fields below are the state of the transform
// feedback object currently bound to GL_TRANSFORM_FEEDBACK; see the object
@@ -407,9 +413,12 @@ namespace MobileGL {
// cannot be captured in MG_State::Init() - that runs BEFORE MG_Backend::Init(),
// so there is no backend to query yet. Re-captured whenever the active backend
// object changes, which also rolls the fingerprint and therefore invalidates
// every P0b preprocess memo keyed against the old one.
// every P0b preprocess memo keyed against the old one. A backend whose dynamic
// capabilities become available without changing object identity must call
// InvalidateCompileEnv() after publishing them.
// GL thread only.
const SharedPtr<const MG_Util::ShaderTranspiler::CompileEnv>& GetCompileEnv();
void InvalidateCompileEnv();
private:
// State Components
@@ -60,6 +60,8 @@ namespace MobileGL::MG_State::GLState {
Uint externalIndex = 0; // logs only
Vector<LinkShaderInput> shaders; // already stage-sorted
SharedPtr<const MG_Util::ShaderTranspiler::CompileEnv> env;
// Startup configuration copied with the task, never read from worker code.
Bool enableSpirvValidation = false;
// The four "takes effect at the next link" request maps. Snapshotted rather than
// referenced, which is precisely what makes glBindAttribLocation and friends
// legal to call over a pending link without cancelling it: the pending link keeps
@@ -494,6 +494,7 @@ namespace MobileGL::MG_State::GLState {
auto task = MakeShared<ProgramLinkTask>();
task->in.externalIndex = m_externalIndex;
task->in.env = MG_Util::ShaderTranspiler::GetCurrentCompileEnv();
task->in.enableSpirvValidation = MG_Config::Features.EnableSpirvValidation;
task->in.explicitAttribLocations = m_explicitAttribLocations;
task->in.explicitFragDataLocation = m_explicitFragDataLocation;
task->in.explicitFragDataIndex = m_explicitFragDataIndex;
@@ -819,6 +819,9 @@ namespace MobileGL::MG_State::GLState {
// backend asks this exactly where it used to ask GetLinkStatus(), i.e. right before
// it builds or draws with the program.
Bool GetSpirvStatus() const { return Spirv().spirvStatus; }
// Copied from the link task that generated this program's SPIR-V. Backends use it for
// their final transforms, which must honor the same diagnostic setting as phase B.
Bool GetSpirvValidationEnabled() const { return Spirv().enableSpirvValidation; }
// The linked glslang reflection itself, for the ONE consumer that needs resource
// lists no typed getter above exposes: the GL program-interface query layer
@@ -985,6 +988,7 @@ namespace MobileGL::MG_State::GLState {
// cannot be lifted out of glslang's reflection instead.
struct SpirvArtifacts {
Vector<Vector<unsigned>> generatedSpirv;
Bool enableSpirvValidation = false;
// Byte offset of each uniform location inside globalUboScratch, or
// kInvalidUniformOffset. Sized maxUniformLocation + 1 by the routing pass.
Vector<Uint> uniformOffsets;
@@ -102,7 +102,11 @@ namespace MobileGL::MG_State::GLState {
}
MGLOG_D("ProgramObject %u: Starting SPIR-V generation", externalIndex);
GenerateSpirv(handoff, externalIndex);
const Bool deferOutputValidationForDirectVulkan =
m_phaseA->in.env != nullptr && m_phaseA->in.env->backend == BackendType::DirectVulkan;
const Bool enableSpirvValidation = m_phaseA->in.enableSpirvValidation;
artifacts.enableSpirvValidation = enableSpirvValidation;
GenerateSpirv(handoff, externalIndex, deferOutputValidationForDirectVulkan, enableSpirvValidation);
// GlslangToSpv was the only consumer of the parsed ASTs; everything after this point
// works on the SPIR-V and on the TProgram's own self-contained reflection pool. Drop
// them here rather than at the end of the body, which is ~87% of this node's runtime
@@ -137,7 +141,9 @@ namespace MobileGL::MG_State::GLState {
artifacts.generatedSpirv.size());
}
void ProgramSpirvTask::GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, const Uint externalIndex) {
void ProgramSpirvTask::GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, const Uint externalIndex,
const Bool deferOutputValidationForDirectVulkan,
const Bool enableSpirvValidation) {
/* As we passed first stage compilation/linking,
* we'll assume all the operations here should
* pass. We may be able to employ some optimizations
@@ -169,7 +175,8 @@ namespace MobileGL::MG_State::GLState {
Bool allOptimized = true;
{
for (auto& spv : artifacts.generatedSpirv) {
auto success = ShaderCompiler::SanitizeAndOptimizeBinary(spv, spv);
auto success = ShaderCompiler::SanitizeAndOptimizeBinary(
spv, spv, !deferOutputValidationForDirectVulkan, enableSpirvValidation);
if (!success) {
// The one genuine phase-B failure mode: one of the seven optimizer passes
// reported failure, so `spv` is whatever the run left behind. A fordebug
@@ -65,7 +65,8 @@ namespace MobileGL::MG_State::GLState {
private:
void RunBody() override;
void GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, Uint externalIndex);
void GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, Uint externalIndex,
Bool deferOutputValidationForDirectVulkan, Bool enableSpirvValidation);
void BuildGlobalUboRouting(const ProgramLinkTask::SpirvHandoff& handoff, Uint externalIndex);
// Worker-side MGLOG replacement, replayed by the join on the GL thread. Same reason as
@@ -25,6 +25,12 @@ namespace MobileGL {
return 0;
}
}
// Every viewport's scissor-test bit set, i.e. what glEnable(GL_SCISSOR_TEST) writes.
constexpr Uint32 kAllViewportsMask =
RenderStateParameters::MAX_VIEWPORTS >= 32
? ~0u
: (1u << RenderStateParameters::MAX_VIEWPORTS) - 1u;
} // namespace
RenderState::RenderState() {
@@ -32,6 +38,15 @@ namespace MobileGL {
for (auto& mask : m_parameters.ColorMasks) {
mask = BoolVec4(true, true, true, true);
}
// Every viewport's depth range starts at (0, 1) - GL 4.6 core table 23.4. The
// viewport and scissor rectangles legitimately start all-zero here: their spec
// initial value is the size of the window the context is first made current to,
// which the frontend does not know yet, so an all-zero rectangle means "never
// written" and the backends resolve it against the live surface (see
// DirectGLES' SyncRenderState and VulkanRenderer's ApplyGLViewportState).
for (auto& range : m_parameters.DepthRanges) {
range = FloatVec2(0.0f, 1.0f);
}
}
Uint RenderState::GetVersion() const {
@@ -47,15 +62,47 @@ namespace MobileGL {
}
// -------------------- Rasterization --------------------
// ARB_viewport_array, "Additions to Chapter 2": Viewport(x, y, w, h) is equivalent to
// ViewportIndexedf(i, x, y, w, h) for every i in [0, MAX_VIEWPORTS) - it is not a
// synonym for "viewport 0".
void RenderState::SetViewport(IntVec4 viewport) {
if (m_parameters.Viewport == viewport) return;
const FloatVec4 asFloat(static_cast<Float>(viewport.x()), static_cast<Float>(viewport.y()),
static_cast<Float>(viewport.z()), static_cast<Float>(viewport.w()));
Bool stateChanged = false;
for (auto& stored : m_parameters.Viewports) {
if (stored == asFloat) continue;
stored = asFloat;
stateChanged = true;
}
if (stateChanged) ++m_version;
}
m_parameters.Viewport = viewport;
IntVec4 RenderState::GetViewport() const {
const FloatVec4& viewport = m_parameters.Viewports[0];
// Round rather than truncate: glGetIntegerv on floating-point state rounds to
// nearest (GL 4.6 core 22.2), and truncating a 63.5-wide viewport to 63 would
// also hand the backends a rectangle one pixel short of what was asked for.
return IntVec4(static_cast<Int>(std::lround(viewport.x())), static_cast<Int>(std::lround(viewport.y())),
static_cast<Int>(std::lround(viewport.z())), static_cast<Int>(std::lround(viewport.w())));
}
void RenderState::SetViewportIndexed(Uint index, FloatVec4 viewport) {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Viewport index out of range: %u", index);
return;
}
if (m_parameters.Viewports[index] == viewport) return;
m_parameters.Viewports[index] = viewport;
++m_version;
}
const IntVec4& RenderState::GetViewport() const {
return m_parameters.Viewport;
const FloatVec4& RenderState::GetViewportIndexed(Uint index) const {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Viewport index out of range: %u", index);
return m_parameters.Viewports[0];
}
return m_parameters.Viewports[index];
}
void RenderState::SetLineWidth(Float width) {
@@ -223,7 +270,6 @@ namespace MobileGL {
SET_CAPABILITY(SampleAlphaToOne, enabled);
SET_CAPABILITY(SampleCoverage, enabled);
SET_CAPABILITY(SampleMask, enabled);
SET_CAPABILITY(ScissorTest, enabled);
SET_CAPABILITY(StencilTest, enabled);
SET_CAPABILITY(ProgramPointSize, enabled);
case CapabilityInput::Blend: {
@@ -236,6 +282,17 @@ namespace MobileGL {
if (stateChanged) BumpVersions();
break;
}
// GL 4.6 core 17.3.2: the non-indexed Enable/Disable(SCISSOR_TEST) enables or
// disables the test for ALL viewports, exactly like glViewport writes all
// viewports. Anything narrower fails KHR-GL43.viewport_array.scissor_test_state_api,
// whose "enable all" phase reads every index back through glIsEnabledi.
case CapabilityInput::ScissorTest: {
const Uint32 updated = enabled ? kAllViewportsMask : 0u;
if (m_parameters.ScissorTestEnabledMask == updated) break;
m_parameters.ScissorTestEnabledMask = updated;
BumpVersions();
break;
}
case CapabilityInput::ClipDistance0:
case CapabilityInput::ClipDistance1:
case CapabilityInput::ClipDistance2:
@@ -287,11 +344,14 @@ namespace MobileGL {
RETURN_CAPABILITY(SampleAlphaToOne);
RETURN_CAPABILITY(SampleCoverage);
RETURN_CAPABILITY(SampleMask);
RETURN_CAPABILITY(ScissorTest);
RETURN_CAPABILITY(StencilTest);
RETURN_CAPABILITY(ProgramPointSize);
case CapabilityInput::Blend:
return m_parameters.BlendStates[0].Enabled;
// The non-indexed query of an indexed capability answers for index 0
// (GL 4.6 core 22.1), which is also the only bit either backend consumes today.
case CapabilityInput::ScissorTest:
return (m_parameters.ScissorTestEnabledMask & 1u) != 0;
case CapabilityInput::ClipDistance0:
case CapabilityInput::ClipDistance1:
case CapabilityInput::ClipDistance2:
@@ -307,13 +367,29 @@ namespace MobileGL {
}
void RenderState::SetCapabilityIndexed(CapabilityInput cap, Uint index, Bool enabled) {
// Only for BlendState currently. The GL entry points (glEnablei/glDisablei) already
// reject every non-GL_BLEND target with GL_INVALID_ENUM before reaching here, so this
// is a backstop - but it must stay a backstop: THROW_UNIMPL_EXCEPTION unwinds a C++
// exception through the C GL ABI and terminates the process.
// GL_BLEND (indexed by draw buffer) and GL_SCISSOR_TEST (indexed by viewport) are
// the only indexed capabilities in GL 4.6 core. The GL entry points
// (glEnablei/glDisablei) already reject every other target with GL_INVALID_ENUM
// and every out-of-range index with GL_INVALID_VALUE before reaching here, so the
// guards below are backstops - but they must stay backstops:
// THROW_UNIMPL_EXCEPTION unwinds a C++ exception through the C GL ABI and
// terminates the process.
if (cap == CapabilityInput::ScissorTest) {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Scissor test capability index out of range: %u", index);
return;
}
const Uint32 bit = 1u << index;
const Uint32 updated = enabled ? (m_parameters.ScissorTestEnabledMask | bit)
: (m_parameters.ScissorTestEnabledMask & ~bit);
if (updated == m_parameters.ScissorTestEnabledMask) return;
m_parameters.ScissorTestEnabledMask = updated;
BumpVersions();
return;
}
if (cap != CapabilityInput::Blend) {
MGLOG_I("RenderState::SetCapabilityIndexed: indexed capability state exists only for "
"GL_BLEND (cap=%d, index=%u); ignoring",
"GL_BLEND and GL_SCISSOR_TEST (cap=%d, index=%u); ignoring",
static_cast<int>(cap), index);
return;
}
@@ -328,9 +404,17 @@ namespace MobileGL {
}
Bool RenderState::IsCapabilityEnabledIndexed(CapabilityInput cap, Uint index) const {
// Only for BlendState currently - same backstop reasoning as SetCapabilityIndexed:
// glIsEnabledi has already answered GL_INVALID_ENUM/GL_FALSE for anything else, and a
// query must never be able to terminate the process.
// GL_BLEND and GL_SCISSOR_TEST only - same backstop reasoning as
// SetCapabilityIndexed: glIsEnabledi has already answered
// GL_INVALID_ENUM/GL_INVALID_VALUE for anything else, and a query must never be
// able to terminate the process.
if (cap == CapabilityInput::ScissorTest) {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Scissor test capability index out of range: %u", index);
return false;
}
return (m_parameters.ScissorTestEnabledMask & (1u << index)) != 0;
}
if (cap != CapabilityInput::Blend) {
MGLOG_I("RenderState::IsCapabilityEnabledIndexed: indexed capability state exists only "
"for GL_BLEND (cap=%d, index=%u); reporting disabled",
@@ -591,15 +675,39 @@ namespace MobileGL {
return m_parameters.BlendColor;
}
// Like Viewport: ARB_viewport_array makes DepthRange(n, f) the same as
// DepthRangeIndexed(i, n, f) for every i.
void RenderState::SetDepthRange(FloatVec2 range) {
if (m_parameters.DepthRange == range) return;
m_parameters.DepthRange = range;
++m_version;
Bool stateChanged = false;
for (auto& stored : m_parameters.DepthRanges) {
if (stored == range) continue;
stored = range;
stateChanged = true;
}
if (stateChanged) ++m_version;
}
const FloatVec2& RenderState::GetDepthRange() const {
return m_parameters.DepthRange;
return m_parameters.DepthRanges[0];
}
void RenderState::SetDepthRangeIndexed(Uint index, FloatVec2 range) {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Depth range index out of range: %u", index);
return;
}
if (m_parameters.DepthRanges[index] == range) return;
m_parameters.DepthRanges[index] = range;
++m_version;
}
const FloatVec2& RenderState::GetDepthRangeIndexed(Uint index) const {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Depth range index out of range: %u", index);
return m_parameters.DepthRanges[0];
}
return m_parameters.DepthRanges[index];
}
void RenderState::SetSampleCoverage(Float value, Bool invert) {
@@ -726,15 +834,39 @@ namespace MobileGL {
}
// --------------------- Scissor ---------------------
// Like Viewport: ARB_viewport_array makes Scissor(x, y, w, h) the same as
// ScissorIndexed(i, x, y, w, h) for every i.
void RenderState::SetScissorBox(IntVec4 box) {
if (m_parameters.ScissorBox == box) return;
m_parameters.ScissorBox = box;
++m_version;
Bool stateChanged = false;
for (auto& stored : m_parameters.ScissorBoxes) {
if (stored == box) continue;
stored = box;
stateChanged = true;
}
if (stateChanged) ++m_version;
}
const IntVec4& RenderState::GetScissorBox() const {
return m_parameters.ScissorBox;
return m_parameters.ScissorBoxes[0];
}
void RenderState::SetScissorBoxIndexed(Uint index, IntVec4 box) {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Scissor box index out of range: %u", index);
return;
}
if (m_parameters.ScissorBoxes[index] == box) return;
m_parameters.ScissorBoxes[index] = box;
++m_version;
}
const IntVec4& RenderState::GetScissorBoxIndexed(Uint index) const {
if (index >= RenderStateParameters::MAX_VIEWPORTS) {
MOBILEGL_ASSERT(false, "Scissor box index out of range: %u", index);
return m_parameters.ScissorBoxes[0];
}
return m_parameters.ScissorBoxes[index];
}
} // namespace GLState
} // namespace MG_State
@@ -220,8 +220,22 @@ namespace MobileGL {
};
struct RenderStateParameters {
// ARB_viewport_array / GL 4.6 core 13.6.1: the viewport, the scissor rectangle, the depth
// range and the scissor-test enable are all arrays indexed by gl_ViewportIndex, and the
// spec floor for MAX_VIEWPORTS is 16. MobileGL advertises exactly 16 on both backends, so
// this is also what GL_MAX_VIEWPORTS reports (see the backend loaders' caps.MaxViewports).
static constexpr Uint MAX_VIEWPORTS = 16;
// Rasterization
IntVec4 Viewport = IntVec4(0, 0, 0, 0); // x, y, width, height
// The viewport rectangle is FLOAT state as of GL 4.1 - ViewportIndexedf writes fractional
// values and GetFloati_v(GL_VIEWPORT) must hand them back bit-exact
// (KHR-GL43.viewport_array.viewport_api compares with ==, no tolerance). glViewport's
// integers are simply one way to write it. Index 0 is what a program that never assigns
// gl_ViewportIndex rasterizes against, and what the classic glViewport /
// glGetIntegerv(GL_VIEWPORT) pair addresses. Both backends rasterize the rectangle
// rounded back to integers; the STATE stays exact, which is the half the conformance
// suite checks (see the KNOWN INFIDELITY note in AdvertisedLimitsScenario.cpp).
Array<FloatVec4, MAX_VIEWPORTS> Viewports{}; // x, y, width, height
Float LineWidth = 1.0f;
Float PointSize = 1.0f;
// GL_PATCH_VERTICES: how many vertices one tessellation patch consumes.
@@ -247,7 +261,13 @@ namespace MobileGL {
Float ClearDepth = 1.0f;
Uint32 ClearStencil = 0;
FloatVec4 BlendColor = FloatVec4(0.0f, 0.0f, 0.0f, 0.0f);
FloatVec2 DepthRange = FloatVec2(0.0f, 1.0f);
// Per-viewport depth range (glDepthRangeIndexed / glDepthRangeArrayv). Every entry is
// initialized to (0, 1) in RenderState's constructor - a default member initializer would
// not survive the Array<> aggregate. Kept float rather than double: DepthRangeArrayv takes
// GLdouble, but the value reaches the hardware as VkViewport::minDepth/maxDepth (float) on
// Magma and glDepthRangef on Espryt, so a double store would only widen the readback and
// then lose it again at the same place.
Array<FloatVec2, MAX_VIEWPORTS> DepthRanges{};
Float SampleCoverageValue = 1.0f;
Bool SampleCoverageInvert = false;
Uint32 SampleMaskValue = 0xffffffffu;
@@ -299,10 +319,15 @@ namespace MobileGL {
Bool SampleAlphaToOneEnabled = false;
Bool SampleCoverageEnabled = false;
Bool SampleMaskEnabled = false;
Bool ScissorTestEnabled = false;
Bool StencilTestEnabled = false;
Bool ProgramPointSizeEnabled = false;
IntVec4 ScissorBox = IntVec4(0, 0, 0, 0); // x, y, width, height
// glEnable(GL_SCISSOR_TEST) enables the test for EVERY viewport, glEnablei for one
// (GL 4.6 core 17.3.2), so this is 16 bits and not a bool. Bit 0 is what the classic
// glIsEnabled(GL_SCISSOR_TEST) reports and what both backends currently consume. Unlike
// ClipDistanceEnabledMask below it DOES bump the pipeline version, because DirectGLES
// turns it into a real glEnable/glDisable.
Uint32 ScissorTestEnabledMask = 0;
Array<IntVec4, MAX_VIEWPORTS> ScissorBoxes{}; // x, y, width, height
// glEnable(GL_CLIP_DISTANCE0 + i) for i in [0, 8), one bit each. A bitmask rather than
// eight bools because every consumer wants the set, not an individual flag, and because
// the SYNC_CAPABILITY/SET_CAPABILITY macros key off a "<Name>Enabled" field name that
@@ -323,8 +348,14 @@ namespace MobileGL {
const RenderStateParameters& GetAllParameters() const;
// Rasterization
// ARB_viewport_array defines glViewport as ViewportIndexedf on EVERY index, so the
// classic setter broadcasts; GetViewport answers for index 0 (rounded to the
// integers glGetIntegerv(GL_VIEWPORT) and both backends want) and is BY VALUE for
// that reason. The indexed pair is the verbatim float state.
void SetViewport(IntVec4 viewport); // x, y, width, height
const IntVec4& GetViewport() const; // x, y, width, height
IntVec4 GetViewport() const; // x, y, width, height, viewport 0, rounded
void SetViewportIndexed(Uint index, FloatVec4 viewport);
const FloatVec4& GetViewportIndexed(Uint index) const;
void SetLineWidth(Float width);
Float GetLineWidth() const;
void SetPointSize(Float size);
@@ -400,8 +431,12 @@ namespace MobileGL {
Uint32 GetClearStencil() const;
void SetBlendColor(FloatVec4 color);
const FloatVec4& GetBlendColor() const;
// glDepthRange(f) writes every viewport's range (ARB_viewport_array); the indexed
// pair is glDepthRangeIndexed / glDepthRangeArrayv. GetDepthRange answers index 0.
void SetDepthRange(FloatVec2 range);
const FloatVec2& GetDepthRange() const;
void SetDepthRangeIndexed(Uint index, FloatVec2 range);
const FloatVec2& GetDepthRangeIndexed(Uint index) const;
void SetSampleCoverage(Float value, Bool invert);
Float GetSampleCoverageValue() const;
Bool GetSampleCoverageInvert() const;
@@ -421,9 +456,12 @@ namespace MobileGL {
void SetProvokingVertexMode(ProvokingVertexMode mode);
ProvokingVertexMode GetProvokingVertexMode() const;
// Scissor
// Scissor. glScissor writes every rectangle (ARB_viewport_array); GetScissorBox
// answers for index 0.
void SetScissorBox(IntVec4 box); // x, y, width, height
const IntVec4& GetScissorBox() const; // x, y, width, height
void SetScissorBoxIndexed(Uint index, IntVec4 box);
const IntVec4& GetScissorBoxIndexed(Uint index) const;
private:
// Bump both: any state change invalidates the draw snapshot, and this one also
@@ -8,9 +8,28 @@
#include <gtest/gtest.h>
#include <iostream>
#include <utility>
#include <vector>
#include <vulkan/vulkan.h>
#include <MG_Backend/DirectVulkan/Renderer/ProgramFactory.h>
TEST(DirectVulkanSanity, ProgramMovePreservesViewportIndexUsage) {
using VkProgramObject = MobileGL::MG_Backend::DirectVulkan::ProgramFactory::VkProgramObject;
VkProgramObject moveConstructedSource;
moveConstructedSource.writesViewportIndexBuiltin = true;
VkProgramObject moveConstructed(std::move(moveConstructedSource));
EXPECT_TRUE(moveConstructed.writesViewportIndexBuiltin);
EXPECT_FALSE(moveConstructedSource.writesViewportIndexBuiltin);
VkProgramObject moveAssignedSource;
moveAssignedSource.writesViewportIndexBuiltin = true;
VkProgramObject moveAssigned;
moveAssigned = std::move(moveAssignedSource);
EXPECT_TRUE(moveAssigned.writesViewportIndexBuiltin);
EXPECT_FALSE(moveAssignedSource.writesViewportIndexBuiltin);
}
TEST(DirectVulkanSanity, ExtensionEnumeration) {
uint32_t extensionCount = 0;
vkEnumerateInstanceExtensionProperties(nullptr, &extensionCount, nullptr);
@@ -725,22 +725,57 @@ TEST(TextureAnisotropyCapabilities, ExtensionIsAdvertisedOnlyWhenTheHostDriverSu
return std::find(extensions.begin(), extensions.end(), wanted) != extensions.end();
};
const auto without = MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false);
const auto without = MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false, false, false);
EXPECT_FALSE(contains(without, MobileGL::E_GL_EXT_texture_filter_anisotropic));
EXPECT_FALSE(contains(without, MobileGL::E_GL_ARB_texture_filter_anisotropic));
const auto with = MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, true);
const auto with = MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, true, false, false);
EXPECT_TRUE(contains(with, MobileGL::E_GL_EXT_texture_filter_anisotropic));
EXPECT_TRUE(contains(with, MobileGL::E_GL_ARB_texture_filter_anisotropic));
// Same rule on the Vulkan backend, where the gate is the samplerAnisotropy device feature.
const auto vkWithout = MobileGL::MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false);
const auto vkWithout = MobileGL::MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false, false);
EXPECT_FALSE(contains(vkWithout, MobileGL::E_GL_EXT_texture_filter_anisotropic));
const auto vkWith = MobileGL::MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, true);
const auto vkWith = MobileGL::MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, true, false);
EXPECT_TRUE(contains(vkWith, MobileGL::E_GL_EXT_texture_filter_anisotropic));
EXPECT_TRUE(contains(vkWith, MobileGL::E_GL_ARB_texture_filter_anisotropic));
}
// Minecraft 26.3 checks ARB_draw_indirect before it considers the already-advertised
// ARB_multi_draw_indirect, then separately requires ARB_base_instance before enabling its terrain
// indirect path. Pin both strings and, just as importantly, the non-zero firstInstance gate.
TEST(IndirectDrawAdvertisement, MatchesEachBackendsUsableCommandSemantics) {
const auto contains = [](const MobileGL::Vector<MobileGL::GLExtension>& extensions,
MobileGL::GLExtension wanted) {
return std::find(extensions.begin(), extensions.end(), wanted) != extensions.end();
};
const auto esWithoutIndirect =
MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false, false, false);
EXPECT_FALSE(contains(esWithoutIndirect, MobileGL::E_GL_ARB_draw_indirect));
EXPECT_FALSE(contains(esWithoutIndirect, MobileGL::E_GL_ARB_base_instance));
const auto esWithoutBaseInstance =
MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false, true, false);
EXPECT_TRUE(contains(esWithoutBaseInstance, MobileGL::E_GL_ARB_draw_indirect));
EXPECT_FALSE(contains(esWithoutBaseInstance, MobileGL::E_GL_ARB_base_instance));
const auto esWithBoth =
MobileGL::MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false, true, true);
EXPECT_TRUE(contains(esWithBoth, MobileGL::E_GL_ARB_draw_indirect));
EXPECT_TRUE(contains(esWithBoth, MobileGL::E_GL_ARB_base_instance));
const auto vkWithoutBaseInstance =
MobileGL::MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false, false);
EXPECT_TRUE(contains(vkWithoutBaseInstance, MobileGL::E_GL_ARB_draw_indirect));
EXPECT_FALSE(contains(vkWithoutBaseInstance, MobileGL::E_GL_ARB_base_instance));
const auto vkWithBoth =
MobileGL::MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false, true);
EXPECT_TRUE(contains(vkWithBoth, MobileGL::E_GL_ARB_draw_indirect));
EXPECT_TRUE(contains(vkWithBoth, MobileGL::E_GL_ARB_base_instance));
}
TEST(TextureAnisotropyCapabilities, MaxAnisotropyIsQueriedOnlyWhenTheExtensionIsPresent) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
@@ -836,3 +871,63 @@ TEST(MultiDrawCapabilities, ExtensionWithoutResolvedPointerIsNotSupport) {
EXPECT_FALSE(caps.SupportsMultiDrawIndirect);
EXPECT_FALSE(caps.SupportsMultiDrawElementsBaseVertex);
}
TEST(DrawIndirectCapabilities, RequiresEs31AndBothCoreEntryPoints) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
auto funcs = MakeFakeGLESFunctions();
funcs.glDrawElementsIndirect = [](GLenum, GLenum, const void*) {};
MobileGL::MG_External::GLESCapabilities supportedCaps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(supportedCaps, funcs));
EXPECT_TRUE(supportedCaps.SupportsDrawIndirect);
// The same pointers on an ES 3.0 context are not core entry points and cannot back the
// desktop extension contract.
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.glesMinorVersion = 0;
MobileGL::MG_External::GLESCapabilities es30Caps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(es30Caps, funcs));
EXPECT_FALSE(es30Caps.SupportsDrawIndirect);
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
const auto missingElements = MakeFakeGLESFunctions();
MobileGL::MG_External::GLESCapabilities missingEntryPointCaps;
ASSERT_TRUE(
MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(missingEntryPointCaps, missingElements));
EXPECT_FALSE(missingEntryPointCaps.SupportsDrawIndirect);
}
TEST(BaseInstanceCapabilities, RequiresTheExtensionAndAllThreeEntryPoints) {
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
auto funcs = MakeFakeGLESFunctions();
funcs.glDrawArraysInstancedBaseInstanceEXT = [](GLenum, GLint, GLsizei, GLsizei, GLuint) {};
funcs.glDrawElementsInstancedBaseInstanceEXT =
[](GLenum, GLsizei, GLenum, const void*, GLsizei, GLuint) {};
funcs.glDrawElementsInstancedBaseVertexBaseInstanceEXT =
[](GLenum, GLsizei, GLenum, const void*, GLsizei, GLint, GLuint) {};
// Resolved stubs alone must never make the capability true.
MobileGL::MG_External::GLESCapabilities pointersOnlyCaps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(pointersOnlyCaps, funcs));
EXPECT_FALSE(pointersOnlyCaps.SupportsBaseInstance);
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.extensions.emplace_back("GL_EXT_base_instance");
MobileGL::MG_External::GLESCapabilities supportedCaps;
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(supportedCaps, funcs));
EXPECT_TRUE(supportedCaps.SupportsBaseInstance);
ResetFakeDriver();
g_fake.maxVertexSsboBlocks = 0;
g_fake.extensions.emplace_back("GL_EXT_base_instance");
funcs.glDrawElementsInstancedBaseInstanceEXT = nullptr;
MobileGL::MG_External::GLESCapabilities missingEntryPointCaps;
ASSERT_TRUE(
MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(missingEntryPointCaps, funcs));
EXPECT_FALSE(missingEntryPointCaps.SupportsBaseInstance);
}
+112
View File
@@ -16,6 +16,7 @@
#include <MG_State/GLState/Core.h>
#include <MG_Impl/GLImpl/Buffer/GL_Buffer.h>
#include <MG_Impl/GetProcAddress.h>
#include <MG_Impl/GLImpl/Getter/GL_Getter.h>
using namespace MobileGL;
@@ -599,6 +600,117 @@ TEST_F(BufferTest, ClearNamedBufferSubDataRepeatsPattern) {
EXPECT_EQ(actual, (Vector<Uint32>{0, pattern, pattern, pattern, 0}));
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
}
TEST_F(BufferTest, ClearBufferSubDataInitializesIrisStaticSsboRange) {
GLuint buffer = 0;
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, buffer);
Vector<Uint8> initial(32, 0x7F);
MobileGL::MG_Impl::GLImpl::BufferData(
GL_SHADER_STORAGE_BUFFER, initial.size(), initial.data(), GL_STATIC_DRAW);
const GLbyte zero = 0;
const auto clear = reinterpret_cast<PFNGLCLEARBUFFERSUBDATAPROC>(
MobileGL::MG_Impl::GetProcAddress("glClearBufferSubData"));
ASSERT_NE(clear, nullptr);
clear(GL_SHADER_STORAGE_BUFFER, GL_R8, 4, 24, GL_RED, GL_BYTE, &zero);
Vector<Uint8> actual(initial.size());
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
ASSERT_NE(bufferObject, nullptr);
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size());
EXPECT_EQ(actual, (Vector<Uint8>{0x7F, 0x7F, 0x7F, 0x7F,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0x7F, 0x7F, 0x7F, 0x7F}));
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
DrainPendingGlErrors();
}
TEST_F(BufferTest, ClearBufferSubDataInitializesCompleteIrisStaticSsbo) {
constexpr SizeT irisStaticSsboSize = 5'000'192;
GLuint buffer = 0;
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, buffer);
Vector<Uint8> initial(irisStaticSsboSize, 0x7F);
MobileGL::MG_Impl::GLImpl::BufferData(
GL_SHADER_STORAGE_BUFFER, initial.size(), initial.data(), GL_STATIC_DRAW);
const GLbyte zero = 0;
MobileGL::MG_Impl::GLImpl::ClearBufferSubData(
GL_SHADER_STORAGE_BUFFER, GL_R8, 0, irisStaticSsboSize, GL_RED, GL_BYTE, &zero);
Vector<Uint8> actual(irisStaticSsboSize);
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
ASSERT_NE(bufferObject, nullptr);
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size());
EXPECT_EQ(actual, Vector<Uint8>(irisStaticSsboSize, 0));
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
DrainPendingGlErrors();
}
TEST_F(BufferTest, ClearBufferDataConvertsOneClientPixelBeforeRepeatingIt) {
GLuint buffer = 0;
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, buffer);
Vector<Uint32> initial(4, 0u);
MobileGL::MG_Impl::GLImpl::BufferData(GL_ARRAY_BUFFER, initial.size() * sizeof(Uint32), initial.data(),
GL_STATIC_DRAW);
const Uint8 value = 0xAB;
MobileGL::MG_Impl::GLImpl::ClearBufferData(
GL_ARRAY_BUFFER, GL_R32UI, GL_RED_INTEGER, GL_UNSIGNED_BYTE, &value);
Vector<Uint32> actual(initial.size());
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
ASSERT_NE(bufferObject, nullptr);
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size() * sizeof(Uint32));
EXPECT_EQ(actual, Vector<Uint32>(initial.size(), value));
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, 0);
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
DrainPendingGlErrors();
}
TEST_F(BufferTest, ClearBufferSubDataRejectsUnboundTarget) {
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
const GLbyte zero = 0;
MobileGL::MG_Impl::GLImpl::ClearBufferSubData(
GL_SHADER_STORAGE_BUFFER, GL_R8, 0, 1, GL_RED, GL_BYTE, &zero);
ExpectSingleGlError(GL_INVALID_OPERATION);
}
TEST_F(BufferTest, ClearBufferDataRejectsInvalidPixelFormatTypePairs) {
GLuint buffer = 0;
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, buffer);
const Vector<Uint8> initial{0x7F, 0x7F};
MobileGL::MG_Impl::GLImpl::BufferData(GL_ARRAY_BUFFER, initial.size(), initial.data(), GL_STATIC_DRAW);
const Uint16 packed = 0;
MobileGL::MG_Impl::GLImpl::ClearBufferData(
GL_ARRAY_BUFFER, GL_R16, GL_RED, GL_UNSIGNED_SHORT_5_6_5, &packed);
ExpectSingleGlError(GL_INVALID_VALUE);
MobileGL::MG_Impl::GLImpl::ClearBufferData(
GL_ARRAY_BUFFER, GL_R16, GL_RED, GL_UNSIGNED_SHORT_5_6_5, nullptr);
ExpectSingleGlError(GL_INVALID_VALUE);
Vector<Uint8> actual(initial.size());
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
ASSERT_NE(bufferObject, nullptr);
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size());
EXPECT_EQ(actual, initial);
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, 0);
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
DrainPendingGlErrors();
}
// GL 4.6 core 6.5: glBufferSubData fails only when the written range OVERLAPS the mapped range.
+1
View File
@@ -78,6 +78,7 @@ add_subdirectory(Query)
add_subdirectory(Pipeline)
add_subdirectory(ShaderTranspiler)
add_subdirectory(Util)
add_subdirectory(SelfTest)
# The DirectGLES post-transpile ESSL passes are pure String -> String, so unlike the
# DirectVulkan suite below this one needs no device and always builds.
add_subdirectory(Backend/DirectGLES)
+1
View File
@@ -4,6 +4,7 @@ add_executable(
PipelineQuirkTest
PipelineQuirkTest.cpp
PassthroughTessControlTest.cpp
ViewportIndexReflectionTest.cpp
)
target_include_directories(PipelineQuirkTest PRIVATE
@@ -0,0 +1,183 @@
// MobileGL - MobileGL/MG_Test/Pipeline/ViewportIndexReflectionTest.cpp
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// ProgramFactory::ReflectedWritesViewportIndexBuiltin is the switch that decides whether a
// DirectVulkan pipeline declares one viewport or all sixteen. Getting it wrong is silent in both
// directions and neither direction is caught by a state test:
//
// - a false NEGATIVE collapses every gl_ViewportIndex onto viewport 0, which is precisely the
// bug the multi-viewport work exists to fix and which a set/get round trip cannot see;
// - a false POSITIVE widens viewportCount for an ordinary Minecraft shader, costing a longer
// vkCmdSetViewport per state change and, on a tiler, possibly a hardware fast path.
//
// So this compiles REAL GLSL through the same glslang path the renderer uses and reflects the
// SPIR-V that comes out, rather than asserting against hand-assembled words: what has to hold is
// that the detector agrees with what glslang actually emits for a shader that writes the builtin,
// including the stage-by-stage question of WHERE it may be written (GL 4.1 allows the geometry
// stage; ARB_shader_viewport_layer_array adds vertex and tessellation evaluation).
//
// The end-to-end claim - that a detected writer really does route pixels to its own viewport -
// lives in MG_IntegrationTest/Scenarios/ViewportArrayScenario.cpp.
#include <gtest/gtest.h>
#include <string>
#include <vector>
#include "Includes.h"
#include "Init.h"
#include <MG_Backend/DirectVulkan/Renderer/ProgramFactory.h>
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/Types.h>
#include <spirv_reflect.h>
using namespace MobileGL;
using MobileGL::MG_Backend::DirectVulkan::ProgramFactory;
using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler;
namespace {
Vector<Uint32> CompileToSpirv(GLenum stage, const String& source) {
using namespace MG_Util::ShaderTranspiler;
ShaderAttrib shaderAttrib{.shaderType = stage, .sourceStr = source};
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
EXPECT_TRUE(shaderResult) << (shaderResult ? String{} : shaderResult.error().log);
if (!shaderResult) return {};
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
EXPECT_TRUE(programResult) << (programResult ? String{} : programResult.error().log);
if (!programResult) return {};
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {stage}, .program = *programResult.value()};
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
EXPECT_TRUE(binaryResult) << (binaryResult ? String{} : binaryResult.error().log);
if (!binaryResult || binaryResult->empty()) return {};
return binaryResult->front();
}
// Owns the reflection module so a failing EXPECT cannot leak it.
class ReflectModule {
public:
explicit ReflectModule(const Vector<Uint32>& spirv) {
if (spirv.empty()) return;
m_created = spvReflectCreateShaderModule(spirv.size() * sizeof(Uint32), spirv.data(), &m_module) ==
SPV_REFLECT_RESULT_SUCCESS;
}
~ReflectModule() {
if (m_created) spvReflectDestroyShaderModule(&m_module);
}
ReflectModule(const ReflectModule&) = delete;
ReflectModule& operator=(const ReflectModule&) = delete;
Bool Created() const { return m_created; }
const SpvReflectShaderModule& Get() const { return m_module; }
private:
SpvReflectShaderModule m_module{};
Bool m_created = false;
};
class ViewportIndexReflectionTest : public ::testing::Test {
protected:
void SetUp() override { MobileGL::Initialize(); }
};
const char* const kGeometryWritesViewportIndex = R"(#version 410 core
layout(points, invocations = 16) in;
layout(triangle_strip, max_vertices = 4) out;
void main() {
gl_ViewportIndex = gl_InvocationID;
gl_Position = vec4(-1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 1.0, 0.0, 1.0); EmitVertex();
EndPrimitive();
}
)";
// Same stage, same shape, writing gl_Layer INSTEAD. Layered rendering and viewport routing
// are different features and the detector must not confuse them: a Minecraft-style cubemap
// pass writes gl_Layer and must keep the one-viewport pipeline.
const char* const kGeometryWritesLayerOnly = R"(#version 410 core
layout(points, invocations = 6) in;
layout(triangle_strip, max_vertices = 4) out;
void main() {
gl_Layer = gl_InvocationID;
gl_Position = vec4(-1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 1.0, 0.0, 1.0); EmitVertex();
EndPrimitive();
}
)";
const char* const kPlainGeometry = R"(#version 410 core
layout(points, invocations = 1) in;
layout(triangle_strip, max_vertices = 4) out;
void main() {
gl_Position = vec4(-1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, -1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4(-1.0, 1.0, 0.0, 1.0); EmitVertex();
gl_Position = vec4( 1.0, 1.0, 0.0, 1.0); EmitVertex();
EndPrimitive();
}
)";
const char* const kPlainVertex = R"(#version 410 core
void main() { gl_Position = vec4(0.0, 0.0, 0.0, 1.0); }
)";
const char* const kPlainFragment = R"(#version 410 core
layout(location = 0) out vec4 fragColor;
void main() { fragColor = vec4(1.0); }
)";
TEST_F(ViewportIndexReflectionTest, TrueForAGeometryShaderThatAssignsViewportIndex) {
const ReflectModule module(CompileToSpirv(GL_GEOMETRY_SHADER, kGeometryWritesViewportIndex));
ASSERT_TRUE(module.Created());
EXPECT_TRUE(ProgramFactory::ReflectedWritesViewportIndexBuiltin(module.Get()))
<< "a shader that assigns gl_ViewportIndex must get a multi-viewport pipeline; missing it is what "
"collapses every index onto viewport 0";
}
TEST_F(ViewportIndexReflectionTest, FalseForAGeometryShaderThatOnlyAssignsLayer) {
const ReflectModule module(CompileToSpirv(GL_GEOMETRY_SHADER, kGeometryWritesLayerOnly));
ASSERT_TRUE(module.Created());
EXPECT_FALSE(ProgramFactory::ReflectedWritesViewportIndexBuiltin(module.Get()))
<< "gl_Layer is layered rendering, not viewport routing; widening viewportCount for it costs the "
"single-viewport fast path for nothing";
}
TEST_F(ViewportIndexReflectionTest, FalseForAPlainGeometryShader) {
const ReflectModule module(CompileToSpirv(GL_GEOMETRY_SHADER, kPlainGeometry));
ASSERT_TRUE(module.Created());
EXPECT_FALSE(ProgramFactory::ReflectedWritesViewportIndexBuiltin(module.Get()));
}
TEST_F(ViewportIndexReflectionTest, FalseForTheOrdinaryVertexAndFragmentStages) {
// The shape every real application ships: neither stage may widen the pipeline.
const ReflectModule vertexModule(CompileToSpirv(GL_VERTEX_SHADER, kPlainVertex));
ASSERT_TRUE(vertexModule.Created());
EXPECT_FALSE(ProgramFactory::ReflectedWritesViewportIndexBuiltin(vertexModule.Get()));
const ReflectModule fragmentModule(CompileToSpirv(GL_FRAGMENT_SHADER, kPlainFragment));
ASSERT_TRUE(fragmentModule.Created());
EXPECT_FALSE(ProgramFactory::ReflectedWritesViewportIndexBuiltin(fragmentModule.Get()));
}
TEST_F(ViewportIndexReflectionTest, FalseForAnEmptyModuleWithoutDereferencing) {
// A default-constructed module has no entry points. The scan runs on every link, so it
// must survive a reflection that never got built rather than walk a null array.
SpvReflectShaderModule emptyModule{};
EXPECT_FALSE(ProgramFactory::ReflectedWritesViewportIndexBuiltin(emptyModule));
}
} // namespace
@@ -516,17 +516,17 @@ TEST_F(ParallelShaderCompileTest, MaxShaderCompilerThreadsIgnoresTheCurrentBudge
TEST_F(ParallelShaderCompileTest, BothBackendsAdvertiseTheExtensionIffAsyncIsEnabled) {
{
const AsyncModeScope async(true);
EXPECT_TRUE(Advertises(MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false),
EXPECT_TRUE(Advertises(MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false, false, false),
E_GL_KHR_parallel_shader_compile));
EXPECT_TRUE(Advertises(MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false),
EXPECT_TRUE(Advertises(MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false, false),
E_GL_KHR_parallel_shader_compile));
}
{
const AsyncModeScope async(false);
EXPECT_FALSE(Advertises(MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false),
EXPECT_FALSE(Advertises(MG_Backend::DirectGLES::BuildAdvertisedExtensions(false, false, false, false),
E_GL_KHR_parallel_shader_compile))
<< "MOBILEGL_ASYNC_SHADER_COMPILE=0 must withdraw the extension, not only the threading";
EXPECT_FALSE(Advertises(MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false),
EXPECT_FALSE(Advertises(MG_Backend::DirectVulkan::BuildAdvertisedExtensions(false, false, false, false),
E_GL_KHR_parallel_shader_compile))
<< "MOBILEGL_ASYNC_SHADER_COMPILE=0 must withdraw the extension, not only the threading";
}
+4 -1
View File
@@ -2692,11 +2692,13 @@ out vec4 fragColor;
float fma
(float a, float b, float c) { return a * b + c; }
float sinh(float x, float y) { return x * y; }
float length_squared(vec3 value) { return dot(value, value); }
float round(float x) { return floor(x + 0.5); }
float min3(float a, float b, float c) { return min(min(a, b), c); }
void main() {
fragColor = vec4(fma(0.1, 0.2, 0.3), sinh(0.4, 2.0), round(1.25), min3(0.1, 0.2, 0.3));
fragColor = vec4(fma(0.1, 0.2, 0.3), sinh(0.4, 2.0), round(1.25),
min3(0.1, 0.2, 0.3) + length_squared(vec3(0.1, 0.2, 0.3)));
}
)";
GLuint vs = CompileShaderChecked(GL_VERTEX_SHADER, vsSource);
@@ -2707,6 +2709,7 @@ void main() {
if (essl.find("fragColor") == String::npos) continue; // fragment module only
EXPECT_NE(essl.find("mg_fma("), String::npos) << essl;
EXPECT_NE(essl.find("mg_sinh("), String::npos) << essl;
EXPECT_NE(essl.find("mg_length_squared("), String::npos) << essl;
EXPECT_NE(essl.find("mg_round("), String::npos) << essl;
EXPECT_NE(essl.find("mg_min3("), String::npos) << essl;
EXPECT_EQ(essl.find("float fma("), String::npos) << essl;
+38 -200
View File
@@ -52,20 +52,24 @@ TEST_F(ProgramUtilTest, RenameSamplerFunctionParameterInSpirvPass) {
OpEntryPoint Fragment %main "main" %outColor
OpExecutionMode %main OriginUpperLeft
OpName %globalSampler "sampler"
OpName %globalNew "new"
OpName %paramSampler "sampler"
OpName %paramNew "new"
OpName %main "main"
OpDecorate %outColor Location 0
%void = OpTypeVoid
%float = OpTypeFloat 32
%v4float = OpTypeVector %float 4
%mainFn = OpTypeFunction %void
%paramFn = OpTypeFunction %void %float
%paramFn = OpTypeFunction %void %float %float
%outV4Ptr = OpTypePointer Output %v4float
%privatePtr = OpTypePointer Private %float
%outColor = OpVariable %outV4Ptr Output
%globalSampler = OpVariable %privatePtr Private
%globalNew = OpVariable %privatePtr Private
%helper = OpFunction %void None %paramFn
%paramSampler = OpFunctionParameter %float
%paramNew = OpFunctionParameter %float
%helperBody = OpLabel
OpReturn
OpFunctionEnd
@@ -91,6 +95,7 @@ TEST_F(ProgramUtilTest, RenameSamplerFunctionParameterInSpirvPass) {
ASSERT_TRUE(tools.Disassemble(outputBinary, &outputText));
EXPECT_NE(outputText.find("\"MGL_COMPAT_sampler\""), String::npos);
EXPECT_NE(outputText.find("\"MGL_COMPAT_new\""), String::npos);
SizeT exactSamplerNameCount = 0;
SizeT searchOffset = 0;
@@ -99,6 +104,14 @@ TEST_F(ProgramUtilTest, RenameSamplerFunctionParameterInSpirvPass) {
searchOffset += std::strlen("\"sampler\"");
}
EXPECT_EQ(exactSamplerNameCount, 1u);
SizeT exactNewNameCount = 0;
searchOffset = 0;
while ((searchOffset = outputText.find("\"new\"", searchOffset)) != String::npos) {
++exactNewNameCount;
searchOffset += std::strlen("\"new\"");
}
EXPECT_EQ(exactNewNameCount, 1u);
}
TEST_F(ProgramUtilTest, UnformattedFloatStorageImagesKeepIntegerAtomicImagesTyped) {
@@ -2060,153 +2073,6 @@ void main() {
EXPECT_NE(source.find("layout(std140) uniform Blk"), String::npos);
}
namespace {
String MakeLinearSubgroupPrefixScanShader() {
return R"(#version 460 core
#extension GL_KHR_shader_subgroup_arithmetic : enable
layout(local_size_x = 1024) in;
shared float prefixSumCache[64];
layout(std430, binding = 0) writeonly buffer OutputBuffer {
float outputValues[];
};
void main() {
float importance = 1.0f;
float prefixSum = subgroupInclusiveAdd(importance);
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = prefixSum;
barrier();
uint loopLength = uint(findMSB(gl_NumSubgroups));
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
for (uint i = 0; i < loopLength; i++) {
if ((gl_SubgroupID & (1u << i)) > 0u) {
prefixSum += prefixSumCache[(gl_SubgroupID >> i << i) - 1u];
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = prefixSum;
}
barrier();
}
if (gl_LocalInvocationID.x == uint(1024 - 1)) prefixSumCache[0] = prefixSum;
barrier();
float sum = prefixSumCache[0];
float warp = (prefixSum - importance) / sum - float(gl_LocalInvocationID.x + 1u) / float(1024);
outputValues[gl_GlobalInvocationID.x] = warp;
}
)";
}
} // namespace
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanUsesSharedMemoryAndProducesValidSpirv) {
using namespace MG_Util::ShaderTranspiler;
String source = MakeLinearSubgroupPrefixScanShader();
ASSERT_TRUE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
EXPECT_NE(source.find("shared float prefixSumCache[1024]"), String::npos) << source;
EXPECT_NE(source.find("mglVirtualSubgroupInvocation"), String::npos) << source;
EXPECT_NE(source.find("for (uint mglPrefixLane"), String::npos) << source;
EXPECT_EQ(source.find("subgroupInclusiveAdd"), String::npos) << source;
EXPECT_EQ(source.find("gl_Subgroup"), String::npos) << source;
const String onceRewritten = source;
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
EXPECT_EQ(source, onceRewritten);
ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source};
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
ASSERT_TRUE(shaderResult) << shaderResult.error().log << "\nsource:\n" << source;
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
ASSERT_TRUE(programResult) << programResult.error().log;
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()};
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
ASSERT_TRUE(binaryResult) << binaryResult.error().log;
ASSERT_EQ(binaryResult->size(), 1u);
String validationDiagnostics;
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
tools.SetMessageConsumer([&](spv_message_level_t, const char*, const spv_position_t&, const char* message) {
validationDiagnostics += message;
validationDiagnostics += '\n';
});
EXPECT_TRUE(tools.Validate(binaryResult->front())) << validationDiagnostics;
String spirvText;
ASSERT_TRUE(tools.Disassemble(binaryResult->front(), &spirvText));
EXPECT_EQ(spirvText.find("OpGroupNonUniform"), String::npos) << spirvText;
}
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsOtherStagesAndSubgroupWidths) {
using namespace MG_Util::ShaderTranspiler;
const String original = MakeLinearSubgroupPrefixScanShader();
for (const auto& [stage, subgroupSize] :
{std::pair{ShaderStage::Compute, Uint32{32}}, std::pair{ShaderStage::Fragment, Uint32{64}},
std::pair{ShaderStage::Compute, Uint32{96}}}) {
String source = original;
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(stage, subgroupSize, source));
EXPECT_EQ(source, original);
}
}
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsPartialOrUnsafeTemplateMatches) {
using namespace MG_Util::ShaderTranspiler;
const auto expectUnchanged = [](String source) {
const String original = source;
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
EXPECT_EQ(source, original);
};
String wrongLocalSize = MakeLinearSubgroupPrefixScanShader();
wrongLocalSize.replace(wrongLocalSize.find("local_size_x = 1024"), std::strlen("local_size_x = 1024"),
"local_size_x = 512");
expectUnchanged(std::move(wrongLocalSize));
String cacheHasAnotherUse = MakeLinearSubgroupPrefixScanShader();
cacheHasAnotherUse.insert(cacheHasAnotherUse.find("float importance"), "prefixSumCache[0] = 0.0f;\n ");
expectUnchanged(std::move(cacheHasAnotherUse));
String extraSubgroupBuiltin = MakeLinearSubgroupPrefixScanShader();
extraSubgroupBuiltin.insert(extraSubgroupBuiltin.find("float importance"),
"uvec4 extraMask = gl_SubgroupEqMask;\n ");
expectUnchanged(std::move(extraSubgroupBuiltin));
String alteredBarrier = MakeLinearSubgroupPrefixScanShader();
alteredBarrier.replace(alteredBarrier.find("barrier();"), std::strlen("barrier();"), "memoryBarrierShared();");
expectUnchanged(std::move(alteredBarrier));
String nestedScan = MakeLinearSubgroupPrefixScanShader();
nestedScan.insert(nestedScan.find("float prefixSum ="), "if (importance > 0.0f) {\n ");
const SizeT consumerEnd = nestedScan.find(';', nestedScan.find("float warp ="));
ASSERT_NE(consumerEnd, String::npos);
nestedScan.insert(consumerEnd + 1, "\n }");
expectUnchanged(std::move(nestedScan));
// ARB/NV spellings of lane-width-sensitive builtins must block the rewrite exactly
// like their KHR counterparts.
String arbSubgroupBuiltin = MakeLinearSubgroupPrefixScanShader();
arbSubgroupBuiltin.insert(arbSubgroupBuiltin.find("float importance"),
"uint arbLane = gl_SubGroupInvocationARB;\n ");
expectUnchanged(std::move(arbSubgroupBuiltin));
String arbBallotCall = MakeLinearSubgroupPrefixScanShader();
arbBallotCall.insert(arbBallotCall.find("float importance"),
"uint64_t arbMask = ballotARB(true);\n ");
expectUnchanged(std::move(arbBallotCall));
String nvWarpBuiltin = MakeLinearSubgroupPrefixScanShader();
nvWarpBuiltin.insert(nvWarpBuiltin.find("float importance"),
"uint warpSize = gl_WarpSizeNV;\n ");
expectUnchanged(std::move(nvWarpBuiltin));
String nvShuffleCall = MakeLinearSubgroupPrefixScanShader();
nvShuffleCall.insert(nvShuffleCall.find("float importance"),
"float other = shuffleNV(1.0f, 0u, 32u);\n ");
expectUnchanged(std::move(nvShuffleCall));
}
// The LEXICAL half must fire at the source level (before the parse) for the
// preempt-list names - the end-to-end ESSL tests cannot tell which half did the
// rename, and for these names the parse would fail without the source rewrite.
@@ -2435,9 +2301,6 @@ TEST_F(ProgramUtilTest, CompileEnvFingerprintTracksEveryInput) {
otherExtensions.advertisedExtensions.push_back(MobileGL::E_GL_ARB_gpu_shader_int64);
EXPECT_NE(ComputeCompileEnvFingerprint(otherExtensions), baseline);
CompileEnv otherQuirk = base;
otherQuirk.subgroupPrefixScanQuirk = MobileGL::MG_Config::QuirkOverride::ForceOn;
EXPECT_NE(ComputeCompileEnvFingerprint(otherQuirk), baseline);
}
// The no-backend fallback must stay exactly what the pipeline used to do inline:
@@ -2836,16 +2699,6 @@ vec4 helperTint() { return vec4(1.0); }
return binaryResult->front();
}
struct SpirvValidationScope {
bool previous;
explicit SpirvValidationScope(bool enabled)
: previous(MG_Util::ShaderTranspiler::ShaderCompiler::SpirvValidationEnabled()) {
MG_Util::ShaderTranspiler::ShaderCompiler::SetSpirvValidationEnabled(enabled);
}
~SpirvValidationScope() {
MG_Util::ShaderTranspiler::ShaderCompiler::SetSpirvValidationEnabled(previous);
}
};
} // namespace
TEST_F(ProgramUtilTest, DeadPrivateChainVertexInputIsEliminatedFromOptimizedBinary) {
@@ -2867,7 +2720,7 @@ TEST_F(ProgramUtilTest, DeadPrivateChainVertexInputIsEliminatedFromOptimizedBina
<< "entry-point-with-calls shape it exists for";
Vector<Uint32> optimized;
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
const SpirvVariableCensus after = TakeVariableCensus(optimized);
EXPECT_EQ(after.inputCount, 1u)
@@ -2905,7 +2758,7 @@ void main() {
ASSERT_GE(before.outputCount, 3u);
Vector<Uint32> optimized;
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
EXPECT_EQ(TakeVariableCensus(optimized).outputCount, before.outputCount)
<< "a declared-but-unwritten output was deleted; a fragment stage reading it now "
<< "fails to link (ES) or breaks the Vulkan stage interface";
@@ -2964,17 +2817,15 @@ void main() {
// succeeds - fail-open call sites downstream must not see a different world),
// and the failure latch is the signal. This is the catch that took a device
// bisect to find when the validator was off everywhere.
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
EXPECT_GT(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
<< "an invalid optimized module must bump the validation-failure latch";
}
{
// The shipping configuration: same result, no validation, latch untouched.
SpirvValidationScope validationOff(false);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, false));
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore);
}
}
@@ -3044,10 +2895,9 @@ void main() {
ASSERT_FALSE(raw.empty());
ASSERT_GE(CountRectImageTypes(raw), 1u) << "glslang no longer emits Dim::Rect for sampler2DRect";
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> optimized;
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
EXPECT_EQ(CountRectImageTypes(optimized), 0u);
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
<< "a rectangle module must leave the chain valid, not latched as a failure";
@@ -3072,10 +2922,9 @@ void main() {
ASSERT_TRUE(AnyLocationOnUniformStorage(raw))
<< "glslang no longer keeps the explicit uniform location; the strip pass may be obsolete";
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> optimized;
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
EXPECT_FALSE(AnyLocationOnUniformStorage(optimized));
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
<< "the stripped module must validate clean";
@@ -3172,11 +3021,10 @@ void main() {
<< "the fixture must reproduce the defect before the fix is asked to remove it:\n"
<< DisassembleSpirv(raw);
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> legalized;
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
ASSERT_FALSE(legalized.empty());
const String disassembly = DisassembleSpirv(legalized);
@@ -3213,11 +3061,10 @@ void main() {
ASSERT_TRUE(LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(raw))
<< DisassembleSpirv(raw);
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> legalized;
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
ASSERT_FALSE(legalized.empty());
const String disassembly = DisassembleSpirv(legalized);
@@ -3257,11 +3104,10 @@ void main() {
ASSERT_FALSE(raw.empty());
ASSERT_TRUE(LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(raw));
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> legalized;
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
ASSERT_FALSE(legalized.empty());
const String disassembly = DisassembleSpirv(legalized);
@@ -3300,7 +3146,7 @@ void main() {
ASSERT_FALSE(LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(raw));
Vector<Uint32> legalized;
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
EXPECT_EQ(legalized, raw) << "the module must not be rewritten - not even re-serialized - when "
"nothing indexes a fragment output dynamically";
}
@@ -3550,11 +3396,10 @@ TEST_F(ProgramUtilTest, Lower1DArrayImagesRewritesTheTypeAndWidensTheCoordinate)
ASSERT_EQ(Count1DArrayStorageImageTypes(spirv), 1u)
<< "the shared chain must leave the 1D-array image for this pass to handle";
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> lowered;
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
ASSERT_FALSE(lowered.empty());
EXPECT_EQ(Count1DArrayStorageImageTypes(lowered), 0u)
@@ -3602,11 +3447,10 @@ void main() { ssb.sum = imageLoad(i0, ivec2(2, 3)).r + imageLoad(i1, ivec3(1, 1,
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv));
ASSERT_EQ(Count1DArrayStorageImageTypes(spirv), 1u);
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> lowered;
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
ASSERT_FALSE(lowered.empty());
EXPECT_EQ(Count1DArrayStorageImageTypes(lowered), 0u) << DisassembleSpirv(lowered);
@@ -3636,7 +3480,7 @@ void main() { ssb.sum = imageLoad(i0, 2).r; }
ASSERT_FALSE(spirv.empty());
Vector<Uint32> lowered;
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
EXPECT_EQ(lowered, spirv) << "a non-arrayed 1D storage image must pass through byte for byte";
const String essl = DecompileToEssl(lowered);
@@ -3660,7 +3504,7 @@ void main() { fragColor = texture(uTex, vUv); }
ASSERT_FALSE(spirv.empty());
Vector<Uint32> lowered;
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
EXPECT_EQ(lowered, spirv) << "a sampled 1D-array image must pass through byte for byte";
}
@@ -3684,7 +3528,7 @@ void main() { ssb.sum = uint(imageSize(i0).x) + imageLoad(i0, ivec2(0, 0)).r; }
<< "the fixture must contain the shape the pass declines";
Vector<Uint32> lowered;
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
EXPECT_EQ(lowered, spirv) << "a declined module must be handed back untouched, not partly rewritten";
EXPECT_EQ(Count1DArrayStorageImageTypes(lowered), 1u)
<< "declining means the 1D-array type is still there for the driver to reject";
@@ -3735,11 +3579,10 @@ void main() { imageStore(uni_image, ivec2(gl_GlobalInvocationID.xy), uvec4(15u,
// Precondition: SPIRV-Cross prints no format for it, which is the ESSL the driver refuses.
EXPECT_EQ(DecompileToEssl(spirv).find("r32ui"), String::npos);
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
ASSERT_FALSE(baked.empty());
EXPECT_FALSE(ShaderCompiler::DeclaresFormatlessStorageImage(baked)) << DisassembleSpirv(baked);
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
@@ -3800,7 +3643,7 @@ void main() { imageStore(uni_image, ivec2(0), uvec4(1u)); }
Vector<Uint32> baked;
// Even asked to, with a format of the right component class.
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
EXPECT_EQ(baked, spirv) << "a module with nothing format-less must pass through byte for byte";
EXPECT_NE(DecompileToEssl(baked).find("rgba32ui"), String::npos);
}
@@ -3822,11 +3665,10 @@ void main() { writeIt(uni_image); }
ASSERT_FALSE(spirv.empty());
ASSERT_TRUE(ShaderCompiler::DeclaresFormatlessStorageImage(spirv));
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
EXPECT_EQ(baked, spirv) << "a shape the retype cannot follow must leave the module untouched, "
"not partly rewritten:\n"
<< DisassembleSpirv(baked);
@@ -3848,18 +3690,17 @@ void main() { imageStore(uni_image, ivec2(0), vec4(1.0)); }
GL_COMPUTE_SHADER);
ASSERT_FALSE(spirv.empty());
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
EXPECT_EQ(baked, spirv) << "a declined module must be handed back untouched, not partly rewritten";
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore);
// ...and the same image with a float bind format is baked, so the decline above is about the
// class and not about the pass refusing float images.
Vector<Uint32> bakedFloat;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32f}}, bakedFloat));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32f}}, bakedFloat, true));
EXPECT_NE(DecompileToEssl(bakedFloat).find("r32f"), String::npos) << DisassembleSpirv(bakedFloat);
}
@@ -3883,12 +3724,11 @@ void main() {
ASSERT_EQ(CountSpirvOpcode(DisassembleSpirv(spirv), "OpTypeImage"), 1u)
<< "the fixture must have the two images sharing one type:\n" << DisassembleSpirv(spirv);
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(
spirv, {{"imgA", kGlR32ui}, {"imgB", kGlRgba32ui}}, baked));
spirv, {{"imgA", kGlR32ui}, {"imgB", kGlRgba32ui}}, baked, true));
ASSERT_FALSE(baked.empty());
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
<< "splitting the shared type must not leave a dangling or duplicate declaration:\n"
@@ -3921,11 +3761,10 @@ void main() {
ASSERT_EQ(CountSpirvOpcode(DisassembleSpirv(spirv), "OpTypeImage"), 2u)
<< "the fixture needs one Unknown-format and one r32ui image type:\n" << DisassembleSpirv(spirv);
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"formatless", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"formatless", kGlR32ui}}, baked, true));
ASSERT_FALSE(baked.empty());
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
<< "the baked image collided with the module's own r32ui image and left a duplicate type:\n"
@@ -3950,11 +3789,10 @@ void main() {
ASSERT_FALSE(spirv.empty());
ASSERT_TRUE(ShaderCompiler::DeclaresFormatlessStorageImage(spirv));
SpirvValidationScope validationOn(true);
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"imgs", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"imgs", kGlR32ui}}, baked, true));
ASSERT_FALSE(baked.empty());
EXPECT_FALSE(ShaderCompiler::DeclaresFormatlessStorageImage(baked)) << DisassembleSpirv(baked);
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
@@ -3980,7 +3818,7 @@ void main() { fragColor = texture(uni_sampler, vUv); }
<< "a sampled image must not read as a format-less STORAGE image:\n" << DisassembleSpirv(spirv);
Vector<Uint32> baked;
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_sampler", kGlR32ui}}, baked));
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_sampler", kGlR32ui}}, baked, true));
EXPECT_EQ(baked, spirv) << "a sampled image must pass through byte for byte";
}
+81
View File
@@ -31,6 +31,7 @@
#include <MG_Backend/DirectVulkan/Renderer/VulkanRenderer.h>
#include <MG_Util/Math/HalfFloat.h>
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/CompileEnv.h>
#include <MG_Util/ShaderTranspiler/ShaderSourceProcessor.h>
#include <MG_Util/Debug/Log.h>
#include <MG_Util/Types.h>
@@ -710,6 +711,37 @@ TEST(DirectVulkanSanity, AdvertisesSubgroupOnlyWhenVulkanReportsUsableSupport) {
EXPECT_TRUE(backend.GetDynamicParameters().SubgroupQuadOperationsInAllStages);
}
TEST(DirectVulkanSanity, CapabilityRefreshInvalidatesTheCachedCompileEnvironment) {
using namespace MobileGL;
auto previousContext = Move(MG_State::pGLContext);
auto previousBackend = Move(MG_Backend::pActiveBackendObject);
MG_State::pGLContext = MakeUnique<MG_State::GLState::GLContext>();
auto backend = MakeUnique<MG_Backend::DirectVulkan::BackendObject_DirectVulkan>();
auto* backendPtr = backend.get();
MG_Backend::pActiveBackendObject = Move(backend);
const auto before = MG_State::pGLContext->GetCompileEnv();
EXPECT_EQ(before->params.SubgroupSize, 0u);
MG_External::VulkanCapabilities caps;
caps.SupportsShaderSubgroup = true;
caps.SubgroupSize = 8;
caps.SubgroupSupportedStages = VK_SHADER_STAGE_COMPUTE_BIT;
caps.SubgroupSupportedOperations = VK_SUBGROUP_FEATURE_BASIC_BIT | VK_SUBGROUP_FEATURE_ARITHMETIC_BIT;
backendPtr->ApplyVulkanCapabilitiesForTesting(caps);
const auto after = MG_State::pGLContext->GetCompileEnv();
EXPECT_NE(after.get(), before.get());
EXPECT_NE(after->fingerprint, before->fingerprint);
EXPECT_EQ(after->backend, BackendType::DirectVulkan);
EXPECT_EQ(after->params.SubgroupSize, 8u);
MG_Backend::pActiveBackendObject = Move(previousBackend);
MG_State::pGLContext = Move(previousContext);
}
TEST(DirectVulkanSanity, KeepsOptionalGpuShaderInt64BranchForVoxyQuadDecode) {
using namespace MobileGL;
@@ -936,6 +968,45 @@ TEST(DirectVulkanSanity, ReadbackUsesTheSourceFormatTexelSize) {
EXPECT_EQ(VulkanRenderer::GetReadbackTexelSize(VK_FORMAT_R32G32B32A32_SFLOAT), 16u);
}
TEST(DirectVulkanSanity, DefaultFramebufferQuarterTurnReadbackMapsRectAndPixels) {
using MobileGL::MG_Backend::DirectVulkan::VulkanRenderer;
using MobileGL::Uint8;
VkOffset2D offset{};
VkExtent2D copyExtent{};
ASSERT_TRUE(VulkanRenderer::MapDefaultFramebufferReadbackRect(
1, 0, 2, 1, VkExtent2D{2, 3}, VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR,
&offset, &copyExtent));
EXPECT_EQ(offset.x, 0);
EXPECT_EQ(offset.y, 1);
EXPECT_EQ(copyExtent.width, 1u);
EXPECT_EQ(copyExtent.height, 2u);
ASSERT_TRUE(VulkanRenderer::MapDefaultFramebufferReadbackRect(
1, 0, 2, 1, VkExtent2D{2, 3}, VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR,
&offset, &copyExtent));
EXPECT_EQ(offset.x, 1);
EXPECT_EQ(offset.y, 0);
EXPECT_EQ(copyExtent.width, 1u);
EXPECT_EQ(copyExtent.height, 2u);
// Logical GL rows, bottom to top, are abc / def. The display-oriented swapchain blocks are
// transposed in opposite directions for 90 and 270 degrees.
const Uint8 raw90[] = {'a', 'd', 'b', 'e', 'c', 'f'};
const Uint8 raw270[] = {'f', 'c', 'e', 'b', 'd', 'a'};
const Uint8 expected[] = {'a', 'b', 'c', 'd', 'e', 'f'};
Uint8 result[sizeof(expected)]{};
ASSERT_TRUE(VulkanRenderer::RemapDefaultFramebufferReadback(
raw90, 3, 2, VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR, 1, result));
EXPECT_TRUE(std::equal(std::begin(expected), std::end(expected), std::begin(result)));
std::fill(std::begin(result), std::end(result), 0);
ASSERT_TRUE(VulkanRenderer::RemapDefaultFramebufferReadback(
raw270, 3, 2, VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR, 1, result));
EXPECT_TRUE(std::equal(std::begin(expected), std::end(expected), std::begin(result)));
}
TEST(DirectVulkanSanity, ReadbackConvertsRgba8AndRgba16fPixels) {
using MobileGL::MG_Backend::DirectVulkan::VulkanRenderer;
using MobileGL::MG_Util::EncodeFloatToHalfBits;
@@ -2316,3 +2387,13 @@ TEST(DirectGLESTextureSync, UnitMemoRefusesToDriveATwinFromAnotherTexture) {
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, 0);
}
TEST(DirectVulkanSanity, GraphicsSamplerFeedbackOnlyAliasesWritableOverlappingMip) {
using MobileGL::MG_Backend::DirectVulkan::UniformManager;
EXPECT_TRUE(UniformManager::SamplerOverlapsWritableImageSubresource(1, 3, 2, GL_WRITE_ONLY));
EXPECT_TRUE(UniformManager::SamplerOverlapsWritableImageSubresource(1, 3, 3, GL_READ_WRITE));
EXPECT_FALSE(UniformManager::SamplerOverlapsWritableImageSubresource(1, 3, 2, GL_READ_ONLY));
EXPECT_FALSE(UniformManager::SamplerOverlapsWritableImageSubresource(1, 3, 0, GL_WRITE_ONLY));
EXPECT_FALSE(UniformManager::SamplerOverlapsWritableImageSubresource(1, 3, 4, GL_WRITE_ONLY));
}
+19
View File
@@ -0,0 +1,19 @@
# MobileGL - MobileGL/MG_Test/SelfTest/CMakeLists.txt
add_executable(
DriverPostProgram203WitnessTest
DriverPostProgram203WitnessTest.cpp
)
target_include_directories(DriverPostProgram203WitnessTest PRIVATE
${MGL_ROOT}/include
${MGL_ROOT}/MobileGL
)
target_link_libraries(DriverPostProgram203WitnessTest PRIVATE
GTest::gtest_main
${LINK_LIBRARIES}
)
include(GoogleTest)
gtest_discover_tests(DriverPostProgram203WitnessTest DISCOVERY_TIMEOUT 30 PROPERTIES LABELS unit)
@@ -0,0 +1,188 @@
// MobileGL - MobileGL/MG_Test/SelfTest/DriverPostProgram203WitnessTest.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include <gtest/gtest.h>
#include <string>
#include "MG_Util/SelfTest/DriverPostProgram203Witness.h"
namespace MobileGL::MG_Util::SelfTest {
namespace {
Program203WitnessOutput MakeValidWitness(std::uint32_t numSubgroups) {
Program203WitnessOutput output{};
output.magic = kProgram203WitnessMagic;
output.numSubgroups = numSubgroups;
output.loopLength = ComputeProgram203WitnessLoopLength(numSubgroups);
output.seenSubgroupMask =
numSubgroups == kProgram203WitnessMaxSubgroups ? 0xffffffffu : (1u << numSubgroups) - 1u;
// Valid test layouts use equal contiguous groups of the indexed
// 1..512 input. The compact witness only needs their independent sums.
const std::uint32_t subgroupSize = kProgram203WitnessInvocationCount / numSubgroups;
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
const std::uint32_t first = subgroup * subgroupSize + 1u;
const std::uint32_t last = first + subgroupSize - 1u;
output.lastLaneWriterCount[subgroup] = 1u;
output.indexedInputTotal[subgroup] = subgroupSize * (first + last) / 2u;
output.rawPrefix[subgroup] = {static_cast<float>(output.indexedInputTotal[subgroup]), 0.0f};
}
output.owner511 = {subgroupSize, numSubgroups, numSubgroups - 1u, subgroupSize - 1u};
auto cache = output.rawPrefix;
for (std::uint32_t scanStage = 0u; scanStage < output.loopLength; ++scanStage) {
auto cacheAfterStage = cache;
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
if ((subgroup & (1u << scanStage)) == 0u) continue;
const std::uint32_t sourceCacheIndex = (subgroup >> scanStage << scanStage) - 1u;
cacheAfterStage[subgroup].x += cache[sourceCacheIndex].x;
cacheAfterStage[subgroup].y += cache[sourceCacheIndex].y;
}
cache = cacheAfterStage;
output.scanCache[scanStage] = cache;
}
output.finalAverage = {256.5f, 0.0f};
return output;
}
Program203WitnessLimits MakeSufficientLimits() {
Program203WitnessLimits limits;
limits.computeStageSupported = true;
limits.basicSubgroupSupported = true;
limits.arithmeticSubgroupSupported = true;
limits.subgroupSize = 32u;
limits.maxComputeWorkGroupInvocations = kProgram203WitnessInvocationCount;
limits.maxComputeWorkGroupSize = {32u, 16u, 1u};
limits.maxComputeSharedMemorySize = kProgram203WitnessSharedMemoryBytes;
limits.maxPerStageDescriptorStorageBuffers = 1u;
limits.maxDescriptorSetStorageBuffers = 1u;
limits.maxBoundDescriptorSets = 1u;
limits.maxStorageBufferRange = sizeof(Program203WitnessOutput);
return limits;
}
} // namespace
TEST(DriverPostProgram203WitnessTest, ValidTwoSubgroupWitness) {
const Program203WitnessValidationResult validation = ValidateProgram203Witness(MakeValidWitness(2u));
ASSERT_TRUE(validation.ok) << validation.detail;
EXPECT_EQ(validation.detail, "N=2, owner511=id1/lane255, 2 scan stages, average=(256.5,0)");
}
TEST(DriverPostProgram203WitnessTest, ValidThirtyTwoSubgroupWitness) {
const Program203WitnessValidationResult validation = ValidateProgram203Witness(MakeValidWitness(32u));
ASSERT_TRUE(validation.ok) << validation.detail;
EXPECT_EQ(validation.detail, "N=32, owner511=id31/lane15, 6 scan stages, average=(256.5,0)");
}
TEST(DriverPostProgram203WitnessTest, RejectsNonuniformNumSubgroups) {
Program203WitnessOutput output = MakeValidWitness(16u);
output.topologyFlags |= Program203WitnessNonuniformNumSubgroups;
const Program203WitnessValidationResult validation = ValidateProgram203Witness(output);
EXPECT_FALSE(validation.ok);
EXPECT_EQ(validation.failure, Program203WitnessValidationFailure::Topology);
EXPECT_NE(validation.detail.find("gl_NumSubgroups differed"), std::string::npos);
}
TEST(DriverPostProgram203WitnessTest, RejectsMissingAndOutOfRangeSubgroupIds) {
Program203WitnessOutput missing = MakeValidWitness(16u);
missing.seenSubgroupMask &= ~(1u << 7u);
Program203WitnessValidationResult validation = ValidateProgram203Witness(missing);
EXPECT_FALSE(validation.ok);
EXPECT_NE(validation.detail.find("seen subgroup-ID mask"), std::string::npos);
Program203WitnessOutput outOfRange = MakeValidWitness(16u);
outOfRange.topologyFlags |= Program203WitnessInvalidSubgroupId;
validation = ValidateProgram203Witness(outOfRange);
EXPECT_FALSE(validation.ok);
EXPECT_NE(validation.detail.find("invalid gl_SubgroupID"), std::string::npos);
}
TEST(DriverPostProgram203WitnessTest, RejectsInvalidMultipleAndMissingLastLaneWriters) {
Program203WitnessOutput invalidLane = MakeValidWitness(16u);
invalidLane.topologyFlags |= Program203WitnessInvalidSubgroupLane;
Program203WitnessValidationResult validation = ValidateProgram203Witness(invalidLane);
EXPECT_FALSE(validation.ok);
EXPECT_NE(validation.detail.find("invalid subgroup lane"), std::string::npos);
Program203WitnessOutput multiple = MakeValidWitness(16u);
multiple.lastLaneWriterCount[4] = 2u;
validation = ValidateProgram203Witness(multiple);
EXPECT_FALSE(validation.ok);
EXPECT_NE(validation.detail.find("subgroup 4 has 2 source last-lane writers"), std::string::npos);
Program203WitnessOutput missing = MakeValidWitness(16u);
missing.lastLaneWriterCount[6] = 0u;
validation = ValidateProgram203Witness(missing);
EXPECT_FALSE(validation.ok);
EXPECT_NE(validation.detail.find("subgroup 6 has 0 source last-lane writers"), std::string::npos);
}
TEST(DriverPostProgram203WitnessTest, ReportsEarliestCorruptSourceScanStage) {
Program203WitnessOutput output = MakeValidWitness(32u);
output.scanCache[0][1].x += 1.0f;
output.scanCache[3][5].x += 1.0f;
Program203WitnessValidationResult validation = ValidateProgram203Witness(output);
EXPECT_FALSE(validation.ok);
EXPECT_EQ(validation.failure, Program203WitnessValidationFailure::SourceScan);
EXPECT_EQ(validation.scanStage, 0u);
EXPECT_NE(validation.detail.find("source scan stage 0, subgroup 1"), std::string::npos);
output = MakeValidWitness(32u);
output.scanCache[3][5].x += 1.0f;
validation = ValidateProgram203Witness(output);
EXPECT_FALSE(validation.ok);
EXPECT_EQ(validation.failure, Program203WitnessValidationFailure::SourceScan);
EXPECT_EQ(validation.scanStage, 3u);
EXPECT_NE(validation.detail.find("source scan stage 3, subgroup 5"), std::string::npos);
}
TEST(DriverPostProgram203WitnessTest, RejectsOwner511OutsideHighestFinalLane) {
Program203WitnessOutput output = MakeValidWitness(16u);
output.owner511.z = 14u;
const Program203WitnessValidationResult validation = ValidateProgram203Witness(output);
EXPECT_FALSE(validation.ok);
EXPECT_EQ(validation.failure, Program203WitnessValidationFailure::FinalOwner);
EXPECT_NE(validation.detail.find("not in the highest subgroup"), std::string::npos);
}
TEST(DriverPostProgram203WitnessTest, RejectsIncorrectVectorFinalAverage) {
Program203WitnessOutput output = MakeValidWitness(16u);
output.finalAverage.y = 1.0f;
const Program203WitnessValidationResult validation = ValidateProgram203Witness(output);
EXPECT_FALSE(validation.ok);
EXPECT_EQ(validation.failure, Program203WitnessValidationFailure::FinalAverage);
EXPECT_NE(validation.detail.find("final average"), std::string::npos);
}
TEST(DriverPostProgram203WitnessTest, MissingNativeFeatureIsTheOnlySkipCondition) {
for (const auto toggleMissingFeature : {0u, 1u, 2u}) {
Program203WitnessLimits limits = MakeSufficientLimits();
if (toggleMissingFeature == 0u) limits.computeStageSupported = false;
if (toggleMissingFeature == 1u) limits.basicSubgroupSupported = false;
if (toggleMissingFeature == 2u) limits.arithmeticSubgroupSupported = false;
const Program203WitnessEligibilityResult eligibility = EvaluateProgram203WitnessEligibility(limits);
EXPECT_EQ(eligibility.eligibility, Program203WitnessEligibility::SkipUnsupportedNativeFeatureSet)
<< eligibility.detail;
}
Program203WitnessLimits zeroSubgroupSize = MakeSufficientLimits();
zeroSubgroupSize.subgroupSize = 0u;
Program203WitnessEligibilityResult eligibility = EvaluateProgram203WitnessEligibility(zeroSubgroupSize);
EXPECT_EQ(eligibility.eligibility, Program203WitnessEligibility::FailInadequateLimits) << eligibility.detail;
Program203WitnessLimits limits = MakeSufficientLimits();
limits.maxComputeWorkGroupInvocations = 511u;
eligibility = EvaluateProgram203WitnessEligibility(limits);
EXPECT_EQ(eligibility.eligibility, Program203WitnessEligibility::FailInadequateLimits) << eligibility.detail;
limits = MakeSufficientLimits();
limits.maxStorageBufferRange = sizeof(Program203WitnessOutput) - 1u;
eligibility = EvaluateProgram203WitnessEligibility(limits);
EXPECT_EQ(eligibility.eligibility, Program203WitnessEligibility::FailInadequateLimits) << eligibility.detail;
}
} // namespace MobileGL::MG_Util::SelfTest
@@ -3,6 +3,7 @@ cmake_minimum_required(VERSION 3.14)
add_executable(
SpirvPassTest
SpirvPassTest.cpp
DeriveNumSubgroupsTest.cpp
DemoteFloat64Test.cpp
FlattenXfbInterfaceBlocksTest.cpp
)
@@ -154,7 +154,6 @@ class DemoteFloat64Test : public ::testing::Test {
protected:
void SetUp() override {
MobileGL::Initialize();
ShaderCompiler::SetSpirvValidationEnabled(true);
m_validationFailuresAtStart = ShaderCompiler::SpirvValidationFailureCount();
}
@@ -175,7 +174,7 @@ TEST_F(DemoteFloat64Test, DemotesEveryWidthAndDropsTheCapability) {
ASSERT_TRUE(DeclaresFloat64Capability(input));
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
EXPECT_EQ(CountFloatTypesOfWidth(output, 64), 0u) << Disassemble(output);
// And exactly one 32-bit float type survives: the merge has to happen, or spirv-val rejects
@@ -210,7 +209,7 @@ void main() {
<< Disassemble(input);
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
// std140 for the demoted members: float at 4, vec2 at 8, vec3 at 16 (aligned like a vec4),
// vec4 at 32, mat4 at 48 with a 16-byte column stride, the array at 112 with the std140
@@ -241,7 +240,7 @@ void main() {
EXPECT_EQ(CollectOffsetsOf(input, "Ssbo"), (Vector<Uint32>{0, 32, 64})) << Disassemble(input);
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
// std430, so the array packs at its element size rather than being rounded to 16: float at 0,
// vec4 at 16, float[4] at 32 with a 4-byte stride. A storage block must NOT come out std140,
@@ -273,7 +272,7 @@ void main() {
ASSERT_FALSE(before.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
// Only the block that actually narrowed is re-laid-out. Touching the other one would be
// churn at best, and a disagreement with glslang's own layout at worst.
@@ -287,7 +286,7 @@ TEST_F(DemoteFloat64Test, FoldsTheConversionsThatBecameIdentities) {
ASSERT_GT(CountFConverts(input), 0u) << "the fixture no longer converts between the two widths";
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
// SPIR-V requires the two component widths of an OpFConvert to differ, so every one of them
// has to be gone: both sides are 32 bits now.
@@ -307,7 +306,7 @@ void main() {
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
// A 64-bit literal is two words wide and a 32-bit one is a single word, so a constant left
// unconverted is not merely imprecise - it is an unparseable instruction. Disassembling both
@@ -326,7 +325,7 @@ void main() { gl_Position = inPos; }
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
// The pass reports SuccessWithoutChange here, and SPIRV-Tools asserts (in assert-enabled
// builds) that such a run round-trips byte-identically.
EXPECT_EQ(output, input);
@@ -351,7 +350,7 @@ void main() {
ASSERT_EQ(CountFloatTypesOfWidth(input, 64), 1u) << Disassemble(input);
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
EXPECT_EQ(output, input) << Disassemble(output);
EXPECT_TRUE(ShaderCompiler::ModuleDeclaresFloat64(output));
}
@@ -362,7 +361,7 @@ TEST_F(DemoteFloat64Test, ModuleDeclaresFloat64AnswersBothWays) {
EXPECT_TRUE(ShaderCompiler::ModuleDeclaresFloat64(wide));
Vector<Uint32> demoted;
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(wide, demoted));
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(wide, demoted, true));
EXPECT_FALSE(ShaderCompiler::ModuleDeclaresFloat64(demoted));
EXPECT_FALSE(ShaderCompiler::ModuleDeclaresFloat64({}));
@@ -375,7 +374,7 @@ TEST_F(DemoteFloat64Test, TheSharedChainDemotesToo) {
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output));
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
EXPECT_FALSE(ShaderCompiler::ModuleDeclaresFloat64(output)) << Disassemble(output);
}
@@ -459,7 +458,7 @@ TEST_P(DemoteFloat64EsslTest, TheDemotedModuleCanBeEmittedAsEssl) {
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output));
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
SpvcSession session(output, SessionUsageBit::Transpile);
spvc_compiler_options options;
@@ -482,7 +481,7 @@ TEST_P(DemoteFloat64EsslTest, TheDemotedModuleCanBeEmittedAsEssl) {
TEST_F(DemoteFloat64Test, RejectsGarbageInput) {
const Vector<Uint32> notSpirv{0xdeadbeefu, 0u, 0u, 0u, 0u};
Vector<Uint32> output;
EXPECT_FALSE(ShaderCompiler::DemoteFloat64ToFloat32(notSpirv, output));
EXPECT_FALSE(ShaderCompiler::DemoteFloat64ToFloat32(notSpirv, output, true));
}
// EliminateFloatEqualsZeroPass turns a comparison against 0.0 into an epsilon test, a
@@ -502,7 +501,7 @@ namespace {
EXPECT_FALSE(input.empty());
if (input.empty()) return false;
Vector<Uint32> output;
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output));
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
return Disassemble(output).find("FAbs") != String::npos;
}
@@ -0,0 +1,147 @@
// MobileGL - MobileGL/MG_Test/ShaderTranspiler/DeriveNumSubgroupsTest.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include <gtest/gtest.h>
#define SPV_ENABLE_UTILITY_CODE
#include "glslang/SPIRV/spirv.hpp11"
#undef SPV_ENABLE_UTILITY_CODE
#include "Includes.h"
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/Types.h>
#include <spirv-tools/libspirv.hpp>
using namespace MobileGL;
using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler;
namespace {
constexpr SizeT kSpirvHeaderWordCount = 5u;
template <typename Visitor>
void ForEachInstruction(const Vector<Uint32>& spirv, Visitor&& visit) {
for (SizeT offset = kSpirvHeaderWordCount; offset < spirv.size();) {
const Uint32 wordCount = spirv[offset] >> 16u;
if (wordCount == 0u || offset + wordCount > spirv.size()) break;
visit(static_cast<spv::Op>(spirv[offset] & 0xffffu), &spirv[offset], wordCount);
offset += wordCount;
}
}
Vector<Uint32> CompileCompute(const String& source) {
using namespace MobileGL::MG_Util::ShaderTranspiler;
ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source};
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
EXPECT_TRUE(shaderResult) << (shaderResult ? String{} : shaderResult.error().log);
if (!shaderResult) return {};
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
EXPECT_TRUE(programResult) << (programResult ? String{} : programResult.error().log);
if (!programResult) return {};
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()};
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
EXPECT_TRUE(binaryResult) << (binaryResult ? String{} : binaryResult.error().log);
if (!binaryResult || binaryResult->empty()) return {};
return binaryResult->front();
}
Uint32 FindBuiltinTarget(const Vector<Uint32>& spirv, spv::BuiltIn builtin) {
Uint32 target = 0u;
ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32* words, Uint32 wordCount) {
if (opcode == spv::Op::OpDecorate && wordCount >= 4u &&
static_cast<spv::Decoration>(words[2]) == spv::Decoration::BuiltIn &&
static_cast<spv::BuiltIn>(words[3]) == builtin) {
target = words[1];
}
});
return target;
}
Uint32 CountLoadsFrom(const Vector<Uint32>& spirv, Uint32 pointerId) {
Uint32 count = 0u;
ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32* words, Uint32 wordCount) {
if (opcode == spv::Op::OpLoad && wordCount >= 4u && words[3] == pointerId) ++count;
});
return count;
}
Uint32 CountOpcode(const Vector<Uint32>& spirv, spv::Op wanted) {
Uint32 count = 0u;
ForEachInstruction(spirv, [&](spv::Op opcode, const Uint32*, Uint32) {
if (opcode == wanted) ++count;
});
return count;
}
bool Validates(const Vector<Uint32>& spirv) {
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
return tools.Validate(spirv);
}
constexpr const char* kNumSubgroupsOnlySource = R"(#version 450 core
#extension GL_KHR_shader_subgroup_basic : require
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
layout(std430, binding = 0) buffer Output { uint value; } outputData;
void main() {
if (gl_LocalInvocationIndex == 0u)
outputData.value = gl_NumSubgroups;
}
)";
constexpr const char* kNoNumSubgroupsSource = R"(#version 450 core
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
layout(std430, binding = 0) buffer Output { uint value; } outputData;
void main() {
if (gl_LocalInvocationIndex == 0u)
outputData.value = gl_WorkGroupSize.x;
}
)";
} // namespace
TEST(DeriveNumSubgroupsPass, ReplacesBuiltinLoadAndSynthesizesSubgroupSize) {
const Vector<Uint32> input = CompileCompute(kNumSubgroupsOnlySource);
ASSERT_FALSE(input.empty());
const Uint32 inputNumSubgroups = FindBuiltinTarget(input, spv::BuiltIn::NumSubgroups);
ASSERT_NE(inputNumSubgroups, 0u);
EXPECT_EQ(CountLoadsFrom(input, inputNumSubgroups), 1u);
EXPECT_EQ(FindBuiltinTarget(input, spv::BuiltIn::SubgroupSize), 0u);
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DeriveNumSubgroupsForVulkan(input, output, true));
ASSERT_TRUE(Validates(output));
const Uint32 outputNumSubgroups = FindBuiltinTarget(output, spv::BuiltIn::NumSubgroups);
const Uint32 outputSubgroupSize = FindBuiltinTarget(output, spv::BuiltIn::SubgroupSize);
ASSERT_NE(outputNumSubgroups, 0u);
ASSERT_NE(outputSubgroupSize, 0u);
EXPECT_EQ(CountLoadsFrom(output, outputNumSubgroups), 0u);
EXPECT_EQ(CountLoadsFrom(output, outputSubgroupSize), 1u);
EXPECT_EQ(CountOpcode(output, spv::Op::OpCompositeExtract), 3u);
EXPECT_EQ(CountOpcode(output, spv::Op::OpIMul), 2u);
EXPECT_EQ(CountOpcode(output, spv::Op::OpUDiv), 1u);
}
TEST(DeriveNumSubgroupsPass, IsIdempotent) {
Vector<Uint32> once;
ASSERT_TRUE(ShaderCompiler::DeriveNumSubgroupsForVulkan(CompileCompute(kNumSubgroupsOnlySource), once, true));
Vector<Uint32> twice;
ASSERT_TRUE(ShaderCompiler::DeriveNumSubgroupsForVulkan(once, twice, true));
EXPECT_EQ(twice, once);
}
TEST(DeriveNumSubgroupsPass, LeavesUnrelatedComputeShaderUntouched) {
const Vector<Uint32> input = CompileCompute(kNoNumSubgroupsSource);
ASSERT_FALSE(input.empty());
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::DeriveNumSubgroupsForVulkan(input, output, true));
EXPECT_EQ(output, input);
EXPECT_EQ(FindBuiltinTarget(output, spv::BuiltIn::SubgroupSize), 0u);
}
@@ -98,7 +98,6 @@ class FlattenXfbInterfaceBlocksTest : public ::testing::Test {
protected:
void SetUp() override {
MobileGL::Initialize();
ShaderCompiler::SetSpirvValidationEnabled(true);
m_validationFailuresAtStart = ShaderCompiler::SpirvValidationFailureCount();
}
@@ -116,7 +115,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, FlattensACapturedBlockIntoOneVariablePerMe
std::set<String> flattened;
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output));
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output, true));
ASSERT_FALSE(output.empty());
EXPECT_EQ(flattened, (std::set<String>{"StageData"}));
@@ -145,7 +144,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, TheEmittedDeclarationIsAPlainArrayNotABloc
std::set<String> flattened;
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output));
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output, true));
const String after = Transpile(output);
EXPECT_NE(after.find("StageData_attrib[16]"), String::npos) << after;
@@ -162,7 +161,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, GivesEachMemberItsOwnConsecutiveLocations)
std::set<String> flattened;
Vector<Uint32> output;
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output));
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output, true));
ASSERT_FALSE(output.empty());
const String dis = Disassemble(output);
@@ -184,7 +183,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, LeavesABlockNoCaptureNamesAlone) {
std::set<String> flattened;
Vector<Uint32> output;
ASSERT_TRUE(
ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"SomeOtherBlock"}, flattened, output));
ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"SomeOtherBlock"}, flattened, output, true));
EXPECT_TRUE(flattened.empty());
const String after = Transpile(output);
@@ -200,7 +199,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, DeclinesAnEmptyRequestWithoutRewriting) {
std::set<String> flattened;
Vector<Uint32> output;
EXPECT_FALSE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {}, flattened, output));
EXPECT_FALSE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {}, flattened, output, true));
EXPECT_TRUE(flattened.empty());
EXPECT_TRUE(output.empty());
}
+477 -7
View File
@@ -6,11 +6,20 @@
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Indexed capability state (glEnablei/glDisablei/glIsEnabledi) exists only for GL_BLEND in this
// stack. Every other capability must come back as GL_INVALID_ENUM per GL 4.6 sec. 17.3.3 - and,
// far more importantly, must come back at all: RenderState::SetCapabilityIndexed and
// IsCapabilityEnabledIndexed used to answer a non-blend capability with THROW_UNIMPL_EXCEPTION,
// Indexed capability state (glEnablei/glDisablei/glIsEnabledi) exists for exactly two
// capabilities: GL_BLEND, indexed by draw buffer, and GL_SCISSOR_TEST, indexed by viewport
// (ARB_viewport_array). Every other capability must come back as GL_INVALID_ENUM per GL 4.6
// sec. 17.3.3 - and, far more importantly, must come back at all: RenderState::SetCapabilityIndexed
// and IsCapabilityEnabledIndexed used to answer a non-blend capability with THROW_UNIMPL_EXCEPTION,
// which unwinds a C++ exception through the C GL ABI and terminates the process.
//
// The second half of this file is the ARB_viewport_array indexed rectangle state. Every one of
// glViewportArrayv/glViewportIndexedf(v)/glScissorArrayv/glScissorIndexed(v)/glDepthRangeArrayv/
// glDepthRangeIndexed was a MGLOG_W_ONCE stub that raised no error and stored nothing, and the
// indexed getters answered EVERY index with viewport 0's value, so a set/get round trip silently
// reported the initial state. The assertions below are deliberately state-shaped rather than
// render-shaped: this IS the state machine, and the rendering half (gl_ViewportIndex routing) is
// asserted separately in MG_IntegrationTest/Scenarios/ViewportArrayScenario.cpp.
#include <gtest/gtest.h>
@@ -21,6 +30,7 @@
#include <MG_Impl/GLImpl/RenderState/GL_RenderState.h>
#include <MG_State/GLState/Core.h>
#include <MG_State/GLState/FramebufferState/FramebufferObject.h>
#include <MG_State/GLState/RenderState/RenderState.h>
using namespace MobileGL;
@@ -50,10 +60,11 @@ namespace {
};
} // namespace
TEST_F(RenderStateTest, IndexedCapabilityTogglesRejectNonBlendCapabilities) {
TEST_F(RenderStateTest, IndexedCapabilityTogglesRejectNonIndexedCapabilities) {
// GL_CLIP_DISTANCE0 is a real capability, just not an indexed one - the shape an application or
// a CTS negative test would hit.
for (const GLenum cap : {GL_CLIP_DISTANCE0, GL_DEPTH_TEST, GL_SCISSOR_TEST}) {
// a CTS negative test would hit. GL_SCISSOR_TEST used to be in this list and is not any more:
// ARB_viewport_array makes it the second indexed capability (see the tests below).
for (const GLenum cap : {GL_CLIP_DISTANCE0, GL_DEPTH_TEST, GL_STENCIL_TEST}) {
MG_Impl::GLImpl::Enablei(cap, 0);
ExpectSingleGlError(GL_INVALID_ENUM);
@@ -89,3 +100,462 @@ TEST_F(RenderStateTest, IndexedBlendTogglesStillWork) {
EXPECT_EQ(MG_Impl::GLImpl::IsEnabledi(GL_BLEND, 1), GL_FALSE);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
}
// ---------------------------------------------------------------------------------------------
// ARB_viewport_array: indexed viewport / scissor / depth-range state
// ---------------------------------------------------------------------------------------------
namespace {
constexpr GLuint kMaxViewports = RenderStateParameters::MAX_VIEWPORTS;
Array<Array<GLfloat, 4>, kMaxViewports> ReadAllViewports() {
Array<Array<GLfloat, 4>, kMaxViewports> out{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, i, out[i].data());
}
return out;
}
Array<Array<GLdouble, 2>, kMaxViewports> ReadAllDepthRanges() {
Array<Array<GLdouble, 2>, kMaxViewports> out{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
MG_Impl::GLImpl::GetDoublei_v(GL_DEPTH_RANGE, i, out[i].data());
}
return out;
}
} // namespace
TEST_F(RenderStateTest, ScissorTestIsIndexedByViewport) {
// The exact shape of KHR-GL43.viewport_array.scissor_test_state_api's toggle loop: one index
// is flipped and EVERY index is read back, so a broadcast masquerading as an indexed write
// cannot pass.
MG_Impl::GLImpl::Disable(GL_SCISSOR_TEST);
ExpectSingleGlError(GL_NO_ERROR);
for (GLuint toggled = 0; toggled < kMaxViewports; ++toggled) {
MG_Impl::GLImpl::Enablei(GL_SCISSOR_TEST, toggled);
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR) << "index " << toggled;
for (GLuint i = 0; i < kMaxViewports; ++i) {
EXPECT_EQ(MG_Impl::GLImpl::IsEnabledi(GL_SCISSOR_TEST, i), i == toggled ? GL_TRUE : GL_FALSE)
<< "enabled index " << toggled << ", read index " << i;
}
MG_Impl::GLImpl::Disablei(GL_SCISSOR_TEST, toggled);
EXPECT_EQ(MG_Impl::GLImpl::IsEnabledi(GL_SCISSOR_TEST, toggled), GL_FALSE);
}
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, NonIndexedScissorTestEnableWritesEveryViewport) {
// GL 4.6 core 17.3.2: Enable/Disable(SCISSOR_TEST) is "for all viewports". Reading only
// index 0 back would let a broadcast-less implementation through, so every index is checked.
MG_Impl::GLImpl::Enable(GL_SCISSOR_TEST);
for (GLuint i = 0; i < kMaxViewports; ++i) {
EXPECT_EQ(MG_Impl::GLImpl::IsEnabledi(GL_SCISSOR_TEST, i), GL_TRUE) << "index " << i;
}
// ... and the non-indexed query answers for viewport 0 (GL 4.6 core 22.1).
EXPECT_EQ(MG_Impl::GLImpl::IsEnabled(GL_SCISSOR_TEST), GL_TRUE);
MG_Impl::GLImpl::Disable(GL_SCISSOR_TEST);
for (GLuint i = 0; i < kMaxViewports; ++i) {
EXPECT_EQ(MG_Impl::GLImpl::IsEnabledi(GL_SCISSOR_TEST, i), GL_FALSE) << "index " << i;
}
EXPECT_EQ(MG_Impl::GLImpl::IsEnabled(GL_SCISSOR_TEST), GL_FALSE);
// An indexed enable on a NON-zero index must not move the non-indexed answer.
MG_Impl::GLImpl::Enablei(GL_SCISSOR_TEST, 3);
EXPECT_EQ(MG_Impl::GLImpl::IsEnabled(GL_SCISSOR_TEST), GL_FALSE);
MG_Impl::GLImpl::Enablei(GL_SCISSOR_TEST, 0);
EXPECT_EQ(MG_Impl::GLImpl::IsEnabled(GL_SCISSOR_TEST), GL_TRUE);
MG_Impl::GLImpl::Disable(GL_SCISSOR_TEST);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, ScissorTestEnableRejectsAnOutOfRangeViewportIndex) {
MG_Impl::GLImpl::Enablei(GL_SCISSOR_TEST, kMaxViewports);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::Disablei(GL_SCISSOR_TEST, kMaxViewports);
ExpectSingleGlError(GL_INVALID_VALUE);
EXPECT_EQ(MG_Impl::GLImpl::IsEnabledi(GL_SCISSOR_TEST, kMaxViewports), GL_FALSE);
ExpectSingleGlError(GL_INVALID_VALUE);
// MAX_VIEWPORTS - 1 is the last LEGAL index and must stay silent.
MG_Impl::GLImpl::Enablei(GL_SCISSOR_TEST, kMaxViewports - 1);
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::Disablei(GL_SCISSOR_TEST, kMaxViewports - 1);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, MaxViewportsMatchesTheIndexedStateWidth) {
// The advertised limit and the width of the state arrays are the same number by
// construction; a divergence would make some index simultaneously legal to the CTS and
// out of range to the setters.
GLint maxViewports = 0;
MG_Impl::GLImpl::GetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
ExpectSingleGlError(GL_NO_ERROR);
EXPECT_EQ(maxViewports, static_cast<GLint>(kMaxViewports));
EXPECT_GE(maxViewports, 16) << "GL 4.3 core requires MAX_VIEWPORTS >= 16";
}
TEST_F(RenderStateTest, ViewportArrayvRoundTripsThroughEveryGetterWidth) {
Array<GLfloat, kMaxViewports * 4> written{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
written[i * 4 + 0] = static_cast<GLfloat>(i) + 0.125f;
written[i * 4 + 1] = static_cast<GLfloat>(i) + 0.25f;
written[i * 4 + 2] = static_cast<GLfloat>(64 + i);
written[i * 4 + 3] = static_cast<GLfloat>(32 + i);
}
MG_Impl::GLImpl::ViewportArrayv(0, kMaxViewports, written.data());
ExpectSingleGlError(GL_NO_ERROR);
for (GLuint i = 0; i < kMaxViewports; ++i) {
GLfloat asFloat[4] = {};
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, i, asFloat);
// Bit-exact: the fractional origin is the whole point of float viewport state, and the
// CTS compares with == (0.125 and 0.25 are exact binary fractions, so this is fair).
EXPECT_EQ(asFloat[0], written[i * 4 + 0]) << "index " << i << " must round-trip verbatim";
EXPECT_EQ(asFloat[1], written[i * 4 + 1]) << "index " << i;
EXPECT_EQ(asFloat[2], written[i * 4 + 2]) << "index " << i;
EXPECT_EQ(asFloat[3], written[i * 4 + 3]) << "index " << i;
GLdouble asDouble[4] = {};
MG_Impl::GLImpl::GetDoublei_v(GL_VIEWPORT, i, asDouble);
for (int c = 0; c < 4; ++c) {
EXPECT_EQ(asDouble[c], static_cast<GLdouble>(written[i * 4 + c])) << "index " << i << " component " << c;
}
// The integer widths round to nearest rather than truncate; the .5+ case is pinned by
// ViewportRoundsRatherThanTruncatesForIntegerQueries below.
GLint asInt[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_VIEWPORT, i, asInt);
EXPECT_EQ(asInt[2], static_cast<GLint>(64 + i)) << "index " << i;
EXPECT_EQ(asInt[3], static_cast<GLint>(32 + i)) << "index " << i;
GLint64 asInt64[4] = {};
MG_Impl::GLImpl::GetInteger64i_v(GL_VIEWPORT, i, asInt64);
for (int c = 0; c < 4; ++c) {
EXPECT_EQ(asInt64[c], static_cast<GLint64>(asInt[c])) << "index " << i << " component " << c;
}
GLboolean asBool[4] = {};
MG_Impl::GLImpl::GetBooleani_v(GL_VIEWPORT, i, asBool);
EXPECT_EQ(asBool[2], GL_TRUE) << "index " << i << ": a non-zero width is GL_TRUE";
}
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, ViewportRoundsRatherThanTruncatesForIntegerQueries) {
MG_Impl::GLImpl::ViewportIndexedf(2, 0.0f, 0.0f, 255.875f, 63.5f);
ExpectSingleGlError(GL_NO_ERROR);
GLint asInt[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_VIEWPORT, 2, asInt);
EXPECT_EQ(asInt[2], 256);
EXPECT_EQ(asInt[3], 64);
GLfloat asFloat[4] = {};
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, 2, asFloat);
EXPECT_EQ(asFloat[2], 255.875f) << "the integer query must not disturb the stored float";
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, ViewportIndexedWritesTouchExactlyOneIndex) {
MG_Impl::GLImpl::Viewport(0, 0, 8, 8);
const auto before = ReadAllViewports();
for (GLuint target = 0; target < kMaxViewports; ++target) {
const GLfloat value[4] = {0.375f, 0.375f, 0.625f, 0.625f};
// Alternate the two indexed entry points so both are covered by the isolation claim.
if (target % 2 == 0) {
MG_Impl::GLImpl::ViewportIndexedf(target, value[0], value[1], value[2], value[3]);
} else {
MG_Impl::GLImpl::ViewportIndexedfv(target, value);
}
ExpectSingleGlError(GL_NO_ERROR);
const auto after = ReadAllViewports();
for (GLuint i = 0; i < kMaxViewports; ++i) {
if (i == target) {
EXPECT_EQ(after[i][0], value[0]) << "index " << i;
EXPECT_EQ(after[i][2], value[2]) << "index " << i;
} else {
EXPECT_EQ(after[i], before[i]) << "write to " << target << " disturbed index " << i;
}
}
MG_Impl::GLImpl::ViewportIndexedf(target, before[target][0], before[target][1], before[target][2],
before[target][3]);
}
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, ClassicViewportWritesEveryIndexAndIsVisibleThroughIndexZero) {
// Both directions of the aliasing. ARB_viewport_array defines glViewport as ViewportIndexedf
// on every index, and glGetIntegerv(GL_VIEWPORT) as viewport 0.
MG_Impl::GLImpl::ViewportIndexedf(5, 1.0f, 2.0f, 3.0f, 4.0f);
MG_Impl::GLImpl::Viewport(0, 0, 1, 1);
ExpectSingleGlError(GL_NO_ERROR);
for (GLuint i = 0; i < kMaxViewports; ++i) {
GLfloat data[4] = {};
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, i, data);
EXPECT_EQ(data[0], 0.0f) << "index " << i;
EXPECT_EQ(data[2], 1.0f) << "glViewport must overwrite index " << i;
}
MG_Impl::GLImpl::ViewportIndexedf(0, 4.0f, 5.0f, 6.0f, 7.0f);
GLint classic[4] = {};
MG_Impl::GLImpl::GetIntegerv(GL_VIEWPORT, classic);
EXPECT_EQ(classic[0], 4);
EXPECT_EQ(classic[2], 6);
GLfloat classicFloat[4] = {};
MG_Impl::GLImpl::GetFloatv(GL_VIEWPORT, classicFloat);
EXPECT_EQ(classicFloat[2], 6.0f);
// Index 5 keeps its own value: writing index 0 is not a broadcast.
GLfloat other[4] = {};
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, 5, other);
EXPECT_EQ(other[2], 1.0f);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, ScissorBoxRoundTripsPerIndexAndAliasesIndexZero) {
Array<GLint, kMaxViewports * 4> written{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
written[i * 4 + 0] = static_cast<GLint>(i);
written[i * 4 + 1] = static_cast<GLint>(i * 2);
written[i * 4 + 2] = static_cast<GLint>(16 + i);
written[i * 4 + 3] = static_cast<GLint>(8 + i);
}
MG_Impl::GLImpl::ScissorArrayv(0, kMaxViewports, written.data());
ExpectSingleGlError(GL_NO_ERROR);
for (GLuint i = 0; i < kMaxViewports; ++i) {
GLint readBack[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, i, readBack);
for (int c = 0; c < 4; ++c) {
EXPECT_EQ(readBack[c], written[i * 4 + c]) << "index " << i << " component " << c;
}
}
// Indexed writes stay indexed; both spellings.
MG_Impl::GLImpl::ScissorIndexed(4, 4, 4, 8, 8);
const GLint indexedV[4] = {9, 9, 12, 12};
MG_Impl::GLImpl::ScissorIndexedv(7, indexedV);
ExpectSingleGlError(GL_NO_ERROR);
GLint probe[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, 4, probe);
EXPECT_EQ(probe[2], 8);
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, 7, probe);
EXPECT_EQ(probe[2], 12);
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, 5, probe);
EXPECT_EQ(probe[2], static_cast<GLint>(16 + 5)) << "index 5 must be untouched";
// glScissor writes every rectangle, and glGetIntegerv(GL_SCISSOR_BOX) reports rectangle 0.
MG_Impl::GLImpl::Scissor(2, 3, 5, 6);
for (GLuint i = 0; i < kMaxViewports; ++i) {
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, i, probe);
EXPECT_EQ(probe[0], 2) << "index " << i;
EXPECT_EQ(probe[2], 5) << "index " << i;
}
GLint classic[4] = {};
MG_Impl::GLImpl::GetIntegerv(GL_SCISSOR_BOX, classic);
EXPECT_EQ(classic[2], 5);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, DepthRangeRoundTripsPerIndexAndAliasesIndexZero) {
Array<GLdouble, kMaxViewports * 2> written{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
// Exact binary fractions, like the CTS uses: a float-backed store round-trips them.
written[i * 2 + 0] = static_cast<GLdouble>(i) / 16.0;
written[i * 2 + 1] = 1.0 - static_cast<GLdouble>(i) / 16.0;
}
MG_Impl::GLImpl::DepthRangeArrayv(0, kMaxViewports, written.data());
ExpectSingleGlError(GL_NO_ERROR);
const auto readBack = ReadAllDepthRanges();
for (GLuint i = 0; i < kMaxViewports; ++i) {
EXPECT_EQ(readBack[i][0], written[i * 2 + 0]) << "index " << i;
EXPECT_EQ(readBack[i][1], written[i * 2 + 1]) << "index " << i;
}
MG_Impl::GLImpl::DepthRangeIndexed(9, 0.25, 0.75);
ExpectSingleGlError(GL_NO_ERROR);
GLdouble probe[2] = {};
MG_Impl::GLImpl::GetDoublei_v(GL_DEPTH_RANGE, 9, probe);
EXPECT_EQ(probe[0], 0.25);
EXPECT_EQ(probe[1], 0.75);
MG_Impl::GLImpl::GetDoublei_v(GL_DEPTH_RANGE, 8, probe);
EXPECT_EQ(probe[0], 8.0 / 16.0) << "index 8 must be untouched";
GLfloat asFloat[2] = {};
MG_Impl::GLImpl::GetFloati_v(GL_DEPTH_RANGE, 9, asFloat);
EXPECT_EQ(asFloat[0], 0.25f);
EXPECT_EQ(asFloat[1], 0.75f);
// glDepthRange writes every range; glGetDoublev(GL_DEPTH_RANGE) reports range 0.
MG_Impl::GLImpl::DepthRange(0.0, 1.0);
for (GLuint i = 0; i < kMaxViewports; ++i) {
MG_Impl::GLImpl::GetDoublei_v(GL_DEPTH_RANGE, i, probe);
EXPECT_EQ(probe[0], 0.0) << "index " << i;
EXPECT_EQ(probe[1], 1.0) << "index " << i;
}
MG_Impl::GLImpl::DepthRangeIndexed(0, 0.125, 0.875);
GLdouble classic[2] = {};
MG_Impl::GLImpl::GetDoublev(GL_DEPTH_RANGE, classic);
EXPECT_EQ(classic[0], 0.125);
EXPECT_EQ(classic[1], 0.875);
MG_Impl::GLImpl::DepthRange(0.0, 1.0);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, IndexedRectangleSettersRejectAnOutOfRangeIndex) {
const GLfloat viewport[4] = {0.0f, 0.0f, 1.0f, 1.0f};
const GLint scissor[4] = {0, 0, 1, 1};
for (const GLuint index : {kMaxViewports, kMaxViewports + 1}) {
MG_Impl::GLImpl::ViewportIndexedf(index, 0.0f, 0.0f, 1.0f, 1.0f);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ViewportIndexedfv(index, viewport);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorIndexed(index, 0, 0, 1, 1);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorIndexedv(index, scissor);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::DepthRangeIndexed(index, 0.0, 1.0);
ExpectSingleGlError(GL_INVALID_VALUE);
}
// The last legal index must stay silent - api_errors checks both sides of the boundary.
MG_Impl::GLImpl::ViewportIndexedf(kMaxViewports - 1, 0.0f, 0.0f, 1.0f, 1.0f);
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::ScissorIndexed(kMaxViewports - 1, 0, 0, 1, 1);
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::DepthRangeIndexed(kMaxViewports - 1, 0.0, 1.0);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, ArraySettersRejectAnOutOfRangeRangeButAcceptAnExactlyFullOne) {
Array<GLfloat, kMaxViewports * 4> viewports{};
Array<GLint, kMaxViewports * 4> scissors{};
Array<GLdouble, kMaxViewports * 2> depths{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
viewports[i * 4 + 2] = 1.0f;
viewports[i * 4 + 3] = 1.0f;
scissors[i * 4 + 2] = 1;
scissors[i * 4 + 3] = 1;
depths[i * 2 + 1] = 1.0;
}
// first == MAX_VIEWPORTS, and first + count > MAX_VIEWPORTS.
MG_Impl::GLImpl::ViewportArrayv(kMaxViewports, 1, viewports.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ViewportArrayv(1, kMaxViewports, viewports.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorArrayv(kMaxViewports, 1, scissors.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorArrayv(1, kMaxViewports, scissors.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::DepthRangeArrayv(kMaxViewports, 1, depths.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::DepthRangeArrayv(1, kMaxViewports, depths.data());
ExpectSingleGlError(GL_INVALID_VALUE);
// first + count == MAX_VIEWPORTS is LEGAL - the off-by-one an ">=" bound would get wrong,
// and one KHR-GL43.viewport_array.api_errors asserts explicitly.
MG_Impl::GLImpl::ViewportArrayv(1, kMaxViewports - 1, viewports.data());
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::ScissorArrayv(1, kMaxViewports - 1, scissors.data());
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::DepthRangeArrayv(1, kMaxViewports - 1, depths.data());
ExpectSingleGlError(GL_NO_ERROR);
// A negative count is GL_INVALID_VALUE and must not be read as a huge unsigned length.
MG_Impl::GLImpl::ViewportArrayv(0, -1, viewports.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorArrayv(0, -1, scissors.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::DepthRangeArrayv(0, -1, depths.data());
ExpectSingleGlError(GL_INVALID_VALUE);
}
TEST_F(RenderStateTest, NegativeExtentsAreRejectedWithoutDisturbingState) {
MG_Impl::GLImpl::Viewport(0, 0, 4, 4);
MG_Impl::GLImpl::Scissor(0, 0, 4, 4);
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::Viewport(0, 0, -1, 1);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::Viewport(0, 0, 1, -1);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::Scissor(0, 0, -1, 1);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::Scissor(0, 0, 1, -1);
ExpectSingleGlError(GL_INVALID_VALUE);
for (GLuint index = 0; index < kMaxViewports; ++index) {
MG_Impl::GLImpl::ViewportIndexedf(index, 0.0f, 0.0f, -1.0f, 1.0f);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ViewportIndexedf(index, 0.0f, 0.0f, 1.0f, -1.0f);
ExpectSingleGlError(GL_INVALID_VALUE);
const GLfloat badW[4] = {0.0f, 0.0f, -1.0f, 1.0f};
MG_Impl::GLImpl::ViewportIndexedfv(index, badW);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorIndexed(index, 0, 0, -1, 1);
ExpectSingleGlError(GL_INVALID_VALUE);
const GLint badH[4] = {0, 0, 1, -1};
MG_Impl::GLImpl::ScissorIndexedv(index, badH);
ExpectSingleGlError(GL_INVALID_VALUE);
// The array form must reject the WHOLE call for one bad element, exactly once, and
// leave every rectangle alone - api_errors submits a full 16-element array with a
// single negative extent and then requires the error queue to hold one entry.
Array<GLfloat, kMaxViewports * 4> viewports{};
Array<GLint, kMaxViewports * 4> scissors{};
for (GLuint i = 0; i < kMaxViewports; ++i) {
viewports[i * 4 + 2] = 1.0f;
viewports[i * 4 + 3] = 1.0f;
scissors[i * 4 + 2] = 1;
scissors[i * 4 + 3] = 1;
}
viewports[index * 4 + 2] = -1.0f;
scissors[index * 4 + 3] = -1;
MG_Impl::GLImpl::ViewportArrayv(0, kMaxViewports, viewports.data());
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::ScissorArrayv(0, kMaxViewports, scissors.data());
ExpectSingleGlError(GL_INVALID_VALUE);
}
// Nothing above may have landed.
GLint viewport[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_VIEWPORT, 0, viewport);
EXPECT_EQ(viewport[2], 4);
EXPECT_EQ(viewport[3], 4);
GLint scissor[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, 0, scissor);
EXPECT_EQ(scissor[2], 4);
EXPECT_EQ(scissor[3], 4);
ExpectSingleGlError(GL_NO_ERROR);
}
TEST_F(RenderStateTest, IndexedRectangleQueriesRejectAnOutOfRangeIndex) {
GLint ints[4] = {};
GLfloat floats[4] = {};
GLdouble doubles[4] = {};
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, kMaxViewports, ints);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, kMaxViewports, floats);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::GetDoublei_v(GL_DEPTH_RANGE, kMaxViewports, doubles);
ExpectSingleGlError(GL_INVALID_VALUE);
MG_Impl::GLImpl::GetIntegeri_v(GL_SCISSOR_BOX, kMaxViewports - 1, ints);
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::GetFloati_v(GL_VIEWPORT, kMaxViewports - 1, floats);
ExpectSingleGlError(GL_NO_ERROR);
MG_Impl::GLImpl::GetDoublei_v(GL_DEPTH_RANGE, kMaxViewports - 1, doubles);
ExpectSingleGlError(GL_NO_ERROR);
}
@@ -955,6 +955,8 @@ namespace MobileGL::MG_Util::BackendLoader {
(caps.GLESVersion.Major == 3 && caps.GLESVersion.Minor >= 2);
const Bool esAtLeast31 = caps.GLESVersion.Major > 3 ||
(caps.GLESVersion.Major == 3 && caps.GLESVersion.Minor >= 1);
caps.SupportsDrawIndirect = esAtLeast31 && glesFuncs.glDrawArraysIndirect != nullptr &&
glesFuncs.glDrawElementsIndirect != nullptr;
caps.SupportsDrawElementsBaseVertex = (esAtLeast32 || hasDrawElementsBaseVertexExtension) &&
glesFuncs.glDrawElementsBaseVertex != nullptr;
caps.SupportsComputeShader = esAtLeast31 && glesFuncs.glDispatchCompute != nullptr &&
@@ -976,6 +978,7 @@ namespace MobileGL::MG_Util::BackendLoader {
MGLOG_I(" indexed glColorMaski: %s", caps.SupportsIndexedColorMask ? "yes" : "no");
MGLOG_I(" dual-source blend (EXT_blend_func_extended): %s",
caps.SupportsDualSourceBlend ? "yes" : "no");
MGLOG_I(" draw indirect (ES 3.1 core): %s", caps.SupportsDrawIndirect ? "yes" : "no");
MGLOG_I(" multi-draw indirect (EXT_multi_draw_indirect): %s",
caps.SupportsMultiDrawIndirect ? "yes" : "no");
MGLOG_I(" multi-draw base vertex (EXT/OES_draw_elements_base_vertex + EXT_multi_draw_arrays): %s",
@@ -1000,7 +1003,11 @@ namespace MobileGL::MG_Util::BackendLoader {
GLfloat smoothLineWidthRange[2] = {1.0f, 1.0f};
GLfloat smoothLineWidthGranularity = 1.0f;
GLfloat aliasedPointSizeRange[2] = {1.0f, 1.0f};
GLfloat viewportBoundsRange[2] = {0.0f, 0.0f};
// GL 4.6 core table 23.60 sets the MINIMUM VIEWPORT_BOUNDS_RANGE at [-32768, 32767], and
// KHR-GL43.viewport_array.queries asserts exactly that floor. GLES has no such query, so
// the glGetFloatv below raises GL_INVALID_ENUM and leaves this untouched - starting it at
// {0, 0} advertised a range that admits no viewport origin at all.
GLfloat viewportBoundsRange[2] = {-32768.0f, 32767.0f};
GLint maxViewportDims[2] = {16384, 16384};
GLint viewportSubpixelBits = 0;
GLint max3DTextureSize = 16384;
@@ -1289,8 +1296,12 @@ namespace MobileGL::MG_Util::BackendLoader {
caps.MaxViewports = maxViewports;
caps.MaxViewportWidth = maxViewportDims[0];
caps.MaxViewportHeight = maxViewportDims[1];
caps.ViewportBoundsRangeMin = viewportBoundsRange[0];
caps.ViewportBoundsRangeMax = viewportBoundsRange[1];
// Only ever WIDER than the core minimum: a driver that answered the query is allowed to
// exceed the floor but never to sit inside it, and a driver that rejected the query left
// the floor in place. Written as a clamp rather than a plain assignment so a partial
// write (one component answered, the other not) cannot narrow the range either.
caps.ViewportBoundsRangeMin = std::min(viewportBoundsRange[0], -32768.0f);
caps.ViewportBoundsRangeMax = std::max(viewportBoundsRange[1], 32767.0f);
caps.ViewportSubpixelBits = viewportSubpixelBits;
caps.MinFragmentInterpolationOffset =
std::isfinite(minFragmentInterpolationOffset) && minFragmentInterpolationOffset <= -0.5f
@@ -1149,6 +1149,10 @@ namespace MobileGL {
// GLES 3.2 core or GL_OES_shader_multisample_interpolation exposes
// interpolateAtOffset and the three fragment-offset limit queries.
Bool SupportsShaderMultisampleInterpolation = false;
// ES 3.1+ exposes glDrawArraysIndirect / glDrawElementsIndirect in core. Keep the
// version and both entry-point checks together so extension advertisement and the
// DirectGLES dispatch path cannot disagree on whether native indirect draws exist.
Bool SupportsDrawIndirect = false;
// GL_EXT_multi_draw_indirect is present AND glMultiDrawArraysIndirectEXT /
// glMultiDrawElementsIndirectEXT both resolved. Multi-draw is not core in any ES
// version, and eglGetProcAddress may return a live-looking stub on drivers without
+2 -2
View File
@@ -33,13 +33,13 @@ namespace MobileGL {
std::string GetThreadName() {
char buffer[64] = {0};
#if defined(_WIN32) && !defined(__MINGW32__)
#if defined(_WIN32)
PWSTR desc = nullptr;
if (SUCCEEDED(GetThreadDescription(GetCurrentThread(), &desc))) {
WideCharToMultiByte(CP_UTF8, 0, desc, -1, buffer, sizeof(buffer), nullptr, nullptr);
LocalFree(desc);
}
#elif defined(__ANDROID__) || defined(__linux__) || defined(__APPLE__) || defined(__MINGW32__)
#elif defined(__ANDROID__) || defined(__linux__) || defined(__APPLE__)
pthread_getname_np(pthread_self(), buffer, sizeof(buffer));
#endif
return buffer[0] ? buffer : "UnknownThread";
+458 -3
View File
@@ -7,6 +7,8 @@
// End of Source File Header
#include "DriverPost.h"
#include "DriverPostProgram203Witness.h"
#include "DriverPostProgram203WitnessSpv.h"
#include "MG_Util/BackendLoaders/OpenGL/Loader.h"
#include <Config.h>
#include <MGGitHash.h>
@@ -24,6 +26,8 @@
#include <MG_Util/Texture/TextureFormatProcessor.h>
#include <MG_Util/Async/ShaderCompilePool.h>
#include <chrono>
#include <cstring>
#include <limits>
#include <thread>
#if !defined(_WIN32)
@@ -1168,7 +1172,9 @@ namespace MobileGL::MG_Util::SelfTest {
backendApiVersionString = MG_Backend::DirectGLES::FormatBackendAPIVersionString(
summary.caps.GLESRendererString, summary.caps.GLESVersion.Major, summary.caps.GLESVersion.Minor);
advertisedExtensions = JoinAdvertisedExtensions(MG_Backend::DirectGLES::BuildAdvertisedExtensions(
summary.caps.SupportsDisjointTimerQuery, summary.caps.SupportsTextureFilterAnisotropy));
summary.caps.SupportsDisjointTimerQuery, summary.caps.SupportsTextureFilterAnisotropy,
summary.caps.SupportsDrawIndirect,
summary.caps.SupportsDrawIndirect && summary.caps.SupportsBaseInstance));
}
AppendMobileGLReportedRows(builder, MG_Backend::DirectGLES::GetRendererIdentity(), backendApiVersionString,
advertisedExtensions);
@@ -1452,6 +1458,437 @@ namespace MobileGL::MG_Util::SelfTest {
disabledNote);
}
// Native Program-203 compute witness. This deliberately uses a separate
// throwaway Vulkan device rather than the real renderer's queues, and it
// treats MOBILEGL_DISABLE_SUBGROUP as irrelevant: the row reports what the
// driver does, not what MobileGL elects to advertise to applications.
void ProbeVulkanProgram203Witness(ReportBuilder& builder, PFN_vkGetInstanceProcAddr getInstanceProcAddr,
VkInstance instance, VkPhysicalDevice physicalDevice,
Uint32 computeQueueFamilyIndex,
const VkPhysicalDeviceProperties& properties,
Bool subgroupPropertiesAvailable,
const VkPhysicalDeviceSubgroupProperties& subgroupProperties) {
constexpr const char* RowName = "Subgroup first-reduction witness";
const auto fail = [&](String detail) { builder.Fail(RowName, Move(detail)); };
if (!subgroupPropertiesAvailable) {
fail("vkGetPhysicalDeviceProperties2 could not provide raw Vulkan subgroup properties");
return;
}
Program203WitnessLimits limits{};
limits.computeStageSupported =
(subgroupProperties.supportedStages & VK_SHADER_STAGE_COMPUTE_BIT) != 0;
limits.basicSubgroupSupported =
(subgroupProperties.supportedOperations & VK_SUBGROUP_FEATURE_BASIC_BIT) != 0;
limits.arithmeticSubgroupSupported =
(subgroupProperties.supportedOperations & VK_SUBGROUP_FEATURE_ARITHMETIC_BIT) != 0;
limits.subgroupSize = subgroupProperties.subgroupSize;
limits.maxComputeWorkGroupInvocations = properties.limits.maxComputeWorkGroupInvocations;
limits.maxComputeWorkGroupSize = {properties.limits.maxComputeWorkGroupSize[0],
properties.limits.maxComputeWorkGroupSize[1],
properties.limits.maxComputeWorkGroupSize[2]};
limits.maxComputeSharedMemorySize = properties.limits.maxComputeSharedMemorySize;
limits.maxPerStageDescriptorStorageBuffers = properties.limits.maxPerStageDescriptorStorageBuffers;
limits.maxDescriptorSetStorageBuffers = properties.limits.maxDescriptorSetStorageBuffers;
limits.maxBoundDescriptorSets = properties.limits.maxBoundDescriptorSets;
limits.maxStorageBufferRange = properties.limits.maxStorageBufferRange;
const Program203WitnessEligibilityResult eligibility = EvaluateProgram203WitnessEligibility(limits);
if (eligibility.eligibility == Program203WitnessEligibility::SkipUnsupportedNativeFeatureSet) {
builder.Info(RowName, eligibility.detail);
return;
}
if (eligibility.eligibility == Program203WitnessEligibility::FailInadequateLimits) {
fail(eligibility.detail);
return;
}
if (computeQueueFamilyIndex == std::numeric_limits<Uint32>::max()) {
fail("no compute queue family is available for the native Vulkan witness");
return;
}
const auto vkGetPhysicalDeviceMemoryPropertiesFn =
reinterpret_cast<PFN_vkGetPhysicalDeviceMemoryProperties>(
getInstanceProcAddr(instance, "vkGetPhysicalDeviceMemoryProperties"));
const auto vkCreateDeviceFn =
reinterpret_cast<PFN_vkCreateDevice>(getInstanceProcAddr(instance, "vkCreateDevice"));
const auto vkDestroyDeviceFn =
reinterpret_cast<PFN_vkDestroyDevice>(getInstanceProcAddr(instance, "vkDestroyDevice"));
const auto vkGetDeviceQueueFn =
reinterpret_cast<PFN_vkGetDeviceQueue>(getInstanceProcAddr(instance, "vkGetDeviceQueue"));
const auto vkCreateBufferFn =
reinterpret_cast<PFN_vkCreateBuffer>(getInstanceProcAddr(instance, "vkCreateBuffer"));
const auto vkDestroyBufferFn =
reinterpret_cast<PFN_vkDestroyBuffer>(getInstanceProcAddr(instance, "vkDestroyBuffer"));
const auto vkGetBufferMemoryRequirementsFn = reinterpret_cast<PFN_vkGetBufferMemoryRequirements>(
getInstanceProcAddr(instance, "vkGetBufferMemoryRequirements"));
const auto vkAllocateMemoryFn =
reinterpret_cast<PFN_vkAllocateMemory>(getInstanceProcAddr(instance, "vkAllocateMemory"));
const auto vkFreeMemoryFn =
reinterpret_cast<PFN_vkFreeMemory>(getInstanceProcAddr(instance, "vkFreeMemory"));
const auto vkBindBufferMemoryFn =
reinterpret_cast<PFN_vkBindBufferMemory>(getInstanceProcAddr(instance, "vkBindBufferMemory"));
const auto vkMapMemoryFn =
reinterpret_cast<PFN_vkMapMemory>(getInstanceProcAddr(instance, "vkMapMemory"));
const auto vkUnmapMemoryFn =
reinterpret_cast<PFN_vkUnmapMemory>(getInstanceProcAddr(instance, "vkUnmapMemory"));
const auto vkCreateDescriptorSetLayoutFn = reinterpret_cast<PFN_vkCreateDescriptorSetLayout>(
getInstanceProcAddr(instance, "vkCreateDescriptorSetLayout"));
const auto vkDestroyDescriptorSetLayoutFn = reinterpret_cast<PFN_vkDestroyDescriptorSetLayout>(
getInstanceProcAddr(instance, "vkDestroyDescriptorSetLayout"));
const auto vkCreateDescriptorPoolFn =
reinterpret_cast<PFN_vkCreateDescriptorPool>(getInstanceProcAddr(instance, "vkCreateDescriptorPool"));
const auto vkDestroyDescriptorPoolFn = reinterpret_cast<PFN_vkDestroyDescriptorPool>(
getInstanceProcAddr(instance, "vkDestroyDescriptorPool"));
const auto vkAllocateDescriptorSetsFn = reinterpret_cast<PFN_vkAllocateDescriptorSets>(
getInstanceProcAddr(instance, "vkAllocateDescriptorSets"));
const auto vkUpdateDescriptorSetsFn =
reinterpret_cast<PFN_vkUpdateDescriptorSets>(getInstanceProcAddr(instance, "vkUpdateDescriptorSets"));
const auto vkCreateShaderModuleFn =
reinterpret_cast<PFN_vkCreateShaderModule>(getInstanceProcAddr(instance, "vkCreateShaderModule"));
const auto vkDestroyShaderModuleFn =
reinterpret_cast<PFN_vkDestroyShaderModule>(getInstanceProcAddr(instance, "vkDestroyShaderModule"));
const auto vkCreatePipelineLayoutFn =
reinterpret_cast<PFN_vkCreatePipelineLayout>(getInstanceProcAddr(instance, "vkCreatePipelineLayout"));
const auto vkDestroyPipelineLayoutFn = reinterpret_cast<PFN_vkDestroyPipelineLayout>(
getInstanceProcAddr(instance, "vkDestroyPipelineLayout"));
const auto vkCreateComputePipelinesFn = reinterpret_cast<PFN_vkCreateComputePipelines>(
getInstanceProcAddr(instance, "vkCreateComputePipelines"));
const auto vkDestroyPipelineFn =
reinterpret_cast<PFN_vkDestroyPipeline>(getInstanceProcAddr(instance, "vkDestroyPipeline"));
const auto vkCreateCommandPoolFn =
reinterpret_cast<PFN_vkCreateCommandPool>(getInstanceProcAddr(instance, "vkCreateCommandPool"));
const auto vkDestroyCommandPoolFn =
reinterpret_cast<PFN_vkDestroyCommandPool>(getInstanceProcAddr(instance, "vkDestroyCommandPool"));
const auto vkAllocateCommandBuffersFn = reinterpret_cast<PFN_vkAllocateCommandBuffers>(
getInstanceProcAddr(instance, "vkAllocateCommandBuffers"));
const auto vkBeginCommandBufferFn =
reinterpret_cast<PFN_vkBeginCommandBuffer>(getInstanceProcAddr(instance, "vkBeginCommandBuffer"));
const auto vkEndCommandBufferFn =
reinterpret_cast<PFN_vkEndCommandBuffer>(getInstanceProcAddr(instance, "vkEndCommandBuffer"));
const auto vkCmdBindPipelineFn =
reinterpret_cast<PFN_vkCmdBindPipeline>(getInstanceProcAddr(instance, "vkCmdBindPipeline"));
const auto vkCmdBindDescriptorSetsFn = reinterpret_cast<PFN_vkCmdBindDescriptorSets>(
getInstanceProcAddr(instance, "vkCmdBindDescriptorSets"));
const auto vkCmdDispatchFn =
reinterpret_cast<PFN_vkCmdDispatch>(getInstanceProcAddr(instance, "vkCmdDispatch"));
const auto vkCmdPipelineBarrierFn =
reinterpret_cast<PFN_vkCmdPipelineBarrier>(getInstanceProcAddr(instance, "vkCmdPipelineBarrier"));
const auto vkCreateFenceFn =
reinterpret_cast<PFN_vkCreateFence>(getInstanceProcAddr(instance, "vkCreateFence"));
const auto vkDestroyFenceFn =
reinterpret_cast<PFN_vkDestroyFence>(getInstanceProcAddr(instance, "vkDestroyFence"));
const auto vkQueueSubmitFn =
reinterpret_cast<PFN_vkQueueSubmit>(getInstanceProcAddr(instance, "vkQueueSubmit"));
const auto vkWaitForFencesFn =
reinterpret_cast<PFN_vkWaitForFences>(getInstanceProcAddr(instance, "vkWaitForFences"));
const auto vkDeviceWaitIdleFn =
reinterpret_cast<PFN_vkDeviceWaitIdle>(getInstanceProcAddr(instance, "vkDeviceWaitIdle"));
if (vkGetPhysicalDeviceMemoryPropertiesFn == nullptr || vkCreateDeviceFn == nullptr ||
vkDestroyDeviceFn == nullptr || vkGetDeviceQueueFn == nullptr || vkCreateBufferFn == nullptr ||
vkDestroyBufferFn == nullptr || vkGetBufferMemoryRequirementsFn == nullptr ||
vkAllocateMemoryFn == nullptr || vkFreeMemoryFn == nullptr || vkBindBufferMemoryFn == nullptr ||
vkMapMemoryFn == nullptr || vkUnmapMemoryFn == nullptr || vkCreateDescriptorSetLayoutFn == nullptr ||
vkDestroyDescriptorSetLayoutFn == nullptr || vkCreateDescriptorPoolFn == nullptr ||
vkDestroyDescriptorPoolFn == nullptr || vkAllocateDescriptorSetsFn == nullptr ||
vkUpdateDescriptorSetsFn == nullptr || vkCreateShaderModuleFn == nullptr ||
vkDestroyShaderModuleFn == nullptr || vkCreatePipelineLayoutFn == nullptr ||
vkDestroyPipelineLayoutFn == nullptr || vkCreateComputePipelinesFn == nullptr ||
vkDestroyPipelineFn == nullptr || vkCreateCommandPoolFn == nullptr || vkDestroyCommandPoolFn == nullptr ||
vkAllocateCommandBuffersFn == nullptr || vkBeginCommandBufferFn == nullptr ||
vkEndCommandBufferFn == nullptr || vkCmdBindPipelineFn == nullptr ||
vkCmdBindDescriptorSetsFn == nullptr || vkCmdDispatchFn == nullptr ||
vkCmdPipelineBarrierFn == nullptr || vkCreateFenceFn == nullptr || vkDestroyFenceFn == nullptr ||
vkQueueSubmitFn == nullptr || vkWaitForFencesFn == nullptr || vkDeviceWaitIdleFn == nullptr) {
fail("vkGetInstanceProcAddr could not resolve the Vulkan entry points required for the witness");
return;
}
const Float queuePriority = 1.0f;
VkDeviceQueueCreateInfo queueInfo{};
queueInfo.sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO;
queueInfo.queueFamilyIndex = computeQueueFamilyIndex;
queueInfo.queueCount = 1;
queueInfo.pQueuePriorities = &queuePriority;
VkDeviceCreateInfo deviceInfo{};
deviceInfo.sType = VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO;
deviceInfo.queueCreateInfoCount = 1;
deviceInfo.pQueueCreateInfos = &queueInfo;
VkDevice device = VK_NULL_HANDLE;
VkResult result = vkCreateDeviceFn(physicalDevice, &deviceInfo, nullptr, &device);
if (result != VK_SUCCESS || device == VK_NULL_HANDLE) {
fail(format("vkCreateDevice failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkBuffer outputBuffer = VK_NULL_HANDLE;
VkDeviceMemory outputMemory = VK_NULL_HANDLE;
VkDescriptorSetLayout descriptorSetLayout = VK_NULL_HANDLE;
VkDescriptorPool descriptorPool = VK_NULL_HANDLE;
VkShaderModule shaderModule = VK_NULL_HANDLE;
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
VkPipeline pipeline = VK_NULL_HANDLE;
VkCommandPool commandPool = VK_NULL_HANDLE;
VkFence fence = VK_NULL_HANDLE;
void* mappedOutput = nullptr;
Bool fenceWaitTimedOut = false;
const ScopeGuard destroyDeviceObjects([&]() {
if (fenceWaitTimedOut) {
// Match ProbeVulkanTimerQuery: the command may still execute
// after a timeout, so intentionally retain every device-owned
// resource rather than risking a forever wait or UAF in the ICD.
return;
}
vkDeviceWaitIdleFn(device);
if (fence != VK_NULL_HANDLE) vkDestroyFenceFn(device, fence, nullptr);
if (commandPool != VK_NULL_HANDLE) vkDestroyCommandPoolFn(device, commandPool, nullptr);
if (pipeline != VK_NULL_HANDLE) vkDestroyPipelineFn(device, pipeline, nullptr);
if (pipelineLayout != VK_NULL_HANDLE) vkDestroyPipelineLayoutFn(device, pipelineLayout, nullptr);
if (shaderModule != VK_NULL_HANDLE) vkDestroyShaderModuleFn(device, shaderModule, nullptr);
if (descriptorPool != VK_NULL_HANDLE) vkDestroyDescriptorPoolFn(device, descriptorPool, nullptr);
if (descriptorSetLayout != VK_NULL_HANDLE) {
vkDestroyDescriptorSetLayoutFn(device, descriptorSetLayout, nullptr);
}
if (mappedOutput != nullptr) vkUnmapMemoryFn(device, outputMemory);
if (outputBuffer != VK_NULL_HANDLE) vkDestroyBufferFn(device, outputBuffer, nullptr);
if (outputMemory != VK_NULL_HANDLE) vkFreeMemoryFn(device, outputMemory, nullptr);
vkDestroyDeviceFn(device, nullptr);
});
VkQueue queue = VK_NULL_HANDLE;
vkGetDeviceQueueFn(device, computeQueueFamilyIndex, 0, &queue);
if (queue == VK_NULL_HANDLE) {
fail("vkGetDeviceQueue returned a null compute queue");
return;
}
VkBufferCreateInfo bufferInfo{};
bufferInfo.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO;
bufferInfo.size = sizeof(Program203WitnessOutput);
bufferInfo.usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
bufferInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
result = vkCreateBufferFn(device, &bufferInfo, nullptr, &outputBuffer);
if (result != VK_SUCCESS) {
fail(format("vkCreateBuffer(output SSBO) failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkMemoryRequirements memoryRequirements{};
vkGetBufferMemoryRequirementsFn(device, outputBuffer, &memoryRequirements);
VkPhysicalDeviceMemoryProperties memoryProperties{};
vkGetPhysicalDeviceMemoryPropertiesFn(physicalDevice, &memoryProperties);
Uint32 memoryTypeIndex = std::numeric_limits<Uint32>::max();
for (Uint32 index = 0; index < memoryProperties.memoryTypeCount; ++index) {
const Bool compatible = (memoryRequirements.memoryTypeBits & (1u << index)) != 0u;
const VkMemoryPropertyFlags required = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
if (compatible && (memoryProperties.memoryTypes[index].propertyFlags & required) == required) {
memoryTypeIndex = index;
break;
}
}
if (memoryTypeIndex == std::numeric_limits<Uint32>::max()) {
fail("no host-visible/coherent memory type is compatible with the output SSBO");
return;
}
VkMemoryAllocateInfo memoryInfo{};
memoryInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
memoryInfo.allocationSize = memoryRequirements.size;
memoryInfo.memoryTypeIndex = memoryTypeIndex;
result = vkAllocateMemoryFn(device, &memoryInfo, nullptr, &outputMemory);
if (result != VK_SUCCESS) {
fail(format("vkAllocateMemory(output SSBO) failed (VkResult = {})", static_cast<Int>(result)));
return;
}
result = vkBindBufferMemoryFn(device, outputBuffer, outputMemory, 0);
if (result != VK_SUCCESS) {
fail(format("vkBindBufferMemory(output SSBO) failed (VkResult = {})", static_cast<Int>(result)));
return;
}
result = vkMapMemoryFn(device, outputMemory, 0, sizeof(Program203WitnessOutput), 0, &mappedOutput);
if (result != VK_SUCCESS || mappedOutput == nullptr) {
fail(format("vkMapMemory(output SSBO) failed (VkResult = {})", static_cast<Int>(result)));
return;
}
std::memset(mappedOutput, 0xa5, sizeof(Program203WitnessOutput));
VkDescriptorSetLayoutBinding outputBinding{};
outputBinding.binding = 0;
outputBinding.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
outputBinding.descriptorCount = 1;
outputBinding.stageFlags = VK_SHADER_STAGE_COMPUTE_BIT;
VkDescriptorSetLayoutCreateInfo descriptorSetLayoutInfo{};
descriptorSetLayoutInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO;
descriptorSetLayoutInfo.bindingCount = 1;
descriptorSetLayoutInfo.pBindings = &outputBinding;
result = vkCreateDescriptorSetLayoutFn(device, &descriptorSetLayoutInfo, nullptr, &descriptorSetLayout);
if (result != VK_SUCCESS) {
fail(format("vkCreateDescriptorSetLayout failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkDescriptorPoolSize poolSize{};
poolSize.type = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
poolSize.descriptorCount = 1;
VkDescriptorPoolCreateInfo descriptorPoolInfo{};
descriptorPoolInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO;
descriptorPoolInfo.maxSets = 1;
descriptorPoolInfo.poolSizeCount = 1;
descriptorPoolInfo.pPoolSizes = &poolSize;
result = vkCreateDescriptorPoolFn(device, &descriptorPoolInfo, nullptr, &descriptorPool);
if (result != VK_SUCCESS) {
fail(format("vkCreateDescriptorPool failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkDescriptorSet descriptorSet = VK_NULL_HANDLE;
VkDescriptorSetAllocateInfo descriptorSetInfo{};
descriptorSetInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO;
descriptorSetInfo.descriptorPool = descriptorPool;
descriptorSetInfo.descriptorSetCount = 1;
descriptorSetInfo.pSetLayouts = &descriptorSetLayout;
result = vkAllocateDescriptorSetsFn(device, &descriptorSetInfo, &descriptorSet);
if (result != VK_SUCCESS) {
fail(format("vkAllocateDescriptorSets failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkDescriptorBufferInfo outputDescriptor{};
outputDescriptor.buffer = outputBuffer;
outputDescriptor.offset = 0;
outputDescriptor.range = sizeof(Program203WitnessOutput);
VkWriteDescriptorSet descriptorWrite{};
descriptorWrite.sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
descriptorWrite.dstSet = descriptorSet;
descriptorWrite.dstBinding = 0;
descriptorWrite.descriptorCount = 1;
descriptorWrite.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
descriptorWrite.pBufferInfo = &outputDescriptor;
vkUpdateDescriptorSetsFn(device, 1, &descriptorWrite, 0, nullptr);
VkShaderModuleCreateInfo shaderModuleInfo{};
shaderModuleInfo.sType = VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO;
shaderModuleInfo.codeSize = sizeof(kDriverPostProgram203WitnessSpv);
shaderModuleInfo.pCode = kDriverPostProgram203WitnessSpv;
result = vkCreateShaderModuleFn(device, &shaderModuleInfo, nullptr, &shaderModule);
if (result != VK_SUCCESS) {
fail(format("vkCreateShaderModule failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkPipelineLayoutCreateInfo pipelineLayoutInfo{};
pipelineLayoutInfo.sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
pipelineLayoutInfo.setLayoutCount = 1;
pipelineLayoutInfo.pSetLayouts = &descriptorSetLayout;
result = vkCreatePipelineLayoutFn(device, &pipelineLayoutInfo, nullptr, &pipelineLayout);
if (result != VK_SUCCESS) {
fail(format("vkCreatePipelineLayout failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkPipelineShaderStageCreateInfo shaderStage{};
shaderStage.sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
shaderStage.stage = VK_SHADER_STAGE_COMPUTE_BIT;
shaderStage.module = shaderModule;
shaderStage.pName = "main";
VkComputePipelineCreateInfo pipelineInfo{};
pipelineInfo.sType = VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO;
pipelineInfo.stage = shaderStage;
pipelineInfo.layout = pipelineLayout;
result = vkCreateComputePipelinesFn(device, VK_NULL_HANDLE, 1, &pipelineInfo, nullptr, &pipeline);
if (result != VK_SUCCESS) {
fail(format("vkCreateComputePipelines failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkCommandPoolCreateInfo commandPoolInfo{};
commandPoolInfo.sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO;
commandPoolInfo.queueFamilyIndex = computeQueueFamilyIndex;
result = vkCreateCommandPoolFn(device, &commandPoolInfo, nullptr, &commandPool);
if (result != VK_SUCCESS) {
fail(format("vkCreateCommandPool failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkCommandBufferAllocateInfo commandBufferInfo{};
commandBufferInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
commandBufferInfo.commandPool = commandPool;
commandBufferInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
commandBufferInfo.commandBufferCount = 1;
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
result = vkAllocateCommandBuffersFn(device, &commandBufferInfo, &commandBuffer);
if (result != VK_SUCCESS) {
fail(format("vkAllocateCommandBuffers failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkCommandBufferBeginInfo commandBufferBeginInfo{};
commandBufferBeginInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO;
commandBufferBeginInfo.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
result = vkBeginCommandBufferFn(commandBuffer, &commandBufferBeginInfo);
if (result != VK_SUCCESS) {
fail(format("vkBeginCommandBuffer failed (VkResult = {})", static_cast<Int>(result)));
return;
}
vkCmdBindPipelineFn(commandBuffer, VK_PIPELINE_BIND_POINT_COMPUTE, pipeline);
vkCmdBindDescriptorSetsFn(commandBuffer, VK_PIPELINE_BIND_POINT_COMPUTE, pipelineLayout, 0, 1,
&descriptorSet, 0, nullptr);
vkCmdDispatchFn(commandBuffer, 1, 1, 1);
VkBufferMemoryBarrier hostReadBarrier{};
hostReadBarrier.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER;
hostReadBarrier.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT;
hostReadBarrier.dstAccessMask = VK_ACCESS_HOST_READ_BIT;
hostReadBarrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
hostReadBarrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
hostReadBarrier.buffer = outputBuffer;
hostReadBarrier.offset = 0;
hostReadBarrier.size = sizeof(Program203WitnessOutput);
vkCmdPipelineBarrierFn(commandBuffer, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_HOST_BIT, 0,
0, nullptr, 1, &hostReadBarrier, 0, nullptr);
result = vkEndCommandBufferFn(commandBuffer);
if (result != VK_SUCCESS) {
fail(format("vkEndCommandBuffer failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkFenceCreateInfo fenceInfo{};
fenceInfo.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO;
result = vkCreateFenceFn(device, &fenceInfo, nullptr, &fence);
if (result != VK_SUCCESS) {
fail(format("vkCreateFence failed (VkResult = {})", static_cast<Int>(result)));
return;
}
VkSubmitInfo submitInfo{};
submitInfo.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO;
submitInfo.commandBufferCount = 1;
submitInfo.pCommandBuffers = &commandBuffer;
result = vkQueueSubmitFn(queue, 1, &submitInfo, fence);
if (result != VK_SUCCESS) {
fail(format("vkQueueSubmit failed (VkResult = {})", static_cast<Int>(result)));
return;
}
constexpr Uint64 FenceTimeoutNs = 5'000'000'000ull;
result = vkWaitForFencesFn(device, 1, &fence, VK_TRUE, FenceTimeoutNs);
if (result != VK_SUCCESS) {
fenceWaitTimedOut = true;
fail(format("vkWaitForFences did not signal within 5 s (VkResult = {})", static_cast<Int>(result)));
return;
}
Program203WitnessOutput output{};
std::memcpy(&output, mappedOutput, sizeof(output));
const Program203WitnessValidationResult validation = ValidateProgram203Witness(output);
if (!validation.ok) {
fail(validation.detail);
return;
}
builder.Pass(RowName, validation.detail);
}
// Everything the "MobileGL reported ..." rows need from the Vulkan device probe.
struct VulkanProbeSummary {
Bool devicePropsValid = false;
@@ -1461,6 +1898,8 @@ namespace MobileGL::MG_Util::SelfTest {
Bool shaderSubgroupUsable = false;
Bool timerQueriesSupported = false;
Bool samplerAnisotropySupported = false;
Bool drawIndirectFirstInstanceSupported = false;
Bool shaderDrawParametersSupported = false;
};
} // namespace
@@ -1665,6 +2104,7 @@ namespace MobileGL::MG_Util::SelfTest {
VkPhysicalDevice physicalDevice = VK_NULL_HANDLE;
Uint32 graphicsQueueFamilyIndex = 0;
Uint32 graphicsQueueTimestampValidBits = 0;
Uint32 computeQueueFamilyIndex = std::numeric_limits<Uint32>::max();
for (VkPhysicalDevice candidate : devices) {
Uint32 queueFamilyCount = 0;
vkGetPhysicalDeviceQueueFamilyPropertiesFn(candidate, &queueFamilyCount, nullptr);
@@ -1680,6 +2120,13 @@ namespace MobileGL::MG_Util::SelfTest {
}
}
if (physicalDevice != VK_NULL_HANDLE) {
for (Uint32 familyIndex = 0; familyIndex < queueFamilyCount; ++familyIndex) {
const VkQueueFamilyProperties& family = queueFamilies[familyIndex];
if (family.queueCount > 0 && (family.queueFlags & VK_QUEUE_COMPUTE_BIT) != 0) {
computeQueueFamilyIndex = familyIndex;
break;
}
}
break;
}
}
@@ -1744,6 +2191,7 @@ namespace MobileGL::MG_Util::SelfTest {
VkPhysicalDeviceFeatures features{};
vkGetPhysicalDeviceFeaturesFn(physicalDevice, &features);
summary.samplerAnisotropySupported = features.samplerAnisotropy == VK_TRUE;
summary.drawIndirectFirstInstanceSupported = features.drawIndirectFirstInstance == VK_TRUE;
if (features.multiDrawIndirect == VK_TRUE) {
builder.Pass("multiDrawIndirect", "indirect multi-draw batches run as single native commands");
} else {
@@ -1910,6 +2358,7 @@ namespace MobileGL::MG_Util::SelfTest {
builder.Warn("shaderDrawParameters",
"unavailable; shaders using gl_DrawID/gl_BaseInstance will not work");
}
summary.shaderDrawParametersSupported = shaderDrawParameters;
Bool provokingVertexLast = false;
Bool transformFeedbackPreservesProvokingVertex = false;
@@ -2014,13 +2463,15 @@ namespace MobileGL::MG_Util::SelfTest {
"change every N instances change every one");
}
VkPhysicalDeviceSubgroupProperties subgroupProperties{};
Bool subgroupPropertiesAvailable = false;
if (vkGetPhysicalDeviceProperties2Fn != nullptr && properties.apiVersion >= VK_API_VERSION_1_1) {
VkPhysicalDeviceSubgroupProperties subgroupProperties{};
subgroupProperties.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SUBGROUP_PROPERTIES;
VkPhysicalDeviceProperties2 properties2{};
properties2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2;
properties2.pNext = &subgroupProperties;
vkGetPhysicalDeviceProperties2Fn(physicalDevice, &properties2);
subgroupPropertiesAvailable = true;
const Bool subgroupUsable = subgroupProperties.subgroupSize > 0 &&
(subgroupProperties.supportedStages & VK_SHADER_STAGE_COMPUTE_BIT) != 0 &&
(subgroupProperties.supportedOperations & VK_SUBGROUP_FEATURE_BASIC_BIT) != 0;
@@ -2040,6 +2491,9 @@ namespace MobileGL::MG_Util::SelfTest {
builder.Warn("Compute shader subgroup", "subgroup properties could not be queried");
}
ProbeVulkanProgram203Witness(builder, getInstanceProcAddr, instance, physicalDevice, computeQueueFamilyIndex,
properties, subgroupPropertiesAvailable, subgroupProperties);
if (HasVkExtension(deviceExtensions, VK_KHR_DRAW_INDIRECT_COUNT_EXTENSION_NAME)) {
builder.Pass("VK_KHR_draw_indirect_count",
"supported (count-buffer indirect draws run as single native "
@@ -2108,7 +2562,8 @@ namespace MobileGL::MG_Util::SelfTest {
backendApiVersionString = MG_Backend::DirectVulkan::FormatBackendAPIVersionString(
summary.deviceName, summary.apiVersionString, summary.driverVersionString);
advertisedExtensions = JoinAdvertisedExtensions(MG_Backend::DirectVulkan::BuildAdvertisedExtensions(
summary.shaderSubgroupUsable, summary.timerQueriesSupported, summary.samplerAnisotropySupported));
summary.shaderSubgroupUsable, summary.timerQueriesSupported, summary.samplerAnisotropySupported,
summary.drawIndirectFirstInstanceSupported && summary.shaderDrawParametersSupported));
}
AppendMobileGLReportedRows(builder, MG_Backend::DirectVulkan::GetRendererIdentity(), backendApiVersionString,
advertisedExtensions);
@@ -0,0 +1,164 @@
// MobileGL - MobileGL/MG_Util/SelfTest/DriverPostProgram203Witness.comp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Native Vulkan GLSL 450 witness for Program 203's first subgroup reduction.
// It is intentionally independent of the GL 430 integration scenario. The body
// below preserves Program 203's source reduction; the surrounding diagnostics
// only observe its topology and cache handoffs.
#version 450
#extension GL_KHR_shader_subgroup_basic : require
#extension GL_KHR_shader_subgroup_arithmetic : require
layout(local_size_x = 32, local_size_y = 16, local_size_z = 1) in;
const uint kTopologyNonuniformNumSubgroups = 1u << 0u;
const uint kTopologyInvalidNumSubgroups = 1u << 1u;
const uint kTopologyInvalidSubgroupId = 1u << 2u;
const uint kTopologyInvalidSubgroupLane = 1u << 3u;
const uint kWitnessMagic = 0x50323033u;
layout(std430, set = 0, binding = 0) buffer Program203WitnessOutput {
uint magic;
uint topologyFlags;
uint numSubgroups;
uint loopLength;
uint seenSubgroupMask;
uvec4 owner511;
uint lastLaneWriterCount[32];
uint indexedInputTotal[32];
vec2 rawPrefix[32];
vec2 scanCache[6][32];
vec2 finalAverage;
} outWitness;
// Program 203's cache stays separate from all diagnostic shared state. In
// particular, no instrumentation stores through prefixSumCache except source
// writes retained below.
shared vec2 prefixSumCache[32];
shared uint canonicalNumSubgroups;
shared uint topologyFlagsShared;
shared uint seenSubgroupMaskShared;
shared uint lastLaneWriterCountShared[32];
shared uint indexedInputTotalShared[32];
void main() {
const uint localInvocationIndex = gl_LocalInvocationIndex;
// Host memory is deliberately poisoned before dispatch. Initialize only
// shared atomic diagnostic state; owner, average, and magic remain poisoned
// until their required post-source-barrier writes below.
if (localInvocationIndex == 0u) {
canonicalNumSubgroups = 0u;
topologyFlagsShared = 0u;
seenSubgroupMaskShared = 0u;
}
if (localInvocationIndex < 32u) {
lastLaneWriterCountShared[localInvocationIndex] = 0u;
indexedInputTotalShared[localInvocationIndex] = 0u;
}
memoryBarrierShared();
barrier();
// Invocation zero defines the canonical domain. It is broadcast through
// shared memory before every invocation records its own raw observation.
if (localInvocationIndex == 0u) {
canonicalNumSubgroups = gl_NumSubgroups;
outWitness.numSubgroups = gl_NumSubgroups;
}
barrier();
const uint canonicalN = canonicalNumSubgroups;
if (gl_NumSubgroups != canonicalN)
atomicOr(topologyFlagsShared, kTopologyNonuniformNumSubgroups);
if (gl_NumSubgroups < 2u || gl_NumSubgroups > 32u)
atomicOr(topologyFlagsShared, kTopologyInvalidNumSubgroups);
if (gl_SubgroupID >= canonicalN || gl_SubgroupID >= 32u)
atomicOr(topologyFlagsShared, kTopologyInvalidSubgroupId);
if (gl_SubgroupSize == 0u || gl_SubgroupInvocationID >= gl_SubgroupSize)
atomicOr(topologyFlagsShared, kTopologyInvalidSubgroupLane);
// Keep all atomic collection bounded by the canonical valid domain. A
// nonuniform/broken report reaches the uniform safety branch below instead
// of making some lanes return before a barrier.
const bool canonicalDomain = canonicalN >= 2u && canonicalN <= 32u;
const bool idInCanonicalDomain = canonicalDomain && gl_SubgroupID < canonicalN;
if (idInCanonicalDomain) {
atomicOr(seenSubgroupMaskShared, 1u << gl_SubgroupID);
atomicAdd(indexedInputTotalShared[gl_SubgroupID], localInvocationIndex + 1u);
if (gl_SubgroupSize != 0u && gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
atomicAdd(lastLaneWriterCountShared[gl_SubgroupID], 1u);
}
memoryBarrierShared();
barrier();
if (localInvocationIndex == 0u)
outWitness.seenSubgroupMask = seenSubgroupMaskShared;
if (localInvocationIndex < 32u) {
outWitness.lastLaneWriterCount[localInvocationIndex] = lastLaneWriterCountShared[localInvocationIndex];
outWitness.indexedInputTotal[localInvocationIndex] = indexedInputTotalShared[localInvocationIndex];
}
// This branch is uniform after collection and is solely a safety guard for
// broken topology reports. The valid side retains Program 203 verbatim.
const bool sourceDomain = canonicalDomain && topologyFlagsShared == 0u;
if (sourceDomain) {
vec2 sampleLuminance = vec2(float(gl_LocalInvocationIndex + 1u), 0.0);
sampleLuminance = subgroupInclusiveAdd(sampleLuminance);
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = sampleLuminance;
barrier();
if (gl_LocalInvocationIndex < gl_NumSubgroups)
outWitness.rawPrefix[gl_LocalInvocationIndex] = prefixSumCache[gl_LocalInvocationIndex];
barrier();
uint loopLength = uint(findMSB(gl_NumSubgroups));
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
if (gl_LocalInvocationIndex == 0u)
outWitness.loopLength = loopLength;
for (uint scanStage = 0u; scanStage < loopLength; ++scanStage) {
if ((gl_SubgroupID & (1u << scanStage)) > 0u) {
sampleLuminance += prefixSumCache[(gl_SubgroupID >> scanStage << scanStage) - 1u];
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u)
prefixSumCache[gl_SubgroupID] = sampleLuminance;
}
barrier();
if (gl_LocalInvocationIndex < gl_NumSubgroups)
outWitness.scanCache[scanStage][gl_LocalInvocationIndex] =
prefixSumCache[gl_LocalInvocationIndex];
// A second, diagnostic-only barrier prevents a faster invocation
// from entering the next source stage while another reads this cache.
barrier();
}
if (gl_LocalInvocationIndex == 511u)
prefixSumCache[0] = sampleLuminance / 512.0;
barrier();
if (gl_LocalInvocationIndex == 511u) {
outWitness.owner511 = uvec4(gl_SubgroupSize, gl_NumSubgroups, gl_SubgroupID,
gl_SubgroupInvocationID);
outWitness.finalAverage = prefixSumCache[0];
}
}
// Both sides of the uniform branch reach this barrier. The magic is the
// completion latch and therefore cannot be written before the final barrier.
barrier();
if (localInvocationIndex == 0u) {
outWitness.topologyFlags = topologyFlagsShared;
outWitness.magic = kWitnessMagic;
}
}
@@ -0,0 +1,277 @@
// MobileGL - MobileGL/MG_Util/SelfTest/DriverPostProgram203Witness.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include "DriverPostProgram203Witness.h"
#include <bit>
#include <sstream>
#include <utility>
#include <vector>
namespace MobileGL::MG_Util::SelfTest {
namespace {
[[nodiscard]] Program203WitnessValidationResult Failure(Program203WitnessValidationFailure failure,
std::string detail,
std::uint32_t scanStage = 0u,
std::uint32_t subgroup = 0u) {
Program203WitnessValidationResult result;
result.ok = false;
result.failure = failure;
result.scanStage = scanStage;
result.subgroup = subgroup;
result.detail = std::move(detail);
return result;
}
[[nodiscard]] std::uint32_t FloatBits(float value) {
return std::bit_cast<std::uint32_t>(value);
}
[[nodiscard]] bool SameBits(float lhs, float rhs) {
return FloatBits(lhs) == FloatBits(rhs);
}
[[nodiscard]] bool SameBits(const Program203WitnessVec2& lhs, const Program203WitnessVec2& rhs) {
return SameBits(lhs.x, rhs.x) && SameBits(lhs.y, rhs.y);
}
[[nodiscard]] std::string Vec2String(const Program203WitnessVec2& value) {
std::ostringstream output;
output << '(' << value.x << ',' << value.y << ')';
return output.str();
}
[[nodiscard]] std::uint32_t ExpectedSeenSubgroupMask(std::uint32_t numSubgroups) {
return numSubgroups == kProgram203WitnessMaxSubgroups ? 0xffffffffu : (1u << numSubgroups) - 1u;
}
[[nodiscard]] std::string JoinRequirements(const std::vector<std::string>& requirements) {
std::ostringstream output;
for (std::size_t i = 0; i < requirements.size(); ++i) {
if (i != 0u) output << "; ";
output << requirements[i];
}
return output.str();
}
} // namespace
Program203WitnessEligibilityResult
EvaluateProgram203WitnessEligibility(const Program203WitnessLimits& limits) {
// This classification deliberately precedes numeric limits. An absent native
// compute/basic/arithmetic subgroup contract means there is nothing to witness,
// whereas every resource/entry-point failure on a capable device is a POST FAIL.
if (!limits.computeStageSupported || !limits.basicSubgroupSupported || !limits.arithmeticSubgroupSupported) {
std::vector<std::string> missing;
if (!limits.computeStageSupported) missing.emplace_back("VK_SHADER_STAGE_COMPUTE_BIT");
if (!limits.basicSubgroupSupported) missing.emplace_back("VK_SUBGROUP_FEATURE_BASIC_BIT");
if (!limits.arithmeticSubgroupSupported) missing.emplace_back("VK_SUBGROUP_FEATURE_ARITHMETIC_BIT");
return {Program203WitnessEligibility::SkipUnsupportedNativeFeatureSet,
"skipped because the native compute/basic/arithmetic subgroup feature set is unsupported (missing " +
JoinRequirements(missing) + ')'};
}
std::vector<std::string> inadequate;
if (limits.subgroupSize == 0u) {
inadequate.emplace_back("subgroupSize == 0");
}
if (limits.maxComputeWorkGroupInvocations < kProgram203WitnessInvocationCount) {
inadequate.emplace_back("maxComputeWorkGroupInvocations < 512");
}
if (limits.maxComputeWorkGroupSize[0] < 32u || limits.maxComputeWorkGroupSize[1] < 16u ||
limits.maxComputeWorkGroupSize[2] < 1u) {
inadequate.emplace_back("maxComputeWorkGroupSize does not cover 32x16x1");
}
if (limits.maxComputeSharedMemorySize < kProgram203WitnessSharedMemoryBytes) {
inadequate.emplace_back("maxComputeSharedMemorySize < " +
std::to_string(kProgram203WitnessSharedMemoryBytes));
}
if (limits.maxPerStageDescriptorStorageBuffers < 1u) {
inadequate.emplace_back("maxPerStageDescriptorStorageBuffers < 1");
}
if (limits.maxDescriptorSetStorageBuffers < 1u) {
inadequate.emplace_back("maxDescriptorSetStorageBuffers < 1");
}
if (limits.maxBoundDescriptorSets < 1u) {
inadequate.emplace_back("maxBoundDescriptorSets < 1");
}
if (limits.maxStorageBufferRange < sizeof(Program203WitnessOutput)) {
inadequate.emplace_back("maxStorageBufferRange < " +
std::to_string(sizeof(Program203WitnessOutput)));
}
if (!inadequate.empty()) {
return {Program203WitnessEligibility::FailInadequateLimits,
"insufficient Vulkan limits for a 32x16x1 workgroup, one output SSBO, and " +
std::to_string(kProgram203WitnessSharedMemoryBytes) + " bytes of shared memory: " +
JoinRequirements(inadequate)};
}
return {Program203WitnessEligibility::Execute, {}};
}
std::uint32_t ComputeProgram203WitnessLoopLength(std::uint32_t numSubgroups) {
if (numSubgroups < 2u || numSubgroups > kProgram203WitnessMaxSubgroups) return 0u;
// Exact C++ spelling of the source's findMSB-based calculation. In
// particular, its final iteration for powers of two is intentional.
std::uint32_t loopLength = 0u;
for (std::uint32_t value = numSubgroups; value > 1u; value >>= 1u) {
++loopLength;
}
loopLength += static_cast<std::uint32_t>(numSubgroups - (1u << (loopLength - 1u)) > 0u);
return loopLength;
}
Program203WitnessValidationResult ValidateProgram203Witness(const Program203WitnessOutput& output) {
// 1. Completion. A poisoned or unwritten result must never turn into a
// topology diagnosis, because it says nothing about execution.
if (output.magic != kProgram203WitnessMagic) {
std::ostringstream detail;
detail << "completion: magic was 0x" << std::hex << output.magic << ", expected 0x"
<< kProgram203WitnessMagic;
return Failure(Program203WitnessValidationFailure::Completion, detail.str());
}
// 2. Observed topology. All checks consume observations written by the
// shader, rather than inferring subgroup layout from invocation indices.
const std::uint32_t numSubgroups = output.numSubgroups;
if (numSubgroups < 2u || numSubgroups > kProgram203WitnessMaxSubgroups) {
std::ostringstream detail;
detail << "topology: canonical gl_NumSubgroups=" << numSubgroups << " is outside [2, 32]";
return Failure(Program203WitnessValidationFailure::Topology, detail.str());
}
if ((output.topologyFlags & Program203WitnessNonuniformNumSubgroups) != 0u) {
return Failure(Program203WitnessValidationFailure::Topology,
"topology: gl_NumSubgroups differed across workgroup");
}
if ((output.topologyFlags & Program203WitnessInvalidNumSubgroups) != 0u) {
return Failure(Program203WitnessValidationFailure::Topology,
"topology: an invocation reported gl_NumSubgroups outside [2, 32]");
}
if ((output.topologyFlags & Program203WitnessInvalidSubgroupId) != 0u) {
return Failure(Program203WitnessValidationFailure::Topology,
"topology: an invocation reported an invalid gl_SubgroupID");
}
if ((output.topologyFlags & Program203WitnessInvalidSubgroupLane) != 0u) {
return Failure(Program203WitnessValidationFailure::Topology,
"topology: an invocation reported an invalid subgroup lane");
}
if ((output.topologyFlags & ~(Program203WitnessNonuniformNumSubgroups |
Program203WitnessInvalidNumSubgroups |
Program203WitnessInvalidSubgroupId |
Program203WitnessInvalidSubgroupLane)) != 0u) {
std::ostringstream detail;
detail << "topology: unknown topology flags 0x" << std::hex << output.topologyFlags;
return Failure(Program203WitnessValidationFailure::Topology, detail.str());
}
const std::uint32_t expectedMask = ExpectedSeenSubgroupMask(numSubgroups);
if (output.seenSubgroupMask != expectedMask) {
std::ostringstream detail;
detail << "topology: seen subgroup-ID mask was 0x" << std::hex << output.seenSubgroupMask
<< ", expected 0x" << expectedMask;
return Failure(Program203WitnessValidationFailure::Topology, detail.str());
}
const std::uint32_t expectedLoopLength = ComputeProgram203WitnessLoopLength(numSubgroups);
if (output.loopLength != expectedLoopLength) {
std::ostringstream detail;
detail << "topology: loopLength was " << std::dec << output.loopLength << ", expected "
<< expectedLoopLength;
return Failure(Program203WitnessValidationFailure::Topology, detail.str());
}
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
if (output.lastLaneWriterCount[subgroup] != 1u) {
std::ostringstream detail;
detail << "topology: subgroup " << subgroup << " has "
<< output.lastLaneWriterCount[subgroup] << " source last-lane writers, expected exactly 1";
return Failure(Program203WitnessValidationFailure::Topology, detail.str(), 0u, subgroup);
}
}
if (output.owner511.y != numSubgroups) {
std::ostringstream detail;
detail << "final owner: invocation 511 reported gl_NumSubgroups=" << output.owner511.y << ", expected "
<< numSubgroups;
return Failure(Program203WitnessValidationFailure::FinalOwner, detail.str());
}
if (output.owner511.z != numSubgroups - 1u) {
std::ostringstream detail;
detail << "final owner: invocation 511 is not in the highest subgroup (id" << output.owner511.z
<< ", expected id" << (numSubgroups - 1u) << ')';
return Failure(Program203WitnessValidationFailure::FinalOwner, detail.str());
}
if (output.owner511.x == 0u || output.owner511.w != output.owner511.x - 1u) {
std::ostringstream detail;
detail << "final owner: invocation 511 is not the last lane of highest subgroup (size "
<< output.owner511.x << ", lane " << output.owner511.w << ')';
return Failure(Program203WitnessValidationFailure::FinalOwner, detail.str());
}
// 3. Initial subgroup handoff. The atomic scalar totals are independent
// of subgroupInclusiveAdd; their sum and the cache values establish that
// the final lanes handed off the native vector inclusive-add results.
std::uint64_t indexedTotal = 0u;
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
indexedTotal += output.indexedInputTotal[subgroup];
}
if (indexedTotal != 131328u) {
std::ostringstream detail;
detail << "initial subgroup handoff: indexed input total was " << indexedTotal << ", expected 131328";
return Failure(Program203WitnessValidationFailure::InitialSubgroupHandoff, detail.str());
}
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
const Program203WitnessVec2 expected = {static_cast<float>(output.indexedInputTotal[subgroup]), 0.0f};
if (!SameBits(output.rawPrefix[subgroup], expected)) {
std::ostringstream detail;
detail << "initial subgroup handoff: subgroup " << subgroup << " rawPrefix observed "
<< Vec2String(output.rawPrefix[subgroup]) << ", expected " << Vec2String(expected);
return Failure(Program203WitnessValidationFailure::InitialSubgroupHandoff, detail.str(), 0u,
subgroup);
}
}
// 4. Source scan. Do not substitute a conventional scan: this reproduces
// the source cache index expression and stage ordering word for word.
std::array<Program203WitnessVec2, kProgram203WitnessMaxSubgroups> expectedCache = output.rawPrefix;
for (std::uint32_t scanStage = 0u; scanStage < expectedLoopLength; ++scanStage) {
auto cacheAfterStage = expectedCache;
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
if ((subgroup & (1u << scanStage)) > 0u) {
const std::uint32_t sourceCacheIndex = (subgroup >> scanStage << scanStage) - 1u;
cacheAfterStage[subgroup].x += expectedCache[sourceCacheIndex].x;
cacheAfterStage[subgroup].y += expectedCache[sourceCacheIndex].y;
}
}
expectedCache = cacheAfterStage;
for (std::uint32_t subgroup = 0u; subgroup < numSubgroups; ++subgroup) {
if (!SameBits(output.scanCache[scanStage][subgroup], expectedCache[subgroup])) {
std::ostringstream detail;
detail << "source scan stage " << scanStage << ", subgroup " << subgroup << ": observed "
<< Vec2String(output.scanCache[scanStage][subgroup]) << ", expected "
<< Vec2String(expectedCache[subgroup]);
return Failure(Program203WitnessValidationFailure::SourceScan, detail.str(), scanStage, subgroup);
}
}
}
// 5. The owner contract was checked above with the other topology facts;
// this final result remains a separate exact-vector check.
const Program203WitnessVec2 expectedAverage = {256.5f, 0.0f};
if (!SameBits(output.finalAverage, expectedAverage)) {
std::ostringstream detail;
detail << "final average: observed " << Vec2String(output.finalAverage) << ", expected "
<< Vec2String(expectedAverage);
return Failure(Program203WitnessValidationFailure::FinalAverage, detail.str());
}
std::ostringstream detail;
detail << "N=" << numSubgroups << ", owner511=id" << output.owner511.z << "/lane" << output.owner511.w
<< ", " << expectedLoopLength << " scan stages, average=" << Vec2String(output.finalAverage);
Program203WitnessValidationResult result;
result.ok = true;
result.failure = Program203WitnessValidationFailure::None;
result.detail = detail.str();
return result;
}
} // namespace MobileGL::MG_Util::SelfTest
@@ -0,0 +1,151 @@
// MobileGL - MobileGL/MG_Util/SelfTest/DriverPostProgram203Witness.h
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Compact, native-Vulkan Program-203 first-reduction witness ABI and its pure
// validator. The types below deliberately mirror DriverPostProgram203Witness.comp's
// single std430 storage block; changing either side requires updating the static
// layout assertions here.
#pragma once
#include <array>
#include <cstddef>
#include <cstdint>
#include <string>
#include <type_traits>
namespace MobileGL::MG_Util::SelfTest {
constexpr std::uint32_t kProgram203WitnessMagic = 0x50323033u; // "P203"
constexpr std::uint32_t kProgram203WitnessInvocationCount = 512u;
constexpr std::uint32_t kProgram203WitnessMaxSubgroups = 32u;
constexpr std::uint32_t kProgram203WitnessMaxScanStages = 6u;
// These bit values are shared with the GLSL source. They document failures in
// topology observations rather than guessing a topology from local IDs on the host.
enum Program203WitnessTopologyFlag : std::uint32_t {
Program203WitnessNonuniformNumSubgroups = 1u << 0u,
Program203WitnessInvalidNumSubgroups = 1u << 1u,
Program203WitnessInvalidSubgroupId = 1u << 2u,
Program203WitnessInvalidSubgroupLane = 1u << 3u,
};
struct alignas(8) Program203WitnessVec2 {
float x;
float y;
};
struct alignas(16) Program203WitnessUVec4 {
std::uint32_t x;
std::uint32_t y;
std::uint32_t z;
std::uint32_t w;
};
// std430 layout of DriverPostProgram203Witness.comp's Program203WitnessOutput block.
struct alignas(16) Program203WitnessOutput {
std::uint32_t magic;
std::uint32_t topologyFlags;
std::uint32_t numSubgroups;
std::uint32_t loopLength;
std::uint32_t seenSubgroupMask;
Program203WitnessUVec4 owner511;
std::array<std::uint32_t, kProgram203WitnessMaxSubgroups> lastLaneWriterCount;
std::array<std::uint32_t, kProgram203WitnessMaxSubgroups> indexedInputTotal;
std::array<Program203WitnessVec2, kProgram203WitnessMaxSubgroups> rawPrefix;
std::array<std::array<Program203WitnessVec2, kProgram203WitnessMaxSubgroups>,
kProgram203WitnessMaxScanStages>
scanCache;
Program203WitnessVec2 finalAverage;
};
static_assert(std::is_standard_layout_v<Program203WitnessVec2>);
static_assert(std::is_standard_layout_v<Program203WitnessUVec4>);
static_assert(std::is_standard_layout_v<Program203WitnessOutput>);
static_assert(sizeof(Program203WitnessVec2) == 8u);
static_assert(alignof(Program203WitnessVec2) == 8u);
static_assert(sizeof(Program203WitnessUVec4) == 16u);
static_assert(alignof(Program203WitnessUVec4) == 16u);
static_assert(offsetof(Program203WitnessOutput, magic) == 0u);
static_assert(offsetof(Program203WitnessOutput, topologyFlags) == 4u);
static_assert(offsetof(Program203WitnessOutput, numSubgroups) == 8u);
static_assert(offsetof(Program203WitnessOutput, loopLength) == 12u);
static_assert(offsetof(Program203WitnessOutput, seenSubgroupMask) == 16u);
static_assert(offsetof(Program203WitnessOutput, owner511) == 32u);
static_assert(offsetof(Program203WitnessOutput, lastLaneWriterCount) == 48u);
static_assert(offsetof(Program203WitnessOutput, indexedInputTotal) == 176u);
static_assert(offsetof(Program203WitnessOutput, rawPrefix) == 304u);
static_assert(offsetof(Program203WitnessOutput, scanCache) == 560u);
static_assert(offsetof(Program203WitnessOutput, finalAverage) == 2096u);
static_assert(sizeof(Program203WitnessOutput) == 2112u);
// The witness uses prefixSumCache[32], three scalar shared diagnostics, and
// two 32-entry scalar diagnostic arrays in the GLSL source. Keep this
// independent of the output SSBO size.
constexpr std::uint32_t kProgram203WitnessSharedMemoryBytes =
kProgram203WitnessMaxSubgroups * sizeof(Program203WitnessVec2) +
3u * sizeof(std::uint32_t) +
2u * kProgram203WitnessMaxSubgroups * sizeof(std::uint32_t);
enum class Program203WitnessEligibility {
Execute,
SkipUnsupportedNativeFeatureSet,
FailInadequateLimits,
};
// The raw physical-device conditions needed by the native witness. This is
// intentionally distinct from MobileGL's advertised-extension policy.
struct Program203WitnessLimits {
bool computeStageSupported = false;
bool basicSubgroupSupported = false;
bool arithmeticSubgroupSupported = false;
std::uint32_t subgroupSize = 0u;
std::uint32_t maxComputeWorkGroupInvocations = 0u;
std::array<std::uint32_t, 3> maxComputeWorkGroupSize{};
std::uint32_t maxComputeSharedMemorySize = 0u;
std::uint32_t maxPerStageDescriptorStorageBuffers = 0u;
std::uint32_t maxDescriptorSetStorageBuffers = 0u;
std::uint32_t maxBoundDescriptorSets = 0u;
std::uint64_t maxStorageBufferRange = 0u;
};
struct Program203WitnessEligibilityResult {
Program203WitnessEligibility eligibility = Program203WitnessEligibility::FailInadequateLimits;
std::string detail;
};
enum class Program203WitnessValidationFailure {
None,
Completion,
Topology,
InitialSubgroupHandoff,
SourceScan,
FinalOwner,
FinalAverage,
};
struct Program203WitnessValidationResult {
bool ok = false;
Program203WitnessValidationFailure failure = Program203WitnessValidationFailure::Completion;
std::uint32_t scanStage = 0u;
std::uint32_t subgroup = 0u;
std::string detail;
};
[[nodiscard]] Program203WitnessEligibilityResult
EvaluateProgram203WitnessEligibility(const Program203WitnessLimits& limits);
// Mirrors the source's findMSB expression for valid N in [2, 32].
[[nodiscard]] std::uint32_t ComputeProgram203WitnessLoopLength(std::uint32_t numSubgroups);
[[nodiscard]] Program203WitnessValidationResult
ValidateProgram203Witness(const Program203WitnessOutput& output);
} // namespace MobileGL::MG_Util::SelfTest
@@ -0,0 +1,291 @@
// MobileGL - MobileGL/MG_Util/SelfTest/DriverPostProgram203WitnessSpv.h
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// Generated from DriverPostProgram203Witness.comp with:
// glslangValidator --target-env vulkan1.1 -V DriverPostProgram203Witness.comp
// Validated with spirv-val --target-env vulkan1.1. Do not edit words by hand.
#pragma once
#include <cstddef>
#include <cstdint>
namespace MobileGL::MG_Util::SelfTest {
inline constexpr std::uint32_t kDriverPostProgram203WitnessSpv[] = {
0x07230203u, 0x00010300u, 0x0008000bu, 0x00000145u, 0x00000000u, 0x00020011u, 0x00000001u, 0x00020011u,
0x0000003du, 0x00020011u, 0x0000003fu, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x000a000fu, 0x00000005u, 0x00000004u, 0x6e69616du,
0x00000000u, 0x0000000au, 0x0000002au, 0x00000050u, 0x0000005eu, 0x00000064u, 0x00060010u, 0x00000004u,
0x00000011u, 0x00000020u, 0x00000010u, 0x00000001u, 0x00030003u, 0x00000002u, 0x000001c2u, 0x000a0004u,
0x4b5f4c47u, 0x735f5248u, 0x65646168u, 0x75735f72u, 0x6f726762u, 0x615f7075u, 0x68746972u, 0x6974656du,
0x00000063u, 0x00090004u, 0x4b5f4c47u, 0x735f5248u, 0x65646168u, 0x75735f72u, 0x6f726762u, 0x625f7075u,
0x63697361u, 0x00000000u, 0x00040005u, 0x00000004u, 0x6e69616du, 0x00000000u, 0x00080005u, 0x00000008u,
0x61636f6cu, 0x766e496cu, 0x7461636fu, 0x496e6f69u, 0x7865646eu, 0x00000000u, 0x00080005u, 0x0000000au,
0x4c5f6c67u, 0x6c61636fu, 0x6f766e49u, 0x69746163u, 0x6e496e6fu, 0x00786564u, 0x00080005u, 0x00000013u,
0x6f6e6163u, 0x6163696eu, 0x6d754e6cu, 0x67627553u, 0x70756f72u, 0x00000073u, 0x00070005u, 0x00000014u,
0x6f706f74u, 0x79676f6cu, 0x67616c46u, 0x61685373u, 0x00646572u, 0x00080005u, 0x00000015u, 0x6e656573u,
0x67627553u, 0x70756f72u, 0x6b73614du, 0x72616853u, 0x00006465u, 0x00090005u, 0x0000001du, 0x7473616cu,
0x656e614cu, 0x74697257u, 0x6f437265u, 0x53746e75u, 0x65726168u, 0x00000064u, 0x00080005u, 0x00000020u,
0x65646e69u, 0x49646578u, 0x7475706eu, 0x61746f54u, 0x6168536cu, 0x00646572u, 0x00060005u, 0x0000002au,
0x4e5f6c67u, 0x75536d75u, 0x6f726762u, 0x00737075u, 0x00080005u, 0x00000035u, 0x676f7250u, 0x326d6172u,
0x69573330u, 0x73656e74u, 0x74754f73u, 0x00747570u, 0x00050006u, 0x00000035u, 0x00000000u, 0x6967616du,
0x00000063u, 0x00070006u, 0x00000035u, 0x00000001u, 0x6f706f74u, 0x79676f6cu, 0x67616c46u, 0x00000073u,
0x00070006u, 0x00000035u, 0x00000002u, 0x536d756eu, 0x72676275u, 0x7370756fu, 0x00000000u, 0x00060006u,
0x00000035u, 0x00000003u, 0x706f6f6cu, 0x676e654cu, 0x00006874u, 0x00080006u, 0x00000035u, 0x00000004u,
0x6e656573u, 0x67627553u, 0x70756f72u, 0x6b73614du, 0x00000000u, 0x00060006u, 0x00000035u, 0x00000005u,
0x656e776fu, 0x31313572u, 0x00000000u, 0x00080006u, 0x00000035u, 0x00000006u, 0x7473616cu, 0x656e614cu,
0x74697257u, 0x6f437265u, 0x00746e75u, 0x00080006u, 0x00000035u, 0x00000007u, 0x65646e69u, 0x49646578u,
0x7475706eu, 0x61746f54u, 0x0000006cu, 0x00060006u, 0x00000035u, 0x00000008u, 0x50776172u, 0x69666572u,
0x00000078u, 0x00060006u, 0x00000035u, 0x00000009u, 0x6e616373u, 0x68636143u, 0x00000065u, 0x00070006u,
0x00000035u, 0x0000000au, 0x616e6966u, 0x6576416cu, 0x65676172u, 0x00000000u, 0x00050005u, 0x00000037u,
0x5774756fu, 0x656e7469u, 0x00007373u, 0x00050005u, 0x0000003du, 0x6f6e6163u, 0x6163696eu, 0x00004e6cu,
0x00060005u, 0x00000050u, 0x535f6c67u, 0x72676275u, 0x4970756fu, 0x00000044u, 0x00060005u, 0x0000005eu,
0x535f6c67u, 0x72676275u, 0x5370756fu, 0x00657a69u, 0x00080005u, 0x00000064u, 0x535f6c67u, 0x72676275u,
0x4970756fu, 0x636f766eu, 0x6f697461u, 0x0044496eu, 0x00060005u, 0x0000006eu, 0x6f6e6163u, 0x6163696eu,
0x6d6f446cu, 0x006e6961u, 0x00070005u, 0x00000074u, 0x6e496469u, 0x6f6e6143u, 0x6163696eu, 0x6d6f446cu,
0x006e6961u, 0x00060005u, 0x000000acu, 0x72756f73u, 0x6f446563u, 0x6e69616du, 0x00000000u, 0x00060005u,
0x000000b7u, 0x706d6173u, 0x754c656cu, 0x616e696du, 0x0065636eu, 0x00060005u, 0x000000c8u, 0x66657270u,
0x75537869u, 0x6361436du, 0x00006568u, 0x00050005u, 0x000000d9u, 0x706f6f6cu, 0x676e654cu, 0x00006874u,
0x00050005u, 0x000000edu, 0x6e616373u, 0x67617453u, 0x00000065u, 0x00040047u, 0x0000000au, 0x0000000bu,
0x0000001du, 0x00040047u, 0x0000002au, 0x0000000bu, 0x00000026u, 0x00040047u, 0x0000002du, 0x00000006u,
0x00000004u, 0x00040047u, 0x0000002eu, 0x00000006u, 0x00000004u, 0x00040047u, 0x00000031u, 0x00000006u,
0x00000008u, 0x00040047u, 0x00000032u, 0x00000006u, 0x00000008u, 0x00040047u, 0x00000034u, 0x00000006u,
0x00000100u, 0x00030047u, 0x00000035u, 0x00000002u, 0x00050048u, 0x00000035u, 0x00000000u, 0x00000023u,
0x00000000u, 0x00050048u, 0x00000035u, 0x00000001u, 0x00000023u, 0x00000004u, 0x00050048u, 0x00000035u,
0x00000002u, 0x00000023u, 0x00000008u, 0x00050048u, 0x00000035u, 0x00000003u, 0x00000023u, 0x0000000cu,
0x00050048u, 0x00000035u, 0x00000004u, 0x00000023u, 0x00000010u, 0x00050048u, 0x00000035u, 0x00000005u,
0x00000023u, 0x00000020u, 0x00050048u, 0x00000035u, 0x00000006u, 0x00000023u, 0x00000030u, 0x00050048u,
0x00000035u, 0x00000007u, 0x00000023u, 0x000000b0u, 0x00050048u, 0x00000035u, 0x00000008u, 0x00000023u,
0x00000130u, 0x00050048u, 0x00000035u, 0x00000009u, 0x00000023u, 0x00000230u, 0x00050048u, 0x00000035u,
0x0000000au, 0x00000023u, 0x00000830u, 0x00040047u, 0x00000037u, 0x00000021u, 0x00000000u, 0x00040047u,
0x00000037u, 0x00000022u, 0x00000000u, 0x00040047u, 0x00000050u, 0x0000000bu, 0x00000028u, 0x00030047u,
0x0000005eu, 0x00000000u, 0x00040047u, 0x0000005eu, 0x0000000bu, 0x00000024u, 0x00030047u, 0x0000005fu,
0x00000000u, 0x00030047u, 0x00000064u, 0x00000000u, 0x00040047u, 0x00000064u, 0x0000000bu, 0x00000029u,
0x00030047u, 0x00000065u, 0x00000000u, 0x00030047u, 0x00000066u, 0x00000000u, 0x00030047u, 0x00000087u,
0x00000000u, 0x00030047u, 0x0000008bu, 0x00000000u, 0x00030047u, 0x0000008cu, 0x00000000u, 0x00030047u,
0x0000008du, 0x00000000u, 0x00030047u, 0x000000c0u, 0x00000000u, 0x00030047u, 0x000000c1u, 0x00000000u,
0x00030047u, 0x000000c2u, 0x00000000u, 0x00030047u, 0x00000107u, 0x00000000u, 0x00030047u, 0x00000108u,
0x00000000u, 0x00030047u, 0x00000109u, 0x00000000u, 0x00030047u, 0x0000012fu, 0x00000000u, 0x00030047u,
0x00000132u, 0x00000000u, 0x00040047u, 0x00000144u, 0x0000000bu, 0x00000019u, 0x00020013u, 0x00000002u,
0x00030021u, 0x00000003u, 0x00000002u, 0x00040015u, 0x00000006u, 0x00000020u, 0x00000000u, 0x00040020u,
0x00000007u, 0x00000007u, 0x00000006u, 0x00040020u, 0x00000009u, 0x00000001u, 0x00000006u, 0x0004003bu,
0x00000009u, 0x0000000au, 0x00000001u, 0x0004002bu, 0x00000006u, 0x0000000du, 0x00000000u, 0x00020014u,
0x0000000eu, 0x00040020u, 0x00000012u, 0x00000004u, 0x00000006u, 0x0004003bu, 0x00000012u, 0x00000013u,
0x00000004u, 0x0004003bu, 0x00000012u, 0x00000014u, 0x00000004u, 0x0004003bu, 0x00000012u, 0x00000015u,
0x00000004u, 0x0004002bu, 0x00000006u, 0x00000017u, 0x00000020u, 0x0004001cu, 0x0000001bu, 0x00000006u,
0x00000017u, 0x00040020u, 0x0000001cu, 0x00000004u, 0x0000001bu, 0x0004003bu, 0x0000001cu, 0x0000001du,
0x00000004u, 0x0004003bu, 0x0000001cu, 0x00000020u, 0x00000004u, 0x0004002bu, 0x00000006u, 0x00000023u,
0x00000001u, 0x0004002bu, 0x00000006u, 0x00000024u, 0x00000108u, 0x0004002bu, 0x00000006u, 0x00000025u,
0x00000002u, 0x0004003bu, 0x00000009u, 0x0000002au, 0x00000001u, 0x00040017u, 0x0000002cu, 0x00000006u,
0x00000004u, 0x0004001cu, 0x0000002du, 0x00000006u, 0x00000017u, 0x0004001cu, 0x0000002eu, 0x00000006u,
0x00000017u, 0x00030016u, 0x0000002fu, 0x00000020u, 0x00040017u, 0x00000030u, 0x0000002fu, 0x00000002u,
0x0004001cu, 0x00000031u, 0x00000030u, 0x00000017u, 0x0004001cu, 0x00000032u, 0x00000030u, 0x00000017u,
0x0004002bu, 0x00000006u, 0x00000033u, 0x00000006u, 0x0004001cu, 0x00000034u, 0x00000032u, 0x00000033u,
0x000d001eu, 0x00000035u, 0x00000006u, 0x00000006u, 0x00000006u, 0x00000006u, 0x00000006u, 0x0000002cu,
0x0000002du, 0x0000002eu, 0x00000031u, 0x00000034u, 0x00000030u, 0x00040020u, 0x00000036u, 0x0000000cu,
0x00000035u, 0x0004003bu, 0x00000036u, 0x00000037u, 0x0000000cu, 0x00040015u, 0x00000038u, 0x00000020u,
0x00000001u, 0x0004002bu, 0x00000038u, 0x00000039u, 0x00000002u, 0x00040020u, 0x0000003bu, 0x0000000cu,
0x00000006u, 0x0004003bu, 0x00000009u, 0x00000050u, 0x00000001u, 0x0004002bu, 0x00000006u, 0x0000005cu,
0x00000004u, 0x0004003bu, 0x00000009u, 0x0000005eu, 0x00000001u, 0x0004003bu, 0x00000009u, 0x00000064u,
0x00000001u, 0x0004002bu, 0x00000006u, 0x0000006bu, 0x00000008u, 0x00040020u, 0x0000006du, 0x00000007u,
0x0000000eu, 0x0004002bu, 0x00000038u, 0x00000099u, 0x00000004u, 0x0004002bu, 0x00000038u, 0x000000a0u,
0x00000006u, 0x0004002bu, 0x00000038u, 0x000000a6u, 0x00000007u, 0x00040020u, 0x000000b6u, 0x00000007u,
0x00000030u, 0x0004002bu, 0x0000002fu, 0x000000bbu, 0x00000000u, 0x0004002bu, 0x00000006u, 0x000000beu,
0x00000003u, 0x0004001cu, 0x000000c6u, 0x00000030u, 0x00000017u, 0x00040020u, 0x000000c7u, 0x00000004u,
0x000000c6u, 0x0004003bu, 0x000000c7u, 0x000000c8u, 0x00000004u, 0x00040020u, 0x000000cbu, 0x00000004u,
0x00000030u, 0x0004002bu, 0x00000038u, 0x000000d2u, 0x00000008u, 0x00040020u, 0x000000d7u, 0x0000000cu,
0x00000030u, 0x0004002bu, 0x00000038u, 0x000000eau, 0x00000003u, 0x0004002bu, 0x00000038u, 0x00000115u,
0x00000009u, 0x0004002bu, 0x00000038u, 0x0000011du, 0x00000001u, 0x0004002bu, 0x00000006u, 0x00000120u,
0x000001ffu, 0x0004002bu, 0x00000038u, 0x00000124u, 0x00000000u, 0x0004002bu, 0x0000002fu, 0x00000126u,
0x44000000u, 0x0004002bu, 0x00000038u, 0x0000012eu, 0x00000005u, 0x00040020u, 0x00000134u, 0x0000000cu,
0x0000002cu, 0x0004002bu, 0x00000038u, 0x00000136u, 0x0000000au, 0x0004002bu, 0x00000006u, 0x00000140u,
0x50323033u, 0x00040017u, 0x00000142u, 0x00000006u, 0x00000003u, 0x0004002bu, 0x00000006u, 0x00000143u,
0x00000010u, 0x0006002cu, 0x00000142u, 0x00000144u, 0x00000017u, 0x00000143u, 0x00000023u, 0x00050036u,
0x00000002u, 0x00000004u, 0x00000000u, 0x00000003u, 0x000200f8u, 0x00000005u, 0x0004003bu, 0x00000007u,
0x00000008u, 0x00000007u, 0x0004003bu, 0x00000007u, 0x0000003du, 0x00000007u, 0x0004003bu, 0x0000006du,
0x0000006eu, 0x00000007u, 0x0004003bu, 0x0000006du, 0x00000074u, 0x00000007u, 0x0004003bu, 0x0000006du,
0x000000acu, 0x00000007u, 0x0004003bu, 0x000000b6u, 0x000000b7u, 0x00000007u, 0x0004003bu, 0x00000007u,
0x000000d9u, 0x00000007u, 0x0004003bu, 0x00000007u, 0x000000edu, 0x00000007u, 0x0004003du, 0x00000006u,
0x0000000bu, 0x0000000au, 0x0003003eu, 0x00000008u, 0x0000000bu, 0x0004003du, 0x00000006u, 0x0000000cu,
0x00000008u, 0x000500aau, 0x0000000eu, 0x0000000fu, 0x0000000cu, 0x0000000du, 0x000300f7u, 0x00000011u,
0x00000000u, 0x000400fau, 0x0000000fu, 0x00000010u, 0x00000011u, 0x000200f8u, 0x00000010u, 0x0003003eu,
0x00000013u, 0x0000000du, 0x0003003eu, 0x00000014u, 0x0000000du, 0x0003003eu, 0x00000015u, 0x0000000du,
0x000200f9u, 0x00000011u, 0x000200f8u, 0x00000011u, 0x0004003du, 0x00000006u, 0x00000016u, 0x00000008u,
0x000500b0u, 0x0000000eu, 0x00000018u, 0x00000016u, 0x00000017u, 0x000300f7u, 0x0000001au, 0x00000000u,
0x000400fau, 0x00000018u, 0x00000019u, 0x0000001au, 0x000200f8u, 0x00000019u, 0x0004003du, 0x00000006u,
0x0000001eu, 0x00000008u, 0x00050041u, 0x00000012u, 0x0000001fu, 0x0000001du, 0x0000001eu, 0x0003003eu,
0x0000001fu, 0x0000000du, 0x0004003du, 0x00000006u, 0x00000021u, 0x00000008u, 0x00050041u, 0x00000012u,
0x00000022u, 0x00000020u, 0x00000021u, 0x0003003eu, 0x00000022u, 0x0000000du, 0x000200f9u, 0x0000001au,
0x000200f8u, 0x0000001au, 0x000300e1u, 0x00000023u, 0x00000024u, 0x000400e0u, 0x00000025u, 0x00000025u,
0x00000024u, 0x0004003du, 0x00000006u, 0x00000026u, 0x00000008u, 0x000500aau, 0x0000000eu, 0x00000027u,
0x00000026u, 0x0000000du, 0x000300f7u, 0x00000029u, 0x00000000u, 0x000400fau, 0x00000027u, 0x00000028u,
0x00000029u, 0x000200f8u, 0x00000028u, 0x0004003du, 0x00000006u, 0x0000002bu, 0x0000002au, 0x0003003eu,
0x00000013u, 0x0000002bu, 0x0004003du, 0x00000006u, 0x0000003au, 0x0000002au, 0x00050041u, 0x0000003bu,
0x0000003cu, 0x00000037u, 0x00000039u, 0x0003003eu, 0x0000003cu, 0x0000003au, 0x000200f9u, 0x00000029u,
0x000200f8u, 0x00000029u, 0x000400e0u, 0x00000025u, 0x00000025u, 0x00000024u, 0x0004003du, 0x00000006u,
0x0000003eu, 0x00000013u, 0x0003003eu, 0x0000003du, 0x0000003eu, 0x0004003du, 0x00000006u, 0x0000003fu,
0x0000002au, 0x0004003du, 0x00000006u, 0x00000040u, 0x0000003du, 0x000500abu, 0x0000000eu, 0x00000041u,
0x0000003fu, 0x00000040u, 0x000300f7u, 0x00000043u, 0x00000000u, 0x000400fau, 0x00000041u, 0x00000042u,
0x00000043u, 0x000200f8u, 0x00000042u, 0x000700f1u, 0x00000006u, 0x00000044u, 0x00000014u, 0x00000023u,
0x0000000du, 0x00000023u, 0x000200f9u, 0x00000043u, 0x000200f8u, 0x00000043u, 0x0004003du, 0x00000006u,
0x00000045u, 0x0000002au, 0x000500b0u, 0x0000000eu, 0x00000046u, 0x00000045u, 0x00000025u, 0x000400a8u,
0x0000000eu, 0x00000047u, 0x00000046u, 0x000300f7u, 0x00000049u, 0x00000000u, 0x000400fau, 0x00000047u,
0x00000048u, 0x00000049u, 0x000200f8u, 0x00000048u, 0x0004003du, 0x00000006u, 0x0000004au, 0x0000002au,
0x000500acu, 0x0000000eu, 0x0000004bu, 0x0000004au, 0x00000017u, 0x000200f9u, 0x00000049u, 0x000200f8u,
0x00000049u, 0x000700f5u, 0x0000000eu, 0x0000004cu, 0x00000046u, 0x00000043u, 0x0000004bu, 0x00000048u,
0x000300f7u, 0x0000004eu, 0x00000000u, 0x000400fau, 0x0000004cu, 0x0000004du, 0x0000004eu, 0x000200f8u,
0x0000004du, 0x000700f1u, 0x00000006u, 0x0000004fu, 0x00000014u, 0x00000023u, 0x0000000du, 0x00000025u,
0x000200f9u, 0x0000004eu, 0x000200f8u, 0x0000004eu, 0x0004003du, 0x00000006u, 0x00000051u, 0x00000050u,
0x0004003du, 0x00000006u, 0x00000052u, 0x0000003du, 0x000500aeu, 0x0000000eu, 0x00000053u, 0x00000051u,
0x00000052u, 0x000400a8u, 0x0000000eu, 0x00000054u, 0x00000053u, 0x000300f7u, 0x00000056u, 0x00000000u,
0x000400fau, 0x00000054u, 0x00000055u, 0x00000056u, 0x000200f8u, 0x00000055u, 0x0004003du, 0x00000006u,
0x00000057u, 0x00000050u, 0x000500aeu, 0x0000000eu, 0x00000058u, 0x00000057u, 0x00000017u, 0x000200f9u,
0x00000056u, 0x000200f8u, 0x00000056u, 0x000700f5u, 0x0000000eu, 0x00000059u, 0x00000053u, 0x0000004eu,
0x00000058u, 0x00000055u, 0x000300f7u, 0x0000005bu, 0x00000000u, 0x000400fau, 0x00000059u, 0x0000005au,
0x0000005bu, 0x000200f8u, 0x0000005au, 0x000700f1u, 0x00000006u, 0x0000005du, 0x00000014u, 0x00000023u,
0x0000000du, 0x0000005cu, 0x000200f9u, 0x0000005bu, 0x000200f8u, 0x0000005bu, 0x0004003du, 0x00000006u,
0x0000005fu, 0x0000005eu, 0x000500aau, 0x0000000eu, 0x00000060u, 0x0000005fu, 0x0000000du, 0x000400a8u,
0x0000000eu, 0x00000061u, 0x00000060u, 0x000300f7u, 0x00000063u, 0x00000000u, 0x000400fau, 0x00000061u,
0x00000062u, 0x00000063u, 0x000200f8u, 0x00000062u, 0x0004003du, 0x00000006u, 0x00000065u, 0x00000064u,
0x0004003du, 0x00000006u, 0x00000066u, 0x0000005eu, 0x000500aeu, 0x0000000eu, 0x00000067u, 0x00000065u,
0x00000066u, 0x000200f9u, 0x00000063u, 0x000200f8u, 0x00000063u, 0x000700f5u, 0x0000000eu, 0x00000068u,
0x00000060u, 0x0000005bu, 0x00000067u, 0x00000062u, 0x000300f7u, 0x0000006au, 0x00000000u, 0x000400fau,
0x00000068u, 0x00000069u, 0x0000006au, 0x000200f8u, 0x00000069u, 0x000700f1u, 0x00000006u, 0x0000006cu,
0x00000014u, 0x00000023u, 0x0000000du, 0x0000006bu, 0x000200f9u, 0x0000006au, 0x000200f8u, 0x0000006au,
0x0004003du, 0x00000006u, 0x0000006fu, 0x0000003du, 0x000500aeu, 0x0000000eu, 0x00000070u, 0x0000006fu,
0x00000025u, 0x0004003du, 0x00000006u, 0x00000071u, 0x0000003du, 0x000500b2u, 0x0000000eu, 0x00000072u,
0x00000071u, 0x00000017u, 0x000500a7u, 0x0000000eu, 0x00000073u, 0x00000070u, 0x00000072u, 0x0003003eu,
0x0000006eu, 0x00000073u, 0x0004003du, 0x0000000eu, 0x00000075u, 0x0000006eu, 0x000300f7u, 0x00000077u,
0x00000000u, 0x000400fau, 0x00000075u, 0x00000076u, 0x00000077u, 0x000200f8u, 0x00000076u, 0x0004003du,
0x00000006u, 0x00000078u, 0x00000050u, 0x0004003du, 0x00000006u, 0x00000079u, 0x0000003du, 0x000500b0u,
0x0000000eu, 0x0000007au, 0x00000078u, 0x00000079u, 0x000200f9u, 0x00000077u, 0x000200f8u, 0x00000077u,
0x000700f5u, 0x0000000eu, 0x0000007bu, 0x00000075u, 0x0000006au, 0x0000007au, 0x00000076u, 0x0003003eu,
0x00000074u, 0x0000007bu, 0x0004003du, 0x0000000eu, 0x0000007cu, 0x00000074u, 0x000300f7u, 0x0000007eu,
0x00000000u, 0x000400fau, 0x0000007cu, 0x0000007du, 0x0000007eu, 0x000200f8u, 0x0000007du, 0x0004003du,
0x00000006u, 0x0000007fu, 0x00000050u, 0x000500c4u, 0x00000006u, 0x00000080u, 0x00000023u, 0x0000007fu,
0x000700f1u, 0x00000006u, 0x00000081u, 0x00000015u, 0x00000023u, 0x0000000du, 0x00000080u, 0x0004003du,
0x00000006u, 0x00000082u, 0x00000050u, 0x00050041u, 0x00000012u, 0x00000083u, 0x00000020u, 0x00000082u,
0x0004003du, 0x00000006u, 0x00000084u, 0x00000008u, 0x00050080u, 0x00000006u, 0x00000085u, 0x00000084u,
0x00000023u, 0x000700eau, 0x00000006u, 0x00000086u, 0x00000083u, 0x00000023u, 0x0000000du, 0x00000085u,
0x0004003du, 0x00000006u, 0x00000087u, 0x0000005eu, 0x000500abu, 0x0000000eu, 0x00000088u, 0x00000087u,
0x0000000du, 0x000300f7u, 0x0000008au, 0x00000000u, 0x000400fau, 0x00000088u, 0x00000089u, 0x0000008au,
0x000200f8u, 0x00000089u, 0x0004003du, 0x00000006u, 0x0000008bu, 0x00000064u, 0x0004003du, 0x00000006u,
0x0000008cu, 0x0000005eu, 0x00050082u, 0x00000006u, 0x0000008du, 0x0000008cu, 0x00000023u, 0x000500aau,
0x0000000eu, 0x0000008eu, 0x0000008bu, 0x0000008du, 0x000200f9u, 0x0000008au, 0x000200f8u, 0x0000008au,
0x000700f5u, 0x0000000eu, 0x0000008fu, 0x00000088u, 0x0000007du, 0x0000008eu, 0x00000089u, 0x000300f7u,
0x00000091u, 0x00000000u, 0x000400fau, 0x0000008fu, 0x00000090u, 0x00000091u, 0x000200f8u, 0x00000090u,
0x0004003du, 0x00000006u, 0x00000092u, 0x00000050u, 0x00050041u, 0x00000012u, 0x00000093u, 0x0000001du,
0x00000092u, 0x000700eau, 0x00000006u, 0x00000094u, 0x00000093u, 0x00000023u, 0x0000000du, 0x00000023u,
0x000200f9u, 0x00000091u, 0x000200f8u, 0x00000091u, 0x000200f9u, 0x0000007eu, 0x000200f8u, 0x0000007eu,
0x000300e1u, 0x00000023u, 0x00000024u, 0x000400e0u, 0x00000025u, 0x00000025u, 0x00000024u, 0x0004003du,
0x00000006u, 0x00000095u, 0x00000008u, 0x000500aau, 0x0000000eu, 0x00000096u, 0x00000095u, 0x0000000du,
0x000300f7u, 0x00000098u, 0x00000000u, 0x000400fau, 0x00000096u, 0x00000097u, 0x00000098u, 0x000200f8u,
0x00000097u, 0x0004003du, 0x00000006u, 0x0000009au, 0x00000015u, 0x00050041u, 0x0000003bu, 0x0000009bu,
0x00000037u, 0x00000099u, 0x0003003eu, 0x0000009bu, 0x0000009au, 0x000200f9u, 0x00000098u, 0x000200f8u,
0x00000098u, 0x0004003du, 0x00000006u, 0x0000009cu, 0x00000008u, 0x000500b0u, 0x0000000eu, 0x0000009du,
0x0000009cu, 0x00000017u, 0x000300f7u, 0x0000009fu, 0x00000000u, 0x000400fau, 0x0000009du, 0x0000009eu,
0x0000009fu, 0x000200f8u, 0x0000009eu, 0x0004003du, 0x00000006u, 0x000000a1u, 0x00000008u, 0x0004003du,
0x00000006u, 0x000000a2u, 0x00000008u, 0x00050041u, 0x00000012u, 0x000000a3u, 0x0000001du, 0x000000a2u,
0x0004003du, 0x00000006u, 0x000000a4u, 0x000000a3u, 0x00060041u, 0x0000003bu, 0x000000a5u, 0x00000037u,
0x000000a0u, 0x000000a1u, 0x0003003eu, 0x000000a5u, 0x000000a4u, 0x0004003du, 0x00000006u, 0x000000a7u,
0x00000008u, 0x0004003du, 0x00000006u, 0x000000a8u, 0x00000008u, 0x00050041u, 0x00000012u, 0x000000a9u,
0x00000020u, 0x000000a8u, 0x0004003du, 0x00000006u, 0x000000aau, 0x000000a9u, 0x00060041u, 0x0000003bu,
0x000000abu, 0x00000037u, 0x000000a6u, 0x000000a7u, 0x0003003eu, 0x000000abu, 0x000000aau, 0x000200f9u,
0x0000009fu, 0x000200f8u, 0x0000009fu, 0x0004003du, 0x0000000eu, 0x000000adu, 0x0000006eu, 0x000300f7u,
0x000000afu, 0x00000000u, 0x000400fau, 0x000000adu, 0x000000aeu, 0x000000afu, 0x000200f8u, 0x000000aeu,
0x0004003du, 0x00000006u, 0x000000b0u, 0x00000014u, 0x000500aau, 0x0000000eu, 0x000000b1u, 0x000000b0u,
0x0000000du, 0x000200f9u, 0x000000afu, 0x000200f8u, 0x000000afu, 0x000700f5u, 0x0000000eu, 0x000000b2u,
0x000000adu, 0x0000009fu, 0x000000b1u, 0x000000aeu, 0x0003003eu, 0x000000acu, 0x000000b2u, 0x0004003du,
0x0000000eu, 0x000000b3u, 0x000000acu, 0x000300f7u, 0x000000b5u, 0x00000000u, 0x000400fau, 0x000000b3u,
0x000000b4u, 0x000000b5u, 0x000200f8u, 0x000000b4u, 0x0004003du, 0x00000006u, 0x000000b8u, 0x0000000au,
0x00050080u, 0x00000006u, 0x000000b9u, 0x000000b8u, 0x00000023u, 0x00040070u, 0x0000002fu, 0x000000bau,
0x000000b9u, 0x00050050u, 0x00000030u, 0x000000bcu, 0x000000bau, 0x000000bbu, 0x0003003eu, 0x000000b7u,
0x000000bcu, 0x0004003du, 0x00000030u, 0x000000bdu, 0x000000b7u, 0x0006015eu, 0x00000030u, 0x000000bfu,
0x000000beu, 0x00000001u, 0x000000bdu, 0x0003003eu, 0x000000b7u, 0x000000bfu, 0x0004003du, 0x00000006u,
0x000000c0u, 0x00000064u, 0x0004003du, 0x00000006u, 0x000000c1u, 0x0000005eu, 0x00050082u, 0x00000006u,
0x000000c2u, 0x000000c1u, 0x00000023u, 0x000500aau, 0x0000000eu, 0x000000c3u, 0x000000c0u, 0x000000c2u,
0x000300f7u, 0x000000c5u, 0x00000000u, 0x000400fau, 0x000000c3u, 0x000000c4u, 0x000000c5u, 0x000200f8u,
0x000000c4u, 0x0004003du, 0x00000006u, 0x000000c9u, 0x00000050u, 0x0004003du, 0x00000030u, 0x000000cau,
0x000000b7u, 0x00050041u, 0x000000cbu, 0x000000ccu, 0x000000c8u, 0x000000c9u, 0x0003003eu, 0x000000ccu,
0x000000cau, 0x000200f9u, 0x000000c5u, 0x000200f8u, 0x000000c5u, 0x000400e0u, 0x00000025u, 0x00000025u,
0x00000024u, 0x0004003du, 0x00000006u, 0x000000cdu, 0x0000000au, 0x0004003du, 0x00000006u, 0x000000ceu,
0x0000002au, 0x000500b0u, 0x0000000eu, 0x000000cfu, 0x000000cdu, 0x000000ceu, 0x000300f7u, 0x000000d1u,
0x00000000u, 0x000400fau, 0x000000cfu, 0x000000d0u, 0x000000d1u, 0x000200f8u, 0x000000d0u, 0x0004003du,
0x00000006u, 0x000000d3u, 0x0000000au, 0x0004003du, 0x00000006u, 0x000000d4u, 0x0000000au, 0x00050041u,
0x000000cbu, 0x000000d5u, 0x000000c8u, 0x000000d4u, 0x0004003du, 0x00000030u, 0x000000d6u, 0x000000d5u,
0x00060041u, 0x000000d7u, 0x000000d8u, 0x00000037u, 0x000000d2u, 0x000000d3u, 0x0003003eu, 0x000000d8u,
0x000000d6u, 0x000200f9u, 0x000000d1u, 0x000200f8u, 0x000000d1u, 0x000400e0u, 0x00000025u, 0x00000025u,
0x00000024u, 0x0004003du, 0x00000006u, 0x000000dau, 0x0000002au, 0x0006000cu, 0x00000038u, 0x000000dbu,
0x00000001u, 0x0000004bu, 0x000000dau, 0x0004007cu, 0x00000006u, 0x000000dcu, 0x000000dbu, 0x0003003eu,
0x000000d9u, 0x000000dcu, 0x0004003du, 0x00000006u, 0x000000ddu, 0x0000002au, 0x0004003du, 0x00000006u,
0x000000deu, 0x000000d9u, 0x00050082u, 0x00000006u, 0x000000dfu, 0x000000deu, 0x00000023u, 0x000500c4u,
0x00000006u, 0x000000e0u, 0x00000023u, 0x000000dfu, 0x00050082u, 0x00000006u, 0x000000e1u, 0x000000ddu,
0x000000e0u, 0x000500acu, 0x0000000eu, 0x000000e2u, 0x000000e1u, 0x0000000du, 0x000600a9u, 0x00000006u,
0x000000e3u, 0x000000e2u, 0x00000023u, 0x0000000du, 0x0004003du, 0x00000006u, 0x000000e4u, 0x000000d9u,
0x00050080u, 0x00000006u, 0x000000e5u, 0x000000e4u, 0x000000e3u, 0x0003003eu, 0x000000d9u, 0x000000e5u,
0x0004003du, 0x00000006u, 0x000000e6u, 0x0000000au, 0x000500aau, 0x0000000eu, 0x000000e7u, 0x000000e6u,
0x0000000du, 0x000300f7u, 0x000000e9u, 0x00000000u, 0x000400fau, 0x000000e7u, 0x000000e8u, 0x000000e9u,
0x000200f8u, 0x000000e8u, 0x0004003du, 0x00000006u, 0x000000ebu, 0x000000d9u, 0x00050041u, 0x0000003bu,
0x000000ecu, 0x00000037u, 0x000000eau, 0x0003003eu, 0x000000ecu, 0x000000ebu, 0x000200f9u, 0x000000e9u,
0x000200f8u, 0x000000e9u, 0x0003003eu, 0x000000edu, 0x0000000du, 0x000200f9u, 0x000000eeu, 0x000200f8u,
0x000000eeu, 0x000400f6u, 0x000000f0u, 0x000000f1u, 0x00000000u, 0x000200f9u, 0x000000f2u, 0x000200f8u,
0x000000f2u, 0x0004003du, 0x00000006u, 0x000000f3u, 0x000000edu, 0x0004003du, 0x00000006u, 0x000000f4u,
0x000000d9u, 0x000500b0u, 0x0000000eu, 0x000000f5u, 0x000000f3u, 0x000000f4u, 0x000400fau, 0x000000f5u,
0x000000efu, 0x000000f0u, 0x000200f8u, 0x000000efu, 0x0004003du, 0x00000006u, 0x000000f6u, 0x00000050u,
0x0004003du, 0x00000006u, 0x000000f7u, 0x000000edu, 0x000500c4u, 0x00000006u, 0x000000f8u, 0x00000023u,
0x000000f7u, 0x000500c7u, 0x00000006u, 0x000000f9u, 0x000000f6u, 0x000000f8u, 0x000500acu, 0x0000000eu,
0x000000fau, 0x000000f9u, 0x0000000du, 0x000300f7u, 0x000000fcu, 0x00000000u, 0x000400fau, 0x000000fau,
0x000000fbu, 0x000000fcu, 0x000200f8u, 0x000000fbu, 0x0004003du, 0x00000006u, 0x000000fdu, 0x00000050u,
0x0004003du, 0x00000006u, 0x000000feu, 0x000000edu, 0x000500c2u, 0x00000006u, 0x000000ffu, 0x000000fdu,
0x000000feu, 0x0004003du, 0x00000006u, 0x00000100u, 0x000000edu, 0x000500c4u, 0x00000006u, 0x00000101u,
0x000000ffu, 0x00000100u, 0x00050082u, 0x00000006u, 0x00000102u, 0x00000101u, 0x00000023u, 0x00050041u,
0x000000cbu, 0x00000103u, 0x000000c8u, 0x00000102u, 0x0004003du, 0x00000030u, 0x00000104u, 0x00000103u,
0x0004003du, 0x00000030u, 0x00000105u, 0x000000b7u, 0x00050081u, 0x00000030u, 0x00000106u, 0x00000105u,
0x00000104u, 0x0003003eu, 0x000000b7u, 0x00000106u, 0x0004003du, 0x00000006u, 0x00000107u, 0x00000064u,
0x0004003du, 0x00000006u, 0x00000108u, 0x0000005eu, 0x00050082u, 0x00000006u, 0x00000109u, 0x00000108u,
0x00000023u, 0x000500aau, 0x0000000eu, 0x0000010au, 0x00000107u, 0x00000109u, 0x000300f7u, 0x0000010cu,
0x00000000u, 0x000400fau, 0x0000010au, 0x0000010bu, 0x0000010cu, 0x000200f8u, 0x0000010bu, 0x0004003du,
0x00000006u, 0x0000010du, 0x00000050u, 0x0004003du, 0x00000030u, 0x0000010eu, 0x000000b7u, 0x00050041u,
0x000000cbu, 0x0000010fu, 0x000000c8u, 0x0000010du, 0x0003003eu, 0x0000010fu, 0x0000010eu, 0x000200f9u,
0x0000010cu, 0x000200f8u, 0x0000010cu, 0x000200f9u, 0x000000fcu, 0x000200f8u, 0x000000fcu, 0x000400e0u,
0x00000025u, 0x00000025u, 0x00000024u, 0x0004003du, 0x00000006u, 0x00000110u, 0x0000000au, 0x0004003du,
0x00000006u, 0x00000111u, 0x0000002au, 0x000500b0u, 0x0000000eu, 0x00000112u, 0x00000110u, 0x00000111u,
0x000300f7u, 0x00000114u, 0x00000000u, 0x000400fau, 0x00000112u, 0x00000113u, 0x00000114u, 0x000200f8u,
0x00000113u, 0x0004003du, 0x00000006u, 0x00000116u, 0x000000edu, 0x0004003du, 0x00000006u, 0x00000117u,
0x0000000au, 0x0004003du, 0x00000006u, 0x00000118u, 0x0000000au, 0x00050041u, 0x000000cbu, 0x00000119u,
0x000000c8u, 0x00000118u, 0x0004003du, 0x00000030u, 0x0000011au, 0x00000119u, 0x00070041u, 0x000000d7u,
0x0000011bu, 0x00000037u, 0x00000115u, 0x00000116u, 0x00000117u, 0x0003003eu, 0x0000011bu, 0x0000011au,
0x000200f9u, 0x00000114u, 0x000200f8u, 0x00000114u, 0x000400e0u, 0x00000025u, 0x00000025u, 0x00000024u,
0x000200f9u, 0x000000f1u, 0x000200f8u, 0x000000f1u, 0x0004003du, 0x00000006u, 0x0000011cu, 0x000000edu,
0x00050080u, 0x00000006u, 0x0000011eu, 0x0000011cu, 0x0000011du, 0x0003003eu, 0x000000edu, 0x0000011eu,
0x000200f9u, 0x000000eeu, 0x000200f8u, 0x000000f0u, 0x0004003du, 0x00000006u, 0x0000011fu, 0x0000000au,
0x000500aau, 0x0000000eu, 0x00000121u, 0x0000011fu, 0x00000120u, 0x000300f7u, 0x00000123u, 0x00000000u,
0x000400fau, 0x00000121u, 0x00000122u, 0x00000123u, 0x000200f8u, 0x00000122u, 0x0004003du, 0x00000030u,
0x00000125u, 0x000000b7u, 0x00050050u, 0x00000030u, 0x00000127u, 0x00000126u, 0x00000126u, 0x00050088u,
0x00000030u, 0x00000128u, 0x00000125u, 0x00000127u, 0x00050041u, 0x000000cbu, 0x00000129u, 0x000000c8u,
0x00000124u, 0x0003003eu, 0x00000129u, 0x00000128u, 0x000200f9u, 0x00000123u, 0x000200f8u, 0x00000123u,
0x000400e0u, 0x00000025u, 0x00000025u, 0x00000024u, 0x0004003du, 0x00000006u, 0x0000012au, 0x0000000au,
0x000500aau, 0x0000000eu, 0x0000012bu, 0x0000012au, 0x00000120u, 0x000300f7u, 0x0000012du, 0x00000000u,
0x000400fau, 0x0000012bu, 0x0000012cu, 0x0000012du, 0x000200f8u, 0x0000012cu, 0x0004003du, 0x00000006u,
0x0000012fu, 0x0000005eu, 0x0004003du, 0x00000006u, 0x00000130u, 0x0000002au, 0x0004003du, 0x00000006u,
0x00000131u, 0x00000050u, 0x0004003du, 0x00000006u, 0x00000132u, 0x00000064u, 0x00070050u, 0x0000002cu,
0x00000133u, 0x0000012fu, 0x00000130u, 0x00000131u, 0x00000132u, 0x00050041u, 0x00000134u, 0x00000135u,
0x00000037u, 0x0000012eu, 0x0003003eu, 0x00000135u, 0x00000133u, 0x00050041u, 0x000000cbu, 0x00000137u,
0x000000c8u, 0x00000124u, 0x0004003du, 0x00000030u, 0x00000138u, 0x00000137u, 0x00050041u, 0x000000d7u,
0x00000139u, 0x00000037u, 0x00000136u, 0x0003003eu, 0x00000139u, 0x00000138u, 0x000200f9u, 0x0000012du,
0x000200f8u, 0x0000012du, 0x000200f9u, 0x000000b5u, 0x000200f8u, 0x000000b5u, 0x000400e0u, 0x00000025u,
0x00000025u, 0x00000024u, 0x0004003du, 0x00000006u, 0x0000013au, 0x00000008u, 0x000500aau, 0x0000000eu,
0x0000013bu, 0x0000013au, 0x0000000du, 0x000300f7u, 0x0000013du, 0x00000000u, 0x000400fau, 0x0000013bu,
0x0000013cu, 0x0000013du, 0x000200f8u, 0x0000013cu, 0x0004003du, 0x00000006u, 0x0000013eu, 0x00000014u,
0x00050041u, 0x0000003bu, 0x0000013fu, 0x00000037u, 0x0000011du, 0x0003003eu, 0x0000013fu, 0x0000013eu,
0x00050041u, 0x0000003bu, 0x00000141u, 0x00000037u, 0x00000124u, 0x0003003eu, 0x00000141u, 0x00000140u,
0x000200f9u, 0x0000013du, 0x000200f8u, 0x0000013du, 0x000100fdu, 0x00010038u,
};
inline constexpr std::size_t kDriverPostProgram203WitnessSpvWordCount =
sizeof(kDriverPostProgram203WitnessSpv) / sizeof(kDriverPostProgram203WitnessSpv[0]);
} // namespace MobileGL::MG_Util::SelfTest
@@ -38,7 +38,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
HashBytes(state, env.advertisedExtensions.data(),
env.advertisedExtensions.size() * sizeof(GLExtension));
}
HashValue(state, env.subgroupPrefixScanQuirk);
return state;
}
@@ -73,8 +72,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
kFrontendMaxComputeWorkGroupInvocations)
: kFrontendMaxComputeWorkGroupInvocations;
env->subgroupPrefixScanQuirk = MG_Config::Features.SubgroupPrefixScanQuirk;
env->fingerprint = ComputeCompileEnvFingerprint(*env);
return env;
}
@@ -84,7 +81,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
// computed, and this must not run before MG_Config is loaded.
static const SharedPtr<const CompileEnv> kDefault = [] {
auto env = MakeShared<CompileEnv>();
env->subgroupPrefixScanQuirk = MG_Config::Features.SubgroupPrefixScanQuirk;
env->fingerprint = ComputeCompileEnvFingerprint(*env);
return SharedPtr<const CompileEnv>(Move(env));
}();
@@ -12,9 +12,9 @@
#include <MG_Backend/BackendObject.h>
namespace MobileGL::MG_Util::ShaderTranspiler {
// Everything the shader compile/link pipeline reads from OUTSIDE its own (stage, source)
// inputs: backend identity, backend limits, the advertised extension list, and the one
// config quirk the source rewriter branches on.
// everything outside (stage, source) this reads - advertised extensions and backend limits -
// so the transformation is a pure function of its three arguments and can run on a worker
// thread.
//
// Why it exists (P1): every one of those reads is a reach-back into
// MG_Backend::pActiveBackendObject / gBackendFunctionsTable, and one of them
@@ -46,9 +46,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
MG_Backend::DynamicBackendParameters params{}; // by value, never by reference
Vector<GLExtension> advertisedExtensions;
// --- config the source rewriter branches on ---
MG_Config::QuirkOverride subgroupPrefixScanQuirk = MG_Config::QuirkOverride::Auto;
Uint64 fingerprint = 0; // set by CaptureCompileEnv()
Bool HasBackend() const { return backend != BackendType::Unknown; }
@@ -61,7 +61,7 @@ namespace MobileGL {
"imageAtomicXor", "imageLoad", "imageSize", "imageStore", "imulExtended",
"intBitsToFloat", "interpolateAtCentroid", "interpolateAtOffset",
"interpolateAtSample", "inverse", "inversesqrt", "isinf", "isnan",
"ldexp", "length", "lessThan", "lessThanEqual", "log", "log2",
"ldexp", "length", "length_squared", "lessThan", "lessThanEqual", "log", "log2",
"matrixCompMult", "max", "max3", "memoryBarrier",
"memoryBarrierAtomicCounter", "memoryBarrierBuffer", "memoryBarrierImage",
"memoryBarrierShared", "mid3", "min", "min3", "mix", "mod", "modf",
@@ -25,6 +25,7 @@
#include "SpirvPasses/SplitArrayVertexInputsPass.h"
#include "SpirvPasses/RebaseInstanceIndexPass.h"
#include "SpirvPasses/ZeroBaseVertexPass.h"
#include "SpirvPasses/DeriveNumSubgroupsPass.h"
#include "SpirvPasses/NormalizeRectCoordinatesPass.h"
#include "SpirvPasses/Lower1DArrayImagesPass.h"
#include "SpirvPasses/BakeImageFormatsPass.h"
@@ -369,12 +370,6 @@ namespace MobileGL {
return allSpirv;
}
// -1 unresolved, 0 off, 1 on. Resolved once from MOBILEGL_VALIDATE_SPIRV on first
// use. A live getenv rather than an MG_Config::Features field, for the same reason
// Config.h already exempts MOBILEGL_LOG_FILE_PATH: suites like SpirvPassTest never
// run MobileGL::Initialize(), and every Initialize() re-runs MG_ConfigLoader::Init,
// which would clobber a programmatic override stored in the feature table.
static std::atomic<int> g_validateSpirv{-1};
// Total validation failures observed this process. This latch - not the wrappers'
// return values - is the test-lane signal: validation must never change what a
// wrapper returns, or the validating lanes would render differently from the
@@ -383,28 +378,6 @@ namespace MobileGL {
static std::atomic<Uint64> g_spirvValidationFailures{0};
namespace {
// Test lanes (desktop/CI/WSL) validate by default; device builds do not -
// validation costs real time per module, and on device the driver is the
// final validator anyway. MOBILEGL_VALIDATE_SPIRV overrides in either
// direction, using the ConfigLoader truthy rule.
constexpr bool kValidateSpirvDefault =
#if defined(__ANDROID__)
false;
#else
true;
#endif
bool IsTruthySpirvEnvValue(const char* value) {
if (value == nullptr || value[0] == '\0') {
return false;
}
String lowered(value);
for (auto& c : lowered) {
c = static_cast<char>(std::tolower(static_cast<unsigned char>(c)));
}
return lowered != "0" && lowered != "false";
}
// spirv-tools' validator lazily constructs function-local static tables on
// its first run, which on this codebase happens on a ShaderCompilePool
// worker. Function-local statics are destroyed in reverse construction
@@ -445,11 +418,6 @@ namespace MobileGL {
tools.Validate(warmup);
}
std::atexit(+[] {
// Flip validation off first: a validator table this warmup does
// not know about (a future spirv-tools bump) would still be
// destroyed before this handler, and workers must stop entering
// Validate before the drain waits for them.
g_validateSpirv.store(0, std::memory_order_release);
Async::ShaderCompilePool::StopAndDrainProcessPoolAtExit();
});
});
@@ -478,10 +446,10 @@ namespace MobileGL {
// Validation is decoupled from control flow on purpose: a failure logs and
// bumps the latch, and the caller proceeds exactly as the shipping (non-
// validating) configuration would. Tests assert on the latch delta.
void ValidateOrLatch(const char* site, const Vector<Uint32>& binary) {
if (!ShaderCompiler::SpirvValidationEnabled()) {
return;
}
void ValidateOrLatch(const char* site, const Vector<Uint32>& binary,
const bool enableSpirvValidation) {
if (!enableSpirvValidation) return;
PinValidatorTablesForProcessExit();
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
tools.SetMessageConsumer(MakeSpirvMessageConsumer(site));
if (!tools.Validate(binary)) {
@@ -502,39 +470,23 @@ namespace MobileGL {
// spirv-tools drops pass diagnostics on the floor.
bool RunOptimizerChecked(const char* site, spvtools::Optimizer& optimizer,
const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary, const bool validateOutput,
const bool enableSpirvValidation) {
spvtools::OptimizerOptions options;
options.set_run_validator(false);
optimizer.SetMessageConsumer(MakeSpirvMessageConsumer(site));
if (!optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options)) {
return false;
}
ValidateOrLatch(site, outputBinary);
if (validateOutput) {
ValidateOrLatch(site, outputBinary, enableSpirvValidation);
}
return true;
}
} // namespace
bool ShaderCompiler::SpirvValidationEnabled() {
int state = g_validateSpirv.load(std::memory_order_acquire);
if (state < 0) {
const char* env = std::getenv("MOBILEGL_VALIDATE_SPIRV");
const bool resolved = env != nullptr ? IsTruthySpirvEnvValue(env) : kValidateSpirvDefault;
int expected = -1;
g_validateSpirv.compare_exchange_strong(expected, resolved ? 1 : 0,
std::memory_order_acq_rel);
state = g_validateSpirv.load(std::memory_order_acquire);
if (state == 1) {
PinValidatorTablesForProcessExit();
}
}
return state == 1;
}
void ShaderCompiler::SetSpirvValidationEnabled(bool enabled) {
g_validateSpirv.store(enabled ? 1 : 0, std::memory_order_release);
if (enabled) {
PinValidatorTablesForProcessExit();
}
void ShaderCompiler::PrepareSpirvValidation() {
PinValidatorTablesForProcessExit();
}
Uint64 ShaderCompiler::NoteSpirvValidationFailure() {
@@ -604,16 +556,19 @@ namespace MobileGL {
}
bool ShaderCompiler::DemoteFloat64ToFloat32(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(DemoteFloat64Pass::CreateDemoteFloat64Pass());
return RunOptimizerChecked("DemoteFloat64ToFloat32", optimizer, inputBinary, outputBinary);
return RunOptimizerChecked("DemoteFloat64ToFloat32", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::SanitizeAndOptimizeBinary(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool validateOutput,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
@@ -663,38 +618,41 @@ namespace MobileGL {
optimizer.RegisterPass(DemoteFloat64Pass::CreateDemoteFloat64Pass());
return RunOptimizerChecked("SanitizeAndOptimizeBinary", optimizer, inputBinary,
outputBinary);
outputBinary, validateOutput, enableSpirvValidation);
}
bool ShaderCompiler::LowerDrawParametersForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(LowerDrawParametersPass::CreateLowerDrawParametersPass());
return RunOptimizerChecked("LowerDrawParametersForEssl", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::SplitArrayVertexInputsForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(SplitArrayVertexInputsPass::CreateSplitArrayVertexInputsPass());
return RunOptimizerChecked("SplitArrayVertexInputsForEssl", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::BakeImageFormatsForEssl(const Vector<Uint32>& inputBinary,
const UnorderedMap<String, Uint>& glFormatByName,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
if (glFormatByName.empty()) return false;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(BakeImageFormatsPass::CreateBakeImageFormatsPass(glFormatByName));
return RunOptimizerChecked("BakeImageFormatsForEssl", optimizer, inputBinary, outputBinary);
return RunOptimizerChecked("BakeImageFormatsForEssl", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::DeclaresFormatlessStorageImage(const Vector<Uint32>& binary) {
@@ -718,7 +676,8 @@ namespace MobileGL {
bool ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(const Vector<Uint32>& inputBinary,
const std::set<String>& blockNames,
std::set<String>& flattenedBlockNames,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
if (blockNames.empty()) return false;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
@@ -726,7 +685,7 @@ namespace MobileGL {
blockNames, &flattenedBlockNames));
return RunOptimizerChecked("FlattenXfbInterfaceBlocksForEssl", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::RewriteXfbCaptureNameForFlattenedBlock(
@@ -736,48 +695,53 @@ namespace MobileGL {
}
bool ShaderCompiler::PackDoubleVertexInputsForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(PackDoubleVertexInputsPass::CreatePackDoubleVertexInputsPass());
return RunOptimizerChecked("PackDoubleVertexInputsForVulkan", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::StripUboMemberRelaxedPrecisionForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(
StripUboMemberRelaxedPrecisionPass::CreateStripUboMemberRelaxedPrecisionPass());
return RunOptimizerChecked("StripUboMemberRelaxedPrecisionForEssl", optimizer,
inputBinary, outputBinary);
inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(StripNoPerspectivePass::CreateStripNoPerspectivePass());
return RunOptimizerChecked("StripNoPerspectiveForEssl", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(EmulateNoPerspectivePass::CreateEmulateNoPerspectivePass());
return RunOptimizerChecked("EmulateNoPerspectiveForEssl", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
// Detection gates everything: a module with no dynamically indexed fragment
@@ -810,7 +774,7 @@ namespace MobileGL {
Vector<uint32_t> folded;
if (!RunOptimizerChecked("LegalizeFragmentOutputIndexingForEssl.fold", folder, inputBinary,
folded) ||
folded, true, enableSpirvValidation) ||
folded.empty()) {
// Fail open onto the fallback rather than onto the illegal module.
folded = inputBinary;
@@ -829,7 +793,7 @@ namespace MobileGL {
lowerer.RegisterPass(CreateAggressiveDCEPass(false));
if (!RunOptimizerChecked("LegalizeFragmentOutputIndexingForEssl.lower", lowerer, folded,
outputBinary) ||
outputBinary, true, enableSpirvValidation) ||
outputBinary.empty()) {
outputBinary = folded;
return true;
@@ -846,16 +810,17 @@ namespace MobileGL {
}
bool ShaderCompiler::LowerRectImages(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(NormalizeRectCoordinatesPass::CreateNormalizeRectCoordinatesPass());
return RunOptimizerChecked("LowerRectImages", optimizer, inputBinary, outputBinary);
return RunOptimizerChecked("LowerRectImages", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::Lower1DArrayImagesForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
using namespace spvtools;
// Declined rather than half-translated: after the rewrite the image is a 2D
@@ -897,40 +862,52 @@ namespace MobileGL {
// second Shader. Deduplicating afterwards collapses all three at once.
optimizer.RegisterPass(CreateRemoveDuplicatesPass());
return RunOptimizerChecked("Lower1DArrayImagesForEssl", optimizer, inputBinary, outputBinary);
return RunOptimizerChecked("Lower1DArrayImagesForEssl", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(RebaseInstanceIndexPass::CreateRebaseInstanceIndexPass());
return RunOptimizerChecked("RebaseInstanceIndexForVulkan", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::ZeroBaseVertexForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(ZeroBaseVertexPass::CreateZeroBaseVertexPass());
return RunOptimizerChecked("ZeroBaseVertexForVulkan", optimizer, inputBinary, outputBinary);
return RunOptimizerChecked("ZeroBaseVertexForVulkan", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::DeriveNumSubgroupsForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(DeriveNumSubgroupsPass::CreateDeriveNumSubgroupsPass());
return RunOptimizerChecked("DeriveNumSubgroupsForVulkan", optimizer, inputBinary,
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
using namespace spvtools;
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
optimizer.RegisterPass(DecoratePositionInvariantPass::CreateDecoratePositionInvariantPass());
return RunOptimizerChecked("DecoratePositionInvariantForVulkan", optimizer, inputBinary,
outputBinary);
outputBinary, true, enableSpirvValidation);
}
bool ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary) {
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
const bool enableSpirvValidation) {
constexpr SizeT kSpirvHeaderWordCount = 5;
outputBinary.clear();
if (inputBinary.size() < kSpirvHeaderWordCount || inputBinary[0] != spv::MagicNumber) {
@@ -1058,7 +1035,8 @@ namespace MobileGL {
addedCapabilities.begin(), addedCapabilities.end());
// Hand-rolled word walk, so no Optimizer wrapper ever sees this rewrite;
// check the modified module explicitly in validating lanes.
ValidateOrLatch("UseUnformattedFloatStorageImagesForVulkan", outputBinary);
ValidateOrLatch("UseUnformattedFloatStorageImagesForVulkan", outputBinary,
enableSpirvValidation);
return true;
}
@@ -23,19 +23,23 @@ namespace MobileGL {
static Result<SharedPtr<glslang::TProgram>> LinkProgram(const ProgramAttrib& attrib);
static Result<Vector<Vector<unsigned>>> GetSpirvBinaryFromProgram(const ProgramBinaryAttrib& attrib);
static bool SanitizeAndOptimizeBinary(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool validateOutput = true,
bool enableSpirvValidation = false);
// Demotes DrawIndex/BaseInstance/BaseVertex builtins to plain Private globals
// (mg_DrawID/mg_BaseInstance/mg_BaseVertex) so SPIRV-Cross can emit ESSL.
// Only for backends without native draw-parameter support (DirectGLES).
static bool LowerDrawParametersForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Replaces an ARRAY vertex input with one input per element at consecutive
// locations, seeding a Private copy of the array so indexed reads still work.
// GLSL ES has no array vertex inputs and SPIRV-Cross refuses the whole module
// rather than emulating them, so without this the stage never reaches the
// driver. Only for the DirectGLES transpile path.
static bool SplitArrayVertexInputsForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Replaces the named interface BLOCKS with one variable per member, named
// "<Block>_<member>", shadowing the block itself so the body is untouched. The
// Adreno ES driver silently captures NOTHING for a transform-feedback varying
@@ -46,7 +50,8 @@ namespace MobileGL {
static bool FlattenXfbInterfaceBlocksForEssl(const Vector<Uint32>& inputBinary,
const std::set<String>& blockNames,
std::set<String>& flattenedBlockNames,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// The capture request "StageData.attrib[0]" as the pass above renamed it,
// "StageData_attrib[0]", or false when it does not name a member of a block
// that was flattened.
@@ -58,17 +63,20 @@ namespace MobileGL {
// drivers reject cross-stage uniform blocks whose member precisions differ.
// Only for the DirectGLES transpile path.
static bool StripUboMemberRelaxedPrecisionForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Removes NoPerspective decorations so SPIRV-Cross emits plain (smooth) ESSL varyings.
// DirectGLES fallback only, for devices lacking GL_NV_shader_noperspective_interpolation
// (SPIRV-Cross would otherwise require that extension and the driver would reject it).
static bool StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Emulates noperspective (screen-linear) interpolation via gl_Position.w / gl_FragCoord.w
// so no NV extension is needed; strips what it cannot emulate. DirectGLES fallback for
// devices lacking GL_NV_shader_noperspective_interpolation. See EmulateNoPerspectivePass.
static bool EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Makes every index into a fragment-output array a constant integral
// expression, which is what GLSL ES requires and SPIR-V does not. Runs the
// stock folding chain first (loop unrolling folds the loop-derived indices
@@ -79,7 +87,8 @@ namespace MobileGL {
// dynamically, which is every shader but a handful.
// See LegalizeFragmentOutputIndexPass.
static bool LegalizeFragmentOutputIndexingForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Rebases loads of the InstanceIndex builtin to (InstanceIndex - BaseInstance) so
// shaders see GL's zero-based gl_InstanceID. Vertex shaders only; DirectVulkan
// backend only (glslang's relaxed mode aliases gl_InstanceID to gl_InstanceIndex,
@@ -88,7 +97,8 @@ namespace MobileGL {
// divides the coordinate of each normalized-coordinate lookup by the texture
// size and rewrites the image type to 2D. See NormalizeRectCoordinatesPass for
// what it declines and why.
static bool LowerRectImages(const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary);
static bool LowerRectImages(const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// GL_TEXTURE_1D_ARRAY storage images rewritten to the 2D-array shape the texture
// is actually stored in on ES, with the layer moved from the coordinate's second
// component to its third. DirectGLES transpile path only - Vulkan binds a real
@@ -96,7 +106,8 @@ namespace MobileGL {
// through untouched when the module declares no such image, which is every shader
// but a handful. See Lower1DArrayImagesPass for what it declines and why.
static bool Lower1DArrayImagesForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Gives each format-less storage image the format bound to its image unit, so
// the emitted ESSL can carry the format layout qualifier GLSL ES requires of
// every image and desktop GLSL lets a writeonly declaration omit. `glFormatByName`
@@ -105,7 +116,8 @@ namespace MobileGL {
// natively. See BakeImageFormatsPass for what it declines and why.
static bool BakeImageFormatsForEssl(const Vector<Uint32>& inputBinary,
const UnorderedMap<String, Uint>& glFormatByName,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Whether the module declares a storage image with no format qualifier at all,
// i.e. whether BakeImageFormatsForEssl could change anything. One module parse,
// so the ~every shader that declares none pays no optimizer run.
@@ -124,27 +136,38 @@ namespace MobileGL {
// emitted text instead.
static bool SpirvCrossCanPrintEsslImageFormat(Uint glInternalFormat);
static bool RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Builds the non-indexed-draw variant of a vertex shader: every gl_BaseVertex
// read becomes zero, which is what GL defines for a command carrying no
// baseVertex parameter while Vulkan's builtin would report firstVertex.
// See ZeroBaseVertexPass.
static bool ZeroBaseVertexForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Replaces compute gl_NumSubgroups loads with the value derived from the local
// workgroup dimensions and gl_SubgroupSize. DirectVulkan only; this avoids a
// driver builtin that can disagree with the subgroup IDs the same dispatch emits.
// See DeriveNumSubgroupsPass.
static bool DeriveNumSubgroupsForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair
// (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no
// VK_FORMAT_R64*_SFLOAT is needed - lavapipe advertises none of them for vertex
// buffers. Vertex stage, DirectVulkan only; pairs with the Float64 case in
// VertexInputStateFactory::ToVkVertexFormat.
static bool PackDoubleVertexInputsForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Adds the Invariant decoration to every Position builtin output. GL apps
// routinely rely on cross-program position invariance for multi-pass
// equality depth tests (e.g. GEQUAL re-draws of the same geometry), and
// mobile drivers that optimize per-pipeline break that without the
// decoration. DirectVulkan only.
static bool DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Replaces the declared format of float storage images with Unknown and adds the
// matching SPIR-V capabilities. DirectVulkan uses this only when both Vulkan
// shaderStorageImage*WithoutFormat features are enabled, allowing the
@@ -152,13 +175,15 @@ namespace MobileGL {
// storage images deliberately keep their declared format for GL-compatible bit
// reinterpretation paths (for example, R32F storage accessed as r32ui).
static bool UseUnformattedFloatStorageImagesForVulkan(
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary);
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
// Rewrites every 64-bit float in the module to a 32-bit one, preserving every
// block offset and stride exactly (see DemoteFloat64Pass). Already part of
// SanitizeAndOptimizeBinary, which is where production reaches it; exposed
// separately so a test can drive the demotion on its own.
static bool DemoteFloat64ToFloat32(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
Vector<uint32_t>& outputBinary,
bool enableSpirvValidation = false);
static Result<String> DecompileShader(SpvcSession& session);
// Parses one trivial shader in each configuration the production path can
@@ -185,18 +210,16 @@ namespace MobileGL {
// no way left to warm it.
static void ResetPrewarmLatch();
// Test-environment SPIR-V validation. When enabled, every Optimizer wrapper
// in this file validates its OUTPUT binary - the bytes a driver can actually
// receive - and a failure logs the VUID (via MGLOG_I; see the consumer for
// why not MGLOG_E) and bumps the failure latch below WITHOUT changing the
// wrapper's return value: control flow must stay identical between the
// validating and shipping configurations, or fail-open call sites would make
// the two render differently. Resolved lazily from MOBILEGL_VALIDATE_SPIRV;
// defaults on for desktop/CI/WSL builds and off for device (__ANDROID__)
// builds. The setter wins over the environment and is safe to call from test
// fixtures at any time.
static bool SpirvValidationEnabled();
static void SetSpirvValidationEnabled(bool enabled);
// Validation is an explicit immutable option of each compiler operation. The
// program-link task snapshots MOBILEGL_ENABLE_SPIRV_VALIDATION before it can run
// on a worker; standalone callers pass true directly. A failure logs the VUID and
// bumps the latch below WITHOUT changing a wrapper's return value, so validating
// and shipping configurations preserve identical rendering control flow.
// Makes validator table lifetime safe before an external final-module validator
// runs. This has no configuration state; callers invoke it only for an enabled
// task-local validation option.
static void PrepareSpirvValidation();
// The test-lane enforcement signal: total validation failures observed this
// process. Tests snapshot it, run the operation under scrutiny, and assert
@@ -23,6 +23,7 @@
namespace {
using MobileGL::SizeT;
using MobileGL::String;
using MobileGL::Uint32;
using MobileGL::Vector;
bool IsIdentifierChar(char ch) {
@@ -181,331 +182,6 @@ namespace {
return std::all_of(token.text.begin() + 1, token.text.end(), IsIdentifierChar);
}
class TokenCursor {
public:
TokenCursor(const Vector<CodeToken>& tokens, SizeT position) : m_tokens(tokens), m_position(position) {}
bool Consume(const char* expected) {
if (m_position >= m_tokens.size() || m_tokens[m_position].text != expected) {
return false;
}
++m_position;
return true;
}
bool ConsumeAnyIdentifier(String& identifier) {
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position])) {
return false;
}
identifier = m_tokens[m_position++].text;
return true;
}
bool ConsumeAnyIdentifier() {
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position])) {
return false;
}
++m_position;
return true;
}
bool ConsumeIdentifier(const String& expected) {
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position]) ||
m_tokens[m_position].text != expected) {
return false;
}
++m_position;
return true;
}
SizeT Position() const { return m_position; }
private:
const Vector<CodeToken>& m_tokens;
SizeT m_position;
};
SizeT CountToken(const Vector<CodeToken>& tokens, const String& tokenText) {
return static_cast<SizeT>(std::count_if(tokens.begin(), tokens.end(),
[&](const CodeToken& token) { return token.text == tokenText; }));
}
bool HasIdentifierWithPrefixOutsideAllowed(const Vector<CodeToken>& tokens, const String& prefix,
std::initializer_list<const char*> allowedIdentifiers) {
return std::any_of(tokens.begin(), tokens.end(), [&](const CodeToken& token) {
if (!IsIdentifierToken(token) || !token.text.starts_with(prefix)) {
return false;
}
return std::none_of(allowedIdentifiers.begin(), allowedIdentifiers.end(),
[&](const char* allowed) { return token.text == allowed; });
});
}
bool MatchTokenSequence(const Vector<CodeToken>& tokens, SizeT position,
std::initializer_list<const char*> expected) {
if (position + expected.size() > tokens.size()) {
return false;
}
for (const char* token : expected) {
if (tokens[position++].text != token) {
return false;
}
}
return true;
}
struct LinearPrefixScanMatch {
SizeT sharedArraySizeBegin = 0;
SizeT sharedArraySizeEnd = 0;
SizeT scanBegin = 0;
SizeT scanEnd = 0;
String cache;
String importance;
String prefixSum;
String loopLength;
String loopIndex;
String sum;
};
bool ParseLinearPrefixScanTemplate(const Vector<CodeToken>& tokens, LinearPrefixScanMatch& match) {
// The workaround deliberately recognizes one complete algorithm, not merely the
// subgroupInclusiveAdd token. Changing scratch storage is only safe when that storage is
// private to this scan and the workgroup has exactly 1024 X invocations.
SizeT localSizeDeclarationCount = 0;
for (SizeT i = 0; i < tokens.size(); ++i) {
if (MatchTokenSequence(tokens, i, {"layout", "(", "local_size_x", "=", "1024", ")", "in", ";"})) {
++localSizeDeclarationCount;
}
}
if (localSizeDeclarationCount != 1) {
return false;
}
SizeT sharedDeclarationIndex = String::npos;
SizeT sharedDeclarationCount = 0;
String cacheName;
for (SizeT i = 0; i + 6 < tokens.size(); ++i) {
if (tokens[i].text != "shared" || tokens[i + 1].text != "float" || !IsIdentifierToken(tokens[i + 2]) ||
tokens[i + 3].text != "[" || tokens[i + 4].text != "64" || tokens[i + 5].text != "]" ||
tokens[i + 6].text != ";") {
continue;
}
++sharedDeclarationCount;
sharedDeclarationIndex = i;
cacheName = tokens[i + 2].text;
}
if (sharedDeclarationCount != 1) {
return false;
}
SizeT scanTokenIndex = String::npos;
SizeT scanCount = 0;
for (SizeT i = 0; i + 7 < tokens.size(); ++i) {
if (tokens[i].text == "float" && IsIdentifierToken(tokens[i + 1]) && tokens[i + 2].text == "=" &&
tokens[i + 3].text == "subgroupInclusiveAdd" && tokens[i + 4].text == "(" &&
IsIdentifierToken(tokens[i + 5]) && tokens[i + 6].text == ")" && tokens[i + 7].text == ";") {
++scanCount;
scanTokenIndex = i;
}
}
if (scanCount != 1 || sharedDeclarationIndex >= scanTokenIndex) {
return false;
}
TokenCursor cursor(tokens, scanTokenIndex);
String prefixSum;
String importance;
String loopLength;
String loopIndex;
String sum;
if (!cursor.Consume("float") || !cursor.ConsumeAnyIdentifier(prefixSum) || !cursor.Consume("=") ||
!cursor.Consume("subgroupInclusiveAdd") || !cursor.Consume("(") ||
!cursor.ConsumeAnyIdentifier(importance) || !cursor.Consume(")") || !cursor.Consume(";") ||
!cursor.Consume("if") || !cursor.Consume("(") || !cursor.Consume("gl_SubgroupInvocationID") ||
!cursor.Consume("==") || !cursor.Consume("gl_SubgroupSize") || !cursor.Consume("-") ||
!cursor.Consume("1u") || !cursor.Consume(")") || !cursor.ConsumeIdentifier(cacheName) ||
!cursor.Consume("[") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume("]") || !cursor.Consume("=") ||
!cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume(";") || !cursor.Consume("barrier") ||
!cursor.Consume("(") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("uint") ||
!cursor.ConsumeAnyIdentifier(loopLength) || !cursor.Consume("=") || !cursor.Consume("uint") ||
!cursor.Consume("(") || !cursor.Consume("findMSB") || !cursor.Consume("(") ||
!cursor.Consume("gl_NumSubgroups") || !cursor.Consume(")") || !cursor.Consume(")") ||
!cursor.Consume(";") || !cursor.ConsumeIdentifier(loopLength) || !cursor.Consume("+=") ||
!cursor.Consume("uint") || !cursor.Consume("(") || !cursor.Consume("gl_NumSubgroups") ||
!cursor.Consume("-") || !cursor.Consume("(") || !cursor.Consume("1u") || !cursor.Consume("<<") ||
!cursor.Consume("(") || !cursor.ConsumeIdentifier(loopLength) || !cursor.Consume("-") ||
!cursor.Consume("1u") || !cursor.Consume(")") || !cursor.Consume(")") || !cursor.Consume(">") ||
!cursor.Consume("0u") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("for") ||
!cursor.Consume("(") || !cursor.Consume("uint") || !cursor.ConsumeAnyIdentifier(loopIndex) ||
!cursor.Consume("=") || !cursor.Consume("0") || !cursor.Consume(";") ||
!cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("<") || !cursor.ConsumeIdentifier(loopLength) ||
!cursor.Consume(";") || !cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("++") ||
!cursor.Consume(")") || !cursor.Consume("{") || !cursor.Consume("if") || !cursor.Consume("(") ||
!cursor.Consume("(") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume("&") || !cursor.Consume("(") ||
!cursor.Consume("1u") || !cursor.Consume("<<") || !cursor.ConsumeIdentifier(loopIndex) ||
!cursor.Consume(")") || !cursor.Consume(")") || !cursor.Consume(">") || !cursor.Consume("0u") ||
!cursor.Consume(")") || !cursor.Consume("{") || !cursor.ConsumeIdentifier(prefixSum) ||
!cursor.Consume("+=") || !cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") ||
!cursor.Consume("(") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume(">>") ||
!cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("<<") || !cursor.ConsumeIdentifier(loopIndex) ||
!cursor.Consume(")") || !cursor.Consume("-") || !cursor.Consume("1u") || !cursor.Consume("]") ||
!cursor.Consume(";") || !cursor.Consume("if") || !cursor.Consume("(") ||
!cursor.Consume("gl_SubgroupInvocationID") || !cursor.Consume("==") || !cursor.Consume("gl_SubgroupSize") ||
!cursor.Consume("-") || !cursor.Consume("1u") || !cursor.Consume(")") ||
!cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") || !cursor.Consume("gl_SubgroupID") ||
!cursor.Consume("]") || !cursor.Consume("=") || !cursor.ConsumeIdentifier(prefixSum) ||
!cursor.Consume(";") || !cursor.Consume("}") || !cursor.Consume("barrier") || !cursor.Consume("(") ||
!cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("}") || !cursor.Consume("if") ||
!cursor.Consume("(") || !cursor.Consume("gl_LocalInvocationID") || !cursor.Consume(".") ||
!cursor.Consume("x") || !cursor.Consume("==") || !cursor.Consume("uint") || !cursor.Consume("(") ||
!cursor.Consume("1024") || !cursor.Consume("-") || !cursor.Consume("1") || !cursor.Consume(")") ||
!cursor.Consume(")") || !cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") ||
!cursor.Consume("0") || !cursor.Consume("]") || !cursor.Consume("=") ||
!cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume(";") || !cursor.Consume("barrier") ||
!cursor.Consume("(") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("float") ||
!cursor.ConsumeAnyIdentifier(sum) || !cursor.Consume("=") || !cursor.ConsumeIdentifier(cacheName) ||
!cursor.Consume("[") || !cursor.Consume("0") || !cursor.Consume("]") || !cursor.Consume(";")) {
return false;
}
const SizeT scanEndToken = cursor.Position() - 1;
// Require the scan's immediate consumer as well. This makes the match specific to a
// linear distribution warp, and avoids changing unrelated prefix scans which may rely on
// the implementation's native subgroup partitioning.
if (!cursor.Consume("float") || !cursor.ConsumeAnyIdentifier() || !cursor.Consume("=") ||
!cursor.Consume("(") || !cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume("-") ||
!cursor.ConsumeIdentifier(importance) || !cursor.Consume(")") || !cursor.Consume("/") ||
!cursor.ConsumeIdentifier(sum) || !cursor.Consume("-") || !cursor.Consume("float") ||
!cursor.Consume("(") || !cursor.Consume("gl_LocalInvocationID") || !cursor.Consume(".") ||
!cursor.Consume("x") || !cursor.Consume("+") || !cursor.Consume("1u") || !cursor.Consume(")") ||
!cursor.Consume("/") || !cursor.Consume("float") || !cursor.Consume("(") || !cursor.Consume("1024") ||
!cursor.Consume(")") || !cursor.Consume(";")) {
return false;
}
// No other use may share the scratch array, and no additional subgroup operation or
// builtin may silently retain native-64 semantics after this module becomes virtual-32.
if (CountToken(tokens, cacheName) != 6 || CountToken(tokens, "subgroupInclusiveAdd") != 1 ||
CountToken(tokens, "gl_SubgroupInvocationID") != 2 || CountToken(tokens, "gl_SubgroupSize") != 2 ||
CountToken(tokens, "gl_SubgroupID") != 4 || CountToken(tokens, "gl_NumSubgroups") != 2 ||
CountToken(tokens, "gl_LocalInvocationID") != 2 || CountToken(tokens, "barrier") != 3 ||
CountToken(tokens, "findMSB") != 1 ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "subgroup", {"subgroupInclusiveAdd"}) ||
HasIdentifierWithPrefixOutsideAllowed(
tokens, "gl_Subgroup",
{"gl_SubgroupInvocationID", "gl_SubgroupSize", "gl_SubgroupID", "gl_NumSubgroups"}) ||
// ARB/NV spellings of lane-width-sensitive builtins and functions
// (gl_SubGroupSizeARB, ballotARB, gl_WarpSizeNV, shuffleNV, ...) must block the
// rewrite just like their KHR counterparts: they would silently keep native-width
// semantics in a module rewritten to the virtual 32-lane model.
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SubGroup", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Warp", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Thread", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SMID", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "ballot", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "shuffle", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "readInvocation", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "readFirstInvocation", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "anyInvocation", {}) ||
HasIdentifierWithPrefixOutsideAllowed(tokens, "allInvocations", {})) {
return false;
}
// The scan must be at the top level of the sole main() body. Its existing barriers already
// require uniform control flow; this check prevents us from introducing extra barriers in
// a nested branch or loop.
SizeT mainOpenBrace = String::npos;
SizeT mainCloseBrace = String::npos;
SizeT mainCount = 0;
for (SizeT i = 0; i + 4 < tokens.size(); ++i) {
if (!MatchTokenSequence(tokens, i, {"void", "main", "(", ")", "{"})) {
continue;
}
++mainCount;
mainOpenBrace = i + 4;
int depth = 1;
for (SizeT j = mainOpenBrace + 1; j < tokens.size(); ++j) {
if (tokens[j].text == "{")
++depth;
else if (tokens[j].text == "}" && --depth == 0) {
mainCloseBrace = j;
break;
}
}
}
if (mainCount != 1 || mainCloseBrace == String::npos || scanTokenIndex <= mainOpenBrace ||
scanEndToken >= mainCloseBrace) {
return false;
}
int depthAtScan = 1;
for (SizeT i = mainOpenBrace + 1; i < scanTokenIndex; ++i) {
if (tokens[i].text == "{")
++depthAtScan;
else if (tokens[i].text == "}")
--depthAtScan;
}
if (depthAtScan != 1) {
return false;
}
constexpr const char* injectedNames[] = {"mglPrefixScanLane", "mglVirtualSubgroupInvocation",
"mglVirtualSubgroup", "mglVirtualSubgroupBase",
"mglPrefixLane", "mglVirtualSubgroupCount"};
for (const char* injectedName : injectedNames) {
if (CountToken(tokens, injectedName) != 0) {
return false;
}
}
match.sharedArraySizeBegin = tokens[sharedDeclarationIndex + 4].begin;
match.sharedArraySizeEnd = tokens[sharedDeclarationIndex + 4].end;
match.scanBegin = tokens[scanTokenIndex].begin;
match.scanEnd = tokens[scanEndToken].end;
match.cache = std::move(cacheName);
match.importance = std::move(importance);
match.prefixSum = std::move(prefixSum);
match.loopLength = std::move(loopLength);
match.loopIndex = std::move(loopIndex);
match.sum = std::move(sum);
return true;
}
String BuildLinearPrefixScanReplacement(const LinearPrefixScanMatch& match) {
String replacement;
replacement.reserve(1800);
replacement += "uint mglPrefixScanLane = gl_LocalInvocationID.x;\n";
replacement += "uint mglVirtualSubgroupInvocation = mglPrefixScanLane & 31u;\n";
replacement += "uint mglVirtualSubgroup = mglPrefixScanLane >> 5u;\n";
replacement += "const uint mglVirtualSubgroupCount = 32u;\n";
replacement += match.cache + "[mglPrefixScanLane] = " + match.importance + ";\n";
replacement += "barrier();\n";
replacement += "float " + match.prefixSum + " = 0.0f;\n";
replacement += "uint mglVirtualSubgroupBase = mglVirtualSubgroup << 5u;\n";
replacement += "for (uint mglPrefixLane = mglVirtualSubgroupBase; "
"mglPrefixLane <= mglPrefixScanLane; ++mglPrefixLane) {\n";
replacement += match.prefixSum + " += " + match.cache + "[mglPrefixLane];\n";
replacement += "}\n";
replacement += "barrier();\n";
replacement += "if (mglVirtualSubgroupInvocation == 31u) " + match.cache +
"[mglVirtualSubgroup] = " + match.prefixSum + ";\n";
replacement += "barrier();\n";
replacement += "uint " + match.loopLength + " = uint(findMSB(mglVirtualSubgroupCount));\n";
replacement +=
match.loopLength + " += uint(mglVirtualSubgroupCount - (1u << (" + match.loopLength + " - 1u)) > 0u);\n";
replacement += "for (uint " + match.loopIndex + " = 0u; " + match.loopIndex + " < " + match.loopLength +
"; ++" + match.loopIndex + ") {\n";
replacement += "if ((mglVirtualSubgroup & (1u << " + match.loopIndex + ")) > 0u) {\n";
replacement += match.prefixSum + " += " + match.cache + "[(mglVirtualSubgroup >> " + match.loopIndex + " << " +
match.loopIndex + ") - 1u];\n";
replacement += "if (mglVirtualSubgroupInvocation == 31u) " + match.cache +
"[mglVirtualSubgroup] = " + match.prefixSum + ";\n";
replacement += "}\nbarrier();\n}\n";
replacement += "if (mglPrefixScanLane == 1023u) " + match.cache + "[0] = " + match.prefixSum + ";\n";
replacement += "barrier();\n";
replacement += "float " + match.sum + " = " + match.cache + "[0];";
return replacement;
}
void SkipDirectiveWhitespace(const MobileGL::String& source, SizeT& pos, SizeT lineEnd) {
while (pos < lineEnd && std::isspace(static_cast<unsigned char>(source[pos]))) {
pos++;
@@ -1253,117 +929,6 @@ namespace {
namespace MobileGL {
namespace MG_Util {
namespace ShaderTranspiler {
Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize,
String& source) {
constexpr Uint32 capturedSubgroupSize = 32;
if (stage != ShaderStage::Compute || nativeSubgroupSize <= capturedSubgroupSize ||
nativeSubgroupSize % capturedSubgroupSize != 0) {
return false;
}
// Vulkan subgroup widths are powers of two. Keep the workaround restricted to
// wider widths which are a power-of-two multiple of the captured 32-lane model.
const Uint32 subgroupScale = nativeSubgroupSize / capturedSubgroupSize;
if ((subgroupScale & (subgroupScale - 1u)) != 0u) {
return false;
}
const Vector<CodeToken> tokens = TokenizeCode(source);
LinearPrefixScanMatch match;
if (!ParseLinearPrefixScanTemplate(tokens, match)) {
// Diagnosability: when the trigger op is present but the template no longer
// matches (e.g. the pack shipped a new shader revision), the affected device
// silently falls back to the driver's miscompiled path. Make that visible.
if (CountToken(tokens, "subgroupInclusiveAdd") > 0) {
MGLOG_W_ONCE("%s: subgroupInclusiveAdd present but the linear prefix-scan template "
"did not match; the wide-subgroup rewrite was NOT applied",
__func__);
}
return false;
}
const String replacement = BuildLinearPrefixScanReplacement(match);
source.replace(match.scanBegin, match.scanEnd - match.scanBegin, replacement);
// The declaration occurs before the replaced scan, so its original offsets remain
// valid after the first replacement.
source.replace(match.sharedArraySizeBegin, match.sharedArraySizeEnd - match.sharedArraySizeBegin,
"1024");
return true;
}
namespace {
struct ShaderSourceQuirkContext {
ShaderStage stage = ShaderStage::Unknown;
BackendType backend = BackendType::Unknown;
MG_Backend::GpuVendorKind vendor = MG_Backend::GpuVendorKind::Unknown;
Uint32 subgroupSize = 0;
};
// Device-quirk registry. Every entry is a narrowly scoped source rewrite that
// works around a specific driver defect. A quirk runs when its env override
// forces it on, or when the override is Auto and DeviceApplies matches the
// detected device. ForceOn bypasses only the device gate - each Apply keeps
// its own structural safety checks. Add new per-device workarounds here
// instead of open-coding them in PreprocessShaderSource.
struct ShaderSourceQuirk {
const char* name;
// Reads the override out of the captured env, never out of the live
// MG_Config table: a worker must see the same config the GL thread saw.
MG_Config::QuirkOverride (*GetOverride)(const CompileEnv&);
Bool (*DeviceApplies)(const ShaderSourceQuirkContext&);
Bool (*Apply)(const ShaderSourceQuirkContext&, String&);
};
constexpr ShaderSourceQuirk kShaderSourceQuirks[] = {
{
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN
"subgroup-prefix-scan-rewrite",
[](const CompileEnv& env) { return env.subgroupPrefixScanQuirk; },
[](const ShaderSourceQuirkContext& ctx) {
// Qualcomm's Vulkan driver miscompiles the recognized float
// InclusiveScan pattern for native subgroups wider than the
// captured 32 lanes; other vendors compile it correctly and
// should keep their native scan.
return ctx.backend == BackendType::DirectVulkan &&
ctx.vendor == MG_Backend::GpuVendorKind::Qualcomm;
},
[](const ShaderSourceQuirkContext& ctx, String& source) {
return RewriteLinearSubgroupPrefixScanForVulkan(ctx.stage, ctx.subgroupSize,
source);
},
},
};
void ApplyShaderSourceQuirks(const CompileEnv& env, ShaderStage stage, String& source) {
// No backend at capture time means no device to match a quirk against,
// and (as before) no quirk can fire - not even a forced one, because
// every Apply reads device parameters that do not exist yet.
if (!env.HasBackend()) {
return;
}
const ShaderSourceQuirkContext quirkContext{
stage,
env.backend,
env.params.GpuVendor,
env.params.SubgroupSize,
};
for (const ShaderSourceQuirk& quirk : kShaderSourceQuirks) {
const MG_Config::QuirkOverride quirkOverride = quirk.GetOverride(env);
if (quirkOverride == MG_Config::QuirkOverride::ForceOff) {
continue;
}
if (quirkOverride == MG_Config::QuirkOverride::Auto &&
!quirk.DeviceApplies(quirkContext)) {
continue;
}
if (quirk.Apply(quirkContext, source)) {
MGLOG_D("ApplyShaderSourceQuirks: applied '%s'%s", quirk.name,
quirkOverride == MG_Config::QuirkOverride::ForceOn ? " (forced on)" : "");
}
}
}
} // namespace
void PreprocessShaderSource(ShaderStage stage, String& source) {
PreprocessShaderSource(stage, source, *GetCurrentCompileEnv());
}
@@ -1403,7 +968,6 @@ namespace MobileGL {
ModernizeLegacyGLSL(stage, source, afterVersion);
InjectDepthRangeBuiltinShim(stage, source, afterVersion);
ApplyShaderSourceQuirks(env, stage, source);
}
Bool RetargetLegacyVersionDirectiveTo460(String& source) {
@@ -30,18 +30,6 @@ namespace MobileGL {
// tests and diagnostics that drive the preprocessor standalone.
void PreprocessShaderSource(ShaderStage stage, String& source);
// Some desktop-captured compute shaders build a workgroup-wide linear prefix scan
// from subgroupInclusiveAdd plus a shared array of subgroup totals. Qualcomm's
// Vulkan driver miscompiles that exact float InclusiveScan path for native subgroups
// wider than the capture's 32 lanes. For the narrowly recognized, uniform-control-
// flow template, replace the subgroup-local scan with a shared-memory, strict
// left-fold over virtual 32-lane segments. Returns true only when the complete safe
// template was recognized and rewritten. PreprocessShaderSource reaches this through
// its device-quirk registry: by default only on detected Qualcomm Vulkan devices,
// overridable either way with MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN=1/0. The explicit
// entry point exists for deterministic tests.
Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize, String& source);
// Rewrites a "#version 330 core" directive that PreprocessShaderSource normalized down
// from a legacy desktop version back up to "#version 460 core". Returns false (leaving
// the source untouched) for anything else: ES, compatibility, or an already-modern
@@ -0,0 +1,251 @@
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DeriveNumSubgroupsPass.cpp
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#include "DeriveNumSubgroupsPass.h"
#include "spirv.hpp"
#include "source/opt/constants.h"
#include "source/opt/def_use_manager.h"
#include "source/opt/instruction.h"
#include "source/opt/ir_context.h"
#include "source/opt/module.h"
#include "source/util/make_unique.h"
#include <vector>
namespace MobileGL {
namespace MG_Util {
namespace ShaderTranspiler {
namespace {
using spvtools::opt::Instruction;
using spvtools::opt::IRContext;
using spvtools::opt::Operand;
Instruction* FindBuiltinDefinition(IRContext* context, spv::BuiltIn builtin) {
auto* defUseMgr = context->get_def_use_mgr();
for (auto& annotation : context->annotations()) {
if (annotation.opcode() != spv::Op::OpDecorate || annotation.NumInOperands() < 3) {
continue;
}
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(1)) !=
spv::Decoration::BuiltIn) {
continue;
}
if (static_cast<spv::BuiltIn>(annotation.GetSingleWordInOperand(2)) != builtin) {
continue;
}
return defUseMgr->GetDef(annotation.GetSingleWordInOperand(0));
}
return nullptr;
}
bool IsInputPointerTo(IRContext* context, const Instruction* variable, uint32_t pointeeTypeId) {
if (variable == nullptr || variable->opcode() != spv::Op::OpVariable ||
variable->NumInOperands() < 1 ||
static_cast<spv::StorageClass>(variable->GetSingleWordInOperand(0)) !=
spv::StorageClass::Input) {
return false;
}
const Instruction* pointerType = context->get_def_use_mgr()->GetDef(variable->type_id());
return pointerType != nullptr && pointerType->opcode() == spv::Op::OpTypePointer &&
pointerType->NumInOperands() >= 2 &&
static_cast<spv::StorageClass>(pointerType->GetSingleWordInOperand(0)) ==
spv::StorageClass::Input &&
pointerType->GetSingleWordInOperand(1) == pointeeTypeId;
}
bool IsUnsignedInt32(IRContext* context, uint32_t typeId) {
const Instruction* type = context->get_def_use_mgr()->GetDef(typeId);
return type != nullptr && type->opcode() == spv::Op::OpTypeInt &&
type->NumInOperands() >= 2 && type->GetSingleWordInOperand(0) == 32u &&
type->GetSingleWordInOperand(1) == 0u;
}
uint32_t SynthesizeSubgroupSizeVariable(IRContext* context, uint32_t pointerTypeId) {
const uint32_t variableId = context->TakeNextId();
context->AddGlobalValue(spvtools::MakeUnique<Instruction>(
context, spv::Op::OpVariable, pointerTypeId, variableId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_STORAGE_CLASS,
{static_cast<uint32_t>(spv::StorageClass::Input)}}}));
context->AddAnnotationInst(spvtools::MakeUnique<Instruction>(
context, spv::Op::OpDecorate, 0, 0,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {variableId}},
{SPV_OPERAND_TYPE_DECORATION,
{static_cast<uint32_t>(spv::Decoration::BuiltIn)}},
{SPV_OPERAND_TYPE_LITERAL_INTEGER,
{static_cast<uint32_t>(spv::BuiltIn::SubgroupSize)}}}));
for (Instruction& entryPoint : context->module()->entry_points()) {
entryPoint.AddOperand({SPV_OPERAND_TYPE_ID, {variableId}});
}
return variableId;
}
} // namespace
spvtools::opt::Pass::Status DeriveNumSubgroupsPass::Process() {
auto* irContext = context();
auto* defUseMgr = irContext->get_def_use_mgr();
Instruction* numSubgroupsVar = FindBuiltinDefinition(irContext, spv::BuiltIn::NumSubgroups);
if (numSubgroupsVar == nullptr) {
return Status::SuccessWithoutChange;
}
std::vector<Instruction*> numSubgroupsLoads;
bool sawUnexpectedUser = false;
const uint32_t numSubgroupsVarId = numSubgroupsVar->result_id();
defUseMgr->ForEachUser(numSubgroupsVar, [&](Instruction* user) {
switch (user->opcode()) {
case spv::Op::OpLoad:
if (user->NumInOperands() >= 1 &&
user->GetSingleWordInOperand(0) == numSubgroupsVarId) {
numSubgroupsLoads.push_back(user);
} else {
sawUnexpectedUser = true;
}
return;
case spv::Op::OpDecorate:
case spv::Op::OpDecorateId:
case spv::Op::OpDecorateString:
case spv::Op::OpName:
case spv::Op::OpEntryPoint:
return;
default:
sawUnexpectedUser = true;
return;
}
});
if (sawUnexpectedUser) {
return Status::Failure;
}
if (numSubgroupsLoads.empty()) {
return Status::SuccessWithoutChange;
}
const uint32_t valueTypeId = numSubgroupsLoads.front()->type_id();
if (!IsUnsignedInt32(irContext, valueTypeId) ||
!IsInputPointerTo(irContext, numSubgroupsVar, valueTypeId)) {
return Status::Failure;
}
for (const Instruction* load : numSubgroupsLoads) {
if (load->type_id() != valueTypeId) {
return Status::Failure;
}
}
Instruction* workgroupSize = FindBuiltinDefinition(irContext, spv::BuiltIn::WorkgroupSize);
if (workgroupSize == nullptr ||
(workgroupSize->opcode() != spv::Op::OpConstantComposite &&
workgroupSize->opcode() != spv::Op::OpSpecConstantComposite)) {
return Status::Failure;
}
const Instruction* workgroupSizeType = defUseMgr->GetDef(workgroupSize->type_id());
if (workgroupSizeType == nullptr || workgroupSizeType->opcode() != spv::Op::OpTypeVector ||
workgroupSizeType->NumInOperands() < 2 ||
workgroupSizeType->GetSingleWordInOperand(0) != valueTypeId ||
workgroupSizeType->GetSingleWordInOperand(1) != 3u) {
return Status::Failure;
}
Instruction* subgroupSizeVar = FindBuiltinDefinition(irContext, spv::BuiltIn::SubgroupSize);
if (subgroupSizeVar != nullptr &&
!IsInputPointerTo(irContext, subgroupSizeVar, valueTypeId)) {
return Status::Failure;
}
auto* constantMgr = irContext->get_constant_mgr();
auto* typeMgr = irContext->get_type_mgr();
const auto* valueType = typeMgr->GetType(valueTypeId);
if (valueType == nullptr) {
return Status::Failure;
}
const auto* one = constantMgr->GetConstant(valueType, {1u});
const Instruction* oneInst =
one != nullptr ? constantMgr->GetDefiningInstruction(one, valueTypeId) : nullptr;
if (oneInst == nullptr) {
return Status::Failure;
}
const uint32_t oneId = oneInst->result_id();
const uint32_t subgroupSizeVarId = subgroupSizeVar != nullptr
? subgroupSizeVar->result_id()
: SynthesizeSubgroupSizeVariable(irContext, numSubgroupsVar->type_id());
const uint32_t workgroupSizeId = workgroupSize->result_id();
// The pipeline never enables ALLOW_VARYING_SUBGROUP_SIZE, so Vulkan's fixed
// subgroup partition is exactly ceil(local invocation count / SubgroupSize).
// `(count - 1) / size + 1` avoids an addition overflow at count + size - 1.
for (Instruction* load : numSubgroupsLoads) {
const uint32_t localSizeXId = irContext->TakeNextId();
const uint32_t localSizeYId = irContext->TakeNextId();
const uint32_t localSizeZId = irContext->TakeNextId();
const uint32_t localSizeXYId = irContext->TakeNextId();
const uint32_t invocationCountId = irContext->TakeNextId();
const uint32_t adjustedCountId = irContext->TakeNextId();
const uint32_t subgroupSizeId = irContext->TakeNextId();
const uint32_t quotientId = irContext->TakeNextId();
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpCompositeExtract, valueTypeId, localSizeXId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {workgroupSizeId}},
{SPV_OPERAND_TYPE_LITERAL_INTEGER, {0u}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpCompositeExtract, valueTypeId, localSizeYId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {workgroupSizeId}},
{SPV_OPERAND_TYPE_LITERAL_INTEGER, {1u}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpCompositeExtract, valueTypeId, localSizeZId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {workgroupSizeId}},
{SPV_OPERAND_TYPE_LITERAL_INTEGER, {2u}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpIMul, valueTypeId, localSizeXYId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {localSizeXId}},
{SPV_OPERAND_TYPE_ID, {localSizeYId}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpIMul, valueTypeId, invocationCountId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {localSizeXYId}},
{SPV_OPERAND_TYPE_ID, {localSizeZId}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpISub, valueTypeId, adjustedCountId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {invocationCountId}},
{SPV_OPERAND_TYPE_ID, {oneId}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpLoad, valueTypeId, subgroupSizeId,
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {subgroupSizeVarId}}}));
load->InsertBefore(spvtools::MakeUnique<Instruction>(
irContext, spv::Op::OpUDiv, valueTypeId, quotientId,
std::initializer_list<Operand>{
{SPV_OPERAND_TYPE_ID, {adjustedCountId}},
{SPV_OPERAND_TYPE_ID, {subgroupSizeId}}}));
// Preserve the original result id so every downstream use automatically sees
// the derived value instead of the driver's NumSubgroups builtin.
load->SetOpcode(spv::Op::OpIAdd);
load->SetInOperands(Instruction::OperandList{
{SPV_OPERAND_TYPE_ID, {quotientId}},
{SPV_OPERAND_TYPE_ID, {oneId}}});
}
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
return Status::SuccessWithChange;
}
spvtools::Optimizer::PassToken DeriveNumSubgroupsPass::CreateDeriveNumSubgroupsPass() {
return spvtools::Optimizer::PassToken(MakeUnique<DeriveNumSubgroupsPass>());
}
} // namespace ShaderTranspiler
} // namespace MG_Util
} // namespace MobileGL
@@ -0,0 +1,37 @@
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DeriveNumSubgroupsPass.h
// Copyright (c) 2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
#pragma once
#include "source/opt/pass.h"
#include "spirv-tools/optimizer.hpp"
#include <Includes.h>
namespace MobileGL {
namespace MG_Util {
namespace ShaderTranspiler {
// Replaces compute-stage NumSubgroups builtin loads with
// ceil(WorkgroupSize.x * WorkgroupSize.y * WorkgroupSize.z / SubgroupSize).
//
// That is the value Vulkan defines for NumSubgroups when the pipeline does not
// enable varying subgroup sizes, which MobileGL never does. Deriving it avoids
// drivers that expose the real SubgroupId topology but return an inconsistent
// NumSubgroups value. This is a DirectVulkan semantic repair, not a source-shader
// rewrite; the application's subgroup arithmetic and shared-memory logic remain
// unchanged.
class DeriveNumSubgroupsPass : public spvtools::opt::Pass {
public:
const char* name() const override { return "derive-num-subgroups"; }
Status Process() override;
static spvtools::Optimizer::PassToken CreateDeriveNumSubgroupsPass();
};
} // namespace ShaderTranspiler
} // namespace MG_Util
} // namespace MobileGL
@@ -19,23 +19,27 @@ namespace MobileGL {
namespace MG_Util {
namespace ShaderTranspiler {
namespace {
constexpr const char* kConflictingName = "sampler";
constexpr const char* kCompatName = "MGL_COMPAT_sampler";
const char* GetCompatName(StringView name) {
if (name == "sampler") return "MGL_COMPAT_sampler";
if (name == "new") return "MGL_COMPAT_new";
return nullptr;
}
Bool IsNamedSamplerFunctionParameter(spvtools::opt::IRContext* context,
spvtools::opt::Instruction& nameInst) {
const char* GetConflictingFunctionParameterCompatName(spvtools::opt::IRContext* context,
spvtools::opt::Instruction& nameInst) {
if (nameInst.opcode() != spv::Op::OpName || nameInst.NumInOperands() < 2) {
return false;
return nullptr;
}
if (nameInst.GetInOperand(1).AsString() != kConflictingName) {
return false;
const char* compatName = GetCompatName(nameInst.GetInOperand(1).AsString());
if (compatName == nullptr) {
return nullptr;
}
auto* defUseMgr = context->get_def_use_mgr();
const Uint32 targetId = nameInst.GetSingleWordInOperand(0);
const auto* target = defUseMgr->GetDef(targetId);
return target != nullptr && target->opcode() == spv::Op::OpFunctionParameter;
return target != nullptr && target->opcode() == spv::Op::OpFunctionParameter ? compatName : nullptr;
}
} // namespace
@@ -44,12 +48,13 @@ namespace MobileGL {
auto* irContext = context();
for (auto& debugInst : irContext->debugs2()) {
if (!IsNamedSamplerFunctionParameter(irContext, debugInst)) {
const char* compatName = GetConflictingFunctionParameterCompatName(irContext, debugInst);
if (compatName == nullptr) {
continue;
}
debugInst.SetInOperand(
1, spvtools::utils::MakeVector<spvtools::opt::Operand::OperandData>(kCompatName));
1, spvtools::utils::MakeVector<spvtools::opt::Operand::OperandData>(compatName));
modified = true;
}
@@ -423,6 +423,26 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
InternalPackedLayout internalPacked;
};
Bool IsValidUnpackPixelPair(TextureInputFormat format, TexturePixelDataType type) {
UnpackChannelMapping mapping{};
if (!GetUnpackChannelMapping(format, mapping)) return false;
PackedTypeLayout packed{};
if (GetPackedTypeLayout(type, packed)) {
return packed.fieldCount == mapping.channelCount;
}
switch (type) {
case TexturePixelDataType::UnsignedInt5999Rev:
case TexturePixelDataType::UnsignedInt101111Rev:
return !mapping.isInteger && mapping.channelCount == 3;
default: {
ShadowComponent component{};
return GetDirectShadowComponentForType(type, mapping.isInteger, component);
}
}
}
// Returns true when the (format, type) -> internal-format upload needs a per-texel conversion;
// returns false both for layouts that already match the shadow bytes (memcpy fast path) and for
// combinations the converter does not support (legacy copy behavior).
@@ -964,6 +984,32 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
return outputPixels;
}
Bool ConvertOnePixelToInternal(TextureInternalFormat targetInternalFormat,
TextureInputFormat textureInputFormat,
TexturePixelDataType inputDataType,
const void* inputPixel,
Vector<Uint8>& outputPixel) {
outputPixel.clear();
if (inputPixel == nullptr || !IsValidUnpackPixelPair(textureInputFormat, inputDataType)) return false;
PixelStoreParameters params{};
params.Alignment = 1;
SizeT convertedSize = 0;
void* converted = ProcessTexturePixelsDataUnpack(
inputPixel, params, targetInternalFormat, textureInputFormat, inputDataType, {1, 1, 1}, false,
convertedSize);
const SizeT expectedSize = MG_Util::GetSizedInternalFormatSizeInBytes(targetInternalFormat);
if (converted == nullptr || convertedSize != expectedSize || expectedSize == 0) {
if (converted != nullptr) free(converted);
return false;
}
outputPixel.resize(convertedSize);
Memcpy(outputPixel.data(), converted, convertedSize);
free(converted);
return true;
}
void* ProcessTexturePixelsDataPack(const void* inputPixels, const PixelStoreParameters& params,
TextureInternalFormat srcInternalFormat, TexturePixelDataType srcDataType,
TextureInputFormat dstInputFormat, TexturePixelDataType dstDataType,
@@ -21,6 +21,12 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
TextureInternalFormat srcInternalFormat, TexturePixelDataType srcDataType,
TextureInputFormat dstInputFormat, TexturePixelDataType dstDataType,
IntVec3 dimension, Bool isBitmap, SizeT& outSize);
Bool ConvertOnePixelToInternal(TextureInternalFormat targetInternalFormat,
TextureInputFormat textureInputFormat,
TexturePixelDataType inputDataType,
const void* inputPixel,
Vector<Uint8>& outputPixel);
void ProcessColorSwizzle(void* data, SizeT pixelCount, const Vector<TextureSwizzleParam>& swizzle);
// True when a packed internal format's 32-bit storage word IS the client (format, type) word,
@@ -223,7 +223,8 @@ target_include_directories(glretrace_common PUBLIC
"${APITRACE_GENERATED_DIR}"
"${APITRACE_ROOT}/dispatch"
"${APITRACE_ROOT}/helpers"
"${APITRACE_ROOT}/retrace")
"${APITRACE_ROOT}/retrace"
"${CMAKE_CURRENT_LIST_DIR}/../../../../../tools/trace_replay")
target_compile_definitions(glretrace_common PRIVATE
main=mobilegl_apitrace_main)
target_redirect_exit(glretrace_common)
@@ -231,14 +232,16 @@ target_link_libraries(glretrace_common PUBLIC retrace_common glhelpers glproc)
add_library(trace_replay_runner SHARED
trace_replay_core.cpp
trace_replay_jni.cpp)
trace_replay_jni.cpp
"${CMAKE_CURRENT_LIST_DIR}/../../../../../tools/trace_replay/apitrace_fbo_dump.cpp")
target_compile_features(trace_replay_runner PRIVATE cxx_std_17)
target_compile_definitions(trace_replay_runner PRIVATE
MOBILEGL_APITRACE_RETRACE_MAIN=mobilegl_apitrace_main)
target_include_directories(trace_replay_runner PRIVATE
"${APITRACE_ROOT}/lib/image")
"${APITRACE_ROOT}/lib/image"
"${CMAKE_CURRENT_LIST_DIR}/../../../../../tools/trace_replay")
target_link_libraries(trace_replay_runner
glretrace_common
@@ -1,3 +1,4 @@
#include "apitrace_fbo_dump.hpp"
#include "glws.hpp"
#include "retrace.hpp"
@@ -375,6 +376,7 @@ bool makeCurrentInternal(Drawable *drawable, Drawable *readable, Context *contex
}
gCurrentDrawable = drawable;
gCurrentContext = eglContext;
mobilegl_trace_dump::InstallIfRequested();
return true;
}
@@ -164,6 +164,11 @@ bool LoadMobileGL(const Request& request, std::string& error) {
} else {
unsetenv("MOBILEGL_COHERENT_AS_FLUSH");
}
if (request.numSubgroupsQuirk) {
setenv("MOBILEGL_NUM_SUBGROUPS_QUIRK", "1", 1);
} else {
unsetenv("MOBILEGL_NUM_SUBGROUPS_QUIRK");
}
if (request.fboAttachmentDumps.empty()) {
unsetenv("MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS");
} else {
@@ -176,6 +181,18 @@ bool LoadMobileGL(const Request& request, std::string& error) {
}
setenv("MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS", dumpPoints.c_str(), 1);
}
if (request.texture2dDumps.empty()) {
unsetenv("MOBILEGL_TRACE_DUMP_TEXTURE_2D");
} else {
std::string dumpPoints;
for (const std::string& dumpPoint : request.texture2dDumps) {
if (!dumpPoints.empty()) {
dumpPoints += ';';
}
dumpPoints += dumpPoint;
}
setenv("MOBILEGL_TRACE_DUMP_TEXTURE_2D", dumpPoints.c_str(), 1);
}
void* handle = dlopen(request.mobileGlLibrary.c_str(), RTLD_NOW | RTLD_GLOBAL);
if (handle == nullptr) {
@@ -383,6 +400,13 @@ std::string SnapshotCallSet(const Request& request) {
callSet += "," + call;
}
}
for (const std::string& dumpPoint : request.texture2dDumps) {
const std::size_t separator = dumpPoint.find(',');
const std::string call = dumpPoint.substr(0, separator);
if (!call.empty() && call != std::to_string(request.targetCall)) {
callSet += "," + call;
}
}
return callSet;
}
@@ -799,6 +823,7 @@ bool WriteResultJson(const Request& request, const Result& result) {
<< (request.avoidAngleLlvmpipeSamplerMipmapMinFilter ? "true" : "false") << ",\n";
file << " \"avoidAngleLlvmpipeExplicitLodBias\": "
<< (request.avoidAngleLlvmpipeExplicitLodBias ? "true" : "false") << ",\n";
file << " \"numSubgroupsQuirk\": " << (request.numSubgroupsQuirk ? "true" : "false") << ",\n";
file << " \"holdMs\": " << request.holdMs << ",\n";
file << " \"mismatchPixels\": " << result.mismatchPixels << "\n";
file << "}\n";
@@ -27,6 +27,9 @@ struct Request {
// Framebuffer-attachment dump points, each `CALL:DIR[:FBO,FBO,...]`. Debug-only; the
// replay behaves exactly as before when this is empty.
std::vector<std::string> fboAttachmentDumps;
// Named GL_TEXTURE_2D dump points, each `CALL,TEXTURE,LEVEL,DIR`. Debug-only; the replay
// behaves exactly as before when this is empty.
std::vector<std::string> texture2dDumps;
int targetFrame = -1;
long long targetCall = -1;
int width = 0;
@@ -41,6 +44,7 @@ struct Request {
bool avoidAngleLlvmpipeSamplerMipmapMinFilter = false;
bool avoidAngleLlvmpipeExplicitLodBias = false;
bool coherentAsFlush = false;
bool numSubgroupsQuirk = false;
int holdMs = 0;
};
@@ -6,6 +6,7 @@
#include <exception>
#include <string>
#include <vector>
#include <sys/stat.h>
extern "C" void mobilegl_trace_set_native_window(ANativeWindow *window);
@@ -25,6 +26,23 @@ std::string ToString(JNIEnv* env, jstring value) {
return out;
}
std::vector<std::string> SplitSemicolonList(const std::string& value) {
std::vector<std::string> values;
std::size_t begin = 0;
while (begin < value.size()) {
const std::size_t end = value.find(';', begin);
const std::string entry = value.substr(begin, end - begin);
if (!entry.empty()) {
values.push_back(entry);
}
if (end == std::string::npos) {
break;
}
begin = end + 1;
}
return values;
}
jobject MakeResult(JNIEnv* env, const mobilegl_trace::Result& result) {
jclass clazz = env->FindClass("top/mobilegl/plugin/trace/TraceReplayActivity$TraceReplayResult");
if (clazz == nullptr) {
@@ -103,7 +121,9 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
jboolean usePbuffer,
jboolean avoidAngleLlvmpipeSamplerMipmapMinFilter,
jboolean avoidAngleLlvmpipeExplicitLodBias,
jboolean coherentAsFlush) {
jboolean coherentAsFlush,
jboolean numSubgroupsQuirk,
jstring texture2dDumps) {
mobilegl_trace::Request request;
request.tracePath = ToString(env, tracePath);
request.goldenPath = ToString(env, goldenPath);
@@ -115,6 +135,7 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
request.diffPath = ToString(env, diffPath);
request.backend = ToString(env, backend);
request.angleVariant = ToString(env, angleVariant);
request.texture2dDumps = SplitSemicolonList(ToString(env, texture2dDumps));
request.targetFrame = targetFrame;
request.targetCall = targetCall;
request.width = width;
@@ -130,6 +151,7 @@ Java_top_mobilegl_plugin_trace_TraceReplayActivity_nativeRunTraceReplay(JNIEnv*
avoidAngleLlvmpipeSamplerMipmapMinFilter == JNI_TRUE;
request.avoidAngleLlvmpipeExplicitLodBias = avoidAngleLlvmpipeExplicitLodBias == JNI_TRUE;
request.coherentAsFlush = coherentAsFlush == JNI_TRUE;
request.numSubgroupsQuirk = numSubgroupsQuirk == JNI_TRUE;
ScopedTraceReplayState replayState;
mobilegl_trace_set_requested_size(request.width, request.height);
@@ -115,7 +115,9 @@ public final class TraceReplayActivity extends Activity {
request.usePbuffer,
request.avoidAngleLlvmpipeSamplerMipmapMinFilter,
request.avoidAngleLlvmpipeExplicitLodBias,
request.coherentAsFlush
request.coherentAsFlush,
request.numSubgroupsQuirk,
request.texture2dDumps
);
Log.i(TAG, result.toString());
TraceReplayResult finalResult = result;
@@ -147,7 +149,9 @@ public final class TraceReplayActivity extends Activity {
boolean usePbuffer,
boolean avoidAngleLlvmpipeSamplerMipmapMinFilter,
boolean avoidAngleLlvmpipeExplicitLodBias,
boolean coherentAsFlush
boolean coherentAsFlush,
boolean numSubgroupsQuirk,
String texture2dDumps
);
private static final class TraceReplayRequest {
@@ -172,6 +176,8 @@ public final class TraceReplayActivity extends Activity {
final boolean avoidAngleLlvmpipeSamplerMipmapMinFilter;
final boolean avoidAngleLlvmpipeExplicitLodBias;
final boolean coherentAsFlush;
final boolean numSubgroupsQuirk;
final String texture2dDumps;
private TraceReplayRequest(
String tracePath,
@@ -194,7 +200,9 @@ public final class TraceReplayActivity extends Activity {
boolean usePbuffer,
boolean avoidAngleLlvmpipeSamplerMipmapMinFilter,
boolean avoidAngleLlvmpipeExplicitLodBias,
boolean coherentAsFlush
boolean coherentAsFlush,
boolean numSubgroupsQuirk,
String texture2dDumps
) {
this.tracePath = tracePath;
this.goldenPath = goldenPath;
@@ -217,6 +225,8 @@ public final class TraceReplayActivity extends Activity {
this.avoidAngleLlvmpipeSamplerMipmapMinFilter = avoidAngleLlvmpipeSamplerMipmapMinFilter;
this.avoidAngleLlvmpipeExplicitLodBias = avoidAngleLlvmpipeExplicitLodBias;
this.coherentAsFlush = coherentAsFlush;
this.numSubgroupsQuirk = numSubgroupsQuirk;
this.texture2dDumps = texture2dDumps;
}
static TraceReplayRequest from(Intent intent, File filesDir, String defaultBackend) {
@@ -243,7 +253,9 @@ public final class TraceReplayActivity extends Activity {
intent.getBooleanExtra("use_pbuffer", false),
intent.getBooleanExtra("avoid_angle_llvmpipe_sampler_mipmap_min_filter", false),
intent.getBooleanExtra("avoid_angle_llvmpipe_explicit_lod_bias", false),
intent.getBooleanExtra("coherent_as_flush", false)
intent.getBooleanExtra("coherent_as_flush", false),
intent.getBooleanExtra("num_subgroups_quirk", false),
readString(intent, "texture_2d_dumps", "")
);
}
+36
View File
@@ -31,6 +31,8 @@ Usage:
[--avoid-angle-llvmpipe-sampler-mipmap-min-filter] \
[--avoid-angle-llvmpipe-explicit-lod-bias] \
[--coherent-as-flush] \
[--num-subgroups-quirk] \
[--dump-texture-2d CALL,TEXTURE,LEVEL,DIR] \
--timeout-seconds N
Set MOBILEGL_USE_ANGLE=1 to run DirectGLES replay with packaged ANGLE
@@ -46,6 +48,8 @@ sample with an explicit LOD that ANGLE llvmpipe cannot take a LOD bias on
(MOBILEGL_AVOID_EXPLICIT_LOD_BIAS=1).
Pass --coherent-as-flush for traces whose engine writes persistent
GL_MAP_FLUSH_EXPLICIT_BIT maps it never flushes (MOBILEGL_COHERENT_AS_FLUSH=1).
Pass --num-subgroups-quirk to derive compute gl_NumSubgroups instead of reading
the Vulkan builtin (MOBILEGL_NUM_SUBGROUPS_QUIRK=1).
EOF
}
@@ -104,6 +108,8 @@ use_pbuffer=0
avoid_angle_llvmpipe_sampler_mipmap_min_filter=0
avoid_angle_llvmpipe_explicit_lod_bias=0
coherent_as_flush=0
num_subgroups_quirk=0
texture_2d_dumps=""
timeout_seconds=""
while [ "$#" -gt 0 ]; do
@@ -144,6 +150,8 @@ while [ "$#" -gt 0 ]; do
shift 1
;;
--coherent-as-flush) coherent_as_flush=1; shift 1 ;;
--num-subgroups-quirk) num_subgroups_quirk=1; shift 1 ;;
--dump-texture-2d) texture_2d_dumps="$(next_arg "$@")"; shift 2 ;;
--timeout-seconds) timeout_seconds="$(next_arg "$@")"; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) die "unknown argument: $1" ;;
@@ -262,6 +270,27 @@ copy_app_artifact() {
fi
}
copy_texture_2d_dumps() {
[ -n "${texture_2d_dumps}" ] || return 0
saved_ifs="${IFS}"
IFS=';'
set -- ${texture_2d_dumps}
IFS="${saved_ifs}"
for dump_point in "$@"; do
dump_dir="${dump_point#*,}"
dump_dir="${dump_dir#*,}"
dump_dir="${dump_dir#*,}"
[ -n "${dump_dir}" ] || continue
dump_name="$(basename "${dump_dir}")"
destination_dir="${result_dir}/${dump_name}"
mkdir -p "${destination_dir}"
if ! adb_device_path exec-out run-as "${package_name}" tar -C "${dump_dir}" -cf - . | tar -xf - -C "${destination_dir}"; then
echo "trace-replay-ci.sh: warning: failed to copy texture dump ${dump_dir}" >&2
rm -rf "${destination_dir}"
fi
done
}
prepare_fixture() {
fixture_dir="${fixture_root}/${safe_case}"
rm -rf "${fixture_dir}"
@@ -339,6 +368,12 @@ run_retrace() {
if [ "${coherent_as_flush}" -eq 1 ]; then
set -- "$@" --ez coherent_as_flush true
fi
if [ "${num_subgroups_quirk}" -eq 1 ]; then
set -- "$@" --ez num_subgroups_quirk true
fi
if [ -n "${texture_2d_dumps}" ]; then
set -- "$@" --es texture_2d_dumps "${texture_2d_dumps}"
fi
set -- "$@" \
--es output_dir "${app_dir}/output" \
--es diff_path "${app_dir}/output/${safe_case}-diff.png" \
@@ -399,6 +434,7 @@ run_retrace() {
copy_app_artifact "${app_dir}/output/${safe_case}-diff.png" "${result_dir}/${safe_case}-${backend}-diff.png"
copy_app_artifact "${app_dir}/output/retrace.log" "${result_dir}/retrace.log"
copy_app_artifact "${app_dir}/output/mobilegl.log" "${result_dir}/mobilegl.log"
copy_texture_2d_dumps
# A replay that wrote result.json but did not pass used to print nothing but
# the JSON, which for a non-zero statusCode says only "retrace failed with
+7 -4
View File
@@ -221,10 +221,13 @@ typedef signed char khronos_int8_t;
typedef unsigned char khronos_uint8_t;
typedef signed short int khronos_int16_t;
typedef unsigned short int khronos_uint16_t;
typedef signed long int khronos_intptr_t;
typedef unsigned long int khronos_uintptr_t;
typedef signed long int khronos_ssize_t;
typedef unsigned long int khronos_usize_t;
/* `long` is 32-bit on LLP64 Windows, including 64-bit MinGW. Use the
* standard pointer-sized integer types so these remain pointer-width there. */
#include <stdint.h>
typedef intptr_t khronos_intptr_t;
typedef uintptr_t khronos_uintptr_t;
typedef intptr_t khronos_ssize_t;
typedef uintptr_t khronos_usize_t;
#if KHRONOS_SUPPORT_FLOAT
/*
+2
View File
@@ -11,6 +11,8 @@ The bundled fixtures cover:
![Minecraft 1.21.4 startup golden](fixtures/minecraft-1.21.4-startup.0000092195.png)
- minecraft-1.21.4-main-menu: captured from Minecraft 1.21.4's main menu.
![Minecraft 1.21.4 main menu golden](fixtures/minecraft-1.21.4-main-menu.0000481787.png)
- minecraft-1.21.11-main-menu: captured from Minecraft 1.21.11's main menu on a Pixel 8 Pro through FCL MobileGL.
![Minecraft 1.21.11 main menu golden](fixtures/minecraft-1.21.11-main-menu.0000205347.png)
- minecraft-1.17-main-menu-854: captured from Minecraft 1.17's 854x480 main menu through FCL MobileGL capture.
![Minecraft 1.17 854x480 main menu golden](fixtures/minecraft-1.17-main-menu-854.0000117757.png)
- minecraft-1.21.4-in-world: captured from Minecraft 1.21.4 after entering a singleplayer world.
+235 -10
View File
@@ -13,8 +13,13 @@
#include <cstring>
#include <fstream>
#include <iostream>
#include <limits>
#include <string>
#if defined(_WIN32)
#include <direct.h>
#else
#include <sys/stat.h>
#endif
#include <vector>
// Dumps every colour attachment (and the depth attachment) of every live framebuffer
@@ -34,6 +39,7 @@ using PfnGetIntegerv = void (*)(GLenum, GLint *);
using PfnGetError = GLenum (*)(void);
constexpr const char *kDumpPointsEnv = "MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS";
constexpr const char *kTexture2dDumpPointsEnv = "MOBILEGL_TRACE_DUMP_TEXTURE_2D";
constexpr const char *kScanLimitEnv = "MOBILEGL_TRACE_DUMP_FBO_SCAN_LIMIT";
constexpr unsigned kDefaultScanLimit = 1024;
@@ -45,6 +51,16 @@ struct DumpPoint {
bool done = false;
};
// CALL,TEXTURE,LEVEL,DIR entries, separated by ';'. This deliberately does not share the
// framebuffer dump grammar: an absolute Windows directory contains ':' but not ','.
struct Texture2dDumpPoint {
unsigned call = 0;
unsigned texture = 0;
unsigned level = 0;
std::string directory;
bool done = false;
};
struct AttachmentDesc {
GLint objectType = GL_NONE;
GLint objectName = 0;
@@ -56,6 +72,7 @@ struct AttachmentDesc {
};
std::vector<DumpPoint> gDumpPoints;
std::vector<Texture2dDumpPoint> gTexture2dDumpPoints;
bool gInstalled = false;
bool gConfigured = false;
retrace::Dumper *gInnerDumper = nullptr;
@@ -105,13 +122,27 @@ bool MakeDirectories(const std::string &path) {
for (std::size_t i = 0; i < path.size(); ++i) {
partial.push_back(path[i]);
const bool last = i + 1 == path.size();
if (path[i] != '/' && !last) {
#if defined(_WIN32)
const bool separator = path[i] == '/' || path[i] == '\\';
#else
const bool separator = path[i] == '/';
#endif
if (!separator && !last) {
continue;
}
if (partial == "/") {
continue;
}
#if defined(_WIN32)
if (separator && partial.size() == 3 && partial[1] == ':') {
continue;
}
#endif
#if defined(_WIN32)
if (_mkdir(partial.c_str()) != 0 && errno != EEXIST) {
#else
if (mkdir(partial.c_str(), 0755) != 0 && errno != EEXIST) {
#endif
return false;
}
}
@@ -161,6 +192,52 @@ void ParseDumpPoints(const char *spec) {
}
}
bool IsDecimal(const std::string &value) {
return !value.empty() && value.find_first_not_of("0123456789") == std::string::npos;
}
void ParseTexture2dDumpPoints(const char *spec) {
for (const std::string &entry : Split(spec, ';')) {
if (entry.empty()) {
continue;
}
const std::vector<std::string> fields = Split(entry, ',');
if (fields.size() != 4 || !IsDecimal(fields[0]) || !IsDecimal(fields[1]) ||
!IsDecimal(fields[2]) || fields[3].empty()) {
std::cerr << "warning: ignoring malformed " << kTexture2dDumpPointsEnv
<< " entry: " << entry << "\n";
continue;
}
char *end = nullptr;
const unsigned long call = std::strtoul(fields[0].c_str(), &end, 10);
if (*end != '\0' || call > std::numeric_limits<unsigned>::max()) {
std::cerr << "warning: ignoring malformed " << kTexture2dDumpPointsEnv
<< " call: " << entry << "\n";
continue;
}
const unsigned long texture = std::strtoul(fields[1].c_str(), &end, 10);
if (*end != '\0' || texture == 0 || texture > std::numeric_limits<unsigned>::max()) {
std::cerr << "warning: ignoring malformed " << kTexture2dDumpPointsEnv
<< " texture: " << entry << "\n";
continue;
}
const unsigned long level = std::strtoul(fields[2].c_str(), &end, 10);
if (*end != '\0' || level > static_cast<unsigned long>(std::numeric_limits<GLint>::max())) {
std::cerr << "warning: ignoring malformed " << kTexture2dDumpPointsEnv
<< " level: " << entry << "\n";
continue;
}
Texture2dDumpPoint point;
point.call = static_cast<unsigned>(call);
point.texture = static_cast<unsigned>(texture);
point.level = static_cast<unsigned>(level);
point.directory = fields[3];
gTexture2dDumpPoints.push_back(point);
}
}
unsigned ScanLimit() {
const char *value = std::getenv(kScanLimitEnv);
if (value == nullptr || value[0] == '\0') {
@@ -235,6 +312,45 @@ bool DescribeAttachment(GLenum attachment, AttachmentDesc &desc) {
return desc.width > 0 && desc.height > 0;
}
bool DescribeTexture2D(GLuint texture, GLint level, AttachmentDesc &desc) {
const GLint savedTexture = GetInteger(GL_TEXTURE_BINDING_2D);
glBindTexture(GL_TEXTURE_2D, texture);
desc.objectType = GL_TEXTURE;
desc.objectName = static_cast<GLint>(texture);
desc.level = level;
glGetTexLevelParameteriv(GL_TEXTURE_2D, level, GL_TEXTURE_WIDTH, &desc.width);
glGetTexLevelParameteriv(GL_TEXTURE_2D, level, GL_TEXTURE_HEIGHT, &desc.height);
glGetTexLevelParameteriv(GL_TEXTURE_2D, level, GL_TEXTURE_INTERNAL_FORMAT, &desc.internalFormat);
glGetTexLevelParameteriv(GL_TEXTURE_2D, level, GL_TEXTURE_RED_TYPE, &desc.componentType);
glBindTexture(GL_TEXTURE_2D, static_cast<GLuint>(savedTexture));
return DrainErrors() == 0 && desc.width > 0 && desc.height > 0;
}
bool ReadTexture2DFloats(const AttachmentDesc &desc, std::vector<float> &pixels) {
const std::size_t count = static_cast<std::size_t>(desc.width) * desc.height * 4;
pixels.assign(count, 0.0f);
const GLint savedTexture = GetInteger(GL_TEXTURE_BINDING_2D);
glBindTexture(GL_TEXTURE_2D, static_cast<GLuint>(desc.objectName));
if (desc.componentType == GL_INT || desc.componentType == GL_UNSIGNED_INT) {
std::vector<std::int32_t> raw(count, 0);
const GLenum type = desc.componentType == GL_INT ? GL_INT : GL_UNSIGNED_INT;
glGetTexImage(GL_TEXTURE_2D, desc.level, GL_RGBA_INTEGER, type, raw.data());
glBindTexture(GL_TEXTURE_2D, static_cast<GLuint>(savedTexture));
if (DrainErrors() != 0) {
return false;
}
for (std::size_t i = 0; i < count; ++i) {
pixels[i] = desc.componentType == GL_INT
? static_cast<float>(raw[i])
: static_cast<float>(static_cast<std::uint32_t>(raw[i]));
}
return true;
}
glGetTexImage(GL_TEXTURE_2D, desc.level, GL_RGBA, GL_FLOAT, pixels.data());
glBindTexture(GL_TEXTURE_2D, static_cast<GLuint>(savedTexture));
return DrainErrors() == 0;
}
// Reads the attachment as floats regardless of its storage: normalised and float targets
// convert on the way out, integer targets are read as integers and widened. The float view
// keeps out-of-[0,1] accumulation buffers legible in the statistics even though the PNG
@@ -308,6 +424,17 @@ std::string FormatStatistics(const std::vector<float> &pixels, unsigned channels
static_cast<double>(minimum[c]), static_cast<double>(maximum[c]), mean);
text += buffer;
}
if (pixelCount > 0) {
text += " first=(";
for (unsigned c = 0; c < channels; ++c) {
if (c > 0) {
text += ",";
}
std::snprintf(buffer, sizeof(buffer), "%.9g", static_cast<double>(pixels[c]));
text += buffer;
}
text += ")";
}
std::snprintf(buffer, sizeof(buffer), " nonfinite=%zu hash=%016llx", nonFinite,
static_cast<unsigned long long>(hash));
text += buffer;
@@ -325,8 +452,8 @@ bool WriteFloatPng(const std::string &path, const AttachmentDesc &desc, unsigned
return snapshot.writePNG(path.c_str());
}
void DumpOneAttachment(std::ofstream &manifest, const std::string &directory, unsigned framebuffer,
GLenum attachment, const char *label, bool depth) {
void DumpOneAttachment(std::ofstream &manifest, const std::string &directory,
const std::string &identity, GLenum attachment, const char *label, bool depth) {
AttachmentDesc desc;
if (!DescribeAttachment(attachment, desc)) {
return;
@@ -343,14 +470,13 @@ void DumpOneAttachment(std::ofstream &manifest, const std::string &directory, un
std::vector<float> pixels;
const bool read = ReadAttachmentFloats(desc, depth, channels, pixels);
const std::string path =
directory + "/fbo" + std::to_string(framebuffer) + "-" + label + ".png";
const std::string path = directory + "/" + identity + "-" + label + ".png";
const bool wrote = read && WriteFloatPng(path, desc, channels, pixels);
char header[512];
std::snprintf(header, sizeof(header),
"fbo %u %s object=%s name=%d level=%d size=%dx%d internalformat=0x%04x component=%s",
framebuffer, label,
"%s %s object=%s name=%d level=%d size=%dx%d internalformat=0x%04x component=%s",
identity.c_str(), label,
desc.objectType == GL_RENDERBUFFER ? "renderbuffer" : "texture", desc.objectName,
desc.level, desc.width, desc.height, static_cast<unsigned>(desc.internalFormat),
ComponentTypeName(desc.componentType));
@@ -381,13 +507,14 @@ void DumpFramebuffer(std::ofstream &manifest, const std::string &directory, unsi
}
const GLint savedReadBuffer = GetInteger(GL_READ_BUFFER);
const std::string identity = "fbo" + std::to_string(framebuffer);
for (GLint index = 0; index < maxColorAttachments; ++index) {
char label[32];
std::snprintf(label, sizeof(label), "att%d", index);
DumpOneAttachment(manifest, directory, framebuffer,
DumpOneAttachment(manifest, directory, identity,
static_cast<GLenum>(GL_COLOR_ATTACHMENT0 + index), label, false);
}
DumpOneAttachment(manifest, directory, framebuffer, GL_DEPTH_ATTACHMENT, "depth", true);
DumpOneAttachment(manifest, directory, identity, GL_DEPTH_ATTACHMENT, "depth", true);
// The read buffer is per-framebuffer state the trace goes on using; put it back.
if (framebuffer != 0 && savedReadBuffer != 0) {
@@ -416,6 +543,7 @@ void RunDumpPoint(DumpPoint &point) {
const GLint savedPackSkipRows = GetInteger(GL_PACK_SKIP_ROWS);
const GLint savedPackImageHeight = GetInteger(GL_PACK_IMAGE_HEIGHT);
const GLint savedPackSkipImages = GetInteger(GL_PACK_SKIP_IMAGES);
const GLint savedPackSwapBytes = GetInteger(GL_PACK_SWAP_BYTES);
DrainErrors();
if (savedPackBuffer != 0) {
@@ -427,6 +555,7 @@ void RunDumpPoint(DumpPoint &point) {
glPixelStorei(GL_PACK_SKIP_ROWS, 0);
glPixelStorei(GL_PACK_IMAGE_HEIGHT, 0);
glPixelStorei(GL_PACK_SKIP_IMAGES, 0);
glPixelStorei(GL_PACK_SWAP_BYTES, GL_FALSE);
DrainErrors();
const GLint maxColorAttachments = GetInteger(GL_MAX_COLOR_ATTACHMENTS);
@@ -462,6 +591,7 @@ void RunDumpPoint(DumpPoint &point) {
glPixelStorei(GL_PACK_SKIP_ROWS, savedPackSkipRows);
glPixelStorei(GL_PACK_IMAGE_HEIGHT, savedPackImageHeight);
glPixelStorei(GL_PACK_SKIP_IMAGES, savedPackSkipImages);
glPixelStorei(GL_PACK_SWAP_BYTES, savedPackSwapBytes);
DrainErrors();
std::cerr << "MOBILEGL_TRACE_FBO_DUMP: call " << retrace::callNo << " -> " << manifestPath
@@ -469,12 +599,98 @@ void RunDumpPoint(DumpPoint &point) {
point.done = true;
}
void RunTexture2dDumpPoint(Texture2dDumpPoint &point) {
if (!MakeDirectories(point.directory)) {
std::cerr << "warning: failed to create texture dump directory " << point.directory << "\n";
point.done = true;
return;
}
// glGetError is destructive. This debug-only snapshot hook deliberately starts from a
// clean error state so diagnostics below identify the dump rather than an earlier trace call.
DrainErrors();
const GLint savedReadFramebuffer = GetInteger(GL_READ_FRAMEBUFFER_BINDING);
const GLint savedPackBuffer = GetInteger(GL_PIXEL_PACK_BUFFER_BINDING);
const GLint savedPackAlignment = GetInteger(GL_PACK_ALIGNMENT);
const GLint savedPackRowLength = GetInteger(GL_PACK_ROW_LENGTH);
const GLint savedPackSkipPixels = GetInteger(GL_PACK_SKIP_PIXELS);
const GLint savedPackSkipRows = GetInteger(GL_PACK_SKIP_ROWS);
const GLint savedPackImageHeight = GetInteger(GL_PACK_IMAGE_HEIGHT);
const GLint savedPackSkipImages = GetInteger(GL_PACK_SKIP_IMAGES);
const GLint savedPackSwapBytes = GetInteger(GL_PACK_SWAP_BYTES);
DrainErrors();
if (savedPackBuffer != 0) {
glBindBuffer(GL_PIXEL_PACK_BUFFER, 0);
}
glPixelStorei(GL_PACK_ALIGNMENT, 1);
glPixelStorei(GL_PACK_ROW_LENGTH, 0);
glPixelStorei(GL_PACK_SKIP_PIXELS, 0);
glPixelStorei(GL_PACK_SKIP_ROWS, 0);
glPixelStorei(GL_PACK_IMAGE_HEIGHT, 0);
glPixelStorei(GL_PACK_SKIP_IMAGES, 0);
glPixelStorei(GL_PACK_SWAP_BYTES, GL_FALSE);
DrainErrors();
const std::string identity = "texture" + std::to_string(point.texture) +
"-level" + std::to_string(point.level);
const std::string manifestPath = point.directory + "/manifest.txt";
std::ofstream manifest(manifestPath, std::ios::trunc);
manifest << "call " << retrace::callNo << " texture " << point.texture << " level "
<< point.level << "\n";
AttachmentDesc desc;
if (glIsTexture(point.texture) == GL_FALSE ||
!DescribeTexture2D(point.texture, static_cast<GLint>(point.level), desc)) {
manifest << identity << " skipped=not-live-2d-texture\n";
} else {
std::vector<float> pixels;
const bool read = ReadTexture2DFloats(desc, pixels);
const bool wrote = read && WriteFloatPng(point.directory + "/" + identity + ".png", desc, 4, pixels);
manifest << identity << " object=texture name=" << desc.objectName << " level=" << desc.level
<< " size=" << desc.width << "x" << desc.height << " internalformat=0x" << std::hex
<< static_cast<unsigned>(desc.internalFormat) << std::dec
<< " component=" << ComponentTypeName(desc.componentType);
if (read) {
manifest << FormatStatistics(pixels, 4);
} else {
manifest << " read=failed";
}
if (!wrote) {
manifest << " png=failed";
}
manifest << "\n";
}
manifest.flush();
glBindFramebuffer(GL_READ_FRAMEBUFFER, static_cast<GLuint>(savedReadFramebuffer));
if (savedPackBuffer != 0) {
glBindBuffer(GL_PIXEL_PACK_BUFFER, static_cast<GLuint>(savedPackBuffer));
}
glPixelStorei(GL_PACK_ALIGNMENT, savedPackAlignment);
glPixelStorei(GL_PACK_ROW_LENGTH, savedPackRowLength);
glPixelStorei(GL_PACK_SKIP_PIXELS, savedPackSkipPixels);
glPixelStorei(GL_PACK_SKIP_ROWS, savedPackSkipRows);
glPixelStorei(GL_PACK_IMAGE_HEIGHT, savedPackImageHeight);
glPixelStorei(GL_PACK_SKIP_IMAGES, savedPackSkipImages);
glPixelStorei(GL_PACK_SWAP_BYTES, savedPackSwapBytes);
DrainErrors();
std::cerr << "MOBILEGL_TRACE_TEXTURE_2D_DUMP: call " << retrace::callNo << " texture "
<< point.texture << " level " << point.level << " -> " << manifestPath << "\n";
point.done = true;
}
void RunPendingDumps() {
for (DumpPoint &point : gDumpPoints) {
if (!point.done && point.call == retrace::callNo) {
RunDumpPoint(point);
}
}
for (Texture2dDumpPoint &point : gTexture2dDumpPoints) {
if (!point.done && point.call == retrace::callNo) {
RunTexture2dDumpPoint(point);
}
}
}
class DumpingDumper final : public retrace::Dumper {
@@ -511,8 +727,12 @@ void InstallIfRequested() {
if (spec != nullptr && spec[0] != '\0') {
ParseDumpPoints(spec);
}
const char *textureSpec = std::getenv(kTexture2dDumpPointsEnv);
if (textureSpec != nullptr && textureSpec[0] != '\0') {
ParseTexture2dDumpPoints(textureSpec);
}
}
if (gDumpPoints.empty()) {
if (gDumpPoints.empty() && gTexture2dDumpPoints.empty()) {
gInstalled = true;
return;
}
@@ -528,6 +748,11 @@ void InstallIfRequested() {
std::cerr << "MOBILEGL_TRACE_FBO_DUMP: armed for call " << point.call << " -> "
<< point.directory << "\n";
}
for (const Texture2dDumpPoint &point : gTexture2dDumpPoints) {
std::cerr << "MOBILEGL_TRACE_TEXTURE_2D_DUMP: armed for call " << point.call
<< " texture " << point.texture << " level " << point.level << " -> "
<< point.directory << "\n";
}
}
} // namespace mobilegl_trace_dump
+4 -3
View File
@@ -2,9 +2,10 @@
namespace mobilegl_trace_dump {
// Installs the framebuffer-attachment dump hook when MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS
// describes at least one dump point. Safe and cheap to call on every makeCurrent: the
// environment is consulted once and the hook is installed at most once.
// Installs the opt-in framebuffer-attachment and named-2D-texture dump hooks. The respective
// environments are MOBILEGL_TRACE_DUMP_FBO_ATTACHMENTS and MOBILEGL_TRACE_DUMP_TEXTURE_2D.
// Safe and cheap to call on every makeCurrent: the environment is consulted once and the hook is
// installed at most once.
void InstallIfRequested();
} // namespace mobilegl_trace_dump
+13 -2
View File
@@ -40,6 +40,13 @@
"target_call": 481787,
"timeout_seconds": 180
},
{
"name": "minecraft-1.21.11-main-menu",
"trace_archive": "minecraft-1.21.11-main-menu.tgz",
"golden": "minecraft-1.21.11-main-menu.0000205347.png",
"target_call": 205347,
"timeout_seconds": 180
},
{
"name": "minecraft-1.17-main-menu-854",
"trace_archive": "minecraft-1.17-main-menu-854.tgz",
@@ -270,11 +277,15 @@
},
{
"name": "minecraft-1.21.4-fabric-iris-iterationrp-in-world",
"ci": false,
"ci_backends": [
"DirectVulkan"
],
"trace_archive": "minecraft-1.21.4-fabric-iris-iterationrp-in-world.tgz",
"golden": "minecraft-1.21.4-fabric-iris-iterationrp-in-world.0000202020.png",
"target_call": 202020,
"timeout_seconds": 1800
"timeout_seconds": 1800,
"ssim_threshold": 0.98,
"num_subgroups_quirk": true
},
{
"name": "minecraft-1.21.4-fabric-iris-bsl-esc-menu-854",
+53 -1
View File
@@ -6,6 +6,7 @@ from pathlib import Path
TRACE_CASES_JSON = Path(__file__).with_name("trace_cases.json")
CI_BACKENDS = ("DirectGLES", "DirectVulkan")
def load_trace_case_manifest(path=TRACE_CASES_JSON):
@@ -71,6 +72,46 @@ def ci_trace_cases(cases):
return [case for case in cases if case.get("ci", True)]
def ci_backends(case):
backends = case.get("ci_backends")
if backends is None:
return CI_BACKENDS
if not isinstance(backends, list) or not backends:
raise ValueError(f"ci_backends must be a non-empty list for {case['name']}")
unknown = [backend for backend in backends if backend not in CI_BACKENDS]
if unknown:
raise ValueError(
f"unknown ci_backends for {case['name']}: {', '.join(unknown)}"
)
if len(set(backends)) != len(backends):
raise ValueError(f"ci_backends contains duplicates for {case['name']}")
return backends
def github_test_matrix(cases):
return {
"include": [
{"backend": backend, "case": case["name"]}
for case in cases
for backend in ci_backends(case)
]
}
def github_apk_matrix(cases):
backends = {
"DirectGLES": {"name": "DirectGLES", "gpu": "software"},
"DirectVulkan": {"name": "DirectVulkan", "gpu": "lavapipe"},
}
return {
"include": [
{"backend": backends[backend], "case": github_apk_case(case)}
for case in cases
for backend in ci_backends(case)
]
}
def cmake_quote(value):
return '"' + str(value).replace("\\", "/").replace('"', '\\"') + '"'
@@ -114,7 +155,14 @@ def parse_args():
parser.add_argument("--fixture-root", default="tools/trace_replay/fixtures")
parser.add_argument(
"--format",
choices=("names", "github-apk", "fixture-files", "cmake"),
choices=(
"names",
"github-test-matrix",
"github-apk",
"github-apk-matrix",
"fixture-files",
"cmake",
),
default="names",
)
return parser.parse_args()
@@ -127,8 +175,12 @@ def main():
cases = ci_trace_cases(cases)
if args.format == "names":
print(json.dumps([case["name"] for case in cases], separators=(",", ":")))
elif args.format == "github-test-matrix":
print(json.dumps(github_test_matrix(cases), separators=(",", ":")))
elif args.format == "github-apk":
print(json.dumps([github_apk_case(case) for case in cases], separators=(",", ":")))
elif args.format == "github-apk-matrix":
print(json.dumps(github_apk_matrix(cases), separators=(",", ":")))
elif args.format == "fixture-files":
if not args.case_name:
print("--case is required for --format fixture-files", file=sys.stderr)