mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-08 04:08:32 +09:00
Compare commits
49
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
24dfbb41f9 | ||
|
|
ca7878bf3a | ||
|
|
e418063b08 | ||
|
|
b43ec25bd7 | ||
|
|
525607bad6 | ||
|
|
b045024b6c | ||
|
|
14dfbeeed9 | ||
|
|
d7e79409b3 | ||
|
|
ca3b524396 | ||
|
|
071c8eb673 | ||
|
|
51a43518ac | ||
|
|
34ff95f6f5 | ||
|
|
b9d1504cc5 | ||
|
|
a223499143 | ||
|
|
5dca617f01 | ||
|
|
eb8ef893be | ||
|
|
20567fba6d | ||
|
|
e3f44e8da1 | ||
|
|
9f79a88af5 | ||
|
|
6bf32acdef | ||
|
|
23b53eacce | ||
|
|
e829e70d8b | ||
|
|
7e765e1535 | ||
|
|
bf6061811f | ||
|
|
87750c3b21 | ||
|
|
45b309db37 | ||
|
|
403c82ac4a | ||
|
|
827d46cad3 | ||
|
|
1748da0443 | ||
|
|
7efee8e3e6 | ||
|
|
08ca897a07 | ||
|
|
98f2a55214 | ||
|
|
cedc257566 | ||
|
|
821c0e0d4e | ||
|
|
bd9680ad67 | ||
|
|
6375e07030 | ||
|
|
53cac39d4e | ||
|
|
f6b1ea635b | ||
|
|
8b2711e32a | ||
|
|
9776cc8047 | ||
|
|
3b0591e0ba | ||
|
|
e62f158c22 | ||
|
|
7769156cfc | ||
|
|
0ecfdff4e7 | ||
|
|
6df5a6137f | ||
|
|
b3794f4e6a | ||
|
|
14d3901d30 | ||
|
|
d4766513e4 | ||
|
|
72dc7aa6aa |
@@ -209,7 +209,7 @@ jobs:
|
||||
- name: Load trace cases
|
||||
id: trace-cases
|
||||
run: |
|
||||
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk)" >> "$GITHUB_OUTPUT"
|
||||
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk-matrix)" >> "$GITHUB_OUTPUT"
|
||||
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
trace-fixtures:
|
||||
@@ -337,13 +337,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
backend:
|
||||
- name: DirectGLES
|
||||
gpu: software
|
||||
- name: DirectVulkan
|
||||
gpu: lavapipe
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.android) }}
|
||||
matrix: ${{ fromJSON(needs.trace-cases.outputs.android) }}
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@v1.0
|
||||
|
||||
@@ -491,6 +491,7 @@ jobs:
|
||||
- benchmark
|
||||
- integration
|
||||
outputs:
|
||||
matrix: ${{ steps.trace-cases.outputs.matrix }}
|
||||
names: ${{ steps.trace-cases.outputs.names }}
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
@@ -498,7 +499,9 @@ jobs:
|
||||
|
||||
- name: Load trace cases
|
||||
id: trace-cases
|
||||
run: echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
run: |
|
||||
echo "matrix=$(python3 tools/trace_replay/trace_cases.py --ci --format github-test-matrix)" >> "$GITHUB_OUTPUT"
|
||||
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
trace-fixtures:
|
||||
name: trace fixture (${{ matrix.case }})
|
||||
@@ -577,11 +580,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
backend:
|
||||
- DirectGLES
|
||||
- DirectVulkan
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.names) }}
|
||||
matrix: ${{ fromJSON(needs.trace-cases.outputs.matrix) }}
|
||||
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
|
||||
+4
-1
@@ -1,4 +1,4 @@
|
||||
################################################################################
|
||||
################################################################################
|
||||
# 此 .gitignore 文件已由 Microsoft(R) Visual Studio 自动创建。
|
||||
################################################################################
|
||||
|
||||
@@ -16,6 +16,9 @@ MobileGLCodeManager
|
||||
MobileGL/MG_Test/build
|
||||
/build_*
|
||||
/cmake-build*
|
||||
/build-*/
|
||||
/local.properties
|
||||
/.jspace/
|
||||
.idea
|
||||
MobileGL/MG*/build*
|
||||
MobileGL/MG*/cmake-build*
|
||||
|
||||
+42
-1
@@ -199,7 +199,6 @@ set(SPIRV_REFLECT_ENABLE_ASSERTS OFF CACHE BOOL "Enable asserts for debugging"
|
||||
set(SPIRV_REFLECT_ENABLE_ASAN OFF CACHE BOOL "Use address sanitization" FORCE)
|
||||
set(SPIRV_REFLECT_INSTALL OFF CACHE BOOL "Whether to install" FORCE)
|
||||
|
||||
# add_subdirectory(3rdparty/DiligentCore)
|
||||
add_subdirectory(3rdparty/glslang)
|
||||
add_subdirectory(3rdparty/SPIRV-Cross)
|
||||
add_subdirectory(3rdparty/VulkanMemoryAllocator)
|
||||
@@ -211,6 +210,23 @@ set(XXHASH_BUILD_XXHSUM OFF)
|
||||
option(BUILD_SHARED_LIBS OFF)
|
||||
add_subdirectory(3rdparty/xxHash/build/cmake xxhash_build EXCLUDE_FROM_ALL)
|
||||
|
||||
# Diligent-based backend. Enabled by default on local builds; only the Vulkan
|
||||
# engine from DiligentCore is built. Added after the other 3rdparty projects so
|
||||
# DiligentCore reuses the glslang / SPIRV-Cross / SPIRV-Tools / xxHash targets
|
||||
# already defined by MobileGL instead of building its bundled copies.
|
||||
option(MOBILEGL_ENABLE_DILIGENT "Enable the Diligent/Vulkan backend" ON)
|
||||
if(MOBILEGL_ENABLE_DILIGENT)
|
||||
set(DILIGENT_NO_DIRECT3D11 ON CACHE BOOL "Disable Direct3D11 backend" FORCE)
|
||||
set(DILIGENT_NO_DIRECT3D12 ON CACHE BOOL "Disable Direct3D12 backend" FORCE)
|
||||
set(DILIGENT_NO_OPENGL ON CACHE BOOL "Disable OpenGL backend" FORCE)
|
||||
set(DILIGENT_NO_METAL ON CACHE BOOL "Disable Metal backend" FORCE)
|
||||
set(DILIGENT_NO_WEBGPU ON CACHE BOOL "Disable WebGPU backend" FORCE)
|
||||
set(DILIGENT_NO_ARCHIVER ON CACHE BOOL "Disable Archiver" FORCE)
|
||||
set(DILIGENT_BUILD_TESTS OFF CACHE BOOL "Build Diligent tests" FORCE)
|
||||
set(DILIGENT_INSTALL_CORE OFF CACHE BOOL "Install DiligentCore" FORCE)
|
||||
add_subdirectory(3rdparty/DiligentCore)
|
||||
endif()
|
||||
|
||||
set(TRACY_ENABLE ${MOBILEGL_ENABLE_TRACY} CACHE BOOL "Enable Tracy, this is an internal variable" FORCE)
|
||||
|
||||
if (TRACY_ENABLE)
|
||||
@@ -396,6 +412,14 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_State/GLState/RenderbufferState/RenderbufferState.cpp
|
||||
)
|
||||
|
||||
if(MOBILEGL_ENABLE_DILIGENT)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Backend/Diligent/BackendObject_Diligent.cpp
|
||||
MobileGL/MG_Backend/Diligent/DiligentVulkan.cpp
|
||||
MobileGL/MG_Backend/Diligent/Renderer/DiligentRenderer.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
if (APPLE AND NOT MOBILEGL_IOS)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Impl/CGLImpl/CGLImpl.cpp
|
||||
@@ -436,6 +460,21 @@ set(MOBILEGL_LINK_LIBRARIES
|
||||
Threads::Threads
|
||||
)
|
||||
|
||||
if(MOBILEGL_ENABLE_DILIGENT)
|
||||
list(APPEND MOBILEGL_LINK_LIBRARIES
|
||||
Diligent-GraphicsEngineVk-static
|
||||
Diligent-GraphicsEngine
|
||||
Diligent-GraphicsEngineNextGenBase
|
||||
Diligent-GraphicsAccessories
|
||||
Diligent-ShaderTools
|
||||
Diligent-GraphicsTools
|
||||
Diligent-Common
|
||||
Diligent-Primitives
|
||||
Diligent-TargetPlatform
|
||||
Vulkan::Headers
|
||||
)
|
||||
endif()
|
||||
|
||||
set(MOBILEGL_COMPILE_DEF
|
||||
-DVMA_STATIC_VULKAN_FUNCTIONS=0
|
||||
-DVMA_DYNAMIC_VULKAN_FUNCTIONS=1
|
||||
@@ -501,6 +540,7 @@ target_compile_definitions(${CMAKE_PROJECT_NAME}
|
||||
${MOBILEGL_COMPILE_DEF}
|
||||
MOBILEGL_LOG_ACTIVE_LEVEL=${MOBILEGL_LOG_ACTIVE_LEVEL}
|
||||
$<$<BOOL:${MOBILEGL_TRACE_ANGLE_VARIANTS}>:MOBILEGL_TRACE_ANGLE_VARIANTS=1>
|
||||
$<$<BOOL:${MOBILEGL_ENABLE_DILIGENT}>:MOBILEGL_ENABLE_DILIGENT=1>
|
||||
)
|
||||
|
||||
if(UNIX AND NOT APPLE AND NOT ANDROID)
|
||||
@@ -559,6 +599,7 @@ if(NOT ANDROID)
|
||||
PUBLIC
|
||||
${MOBILEGL_COMPILE_DEF}
|
||||
MOBILEGL_LOG_ACTIVE_LEVEL=${MOBILEGL_LOG_ACTIVE_LEVEL}
|
||||
$<$<BOOL:${MOBILEGL_ENABLE_DILIGENT}>:MOBILEGL_ENABLE_DILIGENT=1>
|
||||
)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -0,0 +1,278 @@
|
||||
# Handoff: Diligent/Vulkan GL3.2 Backend for MobileGL
|
||||
|
||||
Date: 2026-08-18
|
||||
Branch: `feat/diligent-vulkan-backend`
|
||||
Repo: `~/MobileGL-dev`
|
||||
Status: **Active work-in-progress. Do not mark complete yet.**
|
||||
|
||||
---
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Implement a complete OpenGL 3.2 front-end emulation on a new Diligent/Vulkan backend inside MobileGL, instead of the DirectVulkan / DirectGLES backends.
|
||||
|
||||
Target state:
|
||||
- Fully wire MobileGL front-end `MG_State` (buffers, VAO, program, texture, sampler, framebuffer, render-state) into Diligent.
|
||||
- Implement all GL 3.2 core entry points through the Diligent backend.
|
||||
- Pass local non-Android GL3.2 tests on the Turnip Adreno 750 GPU.
|
||||
|
||||
---
|
||||
|
||||
## 2. Current Branch / Commits
|
||||
|
||||
Latest 12 commits on `feat/diligent-vulkan-backend`:
|
||||
|
||||
```
|
||||
2f5abf83 test(diligent): verify indexed DrawElements path from real frontend state
|
||||
c31b7381 feat(diligent): add basic texture binding and textured state-draw test
|
||||
8945c507 feat(diligent): clear depth in GL Clear when GL_DEPTH_BUFFER_BIT set
|
||||
558d3aea feat(diligent): add offscreen depth target and depth clear
|
||||
7e2f0bc8 feat(diligent): wire stencil and color-mask state into state PSO
|
||||
f99786f6 feat(diligent): wire viewport/scissor state into state draws
|
||||
be4cc3ce feat(diligent): wire blend/depth/cull render state into state PSO
|
||||
c855e6cf feat(diligent): verify state-driven draw with real MobileGL frontend state
|
||||
02e60bfa feat(diligent): add state-driven draw path (VAO/buffer/program to Diligent)
|
||||
a9515c92 feat(diligent): add dynamic vertex buffer upload path
|
||||
2f57582a feat(diligent): wire Clear/Draw/Present into GLFunctionsTable
|
||||
beb21123 feat(diligent): add real offscreen renderer with clear and triangle draw
|
||||
```
|
||||
|
||||
Working tree is clean.
|
||||
|
||||
---
|
||||
|
||||
## 3. Key Files
|
||||
|
||||
### Backend core
|
||||
|
||||
- `MobileGL/MG_Backend/Diligent/BackendObject_Diligent.h/.cpp`
|
||||
- `BackendObject_Diligent`
|
||||
- Creates Diligent Vulkan device/context
|
||||
- Owns `DiligentRenderer`
|
||||
- Wires `GLFunctionsTable`:
|
||||
- `Clear` (color + depth)
|
||||
- `DrawArrays`
|
||||
- `DrawElements`
|
||||
- `Present`
|
||||
- `MobileGL/MG_Backend/Diligent/DiligentVulkan.h/.cpp`
|
||||
- Backend identity helper / translation unit
|
||||
- `MobileGL/MG_Backend/Diligent/Renderer/DiligentRenderer.h/.cpp`
|
||||
- Offscreen RGBA8 + D32F targets
|
||||
- Clear / ClearDepth / DrawTriangle / DrawVertices
|
||||
- `CreateTestTexture` (RGBA8 texture + SRV + sampler)
|
||||
- `DrawFromState` (main front-end emulation draw path)
|
||||
- `CreatePipelineFromState`:
|
||||
- SPIR-V → Diligent shaders via SPIRV-Reflect
|
||||
- VAO attributes → input layout
|
||||
- primitive topology from GL mode
|
||||
- blend / depth / cull / stencil / color-mask state
|
||||
- `UploadVertexDataFromState`:
|
||||
- packs enabled VAO attributes from `BufferObject` into interleaved vertex buffer
|
||||
- supports `DrawArrays`, `DrawElements`, triangle-fan and line-loop expansion
|
||||
- Static texture binding to `g_Texture` through PSO static variables + SRB
|
||||
|
||||
### Integration changes
|
||||
|
||||
- `CMakeLists.txt`
|
||||
- New option `MOBILEGL_ENABLE_DILIGENT` (default ON for local)
|
||||
- DiligentCore added **after** glslang/SPIRV-Cross/xxHash/Vulkan-Headers so it reuses existing CMake targets
|
||||
- Diligent static libraries linked into `MobileGL` / `MobileGL_s`
|
||||
- New Diligent backend sources added
|
||||
- `MobileGL/MG_Backend/BackendObject.h`
|
||||
- New `BackendType::DiligentVulkan`
|
||||
- `MobileGL/MG_Backend/Init.cpp`
|
||||
- New backend switch case
|
||||
- `MobileGL/ConfigLoader.cpp`
|
||||
- `MOBILEGL_BACKEND_TYPE=DiligentVulkan` accepted
|
||||
- `MobileGL/MG_Test/CMakeLists.txt`
|
||||
- New `MobileGL/MG_Test/Backend/Diligent` subdirectory
|
||||
- `MobileGL/MG_Test/Backend/Diligent/`
|
||||
- `CMakeLists.txt`
|
||||
- `SanityTest.cpp`
|
||||
|
||||
### Local test files
|
||||
|
||||
- `MobileGL/MG_Test/Backend/Diligent/SanityTest.cpp`
|
||||
- `CreatesDiligentDeviceAndAdvertisesGL32`
|
||||
- `ClearsAndDrawsTriangleOffscreen`
|
||||
- `DrawsFromMobileGLState`
|
||||
- `DrawsTexturedFromMobileGLState`
|
||||
- `DrawsIndexedFromMobileGLState`
|
||||
- `DrawsRealTexturedFromMobileGLState`
|
||||
- `DrawsUniformFromMobileGLState`
|
||||
- `DrawsToOffscreenFramebufferFromMobileGLState`
|
||||
- `DrawsWithScissorFromMobileGLState`
|
||||
- `DrawsWithBlendFromMobileGLState`
|
||||
- `DrawsWithDepthTestFromMobileGLState`
|
||||
- `DrawsNamedUniformBlockFromMobileGLState`
|
||||
- `DrawsWithStencilTestFromMobileGLState`
|
||||
- `DrawsToRenderbufferFramebufferFromMobileGLState`
|
||||
- `DrawsToMultipleColorAttachmentsFromMobileGLState`
|
||||
- `DrawsIndexedBaseVertexFromMobileGLState`
|
||||
|
||||
---
|
||||
|
||||
## 4. What Works Today
|
||||
|
||||
Verified locally on Turnip Adreno 750:
|
||||
|
||||
- Diligent device/context creation
|
||||
- EGL window-surface swapchain creation path through Diligent `ISwapChain` (offscreen tests still use the offscreen target)
|
||||
- GL 3.2 / GLSL 1.50 capability advertisement
|
||||
- Offscreen color + depth rendering
|
||||
- Clear color and depth
|
||||
- Real mobilegl front-end state-driven drawing:
|
||||
- Program SPIR-V → Diligent shaders
|
||||
- VAO attributes + bound GL buffer → interleaved vertex buffer
|
||||
- `DrawArrays` path
|
||||
- `DrawElements` path (index buffer)
|
||||
- Texture basics:
|
||||
- Offscreen texture creation
|
||||
- CPU → Diligent texture (`CreateTestTexture`)
|
||||
- Static sampler2D binding to `g_Texture`
|
||||
- Textured draw test passes
|
||||
- Render state:
|
||||
- Blend enable/factors/equations
|
||||
- Stencil clear + test enabled on a D24S8 default depth/stencil target
|
||||
- Depth test enable/func/write mask
|
||||
- Cull face enable/mode/front-face winding
|
||||
- Stencil test enable/masks/ops/func/ref
|
||||
- Color write mask
|
||||
- Viewport
|
||||
- Scissor rect
|
||||
- Texture/sampler full integration:
|
||||
- `ITextureObject` → Diligent `ITexture` + SRV with automatic dirty upload
|
||||
- `SamplerObject` / texture-object sampler → Diligent `ISampler`
|
||||
- Real front-end `glTexImage2D` path (not only `CreateTestTexture`) verified
|
||||
- Global UBO upload:
|
||||
- Front-end `glUniform*` shadow → Diligent uniform buffer bound as `MGL_GLOBAL_UBO`
|
||||
- User framebuffer mapping:
|
||||
- Current draw/read FBO resolves texture attachments to Diligent RTV/DSV
|
||||
- `ReadPixels` can read back from a user FBO color attachment
|
||||
- More GL entry points wired:
|
||||
- `DrawRangeElements` / `DrawRangeElementsBaseVertex`
|
||||
- `DrawElementsBaseVertex` with real baseVertex selection
|
||||
- `MultiDrawArrays` / `MultiDrawElements` / `MultiDrawElementsBaseVertex`
|
||||
- `DrawArraysInstanced` / `DrawElementsInstanced` family
|
||||
- Indirect draw CPU fallback: `DrawArraysIndirect`, `DrawElementsIndirect`, `MultiDraw*Indirect`, `*IndirectCount`
|
||||
- `ClearBufferfv` / `ClearBufferfi` / `ClearBufferiv` / `ClearBufferuiv` (incl. stencil clear)
|
||||
- `BlitFramebuffer` / `BlitNamedFramebuffer` (same-size color copy between read/draw FBOs)
|
||||
- `CopyTexImage2D` / `CopyTexSubImage2D` (whole-color copy fallback)
|
||||
- `CopyImageSubData` (whole-texture copy between two texture objects)
|
||||
- `GenerateMipmap` (Diligent GPU mip generation on state textures)
|
||||
- `GetTexImage` / `GetTextureImage` (RGBA8 readback)
|
||||
- Fence sync entries (`FenceSync` / `ClientWaitSync` / `WaitSync` / `DeleteSync` / `GetSyncStatus`) as CPU always-signaled fallback
|
||||
- Timer query entries (`BeginTimeElapsedQuery` / `EndTimeElapsedQuery` / `QueryCounterTimestamp` / `GetQueryResult64` etc.) as CPU `steady_clock` fallback
|
||||
- `ReadPixels` from default and user color attachments
|
||||
- Primitive expansion:
|
||||
- `GL_TRIANGLE_FAN` expanded to triangle list
|
||||
- `GL_LINE_LOOP` expanded to line strip
|
||||
- Local test result:
|
||||
|
||||
```
|
||||
[ PASSED ] 16 tests
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 5. How to Build and Run Locally
|
||||
|
||||
From repo root `~/MobileGL-dev`:
|
||||
|
||||
```bash
|
||||
cmake -S . -B build-diligent -G Ninja \
|
||||
-DCMAKE_BUILD_TYPE=Debug \
|
||||
-DMOBILEGL_ENABLE_DILIGENT=ON \
|
||||
-DMOBILEGL_BUILD_TEST=ON \
|
||||
-DMOBILEGL_BUILD_BENCHMARK=OFF \
|
||||
-DFETCHCONTENT_SOURCE_DIR_GOOGLETEST="$PWD/3rdparty/DiligentCore/ThirdParty/googletest"
|
||||
|
||||
cmake --build build-diligent --target DiligentVulkanSanityTest -j 4
|
||||
|
||||
./build-diligent/MobileGL/MG_Test/Backend/Diligent/DiligentVulkanSanityTest --gtest_color=no
|
||||
```
|
||||
|
||||
Notes:
|
||||
- `MOBILEGL_BUILD_BENCHMARK=OFF` avoids network fetch of google/benchmark in this environment.
|
||||
- `FETCHCONTENT_SOURCE_DIR_GOOGLETEST` pins googletest to DiligentCore's bundled copy, avoiding flaky network clone.
|
||||
- Max 4 cores is intentional: use `-j 4`.
|
||||
|
||||
---
|
||||
|
||||
## 6. Environment Notes
|
||||
|
||||
- Host: Linux `aarch64`, glibc 2.43 (Fedora container on Android/Droidspaces)
|
||||
- GPU: Turnip Adreno 750, Vulkan API 1.4.354
|
||||
- GPU nodes available:
|
||||
- `/dev/dri/renderD128`
|
||||
- `/dev/kgsl-3d0`
|
||||
- Android SDK/NDK: `~/android-sdk` (aarch64 glibc)
|
||||
- NDK `27.3.13750724`
|
||||
- CMake `3.22.1`
|
||||
- JDK/Gradle for APK builds:
|
||||
- `~/android-build-tools/jdk17`
|
||||
- `~/android-build-tools/gradle/gradle-8.10.2`
|
||||
|
||||
---
|
||||
|
||||
## 7. Known Limitations / Not Yet Implemented
|
||||
|
||||
- User framebuffers now support texture color attachments, renderbuffer color readback, multiple simultaneous color targets, and depth/stencil texture or renderbuffer attachments.
|
||||
- Textures auto-sync `ITextureObject` → Diligent resources, including mip levels and sampler state; compressed textures and integer/3-channel formats that Diligent lacks are still skipped.
|
||||
- Global UBO (default-block `glUniform*`) and named application UBO blocks (through `glBindBufferBase`/`glUniformBlockBinding`) now upload and bind; SSBOs are still not fed from frontend buffer bindings.
|
||||
- Swapchain creation and resize are wired for native EGL window surfaces via `Diligent::ISwapChain`; `Present()` presents the active swap chain when present and otherwise flushes the offscreen target. Actual on-screen EGL presentation is still untested in this headless environment, and the X11 display/connection fields are not yet plumbed through `WindowHandle`. `SetSwapInterval` now forwards the requested sync interval to `ISwapChain::Present()`.
|
||||
- No transform feedback / GPU-accelerated queries / non-color readback; fence sync and timer queries use CPU fallbacks.
|
||||
- Draw range, multi-draw, instanced-draw wrappers, clear-buffer, blit, read-pixels, CopyTexImage*, CopyImageSubData, GenerateMipmap, GetTexImage/GetTextureImage and indirect draws are now wired; buffer subdata paths still remain.
|
||||
- A last-PSO cache now avoids recreating the pipeline when program/render-state/topology/VAO layout is unchanged; texture/UBO resources are still rebound dynamically per draw.
|
||||
- The `GLFunctionsTable` is only partially populated.
|
||||
|
||||
---
|
||||
|
||||
## 8. Recommended Next Steps
|
||||
|
||||
1. **Framebuffer / Renderbuffer mapping**
|
||||
- [x] Map `MG_State::GLState::FramebufferObject` attachments to Diligent `ITextureView` / `ITexture`.
|
||||
- [x] Support default framebuffer as current offscreen target.
|
||||
- [x] Support `glBindFramebuffer`, `glFramebufferTexture2D`, renderbuffer color/depth attachments and renderbuffer color readback.
|
||||
- [x] Multiple simultaneous color attachments.
|
||||
|
||||
2. **Texture / Sampler full integration**
|
||||
- [x] Translate MobileGL `ITextureObject` to Diligent `ITexture` and cache by `GetLifetimeId()`.
|
||||
- [x] Propagate texture unit bindings into the PSO SRB.
|
||||
- [x] Translate `SamplerObject` state into Diligent `SamplerDesc`.
|
||||
|
||||
3. **Uniform / UBO support**
|
||||
- [x] Create Diligent buffer for `ProgramObject::GetUBOData()` / `GetUBOSize()`.
|
||||
- [x] Bind the global UBO as a dynamic shader resource.
|
||||
- [x] Handle per-program uniform block bindings / named UBO blocks.
|
||||
|
||||
4. **PSO / resource caching**
|
||||
- [~] Cache PSOs by program + VAO config + render state + topology (single last-PSO fast path).
|
||||
- [~] Cache textures and samplers; buffers/SRBs can still be re-bound per draw.
|
||||
|
||||
5. **More GL 3.2 entry points**
|
||||
- [x] `DrawRangeElements`
|
||||
- [x] `MultiDraw*`
|
||||
- [x] `BlitFramebuffer` (same-size color copy)
|
||||
- [x] `ReadPixels` from non-default framebuffer
|
||||
- [x] `CopyTexImage*` / `CopyImageSubData` wired as whole-resource copies
|
||||
- [x] `GetTexImage` / `GetTextureImage` (RGBA8)
|
||||
- [x] Indirect draws (CPU fallback)
|
||||
|
||||
6. **Expand local test suite**
|
||||
- [x] Scissor test
|
||||
- [x] Blend test
|
||||
- [x] Texture filtering / sampler state test
|
||||
- [x] framebuffer offscreen render-to-texture test
|
||||
- [x] Depth test visual test
|
||||
- [x] Stencil test
|
||||
|
||||
---
|
||||
|
||||
## 9. Handoff Notes for Next Agent
|
||||
|
||||
- Do **not** reference `origin/Deprecated/Feat/Diligent`; that old implementation is intentionally ignored.
|
||||
- Work from this branch, keep tests green.
|
||||
- The command `./build-diligent/.../DiligentVulkanSanityTest` runs all 5 Diligent tests.
|
||||
- If a new test crashes during shader resource binding, remember Diligent texture SRVs need a sampler attached via `ITextureView::SetSampler()` before `InitializeStaticSRBResources()`.
|
||||
- When re-creating a PSO or buffer, call `Release()` (or assign `nullptr`) before the create call to avoid Diligent debug “Overwriting reference” assertions.
|
||||
+3
-9
@@ -66,14 +66,12 @@ namespace MobileGL::MG_Config {
|
||||
// - DISPLAY: X11 session variable, not MobileGL configuration.
|
||||
// - MOBILEGL_LOG_FILE_PATH: log-file init runs before MG_ConfigLoader::Init
|
||||
// (see MG_Util/Debug/Log.cpp).
|
||||
// - MOBILEGL_VALIDATE_SPIRV: test suites like SpirvPassTest exercise
|
||||
// ShaderCompiler without ever running MobileGL::Initialize(), and every
|
||||
// Initialize() re-runs MG_ConfigLoader::Init, which would clobber a
|
||||
// programmatic override stored here (see ShaderCompiler.cpp,
|
||||
// SpirvValidationEnabled).
|
||||
struct FeaturesTable {
|
||||
// MOBILEGL_DISABLE_TIMERQUERY: do not advertise or use GPU timer queries.
|
||||
Bool DisableTimerQuery = false;
|
||||
// MOBILEGL_ENABLE_SPIRV_VALIDATION: validate generated and transformed SPIR-V.
|
||||
// Disabled by default because validation is a diagnostics-only cost.
|
||||
Bool EnableSpirvValidation = false;
|
||||
// MOBILEGL_USE_ANGLE: load ANGLE EGL/GLES libraries.
|
||||
Bool UseAngle = false;
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
@@ -130,10 +128,6 @@ namespace MobileGL::MG_Config {
|
||||
// explicitly request a core profile via EGL_CONTEXT_OPENGL_PROFILE_MASK / a >=3.1
|
||||
// version request.
|
||||
Bool RelaxedSemantics = false;
|
||||
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that
|
||||
// rewrites the recognized workgroup prefix-scan template on Qualcomm devices with
|
||||
// subgroups wider than 32 lanes (see ShaderSourceProcessor's quirk registry).
|
||||
QuirkOverride SubgroupPrefixScanQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE: overrides the DirectVulkan quirk that
|
||||
// strips depth writes from accumulation-blended pipelines (MIN/MAX or additive
|
||||
// ONE+ONE - the multi-pass depth-equality signature) on drivers without
|
||||
|
||||
@@ -162,6 +162,7 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
inline void InitFeatures() {
|
||||
auto& features = MG_Config::Features;
|
||||
features.DisableTimerQuery = QueryEnvFlag("MOBILEGL_DISABLE_TIMERQUERY");
|
||||
features.EnableSpirvValidation = QueryEnvFlag("MOBILEGL_ENABLE_SPIRV_VALIDATION");
|
||||
features.UseAngle = QueryEnvFlag("MOBILEGL_USE_ANGLE");
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
QueryEnvVariable("MOBILEGL_TRACE_ANGLE_VARIANT", features.TraceAngleVariant, "");
|
||||
@@ -179,7 +180,6 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
features.EsprytForceDepthStencilReadbackEmulation =
|
||||
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||
features.SubgroupPrefixScanQuirk = QueryEnvQuirkOverride("MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN");
|
||||
features.MagmaDisableBlendedDepthWriteQuirk =
|
||||
QueryEnvQuirkOverride("MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE");
|
||||
features.DisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS");
|
||||
@@ -202,6 +202,7 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
}
|
||||
ENTRY(DirectGLES)
|
||||
ENTRY(DirectVulkan)
|
||||
ENTRY(DiligentVulkan)
|
||||
ENTRY(Unknown)
|
||||
MG_Config::ActiveBackendType = BackendType::Unknown;
|
||||
#undef ENTRY
|
||||
|
||||
@@ -19,6 +19,7 @@ namespace MobileGL {
|
||||
enum class BackendType {
|
||||
DirectGLES,
|
||||
DirectVulkan,
|
||||
DiligentVulkan,
|
||||
BackendTypeCount,
|
||||
Unknown = -1
|
||||
};
|
||||
|
||||
@@ -0,0 +1,906 @@
|
||||
// MobileGL - MobileGL/MG_Backend/Diligent/BackendObject_Diligent.cpp
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
|
||||
#include "BackendObject_Diligent.h"
|
||||
#include "DiligentVulkan.h"
|
||||
#include "Renderer/DiligentRenderer.h"
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
|
||||
#include <EngineFactoryVk.h>
|
||||
#include <RenderDevice.h>
|
||||
#include <DeviceContext.h>
|
||||
|
||||
#include <exception>
|
||||
#include <chrono>
|
||||
|
||||
namespace MobileGL::MG_Backend::DiligentBackend {
|
||||
namespace {
|
||||
const RendererInfo BuildInitialRendererInfo() {
|
||||
RendererInfo info;
|
||||
info.RendererName = "MobileGL (Diligent/Vulkan)";
|
||||
info.BackendName = "Diligent Vulkan";
|
||||
info.RendererGLInfo.TargetGLVersion = {3, 2, 0};
|
||||
info.RendererGLInfo.TargetGLSLVersion = {1, 50, 0};
|
||||
info.RendererGLInfo.IsCompatibilityProfile = false;
|
||||
return info;
|
||||
}
|
||||
|
||||
DiligentRenderer* GetActiveRenderer() {
|
||||
auto* backend = dynamic_cast<BackendObject_Diligent*>(pActiveBackendObject.get());
|
||||
return backend != nullptr ? backend->GetRenderer() : nullptr;
|
||||
}
|
||||
|
||||
struct DrawArraysIndirectCommand {
|
||||
Uint32 Count = 0;
|
||||
Uint32 InstanceCount = 0;
|
||||
Uint32 First = 0;
|
||||
Uint32 BaseInstance = 0;
|
||||
};
|
||||
|
||||
struct DrawElementsIndirectCommand {
|
||||
Uint32 Count = 0;
|
||||
Uint32 InstanceCount = 0;
|
||||
Uint32 FirstIndex = 0;
|
||||
Int32 BaseVertex = 0;
|
||||
Uint32 BaseInstance = 0;
|
||||
};
|
||||
|
||||
struct CpuTimerQuery {
|
||||
std::chrono::steady_clock::time_point Start;
|
||||
Uint64 TimestampNs = 0;
|
||||
Bool Available = false;
|
||||
};
|
||||
|
||||
const Uint8* ResolveIndirectCommandBytes(const void* indirect, SizeT requiredBytes, const char* label) {
|
||||
auto drawBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (drawBuffer) {
|
||||
drawBuffer->SyncPersistentMappedRange();
|
||||
const SizeT commandOffset = reinterpret_cast<SizeT>(indirect);
|
||||
if (drawBuffer->MappedData() == nullptr || commandOffset + requiredBytes > drawBuffer->GetSize()) {
|
||||
MGLOG_E_ONCE("%s skipped: invalid GL_DRAW_INDIRECT_BUFFER binding or range", label);
|
||||
return nullptr;
|
||||
}
|
||||
return drawBuffer->MappedData() + commandOffset;
|
||||
}
|
||||
if (indirect == nullptr) {
|
||||
MGLOG_E_ONCE("%s skipped: indirect pointer is null", label);
|
||||
return nullptr;
|
||||
}
|
||||
return reinterpret_cast<const Uint8*>(indirect);
|
||||
}
|
||||
|
||||
void Clear(GLbitfield mask) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr) {
|
||||
return;
|
||||
}
|
||||
if ((mask & GL_COLOR_BUFFER_BIT) != 0) {
|
||||
const auto& color = MG_State::pGLContext->GetClearColor();
|
||||
renderer->Clear(color.x(), color.y(), color.z(), color.w());
|
||||
}
|
||||
if ((mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
||||
renderer->ClearDepth(MG_State::pGLContext->GetClearDepth());
|
||||
}
|
||||
if ((mask & GL_STENCIL_BUFFER_BIT) != 0) {
|
||||
renderer->ClearStencil(MG_State::pGLContext->GetClearStencil());
|
||||
}
|
||||
}
|
||||
|
||||
void DrawArrays(GLenum mode, GLint first, GLsizei count) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->DrawFromState(mode, first, count, 0, nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void DrawElements(GLenum mode, GLsizei count, GLenum type, const void* indices) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->DrawFromState(mode, 0, count, type, indices);
|
||||
}
|
||||
}
|
||||
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices) {
|
||||
// The CPU-side UploadVertexDataFromState path already honors the selected index
|
||||
// range. start/end only restrict which indices may be referenced; they do not
|
||||
// change the vertex buffer layout for this backend.
|
||||
(void)start;
|
||||
(void)end;
|
||||
DrawElements(mode, count, type, indices);
|
||||
}
|
||||
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex) {
|
||||
(void)start;
|
||||
(void)end;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->DrawFromState(mode, 0, count, type, indices, basevertex);
|
||||
}
|
||||
}
|
||||
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr) {
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] > 0) {
|
||||
renderer->DrawFromState(mode, first[i], count[i], 0, nullptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void MultiDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr) {
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] > 0) {
|
||||
renderer->DrawFromState(mode, 0, count[i], type, indices[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void DrawElementsBaseVertex(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLint basevertex) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->DrawFromState(mode, 0, count, type, indices, basevertex);
|
||||
}
|
||||
}
|
||||
|
||||
void MultiDrawElementsBaseVertex(GLenum mode, const GLsizei* count, GLenum type,
|
||||
const GLvoid* const* indices, GLsizei drawcount,
|
||||
const GLint* basevertex) {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] > 0) {
|
||||
DrawElementsBaseVertex(mode, count[i], type, indices[i],
|
||||
basevertex != nullptr ? basevertex[i] : 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void DrawArraysIndirect(GLenum mode, const void* indirect) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr) {
|
||||
return;
|
||||
}
|
||||
const auto* bytes = ResolveIndirectCommandBytes(indirect, sizeof(DrawArraysIndirectCommand),
|
||||
"DrawArraysIndirect");
|
||||
if (bytes == nullptr) {
|
||||
return;
|
||||
}
|
||||
DrawArraysIndirectCommand cmd{};
|
||||
std::memcpy(&cmd, bytes, sizeof(cmd));
|
||||
if (cmd.Count == 0 || cmd.InstanceCount == 0) {
|
||||
return;
|
||||
}
|
||||
for (Uint32 i = 0; i < cmd.InstanceCount; ++i) {
|
||||
renderer->DrawFromState(mode, static_cast<GLint>(cmd.First), static_cast<GLsizei>(cmd.Count),
|
||||
0, nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void DrawElementsIndirect(GLenum mode, GLenum type, const void* indirect) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr) {
|
||||
return;
|
||||
}
|
||||
const SizeT indexSize = MG_Util::GetGLTypeSize(type);
|
||||
if (indexSize == 0) {
|
||||
return;
|
||||
}
|
||||
const auto* bytes = ResolveIndirectCommandBytes(indirect, sizeof(DrawElementsIndirectCommand),
|
||||
"DrawElementsIndirect");
|
||||
if (bytes == nullptr) {
|
||||
return;
|
||||
}
|
||||
DrawElementsIndirectCommand cmd{};
|
||||
std::memcpy(&cmd, bytes, sizeof(cmd));
|
||||
if (cmd.Count == 0 || cmd.InstanceCount == 0) {
|
||||
return;
|
||||
}
|
||||
const void* indices = reinterpret_cast<const void*>(static_cast<SizeT>(cmd.FirstIndex) * indexSize);
|
||||
for (Uint32 i = 0; i < cmd.InstanceCount; ++i) {
|
||||
renderer->DrawFromState(mode, 0, static_cast<GLsizei>(cmd.Count), type, indices,
|
||||
static_cast<GLint>(cmd.BaseVertex));
|
||||
}
|
||||
}
|
||||
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr || drawcount <= 0) {
|
||||
return;
|
||||
}
|
||||
const GLsizei realStride = stride == 0 ? static_cast<GLsizei>(sizeof(DrawArraysIndirectCommand)) : stride;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
const auto* bytes = ResolveIndirectCommandBytes(
|
||||
static_cast<const Uint8*>(indirect) + static_cast<SizeT>(i) * static_cast<SizeT>(realStride),
|
||||
sizeof(DrawArraysIndirectCommand), "MultiDrawArraysIndirect");
|
||||
if (bytes == nullptr) {
|
||||
continue;
|
||||
}
|
||||
DrawArraysIndirectCommand cmd{};
|
||||
std::memcpy(&cmd, bytes, sizeof(cmd));
|
||||
if (cmd.Count == 0 || cmd.InstanceCount == 0) {
|
||||
continue;
|
||||
}
|
||||
for (Uint32 instance = 0; instance < cmd.InstanceCount; ++instance) {
|
||||
renderer->DrawFromState(mode, static_cast<GLint>(cmd.First),
|
||||
static_cast<GLsizei>(cmd.Count), 0, nullptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount,
|
||||
GLsizei stride) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr || drawcount <= 0) {
|
||||
return;
|
||||
}
|
||||
const SizeT indexSize = MG_Util::GetGLTypeSize(type);
|
||||
if (indexSize == 0) {
|
||||
return;
|
||||
}
|
||||
const GLsizei realStride = stride == 0 ? static_cast<GLsizei>(sizeof(DrawElementsIndirectCommand)) : stride;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
const auto* bytes = ResolveIndirectCommandBytes(
|
||||
static_cast<const Uint8*>(indirect) + static_cast<SizeT>(i) * static_cast<SizeT>(realStride),
|
||||
sizeof(DrawElementsIndirectCommand), "MultiDrawElementsIndirect");
|
||||
if (bytes == nullptr) {
|
||||
continue;
|
||||
}
|
||||
DrawElementsIndirectCommand cmd{};
|
||||
std::memcpy(&cmd, bytes, sizeof(cmd));
|
||||
if (cmd.Count == 0 || cmd.InstanceCount == 0) {
|
||||
continue;
|
||||
}
|
||||
const void* indices = reinterpret_cast<const void*>(static_cast<SizeT>(cmd.FirstIndex) * indexSize);
|
||||
for (Uint32 instance = 0; instance < cmd.InstanceCount; ++instance) {
|
||||
renderer->DrawFromState(mode, 0, static_cast<GLsizei>(cmd.Count), type, indices,
|
||||
static_cast<GLint>(cmd.BaseVertex));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride) {
|
||||
if (MG_State::pGLContext == nullptr) {
|
||||
return;
|
||||
}
|
||||
auto paramBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
if (!paramBuffer) {
|
||||
return;
|
||||
}
|
||||
paramBuffer->SyncPersistentMappedRange();
|
||||
const Uint8* paramData = paramBuffer->MappedData();
|
||||
if (paramData == nullptr) {
|
||||
return;
|
||||
}
|
||||
Uint32 actualDrawCount = 0;
|
||||
std::memcpy(&actualDrawCount, paramData + static_cast<SizeT>(drawcount), sizeof(actualDrawCount));
|
||||
actualDrawCount = std::min<Uint32>(actualDrawCount, static_cast<Uint32>(maxdrawcount));
|
||||
MultiDrawArraysIndirect(mode, indirect, static_cast<GLsizei>(actualDrawCount), stride);
|
||||
}
|
||||
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect,
|
||||
GLintptr drawcount, GLsizei maxdrawcount, GLsizei stride) {
|
||||
if (MG_State::pGLContext == nullptr) {
|
||||
return;
|
||||
}
|
||||
auto paramBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
if (!paramBuffer) {
|
||||
return;
|
||||
}
|
||||
paramBuffer->SyncPersistentMappedRange();
|
||||
const Uint8* paramData = paramBuffer->MappedData();
|
||||
if (paramData == nullptr) {
|
||||
return;
|
||||
}
|
||||
Uint32 actualDrawCount = 0;
|
||||
std::memcpy(&actualDrawCount, paramData + static_cast<SizeT>(drawcount), sizeof(actualDrawCount));
|
||||
actualDrawCount = std::min<Uint32>(actualDrawCount, static_cast<Uint32>(maxdrawcount));
|
||||
MultiDrawElementsIndirect(mode, type, indirect, static_cast<GLsizei>(actualDrawCount), stride);
|
||||
}
|
||||
|
||||
void DrawArraysInstanced(GLenum mode, GLint first, GLsizei count, GLsizei instancecount) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || instancecount <= 0) {
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < instancecount; ++i) {
|
||||
renderer->DrawFromState(mode, first, count, 0, nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void DrawArraysInstancedBaseInstance(GLenum mode, GLint first, GLsizei count, GLsizei instancecount,
|
||||
GLuint baseinstance) {
|
||||
(void)baseinstance;
|
||||
DrawArraysInstanced(mode, first, count, instancecount);
|
||||
}
|
||||
|
||||
void DrawElementsInstanced(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || instancecount <= 0) {
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < instancecount; ++i) {
|
||||
renderer->DrawFromState(mode, 0, count, type, indices);
|
||||
}
|
||||
}
|
||||
|
||||
void DrawElementsInstancedBaseVertex(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLint basevertex) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || instancecount <= 0) {
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < instancecount; ++i) {
|
||||
renderer->DrawFromState(mode, 0, count, type, indices, basevertex);
|
||||
}
|
||||
}
|
||||
|
||||
void DrawElementsInstancedBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLuint baseinstance) {
|
||||
(void)baseinstance;
|
||||
DrawElementsInstanced(mode, count, type, indices, instancecount);
|
||||
}
|
||||
|
||||
void DrawElementsInstancedBaseVertexBaseInstance(GLenum mode, GLsizei count, GLenum type,
|
||||
const void* indices, GLsizei instancecount,
|
||||
GLint basevertex, GLuint baseinstance) {
|
||||
(void)baseinstance;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || instancecount <= 0) {
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < instancecount; ++i) {
|
||||
renderer->DrawFromState(mode, 0, count, type, indices, basevertex);
|
||||
}
|
||||
}
|
||||
|
||||
void ClearBufferfv(GLenum buffer, GLint drawbuffer, const GLfloat* value) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || value == nullptr) {
|
||||
return;
|
||||
}
|
||||
if (buffer == GL_COLOR && drawbuffer == 0) {
|
||||
renderer->Clear(value[0], value[1], value[2], value[3]);
|
||||
} else if (buffer == GL_DEPTH && drawbuffer == 0) {
|
||||
renderer->ClearDepth(value[0]);
|
||||
}
|
||||
}
|
||||
|
||||
void ClearBufferiv(GLenum buffer, GLint drawbuffer, const GLint* value) {
|
||||
if (value == nullptr) {
|
||||
return;
|
||||
}
|
||||
if (buffer == GL_STENCIL) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->ClearStencil(static_cast<Uint32>(value[0]));
|
||||
}
|
||||
return;
|
||||
}
|
||||
Float color[4] = {
|
||||
static_cast<Float>(value[0]) / 255.0f,
|
||||
static_cast<Float>(value[1]) / 255.0f,
|
||||
static_cast<Float>(value[2]) / 255.0f,
|
||||
static_cast<Float>(value[3]) / 255.0f,
|
||||
};
|
||||
ClearBufferfv(buffer, drawbuffer, color);
|
||||
}
|
||||
|
||||
void ClearBufferuiv(GLenum buffer, GLint drawbuffer, const GLuint* value) {
|
||||
if (value == nullptr) {
|
||||
return;
|
||||
}
|
||||
if (buffer == GL_STENCIL) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->ClearStencil(value[0]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
Float color[4] = {
|
||||
static_cast<Float>(value[0]) / 255.0f,
|
||||
static_cast<Float>(value[1]) / 255.0f,
|
||||
static_cast<Float>(value[2]) / 255.0f,
|
||||
static_cast<Float>(value[3]) / 255.0f,
|
||||
};
|
||||
ClearBufferfv(buffer, drawbuffer, color);
|
||||
}
|
||||
|
||||
void ClearBufferfi(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || buffer != GL_DEPTH_STENCIL) {
|
||||
return;
|
||||
}
|
||||
(void)drawbuffer;
|
||||
renderer->ClearDepth(depth);
|
||||
renderer->ClearStencil(static_cast<Uint32>(stencil));
|
||||
}
|
||||
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || pixels == nullptr) {
|
||||
return;
|
||||
}
|
||||
// The Diligent backend's offscreen targets are RGBA8; the frontend currently
|
||||
// uses this entry for the common GL_RGBA/GL_UNSIGNED_BYTE readback path.
|
||||
if (format != GL_RGBA || type != GL_UNSIGNED_BYTE) {
|
||||
return;
|
||||
}
|
||||
renderer->ReadPixels(static_cast<Uint32>(x), static_cast<Uint32>(y),
|
||||
static_cast<Uint32>(width), static_cast<Uint32>(height), pixels);
|
||||
}
|
||||
|
||||
void BlitFramebuffer(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLbitfield mask, GLenum filter) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->BlitFramebuffer(srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1, dstY1,
|
||||
mask, filter);
|
||||
}
|
||||
}
|
||||
|
||||
void BlitNamedFramebuffer(const SharedPtr<MG_State::GLState::FramebufferObject>& readFramebuffer,
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& drawFramebuffer,
|
||||
GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLbitfield mask, GLenum filter) {
|
||||
(void)srcX0;
|
||||
(void)srcY0;
|
||||
(void)srcX1;
|
||||
(void)srcY1;
|
||||
(void)dstX0;
|
||||
(void)dstY0;
|
||||
(void)dstX1;
|
||||
(void)dstY1;
|
||||
(void)filter;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->BlitNamedFramebuffer(readFramebuffer, drawFramebuffer, mask);
|
||||
}
|
||||
}
|
||||
|
||||
void CopyTexImage2D(GLenum target, GLint level, GLenum internalformat, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height, GLint border) {
|
||||
(void)level;
|
||||
(void)internalformat;
|
||||
(void)x;
|
||||
(void)y;
|
||||
(void)width;
|
||||
(void)height;
|
||||
(void)border;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr || target != GL_TEXTURE_2D) {
|
||||
return;
|
||||
}
|
||||
auto& unit = MG_State::pGLContext->GetTextureUnitObject(MG_State::pGLContext->GetActiveTextureUnit());
|
||||
auto texture = unit.GetBindingSlot(TextureTarget::Texture2D).GetBoundObject();
|
||||
if (texture) {
|
||||
renderer->CopyReadFramebufferToTexture(*texture);
|
||||
}
|
||||
}
|
||||
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height) {
|
||||
(void)level;
|
||||
(void)xoffset;
|
||||
(void)yoffset;
|
||||
(void)x;
|
||||
(void)y;
|
||||
(void)width;
|
||||
(void)height;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr || target != GL_TEXTURE_2D) {
|
||||
return;
|
||||
}
|
||||
auto& unit = MG_State::pGLContext->GetTextureUnitObject(MG_State::pGLContext->GetActiveTextureUnit());
|
||||
auto texture = unit.GetBindingSlot(TextureTarget::Texture2D).GetBoundObject();
|
||||
if (texture) {
|
||||
renderer->CopyReadFramebufferToTexture(*texture);
|
||||
}
|
||||
}
|
||||
|
||||
void GetTexImage(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr || target != GL_TEXTURE_2D ||
|
||||
format != GL_RGBA || type != GL_UNSIGNED_BYTE || pixels == nullptr) {
|
||||
return;
|
||||
}
|
||||
auto& unit = MG_State::pGLContext->GetTextureUnitObject(MG_State::pGLContext->GetActiveTextureUnit());
|
||||
auto texture = unit.GetBindingSlot(TextureTarget::Texture2D).GetBoundObject();
|
||||
if (texture) {
|
||||
renderer->ReadTextureImage(*texture, static_cast<Uint32>(level), pixels);
|
||||
}
|
||||
}
|
||||
|
||||
void GetTextureImage(const SharedPtr<MG_State::GLState::ITextureObject>& texture,
|
||||
TextureUploadTarget uploadTarget, GLint level, GLenum format, GLenum type,
|
||||
GLsizei bufSize, GLvoid* pixels) {
|
||||
(void)uploadTarget;
|
||||
(void)bufSize;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || !texture || format != GL_RGBA || type != GL_UNSIGNED_BYTE ||
|
||||
pixels == nullptr) {
|
||||
return;
|
||||
}
|
||||
renderer->ReadTextureImage(*texture, static_cast<Uint32>(level), pixels);
|
||||
}
|
||||
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth) {
|
||||
(void)srcTarget;
|
||||
(void)srcLevel;
|
||||
(void)srcX;
|
||||
(void)srcY;
|
||||
(void)srcZ;
|
||||
(void)dstTarget;
|
||||
(void)dstLevel;
|
||||
(void)dstX;
|
||||
(void)dstY;
|
||||
(void)dstZ;
|
||||
(void)srcWidth;
|
||||
(void)srcHeight;
|
||||
(void)srcDepth;
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr && srcTexture && dstTexture) {
|
||||
renderer->CopyTextureSubData(*srcTexture, *dstTexture);
|
||||
}
|
||||
}
|
||||
|
||||
void GenerateMipmap(GLenum target) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer == nullptr || MG_State::pGLContext == nullptr || target != GL_TEXTURE_2D) {
|
||||
return;
|
||||
}
|
||||
auto& unit = MG_State::pGLContext->GetTextureUnitObject(MG_State::pGLContext->GetActiveTextureUnit());
|
||||
auto texture = unit.GetBindingSlot(TextureTarget::Texture2D).GetBoundObject();
|
||||
if (texture) {
|
||||
renderer->GenerateMipmap(*texture);
|
||||
}
|
||||
}
|
||||
|
||||
Bool IsTimerQuerySupported() {
|
||||
return true;
|
||||
}
|
||||
|
||||
BackendQueryHandle BeginTimeElapsedQuery() {
|
||||
auto* query = new CpuTimerQuery;
|
||||
query->Start = std::chrono::steady_clock::now();
|
||||
query->Available = false;
|
||||
return query;
|
||||
}
|
||||
|
||||
void EndTimeElapsedQuery(BackendQueryHandle query) {
|
||||
if (query == nullptr) {
|
||||
return;
|
||||
}
|
||||
auto* cpuQuery = static_cast<CpuTimerQuery*>(query);
|
||||
const auto now = std::chrono::steady_clock::now();
|
||||
cpuQuery->TimestampNs = static_cast<Uint64>(
|
||||
std::chrono::duration_cast<std::chrono::nanoseconds>(now - cpuQuery->Start).count());
|
||||
cpuQuery->Available = true;
|
||||
}
|
||||
|
||||
BackendQueryHandle QueryCounterTimestamp() {
|
||||
auto* query = new CpuTimerQuery;
|
||||
query->TimestampNs = static_cast<Uint64>(
|
||||
std::chrono::duration_cast<std::chrono::nanoseconds>(
|
||||
std::chrono::steady_clock::now().time_since_epoch()).count());
|
||||
query->Available = true;
|
||||
return query;
|
||||
}
|
||||
|
||||
Bool IsQueryResultAvailable(BackendQueryHandle query) {
|
||||
return query != nullptr && static_cast<CpuTimerQuery*>(query)->Available;
|
||||
}
|
||||
|
||||
Bool GetQueryResult64(BackendQueryHandle query, Bool wait, Uint64* outNanoseconds) {
|
||||
if (query == nullptr || outNanoseconds == nullptr) {
|
||||
return false;
|
||||
}
|
||||
auto* cpuQuery = static_cast<CpuTimerQuery*>(query);
|
||||
if (!cpuQuery->Available && !wait) {
|
||||
return false;
|
||||
}
|
||||
*outNanoseconds = cpuQuery->TimestampNs;
|
||||
return true;
|
||||
}
|
||||
|
||||
void DeleteBackendQuery(BackendQueryHandle query) {
|
||||
delete static_cast<CpuTimerQuery*>(query);
|
||||
}
|
||||
|
||||
BackendSyncHandle FenceSync() {
|
||||
// CPU fallback fence: always signaled is a valid implementation for a
|
||||
// backend without native sync primitives. The handle still round-trips
|
||||
// through ClientWaitSync/DeleteSync so frontend state stays balanced.
|
||||
return new int(0);
|
||||
}
|
||||
|
||||
GLenum ClientWaitSync(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout) {
|
||||
(void)sync;
|
||||
(void)flags;
|
||||
(void)timeout;
|
||||
return GL_ALREADY_SIGNALED;
|
||||
}
|
||||
|
||||
void WaitSync(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout) {
|
||||
(void)sync;
|
||||
(void)flags;
|
||||
(void)timeout;
|
||||
}
|
||||
|
||||
void DeleteSync(BackendSyncHandle sync) {
|
||||
delete static_cast<int*>(sync);
|
||||
}
|
||||
|
||||
Bool GetSyncStatus(BackendSyncHandle sync) {
|
||||
(void)sync;
|
||||
return true;
|
||||
}
|
||||
|
||||
void SetSwapInterval(Int interval) {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->SetSwapInterval(interval > 0 ? static_cast<Uint32>(interval) : 0);
|
||||
}
|
||||
}
|
||||
|
||||
void Present() {
|
||||
auto* renderer = GetActiveRenderer();
|
||||
if (renderer != nullptr) {
|
||||
renderer->Present();
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
BackendObject_Diligent::BackendObject_Diligent()
|
||||
: m_rendererInfo(BuildInitialRendererInfo()) {}
|
||||
|
||||
BackendObject_Diligent::~BackendObject_Diligent() {
|
||||
m_pRenderer.reset();
|
||||
m_pContext.Release();
|
||||
m_pDevice.Release();
|
||||
m_pFactoryVk = nullptr;
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::CreateDiligentDevice() {
|
||||
if (m_pDevice && m_pContext) {
|
||||
return true;
|
||||
}
|
||||
|
||||
try {
|
||||
if (m_pFactoryVk == nullptr) {
|
||||
m_pFactoryVk = ::Diligent::GetEngineFactoryVk();
|
||||
if (m_pFactoryVk == nullptr) {
|
||||
MGLOG_E("Diligent: failed to load Vulkan engine factory");
|
||||
return false;
|
||||
}
|
||||
m_pFactoryVk->SetBreakOnError(false);
|
||||
}
|
||||
|
||||
::Diligent::Uint32 numAdapters = 0;
|
||||
m_pFactoryVk->EnumerateAdapters(::Diligent::Version{}, numAdapters, nullptr);
|
||||
if (numAdapters == 0) {
|
||||
MGLOG_W("Diligent: no Vulkan adapters available; skipping device creation");
|
||||
return false;
|
||||
}
|
||||
|
||||
::Diligent::EngineVkCreateInfo engineCI;
|
||||
::Diligent::ImmediateContextCreateInfo ctxCI;
|
||||
ctxCI.Name = "MobileGL Diligent Main Context";
|
||||
ctxCI.QueueId = 0;
|
||||
ctxCI.Priority = ::Diligent::QUEUE_PRIORITY_MEDIUM;
|
||||
engineCI.NumImmediateContexts = 1;
|
||||
engineCI.pImmediateContextInfo = &ctxCI;
|
||||
|
||||
::Diligent::IRenderDevice* pDevice = nullptr;
|
||||
::Diligent::IDeviceContext* pContext = nullptr;
|
||||
m_pFactoryVk->CreateDeviceAndContextsVk(engineCI, &pDevice, &pContext);
|
||||
if (pDevice == nullptr || pContext == nullptr) {
|
||||
MGLOG_E("Diligent: failed to create Vulkan device/context");
|
||||
return false;
|
||||
}
|
||||
|
||||
m_pDevice.Attach(pDevice);
|
||||
m_pContext.Attach(pContext);
|
||||
MGLOG_I("Diligent: Vulkan device created");
|
||||
return true;
|
||||
} catch (const std::exception& e) {
|
||||
MGLOG_W("Diligent: Vulkan device creation failed: %s", e.what());
|
||||
return false;
|
||||
} catch (...) {
|
||||
MGLOG_W("Diligent: Vulkan device creation failed");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
void BackendObject_Diligent::Initialize() {
|
||||
if (m_initialized) {
|
||||
return;
|
||||
}
|
||||
if (!CreateDiligentDevice()) {
|
||||
MGLOG_W("Diligent: backend initialization failed");
|
||||
return;
|
||||
}
|
||||
|
||||
m_pRenderer = std::make_unique<DiligentRenderer>(m_pDevice, m_pContext);
|
||||
if (!m_pRenderer->Initialize(256, 256)) {
|
||||
MGLOG_W("Diligent: renderer initialization failed");
|
||||
m_pRenderer.reset();
|
||||
return;
|
||||
}
|
||||
|
||||
m_functions.GL.Clear = Clear;
|
||||
m_functions.GL.DrawArrays = DrawArrays;
|
||||
m_functions.GL.DrawElements = DrawElements;
|
||||
m_functions.GL.DrawElementsBaseVertex = DrawElementsBaseVertex;
|
||||
m_functions.GL.DrawRangeElements = DrawRangeElements;
|
||||
m_functions.GL.DrawRangeElementsBaseVertex = DrawRangeElementsBaseVertex;
|
||||
m_functions.GL.MultiDrawArrays = MultiDrawArrays;
|
||||
m_functions.GL.MultiDrawElements = MultiDrawElements;
|
||||
m_functions.GL.MultiDrawElementsBaseVertex = MultiDrawElementsBaseVertex;
|
||||
m_functions.GL.DrawArraysInstanced = DrawArraysInstanced;
|
||||
m_functions.GL.DrawArraysInstancedBaseInstance = DrawArraysInstancedBaseInstance;
|
||||
m_functions.GL.DrawElementsInstanced = DrawElementsInstanced;
|
||||
m_functions.GL.DrawElementsInstancedBaseVertex = DrawElementsInstancedBaseVertex;
|
||||
m_functions.GL.DrawElementsInstancedBaseInstance = DrawElementsInstancedBaseInstance;
|
||||
m_functions.GL.DrawElementsInstancedBaseVertexBaseInstance = DrawElementsInstancedBaseVertexBaseInstance;
|
||||
m_functions.GL.DrawArraysIndirect = DrawArraysIndirect;
|
||||
m_functions.GL.DrawElementsIndirect = DrawElementsIndirect;
|
||||
m_functions.GL.MultiDrawArraysIndirect = MultiDrawArraysIndirect;
|
||||
m_functions.GL.MultiDrawElementsIndirect = MultiDrawElementsIndirect;
|
||||
m_functions.GL.MultiDrawArraysIndirectCount = MultiDrawArraysIndirectCount;
|
||||
m_functions.GL.MultiDrawElementsIndirectCount = MultiDrawElementsIndirectCount;
|
||||
m_functions.GL.ClearBufferfv = ClearBufferfv;
|
||||
m_functions.GL.ClearBufferfi = ClearBufferfi;
|
||||
m_functions.GL.ClearBufferiv = ClearBufferiv;
|
||||
m_functions.GL.ClearBufferuiv = ClearBufferuiv;
|
||||
m_functions.GL.BlitFramebuffer = BlitFramebuffer;
|
||||
m_functions.GL.BlitNamedFramebuffer = BlitNamedFramebuffer;
|
||||
m_functions.GL.CopyTexImage2D = CopyTexImage2D;
|
||||
m_functions.GL.CopyTexSubImage2D = CopyTexSubImage2D;
|
||||
m_functions.GL.CopyImageSubData = CopyImageSubData;
|
||||
m_functions.GL.GenerateMipmap = GenerateMipmap;
|
||||
m_functions.GL.GetTexImage = GetTexImage;
|
||||
m_functions.GL.GetTextureImage = GetTextureImage;
|
||||
m_functions.GL.ReadPixels = ReadPixels;
|
||||
m_functions.GL.FenceSync = FenceSync;
|
||||
m_functions.GL.ClientWaitSync = ClientWaitSync;
|
||||
m_functions.GL.WaitSync = WaitSync;
|
||||
m_functions.GL.DeleteSync = DeleteSync;
|
||||
m_functions.GL.GetSyncStatus = GetSyncStatus;
|
||||
m_functions.GL.IsTimerQuerySupported = IsTimerQuerySupported;
|
||||
m_functions.GL.BeginTimeElapsedQuery = BeginTimeElapsedQuery;
|
||||
m_functions.GL.EndTimeElapsedQuery = EndTimeElapsedQuery;
|
||||
m_functions.GL.QueryCounterTimestamp = QueryCounterTimestamp;
|
||||
m_functions.GL.IsQueryResultAvailable = IsQueryResultAvailable;
|
||||
m_functions.GL.GetQueryResult64 = GetQueryResult64;
|
||||
m_functions.GL.DeleteBackendQuery = DeleteBackendQuery;
|
||||
m_functions.Present = Present;
|
||||
m_functions.SetSwapInterval = SetSwapInterval;
|
||||
|
||||
m_initialized = true;
|
||||
}
|
||||
|
||||
DiligentRenderer* BackendObject_Diligent::GetRenderer() {
|
||||
return m_pRenderer.get();
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::InitCapabilities() {
|
||||
// Skeleton: no format probing yet. The backend advertises GL 3.2 core
|
||||
// capability, and the capability tables will be filled as resource
|
||||
// creation paths are ported.
|
||||
m_backendCapabilitiesInitialized = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::InitWindowSurface() {
|
||||
if (!m_windowHandle.Handle) {
|
||||
MGLOG_E("BackendObject_Diligent::InitWindowSurface failed: native window handle is null");
|
||||
return false;
|
||||
}
|
||||
if (m_pRenderer == nullptr || m_pFactoryVk == nullptr) {
|
||||
MGLOG_E("BackendObject_Diligent::InitWindowSurface failed: renderer/factory is not ready");
|
||||
return false;
|
||||
}
|
||||
return m_pRenderer->CreateSwapChain(m_pFactoryVk, m_windowHandle,
|
||||
m_windowHandle.Width, m_windowHandle.Height);
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::InitPbufferSurface(EGLint width, EGLint height) {
|
||||
// The Diligent backend keeps its offscreen target for pbuffer EGL surfaces.
|
||||
// A future enhancement can resize/recreate the offscreen target to match the
|
||||
// pbuffer dimensions.
|
||||
(void)width;
|
||||
(void)height;
|
||||
return m_pRenderer != nullptr;
|
||||
}
|
||||
|
||||
void BackendObject_Diligent::ReleaseEGLResources() {
|
||||
if (m_pRenderer != nullptr) {
|
||||
m_pRenderer->ReleaseSwapChain();
|
||||
}
|
||||
BackendObject::ReleaseEGLResources();
|
||||
}
|
||||
|
||||
void BackendObject_Diligent::OnEGLSurfaceReleased(EGLSurface surface) {
|
||||
(void)surface;
|
||||
if (m_pRenderer != nullptr) {
|
||||
m_pRenderer->ReleaseSwapChain();
|
||||
}
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_initialized) {
|
||||
MGLOG_E("BackendObject_Diligent::CreateEGLWindowSurface failed: backend not initialized");
|
||||
return false;
|
||||
}
|
||||
if (!handle.Handle || (handle.Backend != WindowBackend::Android && handle.Backend != WindowBackend::X11 &&
|
||||
handle.Backend != WindowBackend::MetalLayer && handle.Backend != WindowBackend::Win32)) {
|
||||
MGLOG_E("BackendObject_Diligent::CreateEGLWindowSurface failed: unsupported native window backend");
|
||||
return false;
|
||||
}
|
||||
return RegisterEGLWindowSurface(surface, handle);
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_initialized) {
|
||||
MGLOG_E("BackendObject_Diligent::CreateEGLPbufferSurface failed: backend not initialized");
|
||||
return false;
|
||||
}
|
||||
return RegisterEGLPbufferSurface(surface, width, height);
|
||||
}
|
||||
|
||||
Bool BackendObject_Diligent::ResizeEGLWindowSurface(EGLSurface surface, Uint32 width, Uint32 height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!BackendObject::ResizeEGLWindowSurface(surface, width, height)) {
|
||||
return false;
|
||||
}
|
||||
if (m_eglSurface == surface && m_pRenderer != nullptr) {
|
||||
return m_pRenderer->ResizeSwapChain(width, height);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
const RendererInfo& BackendObject_Diligent::GetRendererInfo() const {
|
||||
return m_rendererInfo;
|
||||
}
|
||||
|
||||
String BackendObject_Diligent::GetBackendAPIVersionString() const {
|
||||
return "Diligent Vulkan 0.1 (GL 3.2 skeleton)";
|
||||
}
|
||||
|
||||
const GlobalBackendFunctionsTable& BackendObject_Diligent::GetBackendFunctions() const {
|
||||
return m_functions;
|
||||
}
|
||||
|
||||
const DynamicBackendParameters& BackendObject_Diligent::GetDynamicParameters() const {
|
||||
return m_dynamicParameters;
|
||||
}
|
||||
|
||||
BackendType BackendObject_Diligent::GetBackendType() const {
|
||||
return BackendType::DiligentVulkan;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DiligentBackend
|
||||
@@ -0,0 +1,72 @@
|
||||
// MobileGL - MobileGL/MG_Backend/Diligent/BackendObject_Diligent.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <Includes.h>
|
||||
#include "../BackendObject.h"
|
||||
|
||||
// X11 (pulled in by Includes.h through Vulkan-Headers) defines True/False as
|
||||
// macros, which collide with Diligent's Bool constants in BasicTypes.h.
|
||||
#if defined(True)
|
||||
#undef True
|
||||
#endif
|
||||
#if defined(False)
|
||||
#undef False
|
||||
#endif
|
||||
|
||||
#include <RefCntAutoPtr.hpp>
|
||||
|
||||
namespace Diligent {
|
||||
struct IEngineFactoryVk;
|
||||
struct IRenderDevice;
|
||||
struct IDeviceContext;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DiligentBackend {
|
||||
class DiligentRenderer;
|
||||
|
||||
// New Diligent/Vulkan backend, implemented from scratch on top of
|
||||
// DiligentCore. The backend object owns the Diligent device/context and
|
||||
// currently advertises OpenGL 3.2 core capability; the GL function table
|
||||
// is intentionally empty until drawing/resource paths are ported.
|
||||
class BackendObject_Diligent : public BackendObject {
|
||||
public:
|
||||
BackendObject_Diligent();
|
||||
~BackendObject_Diligent() override;
|
||||
|
||||
void Initialize() override;
|
||||
Bool InitCapabilities() override;
|
||||
Bool InitWindowSurface() override;
|
||||
Bool InitPbufferSurface(EGLint width, EGLint height) override;
|
||||
Bool CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) override;
|
||||
Bool CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) override;
|
||||
Bool ResizeEGLWindowSurface(EGLSurface surface, Uint32 width, Uint32 height) override;
|
||||
void OnEGLSurfaceReleased(EGLSurface surface) override;
|
||||
|
||||
const RendererInfo& GetRendererInfo() const override;
|
||||
String GetBackendAPIVersionString() const override;
|
||||
const GlobalBackendFunctionsTable& GetBackendFunctions() const override;
|
||||
const DynamicBackendParameters& GetDynamicParameters() const override;
|
||||
BackendType GetBackendType() const override;
|
||||
void ReleaseEGLResources() override;
|
||||
|
||||
DiligentRenderer* GetRenderer();
|
||||
|
||||
private:
|
||||
Bool CreateDiligentDevice();
|
||||
|
||||
RendererInfo m_rendererInfo;
|
||||
DynamicBackendParameters m_dynamicParameters;
|
||||
GlobalBackendFunctionsTable m_functions{};
|
||||
::Diligent::IEngineFactoryVk* m_pFactoryVk = nullptr;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::IRenderDevice> m_pDevice;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::IDeviceContext> m_pContext;
|
||||
std::unique_ptr<DiligentRenderer> m_pRenderer;
|
||||
Bool m_initialized = false;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DiligentBackend
|
||||
@@ -0,0 +1,8 @@
|
||||
// MobileGL - MobileGL/MG_Backend/Diligent/DiligentVulkan.cpp
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
|
||||
#include "DiligentVulkan.h"
|
||||
@@ -0,0 +1,17 @@
|
||||
// MobileGL - MobileGL/MG_Backend/Diligent/DiligentVulkan.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DiligentBackend {
|
||||
// Backend identity string used by the backend object and local smoke tests.
|
||||
inline String GetDiligentVulkanBackendName() {
|
||||
return "DiligentVulkan";
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DiligentBackend
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,153 @@
|
||||
// MobileGL - MobileGL/MG_Backend/Diligent/Renderer/DiligentRenderer.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
// X11 (pulled in by Includes.h through Vulkan-Headers) defines True/False as
|
||||
// macros, which collide with Diligent's Bool constants in BasicTypes.h.
|
||||
#if defined(True)
|
||||
#undef True
|
||||
#endif
|
||||
#if defined(False)
|
||||
#undef False
|
||||
#endif
|
||||
|
||||
#include <RefCntAutoPtr.hpp>
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
struct WindowHandle;
|
||||
}
|
||||
|
||||
namespace Diligent {
|
||||
struct IRenderDevice;
|
||||
struct IDeviceContext;
|
||||
struct ITexture;
|
||||
struct ITextureView;
|
||||
struct IPipelineState;
|
||||
struct IBuffer;
|
||||
struct ISampler;
|
||||
struct IShaderResourceBinding;
|
||||
struct ISwapChain;
|
||||
struct IEngineFactoryVk;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject;
|
||||
class SamplerObject;
|
||||
class ProgramObject;
|
||||
class RenderbufferObject;
|
||||
class FramebufferObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DiligentBackend {
|
||||
// Minimal real Diligent renderer used to prove the GL 3.2 basic path:
|
||||
// clear an offscreen color target, draw a hardcoded triangle, and read
|
||||
// pixels back. This is the first concrete rendering layer on top of the
|
||||
// Diligent device; it will be expanded into the full MobileGL backend.
|
||||
class DiligentRenderer {
|
||||
public:
|
||||
DiligentRenderer(::Diligent::IRenderDevice* device, ::Diligent::IDeviceContext* context);
|
||||
~DiligentRenderer();
|
||||
|
||||
Bool Initialize(Uint32 width, Uint32 height);
|
||||
void Clear(Float r, Float g, Float b, Float a);
|
||||
void ClearDepth(Float depth);
|
||||
void ClearStencil(Uint32 stencil);
|
||||
void DrawTriangle();
|
||||
void DrawVertices(const Float* vertices, Uint32 vertexCount);
|
||||
// Creates a real Diligent swap chain for a native EGL window surface.
|
||||
Bool CreateSwapChain(::Diligent::IEngineFactoryVk* factory, const WindowHandle& handle,
|
||||
Uint32 width, Uint32 height);
|
||||
Bool ResizeSwapChain(Uint32 width, Uint32 height);
|
||||
void SetSwapInterval(Uint32 interval);
|
||||
// Creates a simple 2D RGBA8 texture from CPU data and makes it available
|
||||
// to state PSOs under the shader variable name "g_Texture".
|
||||
Bool CreateTestTexture(const void* data, Uint32 width, Uint32 height);
|
||||
// Draws using the live MG_State GL context: current program, VAO and
|
||||
// bound buffers. This is the front-end emulation entry point.
|
||||
void DrawFromState(GLenum mode, GLint first, GLsizei count, GLenum type, const void* indices,
|
||||
GLint baseVertex = 0);
|
||||
void ReadPixels(Uint32 x, Uint32 y, Uint32 width, Uint32 height, void* pixels);
|
||||
void BlitFramebuffer(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLbitfield mask, GLenum filter);
|
||||
void BlitNamedFramebuffer(const SharedPtr<MG_State::GLState::FramebufferObject>& readFbo,
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& drawFbo,
|
||||
GLbitfield mask);
|
||||
void CopyReadFramebufferToTexture(MG_State::GLState::ITextureObject& dst);
|
||||
void CopyTextureSubData(MG_State::GLState::ITextureObject& src, MG_State::GLState::ITextureObject& dst);
|
||||
void GenerateMipmap(MG_State::GLState::ITextureObject& texture);
|
||||
Bool ReadTextureImage(MG_State::GLState::ITextureObject& texture, Uint32 level, void* pixels);
|
||||
void ReleaseSwapChain();
|
||||
void Present();
|
||||
|
||||
::Diligent::IRenderDevice* GetDevice() const { return m_pDevice; }
|
||||
::Diligent::IDeviceContext* GetContext() const { return m_pContext; }
|
||||
|
||||
private:
|
||||
struct TextureResource {
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITexture> Texture;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITextureView> SRV;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITextureView> RTV;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITextureView> DSV;
|
||||
Uint64 ContentVersion = 0;
|
||||
Uint16 ParamsVersion = 0;
|
||||
Bool IsDepth = false;
|
||||
};
|
||||
|
||||
struct SamplerResource {
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ISampler> Sampler;
|
||||
Uint16 Version = 0;
|
||||
};
|
||||
|
||||
Bool CreateOffscreenTargets();
|
||||
Bool CreatePipeline();
|
||||
Bool CreateVertexBuffer();
|
||||
Bool CreatePipelineFromState(GLenum mode);
|
||||
Bool UploadVertexDataFromState(GLenum mode, GLint first, GLsizei count, GLenum type, const void* indices,
|
||||
GLint baseVertex = 0);
|
||||
::Diligent::ITextureView* SyncTexture(MG_State::GLState::ITextureObject& texture);
|
||||
::Diligent::ITextureView* SyncTextureForAttachment(MG_State::GLState::ITextureObject& texture, Bool depth);
|
||||
::Diligent::ITextureView* SyncRenderbuffer(MG_State::GLState::RenderbufferObject& renderbuffer);
|
||||
::Diligent::ISampler* SyncSampler(const MG_State::GLState::SamplerObject& sampler);
|
||||
Bool BindShaderResourcesFromState(const MG_State::GLState::ProgramObject& program);
|
||||
Bool UploadUBOFromState(const MG_State::GLState::ProgramObject& program);
|
||||
Bool ResolveCurrentRenderTargets(Vector<::Diligent::ITextureView*>& rtvs,
|
||||
::Diligent::ITextureView*& dsv);
|
||||
|
||||
::Diligent::IRenderDevice* m_pDevice = nullptr;
|
||||
::Diligent::IDeviceContext* m_pContext = nullptr;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITexture> m_pColorTarget;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITextureView> m_pColorRTV;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITexture> m_pDepthTarget;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITextureView> m_pDepthDSV;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ISwapChain> m_pSwapChain;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITexture> m_pTestTexture;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ITextureView> m_pTestSRV;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::ISampler> m_pTestSampler;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::IShaderResourceBinding> m_pStateSRB;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::IPipelineState> m_pPSO;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::IBuffer> m_pVertexBuffer;
|
||||
::Diligent::RefCntAutoPtr<::Diligent::IBuffer> m_pUBO;
|
||||
Uint32 m_uboSize = 0;
|
||||
Uint32 m_uboContentVersion = 0;
|
||||
Uint64 m_uboProgramLifetimeId = 0;
|
||||
UnorderedMap<Uint64, TextureResource> m_textureCache;
|
||||
UnorderedMap<Uint64, SamplerResource> m_samplerCache;
|
||||
UnorderedMap<Uint32, TextureResource> m_renderbufferCache;
|
||||
UnorderedMap<Uint64, ::Diligent::RefCntAutoPtr<::Diligent::IBuffer>> m_namedUboCache;
|
||||
Uint32 m_width = 256;
|
||||
Uint32 m_height = 256;
|
||||
Uint32 m_swapInterval = 0;
|
||||
Uint32 m_lastDrawVertexCount = 0;
|
||||
Uint64 m_lastPSOKey = 0;
|
||||
Bool m_hasCachedPSO = false;
|
||||
Bool m_initialized = false;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DiligentBackend
|
||||
@@ -932,7 +932,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Vector<GLExtension> extensions = {
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
|
||||
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
|
||||
E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage, E_GL_ARB_texture_storage,
|
||||
E_GL_ARB_texture_storage_multisample, E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access,
|
||||
E_GL_ARB_multi_draw_indirect, E_GL_ARB_indirect_parameters, E_GL_ARB_shader_draw_parameters,
|
||||
|
||||
@@ -729,8 +729,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void Ops_ReadbackFromGpu(BufferObject& bufferObject) {
|
||||
auto* resource = ResourceOf(bufferObject);
|
||||
if (!resource || resource->id == 0 || !resource->storageInitialized) return;
|
||||
if (resource->persistentMapped) return; // shadow already IS the GPU storage
|
||||
if (!CanTouchGLNow() || resource->contextGeneration != g_bufferContextGeneration) return;
|
||||
if (resource->persistentMapped) {
|
||||
// Host writes to a persistent map must not race shader writes already queued
|
||||
// on this context. There is no backend copy to read back in this case.
|
||||
if (g_GLESFuncs.glFinish) g_GLESFuncs.glFinish();
|
||||
return;
|
||||
}
|
||||
if (!g_GLESFuncs.glMapBufferRange || !g_GLESFuncs.glUnmapBuffer) return;
|
||||
const SizeT size = std::min<SizeT>(bufferObject.GetSize(), resource->storageSize);
|
||||
if (size == 0) return;
|
||||
@@ -4741,6 +4746,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
MGLOG_D("%s:", src.empty() ? "" : src.c_str());
|
||||
}
|
||||
auto& shaderSpirvs = stateProgramObject->GetGeneratedSpirv();
|
||||
const Bool enableSpirvValidation = stateProgramObject->GetSpirvValidationEnabled();
|
||||
|
||||
// Blocks a transform-feedback capture request names a member of ("StageData" of
|
||||
// "StageData.attrib[0]"). The Adreno ES driver accepts such a request, links, and
|
||||
@@ -4794,7 +4800,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Vector<unsigned int> loweredSpirv;
|
||||
const Vector<unsigned int>* effectiveSpirv = &spirvCode;
|
||||
if (glShaderType == GL_VERTEX_SHADER &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::LowerDrawParametersForEssl(spirvCode, loweredSpirv) &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::LowerDrawParametersForEssl(spirvCode, loweredSpirv, enableSpirvValidation) &&
|
||||
!loweredSpirv.empty()) {
|
||||
effectiveSpirv = &loweredSpirv;
|
||||
}
|
||||
@@ -4804,7 +4810,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Vector<unsigned int> splitArrayInputSpirv;
|
||||
if (glShaderType == GL_VERTEX_SHADER &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::SplitArrayVertexInputsForEssl(
|
||||
*effectiveSpirv, splitArrayInputSpirv) &&
|
||||
*effectiveSpirv, splitArrayInputSpirv, enableSpirvValidation) &&
|
||||
!splitArrayInputSpirv.empty() && splitArrayInputSpirv != *effectiveSpirv) {
|
||||
// Only when the pass ACTUALLY split something. The optimizer hands back a
|
||||
// re-serialised copy either way, and adopting that copy for every vertex
|
||||
@@ -4826,7 +4832,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (!xfbCaptureBlockNames.empty() &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(
|
||||
*effectiveSpirv, xfbCaptureBlockNames, stageFlattenedXfbBlockNames,
|
||||
flattenedXfbSpirv) &&
|
||||
flattenedXfbSpirv, enableSpirvValidation) &&
|
||||
!flattenedXfbSpirv.empty() && !stageFlattenedXfbBlockNames.empty()) {
|
||||
effectiveSpirv = &flattenedXfbSpirv;
|
||||
flattenedXfbBlockNames.insert(stageFlattenedXfbBlockNames.begin(),
|
||||
@@ -4842,7 +4848,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// declare the member highp; nothing else about emission changes.
|
||||
Vector<unsigned int> uboPrecisionSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::StripUboMemberRelaxedPrecisionForEssl(
|
||||
*effectiveSpirv, uboPrecisionSpirv) &&
|
||||
*effectiveSpirv, uboPrecisionSpirv, enableSpirvValidation) &&
|
||||
!uboPrecisionSpirv.empty()) {
|
||||
effectiveSpirv = &uboPrecisionSpirv;
|
||||
}
|
||||
@@ -4857,7 +4863,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Vector<unsigned int> noperspectiveSpirv;
|
||||
if (!g_GLESCapabilities.SupportsNoperspectiveInterpolation &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::EmulateNoPerspectiveForEssl(
|
||||
*effectiveSpirv, noperspectiveSpirv) &&
|
||||
*effectiveSpirv, noperspectiveSpirv, enableSpirvValidation) &&
|
||||
!noperspectiveSpirv.empty()) {
|
||||
effectiveSpirv = &noperspectiveSpirv;
|
||||
}
|
||||
@@ -4867,7 +4873,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// divides the coordinate of every normalized-coordinate lookup by the texture
|
||||
// size, which is the whole of the difference between the two.
|
||||
Vector<unsigned int> rectLoweredSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(*effectiveSpirv, rectLoweredSpirv) &&
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(*effectiveSpirv, rectLoweredSpirv, enableSpirvValidation) &&
|
||||
!rectLoweredSpirv.empty()) {
|
||||
effectiveSpirv = &rectLoweredSpirv;
|
||||
}
|
||||
@@ -4881,7 +4887,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// coordinate to (u, 0, layer) - before SPIRV-Cross can apply its own.
|
||||
Vector<unsigned int> arrayImageSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::Lower1DArrayImagesForEssl(*effectiveSpirv,
|
||||
arrayImageSpirv) &&
|
||||
arrayImageSpirv, enableSpirvValidation) &&
|
||||
!arrayImageSpirv.empty()) {
|
||||
effectiveSpirv = &arrayImageSpirv;
|
||||
}
|
||||
@@ -4900,7 +4906,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (!imageFormatBake.glFormatByUniformName.empty() &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::DeclaresFormatlessStorageImage(*effectiveSpirv) &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::BakeImageFormatsForEssl(
|
||||
*effectiveSpirv, imageFormatBake.glFormatByUniformName, imageFormatSpirv) &&
|
||||
*effectiveSpirv, imageFormatBake.glFormatByUniformName, imageFormatSpirv,
|
||||
enableSpirvValidation) &&
|
||||
!imageFormatSpirv.empty()) {
|
||||
effectiveSpirv = &imageFormatSpirv;
|
||||
}
|
||||
@@ -4916,7 +4923,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Vector<unsigned int> outputIndexSpirv;
|
||||
if (glShaderType == GL_FRAGMENT_SHADER &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(
|
||||
*effectiveSpirv, outputIndexSpirv) &&
|
||||
*effectiveSpirv, outputIndexSpirv, enableSpirvValidation) &&
|
||||
!outputIndexSpirv.empty()) {
|
||||
effectiveSpirv = &outputIndexSpirv;
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
#include "MG_Backend/BackendObject.h"
|
||||
#include "DirectVulkan.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include "MG_State/GLState/TextureState/TextureState.h"
|
||||
#include "MG_Util/Classifiers/TextureEnumClassifier.h"
|
||||
#include "MG_Util/Converters/MGToGL/TextureEnumConverter.h"
|
||||
@@ -383,6 +384,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
if (MG_State::pGLContext) {
|
||||
MG_State::pGLContext->InvalidateCompileEnv();
|
||||
}
|
||||
PopulateFormatCapabilities(physicalDevice.handle, vkGetPhysicalDeviceFormatProperties, m_vulkanCaps,
|
||||
MutableFormatCapabilities());
|
||||
PrintFormatCapabilities(GetFormatCapabilities());
|
||||
@@ -511,7 +515,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<GLExtension> extensions = {
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_draw_indirect,
|
||||
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_draw_indirect,
|
||||
E_GL_ARB_multi_draw_indirect,
|
||||
E_GL_ARB_indirect_parameters, E_GL_EXT_framebuffer_object, E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage,
|
||||
E_GL_ARB_texture_storage, E_GL_ARB_texture_storage_multisample, E_GL_ARB_texture_multisample,
|
||||
@@ -687,6 +691,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_vulkanCaps = capabilities;
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
if (MG_State::pGLContext) {
|
||||
MG_State::pGLContext->InvalidateCompileEnv();
|
||||
}
|
||||
MutableFormatCapabilities().Clear();
|
||||
}
|
||||
|
||||
|
||||
@@ -2973,8 +2973,73 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bindings.push_back(layoutBinding);
|
||||
}
|
||||
|
||||
// UPDATE_AFTER_BIND is strictly an optional per-layout acceleration. The GL
|
||||
// descriptor model still resolves every sampler uniform element independently
|
||||
// (including its texture-unit sampler-object override); selecting this path
|
||||
// changes neither that resolution nor the set versioning in UniformManager.
|
||||
// A conservative count keeps a layout on ordinary descriptors whenever any
|
||||
// relevant update-after-bind limit is not large enough, rather than asking a
|
||||
// driver to reject it during vkCreateDescriptorSetLayout.
|
||||
Uint32 updateAfterBindSamplers = 0;
|
||||
Uint32 updateAfterBindUniformBuffers = 0;
|
||||
Uint32 updateAfterBindStorageBuffers = 0;
|
||||
Uint32 updateAfterBindSampledImages = 0;
|
||||
Uint32 updateAfterBindStorageImages = 0;
|
||||
for (Uint32 binding = 0; binding < m_maxBindings; ++binding) {
|
||||
const Uint32 count = entry.bindingDescriptorCounts[binding];
|
||||
switch (entry.bindingKinds[binding]) {
|
||||
case DescriptorBindingKind::UniformBufferDynamic:
|
||||
updateAfterBindUniformBuffers += count;
|
||||
break;
|
||||
case DescriptorBindingKind::CombinedImageSampler:
|
||||
updateAfterBindSamplers += count;
|
||||
updateAfterBindSampledImages += count;
|
||||
break;
|
||||
case DescriptorBindingKind::UniformTexelBuffer:
|
||||
updateAfterBindSampledImages += count;
|
||||
break;
|
||||
case DescriptorBindingKind::StorageBuffer:
|
||||
case DescriptorBindingKind::StorageTexelBuffer:
|
||||
updateAfterBindStorageBuffers += count;
|
||||
break;
|
||||
case DescriptorBindingKind::StorageImage:
|
||||
updateAfterBindStorageImages += count;
|
||||
break;
|
||||
case DescriptorBindingKind::None:
|
||||
break;
|
||||
}
|
||||
}
|
||||
const Uint32 updateAfterBindResources = updateAfterBindUniformBuffers + updateAfterBindStorageBuffers +
|
||||
updateAfterBindSampledImages + updateAfterBindStorageImages;
|
||||
const auto& uab = m_updateAfterBindLimits;
|
||||
entry.usesUpdateAfterBind =
|
||||
uab.enabled && updateAfterBindSamplers <= uab.maxPerStageSamplers &&
|
||||
updateAfterBindUniformBuffers <= uab.maxPerStageUniformBuffers &&
|
||||
updateAfterBindStorageBuffers <= uab.maxPerStageStorageBuffers &&
|
||||
updateAfterBindSampledImages <= uab.maxPerStageSampledImages &&
|
||||
updateAfterBindStorageImages <= uab.maxPerStageStorageImages &&
|
||||
updateAfterBindResources <= uab.maxPerStageResources &&
|
||||
updateAfterBindSamplers <= uab.maxSetSamplers &&
|
||||
updateAfterBindUniformBuffers <= uab.maxSetUniformBuffers &&
|
||||
updateAfterBindUniformBuffers <= uab.maxSetUniformBuffersDynamic &&
|
||||
updateAfterBindStorageBuffers <= uab.maxSetStorageBuffers &&
|
||||
updateAfterBindStorageBuffers <= uab.maxSetStorageBuffersDynamic &&
|
||||
updateAfterBindSampledImages <= uab.maxSetSampledImages &&
|
||||
updateAfterBindStorageImages <= uab.maxSetStorageImages;
|
||||
|
||||
Vector<VkDescriptorBindingFlags> bindingFlags;
|
||||
VkDescriptorSetLayoutBindingFlagsCreateInfo bindingFlagsInfo{};
|
||||
if (entry.usesUpdateAfterBind) {
|
||||
bindingFlags.assign(bindings.size(), VK_DESCRIPTOR_BINDING_UPDATE_AFTER_BIND_BIT);
|
||||
bindingFlagsInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_BINDING_FLAGS_CREATE_INFO;
|
||||
bindingFlagsInfo.bindingCount = static_cast<Uint32>(bindingFlags.size());
|
||||
bindingFlagsInfo.pBindingFlags = bindingFlags.data();
|
||||
}
|
||||
|
||||
VkDescriptorSetLayoutCreateInfo setLayoutInfo{};
|
||||
setLayoutInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO;
|
||||
setLayoutInfo.flags = entry.usesUpdateAfterBind ? VK_DESCRIPTOR_SET_LAYOUT_CREATE_UPDATE_AFTER_BIND_POOL_BIT : 0;
|
||||
setLayoutInfo.pNext = entry.usesUpdateAfterBind ? &bindingFlagsInfo : nullptr;
|
||||
setLayoutInfo.bindingCount = static_cast<Uint32>(bindings.size());
|
||||
setLayoutInfo.pBindings = bindings.data();
|
||||
VK_VERIFY(vkCreateDescriptorSetLayout(m_device, &setLayoutInfo, nullptr, &entry.descriptorSetLayout),
|
||||
@@ -3054,6 +3119,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto& shaders = program.GetAttachedShaders();
|
||||
auto& spirv = program.GetGeneratedSpirv();
|
||||
Vector<Vector<Uint>> moduleSpirvs(spirv.size());
|
||||
const Bool enableSpirvValidation = program.GetSpirvValidationEnabled();
|
||||
if (enableSpirvValidation) {
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::PrepareSpirvValidation();
|
||||
}
|
||||
|
||||
const ShaderStage fixupStage = PickClipFixupStage(shaders);
|
||||
|
||||
@@ -3099,7 +3168,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// stored as - which addresses [0,1] where the application addressed texels.
|
||||
{
|
||||
Vector<Uint> rectLoweredSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(moduleSpirvs[i], rectLoweredSpirv) &&
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::LowerRectImages(moduleSpirvs[i], rectLoweredSpirv, enableSpirvValidation) &&
|
||||
!rectLoweredSpirv.empty()) {
|
||||
moduleSpirvs[i] = Move(rectLoweredSpirv);
|
||||
}
|
||||
@@ -3112,7 +3181,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
{
|
||||
Vector<Uint> invariantSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::DecoratePositionInvariantForVulkan(
|
||||
moduleSpirvs[i], invariantSpirv)) {
|
||||
moduleSpirvs[i], invariantSpirv, enableSpirvValidation)) {
|
||||
moduleSpirvs[i] = std::move(invariantSpirv);
|
||||
} else {
|
||||
// The pass round-trips through SPIRV-Tools IR, so an unparseable module
|
||||
@@ -3136,7 +3205,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_shaderDrawParametersEnabled) {
|
||||
Vector<Uint> rebasedSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::RebaseInstanceIndexForVulkan(moduleSpirvs[i],
|
||||
rebasedSpirv)) {
|
||||
rebasedSpirv, enableSpirvValidation)) {
|
||||
moduleSpirvs[i] = std::move(rebasedSpirv);
|
||||
} else {
|
||||
MGLOG_E("ProgramFactory: failed to rebase gl_InstanceID for program %u; "
|
||||
@@ -3154,7 +3223,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
(flags & CompileOptionBit::ZeroBaseVertex)) {
|
||||
Vector<Uint> zeroedSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::ZeroBaseVertexForVulkan(moduleSpirvs[i],
|
||||
zeroedSpirv)) {
|
||||
zeroedSpirv, enableSpirvValidation)) {
|
||||
moduleSpirvs[i] = std::move(zeroedSpirv);
|
||||
} else {
|
||||
// Failing open keeps the native builtin, which is the pre-fix behavior:
|
||||
@@ -3177,7 +3246,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (shaders[i] && shaders[i]->GetShaderStage() == ShaderStage::Vertex) {
|
||||
Vector<Uint> packedSpirv;
|
||||
const Bool packOk = MG_Util::ShaderTranspiler::ShaderCompiler::PackDoubleVertexInputsForVulkan(
|
||||
moduleSpirvs[i], packedSpirv);
|
||||
moduleSpirvs[i], packedSpirv, enableSpirvValidation);
|
||||
MOBILEGL_ASSERT(packOk,
|
||||
"ProgramFactory: 64-bit vertex input packing failed for program %u; the "
|
||||
"vertex-input format and the shader input type now disagree",
|
||||
@@ -3201,7 +3270,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (m_unformattedFloatStorageImagesEnabled) {
|
||||
Vector<Uint> unformattedSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(
|
||||
moduleSpirvs[i], unformattedSpirv)) {
|
||||
moduleSpirvs[i], unformattedSpirv, enableSpirvValidation)) {
|
||||
moduleSpirvs[i] = std::move(unformattedSpirv);
|
||||
} else {
|
||||
MGLOG_E("ProgramFactory: failed to make float storage images unformatted for program %u",
|
||||
@@ -3222,7 +3291,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
#else
|
||||
// Final module the driver receives; also checked in the INFO-level CI/test
|
||||
// lanes, where the DEBUG gate above is compiled out.
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::SpirvValidationEnabled()) {
|
||||
if (enableSpirvValidation) {
|
||||
ValidateTransformedSpirv(moduleSpv, shaders[i]->GetShaderStage(), program.GetExternalIndex());
|
||||
}
|
||||
#endif
|
||||
@@ -3299,14 +3368,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VkDescriptorSetLayout descriptorSetLayout = it->second.descriptorSetLayout;
|
||||
MGLOG_D("ProgramFactory::OnFrameBoundary: evicting idle program entry hash=0x%llx",
|
||||
static_cast<unsigned long long>(hash));
|
||||
// erase runs ~VkProgramObject (modules/layouts destroyed); notify after
|
||||
// so an observer never observes a half-destroyed entry through a lookup.
|
||||
// Observers only need the handle values to purge their keyed caches.
|
||||
++m_cacheStructureEpoch; // erase moves/kills entries: memoised pointers die
|
||||
it = m_cache.erase(it);
|
||||
// The observer destroys dependent pipelines and frees descriptor sets while
|
||||
// this entry still owns its layout. Vulkan requires every descriptor set to be
|
||||
// freed before its VkDescriptorSetLayout is destroyed.
|
||||
if (m_evictionObserver != nullptr) {
|
||||
m_evictionObserver->OnProgramEvicted(hash, descriptorSetLayout);
|
||||
}
|
||||
++m_cacheStructureEpoch; // erase moves/kills entries: memoised pointers die
|
||||
it = m_cache.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
@@ -3440,7 +3509,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG
|
||||
ValidateTransformedSpirv(spirv, ShaderStage::TessControl, 0);
|
||||
#else
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::SpirvValidationEnabled()) {
|
||||
if (m_enableSpirvValidation) {
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::PrepareSpirvValidation();
|
||||
ValidateTransformedSpirv(spirv, ShaderStage::TessControl, 0);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -76,6 +76,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
|
||||
struct UpdateAfterBindLimits {
|
||||
Bool enabled = false;
|
||||
Uint32 maxPerStageSamplers = 0;
|
||||
Uint32 maxPerStageUniformBuffers = 0;
|
||||
Uint32 maxPerStageStorageBuffers = 0;
|
||||
Uint32 maxPerStageSampledImages = 0;
|
||||
Uint32 maxPerStageStorageImages = 0;
|
||||
Uint32 maxPerStageResources = 0;
|
||||
Uint32 maxSetSamplers = 0;
|
||||
Uint32 maxSetUniformBuffers = 0;
|
||||
Uint32 maxSetUniformBuffersDynamic = 0;
|
||||
Uint32 maxSetStorageBuffers = 0;
|
||||
Uint32 maxSetStorageBuffersDynamic = 0;
|
||||
Uint32 maxSetSampledImages = 0;
|
||||
Uint32 maxSetStorageImages = 0;
|
||||
};
|
||||
|
||||
struct VkProgramObject {
|
||||
static constexpr Uint32 kMaxVertexInputLocations = 32;
|
||||
|
||||
@@ -88,6 +105,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
// Layout data (previously in separate VkProgramLayout)
|
||||
VkDescriptorSetLayout descriptorSetLayout = VK_NULL_HANDLE;
|
||||
// True only when this layout passed every descriptor-indexing feature and
|
||||
// update-after-bind limit gate at reflection time. It controls both the
|
||||
// layout/binding flags and the pool class used by UniformManager.
|
||||
Bool usesUpdateAfterBind = false;
|
||||
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
|
||||
Vector<DescriptorBindingKind> bindingKinds;
|
||||
// The bindings this program actually declares, ascending. bindingKinds is sized to the
|
||||
@@ -196,6 +217,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// a pipeline failure would be reported against the wrong SPIR-V.
|
||||
stageSpirvDigests = std::move(other.stageSpirvDigests);
|
||||
descriptorSetLayout = other.descriptorSetLayout;
|
||||
usesUpdateAfterBind = other.usesUpdateAfterBind;
|
||||
pipelineLayout = other.pipelineLayout;
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
activeBindings = std::move(other.activeBindings);
|
||||
@@ -230,6 +252,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.usesUpdateAfterBind = false;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.declinedDescriptors = false;
|
||||
@@ -256,6 +279,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
modules = std::move(other.modules);
|
||||
stageSpirvDigests = std::move(other.stageSpirvDigests); // travels with `modules` - see the move ctor
|
||||
descriptorSetLayout = other.descriptorSetLayout;
|
||||
usesUpdateAfterBind = other.usesUpdateAfterBind;
|
||||
pipelineLayout = other.pipelineLayout;
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
activeBindings = std::move(other.activeBindings);
|
||||
@@ -290,6 +314,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.usesUpdateAfterBind = false;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.declinedDescriptors = false;
|
||||
@@ -347,12 +372,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
virtual void OnProgramEvicted(HashType programHash, VkDescriptorSetLayout descriptorSetLayout) = 0;
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings = 16,
|
||||
Bool shaderDrawParametersEnabled = false,
|
||||
Bool unformattedFloatStorageImagesEnabled = false)
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings,
|
||||
Bool shaderDrawParametersEnabled,
|
||||
Bool unformattedFloatStorageImagesEnabled,
|
||||
Bool enableSpirvValidation,
|
||||
UpdateAfterBindLimits updateAfterBindLimits)
|
||||
: m_device(device), m_maxBindings(maxBindings), m_config(config),
|
||||
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled),
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled) {
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled),
|
||||
m_enableSpirvValidation(enableSpirvValidation),
|
||||
m_updateAfterBindLimits(updateAfterBindLimits) {
|
||||
VkProgramObject::s_device = device;
|
||||
}
|
||||
// Destroys the pass-through tessellation control modules. Runs while the device is
|
||||
@@ -475,6 +504,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// True only when the logical device enabled both
|
||||
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// Startup snapshot used only by internally synthesized shader modules, which do not
|
||||
// originate from a ProgramLinkTask.
|
||||
Bool m_enableSpirvValidation = false;
|
||||
// Device feature and limit gate resolved before vkCreateDevice. Keeping it in
|
||||
// the factory lets each reflected layout choose ordinary descriptors when its
|
||||
// own counts would exceed the update-after-bind budget.
|
||||
UpdateAfterBindLimits m_updateAfterBindLimits{};
|
||||
// See SetDefaultFramebufferHeight. 0 means "not known yet"; the FragCoordYFlip bit is
|
||||
// never set before the swapchain exists, so no variant can be compiled against it.
|
||||
Uint32 m_defaultFramebufferHeight = 0;
|
||||
|
||||
@@ -156,13 +156,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frame.descriptorPools.clear();
|
||||
|
||||
VkDescriptorPool initialPool = VK_NULL_HANDLE;
|
||||
if (!CreateDescriptorPool(m_setsPerFrame, initialPool)) {
|
||||
if (!CreateDescriptorPool(m_setsPerFrame, false, initialPool)) {
|
||||
MGLOG_E_ONCE("UniformDescriptorBinder::Initialize failed: cannot create frame descriptor pool %u",
|
||||
frameIndex);
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
frame.descriptorPools.push_back({initialPool, m_setsPerFrame, 0});
|
||||
frame.descriptorPools.push_back({initialPool, m_setsPerFrame, 0, false});
|
||||
MGLOG_D("UniformDescriptorBinder: frame %u descriptor pool created (maxSets=%u)", frameIndex,
|
||||
m_setsPerFrame);
|
||||
}
|
||||
@@ -1390,7 +1390,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const {
|
||||
Bool UniformManager::CreateDescriptorPool(Uint32 maxSets, Bool updateAfterBind, VkDescriptorPool& outPool) const {
|
||||
outPool = VK_NULL_HANDLE;
|
||||
if (m_device == VK_NULL_HANDLE || maxSets == 0 || m_maxBindings == 0) {
|
||||
return false;
|
||||
@@ -1433,7 +1433,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// (OnDescriptorSetLayoutDestroyed) so program churn recycles pool capacity.
|
||||
// The cost is on set allocation only, which happens when a layout's per-frame
|
||||
// cache grows - never on the per-draw reuse path.
|
||||
poolInfo.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT;
|
||||
poolInfo.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT |
|
||||
(updateAfterBind ? VK_DESCRIPTOR_POOL_CREATE_UPDATE_AFTER_BIND_BIT : 0);
|
||||
poolInfo.maxSets = maxSets;
|
||||
poolInfo.poolSizeCount = static_cast<Uint32>(std::size(poolSizes));
|
||||
poolInfo.pPoolSizes = poolSizes;
|
||||
@@ -1447,24 +1448,28 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex) {
|
||||
Bool UniformManager::GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex, Bool updateAfterBind) {
|
||||
if (frame.descriptorPools.empty()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto& currentBucket = frame.descriptorPools[frame.activeDescriptorPoolIndex];
|
||||
const Uint32 currentMaxSets = std::max<Uint32>(1, currentBucket.maxSets);
|
||||
const auto matchingBucket = std::find_if(
|
||||
frame.descriptorPools.begin(), frame.descriptorPools.end(),
|
||||
[updateAfterBind](const DescriptorPoolBucket& candidate) { return candidate.updateAfterBind == updateAfterBind; });
|
||||
const Uint32 currentMaxSets = matchingBucket != frame.descriptorPools.end()
|
||||
? std::max<Uint32>(1, matchingBucket->maxSets)
|
||||
: m_setsPerFrame;
|
||||
const Uint32 grownMaxSets = currentMaxSets <= (std::numeric_limits<Uint32>::max() / 2) ? (currentMaxSets * 2)
|
||||
: currentMaxSets;
|
||||
|
||||
VkDescriptorPool grownPool = VK_NULL_HANDLE;
|
||||
if (!CreateDescriptorPool(grownMaxSets, grownPool)) {
|
||||
if (!CreateDescriptorPool(grownMaxSets, updateAfterBind, grownPool)) {
|
||||
MGLOG_E_ONCE("UniformDescriptorBinder::GrowFrameDescriptorPool failed: cannot create grown pool (%u -> %u sets)",
|
||||
currentMaxSets, grownMaxSets);
|
||||
return false;
|
||||
}
|
||||
|
||||
frame.descriptorPools.push_back({grownPool, grownMaxSets, 0});
|
||||
frame.descriptorPools.push_back({grownPool, grownMaxSets, 0, updateAfterBind});
|
||||
frame.activeDescriptorPoolIndex = static_cast<Uint32>(frame.descriptorPools.size() - 1);
|
||||
MGLOG_D(
|
||||
"UniformDescriptorBinder: frame %u descriptor pool exhausted, grew pool (%u -> %u sets), poolCount=%zu",
|
||||
@@ -1474,14 +1479,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkResult UniformManager::AllocateDescriptorSetsFromActivePool(Uint32 frameIndex, const ProgramFactory::VkProgramObject& programObj, VkDescriptorSet& outDescriptorSet) {
|
||||
auto& frame = m_frames[frameIndex];
|
||||
if (frame.activeDescriptorPoolIndex >= frame.descriptorPools.size()) {
|
||||
frame.activeDescriptorPoolIndex = 0;
|
||||
}
|
||||
if (frame.descriptorPools[frame.activeDescriptorPoolIndex].allocatedSets >=
|
||||
frame.descriptorPools[frame.activeDescriptorPoolIndex].maxSets) {
|
||||
const Bool updateAfterBind = programObj.usesUpdateAfterBind;
|
||||
if (frame.activeDescriptorPoolIndex >= frame.descriptorPools.size() ||
|
||||
frame.descriptorPools[frame.activeDescriptorPoolIndex].updateAfterBind != updateAfterBind ||
|
||||
frame.descriptorPools[frame.activeDescriptorPoolIndex].allocatedSets >=
|
||||
frame.descriptorPools[frame.activeDescriptorPoolIndex].maxSets) {
|
||||
const auto availableBucket = std::find_if(
|
||||
frame.descriptorPools.begin(), frame.descriptorPools.end(),
|
||||
[](const DescriptorPoolBucket& candidate) { return candidate.allocatedSets < candidate.maxSets; });
|
||||
[updateAfterBind](const DescriptorPoolBucket& candidate) {
|
||||
return candidate.updateAfterBind == updateAfterBind && candidate.allocatedSets < candidate.maxSets;
|
||||
});
|
||||
if (availableBucket == frame.descriptorPools.end()) {
|
||||
outDescriptorSet = VK_NULL_HANDLE;
|
||||
return VK_ERROR_OUT_OF_POOL_MEMORY;
|
||||
@@ -1517,7 +1524,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
} else {
|
||||
VkResult allocResult = AllocateDescriptorSetsFromActivePool(frameIndex, programObj, outDescriptorSet);
|
||||
if (allocResult == VK_ERROR_OUT_OF_POOL_MEMORY || allocResult == VK_ERROR_FRAGMENTED_POOL) {
|
||||
if (!GrowFrameDescriptorPool(frame, frameIndex)) {
|
||||
if (!GrowFrameDescriptorPool(frame, frameIndex, programObj.usesUpdateAfterBind)) {
|
||||
MGLOG_E_ONCE("UniformDescriptorBinder::AcquireDescriptorSet failed: descriptor pool growth failed");
|
||||
return allocResult;
|
||||
}
|
||||
|
||||
@@ -114,6 +114,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDescriptorPool handle = VK_NULL_HANDLE;
|
||||
Uint32 maxSets = 0;
|
||||
Uint32 allocatedSets = 0;
|
||||
Bool updateAfterBind = false;
|
||||
};
|
||||
|
||||
// A cached descriptor set together with the pool it was allocated from, so a
|
||||
@@ -223,8 +224,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void BindDescriptorSetDeduped(VkCommandBuffer commandBuffer, VkPipelineBindPoint bindPoint,
|
||||
VkPipelineLayout pipelineLayout, VkDescriptorSet descriptorSet,
|
||||
const Vector<Uint32>& dynamicOffsets);
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, Bool updateAfterBind, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex, Bool updateAfterBind);
|
||||
VkResult AllocateDescriptorSetsFromActivePool(
|
||||
Uint32 frameIndex, const ProgramFactory::VkProgramObject& programObj, VkDescriptorSet& outDescriptorSet);
|
||||
VkResult AcquireDescriptorSet(Uint32 frameIndex,
|
||||
|
||||
@@ -1480,7 +1480,8 @@ void main() {
|
||||
const char* label = nullptr;
|
||||
};
|
||||
|
||||
static Uint32 ComputeMaxProgramBindings(const VkPhysicalDeviceProperties& properties) {
|
||||
static Uint32 ComputeMaxProgramBindings(const VkPhysicalDeviceProperties& properties,
|
||||
const ProgramFactory::UpdateAfterBindLimits& updateAfterBindLimits) {
|
||||
const auto& limits = properties.limits;
|
||||
static constexpr Uint32 kMinProgramBindings = 16;
|
||||
static constexpr Uint32 kMaxProgramBindingsCap = 256;
|
||||
@@ -1495,6 +1496,21 @@ void main() {
|
||||
maxBindings = std::min(maxBindings, maxCombinedImageSamplers);
|
||||
maxBindings = std::min(maxBindings, maxSampledImages + maxDynamicUniformBuffers);
|
||||
|
||||
if (updateAfterBindLimits.enabled) {
|
||||
const Uint32 updateAfterBindSamplers = std::min(updateAfterBindLimits.maxPerStageSamplers,
|
||||
updateAfterBindLimits.maxSetSamplers);
|
||||
const Uint32 updateAfterBindSampledImages = std::min(updateAfterBindLimits.maxPerStageSampledImages,
|
||||
updateAfterBindLimits.maxSetSampledImages);
|
||||
const Uint32 updateAfterBindDynamicUniformBuffers =
|
||||
std::min(updateAfterBindLimits.maxPerStageUniformBuffers,
|
||||
updateAfterBindLimits.maxSetUniformBuffersDynamic);
|
||||
Uint32 updateAfterBindBindings = updateAfterBindLimits.maxPerStageResources;
|
||||
updateAfterBindBindings = std::min(updateAfterBindBindings, updateAfterBindSamplers);
|
||||
updateAfterBindBindings =
|
||||
std::min(updateAfterBindBindings, updateAfterBindSampledImages + updateAfterBindDynamicUniformBuffers);
|
||||
maxBindings = std::max(maxBindings, updateAfterBindBindings);
|
||||
}
|
||||
|
||||
maxBindings = std::max(kMinProgramBindings, maxBindings);
|
||||
maxBindings = std::min(kMaxProgramBindingsCap, maxBindings);
|
||||
return maxBindings;
|
||||
@@ -3010,7 +3026,7 @@ void main() {
|
||||
succeeded = m_renderPassManager->Initialize();
|
||||
MOBILEGL_ASSERT(succeeded, "VkRenderPassManager initialization failed.");
|
||||
|
||||
const Uint32 maxProgramBindings = ComputeMaxProgramBindings(m_physicalDevice.properties);
|
||||
const Uint32 maxProgramBindings = ComputeMaxProgramBindings(m_physicalDevice.properties, m_updateAfterBindLimits);
|
||||
MGLOG_I("DirectVulkan: using %u program descriptor bindings", maxProgramBindings);
|
||||
if (IsPowerVRDevice(m_physicalDevice.properties)) {
|
||||
m_config.DisablePipelineCache = true;
|
||||
@@ -3044,7 +3060,9 @@ void main() {
|
||||
}
|
||||
m_programFactory = MakeUnique<ProgramFactory>(m_device, m_config, maxProgramBindings,
|
||||
m_shaderDrawParametersFeatureEnabled,
|
||||
m_unformattedFloatStorageImagesEnabled);
|
||||
m_unformattedFloatStorageImagesEnabled,
|
||||
MG_Config::Features.EnableSpirvValidation,
|
||||
m_updateAfterBindLimits);
|
||||
MOBILEGL_ASSERT(m_programFactory != nullptr, "ProgramFactory creation failed.");
|
||||
// The swapchain already exists at this point (Initialize creates it first), so seed the
|
||||
// height the factory could not be told about from CreateSwapchain.
|
||||
@@ -11899,6 +11917,17 @@ void main() {
|
||||
}
|
||||
|
||||
void VulkanRenderer::CreateInstance() {
|
||||
#if defined(VK_USE_PLATFORM_METAL_EXT)
|
||||
// MoltenVK snapshots its configuration when the loader first discovers the ICD. Set
|
||||
// this before instance-extension enumeration, while preserving an explicit user value.
|
||||
if (std::getenv("MVK_CONFIG_USE_METAL_ARGUMENT_BUFFERS") == nullptr) {
|
||||
if (::setenv("MVK_CONFIG_USE_METAL_ARGUMENT_BUFFERS", "1", 0) == 0) {
|
||||
MGLOG_I("MoltenVK: enabling Metal argument buffers");
|
||||
} else {
|
||||
MGLOG_W("MoltenVK: could not enable Metal argument buffers before ICD discovery");
|
||||
}
|
||||
}
|
||||
#endif
|
||||
m_extensions = EnumerateInstanceExtensions();
|
||||
MGLOG_I("Got %d Vulkan instance extensions: ", m_extensions.size());
|
||||
for (auto& extension : m_extensions) {
|
||||
@@ -12040,17 +12069,19 @@ void main() {
|
||||
|
||||
auto debugMessengerCreateInfo = PopulateDebugMessengerCreateInfo();
|
||||
// Layers
|
||||
const void* instanceCreatePNext = nullptr;
|
||||
if (m_validationLayersEnabled) {
|
||||
MGLOG_I("Enabling validation layer...");
|
||||
instanceInfo.enabledLayerCount = static_cast<uint32_t>(std::size(s_validationLayerNames));
|
||||
instanceInfo.ppEnabledLayerNames = s_validationLayerNames;
|
||||
// Chaining the messenger create-info is only legal with the extension on.
|
||||
instanceInfo.pNext = debugUtilsAvailable ? &debugMessengerCreateInfo : nullptr;
|
||||
instanceCreatePNext = debugUtilsAvailable ? &debugMessengerCreateInfo : nullptr;
|
||||
} else {
|
||||
instanceInfo.enabledLayerCount = 0;
|
||||
instanceInfo.pNext = nullptr;
|
||||
}
|
||||
|
||||
instanceInfo.pNext = instanceCreatePNext;
|
||||
|
||||
VK_VERIFY(vkCreateInstance(&instanceInfo, nullptr, &m_instance), "vkCreateInstance failed");
|
||||
|
||||
if (debugUtilsAvailable) {
|
||||
@@ -12476,6 +12507,75 @@ void main() {
|
||||
vkGetInstanceProcAddr(m_instance, "vkGetPhysicalDeviceFeatures2KHR"));
|
||||
}
|
||||
|
||||
m_updateAfterBindLimits = {};
|
||||
VkPhysicalDeviceDescriptorIndexingFeatures descriptorIndexingFeatures{};
|
||||
descriptorIndexingFeatures.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DESCRIPTOR_INDEXING_FEATURES;
|
||||
VkPhysicalDeviceDescriptorIndexingProperties descriptorIndexingProperties{};
|
||||
descriptorIndexingProperties.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DESCRIPTOR_INDEXING_PROPERTIES;
|
||||
const Bool descriptorIndexingCore = m_physicalDevice.properties.apiVersion >= VK_API_VERSION_1_2;
|
||||
const Bool descriptorIndexingExtension =
|
||||
IsExtensionSupported(availableExtensions, VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME);
|
||||
auto getPhysicalDeviceProperties2 = reinterpret_cast<PFN_vkGetPhysicalDeviceProperties2>(
|
||||
vkGetInstanceProcAddr(m_instance, "vkGetPhysicalDeviceProperties2"));
|
||||
if (getPhysicalDeviceProperties2 == nullptr) {
|
||||
getPhysicalDeviceProperties2 = reinterpret_cast<PFN_vkGetPhysicalDeviceProperties2>(
|
||||
vkGetInstanceProcAddr(m_instance, "vkGetPhysicalDeviceProperties2KHR"));
|
||||
}
|
||||
if ((descriptorIndexingCore || descriptorIndexingExtension) && getPhysicalDeviceFeatures2 != nullptr &&
|
||||
getPhysicalDeviceProperties2 != nullptr) {
|
||||
VkPhysicalDeviceFeatures2 featureQuery{};
|
||||
featureQuery.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2;
|
||||
featureQuery.pNext = &descriptorIndexingFeatures;
|
||||
getPhysicalDeviceFeatures2(m_physicalDevice.handle, &featureQuery);
|
||||
VkPhysicalDeviceProperties2 propertyQuery{};
|
||||
propertyQuery.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2;
|
||||
propertyQuery.pNext = &descriptorIndexingProperties;
|
||||
getPhysicalDeviceProperties2(m_physicalDevice.handle, &propertyQuery);
|
||||
|
||||
// This renderer emits every descriptor category listed below, including
|
||||
// dynamic UBOs and combined image samplers. Do not enable a partial
|
||||
// descriptor-indexing contract: it would make a later reflected program
|
||||
// fail in the driver instead of choosing its ordinary descriptor layout.
|
||||
const Bool allUpdateAfterBindFeatures =
|
||||
descriptorIndexingFeatures.descriptorBindingUniformBufferUpdateAfterBind == VK_TRUE &&
|
||||
descriptorIndexingFeatures.descriptorBindingSampledImageUpdateAfterBind == VK_TRUE &&
|
||||
descriptorIndexingFeatures.descriptorBindingStorageImageUpdateAfterBind == VK_TRUE &&
|
||||
descriptorIndexingFeatures.descriptorBindingStorageBufferUpdateAfterBind == VK_TRUE &&
|
||||
descriptorIndexingFeatures.descriptorBindingUniformTexelBufferUpdateAfterBind == VK_TRUE &&
|
||||
descriptorIndexingFeatures.descriptorBindingStorageTexelBufferUpdateAfterBind == VK_TRUE &&
|
||||
(!deviceFeatures.robustBufferAccess || descriptorIndexingProperties.robustBufferAccessUpdateAfterBind);
|
||||
if (allUpdateAfterBindFeatures) {
|
||||
if (!descriptorIndexingCore && !IsExtensionAlreadyEnabled(
|
||||
enabledDeviceExtensions,
|
||||
VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME)) {
|
||||
enabledDeviceExtensions.push_back(VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME);
|
||||
}
|
||||
descriptorIndexingFeatures.pNext = const_cast<void*>(deviceCreateInfo.pNext);
|
||||
deviceCreateInfo.pNext = &descriptorIndexingFeatures;
|
||||
m_updateAfterBindLimits = {
|
||||
true,
|
||||
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindSamplers,
|
||||
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindUniformBuffers,
|
||||
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindStorageBuffers,
|
||||
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindSampledImages,
|
||||
descriptorIndexingProperties.maxPerStageDescriptorUpdateAfterBindStorageImages,
|
||||
descriptorIndexingProperties.maxPerStageUpdateAfterBindResources,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindSamplers,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindUniformBuffers,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindUniformBuffersDynamic,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindStorageBuffers,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindStorageBuffersDynamic,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindSampledImages,
|
||||
descriptorIndexingProperties.maxDescriptorSetUpdateAfterBindStorageImages};
|
||||
MGLOG_I("Vulkan: update-after-bind descriptor layouts enabled");
|
||||
} else {
|
||||
MGLOG_I("Vulkan: descriptor indexing is present but lacks the complete update-after-bind feature set; "
|
||||
"using ordinary descriptor layouts");
|
||||
}
|
||||
} else {
|
||||
MGLOG_I("Vulkan: descriptor indexing unavailable; using ordinary descriptor layouts");
|
||||
}
|
||||
|
||||
VkPhysicalDeviceIndexTypeUint8Features indexTypeUint8Features{};
|
||||
indexTypeUint8Features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_INDEX_TYPE_UINT8_FEATURES;
|
||||
if (indexTypeUint8ExtensionName != nullptr) {
|
||||
|
||||
@@ -555,6 +555,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool m_shaderDrawParametersExtensionEnabled = false;
|
||||
Bool m_shaderDrawParametersFeatureEnabled = false;
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// Set only after descriptor-indexing feature AND property queries prove that
|
||||
// update-after-bind is legal for every descriptor category this renderer emits.
|
||||
ProgramFactory::UpdateAfterBindLimits m_updateAfterBindLimits{};
|
||||
// fillModeNonSolid gates VK_POLYGON_MODE_LINE/_POINT (glPolygonMode); independentBlend gates
|
||||
// per-draw-buffer color write masks (glColorMaski). Both are cached at device creation and
|
||||
// drive a runtime fallback when the device lacks them.
|
||||
|
||||
@@ -10,6 +10,9 @@
|
||||
#include <Config.h>
|
||||
#include <MG_Util/BackendLoaders/OpenGL/Loader.h>
|
||||
#include <MG_Util/Converters/MGToStr/GLExtensionConverter.h>
|
||||
#if defined(MOBILEGL_ENABLE_DILIGENT)
|
||||
#include <MG_Backend/Diligent/BackendObject_Diligent.h>
|
||||
#endif
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
void LogBackendInfo() {
|
||||
@@ -55,6 +58,11 @@ namespace MobileGL::MG_Backend {
|
||||
case BackendType::DirectVulkan:
|
||||
pActiveBackendObject = MakeUnique<DirectVulkan::BackendObject_DirectVulkan>();
|
||||
break;
|
||||
#if defined(MOBILEGL_ENABLE_DILIGENT)
|
||||
case BackendType::DiligentVulkan:
|
||||
pActiveBackendObject = MakeUnique<DiligentBackend::BackendObject_Diligent>();
|
||||
break;
|
||||
#endif
|
||||
case BackendType::Unknown:
|
||||
default:
|
||||
MGLOG_W("Unknown backend type, defaulting to unknown backend");
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/BufferEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/BufferEnumConverter.h>
|
||||
#include <MG_Util/Texture/PixelStoreProcessor.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
@@ -31,6 +32,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
NamedBufferData,
|
||||
NamedBufferSubData,
|
||||
CopyNamedBufferSubData,
|
||||
ClearBufferData,
|
||||
ClearBufferSubData,
|
||||
ClearNamedBufferData,
|
||||
ClearNamedBufferSubData,
|
||||
MapBufferRange,
|
||||
@@ -65,6 +68,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return "NamedBufferSubData";
|
||||
case BufferOp::CopyNamedBufferSubData:
|
||||
return "CopyNamedBufferSubData";
|
||||
case BufferOp::ClearBufferData:
|
||||
return "ClearBufferData";
|
||||
case BufferOp::ClearBufferSubData:
|
||||
return "ClearBufferSubData";
|
||||
case BufferOp::ClearNamedBufferData:
|
||||
return "ClearNamedBufferData";
|
||||
case BufferOp::ClearNamedBufferSubData:
|
||||
@@ -143,16 +150,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// The pattern is replicated verbatim, which is only the whole story while the client
|
||||
// layout already matches the internal format - the case every entry point in practice
|
||||
// uses, and the only one the conversion machinery here can express. Say so rather than
|
||||
// quietly writing a differently-sized pattern.
|
||||
const SizeT sourceSize = MG_Util::GetInputBytesPerPixel(inputFormat, pixelType);
|
||||
if (sourceSize != elementSize) {
|
||||
MGLOG_W_ONCE("%s: clear pattern is %zu bytes but internalformat 0x%X stores %zu; "
|
||||
"converting between them is not implemented",
|
||||
GetBufferOpName(op), sourceSize, internalformat, elementSize);
|
||||
}
|
||||
return elementSize;
|
||||
}
|
||||
|
||||
@@ -194,27 +191,59 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearNamedBufferRange_State(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data, BufferOp op) {
|
||||
Bool BuildClearPattern(GLenum internalformat, GLenum format, GLenum type, const void* data,
|
||||
SizeT patternSize, BufferOp op, Vector<Uint8>& pattern) {
|
||||
const TextureInternalFormat internal = MG_Util::ConvertGLEnumToTextureInternalFormat(internalformat);
|
||||
const TextureInputFormat inputFormat = MG_Util::ConvertGLEnumToTextureInputFormat(format);
|
||||
const TexturePixelDataType inputType = MG_Util::ConvertGLEnumToTexturePixelDataType(type);
|
||||
|
||||
Vector<Uint8> zeroInput;
|
||||
const void* inputPixel = data;
|
||||
if (inputPixel == nullptr) {
|
||||
const SizeT inputSize = MG_Util::GetInputBytesPerPixel(inputFormat, inputType);
|
||||
if (inputSize == 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", GetBufferOpName(op),
|
||||
"format and type do not describe a source pixel."));
|
||||
return false;
|
||||
}
|
||||
zeroInput.resize(inputSize);
|
||||
inputPixel = zeroInput.data();
|
||||
}
|
||||
|
||||
if (!MG_Util::PixelStoreProcessor::ConvertOnePixelToInternal(
|
||||
internal, inputFormat, inputType, inputPixel, pattern)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", GetBufferOpName(op),
|
||||
std::format("Cannot convert one ({}, {}) pixel into internalformat 0x{:X}.",
|
||||
MG_Util::ConvertGLEnumToString(format), MG_Util::ConvertGLEnumToString(type),
|
||||
internalformat)));
|
||||
return false;
|
||||
}
|
||||
|
||||
if (data == nullptr) {
|
||||
// GL defines a null clear value as all zero bits in the destination store, while
|
||||
// retaining the format/type validation above.
|
||||
pattern.assign(patternSize, 0);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearBufferRange_State(const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data, BufferOp op) {
|
||||
const SizeT patternSize = GetClearPatternSize(internalformat, format, type, op);
|
||||
if (patternSize == 0) return;
|
||||
|
||||
auto bufferObject = GetNamedBufferObject(buffer, op);
|
||||
if (!bufferObject) return;
|
||||
if (!ValidateBufferClearRange(bufferObject, offset, size, patternSize, op)) return;
|
||||
if (size == 0) return;
|
||||
|
||||
Vector<Uint8> clearData(static_cast<SizeT>(size));
|
||||
if (data) {
|
||||
const auto* pattern = static_cast<const Uint8*>(data);
|
||||
for (SizeT at = 0; at < clearData.size(); at += patternSize) {
|
||||
Memcpy(clearData.data() + at, pattern, patternSize);
|
||||
}
|
||||
} else {
|
||||
Memset(clearData.data(), 0, clearData.size());
|
||||
}
|
||||
|
||||
bufferObject->UploadSubData({clearData.data(), clearData.size()}, static_cast<SizeT>(offset));
|
||||
Vector<Uint8> pattern;
|
||||
if (!BuildClearPattern(internalformat, format, type, data, patternSize, op, pattern)) return;
|
||||
bufferObject->FillSubData({pattern.data(), pattern.size()}, static_cast<SizeT>(offset),
|
||||
static_cast<SizeT>(size));
|
||||
}
|
||||
|
||||
auto& GetBufferBindingSlot(BufferTarget target) {
|
||||
@@ -1197,17 +1226,34 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
static_cast<SizeT>(writeOffset), static_cast<SizeT>(size));
|
||||
}
|
||||
|
||||
void ClearBufferData_State(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
auto bufferObject = GetBoundBufferObject(target, BufferOp::ClearBufferData);
|
||||
if (!bufferObject) return;
|
||||
ClearBufferRange_State(bufferObject, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
|
||||
type, data, BufferOp::ClearBufferData);
|
||||
}
|
||||
|
||||
void ClearBufferSubData_State(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data) {
|
||||
auto bufferObject = GetBoundBufferObject(target, BufferOp::ClearBufferSubData);
|
||||
if (!bufferObject) return;
|
||||
ClearBufferRange_State(bufferObject, internalformat, offset, size, format, type, data,
|
||||
BufferOp::ClearBufferSubData);
|
||||
}
|
||||
|
||||
void ClearNamedBufferData_State(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
auto bufferObject = GetNamedBufferObject(buffer, BufferOp::ClearNamedBufferData);
|
||||
if (!bufferObject) return;
|
||||
ClearNamedBufferRange_State(buffer, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
|
||||
type, data, BufferOp::ClearNamedBufferData);
|
||||
ClearBufferRange_State(bufferObject, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
|
||||
type, data, BufferOp::ClearNamedBufferData);
|
||||
}
|
||||
|
||||
void ClearNamedBufferSubData_State(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data) {
|
||||
ClearNamedBufferRange_State(buffer, internalformat, offset, size, format, type, data,
|
||||
BufferOp::ClearNamedBufferSubData);
|
||||
auto bufferObject = GetNamedBufferObject(buffer, BufferOp::ClearNamedBufferSubData);
|
||||
if (!bufferObject) return;
|
||||
ClearBufferRange_State(bufferObject, internalformat, offset, size, format, type, data,
|
||||
BufferOp::ClearNamedBufferSubData);
|
||||
}
|
||||
|
||||
void* MapNamedBuffer_State(GLuint buffer, GLenum access) {
|
||||
@@ -1662,6 +1708,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
CopyNamedBufferSubData_State(readBuffer, writeBuffer, readOffset, writeOffset, size);
|
||||
}
|
||||
|
||||
void ClearBufferData(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
ClearBufferData_State(target, internalformat, format, type, data);
|
||||
}
|
||||
|
||||
void ClearBufferSubData(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
|
||||
GLenum type, const void* data) {
|
||||
ClearBufferSubData_State(target, internalformat, offset, size, format, type, data);
|
||||
}
|
||||
|
||||
void ClearNamedBufferData(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
ClearNamedBufferData_State(buffer, internalformat, format, type, data);
|
||||
}
|
||||
|
||||
@@ -27,6 +27,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void NamedBufferSubData(GLuint buffer, GLintptr offset, GLsizeiptr size, const void* data);
|
||||
void CopyNamedBufferSubData(GLuint readBuffer, GLuint writeBuffer, GLintptr readOffset, GLintptr writeOffset,
|
||||
GLsizeiptr size);
|
||||
void ClearBufferData(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data);
|
||||
void ClearBufferSubData(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
|
||||
GLenum type, const void* data);
|
||||
void ClearNamedBufferData(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data);
|
||||
void ClearNamedBufferSubData(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
|
||||
GLenum type, const void* data);
|
||||
|
||||
@@ -985,8 +985,8 @@ DECLARE_GL_FUNCTION_HEAD(void, DrawElementsInstancedBaseVertexBaseInstance, GLen
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetActiveAtomicCounterBufferiv, GLuint program, GLuint bufferIndex, GLenum pname, GLint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetActiveAtomicCounterBufferiv, program, bufferIndex, pname, params)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackInstanced, GLenum mode, GLuint id, GLsizei instancecount) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackInstanced, mode, id, instancecount)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackStreamInstanced, GLenum mode, GLuint id, GLuint stream, GLsizei instancecount) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackStreamInstanced, mode, id, stream, instancecount)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearBufferData, GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearBufferData, target, internalformat, format, type, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearBufferSubData, GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearBufferSubData, target, internalformat, offset, size, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearBufferData, GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearBufferData, target, internalformat, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearBufferSubData, GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearBufferSubData, target, internalformat, offset, size, format, type, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetInternalformati64v, GLenum target, GLenum internalformat, GLenum pname, GLsizei count, GLint64* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetInternalformati64v, target, internalformat, pname, count, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, InvalidateTexSubImage, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, InvalidateTexSubImage, texture, level, xoffset, yoffset, zoffset, width, height, depth)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, InvalidateTexImage, GLuint texture, GLint level) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, InvalidateTexImage, texture, level)
|
||||
|
||||
@@ -175,14 +175,14 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
|
||||
struct ContextObject {
|
||||
Display* XDisplay = nullptr;
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLDisplay Dpy = EGL_NO_DISPLAY;
|
||||
EGLConfig Config = nullptr;
|
||||
EGLContext Context = EGL_NO_CONTEXT;
|
||||
const FBConfigInfo* FBConfig = nullptr;
|
||||
};
|
||||
|
||||
struct DrawableSurface {
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLDisplay Dpy = EGL_NO_DISPLAY;
|
||||
EGLSurface Surface = EGL_NO_SURFACE;
|
||||
Uint32 Width = 0;
|
||||
Uint32 Height = 0;
|
||||
@@ -294,7 +294,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
if (width == surface.Width && height == surface.Height) {
|
||||
return;
|
||||
}
|
||||
if (EGLImpl::ResizePlatformWindowSurface(surface.Display, surface.Surface,
|
||||
if (EGLImpl::ResizePlatformWindowSurface(surface.Dpy, surface.Surface,
|
||||
static_cast<EGLint>(width),
|
||||
static_cast<EGLint>(height))) {
|
||||
surface.Width = width;
|
||||
@@ -324,7 +324,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLSurface surface = EGLImpl::CreatePlatformWindowSurface(
|
||||
context.Display, context.Config, reinterpret_cast<void*>(drawable), attribs);
|
||||
context.Dpy, context.Config, reinterpret_cast<void*>(drawable), attribs);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E_ONCE("glx: failed to create window surface for drawable 0x%lx (%ux%u)", drawable,
|
||||
width, height);
|
||||
@@ -332,7 +332,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
}
|
||||
|
||||
DrawableSurface record;
|
||||
record.Display = context.Display;
|
||||
record.Dpy = context.Dpy;
|
||||
record.Surface = surface;
|
||||
record.Width = width;
|
||||
record.Height = height;
|
||||
@@ -388,7 +388,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
|
||||
ContextObject object;
|
||||
object.XDisplay = dpy;
|
||||
object.Display = display;
|
||||
object.Dpy = display;
|
||||
object.Config = config;
|
||||
object.Context = eglContext;
|
||||
object.FBConfig = fbconfig;
|
||||
@@ -895,7 +895,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
return;
|
||||
}
|
||||
if (object->Context != EGL_NO_CONTEXT) {
|
||||
EGLImpl::DestroyContext(object->Display, object->Context);
|
||||
EGLImpl::DestroyContext(object->Dpy, object->Context);
|
||||
}
|
||||
Contexts().erase(context);
|
||||
}
|
||||
@@ -929,7 +929,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (!EGLImpl::MakeCurrent(object->Display, surface->Surface, surface->Surface,
|
||||
if (!EGLImpl::MakeCurrent(object->Dpy, surface->Surface, surface->Surface,
|
||||
object->Context)) {
|
||||
MGLOG_E_ONCE("glx: eglMakeCurrent failed (drawable=0x%lx, ctx=%p)", drawable, context);
|
||||
return 0;
|
||||
@@ -962,7 +962,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
return;
|
||||
}
|
||||
SyncSurfaceSize(dpy, drawable, it->second);
|
||||
EGLImpl::SwapBuffers(it->second.Display, it->second.Surface);
|
||||
EGLImpl::SwapBuffers(it->second.Dpy, it->second.Surface);
|
||||
}
|
||||
|
||||
GLXDrawableHandle CreateWindow(Display*, GLXFBConfigHandle config, GLXDrawableHandle window,
|
||||
@@ -988,7 +988,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
if (it == surfaces.end()) {
|
||||
return;
|
||||
}
|
||||
EGLImpl::DestroySurface(it->second.Display, it->second.Surface);
|
||||
EGLImpl::DestroySurface(it->second.Dpy, it->second.Surface);
|
||||
surfaces.erase(it);
|
||||
}
|
||||
|
||||
|
||||
@@ -250,6 +250,34 @@ namespace MobileGL::MG_State::GLState {
|
||||
NotifyContentWrite(atOffset, data.size);
|
||||
}
|
||||
|
||||
void BufferObject::FillSubData(DataPtr pattern, SizeT atOffset, SizeT size) {
|
||||
MOBILEGL_ASSERT(pattern.data != nullptr && pattern.size > 0,
|
||||
"FillSubData requires a non-empty pattern.");
|
||||
MOBILEGL_ASSERT(size % pattern.size == 0,
|
||||
"FillSubData size (%zu) must be a multiple of pattern size (%zu).", size, pattern.size);
|
||||
MOBILEGL_ASSERT(atOffset <= m_size && size <= m_size - atOffset,
|
||||
"FillSubData out of bounds: atOffset (%zu) + size (%zu) > m_size (%zu)", atOffset, size,
|
||||
m_size);
|
||||
MOBILEGL_ASSERT(!m_isMapped || (m_mappingAccess & BufferMappingAccessBit::Persistent),
|
||||
"Cannot fill data while buffer is non-persistently mapped.");
|
||||
if (size == 0) return;
|
||||
|
||||
// A clear is ordered after all earlier GPU writes. Partial clears additionally need the
|
||||
// retained shadow bytes; whole-store clears need the same synchronization before writing
|
||||
// an adopted persistent mapping that the GPU may still be accessing.
|
||||
SyncGpuWrites();
|
||||
|
||||
Uint8* dst = m_resource.Bytes() + atOffset;
|
||||
if (pattern.size == 1) {
|
||||
Memset(dst, *static_cast<const Uint8*>(pattern.data), size);
|
||||
} else {
|
||||
for (SizeT at = 0; at < size; at += pattern.size) {
|
||||
Memcpy(dst + at, pattern.data, pattern.size);
|
||||
}
|
||||
}
|
||||
NotifyContentWrite(atOffset, size);
|
||||
}
|
||||
|
||||
void BufferObject::DownloadSubData(void* dst, SizeT atOffset, SizeT size) const {
|
||||
MOBILEGL_ASSERT(atOffset + size <= m_size,
|
||||
"DownloadSubData out of bounds: atOffset (%zu) + size (%zu) > m_size (%zu)", atOffset, size,
|
||||
|
||||
@@ -132,6 +132,9 @@ namespace MobileGL {
|
||||
|
||||
void UploadData(DataPtr data, SizeT atOffset);
|
||||
void UploadSubData(DataPtr data, SizeT atOffset);
|
||||
// Repeats one already-converted element through [atOffset, atOffset + size) and
|
||||
// publishes the range as one content mutation.
|
||||
void FillSubData(DataPtr pattern, SizeT atOffset, SizeT size);
|
||||
// Reads `size` bytes from the CPU shadow at `atOffset` into `dst` (glGetBufferSubData).
|
||||
// The shadow reflects CPU writes (BufferData/SubData/maps) and backend write-backs, but not
|
||||
// arbitrary GPU-side writes.
|
||||
|
||||
@@ -39,6 +39,11 @@ namespace MobileGL::MG_State {
|
||||
return m_compileEnv;
|
||||
}
|
||||
|
||||
void GLContext::InvalidateCompileEnv() {
|
||||
m_compileEnv.reset();
|
||||
m_compileEnvBackend = nullptr;
|
||||
}
|
||||
|
||||
// Error
|
||||
void GLContext::RecordError(ErrorCode code, UniquePtr<ErrorInfo> info) {
|
||||
// Invariant I1, mechanically enforced: the GL error state is GL-thread-owned.
|
||||
|
||||
@@ -413,9 +413,12 @@ namespace MobileGL {
|
||||
// cannot be captured in MG_State::Init() - that runs BEFORE MG_Backend::Init(),
|
||||
// so there is no backend to query yet. Re-captured whenever the active backend
|
||||
// object changes, which also rolls the fingerprint and therefore invalidates
|
||||
// every P0b preprocess memo keyed against the old one.
|
||||
// every P0b preprocess memo keyed against the old one. A backend whose dynamic
|
||||
// capabilities become available without changing object identity must call
|
||||
// InvalidateCompileEnv() after publishing them.
|
||||
// GL thread only.
|
||||
const SharedPtr<const MG_Util::ShaderTranspiler::CompileEnv>& GetCompileEnv();
|
||||
void InvalidateCompileEnv();
|
||||
|
||||
private:
|
||||
// State Components
|
||||
|
||||
@@ -60,6 +60,8 @@ namespace MobileGL::MG_State::GLState {
|
||||
Uint externalIndex = 0; // logs only
|
||||
Vector<LinkShaderInput> shaders; // already stage-sorted
|
||||
SharedPtr<const MG_Util::ShaderTranspiler::CompileEnv> env;
|
||||
// Startup configuration copied with the task, never read from worker code.
|
||||
Bool enableSpirvValidation = false;
|
||||
// The four "takes effect at the next link" request maps. Snapshotted rather than
|
||||
// referenced, which is precisely what makes glBindAttribLocation and friends
|
||||
// legal to call over a pending link without cancelling it: the pending link keeps
|
||||
|
||||
@@ -494,6 +494,7 @@ namespace MobileGL::MG_State::GLState {
|
||||
auto task = MakeShared<ProgramLinkTask>();
|
||||
task->in.externalIndex = m_externalIndex;
|
||||
task->in.env = MG_Util::ShaderTranspiler::GetCurrentCompileEnv();
|
||||
task->in.enableSpirvValidation = MG_Config::Features.EnableSpirvValidation;
|
||||
task->in.explicitAttribLocations = m_explicitAttribLocations;
|
||||
task->in.explicitFragDataLocation = m_explicitFragDataLocation;
|
||||
task->in.explicitFragDataIndex = m_explicitFragDataIndex;
|
||||
|
||||
@@ -819,6 +819,9 @@ namespace MobileGL::MG_State::GLState {
|
||||
// backend asks this exactly where it used to ask GetLinkStatus(), i.e. right before
|
||||
// it builds or draws with the program.
|
||||
Bool GetSpirvStatus() const { return Spirv().spirvStatus; }
|
||||
// Copied from the link task that generated this program's SPIR-V. Backends use it for
|
||||
// their final transforms, which must honor the same diagnostic setting as phase B.
|
||||
Bool GetSpirvValidationEnabled() const { return Spirv().enableSpirvValidation; }
|
||||
|
||||
// The linked glslang reflection itself, for the ONE consumer that needs resource
|
||||
// lists no typed getter above exposes: the GL program-interface query layer
|
||||
@@ -985,6 +988,7 @@ namespace MobileGL::MG_State::GLState {
|
||||
// cannot be lifted out of glslang's reflection instead.
|
||||
struct SpirvArtifacts {
|
||||
Vector<Vector<unsigned>> generatedSpirv;
|
||||
Bool enableSpirvValidation = false;
|
||||
// Byte offset of each uniform location inside globalUboScratch, or
|
||||
// kInvalidUniformOffset. Sized maxUniformLocation + 1 by the routing pass.
|
||||
Vector<Uint> uniformOffsets;
|
||||
|
||||
@@ -102,7 +102,11 @@ namespace MobileGL::MG_State::GLState {
|
||||
}
|
||||
|
||||
MGLOG_D("ProgramObject %u: Starting SPIR-V generation", externalIndex);
|
||||
GenerateSpirv(handoff, externalIndex);
|
||||
const Bool deferOutputValidationForDirectVulkan =
|
||||
m_phaseA->in.env != nullptr && m_phaseA->in.env->backend == BackendType::DirectVulkan;
|
||||
const Bool enableSpirvValidation = m_phaseA->in.enableSpirvValidation;
|
||||
artifacts.enableSpirvValidation = enableSpirvValidation;
|
||||
GenerateSpirv(handoff, externalIndex, deferOutputValidationForDirectVulkan, enableSpirvValidation);
|
||||
// GlslangToSpv was the only consumer of the parsed ASTs; everything after this point
|
||||
// works on the SPIR-V and on the TProgram's own self-contained reflection pool. Drop
|
||||
// them here rather than at the end of the body, which is ~87% of this node's runtime
|
||||
@@ -137,7 +141,9 @@ namespace MobileGL::MG_State::GLState {
|
||||
artifacts.generatedSpirv.size());
|
||||
}
|
||||
|
||||
void ProgramSpirvTask::GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, const Uint externalIndex) {
|
||||
void ProgramSpirvTask::GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, const Uint externalIndex,
|
||||
const Bool deferOutputValidationForDirectVulkan,
|
||||
const Bool enableSpirvValidation) {
|
||||
/* As we passed first stage compilation/linking,
|
||||
* we'll assume all the operations here should
|
||||
* pass. We may be able to employ some optimizations
|
||||
@@ -169,7 +175,8 @@ namespace MobileGL::MG_State::GLState {
|
||||
Bool allOptimized = true;
|
||||
{
|
||||
for (auto& spv : artifacts.generatedSpirv) {
|
||||
auto success = ShaderCompiler::SanitizeAndOptimizeBinary(spv, spv);
|
||||
auto success = ShaderCompiler::SanitizeAndOptimizeBinary(
|
||||
spv, spv, !deferOutputValidationForDirectVulkan, enableSpirvValidation);
|
||||
if (!success) {
|
||||
// The one genuine phase-B failure mode: one of the seven optimizer passes
|
||||
// reported failure, so `spv` is whatever the run left behind. A fordebug
|
||||
|
||||
@@ -65,7 +65,8 @@ namespace MobileGL::MG_State::GLState {
|
||||
private:
|
||||
void RunBody() override;
|
||||
|
||||
void GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, Uint externalIndex);
|
||||
void GenerateSpirv(const ProgramLinkTask::SpirvHandoff& handoff, Uint externalIndex,
|
||||
Bool deferOutputValidationForDirectVulkan, Bool enableSpirvValidation);
|
||||
void BuildGlobalUboRouting(const ProgramLinkTask::SpirvHandoff& handoff, Uint externalIndex);
|
||||
|
||||
// Worker-side MGLOG replacement, replayed by the join on the GL thread. Same reason as
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
|
||||
message(STATUS "Generating build files for MobileGL Diligent Backend Test...")
|
||||
|
||||
add_executable(
|
||||
DiligentVulkanSanityTest
|
||||
SanityTest.cpp
|
||||
)
|
||||
|
||||
target_include_directories(DiligentVulkanSanityTest PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
DiligentVulkanSanityTest PRIVATE
|
||||
GTest::gtest_main
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
include(GoogleTest)
|
||||
gtest_discover_tests(DiligentVulkanSanityTest DISCOVERY_TIMEOUT 30 PROPERTIES LABELS integration)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -16,6 +16,7 @@
|
||||
#include <MG_State/GLState/Core.h>
|
||||
|
||||
#include <MG_Impl/GLImpl/Buffer/GL_Buffer.h>
|
||||
#include <MG_Impl/GetProcAddress.h>
|
||||
#include <MG_Impl/GLImpl/Getter/GL_Getter.h>
|
||||
|
||||
using namespace MobileGL;
|
||||
@@ -599,6 +600,117 @@ TEST_F(BufferTest, ClearNamedBufferSubDataRepeatsPattern) {
|
||||
EXPECT_EQ(actual, (Vector<Uint32>{0, pattern, pattern, pattern, 0}));
|
||||
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
TEST_F(BufferTest, ClearBufferSubDataInitializesIrisStaticSsboRange) {
|
||||
GLuint buffer = 0;
|
||||
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, buffer);
|
||||
|
||||
Vector<Uint8> initial(32, 0x7F);
|
||||
MobileGL::MG_Impl::GLImpl::BufferData(
|
||||
GL_SHADER_STORAGE_BUFFER, initial.size(), initial.data(), GL_STATIC_DRAW);
|
||||
const GLbyte zero = 0;
|
||||
const auto clear = reinterpret_cast<PFNGLCLEARBUFFERSUBDATAPROC>(
|
||||
MobileGL::MG_Impl::GetProcAddress("glClearBufferSubData"));
|
||||
ASSERT_NE(clear, nullptr);
|
||||
clear(GL_SHADER_STORAGE_BUFFER, GL_R8, 4, 24, GL_RED, GL_BYTE, &zero);
|
||||
|
||||
Vector<Uint8> actual(initial.size());
|
||||
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
|
||||
ASSERT_NE(bufferObject, nullptr);
|
||||
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size());
|
||||
EXPECT_EQ(actual, (Vector<Uint8>{0x7F, 0x7F, 0x7F, 0x7F,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0x7F, 0x7F, 0x7F, 0x7F}));
|
||||
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
|
||||
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
|
||||
DrainPendingGlErrors();
|
||||
}
|
||||
|
||||
TEST_F(BufferTest, ClearBufferSubDataInitializesCompleteIrisStaticSsbo) {
|
||||
constexpr SizeT irisStaticSsboSize = 5'000'192;
|
||||
GLuint buffer = 0;
|
||||
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, buffer);
|
||||
|
||||
Vector<Uint8> initial(irisStaticSsboSize, 0x7F);
|
||||
MobileGL::MG_Impl::GLImpl::BufferData(
|
||||
GL_SHADER_STORAGE_BUFFER, initial.size(), initial.data(), GL_STATIC_DRAW);
|
||||
const GLbyte zero = 0;
|
||||
MobileGL::MG_Impl::GLImpl::ClearBufferSubData(
|
||||
GL_SHADER_STORAGE_BUFFER, GL_R8, 0, irisStaticSsboSize, GL_RED, GL_BYTE, &zero);
|
||||
|
||||
Vector<Uint8> actual(irisStaticSsboSize);
|
||||
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
|
||||
ASSERT_NE(bufferObject, nullptr);
|
||||
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size());
|
||||
EXPECT_EQ(actual, Vector<Uint8>(irisStaticSsboSize, 0));
|
||||
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
|
||||
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
|
||||
DrainPendingGlErrors();
|
||||
}
|
||||
|
||||
TEST_F(BufferTest, ClearBufferDataConvertsOneClientPixelBeforeRepeatingIt) {
|
||||
GLuint buffer = 0;
|
||||
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, buffer);
|
||||
|
||||
Vector<Uint32> initial(4, 0u);
|
||||
MobileGL::MG_Impl::GLImpl::BufferData(GL_ARRAY_BUFFER, initial.size() * sizeof(Uint32), initial.data(),
|
||||
GL_STATIC_DRAW);
|
||||
const Uint8 value = 0xAB;
|
||||
MobileGL::MG_Impl::GLImpl::ClearBufferData(
|
||||
GL_ARRAY_BUFFER, GL_R32UI, GL_RED_INTEGER, GL_UNSIGNED_BYTE, &value);
|
||||
|
||||
Vector<Uint32> actual(initial.size());
|
||||
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
|
||||
ASSERT_NE(bufferObject, nullptr);
|
||||
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size() * sizeof(Uint32));
|
||||
EXPECT_EQ(actual, Vector<Uint32>(initial.size(), value));
|
||||
EXPECT_EQ(MobileGL::MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, 0);
|
||||
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
|
||||
DrainPendingGlErrors();
|
||||
}
|
||||
|
||||
TEST_F(BufferTest, ClearBufferSubDataRejectsUnboundTarget) {
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_SHADER_STORAGE_BUFFER, 0);
|
||||
const GLbyte zero = 0;
|
||||
MobileGL::MG_Impl::GLImpl::ClearBufferSubData(
|
||||
GL_SHADER_STORAGE_BUFFER, GL_R8, 0, 1, GL_RED, GL_BYTE, &zero);
|
||||
ExpectSingleGlError(GL_INVALID_OPERATION);
|
||||
}
|
||||
|
||||
TEST_F(BufferTest, ClearBufferDataRejectsInvalidPixelFormatTypePairs) {
|
||||
GLuint buffer = 0;
|
||||
MobileGL::MG_Impl::GLImpl::GenBuffers(1, &buffer);
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, buffer);
|
||||
|
||||
const Vector<Uint8> initial{0x7F, 0x7F};
|
||||
MobileGL::MG_Impl::GLImpl::BufferData(GL_ARRAY_BUFFER, initial.size(), initial.data(), GL_STATIC_DRAW);
|
||||
const Uint16 packed = 0;
|
||||
MobileGL::MG_Impl::GLImpl::ClearBufferData(
|
||||
GL_ARRAY_BUFFER, GL_R16, GL_RED, GL_UNSIGNED_SHORT_5_6_5, &packed);
|
||||
ExpectSingleGlError(GL_INVALID_VALUE);
|
||||
MobileGL::MG_Impl::GLImpl::ClearBufferData(
|
||||
GL_ARRAY_BUFFER, GL_R16, GL_RED, GL_UNSIGNED_SHORT_5_6_5, nullptr);
|
||||
ExpectSingleGlError(GL_INVALID_VALUE);
|
||||
|
||||
Vector<Uint8> actual(initial.size());
|
||||
auto bufferObject = MobileGL::MG_State::pGLContext->GetBufferObject(buffer);
|
||||
ASSERT_NE(bufferObject, nullptr);
|
||||
Memcpy(actual.data(), bufferObject->AcquireMemory(false, true, false), actual.size());
|
||||
EXPECT_EQ(actual, initial);
|
||||
|
||||
MobileGL::MG_Impl::GLImpl::BindBuffer(GL_ARRAY_BUFFER, 0);
|
||||
MobileGL::MG_Impl::GLImpl::DeleteBuffers(1, &buffer);
|
||||
DrainPendingGlErrors();
|
||||
}
|
||||
|
||||
|
||||
// GL 4.6 core 6.5: glBufferSubData fails only when the written range OVERLAPS the mapped range.
|
||||
|
||||
@@ -84,3 +84,6 @@ add_subdirectory(Backend/DirectGLES)
|
||||
if (ENABLE_INTEGRATION_TESTS)
|
||||
add_subdirectory(Backend/DirectVulkan)
|
||||
endif()
|
||||
if (MOBILEGL_ENABLE_DILIGENT)
|
||||
add_subdirectory(Backend/Diligent)
|
||||
endif()
|
||||
|
||||
@@ -2073,153 +2073,6 @@ void main() {
|
||||
EXPECT_NE(source.find("layout(std140) uniform Blk"), String::npos);
|
||||
}
|
||||
|
||||
namespace {
|
||||
String MakeLinearSubgroupPrefixScanShader() {
|
||||
return R"(#version 460 core
|
||||
#extension GL_KHR_shader_subgroup_arithmetic : enable
|
||||
layout(local_size_x = 1024) in;
|
||||
shared float prefixSumCache[64];
|
||||
|
||||
layout(std430, binding = 0) writeonly buffer OutputBuffer {
|
||||
float outputValues[];
|
||||
};
|
||||
|
||||
void main() {
|
||||
float importance = 1.0f;
|
||||
float prefixSum = subgroupInclusiveAdd(importance);
|
||||
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||
barrier();
|
||||
uint loopLength = uint(findMSB(gl_NumSubgroups));
|
||||
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
|
||||
for (uint i = 0; i < loopLength; i++) {
|
||||
if ((gl_SubgroupID & (1u << i)) > 0u) {
|
||||
prefixSum += prefixSumCache[(gl_SubgroupID >> i << i) - 1u];
|
||||
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||
}
|
||||
barrier();
|
||||
}
|
||||
if (gl_LocalInvocationID.x == uint(1024 - 1)) prefixSumCache[0] = prefixSum;
|
||||
barrier();
|
||||
float sum = prefixSumCache[0];
|
||||
float warp = (prefixSum - importance) / sum - float(gl_LocalInvocationID.x + 1u) / float(1024);
|
||||
outputValues[gl_GlobalInvocationID.x] = warp;
|
||||
}
|
||||
)";
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanUsesSharedMemoryAndProducesValidSpirv) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = MakeLinearSubgroupPrefixScanShader();
|
||||
ASSERT_TRUE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
|
||||
|
||||
EXPECT_NE(source.find("shared float prefixSumCache[1024]"), String::npos) << source;
|
||||
EXPECT_NE(source.find("mglVirtualSubgroupInvocation"), String::npos) << source;
|
||||
EXPECT_NE(source.find("for (uint mglPrefixLane"), String::npos) << source;
|
||||
EXPECT_EQ(source.find("subgroupInclusiveAdd"), String::npos) << source;
|
||||
EXPECT_EQ(source.find("gl_Subgroup"), String::npos) << source;
|
||||
|
||||
const String onceRewritten = source;
|
||||
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
|
||||
EXPECT_EQ(source, onceRewritten);
|
||||
|
||||
ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source};
|
||||
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
|
||||
ASSERT_TRUE(shaderResult) << shaderResult.error().log << "\nsource:\n" << source;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
|
||||
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
|
||||
ASSERT_TRUE(programResult) << programResult.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()};
|
||||
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
ASSERT_TRUE(binaryResult) << binaryResult.error().log;
|
||||
ASSERT_EQ(binaryResult->size(), 1u);
|
||||
|
||||
String validationDiagnostics;
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
tools.SetMessageConsumer([&](spv_message_level_t, const char*, const spv_position_t&, const char* message) {
|
||||
validationDiagnostics += message;
|
||||
validationDiagnostics += '\n';
|
||||
});
|
||||
EXPECT_TRUE(tools.Validate(binaryResult->front())) << validationDiagnostics;
|
||||
|
||||
String spirvText;
|
||||
ASSERT_TRUE(tools.Disassemble(binaryResult->front(), &spirvText));
|
||||
EXPECT_EQ(spirvText.find("OpGroupNonUniform"), String::npos) << spirvText;
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsOtherStagesAndSubgroupWidths) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const String original = MakeLinearSubgroupPrefixScanShader();
|
||||
for (const auto& [stage, subgroupSize] :
|
||||
{std::pair{ShaderStage::Compute, Uint32{32}}, std::pair{ShaderStage::Fragment, Uint32{64}},
|
||||
std::pair{ShaderStage::Compute, Uint32{96}}}) {
|
||||
String source = original;
|
||||
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(stage, subgroupSize, source));
|
||||
EXPECT_EQ(source, original);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsPartialOrUnsafeTemplateMatches) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const auto expectUnchanged = [](String source) {
|
||||
const String original = source;
|
||||
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
|
||||
EXPECT_EQ(source, original);
|
||||
};
|
||||
|
||||
String wrongLocalSize = MakeLinearSubgroupPrefixScanShader();
|
||||
wrongLocalSize.replace(wrongLocalSize.find("local_size_x = 1024"), std::strlen("local_size_x = 1024"),
|
||||
"local_size_x = 512");
|
||||
expectUnchanged(std::move(wrongLocalSize));
|
||||
|
||||
String cacheHasAnotherUse = MakeLinearSubgroupPrefixScanShader();
|
||||
cacheHasAnotherUse.insert(cacheHasAnotherUse.find("float importance"), "prefixSumCache[0] = 0.0f;\n ");
|
||||
expectUnchanged(std::move(cacheHasAnotherUse));
|
||||
|
||||
String extraSubgroupBuiltin = MakeLinearSubgroupPrefixScanShader();
|
||||
extraSubgroupBuiltin.insert(extraSubgroupBuiltin.find("float importance"),
|
||||
"uvec4 extraMask = gl_SubgroupEqMask;\n ");
|
||||
expectUnchanged(std::move(extraSubgroupBuiltin));
|
||||
|
||||
String alteredBarrier = MakeLinearSubgroupPrefixScanShader();
|
||||
alteredBarrier.replace(alteredBarrier.find("barrier();"), std::strlen("barrier();"), "memoryBarrierShared();");
|
||||
expectUnchanged(std::move(alteredBarrier));
|
||||
|
||||
String nestedScan = MakeLinearSubgroupPrefixScanShader();
|
||||
nestedScan.insert(nestedScan.find("float prefixSum ="), "if (importance > 0.0f) {\n ");
|
||||
const SizeT consumerEnd = nestedScan.find(';', nestedScan.find("float warp ="));
|
||||
ASSERT_NE(consumerEnd, String::npos);
|
||||
nestedScan.insert(consumerEnd + 1, "\n }");
|
||||
expectUnchanged(std::move(nestedScan));
|
||||
|
||||
// ARB/NV spellings of lane-width-sensitive builtins must block the rewrite exactly
|
||||
// like their KHR counterparts.
|
||||
String arbSubgroupBuiltin = MakeLinearSubgroupPrefixScanShader();
|
||||
arbSubgroupBuiltin.insert(arbSubgroupBuiltin.find("float importance"),
|
||||
"uint arbLane = gl_SubGroupInvocationARB;\n ");
|
||||
expectUnchanged(std::move(arbSubgroupBuiltin));
|
||||
|
||||
String arbBallotCall = MakeLinearSubgroupPrefixScanShader();
|
||||
arbBallotCall.insert(arbBallotCall.find("float importance"),
|
||||
"uint64_t arbMask = ballotARB(true);\n ");
|
||||
expectUnchanged(std::move(arbBallotCall));
|
||||
|
||||
String nvWarpBuiltin = MakeLinearSubgroupPrefixScanShader();
|
||||
nvWarpBuiltin.insert(nvWarpBuiltin.find("float importance"),
|
||||
"uint warpSize = gl_WarpSizeNV;\n ");
|
||||
expectUnchanged(std::move(nvWarpBuiltin));
|
||||
|
||||
String nvShuffleCall = MakeLinearSubgroupPrefixScanShader();
|
||||
nvShuffleCall.insert(nvShuffleCall.find("float importance"),
|
||||
"float other = shuffleNV(1.0f, 0u, 32u);\n ");
|
||||
expectUnchanged(std::move(nvShuffleCall));
|
||||
}
|
||||
|
||||
// The LEXICAL half must fire at the source level (before the parse) for the
|
||||
// preempt-list names - the end-to-end ESSL tests cannot tell which half did the
|
||||
// rename, and for these names the parse would fail without the source rewrite.
|
||||
@@ -2448,9 +2301,6 @@ TEST_F(ProgramUtilTest, CompileEnvFingerprintTracksEveryInput) {
|
||||
otherExtensions.advertisedExtensions.push_back(MobileGL::E_GL_ARB_gpu_shader_int64);
|
||||
EXPECT_NE(ComputeCompileEnvFingerprint(otherExtensions), baseline);
|
||||
|
||||
CompileEnv otherQuirk = base;
|
||||
otherQuirk.subgroupPrefixScanQuirk = MobileGL::MG_Config::QuirkOverride::ForceOn;
|
||||
EXPECT_NE(ComputeCompileEnvFingerprint(otherQuirk), baseline);
|
||||
}
|
||||
|
||||
// The no-backend fallback must stay exactly what the pipeline used to do inline:
|
||||
@@ -2849,16 +2699,6 @@ vec4 helperTint() { return vec4(1.0); }
|
||||
return binaryResult->front();
|
||||
}
|
||||
|
||||
struct SpirvValidationScope {
|
||||
bool previous;
|
||||
explicit SpirvValidationScope(bool enabled)
|
||||
: previous(MG_Util::ShaderTranspiler::ShaderCompiler::SpirvValidationEnabled()) {
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::SetSpirvValidationEnabled(enabled);
|
||||
}
|
||||
~SpirvValidationScope() {
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::SetSpirvValidationEnabled(previous);
|
||||
}
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST_F(ProgramUtilTest, DeadPrivateChainVertexInputIsEliminatedFromOptimizedBinary) {
|
||||
@@ -2880,7 +2720,7 @@ TEST_F(ProgramUtilTest, DeadPrivateChainVertexInputIsEliminatedFromOptimizedBina
|
||||
<< "entry-point-with-calls shape it exists for";
|
||||
|
||||
Vector<Uint32> optimized;
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
|
||||
|
||||
const SpirvVariableCensus after = TakeVariableCensus(optimized);
|
||||
EXPECT_EQ(after.inputCount, 1u)
|
||||
@@ -2918,7 +2758,7 @@ void main() {
|
||||
ASSERT_GE(before.outputCount, 3u);
|
||||
|
||||
Vector<Uint32> optimized;
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
|
||||
EXPECT_EQ(TakeVariableCensus(optimized).outputCount, before.outputCount)
|
||||
<< "a declared-but-unwritten output was deleted; a fragment stage reading it now "
|
||||
<< "fails to link (ES) or breaks the Vulkan stage interface";
|
||||
@@ -2977,17 +2817,15 @@ void main() {
|
||||
// succeeds - fail-open call sites downstream must not see a different world),
|
||||
// and the failure latch is the signal. This is the catch that took a device
|
||||
// bisect to find when the validator was off everywhere.
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
|
||||
EXPECT_GT(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
<< "an invalid optimized module must bump the validation-failure latch";
|
||||
}
|
||||
{
|
||||
// The shipping configuration: same result, no validation, latch untouched.
|
||||
SpirvValidationScope validationOff(false);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, false));
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore);
|
||||
}
|
||||
}
|
||||
@@ -3057,10 +2895,9 @@ void main() {
|
||||
ASSERT_FALSE(raw.empty());
|
||||
ASSERT_GE(CountRectImageTypes(raw), 1u) << "glslang no longer emits Dim::Rect for sampler2DRect";
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
Vector<Uint32> optimized;
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
|
||||
EXPECT_EQ(CountRectImageTypes(optimized), 0u);
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
<< "a rectangle module must leave the chain valid, not latched as a failure";
|
||||
@@ -3085,10 +2922,9 @@ void main() {
|
||||
ASSERT_TRUE(AnyLocationOnUniformStorage(raw))
|
||||
<< "glslang no longer keeps the explicit uniform location; the strip pass may be obsolete";
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
Vector<Uint32> optimized;
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized));
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, optimized, true, true));
|
||||
EXPECT_FALSE(AnyLocationOnUniformStorage(optimized));
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
<< "the stripped module must validate clean";
|
||||
@@ -3185,11 +3021,10 @@ void main() {
|
||||
<< "the fixture must reproduce the defect before the fix is asked to remove it:\n"
|
||||
<< DisassembleSpirv(raw);
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> legalized;
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
|
||||
ASSERT_FALSE(legalized.empty());
|
||||
|
||||
const String disassembly = DisassembleSpirv(legalized);
|
||||
@@ -3226,11 +3061,10 @@ void main() {
|
||||
ASSERT_TRUE(LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(raw))
|
||||
<< DisassembleSpirv(raw);
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> legalized;
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
|
||||
ASSERT_FALSE(legalized.empty());
|
||||
|
||||
const String disassembly = DisassembleSpirv(legalized);
|
||||
@@ -3270,11 +3104,10 @@ void main() {
|
||||
ASSERT_FALSE(raw.empty());
|
||||
ASSERT_TRUE(LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(raw));
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> legalized;
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
|
||||
ASSERT_FALSE(legalized.empty());
|
||||
|
||||
const String disassembly = DisassembleSpirv(legalized);
|
||||
@@ -3313,7 +3146,7 @@ void main() {
|
||||
ASSERT_FALSE(LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(raw));
|
||||
|
||||
Vector<Uint32> legalized;
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized));
|
||||
ASSERT_TRUE(ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(raw, legalized, true));
|
||||
EXPECT_EQ(legalized, raw) << "the module must not be rewritten - not even re-serialized - when "
|
||||
"nothing indexes a fragment output dynamically";
|
||||
}
|
||||
@@ -3563,11 +3396,10 @@ TEST_F(ProgramUtilTest, Lower1DArrayImagesRewritesTheTypeAndWidensTheCoordinate)
|
||||
ASSERT_EQ(Count1DArrayStorageImageTypes(spirv), 1u)
|
||||
<< "the shared chain must leave the 1D-array image for this pass to handle";
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> lowered;
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
|
||||
ASSERT_FALSE(lowered.empty());
|
||||
|
||||
EXPECT_EQ(Count1DArrayStorageImageTypes(lowered), 0u)
|
||||
@@ -3615,11 +3447,10 @@ void main() { ssb.sum = imageLoad(i0, ivec2(2, 3)).r + imageLoad(i1, ivec3(1, 1,
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(raw, spirv));
|
||||
ASSERT_EQ(Count1DArrayStorageImageTypes(spirv), 1u);
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> lowered;
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
|
||||
ASSERT_FALSE(lowered.empty());
|
||||
|
||||
EXPECT_EQ(Count1DArrayStorageImageTypes(lowered), 0u) << DisassembleSpirv(lowered);
|
||||
@@ -3649,7 +3480,7 @@ void main() { ssb.sum = imageLoad(i0, 2).r; }
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<Uint32> lowered;
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
|
||||
EXPECT_EQ(lowered, spirv) << "a non-arrayed 1D storage image must pass through byte for byte";
|
||||
|
||||
const String essl = DecompileToEssl(lowered);
|
||||
@@ -3673,7 +3504,7 @@ void main() { fragColor = texture(uTex, vUv); }
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<Uint32> lowered;
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
|
||||
EXPECT_EQ(lowered, spirv) << "a sampled 1D-array image must pass through byte for byte";
|
||||
}
|
||||
|
||||
@@ -3697,7 +3528,7 @@ void main() { ssb.sum = uint(imageSize(i0).x) + imageLoad(i0, ivec2(0, 0)).r; }
|
||||
<< "the fixture must contain the shape the pass declines";
|
||||
|
||||
Vector<Uint32> lowered;
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered));
|
||||
ASSERT_TRUE(ShaderCompiler::Lower1DArrayImagesForEssl(spirv, lowered, true));
|
||||
EXPECT_EQ(lowered, spirv) << "a declined module must be handed back untouched, not partly rewritten";
|
||||
EXPECT_EQ(Count1DArrayStorageImageTypes(lowered), 1u)
|
||||
<< "declining means the 1D-array type is still there for the driver to reject";
|
||||
@@ -3748,11 +3579,10 @@ void main() { imageStore(uni_image, ivec2(gl_GlobalInvocationID.xy), uvec4(15u,
|
||||
// Precondition: SPIRV-Cross prints no format for it, which is the ESSL the driver refuses.
|
||||
EXPECT_EQ(DecompileToEssl(spirv).find("r32ui"), String::npos);
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
|
||||
ASSERT_FALSE(baked.empty());
|
||||
EXPECT_FALSE(ShaderCompiler::DeclaresFormatlessStorageImage(baked)) << DisassembleSpirv(baked);
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
@@ -3813,7 +3643,7 @@ void main() { imageStore(uni_image, ivec2(0), uvec4(1u)); }
|
||||
|
||||
Vector<Uint32> baked;
|
||||
// Even asked to, with a format of the right component class.
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
|
||||
EXPECT_EQ(baked, spirv) << "a module with nothing format-less must pass through byte for byte";
|
||||
EXPECT_NE(DecompileToEssl(baked).find("rgba32ui"), String::npos);
|
||||
}
|
||||
@@ -3835,11 +3665,10 @@ void main() { writeIt(uni_image); }
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
ASSERT_TRUE(ShaderCompiler::DeclaresFormatlessStorageImage(spirv));
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
|
||||
EXPECT_EQ(baked, spirv) << "a shape the retype cannot follow must leave the module untouched, "
|
||||
"not partly rewritten:\n"
|
||||
<< DisassembleSpirv(baked);
|
||||
@@ -3861,18 +3690,17 @@ void main() { imageStore(uni_image, ivec2(0), vec4(1.0)); }
|
||||
GL_COMPUTE_SHADER);
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32ui}}, baked, true));
|
||||
EXPECT_EQ(baked, spirv) << "a declined module must be handed back untouched, not partly rewritten";
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore);
|
||||
|
||||
// ...and the same image with a float bind format is baked, so the decline above is about the
|
||||
// class and not about the pass refusing float images.
|
||||
Vector<Uint32> bakedFloat;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32f}}, bakedFloat));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_image", kGlR32f}}, bakedFloat, true));
|
||||
EXPECT_NE(DecompileToEssl(bakedFloat).find("r32f"), String::npos) << DisassembleSpirv(bakedFloat);
|
||||
}
|
||||
|
||||
@@ -3896,12 +3724,11 @@ void main() {
|
||||
ASSERT_EQ(CountSpirvOpcode(DisassembleSpirv(spirv), "OpTypeImage"), 1u)
|
||||
<< "the fixture must have the two images sharing one type:\n" << DisassembleSpirv(spirv);
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(
|
||||
spirv, {{"imgA", kGlR32ui}, {"imgB", kGlRgba32ui}}, baked));
|
||||
spirv, {{"imgA", kGlR32ui}, {"imgB", kGlRgba32ui}}, baked, true));
|
||||
ASSERT_FALSE(baked.empty());
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
<< "splitting the shared type must not leave a dangling or duplicate declaration:\n"
|
||||
@@ -3934,11 +3761,10 @@ void main() {
|
||||
ASSERT_EQ(CountSpirvOpcode(DisassembleSpirv(spirv), "OpTypeImage"), 2u)
|
||||
<< "the fixture needs one Unknown-format and one r32ui image type:\n" << DisassembleSpirv(spirv);
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"formatless", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"formatless", kGlR32ui}}, baked, true));
|
||||
ASSERT_FALSE(baked.empty());
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
<< "the baked image collided with the module's own r32ui image and left a duplicate type:\n"
|
||||
@@ -3963,11 +3789,10 @@ void main() {
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
ASSERT_TRUE(ShaderCompiler::DeclaresFormatlessStorageImage(spirv));
|
||||
|
||||
SpirvValidationScope validationOn(true);
|
||||
const Uint64 failuresBefore = ShaderCompiler::SpirvValidationFailureCount();
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"imgs", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"imgs", kGlR32ui}}, baked, true));
|
||||
ASSERT_FALSE(baked.empty());
|
||||
EXPECT_FALSE(ShaderCompiler::DeclaresFormatlessStorageImage(baked)) << DisassembleSpirv(baked);
|
||||
EXPECT_EQ(ShaderCompiler::SpirvValidationFailureCount(), failuresBefore)
|
||||
@@ -3993,7 +3818,7 @@ void main() { fragColor = texture(uni_sampler, vUv); }
|
||||
<< "a sampled image must not read as a format-less STORAGE image:\n" << DisassembleSpirv(spirv);
|
||||
|
||||
Vector<Uint32> baked;
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_sampler", kGlR32ui}}, baked));
|
||||
ASSERT_TRUE(ShaderCompiler::BakeImageFormatsForEssl(spirv, {{"uni_sampler", kGlR32ui}}, baked, true));
|
||||
EXPECT_EQ(baked, spirv) << "a sampled image must pass through byte for byte";
|
||||
}
|
||||
|
||||
|
||||
@@ -31,6 +31,7 @@
|
||||
#include <MG_Backend/DirectVulkan/Renderer/VulkanRenderer.h>
|
||||
#include <MG_Util/Math/HalfFloat.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
#include <MG_Util/ShaderTranspiler/CompileEnv.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderSourceProcessor.h>
|
||||
#include <MG_Util/Debug/Log.h>
|
||||
#include <MG_Util/Types.h>
|
||||
@@ -710,6 +711,37 @@ TEST(DirectVulkanSanity, AdvertisesSubgroupOnlyWhenVulkanReportsUsableSupport) {
|
||||
EXPECT_TRUE(backend.GetDynamicParameters().SubgroupQuadOperationsInAllStages);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, CapabilityRefreshInvalidatesTheCachedCompileEnvironment) {
|
||||
using namespace MobileGL;
|
||||
|
||||
auto previousContext = Move(MG_State::pGLContext);
|
||||
auto previousBackend = Move(MG_Backend::pActiveBackendObject);
|
||||
MG_State::pGLContext = MakeUnique<MG_State::GLState::GLContext>();
|
||||
|
||||
auto backend = MakeUnique<MG_Backend::DirectVulkan::BackendObject_DirectVulkan>();
|
||||
auto* backendPtr = backend.get();
|
||||
MG_Backend::pActiveBackendObject = Move(backend);
|
||||
|
||||
const auto before = MG_State::pGLContext->GetCompileEnv();
|
||||
EXPECT_EQ(before->params.SubgroupSize, 0u);
|
||||
|
||||
MG_External::VulkanCapabilities caps;
|
||||
caps.SupportsShaderSubgroup = true;
|
||||
caps.SubgroupSize = 8;
|
||||
caps.SubgroupSupportedStages = VK_SHADER_STAGE_COMPUTE_BIT;
|
||||
caps.SubgroupSupportedOperations = VK_SUBGROUP_FEATURE_BASIC_BIT | VK_SUBGROUP_FEATURE_ARITHMETIC_BIT;
|
||||
backendPtr->ApplyVulkanCapabilitiesForTesting(caps);
|
||||
|
||||
const auto after = MG_State::pGLContext->GetCompileEnv();
|
||||
EXPECT_NE(after.get(), before.get());
|
||||
EXPECT_NE(after->fingerprint, before->fingerprint);
|
||||
EXPECT_EQ(after->backend, BackendType::DirectVulkan);
|
||||
EXPECT_EQ(after->params.SubgroupSize, 8u);
|
||||
|
||||
MG_Backend::pActiveBackendObject = Move(previousBackend);
|
||||
MG_State::pGLContext = Move(previousContext);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, KeepsOptionalGpuShaderInt64BranchForVoxyQuadDecode) {
|
||||
using namespace MobileGL;
|
||||
|
||||
|
||||
@@ -154,7 +154,6 @@ class DemoteFloat64Test : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
MobileGL::Initialize();
|
||||
ShaderCompiler::SetSpirvValidationEnabled(true);
|
||||
m_validationFailuresAtStart = ShaderCompiler::SpirvValidationFailureCount();
|
||||
}
|
||||
|
||||
@@ -175,7 +174,7 @@ TEST_F(DemoteFloat64Test, DemotesEveryWidthAndDropsTheCapability) {
|
||||
ASSERT_TRUE(DeclaresFloat64Capability(input));
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
|
||||
EXPECT_EQ(CountFloatTypesOfWidth(output, 64), 0u) << Disassemble(output);
|
||||
// And exactly one 32-bit float type survives: the merge has to happen, or spirv-val rejects
|
||||
@@ -210,7 +209,7 @@ void main() {
|
||||
<< Disassemble(input);
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
|
||||
// std140 for the demoted members: float at 4, vec2 at 8, vec3 at 16 (aligned like a vec4),
|
||||
// vec4 at 32, mat4 at 48 with a 16-byte column stride, the array at 112 with the std140
|
||||
@@ -241,7 +240,7 @@ void main() {
|
||||
EXPECT_EQ(CollectOffsetsOf(input, "Ssbo"), (Vector<Uint32>{0, 32, 64})) << Disassemble(input);
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
|
||||
// std430, so the array packs at its element size rather than being rounded to 16: float at 0,
|
||||
// vec4 at 16, float[4] at 32 with a 4-byte stride. A storage block must NOT come out std140,
|
||||
@@ -273,7 +272,7 @@ void main() {
|
||||
ASSERT_FALSE(before.empty());
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
|
||||
// Only the block that actually narrowed is re-laid-out. Touching the other one would be
|
||||
// churn at best, and a disagreement with glslang's own layout at worst.
|
||||
@@ -287,7 +286,7 @@ TEST_F(DemoteFloat64Test, FoldsTheConversionsThatBecameIdentities) {
|
||||
ASSERT_GT(CountFConverts(input), 0u) << "the fixture no longer converts between the two widths";
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
|
||||
// SPIR-V requires the two component widths of an OpFConvert to differ, so every one of them
|
||||
// has to be gone: both sides are 32 bits now.
|
||||
@@ -307,7 +306,7 @@ void main() {
|
||||
ASSERT_FALSE(input.empty());
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
|
||||
// A 64-bit literal is two words wide and a 32-bit one is a single word, so a constant left
|
||||
// unconverted is not merely imprecise - it is an unparseable instruction. Disassembling both
|
||||
@@ -326,7 +325,7 @@ void main() { gl_Position = inPos; }
|
||||
ASSERT_FALSE(input.empty());
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
// The pass reports SuccessWithoutChange here, and SPIRV-Tools asserts (in assert-enabled
|
||||
// builds) that such a run round-trips byte-identically.
|
||||
EXPECT_EQ(output, input);
|
||||
@@ -351,7 +350,7 @@ void main() {
|
||||
ASSERT_EQ(CountFloatTypesOfWidth(input, 64), 1u) << Disassemble(input);
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(input, output, true));
|
||||
EXPECT_EQ(output, input) << Disassemble(output);
|
||||
EXPECT_TRUE(ShaderCompiler::ModuleDeclaresFloat64(output));
|
||||
}
|
||||
@@ -362,7 +361,7 @@ TEST_F(DemoteFloat64Test, ModuleDeclaresFloat64AnswersBothWays) {
|
||||
EXPECT_TRUE(ShaderCompiler::ModuleDeclaresFloat64(wide));
|
||||
|
||||
Vector<Uint32> demoted;
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(wide, demoted));
|
||||
ASSERT_TRUE(ShaderCompiler::DemoteFloat64ToFloat32(wide, demoted, true));
|
||||
EXPECT_FALSE(ShaderCompiler::ModuleDeclaresFloat64(demoted));
|
||||
|
||||
EXPECT_FALSE(ShaderCompiler::ModuleDeclaresFloat64({}));
|
||||
@@ -375,7 +374,7 @@ TEST_F(DemoteFloat64Test, TheSharedChainDemotesToo) {
|
||||
ASSERT_FALSE(input.empty());
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
|
||||
EXPECT_FALSE(ShaderCompiler::ModuleDeclaresFloat64(output)) << Disassemble(output);
|
||||
}
|
||||
|
||||
@@ -459,7 +458,7 @@ TEST_P(DemoteFloat64EsslTest, TheDemotedModuleCanBeEmittedAsEssl) {
|
||||
ASSERT_FALSE(input.empty());
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output));
|
||||
ASSERT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
|
||||
|
||||
SpvcSession session(output, SessionUsageBit::Transpile);
|
||||
spvc_compiler_options options;
|
||||
@@ -482,7 +481,7 @@ TEST_P(DemoteFloat64EsslTest, TheDemotedModuleCanBeEmittedAsEssl) {
|
||||
TEST_F(DemoteFloat64Test, RejectsGarbageInput) {
|
||||
const Vector<Uint32> notSpirv{0xdeadbeefu, 0u, 0u, 0u, 0u};
|
||||
Vector<Uint32> output;
|
||||
EXPECT_FALSE(ShaderCompiler::DemoteFloat64ToFloat32(notSpirv, output));
|
||||
EXPECT_FALSE(ShaderCompiler::DemoteFloat64ToFloat32(notSpirv, output, true));
|
||||
}
|
||||
|
||||
// EliminateFloatEqualsZeroPass turns a comparison against 0.0 into an epsilon test, a
|
||||
@@ -502,7 +501,7 @@ namespace {
|
||||
EXPECT_FALSE(input.empty());
|
||||
if (input.empty()) return false;
|
||||
Vector<Uint32> output;
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output));
|
||||
EXPECT_TRUE(ShaderCompiler::SanitizeAndOptimizeBinary(input, output, true, true));
|
||||
return Disassemble(output).find("FAbs") != String::npos;
|
||||
}
|
||||
|
||||
|
||||
@@ -98,7 +98,6 @@ class FlattenXfbInterfaceBlocksTest : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
MobileGL::Initialize();
|
||||
ShaderCompiler::SetSpirvValidationEnabled(true);
|
||||
m_validationFailuresAtStart = ShaderCompiler::SpirvValidationFailureCount();
|
||||
}
|
||||
|
||||
@@ -116,7 +115,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, FlattensACapturedBlockIntoOneVariablePerMe
|
||||
|
||||
std::set<String> flattened;
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output));
|
||||
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output, true));
|
||||
ASSERT_FALSE(output.empty());
|
||||
EXPECT_EQ(flattened, (std::set<String>{"StageData"}));
|
||||
|
||||
@@ -145,7 +144,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, TheEmittedDeclarationIsAPlainArrayNotABloc
|
||||
|
||||
std::set<String> flattened;
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output));
|
||||
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output, true));
|
||||
|
||||
const String after = Transpile(output);
|
||||
EXPECT_NE(after.find("StageData_attrib[16]"), String::npos) << after;
|
||||
@@ -162,7 +161,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, GivesEachMemberItsOwnConsecutiveLocations)
|
||||
|
||||
std::set<String> flattened;
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output));
|
||||
ASSERT_TRUE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"StageData"}, flattened, output, true));
|
||||
ASSERT_FALSE(output.empty());
|
||||
|
||||
const String dis = Disassemble(output);
|
||||
@@ -184,7 +183,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, LeavesABlockNoCaptureNamesAlone) {
|
||||
std::set<String> flattened;
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(
|
||||
ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"SomeOtherBlock"}, flattened, output));
|
||||
ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {"SomeOtherBlock"}, flattened, output, true));
|
||||
EXPECT_TRUE(flattened.empty());
|
||||
|
||||
const String after = Transpile(output);
|
||||
@@ -200,7 +199,7 @@ TEST_F(FlattenXfbInterfaceBlocksTest, DeclinesAnEmptyRequestWithoutRewriting) {
|
||||
|
||||
std::set<String> flattened;
|
||||
Vector<Uint32> output;
|
||||
EXPECT_FALSE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {}, flattened, output));
|
||||
EXPECT_FALSE(ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(input, {}, flattened, output, true));
|
||||
EXPECT_TRUE(flattened.empty());
|
||||
EXPECT_TRUE(output.empty());
|
||||
}
|
||||
|
||||
@@ -33,13 +33,13 @@ namespace MobileGL {
|
||||
|
||||
std::string GetThreadName() {
|
||||
char buffer[64] = {0};
|
||||
#if defined(_WIN32) && !defined(__MINGW32__)
|
||||
#if defined(_WIN32)
|
||||
PWSTR desc = nullptr;
|
||||
if (SUCCEEDED(GetThreadDescription(GetCurrentThread(), &desc))) {
|
||||
WideCharToMultiByte(CP_UTF8, 0, desc, -1, buffer, sizeof(buffer), nullptr, nullptr);
|
||||
LocalFree(desc);
|
||||
}
|
||||
#elif defined(__ANDROID__) || defined(__linux__) || defined(__APPLE__) || defined(__MINGW32__)
|
||||
#elif defined(__ANDROID__) || defined(__linux__) || defined(__APPLE__)
|
||||
pthread_getname_np(pthread_self(), buffer, sizeof(buffer));
|
||||
#endif
|
||||
return buffer[0] ? buffer : "UnknownThread";
|
||||
|
||||
@@ -38,7 +38,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
|
||||
HashBytes(state, env.advertisedExtensions.data(),
|
||||
env.advertisedExtensions.size() * sizeof(GLExtension));
|
||||
}
|
||||
HashValue(state, env.subgroupPrefixScanQuirk);
|
||||
return state;
|
||||
}
|
||||
|
||||
@@ -73,8 +72,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
|
||||
kFrontendMaxComputeWorkGroupInvocations)
|
||||
: kFrontendMaxComputeWorkGroupInvocations;
|
||||
|
||||
env->subgroupPrefixScanQuirk = MG_Config::Features.SubgroupPrefixScanQuirk;
|
||||
|
||||
env->fingerprint = ComputeCompileEnvFingerprint(*env);
|
||||
return env;
|
||||
}
|
||||
@@ -84,7 +81,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
|
||||
// computed, and this must not run before MG_Config is loaded.
|
||||
static const SharedPtr<const CompileEnv> kDefault = [] {
|
||||
auto env = MakeShared<CompileEnv>();
|
||||
env->subgroupPrefixScanQuirk = MG_Config::Features.SubgroupPrefixScanQuirk;
|
||||
env->fingerprint = ComputeCompileEnvFingerprint(*env);
|
||||
return SharedPtr<const CompileEnv>(Move(env));
|
||||
}();
|
||||
|
||||
@@ -12,9 +12,9 @@
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
|
||||
namespace MobileGL::MG_Util::ShaderTranspiler {
|
||||
// Everything the shader compile/link pipeline reads from OUTSIDE its own (stage, source)
|
||||
// inputs: backend identity, backend limits, the advertised extension list, and the one
|
||||
// config quirk the source rewriter branches on.
|
||||
// everything outside (stage, source) this reads - advertised extensions and backend limits -
|
||||
// so the transformation is a pure function of its three arguments and can run on a worker
|
||||
// thread.
|
||||
//
|
||||
// Why it exists (P1): every one of those reads is a reach-back into
|
||||
// MG_Backend::pActiveBackendObject / gBackendFunctionsTable, and one of them
|
||||
@@ -46,9 +46,6 @@ namespace MobileGL::MG_Util::ShaderTranspiler {
|
||||
MG_Backend::DynamicBackendParameters params{}; // by value, never by reference
|
||||
Vector<GLExtension> advertisedExtensions;
|
||||
|
||||
// --- config the source rewriter branches on ---
|
||||
MG_Config::QuirkOverride subgroupPrefixScanQuirk = MG_Config::QuirkOverride::Auto;
|
||||
|
||||
Uint64 fingerprint = 0; // set by CaptureCompileEnv()
|
||||
|
||||
Bool HasBackend() const { return backend != BackendType::Unknown; }
|
||||
|
||||
@@ -369,12 +369,6 @@ namespace MobileGL {
|
||||
return allSpirv;
|
||||
}
|
||||
|
||||
// -1 unresolved, 0 off, 1 on. Resolved once from MOBILEGL_VALIDATE_SPIRV on first
|
||||
// use. A live getenv rather than an MG_Config::Features field, for the same reason
|
||||
// Config.h already exempts MOBILEGL_LOG_FILE_PATH: suites like SpirvPassTest never
|
||||
// run MobileGL::Initialize(), and every Initialize() re-runs MG_ConfigLoader::Init,
|
||||
// which would clobber a programmatic override stored in the feature table.
|
||||
static std::atomic<int> g_validateSpirv{-1};
|
||||
// Total validation failures observed this process. This latch - not the wrappers'
|
||||
// return values - is the test-lane signal: validation must never change what a
|
||||
// wrapper returns, or the validating lanes would render differently from the
|
||||
@@ -383,28 +377,6 @@ namespace MobileGL {
|
||||
static std::atomic<Uint64> g_spirvValidationFailures{0};
|
||||
|
||||
namespace {
|
||||
// Test lanes (desktop/CI/WSL) validate by default; device builds do not -
|
||||
// validation costs real time per module, and on device the driver is the
|
||||
// final validator anyway. MOBILEGL_VALIDATE_SPIRV overrides in either
|
||||
// direction, using the ConfigLoader truthy rule.
|
||||
constexpr bool kValidateSpirvDefault =
|
||||
#if defined(__ANDROID__)
|
||||
false;
|
||||
#else
|
||||
true;
|
||||
#endif
|
||||
|
||||
bool IsTruthySpirvEnvValue(const char* value) {
|
||||
if (value == nullptr || value[0] == '\0') {
|
||||
return false;
|
||||
}
|
||||
String lowered(value);
|
||||
for (auto& c : lowered) {
|
||||
c = static_cast<char>(std::tolower(static_cast<unsigned char>(c)));
|
||||
}
|
||||
return lowered != "0" && lowered != "false";
|
||||
}
|
||||
|
||||
// spirv-tools' validator lazily constructs function-local static tables on
|
||||
// its first run, which on this codebase happens on a ShaderCompilePool
|
||||
// worker. Function-local statics are destroyed in reverse construction
|
||||
@@ -445,11 +417,6 @@ namespace MobileGL {
|
||||
tools.Validate(warmup);
|
||||
}
|
||||
std::atexit(+[] {
|
||||
// Flip validation off first: a validator table this warmup does
|
||||
// not know about (a future spirv-tools bump) would still be
|
||||
// destroyed before this handler, and workers must stop entering
|
||||
// Validate before the drain waits for them.
|
||||
g_validateSpirv.store(0, std::memory_order_release);
|
||||
Async::ShaderCompilePool::StopAndDrainProcessPoolAtExit();
|
||||
});
|
||||
});
|
||||
@@ -478,10 +445,10 @@ namespace MobileGL {
|
||||
// Validation is decoupled from control flow on purpose: a failure logs and
|
||||
// bumps the latch, and the caller proceeds exactly as the shipping (non-
|
||||
// validating) configuration would. Tests assert on the latch delta.
|
||||
void ValidateOrLatch(const char* site, const Vector<Uint32>& binary) {
|
||||
if (!ShaderCompiler::SpirvValidationEnabled()) {
|
||||
return;
|
||||
}
|
||||
void ValidateOrLatch(const char* site, const Vector<Uint32>& binary,
|
||||
const bool enableSpirvValidation) {
|
||||
if (!enableSpirvValidation) return;
|
||||
PinValidatorTablesForProcessExit();
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
tools.SetMessageConsumer(MakeSpirvMessageConsumer(site));
|
||||
if (!tools.Validate(binary)) {
|
||||
@@ -502,39 +469,23 @@ namespace MobileGL {
|
||||
// spirv-tools drops pass diagnostics on the floor.
|
||||
bool RunOptimizerChecked(const char* site, spvtools::Optimizer& optimizer,
|
||||
const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary, const bool validateOutput,
|
||||
const bool enableSpirvValidation) {
|
||||
spvtools::OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
optimizer.SetMessageConsumer(MakeSpirvMessageConsumer(site));
|
||||
if (!optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options)) {
|
||||
return false;
|
||||
}
|
||||
ValidateOrLatch(site, outputBinary);
|
||||
if (validateOutput) {
|
||||
ValidateOrLatch(site, outputBinary, enableSpirvValidation);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool ShaderCompiler::SpirvValidationEnabled() {
|
||||
int state = g_validateSpirv.load(std::memory_order_acquire);
|
||||
if (state < 0) {
|
||||
const char* env = std::getenv("MOBILEGL_VALIDATE_SPIRV");
|
||||
const bool resolved = env != nullptr ? IsTruthySpirvEnvValue(env) : kValidateSpirvDefault;
|
||||
int expected = -1;
|
||||
g_validateSpirv.compare_exchange_strong(expected, resolved ? 1 : 0,
|
||||
std::memory_order_acq_rel);
|
||||
state = g_validateSpirv.load(std::memory_order_acquire);
|
||||
if (state == 1) {
|
||||
PinValidatorTablesForProcessExit();
|
||||
}
|
||||
}
|
||||
return state == 1;
|
||||
}
|
||||
|
||||
void ShaderCompiler::SetSpirvValidationEnabled(bool enabled) {
|
||||
g_validateSpirv.store(enabled ? 1 : 0, std::memory_order_release);
|
||||
if (enabled) {
|
||||
PinValidatorTablesForProcessExit();
|
||||
}
|
||||
void ShaderCompiler::PrepareSpirvValidation() {
|
||||
PinValidatorTablesForProcessExit();
|
||||
}
|
||||
|
||||
Uint64 ShaderCompiler::NoteSpirvValidationFailure() {
|
||||
@@ -604,16 +555,19 @@ namespace MobileGL {
|
||||
}
|
||||
|
||||
bool ShaderCompiler::DemoteFloat64ToFloat32(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(DemoteFloat64Pass::CreateDemoteFloat64Pass());
|
||||
|
||||
return RunOptimizerChecked("DemoteFloat64ToFloat32", optimizer, inputBinary, outputBinary);
|
||||
return RunOptimizerChecked("DemoteFloat64ToFloat32", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::SanitizeAndOptimizeBinary(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool validateOutput,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
|
||||
@@ -663,38 +617,41 @@ namespace MobileGL {
|
||||
optimizer.RegisterPass(DemoteFloat64Pass::CreateDemoteFloat64Pass());
|
||||
|
||||
return RunOptimizerChecked("SanitizeAndOptimizeBinary", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, validateOutput, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::LowerDrawParametersForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(LowerDrawParametersPass::CreateLowerDrawParametersPass());
|
||||
|
||||
return RunOptimizerChecked("LowerDrawParametersForEssl", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::SplitArrayVertexInputsForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(SplitArrayVertexInputsPass::CreateSplitArrayVertexInputsPass());
|
||||
|
||||
return RunOptimizerChecked("SplitArrayVertexInputsForEssl", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::BakeImageFormatsForEssl(const Vector<Uint32>& inputBinary,
|
||||
const UnorderedMap<String, Uint>& glFormatByName,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
if (glFormatByName.empty()) return false;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(BakeImageFormatsPass::CreateBakeImageFormatsPass(glFormatByName));
|
||||
|
||||
return RunOptimizerChecked("BakeImageFormatsForEssl", optimizer, inputBinary, outputBinary);
|
||||
return RunOptimizerChecked("BakeImageFormatsForEssl", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::DeclaresFormatlessStorageImage(const Vector<Uint32>& binary) {
|
||||
@@ -718,7 +675,8 @@ namespace MobileGL {
|
||||
bool ShaderCompiler::FlattenXfbInterfaceBlocksForEssl(const Vector<Uint32>& inputBinary,
|
||||
const std::set<String>& blockNames,
|
||||
std::set<String>& flattenedBlockNames,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
if (blockNames.empty()) return false;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
@@ -726,7 +684,7 @@ namespace MobileGL {
|
||||
blockNames, &flattenedBlockNames));
|
||||
|
||||
return RunOptimizerChecked("FlattenXfbInterfaceBlocksForEssl", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::RewriteXfbCaptureNameForFlattenedBlock(
|
||||
@@ -736,48 +694,53 @@ namespace MobileGL {
|
||||
}
|
||||
|
||||
bool ShaderCompiler::PackDoubleVertexInputsForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(PackDoubleVertexInputsPass::CreatePackDoubleVertexInputsPass());
|
||||
|
||||
return RunOptimizerChecked("PackDoubleVertexInputsForVulkan", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::StripUboMemberRelaxedPrecisionForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(
|
||||
StripUboMemberRelaxedPrecisionPass::CreateStripUboMemberRelaxedPrecisionPass());
|
||||
|
||||
return RunOptimizerChecked("StripUboMemberRelaxedPrecisionForEssl", optimizer,
|
||||
inputBinary, outputBinary);
|
||||
inputBinary, outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(StripNoPerspectivePass::CreateStripNoPerspectivePass());
|
||||
|
||||
return RunOptimizerChecked("StripNoPerspectiveForEssl", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(EmulateNoPerspectivePass::CreateEmulateNoPerspectivePass());
|
||||
|
||||
return RunOptimizerChecked("EmulateNoPerspectiveForEssl", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
|
||||
// Detection gates everything: a module with no dynamically indexed fragment
|
||||
@@ -810,7 +773,7 @@ namespace MobileGL {
|
||||
|
||||
Vector<uint32_t> folded;
|
||||
if (!RunOptimizerChecked("LegalizeFragmentOutputIndexingForEssl.fold", folder, inputBinary,
|
||||
folded) ||
|
||||
folded, true, enableSpirvValidation) ||
|
||||
folded.empty()) {
|
||||
// Fail open onto the fallback rather than onto the illegal module.
|
||||
folded = inputBinary;
|
||||
@@ -829,7 +792,7 @@ namespace MobileGL {
|
||||
lowerer.RegisterPass(CreateAggressiveDCEPass(false));
|
||||
|
||||
if (!RunOptimizerChecked("LegalizeFragmentOutputIndexingForEssl.lower", lowerer, folded,
|
||||
outputBinary) ||
|
||||
outputBinary, true, enableSpirvValidation) ||
|
||||
outputBinary.empty()) {
|
||||
outputBinary = folded;
|
||||
return true;
|
||||
@@ -846,16 +809,17 @@ namespace MobileGL {
|
||||
}
|
||||
|
||||
bool ShaderCompiler::LowerRectImages(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(NormalizeRectCoordinatesPass::CreateNormalizeRectCoordinatesPass());
|
||||
|
||||
return RunOptimizerChecked("LowerRectImages", optimizer, inputBinary, outputBinary);
|
||||
return RunOptimizerChecked("LowerRectImages", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::Lower1DArrayImagesForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
|
||||
// Declined rather than half-translated: after the rewrite the image is a 2D
|
||||
@@ -897,40 +861,41 @@ namespace MobileGL {
|
||||
// second Shader. Deduplicating afterwards collapses all three at once.
|
||||
optimizer.RegisterPass(CreateRemoveDuplicatesPass());
|
||||
|
||||
return RunOptimizerChecked("Lower1DArrayImagesForEssl", optimizer, inputBinary, outputBinary);
|
||||
return RunOptimizerChecked("Lower1DArrayImagesForEssl", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(RebaseInstanceIndexPass::CreateRebaseInstanceIndexPass());
|
||||
|
||||
return RunOptimizerChecked("RebaseInstanceIndexForVulkan", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::ZeroBaseVertexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(ZeroBaseVertexPass::CreateZeroBaseVertexPass());
|
||||
|
||||
return RunOptimizerChecked("ZeroBaseVertexForVulkan", optimizer, inputBinary, outputBinary);
|
||||
return RunOptimizerChecked("ZeroBaseVertexForVulkan", optimizer, inputBinary, outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
Vector<uint32_t>& outputBinary, const bool enableSpirvValidation) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(DecoratePositionInvariantPass::CreateDecoratePositionInvariantPass());
|
||||
|
||||
return RunOptimizerChecked("DecoratePositionInvariantForVulkan", optimizer, inputBinary,
|
||||
outputBinary);
|
||||
outputBinary, true, enableSpirvValidation);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(
|
||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary) {
|
||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
|
||||
const bool enableSpirvValidation) {
|
||||
constexpr SizeT kSpirvHeaderWordCount = 5;
|
||||
outputBinary.clear();
|
||||
if (inputBinary.size() < kSpirvHeaderWordCount || inputBinary[0] != spv::MagicNumber) {
|
||||
@@ -1058,7 +1023,8 @@ namespace MobileGL {
|
||||
addedCapabilities.begin(), addedCapabilities.end());
|
||||
// Hand-rolled word walk, so no Optimizer wrapper ever sees this rewrite;
|
||||
// check the modified module explicitly in validating lanes.
|
||||
ValidateOrLatch("UseUnformattedFloatStorageImagesForVulkan", outputBinary);
|
||||
ValidateOrLatch("UseUnformattedFloatStorageImagesForVulkan", outputBinary,
|
||||
enableSpirvValidation);
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -23,19 +23,23 @@ namespace MobileGL {
|
||||
static Result<SharedPtr<glslang::TProgram>> LinkProgram(const ProgramAttrib& attrib);
|
||||
static Result<Vector<Vector<unsigned>>> GetSpirvBinaryFromProgram(const ProgramBinaryAttrib& attrib);
|
||||
static bool SanitizeAndOptimizeBinary(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool validateOutput = true,
|
||||
bool enableSpirvValidation = false);
|
||||
// Demotes DrawIndex/BaseInstance/BaseVertex builtins to plain Private globals
|
||||
// (mg_DrawID/mg_BaseInstance/mg_BaseVertex) so SPIRV-Cross can emit ESSL.
|
||||
// Only for backends without native draw-parameter support (DirectGLES).
|
||||
static bool LowerDrawParametersForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Replaces an ARRAY vertex input with one input per element at consecutive
|
||||
// locations, seeding a Private copy of the array so indexed reads still work.
|
||||
// GLSL ES has no array vertex inputs and SPIRV-Cross refuses the whole module
|
||||
// rather than emulating them, so without this the stage never reaches the
|
||||
// driver. Only for the DirectGLES transpile path.
|
||||
static bool SplitArrayVertexInputsForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Replaces the named interface BLOCKS with one variable per member, named
|
||||
// "<Block>_<member>", shadowing the block itself so the body is untouched. The
|
||||
// Adreno ES driver silently captures NOTHING for a transform-feedback varying
|
||||
@@ -46,7 +50,8 @@ namespace MobileGL {
|
||||
static bool FlattenXfbInterfaceBlocksForEssl(const Vector<Uint32>& inputBinary,
|
||||
const std::set<String>& blockNames,
|
||||
std::set<String>& flattenedBlockNames,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// The capture request "StageData.attrib[0]" as the pass above renamed it,
|
||||
// "StageData_attrib[0]", or false when it does not name a member of a block
|
||||
// that was flattened.
|
||||
@@ -58,17 +63,20 @@ namespace MobileGL {
|
||||
// drivers reject cross-stage uniform blocks whose member precisions differ.
|
||||
// Only for the DirectGLES transpile path.
|
||||
static bool StripUboMemberRelaxedPrecisionForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Removes NoPerspective decorations so SPIRV-Cross emits plain (smooth) ESSL varyings.
|
||||
// DirectGLES fallback only, for devices lacking GL_NV_shader_noperspective_interpolation
|
||||
// (SPIRV-Cross would otherwise require that extension and the driver would reject it).
|
||||
static bool StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Emulates noperspective (screen-linear) interpolation via gl_Position.w / gl_FragCoord.w
|
||||
// so no NV extension is needed; strips what it cannot emulate. DirectGLES fallback for
|
||||
// devices lacking GL_NV_shader_noperspective_interpolation. See EmulateNoPerspectivePass.
|
||||
static bool EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Makes every index into a fragment-output array a constant integral
|
||||
// expression, which is what GLSL ES requires and SPIR-V does not. Runs the
|
||||
// stock folding chain first (loop unrolling folds the loop-derived indices
|
||||
@@ -79,7 +87,8 @@ namespace MobileGL {
|
||||
// dynamically, which is every shader but a handful.
|
||||
// See LegalizeFragmentOutputIndexPass.
|
||||
static bool LegalizeFragmentOutputIndexingForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Rebases loads of the InstanceIndex builtin to (InstanceIndex - BaseInstance) so
|
||||
// shaders see GL's zero-based gl_InstanceID. Vertex shaders only; DirectVulkan
|
||||
// backend only (glslang's relaxed mode aliases gl_InstanceID to gl_InstanceIndex,
|
||||
@@ -88,7 +97,8 @@ namespace MobileGL {
|
||||
// divides the coordinate of each normalized-coordinate lookup by the texture
|
||||
// size and rewrites the image type to 2D. See NormalizeRectCoordinatesPass for
|
||||
// what it declines and why.
|
||||
static bool LowerRectImages(const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary);
|
||||
static bool LowerRectImages(const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// GL_TEXTURE_1D_ARRAY storage images rewritten to the 2D-array shape the texture
|
||||
// is actually stored in on ES, with the layer moved from the coordinate's second
|
||||
// component to its third. DirectGLES transpile path only - Vulkan binds a real
|
||||
@@ -96,7 +106,8 @@ namespace MobileGL {
|
||||
// through untouched when the module declares no such image, which is every shader
|
||||
// but a handful. See Lower1DArrayImagesPass for what it declines and why.
|
||||
static bool Lower1DArrayImagesForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Gives each format-less storage image the format bound to its image unit, so
|
||||
// the emitted ESSL can carry the format layout qualifier GLSL ES requires of
|
||||
// every image and desktop GLSL lets a writeonly declaration omit. `glFormatByName`
|
||||
@@ -105,7 +116,8 @@ namespace MobileGL {
|
||||
// natively. See BakeImageFormatsPass for what it declines and why.
|
||||
static bool BakeImageFormatsForEssl(const Vector<Uint32>& inputBinary,
|
||||
const UnorderedMap<String, Uint>& glFormatByName,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Whether the module declares a storage image with no format qualifier at all,
|
||||
// i.e. whether BakeImageFormatsForEssl could change anything. One module parse,
|
||||
// so the ~every shader that declares none pays no optimizer run.
|
||||
@@ -124,27 +136,31 @@ namespace MobileGL {
|
||||
// emitted text instead.
|
||||
static bool SpirvCrossCanPrintEsslImageFormat(Uint glInternalFormat);
|
||||
static bool RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Builds the non-indexed-draw variant of a vertex shader: every gl_BaseVertex
|
||||
// read becomes zero, which is what GL defines for a command carrying no
|
||||
// baseVertex parameter while Vulkan's builtin would report firstVertex.
|
||||
// See ZeroBaseVertexPass.
|
||||
static bool ZeroBaseVertexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair
|
||||
// (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no
|
||||
// VK_FORMAT_R64*_SFLOAT is needed - lavapipe advertises none of them for vertex
|
||||
// buffers. Vertex stage, DirectVulkan only; pairs with the Float64 case in
|
||||
// VertexInputStateFactory::ToVkVertexFormat.
|
||||
static bool PackDoubleVertexInputsForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Adds the Invariant decoration to every Position builtin output. GL apps
|
||||
// routinely rely on cross-program position invariance for multi-pass
|
||||
// equality depth tests (e.g. GEQUAL re-draws of the same geometry), and
|
||||
// mobile drivers that optimize per-pipeline break that without the
|
||||
// decoration. DirectVulkan only.
|
||||
static bool DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Replaces the declared format of float storage images with Unknown and adds the
|
||||
// matching SPIR-V capabilities. DirectVulkan uses this only when both Vulkan
|
||||
// shaderStorageImage*WithoutFormat features are enabled, allowing the
|
||||
@@ -152,13 +168,15 @@ namespace MobileGL {
|
||||
// storage images deliberately keep their declared format for GL-compatible bit
|
||||
// reinterpretation paths (for example, R32F storage accessed as r32ui).
|
||||
static bool UseUnformattedFloatStorageImagesForVulkan(
|
||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary);
|
||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
// Rewrites every 64-bit float in the module to a 32-bit one, preserving every
|
||||
// block offset and stride exactly (see DemoteFloat64Pass). Already part of
|
||||
// SanitizeAndOptimizeBinary, which is where production reaches it; exposed
|
||||
// separately so a test can drive the demotion on its own.
|
||||
static bool DemoteFloat64ToFloat32(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
Vector<uint32_t>& outputBinary,
|
||||
bool enableSpirvValidation = false);
|
||||
static Result<String> DecompileShader(SpvcSession& session);
|
||||
|
||||
// Parses one trivial shader in each configuration the production path can
|
||||
@@ -185,18 +203,16 @@ namespace MobileGL {
|
||||
// no way left to warm it.
|
||||
static void ResetPrewarmLatch();
|
||||
|
||||
// Test-environment SPIR-V validation. When enabled, every Optimizer wrapper
|
||||
// in this file validates its OUTPUT binary - the bytes a driver can actually
|
||||
// receive - and a failure logs the VUID (via MGLOG_I; see the consumer for
|
||||
// why not MGLOG_E) and bumps the failure latch below WITHOUT changing the
|
||||
// wrapper's return value: control flow must stay identical between the
|
||||
// validating and shipping configurations, or fail-open call sites would make
|
||||
// the two render differently. Resolved lazily from MOBILEGL_VALIDATE_SPIRV;
|
||||
// defaults on for desktop/CI/WSL builds and off for device (__ANDROID__)
|
||||
// builds. The setter wins over the environment and is safe to call from test
|
||||
// fixtures at any time.
|
||||
static bool SpirvValidationEnabled();
|
||||
static void SetSpirvValidationEnabled(bool enabled);
|
||||
// Validation is an explicit immutable option of each compiler operation. The
|
||||
// program-link task snapshots MOBILEGL_ENABLE_SPIRV_VALIDATION before it can run
|
||||
// on a worker; standalone callers pass true directly. A failure logs the VUID and
|
||||
// bumps the latch below WITHOUT changing a wrapper's return value, so validating
|
||||
// and shipping configurations preserve identical rendering control flow.
|
||||
|
||||
// Makes validator table lifetime safe before an external final-module validator
|
||||
// runs. This has no configuration state; callers invoke it only for an enabled
|
||||
// task-local validation option.
|
||||
static void PrepareSpirvValidation();
|
||||
|
||||
// The test-lane enforcement signal: total validation failures observed this
|
||||
// process. Tests snapshot it, run the operation under scrutiny, and assert
|
||||
|
||||
@@ -23,6 +23,7 @@
|
||||
namespace {
|
||||
using MobileGL::SizeT;
|
||||
using MobileGL::String;
|
||||
using MobileGL::Uint32;
|
||||
using MobileGL::Vector;
|
||||
|
||||
bool IsIdentifierChar(char ch) {
|
||||
@@ -181,331 +182,6 @@ namespace {
|
||||
return std::all_of(token.text.begin() + 1, token.text.end(), IsIdentifierChar);
|
||||
}
|
||||
|
||||
class TokenCursor {
|
||||
public:
|
||||
TokenCursor(const Vector<CodeToken>& tokens, SizeT position) : m_tokens(tokens), m_position(position) {}
|
||||
|
||||
bool Consume(const char* expected) {
|
||||
if (m_position >= m_tokens.size() || m_tokens[m_position].text != expected) {
|
||||
return false;
|
||||
}
|
||||
++m_position;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ConsumeAnyIdentifier(String& identifier) {
|
||||
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position])) {
|
||||
return false;
|
||||
}
|
||||
identifier = m_tokens[m_position++].text;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ConsumeAnyIdentifier() {
|
||||
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position])) {
|
||||
return false;
|
||||
}
|
||||
++m_position;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ConsumeIdentifier(const String& expected) {
|
||||
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position]) ||
|
||||
m_tokens[m_position].text != expected) {
|
||||
return false;
|
||||
}
|
||||
++m_position;
|
||||
return true;
|
||||
}
|
||||
|
||||
SizeT Position() const { return m_position; }
|
||||
|
||||
private:
|
||||
const Vector<CodeToken>& m_tokens;
|
||||
SizeT m_position;
|
||||
};
|
||||
|
||||
SizeT CountToken(const Vector<CodeToken>& tokens, const String& tokenText) {
|
||||
return static_cast<SizeT>(std::count_if(tokens.begin(), tokens.end(),
|
||||
[&](const CodeToken& token) { return token.text == tokenText; }));
|
||||
}
|
||||
|
||||
bool HasIdentifierWithPrefixOutsideAllowed(const Vector<CodeToken>& tokens, const String& prefix,
|
||||
std::initializer_list<const char*> allowedIdentifiers) {
|
||||
return std::any_of(tokens.begin(), tokens.end(), [&](const CodeToken& token) {
|
||||
if (!IsIdentifierToken(token) || !token.text.starts_with(prefix)) {
|
||||
return false;
|
||||
}
|
||||
return std::none_of(allowedIdentifiers.begin(), allowedIdentifiers.end(),
|
||||
[&](const char* allowed) { return token.text == allowed; });
|
||||
});
|
||||
}
|
||||
|
||||
bool MatchTokenSequence(const Vector<CodeToken>& tokens, SizeT position,
|
||||
std::initializer_list<const char*> expected) {
|
||||
if (position + expected.size() > tokens.size()) {
|
||||
return false;
|
||||
}
|
||||
for (const char* token : expected) {
|
||||
if (tokens[position++].text != token) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
struct LinearPrefixScanMatch {
|
||||
SizeT sharedArraySizeBegin = 0;
|
||||
SizeT sharedArraySizeEnd = 0;
|
||||
SizeT scanBegin = 0;
|
||||
SizeT scanEnd = 0;
|
||||
String cache;
|
||||
String importance;
|
||||
String prefixSum;
|
||||
String loopLength;
|
||||
String loopIndex;
|
||||
String sum;
|
||||
};
|
||||
|
||||
bool ParseLinearPrefixScanTemplate(const Vector<CodeToken>& tokens, LinearPrefixScanMatch& match) {
|
||||
// The workaround deliberately recognizes one complete algorithm, not merely the
|
||||
// subgroupInclusiveAdd token. Changing scratch storage is only safe when that storage is
|
||||
// private to this scan and the workgroup has exactly 1024 X invocations.
|
||||
SizeT localSizeDeclarationCount = 0;
|
||||
for (SizeT i = 0; i < tokens.size(); ++i) {
|
||||
if (MatchTokenSequence(tokens, i, {"layout", "(", "local_size_x", "=", "1024", ")", "in", ";"})) {
|
||||
++localSizeDeclarationCount;
|
||||
}
|
||||
}
|
||||
if (localSizeDeclarationCount != 1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
SizeT sharedDeclarationIndex = String::npos;
|
||||
SizeT sharedDeclarationCount = 0;
|
||||
String cacheName;
|
||||
for (SizeT i = 0; i + 6 < tokens.size(); ++i) {
|
||||
if (tokens[i].text != "shared" || tokens[i + 1].text != "float" || !IsIdentifierToken(tokens[i + 2]) ||
|
||||
tokens[i + 3].text != "[" || tokens[i + 4].text != "64" || tokens[i + 5].text != "]" ||
|
||||
tokens[i + 6].text != ";") {
|
||||
continue;
|
||||
}
|
||||
++sharedDeclarationCount;
|
||||
sharedDeclarationIndex = i;
|
||||
cacheName = tokens[i + 2].text;
|
||||
}
|
||||
if (sharedDeclarationCount != 1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
SizeT scanTokenIndex = String::npos;
|
||||
SizeT scanCount = 0;
|
||||
for (SizeT i = 0; i + 7 < tokens.size(); ++i) {
|
||||
if (tokens[i].text == "float" && IsIdentifierToken(tokens[i + 1]) && tokens[i + 2].text == "=" &&
|
||||
tokens[i + 3].text == "subgroupInclusiveAdd" && tokens[i + 4].text == "(" &&
|
||||
IsIdentifierToken(tokens[i + 5]) && tokens[i + 6].text == ")" && tokens[i + 7].text == ";") {
|
||||
++scanCount;
|
||||
scanTokenIndex = i;
|
||||
}
|
||||
}
|
||||
if (scanCount != 1 || sharedDeclarationIndex >= scanTokenIndex) {
|
||||
return false;
|
||||
}
|
||||
|
||||
TokenCursor cursor(tokens, scanTokenIndex);
|
||||
String prefixSum;
|
||||
String importance;
|
||||
String loopLength;
|
||||
String loopIndex;
|
||||
String sum;
|
||||
if (!cursor.Consume("float") || !cursor.ConsumeAnyIdentifier(prefixSum) || !cursor.Consume("=") ||
|
||||
!cursor.Consume("subgroupInclusiveAdd") || !cursor.Consume("(") ||
|
||||
!cursor.ConsumeAnyIdentifier(importance) || !cursor.Consume(")") || !cursor.Consume(";") ||
|
||||
!cursor.Consume("if") || !cursor.Consume("(") || !cursor.Consume("gl_SubgroupInvocationID") ||
|
||||
!cursor.Consume("==") || !cursor.Consume("gl_SubgroupSize") || !cursor.Consume("-") ||
|
||||
!cursor.Consume("1u") || !cursor.Consume(")") || !cursor.ConsumeIdentifier(cacheName) ||
|
||||
!cursor.Consume("[") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume("]") || !cursor.Consume("=") ||
|
||||
!cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume(";") || !cursor.Consume("barrier") ||
|
||||
!cursor.Consume("(") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("uint") ||
|
||||
!cursor.ConsumeAnyIdentifier(loopLength) || !cursor.Consume("=") || !cursor.Consume("uint") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("findMSB") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("gl_NumSubgroups") || !cursor.Consume(")") || !cursor.Consume(")") ||
|
||||
!cursor.Consume(";") || !cursor.ConsumeIdentifier(loopLength) || !cursor.Consume("+=") ||
|
||||
!cursor.Consume("uint") || !cursor.Consume("(") || !cursor.Consume("gl_NumSubgroups") ||
|
||||
!cursor.Consume("-") || !cursor.Consume("(") || !cursor.Consume("1u") || !cursor.Consume("<<") ||
|
||||
!cursor.Consume("(") || !cursor.ConsumeIdentifier(loopLength) || !cursor.Consume("-") ||
|
||||
!cursor.Consume("1u") || !cursor.Consume(")") || !cursor.Consume(")") || !cursor.Consume(">") ||
|
||||
!cursor.Consume("0u") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("for") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("uint") || !cursor.ConsumeAnyIdentifier(loopIndex) ||
|
||||
!cursor.Consume("=") || !cursor.Consume("0") || !cursor.Consume(";") ||
|
||||
!cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("<") || !cursor.ConsumeIdentifier(loopLength) ||
|
||||
!cursor.Consume(";") || !cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("++") ||
|
||||
!cursor.Consume(")") || !cursor.Consume("{") || !cursor.Consume("if") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume("&") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("1u") || !cursor.Consume("<<") || !cursor.ConsumeIdentifier(loopIndex) ||
|
||||
!cursor.Consume(")") || !cursor.Consume(")") || !cursor.Consume(">") || !cursor.Consume("0u") ||
|
||||
!cursor.Consume(")") || !cursor.Consume("{") || !cursor.ConsumeIdentifier(prefixSum) ||
|
||||
!cursor.Consume("+=") || !cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume(">>") ||
|
||||
!cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("<<") || !cursor.ConsumeIdentifier(loopIndex) ||
|
||||
!cursor.Consume(")") || !cursor.Consume("-") || !cursor.Consume("1u") || !cursor.Consume("]") ||
|
||||
!cursor.Consume(";") || !cursor.Consume("if") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("gl_SubgroupInvocationID") || !cursor.Consume("==") || !cursor.Consume("gl_SubgroupSize") ||
|
||||
!cursor.Consume("-") || !cursor.Consume("1u") || !cursor.Consume(")") ||
|
||||
!cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") || !cursor.Consume("gl_SubgroupID") ||
|
||||
!cursor.Consume("]") || !cursor.Consume("=") || !cursor.ConsumeIdentifier(prefixSum) ||
|
||||
!cursor.Consume(";") || !cursor.Consume("}") || !cursor.Consume("barrier") || !cursor.Consume("(") ||
|
||||
!cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("}") || !cursor.Consume("if") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_LocalInvocationID") || !cursor.Consume(".") ||
|
||||
!cursor.Consume("x") || !cursor.Consume("==") || !cursor.Consume("uint") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("1024") || !cursor.Consume("-") || !cursor.Consume("1") || !cursor.Consume(")") ||
|
||||
!cursor.Consume(")") || !cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") ||
|
||||
!cursor.Consume("0") || !cursor.Consume("]") || !cursor.Consume("=") ||
|
||||
!cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume(";") || !cursor.Consume("barrier") ||
|
||||
!cursor.Consume("(") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("float") ||
|
||||
!cursor.ConsumeAnyIdentifier(sum) || !cursor.Consume("=") || !cursor.ConsumeIdentifier(cacheName) ||
|
||||
!cursor.Consume("[") || !cursor.Consume("0") || !cursor.Consume("]") || !cursor.Consume(";")) {
|
||||
return false;
|
||||
}
|
||||
const SizeT scanEndToken = cursor.Position() - 1;
|
||||
|
||||
// Require the scan's immediate consumer as well. This makes the match specific to a
|
||||
// linear distribution warp, and avoids changing unrelated prefix scans which may rely on
|
||||
// the implementation's native subgroup partitioning.
|
||||
if (!cursor.Consume("float") || !cursor.ConsumeAnyIdentifier() || !cursor.Consume("=") ||
|
||||
!cursor.Consume("(") || !cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume("-") ||
|
||||
!cursor.ConsumeIdentifier(importance) || !cursor.Consume(")") || !cursor.Consume("/") ||
|
||||
!cursor.ConsumeIdentifier(sum) || !cursor.Consume("-") || !cursor.Consume("float") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_LocalInvocationID") || !cursor.Consume(".") ||
|
||||
!cursor.Consume("x") || !cursor.Consume("+") || !cursor.Consume("1u") || !cursor.Consume(")") ||
|
||||
!cursor.Consume("/") || !cursor.Consume("float") || !cursor.Consume("(") || !cursor.Consume("1024") ||
|
||||
!cursor.Consume(")") || !cursor.Consume(";")) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// No other use may share the scratch array, and no additional subgroup operation or
|
||||
// builtin may silently retain native-64 semantics after this module becomes virtual-32.
|
||||
if (CountToken(tokens, cacheName) != 6 || CountToken(tokens, "subgroupInclusiveAdd") != 1 ||
|
||||
CountToken(tokens, "gl_SubgroupInvocationID") != 2 || CountToken(tokens, "gl_SubgroupSize") != 2 ||
|
||||
CountToken(tokens, "gl_SubgroupID") != 4 || CountToken(tokens, "gl_NumSubgroups") != 2 ||
|
||||
CountToken(tokens, "gl_LocalInvocationID") != 2 || CountToken(tokens, "barrier") != 3 ||
|
||||
CountToken(tokens, "findMSB") != 1 ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "subgroup", {"subgroupInclusiveAdd"}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(
|
||||
tokens, "gl_Subgroup",
|
||||
{"gl_SubgroupInvocationID", "gl_SubgroupSize", "gl_SubgroupID", "gl_NumSubgroups"}) ||
|
||||
// ARB/NV spellings of lane-width-sensitive builtins and functions
|
||||
// (gl_SubGroupSizeARB, ballotARB, gl_WarpSizeNV, shuffleNV, ...) must block the
|
||||
// rewrite just like their KHR counterparts: they would silently keep native-width
|
||||
// semantics in a module rewritten to the virtual 32-lane model.
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SubGroup", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Warp", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Thread", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SMID", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "ballot", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "shuffle", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "readInvocation", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "readFirstInvocation", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "anyInvocation", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "allInvocations", {})) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// The scan must be at the top level of the sole main() body. Its existing barriers already
|
||||
// require uniform control flow; this check prevents us from introducing extra barriers in
|
||||
// a nested branch or loop.
|
||||
SizeT mainOpenBrace = String::npos;
|
||||
SizeT mainCloseBrace = String::npos;
|
||||
SizeT mainCount = 0;
|
||||
for (SizeT i = 0; i + 4 < tokens.size(); ++i) {
|
||||
if (!MatchTokenSequence(tokens, i, {"void", "main", "(", ")", "{"})) {
|
||||
continue;
|
||||
}
|
||||
++mainCount;
|
||||
mainOpenBrace = i + 4;
|
||||
int depth = 1;
|
||||
for (SizeT j = mainOpenBrace + 1; j < tokens.size(); ++j) {
|
||||
if (tokens[j].text == "{")
|
||||
++depth;
|
||||
else if (tokens[j].text == "}" && --depth == 0) {
|
||||
mainCloseBrace = j;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (mainCount != 1 || mainCloseBrace == String::npos || scanTokenIndex <= mainOpenBrace ||
|
||||
scanEndToken >= mainCloseBrace) {
|
||||
return false;
|
||||
}
|
||||
int depthAtScan = 1;
|
||||
for (SizeT i = mainOpenBrace + 1; i < scanTokenIndex; ++i) {
|
||||
if (tokens[i].text == "{")
|
||||
++depthAtScan;
|
||||
else if (tokens[i].text == "}")
|
||||
--depthAtScan;
|
||||
}
|
||||
if (depthAtScan != 1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
constexpr const char* injectedNames[] = {"mglPrefixScanLane", "mglVirtualSubgroupInvocation",
|
||||
"mglVirtualSubgroup", "mglVirtualSubgroupBase",
|
||||
"mglPrefixLane", "mglVirtualSubgroupCount"};
|
||||
for (const char* injectedName : injectedNames) {
|
||||
if (CountToken(tokens, injectedName) != 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
match.sharedArraySizeBegin = tokens[sharedDeclarationIndex + 4].begin;
|
||||
match.sharedArraySizeEnd = tokens[sharedDeclarationIndex + 4].end;
|
||||
match.scanBegin = tokens[scanTokenIndex].begin;
|
||||
match.scanEnd = tokens[scanEndToken].end;
|
||||
match.cache = std::move(cacheName);
|
||||
match.importance = std::move(importance);
|
||||
match.prefixSum = std::move(prefixSum);
|
||||
match.loopLength = std::move(loopLength);
|
||||
match.loopIndex = std::move(loopIndex);
|
||||
match.sum = std::move(sum);
|
||||
return true;
|
||||
}
|
||||
|
||||
String BuildLinearPrefixScanReplacement(const LinearPrefixScanMatch& match) {
|
||||
String replacement;
|
||||
replacement.reserve(1800);
|
||||
replacement += "uint mglPrefixScanLane = gl_LocalInvocationID.x;\n";
|
||||
replacement += "uint mglVirtualSubgroupInvocation = mglPrefixScanLane & 31u;\n";
|
||||
replacement += "uint mglVirtualSubgroup = mglPrefixScanLane >> 5u;\n";
|
||||
replacement += "const uint mglVirtualSubgroupCount = 32u;\n";
|
||||
replacement += match.cache + "[mglPrefixScanLane] = " + match.importance + ";\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "float " + match.prefixSum + " = 0.0f;\n";
|
||||
replacement += "uint mglVirtualSubgroupBase = mglVirtualSubgroup << 5u;\n";
|
||||
replacement += "for (uint mglPrefixLane = mglVirtualSubgroupBase; "
|
||||
"mglPrefixLane <= mglPrefixScanLane; ++mglPrefixLane) {\n";
|
||||
replacement += match.prefixSum + " += " + match.cache + "[mglPrefixLane];\n";
|
||||
replacement += "}\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "if (mglVirtualSubgroupInvocation == 31u) " + match.cache +
|
||||
"[mglVirtualSubgroup] = " + match.prefixSum + ";\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "uint " + match.loopLength + " = uint(findMSB(mglVirtualSubgroupCount));\n";
|
||||
replacement +=
|
||||
match.loopLength + " += uint(mglVirtualSubgroupCount - (1u << (" + match.loopLength + " - 1u)) > 0u);\n";
|
||||
replacement += "for (uint " + match.loopIndex + " = 0u; " + match.loopIndex + " < " + match.loopLength +
|
||||
"; ++" + match.loopIndex + ") {\n";
|
||||
replacement += "if ((mglVirtualSubgroup & (1u << " + match.loopIndex + ")) > 0u) {\n";
|
||||
replacement += match.prefixSum + " += " + match.cache + "[(mglVirtualSubgroup >> " + match.loopIndex + " << " +
|
||||
match.loopIndex + ") - 1u];\n";
|
||||
replacement += "if (mglVirtualSubgroupInvocation == 31u) " + match.cache +
|
||||
"[mglVirtualSubgroup] = " + match.prefixSum + ";\n";
|
||||
replacement += "}\nbarrier();\n}\n";
|
||||
replacement += "if (mglPrefixScanLane == 1023u) " + match.cache + "[0] = " + match.prefixSum + ";\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "float " + match.sum + " = " + match.cache + "[0];";
|
||||
return replacement;
|
||||
}
|
||||
|
||||
void SkipDirectiveWhitespace(const MobileGL::String& source, SizeT& pos, SizeT lineEnd) {
|
||||
while (pos < lineEnd && std::isspace(static_cast<unsigned char>(source[pos]))) {
|
||||
pos++;
|
||||
@@ -1253,117 +929,6 @@ namespace {
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize,
|
||||
String& source) {
|
||||
constexpr Uint32 capturedSubgroupSize = 32;
|
||||
if (stage != ShaderStage::Compute || nativeSubgroupSize <= capturedSubgroupSize ||
|
||||
nativeSubgroupSize % capturedSubgroupSize != 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Vulkan subgroup widths are powers of two. Keep the workaround restricted to
|
||||
// wider widths which are a power-of-two multiple of the captured 32-lane model.
|
||||
const Uint32 subgroupScale = nativeSubgroupSize / capturedSubgroupSize;
|
||||
if ((subgroupScale & (subgroupScale - 1u)) != 0u) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Vector<CodeToken> tokens = TokenizeCode(source);
|
||||
LinearPrefixScanMatch match;
|
||||
if (!ParseLinearPrefixScanTemplate(tokens, match)) {
|
||||
// Diagnosability: when the trigger op is present but the template no longer
|
||||
// matches (e.g. the pack shipped a new shader revision), the affected device
|
||||
// silently falls back to the driver's miscompiled path. Make that visible.
|
||||
if (CountToken(tokens, "subgroupInclusiveAdd") > 0) {
|
||||
MGLOG_W_ONCE("%s: subgroupInclusiveAdd present but the linear prefix-scan template "
|
||||
"did not match; the wide-subgroup rewrite was NOT applied",
|
||||
__func__);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
const String replacement = BuildLinearPrefixScanReplacement(match);
|
||||
source.replace(match.scanBegin, match.scanEnd - match.scanBegin, replacement);
|
||||
// The declaration occurs before the replaced scan, so its original offsets remain
|
||||
// valid after the first replacement.
|
||||
source.replace(match.sharedArraySizeBegin, match.sharedArraySizeEnd - match.sharedArraySizeBegin,
|
||||
"1024");
|
||||
return true;
|
||||
}
|
||||
|
||||
namespace {
|
||||
struct ShaderSourceQuirkContext {
|
||||
ShaderStage stage = ShaderStage::Unknown;
|
||||
BackendType backend = BackendType::Unknown;
|
||||
MG_Backend::GpuVendorKind vendor = MG_Backend::GpuVendorKind::Unknown;
|
||||
Uint32 subgroupSize = 0;
|
||||
};
|
||||
|
||||
// Device-quirk registry. Every entry is a narrowly scoped source rewrite that
|
||||
// works around a specific driver defect. A quirk runs when its env override
|
||||
// forces it on, or when the override is Auto and DeviceApplies matches the
|
||||
// detected device. ForceOn bypasses only the device gate - each Apply keeps
|
||||
// its own structural safety checks. Add new per-device workarounds here
|
||||
// instead of open-coding them in PreprocessShaderSource.
|
||||
struct ShaderSourceQuirk {
|
||||
const char* name;
|
||||
// Reads the override out of the captured env, never out of the live
|
||||
// MG_Config table: a worker must see the same config the GL thread saw.
|
||||
MG_Config::QuirkOverride (*GetOverride)(const CompileEnv&);
|
||||
Bool (*DeviceApplies)(const ShaderSourceQuirkContext&);
|
||||
Bool (*Apply)(const ShaderSourceQuirkContext&, String&);
|
||||
};
|
||||
|
||||
constexpr ShaderSourceQuirk kShaderSourceQuirks[] = {
|
||||
{
|
||||
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN
|
||||
"subgroup-prefix-scan-rewrite",
|
||||
[](const CompileEnv& env) { return env.subgroupPrefixScanQuirk; },
|
||||
[](const ShaderSourceQuirkContext& ctx) {
|
||||
// Qualcomm's Vulkan driver miscompiles the recognized float
|
||||
// InclusiveScan pattern for native subgroups wider than the
|
||||
// captured 32 lanes; other vendors compile it correctly and
|
||||
// should keep their native scan.
|
||||
return ctx.backend == BackendType::DirectVulkan &&
|
||||
ctx.vendor == MG_Backend::GpuVendorKind::Qualcomm;
|
||||
},
|
||||
[](const ShaderSourceQuirkContext& ctx, String& source) {
|
||||
return RewriteLinearSubgroupPrefixScanForVulkan(ctx.stage, ctx.subgroupSize,
|
||||
source);
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
void ApplyShaderSourceQuirks(const CompileEnv& env, ShaderStage stage, String& source) {
|
||||
// No backend at capture time means no device to match a quirk against,
|
||||
// and (as before) no quirk can fire - not even a forced one, because
|
||||
// every Apply reads device parameters that do not exist yet.
|
||||
if (!env.HasBackend()) {
|
||||
return;
|
||||
}
|
||||
const ShaderSourceQuirkContext quirkContext{
|
||||
stage,
|
||||
env.backend,
|
||||
env.params.GpuVendor,
|
||||
env.params.SubgroupSize,
|
||||
};
|
||||
for (const ShaderSourceQuirk& quirk : kShaderSourceQuirks) {
|
||||
const MG_Config::QuirkOverride quirkOverride = quirk.GetOverride(env);
|
||||
if (quirkOverride == MG_Config::QuirkOverride::ForceOff) {
|
||||
continue;
|
||||
}
|
||||
if (quirkOverride == MG_Config::QuirkOverride::Auto &&
|
||||
!quirk.DeviceApplies(quirkContext)) {
|
||||
continue;
|
||||
}
|
||||
if (quirk.Apply(quirkContext, source)) {
|
||||
MGLOG_D("ApplyShaderSourceQuirks: applied '%s'%s", quirk.name,
|
||||
quirkOverride == MG_Config::QuirkOverride::ForceOn ? " (forced on)" : "");
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void PreprocessShaderSource(ShaderStage stage, String& source) {
|
||||
PreprocessShaderSource(stage, source, *GetCurrentCompileEnv());
|
||||
}
|
||||
@@ -1403,7 +968,6 @@ namespace MobileGL {
|
||||
ModernizeLegacyGLSL(stage, source, afterVersion);
|
||||
InjectDepthRangeBuiltinShim(stage, source, afterVersion);
|
||||
|
||||
ApplyShaderSourceQuirks(env, stage, source);
|
||||
}
|
||||
|
||||
Bool RetargetLegacyVersionDirectiveTo460(String& source) {
|
||||
|
||||
@@ -30,18 +30,6 @@ namespace MobileGL {
|
||||
// tests and diagnostics that drive the preprocessor standalone.
|
||||
void PreprocessShaderSource(ShaderStage stage, String& source);
|
||||
|
||||
// Some desktop-captured compute shaders build a workgroup-wide linear prefix scan
|
||||
// from subgroupInclusiveAdd plus a shared array of subgroup totals. Qualcomm's
|
||||
// Vulkan driver miscompiles that exact float InclusiveScan path for native subgroups
|
||||
// wider than the capture's 32 lanes. For the narrowly recognized, uniform-control-
|
||||
// flow template, replace the subgroup-local scan with a shared-memory, strict
|
||||
// left-fold over virtual 32-lane segments. Returns true only when the complete safe
|
||||
// template was recognized and rewritten. PreprocessShaderSource reaches this through
|
||||
// its device-quirk registry: by default only on detected Qualcomm Vulkan devices,
|
||||
// overridable either way with MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN=1/0. The explicit
|
||||
// entry point exists for deterministic tests.
|
||||
Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize, String& source);
|
||||
|
||||
// Rewrites a "#version 330 core" directive that PreprocessShaderSource normalized down
|
||||
// from a legacy desktop version back up to "#version 460 core". Returns false (leaving
|
||||
// the source untouched) for anything else: ES, compatibility, or an already-modern
|
||||
|
||||
@@ -423,6 +423,26 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
|
||||
InternalPackedLayout internalPacked;
|
||||
};
|
||||
|
||||
Bool IsValidUnpackPixelPair(TextureInputFormat format, TexturePixelDataType type) {
|
||||
UnpackChannelMapping mapping{};
|
||||
if (!GetUnpackChannelMapping(format, mapping)) return false;
|
||||
|
||||
PackedTypeLayout packed{};
|
||||
if (GetPackedTypeLayout(type, packed)) {
|
||||
return packed.fieldCount == mapping.channelCount;
|
||||
}
|
||||
|
||||
switch (type) {
|
||||
case TexturePixelDataType::UnsignedInt5999Rev:
|
||||
case TexturePixelDataType::UnsignedInt101111Rev:
|
||||
return !mapping.isInteger && mapping.channelCount == 3;
|
||||
default: {
|
||||
ShadowComponent component{};
|
||||
return GetDirectShadowComponentForType(type, mapping.isInteger, component);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Returns true when the (format, type) -> internal-format upload needs a per-texel conversion;
|
||||
// returns false both for layouts that already match the shadow bytes (memcpy fast path) and for
|
||||
// combinations the converter does not support (legacy copy behavior).
|
||||
@@ -964,6 +984,32 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
|
||||
return outputPixels;
|
||||
}
|
||||
|
||||
Bool ConvertOnePixelToInternal(TextureInternalFormat targetInternalFormat,
|
||||
TextureInputFormat textureInputFormat,
|
||||
TexturePixelDataType inputDataType,
|
||||
const void* inputPixel,
|
||||
Vector<Uint8>& outputPixel) {
|
||||
outputPixel.clear();
|
||||
if (inputPixel == nullptr || !IsValidUnpackPixelPair(textureInputFormat, inputDataType)) return false;
|
||||
|
||||
PixelStoreParameters params{};
|
||||
params.Alignment = 1;
|
||||
SizeT convertedSize = 0;
|
||||
void* converted = ProcessTexturePixelsDataUnpack(
|
||||
inputPixel, params, targetInternalFormat, textureInputFormat, inputDataType, {1, 1, 1}, false,
|
||||
convertedSize);
|
||||
const SizeT expectedSize = MG_Util::GetSizedInternalFormatSizeInBytes(targetInternalFormat);
|
||||
if (converted == nullptr || convertedSize != expectedSize || expectedSize == 0) {
|
||||
if (converted != nullptr) free(converted);
|
||||
return false;
|
||||
}
|
||||
|
||||
outputPixel.resize(convertedSize);
|
||||
Memcpy(outputPixel.data(), converted, convertedSize);
|
||||
free(converted);
|
||||
return true;
|
||||
}
|
||||
|
||||
void* ProcessTexturePixelsDataPack(const void* inputPixels, const PixelStoreParameters& params,
|
||||
TextureInternalFormat srcInternalFormat, TexturePixelDataType srcDataType,
|
||||
TextureInputFormat dstInputFormat, TexturePixelDataType dstDataType,
|
||||
|
||||
@@ -21,6 +21,12 @@ namespace MobileGL::MG_Util::PixelStoreProcessor {
|
||||
TextureInternalFormat srcInternalFormat, TexturePixelDataType srcDataType,
|
||||
TextureInputFormat dstInputFormat, TexturePixelDataType dstDataType,
|
||||
IntVec3 dimension, Bool isBitmap, SizeT& outSize);
|
||||
Bool ConvertOnePixelToInternal(TextureInternalFormat targetInternalFormat,
|
||||
TextureInputFormat textureInputFormat,
|
||||
TexturePixelDataType inputDataType,
|
||||
const void* inputPixel,
|
||||
Vector<Uint8>& outputPixel);
|
||||
|
||||
void ProcessColorSwizzle(void* data, SizeT pixelCount, const Vector<TextureSwizzleParam>& swizzle);
|
||||
|
||||
// True when a packed internal format's 32-bit storage word IS the client (format, type) word,
|
||||
|
||||
@@ -86,7 +86,7 @@ val pluginRendererConfig = buildJsonValue {
|
||||
selectable(
|
||||
key = "MOBILEGL_BACKEND_TYPE",
|
||||
title = RendererConfig.MetaString("mobilegl_backend_type_title"),
|
||||
items = RendererConfig.EnvItems("DirectGLES", listOf("DirectVulkan")),
|
||||
items = RendererConfig.EnvItems("DirectGLES", listOf("DirectVulkan", "DiligentVulkan")),
|
||||
)
|
||||
toggleable("MOBILEGL_DISABLE_TIMERQUERY", "1", false, RendererConfig.MetaString("mobilegl_disable_timerquery_title"))
|
||||
toggleable("MOBILEGL_DISABLE_SUBGROUP", "1", false, RendererConfig.MetaString("mobilegl_disable_subgroup_title"))
|
||||
|
||||
@@ -221,10 +221,13 @@ typedef signed char khronos_int8_t;
|
||||
typedef unsigned char khronos_uint8_t;
|
||||
typedef signed short int khronos_int16_t;
|
||||
typedef unsigned short int khronos_uint16_t;
|
||||
typedef signed long int khronos_intptr_t;
|
||||
typedef unsigned long int khronos_uintptr_t;
|
||||
typedef signed long int khronos_ssize_t;
|
||||
typedef unsigned long int khronos_usize_t;
|
||||
/* `long` is 32-bit on LLP64 Windows, including 64-bit MinGW. Use the
|
||||
* standard pointer-sized integer types so these remain pointer-width there. */
|
||||
#include <stdint.h>
|
||||
typedef intptr_t khronos_intptr_t;
|
||||
typedef uintptr_t khronos_uintptr_t;
|
||||
typedef intptr_t khronos_ssize_t;
|
||||
typedef uintptr_t khronos_usize_t;
|
||||
|
||||
#if KHRONOS_SUPPORT_FLOAT
|
||||
/*
|
||||
|
||||
@@ -277,11 +277,14 @@
|
||||
},
|
||||
{
|
||||
"name": "minecraft-1.21.4-fabric-iris-iterationrp-in-world",
|
||||
"ci": false,
|
||||
"ci_backends": [
|
||||
"DirectVulkan"
|
||||
],
|
||||
"trace_archive": "minecraft-1.21.4-fabric-iris-iterationrp-in-world.tgz",
|
||||
"golden": "minecraft-1.21.4-fabric-iris-iterationrp-in-world.0000202020.png",
|
||||
"target_call": 202020,
|
||||
"timeout_seconds": 1800
|
||||
"timeout_seconds": 1800,
|
||||
"ssim_threshold": 0.98
|
||||
},
|
||||
{
|
||||
"name": "minecraft-1.21.4-fabric-iris-bsl-esc-menu-854",
|
||||
|
||||
@@ -6,6 +6,7 @@ from pathlib import Path
|
||||
|
||||
|
||||
TRACE_CASES_JSON = Path(__file__).with_name("trace_cases.json")
|
||||
CI_BACKENDS = ("DirectGLES", "DirectVulkan")
|
||||
|
||||
|
||||
def load_trace_case_manifest(path=TRACE_CASES_JSON):
|
||||
@@ -71,6 +72,46 @@ def ci_trace_cases(cases):
|
||||
return [case for case in cases if case.get("ci", True)]
|
||||
|
||||
|
||||
def ci_backends(case):
|
||||
backends = case.get("ci_backends")
|
||||
if backends is None:
|
||||
return CI_BACKENDS
|
||||
if not isinstance(backends, list) or not backends:
|
||||
raise ValueError(f"ci_backends must be a non-empty list for {case['name']}")
|
||||
unknown = [backend for backend in backends if backend not in CI_BACKENDS]
|
||||
if unknown:
|
||||
raise ValueError(
|
||||
f"unknown ci_backends for {case['name']}: {', '.join(unknown)}"
|
||||
)
|
||||
if len(set(backends)) != len(backends):
|
||||
raise ValueError(f"ci_backends contains duplicates for {case['name']}")
|
||||
return backends
|
||||
|
||||
|
||||
def github_test_matrix(cases):
|
||||
return {
|
||||
"include": [
|
||||
{"backend": backend, "case": case["name"]}
|
||||
for case in cases
|
||||
for backend in ci_backends(case)
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def github_apk_matrix(cases):
|
||||
backends = {
|
||||
"DirectGLES": {"name": "DirectGLES", "gpu": "software"},
|
||||
"DirectVulkan": {"name": "DirectVulkan", "gpu": "lavapipe"},
|
||||
}
|
||||
return {
|
||||
"include": [
|
||||
{"backend": backends[backend], "case": github_apk_case(case)}
|
||||
for case in cases
|
||||
for backend in ci_backends(case)
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def cmake_quote(value):
|
||||
return '"' + str(value).replace("\\", "/").replace('"', '\\"') + '"'
|
||||
|
||||
@@ -114,7 +155,14 @@ def parse_args():
|
||||
parser.add_argument("--fixture-root", default="tools/trace_replay/fixtures")
|
||||
parser.add_argument(
|
||||
"--format",
|
||||
choices=("names", "github-apk", "fixture-files", "cmake"),
|
||||
choices=(
|
||||
"names",
|
||||
"github-test-matrix",
|
||||
"github-apk",
|
||||
"github-apk-matrix",
|
||||
"fixture-files",
|
||||
"cmake",
|
||||
),
|
||||
default="names",
|
||||
)
|
||||
return parser.parse_args()
|
||||
@@ -127,8 +175,12 @@ def main():
|
||||
cases = ci_trace_cases(cases)
|
||||
if args.format == "names":
|
||||
print(json.dumps([case["name"] for case in cases], separators=(",", ":")))
|
||||
elif args.format == "github-test-matrix":
|
||||
print(json.dumps(github_test_matrix(cases), separators=(",", ":")))
|
||||
elif args.format == "github-apk":
|
||||
print(json.dumps([github_apk_case(case) for case in cases], separators=(",", ":")))
|
||||
elif args.format == "github-apk-matrix":
|
||||
print(json.dumps(github_apk_matrix(cases), separators=(",", ":")))
|
||||
elif args.format == "fixture-files":
|
||||
if not args.case_name:
|
||||
print("--case is required for --format fixture-files", file=sys.stderr)
|
||||
|
||||
Reference in New Issue
Block a user