mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-12 22:28:32 +09:00
Compare commits
44
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3223ecb14e | ||
|
|
fc4cd980f2 | ||
|
|
992d16267c | ||
|
|
0ea9e6de5f | ||
|
|
241ed377b4 | ||
|
|
bf312a4b67 | ||
|
|
56b31a9587 | ||
|
|
8a0a8a0274 | ||
|
|
d076c29146 | ||
|
|
930a607bdf | ||
|
|
34685b4bb0 | ||
|
|
c540fb88ee | ||
|
|
6ae3245a0d | ||
|
|
7e048fc2bf | ||
|
|
83cdfd6bdd | ||
|
|
1c76f886cf | ||
|
|
a2e109beff | ||
|
|
63f0756644 | ||
|
|
450215d12c | ||
|
|
3a9e520170 | ||
|
|
d2996ba1cf | ||
|
|
c8632dfefe | ||
|
|
b8a8a660e1 | ||
|
|
7ab83861ca | ||
|
|
eaeba556a3 | ||
|
|
72fa1221a5 | ||
|
|
c4254c4bbd | ||
|
|
199164c2e0 | ||
|
|
e87063e90c | ||
|
|
122da27249 | ||
|
|
bc2d698b3e | ||
|
|
3049c4b82b | ||
|
|
2b3850b76b | ||
|
|
2a0ae743a0 | ||
|
|
6839219c10 | ||
|
|
52ddb440ca | ||
|
|
79feeffd25 | ||
|
|
202037b5a3 | ||
|
|
bce9c48c8e | ||
|
|
b6a7807a3a | ||
|
|
48ba622387 | ||
|
|
05260d1262 | ||
|
|
6eb5ff51c5 | ||
|
|
e526f8e8ac |
+44
-1
@@ -192,6 +192,8 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerDrawParametersPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUboMemberRelaxedPrecisionPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripNoPerspectivePass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateNoPerspectivePass.cpp
|
||||
|
||||
MobileGL/MG_Util/BackendLoaders/OpenGL/Loader.cpp
|
||||
MobileGL/MG_Util/BackendLoaders/Vulkan/Loader.cpp
|
||||
@@ -301,6 +303,13 @@ if (ANDROID)
|
||||
)
|
||||
endif()
|
||||
|
||||
if (WIN32)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Impl/WGLImpl/WGLImpl.cpp
|
||||
MobileGL/MG_Impl/WGLImpl/Exporting/Definitions.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
set(MOBILEGL_LINK_LIBRARIES
|
||||
glslang::glslang
|
||||
spirv-cross-c
|
||||
@@ -329,10 +338,18 @@ set(MOBILEGL_INCLUDE_DIR
|
||||
${SPIRV-Headers_SOURCE_DIR}/include
|
||||
)
|
||||
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
${SOURCE_FILES}
|
||||
)
|
||||
|
||||
if (WIN32)
|
||||
# The wgl* entry points are exported via .def (see the comment in wgl.def);
|
||||
# only the shared library links it.
|
||||
target_sources(${CMAKE_PROJECT_NAME} PRIVATE
|
||||
MobileGL/MG_Impl/WGLImpl/Exporting/wgl.def
|
||||
)
|
||||
endif()
|
||||
|
||||
if (CMAKE_BUILD_TYPE STREQUAL "Debug")
|
||||
set_target_properties(${CMAKE_PROJECT_NAME} PROPERTIES
|
||||
C_VISIBILITY_PRESET default
|
||||
@@ -375,6 +392,18 @@ if(UNIX AND NOT APPLE AND NOT ANDROID)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if(WIN32)
|
||||
# Drop-in for the classic GL loader path: a copy named opengl32.dll placed
|
||||
# next to a host executable is what LoadLibrary("opengl32.dll") and gdi32's
|
||||
# pixel-format forwarding will resolve.
|
||||
add_custom_command(TARGET ${CMAKE_PROJECT_NAME} POST_BUILD
|
||||
COMMAND ${CMAKE_COMMAND} -E copy_if_different
|
||||
"$<TARGET_FILE:${CMAKE_PROJECT_NAME}>"
|
||||
"$<TARGET_FILE_DIR:${CMAKE_PROJECT_NAME}>/opengl32.dll"
|
||||
COMMENT "Creating opengl32.dll drop-in copy"
|
||||
)
|
||||
endif()
|
||||
|
||||
if(NOT ANDROID)
|
||||
add_library(${CMAKE_PROJECT_NAME}_s STATIC
|
||||
${SOURCE_FILES}
|
||||
@@ -426,8 +455,21 @@ if (ANDROID)
|
||||
endif()
|
||||
|
||||
if (APPLE AND NOT MOBILEGL_IOS)
|
||||
# MobileGL statically embeds glslang, SPIRV-Tools, and SPIRV-Cross. When
|
||||
# this dylib is injected with DYLD_INSERT_LIBRARIES, exporting those C++
|
||||
# symbols interposes incompatible copies embedded by host libraries such
|
||||
# as shaderc. Keep only the public GL/EGL/CGL loader surface globally
|
||||
# visible; GetProcAddress can still return pointers to hidden internals.
|
||||
set(MOBILEGL_MACOS_EXPORTED_SYMBOLS
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/MobileGL/MG_Impl/DyldInterpose/ExportedSymbols.txt")
|
||||
target_link_options(${CMAKE_PROJECT_NAME} PRIVATE
|
||||
"LINKER:-exported_symbols_list,${MOBILEGL_MACOS_EXPORTED_SYMBOLS}")
|
||||
set_property(TARGET ${CMAKE_PROJECT_NAME} APPEND PROPERTY
|
||||
LINK_DEPENDS "${MOBILEGL_MACOS_EXPORTED_SYMBOLS}")
|
||||
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME} PUBLIC
|
||||
"-framework Cocoa"
|
||||
"-framework CoreVideo"
|
||||
"-framework QuartzCore"
|
||||
"-framework Foundation"
|
||||
"-framework OpenGL"
|
||||
@@ -435,6 +477,7 @@ if (APPLE AND NOT MOBILEGL_IOS)
|
||||
if(TARGET ${CMAKE_PROJECT_NAME}_s)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME}_s PUBLIC
|
||||
"-framework Cocoa"
|
||||
"-framework CoreVideo"
|
||||
"-framework QuartzCore"
|
||||
"-framework Foundation"
|
||||
"-framework OpenGL"
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
#define MOBILEGL_EGL_API MOBILEGL_API
|
||||
#define MOBILEGL_CGL_API MOBILEGL_API
|
||||
#define MOBILEGL_NSOPENGL_API MOBILEGL_API
|
||||
#define MOBILEGL_WGL_API MOBILEGL_API
|
||||
|
||||
// ====================== MobileGL configurations ======================= //
|
||||
#ifndef MOBILEGL_LOG_ACTIVE_LEVEL
|
||||
|
||||
@@ -14,7 +14,13 @@ namespace MobileGL {
|
||||
} // namespace MG_Config
|
||||
|
||||
namespace MG_Backend {
|
||||
UniquePtr<BackendObject> pActiveBackendObject;
|
||||
// Leak-at-exit storage: the UniquePtr itself lives on the heap and is
|
||||
// never destroyed by the runtime, so process exit runs no backend
|
||||
// destructors (static destruction order across TUs is undefined).
|
||||
// Deterministic teardown happens inside the EGL lifecycle instead:
|
||||
// the last eglTerminate calls MobileGL::Destroy(), which .reset()s
|
||||
// these singletons while the process is still healthy.
|
||||
UniquePtr<BackendObject>& pActiveBackendObject = *new UniquePtr<BackendObject>();
|
||||
GlobalBackendFunctionsTable gBackendFunctionsTable;
|
||||
} // namespace MG_Backend
|
||||
} // namespace MobileGL
|
||||
|
||||
+47
-33
@@ -9,14 +9,25 @@
|
||||
#include "Init.h"
|
||||
#include "Config.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Backend/DirectVulkan/DirectVulkan.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <MG_Impl/GLImpl/Texture/ProxyTexture.h>
|
||||
#include <MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h>
|
||||
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace {
|
||||
Bool g_isInitialized = false;
|
||||
std::atomic<Bool> g_isInitialized = false;
|
||||
thread_local Bool tl_initializing = false;
|
||||
|
||||
std::mutex& InitMutex() {
|
||||
static std::mutex mutex;
|
||||
return mutex;
|
||||
}
|
||||
|
||||
void DestroyImpl(Bool logLifecycle) {
|
||||
if (!g_isInitialized) {
|
||||
@@ -27,6 +38,12 @@ namespace MobileGL {
|
||||
MGLOG_I("MobileGL closing...");
|
||||
}
|
||||
glslang::FinalizeProcess();
|
||||
// GL syncs die with their contexts, and every context is gone by the
|
||||
// time full teardown runs: drain the live-sync registry while the
|
||||
// backend function table can still release the backend handles (and
|
||||
// before a re-initialized library could pair them with the wrong
|
||||
// backend's DeleteSync).
|
||||
MG_Impl::GLImpl::DestroyAllSyncObjects();
|
||||
MG_Backend::pActiveBackendObject.reset();
|
||||
MG_State::pGLContext.reset();
|
||||
MG_State::pEGLContext.reset();
|
||||
@@ -64,40 +81,37 @@ namespace MobileGL {
|
||||
MGLOG_I("MobileGL initialized");
|
||||
}
|
||||
|
||||
void EnsureInitialized() {
|
||||
if (g_isInitialized.load(std::memory_order_acquire)) {
|
||||
return;
|
||||
}
|
||||
// Re-entrant call while this thread is already inside Initialize()
|
||||
// (e.g. an init step routing back through a public entry point).
|
||||
if (tl_initializing) {
|
||||
return;
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(InitMutex());
|
||||
if (g_isInitialized.load(std::memory_order_acquire)) {
|
||||
return;
|
||||
}
|
||||
tl_initializing = true;
|
||||
Initialize();
|
||||
tl_initializing = false;
|
||||
}
|
||||
|
||||
void Destroy() {
|
||||
DestroyImpl(true);
|
||||
}
|
||||
|
||||
#if defined(__linux__) || defined(__APPLE__)
|
||||
__attribute__((constructor)) static void AutoInit() {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
__attribute__((destructor)) static void AutoDestroy() {
|
||||
if (MG_Config::Features.TraceSkipAutodestroy) {
|
||||
return;
|
||||
}
|
||||
#if defined(__APPLE__)
|
||||
// macOS injected dylibs can run destructors after logging/backend static state is already torn down.
|
||||
return;
|
||||
#else
|
||||
DestroyImpl(false);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef _WIN32
|
||||
BOOL WINAPI DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved) {
|
||||
switch (ul_reason_for_call) {
|
||||
case DLL_PROCESS_ATTACH:
|
||||
Initialize();
|
||||
break;
|
||||
|
||||
case DLL_PROCESS_DETACH:
|
||||
Destroy();
|
||||
break;
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
#endif
|
||||
// MobileGL's lifecycle is owned entirely by the host-API layers
|
||||
// (EGL/WGL/CGL): initialization happens lazily on the first entry point
|
||||
// via EnsureInitialized(), and full teardown happens deterministically
|
||||
// when the last EGL display is terminated with nothing current (EGLImpl
|
||||
// calls Destroy()). There is intentionally no backend-initializing static
|
||||
// constructor, no static destructor, and no DllMain: the global singletons
|
||||
// use leak-at-exit storage (see GlobalObjects.cpp), so a process that exits
|
||||
// without eglTerminate simply leaks them to the OS instead of running
|
||||
// backend destructors during static teardown. macOS has a lightweight
|
||||
// dyld constructor that installs NSOpenGL dispatch hooks only; full backend
|
||||
// initialization still enters here from the first hooked CGL context.
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -11,6 +11,13 @@
|
||||
|
||||
namespace MobileGL {
|
||||
void Initialize();
|
||||
// Thread-safe, idempotent, and re-entrant wrapper around Initialize().
|
||||
// Host layers (EGL/WGL/CGL entry points) call this lazily on first use so
|
||||
// full backend initialization never depends on ELF/DLL static constructors,
|
||||
// and so a fresh init can follow a full Destroy() (e.g. after the last
|
||||
// eglTerminate). The macOS dyld bootstrap installs only lightweight
|
||||
// NSOpenGL method hooks.
|
||||
void EnsureInitialized();
|
||||
void Destroy();
|
||||
|
||||
namespace MG_Util::Debug {
|
||||
|
||||
@@ -313,7 +313,8 @@ namespace MobileGL {
|
||||
Android,
|
||||
X11,
|
||||
MetalLayer,
|
||||
// TODO: Wayland, Windows, etc.
|
||||
Win32, // Handle is an HWND
|
||||
// TODO: Wayland, etc.
|
||||
WindowBackendCount,
|
||||
Unknown = -1
|
||||
};
|
||||
|
||||
@@ -13,6 +13,6 @@
|
||||
#include "DirectVulkan/BackendObject_DirectVulkan.h"
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
extern UniquePtr<BackendObject> pActiveBackendObject;
|
||||
extern UniquePtr<BackendObject>& pActiveBackendObject;
|
||||
extern GlobalBackendFunctionsTable gBackendFunctionsTable;
|
||||
} // namespace MobileGL::MG_Backend
|
||||
|
||||
@@ -701,9 +701,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
if ((handle.Backend != WindowBackend::Android &&
|
||||
handle.Backend != WindowBackend::X11 &&
|
||||
handle.Backend != WindowBackend::MetalLayer) ||
|
||||
handle.Backend != WindowBackend::MetalLayer &&
|
||||
handle.Backend != WindowBackend::Win32) ||
|
||||
!handle.Handle) {
|
||||
MGLOG_E("DirectGLES backend only supports Android, X11, and CAMetalLayer native windows");
|
||||
MGLOG_E("DirectGLES backend only supports Android, X11, CAMetalLayer, and Win32 native windows");
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -267,16 +267,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Uint g_boundArrayBufferId = 0;
|
||||
Bool g_boundArrayBufferKnown = false;
|
||||
|
||||
// Driver-level GL_PIXEL_PACK/UNPACK_BUFFER binding shadows (see
|
||||
// Managers.h). Resting state between operations is 0; scopes in the
|
||||
// readback/upload paths bind what they need through the cache and
|
||||
// return to 0, so a stale user PBO can never capture a later
|
||||
// readback that meant to target client memory.
|
||||
Uint g_boundPixelPackBufferId = 0;
|
||||
Bool g_boundPixelPackBufferKnown = false;
|
||||
Uint g_boundPixelUnpackBufferId = 0;
|
||||
Bool g_boundPixelUnpackBufferKnown = false;
|
||||
|
||||
// Bumped whenever the backend ES context is destroyed; resources with
|
||||
// an older generation hold ids from a dead context.
|
||||
Uint g_bufferContextGeneration = 1;
|
||||
|
||||
// Defined next to the indexed-binding shadow below; forward-declared so
|
||||
// every glDeleteBuffers site in this namespace can scrub stale shadow
|
||||
// entries (GL resets a deleted buffer's indexed bindings to 0, and a
|
||||
// recycled name matching a stale shadow entry would otherwise
|
||||
// false-skip the rebind).
|
||||
void ScrubIndexedBufferBindingShadowForId(Uint id);
|
||||
// entries (GL resets a deleted buffer's bindings - indexed and pixel
|
||||
// pack/unpack alike - to 0, and a recycled name matching a stale shadow
|
||||
// entry would otherwise false-skip the rebind).
|
||||
void ScrubBufferBindingShadowsForId(Uint id);
|
||||
|
||||
// Resources whose owning BufferObject died; ids deleted at the next
|
||||
// sync point with a current ES context.
|
||||
@@ -318,10 +328,16 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == r.id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
// Pooling keeps the id alive (and thus any driver binding of it);
|
||||
// drop to unknown rather than claiming the post-delete 0 state.
|
||||
if ((g_boundPixelPackBufferKnown && g_boundPixelPackBufferId == r.id) ||
|
||||
(g_boundPixelUnpackBufferKnown && g_boundPixelUnpackBufferId == r.id)) {
|
||||
InvalidatePixelBufferBindingCaches();
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(g_poolMutex);
|
||||
auto& bucket = g_bufferPool[r.storageSize];
|
||||
if (bucket.size() >= kMaxEntriesPerBucket || g_pooledBytes + r.storageSize > kMaxPoolBytes) {
|
||||
ScrubIndexedBufferBindingShadowForId(r.id);
|
||||
ScrubBufferBindingShadowsForId(r.id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &r.id); // over budget: don't pool
|
||||
r.id = 0;
|
||||
return;
|
||||
@@ -502,7 +518,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Need a fresh id: glBufferStorage fails on a buffer that already has
|
||||
// immutable storage, and any prior mutable store is replaced anyway.
|
||||
if (resource->id != 0) {
|
||||
ScrubIndexedBufferBindingShadowForId(resource->id);
|
||||
ScrubBufferBindingShadowsForId(resource->id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &resource->id);
|
||||
resource->id = 0;
|
||||
}
|
||||
@@ -630,7 +646,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == glesResource->id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
ScrubIndexedBufferBindingShadowForId(glesResource->id);
|
||||
ScrubBufferBindingShadowsForId(glesResource->id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &glesResource->id);
|
||||
glesResource->id = 0;
|
||||
}
|
||||
@@ -668,6 +684,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void OnBackendContextDestroyed() {
|
||||
UnregisterBufferBackendOps();
|
||||
++g_bufferContextGeneration;
|
||||
InvalidateArrayBufferBindingCache();
|
||||
InvalidateIndexedBufferBindingCache();
|
||||
InvalidatePixelBufferBindingCaches();
|
||||
// The global-UBO ring's id and persistent map died with the context;
|
||||
// drop the handles (no GL) and let the next draw recreate the ring.
|
||||
ResetUboRingForNewContext();
|
||||
@@ -694,7 +713,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == glesResource->id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
ScrubIndexedBufferBindingShadowForId(glesResource->id);
|
||||
ScrubBufferBindingShadowsForId(glesResource->id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &glesResource->id);
|
||||
glesResource->id = 0;
|
||||
}
|
||||
@@ -819,6 +838,41 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
g_boundArrayBufferKnown = false;
|
||||
}
|
||||
|
||||
void BindPixelPackBufferId(Uint id) {
|
||||
if (g_boundPixelPackBufferKnown && g_boundPixelPackBufferId == id) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_PACK_BUFFER, id);
|
||||
g_boundPixelPackBufferId = id;
|
||||
g_boundPixelPackBufferKnown = true;
|
||||
}
|
||||
|
||||
void BindPixelUnpackBufferId(Uint id) {
|
||||
if (g_boundPixelUnpackBufferKnown && g_boundPixelUnpackBufferId == id) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, id);
|
||||
g_boundPixelUnpackBufferId = id;
|
||||
g_boundPixelUnpackBufferKnown = true;
|
||||
}
|
||||
|
||||
void InvalidatePixelBufferBindingCaches() {
|
||||
g_boundPixelPackBufferId = 0;
|
||||
g_boundPixelPackBufferKnown = false;
|
||||
g_boundPixelUnpackBufferId = 0;
|
||||
g_boundPixelUnpackBufferKnown = false;
|
||||
}
|
||||
|
||||
void NoteBufferIdDeleted(Uint id) {
|
||||
if (id == 0) {
|
||||
return;
|
||||
}
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
ScrubBufferBindingShadowsForId(id);
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Shadow of the GL indexed buffer bindings so redundant glBindBufferBase/Range
|
||||
// (same index + id + range) are skipped. isBase distinguishes a whole-buffer
|
||||
@@ -840,12 +894,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// glDeleteBuffers resets the deleted buffer's bindings (indexed ones
|
||||
// included) to 0 in the current context; mirror that in the shadow, or a
|
||||
// later buffer recycling the same name with a matching range would
|
||||
// false-skip its rebind. Default IndexedBufferBinding{} == base(0) ==
|
||||
// the post-delete GL state.
|
||||
void ScrubIndexedBufferBindingShadowForId(Uint id) {
|
||||
// glDeleteBuffers resets the deleted buffer's bindings (indexed and
|
||||
// pixel pack/unpack ones included) to 0 in the current context; mirror
|
||||
// that in the shadows, or a later buffer recycling the same name with a
|
||||
// matching shadow entry would false-skip its rebind. Default
|
||||
// IndexedBufferBinding{} == base(0) == the post-delete GL state.
|
||||
void ScrubBufferBindingShadowsForId(Uint id) {
|
||||
if (id == 0) return;
|
||||
for (auto& binding : g_indexedUBOBindings) {
|
||||
if (binding.id == id) binding = {};
|
||||
@@ -853,6 +907,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
for (auto& binding : g_indexedSSBOBindings) {
|
||||
if (binding.id == id) binding = {};
|
||||
}
|
||||
if (g_boundPixelPackBufferKnown && g_boundPixelPackBufferId == id) {
|
||||
g_boundPixelPackBufferId = 0;
|
||||
}
|
||||
if (g_boundPixelUnpackBufferKnown && g_boundPixelUnpackBufferId == id) {
|
||||
g_boundPixelUnpackBufferId = 0;
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
@@ -897,7 +957,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
auto& bucket = g_bufferPool[oldestKey];
|
||||
PooledBuffer& e = bucket[oldestIdx];
|
||||
if (e.contextGeneration == g_bufferContextGeneration && e.id != 0) {
|
||||
ScrubIndexedBufferBindingShadowForId(e.id);
|
||||
ScrubBufferBindingShadowsForId(e.id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &e.id);
|
||||
}
|
||||
g_pooledBytes -= e.size;
|
||||
@@ -1064,7 +1124,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const Bool staleContext = entry.contextGeneration != g_bufferContextGeneration;
|
||||
if (!staleContext && entry.retireSerial > completed) continue;
|
||||
if (!staleContext && entry.id != 0) {
|
||||
ScrubIndexedBufferBindingShadowForId(entry.id);
|
||||
ScrubBufferBindingShadowsForId(entry.id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &entry.id);
|
||||
}
|
||||
g_retiredUboRings[i] = g_retiredUboRings.back();
|
||||
@@ -1152,6 +1212,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
for (auto& bufferId : m_clientAttributeBufferIds) {
|
||||
if (bufferId != 0) {
|
||||
BufferImpl::NoteBufferIdDeleted(bufferId);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &bufferId);
|
||||
bufferId = 0;
|
||||
}
|
||||
@@ -1324,6 +1385,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
g_GLESFuncs.glGenTextures(1, &m_backendTextureId);
|
||||
m_contextGeneration = g_textureContextGeneration;
|
||||
if (m_backendTextureId == 0) {
|
||||
MGLOG_E("Failed to generate texture object.");
|
||||
MGLOG_E("ES glGetError(): %s", MG_Util::ConvertGLEnumToString(g_GLESFuncs.glGetError()).c_str());
|
||||
@@ -1332,6 +1394,27 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
}
|
||||
|
||||
BackendTextureObject::~BackendTextureObject() {
|
||||
if (m_backendTextureId == 0) {
|
||||
return;
|
||||
}
|
||||
// Scrub every driver-state shadow that could false-skip when the name
|
||||
// or this heap address is recycled - regardless of whether the id can
|
||||
// still be deleted.
|
||||
ScratchFBOImpl::NoteTextureIdDeleted(m_backendTextureId);
|
||||
for (auto& unitCache : g_boundTexturesCache) {
|
||||
for (auto& boundTexture : unitCache) {
|
||||
if (boundTexture == this) {
|
||||
boundTexture = nullptr;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (m_contextGeneration == g_textureContextGeneration && g_GLESFuncs.glDeleteTextures) {
|
||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||
}
|
||||
m_backendTextureId = 0;
|
||||
}
|
||||
|
||||
void BackendTextureObject::Bind(GLenum target, Uint unit) {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
@@ -1364,7 +1447,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
void BackendTextureObject::RecreateBackendTexture() {
|
||||
if (m_backendTextureId != 0) {
|
||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||
ScratchFBOImpl::NoteTextureIdDeleted(m_backendTextureId);
|
||||
if (m_contextGeneration == g_textureContextGeneration) {
|
||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||
}
|
||||
for (auto& unitCache : g_boundTexturesCache) {
|
||||
for (auto& boundTexture : unitCache) {
|
||||
if (boundTexture == this) {
|
||||
@@ -1375,6 +1461,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
|
||||
g_GLESFuncs.glGenTextures(1, &m_backendTextureId);
|
||||
m_contextGeneration = g_textureContextGeneration;
|
||||
if (m_backendTextureId == 0) {
|
||||
MGLOG_E("Failed to regenerate texture object.");
|
||||
MGLOG_E("ES glGetError(): %s", MG_Util::ConvertGLEnumToString(g_GLESFuncs.glGetError()).c_str());
|
||||
@@ -1391,7 +1478,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// glGetIntegerv - that query forces a driver pipeline sync and, because texture
|
||||
// uploads run it per dirty texture per frame, it dominated the DirectGLES draw
|
||||
// path. The backend unpack state is set ONLY by MobileGL's own save/restore
|
||||
// helpers (this class, TempPixelStoreParameterSync, the R32F copy path), all of
|
||||
// helpers (this class and, historically, the R32F copy path), all of
|
||||
// which restore to the resting default, so the shadow stays accurate; a one-time
|
||||
// forced sync pins the backend to that known default up front. Apply() is
|
||||
// compare-and-set, so the (now redundant) glPixelStorei calls also usually no-op.
|
||||
@@ -1563,6 +1650,56 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return convertedData.data();
|
||||
}
|
||||
|
||||
// RGB565/RGB5_A1 shadow data is stored as 8-bit unorm; uploading it as GL_UNSIGNED_BYTE
|
||||
// leaves the 8-bit -> 5/6-bit requantization to the driver, whose rounding direction is
|
||||
// implementation-defined: Adreno rounds to nearest (lossless round trip) but Mali floors,
|
||||
// drifting mid-range texels one 5-bit step down and failing the KHR-GL3x
|
||||
// pixelstoragemodes.teximage3d rgb565/rgb5a1 1/32-eps checks. Repack the shadow rows into
|
||||
// the packed 16-bit client type with round-to-nearest instead - that recovers the original
|
||||
// 5/6-bit values exactly (the shadow expansion round(v * 255 / max) is injective), so the
|
||||
// driver stores them verbatim with no requantization left to its discretion. 4-bit formats
|
||||
// (RGBA4) are exempt: their 8-bit expansion (v * 17) is exact under either rounding.
|
||||
// Always retargets *inOutType for these formats (even for null data) so every upload of a
|
||||
// level uses the same client type.
|
||||
static const void* PreparePackedNormUpload(TextureInternalFormat format, const IntVec3& texelSize,
|
||||
const void* data, SizeT byteSize, GLenum* inOutType,
|
||||
Vector<Uint8>& packedData) {
|
||||
if (format != TextureInternalFormat::RGB5 && format != TextureInternalFormat::RGB5A1) {
|
||||
return data;
|
||||
}
|
||||
const Bool hasAlpha = format == TextureInternalFormat::RGB5A1;
|
||||
const GLenum packedType = hasAlpha ? GL_UNSIGNED_SHORT_5_5_5_1 : GL_UNSIGNED_SHORT_5_6_5;
|
||||
// Idempotent across a region's level loop: glType is shared, so later levels arrive with
|
||||
// the already-retargeted packed type and must still be converted.
|
||||
if (*inOutType != GL_UNSIGNED_BYTE && *inOutType != packedType) {
|
||||
return data;
|
||||
}
|
||||
*inOutType = packedType;
|
||||
if (data == nullptr || byteSize == 0) {
|
||||
return data;
|
||||
}
|
||||
const SizeT srcPixelBytes = hasAlpha ? 4 : 3;
|
||||
const SizeT texelCount = std::min(static_cast<SizeT>(std::max(texelSize.x(), 0)) *
|
||||
static_cast<SizeT>(std::max(texelSize.y(), 0)) *
|
||||
static_cast<SizeT>(std::max(texelSize.z(), 1)),
|
||||
byteSize / srcPixelBytes);
|
||||
packedData.resize(texelCount * sizeof(Uint16));
|
||||
const Uint8* src = static_cast<const Uint8*>(data);
|
||||
auto* dst = reinterpret_cast<Uint16*>(packedData.data());
|
||||
for (SizeT i = 0; i < texelCount; ++i, src += srcPixelBytes) {
|
||||
const Uint32 r = (static_cast<Uint32>(src[0]) * 31u + 127u) / 255u;
|
||||
const Uint32 b = (static_cast<Uint32>(src[2]) * 31u + 127u) / 255u;
|
||||
if (hasAlpha) {
|
||||
const Uint32 g = (static_cast<Uint32>(src[1]) * 31u + 127u) / 255u;
|
||||
dst[i] = static_cast<Uint16>((r << 11) | (g << 6) | (b << 1) | (src[3] >= 128 ? 1u : 0u));
|
||||
} else {
|
||||
const Uint32 g = (static_cast<Uint32>(src[1]) * 63u + 127u) / 255u;
|
||||
dst[i] = static_cast<Uint16>((r << 11) | (g << 5) | b);
|
||||
}
|
||||
}
|
||||
return packedData.data();
|
||||
}
|
||||
|
||||
void BackendTextureObject::SyncMipmapsToBackend(
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject) {
|
||||
if (!stateTextureObject) {
|
||||
@@ -1706,9 +1843,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), levelTexelSize, pData, levelByteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData = PreparePackedNormUpload(textureMipmapObject->GetFormat(), levelTexelSize,
|
||||
uploadData, levelByteSize, &glType, packedUploadData);
|
||||
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
const IntVec3 uploadSize =
|
||||
GetBackendUploadSize(stateTextureObject->GetTarget(), levelTexelSize);
|
||||
switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) {
|
||||
@@ -1760,7 +1900,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const auto& uploadTargets = textureMipmapObject->GetUploadTargets();
|
||||
if (TextureImpl::IsMultisampleTextureTarget(targetInternal)) {
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
switch (targetInternal) {
|
||||
case TextureTarget::Texture2DMultisample:
|
||||
g_GLESFuncs.glTexStorage2DMultisample(
|
||||
@@ -1787,7 +1927,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
} else if (stateTextureObject->IsImmutable() || m_imageBindableStorageRequired) {
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
const IntVec3 storageSize = GetBackendUploadSize(targetInternal, baseSize);
|
||||
switch (MapToBackendTextureTarget(targetInternal)) {
|
||||
case TextureTarget::Texture2D:
|
||||
@@ -1831,9 +1971,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), levelTexelSize, pData, levelByteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData =
|
||||
PreparePackedNormUpload(textureMipmapObject->GetFormat(), levelTexelSize,
|
||||
uploadData, levelByteSize, &glType, packedUploadData);
|
||||
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
const IntVec3 uploadSize =
|
||||
GetBackendUploadSize(targetInternal, levelTexelSize);
|
||||
switch (MapToBackendTextureTarget(targetInternal)) {
|
||||
@@ -1885,6 +2029,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), levelTexelSize, pData, levelByteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData =
|
||||
PreparePackedNormUpload(textureMipmapObject->GetFormat(), levelTexelSize,
|
||||
uploadData, levelByteSize, &glType, packedUploadData);
|
||||
MGLOG_D("%s: target: %s: syncing mip %d: %dx%dx%d, byteSize = %d, pData = %p, "
|
||||
"levelDirty = %s",
|
||||
__func__, MG_Util::ConvertTextureUploadTargetToString(uploadTarget).c_str(),
|
||||
@@ -1892,7 +2040,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
levelByteSize, pData, levelDirty ? "true" : "false");
|
||||
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
auto textureTarget = stateTextureObject->GetTarget();
|
||||
const IntVec3 uploadSize = GetBackendUploadSize(textureTarget, levelTexelSize);
|
||||
switch (MapToBackendTextureTarget(textureTarget)) {
|
||||
@@ -1978,7 +2126,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
textureMipmapObject->GetMipmapTexelSize(uploadTarget, level).y(), byteSize);
|
||||
|
||||
auto glUploadTarget = ConvertTextureUploadTargetToBackendGLEnum(uploadTarget);
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
DebugImpl::ErrorLopper::Loop(
|
||||
[file = __FILE__, line = __LINE__, func = __func__](GLenum err) {
|
||||
MGLOG_D("%s(%s:%d) ES error: %s", func, file, line,
|
||||
@@ -1990,6 +2138,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), texelSize, mipData, byteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData = PreparePackedNormUpload(textureMipmapObject->GetFormat(), texelSize,
|
||||
uploadData, byteSize, &glType, packedUploadData);
|
||||
const IntVec3 uploadSize =
|
||||
GetBackendUploadSize(stateTextureObject->GetTarget(), texelSize);
|
||||
switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) {
|
||||
@@ -2301,6 +2452,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
|
||||
Uint g_activeTextureUnit = 0;
|
||||
Uint g_textureContextGeneration = 1;
|
||||
Array<Array<BackendTextureObject*, (SizeT)TextureTarget::TextureTargetCount>,
|
||||
MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS>
|
||||
g_boundTexturesCache;
|
||||
@@ -2326,9 +2478,59 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (target == FramebufferTarget::Read)
|
||||
g_GLESFuncs.glBindFramebuffer(GL_READ_FRAMEBUFFER, m_backendFBOId);
|
||||
BindFramebufferId(GL_READ_FRAMEBUFFER, m_backendFBOId);
|
||||
else
|
||||
g_GLESFuncs.glBindFramebuffer(GL_DRAW_FRAMEBUFFER, m_backendFBOId);
|
||||
BindFramebufferId(GL_DRAW_FRAMEBUFFER, m_backendFBOId);
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Driver-level framebuffer-binding shadow (see Managers.h). Indexed by
|
||||
// FramebufferTarget {Draw, Read}.
|
||||
Array<Uint, SizeT(FramebufferTarget::FramebufferTargetCount)> g_driverFBOBindings = {0, 0};
|
||||
Array<Bool, SizeT(FramebufferTarget::FramebufferTargetCount)> g_driverFBOBindingKnown = {false, false};
|
||||
} // namespace
|
||||
|
||||
void BindFramebufferId(GLenum fbTarget, Uint id) {
|
||||
const Bool bindsDraw = fbTarget == GL_DRAW_FRAMEBUFFER || fbTarget == GL_FRAMEBUFFER;
|
||||
const Bool bindsRead = fbTarget == GL_READ_FRAMEBUFFER || fbTarget == GL_FRAMEBUFFER;
|
||||
const SizeT drawIdx = SizeT(FramebufferTarget::Draw);
|
||||
const SizeT readIdx = SizeT(FramebufferTarget::Read);
|
||||
const Bool drawMatches =
|
||||
!bindsDraw || (g_driverFBOBindingKnown[drawIdx] && g_driverFBOBindings[drawIdx] == id);
|
||||
const Bool readMatches =
|
||||
!bindsRead || (g_driverFBOBindingKnown[readIdx] && g_driverFBOBindings[readIdx] == id);
|
||||
if (drawMatches && readMatches) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glBindFramebuffer(fbTarget, id);
|
||||
if (bindsDraw) {
|
||||
g_driverFBOBindings[drawIdx] = id;
|
||||
g_driverFBOBindingKnown[drawIdx] = true;
|
||||
}
|
||||
if (bindsRead) {
|
||||
g_driverFBOBindings[readIdx] = id;
|
||||
g_driverFBOBindingKnown[readIdx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
Uint CurrentFramebufferBinding(FramebufferTarget target) {
|
||||
const SizeT idx = SizeT(target);
|
||||
if (!g_driverFBOBindingKnown[idx]) {
|
||||
// Cold path: pin the shadow from the driver once (init probes and
|
||||
// pre-shadow code bind raw but restore what they found).
|
||||
GLint binding = 0;
|
||||
g_GLESFuncs.glGetIntegerv(
|
||||
target == FramebufferTarget::Read ? GL_READ_FRAMEBUFFER_BINDING : GL_DRAW_FRAMEBUFFER_BINDING,
|
||||
&binding);
|
||||
g_driverFBOBindings[idx] = static_cast<Uint>(binding);
|
||||
g_driverFBOBindingKnown[idx] = true;
|
||||
}
|
||||
return g_driverFBOBindings[idx];
|
||||
}
|
||||
|
||||
void InvalidateFramebufferBindingCache() {
|
||||
g_driverFBOBindings = {0, 0};
|
||||
g_driverFBOBindingKnown = {false, false};
|
||||
}
|
||||
|
||||
void BackendFramebufferObject::InvalidateSyncedState() {
|
||||
@@ -2382,7 +2584,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (glTextureTarget == GL_UNKNOWN_MGL) {
|
||||
glTextureTarget = TextureImpl::ConvertTextureTargetToBackendGLEnum(textureObject->GetTarget());
|
||||
}
|
||||
backendTextureObject->Bind(glTextureTarget);
|
||||
// glBindTexture rejects cube-face enums (INVALID_ENUM with no
|
||||
// bind, while Bind() would still record the cube-map cache slot
|
||||
// as bound): bind via the owning cube target; the attach below
|
||||
// keeps the face target.
|
||||
const Bool isCubeFace = glTextureTarget >= GL_TEXTURE_CUBE_MAP_POSITIVE_X &&
|
||||
glTextureTarget <= GL_TEXTURE_CUBE_MAP_NEGATIVE_Z;
|
||||
backendTextureObject->Bind(isCubeFace ? GL_TEXTURE_CUBE_MAP : glTextureTarget);
|
||||
g_GLESFuncs.glFramebufferTexture2D(glFBOTarget, glBackendAttachment, glTextureTarget,
|
||||
backendTextureObject->GetBackendTextureId(),
|
||||
static_cast<GLint>(attachmentObject.GetTextureLevel()));
|
||||
@@ -2471,6 +2679,28 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return false;
|
||||
}
|
||||
|
||||
void BackendFramebufferObject::SyncReadBufferToBackend(
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject) {
|
||||
if (!stateFBOObject) {
|
||||
return;
|
||||
}
|
||||
auto frontendReadBuf = stateFBOObject->GetReadBuffer();
|
||||
if (frontendReadBuf == m_frontendReadBuffer) {
|
||||
return;
|
||||
}
|
||||
m_frontendReadBuffer = frontendReadBuf;
|
||||
|
||||
GLenum glBackendReadBuffer = GetBackendAttachmentType(frontendReadBuf);
|
||||
if (m_backendReadBuffer != glBackendReadBuffer) {
|
||||
m_backendReadBuffer = glBackendReadBuffer;
|
||||
// glReadBuffer targets whatever FBO is bound to GL_READ_FRAMEBUFFER. When this is
|
||||
// reached from SyncCurrentFBO's "same FBO as draw" skip path the backend FBO was
|
||||
// only bound as DRAW, so bind it as READ first to route the read buffer correctly.
|
||||
Bind(FramebufferTarget::Read);
|
||||
g_GLESFuncs.glReadBuffer(glBackendReadBuffer);
|
||||
}
|
||||
}
|
||||
|
||||
void BackendFramebufferObject::SyncToBackend(
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject, FramebufferTarget asTarget) {
|
||||
#ifdef TRACY_ENABLE
|
||||
@@ -2552,16 +2782,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
// 2. Remap read buffer. glReadBuffer writes the READ-bound FBO's state, so
|
||||
// only apply (and stamp the memo) when this object is bound as READ.
|
||||
auto frontendReadBuf = stateFBOObject->GetReadBuffer();
|
||||
if (frontendReadBuf != m_frontendReadBuffer && asTarget == FramebufferTarget::Read) {
|
||||
m_frontendReadBuffer = frontendReadBuf;
|
||||
|
||||
GLenum glBackendReadBuffer = GetBackendAttachmentType(frontendReadBuf);
|
||||
|
||||
if (m_backendReadBuffer != glBackendReadBuffer) {
|
||||
m_backendReadBuffer = glBackendReadBuffer;
|
||||
g_GLESFuncs.glReadBuffer(glBackendReadBuffer);
|
||||
}
|
||||
if (asTarget == FramebufferTarget::Read) {
|
||||
SyncReadBufferToBackend(stateFBOObject);
|
||||
}
|
||||
|
||||
// -------------------- Attach texture to backend FBO -----------------------
|
||||
@@ -2668,6 +2890,295 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
g_fboSyncedObjects = {};
|
||||
} // namespace FramebufferImpl
|
||||
|
||||
namespace ScratchFBOImpl {
|
||||
namespace {
|
||||
ScratchFramebuffer g_tempFramebuffer;
|
||||
ScratchFramebuffer g_blitReadFramebuffer;
|
||||
ScratchFramebuffer g_blitDrawFramebuffer;
|
||||
Uint g_completeTinyFBOId = 0;
|
||||
Uint g_completeTinyRBOId = 0;
|
||||
|
||||
// Detach every point the shadow no longer vouches for. Used when the
|
||||
// shadow is unknown (context reset, texture id deleted while attached).
|
||||
void ScrubAllAttachments(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
fb.colorTex = 0;
|
||||
fb.colorTarget = 0;
|
||||
fb.colorLevel = 0;
|
||||
fb.colorLayer = -1;
|
||||
fb.depthTex = 0;
|
||||
fb.depthTarget = 0;
|
||||
fb.depthLevel = 0;
|
||||
fb.depthHasStencil = false;
|
||||
fb.attachmentsKnown = true;
|
||||
}
|
||||
|
||||
void PrepareForUse(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
if (!fb.attachmentsKnown) {
|
||||
ScrubAllAttachments(fb, fbTarget);
|
||||
}
|
||||
}
|
||||
|
||||
// The post-attach glGetError probe below must not misread an error some
|
||||
// earlier operation left queued; drain before attaching (rare path -
|
||||
// only runs when the attachment actually changes).
|
||||
void DrainPendingGLErrors() {
|
||||
while (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
}
|
||||
}
|
||||
|
||||
// Record the color point as detached when the shadow said something was
|
||||
// there; the actual detach call is the caller's (it may be replaced by
|
||||
// the new attach directly when the point is being overwritten).
|
||||
void RecordNoColor(ScratchFramebuffer& fb) {
|
||||
fb.colorTex = 0;
|
||||
fb.colorTarget = 0;
|
||||
fb.colorLevel = 0;
|
||||
fb.colorLayer = -1;
|
||||
}
|
||||
|
||||
void RecordNoDepth(ScratchFramebuffer& fb) {
|
||||
fb.depthTex = 0;
|
||||
fb.depthTarget = 0;
|
||||
fb.depthLevel = 0;
|
||||
fb.depthHasStencil = false;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
ScratchFramebuffer& TempFramebuffer() {
|
||||
return g_tempFramebuffer;
|
||||
}
|
||||
ScratchFramebuffer& BlitReadFramebuffer() {
|
||||
return g_blitReadFramebuffer;
|
||||
}
|
||||
ScratchFramebuffer& BlitDrawFramebuffer() {
|
||||
return g_blitDrawFramebuffer;
|
||||
}
|
||||
|
||||
Uint EnsureId(ScratchFramebuffer& fb) {
|
||||
if (fb.id == 0) {
|
||||
g_GLESFuncs.glGenFramebuffers(1, &fb.id);
|
||||
// A fresh FBO has nothing attached and COLOR_ATTACHMENT0 read/draw
|
||||
// buffers (the ES defaults for a non-default framebuffer).
|
||||
fb.attachmentsKnown = true;
|
||||
RecordNoColor(fb);
|
||||
RecordNoDepth(fb);
|
||||
fb.readBuffer = GL_COLOR_ATTACHMENT0;
|
||||
fb.drawBuffer = GL_COLOR_ATTACHMENT0;
|
||||
}
|
||||
return fb.id;
|
||||
}
|
||||
|
||||
void EnsureColorAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget,
|
||||
GLint level) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.depthTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoDepth(fb);
|
||||
}
|
||||
if (fb.colorTex == tex && fb.colorTarget == texTarget && fb.colorLevel == level && fb.colorLayer < 0) {
|
||||
return;
|
||||
}
|
||||
if (fb.colorTex != 0) {
|
||||
// Detach first: if the new attach fails, the point must read as
|
||||
// missing (incomplete FBO), not silently keep the old texture.
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
}
|
||||
DrainPendingGLErrors();
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, texTarget, tex, level);
|
||||
if (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
RecordNoColor(fb);
|
||||
return;
|
||||
}
|
||||
fb.colorTex = tex;
|
||||
fb.colorTarget = texTarget;
|
||||
fb.colorLevel = level;
|
||||
fb.colorLayer = -1;
|
||||
}
|
||||
|
||||
void EnsureColorAttachmentLayer(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLint level, GLint layer) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.depthTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoDepth(fb);
|
||||
}
|
||||
if (fb.colorTex == tex && fb.colorTarget == 0 && fb.colorLevel == level && fb.colorLayer == layer) {
|
||||
return;
|
||||
}
|
||||
if (fb.colorTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
}
|
||||
DrainPendingGLErrors();
|
||||
g_GLESFuncs.glFramebufferTextureLayer(fbTarget, GL_COLOR_ATTACHMENT0, tex, level, layer);
|
||||
if (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
RecordNoColor(fb);
|
||||
return;
|
||||
}
|
||||
fb.colorTex = tex;
|
||||
fb.colorTarget = 0;
|
||||
fb.colorLevel = level;
|
||||
fb.colorLayer = layer;
|
||||
}
|
||||
|
||||
void EnsureDepthAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget, GLint level,
|
||||
Bool withStencil) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.colorTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoColor(fb);
|
||||
}
|
||||
if (fb.depthTex == tex && fb.depthTarget == texTarget && fb.depthLevel == level &&
|
||||
fb.depthHasStencil == withStencil) {
|
||||
return;
|
||||
}
|
||||
if (fb.depthTex != 0) {
|
||||
// One call clears both depth and stencil points regardless of how
|
||||
// the previous attachment was made.
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
}
|
||||
DrainPendingGLErrors();
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget,
|
||||
withStencil ? GL_DEPTH_STENCIL_ATTACHMENT : GL_DEPTH_ATTACHMENT,
|
||||
texTarget, tex, level);
|
||||
if (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
RecordNoDepth(fb);
|
||||
return;
|
||||
}
|
||||
fb.depthTex = tex;
|
||||
fb.depthTarget = texTarget;
|
||||
fb.depthLevel = level;
|
||||
fb.depthHasStencil = withStencil;
|
||||
}
|
||||
|
||||
void EnsureNoColorAttachment(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.colorTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoColor(fb);
|
||||
}
|
||||
}
|
||||
|
||||
void EnsureNoDepthAttachment(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.depthTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoDepth(fb);
|
||||
}
|
||||
}
|
||||
|
||||
void EnsureReadBuffer(ScratchFramebuffer& fb, GLenum readBuffer) {
|
||||
if (fb.readBuffer == readBuffer) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glReadBuffer(readBuffer);
|
||||
fb.readBuffer = readBuffer;
|
||||
}
|
||||
|
||||
void EnsureDrawBuffer(ScratchFramebuffer& fb, GLenum drawBuffer) {
|
||||
if (fb.drawBuffer == drawBuffer) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glDrawBuffers(1, &drawBuffer);
|
||||
fb.drawBuffer = drawBuffer;
|
||||
}
|
||||
|
||||
Uint EnsureCompleteTinyFramebufferId() {
|
||||
if (g_completeTinyFBOId != 0) {
|
||||
return g_completeTinyFBOId;
|
||||
}
|
||||
// One-time creation: the renderbuffer binding is context state with no
|
||||
// shadow, so save/restore it by query here (cold path only).
|
||||
GLint prevRenderbuffer = 0;
|
||||
g_GLESFuncs.glGetIntegerv(GL_RENDERBUFFER_BINDING, &prevRenderbuffer);
|
||||
g_GLESFuncs.glGenFramebuffers(1, &g_completeTinyFBOId);
|
||||
g_GLESFuncs.glGenRenderbuffers(1, &g_completeTinyRBOId);
|
||||
FramebufferImpl::BindFramebufferId(GL_FRAMEBUFFER, g_completeTinyFBOId);
|
||||
g_GLESFuncs.glBindRenderbuffer(GL_RENDERBUFFER, g_completeTinyRBOId);
|
||||
g_GLESFuncs.glRenderbufferStorage(GL_RENDERBUFFER, GL_RGBA8, 1, 1);
|
||||
g_GLESFuncs.glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER,
|
||||
g_completeTinyRBOId);
|
||||
const GLenum drawBuffer = GL_COLOR_ATTACHMENT0;
|
||||
g_GLESFuncs.glDrawBuffers(1, &drawBuffer);
|
||||
g_GLESFuncs.glReadBuffer(GL_COLOR_ATTACHMENT0);
|
||||
MOBILEGL_ASSERT(g_GLESFuncs.glCheckFramebufferStatus(GL_FRAMEBUFFER) == GL_FRAMEBUFFER_COMPLETE,
|
||||
"Scratch 1x1 framebuffer is incomplete.");
|
||||
g_GLESFuncs.glBindRenderbuffer(GL_RENDERBUFFER, static_cast<Uint>(prevRenderbuffer));
|
||||
return g_completeTinyFBOId;
|
||||
}
|
||||
|
||||
void NoteTextureIdDeleted(Uint textureId) {
|
||||
if (textureId == 0) {
|
||||
return;
|
||||
}
|
||||
for (ScratchFramebuffer* fb : {&g_tempFramebuffer, &g_blitReadFramebuffer, &g_blitDrawFramebuffer}) {
|
||||
if (fb->colorTex == textureId || fb->depthTex == textureId) {
|
||||
fb->attachmentsKnown = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OnBackendContextDestroyed() {
|
||||
g_tempFramebuffer = {};
|
||||
g_blitReadFramebuffer = {};
|
||||
g_blitDrawFramebuffer = {};
|
||||
g_completeTinyFBOId = 0;
|
||||
g_completeTinyRBOId = 0;
|
||||
}
|
||||
} // namespace ScratchFBOImpl
|
||||
|
||||
namespace PixelStoreImpl {
|
||||
namespace {
|
||||
PackState g_packState;
|
||||
Bool g_packStateKnown = false;
|
||||
|
||||
void PinPackState(const PackState& value) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ALIGNMENT, value.Alignment);
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ROW_LENGTH, value.RowLength);
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_ROWS, value.SkipRows);
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_PIXELS, value.SkipPixels);
|
||||
g_packState = value;
|
||||
g_packStateKnown = true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void ApplyPackState(const PackState& desired) {
|
||||
if (!g_packStateKnown) {
|
||||
PinPackState(desired);
|
||||
return;
|
||||
}
|
||||
if (desired.Alignment != g_packState.Alignment) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ALIGNMENT, desired.Alignment);
|
||||
g_packState.Alignment = desired.Alignment;
|
||||
}
|
||||
if (desired.RowLength != g_packState.RowLength) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ROW_LENGTH, desired.RowLength);
|
||||
g_packState.RowLength = desired.RowLength;
|
||||
}
|
||||
if (desired.SkipRows != g_packState.SkipRows) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_ROWS, desired.SkipRows);
|
||||
g_packState.SkipRows = desired.SkipRows;
|
||||
}
|
||||
if (desired.SkipPixels != g_packState.SkipPixels) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_PIXELS, desired.SkipPixels);
|
||||
g_packState.SkipPixels = desired.SkipPixels;
|
||||
}
|
||||
}
|
||||
|
||||
PackState CurrentPackState() {
|
||||
if (!g_packStateKnown) {
|
||||
// Fresh/unknown context: pin to the GL defaults (what a new context
|
||||
// starts with; writing them makes the shadow authoritative either way).
|
||||
PinPackState(PackState{});
|
||||
}
|
||||
return g_packState;
|
||||
}
|
||||
|
||||
void InvalidatePackStateCache() {
|
||||
g_packStateKnown = false;
|
||||
}
|
||||
} // namespace PixelStoreImpl
|
||||
|
||||
namespace PrgramImpl {
|
||||
Uint32 g_snormFallbackClampOutputMask = 0;
|
||||
Uint32 g_unormFallbackClampOutputMask = 0;
|
||||
@@ -2790,6 +3301,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
effectiveSpirv = &uboPrecisionSpirv;
|
||||
}
|
||||
|
||||
// noperspective is core desktop GLSL and reaches here as the SPIR-V NoPerspective
|
||||
// decoration. SPIRV-Cross renders it as ESSL `noperspective` + `#extension
|
||||
// GL_NV_shader_noperspective_interpolation : require`; a driver without that extension
|
||||
// rejects the require. So on such devices emulate screen-linear interpolation instead
|
||||
// (pre-multiply outputs by gl_Position.w, recover inputs via gl_FragCoord.w) and drop
|
||||
// the decoration - exact, extension-free. Devices that have the extension keep the
|
||||
// decoration and let the hardware do it natively.
|
||||
Vector<unsigned int> noperspectiveSpirv;
|
||||
if (!g_GLESCapabilities.SupportsNoperspectiveInterpolation &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::EmulateNoPerspectiveForEssl(
|
||||
*effectiveSpirv, noperspectiveSpirv) &&
|
||||
!noperspectiveSpirv.empty()) {
|
||||
effectiveSpirv = &noperspectiveSpirv;
|
||||
}
|
||||
|
||||
MG_Util::ShaderTranspiler::SpvcSession spvcSession(*effectiveSpirv,
|
||||
MG_Util::ShaderTranspiler::SessionUsageBit::Transpile);
|
||||
|
||||
|
||||
@@ -178,6 +178,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// glBindBuffer with a redundant-bind cache for GL_ARRAY_BUFFER.
|
||||
void BindBufferId(GLenum target, Uint id);
|
||||
void InvalidateArrayBufferBindingCache();
|
||||
// Redundant-bind caches for the driver-level GL_PIXEL_PACK/UNPACK_BUFFER
|
||||
// bindings. Every backend readback (glReadPixels / pack-PBO map) and pixel
|
||||
// upload site routes its binding through these so the shadow always matches
|
||||
// the driver; the resting state between operations is 0, which keeps any
|
||||
// path that implicitly assumes "no PBO bound" correct. Scrubbed when a
|
||||
// buffer id is deleted/pooled (GL resets a deleted buffer's bindings to 0,
|
||||
// and a recycled name matching the shadow would false-skip the rebind) and
|
||||
// invalidated on MakeCurrent (context may reset).
|
||||
void BindPixelPackBufferId(Uint id);
|
||||
void BindPixelUnpackBufferId(Uint id);
|
||||
void InvalidatePixelBufferBindingCaches();
|
||||
// A GL buffer id is being deleted by code outside BufferImpl (e.g. the VAO
|
||||
// client-attribute staging buffers): scrub every buffer-binding shadow that
|
||||
// could false-skip when the name is recycled.
|
||||
void NoteBufferIdDeleted(Uint id);
|
||||
// Redundant-bind cache for INDEXED buffer bindings (glBindBufferBase/Range on
|
||||
// GL_UNIFORM_BUFFER / GL_SHADER_STORAGE_BUFFER): skips the GL call when the
|
||||
// (id, range) already at that index matches, like the array-buffer/texture/
|
||||
@@ -332,6 +347,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
class BackendTextureObject {
|
||||
public:
|
||||
BackendTextureObject();
|
||||
// Deletes the GL texture (frontend glDeleteTextures used to leak every
|
||||
// backend id for the context lifetime) and scrubs the binding/scratch-FBO
|
||||
// shadows so a recycled name or heap address cannot false-skip a rebind.
|
||||
~BackendTextureObject();
|
||||
BackendTextureObject(const BackendTextureObject&) = delete;
|
||||
BackendTextureObject& operator=(const BackendTextureObject&) = delete;
|
||||
void SyncMipmapsToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void SyncBuiltinSamplerToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void SyncTextureParamsToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
@@ -343,6 +364,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void RecreateBackendTexture();
|
||||
|
||||
Uint m_backendTextureId = 0;
|
||||
// ES context generation the id was created under; a dtor running after
|
||||
// that context died must not delete a foreign (recycled) name.
|
||||
Uint m_contextGeneration = 0;
|
||||
Bool m_isInitialized = false;
|
||||
Bool m_imageBindableStorageRequired = false;
|
||||
Bool m_backendStorageImmutable = false;
|
||||
@@ -367,6 +391,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS>
|
||||
g_boundTexturesCache;
|
||||
extern Uint g_activeTextureUnit;
|
||||
// Bumped when the backend ES context is destroyed; texture ids stamped with
|
||||
// an older generation belong to a dead context and must not be deleted.
|
||||
extern Uint g_textureContextGeneration;
|
||||
} // namespace TextureImpl
|
||||
|
||||
namespace FramebufferImpl {
|
||||
@@ -375,6 +402,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
BackendFramebufferObject();
|
||||
void SyncToBackend(const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject,
|
||||
FramebufferTarget asTarget);
|
||||
// Apply only this FBO's read buffer (glReadBuffer) to the backend. Split out so it can
|
||||
// still run when SyncCurrentFBO skips the READ-target sync because the same GL FBO is
|
||||
// bound as both draw and read (otherwise glReadBuffer changes would be silently dropped).
|
||||
void SyncReadBufferToBackend(const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject);
|
||||
void InvalidateSyncedState();
|
||||
Uint GetBackendFramebufferId() const { return m_backendFBOId; }
|
||||
void Bind(FramebufferTarget target) const;
|
||||
@@ -414,8 +445,99 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
extern Array<Uint16, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedObjectVersions;
|
||||
extern Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
||||
g_fboSyncedObjects;
|
||||
|
||||
// Driver-level READ/DRAW framebuffer-binding shadow. Every backend
|
||||
// glBindFramebuffer routes through BindFramebufferId so scoped helpers can
|
||||
// save/restore the current binding without a glGetIntegerv round-trip (that
|
||||
// query forces a driver pipeline sync) and so redundant rebinds no-op.
|
||||
// Starts unknown; the first CurrentFramebufferBinding() query pins it from
|
||||
// the driver once. Invalidated on MakeCurrent (context may reset).
|
||||
// GL_FRAMEBUFFER binds both targets.
|
||||
void BindFramebufferId(GLenum fbTarget, Uint id);
|
||||
Uint CurrentFramebufferBinding(FramebufferTarget target);
|
||||
void InvalidateFramebufferBindingCache();
|
||||
} // namespace FramebufferImpl
|
||||
|
||||
// Shared scratch framebuffers for the readback/copy/blit emulation paths, with a
|
||||
// driver-side attachment shadow: repeated uses skip redundant detach/attach GL
|
||||
// calls, and an attachment left by one use (e.g. a depth copy's DEPTH_STENCIL
|
||||
// texture) is detached exactly when a later use of another aspect would
|
||||
// otherwise inherit it (stale cross-aspect attachments made the shared temp FBO
|
||||
// incomplete and silently degraded later readbacks).
|
||||
namespace ScratchFBOImpl {
|
||||
struct ScratchFramebuffer {
|
||||
Uint id = 0;
|
||||
// false => attachment state unknown; scrub every point on next use.
|
||||
// A fresh FBO starts with nothing attached, so creation sets it true.
|
||||
Bool attachmentsKnown = false;
|
||||
Uint colorTex = 0;
|
||||
GLenum colorTarget = 0;
|
||||
GLint colorLevel = 0;
|
||||
GLint colorLayer = -1; // >= 0 => attached via glFramebufferTextureLayer
|
||||
Uint depthTex = 0;
|
||||
GLenum depthTarget = 0;
|
||||
GLint depthLevel = 0;
|
||||
Bool depthHasStencil = false;
|
||||
// Per-FBO read/draw buffer state (0 = unknown, set on first use).
|
||||
GLenum readBuffer = 0;
|
||||
GLenum drawBuffer = 0;
|
||||
};
|
||||
ScratchFramebuffer& TempFramebuffer(); // GetTexImage READ / CopyTex*Image2D depth DRAW
|
||||
ScratchFramebuffer& BlitReadFramebuffer(); // texture-to-texture blit source
|
||||
ScratchFramebuffer& BlitDrawFramebuffer(); // texture-to-texture blit destination
|
||||
// Returns the GL id, generating it if needed (requires a current ES context).
|
||||
Uint EnsureId(ScratchFramebuffer& fb);
|
||||
// The fb must currently be bound at fbTarget (glReadBuffer/glDrawBuffers
|
||||
// target the READ/DRAW binding respectively). Each Ensure* performs the
|
||||
// minimal detach/attach set and keeps the shadow in sync; a failed attach
|
||||
// records the point as detached so the completeness check fails instead of
|
||||
// silently reading a stale attachment.
|
||||
void EnsureColorAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget, GLint level);
|
||||
void EnsureColorAttachmentLayer(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLint level, GLint layer);
|
||||
void EnsureDepthAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget, GLint level,
|
||||
Bool withStencil);
|
||||
void EnsureNoColorAttachment(ScratchFramebuffer& fb, GLenum fbTarget);
|
||||
void EnsureNoDepthAttachment(ScratchFramebuffer& fb, GLenum fbTarget);
|
||||
void EnsureReadBuffer(ScratchFramebuffer& fb, GLenum readBuffer);
|
||||
void EnsureDrawBuffer(ScratchFramebuffer& fb, GLenum drawBuffer);
|
||||
// A 1x1 RGBA8-renderbuffer-complete FBO (GenerateMipmap needs a complete
|
||||
// binding while respecifying texture storage). Attachment is set once at
|
||||
// creation and never changes.
|
||||
Uint EnsureCompleteTinyFramebufferId();
|
||||
// A backend texture id is being deleted or respecified: a scratch FBO still
|
||||
// referencing it would hold a dangling attachment (ES only auto-detaches
|
||||
// from the *bound* framebuffer), and a recycled name could false-skip a
|
||||
// re-attach; force a full scrub on next use.
|
||||
void NoteTextureIdDeleted(Uint textureId);
|
||||
// The ES context (and the scratch FBO ids with it) is going away.
|
||||
void OnBackendContextDestroyed();
|
||||
} // namespace ScratchFBOImpl
|
||||
|
||||
// Driver-level GL_PACK_* pixel-store shadow, the readback-side sibling of the
|
||||
// upload path's ScopedDefaultUnpackState (Managers.cpp): the backend PACK state
|
||||
// is written ONLY through ApplyPackState, so scoped helpers can save/restore it
|
||||
// from the shadow instead of glGetIntegerv (which forces a driver pipeline
|
||||
// sync), and redundant glPixelStorei calls no-op. The first Apply/Current call
|
||||
// pins the driver to the shadow by writing all fields once. Invalidated on
|
||||
// MakeCurrent (context may reset). PACK_IMAGE_HEIGHT/SKIP_IMAGES/SWAP_BYTES/
|
||||
// LSB_FIRST have no ES equivalents; readbacks honor them on the CPU from the
|
||||
// frontend context state instead.
|
||||
namespace PixelStoreImpl {
|
||||
struct PackState {
|
||||
GLint Alignment = 4;
|
||||
GLint RowLength = 0;
|
||||
GLint SkipRows = 0;
|
||||
GLint SkipPixels = 0;
|
||||
Bool operator==(const PackState& o) const {
|
||||
return Alignment == o.Alignment && RowLength == o.RowLength && SkipRows == o.SkipRows &&
|
||||
SkipPixels == o.SkipPixels;
|
||||
}
|
||||
};
|
||||
void ApplyPackState(const PackState& desired);
|
||||
PackState CurrentPackState();
|
||||
void InvalidatePackStateCache();
|
||||
} // namespace PixelStoreImpl
|
||||
|
||||
// Image uniforms take their unit from the layout(binding=N) qualifier baked into
|
||||
// the transpiled ESSL; unlike samplers they must not (and in ES cannot) be
|
||||
// assigned through glUniform1i.
|
||||
|
||||
@@ -764,5 +764,95 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static SizeT AlignReadbackRow(SizeT rowBytes, Int alignment) {
|
||||
const SizeT align = alignment > 0 ? static_cast<SizeT>(alignment) : 1;
|
||||
return (rowBytes + align - 1) / align * align;
|
||||
}
|
||||
|
||||
// Repacks wide RGBA(_INTEGER) rows into the client's (format, type) layout, honoring the
|
||||
// client-side PACK parameters and the bound pixel-pack buffer. `wide` holds
|
||||
// `sliceHeight * sliceCount` rows of `width` texels (slice-major, tightly stacked),
|
||||
// 4 components x GetReadbackComponentSize(wideType) bytes each.
|
||||
// applyPackImageParams: GL_PACK_IMAGE_HEIGHT / GL_PACK_SKIP_IMAGES apply only to GetTexImage
|
||||
// of 3D/array images; ReadPixels and 2D GetTexImage ignore them (GL 3.3 sections 4.3.1, 6.1.4).
|
||||
// Per the GL addressing rules, slice k row j lands at
|
||||
// SKIP_IMAGES*imageStride + SKIP_ROWS*rowStride + SKIP_PIXELS*pixelBytes
|
||||
// + k*imageStride + j*rowStride, with imageStride = max(IMAGE_HEIGHT, sliceHeight)*rowStride.
|
||||
Bool StoreWideRowsToClient(const Uint8* wide, GLenum wideType, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, const ReadbackChannelMapping& mapping, GLenum type,
|
||||
void* pixels, Bool applyPackImageParams) {
|
||||
const SizeT dstPixelBytes = GetReadbackDstPixelSize(mapping, type);
|
||||
if (dstPixelBytes == 0) {
|
||||
return false;
|
||||
}
|
||||
PackedReadbackLayout packedLayout{};
|
||||
const Bool isPackedType = GetPackedReadbackLayout(type, packedLayout);
|
||||
const SizeT dstComponentSize = GetReadbackComponentSize(type);
|
||||
|
||||
const auto& pixelPackBufferObject =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::PixelPack).GetBoundObject();
|
||||
|
||||
// Destination layout is computed from the client-side PACK parameters; only the actual pixel
|
||||
// rows are written so skip regions of the destination stay untouched.
|
||||
const auto packParams = MG_State::pGLContext->GetPixelStoreParameters(false);
|
||||
const SizeT rowPixels = static_cast<SizeT>(packParams.RowLength > 0 ? packParams.RowLength : width);
|
||||
const SizeT dstRowStride = AlignReadbackRow(rowPixels * dstPixelBytes, packParams.Alignment);
|
||||
const SizeT imageRows =
|
||||
applyPackImageParams && packParams.ImageHeight > 0
|
||||
? static_cast<SizeT>(packParams.ImageHeight)
|
||||
: static_cast<SizeT>(sliceHeight);
|
||||
const SizeT dstImageStride = imageRows * dstRowStride;
|
||||
const SizeT skipImages =
|
||||
applyPackImageParams ? static_cast<SizeT>(std::max(packParams.SkipImages, 0)) : SizeT{0};
|
||||
const SizeT dstSkipOffset = skipImages * dstImageStride +
|
||||
static_cast<SizeT>(std::max(packParams.SkipRows, 0)) * dstRowStride +
|
||||
static_cast<SizeT>(std::max(packParams.SkipPixels, 0)) * dstPixelBytes;
|
||||
const SizeT dstRowBytes = static_cast<SizeT>(width) * dstPixelBytes;
|
||||
|
||||
const SizeT pboBaseOffset = reinterpret_cast<SizeT>(pixels); // with a PBO, `pixels` is an offset
|
||||
if (pixelPackBufferObject) {
|
||||
const SizeT requiredSize = pboBaseOffset + dstSkipOffset +
|
||||
static_cast<SizeT>(sliceCount - 1) * dstImageStride +
|
||||
static_cast<SizeT>(sliceHeight - 1) * dstRowStride + dstRowBytes;
|
||||
if (requiredSize > pixelPackBufferObject->GetSize()) {
|
||||
MGLOG_E("Readback conversion: pixel pack buffer is too small");
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT srcComponentSize = GetReadbackComponentSize(wideType);
|
||||
const SizeT srcPixelBytes = 4 * srcComponentSize;
|
||||
Vector<Uint8> convertedRow(dstRowBytes);
|
||||
|
||||
for (GLsizei slice = 0; slice < sliceCount; ++slice) {
|
||||
for (GLsizei row = 0; row < sliceHeight; ++row) {
|
||||
const SizeT flatRow = static_cast<SizeT>(slice) * static_cast<SizeT>(sliceHeight) +
|
||||
static_cast<SizeT>(row);
|
||||
const Uint8* srcRow = wide + flatRow * static_cast<SizeT>(width) * srcPixelBytes;
|
||||
ConvertWideReadbackRow(srcRow, convertedRow.data(), static_cast<SizeT>(width), wideType,
|
||||
mapping, type);
|
||||
|
||||
if (packParams.SwapBytes) {
|
||||
const SizeT groupSize = isPackedType ? packedLayout.byteSize : dstComponentSize;
|
||||
if (groupSize > 1) {
|
||||
for (SizeT offset = 0; offset + groupSize <= dstRowBytes; offset += groupSize) {
|
||||
std::reverse(convertedRow.data() + offset, convertedRow.data() + offset + groupSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT dstOffset = dstSkipOffset + static_cast<SizeT>(slice) * dstImageStride +
|
||||
static_cast<SizeT>(row) * dstRowStride;
|
||||
if (pixelPackBufferObject) {
|
||||
pixelPackBufferObject->WritebackFromBackend({convertedRow.data(), dstRowBytes},
|
||||
pboBaseOffset + dstOffset);
|
||||
} else {
|
||||
Memcpy(static_cast<Uint8*>(pixels) + dstOffset, convertedRow.data(), dstRowBytes);
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
} // namespace ReadbackImpl
|
||||
} // namespace MobileGL::MG_Backend::DirectGLES
|
||||
|
||||
@@ -88,6 +88,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// bytes, dst receives width * GetReadbackDstPixelSize(mapping, type) bytes.
|
||||
void ConvertWideReadbackRow(const Uint8* src, Uint8* dst, SizeT width, GLenum wideType,
|
||||
const ReadbackChannelMapping& mapping, GLenum type);
|
||||
|
||||
// Stores wide RGBA(_INTEGER) rows into the client pointer or the bound PACK pixel buffer,
|
||||
// honoring the client-side PACK pixel-store parameters (row length, alignment, skips,
|
||||
// swap-bytes, and - when applyPackImageParams - image height/skip images). Shared by the
|
||||
// DirectGLES and DirectVulkan readback conversion paths.
|
||||
Bool StoreWideRowsToClient(const Uint8* wide, GLenum wideType, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, const ReadbackChannelMapping& mapping, GLenum type,
|
||||
void* pixels, Bool applyPackImageParams);
|
||||
} // namespace ReadbackImpl
|
||||
|
||||
namespace PrgramImpl {
|
||||
|
||||
@@ -140,6 +140,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
// Legacy low-bit-depth formats with no (or rarely supported) native Vulkan
|
||||
// encoding; a wider normalized fallback keeps at least the required precision.
|
||||
case TextureInternalFormat::R3G3B2:
|
||||
case TextureInternalFormat::RGB4:
|
||||
case TextureInternalFormat::RGB5:
|
||||
case TextureInternalFormat::RGBA2:
|
||||
case TextureInternalFormat::RGBA4:
|
||||
case TextureInternalFormat::RGB5A1:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
case TextureInternalFormat::RGB10:
|
||||
return TextureInternalFormat::RGB10A2;
|
||||
case TextureInternalFormat::RGB12:
|
||||
case TextureInternalFormat::RGBA12:
|
||||
return TextureInternalFormat::RGBA16;
|
||||
case TextureInternalFormat::SRGB8:
|
||||
return TextureInternalFormat::SRGB8Alpha8;
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
@@ -397,8 +411,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
if (!handle.Handle || (handle.Backend != WindowBackend::Android &&
|
||||
handle.Backend != WindowBackend::X11 &&
|
||||
handle.Backend != WindowBackend::MetalLayer)) {
|
||||
MGLOG_E("DirectVulkan backend only supports Android, X11, and CAMetalLayer native windows");
|
||||
handle.Backend != WindowBackend::MetalLayer &&
|
||||
handle.Backend != WindowBackend::Win32)) {
|
||||
MGLOG_E("DirectVulkan backend only supports Android, X11, CAMetalLayer, and Win32 native windows");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -454,6 +469,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// treat them as signaled/available with zero results from here on.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer.reset();
|
||||
// The reflection cache is file-scope, not renderer-owned; without this the
|
||||
// deleted programs' reflection strings survive full context teardown.
|
||||
ClearProgramResourceCaches();
|
||||
BackendObject::ReleaseEGLResources();
|
||||
}
|
||||
|
||||
@@ -463,6 +481,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// treat them as signaled/available with zero results from here on.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer.reset();
|
||||
// The reflection cache is file-scope, not renderer-owned; without this the
|
||||
// deleted programs' reflection strings survive full context teardown.
|
||||
ClearProgramResourceCaches();
|
||||
}
|
||||
|
||||
const RendererInfo& BackendObject_DirectVulkan::GetRendererInfo() const {
|
||||
|
||||
@@ -20,7 +20,8 @@
|
||||
#include <spirv_reflect.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
UniquePtr<VulkanRenderer> pVulkanRenderer = nullptr;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<VulkanRenderer>& pVulkanRenderer = *new UniquePtr<VulkanRenderer>();
|
||||
|
||||
namespace {
|
||||
// Generation of the live VulkanRenderer instance, mirroring
|
||||
@@ -60,6 +61,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
|
||||
struct ProgramResourceCache {
|
||||
// Lifetime id of the program the cached reflection belongs to. GL names are
|
||||
// recycled (IndexGenerator hands freed indices straight back), and a
|
||||
// recreated program's backendStateVersion restarts at the same small values,
|
||||
// so the version alone can collide; the never-reused lifetime id makes the
|
||||
// slot's ownership unambiguous.
|
||||
Uint64 programLifetimeId = 0;
|
||||
Uint32 backendStateVersion = 0;
|
||||
Vector<StorageBlockResource> storageBlocks;
|
||||
Vector<BufferVariableResource> bufferVariables;
|
||||
@@ -81,6 +88,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 baseInstance = 0;
|
||||
};
|
||||
|
||||
// Keyed by GL program name so the freed-name reuse in IndexGenerator bounds the
|
||||
// map at the peak-simultaneous-program high-water mark; each slot's ownership is
|
||||
// checked against the program's lifetime id before it is served (see
|
||||
// GetProgramResourceCache). Cleared wholesale at EGL teardown via
|
||||
// ClearProgramResourceCaches.
|
||||
UnorderedMap<GLuint, ProgramResourceCache> g_programResourceCaches;
|
||||
|
||||
void ClearReadPixelsOutput(GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels) {
|
||||
@@ -141,13 +153,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
ProgramResourceCache& GetProgramResourceCache(const MG_State::GLState::ProgramObject& program) {
|
||||
auto& cache = g_programResourceCaches[program.GetExternalIndex()];
|
||||
const Uint64 programLifetimeId = program.GetLifetimeId();
|
||||
const Uint32 backendStateVersion = program.GetBackendStateVersion();
|
||||
if (cache.backendStateVersion == backendStateVersion &&
|
||||
// The lifetime id must match too: a new program that reuses a deleted
|
||||
// program's name and happens to land on the same backendStateVersion (both
|
||||
// count from zero) would otherwise be served the dead program's reflection.
|
||||
if (cache.programLifetimeId == programLifetimeId &&
|
||||
cache.backendStateVersion == backendStateVersion &&
|
||||
(!cache.storageBlocks.empty() || !cache.bufferVariables.empty())) {
|
||||
return cache;
|
||||
}
|
||||
|
||||
cache = {};
|
||||
cache.programLifetimeId = programLifetimeId;
|
||||
cache.backendStateVersion = backendStateVersion;
|
||||
|
||||
Vector<SpvReflectShaderModule> modules;
|
||||
@@ -365,6 +383,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void ClearProgramResourceCaches() {
|
||||
// Called from EGL teardown while the backend's m_eglStateMutex is held; GL
|
||||
// calls are serialized in this codebase (contexts migrate threads but never
|
||||
// run concurrently), so no other thread can be inside the unsynchronized map.
|
||||
// Live programs in another context self-heal: their entry rebuilds from the
|
||||
// retained generated SPIR-V on the next resource query.
|
||||
g_programResourceCaches.clear();
|
||||
}
|
||||
|
||||
GLuint GetShaderStorageBlockIndex(const MG_State::GLState::ProgramObject& program, const String& name) {
|
||||
auto& cache = GetProgramResourceCache(program);
|
||||
const auto it = std::find_if(cache.storageBlocks.begin(), cache.storageBlocks.end(),
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
#include "Renderer/VulkanRenderer.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
extern UniquePtr<VulkanRenderer> pVulkanRenderer;
|
||||
extern UniquePtr<VulkanRenderer>& pVulkanRenderer;
|
||||
|
||||
// Generation of the live VulkanRenderer instance, mirroring DirectGLES's
|
||||
// g_syncContextGeneration. BackendObject_DirectVulkan bumps it wherever
|
||||
@@ -23,6 +23,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 GetRendererGeneration();
|
||||
void BumpRendererGeneration();
|
||||
|
||||
// Drops every cached program-resource reflection entry (CPU-side strings/vectors
|
||||
// only, no Vulkan handles). Called at EGL teardown next to the renderer reset;
|
||||
// safe because GL calls are serialized in this codebase, and any still-live
|
||||
// program rebuilds its entry from the retained generated SPIR-V on demand.
|
||||
void ClearProgramResourceCaches();
|
||||
|
||||
void ClearBufferfi(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void ClearBufferfv(GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void ClearBufferuiv(GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
|
||||
@@ -150,12 +150,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
Bool FrameContext::TransitionToPresent(VkImage image, VkImageLayout oldLayout, VkImageLayout presentLayout) {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.hasCommandBufferRecorded || frame.isCommandRecording || oldLayout == presentLayout ||
|
||||
oldLayout == VK_IMAGE_LAYOUT_SHARED_PRESENT_KHR) {
|
||||
if (oldLayout == presentLayout || oldLayout == VK_IMAGE_LAYOUT_SHARED_PRESENT_KHR) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& commandBuffer = BeginCommandRecording();
|
||||
// The barrier belongs in the frame's own recording. Bailing out because
|
||||
// something was already recorded (the previous behaviour) dropped the
|
||||
// transition entirely for every frame that never ran a default-framebuffer
|
||||
// render pass - the only other thing that carries the image to
|
||||
// PRESENT_SRC_KHR, via that pass's finalLayout - so the swapchain image was
|
||||
// handed to the WSI still in the layout it was acquired in.
|
||||
// A closed-but-unsubmitted buffer can only come from a submit that already
|
||||
// failed (SubmitPendingCommandBuffer leaves the flag set on error), and
|
||||
// appending to it is illegal while reopening would reset the frame's own
|
||||
// commands away. The device is gone on that path anyway - stay silent-safe
|
||||
// rather than trade a lost device for a barrier into a closed buffer.
|
||||
if (frame.hasCommandBufferRecorded) {
|
||||
MGLOG_E("TransitionToPresent: command buffer already closed; skipping the present barrier");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Reopening a recording here would vkResetCommandBuffer this frame's own
|
||||
// commands away, so append to the open one and let the caller close it.
|
||||
const Bool openedRecording = !frame.isCommandRecording;
|
||||
VkCommandBuffer commandBuffer = openedRecording ? BeginCommandRecording() : frame.commandBuffer;
|
||||
|
||||
VkImageMemoryBarrier presentBarrier{};
|
||||
presentBarrier.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
@@ -174,7 +192,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0,
|
||||
nullptr, 0, nullptr, 1, &presentBarrier);
|
||||
|
||||
EndCommandRecording();
|
||||
if (openedRecording) {
|
||||
EndCommandRecording();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -227,12 +247,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
result = vkAcquireNextImageKHR(device, swapchain, timeout, frame.imageAvailableSemaphore, acquireFence,
|
||||
&outImageIndex);
|
||||
if (result != VK_SUCCESS) {
|
||||
// VK_SUBOPTIMAL_KHR is a success code: an image *was* acquired and
|
||||
// imageAvailableSemaphore *will* be signaled. Bailing out on it skipped both
|
||||
// the consumed-flag reset (leaving a stale "already consumed", so the next
|
||||
// submit never waited on the pending signal) and the fence reset (leaving
|
||||
// the slot's fence signaled for the next submit to reuse). Only a genuine
|
||||
// failure - VK_ERROR_OUT_OF_DATE_KHR and friends, where nothing is acquired
|
||||
// and nothing is signaled - skips the bookkeeping.
|
||||
if (result != VK_SUCCESS && result != VK_SUBOPTIMAL_KHR) {
|
||||
return result;
|
||||
}
|
||||
|
||||
frame.imageAvailableSemaphoreConsumed = false;
|
||||
return vkResetFences(device, 1, &frame.imageInFlightFence);
|
||||
const VkResult resetResult = vkResetFences(device, 1, &frame.imageInFlightFence);
|
||||
// Hand the acquire's own code back so the caller can schedule a rebuild.
|
||||
return resetResult == VK_SUCCESS ? result : resetResult;
|
||||
}
|
||||
|
||||
Uint32 FrameContext::GetCurrentFrameIndex() const {
|
||||
@@ -264,7 +293,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
frame.retiredCommandBuffers.push_back(frame.commandBuffer);
|
||||
// lastSubmitIndex was just written by the renderer for the submission
|
||||
// that carried this command buffer.
|
||||
frame.retiredCommandBuffers.push_back({frame.commandBuffer, frame.lastSubmitIndex});
|
||||
frame.commandBuffer = replacement;
|
||||
return VK_SUCCESS;
|
||||
}
|
||||
@@ -274,12 +305,40 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && m_commandPool != VK_NULL_HANDLE) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, static_cast<Uint32>(frame.retiredCommandBuffers.size()),
|
||||
frame.retiredCommandBuffers.data());
|
||||
for (const auto& retired : frame.retiredCommandBuffers) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &retired.commandBuffer);
|
||||
}
|
||||
}
|
||||
frame.retiredCommandBuffers.clear();
|
||||
}
|
||||
|
||||
void FrameContext::FreeRetiredCommandBuffersCompletedUpTo(Uint64 completedSubmitIndex) {
|
||||
if (m_device == VK_NULL_HANDLE || m_commandPool == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto& frame : m_frames) {
|
||||
// Retired buffers are appended in submit order, so the completed
|
||||
// ones form a prefix.
|
||||
SizeT completedCount = 0;
|
||||
while (completedCount < frame.retiredCommandBuffers.size() &&
|
||||
frame.retiredCommandBuffers[completedCount].submitIndex <= completedSubmitIndex) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1,
|
||||
&frame.retiredCommandBuffers[completedCount].commandBuffer);
|
||||
++completedCount;
|
||||
}
|
||||
if (completedCount > 0) {
|
||||
frame.retiredCommandBuffers.erase(frame.retiredCommandBuffers.begin(),
|
||||
frame.retiredCommandBuffers.begin() + completedCount);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void FrameContext::FreeAllRetiredCommandBuffers() {
|
||||
for (auto& frame : m_frames) {
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
}
|
||||
}
|
||||
|
||||
void FrameContext::AssertValidFrameIndex(Uint32 frameIndex) const {
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(), "FrameContext index out of range");
|
||||
}
|
||||
|
||||
@@ -40,6 +40,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPresentInfoKHR presentInfo{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR};
|
||||
};
|
||||
|
||||
// A command buffer submitted mid-frame (FlushPendingCommands), tagged
|
||||
// with the submit-tracker index it was submitted under so it can be
|
||||
// freed as soon as that submission is observed complete - without
|
||||
// waiting for the slot's fence to be waited again (present-less flush
|
||||
// loops never wait it).
|
||||
struct RetiredCommandBuffer {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
Uint64 submitIndex = 0;
|
||||
};
|
||||
|
||||
struct FrameData {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
VkSemaphore imageAvailableSemaphore = VK_NULL_HANDLE;
|
||||
@@ -47,10 +57,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool isCommandRecording = false;
|
||||
Bool hasCommandBufferRecorded = false;
|
||||
Bool imageAvailableSemaphoreConsumed = false;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands) whose
|
||||
// execution is only known complete once this slot's fence has been
|
||||
// waited again; freed at that point.
|
||||
Vector<VkCommandBuffer> retiredCommandBuffers;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands),
|
||||
// appended in submit order; freed once their submission is known
|
||||
// complete (fence wait or completion poll).
|
||||
Vector<RetiredCommandBuffer> retiredCommandBuffers;
|
||||
// Submit-tracker index of this slot's most recent queue submission
|
||||
// (written by the renderer at submit time).
|
||||
Uint64 lastSubmitIndex = 0;
|
||||
@@ -79,9 +89,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Parks the current (already ended and submitted) command buffer on the
|
||||
// slot's retired list and installs a freshly allocated one, so recording
|
||||
// can restart while the submitted buffer is still executing. Retired
|
||||
// buffers are freed after the slot's fence is next waited.
|
||||
// buffers are freed after the slot's fence is next waited, or as soon
|
||||
// as their submission is observed complete.
|
||||
VkResult RetireCurrentCommandBuffer();
|
||||
|
||||
// Frees every retired command buffer whose tagged submission index is
|
||||
// known complete. Driven by the renderer's submit tracker on completion
|
||||
// events (fence waits and non-blocking polls), so present-less flush
|
||||
// loops reclaim their buffers without any extra wait.
|
||||
void FreeRetiredCommandBuffersCompletedUpTo(Uint64 completedSubmitIndex);
|
||||
// Frees every slot's retired command buffers. Only valid when the
|
||||
// caller has proven every queue submission complete.
|
||||
void FreeAllRetiredCommandBuffers();
|
||||
|
||||
Uint32 GetCurrentFrameIndex() const;
|
||||
Uint32 GetFrameCount() const;
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "PipelineFactory.h"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static const char* PrimitiveTopologyToString(VkPrimitiveTopology topology) {
|
||||
@@ -131,25 +132,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Order-independent accumulation blending: the write order of overlapping fragments
|
||||
// does not change the result, which is what lets multi-pass chains re-rasterize the
|
||||
// same geometry and combine per-pass contributions (MC 26.3 OIT: GL_MAX depth
|
||||
// bounds, additive ONE+ONE transmittance/accumulate). Sorted-transparency "over"
|
||||
// compositing (SRC_ALPHA-style factors) is order-dependent, drawn once per surface,
|
||||
// and relies on its depth writes for occlusion - it must not be treated as hazardous.
|
||||
// MIN/MAX ignore blend factors entirely per the Vulkan spec.
|
||||
// MIN/MAX extremum blending: the signature of a depth-bounds accumulation pass
|
||||
// (MC 26.3 OIT writes vec4(-linD, linD, deviceZ, 0) under GL_MAX while writing
|
||||
// depth for its equality chain). MIN/MAX ignore blend factors per the Vulkan spec.
|
||||
//
|
||||
// Deliberately color-channel only. A separate-alpha accumulation
|
||||
// (glBlendEquationSeparate(GL_FUNC_ADD, GL_MAX)) whose color channel is an ordinary
|
||||
// over-blend is not treated as hazardous: no known content pairs that shape with a
|
||||
// depth-equality chain, and widening the test would re-capture sorted transparency.
|
||||
// Deliberately the ONLY shape stripped. A quirk should touch as little unrelated
|
||||
// content as possible, and a trace sweep of every fixture showed the wider
|
||||
// alternatives all cost more than they fix:
|
||||
// - additive ONE+ONE with a depth write matched zero draws of the 26.3 chain
|
||||
// (its transmittance/accumulate passes disable depth writes themselves) - the
|
||||
// only real content it caught was harmless additive glow effects (Create);
|
||||
// - sorted-transparency "over" blends (SRC_ALPHA-style) are order-dependent,
|
||||
// drawn once per surface, and rely on their depth writes for occlusion;
|
||||
// - separate-alpha accumulation over an over-blending color channel has no
|
||||
// known pairing with a depth-equality chain (color channel only, see tests).
|
||||
// If a future workload pairs another blend shape with an equality chain, widen
|
||||
// this with that evidence in hand rather than pre-emptively.
|
||||
Bool IsAccumulationBlend(const VkPipelineColorBlendAttachmentState& attachment) {
|
||||
if (attachment.colorBlendOp == VK_BLEND_OP_MIN || attachment.colorBlendOp == VK_BLEND_OP_MAX) {
|
||||
return true;
|
||||
}
|
||||
return attachment.colorBlendOp == VK_BLEND_OP_ADD &&
|
||||
attachment.srcColorBlendFactor == VK_BLEND_FACTOR_ONE &&
|
||||
attachment.dstColorBlendFactor == VK_BLEND_FACTOR_ONE;
|
||||
return attachment.colorBlendOp == VK_BLEND_OP_MIN ||
|
||||
attachment.colorBlendOp == VK_BLEND_OP_MAX;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
@@ -243,23 +244,108 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const HashType hash = ComputeHash(payload);
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
return it->second;
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
return it->second.pipeline;
|
||||
}
|
||||
|
||||
VkPipeline pipeline = CreatePipeline(payload);
|
||||
m_cache.emplace(hash, pipeline);
|
||||
m_cache.emplace(hash, PipelineCacheEntry{pipeline, payload.programHash, payload.renderPass,
|
||||
m_frameCounter});
|
||||
return pipeline;
|
||||
}
|
||||
|
||||
void PipelineFactory::DestroyAll() {
|
||||
for (auto& pair : m_cache) {
|
||||
if (pair.second != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, pair.second, nullptr);
|
||||
if (pair.second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, pair.second.pipeline, nullptr);
|
||||
}
|
||||
}
|
||||
m_cache.clear();
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::OnFrameBoundary() {
|
||||
++m_frameCounter;
|
||||
|
||||
// Sweep cadence and retire age mirror VkRenderPassManager::OnPresent: an entry
|
||||
// idle for more than kRetireAgeFrames frame boundaries cannot be referenced by
|
||||
// any in-flight command buffer (frames-in-flight <= MOBILEGL_MAGMA_FRAMESINFLIGHT),
|
||||
// so immediate vkDestroyPipeline is safe. The caller must drop its "last
|
||||
// pipeline" memo when this returns non-zero: the memo can return a cached
|
||||
// handle without touching this cache, so an evicted pipeline may still be
|
||||
// memoized (present-less flush loops never reset the memo per frame).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeFrames = 1024;
|
||||
if ((m_frameCounter % kSweepInterval) != 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameCounter - it->second.lastUsedFrame > kRetireAgeFrames) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::OnFrameBoundary: evicted %u idle pipelines (%zu remain)", evicted,
|
||||
m_cache.size());
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::EvictByRenderPasses(const Vector<VkRenderPass>& renderPasses) {
|
||||
if (renderPasses.empty() || m_cache.empty()) {
|
||||
return 0;
|
||||
}
|
||||
// Sorted-batch membership test keeps a mass eviction (shader-pack switch,
|
||||
// dimension exit) at one O(cache * log batch) scan instead of one full scan
|
||||
// per dying pass.
|
||||
Vector<VkRenderPass> sortedPasses = renderPasses;
|
||||
std::sort(sortedPasses.begin(), sortedPasses.end());
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (std::binary_search(sortedPasses.begin(), sortedPasses.end(), it->second.renderPass)) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::EvictByRenderPasses: evicted %u pipelines for %zu destroyed render passes",
|
||||
evicted, sortedPasses.size());
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::EvictByProgramHash(HashType programHash) {
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (it->second.programHash == programHash) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::EvictByProgramHash: evicted %u pipelines for program hash 0x%llx",
|
||||
evicted, static_cast<unsigned long long>(programHash));
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
VkPipeline PipelineFactory::CreatePipeline(const PipelineCreatePayload& payload) const {
|
||||
MOBILEGL_ASSERT(payload.stages != nullptr && !payload.stages->empty(), "PipelineFactory: stages are empty");
|
||||
MOBILEGL_ASSERT(payload.vertexInputState != nullptr, "PipelineFactory: vertexInputState is null");
|
||||
|
||||
@@ -65,16 +65,36 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPipeline GetOrCreatePipeline(const PipelineCreatePayload& payload);
|
||||
void DestroyAll();
|
||||
|
||||
// Frame boundary hook: ages the pipeline cache and destroys long-unused entries
|
||||
// (their command buffers retired many frames ago), mirroring
|
||||
// VkRenderPassManager::OnPresent's sweep. Returns the number of pipelines
|
||||
// destroyed so the caller can drop any memoized VkPipeline handle.
|
||||
Uint32 OnFrameBoundary();
|
||||
// Destroys every cached pipeline hashed on one of `renderPasses`. Only safe
|
||||
// when the caller guarantees GPU idleness for them - the render-pass manager
|
||||
// calls this (via the renderer) for passes its own >1024-boundary-idle sweep
|
||||
// just evicted, and a pipeline hashed on those handles is only ever bound by
|
||||
// draws that also hit the render-pass entries. Also closes the handle-recycling
|
||||
// hazard: a recycled VkRenderPass value must never serve a stale pipeline.
|
||||
// Batched: one cache scan regardless of how many passes died in the sweep.
|
||||
// Returns the number destroyed (callers invalidate memos when non-zero).
|
||||
Uint32 EvictByRenderPasses(const Vector<VkRenderPass>& renderPasses);
|
||||
// Destroys every cached pipeline built from the program with content hash
|
||||
// `programHash`. Called from the ProgramFactory eviction path, which proves the
|
||||
// same >1024-boundary idleness (the program's pipelines are only bound by draws
|
||||
// that stamp its factory entry). Returns the number destroyed.
|
||||
Uint32 EvictByProgramHash(HashType programHash);
|
||||
|
||||
// Driver quirk: suppress depth writes on accumulation-blended pipelines. Multi-pass
|
||||
// depth-equality rendering (a blended prepass writes depth that later passes re-test
|
||||
// with an equality-inclusive compare on the re-rasterized geometry) requires
|
||||
// cross-pipeline position invariance that some mobile compilers do not provide, even
|
||||
// with the SPIR-V Invariant decoration; whole primitives then drop out of the later
|
||||
// passes. Only order-independent accumulation blends (MIN/MAX, additive ONE+ONE) are
|
||||
// stripped - that is the signature of such equality chains (MC 26.3 OIT) - while
|
||||
// sorted-transparency "over" compositing (e.g. vanilla MC water, SRC_ALPHA factors),
|
||||
// which draws each surface once and depends on its depth writes to occlude later
|
||||
// passes, keeps them. Set at renderer initialization based on the active driver.
|
||||
// passes. Only MIN/MAX extremum blends are stripped - the signature of such a
|
||||
// chain's depth-bounds pass (MC 26.3 OIT), and per a fixture-wide trace sweep the
|
||||
// only depth-writing shape the chain actually uses - so every other blend
|
||||
// (sorted-transparency "over" like vanilla MC water, additive glows, ...) keeps
|
||||
// its depth writes. Set at renderer initialization based on the active driver.
|
||||
static void SetSuppressBlendedDepthWrite(Bool enabled);
|
||||
static Bool IsSuppressBlendedDepthWriteEnabled() { return s_suppressBlendedDepthWrite; }
|
||||
// Device gate for the quirk: ForceOn/ForceOff bypass detection, Auto enables it on
|
||||
@@ -87,12 +107,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static Bool ShouldSuppressDepthWrite(const PipelineCreatePayload& payload);
|
||||
|
||||
private:
|
||||
struct PipelineCacheEntry {
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
// The hashed inputs the eviction paths key on: programHash ties the entry to
|
||||
// its ProgramFactory entry, renderPass records the exact handle the hash
|
||||
// folded in (the hash is one-way, so targeted eviction needs them verbatim).
|
||||
HashType programHash = 0;
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
// Frame-boundary counter value of the last GetOrCreatePipeline hit; drives
|
||||
// cache eviction (see OnFrameBoundary).
|
||||
Uint64 lastUsedFrame = 0;
|
||||
};
|
||||
|
||||
VkPipeline CreatePipeline(const PipelineCreatePayload& payload) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPipelineCache m_pipelineCache = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, VkPipeline> m_cache;
|
||||
UnorderedMap<HashType, PipelineCacheEntry> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
static inline Bool s_suppressBlendedDepthWrite = false;
|
||||
};
|
||||
|
||||
@@ -12,7 +12,10 @@
|
||||
#include "MG_Util/ShaderTranspiler/ShaderCompiler.h"
|
||||
#include "MG_Util/ShaderTranspiler/SpvcSession.h"
|
||||
#include "MG_Util/ShaderTranspiler/Types.h"
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <unordered_set>
|
||||
#include <spirv-tools/libspirv.h>
|
||||
#include <spirv-tools/optimizer.hpp>
|
||||
#include <source/opt/build_module.h>
|
||||
@@ -923,11 +926,610 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
ProgramFactory::CompileOptionFlags m_transformFlags;
|
||||
};
|
||||
|
||||
// Adreno 650 (driver 512.502) faults the GPU on an implicit-LOD sample of a full-screen
|
||||
// colour render target: the texture unit's derivative path reads outside the image's
|
||||
// allocation even though the sampler clamps LOD to 0 and the mapping is 1:1. MobileGL's
|
||||
// own default-framebuffer blit shader works around it with textureLod, but an
|
||||
// application's shader (Minecraft's blit.fsh is `texture(InSampler, texCoord)`) cannot be
|
||||
// edited - so rewrite the sample at the SPIR-V level instead.
|
||||
//
|
||||
// The rewrite is only requested for draws whose every sampler binding is clamped to one
|
||||
// mip level, where explicit LOD 0 is exactly what the implicit form must already produce:
|
||||
// lambda' = clamp(lambda + bias, minLod, maxLod) with minLod = maxLod = 0. Bias and MinLod
|
||||
// operands are therefore dropped rather than translated.
|
||||
class ForceExplicitLod0SamplePass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "force-explicit-lod0-sample"; }
|
||||
|
||||
Status Process() override {
|
||||
Bool isFragment = false;
|
||||
for (auto& entryPoint : get_module()->entry_points()) {
|
||||
if (entryPoint.opcode() != spv::Op::OpEntryPoint) continue;
|
||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) ==
|
||||
spv::ExecutionModel::Fragment) {
|
||||
isFragment = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!isFragment) return Status::SuccessWithoutChange;
|
||||
|
||||
// Plan first, mutate second. Materializing the LOD constant is itself a module
|
||||
// change, so it must not happen unless at least one rewrite is going to follow -
|
||||
// otherwise the pass would grow the binary while reporting SuccessWithoutChange.
|
||||
Vector<RewritePlan> plans;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
RewritePlan plan{};
|
||||
if (PlanRewrite(&inst, plan)) plans.push_back(Move(plan));
|
||||
}
|
||||
}
|
||||
}
|
||||
if (plans.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
const Uint32 zeroId = GetFloatZeroId();
|
||||
if (zeroId == 0) return Status::SuccessWithoutChange;
|
||||
|
||||
for (auto& plan : plans) {
|
||||
plan.operands.push_back({SPV_OPERAND_TYPE_ID, {zeroId}});
|
||||
for (auto& operand : plan.trailingOperands) {
|
||||
plan.operands.push_back(operand);
|
||||
}
|
||||
plan.instruction->SetOpcode(plan.opcode);
|
||||
plan.instruction->SetInOperands(Move(plan.operands));
|
||||
}
|
||||
// Opcodes and operand lists changed underneath every cached analysis.
|
||||
context()->InvalidateAnalysesExceptFor(spvtools::opt::IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
private:
|
||||
struct RewritePlan {
|
||||
spvtools::opt::Instruction* instruction = nullptr;
|
||||
spv::Op opcode = spv::Op::OpNop;
|
||||
// Everything up to and including the Image Operands mask; the Lod id and the
|
||||
// trailing operand values are appended once the constant exists.
|
||||
Vector<spvtools::opt::Operand> operands;
|
||||
Vector<spvtools::opt::Operand> trailingOperands;
|
||||
};
|
||||
|
||||
// Image Operands bits that may accompany an implicit-LOD sample, in the canonical
|
||||
// ascending order SPIR-V requires the operand values to appear in.
|
||||
static constexpr Uint32 kBias = 0x1;
|
||||
static constexpr Uint32 kLod = 0x2;
|
||||
static constexpr Uint32 kGrad = 0x4;
|
||||
static constexpr Uint32 kConstOffset = 0x8;
|
||||
static constexpr Uint32 kOffset = 0x10;
|
||||
static constexpr Uint32 kConstOffsets = 0x20;
|
||||
static constexpr Uint32 kSample = 0x40;
|
||||
static constexpr Uint32 kMinLod = 0x80;
|
||||
static constexpr Uint32 kKnownMask = 0xFF;
|
||||
|
||||
Uint32 GetFloatZeroId() {
|
||||
// Reuse a 32-bit float type already in the module; a shader that samples always has
|
||||
// one, and looking it up avoids depending on type-creation API details.
|
||||
Uint32 floatTypeId = 0;
|
||||
for (auto& inst : get_module()->types_values()) {
|
||||
if (inst.opcode() == spv::Op::OpTypeFloat && inst.NumInOperands() >= 1 &&
|
||||
inst.GetSingleWordInOperand(0) == 32) {
|
||||
floatTypeId = inst.result_id();
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (floatTypeId == 0) return 0;
|
||||
|
||||
const auto* floatType = context()->get_type_mgr()->GetType(floatTypeId);
|
||||
if (floatType == nullptr) return 0;
|
||||
const auto zeroBits = std::bit_cast<Uint32>(0.0f);
|
||||
const auto* zeroConst = context()->get_constant_mgr()->GetConstant(floatType, {zeroBits});
|
||||
if (zeroConst == nullptr) return 0;
|
||||
auto* zeroInst = context()->get_constant_mgr()->GetDefiningInstruction(zeroConst);
|
||||
return zeroInst != nullptr ? zeroInst->result_id() : 0;
|
||||
}
|
||||
|
||||
static Bool MapOpcode(spv::Op op, spv::Op& outOpcode, Uint32& outFixedOperandCount) {
|
||||
switch (op) {
|
||||
case spv::Op::OpImageSampleImplicitLod:
|
||||
outOpcode = spv::Op::OpImageSampleExplicitLod;
|
||||
outFixedOperandCount = 2; // sampled image, coordinate
|
||||
return true;
|
||||
case spv::Op::OpImageSampleProjImplicitLod:
|
||||
outOpcode = spv::Op::OpImageSampleProjExplicitLod;
|
||||
outFixedOperandCount = 2;
|
||||
return true;
|
||||
case spv::Op::OpImageSampleDrefImplicitLod:
|
||||
outOpcode = spv::Op::OpImageSampleDrefExplicitLod;
|
||||
outFixedOperandCount = 3; // sampled image, coordinate, Dref
|
||||
return true;
|
||||
case spv::Op::OpImageSampleProjDrefImplicitLod:
|
||||
outOpcode = spv::Op::OpImageSampleProjDrefExplicitLod;
|
||||
outFixedOperandCount = 3;
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
static Bool PlanRewrite(spvtools::opt::Instruction* inst, RewritePlan& outPlan) {
|
||||
spv::Op newOpcode = spv::Op::OpNop;
|
||||
Uint32 fixedCount = 0;
|
||||
if (!MapOpcode(inst->opcode(), newOpcode, fixedCount)) return false;
|
||||
if (inst->NumInOperands() < fixedCount) return false;
|
||||
|
||||
Uint32 mask = 0;
|
||||
Uint32 next = fixedCount;
|
||||
if (inst->NumInOperands() > fixedCount) {
|
||||
mask = inst->GetSingleWordInOperand(fixedCount);
|
||||
next = fixedCount + 1;
|
||||
}
|
||||
// An operand this pass does not model would be silently reordered or dropped, and
|
||||
// Grad cannot legally accompany an implicit-LOD sample: leave such an instruction be.
|
||||
if ((mask & ~kKnownMask) != 0 || (mask & kGrad) != 0) return false;
|
||||
|
||||
Vector<spvtools::opt::Operand> fixedOperands;
|
||||
fixedOperands.reserve(fixedCount + 1);
|
||||
for (Uint32 i = 0; i < fixedCount; ++i) {
|
||||
fixedOperands.push_back(inst->GetInOperand(i));
|
||||
}
|
||||
|
||||
// Collect the surviving operand values in the same ascending-bit order they were
|
||||
// encoded in, so the rebuilt list stays canonical.
|
||||
Uint32 keptMask = kLod;
|
||||
Vector<spvtools::opt::Operand> keptOperands;
|
||||
static constexpr Uint32 kOrderedBits[] = {kBias, kLod, kGrad, kConstOffset,
|
||||
kOffset, kConstOffsets, kSample, kMinLod};
|
||||
for (const Uint32 bit : kOrderedBits) {
|
||||
if ((mask & bit) == 0) continue;
|
||||
if (next >= inst->NumInOperands()) return false;
|
||||
const spvtools::opt::Operand value = inst->GetInOperand(next++);
|
||||
// Bias and MinLod only shift a lambda that is already clamped to 0, and any
|
||||
// original Lod is replaced by the constant the caller appends.
|
||||
if (bit == kBias || bit == kMinLod || bit == kLod) continue;
|
||||
keptMask |= bit;
|
||||
keptOperands.push_back(value);
|
||||
}
|
||||
|
||||
fixedOperands.push_back({SPV_OPERAND_TYPE_IMAGE, {keptMask}});
|
||||
outPlan.instruction = inst;
|
||||
outPlan.opcode = newOpcode;
|
||||
outPlan.operands = Move(fixedOperands);
|
||||
outPlan.trailingOperands = Move(keptOperands);
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
spvtools::Optimizer::PassToken CreateForceExplicitLod0SamplePass() {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<ForceExplicitLod0SamplePass>());
|
||||
}
|
||||
|
||||
// TEMP-PERFDIAG: measure what fragment-stage fp32 costs on this GPU. Desktop GLSL carries
|
||||
// no precision qualifiers, so everything reaches the driver as full fp32 while Adreno runs
|
||||
// fp16 at twice the rate. Decorating every float-typed result in a fragment entry point
|
||||
// with RelaxedPrecision is the blunt "all mediump" upper bound - it changes results, so it
|
||||
// is a probe, not a shipping transform. Toggled by /sdcard/MG/exp_relaxed_precision.
|
||||
class RelaxedPrecisionProbePass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "relaxed-precision-probe"; }
|
||||
|
||||
Status Process() override {
|
||||
Bool isFragment = false;
|
||||
for (auto& entryPoint : get_module()->entry_points()) {
|
||||
if (entryPoint.opcode() != spv::Op::OpEntryPoint) continue;
|
||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) ==
|
||||
spv::ExecutionModel::Fragment) {
|
||||
isFragment = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!isFragment) return Status::SuccessWithoutChange;
|
||||
|
||||
// Every 32-bit-float scalar/vector/matrix type in the module. Anything wider (f64)
|
||||
// or narrower is left alone: RelaxedPrecision only has meaning for 32-bit floats.
|
||||
std::unordered_set<Uint32> relaxableTypes;
|
||||
for (auto& type : get_module()->types_values()) {
|
||||
const Uint32 typeId = type.result_id();
|
||||
if (typeId == 0) continue;
|
||||
switch (type.opcode()) {
|
||||
case spv::Op::OpTypeFloat:
|
||||
if (type.GetSingleWordInOperand(0) == 32) relaxableTypes.insert(typeId);
|
||||
break;
|
||||
case spv::Op::OpTypeVector:
|
||||
case spv::Op::OpTypeMatrix:
|
||||
if (relaxableTypes.count(type.GetSingleWordInOperand(0)) != 0) {
|
||||
relaxableTypes.insert(typeId);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (relaxableTypes.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
Vector<Uint32> targets;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0) continue;
|
||||
if (relaxableTypes.count(inst.type_id()) == 0) continue;
|
||||
targets.push_back(resultId);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (targets.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
for (const Uint32 id : targets) {
|
||||
context()->get_decoration_mgr()->AddDecoration(
|
||||
id, static_cast<Uint32>(spv::Decoration::RelaxedPrecision));
|
||||
}
|
||||
context()->InvalidateAnalysesExceptFor(spvtools::opt::IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
};
|
||||
|
||||
// Relax fragment-stage arithmetic that provably came out of a texture read. Desktop GLSL
|
||||
// has no precision qualifiers, so every fragment value reaches the driver as fp32 while
|
||||
// Adreno runs fp16 at twice the rate - and a texel is at most 8 bits per channel, which
|
||||
// fp16's 11-bit mantissa carries exactly. Seeding at image reads and propagating only
|
||||
// through operations whose every input is already relaxed keeps everything the shader
|
||||
// computes from other sources (screen coordinates, depth, wide-range uniforms) at full
|
||||
// precision, which is where fp16 would actually go wrong: fp16 cannot even represent a
|
||||
// 3044-pixel gl_FragCoord.x exactly.
|
||||
class RelaxTextureDerivedPrecisionPass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "relax-texture-derived-precision"; }
|
||||
|
||||
Status Process() override {
|
||||
if (!IsFragmentEntryPoint()) return Status::SuccessWithoutChange;
|
||||
// A shader that drives depth or coverage itself is out of scope: those values must
|
||||
// stay exact, and proving which computations feed them is not worth it here.
|
||||
if (WritesDepthOrSampleMask()) return Status::SuccessWithoutChange;
|
||||
|
||||
CollectRelaxableFloatTypes();
|
||||
if (m_relaxableTypes.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
// Whitelisting from texture reads captures nothing in practice: MC's fragment
|
||||
// shaders multiply every texel by an interpolated colour and a UBO value, so one
|
||||
// un-relaxed operand vetoes the whole expression (measured: no fps change).
|
||||
// Taint the few genuinely precision-critical sources instead and relax the rest.
|
||||
std::unordered_set<Uint32> tainted;
|
||||
CollectPrecisionCriticalSeeds(tainted);
|
||||
Bool grew = true;
|
||||
while (grew) {
|
||||
grew = false;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0 || tainted.count(resultId) != 0) continue;
|
||||
if (!AnyOperandTainted(inst, tainted)) continue;
|
||||
tainted.insert(resultId);
|
||||
grew = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unordered_set<Uint32> relaxed;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0 || tainted.count(resultId) != 0) continue;
|
||||
if (m_relaxableTypes.count(inst.type_id()) == 0) continue;
|
||||
relaxed.insert(resultId);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (relaxed.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
for (const Uint32 id : relaxed) {
|
||||
context()->get_decoration_mgr()->AddDecoration(
|
||||
id, static_cast<Uint32>(spv::Decoration::RelaxedPrecision));
|
||||
}
|
||||
context()->InvalidateAnalysesExceptFor(spvtools::opt::IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
private:
|
||||
std::unordered_set<Uint32> m_relaxableTypes;
|
||||
|
||||
Bool IsFragmentEntryPoint() const {
|
||||
for (auto& entryPoint : get_module()->entry_points()) {
|
||||
if (entryPoint.opcode() != spv::Op::OpEntryPoint) continue;
|
||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) ==
|
||||
spv::ExecutionModel::Fragment) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool WritesDepthOrSampleMask() const {
|
||||
for (auto& annotation : get_module()->annotations()) {
|
||||
if (annotation.opcode() != spv::Op::OpDecorate) continue;
|
||||
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(1)) !=
|
||||
spv::Decoration::BuiltIn) {
|
||||
continue;
|
||||
}
|
||||
const auto builtIn = static_cast<spv::BuiltIn>(annotation.GetSingleWordInOperand(2));
|
||||
if (builtIn == spv::BuiltIn::FragDepth || builtIn == spv::BuiltIn::SampleMask) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void CollectRelaxableFloatTypes() {
|
||||
m_relaxableTypes.clear();
|
||||
for (auto& type : get_module()->types_values()) {
|
||||
const Uint32 typeId = type.result_id();
|
||||
if (typeId == 0) continue;
|
||||
switch (type.opcode()) {
|
||||
case spv::Op::OpTypeFloat:
|
||||
if (type.GetSingleWordInOperand(0) == 32) m_relaxableTypes.insert(typeId);
|
||||
break;
|
||||
case spv::Op::OpTypeVector:
|
||||
if (m_relaxableTypes.count(type.GetSingleWordInOperand(0)) != 0) {
|
||||
m_relaxableTypes.insert(typeId);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void CollectImageReadSeeds(std::unordered_set<Uint32>& relaxed) const {
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0 || m_relaxableTypes.count(inst.type_id()) == 0) continue;
|
||||
// Interpolated user varyings seed too, or propagation dies at the
|
||||
// first `texel * vertexColour`: the load of an Input can never be
|
||||
// relaxed by the rule below (its operand is a pointer), so a single
|
||||
// varying vetoes every downstream operation. This is what ESSL's
|
||||
// mediump varyings already mean. Built-ins are excluded - gl_FragCoord
|
||||
// carries pixel coordinates that fp16 cannot represent exactly.
|
||||
if (inst.opcode() == spv::Op::OpLoad && IsNonBuiltInFragmentInput(inst)) {
|
||||
relaxed.insert(resultId);
|
||||
continue;
|
||||
}
|
||||
switch (inst.opcode()) {
|
||||
case spv::Op::OpImageSampleImplicitLod:
|
||||
case spv::Op::OpImageSampleExplicitLod:
|
||||
case spv::Op::OpImageSampleProjImplicitLod:
|
||||
case spv::Op::OpImageSampleProjExplicitLod:
|
||||
case spv::Op::OpImageSampleDrefImplicitLod:
|
||||
case spv::Op::OpImageSampleDrefExplicitLod:
|
||||
case spv::Op::OpImageFetch:
|
||||
case spv::Op::OpImageRead:
|
||||
case spv::Op::OpImageGather:
|
||||
relaxed.insert(resultId);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// OpLoad straight out of a fragment Input variable that carries no BuiltIn decoration.
|
||||
// Only a direct load counts: a load through an access chain could be indexing a
|
||||
// structure whose other members are not interpolated colour data.
|
||||
Bool IsNonBuiltInFragmentInput(const spvtools::opt::Instruction& load) const {
|
||||
const Uint32 pointerId = load.GetSingleWordInOperand(0);
|
||||
const auto* pointer = context()->get_def_use_mgr()->GetDef(pointerId);
|
||||
if (pointer == nullptr || pointer->opcode() != spv::Op::OpVariable) return false;
|
||||
if (static_cast<spv::StorageClass>(pointer->GetSingleWordInOperand(0)) !=
|
||||
spv::StorageClass::Input) {
|
||||
return false;
|
||||
}
|
||||
Bool isBuiltIn = false;
|
||||
context()->get_decoration_mgr()->ForEachDecoration(
|
||||
pointerId, static_cast<Uint32>(spv::Decoration::BuiltIn),
|
||||
[&isBuiltIn](const spvtools::opt::Instruction&) { isBuiltIn = true; });
|
||||
return !isBuiltIn;
|
||||
}
|
||||
|
||||
// A float constant small enough that fp16 represents it without surprise. Colour math
|
||||
// constants (0, 1, 0.5, 255, gamma exponents) all live here; anything larger is
|
||||
// treated as unknown so it stops propagation.
|
||||
Bool IsBoundedFloatConstant(Uint32 id) const {
|
||||
const auto* constant = context()->get_constant_mgr()->FindDeclaredConstant(id);
|
||||
if (constant == nullptr) return false;
|
||||
if (const auto* scalar = constant->AsFloatConstant()) {
|
||||
const float value = scalar->GetFloat();
|
||||
return std::isfinite(value) && std::fabs(value) <= 1024.0f;
|
||||
}
|
||||
if (const auto* composite = constant->AsVectorConstant()) {
|
||||
for (const auto* component : composite->GetComponents()) {
|
||||
const auto* scalar = component->AsFloatConstant();
|
||||
if (scalar == nullptr) return false;
|
||||
const float value = scalar->GetFloat();
|
||||
if (!std::isfinite(value) || std::fabs(value) > 1024.0f) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Precision-critical sources: a built-in fragment input. gl_FragCoord is the one that
|
||||
// matters - fp16 cannot represent a 3044-pixel x coordinate exactly, and anything
|
||||
// derived from it (screen-space effects, manual depth reconstruction) would visibly
|
||||
// quantise. Everything else a fragment shader reads is colour-range data.
|
||||
void CollectPrecisionCriticalSeeds(std::unordered_set<Uint32>& tainted) const {
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
if (inst.opcode() != spv::Op::OpLoad || inst.result_id() == 0) continue;
|
||||
if (IsBuiltInInputLoad(inst)) tainted.insert(inst.result_id());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Bool IsBuiltInInputLoad(const spvtools::opt::Instruction& load) const {
|
||||
const Uint32 pointerId = load.GetSingleWordInOperand(0);
|
||||
const auto* pointer = context()->get_def_use_mgr()->GetDef(pointerId);
|
||||
if (pointer == nullptr || pointer->opcode() != spv::Op::OpVariable) return false;
|
||||
if (static_cast<spv::StorageClass>(pointer->GetSingleWordInOperand(0)) !=
|
||||
spv::StorageClass::Input) {
|
||||
return false;
|
||||
}
|
||||
Bool isBuiltIn = false;
|
||||
context()->get_decoration_mgr()->ForEachDecoration(
|
||||
pointerId, static_cast<Uint32>(spv::Decoration::BuiltIn),
|
||||
[&isBuiltIn](const spvtools::opt::Instruction&) { isBuiltIn = true; });
|
||||
return isBuiltIn;
|
||||
}
|
||||
|
||||
Bool AnyOperandTainted(const spvtools::opt::Instruction& inst,
|
||||
const std::unordered_set<Uint32>& tainted) const {
|
||||
const Uint32 operandCount = inst.NumInOperands();
|
||||
for (Uint32 i = 0; i < operandCount; ++i) {
|
||||
const auto& operand = inst.GetInOperand(i);
|
||||
if (!spvIsIdType(operand.type)) continue;
|
||||
if (IsNonNumericOperand(inst, i)) continue;
|
||||
if (tainted.count(operand.words[0]) != 0) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool AllValueOperandsRelaxed(const spvtools::opt::Instruction& inst,
|
||||
const std::unordered_set<Uint32>& relaxed) const {
|
||||
switch (inst.opcode()) {
|
||||
// Pointer-typed plumbing: relaxing the loaded value would say nothing about the
|
||||
// memory it came from, and the pointer operand can never be in the set.
|
||||
case spv::Op::OpLoad:
|
||||
case spv::Op::OpStore:
|
||||
case spv::Op::OpAccessChain:
|
||||
case spv::Op::OpInBoundsAccessChain:
|
||||
case spv::Op::OpFunctionCall:
|
||||
return false;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
Bool sawValueOperand = false;
|
||||
Bool allRelaxed = true;
|
||||
const Uint32 operandCount = inst.NumInOperands();
|
||||
for (Uint32 i = 0; i < operandCount; ++i) {
|
||||
const auto& operand = inst.GetInOperand(i);
|
||||
if (!spvIsIdType(operand.type)) continue; // literals: selectors, swizzle indices
|
||||
const Uint32 id = operand.words[0];
|
||||
// OpPhi's block labels, OpSelect's condition and OpExtInst's instruction-set id
|
||||
// are ids that carry no numeric precision; skip them rather than let them veto.
|
||||
if (IsNonNumericOperand(inst, i)) continue;
|
||||
sawValueOperand = true;
|
||||
if (relaxed.count(id) != 0) continue;
|
||||
if (IsBoundedFloatConstant(id)) continue;
|
||||
allRelaxed = false;
|
||||
break;
|
||||
}
|
||||
return sawValueOperand && allRelaxed;
|
||||
}
|
||||
|
||||
static Bool IsNonNumericOperand(const spvtools::opt::Instruction& inst, Uint32 index) {
|
||||
switch (inst.opcode()) {
|
||||
case spv::Op::OpPhi:
|
||||
return (index % 2) == 1; // parent block labels
|
||||
case spv::Op::OpSelect:
|
||||
return index == 0; // condition
|
||||
case spv::Op::OpExtInst:
|
||||
return index == 0; // extended instruction set
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// TEMP-PERFDIAG: A/B switch between the scoped transform and the all-float upper bound.
|
||||
Bool PerfDiagRelaxAllPrecision() {
|
||||
static const Bool enabled = [] {
|
||||
std::FILE* probe = std::fopen("/sdcard/MG/exp_relaxed_precision_all", "rb");
|
||||
if (probe == nullptr) return false;
|
||||
std::fclose(probe);
|
||||
MGLOG_I("[PERFDIAG] fragment RelaxedPrecision: ALL floats (upper-bound probe)");
|
||||
return true;
|
||||
}();
|
||||
return enabled;
|
||||
}
|
||||
|
||||
// TEMP-PERFDIAG: lets a run turn the transform off entirely for an A/B baseline.
|
||||
Bool PerfDiagRelaxedPrecisionEnabled() {
|
||||
static const Bool disabled = [] {
|
||||
std::FILE* probe = std::fopen("/sdcard/MG/exp_no_relaxed_precision", "rb");
|
||||
if (probe == nullptr) return false;
|
||||
std::fclose(probe);
|
||||
MGLOG_I("[PERFDIAG] fragment RelaxedPrecision DISABLED");
|
||||
return true;
|
||||
}();
|
||||
return !disabled;
|
||||
}
|
||||
|
||||
Bool TransformSpirvForExplicitLod0Sampling(const Vector<Uint>& input, Vector<Uint>& output) {
|
||||
if (input.empty()) {
|
||||
output.clear();
|
||||
return true;
|
||||
}
|
||||
spvtools::Optimizer optimizer(SPV_ENV_VULKAN_1_3);
|
||||
spvtools::OptimizerOptions options;
|
||||
// Matches the position-fix pass: this build of spirv-tools asserts rather than
|
||||
// reporting, so validation stays off in the shipping path.
|
||||
options.set_run_validator(false);
|
||||
optimizer.SetMessageConsumer([](spv_message_level_t, const char*, const spv_position_t&,
|
||||
const char* message) {
|
||||
MGLOG_E("Vulkan: explicit-LOD0 pass: %s", message != nullptr ? message : "");
|
||||
});
|
||||
optimizer.RegisterPass(CreateForceExplicitLod0SamplePass());
|
||||
|
||||
const Bool success = optimizer.Run(input.data(), input.size(), &output, options);
|
||||
if (!success) {
|
||||
MGLOG_E("Vulkan: explicit-LOD0 sampling pass failed; keeping the original module");
|
||||
output = input;
|
||||
}
|
||||
return success;
|
||||
}
|
||||
|
||||
spvtools::Optimizer::PassToken CreateGlToVulkanPositionFixPass(
|
||||
ProgramFactory::CompileOptionFlags transformFlags) {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<GlToVulkanPositionFixPass>(transformFlags));
|
||||
}
|
||||
|
||||
// TEMP-PERFDIAG
|
||||
Bool TransformSpirvForRelaxedPrecisionProbe(const Vector<Uint>& input, Vector<Uint>& output) {
|
||||
if (input.empty()) {
|
||||
output.clear();
|
||||
return true;
|
||||
}
|
||||
spvtools::Optimizer optimizer(SPV_ENV_VULKAN_1_3);
|
||||
spvtools::OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
optimizer.SetMessageConsumer([](spv_message_level_t, const char*, const spv_position_t&,
|
||||
const char* message) {
|
||||
MGLOG_E("Vulkan: relaxed-precision probe: %s", message != nullptr ? message : "");
|
||||
});
|
||||
// SSA promotion first: glslang emits function-local variables with stores and loads,
|
||||
// and a load can never be relaxed (its operand is a pointer), so without this the
|
||||
// propagation below dies at the first temporary.
|
||||
optimizer.RegisterPass(spvtools::CreateLocalMultiStoreElimPass());
|
||||
if (PerfDiagRelaxAllPrecision()) {
|
||||
optimizer.RegisterPass(spvtools::Optimizer::PassToken(MakeUnique<RelaxedPrecisionProbePass>()));
|
||||
} else {
|
||||
optimizer.RegisterPass(
|
||||
spvtools::Optimizer::PassToken(MakeUnique<RelaxTextureDerivedPrecisionPass>()));
|
||||
}
|
||||
const Bool success = optimizer.Run(input.data(), input.size(), &output, options);
|
||||
if (!success) {
|
||||
MGLOG_E("Vulkan: relaxed-precision probe failed; keeping the original module");
|
||||
output = input;
|
||||
}
|
||||
return success;
|
||||
}
|
||||
|
||||
Bool TransformSpirvForVulkanPositionFix(const Vector<Uint>& input, Vector<Uint>& output,
|
||||
ProgramFactory::CompileOptionFlags transformFlags) {
|
||||
if (input.empty()) {
|
||||
@@ -1094,9 +1696,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (auto* binding : bindings) {
|
||||
MOBILEGL_ASSERT(binding != nullptr, "ProgramFactory: null descriptor binding reflection record");
|
||||
const auto kind = ReflectDescriptorTypeToBindingKind(binding->descriptor_type);
|
||||
MOBILEGL_ASSERT(binding->count == 1,
|
||||
"ProgramFactory: descriptor arrays are unsupported (name='%s' count=%u)",
|
||||
binding->name ? binding->name : "<null>", binding->count);
|
||||
// UBO instance arrays (uniform Block {...} b[N];) occupy one binding with
|
||||
// descriptorCount = N; other descriptor arrays stay unsupported and must
|
||||
// fail program creation cleanly rather than continue with corrupt state.
|
||||
if (binding->count != 1 && kind != ProgramFactory::DescriptorBindingKind::UniformBufferDynamic) {
|
||||
MGLOG_E("ProgramFactory: descriptor arrays are unsupported for this descriptor "
|
||||
"kind (name='%s' count=%u type=%d)",
|
||||
binding->name ? binding->name : "<null>", binding->count,
|
||||
static_cast<Int>(binding->descriptor_type));
|
||||
destroyReflectModules();
|
||||
return false;
|
||||
}
|
||||
|
||||
DescriptorKey key{};
|
||||
key.kind = kind;
|
||||
@@ -1547,6 +2157,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
"ProgramFactory::ReflectFragmentOutputs: failed to create reflection module (result=%d)",
|
||||
static_cast<Int>(createResult));
|
||||
if (createResult != SPV_REFLECT_RESULT_SUCCESS) {
|
||||
// Fail toward the exemption: stripping a genuine gl_FragDepth writer would
|
||||
// corrupt its depth output outright, while wrongly exempting an accumulation
|
||||
// pass merely reverts that one program to the pre-quirk behavior.
|
||||
entry.fragmentReplacesDepth = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -1609,6 +2223,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
entry.storageBlockIndexByBinding.assign(m_maxBindings, -1);
|
||||
entry.globalUboBinding = -1;
|
||||
entry.dynamicBindings.clear();
|
||||
entry.bindingDescriptorCounts.assign(m_maxBindings, 1);
|
||||
entry.arrayedUniformBlockIndicesByBinding.clear();
|
||||
|
||||
// Use SpvcSession (Reflection mode) to reflect all SPIR-V modules in a single pass per module
|
||||
for (const auto& module : spirv) {
|
||||
@@ -1624,6 +2240,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
"ProgramFactory::ReflectLayout: failed to create reflection module (result=%d)",
|
||||
static_cast<Int>(createReflectResult));
|
||||
|
||||
// Descriptor counts per binding (UBO instance arrays reflect count > 1).
|
||||
UnorderedMap<Uint32, Uint32> descriptorCountByBinding;
|
||||
{
|
||||
uint32_t countProbe = 0;
|
||||
if (spvReflectEnumerateDescriptorBindings(&reflectModule, &countProbe, nullptr) ==
|
||||
SPV_REFLECT_RESULT_SUCCESS &&
|
||||
countProbe > 0) {
|
||||
Vector<SpvReflectDescriptorBinding*> probeBindings(countProbe);
|
||||
if (spvReflectEnumerateDescriptorBindings(&reflectModule, &countProbe,
|
||||
probeBindings.data()) ==
|
||||
SPV_REFLECT_RESULT_SUCCESS) {
|
||||
for (const auto* probeBinding : probeBindings) {
|
||||
if (probeBinding != nullptr) {
|
||||
descriptorCountByBinding[probeBinding->binding] =
|
||||
std::max<Uint32>(1, probeBinding->count);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Reflect uniform buffers
|
||||
auto ubos = session.GetShaderInterface(SPVC_RESOURCE_TYPE_UNIFORM_BUFFER);
|
||||
for (const auto& ubo : ubos) {
|
||||
@@ -1649,9 +2286,69 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
}
|
||||
|
||||
const Uint blockIndex = program.GetUniformBlockIndex(ubo.name.c_str());
|
||||
if (blockIndex == 0xFFFFFFFFu) {
|
||||
MGLOG_D("ProgramFactory::ReflectLayout: skipping inactive UBO '%s' at binding %u",
|
||||
const auto countIt = descriptorCountByBinding.find(binding);
|
||||
const Uint32 descriptorCount =
|
||||
countIt != descriptorCountByBinding.end() ? countIt->second : 1u;
|
||||
|
||||
if (descriptorCount <= 1) {
|
||||
const Uint blockIndex = program.GetUniformBlockIndex(ubo.name.c_str());
|
||||
if (blockIndex == 0xFFFFFFFFu) {
|
||||
MGLOG_D("ProgramFactory::ReflectLayout: skipping inactive UBO '%s' at binding %u",
|
||||
ubo.name.c_str(), binding);
|
||||
continue;
|
||||
}
|
||||
|
||||
MOBILEGL_ASSERT(entry.bindingKinds[binding] == DescriptorBindingKind::None ||
|
||||
entry.bindingKinds[binding] == DescriptorBindingKind::UniformBufferDynamic,
|
||||
"ProgramFactory::ReflectLayout: descriptor binding %u has conflicting kinds for UBO '%s'",
|
||||
binding, ubo.name.c_str());
|
||||
entry.bindingKinds[binding] = DescriptorBindingKind::UniformBufferDynamic;
|
||||
MOBILEGL_ASSERT(entry.globalUboBinding != static_cast<Int>(binding),
|
||||
"ProgramFactory::ReflectLayout: regular UBO '%s' collides with global UBO binding %u",
|
||||
ubo.name.c_str(), binding);
|
||||
MOBILEGL_ASSERT(entry.uniformBlockIndexByBinding[binding] < 0 ||
|
||||
entry.uniformBlockIndexByBinding[binding] == static_cast<Int>(blockIndex),
|
||||
"ProgramFactory::ReflectLayout: descriptor binding %u maps to conflicting UBO blocks (%d vs %u)",
|
||||
binding, entry.uniformBlockIndexByBinding[binding], blockIndex);
|
||||
entry.uniformBlockIndexByBinding[binding] = static_cast<Int>(blockIndex);
|
||||
continue;
|
||||
}
|
||||
|
||||
// UBO instance array: one binding, descriptorCount elements. GL exposes each
|
||||
// element as its own active block named "Name[i]"; map every element to its
|
||||
// GL block index so the descriptor write can gather per-element buffer ranges.
|
||||
if (descriptorCount > m_maxBindings) {
|
||||
MGLOG_E("ProgramFactory::ReflectLayout: UBO array '%s' count %u exceeds maxBindings=%u; "
|
||||
"leaving binding %u unmapped",
|
||||
ubo.name.c_str(), descriptorCount, m_maxBindings, binding);
|
||||
continue;
|
||||
}
|
||||
Vector<Int> elementBlockIndices;
|
||||
elementBlockIndices.reserve(descriptorCount);
|
||||
for (Uint32 element = 0; element < descriptorCount; ++element) {
|
||||
String elementName = ubo.name + "[" + std::to_string(element) + "]";
|
||||
Uint elementBlockIndex = program.GetUniformBlockIndex(elementName.c_str());
|
||||
if (elementBlockIndex == 0xFFFFFFFFu && element == 0) {
|
||||
// Some frontends report the first element under the bare block name.
|
||||
elementBlockIndex = program.GetUniformBlockIndex(ubo.name.c_str());
|
||||
}
|
||||
if (elementBlockIndex == 0xFFFFFFFFu) {
|
||||
// Degrade rather than corrupt: reuse element 0's block if we have one,
|
||||
// otherwise give up on the binding (same observable behavior as an
|
||||
// inactive block: wrong values, but no crash).
|
||||
MGLOG_E("ProgramFactory::ReflectLayout: UBO array '%s' element %u has no active "
|
||||
"GL uniform block",
|
||||
ubo.name.c_str(), element);
|
||||
if (!elementBlockIndices.empty()) {
|
||||
elementBlockIndex = static_cast<Uint>(elementBlockIndices.front());
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
elementBlockIndices.push_back(static_cast<Int>(elementBlockIndex));
|
||||
}
|
||||
if (elementBlockIndices.size() != descriptorCount) {
|
||||
MGLOG_E("ProgramFactory::ReflectLayout: skipping unresolved UBO array '%s' at binding %u",
|
||||
ubo.name.c_str(), binding);
|
||||
continue;
|
||||
}
|
||||
@@ -1661,14 +2358,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
"ProgramFactory::ReflectLayout: descriptor binding %u has conflicting kinds for UBO '%s'",
|
||||
binding, ubo.name.c_str());
|
||||
entry.bindingKinds[binding] = DescriptorBindingKind::UniformBufferDynamic;
|
||||
MOBILEGL_ASSERT(entry.globalUboBinding != static_cast<Int>(binding),
|
||||
"ProgramFactory::ReflectLayout: regular UBO '%s' collides with global UBO binding %u",
|
||||
ubo.name.c_str(), binding);
|
||||
MOBILEGL_ASSERT(entry.uniformBlockIndexByBinding[binding] < 0 ||
|
||||
entry.uniformBlockIndexByBinding[binding] == static_cast<Int>(blockIndex),
|
||||
"ProgramFactory::ReflectLayout: descriptor binding %u maps to conflicting UBO blocks (%d vs %u)",
|
||||
binding, entry.uniformBlockIndexByBinding[binding], blockIndex);
|
||||
entry.uniformBlockIndexByBinding[binding] = static_cast<Int>(blockIndex);
|
||||
entry.bindingDescriptorCounts[binding] = static_cast<Uint16>(descriptorCount);
|
||||
entry.uniformBlockIndexByBinding[binding] = elementBlockIndices[0];
|
||||
entry.arrayedUniformBlockIndicesByBinding[binding] = Move(elementBlockIndices);
|
||||
}
|
||||
|
||||
// Reflect sampled images, storage images, samplerBuffer uniforms, and SSBOs.
|
||||
@@ -1815,7 +2507,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkDescriptorSetLayoutBinding layoutBinding{};
|
||||
layoutBinding.binding = binding;
|
||||
layoutBinding.descriptorCount = 1;
|
||||
layoutBinding.descriptorCount = entry.bindingDescriptorCounts[binding];
|
||||
layoutBinding.stageFlags = VK_SHADER_STAGE_ALL;
|
||||
layoutBinding.pImmutableSamplers = nullptr;
|
||||
if (kind == DescriptorBindingKind::UniformBufferDynamic) {
|
||||
@@ -1860,11 +2552,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
// Every draw/dispatch funnels through this lookup (the renderer memos only
|
||||
// skip re-hashing, never the factory lookup), so an actively-used entry is
|
||||
// stamped at least once per frame boundary and can never be aged out while
|
||||
// any in-flight command buffer still references it.
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
return it->second;
|
||||
}
|
||||
|
||||
auto& entry = m_cache[hash];
|
||||
entry.hash = hash;
|
||||
entry.lastUsedFrame = m_frameCounter;
|
||||
auto& shaders = program.GetAttachedShaders();
|
||||
auto& spirv = program.GetGeneratedSpirv();
|
||||
Vector<Vector<Uint>> moduleSpirvs(spirv.size());
|
||||
@@ -1882,6 +2580,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
moduleSpirvs[i] = spv;
|
||||
}
|
||||
|
||||
if ((flags & ProgramFactory::CompileOptionBit::ExplicitLod0Sampling) && shaders[i] &&
|
||||
shaders[i]->GetShaderStage() == ShaderStage::Fragment) {
|
||||
Vector<Uint> explicitLodSpirv;
|
||||
if (TransformSpirvForExplicitLod0Sampling(moduleSpirvs[i], explicitLodSpirv)) {
|
||||
moduleSpirvs[i] = Move(explicitLodSpirv);
|
||||
}
|
||||
}
|
||||
|
||||
if ((flags & ProgramFactory::CompileOptionBit::RelaxedFragmentPrecision) &&
|
||||
PerfDiagRelaxedPrecisionEnabled() && shaders[i] &&
|
||||
shaders[i]->GetShaderStage() == ShaderStage::Fragment) {
|
||||
Vector<Uint> relaxedSpirv;
|
||||
if (TransformSpirvForRelaxedPrecisionProbe(moduleSpirvs[i], relaxedSpirv)) {
|
||||
moduleSpirvs[i] = Move(relaxedSpirv);
|
||||
}
|
||||
}
|
||||
|
||||
// GL apps depend on cross-program position invariance for multi-pass equality
|
||||
// depth tests (MC 26.3's OIT re-draws the cloud geometry with GEQUAL against the
|
||||
// depth its own first pass wrote); decorate Position outputs Invariant so
|
||||
@@ -1978,4 +2693,42 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
return entry;
|
||||
}
|
||||
|
||||
void ProgramFactory::OnFrameBoundary() {
|
||||
++m_frameCounter;
|
||||
|
||||
// Sweep cadence and retire age mirror VkRenderPassManager::OnPresent: an entry
|
||||
// idle for more than kRetireAgeFrames frame boundaries cannot be referenced by
|
||||
// any in-flight command buffer (frames-in-flight <= MOBILEGL_MAGMA_FRAMESINFLIGHT),
|
||||
// so its shader modules and layouts are destroyed immediately - no deferred-
|
||||
// destroy machinery needed. Eviction is content-based, never tied to
|
||||
// glDeleteProgram: the cache is content-hash-shared across GL programs, so a
|
||||
// delete-driven erase could free an entry another live program still resolves.
|
||||
// An evicted entry self-heals - the frontend program keeps its generated
|
||||
// SPIR-V, so the next GetOrCreateProgram rebuilds it (this also covers the
|
||||
// renderer's internal blit/depth-mipmap programs).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeFrames = 1024;
|
||||
if ((m_frameCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameCounter - it->second.lastUsedFrame > kRetireAgeFrames) {
|
||||
const HashType hash = it->first;
|
||||
const VkDescriptorSetLayout descriptorSetLayout = it->second.descriptorSetLayout;
|
||||
MGLOG_D("ProgramFactory::OnFrameBoundary: evicting idle program entry hash=0x%llx",
|
||||
static_cast<unsigned long long>(hash));
|
||||
// erase runs ~VkProgramObject (modules/layouts destroyed); notify after
|
||||
// so an observer never observes a half-destroyed entry through a lookup.
|
||||
// Observers only need the handle values to purge their keyed caches.
|
||||
it = m_cache.erase(it);
|
||||
if (m_evictionObserver != nullptr) {
|
||||
m_evictionObserver->OnProgramEvicted(hash, descriptorSetLayout);
|
||||
}
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -42,6 +42,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
SurfaceRotate90 = 1 << 2,
|
||||
SurfaceRotate180 = 1 << 3,
|
||||
SurfaceRotate270 = 1 << 4,
|
||||
// Rewrites the fragment stage's implicit-LOD image samples to explicit LOD 0.
|
||||
// Only ever set for a draw whose every sampler binding is clamped to a single mip
|
||||
// level, which makes the two forms produce identical texels (the implicit lambda is
|
||||
// clamped into [minLod, maxLod] = [0, 0] regardless of derivatives or bias).
|
||||
ExplicitLod0Sampling = 1 << 5,
|
||||
// Fragment arithmetic may run at relaxed (fp16) precision. Only requested for draws
|
||||
// where every sampled texture and every colour attachment is an 8-bit-or-less
|
||||
// normalized format, so nothing the shader reads or writes carries more precision
|
||||
// than fp16 already represents exactly.
|
||||
RelaxedFragmentPrecision = 1 << 6,
|
||||
};
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
@@ -59,6 +69,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<DescriptorBindingKind> bindingKinds;
|
||||
Vector<Uint32> dynamicBindings;
|
||||
Vector<Int> uniformBlockIndexByBinding;
|
||||
// Descriptor count per binding (1 except for UBO instance arrays, which occupy one
|
||||
// binding with descriptorCount = N).
|
||||
Vector<Uint16> bindingDescriptorCounts;
|
||||
// Per-element GL uniform block indices for arrayed UBO bindings (count > 1);
|
||||
// element 0 of a non-arrayed binding stays in uniformBlockIndexByBinding.
|
||||
UnorderedMap<Uint32, Vector<Int>> arrayedUniformBlockIndicesByBinding;
|
||||
Vector<String> samplerNameByBinding;
|
||||
Vector<Int> samplerUniformLocationByBinding;
|
||||
Vector<TextureTarget> samplerTextureTargetByBinding;
|
||||
@@ -82,6 +98,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// gl_FragDepth); shader-computed depth is immune to the cross-pipeline
|
||||
// position-invariance quirk (see PipelineFactory::ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
// Frame-boundary counter value of the last GetOrCreateProgram hit; drives
|
||||
// cache eviction (see OnFrameBoundary).
|
||||
Uint64 lastUsedFrame = 0;
|
||||
|
||||
static inline VkDevice s_device = VK_NULL_HANDLE;
|
||||
|
||||
@@ -97,6 +116,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
dynamicBindings = std::move(other.dynamicBindings);
|
||||
uniformBlockIndexByBinding = std::move(other.uniformBlockIndexByBinding);
|
||||
bindingDescriptorCounts = std::move(other.bindingDescriptorCounts);
|
||||
arrayedUniformBlockIndicesByBinding = std::move(other.arrayedUniformBlockIndicesByBinding);
|
||||
samplerNameByBinding = std::move(other.samplerNameByBinding);
|
||||
samplerUniformLocationByBinding = std::move(other.samplerUniformLocationByBinding);
|
||||
samplerTextureTargetByBinding = std::move(other.samplerTextureTargetByBinding);
|
||||
@@ -116,6 +137,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
@@ -127,6 +149,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.lastUsedFrame = 0;
|
||||
}
|
||||
VkProgramObject& operator=(VkProgramObject&& other) noexcept {
|
||||
if (this == &other) {
|
||||
@@ -141,6 +164,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
dynamicBindings = std::move(other.dynamicBindings);
|
||||
uniformBlockIndexByBinding = std::move(other.uniformBlockIndexByBinding);
|
||||
bindingDescriptorCounts = std::move(other.bindingDescriptorCounts);
|
||||
arrayedUniformBlockIndicesByBinding = std::move(other.arrayedUniformBlockIndicesByBinding);
|
||||
samplerNameByBinding = std::move(other.samplerNameByBinding);
|
||||
samplerUniformLocationByBinding = std::move(other.samplerUniformLocationByBinding);
|
||||
samplerTextureTargetByBinding = std::move(other.samplerTextureTargetByBinding);
|
||||
@@ -160,6 +185,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
@@ -171,6 +197,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.lastUsedFrame = 0;
|
||||
return *this;
|
||||
}
|
||||
|
||||
@@ -200,6 +227,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
};
|
||||
|
||||
// Notified when the OnFrameBoundary sweep destroys an aged-out cache entry,
|
||||
// carrying the entry's content hash and the VkDescriptorSetLayout it owned.
|
||||
// Dependent caches (compute pipelines, PipelineFactory entries, UniformManager's
|
||||
// per-layout descriptor sets) must purge in the same step: after vkDestroy the
|
||||
// layout handle value may be recycled for an unrelated layout, and the program
|
||||
// hash may be re-inserted by a later rebuild of the same content.
|
||||
class IEvictionObserver {
|
||||
public:
|
||||
virtual ~IEvictionObserver() = default;
|
||||
virtual void OnProgramEvicted(HashType programHash, VkDescriptorSetLayout descriptorSetLayout) = 0;
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings = 16,
|
||||
Bool shaderDrawParametersEnabled = false,
|
||||
Bool unformattedFloatStorageImagesEnabled = false)
|
||||
@@ -215,6 +254,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VkProgramObject& GetOrCreateProgram(
|
||||
const MG_State::GLState::ProgramObject& program, CompileOptionFlags flags);
|
||||
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetEvictionObserver(IEvictionObserver* observer) { m_evictionObserver = observer; }
|
||||
// Frame boundary hook: ages the program cache and evicts long-unused entries
|
||||
// (their command buffers retired many frames ago), mirroring
|
||||
// VkRenderPassManager::OnPresent's sweep.
|
||||
void OnFrameBoundary();
|
||||
|
||||
static VkShaderStageFlagBits ToVkStage(ShaderStage stage);
|
||||
static VkFormat ConvertSpirvImageFormatToVkFormat(SpvImageFormat format);
|
||||
static SamplerNumericDomain UniformTypeToSamplerNumericDomain(GLenum glType);
|
||||
@@ -256,6 +302,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
mutable ProgramLookupCache m_lastLookup;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -247,6 +247,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_surfaceFormat = {createInfo.imageFormat, createInfo.imageColorSpace};
|
||||
m_extent = createInfo.imageExtent;
|
||||
// The surface-space extent this swapchain was built from, i.e. before the
|
||||
// quarter-turn swap above. Out-of-date checks must compare in THIS space: comparing a
|
||||
// freshly queried currentExtent against the swapped m_extent flips axes every rotation
|
||||
// and makes the comparison alternate forever.
|
||||
m_surfaceExtent = defaultFramebufferExtent;
|
||||
m_preTransform = createInfo.preTransform;
|
||||
|
||||
VK_VERIFY(vkCreateSwapchainKHR(device, &createInfo, nullptr, &m_swapchain));
|
||||
|
||||
@@ -35,6 +35,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSwapchainKHR GetHandle() const { return m_swapchain; }
|
||||
const VkSurfaceFormatKHR& GetSurfaceFormat() const { return m_surfaceFormat; }
|
||||
VkExtent2D GetExtent() const { return m_extent; }
|
||||
// Surface-space extent (before the pre-rotation quarter-turn swap) this swapchain was
|
||||
// created from - the value to compare a freshly queried currentExtent against.
|
||||
VkExtent2D GetSurfaceExtent() const { return m_surfaceExtent; }
|
||||
VkSurfaceTransformFlagBitsKHR GetPreTransform() const { return m_preTransform; }
|
||||
const Vector<VkImage>& GetImages() const { return m_images; }
|
||||
const Vector<VkImageView>& GetImageViews() const { return m_imageViews; }
|
||||
@@ -63,6 +66,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSwapchainKHR m_swapchain = VK_NULL_HANDLE;
|
||||
VkSurfaceFormatKHR m_surfaceFormat{};
|
||||
VkExtent2D m_extent{};
|
||||
VkExtent2D m_surfaceExtent{};
|
||||
VkSurfaceTransformFlagBitsKHR m_preTransform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
|
||||
Vector<VkImage> m_images;
|
||||
Vector<VkImageView> m_imageViews;
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
#include "MG_Util/Converters/GLToMG/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include <vulkan/utility/vk_format_utils.h>
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include <Config.h>
|
||||
#include <cstdio>
|
||||
@@ -211,6 +212,40 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
void UniformManager::OnDescriptorSetLayoutDestroyed(VkDescriptorSetLayout descriptorSetLayout) {
|
||||
SizeT purgedSets = 0;
|
||||
for (auto& frame : m_frames) {
|
||||
const auto it = frame.descriptorSetCacheByLayout.find(descriptorSetLayout);
|
||||
if (it == frame.descriptorSetCacheByLayout.end()) {
|
||||
continue;
|
||||
}
|
||||
// Free the sets back to their pools and credit the bucket accounting, so
|
||||
// program churn recycles pool capacity instead of abandoning the slots.
|
||||
// GPU-safe: the layout only dies after >1024 idle frame boundaries, so no
|
||||
// in-flight command buffer references these sets.
|
||||
for (const auto& cached : it->second.sets) {
|
||||
if (cached.set == VK_NULL_HANDLE) {
|
||||
continue;
|
||||
}
|
||||
vkFreeDescriptorSets(m_device, cached.pool, 1, &cached.set);
|
||||
const auto bucket = std::find_if(
|
||||
frame.descriptorPools.begin(), frame.descriptorPools.end(),
|
||||
[&cached](const DescriptorPoolBucket& candidate) { return candidate.handle == cached.pool; });
|
||||
if (bucket != frame.descriptorPools.end() && bucket->allocatedSets > 0) {
|
||||
--bucket->allocatedSets;
|
||||
}
|
||||
}
|
||||
purgedSets += it->second.sets.size();
|
||||
frame.descriptorSetCacheByLayout.erase(it);
|
||||
}
|
||||
if (purgedSets > 0) {
|
||||
// The per-draw reuse memo folds the layout handle into its signature; drop
|
||||
// it so a recycled handle value cannot revive a purged set mid-frame.
|
||||
m_hasLastDescriptor = false;
|
||||
MGLOG_D("UniformDescriptorBinder: freed %zu descriptor sets for destroyed layout", purgedSets);
|
||||
}
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveSamplerDescriptor(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
@@ -350,24 +385,28 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const Uint16 samplerVersion = samplerToUse->GetVersion();
|
||||
const Uint64 textureLifetimeId = texture->GetLifetimeId();
|
||||
const Uint16 textureParamsVersion = texture->GetTextureParamsVersion();
|
||||
// The sampler's LOD clamp depends on how many levels the sampled view exposes, and that
|
||||
// follows uploads as well as GL parameters - so it belongs in the memo key too.
|
||||
const Uint32 viewLevelCount = resource->sampledLevelCount;
|
||||
if (memo.valid && memo.samplerLifetimeId == samplerLifetimeId && memo.samplerVersion == samplerVersion &&
|
||||
memo.textureLifetimeId == textureLifetimeId && memo.textureParamsVersion == textureParamsVersion &&
|
||||
memo.forceNearestFiltering == forceNearestFiltering) {
|
||||
memo.forceNearestFiltering == forceNearestFiltering && memo.viewLevelCount == viewLevelCount) {
|
||||
resolvedSampler = memo.sampler;
|
||||
} else {
|
||||
resolvedSampler =
|
||||
m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture, forceNearestFiltering);
|
||||
resolvedSampler = m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture,
|
||||
forceNearestFiltering, viewLevelCount);
|
||||
memo.samplerLifetimeId = samplerLifetimeId;
|
||||
memo.samplerVersion = samplerVersion;
|
||||
memo.textureLifetimeId = textureLifetimeId;
|
||||
memo.textureParamsVersion = textureParamsVersion;
|
||||
memo.forceNearestFiltering = forceNearestFiltering;
|
||||
memo.viewLevelCount = viewLevelCount;
|
||||
memo.sampler = resolvedSampler;
|
||||
memo.valid = true;
|
||||
}
|
||||
} else {
|
||||
resolvedSampler =
|
||||
m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture, forceNearestFiltering);
|
||||
resolvedSampler = m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture, forceNearestFiltering,
|
||||
resource->sampledLevelCount);
|
||||
}
|
||||
outImageInfo = {
|
||||
.sampler = resolvedSampler,
|
||||
@@ -408,6 +447,106 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return outImageInfo.sampler != VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
namespace {
|
||||
// fp16 carries an 11-bit mantissa, so an 8-bit normalized channel round-trips exactly.
|
||||
// Anything wider - 16-bit normalized, half float, full float, and every packed HDR
|
||||
// encoding - holds precision or range that relaxing the arithmetic would throw away.
|
||||
Bool IsLowPrecisionNormalizedFormat(VkFormat format) {
|
||||
if (format == VK_FORMAT_UNDEFINED) return false;
|
||||
if (!vkuFormatIsUNORM(format) && !vkuFormatIsSNORM(format) && !vkuFormatIsSRGB(format)) {
|
||||
return false;
|
||||
}
|
||||
const struct VKU_FORMAT_INFO info = vkuGetFormatInfo(format);
|
||||
for (Uint32 i = 0; i < info.component_count; ++i) {
|
||||
if (info.components[i].size > 8) return false;
|
||||
}
|
||||
return info.component_count > 0;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool UniformManager::DrawTargetIsLowPrecision(const MG_State::GLState::FramebufferObject* drawFramebuffer) {
|
||||
// Default framebuffer: the swapchain is an 8-bit normalized surface.
|
||||
if (drawFramebuffer == nullptr) return true;
|
||||
|
||||
Bool sawColour = false;
|
||||
for (Int i = static_cast<Int>(FramebufferAttachmentType::Color0);
|
||||
i < static_cast<Int>(FramebufferAttachmentType::FramebufferAttachmentTypeCount);
|
||||
++i) {
|
||||
const auto& attachment =
|
||||
drawFramebuffer->GetAttachment(static_cast<FramebufferAttachmentType>(i));
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
if (const auto& texture = attachment.GetTexture()) {
|
||||
format = MG_Util::ConvertTextureInternalFormatToVkEnum(texture->GetFormat());
|
||||
} else if (const auto& renderbuffer = attachment.GetRenderbuffer()) {
|
||||
format = MG_Util::ConvertTextureInternalFormatToVkEnum(
|
||||
renderbuffer->GetInternalFormat());
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
if (!IsLowPrecisionNormalizedFormat(format)) return false;
|
||||
sawColour = true;
|
||||
}
|
||||
return sawColour;
|
||||
}
|
||||
|
||||
Bool UniformManager::ProgramSamplesOnlyLowPrecisionTextures(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj) {
|
||||
for (Uint32 binding = 0; binding < programObj.bindingKinds.size(); ++binding) {
|
||||
if (programObj.bindingKinds[binding] != ProgramFactory::DescriptorBindingKind::CombinedImageSampler) {
|
||||
continue;
|
||||
}
|
||||
const auto* texture = ResolveSamplerTextureRaw(program, programObj, binding);
|
||||
// An unresolvable binding is unknown territory, not licence to relax.
|
||||
if (texture == nullptr) return false;
|
||||
const VkFormat format =
|
||||
MG_Util::ConvertTextureInternalFormatToVkEnum(texture->GetFormat());
|
||||
if (!IsLowPrecisionNormalizedFormat(format)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::ProgramSamplesOnlySingleLevelTextures(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj) {
|
||||
Bool sawSampler = false;
|
||||
for (Uint32 binding = 0; binding < programObj.bindingKinds.size(); ++binding) {
|
||||
if (programObj.bindingKinds[binding] != ProgramFactory::DescriptorBindingKind::CombinedImageSampler) {
|
||||
continue;
|
||||
}
|
||||
const auto* texture = ResolveSamplerTextureRaw(program, programObj, binding);
|
||||
if (texture == nullptr) return false;
|
||||
const auto& levelRange = texture->GetLevelRange();
|
||||
if (levelRange.x() != levelRange.y()) return false;
|
||||
|
||||
// An explicit-LOD sample is a single filtered tap, so it also gives up anisotropic
|
||||
// filtering - which a single-level view can still have. Resolve the sampler exactly
|
||||
// the way ResolveSamplerDescriptor does and bail if anisotropy would apply.
|
||||
const Int location = programObj.samplerUniformLocationByBinding[binding];
|
||||
const Int unit = ResolveSamplerUnitIndex(program, location, binding);
|
||||
const auto& samplerOverride = MG_State::pGLContext->GetTextureUnitObject(unit).GetSamplerObject();
|
||||
const auto* effectiveSampler =
|
||||
samplerOverride ? samplerOverride.get() : texture->GetSamplerObject().get();
|
||||
if (effectiveSampler == nullptr) return false;
|
||||
if (effectiveSampler->GetMaxAnisotropy() > 1.0f &&
|
||||
effectiveSampler->GetMinFilter() == SamplerFilterMode::Linear &&
|
||||
effectiveSampler->GetMagFilter() == SamplerFilterMode::Linear) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// An explicit LOD 0 makes lambda exactly 0, which is the magnification side of the
|
||||
// min/mag decision. That only matches the implicit form when lambda could not have been
|
||||
// positive anyway (the LOD clamp already pins it at or below 0), or when the two
|
||||
// filters are the same and the choice cannot be observed.
|
||||
const Float effectiveMaxLod = effectiveSampler->GetMipmapMode() == SamplerMipmapMode::None
|
||||
? 0.0f
|
||||
: effectiveSampler->GetMaxLod();
|
||||
if (effectiveMaxLod > 0.0f && effectiveSampler->GetMinFilter() != effectiveSampler->GetMagFilter()) {
|
||||
return false;
|
||||
}
|
||||
sawSampler = true;
|
||||
}
|
||||
return sawSampler;
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveSamplerTexture(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture) {
|
||||
@@ -765,7 +904,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
Bool UniformManager::ResolveUniformBufferPayload(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
UboBindResult& out) const {
|
||||
Uint32 arrayElement, UboBindResult& out) const {
|
||||
const void* outData = nullptr;
|
||||
VkDeviceSize outSize = 0;
|
||||
|
||||
@@ -791,7 +930,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
MOBILEGL_ASSERT(binding < programObj.uniformBlockIndexByBinding.size(),
|
||||
"ResolveUniformBufferPayload: UBO mapping binding %u out of range", binding);
|
||||
const Int blockIndex = programObj.uniformBlockIndexByBinding[binding];
|
||||
Int blockIndex = programObj.uniformBlockIndexByBinding[binding];
|
||||
if (arrayElement > 0) {
|
||||
const auto arrayIt = programObj.arrayedUniformBlockIndicesByBinding.find(binding);
|
||||
const Bool elementValid = arrayIt != programObj.arrayedUniformBlockIndicesByBinding.end() &&
|
||||
arrayElement < arrayIt->second.size();
|
||||
MOBILEGL_ASSERT(elementValid,
|
||||
"ResolveUniformBufferPayload: UBO binding %u has no array element %u", binding,
|
||||
arrayElement);
|
||||
if (!elementValid) {
|
||||
return false;
|
||||
}
|
||||
blockIndex = arrayIt->second[arrayElement];
|
||||
}
|
||||
MOBILEGL_ASSERT(blockIndex >= 0,
|
||||
"ResolveUniformBufferPayload: no uniform block mapped to descriptor binding %u", binding);
|
||||
|
||||
@@ -899,6 +1050,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkDescriptorPoolCreateInfo poolInfo{};
|
||||
poolInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO;
|
||||
// FREE_DESCRIPTOR_SET_BIT lets a destroyed layout's cached sets be freed back
|
||||
// (OnDescriptorSetLayoutDestroyed) so program churn recycles pool capacity.
|
||||
// The cost is on set allocation only, which happens when a layout's per-frame
|
||||
// cache grows - never on the per-draw reuse path.
|
||||
poolInfo.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT;
|
||||
poolInfo.maxSets = maxSets;
|
||||
poolInfo.poolSizeCount = static_cast<Uint32>(std::size(poolSizes));
|
||||
poolInfo.pPoolSizes = poolSizes;
|
||||
@@ -978,7 +1134,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto& frame = m_frames[frameIndex];
|
||||
auto& cache = frame.descriptorSetCacheByLayout[programObj.descriptorSetLayout];
|
||||
if (cache.cursor < cache.sets.size()) {
|
||||
outDescriptorSet = cache.sets[cache.cursor++];
|
||||
outDescriptorSet = cache.sets[cache.cursor++].set;
|
||||
} else {
|
||||
VkResult allocResult = AllocateDescriptorSetsFromActivePool(frameIndex, programObj, outDescriptorSet);
|
||||
if (allocResult == VK_ERROR_OUT_OF_POOL_MEMORY || allocResult == VK_ERROR_FRAGMENTED_POOL) {
|
||||
@@ -992,7 +1148,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return allocResult;
|
||||
}
|
||||
|
||||
cache.sets.push_back(outDescriptorSet);
|
||||
// The successful allocation came from the bucket the alloc helper left
|
||||
// active; record it so a layout-destroyed purge can free the set back.
|
||||
cache.sets.push_back({outDescriptorSet, frame.descriptorPools[frame.activeDescriptorPoolIndex].handle});
|
||||
++cache.cursor;
|
||||
MGLOG_D("UniformDescriptorBinder: cached descriptor set count for frame=%u grew to %zu", frameIndex,
|
||||
cache.sets.size());
|
||||
@@ -1038,11 +1196,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
imageInfos.clear();
|
||||
texelBufferViews.clear();
|
||||
dynamicOffsets.clear();
|
||||
// Arrayed UBO bindings contribute extra buffer infos and dynamic offsets; reserve for
|
||||
// the worst case so the pBufferInfo pointers taken below never dangle on reallocation.
|
||||
Uint32 uboArrayExtra = 0;
|
||||
for (const auto& arrayEntry : programObj.arrayedUniformBlockIndicesByBinding) {
|
||||
uboArrayExtra += static_cast<Uint32>(arrayEntry.second.size()) - 1u;
|
||||
}
|
||||
writes.reserve(m_maxBindings);
|
||||
bufferInfos.reserve(m_maxBindings);
|
||||
bufferInfos.reserve(m_maxBindings + uboArrayExtra);
|
||||
imageInfos.reserve(m_maxBindings);
|
||||
texelBufferViews.reserve(m_maxBindings);
|
||||
dynamicOffsets.reserve(programObj.dynamicBindings.size());
|
||||
dynamicOffsets.reserve(programObj.dynamicBindings.size() + uboArrayExtra);
|
||||
|
||||
const Uint32 bindingCount =
|
||||
std::min<Uint32>(m_maxBindings, static_cast<Uint32>(programObj.bindingKinds.size()));
|
||||
@@ -1060,40 +1224,51 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
write.descriptorCount = 1;
|
||||
|
||||
if (kind == ProgramFactory::DescriptorBindingKind::UniformBufferDynamic) {
|
||||
UboBindResult ubo{};
|
||||
const Bool hasPayload = ResolveUniformBufferPayload(program, programObj, binding, ubo);
|
||||
MOBILEGL_ASSERT(hasPayload && ubo.payload != nullptr && ubo.payloadSize > 0,
|
||||
"UniformDescriptorBinder::BindProgramUniformBuffers failed: missing UBO payload on binding %u",
|
||||
binding);
|
||||
const Uint32 descriptorCount =
|
||||
binding < programObj.bindingDescriptorCounts.size()
|
||||
? std::max<Uint32>(1, programObj.bindingDescriptorCounts[binding])
|
||||
: 1u;
|
||||
const SizeT firstBufferInfoIndex = bufferInfos.size();
|
||||
for (Uint32 element = 0; element < descriptorCount; ++element) {
|
||||
UboBindResult ubo{};
|
||||
const Bool hasPayload =
|
||||
ResolveUniformBufferPayload(program, programObj, binding, element, ubo);
|
||||
MOBILEGL_ASSERT(hasPayload && ubo.payload != nullptr && ubo.payloadSize > 0,
|
||||
"UniformDescriptorBinder::BindProgramUniformBuffers failed: missing UBO payload on binding %u element %u",
|
||||
binding, element);
|
||||
|
||||
VkDescriptorBufferInfo bufferInfo{};
|
||||
// Keep offset 0 (sub-range selected via the dynamic offset) so the hashed bufferInfo
|
||||
// is stable across draws and the descriptor-set reuse cache keeps hitting.
|
||||
bufferInfo.offset = 0;
|
||||
Uint32 dynOffset;
|
||||
if (ubo.directBindable) {
|
||||
// Zero-copy: bind the app's resident VkBuffer directly, no per-draw memcpy.
|
||||
bufferInfo.buffer = ubo.buffer;
|
||||
bufferInfo.range = ubo.range;
|
||||
dynOffset = static_cast<Uint32>(ubo.dynamicOffset);
|
||||
} else {
|
||||
BufferSlice slice{};
|
||||
if (!m_bufferManager->UploadTransient(BufferKind::Uniform, frameIndex, ubo.payload,
|
||||
ubo.payloadSize, m_minDynamicOffsetAlignment, slice)) {
|
||||
MOBILEGL_ASSERT(false, "UniformDescriptorBinder::BindProgramUniformBuffers failed: UBO upload failed on binding %u",
|
||||
binding);
|
||||
return false;
|
||||
VkDescriptorBufferInfo bufferInfo{};
|
||||
// Keep offset 0 (sub-range selected via the dynamic offset) so the hashed bufferInfo
|
||||
// is stable across draws and the descriptor-set reuse cache keeps hitting.
|
||||
bufferInfo.offset = 0;
|
||||
Uint32 dynOffset;
|
||||
if (ubo.directBindable) {
|
||||
// Zero-copy: bind the app's resident VkBuffer directly, no per-draw memcpy.
|
||||
bufferInfo.buffer = ubo.buffer;
|
||||
bufferInfo.range = ubo.range;
|
||||
dynOffset = static_cast<Uint32>(ubo.dynamicOffset);
|
||||
} else {
|
||||
BufferSlice slice{};
|
||||
if (!m_bufferManager->UploadTransient(BufferKind::Uniform, frameIndex, ubo.payload,
|
||||
ubo.payloadSize, m_minDynamicOffsetAlignment, slice)) {
|
||||
MOBILEGL_ASSERT(false, "UniformDescriptorBinder::BindProgramUniformBuffers failed: UBO upload failed on binding %u element %u",
|
||||
binding, element);
|
||||
return false;
|
||||
}
|
||||
bufferInfo.buffer = slice.buffer;
|
||||
bufferInfo.range = ubo.payloadSize;
|
||||
dynOffset = static_cast<Uint32>(slice.offset);
|
||||
}
|
||||
bufferInfo.buffer = slice.buffer;
|
||||
bufferInfo.range = ubo.payloadSize;
|
||||
dynOffset = static_cast<Uint32>(slice.offset);
|
||||
bufferInfos.push_back(bufferInfo);
|
||||
// Dynamic offsets are consumed in binding order, then array element order,
|
||||
// matching Vulkan's dynamic-offset consumption rules.
|
||||
dynamicOffsets.push_back(dynOffset);
|
||||
}
|
||||
bufferInfos.push_back(bufferInfo);
|
||||
|
||||
write.descriptorType = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC;
|
||||
write.pBufferInfo = &bufferInfos.back();
|
||||
write.descriptorCount = descriptorCount;
|
||||
write.pBufferInfo = &bufferInfos[firstBufferInfoIndex];
|
||||
writes.push_back(write);
|
||||
dynamicOffsets.push_back(dynOffset);
|
||||
} else if (kind == ProgramFactory::DescriptorBindingKind::UniformTexelBuffer) {
|
||||
VkBufferView bufferView = VK_NULL_HANDLE;
|
||||
if (!ResolveTexelBufferDescriptor(program, programObj, binding, frameIndex, bufferView) ||
|
||||
|
||||
@@ -39,6 +39,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void Shutdown();
|
||||
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// A ProgramFactory eviction just destroyed this layout: purge every frame
|
||||
// slot's cached descriptor sets for it, so a recycled handle value can never
|
||||
// stale-hit sets written for the dead layout's bindings. The sets are
|
||||
// vkFreeDescriptorSets'd back to their pools (created with
|
||||
// FREE_DESCRIPTOR_SET_BIT) and the pool accounting is credited, so program
|
||||
// churn recycles pool capacity instead of abandoning it. GPU-safe: the layout
|
||||
// only dies after >1024 idle frame boundaries, so no in-flight command buffer
|
||||
// references its sets. This is the only eviction path for the per-layout
|
||||
// caches - a live layout's entry must never be purged (its sets would be
|
||||
// unreachable pool slots), so there is deliberately no age-based sweep here.
|
||||
void OnDescriptorSetLayoutDestroyed(VkDescriptorSetLayout descriptorSetLayout);
|
||||
Bool CollectSampledTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures);
|
||||
@@ -58,6 +69,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static VkFormat ResolveStorageImageViewFormat(VkFormat reflectedFormat, GLenum bindingFormat,
|
||||
VkFormat resourceFormat, Bool useBindingFormat);
|
||||
|
||||
// True when the program reads at least one sampler and every one of them is bound to a
|
||||
// texture whose GL level range is a single level. Such a sampler resolves to
|
||||
// minLod = maxLod = 0 (see VkSamplerManager::GetOrCreateSampler), so an implicit-LOD sample
|
||||
// and an explicit LOD 0 sample must read the same texel - which is what makes the
|
||||
// ExplicitLod0Sampling SPIR-V rewrite safe to request. Deliberately conservative: it reads
|
||||
// only GL state, so a texture that ends up single-level for another reason (one uploaded
|
||||
// level under a wide level range) merely misses the rewrite.
|
||||
// True when every texture this program samples is an 8-bit-or-less normalized format, so
|
||||
// relaxing the fragment stage to fp16 cannot lose a bit the texel ever carried. Says
|
||||
// nothing about the render target - the caller must check that too.
|
||||
static Bool ProgramSamplesOnlyLowPrecisionTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
// True when every colour attachment the draw writes is an 8-bit-or-less normalized
|
||||
// format (nullptr = default framebuffer, which is). Blending happens at attachment
|
||||
// precision, so a wider target must keep the fragment stage at full precision.
|
||||
static Bool DrawTargetIsLowPrecision(const MG_State::GLState::FramebufferObject* drawFramebuffer);
|
||||
static Bool ProgramSamplesOnlySingleLevelTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
|
||||
private:
|
||||
struct DescriptorPoolBucket {
|
||||
VkDescriptorPool handle = VK_NULL_HANDLE;
|
||||
@@ -65,8 +95,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 allocatedSets = 0;
|
||||
};
|
||||
|
||||
// A cached descriptor set together with the pool it was allocated from, so a
|
||||
// layout-destroyed purge can vkFreeDescriptorSets it back and credit the
|
||||
// owning bucket's accounting.
|
||||
struct CachedDescriptorSet {
|
||||
VkDescriptorSet set = VK_NULL_HANDLE;
|
||||
VkDescriptorPool pool = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
struct DescriptorSetCacheEntry {
|
||||
Vector<VkDescriptorSet> sets;
|
||||
Vector<CachedDescriptorSet> sets;
|
||||
Uint32 cursor = 0;
|
||||
};
|
||||
|
||||
@@ -116,7 +154,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
Bool ResolveUniformBufferPayload(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
UboBindResult& out) const;
|
||||
Uint32 arrayElement, UboBindResult& out) const;
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
|
||||
VkResult AllocateDescriptorSetsFromActivePool(
|
||||
@@ -172,6 +210,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 samplerLifetimeId = 0;
|
||||
Uint64 textureLifetimeId = 0;
|
||||
VkSampler sampler = VK_NULL_HANDLE;
|
||||
Uint32 viewLevelCount = 0;
|
||||
Uint16 samplerVersion = 0;
|
||||
Uint16 textureParamsVersion = 0;
|
||||
Bool forceNearestFiltering = false;
|
||||
|
||||
@@ -32,6 +32,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.IsBgra, sizeof(attr.IsBgra)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.Divisor, sizeof(attr.Divisor)));
|
||||
|
||||
// The buffer's heap address is an identity component of the key: a freed
|
||||
// buffer's reused address can alias an old cache entry, but only under a
|
||||
// byte-identical attribute layout - and the entry payload is a pure function
|
||||
// of the hashed inputs, with the draw path re-resolving bindingBufferKeys
|
||||
// against the live VAO attribute pointers, so an aliased hit returns exactly
|
||||
// what a rebuild would. Address drift only grows the map; the OnFrameBoundary
|
||||
// aging sweep bounds that.
|
||||
const SizeT bufferKey = reinterpret_cast<SizeT>(attr.Buffer.get());
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &bufferKey, sizeof(bufferKey)));
|
||||
}
|
||||
@@ -58,6 +65,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash) {
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return it->second;
|
||||
}
|
||||
|
||||
@@ -166,6 +174,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
auto& entry = m_cache[hash];
|
||||
entry.hash = hash;
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
entry.bindings = builder.GetBindings();
|
||||
entry.attributes = builder.GetAttributes();
|
||||
entry.bindingBufferKeys = std::move(bindingBufferKeys);
|
||||
@@ -180,6 +189,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return entry;
|
||||
}
|
||||
|
||||
void VertexInputStateFactory::OnFrameBoundary() {
|
||||
++m_frameBoundaryCounter;
|
||||
|
||||
// Sweep occasionally; evict entries whose last hit is far in the past.
|
||||
// Erasure happens only here, never mid-frame: the draw path holds a
|
||||
// reference into the current entry across its setup, and unordered_map
|
||||
// erase would invalidate it. Entries are CPU-side only, so no GPU-idle
|
||||
// proof is needed; an evicted entry that is used again is simply rebuilt
|
||||
// from the VAO state (same hash, same content).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
if ((m_frameBoundaryCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameBoundaryCounter - it->second.lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
it = m_cache.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VertexInputStateFactory::ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger,
|
||||
Bool isBgra) {
|
||||
if (isBgra) {
|
||||
|
||||
@@ -27,6 +27,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
struct BackendVertexInputState {
|
||||
HashType hash = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age are evicted (CPU heap only).
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
Vector<VkVertexInputBindingDescription> bindings;
|
||||
Vector<VkVertexInputAttributeDescription> attributes;
|
||||
Vector<SizeT> bindingBufferKeys;
|
||||
@@ -55,6 +58,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash);
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(const MG_State::GLState::VertexArrayObject& vao);
|
||||
// Frame boundary hook: ages the cache and evicts entries not hit for many
|
||||
// frames. The key mixes buffer heap addresses, so buffer/VAO churn keeps
|
||||
// minting fresh keys; without eviction the map grows for the whole session.
|
||||
// Entries hold no Vulkan handles (pipeline creation copies the descriptions)
|
||||
// and the draw path's entry reference never spans a frame boundary, so
|
||||
// eviction here needs no GPU-idle proof. Self-gated: one counter bump and
|
||||
// compare except on sweep boundaries.
|
||||
void OnFrameBoundary();
|
||||
static SizeT GetComponentSize(DataType type);
|
||||
// Tightly-packed byte size of one vertex element for this attribute: componentSize * size for
|
||||
// normal types, and 4 (one packed word) for the 2_10_10_10 types and GL_BGRA. Returns 0 for
|
||||
@@ -70,6 +81,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, BackendVertexInputState> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -141,6 +141,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_transientUploadArena.BeginFrame(frameIndex);
|
||||
}
|
||||
|
||||
void VkBufferManager::CollectAllDeferredReleases() {
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_deferredBufferReleases.size(); ++frameIndex) {
|
||||
CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_transientUploadArena.GetFrameCount(); ++frameIndex) {
|
||||
m_transientUploadArena.CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::NotifyDeviceIdle() {
|
||||
// Everything submitted so far has completed. Work recorded for the
|
||||
// current frame has not been submitted yet, so the current serial
|
||||
|
||||
@@ -77,6 +77,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Recreate all per-frame transient arenas
|
||||
Bool RecreateTransientArenas(Uint32 frameCount);
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Drains every frame slot's deferred buffer/resource releases (and the
|
||||
// transient arena's parked superseded blocks). Only valid when the
|
||||
// caller has proven every queue submission complete; used by the
|
||||
// present-less frame-boundary drain.
|
||||
void CollectAllDeferredReleases();
|
||||
// All previously submitted GPU work has completed (vkDeviceWaitIdle).
|
||||
void NotifyDeviceIdle();
|
||||
// A frame slot's submission fence has been waited: every serial up to
|
||||
|
||||
@@ -180,6 +180,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
internalFormat = TextureInternalFormat::Unknown;
|
||||
samples = 0;
|
||||
deadSinceFrame = kNeverObservedDead;
|
||||
}
|
||||
|
||||
VkRenderPassManager::VkRenderPassManager(VkDevice device,
|
||||
@@ -206,6 +207,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.Destroy(m_device, m_allocator);
|
||||
}
|
||||
m_renderbufferResources.clear();
|
||||
CollectDeferredRenderbufferReleases(/*destroyAll=*/true); // caller guarantees device idle
|
||||
m_pendingRenderbufferClears.clear();
|
||||
RenderPassEntry::s_textureResourcesScratch.clear();
|
||||
s_activeRenderPass = {};
|
||||
@@ -213,22 +215,75 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastValid = false;
|
||||
}
|
||||
|
||||
void VkRenderPassManager::CollectRenderbufferGarbage() {
|
||||
Vector<MG_State::GLState::RenderbufferObject*> deadRenderbuffers;
|
||||
deadRenderbuffers.reserve(m_renderbufferResources.size());
|
||||
for (auto& [renderbuffer, resource] : m_renderbufferResources) {
|
||||
const auto liveRenderbuffer = resource.renderbuffer.lock();
|
||||
if (!liveRenderbuffer || liveRenderbuffer.get() != renderbuffer) {
|
||||
deadRenderbuffers.emplace_back(renderbuffer);
|
||||
}
|
||||
Uint64 VkRenderPassManager::RetireAgeFrames() const {
|
||||
// MaxFramesInFlight + 2 covers the frame ring plus one boundary for the
|
||||
// recording-to-submit gap and one because OnPresent runs ahead of Present's
|
||||
// fence wait; the floor of 8 keeps a margin over the default ring of 3 while
|
||||
// still releasing multi-MB attachment memory promptly (the render-pass cache's
|
||||
// 1024-frame retirement would pin it for no additional safety).
|
||||
return std::max<Uint64>(8, static_cast<Uint64>(m_config.MaxFramesInFlight) + 2);
|
||||
}
|
||||
|
||||
void VkRenderPassManager::DeferRenderbufferBackingRelease(RenderbufferResource& resource) {
|
||||
// The superseded backing may still be referenced by in-flight command buffers
|
||||
// (glRenderbufferStorage can respecify a renderbuffer drawn this very frame),
|
||||
// so it is parked and destroyed only after RetireAgeFrames() boundaries.
|
||||
if (resource.image == VK_NULL_HANDLE && resource.view == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto* renderbuffer : deadRenderbuffers) {
|
||||
auto resourceIt = m_renderbufferResources.find(renderbuffer);
|
||||
if (resourceIt != m_renderbufferResources.end()) {
|
||||
resourceIt->second.Destroy(m_device, m_allocator);
|
||||
m_renderbufferResources.erase(resourceIt);
|
||||
m_deferredRenderbufferReleases.push_back({resource.image, resource.allocation, resource.view, m_frameCounter});
|
||||
resource.image = VK_NULL_HANDLE;
|
||||
resource.allocation = nullptr;
|
||||
resource.view = VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
void VkRenderPassManager::CollectDeferredRenderbufferReleases(Bool destroyAll) {
|
||||
if (m_deferredRenderbufferReleases.empty()) {
|
||||
return;
|
||||
}
|
||||
const Uint64 retireAgeFrames = RetireAgeFrames();
|
||||
std::erase_if(m_deferredRenderbufferReleases, [&](DeferredRenderbufferRelease& release) {
|
||||
if (!destroyAll && m_frameCounter - release.deferredAtFrame < retireAgeFrames) {
|
||||
return false;
|
||||
}
|
||||
m_pendingRenderbufferClears.erase(renderbuffer);
|
||||
if (release.view != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(m_device, release.view, nullptr);
|
||||
}
|
||||
if (release.image != VK_NULL_HANDLE) {
|
||||
vmaDestroyImage(m_allocator, release.image, release.allocation);
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
void VkRenderPassManager::CollectRenderbufferGarbage() {
|
||||
// Two-phase reclamation: a dead renderbuffer's VkImage may still be referenced by
|
||||
// command buffers submitted up to frames-in-flight frames ago (it was legally
|
||||
// attached and drawn right up to its deletion), so the first observation of an
|
||||
// expired weak reference only stamps the current frame counter; Destroy runs once
|
||||
// enough frame boundaries have passed that the stamping frame's submission fence
|
||||
// has provably been waited (see RetireAgeFrames).
|
||||
const Uint64 retireAgeFrames = RetireAgeFrames();
|
||||
for (auto it = m_renderbufferResources.begin(); it != m_renderbufferResources.end();) {
|
||||
auto& resource = it->second;
|
||||
const auto liveRenderbuffer = resource.renderbuffer.lock();
|
||||
if (liveRenderbuffer && liveRenderbuffer.get() == it->first) {
|
||||
resource.deadSinceFrame = RenderbufferResource::kNeverObservedDead;
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
if (resource.deadSinceFrame == RenderbufferResource::kNeverObservedDead) {
|
||||
resource.deadSinceFrame = m_frameCounter;
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
if (m_frameCounter - resource.deadSinceFrame < retireAgeFrames) {
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
m_pendingRenderbufferClears.erase(it->first);
|
||||
resource.Destroy(m_device, m_allocator);
|
||||
it = m_renderbufferResources.erase(it);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -251,11 +306,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const auto internalFormat = renderbuffer->GetInternalFormat();
|
||||
const VkFormat format = MG_Util::ConvertTextureInternalFormatToVkEnum(internalFormat);
|
||||
const VkImageAspectFlags aspect = ResolveImageAspectMaskForFormat(format);
|
||||
if ((aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0) {
|
||||
MGLOG_E("GetOrCreateRenderbufferResource: color renderbuffer %u is not supported by DirectVulkan render passes yet",
|
||||
renderbuffer->GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
// Renderbuffers are never sampled (GL has no way to bind one to a sampler), so the
|
||||
// usage set is attachment + transfer: transfer covers readback (vkCmdCopyImageToBuffer),
|
||||
// BlitFramebuffer, CopyTexImage sources, and out-of-render-pass clear materialization.
|
||||
const VkImageUsageFlags imageUsage =
|
||||
((aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0 ? VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT
|
||||
: VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT) |
|
||||
VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT;
|
||||
|
||||
auto& resource = m_renderbufferResources[renderbuffer.get()];
|
||||
const Bool needsCreate =
|
||||
@@ -268,9 +325,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.samples != renderbuffer->GetSamples();
|
||||
if (!needsCreate) {
|
||||
resource.renderbuffer = renderbuffer;
|
||||
// A new renderbuffer at a recycled address may adopt a compatible entry that
|
||||
// was already stamped dead; it is alive again, so cancel the aging.
|
||||
resource.deadSinceFrame = RenderbufferResource::kNeverObservedDead;
|
||||
return &resource;
|
||||
}
|
||||
|
||||
// Respecify: park the old backing for aged destruction instead of destroying
|
||||
// inline - it may still be referenced by in-flight command buffers.
|
||||
DeferRenderbufferBackingRelease(resource);
|
||||
resource.Destroy(m_device, m_allocator);
|
||||
resource.renderbuffer = renderbuffer;
|
||||
|
||||
@@ -285,7 +348,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
imageInfo.format = format;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
|
||||
imageInfo.usage = imageUsage;
|
||||
imageInfo.samples = sampleCount;
|
||||
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
|
||||
@@ -386,6 +449,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void VkRenderPassManager::QueueRenderbufferClear(
|
||||
GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo) {
|
||||
if ((mask & GL_COLOR_BUFFER_BIT) != 0) {
|
||||
// Color renderbuffer draw buffers take the framebuffer-level clear too; texture
|
||||
// attachments are skipped by the per-attachment overload's IsRenderbuffer guard.
|
||||
for (const auto attachmentType : drawFbo.GetDrawBuffers()) {
|
||||
if (attachmentType == FramebufferAttachmentType::None) {
|
||||
continue;
|
||||
}
|
||||
QueueRenderbufferClear(
|
||||
ClearAttachmentPayload{.mask = GL_COLOR_BUFFER_BIT, .color = clearPayload.color},
|
||||
drawFbo.GetAttachment(attachmentType));
|
||||
}
|
||||
}
|
||||
if ((mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
||||
QueueRenderbufferClear(
|
||||
ClearAttachmentPayload{.mask = GL_DEPTH_BUFFER_BIT, .depth = clearPayload.depth},
|
||||
@@ -682,6 +757,83 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// assuming default FBO has the right param
|
||||
for (Uint32 i = 0; i < colorAttachmentSlotCount; ++i) {
|
||||
auto drawbuf = drawbufs[i];
|
||||
|
||||
// Renderbuffer color attachments mirror the texture path below, with the
|
||||
// resource (image/view/format/layout) coming from the render-pass manager's
|
||||
// renderbuffer store instead of the texture manager.
|
||||
if (drawbuf != FramebufferAttachmentType::None && !isDefaultFbo) {
|
||||
const auto& rbAtt = fbo.GetAttachment(drawbuf);
|
||||
if (rbAtt.IsRenderbuffer() && rbAtt.IsComplete()) {
|
||||
const auto& renderbuffer = rbAtt.GetRenderbuffer();
|
||||
auto* rbResource = GetOrCreateRenderbufferResource(renderbuffer);
|
||||
if (rbResource == nullptr || (rbResource->aspect & VK_IMAGE_ASPECT_COLOR_BIT) == 0) {
|
||||
MGLOG_E("GetOrCreateRenderPass: draw buffer slot %u on FBO %u has an unsupported color "
|
||||
"renderbuffer %u; using VK_ATTACHMENT_UNUSED",
|
||||
i, fbo.GetExternalIndex(), renderbuffer->GetExternalIndex());
|
||||
continue;
|
||||
}
|
||||
|
||||
const Uint32 rbAttachmentIndex = static_cast<Uint32>(attachmentDescriptions.size());
|
||||
attachmentDescriptions.emplace_back();
|
||||
VkAttachmentDescription& rbDesc = attachmentDescriptions.back();
|
||||
|
||||
ClearAttachmentPayload rbClearPayload{};
|
||||
Bool rbHasClear = GetPendingRenderbufferClear(renderbuffer.get(), rbClearPayload) &&
|
||||
(rbClearPayload.mask & GL_COLOR_BUFFER_BIT) != 0;
|
||||
if (rbHasClear &&
|
||||
MG_Util::GetBaseInternalFormatComponentCount(renderbuffer->GetInternalFormat()) == 3) {
|
||||
// RGB renderbuffers are backed by an RGBA image; the missing alpha reads as 1.
|
||||
rbClearPayload.color =
|
||||
FloatVec4(rbClearPayload.color.x(), rbClearPayload.color.y(),
|
||||
rbClearPayload.color.z(), 1.0f);
|
||||
}
|
||||
|
||||
const VkImageLayout trackedRbLayout = rbResource->layout;
|
||||
rbDesc.flags = 0;
|
||||
rbDesc.format = rbResource->format;
|
||||
rbDesc.samples = rbResource->sampleCount;
|
||||
rbDesc.loadOp = rbHasClear ? VK_ATTACHMENT_LOAD_OP_CLEAR :
|
||||
(trackedRbLayout == VK_IMAGE_LAYOUT_UNDEFINED ? VK_ATTACHMENT_LOAD_OP_DONT_CARE
|
||||
: VK_ATTACHMENT_LOAD_OP_LOAD);
|
||||
rbDesc.storeOp = VK_ATTACHMENT_STORE_OP_STORE;
|
||||
rbDesc.stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
|
||||
rbDesc.stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
|
||||
rbDesc.initialLayout = (rbHasClear || trackedRbLayout == VK_IMAGE_LAYOUT_UNDEFINED) ?
|
||||
VK_IMAGE_LAYOUT_UNDEFINED : trackedRbLayout;
|
||||
rbDesc.finalLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL;
|
||||
adoptRenderPassSampleCount(rbResource->sampleCount, "color",
|
||||
static_cast<Int>(renderbuffer->GetExternalIndex()));
|
||||
|
||||
if (rbHasClear) {
|
||||
pendingClearAttachments.emplace_back(PendingClearAttachmentInfo {
|
||||
.attachmentIndex = rbAttachmentIndex,
|
||||
.colorAttachmentSlot = i,
|
||||
.renderbuffer = renderbuffer.get(),
|
||||
.hasInlinePayload = true,
|
||||
.inlinePayload = rbClearPayload,
|
||||
});
|
||||
}
|
||||
|
||||
if (width == 0)
|
||||
width = static_cast<Int>(rbResource->extent.width);
|
||||
if (height == 0)
|
||||
height = static_cast<Int>(rbResource->extent.height);
|
||||
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
.target = TrackedAttachmentTarget::Renderbuffer,
|
||||
.renderbuffer = renderbuffer,
|
||||
.finalLayout = rbDesc.finalLayout,
|
||||
});
|
||||
textureResources.emplace_back(nullptr);
|
||||
attachmentViews.emplace_back(rbResource->view);
|
||||
MOBILEGL_ASSERT(attachmentViews.back() != VK_NULL_HANDLE,
|
||||
"GetOrCreateRenderPass: renderbuffer view missing at color attachment %d", i);
|
||||
|
||||
colorAttachmentRefs[i].attachment = rbAttachmentIndex;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
auto* texture = ResolveCompleteColorAttachmentTexture(fbo, drawbuf, i);
|
||||
if (texture == nullptr)
|
||||
continue;
|
||||
@@ -700,6 +852,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case TextureTarget::Texture2D:
|
||||
case TextureTarget::Texture2DArray:
|
||||
case TextureTarget::Texture2DMultisample:
|
||||
case TextureTarget::Texture2DMultisampleArray:
|
||||
case TextureTarget::Texture3D:
|
||||
case TextureTarget::TextureCubeMap:
|
||||
case TextureTarget::TextureCubeMapArray:
|
||||
case TextureTarget::TextureRectangle: {
|
||||
desc.flags = 0;
|
||||
desc.format = isDefaultFbo ?
|
||||
@@ -1083,6 +1239,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void VkRenderPassManager::OnPresent() {
|
||||
++m_frameCounter;
|
||||
|
||||
// Runs every frame boundary, ahead of the render-pass sweep gate below: the walk
|
||||
// is O(#renderbuffer resources) — single digits in practice — and per-frame
|
||||
// invocation keeps dead-resource reclaim latency at the aging bound instead of
|
||||
// coupling it to renderbuffer *use* (the GetOrCreateRenderbufferResource call
|
||||
// site never runs again once an app stops using renderbuffers).
|
||||
CollectRenderbufferGarbage();
|
||||
CollectDeferredRenderbufferReleases(/*destroyAll=*/false);
|
||||
|
||||
// Sweep occasionally; evict entries whose last use is far past every
|
||||
// in-flight frame so their VkRenderPass/VkFramebuffer can be destroyed
|
||||
// safely (RenderPassEntry's destructor releases the handles).
|
||||
@@ -1092,6 +1256,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
|
||||
// Collect the dying handles and notify once after the loop: pipelines hashed
|
||||
// on them share the entries' >kRetireAgeFrames idleness (they are only bound
|
||||
// by draws that hit those entries), so the observer may destroy them
|
||||
// immediately - and a single batched notification costs one pipeline-cache
|
||||
// scan instead of one per evicted pass.
|
||||
Vector<VkRenderPass> destroyedRenderPasses;
|
||||
const Uint64 activeHash = s_hasActiveRenderPass ? s_activeRenderPass.hash : 0;
|
||||
for (auto it = m_renderPasses.begin(); it != m_renderPasses.end();) {
|
||||
const Bool isActive = s_hasActiveRenderPass && it->first == activeHash;
|
||||
@@ -1099,11 +1269,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (m_rpFastValid && m_rpFastRenderPassHash == it->first) {
|
||||
m_rpFastValid = false;
|
||||
}
|
||||
destroyedRenderPasses.push_back(it->second.renderPass);
|
||||
it = m_renderPasses.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (!destroyedRenderPasses.empty() && m_evictionObserver != nullptr) {
|
||||
m_evictionObserver->OnRenderPassesDestroyed(destroyedRenderPasses);
|
||||
}
|
||||
}
|
||||
|
||||
Bool VkRenderPassManager::BeginRenderPass(VkCommandBuffer commandBuffer, RenderPassEntry& renderPassEntry) {
|
||||
|
||||
@@ -157,11 +157,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkRenderPassManager {
|
||||
public:
|
||||
using HashType = Uint64;
|
||||
|
||||
// Notified once per OnPresent sweep with every aged-out entry's VkRenderPass
|
||||
// value: pipelines are hashed on the raw handle, and once destroyed the value
|
||||
// may be recycled for an incompatible pass, so dependent caches must purge
|
||||
// everything keyed on them before any new pass can be created (the sweep and
|
||||
// the notification run back-to-back with no creation in between; observers
|
||||
// compare the values, never dereference them). Batched so a mass-idle cohort
|
||||
// (shader-pack switch, dimension exit) costs the observer one pipeline-cache
|
||||
// scan, not one per dying pass. The wholesale paths
|
||||
// (Shutdown/RecreateSwapchain) do not notify - their callers already drop
|
||||
// every pipeline outright.
|
||||
class IEvictionObserver {
|
||||
public:
|
||||
virtual ~IEvictionObserver() = default;
|
||||
virtual void OnRenderPassesDestroyed(const Vector<VkRenderPass>& renderPasses) = 0;
|
||||
};
|
||||
|
||||
VkRenderPassManager(VkDevice device,
|
||||
VkPhysicalDevice physicalDevice, VmaAllocator allocator, const VulkanRendererConfig& config,
|
||||
VkClearManager& clearManager, VkTextureManager& textureManager, SwapchainObject& swapchainObject);
|
||||
~VkRenderPassManager();
|
||||
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetEvictionObserver(IEvictionObserver* observer) { m_evictionObserver = observer; }
|
||||
|
||||
Bool Initialize();
|
||||
void Shutdown();
|
||||
|
||||
@@ -192,6 +212,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
UnorderedMap<Uint64, RenderPassEntry> m_renderPasses;
|
||||
// Monotonic frame counter (bumped in OnPresent) for render-pass cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
|
||||
// Bumped whenever a renderbuffer VkImage is (re)created; together with the texture
|
||||
// manager's image epoch this invalidates the render-pass fast path on any attachment
|
||||
@@ -211,7 +232,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 m_rpFastRbEpoch = 0;
|
||||
Uint64 m_rpFastRenderPassHash = 0;
|
||||
|
||||
public:
|
||||
struct RenderbufferResource {
|
||||
// deadSinceFrame sentinel: the owning weak reference has not been observed
|
||||
// expired. Dead resources age past every in-flight frame before Destroy
|
||||
// (see CollectRenderbufferGarbage); the GPU may still reference the image
|
||||
// for frames-in-flight frames after the GL object dies.
|
||||
static constexpr Uint64 kNeverObservedDead = UINT64_MAX;
|
||||
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
@@ -223,25 +251,47 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::Unknown;
|
||||
Int samples = 0;
|
||||
// m_frameCounter value at which the weak reference was first seen expired.
|
||||
Uint64 deadSinceFrame = kNeverObservedDead;
|
||||
|
||||
void Destroy(VkDevice device, VmaAllocator allocator);
|
||||
};
|
||||
|
||||
// Public so the renderer's blit/copy/readback bindings can source renderbuffer
|
||||
// attachments the same way texture attachments go through the texture manager.
|
||||
RenderbufferResource* GetOrCreateRenderbufferResource(
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
Bool GetPendingRenderbufferClear(MG_State::GLState::RenderbufferObject* renderbuffer,
|
||||
ClearAttachmentPayload& outPayload) const;
|
||||
|
||||
private:
|
||||
struct PendingRenderbufferClear {
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
ClearAttachmentPayload payload{};
|
||||
};
|
||||
|
||||
// A superseded renderbuffer backing (glRenderbufferStorage respecify) parked
|
||||
// until enough frame boundaries have passed that no in-flight command buffer
|
||||
// can still reference it; destroyed in OnPresent (see RetireAgeFrames).
|
||||
struct DeferredRenderbufferRelease {
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
VkImageView view = VK_NULL_HANDLE;
|
||||
Uint64 deferredAtFrame = 0;
|
||||
};
|
||||
|
||||
UnorderedMap<MG_State::GLState::RenderbufferObject*, RenderbufferResource> m_renderbufferResources;
|
||||
UnorderedMap<MG_State::GLState::RenderbufferObject*, PendingRenderbufferClear> m_pendingRenderbufferClears;
|
||||
Vector<DeferredRenderbufferRelease> m_deferredRenderbufferReleases;
|
||||
|
||||
RenderbufferResource* GetOrCreateRenderbufferResource(
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
Bool GetPendingRenderbufferClear(MG_State::GLState::RenderbufferObject* renderbuffer,
|
||||
ClearAttachmentPayload& outPayload) const;
|
||||
Bool HasPendingRenderbufferClear(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) const;
|
||||
void CollectRenderbufferGarbage();
|
||||
// Frame-boundary margin after which a resource last referenced by a retired
|
||||
// GL object (or superseded backing) is provably past every in-flight frame.
|
||||
Uint64 RetireAgeFrames() const;
|
||||
void DeferRenderbufferBackingRelease(RenderbufferResource& resource);
|
||||
void CollectDeferredRenderbufferReleases(Bool destroyAll);
|
||||
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
static inline ActiveRenderPassInfo s_activeRenderPass{};
|
||||
|
||||
@@ -51,6 +51,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Float ResolveEffectiveMinLod(const MG_State::GLState::SamplerObject& sampler, Float effectiveMaxLod) {
|
||||
return std::min(sampler.GetMinLod(), effectiveMaxLod);
|
||||
}
|
||||
|
||||
// A single-level view can only ever deliver the base level, but the LOD clamp must not be
|
||||
// collapsed to exactly 0: both GL and Vulkan pick magFilter over minFilter from the
|
||||
// *clamped* lambda, so maxLod = 0 would make every fragment magnify and quietly retire the
|
||||
// min filter. 0.25 is the value VkSamplerCreateInfo's own note prescribes for emulating
|
||||
// GL's non-mipmapped minification - large enough for lambda to stay positive, small enough
|
||||
// that a NEAREST mip mode still rounds down to level 0. Clamped rather than assigned, so a
|
||||
// texture whose GL_TEXTURE_MAX_LOD really is 0 keeps magnifying as GL says it must.
|
||||
Float ResolveSingleLevelMaxLod(const MG_State::GLState::SamplerObject& sampler, Bool singleLevelView) {
|
||||
const Float maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
return singleLevelView ? std::min(maxLod, 0.25f) : maxLod;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool VkSamplerManager::Initialize(const InitInfo& initInfo) {
|
||||
@@ -89,15 +101,43 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_config = nullptr;
|
||||
m_frameBoundaryCounter = 0;
|
||||
}
|
||||
|
||||
void VkSamplerManager::OnFrameBoundary() {
|
||||
++m_frameBoundaryCounter;
|
||||
|
||||
// Sweep occasionally; destroy samplers whose last use is far past every
|
||||
// in-flight frame. Destroy and erase must stay atomic, or Shutdown would
|
||||
// double-free the handle; an evicted key that recurs simply re-creates
|
||||
// its sampler on the next miss.
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
if ((m_frameBoundaryCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_samplers.begin(); it != m_samplers.end();) {
|
||||
auto& entry = it->second;
|
||||
if (m_frameBoundaryCounter - entry.lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
if (m_device != VK_NULL_HANDLE && entry.handle != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, entry.handle, nullptr);
|
||||
}
|
||||
it = m_samplers.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Uint64 VkSamplerManager::BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering) const {
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const {
|
||||
MOBILEGL_ASSERT(m_config != nullptr, "VkSamplerManager::BuildSamplerKey: m_config is null");
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, m_config->CacheVersion));
|
||||
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &forceNearestFiltering, sizeof(forceNearestFiltering)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &singleLevelView, sizeof(singleLevelView)));
|
||||
|
||||
const auto minFilter = sampler.GetMinFilter();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &minFilter, sizeof(minFilter)));
|
||||
@@ -111,7 +151,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapT, sizeof(wrapT)));
|
||||
const auto wrapR = sampler.GetWrapR();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapR, sizeof(wrapR)));
|
||||
const auto maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
const auto maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
const auto minLod = ResolveEffectiveMinLod(sampler, maxLod);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &minLod, sizeof(minLod)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &maxLod, sizeof(maxLod)));
|
||||
@@ -133,10 +173,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkSampler VkSamplerManager::GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering) {
|
||||
const Uint64 key = BuildSamplerKey(sampler, texture, forceNearestFiltering);
|
||||
Bool forceNearestFiltering, Uint32 viewLevelCount) {
|
||||
// A view that exposes a single mip level has no second level to blend with, so GL's
|
||||
// *_MIPMAP_* minification filters degenerate to plain filtering on the base level -
|
||||
// sampling is unchanged by pinning the Vulkan sampler to NEAREST mip mode at LOD 0.
|
||||
// It is not cosmetic: MobileGL backs such a view with a fully allocated mip chain whose
|
||||
// tail is never written, and a LINEAR mip mode lets the texture unit issue the level+1
|
||||
// fetch anyway. On Adreno that fetch lands in uninitialized UBWC pages (or past the
|
||||
// allocation for a genuinely single-level image) and faults the GPU - the same failure
|
||||
// the default-framebuffer blit shader had to work around with an explicit-LOD sample.
|
||||
const Bool singleLevelView = viewLevelCount == 1;
|
||||
const Uint64 key = BuildSamplerKey(sampler, texture, forceNearestFiltering, singleLevelView);
|
||||
auto it = m_samplers.find(key);
|
||||
if (it != m_samplers.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return it->second.handle;
|
||||
}
|
||||
|
||||
@@ -144,8 +194,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
samplerInfo.sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO;
|
||||
samplerInfo.magFilter = forceNearestFiltering ? VK_FILTER_NEAREST : ToVkFilter(sampler.GetMagFilter());
|
||||
samplerInfo.minFilter = forceNearestFiltering ? VK_FILTER_NEAREST : ToVkFilter(sampler.GetMinFilter());
|
||||
samplerInfo.mipmapMode = forceNearestFiltering ? VK_SAMPLER_MIPMAP_MODE_NEAREST
|
||||
: ToVkMipmapMode(sampler.GetMipmapMode());
|
||||
samplerInfo.mipmapMode = (forceNearestFiltering || singleLevelView)
|
||||
? VK_SAMPLER_MIPMAP_MODE_NEAREST
|
||||
: ToVkMipmapMode(sampler.GetMipmapMode());
|
||||
samplerInfo.addressModeU = ToVkAddressMode(sampler.GetWrapS());
|
||||
samplerInfo.addressModeV = ToVkAddressMode(sampler.GetWrapT());
|
||||
samplerInfo.addressModeW = ToVkAddressMode(sampler.GetWrapR());
|
||||
@@ -157,7 +208,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
samplerInfo.maxAnisotropy = maxAnisotropy;
|
||||
samplerInfo.compareEnable = sampler.GetCompareMode() == SamplerCompareMode::CompareToTexture ? VK_TRUE : VK_FALSE;
|
||||
samplerInfo.compareOp = ToVkCompareOp(ResolveCompareFunc(sampler, texture));
|
||||
samplerInfo.maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
// Must match BuildSamplerKey's resolution exactly.
|
||||
samplerInfo.maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
samplerInfo.minLod = ResolveEffectiveMinLod(sampler, samplerInfo.maxLod);
|
||||
samplerInfo.borderColor = ResolveVkBorderColor(sampler, texture);
|
||||
samplerInfo.unnormalizedCoordinates = VK_FALSE;
|
||||
@@ -169,6 +221,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
entry.handle = vkSampler;
|
||||
entry.externalIndex = sampler.GetExternalIndex();
|
||||
entry.version = sampler.GetVersion();
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
m_samplers[key] = entry;
|
||||
return vkSampler;
|
||||
}
|
||||
|
||||
@@ -33,20 +33,38 @@ public:
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
|
||||
// viewLevelCount is the mip-level count of the image view this sampler will be paired
|
||||
// with; 0 means "unknown, do not narrow". See GetOrCreateSampler for why it matters.
|
||||
VkSampler GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering = false);
|
||||
Bool forceNearestFiltering = false,
|
||||
Uint32 viewLevelCount = 0);
|
||||
// Frame boundary hook: ages the sampler cache and destroys samplers not used
|
||||
// for many frames. The key hashes continuous float state (lodBias, LOD clamps,
|
||||
// anisotropy), so an app animating those would otherwise mint an unbounded
|
||||
// stream of never-destroyed VkSamplers and eventually exhaust the device's
|
||||
// maxSamplerAllocationCount. A sampler idle for over a thousand frame
|
||||
// boundaries cannot be referenced by any in-flight command buffer (frames in
|
||||
// flight are single digits), and every descriptor set the GPU consumes is
|
||||
// written that same frame with live handles (the per-binding resolve memo and
|
||||
// descriptor-set reuse are both frame-reset), so destruction here needs no
|
||||
// fence wait. Self-gated: one counter bump and compare except on sweep
|
||||
// boundaries.
|
||||
void OnFrameBoundary();
|
||||
|
||||
private:
|
||||
struct SamplerCacheEntry {
|
||||
VkSampler handle = VK_NULL_HANDLE;
|
||||
Uint externalIndex = 0;
|
||||
Uint16 version = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age have their VkSampler destroyed.
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
};
|
||||
|
||||
Uint64 BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering) const;
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const;
|
||||
static VkFilter ToVkFilter(SamplerFilterMode mode);
|
||||
static VkSamplerMipmapMode ToVkMipmapMode(SamplerMipmapMode mode);
|
||||
static VkSamplerAddressMode ToVkAddressMode(SamplerWrapMode mode);
|
||||
@@ -67,6 +85,8 @@ private:
|
||||
Bool m_samplerAnisotropySupported = false;
|
||||
Float m_maxSamplerAnisotropy = 1.0f;
|
||||
UnorderedMap<Uint64, SamplerCacheEntry> m_samplers;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -375,7 +375,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
// Legacy low-bit RGB formats share the UNorm8 canonical shadow layout (see
|
||||
// TextureFormatProcessor), so they upload exactly like RGB8 with an alpha expand.
|
||||
case TextureInternalFormat::R3G3B2:
|
||||
case TextureInternalFormat::RGB4:
|
||||
case TextureInternalFormat::RGB5:
|
||||
return {VK_FORMAT_R8G8B8A8_UNORM, true, 1, {0xFF, 0x00, 0x00, 0x00}};
|
||||
// Low-bit RGBA formats: UNorm8x4 canonical shadow, no expansion needed.
|
||||
case TextureInternalFormat::RGBA2:
|
||||
case TextureInternalFormat::RGBA4:
|
||||
case TextureInternalFormat::RGB5A1:
|
||||
return {VK_FORMAT_R8G8B8A8_UNORM, false, 0, {0, 0, 0, 0}};
|
||||
// 10/12-bit RGB(A): UNorm16 canonical shadow.
|
||||
case TextureInternalFormat::RGB10:
|
||||
case TextureInternalFormat::RGB12:
|
||||
return {VK_FORMAT_R16G16B16A16_UNORM, true, 2, {0xFF, 0xFF, 0x00, 0x00}};
|
||||
case TextureInternalFormat::RGBA12:
|
||||
return {VK_FORMAT_R16G16B16A16_UNORM, false, 0, {0, 0, 0, 0}};
|
||||
case TextureInternalFormat::SRGB8:
|
||||
return {VK_FORMAT_R8G8B8A8_SRGB, true, 1, {0xFF, 0x00, 0x00, 0x00}};
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
@@ -571,6 +587,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_allocator = initInfo.allocator;
|
||||
m_commandPool = initInfo.commandPool;
|
||||
m_graphicsQueue = initInfo.graphicsQueue;
|
||||
m_imageFormatListSupported = initInfo.imageFormatListSupported;
|
||||
m_currentFrameIndex = 0;
|
||||
m_deferredReleases.clear();
|
||||
m_deferredReleases.resize(initInfo.frameCount);
|
||||
@@ -593,6 +610,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
DestroyDeferredReleases();
|
||||
m_textureResources.clear();
|
||||
m_aliveObjects.clear();
|
||||
m_storageImageTextures.clear();
|
||||
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_physicalDevice = VK_NULL_HANDLE;
|
||||
@@ -611,6 +629,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frameIndex, m_deferredViewReleases.size());
|
||||
m_currentFrameIndex = frameIndex;
|
||||
CollectDeferredReleases(frameIndex);
|
||||
|
||||
// Frame-boundary GC: every 64 frame boundaries (~1 s at 60 fps) bounds the reclaim
|
||||
// latency for dead textures regardless of draw traffic — workloads that churn
|
||||
// textures through clears/readbacks alone never reach the draw-gated
|
||||
// CollectGarbage. Must run after CollectDeferredReleases above: the prune defers
|
||||
// its releases into this frame's slot, which was just drained, so they are
|
||||
// destroyed only after the slot's fence has been waited again one full frame-ring
|
||||
// cycle from now (never while an in-flight frame may still reference them).
|
||||
constexpr Uint32 kGcFrameInterval = 64;
|
||||
++m_gcFrameCounter;
|
||||
if (m_gcFrameCounter % kGcFrameInterval == 0) {
|
||||
PruneDeadTextures();
|
||||
}
|
||||
}
|
||||
|
||||
void VkTextureManager::CollectAllDeferredReleases() {
|
||||
const SizeT frameCount = std::min(m_deferredReleases.size(), m_deferredViewReleases.size());
|
||||
for (SizeT frameIndex = 0; frameIndex < frameCount; ++frameIndex) {
|
||||
CollectDeferredReleases(static_cast<Uint32>(frameIndex));
|
||||
}
|
||||
}
|
||||
|
||||
void VkTextureManager::EraseTrackedTexture(const TextureIdentity& identity) {
|
||||
@@ -620,6 +658,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_textureResources.erase(resourceIt);
|
||||
}
|
||||
m_aliveObjects.erase(identity);
|
||||
m_storageImageTextures.erase(identity);
|
||||
}
|
||||
|
||||
void VkTextureManager::PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture) {
|
||||
@@ -686,9 +725,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// construction introduces a new identity. Doing this unconditionally made every
|
||||
// sampled-texture sync scan the entire alive-texture map per draw.
|
||||
if (aliveIt == m_aliveObjects.end()) {
|
||||
WeakPtr<MG_State::GLState::ITextureObject> aliveTexture;
|
||||
const auto& liveTexture = MG_State::pGLContext->GetTextureObject(texture.GetExternalIndex());
|
||||
if (liveTexture && liveTexture.get() == &texture) {
|
||||
m_aliveObjects[identity] = WeakPtr<MG_State::GLState::ITextureObject>(liveTexture);
|
||||
aliveTexture = liveTexture;
|
||||
} else {
|
||||
// The name lookup legally fails while the object is alive: the name was
|
||||
// deleted with the texture still attached to an FBO (the attachment's
|
||||
// SharedPtr keeps it alive), or the name was reused by a new texture, or
|
||||
// this is a default texture object (name 0 lives outside the name map).
|
||||
// Register through the object's own control block so the resource created
|
||||
// below still participates in weak-expiry GC instead of becoming an
|
||||
// orphan no reclamation path can reach until Shutdown.
|
||||
aliveTexture = texture.weak_from_this();
|
||||
}
|
||||
if (!aliveTexture.expired()) {
|
||||
m_aliveObjects[identity] = Move(aliveTexture);
|
||||
PruneStaleTextureAliases(&texture);
|
||||
}
|
||||
}
|
||||
@@ -1140,8 +1192,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return ok;
|
||||
}
|
||||
|
||||
void VkTextureManager::MarkStorageImageTexture(MG_State::GLState::ITextureObject& texture) {
|
||||
m_storageImageTextures.insert(MakeTextureIdentity(&texture));
|
||||
}
|
||||
|
||||
Bool VkTextureManager::NeedsStorageUsageUpgrade(MG_State::GLState::ITextureObject& texture) const {
|
||||
const TextureIdentity identity = MakeTextureIdentity(&texture);
|
||||
if (m_storageImageTextures.find(identity) == m_storageImageTextures.end()) {
|
||||
return false;
|
||||
}
|
||||
const auto it = m_textureResources.find(identity);
|
||||
// No image yet: the first sync creates it with STORAGE straight away, so there is nothing
|
||||
// to preserve and nothing to order against.
|
||||
return it != m_textureResources.end() && it->second.image != VK_NULL_HANDLE &&
|
||||
!it->second.storageUsageResolved;
|
||||
}
|
||||
|
||||
Bool VkTextureManager::NeedsStorageImagePreparation(MG_State::GLState::ITextureObject& texture) const {
|
||||
const auto it = m_textureResources.find(MakeTextureIdentity(&texture));
|
||||
const TextureIdentity identity = MakeTextureIdentity(&texture);
|
||||
const auto it = m_textureResources.find(identity);
|
||||
if (it == m_textureResources.end()) {
|
||||
return true;
|
||||
}
|
||||
@@ -1149,6 +1218,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (resource.image == VK_NULL_HANDLE || resource.layout != VK_IMAGE_LAYOUT_GENERAL) {
|
||||
return true;
|
||||
}
|
||||
// The image predates this texture's first image-unit binding, so it was created without
|
||||
// STORAGE usage and has to be recreated - which is illegal inside a render pass.
|
||||
if (!resource.storageUsageResolved &&
|
||||
m_storageImageTextures.find(identity) != m_storageImageTextures.end()) {
|
||||
return true;
|
||||
}
|
||||
// Mirror SyncTexture's cross-draw skip condition: any version drift means the sync
|
||||
// path may upload or rebuild, both of which need the render pass ended first.
|
||||
const auto* mipTexture = MG_State::GLState::AsMipmapTexture(&texture);
|
||||
@@ -1196,10 +1271,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
SizeT VkTextureManager::CollectGarbage() {
|
||||
// Draw-gated stagger (1 in 256 calls): keeps the per-draw cost at one counter
|
||||
// bump. The guaranteed reclaim path is the frame-boundary prune in BeginFrame;
|
||||
// this remains as a cheap assist so draw-heavy workloads reclaim sooner.
|
||||
m_gcCounter++;
|
||||
if (m_gcCounter != 0) {
|
||||
return 0;
|
||||
}
|
||||
return PruneDeadTextures();
|
||||
}
|
||||
|
||||
SizeT VkTextureManager::PruneDeadTextures() {
|
||||
// Erasing entries would dangle the raw TextureResource pointers memoized for the
|
||||
// current draw; every call path (BeginFrame, and CollectGarbage at the top of a
|
||||
// freshly opened draw-sync scope) runs before any memo entry is recorded.
|
||||
MOBILEGL_ASSERT(m_drawSyncedThisDraw.empty(),
|
||||
"PruneDeadTextures: draw-sync memo holds raw resource pointers an erase would dangle");
|
||||
|
||||
Vector<MG_State::GLState::ITextureObject*> expiredTextures;
|
||||
expiredTextures.reserve(m_aliveObjects.size());
|
||||
@@ -1211,7 +1298,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (auto* texture : expiredTextures) {
|
||||
PruneStaleTextureAliases(texture);
|
||||
}
|
||||
return expiredTextures.size();
|
||||
SizeT prunedCount = expiredTextures.size();
|
||||
|
||||
// Orphan sweep: after the pass above, m_aliveObjects holds only live entries.
|
||||
// Registration in SyncTextureAndGetDescriptor cannot fail for a SharedPtr-owned
|
||||
// texture (weak_from_this fallback), so a resource whose identity has no alive
|
||||
// entry has no trackable owner: its GL-side object is gone, or was never
|
||||
// shared-owned, in which case recreation on a later sync is the safe fallback.
|
||||
// Destruction goes through the per-frame deferred queues, never immediate.
|
||||
Vector<TextureIdentity> orphanIdentities;
|
||||
for (auto it = m_textureResources.begin(); it != m_textureResources.end(); ++it) {
|
||||
if (m_aliveObjects.find(it->first) == m_aliveObjects.end()) {
|
||||
orphanIdentities.emplace_back(it->first);
|
||||
}
|
||||
}
|
||||
for (const auto& identity : orphanIdentities) {
|
||||
EraseTrackedTexture(identity);
|
||||
}
|
||||
prunedCount += orphanIdentities.size();
|
||||
return prunedCount;
|
||||
}
|
||||
|
||||
Bool VkTextureManager::SyncTexture(MG_State::GLState::ITextureObject &texture,
|
||||
@@ -1225,7 +1330,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const auto* syncingMipTexture = MG_State::GLState::AsMipmapTexture(&texture);
|
||||
const Uint32 syncingMipLevelCount =
|
||||
syncingMipTexture != nullptr ? syncingMipTexture->GetMipmapLevelCount() : 0u;
|
||||
if (outResource.image != VK_NULL_HANDLE &&
|
||||
// A pending storage-usage upgrade also has to bust the skip: nothing about the texture's
|
||||
// content or params changed, but the image itself must be recreated with STORAGE usage
|
||||
// before it can back an image-unit descriptor.
|
||||
const Bool storageUpgradePending =
|
||||
!outResource.storageUsageResolved &&
|
||||
m_storageImageTextures.find(MakeTextureIdentity(&texture)) != m_storageImageTextures.end();
|
||||
if (outResource.image != VK_NULL_HANDLE && !storageUpgradePending &&
|
||||
outResource.syncedContentVersion == syncingContentVersion &&
|
||||
outResource.syncedTextureParamsVersion == texture.GetTextureParamsVersion() &&
|
||||
outResource.syncedMipLevelCount == syncingMipLevelCount) {
|
||||
@@ -1336,16 +1447,44 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VkImageAspectFlags aspect = GetAspectMaskForFormat(format);
|
||||
VkFormatProperties formatProperties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &formatProperties);
|
||||
const Bool supportsStorageImage =
|
||||
// Only textures that have actually been bound to a GL image unit get STORAGE usage (and
|
||||
// the MUTABLE_FORMAT it drags in for format-reinterpreting image views). Requesting it
|
||||
// for every storage-capable colour texture costs real bandwidth: Adreno cannot keep UBWC
|
||||
// compression on an image that may be written through a storage descriptor, so the whole
|
||||
// render target - MC's included - runs uncompressed. MarkStorageImageTexture upgrades a
|
||||
// texture before its first image-unit draw, and the usage below feeds the compatibility
|
||||
// check so the upgrade recreates the image.
|
||||
const Bool markedAsStorageImage =
|
||||
m_storageImageTextures.find(MakeTextureIdentity(
|
||||
const_cast<MG_State::GLState::ITextureObject*>(&texture))) != m_storageImageTextures.end();
|
||||
// Storage-image CAPABILITY (does the format allow it at all) is deliberately separate from
|
||||
// whether this texture actually needs the usage. MUTABLE_FORMAT keys off capability, as
|
||||
// before: format-reinterpreting views are not a storage-only concern - the SAMPLED path
|
||||
// needs them too (GetOrCreateSampledImageView bails out without it, see ~line 892), so
|
||||
// tying MUTABLE_FORMAT to the image-unit mark would break sampled format reinterpretation
|
||||
// for every texture that never becomes a storage image.
|
||||
const Bool storageImageCapable =
|
||||
!isMultisampleTexture &&
|
||||
(aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0 &&
|
||||
(formatProperties.optimalTilingFeatures & VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0;
|
||||
const Bool supportsStorageImage = storageImageCapable && markedAsStorageImage;
|
||||
VkImageCreateFlags imageCreateFlags = shapeInfo.imageFlags;
|
||||
if (supportsStorageImage && IsMutableStorageImageFormat(format) &&
|
||||
if (storageImageCapable && IsMutableStorageImageFormat(format) &&
|
||||
m_mutableFormatUnsupported.find(format) == m_mutableFormatUnsupported.end()) {
|
||||
imageCreateFlags |= VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT;
|
||||
}
|
||||
|
||||
VkImageUsageFlags desiredUsage =
|
||||
VK_IMAGE_USAGE_SAMPLED_BIT |
|
||||
(supportsStorageImage ? VK_IMAGE_USAGE_STORAGE_BIT : 0) |
|
||||
((aspect & VK_IMAGE_ASPECT_COLOR_BIT) ? VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT : 0) |
|
||||
(((aspect & VK_IMAGE_ASPECT_DEPTH_BIT) || (aspect & VK_IMAGE_ASPECT_STENCIL_BIT)) ?
|
||||
VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT :
|
||||
0);
|
||||
if (!isMultisampleTexture) {
|
||||
desiredUsage |= VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
|
||||
}
|
||||
|
||||
const Bool compatible = resource.image != VK_NULL_HANDLE && resource.format == format &&
|
||||
resource.extent.width == static_cast<Uint32>(texelSize.x()) &&
|
||||
resource.extent.height == static_cast<Uint32>(texelSize.y()) &&
|
||||
@@ -1354,6 +1493,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.viewType == shapeInfo.viewType &&
|
||||
resource.sampleCount == resolvedSampleCount &&
|
||||
resource.imageCreateFlags == imageCreateFlags &&
|
||||
resource.usageFlags == desiredUsage &&
|
||||
resource.mipLevels == backingMipLevels;
|
||||
if (compatible) {
|
||||
if (resource.perMipViews.size() != backingMipLevels) {
|
||||
@@ -1362,6 +1502,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (resource.perMipSampledViews.size() != backingMipLevels) {
|
||||
resource.perMipSampledViews.resize(backingMipLevels, VK_NULL_HANDLE);
|
||||
}
|
||||
// Keeping the image is itself the answer to the mark: either it already carries
|
||||
// STORAGE, or this format can never carry it. Either way there is nothing left to
|
||||
// recreate, so stop reporting the texture as needing preparation.
|
||||
resource.storageUsageResolved = markedAsStorageImage;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -1376,7 +1520,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.sampleCount == resolvedSampleCount &&
|
||||
resource.imageCreateFlags == imageCreateFlags &&
|
||||
resolvedSampleCount == VK_SAMPLE_COUNT_1_BIT &&
|
||||
resource.mipLevels < backingMipLevels &&
|
||||
// '<=' rather than '<': a storage-usage upgrade recreates the image with an
|
||||
// unchanged mip count, and its contents (a render target's pixels live only on the
|
||||
// GPU) still have to survive. The vkCmdCopyImage below copies min(mipLevels).
|
||||
resource.mipLevels <= backingMipLevels &&
|
||||
resource.layout != VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
|
||||
std::unique_ptr<TextureResource> preservedResource;
|
||||
@@ -1398,16 +1545,37 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
imageInfo.format = format;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_SAMPLED_BIT |
|
||||
(supportsStorageImage ? VK_IMAGE_USAGE_STORAGE_BIT : 0) |
|
||||
((aspect & VK_IMAGE_ASPECT_COLOR_BIT) ? VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT : 0) |
|
||||
(((aspect & VK_IMAGE_ASPECT_DEPTH_BIT) || (aspect & VK_IMAGE_ASPECT_STENCIL_BIT)) ?
|
||||
VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT :
|
||||
0);
|
||||
if (!isMultisampleTexture) {
|
||||
imageInfo.usage |= VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
|
||||
}
|
||||
imageInfo.usage = desiredUsage;
|
||||
imageInfo.samples = resolvedSampleCount;
|
||||
|
||||
// Bound the mutability. A blindly-mutable image has to be laid out so that ANY format in
|
||||
// its compatibility class can be viewed, which costs bandwidth compression on tilers;
|
||||
// naming the exact set instead lets the driver keep it. Only safe when that set really is
|
||||
// exhaustive, so it is restricted to textures that are not image-unit bound: sampled views
|
||||
// can only ever ask for ResolveSampledImageViewFormat's output, whereas glBindImageTexture
|
||||
// may name any compatible format, which nothing here can enumerate ahead of time.
|
||||
Vector<VkFormat> viewFormats;
|
||||
VkImageFormatListCreateInfo formatListInfo{};
|
||||
if (m_imageFormatListSupported && !supportsStorageImage &&
|
||||
(imageInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) {
|
||||
viewFormats.push_back(format);
|
||||
for (const SamplerNumericDomain domain : {SamplerNumericDomain::Float,
|
||||
SamplerNumericDomain::SignedInteger,
|
||||
SamplerNumericDomain::UnsignedInteger}) {
|
||||
const VkFormat viewFormat = ResolveSampledImageViewFormat(format, domain);
|
||||
if (viewFormat == VK_FORMAT_UNDEFINED) {
|
||||
continue;
|
||||
}
|
||||
if (std::find(viewFormats.begin(), viewFormats.end(), viewFormat) == viewFormats.end()) {
|
||||
viewFormats.push_back(viewFormat);
|
||||
}
|
||||
}
|
||||
formatListInfo.sType = VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO;
|
||||
formatListInfo.viewFormatCount = static_cast<Uint32>(viewFormats.size());
|
||||
formatListInfo.pViewFormats = viewFormats.data();
|
||||
imageInfo.pNext = &formatListInfo;
|
||||
}
|
||||
|
||||
if (isMultisampleTexture || (imageInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) {
|
||||
VkImageFormatProperties imageFormatProperties{};
|
||||
VkResult imageFormatResult = vkGetPhysicalDeviceImageFormatProperties(
|
||||
@@ -1464,6 +1632,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.viewType = shapeInfo.viewType;
|
||||
resource.sampleCount = resolvedSampleCount;
|
||||
resource.imageCreateFlags = imageCreateFlags;
|
||||
resource.usageFlags = imageInfo.usage;
|
||||
resource.storageUsageResolved = markedAsStorageImage;
|
||||
resource.syncedTextureParamsVersion = 0;
|
||||
|
||||
if (preservedResource) {
|
||||
|
||||
@@ -53,6 +53,9 @@ public:
|
||||
VkCommandPool commandPool = VK_NULL_HANDLE;
|
||||
VkQueue graphicsQueue = VK_NULL_HANDLE;
|
||||
Uint32 frameCount = 0;
|
||||
// VK_KHR_image_format_list is enabled: MUTABLE_FORMAT images can name the exact set of
|
||||
// formats they will be viewed as, which is what lets a tiler keep them compressed.
|
||||
Bool imageFormatListSupported = false;
|
||||
};
|
||||
|
||||
struct TextureResource {
|
||||
@@ -157,6 +160,17 @@ public:
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
VkImageCreateFlags imageCreateFlags = 0;
|
||||
// Usage the live image was created with. STORAGE is only requested for textures that
|
||||
// have actually been bound to a GL image unit, because on Adreno a storage-capable
|
||||
// image loses UBWC bandwidth compression; a later image binding upgrades the usage
|
||||
// and recreates the image, so the resolved usage has to be part of the compatibility
|
||||
// check that decides whether the existing image can be kept.
|
||||
VkImageUsageFlags usageFlags = 0;
|
||||
// True once this image was (re)resolved while the texture was already marked as an
|
||||
// image-unit texture. Distinguishes "not upgraded yet" from "cannot be upgraded"
|
||||
// (a format whose optimalTilingFeatures lack STORAGE_IMAGE never gains the bit), so
|
||||
// NeedsStorageImagePreparation cannot ask for a recreate that will never happen.
|
||||
Bool storageUsageResolved = false;
|
||||
Uint16 syncedTextureParamsVersion = 0;
|
||||
// Snapshot of ITextureObject::GetContentVersion() at the last successful sync;
|
||||
// lets SyncTexture skip the whole re-check/re-upload when content is unchanged.
|
||||
@@ -190,6 +204,8 @@ public:
|
||||
std::swap(this->viewType, that.viewType);
|
||||
std::swap(this->sampleCount, that.sampleCount);
|
||||
std::swap(this->imageCreateFlags, that.imageCreateFlags);
|
||||
std::swap(this->usageFlags, that.usageFlags);
|
||||
std::swap(this->storageUsageResolved, that.storageUsageResolved);
|
||||
std::swap(this->syncedTextureParamsVersion, that.syncedTextureParamsVersion);
|
||||
std::swap(this->syncedContentVersion, that.syncedContentVersion);
|
||||
std::swap(this->syncedMipLevelCount, that.syncedMipLevelCount);
|
||||
@@ -251,6 +267,8 @@ public:
|
||||
viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageCreateFlags = 0;
|
||||
usageFlags = 0;
|
||||
storageUsageResolved = false;
|
||||
syncedTextureParamsVersion = 0;
|
||||
syncedContentVersion = 0;
|
||||
syncedMipLevelCount = 0;
|
||||
@@ -267,6 +285,10 @@ public:
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Drains every frame slot's deferred image/view releases. Only valid when
|
||||
// the caller has proven every queue submission complete; used by the
|
||||
// present-less frame-boundary drain.
|
||||
void CollectAllDeferredReleases();
|
||||
|
||||
TextureResource* SyncTextureAndGetDescriptor(
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
@@ -285,6 +307,17 @@ public:
|
||||
VkImageLayout newLayout);
|
||||
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
// Records that this texture is bound to a GL image unit, so its image must carry
|
||||
// VK_IMAGE_USAGE_STORAGE_BIT. Must be called before NeedsStorageImagePreparation, and
|
||||
// therefore before the render pass is committed: an image that has to be upgraded is
|
||||
// recreated, which is illegal inside a render pass. Sticky for the texture's lifetime -
|
||||
// GL lets an image binding come and go, and re-creating the image every time it does
|
||||
// would cost far more than the compression it wins back.
|
||||
void MarkStorageImageTexture(MG_State::GLState::ITextureObject& texture);
|
||||
// True when this texture is marked but its live image predates the mark, i.e. the next sync
|
||||
// will recreate it with STORAGE usage and copy the old contents forward. Callers use this to
|
||||
// submit their pending recording first, so that copy cannot read pre-flush content.
|
||||
Bool NeedsStorageUsageUpgrade(MG_State::GLState::ITextureObject& texture) const;
|
||||
// Non-mutating probe for the per-draw storage-image fast path: true when preparing this
|
||||
// texture as a storage image may need work that is illegal inside a render pass (resource
|
||||
// creation, dirty-content upload, or a layout transition to GENERAL). Unknown state reports
|
||||
@@ -364,15 +397,20 @@ private:
|
||||
static TextureIdentity MakeTextureIdentity(MG_State::GLState::ITextureObject* texture);
|
||||
void EraseTrackedTexture(const TextureIdentity& identity);
|
||||
void PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture);
|
||||
SizeT PruneDeadTextures();
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
VkCommandPool m_commandPool = VK_NULL_HANDLE;
|
||||
VkQueue m_graphicsQueue = VK_NULL_HANDLE;
|
||||
Bool m_imageFormatListSupported = false;
|
||||
Uint32 m_currentFrameIndex = 0;
|
||||
|
||||
Uint8 m_gcCounter = 0;
|
||||
// Frame-boundary GC gate: counts BeginFrame calls, not draws, so texture churn
|
||||
// through non-draw paths (FBO clears, readbacks) still reaches the prune.
|
||||
Uint32 m_gcFrameCounter = 0;
|
||||
// Active only between BeginDrawSyncScope/EndDrawSyncScope; identities of
|
||||
// textures already fully synced in the current draw (small N -> flat scan).
|
||||
Bool m_drawSyncScopeActive = false;
|
||||
@@ -390,6 +428,8 @@ private:
|
||||
std::unordered_set<VkFormat> m_mutableFormatUnsupported;
|
||||
std::unordered_map<TextureIdentity, WeakPtr<MG_State::GLState::ITextureObject>, TextureIdentityHash> m_aliveObjects;
|
||||
std::unordered_map<TextureIdentity, TextureResource, TextureIdentityHash> m_textureResources;
|
||||
// Textures that have been bound to a GL image unit (see MarkStorageImageTexture).
|
||||
std::unordered_set<TextureIdentity, TextureIdentityHash> m_storageImageTextures;
|
||||
Vector<Vector<TextureResource>> m_deferredReleases;
|
||||
Vector<Vector<VkImageView>> m_deferredViewReleases;
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -114,7 +114,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
};
|
||||
|
||||
class VulkanRenderer : public IBufferCopyCommandProvider, public FrameContext::IRecordingObserver {
|
||||
class VulkanRenderer : public IBufferCopyCommandProvider,
|
||||
public FrameContext::IRecordingObserver,
|
||||
public VkRenderPassManager::IEvictionObserver,
|
||||
public ProgramFactory::IEvictionObserver {
|
||||
public:
|
||||
VulkanRenderer(NativeWindowType window, const VulkanRendererConfig& cfg = {});
|
||||
~VulkanRenderer();
|
||||
@@ -131,6 +134,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// recording, before any render pass.
|
||||
void OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer) override;
|
||||
|
||||
// VkRenderPassManager::IEvictionObserver: the render-pass aging sweep just
|
||||
// destroyed these VkRenderPasses; evict every graphics pipeline hashed on a
|
||||
// dying handle (they share its >1024-boundary idleness, so immediate
|
||||
// destruction is safe) and drop the last-pipeline memo if any went.
|
||||
void OnRenderPassesDestroyed(const Vector<VkRenderPass>& renderPasses) override;
|
||||
|
||||
// ProgramFactory::IEvictionObserver: an aged-out program entry was
|
||||
// destroyed; evict its compute pipeline and graphics pipelines (same
|
||||
// idleness guarantee - they are only bound through draws/dispatches that
|
||||
// stamp the program entry) and purge the descriptor-set cache entries
|
||||
// keyed by its now-recyclable VkDescriptorSetLayout handle.
|
||||
void OnProgramEvicted(ProgramFactory::HashType programHash,
|
||||
VkDescriptorSetLayout descriptorSetLayout) override;
|
||||
|
||||
Bool SetupDraw(FrameContext::FrameData& frame, GLenum mode, Flags<DrawSetupAspect> aspects,
|
||||
const DrawCmdParam& drawParams,
|
||||
const IndexBufferView* pIndexBufferView = nullptr);
|
||||
@@ -257,7 +274,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 GetTimerQueryTimestampNs(const VkTimerQueryManager::TimestampRecord& record) const;
|
||||
|
||||
void RequestSwapchainResize(Uint32 width, Uint32 height);
|
||||
void RecreateSwapchain();
|
||||
// Re-query the surface and report whether the live swapchain no longer matches it
|
||||
// (size or orientation). This - not a VK_SUBOPTIMAL_KHR result - is what decides a
|
||||
// rebuild, so a surface the driver merely considers suboptimal cannot thrash.
|
||||
Bool SwapchainIsOutOfDate();
|
||||
// Returns false when the surface is zero-area (minimized/hidden window):
|
||||
// no new swapchain is installed and presentation must stay suspended.
|
||||
Bool RecreateSwapchain();
|
||||
|
||||
private:
|
||||
struct BlitUniformData {
|
||||
@@ -343,11 +366,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkFence AcquirePooledSubmitFence();
|
||||
void DestroySubmitFencePool();
|
||||
Bool HasPendingRecordedWork() const;
|
||||
// Frame-boundary housekeeping for paths that never reach Present's
|
||||
// tail (present-less readback loops, suspended presentation, blocking
|
||||
// sync waits): runs the same per-frame drains Present performs, but
|
||||
// only when every queue submission has been observed complete AND no
|
||||
// recorded-but-unsubmitted commands exist - i.e. when CPU-GPU overlap
|
||||
// is provably already zero. Never blocks (non-blocking fence poll
|
||||
// only), so the presenting path's frames-in-flight pipelining is
|
||||
// untouched. Returns true when the drain ran.
|
||||
Bool TryDrainFrameTransients();
|
||||
|
||||
Vector<SubmitRecord> m_inFlightSubmits;
|
||||
Vector<VkFence> m_freeSubmitFences;
|
||||
Uint64 m_submitCounter = 0;
|
||||
Uint64 m_completedSubmitCounter = 0;
|
||||
// Drains since the last Present, gating the drain's frame-boundary-equivalent
|
||||
// work (arena rewind + cache aging): a presenting app's mid-frame
|
||||
// readbacks/waits must neither churn the transient caches nor accelerate the
|
||||
// aging clocks, while present-less loops still cross a boundary every few
|
||||
// iterations. Reset in Present.
|
||||
Uint32 m_drainsSinceLastPresent = 0;
|
||||
|
||||
NativeWindowType m_window = 0;
|
||||
void* m_platformDisplay = nullptr;
|
||||
@@ -355,12 +393,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void* m_platformCloseDisplay = nullptr;
|
||||
VulkanRendererConfig m_config;
|
||||
Bool m_swapchainResizeRequested = false;
|
||||
// Presentation is suspended while the window is zero-area (minimized): the
|
||||
// swapchain is unusable/out of date, so Present drops frames instead of
|
||||
// submitting on a signaled fence / presenting never-acquired images.
|
||||
Bool m_presentSuspended = false;
|
||||
|
||||
// Vulkan objects
|
||||
Bool m_validationLayersEnabled = false;
|
||||
Vector<VkExtensionProperties> m_extensions;
|
||||
VkInstance m_instance = VK_NULL_HANDLE;
|
||||
VkDebugUtilsMessengerEXT m_debugMessenger = VK_NULL_HANDLE;
|
||||
// Fallback reporting channel for drivers that ship the validation layers but
|
||||
// only expose the older VK_EXT_debug_report (Adreno 650 / Vulkan 1.1.128).
|
||||
VkDebugReportCallbackEXT m_debugReportCallback = VK_NULL_HANDLE;
|
||||
PhysicalDevice m_physicalDevice;
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
@@ -511,6 +556,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void CreateInstance();
|
||||
VkResult SetupDebugMessenger();
|
||||
VkResult DestroyDebugMessenger();
|
||||
VkResult SetupDebugReportCallback();
|
||||
void DestroyDebugReportCallback();
|
||||
VkDebugUtilsMessengerCreateInfoEXT PopulateDebugMessengerCreateInfo();
|
||||
void CreateSurface();
|
||||
void PickPhysicalDevice();
|
||||
@@ -529,8 +576,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const RenderPassEntry& renderPassEntry);
|
||||
VkPipeline GetOrCreateComputePipeline(const ProgramFactory::VkProgramObject& programObj);
|
||||
void DestroyComputePipelines();
|
||||
// Takes the frame rather than a command buffer: a first-time storage-usage upgrade has to
|
||||
// flush the pending recording (see the body), which retires the current command buffer.
|
||||
Bool PrepareStorageImageTextures(
|
||||
VkCommandBuffer commandBuffer,
|
||||
FrameContext::FrameData& frame,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
|
||||
@@ -555,6 +604,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLenum filter);
|
||||
Bool MaterializePendingClearForTexture(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
Bool MaterializePendingClearForRenderbuffer(
|
||||
VkCommandBuffer commandBuffer,
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
VkPipeline GetOrCreateBlitPipeline(const RenderPassEntry& renderPassEntry);
|
||||
Bool GenerateDepthMipmapWithShader(FrameContext::FrameData& frame,
|
||||
MG_State::GLState::ITextureObject& texture,
|
||||
@@ -589,6 +641,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const PhysicalDevice& compareWithDevice,
|
||||
PhysicalDevice& outBetterDevice);
|
||||
static constexpr const char* s_validationLayerNames[] = {"VK_LAYER_KHRONOS_validation"};
|
||||
// VK_KHR_image_format_list: lets MUTABLE_FORMAT images declare their exact view-format
|
||||
// set so the driver can keep bandwidth compression (see CreateLogicalDeviceAndQueues).
|
||||
Bool m_imageFormatListExtensionEnabled = false;
|
||||
|
||||
static constexpr const char* s_deviceExtensionNames[] = {VK_KHR_SWAPCHAIN_EXTENSION_NAME};
|
||||
static Bool CheckValidationLayerSupport();
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
GLint Samples = 0;
|
||||
GLint Profile = kCGLOGLPVersion_3_2_Core;
|
||||
GLint RendererId = 0x4d474c;
|
||||
GLint DisplayMask = 0;
|
||||
};
|
||||
|
||||
struct ContextObject {
|
||||
@@ -134,6 +135,9 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
case kCGLPFARendererID:
|
||||
pixelFormat.RendererId = value;
|
||||
break;
|
||||
case kCGLPFADisplayMask:
|
||||
pixelFormat.DisplayMask = value;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -343,6 +347,9 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
case kCGLPFARendererID:
|
||||
*value = pixelFormat->RendererId;
|
||||
return kCGLNoError;
|
||||
case kCGLPFADisplayMask:
|
||||
*value = pixelFormat->DisplayMask;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAOpenGLProfile:
|
||||
*value = pixelFormat->Profile;
|
||||
return kCGLNoError;
|
||||
@@ -481,6 +488,32 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
return it == currentContexts.end() ? nullptr : it->second;
|
||||
}
|
||||
|
||||
CGLError SetVirtualScreen(CGLContextObj ctx, GLint screen) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (screen != 0) {
|
||||
return kCGLBadValue;
|
||||
}
|
||||
object->VirtualScreen = screen;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError GetVirtualScreen(CGLContextObj ctx, GLint* screen) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!screen) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
*screen = object->VirtualScreen;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError SetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
|
||||
@@ -32,6 +32,8 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
|
||||
CGLError SetCurrentContext(CGLContextObj ctx);
|
||||
CGLContextObj GetCurrentContext();
|
||||
CGLError SetVirtualScreen(CGLContextObj ctx, GLint screen);
|
||||
CGLError GetVirtualScreen(CGLContextObj ctx, GLint* screen);
|
||||
CGLError SetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params);
|
||||
CGLError GetParameter(CGLContextObj ctx, CGLContextParameter pname, GLint* params);
|
||||
CGLError UpdateContext(CGLContextObj ctx);
|
||||
|
||||
@@ -71,6 +71,14 @@ MOBILEGL_CGL_API CGLContextObj CGLGetCurrentContext(void) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetCurrentContext();
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetVirtualScreen(CGLContextObj ctx, GLint screen) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetVirtualScreen(ctx, screen);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLGetVirtualScreen(CGLContextObj ctx, GLint* screen) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetVirtualScreen(ctx, screen);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetParameter(ctx, pname, params);
|
||||
}
|
||||
|
||||
@@ -10,8 +10,12 @@
|
||||
|
||||
#if defined(__APPLE__)
|
||||
|
||||
#include "MG_Impl/CGLImpl/CGLImpl.h"
|
||||
#include "MG_Impl/GetProcAddress.h"
|
||||
|
||||
#include <CoreGraphics/CoreGraphics.h>
|
||||
#include <CoreVideo/CVDisplayLink.h>
|
||||
#include <cstdint>
|
||||
#include <dlfcn.h>
|
||||
|
||||
namespace {
|
||||
@@ -47,10 +51,52 @@ namespace {
|
||||
return dlsym(handle, symbol);
|
||||
}
|
||||
|
||||
CGDirectDisplayID DisplayForMask(GLint displayMask) {
|
||||
constexpr std::uint32_t MaxDisplays = sizeof(CGOpenGLDisplayMask) * 8;
|
||||
CGDirectDisplayID displays[MaxDisplays] = {};
|
||||
std::uint32_t displayCount = 0;
|
||||
if (displayMask != 0 &&
|
||||
CGGetActiveDisplayList(MaxDisplays, displays, &displayCount) == kCGErrorSuccess) {
|
||||
const auto mask = static_cast<CGOpenGLDisplayMask>(displayMask);
|
||||
for (std::uint32_t i = 0; i < displayCount; ++i) {
|
||||
if ((CGDisplayIDToOpenGLDisplayMask(displays[i]) & mask) != 0) {
|
||||
return displays[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
return CGMainDisplayID();
|
||||
}
|
||||
|
||||
#pragma clang diagnostic push
|
||||
#pragma clang diagnostic ignored "-Wdeprecated-declarations"
|
||||
CVReturn MobileGLCVDisplayLinkSetCurrentCGDisplayFromOpenGLContext(
|
||||
CVDisplayLinkRef displayLink,
|
||||
CGLContextObj context,
|
||||
CGLPixelFormatObj pixelFormat) {
|
||||
GLint virtualScreen = 0;
|
||||
if (MobileGL::MG_Impl::CGLImpl::GetVirtualScreen(context, &virtualScreen) == kCGLNoError) {
|
||||
GLint displayMask = 0;
|
||||
if (!displayLink ||
|
||||
MobileGL::MG_Impl::CGLImpl::DescribePixelFormat(
|
||||
pixelFormat, virtualScreen, kCGLPFADisplayMask, &displayMask) != kCGLNoError) {
|
||||
return kCVReturnInvalidArgument;
|
||||
}
|
||||
return CVDisplayLinkSetCurrentCGDisplay(displayLink, DisplayForMask(displayMask));
|
||||
}
|
||||
|
||||
using OriginalFunction = CVReturn (*)(CVDisplayLinkRef, CGLContextObj, CGLPixelFormatObj);
|
||||
static const auto original = reinterpret_cast<OriginalFunction>(
|
||||
dlsym(RTLD_NEXT, "CVDisplayLinkSetCurrentCGDisplayFromOpenGLContext"));
|
||||
return original ? original(displayLink, context, pixelFormat) : kCVReturnError;
|
||||
}
|
||||
|
||||
__attribute__((used)) static const DyldInterposeEntry kMobileGLDyldInterpose[]
|
||||
__attribute__((section("__DATA,__interpose"))) = {
|
||||
{reinterpret_cast<const void*>(MobileGLDlsym), reinterpret_cast<const void*>(dlsym)},
|
||||
{reinterpret_cast<const void*>(MobileGLCVDisplayLinkSetCurrentCGDisplayFromOpenGLContext),
|
||||
reinterpret_cast<const void*>(CVDisplayLinkSetCurrentCGDisplayFromOpenGLContext)},
|
||||
};
|
||||
#pragma clang diagnostic pop
|
||||
} // namespace
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
# Public CGL entry points.
|
||||
_CGL*
|
||||
|
||||
# Public EGL entry points.
|
||||
_egl*
|
||||
|
||||
# Public OpenGL and GLX entry points. OpenGL function names always use an
|
||||
# uppercase letter or digit after the "gl" prefix; excluding lowercase here
|
||||
# deliberately prevents glslang_* from matching this pattern.
|
||||
_gl[A-Z0-9]*
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "EGLImpl.h"
|
||||
#include "../GetProcAddress.h"
|
||||
#include <Init.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <mutex>
|
||||
@@ -25,6 +26,17 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return MG_State::pEGLContext.get();
|
||||
}
|
||||
|
||||
// Entry points that can legitimately be an application's FIRST EGL
|
||||
// call (display/proc-address/string queries) lazily bring MobileGL
|
||||
// up here, so the library needs no static constructor and can
|
||||
// re-initialize after the last eglTerminate tore everything down.
|
||||
// Teardown-ish entry points keep using GetState() and fail benignly
|
||||
// when MobileGL is not initialized.
|
||||
EGLStateContext* GetStateEnsureInitialized() {
|
||||
MobileGL::EnsureInitialized();
|
||||
return GetState();
|
||||
}
|
||||
|
||||
MG_Backend::BackendObject* GetBackendObject(EGLStateContext* state) {
|
||||
auto* backendObject = MG_Backend::pActiveBackendObject.get();
|
||||
if (!backendObject && state) {
|
||||
@@ -49,6 +61,8 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return MG_Backend::WindowBackend::Android;
|
||||
#elif defined(__APPLE__)
|
||||
return MG_Backend::WindowBackend::MetalLayer;
|
||||
#elif defined(_WIN32)
|
||||
return MG_Backend::WindowBackend::Win32;
|
||||
#elif defined(__linux__)
|
||||
return MG_Backend::WindowBackend::X11;
|
||||
#else
|
||||
@@ -187,7 +201,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean Initialize(EGLDisplay dpy, EGLint* major, EGLint* minor) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -208,7 +222,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLDisplay GetDisplay(NativeDisplayType display) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
@@ -313,6 +327,14 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
backendObject->ReleaseEGLResources();
|
||||
}
|
||||
// The last initialized display is gone and nothing is current on any
|
||||
// thread: tear the whole library down deterministically inside the
|
||||
// EGL lifecycle (backend, GL/EGL state, glslang). A later EGL call
|
||||
// re-initializes lazily via GetStateEnsureInitialized(); process exit
|
||||
// then has nothing left to destroy.
|
||||
if (!state->HasAnyInitializedDisplay() && !state->HasAnyCurrentContext()) {
|
||||
MobileGL::Destroy();
|
||||
}
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
@@ -345,7 +367,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean BindAPI(EGLenum api) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -378,7 +400,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
char const* QueryString(EGLDisplay display, EGLint name) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return nullptr;
|
||||
}
|
||||
@@ -641,7 +663,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
EGLDisplay GetPlatformDisplay(EGLenum platform, void* native_display, const EGLAttrib* attrib_list) {
|
||||
(void)attrib_list;
|
||||
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
@@ -737,6 +759,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!name) {
|
||||
return nullptr;
|
||||
}
|
||||
MobileGL::EnsureInitialized();
|
||||
|
||||
MGLOG_D("eglGetProcAddress(%s)", name);
|
||||
void* proc = MG_Impl::GetProcAddress(name);
|
||||
|
||||
@@ -424,7 +424,7 @@ DECLARE_GL_FUNCTION_HEAD(void, DrawRangeElementsBaseVertex, GLenum mode, GLuint
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawElementsInstancedBaseVertex, GLenum mode, GLsizei count, GLenum type, const void* indices, GLsizei instancecount, GLint basevertex) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawElementsInstancedBaseVertex, mode, count, type, indices, instancecount, basevertex)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, FramebufferTexture, GLenum target, GLenum attachment, GLuint texture, GLint level) DECLARE_GL_FUNCTION_END_NO_RETURN(void, FramebufferTexture, target, attachment, texture, level)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PrimitiveBoundingBox, GLfloat minX, GLfloat minY, GLfloat minZ, GLfloat minW, GLfloat maxX, GLfloat maxY, GLfloat maxZ, GLfloat maxW) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PrimitiveBoundingBox, minX, minY, minZ, minW, maxX, maxY, maxZ, maxW)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLenum, GetGraphicsResetStatus) DECLARE_GL_FUNCTION_STUB_END(GLenum, GetGraphicsResetStatus)
|
||||
DECLARE_GL_FUNCTION_HEAD(GLenum, GetGraphicsResetStatus) DECLARE_GL_FUNCTION_END(GLenum, GetGraphicsResetStatus)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ReadnPixels, GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, GLsizei bufSize, void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ReadnPixels, x, y, width, height, format, type, bufSize, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformfv, GLuint program, GLint location, GLsizei bufSize, GLfloat* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformfv, program, location, bufSize, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformiv, GLuint program, GLint location, GLsizei bufSize, GLint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformiv, program, location, bufSize, params)
|
||||
|
||||
@@ -2144,6 +2144,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
namespace FramebufferImpl {
|
||||
UniquePtr<DefaultFramebufferInfo> pDefaultFramebufferInfo;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<DefaultFramebufferInfo>& pDefaultFramebufferInfo = *new UniquePtr<DefaultFramebufferInfo>();
|
||||
} // namespace FramebufferImpl
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -78,6 +78,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
SharedPtr<MG_State::GLState::ITextureObject> stencilAttachment;
|
||||
};
|
||||
|
||||
extern UniquePtr<DefaultFramebufferInfo> pDefaultFramebufferInfo;
|
||||
extern UniquePtr<DefaultFramebufferInfo>& pDefaultFramebufferInfo;
|
||||
} // namespace FramebufferImpl
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -1996,4 +1996,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
return MG_Util::ConvertErrorCodeToGLEnum(error->get()->code);
|
||||
}
|
||||
|
||||
GLenum GetGraphicsResetStatus() {
|
||||
// MobileGL does not implement robustness reset notification, so report GL_NO_ERROR
|
||||
// ("no reset detected"). Returning the generic stub's (GLenum)1 makes dEQP read a lost
|
||||
// device after every case (gl3cTestPackages.cpp:121) and, under the default
|
||||
// --deqp-terminate-on-device-lost=enable, tear the whole CTS run down.
|
||||
return GL_NO_ERROR;
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -21,4 +21,5 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data);
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
GLenum GetError();
|
||||
GLenum GetGraphicsResetStatus();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -133,4 +133,33 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
values[0] = value;
|
||||
}
|
||||
}
|
||||
|
||||
void DestroyAllSyncObjects() {
|
||||
// Detach the registry under the lock, release outside it. Entries the app
|
||||
// already deleted were erased by DeleteSync, so nothing here double-frees;
|
||||
// a DeleteSync racing this sweep finds an empty registry and returns. A
|
||||
// thread still blocked inside ClientWaitSync/GetSynciv during teardown
|
||||
// holds a raw SyncObject* these deletes invalidate - the same undefined
|
||||
// race an app-driven DeleteSync already has.
|
||||
UnorderedMap<GLsync, SyncObject*> orphans;
|
||||
{
|
||||
const std::lock_guard<std::mutex> lock(g_syncObjectsMutex);
|
||||
orphans.swap(g_liveSyncObjects);
|
||||
}
|
||||
if (orphans.empty()) {
|
||||
return;
|
||||
}
|
||||
// Both backends' DeleteSync only free the heap wrapper once their GL
|
||||
// context/renderer is gone (generation/current-thread guards), so this is
|
||||
// safe after the backend has released its EGL resources - but not after
|
||||
// the function table itself is cleared.
|
||||
const auto backendDeleteSync = MG_Backend::gBackendFunctionsTable.GL.DeleteSync;
|
||||
for (const auto& [_, syncObject] : orphans) {
|
||||
if (backendDeleteSync && syncObject->backendHandle) {
|
||||
backendDeleteSync(syncObject->backendHandle);
|
||||
}
|
||||
delete syncObject;
|
||||
}
|
||||
MGLOG_D("DestroyAllSyncObjects: reclaimed %zu sync object(s) the app left undeleted", orphans.size());
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -16,4 +16,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void WaitSync(GLsync sync, GLbitfield flags, GLuint64 timeout);
|
||||
void DeleteSync(GLsync sync);
|
||||
void GetSynciv(GLsync sync, GLenum pname, GLsizei bufSize, GLsizei* length, GLint* values);
|
||||
// Destroys every still-registered sync object exactly as DeleteSync would.
|
||||
// GL requires syncs to die with their context; called only from full library
|
||||
// teardown (DestroyImpl), where no context survives on any thread, so the
|
||||
// process-global registry can be drained wholesale. Must run while the
|
||||
// backend function table is still populated: each backend handle has to be
|
||||
// released by the backend that created it, never by a later re-initialized
|
||||
// one.
|
||||
void DestroyAllSyncObjects();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -1724,6 +1724,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
// RGTC is a 2D-only compression scheme, so a 3D target rejects it. This has to be tested on
|
||||
// the raw enum: the RGTC formats resolve to plain R8/RG8/SNORM storage on the way in (see
|
||||
// GLToMG's TextureEnumConverter), so once the internal format is converted there is nothing
|
||||
// left to distinguish them from an ordinary one- or two-channel upload.
|
||||
if ((textureUploadTarget == TextureUploadTarget::Texture3D ||
|
||||
textureUploadTarget == TextureUploadTarget::ProxyTexture3D) &&
|
||||
(internalformat == GL_COMPRESSED_RED_RGTC1 || internalformat == GL_COMPRESSED_SIGNED_RED_RGTC1 ||
|
||||
internalformat == GL_COMPRESSED_RG_RGTC2 || internalformat == GL_COMPRESSED_SIGNED_RG_RGTC2)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"RGTC compressed formats are invalid for 3D texture targets"));
|
||||
return;
|
||||
}
|
||||
|
||||
// TODO: GL_INVALID_OPERATION is generated if a non-zero buffer object name is bound to the
|
||||
// GL_PIXEL_UNPACK_BUFFER target and the buffer object's data store is currently mapped.
|
||||
// GL_INVALID_OPERATION is generated if a non-zero buffer object name is bound to the GL_PIXEL_UNPACK_BUFFER
|
||||
|
||||
@@ -14,7 +14,8 @@
|
||||
#include <MG_State/GLState/TextureState/TextureObjectStubs.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
UniquePtr<ProxyTextureManager> pProxyTextureManager;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<ProxyTextureManager>& pProxyTextureManager = *new UniquePtr<ProxyTextureManager>();
|
||||
|
||||
Bool IsProxyTextureTarget(TextureUploadTarget target) {
|
||||
switch (target) {
|
||||
|
||||
@@ -23,5 +23,5 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
UnorderedMap<TextureUploadTarget, SharedPtr<MG_State::GLState::ITextureObject>> m_proxyTexturesMap;
|
||||
};
|
||||
|
||||
extern UniquePtr<ProxyTextureManager> pProxyTextureManager;
|
||||
extern UniquePtr<ProxyTextureManager>& pProxyTextureManager;
|
||||
} // namespace MobileGL::MG_Impl::GLImpl::TextureImpl
|
||||
|
||||
@@ -85,6 +85,8 @@ namespace MobileGL::MG_Impl {
|
||||
GETPROC(CGLGetPixelFormat, name);
|
||||
GETPROC(CGLSetCurrentContext, name);
|
||||
GETPROC(CGLGetCurrentContext, name);
|
||||
GETPROC(CGLSetVirtualScreen, name);
|
||||
GETPROC(CGLGetVirtualScreen, name);
|
||||
GETPROC(CGLSetParameter, name);
|
||||
GETPROC(CGLGetParameter, name);
|
||||
GETPROC(CGLUpdateContext, name);
|
||||
|
||||
@@ -29,10 +29,19 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
char kContextViewKey;
|
||||
char kContextLayerKey;
|
||||
|
||||
std::once_flag g_installOnce;
|
||||
IMP g_pixelFormatDealloc = nullptr;
|
||||
IMP g_contextDealloc = nullptr;
|
||||
|
||||
std::mutex& HookInstallMutex() {
|
||||
static auto* mutex = new std::mutex();
|
||||
return *mutex;
|
||||
}
|
||||
|
||||
Bool& HooksInstalled() {
|
||||
static auto* installed = new Bool(false);
|
||||
return *installed;
|
||||
}
|
||||
|
||||
template <typename Fn>
|
||||
Fn ObjcMsgSend() {
|
||||
return reinterpret_cast<Fn>(objc_msgSend);
|
||||
@@ -431,12 +440,12 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
method_setImplementation(method, replacement);
|
||||
}
|
||||
|
||||
void InstallHooksOnce() {
|
||||
Bool InstallHooksOnce() {
|
||||
Class pixelFormatClass = objc_getClass("NSOpenGLPixelFormat");
|
||||
Class contextClass = objc_getClass("NSOpenGLContext");
|
||||
if (!pixelFormatClass || !contextClass) {
|
||||
MGLOG_W("NSOpenGLImpl: NSOpenGL classes are not loaded; hooks not installed");
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
|
||||
ReplaceInstanceMethod(pixelFormatClass, "initWithAttributes:",
|
||||
@@ -471,11 +480,34 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
ReplaceInstanceMethod(contextClass, "dealloc", reinterpret_cast<IMP>(ContextDealloc), &g_contextDealloc);
|
||||
|
||||
MGLOG_I("NSOpenGLImpl hooks installed");
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void InstallHooks() {
|
||||
std::call_once(g_installOnce, InstallHooksOnce);
|
||||
const std::lock_guard<std::mutex> lock(HookInstallMutex());
|
||||
if (!HooksInstalled()) {
|
||||
// Do not permanently consume the install attempt when the OpenGL
|
||||
// framework has not registered its Objective-C classes yet. The
|
||||
// dyld bootstrap normally runs after framework dependencies, but
|
||||
// an explicitly loaded/static-linked MobileGL can arrive earlier.
|
||||
HooksInstalled() = InstallHooksOnce();
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::NSOpenGLImpl
|
||||
|
||||
namespace {
|
||||
// SDL's Cocoa backend creates NSOpenGLPixelFormat/NSOpenGLContext before
|
||||
// its first dlsym("glGetString") or other MobileGL host-API call. Install
|
||||
// only the lightweight Objective-C dispatch hooks while the injected dylib
|
||||
// is loading so those first Cocoa objects are routed through CGLImpl. The
|
||||
// hooked context constructor reaches EGLImpl::GetDisplay(), which performs
|
||||
// the full, thread-safe MobileGL initialization outside this bootstrap.
|
||||
//
|
||||
// There is intentionally no matching destructor: backend teardown remains
|
||||
// owned by the EGL lifecycle and process-exit globals remain leak-at-exit.
|
||||
__attribute__((constructor)) void BootstrapNSOpenGLHooks() {
|
||||
MobileGL::MG_Impl::NSOpenGLImpl::InstallHooks();
|
||||
}
|
||||
} // namespace
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
// MobileGL - MobileGL/MG_Impl/WGLImpl/Exporting/Definitions.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
// wingdi.h declares most wgl* entry points as WINGDIAPI (__declspec(dllimport)),
|
||||
// which would reject our definitions. _GDI32_ is the SDK's "I am the module that
|
||||
// implements these" switch: it turns WINGDIAPI into a plain declaration. It must
|
||||
// be defined before the first windows.h inclusion in this translation unit.
|
||||
#if defined(_WIN32) && !defined(_GDI32_)
|
||||
#define _GDI32_ 1
|
||||
#endif
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include "../WGLImpl.h"
|
||||
|
||||
namespace WGL = MobileGL::MG_Impl::WGLImpl;
|
||||
|
||||
// ---- Pixel-format entry points (gdi32 forwards ChoosePixelFormat/SetPixelFormat/
|
||||
// ---- DescribePixelFormat/GetPixelFormat/SwapBuffers into these exports) ----
|
||||
|
||||
extern "C" int WINAPI wglChoosePixelFormat(HDC hdc, CONST PIXELFORMATDESCRIPTOR* ppfd) {
|
||||
return WGL::ChoosePixelFormat(hdc, ppfd);
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglDescribePixelFormat(HDC hdc, int iPixelFormat, UINT nBytes,
|
||||
LPPIXELFORMATDESCRIPTOR ppfd) {
|
||||
return WGL::DescribePixelFormat(hdc, iPixelFormat, nBytes, ppfd);
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglGetPixelFormat(HDC hdc) {
|
||||
return WGL::GetPixelFormat(hdc);
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglSetPixelFormat(HDC hdc, int iPixelFormat, CONST PIXELFORMATDESCRIPTOR* ppfd) {
|
||||
return WGL::SetPixelFormat(hdc, iPixelFormat, ppfd);
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglSwapBuffers(HDC hdc) {
|
||||
return WGL::SwapBuffers(hdc);
|
||||
}
|
||||
|
||||
// ---- Context management ----
|
||||
|
||||
extern "C" HGLRC WINAPI wglCreateContext(HDC hdc) {
|
||||
return WGL::CreateContext(hdc);
|
||||
}
|
||||
|
||||
extern "C" HGLRC WINAPI wglCreateLayerContext(HDC hdc, int iLayerPlane) {
|
||||
return iLayerPlane == 0 ? WGL::CreateContext(hdc) : nullptr;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglCopyContext(HGLRC, HGLRC, UINT) {
|
||||
MGLOG_W("wglCopyContext is not supported");
|
||||
SetLastError(ERROR_NOT_SUPPORTED);
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglDeleteContext(HGLRC hglrc) {
|
||||
return WGL::DeleteContext(hglrc);
|
||||
}
|
||||
|
||||
extern "C" HGLRC WINAPI wglGetCurrentContext(VOID) {
|
||||
return WGL::GetCurrentContext();
|
||||
}
|
||||
|
||||
extern "C" HDC WINAPI wglGetCurrentDC(VOID) {
|
||||
return WGL::GetCurrentDC();
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglMakeCurrent(HDC hdc, HGLRC hglrc) {
|
||||
return WGL::MakeCurrent(hdc, hglrc);
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglShareLists(HGLRC hglrcShare, HGLRC hglrcDest) {
|
||||
return WGL::ShareLists(hglrcShare, hglrcDest);
|
||||
}
|
||||
|
||||
// ---- Proc address ----
|
||||
|
||||
extern "C" PROC WINAPI wglGetProcAddress(LPCSTR lpszProc) {
|
||||
return WGL::GetProcAddress(lpszProc);
|
||||
}
|
||||
|
||||
extern "C" PROC WINAPI wglGetDefaultProcAddress(LPCSTR lpszProc) {
|
||||
return WGL::GetProcAddress(lpszProc);
|
||||
}
|
||||
|
||||
// ---- Layer planes and palettes (unsupported; overlay planes do not exist here) ----
|
||||
|
||||
extern "C" BOOL WINAPI wglDescribeLayerPlane(HDC, int, int, UINT, LPLAYERPLANEDESCRIPTOR) {
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglSetLayerPaletteEntries(HDC, int, int, int, CONST COLORREF*) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglGetLayerPaletteEntries(HDC, int, int, int, COLORREF*) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglRealizeLayerPalette(HDC, int, BOOL) {
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglSwapLayerBuffers(HDC hdc, UINT fuPlanes) {
|
||||
if (fuPlanes & WGL_SWAP_MAIN_PLANE) {
|
||||
return WGL::SwapBuffers(hdc);
|
||||
}
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" DWORD WINAPI wglSwapMultipleBuffers(UINT n, CONST WGLSWAP* ps) {
|
||||
if (!ps) {
|
||||
return 0;
|
||||
}
|
||||
DWORD swapped = 0;
|
||||
for (UINT i = 0; i < n; ++i) {
|
||||
if (WGL::SwapBuffers(ps[i].hdc)) {
|
||||
++swapped;
|
||||
}
|
||||
}
|
||||
return swapped;
|
||||
}
|
||||
|
||||
// ---- Font rendering (legacy immediate-mode feature; not supported) ----
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontBitmapsA(HDC, DWORD, DWORD, DWORD) {
|
||||
MGLOG_W("wglUseFontBitmapsA is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontBitmapsW(HDC, DWORD, DWORD, DWORD) {
|
||||
MGLOG_W("wglUseFontBitmapsW is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontOutlinesA(HDC, DWORD, DWORD, DWORD, FLOAT, FLOAT, int,
|
||||
LPGLYPHMETRICSFLOAT) {
|
||||
MGLOG_W("wglUseFontOutlinesA is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontOutlinesW(HDC, DWORD, DWORD, DWORD, FLOAT, FLOAT, int,
|
||||
LPGLYPHMETRICSFLOAT) {
|
||||
MGLOG_W("wglUseFontOutlinesW is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
#endif // _WIN32
|
||||
@@ -0,0 +1,30 @@
|
||||
; MobileGL WGL exports. The wgl* entry points are defined without
|
||||
; __declspec(dllexport) because wingdi.h pre-declares them (with _GDI32_ they
|
||||
; become plain declarations, and MSVC rejects adding dllexport afterwards),
|
||||
; so this .def file is what actually exports them from the DLL.
|
||||
EXPORTS
|
||||
wglChoosePixelFormat
|
||||
wglCopyContext
|
||||
wglCreateContext
|
||||
wglCreateLayerContext
|
||||
wglDeleteContext
|
||||
wglDescribeLayerPlane
|
||||
wglDescribePixelFormat
|
||||
wglGetCurrentContext
|
||||
wglGetCurrentDC
|
||||
wglGetDefaultProcAddress
|
||||
wglGetLayerPaletteEntries
|
||||
wglGetPixelFormat
|
||||
wglGetProcAddress
|
||||
wglMakeCurrent
|
||||
wglRealizeLayerPalette
|
||||
wglSetLayerPaletteEntries
|
||||
wglSetPixelFormat
|
||||
wglShareLists
|
||||
wglSwapBuffers
|
||||
wglSwapLayerBuffers
|
||||
wglSwapMultipleBuffers
|
||||
wglUseFontBitmapsA
|
||||
wglUseFontBitmapsW
|
||||
wglUseFontOutlinesA
|
||||
wglUseFontOutlinesW
|
||||
@@ -0,0 +1,732 @@
|
||||
// MobileGL - MobileGL/MG_Impl/WGLImpl/WGLImpl.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "WGLImpl.h"
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include "../EGLImpl/EGLImpl.h"
|
||||
#include "../GetProcAddress.h"
|
||||
#include <Init.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::WGLImpl {
|
||||
namespace {
|
||||
// WGL_ARB_pixel_format
|
||||
constexpr int WGL_NUMBER_PIXEL_FORMATS_ARB = 0x2000;
|
||||
constexpr int WGL_DRAW_TO_WINDOW_ARB = 0x2001;
|
||||
constexpr int WGL_DRAW_TO_BITMAP_ARB = 0x2002;
|
||||
constexpr int WGL_ACCELERATION_ARB = 0x2003;
|
||||
constexpr int WGL_NEED_PALETTE_ARB = 0x2004;
|
||||
constexpr int WGL_NEED_SYSTEM_PALETTE_ARB = 0x2005;
|
||||
constexpr int WGL_SWAP_LAYER_BUFFERS_ARB = 0x2006;
|
||||
constexpr int WGL_SWAP_METHOD_ARB = 0x2007;
|
||||
constexpr int WGL_NUMBER_OVERLAYS_ARB = 0x2008;
|
||||
constexpr int WGL_NUMBER_UNDERLAYS_ARB = 0x2009;
|
||||
constexpr int WGL_TRANSPARENT_ARB = 0x200A;
|
||||
constexpr int WGL_SHARE_DEPTH_ARB = 0x200C;
|
||||
constexpr int WGL_SHARE_STENCIL_ARB = 0x200D;
|
||||
constexpr int WGL_SHARE_ACCUM_ARB = 0x200E;
|
||||
constexpr int WGL_SUPPORT_GDI_ARB = 0x200F;
|
||||
constexpr int WGL_SUPPORT_OPENGL_ARB = 0x2010;
|
||||
constexpr int WGL_DOUBLE_BUFFER_ARB = 0x2011;
|
||||
constexpr int WGL_STEREO_ARB = 0x2012;
|
||||
constexpr int WGL_PIXEL_TYPE_ARB = 0x2013;
|
||||
constexpr int WGL_COLOR_BITS_ARB = 0x2014;
|
||||
constexpr int WGL_RED_BITS_ARB = 0x2015;
|
||||
constexpr int WGL_RED_SHIFT_ARB = 0x2016;
|
||||
constexpr int WGL_GREEN_BITS_ARB = 0x2017;
|
||||
constexpr int WGL_GREEN_SHIFT_ARB = 0x2018;
|
||||
constexpr int WGL_BLUE_BITS_ARB = 0x2019;
|
||||
constexpr int WGL_BLUE_SHIFT_ARB = 0x201A;
|
||||
constexpr int WGL_ALPHA_BITS_ARB = 0x201B;
|
||||
constexpr int WGL_ALPHA_SHIFT_ARB = 0x201C;
|
||||
constexpr int WGL_ACCUM_BITS_ARB = 0x201D;
|
||||
constexpr int WGL_ACCUM_RED_BITS_ARB = 0x201E;
|
||||
constexpr int WGL_ACCUM_GREEN_BITS_ARB = 0x201F;
|
||||
constexpr int WGL_ACCUM_BLUE_BITS_ARB = 0x2020;
|
||||
constexpr int WGL_ACCUM_ALPHA_BITS_ARB = 0x2021;
|
||||
constexpr int WGL_DEPTH_BITS_ARB = 0x2022;
|
||||
constexpr int WGL_STENCIL_BITS_ARB = 0x2023;
|
||||
constexpr int WGL_AUX_BUFFERS_ARB = 0x2024;
|
||||
constexpr int WGL_NO_ACCELERATION_ARB = 0x2025;
|
||||
constexpr int WGL_FULL_ACCELERATION_ARB = 0x2027;
|
||||
constexpr int WGL_SWAP_EXCHANGE_ARB = 0x2028;
|
||||
constexpr int WGL_TYPE_RGBA_ARB = 0x202B;
|
||||
// WGL_ARB_multisample
|
||||
constexpr int WGL_SAMPLE_BUFFERS_ARB = 0x2041;
|
||||
constexpr int WGL_SAMPLES_ARB = 0x2042;
|
||||
// WGL_ARB_create_context / _profile / _no_error
|
||||
constexpr int WGL_CONTEXT_MAJOR_VERSION_ARB = 0x2091;
|
||||
constexpr int WGL_CONTEXT_MINOR_VERSION_ARB = 0x2092;
|
||||
constexpr int WGL_CONTEXT_LAYER_PLANE_ARB = 0x2093;
|
||||
constexpr int WGL_CONTEXT_FLAGS_ARB = 0x2094;
|
||||
constexpr int WGL_CONTEXT_PROFILE_MASK_ARB = 0x9126;
|
||||
constexpr int WGL_CONTEXT_DEBUG_BIT_ARB = 0x0001;
|
||||
constexpr int WGL_CONTEXT_FORWARD_COMPATIBLE_BIT_ARB = 0x0002;
|
||||
constexpr int WGL_CONTEXT_CORE_PROFILE_BIT_ARB = 0x00000001;
|
||||
constexpr int WGL_CONTEXT_COMPATIBILITY_PROFILE_BIT_ARB = 0x00000002;
|
||||
constexpr int WGL_CONTEXT_OPENGL_NO_ERROR_ARB = 0x31B3;
|
||||
constexpr DWORD ERROR_INVALID_VERSION_ARB = 0x2095;
|
||||
constexpr DWORD ERROR_INVALID_PROFILE_ARB = 0x2096;
|
||||
|
||||
struct PixelFormatInfo {
|
||||
GLint AlphaBits;
|
||||
GLint DepthBits;
|
||||
GLint StencilBits;
|
||||
};
|
||||
|
||||
// Mirrors the two EGLState configs (RGBA8 + depth24, stencil 8 / stencil 0).
|
||||
constexpr PixelFormatInfo kPixelFormats[] = {
|
||||
{8, 24, 8},
|
||||
{8, 24, 0},
|
||||
};
|
||||
constexpr int kPixelFormatCount = static_cast<int>(std::size(kPixelFormats));
|
||||
|
||||
struct ContextObject {
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLConfig Config = nullptr;
|
||||
EGLContext Context = EGL_NO_CONTEXT;
|
||||
};
|
||||
|
||||
struct WindowSurface {
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLSurface Surface = EGL_NO_SURFACE;
|
||||
Uint32 Width = 0;
|
||||
Uint32 Height = 0;
|
||||
};
|
||||
|
||||
std::recursive_mutex& RegistryMutex() {
|
||||
static auto* mutex = new std::recursive_mutex();
|
||||
return *mutex;
|
||||
}
|
||||
|
||||
UnorderedMap<HGLRC, ContextObject>& Contexts() {
|
||||
static auto* contexts = new UnorderedMap<HGLRC, ContextObject>();
|
||||
return *contexts;
|
||||
}
|
||||
|
||||
UnorderedMap<HWND, WindowSurface>& WindowSurfaces() {
|
||||
static auto* surfaces = new UnorderedMap<HWND, WindowSurface>();
|
||||
return *surfaces;
|
||||
}
|
||||
|
||||
UnorderedMap<HWND, int>& WindowPixelFormats() {
|
||||
static auto* formats = new UnorderedMap<HWND, int>();
|
||||
return *formats;
|
||||
}
|
||||
|
||||
Uint64& NextContextHandle() {
|
||||
static auto* handle = new Uint64(0x10000);
|
||||
return *handle;
|
||||
}
|
||||
|
||||
struct ThreadCurrent {
|
||||
HDC DC = nullptr;
|
||||
HGLRC Context = nullptr;
|
||||
};
|
||||
thread_local ThreadCurrent t_current;
|
||||
|
||||
Int& SwapIntervalShadow() {
|
||||
static auto* interval = new Int(1);
|
||||
return *interval;
|
||||
}
|
||||
|
||||
void EnsureInitialized() {
|
||||
// Initialize() loads backend libraries and glslang, which must not run
|
||||
// under the loader lock; first WGL call is the earliest safe moment.
|
||||
// MobileGL::EnsureInitialized (not a local once_flag) so a fresh init
|
||||
// can follow a full teardown from the last eglTerminate.
|
||||
MobileGL::EnsureInitialized();
|
||||
}
|
||||
|
||||
EGLDisplay EnsureDisplay() {
|
||||
EnsureInitialized();
|
||||
EGLDisplay display = EGLImpl::GetDisplay(EGL_DEFAULT_DISPLAY);
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
if (!EGLImpl::Initialize(display, nullptr, nullptr)) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
return display;
|
||||
}
|
||||
|
||||
HGLRC EncodeContext(Uint64 handle) {
|
||||
return reinterpret_cast<HGLRC>(static_cast<SizeT>(handle));
|
||||
}
|
||||
|
||||
ContextObject* TryGetContext(HGLRC hglrc) {
|
||||
auto& contexts = Contexts();
|
||||
auto it = contexts.find(hglrc);
|
||||
return it == contexts.end() ? nullptr : &it->second;
|
||||
}
|
||||
|
||||
const PixelFormatInfo& PixelFormatForWindow(HWND hwnd) {
|
||||
auto& formats = WindowPixelFormats();
|
||||
auto it = formats.find(hwnd);
|
||||
int index = it == formats.end() ? 1 : it->second;
|
||||
if (index < 1 || index > kPixelFormatCount) {
|
||||
index = 1;
|
||||
}
|
||||
return kPixelFormats[index - 1];
|
||||
}
|
||||
|
||||
Bool QueryClientSize(HWND hwnd, Uint32& width, Uint32& height) {
|
||||
RECT rect{};
|
||||
if (!GetClientRect(hwnd, &rect)) {
|
||||
return false;
|
||||
}
|
||||
width = static_cast<Uint32>(std::max<LONG>(rect.right - rect.left, 1));
|
||||
height = static_cast<Uint32>(std::max<LONG>(rect.bottom - rect.top, 1));
|
||||
return true;
|
||||
}
|
||||
|
||||
// The backends never query the HWND client size themselves; the WGL layer
|
||||
// owns size discovery and pushes changes through the internal resize hook
|
||||
// (same contract as the macOS CGL layer).
|
||||
void SyncSurfaceSize(HWND hwnd, WindowSurface& surface) {
|
||||
Uint32 width = 0;
|
||||
Uint32 height = 0;
|
||||
if (!QueryClientSize(hwnd, width, height)) {
|
||||
return;
|
||||
}
|
||||
if (width == surface.Width && height == surface.Height) {
|
||||
return;
|
||||
}
|
||||
if (EGLImpl::ResizePlatformWindowSurface(surface.Display, surface.Surface,
|
||||
static_cast<EGLint>(width), static_cast<EGLint>(height))) {
|
||||
surface.Width = width;
|
||||
surface.Height = height;
|
||||
}
|
||||
}
|
||||
|
||||
WindowSurface* EnsureWindowSurface(HWND hwnd, const ContextObject& context) {
|
||||
auto& surfaces = WindowSurfaces();
|
||||
auto it = surfaces.find(hwnd);
|
||||
if (it != surfaces.end()) {
|
||||
SyncSurfaceSize(hwnd, it->second);
|
||||
return &it->second;
|
||||
}
|
||||
|
||||
Uint32 width = 0;
|
||||
Uint32 height = 0;
|
||||
if (!QueryClientSize(hwnd, width, height)) {
|
||||
MGLOG_E("wgl: GetClientRect failed for HWND %p", hwnd);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const EGLAttrib attribs[] = {
|
||||
EGL_WIDTH, static_cast<EGLAttrib>(width),
|
||||
EGL_HEIGHT, static_cast<EGLAttrib>(height),
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLSurface surface =
|
||||
EGLImpl::CreatePlatformWindowSurface(context.Display, context.Config, hwnd, attribs);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E("wgl: failed to create window surface for HWND %p (%ux%u)", hwnd, width, height);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
WindowSurface record;
|
||||
record.Display = context.Display;
|
||||
record.Surface = surface;
|
||||
record.Width = width;
|
||||
record.Height = height;
|
||||
auto [inserted, _] = surfaces.emplace(hwnd, record);
|
||||
return &inserted->second;
|
||||
}
|
||||
|
||||
HGLRC CreateContextFromEGLAttribs(HDC hdc, HGLRC share, const EGLint* contextAttribs) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
EGLDisplay display = EnsureDisplay();
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
MGLOG_E("wgl: no EGL display");
|
||||
return nullptr;
|
||||
}
|
||||
EGLImpl::BindAPI(EGL_OPENGL_API);
|
||||
|
||||
EGLContext shareContext = EGL_NO_CONTEXT;
|
||||
if (share) {
|
||||
auto* shareObject = TryGetContext(share);
|
||||
if (!shareObject) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return nullptr;
|
||||
}
|
||||
shareContext = shareObject->Context;
|
||||
}
|
||||
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
const PixelFormatInfo& pixelFormat = PixelFormatForWindow(hwnd);
|
||||
const EGLint configAttribs[] = {
|
||||
EGL_RED_SIZE, 8,
|
||||
EGL_GREEN_SIZE, 8,
|
||||
EGL_BLUE_SIZE, 8,
|
||||
EGL_ALPHA_SIZE, pixelFormat.AlphaBits,
|
||||
EGL_DEPTH_SIZE, pixelFormat.DepthBits,
|
||||
EGL_STENCIL_SIZE, pixelFormat.StencilBits,
|
||||
EGL_SURFACE_TYPE, EGL_WINDOW_BIT | EGL_PBUFFER_BIT,
|
||||
EGL_RENDERABLE_TYPE, EGL_OPENGL_BIT,
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLConfig config = nullptr;
|
||||
EGLint configCount = 0;
|
||||
if (!EGLImpl::ChooseConfig(display, configAttribs, &config, 1, &configCount) || configCount <= 0) {
|
||||
MGLOG_E("wgl: eglChooseConfig failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
EGLContext eglContext = EGLImpl::CreateContext(display, config, shareContext, contextAttribs);
|
||||
if (eglContext == EGL_NO_CONTEXT) {
|
||||
MGLOG_E("wgl: eglCreateContext failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ContextObject object;
|
||||
object.Display = display;
|
||||
object.Config = config;
|
||||
object.Context = eglContext;
|
||||
const auto handle = EncodeContext(NextContextHandle()++);
|
||||
Contexts()[handle] = object;
|
||||
MGLOG_I("wgl: created context %p (EGL context %p)", handle, eglContext);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// ---- WGL extension entry points (resolved via wglGetProcAddress only) ----
|
||||
|
||||
const char* WINAPI Ext_GetExtensionsStringARB(HDC) {
|
||||
return "WGL_ARB_create_context WGL_ARB_create_context_no_error WGL_ARB_create_context_profile "
|
||||
"WGL_ARB_extensions_string WGL_ARB_pixel_format WGL_EXT_extensions_string WGL_EXT_swap_control";
|
||||
}
|
||||
|
||||
const char* WINAPI Ext_GetExtensionsStringEXT() {
|
||||
return Ext_GetExtensionsStringARB(nullptr);
|
||||
}
|
||||
|
||||
HGLRC WINAPI Ext_CreateContextAttribsARB(HDC hdc, HGLRC hShareContext, const int* attribList) {
|
||||
EnsureInitialized();
|
||||
int major = 1;
|
||||
int minor = 0;
|
||||
int profileMask = 0;
|
||||
int flags = 0;
|
||||
if (attribList) {
|
||||
for (SizeT i = 0; attribList[i] != 0; i += 2) {
|
||||
const int attrib = attribList[i];
|
||||
const int value = attribList[i + 1];
|
||||
switch (attrib) {
|
||||
case WGL_CONTEXT_MAJOR_VERSION_ARB:
|
||||
major = value;
|
||||
break;
|
||||
case WGL_CONTEXT_MINOR_VERSION_ARB:
|
||||
minor = value;
|
||||
break;
|
||||
case WGL_CONTEXT_PROFILE_MASK_ARB:
|
||||
profileMask = value;
|
||||
break;
|
||||
case WGL_CONTEXT_FLAGS_ARB:
|
||||
flags = value;
|
||||
break;
|
||||
case WGL_CONTEXT_LAYER_PLANE_ARB:
|
||||
if (value != 0) {
|
||||
SetLastError(ERROR_INVALID_PARAMETER);
|
||||
return nullptr;
|
||||
}
|
||||
break;
|
||||
case WGL_CONTEXT_OPENGL_NO_ERROR_ARB:
|
||||
// Accepted and ignored: MobileGL always validates.
|
||||
break;
|
||||
default:
|
||||
MGLOG_D("wglCreateContextAttribsARB: ignoring attrib 0x%04x = 0x%x", attrib, value);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (major < 1 || (profileMask & ~(WGL_CONTEXT_CORE_PROFILE_BIT_ARB |
|
||||
WGL_CONTEXT_COMPATIBILITY_PROFILE_BIT_ARB))) {
|
||||
SetLastError(profileMask ? ERROR_INVALID_PROFILE_ARB : ERROR_INVALID_VERSION_ARB);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
Vector<EGLint> attribs = {
|
||||
EGL_CONTEXT_MAJOR_VERSION, major,
|
||||
EGL_CONTEXT_MINOR_VERSION, minor,
|
||||
};
|
||||
const Bool wantsCompat = (profileMask & WGL_CONTEXT_COMPATIBILITY_PROFILE_BIT_ARB) != 0;
|
||||
if (major > 3 || (major == 3 && minor >= 2) || profileMask != 0) {
|
||||
attribs.push_back(EGL_CONTEXT_OPENGL_PROFILE_MASK);
|
||||
attribs.push_back(wantsCompat ? EGL_CONTEXT_OPENGL_COMPATIBILITY_PROFILE_BIT
|
||||
: EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT);
|
||||
}
|
||||
if (flags & WGL_CONTEXT_FORWARD_COMPATIBLE_BIT_ARB) {
|
||||
attribs.push_back(EGL_CONTEXT_OPENGL_FORWARD_COMPATIBLE);
|
||||
attribs.push_back(EGL_TRUE);
|
||||
}
|
||||
if (flags & WGL_CONTEXT_DEBUG_BIT_ARB) {
|
||||
attribs.push_back(EGL_CONTEXT_OPENGL_DEBUG);
|
||||
attribs.push_back(EGL_TRUE);
|
||||
}
|
||||
attribs.push_back(EGL_NONE);
|
||||
|
||||
return CreateContextFromEGLAttribs(hdc, hShareContext, attribs.data());
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_SwapIntervalEXT(int interval) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
EGLDisplay display = EnsureDisplay();
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
return FALSE;
|
||||
}
|
||||
if (interval < 0) {
|
||||
// Adaptive vsync is not supported; clamp to regular vsync.
|
||||
interval = 1;
|
||||
}
|
||||
EGLImpl::SwapInterval(display, interval);
|
||||
SwapIntervalShadow() = interval;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
int WINAPI Ext_GetSwapIntervalEXT() {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
return SwapIntervalShadow();
|
||||
}
|
||||
|
||||
int PixelFormatAttribValue(int format, int attrib) {
|
||||
const PixelFormatInfo& info = kPixelFormats[format - 1];
|
||||
switch (attrib) {
|
||||
case WGL_NUMBER_PIXEL_FORMATS_ARB:
|
||||
return kPixelFormatCount;
|
||||
case WGL_SUPPORT_OPENGL_ARB:
|
||||
case WGL_DRAW_TO_WINDOW_ARB:
|
||||
case WGL_DOUBLE_BUFFER_ARB:
|
||||
return 1;
|
||||
case WGL_ACCELERATION_ARB:
|
||||
return WGL_FULL_ACCELERATION_ARB;
|
||||
case WGL_PIXEL_TYPE_ARB:
|
||||
return WGL_TYPE_RGBA_ARB;
|
||||
case WGL_COLOR_BITS_ARB:
|
||||
return 32;
|
||||
case WGL_RED_BITS_ARB:
|
||||
case WGL_GREEN_BITS_ARB:
|
||||
case WGL_BLUE_BITS_ARB:
|
||||
return 8;
|
||||
case WGL_RED_SHIFT_ARB:
|
||||
return 16;
|
||||
case WGL_GREEN_SHIFT_ARB:
|
||||
return 8;
|
||||
case WGL_BLUE_SHIFT_ARB:
|
||||
return 0;
|
||||
case WGL_ALPHA_BITS_ARB:
|
||||
return info.AlphaBits;
|
||||
case WGL_ALPHA_SHIFT_ARB:
|
||||
return 24;
|
||||
case WGL_DEPTH_BITS_ARB:
|
||||
return info.DepthBits;
|
||||
case WGL_STENCIL_BITS_ARB:
|
||||
return info.StencilBits;
|
||||
case WGL_SWAP_METHOD_ARB:
|
||||
return WGL_SWAP_EXCHANGE_ARB;
|
||||
case WGL_DRAW_TO_BITMAP_ARB:
|
||||
case WGL_NEED_PALETTE_ARB:
|
||||
case WGL_NEED_SYSTEM_PALETTE_ARB:
|
||||
case WGL_SWAP_LAYER_BUFFERS_ARB:
|
||||
case WGL_NUMBER_OVERLAYS_ARB:
|
||||
case WGL_NUMBER_UNDERLAYS_ARB:
|
||||
case WGL_TRANSPARENT_ARB:
|
||||
case WGL_SHARE_DEPTH_ARB:
|
||||
case WGL_SHARE_STENCIL_ARB:
|
||||
case WGL_SHARE_ACCUM_ARB:
|
||||
case WGL_SUPPORT_GDI_ARB:
|
||||
case WGL_STEREO_ARB:
|
||||
case WGL_ACCUM_BITS_ARB:
|
||||
case WGL_ACCUM_RED_BITS_ARB:
|
||||
case WGL_ACCUM_GREEN_BITS_ARB:
|
||||
case WGL_ACCUM_BLUE_BITS_ARB:
|
||||
case WGL_ACCUM_ALPHA_BITS_ARB:
|
||||
case WGL_AUX_BUFFERS_ARB:
|
||||
case WGL_SAMPLE_BUFFERS_ARB:
|
||||
case WGL_SAMPLES_ARB:
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_GetPixelFormatAttribivARB(HDC, int iPixelFormat, int iLayerPlane, UINT nAttributes,
|
||||
const int* piAttributes, int* piValues) {
|
||||
if (iLayerPlane != 0 || !piAttributes || !piValues) {
|
||||
return FALSE;
|
||||
}
|
||||
// Format 0 is only valid for WGL_NUMBER_PIXEL_FORMATS_ARB queries.
|
||||
if (iPixelFormat < 0 || iPixelFormat > kPixelFormatCount) {
|
||||
return FALSE;
|
||||
}
|
||||
const int format = iPixelFormat == 0 ? 1 : iPixelFormat;
|
||||
for (UINT i = 0; i < nAttributes; ++i) {
|
||||
piValues[i] = PixelFormatAttribValue(format, piAttributes[i]);
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_GetPixelFormatAttribfvARB(HDC hdc, int iPixelFormat, int iLayerPlane, UINT nAttributes,
|
||||
const int* piAttributes, FLOAT* pfValues) {
|
||||
if (!pfValues) {
|
||||
return FALSE;
|
||||
}
|
||||
Vector<int> values(nAttributes);
|
||||
if (!Ext_GetPixelFormatAttribivARB(hdc, iPixelFormat, iLayerPlane, nAttributes, piAttributes,
|
||||
values.data())) {
|
||||
return FALSE;
|
||||
}
|
||||
for (UINT i = 0; i < nAttributes; ++i) {
|
||||
pfValues[i] = static_cast<FLOAT>(values[i]);
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_ChoosePixelFormatARB(HDC, const int* piAttribIList, const FLOAT*, UINT nMaxFormats,
|
||||
int* piFormats, UINT* nNumFormats) {
|
||||
if (!piFormats || !nNumFormats) {
|
||||
return FALSE;
|
||||
}
|
||||
int wantedStencil = 0;
|
||||
if (piAttribIList) {
|
||||
for (SizeT i = 0; piAttribIList[i] != 0; i += 2) {
|
||||
if (piAttribIList[i] == WGL_STENCIL_BITS_ARB) {
|
||||
wantedStencil = piAttribIList[i + 1];
|
||||
}
|
||||
}
|
||||
}
|
||||
UINT count = 0;
|
||||
const int preferred = wantedStencil > 0 ? 1 : 2;
|
||||
const int fallback = wantedStencil > 0 ? 2 : 1;
|
||||
if (count < nMaxFormats) {
|
||||
piFormats[count++] = preferred;
|
||||
}
|
||||
if (count < nMaxFormats) {
|
||||
piFormats[count++] = fallback;
|
||||
}
|
||||
*nNumFormats = count;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
struct WGLExtensionProc {
|
||||
const char* Name;
|
||||
PROC Proc;
|
||||
};
|
||||
|
||||
const WGLExtensionProc kWGLExtensionProcs[] = {
|
||||
{"wglGetExtensionsStringARB", reinterpret_cast<PROC>(Ext_GetExtensionsStringARB)},
|
||||
{"wglGetExtensionsStringEXT", reinterpret_cast<PROC>(Ext_GetExtensionsStringEXT)},
|
||||
{"wglCreateContextAttribsARB", reinterpret_cast<PROC>(Ext_CreateContextAttribsARB)},
|
||||
{"wglSwapIntervalEXT", reinterpret_cast<PROC>(Ext_SwapIntervalEXT)},
|
||||
{"wglGetSwapIntervalEXT", reinterpret_cast<PROC>(Ext_GetSwapIntervalEXT)},
|
||||
{"wglGetPixelFormatAttribivARB", reinterpret_cast<PROC>(Ext_GetPixelFormatAttribivARB)},
|
||||
{"wglGetPixelFormatAttribfvARB", reinterpret_cast<PROC>(Ext_GetPixelFormatAttribfvARB)},
|
||||
{"wglChoosePixelFormatARB", reinterpret_cast<PROC>(Ext_ChoosePixelFormatARB)},
|
||||
};
|
||||
} // namespace
|
||||
|
||||
int ChoosePixelFormat(HDC hdc, const PIXELFORMATDESCRIPTOR* pfd) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglChoosePixelFormat(hdc=%p)", hdc);
|
||||
// Format 1 (RGBA8 + depth24/stencil8) satisfies every request; a format
|
||||
// exceeding the asked-for capabilities is a legal ChoosePixelFormat answer.
|
||||
(void)pfd;
|
||||
return 1;
|
||||
}
|
||||
|
||||
int DescribePixelFormat(HDC hdc, int format, UINT size, PIXELFORMATDESCRIPTOR* pfd) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglDescribePixelFormat(hdc=%p, format=%d)", hdc, format);
|
||||
if (!pfd) {
|
||||
return kPixelFormatCount;
|
||||
}
|
||||
if (size < sizeof(PIXELFORMATDESCRIPTOR) || format < 1 || format > kPixelFormatCount) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const PixelFormatInfo& info = kPixelFormats[format - 1];
|
||||
std::memset(pfd, 0, sizeof(PIXELFORMATDESCRIPTOR));
|
||||
pfd->nSize = sizeof(PIXELFORMATDESCRIPTOR);
|
||||
pfd->nVersion = 1;
|
||||
pfd->dwFlags = PFD_DRAW_TO_WINDOW | PFD_SUPPORT_OPENGL | PFD_DOUBLEBUFFER | PFD_SWAP_EXCHANGE
|
||||
#if defined(PFD_SUPPORT_COMPOSITION)
|
||||
| PFD_SUPPORT_COMPOSITION
|
||||
#endif
|
||||
;
|
||||
pfd->iPixelType = PFD_TYPE_RGBA;
|
||||
pfd->cColorBits = 32;
|
||||
pfd->cRedBits = 8;
|
||||
pfd->cRedShift = 16;
|
||||
pfd->cGreenBits = 8;
|
||||
pfd->cGreenShift = 8;
|
||||
pfd->cBlueBits = 8;
|
||||
pfd->cBlueShift = 0;
|
||||
pfd->cAlphaBits = static_cast<BYTE>(info.AlphaBits);
|
||||
pfd->cAlphaShift = 24;
|
||||
pfd->cDepthBits = static_cast<BYTE>(info.DepthBits);
|
||||
pfd->cStencilBits = static_cast<BYTE>(info.StencilBits);
|
||||
pfd->iLayerType = PFD_MAIN_PLANE;
|
||||
return kPixelFormatCount;
|
||||
}
|
||||
|
||||
int GetPixelFormat(HDC hdc) {
|
||||
EnsureInitialized();
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
return 0;
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto& formats = WindowPixelFormats();
|
||||
auto it = formats.find(hwnd);
|
||||
return it == formats.end() ? 0 : it->second;
|
||||
}
|
||||
|
||||
BOOL SetPixelFormat(HDC hdc, int format, const PIXELFORMATDESCRIPTOR*) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglSetPixelFormat(hdc=%p, format=%d)", hdc, format);
|
||||
if (format < 1 || format > kPixelFormatCount) {
|
||||
SetLastError(ERROR_INVALID_PARAMETER);
|
||||
return FALSE;
|
||||
}
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
WindowPixelFormats()[hwnd] = format;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL SwapBuffers(HDC hdc) {
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto& surfaces = WindowSurfaces();
|
||||
auto it = surfaces.find(hwnd);
|
||||
if (it == surfaces.end()) {
|
||||
MGLOG_W("wglSwapBuffers: no surface for HWND %p", hwnd);
|
||||
return FALSE;
|
||||
}
|
||||
SyncSurfaceSize(hwnd, it->second);
|
||||
return EGLImpl::SwapBuffers(it->second.Display, it->second.Surface) == EGL_TRUE ? TRUE : FALSE;
|
||||
}
|
||||
|
||||
HGLRC CreateContext(HDC hdc) {
|
||||
EnsureInitialized();
|
||||
MGLOG_I("wglCreateContext(hdc=%p)", hdc);
|
||||
// A legacy WGL context is a compatibility-profile context; MobileGL keys
|
||||
// its relaxed-semantics mode off the explicit compatibility bit.
|
||||
const EGLint attribs[] = {
|
||||
EGL_CONTEXT_MAJOR_VERSION, 3,
|
||||
EGL_CONTEXT_MINOR_VERSION, 3,
|
||||
EGL_CONTEXT_OPENGL_PROFILE_MASK, EGL_CONTEXT_OPENGL_COMPATIBILITY_PROFILE_BIT,
|
||||
EGL_NONE,
|
||||
};
|
||||
return CreateContextFromEGLAttribs(hdc, nullptr, attribs);
|
||||
}
|
||||
|
||||
BOOL DeleteContext(HGLRC hglrc) {
|
||||
EnsureInitialized();
|
||||
MGLOG_I("wglDeleteContext(%p)", hglrc);
|
||||
if (t_current.Context == hglrc) {
|
||||
MakeCurrent(nullptr, nullptr);
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(hglrc);
|
||||
if (!object) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
if (object->Context != EGL_NO_CONTEXT) {
|
||||
EGLImpl::DestroyContext(object->Display, object->Context);
|
||||
}
|
||||
Contexts().erase(hglrc);
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL MakeCurrent(HDC hdc, HGLRC hglrc) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglMakeCurrent(hdc=%p, hglrc=%p)", hdc, hglrc);
|
||||
if (!hglrc) {
|
||||
if (!t_current.Context) {
|
||||
t_current = {};
|
||||
return TRUE;
|
||||
}
|
||||
const EGLBoolean released =
|
||||
EGLImpl::MakeCurrent(EGL_NO_DISPLAY, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT);
|
||||
t_current = {};
|
||||
return released == EGL_TRUE ? TRUE : FALSE;
|
||||
}
|
||||
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(hglrc);
|
||||
if (!object) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
WindowSurface* surface = EnsureWindowSurface(hwnd, *object);
|
||||
if (!surface) {
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
if (!EGLImpl::MakeCurrent(object->Display, surface->Surface, surface->Surface, object->Context)) {
|
||||
MGLOG_E("wglMakeCurrent: eglMakeCurrent failed (hdc=%p, hglrc=%p)", hdc, hglrc);
|
||||
return FALSE;
|
||||
}
|
||||
t_current = {hdc, hglrc};
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
HGLRC GetCurrentContext() {
|
||||
return t_current.Context;
|
||||
}
|
||||
|
||||
HDC GetCurrentDC() {
|
||||
return t_current.DC;
|
||||
}
|
||||
|
||||
BOOL ShareLists(HGLRC hglrcShare, HGLRC hglrcDest) {
|
||||
// All MobileGL contexts alias one global GL object namespace, so every
|
||||
// pair of contexts already shares.
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
if (!TryGetContext(hglrcShare) || !TryGetContext(hglrcDest)) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
PROC GetProcAddress(const char* name) {
|
||||
EnsureInitialized();
|
||||
if (!name) {
|
||||
return nullptr;
|
||||
}
|
||||
if (name[0] == 'w' && name[1] == 'g' && name[2] == 'l') {
|
||||
for (const auto& entry : kWGLExtensionProcs) {
|
||||
if (std::strcmp(entry.Name, name) == 0) {
|
||||
return entry.Proc;
|
||||
}
|
||||
}
|
||||
MGLOG_D("wglGetProcAddress: unknown wgl entry point %s", name);
|
||||
return nullptr;
|
||||
}
|
||||
return reinterpret_cast<PROC>(MG_Impl::GetProcAddress(name));
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::WGLImpl
|
||||
|
||||
#endif // _WIN32
|
||||
@@ -0,0 +1,34 @@
|
||||
// MobileGL - MobileGL/MG_Impl/WGLImpl/WGLImpl.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
#if defined(_WIN32)
|
||||
|
||||
namespace MobileGL::MG_Impl::WGLImpl {
|
||||
// Classic opengl32.dll surface. gdi32's ChoosePixelFormat/SetPixelFormat/
|
||||
// DescribePixelFormat/GetPixelFormat/SwapBuffers forward into the loaded
|
||||
// opengl32.dll's wgl* exports, so these back both call paths.
|
||||
int ChoosePixelFormat(HDC hdc, const PIXELFORMATDESCRIPTOR* pfd);
|
||||
int DescribePixelFormat(HDC hdc, int format, UINT size, PIXELFORMATDESCRIPTOR* pfd);
|
||||
int GetPixelFormat(HDC hdc);
|
||||
BOOL SetPixelFormat(HDC hdc, int format, const PIXELFORMATDESCRIPTOR* pfd);
|
||||
BOOL SwapBuffers(HDC hdc);
|
||||
|
||||
HGLRC CreateContext(HDC hdc);
|
||||
BOOL DeleteContext(HGLRC hglrc);
|
||||
BOOL MakeCurrent(HDC hdc, HGLRC hglrc);
|
||||
HGLRC GetCurrentContext();
|
||||
HDC GetCurrentDC();
|
||||
BOOL ShareLists(HGLRC hglrcShare, HGLRC hglrcDest);
|
||||
|
||||
PROC GetProcAddress(const char* name);
|
||||
} // namespace MobileGL::MG_Impl::WGLImpl
|
||||
|
||||
#endif // _WIN32
|
||||
@@ -368,6 +368,26 @@ namespace MobileGL {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool EGLContext::HasAnyInitializedDisplay() const {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_mutex);
|
||||
for (const auto& [handle, displayObject] : m_displays) {
|
||||
if (displayObject.Initialized) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool EGLContext::HasAnyCurrentContext() const {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_mutex);
|
||||
for (const auto& [threadId, current] : m_threadCurrents) {
|
||||
if (current.Context != nullptr) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool EGLContext::ChooseConfig(EGLDisplayHandle display, const EGLint* attribList, EGLConfigHandle* configs,
|
||||
EGLint configSize, EGLint* numConfig) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_mutex);
|
||||
@@ -1415,6 +1435,7 @@ namespace MobileGL {
|
||||
}
|
||||
} // namespace EGLState
|
||||
|
||||
UniquePtr<EGLState::EGLContext> pEGLContext;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<EGLState::EGLContext>& pEGLContext = *new UniquePtr<EGLState::EGLContext>();
|
||||
} // namespace MG_State
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -39,6 +39,10 @@ namespace MobileGL {
|
||||
Bool IsDisplayInitialized(EGLDisplayHandle display) const;
|
||||
Bool InitializeDisplay(EGLDisplayHandle display, EGLint* major, EGLint* minor);
|
||||
Bool TerminateDisplay(EGLDisplayHandle display);
|
||||
// Whole-library idle checks used by EGLImpl::Terminate to decide
|
||||
// when the last eglTerminate may tear MobileGL down entirely.
|
||||
Bool HasAnyInitializedDisplay() const;
|
||||
Bool HasAnyCurrentContext() const;
|
||||
|
||||
// Config
|
||||
Bool ChooseConfig(EGLDisplayHandle display, const EGLint* attribList, EGLConfigHandle* configs,
|
||||
@@ -262,6 +266,6 @@ namespace MobileGL {
|
||||
};
|
||||
} // namespace EGLState
|
||||
|
||||
extern UniquePtr<EGLState::EGLContext> pEGLContext;
|
||||
extern UniquePtr<EGLState::EGLContext>& pEGLContext;
|
||||
} // namespace MG_State
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -721,5 +721,6 @@ namespace MobileGL::MG_State {
|
||||
}
|
||||
} // namespace GLState
|
||||
|
||||
UniquePtr<GLState::GLContext> pGLContext;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<GLState::GLContext>& pGLContext = *new UniquePtr<GLState::GLContext>();
|
||||
} // namespace MobileGL::MG_State
|
||||
|
||||
@@ -252,7 +252,7 @@ namespace MobileGL {
|
||||
};
|
||||
} // namespace GLState
|
||||
|
||||
extern UniquePtr<GLState::GLContext> pGLContext;
|
||||
extern UniquePtr<GLState::GLContext>& pGLContext;
|
||||
|
||||
// True when relaxed GL semantics apply. Strict core rules are enforced only when the
|
||||
// current EGL context explicitly requested a core profile (core bit in
|
||||
|
||||
@@ -334,16 +334,32 @@ namespace MobileGL::MG_State::GLState {
|
||||
// draw. The memo is keyed by (backendStateVersion, flags); ResetLinkArtifacts and
|
||||
// the binding setters below invalidate it by bumping m_backendStateVersion.
|
||||
Bool GetBackendHashMemo(Uint flags, Uint64& outHash) const {
|
||||
if (m_backendHashMemoVersion != m_backendStateVersion || m_backendHashMemoFlags != flags) {
|
||||
return false;
|
||||
if (m_backendHashMemoVersion != m_backendStateVersion) return false;
|
||||
for (const auto& slot : m_backendHashMemoSlots) {
|
||||
if (slot.valid && slot.flags == flags) {
|
||||
outHash = slot.hash;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
outHash = m_backendHashMemo;
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
void SetBackendHashMemo(Uint flags, Uint64 hash) const {
|
||||
m_backendHashMemo = hash;
|
||||
m_backendHashMemoVersion = m_backendStateVersion;
|
||||
m_backendHashMemoFlags = flags;
|
||||
if (m_backendHashMemoVersion != m_backendStateVersion) {
|
||||
for (auto& slot : m_backendHashMemoSlots) slot.valid = false;
|
||||
m_backendHashMemoVersion = m_backendStateVersion;
|
||||
m_backendHashMemoNextSlot = 0;
|
||||
}
|
||||
for (auto& slot : m_backendHashMemoSlots) {
|
||||
if (slot.valid && slot.flags == flags) {
|
||||
slot.hash = hash;
|
||||
return;
|
||||
}
|
||||
}
|
||||
auto& slot = m_backendHashMemoSlots[m_backendHashMemoNextSlot];
|
||||
slot.flags = flags;
|
||||
slot.hash = hash;
|
||||
slot.valid = true;
|
||||
m_backendHashMemoNextSlot = (m_backendHashMemoNextSlot + 1) % kBackendHashMemoSlotCount;
|
||||
}
|
||||
|
||||
void SetUniformSamplerOrImageUnitIndex(Uint location, Int unit) {
|
||||
@@ -527,10 +543,19 @@ namespace MobileGL::MG_State::GLState {
|
||||
Uint32 m_backendStateVersion = 0;
|
||||
|
||||
// Backend-owned content-hash memo (see GetBackendHashMemo): valid only while
|
||||
// m_backendStateVersion and the compile flags match the recorded values.
|
||||
mutable Uint64 m_backendHashMemo = 0;
|
||||
// m_backendStateVersion matches. Several slots, not one: a backend may resolve the same
|
||||
// program under more than one compile-flag set within a frame (surface rotation, and the
|
||||
// explicit-LOD sampling variant), and a single slot would then miss on every lookup and
|
||||
// re-hash the program's whole SPIR-V once per draw.
|
||||
static constexpr SizeT kBackendHashMemoSlotCount = 4;
|
||||
struct BackendHashMemoSlot {
|
||||
Uint64 hash = 0;
|
||||
Uint flags = 0;
|
||||
Bool valid = false;
|
||||
};
|
||||
mutable Array<BackendHashMemoSlot, kBackendHashMemoSlotCount> m_backendHashMemoSlots{};
|
||||
mutable SizeT m_backendHashMemoNextSlot = 0;
|
||||
mutable Uint32 m_backendHashMemoVersion = ~0u;
|
||||
mutable Uint m_backendHashMemoFlags = 0;
|
||||
Uint32 m_uboContentVersion = 0;
|
||||
Uint32 m_linkVersion = 0;
|
||||
};
|
||||
|
||||
@@ -15,7 +15,11 @@
|
||||
#include <MG_Util/Math/VectorTypes.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject {
|
||||
// Texture objects are always SharedPtr-owned (TextureState creates every instance via
|
||||
// MakeShared, including the per-target default objects). enable_shared_from_this lets
|
||||
// backends that only receive a reference (e.g. syncing a name-deleted texture kept
|
||||
// alive by an FBO attachment) still register a weak liveness reference for GC.
|
||||
class ITextureObject : public std::enable_shared_from_this<ITextureObject> {
|
||||
public:
|
||||
using TargetEnum = TextureTarget;
|
||||
virtual ~ITextureObject() = default;
|
||||
|
||||
@@ -258,11 +258,14 @@ TEST(PipelineQuirkStripDecision, MinBlendIsStripped) {
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, AdditiveOnePlusOneIsStripped) {
|
||||
// MC 26.3 OIT transmittance/accumulate: ONE+ONE additive accumulation writing depth.
|
||||
TEST(PipelineQuirkStripDecision, AdditiveOnePlusOneIsNotStripped) {
|
||||
// ONE+ONE additive with a depth write matched zero draws of the 26.3 chain in the
|
||||
// fixture sweep (transmittance/accumulate disable depth writes themselves); the only
|
||||
// real content with this shape was harmless additive glow effects (Create). A quirk
|
||||
// touches as little unrelated content as possible, so the shape stays exempt.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, kFullColorWriteMask));
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, SortedTransparencyOverBlendIsNotStripped) {
|
||||
@@ -285,9 +288,10 @@ TEST(PipelineQuirkStripDecision, EffectivelyOpaqueBlendIsNotStripped) {
|
||||
|
||||
TEST(PipelineQuirkStripDecision, FullyMaskedAccumulationBlendIsNotStripped) {
|
||||
// Depth-prepass pattern: colorMask(0,0,0,0) with blending left enabled - blending is
|
||||
// moot, and stripping would delete the entire prepass.
|
||||
// moot, and stripping would delete the entire prepass. MAX so the exemption, not the
|
||||
// blend-op filter, is what keeps the depth write.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, 0));
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, 0));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
@@ -314,8 +318,8 @@ TEST(PipelineQuirkStripDecision, FragDepthWriterIsExempt) {
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, AccumulationOnSecondaryAttachmentIsStripped) {
|
||||
// The hazard is not limited to attachment 0: the 26.3 transmittance pass accumulates
|
||||
// into a 2-target MRT.
|
||||
// The scan is not limited to attachment 0: an extremum accumulation on any live
|
||||
// attachment marks the pipeline.
|
||||
PipelineFactory::PipelineCreatePayload payload{};
|
||||
payload.colorAttachmentCount = 2;
|
||||
payload.depthTestEnable = true;
|
||||
@@ -323,21 +327,20 @@ TEST(PipelineQuirkStripDecision, AccumulationOnSecondaryAttachmentIsStripped) {
|
||||
payload.colorBlendAttachments[0] = MakeBlendAttachment(
|
||||
false, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ZERO, VK_BLEND_OP_ADD, kFullColorWriteMask);
|
||||
payload.colorBlendAttachments[1] = MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, kFullColorWriteMask);
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask);
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, AlphaWeightedAdditiveIsNotStripped) {
|
||||
// SRC_ALPHA,ONE additive is order-independent in the color channel but is the classic
|
||||
// *sorted* particle/glow blend, not an OIT accumulation pass. Pins the src==ONE clause:
|
||||
// without it this state would be stripped.
|
||||
// SRC_ALPHA,ONE additive: the classic *sorted* particle/glow blend. Kept exempt like
|
||||
// every other ADD-op shape now that the strip is extremum-only.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_SRC_ALPHA, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, ReverseSubtractIsNotStripped) {
|
||||
// Deliberate narrowing: only MIN/MAX and ONE+ONE ADD carry the equality-chain
|
||||
// Deliberate narrowing: only the MIN/MAX extremum ops carry the depth-bounds
|
||||
// signature. SUBTRACT-class ops stay outside the quirk until content demands them.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_REVERSE_SUBTRACT,
|
||||
@@ -348,7 +351,7 @@ TEST(PipelineQuirkStripDecision, ReverseSubtractIsNotStripped) {
|
||||
TEST(PipelineQuirkStripDecision, PartiallyMaskedAccumulationIsStripped) {
|
||||
// Only a fully masked attachment is exempt; a live alpha channel still accumulates.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, VK_COLOR_COMPONENT_A_BIT));
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, VK_COLOR_COMPONENT_A_BIT));
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
|
||||
@@ -238,6 +238,77 @@ void main() {
|
||||
}
|
||||
}
|
||||
|
||||
// KHR-GL33.shaders.preprocessor.* — a block comment is one preprocessing token that the C/GLSL
|
||||
// preprocessor replaces with a single space, even when it spans newlines inside a directive. glslang
|
||||
// handles this natively, so MobileGL must not mangle it. These reproduce the CTS cases that failed
|
||||
// because comment blanking preserved the interior newline, truncating multi-line #define bodies.
|
||||
static void ExpectCompiles(MobileGL::ShaderStage stage, GLenum glStage, MobileGL::String source) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
PreprocessShaderSource(stage, source);
|
||||
ShaderAttrib attrib{.shaderType = glStage, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessMultilineCommentInDefineBodyCompiles) {
|
||||
ExpectCompiles(ShaderStage::Fragment, GL_FRAGMENT_SHADER,
|
||||
R"(#version 330
|
||||
precision mediump float;
|
||||
out float out0;
|
||||
#define VALUE /* current
|
||||
value */ 4.2
|
||||
|
||||
void main()
|
||||
{
|
||||
out0 = VALUE;
|
||||
})");
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessRedefineObjectMultilineCommentCompiles) {
|
||||
ExpectCompiles(ShaderStage::Fragment, GL_FRAGMENT_SHADER,
|
||||
R"(#version 330
|
||||
precision mediump float;
|
||||
out float out0;
|
||||
# define VAL1 1.0
|
||||
#define VAL2 2.0
|
||||
|
||||
#define RES2 /* fdsjklfdsjkl
|
||||
dsfjkhfdsjkh
|
||||
fdsjklhfdsjkh */ (RES1 * VAL2)
|
||||
#define RES1 (VAL2 / VAL1)
|
||||
#define RES2 /* ewrlkjhsadf */ (RES1 * VAL2)
|
||||
#define VALUE (RES2 + RES1)
|
||||
|
||||
void main()
|
||||
{
|
||||
out0 = VALUE;
|
||||
})");
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessFunctionMacroRedefinitionMultilineCommentCompiles) {
|
||||
ExpectCompiles(ShaderStage::Fragment, GL_FRAGMENT_SHADER,
|
||||
R"(#version 330
|
||||
precision mediump float;
|
||||
out float out0;
|
||||
# define FUNC(a,b) (a +b)
|
||||
# define FUNC(a,b)(a /* comment
|
||||
*/ +b)
|
||||
|
||||
void main()
|
||||
{
|
||||
out0 = FUNC(1.0, 2.0);
|
||||
})");
|
||||
}
|
||||
|
||||
// Note: KHR-GL3x.shaders.preprocessor.conditional_inclusion.basic_2 (`#define AAA defined(BBB)` used
|
||||
// in `#if !AAA`) is intentionally NOT handled here. Generating the `defined` operator via macro
|
||||
// expansion is undefined per the C/GLSL preprocessor spec, and glslang deliberately rejects it
|
||||
// ("'defined' : cannot use in preprocessor expression when expanded from macros"). Making it pass
|
||||
// would require MobileGL to run its own macro expansion ahead of glslang, which is exactly the
|
||||
// preprocessing we defer to glslang; the two cases stay failing by design.
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessLegacyFragmentShaderModernizesGlmarkStyleSource) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -426,6 +497,62 @@ void main() {
|
||||
verifyVersion("#version 460 core");
|
||||
}
|
||||
|
||||
// KHR-GL33.shaders.preprocessor.directive.version_* (also re-run verbatim under GL40-GL44): the
|
||||
// compiler must REJECT a malformed #version line. MobileGL used to rewrite the whole line to
|
||||
// "#version 330 core" whenever it could scrape a leading integer - or treat an unknown profile token
|
||||
// as core - which silently legalized every form below. CTS compiles the shader's own #version
|
||||
// verbatim, so the rejection has to survive preprocessing (and the 460 retry).
|
||||
TEST_F(ProgramUtilTest, PreprocessRejectsMalformedVersionDirectives) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const char* body = "\nout vec4 fragColor;\nvoid main() { fragColor = vec4(1.0); }\n";
|
||||
const auto rejects = [](const String& fullSource) {
|
||||
String src = fullSource;
|
||||
PreprocessShaderSource(ShaderStage::Fragment, src);
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = src};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
return res ? false : true; // "rejects" == compile failed
|
||||
};
|
||||
|
||||
// Silently legalized today - the five this fix must flip to rejection:
|
||||
EXPECT_TRUE(rejects(String("#version 329") + body)) << "329 is not a real version";
|
||||
EXPECT_TRUE(rejects(String("#version 331") + body)) << "331 is not a real version";
|
||||
EXPECT_TRUE(rejects(String("#version 330 foo") + body)) << "unknown profile keyword";
|
||||
EXPECT_TRUE(rejects(String("#version 330.0") + body)) << "float literal, not an int token";
|
||||
EXPECT_TRUE(rejects(String("#version 330 foobar") + body)) << "trailing tokens after a valid decl";
|
||||
|
||||
// Already rejected (no leading integer, or #version is not the first token) - pinned so a future
|
||||
// change to the normalizer cannot start legalizing them either:
|
||||
EXPECT_TRUE(rejects(String("#version") + body)) << "missing version number";
|
||||
EXPECT_TRUE(rejects(String("#version foobar") + body)) << "identifier where the int belongs";
|
||||
EXPECT_TRUE(rejects(String("#version AAA") + body)) << "identifier where the int belongs";
|
||||
EXPECT_TRUE(rejects(String("precision mediump float;\n#version 330") + body))
|
||||
<< "#version must be the first statement";
|
||||
EXPECT_TRUE(rejects(String("#define FOO BAR\n#version 330") + body))
|
||||
<< "#version must precede a #define";
|
||||
}
|
||||
|
||||
// The PASS half of the same CTS group: a valid decl, and #version preceded only by whitespace or a
|
||||
// comment, must still compile. Guards the fix above from over-rejecting.
|
||||
TEST_F(ProgramUtilTest, PreprocessKeepsValidVersionDirectivesCompiling) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const char* body = "\nout vec4 fragColor;\nvoid main() { fragColor = vec4(1.0); }\n";
|
||||
const auto compiles = [](const String& fullSource) {
|
||||
String src = fullSource;
|
||||
PreprocessShaderSource(ShaderStage::Fragment, src);
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = src};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
return res ? true : false;
|
||||
};
|
||||
|
||||
EXPECT_TRUE(compiles(String("#version 330 core") + body));
|
||||
EXPECT_TRUE(compiles(String("\n#version 330 core") + body))
|
||||
<< "leading whitespace is legal before #version";
|
||||
EXPECT_TRUE(compiles(String("// test\n#version 330 core") + body))
|
||||
<< "a leading comment is legal before #version";
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessUsesRealSpacedVersionDirectiveForInjectedOutput) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -446,6 +573,9 @@ void main() {
|
||||
EXPECT_NE(versionPos, String::npos);
|
||||
EXPECT_EQ(outputPos, versionPos + std::strlen("#version 330 core\n"));
|
||||
EXPECT_NE(source.find("// #version 460 core"), String::npos);
|
||||
// This #line sits ahead of the version directive, where GLSL would never have honoured it, so
|
||||
// it is still dropped. Directives that follow the version line are kept - see
|
||||
// PreprocessKeepsPlainLineDirectivesAndSparesLookalikeIdentifiers.
|
||||
EXPECT_EQ(source.find("#line"), String::npos);
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
@@ -455,6 +585,105 @@ void main() {
|
||||
}
|
||||
}
|
||||
|
||||
// A banner line like "//*** NOTE ***" contains "/*" at offset 1 and no "*/" anywhere after it. The
|
||||
// old hand-rolled comment stripper searched for "/*" with no lexical state, found that, failed to
|
||||
// find a terminator, and erased everything from there to the end of the file - deleting the entire
|
||||
// shader. Banner comments in that exact shape are common in Iris and OptiFine packs.
|
||||
TEST_F(ProgramUtilTest, PreprocessKeepsShaderBodyAfterAStarredLineComment) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
//*** lighting pass ***
|
||||
out vec4 fragColor;
|
||||
void main() {
|
||||
fragColor = vec4(1.0);
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_NE(source.find("void main()"), String::npos) << "shader body was truncated:\n" << source;
|
||||
EXPECT_NE(source.find("fragColor = vec4(1.0);"), String::npos);
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
// The builtin-shadowing rename only fires when the shader really defines its own round/tanh/etc.
|
||||
// Deciding that from a commented-out definition renames every genuine call to the builtin to a
|
||||
// mg_ name that nothing defines, which fails to link.
|
||||
TEST_F(ProgramUtilTest, PreprocessIgnoresCommentedOutBuiltinShadowingDefinition) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
// float round(float x) { return floor(x + 0.5); }
|
||||
out vec4 fragColor;
|
||||
void main() {
|
||||
fragColor = vec4(round(1.25));
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_NE(source.find("round(1.25)"), String::npos) << "call was renamed from a comment:\n" << source;
|
||||
EXPECT_EQ(source.find("mg_round"), String::npos);
|
||||
}
|
||||
|
||||
// A block-commented extension directive must not be treated as a real one - the int64 filter turns
|
||||
// unsupported directives into #error, so reading one out of a comment manufactures a compile
|
||||
// failure for a shader that never asked for the extension.
|
||||
TEST_F(ProgramUtilTest, PreprocessIgnoresBlockCommentedExtensionDirectives) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
/*
|
||||
#extension GL_ARB_gpu_shader_int64 : require
|
||||
*/
|
||||
out vec4 fragColor;
|
||||
void main() {
|
||||
fragColor = vec4(1.0);
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_EQ(source.find("#error"), String::npos) << "#error synthesized from a comment:\n" << source;
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
// KHR-GL33.shaders.preprocessor.builtin.line_* checks that __LINE__ follows #line. That only works
|
||||
// if the directive reaches glslang, so a plain integer form must pass through untouched - while
|
||||
// "#linear" and friends must not be mistaken for it.
|
||||
TEST_F(ProgramUtilTest, PreprocessKeepsPlainLineDirectivesAndSparesLookalikeIdentifiers) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
out vec4 fragColor;
|
||||
#line 42
|
||||
float linear(float x) { return x; }
|
||||
void main() {
|
||||
#line 100
|
||||
fragColor = vec4(linear(float(__LINE__)));
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_NE(source.find("#line 42"), String::npos) << source;
|
||||
EXPECT_NE(source.find("#line 100"), String::npos) << source;
|
||||
EXPECT_NE(source.find("float linear(float x)"), String::npos) << "identifier lookalike was eaten:\n" << source;
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessModernSampleQualifierStaysAtVersion460) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -745,6 +974,16 @@ TEST_F(ProgramUtilTest, RetargetLegacyVersionDirectiveOnlyTouchesNormalizedDeskt
|
||||
String commented = "// #version 330 core\nvoid main() {}\n";
|
||||
EXPECT_FALSE(RetargetLegacyVersionDirectiveTo460(commented));
|
||||
EXPECT_EQ(commented.find("#version 460"), String::npos);
|
||||
|
||||
// A malformed directive must NOT be rescued to 460 - that is what silently legalized the CTS
|
||||
// directive.version_* rejection cases. The bad version stays put so glslang keeps rejecting it.
|
||||
String badNumber = "#version 331\nvoid main() {}\n";
|
||||
EXPECT_FALSE(RetargetLegacyVersionDirectiveTo460(badNumber));
|
||||
EXPECT_EQ(badNumber.find("#version 460"), String::npos);
|
||||
|
||||
String badProfile = "#version 330 foo\nvoid main() {}\n";
|
||||
EXPECT_FALSE(RetargetLegacyVersionDirectiveTo460(badProfile));
|
||||
EXPECT_EQ(badProfile.find("#version 460"), String::npos);
|
||||
}
|
||||
|
||||
const char* fs = R"(#version 150
|
||||
@@ -921,6 +1160,353 @@ TEST_F(ProgramUtilTest, CompileFragmentShaderWithDiscard) {
|
||||
}
|
||||
}
|
||||
|
||||
// noperspective is core desktop GLSL (1.30+) and maps to the SPIR-V NoPerspective decoration. It must
|
||||
// reach glslang (not be stripped as text) so the SPIR-V carries the decoration; SPIRV-Cross then emits
|
||||
// ESSL `noperspective` + the GL_NV_shader_noperspective_interpolation extension. Shader packs
|
||||
// (Iris/Complementary) depend on it, and KHR-GL33.glsl_noperspective fails if the result matches
|
||||
// smooth. This is the DirectGLES path with the NV extension available (SPIRV-Cross's default).
|
||||
TEST_F(ProgramUtilTest, NoperspectiveInterpolationSurvivesToEssl) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String fs = R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 fragColor;
|
||||
void main() { fragColor = vColor; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile errc: " << res.error().errc << "\nlog: " << res.error().log;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {res.value()}};
|
||||
auto program_res = ShaderCompiler::LinkProgram(programAttrib);
|
||||
if (!program_res) FAIL() << "link errc: " << program_res.error().errc << "\nlog: " << program_res.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *program_res.value()};
|
||||
auto bin_res = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
if (!bin_res) FAIL() << "spirv errc: " << bin_res.error().errc << "\nlog: " << bin_res.error().log;
|
||||
ASSERT_EQ(bin_res.value().size(), 1u);
|
||||
|
||||
SpvcSession session(bin_res.value()[0], SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile errc: " << essl.error().errc << "\nlog: " << essl.error().log;
|
||||
|
||||
EXPECT_NE(essl.value().find("noperspective"), String::npos)
|
||||
<< "noperspective was lost before it reached SPIR-V:\n" << essl.value();
|
||||
EXPECT_NE(essl.value().find("GL_NV_shader_noperspective_interpolation"), String::npos)
|
||||
<< "SPIRV-Cross must require the NV extension for ES noperspective:\n" << essl.value();
|
||||
}
|
||||
|
||||
// The old handling was a naked substring erase of "noperspective", so any identifier that merely
|
||||
// contained those characters (a uniform named noperspectiveBlend, say) got mangled. Removing the
|
||||
// strip fixes it - glslang, which is identifier-aware, is the only thing that should see the keyword.
|
||||
TEST_F(ProgramUtilTest, PreprocessDoesNotCorruptIdentifiersContainingNoperspective) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
uniform float noperspectiveBlend;
|
||||
out vec4 fragColor;
|
||||
void main() { fragColor = vec4(noperspectiveBlend); }
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
EXPECT_NE(source.find("noperspectiveBlend"), String::npos)
|
||||
<< "identifier was corrupted by substring stripping:\n" << source;
|
||||
}
|
||||
|
||||
// The DirectGLES fallback for devices without GL_NV_shader_noperspective_interpolation: stripping the
|
||||
// NoPerspective decoration makes SPIRV-Cross emit a plain smooth varying with no `#extension … :
|
||||
// require`, so the shader still compiles (rendering as smooth) instead of being rejected by the driver.
|
||||
TEST_F(ProgramUtilTest, StripNoPerspectiveFallbackProducesPlainEssl) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String fs = R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 fragColor;
|
||||
void main() { fragColor = vColor; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile errc: " << res.error().errc << "\nlog: " << res.error().log;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {res.value()}};
|
||||
auto program_res = ShaderCompiler::LinkProgram(programAttrib);
|
||||
if (!program_res) FAIL() << "link errc: " << program_res.error().errc << "\nlog: " << program_res.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *program_res.value()};
|
||||
auto bin_res = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
if (!bin_res) FAIL() << "spirv errc: " << bin_res.error().errc << "\nlog: " << bin_res.error().log;
|
||||
ASSERT_EQ(bin_res.value().size(), 1u);
|
||||
|
||||
// Precondition: with the decoration present the default decompile requires the NV extension.
|
||||
{
|
||||
SpvcSession session(bin_res.value()[0], SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile errc: " << essl.error().errc;
|
||||
ASSERT_NE(essl.value().find("noperspective"), String::npos) << essl.value();
|
||||
}
|
||||
|
||||
// The fallback strips the decoration -> plain smooth ESSL, no extension require.
|
||||
Vector<Uint32> stripped;
|
||||
ASSERT_TRUE(ShaderCompiler::StripNoPerspectiveForEssl(bin_res.value()[0], stripped));
|
||||
ASSERT_FALSE(stripped.empty());
|
||||
|
||||
SpvcSession session(stripped, SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile errc: " << essl.error().errc << "\nlog: " << essl.error().log;
|
||||
EXPECT_EQ(essl.value().find("noperspective"), String::npos)
|
||||
<< "the decoration should be gone:\n" << essl.value();
|
||||
EXPECT_EQ(essl.value().find("GL_NV_shader_noperspective_interpolation"), String::npos)
|
||||
<< "no extension require without the decoration:\n" << essl.value();
|
||||
}
|
||||
|
||||
// Directly exercises BOTH decoration forms StripNoPerspectivePass handles: a plain-variable
|
||||
// OpDecorate NoPerspective (in-operand 1) and an interface-block-member OpMemberDecorate NoPerspective
|
||||
// (in-operand 2). The ESSL round-trip tests above use only a scalar input, so they never reach the
|
||||
// member-decorate branch, which a block varying like `in Block { noperspective vec4 c; }` (common in
|
||||
// shader packs) produces. Unrelated decorations (Flat, Location) must survive untouched.
|
||||
TEST_F(ProgramUtilTest, StripNoPerspectivePassRemovesBothDecorateForms) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const String spirvText = R"(
|
||||
OpCapability Shader
|
||||
OpMemoryModel Logical GLSL450
|
||||
OpEntryPoint Fragment %main "main" %plainVar %blockVar %flatVar
|
||||
OpExecutionMode %main OriginUpperLeft
|
||||
OpName %main "main"
|
||||
OpDecorate %plainVar Location 0
|
||||
OpDecorate %plainVar NoPerspective
|
||||
OpMemberDecorate %Block 0 NoPerspective
|
||||
OpDecorate %blockVar Location 1
|
||||
OpDecorate %flatVar Location 2
|
||||
OpDecorate %flatVar Flat
|
||||
%void = OpTypeVoid
|
||||
%mainFn = OpTypeFunction %void
|
||||
%float = OpTypeFloat 32
|
||||
%v4float = OpTypeVector %float 4
|
||||
%int = OpTypeInt 32 1
|
||||
%inV4Ptr = OpTypePointer Input %v4float
|
||||
%plainVar = OpVariable %inV4Ptr Input
|
||||
%Block = OpTypeStruct %v4float
|
||||
%inBlockPtr = OpTypePointer Input %Block
|
||||
%blockVar = OpVariable %inBlockPtr Input
|
||||
%inIntPtr = OpTypePointer Input %int
|
||||
%flatVar = OpVariable %inIntPtr Input
|
||||
%main = OpFunction %void None %mainFn
|
||||
%mainBody = OpLabel
|
||||
OpReturn
|
||||
OpFunctionEnd
|
||||
)";
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
Vector<uint32_t> inputBinary;
|
||||
ASSERT_TRUE(tools.Assemble(spirvText, &inputBinary));
|
||||
|
||||
const auto countNoPerspective = [](const String& text) {
|
||||
SizeT count = 0, offset = 0;
|
||||
while ((offset = text.find("NoPerspective", offset)) != String::npos) {
|
||||
++count;
|
||||
offset += std::strlen("NoPerspective");
|
||||
}
|
||||
return count;
|
||||
};
|
||||
|
||||
String inputText;
|
||||
ASSERT_TRUE(tools.Disassemble(inputBinary, &inputText));
|
||||
ASSERT_EQ(countNoPerspective(inputText), 2u)
|
||||
<< "fixture must carry both a plain and a member NoPerspective:\n" << inputText;
|
||||
|
||||
Vector<uint32_t> outputBinary;
|
||||
ASSERT_TRUE(ShaderCompiler::StripNoPerspectiveForEssl(inputBinary, outputBinary));
|
||||
ASSERT_FALSE(outputBinary.empty());
|
||||
|
||||
String outputText;
|
||||
ASSERT_TRUE(tools.Disassemble(outputBinary, &outputText));
|
||||
EXPECT_EQ(countNoPerspective(outputText), 0u)
|
||||
<< "both NoPerspective decorations (OpDecorate and OpMemberDecorate) must be stripped:\n" << outputText;
|
||||
EXPECT_NE(outputText.find("Flat"), String::npos)
|
||||
<< "the unrelated Flat decoration must survive:\n" << outputText;
|
||||
EXPECT_NE(outputText.find("Location"), String::npos)
|
||||
<< "Location decorations must survive:\n" << outputText;
|
||||
}
|
||||
|
||||
// Phase 2 emulation - fragment side. On a device without the NV extension the NoPerspective input is
|
||||
// recovered as `load * gl_FragCoord.w` and the decoration removed; gl_FragCoord is synthesized because
|
||||
// the shader did not otherwise use it. The emulated SPIR-V must validate and decompile without the
|
||||
// extension require.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveFragmentRecoversWithFragCoordW) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String fs = R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 f;
|
||||
void main() { f = vColor; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile: " << res.error().log;
|
||||
ProgramAttrib pa{.shaders = {res.value()}};
|
||||
auto pr = ShaderCompiler::LinkProgram(pa);
|
||||
if (!pr) FAIL() << "link: " << pr.error().log;
|
||||
ProgramBinaryAttrib ba{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *pr.value()};
|
||||
auto br = ShaderCompiler::GetSpirvBinaryFromProgram(ba);
|
||||
if (!br) FAIL() << "spirv: " << br.error().log;
|
||||
ASSERT_EQ(br.value().size(), 1u);
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(br.value()[0], emulated));
|
||||
ASSERT_FALSE(emulated.empty());
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << "emulated SPIR-V must be valid:\n" << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << "decoration must be stripped:\n" << dis;
|
||||
EXPECT_NE(dis.find("FragCoord"), String::npos) << "gl_FragCoord must be synthesized:\n" << dis;
|
||||
EXPECT_NE(dis.find("OpVectorTimesScalar"), String::npos) << "the recovery multiply must be present:\n" << dis;
|
||||
|
||||
SpvcSession session(emulated, SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile: " << essl.error().log;
|
||||
EXPECT_EQ(essl.value().find("noperspective"), String::npos) << essl.value();
|
||||
EXPECT_EQ(essl.value().find("GL_NV_shader_noperspective_interpolation"), String::npos) << essl.value();
|
||||
EXPECT_NE(essl.value().find("gl_FragCoord"), String::npos) << "recovery must reference gl_FragCoord:\n" << essl.value();
|
||||
}
|
||||
|
||||
// Phase 2 emulation - vertex side. The NoPerspective output is pre-multiplied by gl_Position.w before
|
||||
// return and the decoration removed. Emulated SPIR-V must validate and decompile without the extension.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveVertexPreMultipliesByPositionW) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String vs = R"(#version 330 core
|
||||
in vec4 pos;
|
||||
noperspective out vec4 vColor;
|
||||
void main() { gl_Position = pos; vColor = pos; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_VERTEX_SHADER, .sourceStr = vs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile: " << res.error().log;
|
||||
ProgramAttrib pa{.shaders = {res.value()}};
|
||||
auto pr = ShaderCompiler::LinkProgram(pa);
|
||||
if (!pr) FAIL() << "link: " << pr.error().log;
|
||||
ProgramBinaryAttrib ba{.shaderTypes = {GL_VERTEX_SHADER}, .program = *pr.value()};
|
||||
auto br = ShaderCompiler::GetSpirvBinaryFromProgram(ba);
|
||||
if (!br) FAIL() << "spirv: " << br.error().log;
|
||||
ASSERT_EQ(br.value().size(), 1u);
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(br.value()[0], emulated));
|
||||
ASSERT_FALSE(emulated.empty());
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << "emulated SPIR-V must be valid:\n" << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << "decoration must be stripped:\n" << dis;
|
||||
EXPECT_NE(dis.find("OpVectorTimesScalar"), String::npos) << "the pre-multiply must be present:\n" << dis;
|
||||
|
||||
SpvcSession session(emulated, SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile: " << essl.error().log;
|
||||
EXPECT_EQ(essl.value().find("noperspective"), String::npos) << essl.value();
|
||||
EXPECT_NE(essl.value().find("gl_Position"), String::npos) << "pre-multiply must reference gl_Position:\n" << essl.value();
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Compiles one shader stage through the full pipeline and returns its SPIR-V, or fails the test.
|
||||
MobileGL::Vector<uint32_t> CompileStageSpirv(GLenum type, const char* src) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
ShaderAttrib attrib{.shaderType = type, .sourceStr = src};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
EXPECT_TRUE(static_cast<bool>(res)) << (res ? "" : res.error().log);
|
||||
if (!res) return {};
|
||||
ProgramAttrib pa{.shaders = {res.value()}};
|
||||
auto pr = ShaderCompiler::LinkProgram(pa);
|
||||
EXPECT_TRUE(static_cast<bool>(pr)) << (pr ? "" : pr.error().log);
|
||||
if (!pr) return {};
|
||||
ProgramBinaryAttrib ba{.shaderTypes = {type}, .program = *pr.value()};
|
||||
auto br = ShaderCompiler::GetSpirvBinaryFromProgram(ba);
|
||||
EXPECT_TRUE(static_cast<bool>(br)) << (br ? "" : br.error().log);
|
||||
if (!br || br.value().empty()) return {};
|
||||
return br.value()[0];
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// Regression: the vertex pre-multiply must be applied exactly once (in main), not once per function.
|
||||
// glslang does not inline, so a helper function survives as its own OpFunction; instrumenting its
|
||||
// return too would scale the varying by gl_Position.w twice (w^2).
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveVertexWithHelperScalesExactlyOnce) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
// helper() returns via OpReturnValue and adds (no vector*scalar), so the ONLY OpVectorTimesScalar
|
||||
// in the module is the emulation's pre-multiply. The old all-functions code injected it at both
|
||||
// helper's and main's return -> count 2; restricted to the entry function it is 1.
|
||||
auto spirv = CompileStageSpirv(GL_VERTEX_SHADER, R"(#version 330 core
|
||||
in vec4 pos;
|
||||
noperspective out vec4 vColor;
|
||||
vec4 helper(vec4 x) { return x + vec4(1.0); }
|
||||
void main() { gl_Position = pos; vColor = helper(pos); }
|
||||
)");
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(spirv, emulated));
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << dis;
|
||||
|
||||
SizeT count = 0, off = 0;
|
||||
while ((off = dis.find("OpVectorTimesScalar", off)) != String::npos) {
|
||||
++count;
|
||||
off += std::strlen("OpVectorTimesScalar");
|
||||
}
|
||||
EXPECT_EQ(count, 1u) << "the gl_Position.w pre-multiply must happen exactly once, not per function:\n" << dis;
|
||||
}
|
||||
|
||||
// Regression: a single-component read (vColor.x), which glslang lowers via OpAccessChain, must still be
|
||||
// recovered with gl_FragCoord.w - not silently left un-scaled.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveFragmentComponentReadIsRecovered) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
auto spirv = CompileStageSpirv(GL_FRAGMENT_SHADER, R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 f;
|
||||
void main() { f = vec4(vColor.x); }
|
||||
)");
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(spirv, emulated));
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << dis;
|
||||
EXPECT_NE(dis.find("FragCoord"), String::npos)
|
||||
<< "the component read must still be recovered via gl_FragCoord.w:\n" << dis;
|
||||
}
|
||||
|
||||
// Coverage: a scalar float varying exercises the OpFMul path; a vector varying the OpVectorTimesScalar
|
||||
// path; multiple noperspective varyings in one stage are all handled.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveHandlesScalarAndMultipleVaryings) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
auto spirv = CompileStageSpirv(GL_FRAGMENT_SHADER, R"(#version 330 core
|
||||
noperspective in float a;
|
||||
noperspective in vec2 b;
|
||||
out vec4 f;
|
||||
void main() { f = vec4(a, b, 1.0); }
|
||||
)");
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(spirv, emulated));
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << dis;
|
||||
EXPECT_NE(dis.find("OpFMul"), String::npos) << "the scalar varying must scale with OpFMul:\n" << dis;
|
||||
EXPECT_NE(dis.find("OpVectorTimesScalar"), String::npos)
|
||||
<< "the vector varying must scale with OpVectorTimesScalar:\n" << dis;
|
||||
}
|
||||
|
||||
const char* vs_location = R"(#version 460
|
||||
|
||||
in vec4 Position;
|
||||
|
||||
@@ -33,6 +33,7 @@
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderSourceProcessor.h>
|
||||
#include <MG_Util/Debug/Log.h>
|
||||
#include <FastSTL/UnorderedMap.h>
|
||||
|
||||
namespace {
|
||||
class DynamicParameterBackend final : public MobileGL::MG_Backend::BackendObject {
|
||||
@@ -1431,3 +1432,366 @@ TEST(RenderStateSanity, PrimitiveRestartIndexStoresAndReadsBack) {
|
||||
|
||||
MG_State::pGLContext.reset();
|
||||
}
|
||||
|
||||
|
||||
// ---- DirectGLES readback driver-state shadows ----------------------------------------------------
|
||||
// Regression coverage for the readback-path state-leak overhaul: the pixel-PBO
|
||||
// binding cache, the framebuffer-binding shadow, the PACK pixel-store shadow and
|
||||
// the scratch-FBO attachment shadow must (a) leave the driver in the documented
|
||||
// resting state, (b) skip redundant GL calls, and (c) scrub correctly on
|
||||
// deletion. All drive the real Managers.cpp implementations against a recording
|
||||
// mock GLES table.
|
||||
namespace {
|
||||
struct StateGuardCallLog {
|
||||
MobileGL::Vector<MobileGL::String> calls;
|
||||
|
||||
MobileGL::SizeT Count(const MobileGL::String& prefix) const {
|
||||
MobileGL::SizeT n = 0;
|
||||
for (const auto& c : calls) {
|
||||
if (c.compare(0, prefix.size(), prefix) == 0) ++n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
};
|
||||
|
||||
StateGuardCallLog* g_stateGuardLog = nullptr;
|
||||
GLuint g_nextStateGuardFBOId = 201;
|
||||
|
||||
void SG_Log(MobileGL::String entry) {
|
||||
if (g_stateGuardLog) g_stateGuardLog->calls.push_back(MobileGL::Move(entry));
|
||||
}
|
||||
void SG_BindBuffer(GLenum target, GLuint buffer) {
|
||||
SG_Log("BindBuffer:" + std::to_string(target) + ":" + std::to_string(buffer));
|
||||
}
|
||||
void SG_BindFramebuffer(GLenum target, GLuint framebuffer) {
|
||||
SG_Log("BindFramebuffer:" + std::to_string(target) + ":" + std::to_string(framebuffer));
|
||||
}
|
||||
void SG_GetIntegerv(GLenum pname, GLint* data) {
|
||||
SG_Log("GetIntegerv:" + std::to_string(pname));
|
||||
if (data) *data = 0;
|
||||
}
|
||||
void SG_PixelStorei(GLenum pname, GLint param) {
|
||||
SG_Log("PixelStorei:" + std::to_string(pname) + ":" + std::to_string(param));
|
||||
}
|
||||
void SG_GenFramebuffers(GLsizei count, GLuint* framebuffers) {
|
||||
for (GLsizei i = 0; i < count; ++i) framebuffers[i] = g_nextStateGuardFBOId++;
|
||||
}
|
||||
void SG_FramebufferTexture2D(GLenum target, GLenum attachment, GLenum textarget, GLuint texture, GLint level) {
|
||||
SG_Log("FramebufferTexture2D:" + std::to_string(target) + ":" + std::to_string(attachment) + ":" +
|
||||
std::to_string(textarget) + ":" + std::to_string(texture) + ":" + std::to_string(level));
|
||||
}
|
||||
void SG_FramebufferTextureLayer(GLenum target, GLenum attachment, GLuint texture, GLint level, GLint layer) {
|
||||
SG_Log("FramebufferTextureLayer:" + std::to_string(target) + ":" + std::to_string(attachment) + ":" +
|
||||
std::to_string(texture) + ":" + std::to_string(level) + ":" + std::to_string(layer));
|
||||
}
|
||||
void SG_ReadBuffer(GLenum src) {
|
||||
SG_Log("ReadBuffer:" + std::to_string(src));
|
||||
}
|
||||
void SG_DrawBuffers(GLsizei n, const GLenum* bufs) {
|
||||
SG_Log("DrawBuffers:" + std::to_string(n) + ":" + std::to_string(n > 0 && bufs ? bufs[0] : 0));
|
||||
}
|
||||
GLenum SG_NoError() {
|
||||
return GL_NO_ERROR;
|
||||
}
|
||||
|
||||
// Installs the recording table and resets every readback driver-state shadow on
|
||||
// both ends, so these tests cannot bleed into (or inherit from) other tests.
|
||||
struct ScopedStateGuardMocks {
|
||||
ScopedStateGuardMocks(): previousFunctions(MobileGL::MG_Backend::DirectGLES::g_GLESFuncs) {
|
||||
ResetShadows();
|
||||
MobileGL::MG_External::GLESFunctionsTable functions{};
|
||||
functions.glBindBuffer = SG_BindBuffer;
|
||||
functions.glBindFramebuffer = SG_BindFramebuffer;
|
||||
functions.glGetIntegerv = SG_GetIntegerv;
|
||||
functions.glPixelStorei = SG_PixelStorei;
|
||||
functions.glGenFramebuffers = SG_GenFramebuffers;
|
||||
functions.glFramebufferTexture2D = SG_FramebufferTexture2D;
|
||||
functions.glFramebufferTextureLayer = SG_FramebufferTextureLayer;
|
||||
functions.glReadBuffer = SG_ReadBuffer;
|
||||
functions.glDrawBuffers = SG_DrawBuffers;
|
||||
functions.glGetError = SG_NoError;
|
||||
MobileGL::MG_Backend::DirectGLES::SetGLESFuncsTable(functions);
|
||||
g_stateGuardLog = &log;
|
||||
}
|
||||
|
||||
~ScopedStateGuardMocks() {
|
||||
g_stateGuardLog = nullptr;
|
||||
MobileGL::MG_Backend::DirectGLES::SetGLESFuncsTable(previousFunctions);
|
||||
ResetShadows();
|
||||
}
|
||||
|
||||
ScopedStateGuardMocks(const ScopedStateGuardMocks&) = delete;
|
||||
ScopedStateGuardMocks& operator=(const ScopedStateGuardMocks&) = delete;
|
||||
|
||||
static void ResetShadows() {
|
||||
MobileGL::MG_Backend::DirectGLES::BufferImpl::InvalidatePixelBufferBindingCaches();
|
||||
MobileGL::MG_Backend::DirectGLES::FramebufferImpl::InvalidateFramebufferBindingCache();
|
||||
MobileGL::MG_Backend::DirectGLES::PixelStoreImpl::InvalidatePackStateCache();
|
||||
MobileGL::MG_Backend::DirectGLES::ScratchFBOImpl::OnBackendContextDestroyed();
|
||||
}
|
||||
|
||||
StateGuardCallLog log;
|
||||
MobileGL::MG_External::GLESFunctionsTable previousFunctions;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST(DirectGLESStateGuards, PixelPackBindingCacheSkipsRedundantBindsAndRestsAtZero) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
BufferImpl::BindPixelPackBufferId(5);
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 1u);
|
||||
BufferImpl::BindPixelPackBufferId(5); // redundant: must not reach the driver
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 1u);
|
||||
BufferImpl::BindPixelPackBufferId(0); // scope exit: resting state
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 2u);
|
||||
BufferImpl::BindPixelPackBufferId(0);
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 2u);
|
||||
|
||||
// After invalidation (MakeCurrent / context reset) the first bind must reach
|
||||
// the driver again even for the same value.
|
||||
BufferImpl::InvalidatePixelBufferBindingCaches();
|
||||
BufferImpl::BindPixelPackBufferId(0);
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 3u);
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, FramebufferBindingShadowPinsOnceThenSkips) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
// Cold path: one driver query pins the shadow; further reads are free.
|
||||
(void)FramebufferImpl::CurrentFramebufferBinding(MobileGL::FramebufferTarget::Read);
|
||||
EXPECT_EQ(mocks.log.Count("GetIntegerv:"), 1u);
|
||||
(void)FramebufferImpl::CurrentFramebufferBinding(MobileGL::FramebufferTarget::Read);
|
||||
EXPECT_EQ(mocks.log.Count("GetIntegerv:"), 1u);
|
||||
|
||||
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 1u);
|
||||
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 1u);
|
||||
// GL_FRAMEBUFFER touches both targets; DRAW is still unknown so it must bind.
|
||||
FramebufferImpl::BindFramebufferId(GL_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 2u);
|
||||
// Both halves now match: no further calls for either single target.
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 7);
|
||||
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, 7);
|
||||
FramebufferImpl::BindFramebufferId(GL_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 2u);
|
||||
EXPECT_EQ(FramebufferImpl::CurrentFramebufferBinding(MobileGL::FramebufferTarget::Draw), 7u);
|
||||
EXPECT_EQ(mocks.log.Count("GetIntegerv:"), 1u); // shadow answered, no new query
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, PackStateShadowAppliesMinimalDeltas) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
// First application pins all four parameters.
|
||||
PixelStoreImpl::ApplyPackState(PixelStoreImpl::PackState{4, 0, 0, 0});
|
||||
EXPECT_EQ(mocks.log.Count("PixelStorei:"), 4u);
|
||||
// Identical state: zero driver calls.
|
||||
PixelStoreImpl::ApplyPackState(PixelStoreImpl::PackState{4, 0, 0, 0});
|
||||
EXPECT_EQ(mocks.log.Count("PixelStorei:"), 4u);
|
||||
// One field changed: exactly one driver call.
|
||||
PixelStoreImpl::ApplyPackState(PixelStoreImpl::PackState{1, 0, 0, 0});
|
||||
EXPECT_EQ(mocks.log.Count("PixelStorei:"), 5u);
|
||||
|
||||
const auto current = PixelStoreImpl::CurrentPackState();
|
||||
EXPECT_EQ(current.Alignment, 1);
|
||||
EXPECT_EQ(current.RowLength, 0);
|
||||
EXPECT_EQ(current.SkipRows, 0);
|
||||
EXPECT_EQ(current.SkipPixels, 0);
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, ScratchFBODetachesCrossAspectResidue) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
auto& fb = ScratchFBOImpl::TempFramebuffer();
|
||||
EXPECT_NE(ScratchFBOImpl::EnsureId(fb), 0u);
|
||||
|
||||
// A depth copy leaves a DEPTH_STENCIL attachment (the pre-fix code never
|
||||
// detached it, wedging every later color readback through this FBO).
|
||||
ScratchFBOImpl::EnsureDepthAttachment2D(fb, GL_DRAW_FRAMEBUFFER, 11, GL_TEXTURE_2D, 0, /*withStencil=*/true);
|
||||
const MobileGL::String dsAttach = "FramebufferTexture2D:" + std::to_string(GL_DRAW_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_DEPTH_STENCIL_ATTACHMENT);
|
||||
EXPECT_EQ(mocks.log.Count(dsAttach), 1u);
|
||||
|
||||
// The next color use must detach the stale depth-stencil attachment exactly once.
|
||||
mocks.log.calls.clear();
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
const MobileGL::String dsDetach = "FramebufferTexture2D:" + std::to_string(GL_READ_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_DEPTH_STENCIL_ATTACHMENT) + ":" +
|
||||
std::to_string(GL_TEXTURE_2D) + ":0:0";
|
||||
const MobileGL::String colorAttach = "FramebufferTexture2D:" + std::to_string(GL_READ_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_COLOR_ATTACHMENT0) + ":" +
|
||||
std::to_string(GL_TEXTURE_2D) + ":22:0";
|
||||
EXPECT_EQ(mocks.log.Count(dsDetach), 1u);
|
||||
EXPECT_EQ(mocks.log.Count(colorAttach), 1u);
|
||||
|
||||
// Back-to-back identical color use: no driver traffic at all.
|
||||
mocks.log.calls.clear();
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
EXPECT_EQ(mocks.log.Count("FramebufferTexture2D:"), 0u);
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, ScratchFBOTextureDeletionForcesFullScrub) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
auto& fb = ScratchFBOImpl::TempFramebuffer();
|
||||
ScratchFBOImpl::EnsureId(fb);
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
|
||||
// The attached texture id dies: the shadow can no longer vouch for the FBO
|
||||
// (ES does not auto-detach from unbound FBOs, and the name may be recycled),
|
||||
// so the next use must scrub and re-attach instead of skipping.
|
||||
ScratchFBOImpl::NoteTextureIdDeleted(22);
|
||||
mocks.log.calls.clear();
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
EXPECT_GE(mocks.log.Count("FramebufferTexture2D:"), 2u); // scrub (color + depth) ...
|
||||
const MobileGL::String colorAttach = "FramebufferTexture2D:" + std::to_string(GL_READ_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_COLOR_ATTACHMENT0) + ":" +
|
||||
std::to_string(GL_TEXTURE_2D) + ":22:0";
|
||||
EXPECT_EQ(mocks.log.Count(colorAttach), 1u); // ... then the real re-attach
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, ScratchFBOReadDrawBufferStateCached) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
auto& fb = ScratchFBOImpl::BlitReadFramebuffer();
|
||||
ScratchFBOImpl::EnsureId(fb);
|
||||
|
||||
// Fresh FBOs default to COLOR_ATTACHMENT0 for both buffers: no call needed.
|
||||
ScratchFBOImpl::EnsureReadBuffer(fb, GL_COLOR_ATTACHMENT0);
|
||||
EXPECT_EQ(mocks.log.Count("ReadBuffer:"), 0u);
|
||||
// Depth blits want GL_NONE; the transition costs one call, repeats are free.
|
||||
ScratchFBOImpl::EnsureReadBuffer(fb, GL_NONE);
|
||||
ScratchFBOImpl::EnsureReadBuffer(fb, GL_NONE);
|
||||
EXPECT_EQ(mocks.log.Count("ReadBuffer:"), 1u);
|
||||
ScratchFBOImpl::EnsureDrawBuffer(fb, GL_NONE);
|
||||
ScratchFBOImpl::EnsureDrawBuffer(fb, GL_NONE);
|
||||
EXPECT_EQ(mocks.log.Count("DrawBuffers:"), 1u);
|
||||
}
|
||||
|
||||
namespace {
|
||||
MobileGL::Vector<GLuint>* g_deletedTextureIds = nullptr;
|
||||
|
||||
void SG_DeleteTextures(GLsizei count, const GLuint* textures) {
|
||||
if (!g_deletedTextureIds) return;
|
||||
for (GLsizei i = 0; i < count; ++i) g_deletedTextureIds->push_back(textures[i]);
|
||||
}
|
||||
|
||||
// Clears the recording hook even when a gtest assertion unwinds the test body
|
||||
// (a dangling pointer to the dead stack vector would corrupt later tests).
|
||||
struct ScopedDeletedTextureRecording {
|
||||
explicit ScopedDeletedTextureRecording(MobileGL::Vector<GLuint>& sink) { g_deletedTextureIds = &sink; }
|
||||
~ScopedDeletedTextureRecording() { g_deletedTextureIds = nullptr; }
|
||||
ScopedDeletedTextureRecording(const ScopedDeletedTextureRecording&) = delete;
|
||||
ScopedDeletedTextureRecording& operator=(const ScopedDeletedTextureRecording&) = delete;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST(DirectGLESBackendTexture, DestructorDeletesIdAndScrubsBindingCache) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedDirectGLESTextureBindings scoped; // installs glGenTextures/glBindTexture mocks + resets caches
|
||||
MobileGL::Vector<GLuint> deleted;
|
||||
ScopedDeletedTextureRecording recording(deleted);
|
||||
auto functions = g_GLESFuncs;
|
||||
functions.glDeleteTextures = SG_DeleteTextures;
|
||||
SetGLESFuncsTable(functions);
|
||||
|
||||
const auto texture2DSlot = static_cast<MobileGL::SizeT>(MobileGL::TextureTarget::Texture2D);
|
||||
GLuint id = 0;
|
||||
{
|
||||
auto backendTexture = MobileGL::MakeShared<TextureImpl::BackendTextureObject>();
|
||||
id = backendTexture->GetBackendTextureId();
|
||||
ASSERT_NE(id, 0u);
|
||||
backendTexture->Bind(GL_TEXTURE_2D, 0);
|
||||
ASSERT_EQ(TextureImpl::g_boundTexturesCache[0][texture2DSlot], backendTexture.get());
|
||||
}
|
||||
// Frontend glDeleteTextures used to leak the backend id forever and leave the
|
||||
// cache pointer dangling (heap-address reuse then false-skips a later Bind).
|
||||
ASSERT_EQ(deleted.size(), 1u);
|
||||
EXPECT_EQ(deleted[0], id);
|
||||
EXPECT_EQ(TextureImpl::g_boundTexturesCache[0][texture2DSlot], nullptr);
|
||||
|
||||
// A wrapper whose context died must NOT delete a foreign (recycled) name.
|
||||
{
|
||||
auto backendTexture = MobileGL::MakeShared<TextureImpl::BackendTextureObject>();
|
||||
++TextureImpl::g_textureContextGeneration;
|
||||
backendTexture.reset();
|
||||
--TextureImpl::g_textureContextGeneration; // restore for later tests
|
||||
EXPECT_EQ(deleted.size(), 1u);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, DefaultFramebufferBindGoesThroughShadow) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
// The regression this guards against: binding framebuffer 0 raw while the
|
||||
// shadow keeps a user-FBO id makes the next re-bind of that FBO false-skip.
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 7);
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 0); // default-FBO path must use this API
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 7); // must reach the driver again
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 3u);
|
||||
}
|
||||
|
||||
// FastSTL::unordered_map::erase(iterator) regression coverage. The open-addressing
|
||||
// iterator constructor snaps forward from a tombstoned slot to the successor, so
|
||||
// erase must NOT advance the rebuilt iterator again: the old double-advance skipped
|
||||
// one live element per erase, and erasing the element in the highest occupied
|
||||
// bucket pushed the returned index past bucket_count where it never compared equal
|
||||
// to end() again - erase-while-iterating sweeps (pipeline/program cache eviction)
|
||||
// then ran off the bucket array and fed garbage handles to vkDestroyPipeline
|
||||
// (device crash on first mass eviction during world load).
|
||||
TEST(FastSTLSanity, EraseWhileIteratingVisitsEveryElementExactlyOnce) {
|
||||
FastSTL::unordered_map<MobileGL::Uint64, MobileGL::Uint64> map;
|
||||
constexpr MobileGL::Uint64 kCount = 1000;
|
||||
for (MobileGL::Uint64 key = 0; key < kCount; ++key) {
|
||||
map.emplace(key * 0x9e3779b97f4a7c15ull, key);
|
||||
}
|
||||
ASSERT_EQ(map.size(), kCount);
|
||||
|
||||
MobileGL::SizeT visited = 0;
|
||||
for (auto it = map.begin(); it != map.end();) {
|
||||
it = map.erase(it);
|
||||
++visited;
|
||||
ASSERT_LE(visited, kCount); // old code: runaway past end / skipped entries
|
||||
}
|
||||
EXPECT_EQ(visited, kCount);
|
||||
EXPECT_EQ(map.size(), 0u);
|
||||
}
|
||||
|
||||
TEST(FastSTLSanity, EraseReturnsTheSuccessorElement) {
|
||||
FastSTL::unordered_map<MobileGL::Uint32, MobileGL::Uint32> map;
|
||||
for (MobileGL::Uint32 key = 1; key <= 64; ++key) {
|
||||
map.emplace(key, key);
|
||||
}
|
||||
|
||||
// Erasing every other visited element must still visit all 64 exactly once:
|
||||
// the iterator returned by erase names the very next element, not one past it.
|
||||
MobileGL::SizeT visited = 0;
|
||||
MobileGL::SizeT erased = 0;
|
||||
for (auto it = map.begin(); it != map.end();) {
|
||||
++visited;
|
||||
if ((visited & 1) != 0) {
|
||||
it = map.erase(it);
|
||||
++erased;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
ASSERT_LE(visited, 64u);
|
||||
}
|
||||
EXPECT_EQ(visited, 64u);
|
||||
EXPECT_EQ(map.size(), 64u - erased);
|
||||
}
|
||||
|
||||
TEST(FastSTLSanity, ErasingTheOnlyElementReturnsEnd) {
|
||||
FastSTL::unordered_map<MobileGL::Uint32, MobileGL::Uint32> map;
|
||||
map.emplace(42u, 1u);
|
||||
auto next = map.erase(map.begin());
|
||||
EXPECT_EQ(next, map.end());
|
||||
EXPECT_TRUE(map.empty());
|
||||
}
|
||||
|
||||
@@ -1516,6 +1516,69 @@ TEST_F(TextureTest, TexStorage2DTrimsALongerPreExistingMipChain) {
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
// glTexImage2D used to reject every GL_COMPRESSED_* internal format with GL_INVALID_ENUM, because
|
||||
// none of them mapped to a TextureInternalFormat and the "unknown format" gate fired. They now
|
||||
// resolve to the uncompressed storage that backs them - what GL prescribes for the generic formats,
|
||||
// and a deliberate deviation for RGTC, which ES cannot compress. The (format, type) pairs below are
|
||||
// the ones KHR-GL33.packed_pixels uploads with, so this table doubles as a pin for those 480 cases.
|
||||
TEST_F(TextureTest, CompressedInternalFormatsResolveToTheirUncompressedStorage) {
|
||||
struct Case {
|
||||
GLenum internalFormat;
|
||||
GLenum format;
|
||||
GLenum type;
|
||||
TextureInternalFormat expected;
|
||||
};
|
||||
const Case cases[] = {
|
||||
{GL_COMPRESSED_RED, GL_RED, GL_UNSIGNED_BYTE, TextureInternalFormat::R8},
|
||||
{GL_COMPRESSED_RG, GL_RG, GL_UNSIGNED_BYTE, TextureInternalFormat::RG8},
|
||||
{GL_COMPRESSED_RGB, GL_RGB, GL_UNSIGNED_BYTE, TextureInternalFormat::RGB8},
|
||||
{GL_COMPRESSED_RGBA, GL_RGBA, GL_UNSIGNED_BYTE, TextureInternalFormat::RGBA8},
|
||||
{GL_COMPRESSED_SRGB, GL_RGB, GL_UNSIGNED_BYTE, TextureInternalFormat::SRGB8},
|
||||
{GL_COMPRESSED_SRGB_ALPHA, GL_RGBA, GL_UNSIGNED_BYTE, TextureInternalFormat::SRGB8Alpha8},
|
||||
{GL_COMPRESSED_RED_RGTC1, GL_RED, GL_UNSIGNED_BYTE, TextureInternalFormat::R8},
|
||||
{GL_COMPRESSED_RG_RGTC2, GL_RG, GL_UNSIGNED_BYTE, TextureInternalFormat::RG8},
|
||||
// The signed RGTC pair is uploaded as GL_BYTE and must land on SNORM storage - resolving
|
||||
// them to plain R8/RG8 would silently reinterpret negative texels.
|
||||
{GL_COMPRESSED_SIGNED_RED_RGTC1, GL_RED, GL_BYTE, TextureInternalFormat::R8Snorm},
|
||||
{GL_COMPRESSED_SIGNED_RG_RGTC2, GL_RG, GL_BYTE, TextureInternalFormat::RG8Snorm},
|
||||
};
|
||||
|
||||
MG_Impl::GLImpl::PixelStorei(GL_UNPACK_ALIGNMENT, 1);
|
||||
for (const auto& c : cases) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, c.internalFormat, 4, 4, 0, c.format, c.type, nullptr);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
ASSERT_NE(textureObject, nullptr) << "internalFormat 0x" << std::hex << c.internalFormat;
|
||||
EXPECT_EQ(textureObject->GetFormat(), c.expected) << "internalFormat 0x" << std::hex << c.internalFormat;
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR) << "internalFormat 0x" << std::hex << c.internalFormat;
|
||||
}
|
||||
}
|
||||
|
||||
// RGTC compresses 4x4 blocks of a 2D image and has no 3D form, so glTexImage3D must reject it even
|
||||
// though the same enum is accepted on a 2D target. The generic compressed formats carry no such
|
||||
// restriction and stay legal in 3D.
|
||||
TEST_F(TextureTest, RgtcInternalFormatsAreRejectedOnThreeDimensionalTargets) {
|
||||
const GLenum rgtc[] = {GL_COMPRESSED_RED_RGTC1, GL_COMPRESSED_SIGNED_RED_RGTC1, GL_COMPRESSED_RG_RGTC2,
|
||||
GL_COMPRESSED_SIGNED_RG_RGTC2};
|
||||
for (const GLenum internalFormat : rgtc) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_3D, texture);
|
||||
MG_Impl::GLImpl::TexImage3D(GL_TEXTURE_3D, 0, internalFormat, 4, 4, 4, 0, GL_RED, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_INVALID_OPERATION)
|
||||
<< "internalFormat 0x" << std::hex << internalFormat;
|
||||
}
|
||||
|
||||
GLuint generic = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &generic);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_3D, generic);
|
||||
MG_Impl::GLImpl::TexImage3D(GL_TEXTURE_3D, 0, GL_COMPRESSED_RGBA, 4, 4, 4, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, TextureStorage3DAndSubImageModifyNamedObjectOnly) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_3D, 1, &texture);
|
||||
|
||||
@@ -9,7 +9,12 @@
|
||||
#include "Loader.h"
|
||||
#include "MG_Util/Types.h"
|
||||
#include <Config.h>
|
||||
#if !defined(__WIN32) && !defined(_WIN32)
|
||||
#if defined(_WIN32)
|
||||
#ifndef WIN32_LEAN_AND_MEAN
|
||||
#define WIN32_LEAN_AND_MEAN 1
|
||||
#endif
|
||||
#include <windows.h>
|
||||
#else
|
||||
#include <dlfcn.h>
|
||||
#endif
|
||||
|
||||
@@ -60,7 +65,14 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
#endif
|
||||
|
||||
static void* OpenLib(const Vector<String>& names) {
|
||||
#if !defined(__WIN32) && !defined(_WIN32) && (!defined(__APPLE__) || defined(MOBILEGL_IOS))
|
||||
#if defined(_WIN32)
|
||||
for (const auto& name : names) {
|
||||
if (HMODULE lib = LoadLibraryA(name.c_str())) {
|
||||
MGLOG_I("Loaded GL backend library: %s", name.c_str());
|
||||
return reinterpret_cast<void*>(lib);
|
||||
}
|
||||
}
|
||||
#elif !defined(__APPLE__) || defined(MOBILEGL_IOS)
|
||||
static const String LibPathPrefixes[] = {
|
||||
#if defined(MOBILEGL_IOS)
|
||||
"@rpath/", "@executable_path/Frameworks/", "@loader_path/Frameworks/",
|
||||
@@ -104,7 +116,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
}
|
||||
|
||||
inline void* ProcAddress(void* lib, const char* name) {
|
||||
#if !defined(__WIN32) && !defined(_WIN32) && (!defined(__APPLE__) || defined(MOBILEGL_IOS))
|
||||
#if defined(_WIN32)
|
||||
return reinterpret_cast<void*>(::GetProcAddress(reinterpret_cast<HMODULE>(lib), name));
|
||||
#elif !defined(__APPLE__) || defined(MOBILEGL_IOS)
|
||||
return dlsym(lib, name);
|
||||
#else
|
||||
return nullptr;
|
||||
@@ -520,6 +534,15 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS) && defined(__ANDROID__)
|
||||
void* angleGlesLib = nullptr;
|
||||
#endif
|
||||
#if defined(_WIN32)
|
||||
// ANGLE is the GLES provider on Windows regardless of UseAngle(). Preload
|
||||
// libGLESv2.dll so libEGL.dll resolves its dependency from the same directory.
|
||||
if (!OpenLib({"libGLESv2.dll"})) {
|
||||
MGLOG_E("Failed to open ANGLE libGLESv2.dll");
|
||||
return;
|
||||
}
|
||||
eglLib = OpenLib({"libEGL.dll"});
|
||||
#else
|
||||
if (UseAngle()) {
|
||||
void* glesLib = OpenLib({"libGLESv2_angle.so"});
|
||||
if (!glesLib) {
|
||||
@@ -541,6 +564,7 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
eglLib = OpenLib({"libEGL.so"});
|
||||
#endif
|
||||
}
|
||||
#endif // !_WIN32
|
||||
|
||||
if (!eglLib) {
|
||||
MGLOG_E("Failed to open EGL library");
|
||||
@@ -811,6 +835,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
if (std::strcmp(extension, "GL_EXT_blend_func_extended") == 0) {
|
||||
caps.SupportsDualSourceBlend = true;
|
||||
}
|
||||
if (std::strcmp(extension, "GL_NV_shader_noperspective_interpolation") == 0) {
|
||||
caps.SupportsNoperspectiveInterpolation = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1050,6 +1050,11 @@ namespace MobileGL {
|
||||
// factors and layout(index = 1) fragment outputs. GLES core has no dual-source blending,
|
||||
// so without this a draw using a SRC1 factor cannot proceed.
|
||||
Bool SupportsDualSourceBlend = false;
|
||||
// GL_NV_shader_noperspective_interpolation is present: the driver accepts the
|
||||
// `noperspective` interpolation qualifier in ESSL. GLES core has none, so without this
|
||||
// SPIRV-Cross's `#extension ... : require` would fail to compile and MobileGL falls back
|
||||
// to stripping the NoPerspective decoration (smooth interpolation) via StripNoPerspectivePass.
|
||||
Bool SupportsNoperspectiveInterpolation = false;
|
||||
// GL_RENDERER contains "ANGLE".
|
||||
Bool IsAngleRenderer = false;
|
||||
// GL_RENDERER contains both "ANGLE" and "llvmpipe".
|
||||
|
||||
@@ -255,6 +255,37 @@ namespace MobileGL {
|
||||
return TextureInternalFormat::DepthComponent;
|
||||
case GL_DEPTH_STENCIL:
|
||||
return TextureInternalFormat::DepthStencil;
|
||||
// Compressed internal formats resolve to the uncompressed storage that backs them.
|
||||
//
|
||||
// For the six generic formats this is exactly what GL prescribes: the implementation
|
||||
// picks a specific compressed format, and when none is available it falls back to the
|
||||
// corresponding base format. Nothing downstream ever sees a compressed enum, so the
|
||||
// metrics, pixel-store and backend tables keep their "one format, N bytes per texel"
|
||||
// invariant instead of each needing a compressed-aware arm.
|
||||
//
|
||||
// The four RGTC formats are a deliberate deviation: they are specific formats that GL
|
||||
// 3.3 requires, but ES exposes no RGTC compressor to hand the data to. Storing the
|
||||
// texels uncompressed keeps them renderable at the cost of the memory saving, which is
|
||||
// strictly better than the INVALID_ENUM the application used to get. Note the signed
|
||||
// variants must land on SNORM storage - CTS uploads them as GL_BYTE.
|
||||
case GL_COMPRESSED_RED:
|
||||
case GL_COMPRESSED_RED_RGTC1:
|
||||
return TextureInternalFormat::R8;
|
||||
case GL_COMPRESSED_SIGNED_RED_RGTC1:
|
||||
return TextureInternalFormat::R8Snorm;
|
||||
case GL_COMPRESSED_RG:
|
||||
case GL_COMPRESSED_RG_RGTC2:
|
||||
return TextureInternalFormat::RG8;
|
||||
case GL_COMPRESSED_SIGNED_RG_RGTC2:
|
||||
return TextureInternalFormat::RG8Snorm;
|
||||
case GL_COMPRESSED_RGB:
|
||||
return TextureInternalFormat::RGB8;
|
||||
case GL_COMPRESSED_RGBA:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
case GL_COMPRESSED_SRGB:
|
||||
return TextureInternalFormat::SRGB8;
|
||||
case GL_COMPRESSED_SRGB_ALPHA:
|
||||
return TextureInternalFormat::SRGB8Alpha8;
|
||||
case GL_ALPHA:
|
||||
case GL_RED:
|
||||
return TextureInternalFormat::Red;
|
||||
|
||||
@@ -133,9 +133,11 @@ namespace MobileGL {
|
||||
case TextureInternalFormat::RGBA8Snorm:
|
||||
return VK_FORMAT_R8G8B8A8_SNORM;
|
||||
case TextureInternalFormat::RGB10A2:
|
||||
return VK_FORMAT_A2R10G10B10_UNORM_PACK32;
|
||||
// GL_UNSIGNED_INT_2_10_10_10_REV puts R in bits 0-9, which is Vulkan's
|
||||
// A2B10G10R10 layout - A2R10G10B10 silently swaps R and B on upload.
|
||||
return VK_FORMAT_A2B10G10R10_UNORM_PACK32;
|
||||
case TextureInternalFormat::RGB10A2UI:
|
||||
return VK_FORMAT_A2R10G10B10_UINT_PACK32;
|
||||
return VK_FORMAT_A2B10G10R10_UINT_PACK32;
|
||||
case TextureInternalFormat::RGBA16:
|
||||
return VK_FORMAT_R16G16B16A16_UNORM;
|
||||
case TextureInternalFormat::RGBA16Snorm:
|
||||
|
||||
@@ -401,6 +401,230 @@ namespace MobileGL::MG_Util::SelfTest {
|
||||
disabledNote);
|
||||
}
|
||||
|
||||
// Compiles + links a two-stage program on the probe context. Returns 0 on failure and writes a
|
||||
// human-readable reason into |detail|.
|
||||
GLuint CompileLinkProgram(const MG_External::GLESFunctionsTable& g, const char* vs, const char* fs,
|
||||
String& detail) {
|
||||
const auto compile = [&](GLenum stage, const char* src, GLuint& out) -> bool {
|
||||
out = g.glCreateShader(stage);
|
||||
if (out == 0) {
|
||||
detail = "glCreateShader returned 0";
|
||||
return false;
|
||||
}
|
||||
g.glShaderSource(out, 1, &src, nullptr);
|
||||
g.glCompileShader(out);
|
||||
GLint ok = GL_FALSE;
|
||||
g.glGetShaderiv(out, GL_COMPILE_STATUS, &ok);
|
||||
if (ok != GL_TRUE) {
|
||||
GLchar log[512] = {};
|
||||
GLsizei len = 0;
|
||||
g.glGetShaderInfoLog(out, static_cast<GLsizei>(sizeof(log) - 1), &len, log);
|
||||
detail = format("{} shader compile failed: {}",
|
||||
stage == GL_VERTEX_SHADER ? "vertex" : "fragment",
|
||||
len > 0 ? log : "(no info log)");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
};
|
||||
GLuint v = 0, f = 0;
|
||||
const ScopeGuard delV([&]() { if (v) g.glDeleteShader(v); });
|
||||
const ScopeGuard delF([&]() { if (f) g.glDeleteShader(f); });
|
||||
if (!compile(GL_VERTEX_SHADER, vs, v) || !compile(GL_FRAGMENT_SHADER, fs, f)) {
|
||||
return 0;
|
||||
}
|
||||
const GLuint prog = g.glCreateProgram();
|
||||
if (prog == 0) {
|
||||
detail = "glCreateProgram returned 0";
|
||||
return 0;
|
||||
}
|
||||
g.glAttachShader(prog, v);
|
||||
g.glAttachShader(prog, f);
|
||||
g.glLinkProgram(prog);
|
||||
GLint linked = GL_FALSE;
|
||||
g.glGetProgramiv(prog, GL_LINK_STATUS, &linked);
|
||||
if (linked != GL_TRUE) {
|
||||
detail = "program link failed";
|
||||
g.glDeleteProgram(prog);
|
||||
return 0;
|
||||
}
|
||||
return prog;
|
||||
}
|
||||
|
||||
// "noperspective interpolation" row - a real correctness render, not just a compile. A viewport-
|
||||
// filling quad is drawn with strong perspective (left clip-w 1, right clip-w 8) and a varying that
|
||||
// runs 0..1 across it. At the screen centre screen-linear interpolation gives 0.5 while perspective-
|
||||
// correct gives 1/(w+1) ~= 0.11, so reading the centre texel tells the two apart. The varying is
|
||||
// carried either through the native `noperspective` qualifier (extension present) or through the
|
||||
// exact gl_Position.w / gl_FragCoord.w rewrite MobileGL applies when it is absent. Verdict:
|
||||
// PASS - extension present and the native noperspective result is screen-linear;
|
||||
// WARN - extension absent but the gl_Position.w/gl_FragCoord.w emulation renders screen-linear
|
||||
// (correct, just the fallback path shipping shader packs hit on such devices);
|
||||
// FAIL - either path renders perspective-correct / wrong (noperspective does not actually work),
|
||||
// or the program will not compile/link, or the render errors.
|
||||
// Requires the probe context to still be current.
|
||||
void ProbeGlesNoperspective(ReportBuilder& builder, const MG_External::GLESCapabilities& caps,
|
||||
const MG_External::GLESFunctionsTable& g) {
|
||||
const Bool native = caps.SupportsNoperspectiveInterpolation;
|
||||
const String pathNote = native ? "GL_NV_shader_noperspective_interpolation present (native path)"
|
||||
: "GL_NV_shader_noperspective_interpolation absent (gl_Position.w / "
|
||||
"gl_FragCoord.w emulation path)";
|
||||
const auto fail = [&](const String& detail) {
|
||||
builder.Fail("noperspective interpolation", pathNote + "; " + detail);
|
||||
};
|
||||
|
||||
if (!g.glCreateShader || !g.glShaderSource || !g.glCompileShader || !g.glGetShaderiv ||
|
||||
!g.glGetShaderInfoLog || !g.glDeleteShader || !g.glCreateProgram || !g.glAttachShader ||
|
||||
!g.glLinkProgram || !g.glGetProgramiv || !g.glUseProgram || !g.glDeleteProgram ||
|
||||
!g.glGenFramebuffers || !g.glBindFramebuffer || !g.glDeleteFramebuffers ||
|
||||
!g.glGenRenderbuffers || !g.glBindRenderbuffer || !g.glRenderbufferStorage ||
|
||||
!g.glFramebufferRenderbuffer || !g.glDeleteRenderbuffers || !g.glCheckFramebufferStatus ||
|
||||
!g.glGenBuffers || !g.glBindBuffer || !g.glBufferData || !g.glDeleteBuffers ||
|
||||
!g.glGetAttribLocation || !g.glVertexAttribPointer || !g.glEnableVertexAttribArray ||
|
||||
!g.glViewport || !g.glClearColor || !g.glClear || !g.glDrawArrays || !g.glReadPixels ||
|
||||
!g.glFinish || !g.glGetError) {
|
||||
fail("the render entry points did not resolve through eglGetProcAddress");
|
||||
return;
|
||||
}
|
||||
|
||||
// Match MobileGL's own ESSL target (the device's version). At #version 300 es some drivers
|
||||
// (Adreno) still treat `noperspective` as reserved even with the extension enabled; the ES 3.2
|
||||
// form the backend actually emits compiles. Emulated shaders are version-agnostic but use the
|
||||
// same header for consistency.
|
||||
const Int esslVer = caps.GLESVersion.Major * 100 + caps.GLESVersion.Minor * 10;
|
||||
const String header = format("#version {} es\n", esslVer >= 300 ? esslVer : 300);
|
||||
static const char* const kVsNativeBody =
|
||||
"#extension GL_NV_shader_noperspective_interpolation : require\n"
|
||||
"in vec4 a_pos;\n"
|
||||
"in float a_v;\n"
|
||||
"noperspective out highp float v_out;\n"
|
||||
"void main() { gl_Position = a_pos; v_out = a_v; }\n";
|
||||
static const char* const kFsNativeBody =
|
||||
"#extension GL_NV_shader_noperspective_interpolation : require\n"
|
||||
"precision highp float;\n"
|
||||
"noperspective in highp float v_out;\n"
|
||||
"out vec4 fragColor;\n"
|
||||
"void main() { fragColor = vec4(v_out, 0.0, 0.0, 1.0); }\n";
|
||||
// Exactly MobileGL's emulation (verified against EmulateNoPerspectivePass output): pre-multiply
|
||||
// the varying by clip-w in the vertex stage, recover with gl_FragCoord.w in the fragment stage,
|
||||
// no noperspective qualifier (so the driver interpolates it perspective-correct).
|
||||
static const char* const kVsEmuBody =
|
||||
"in vec4 a_pos;\n"
|
||||
"in float a_v;\n"
|
||||
"out highp float v_out;\n"
|
||||
"void main() { gl_Position = a_pos; v_out = a_v * gl_Position.w; }\n";
|
||||
static const char* const kFsEmuBody =
|
||||
"precision highp float;\n"
|
||||
"in highp float v_out;\n"
|
||||
"out vec4 fragColor;\n"
|
||||
"void main() { fragColor = vec4(v_out * gl_FragCoord.w, 0.0, 0.0, 1.0); }\n";
|
||||
|
||||
while (g.glGetError() != GL_NO_ERROR) {
|
||||
}
|
||||
|
||||
const String vsSrc = header + (native ? kVsNativeBody : kVsEmuBody);
|
||||
const String fsSrc = header + (native ? kFsNativeBody : kFsEmuBody);
|
||||
String linkDetail;
|
||||
const GLuint prog = CompileLinkProgram(g, vsSrc.c_str(), fsSrc.c_str(), linkDetail);
|
||||
if (prog == 0) {
|
||||
fail(native ? "a noperspective program failed to build though the extension is advertised: " +
|
||||
linkDetail
|
||||
: "the emulation program failed to build: " + linkDetail);
|
||||
return;
|
||||
}
|
||||
const ScopeGuard delProg([&]() { g.glDeleteProgram(prog); });
|
||||
|
||||
// 9x9 so the centre texel (4,4) sits exactly at NDC (0,0).
|
||||
constexpr GLsizei kDim = 9;
|
||||
GLuint rbo = 0, fbo = 0, vbo = 0;
|
||||
g.glGenRenderbuffers(1, &rbo);
|
||||
const ScopeGuard delRbo([&]() { if (rbo) g.glDeleteRenderbuffers(1, &rbo); });
|
||||
g.glBindRenderbuffer(GL_RENDERBUFFER, rbo);
|
||||
g.glRenderbufferStorage(GL_RENDERBUFFER, GL_RGBA8, kDim, kDim);
|
||||
g.glGenFramebuffers(1, &fbo);
|
||||
const ScopeGuard delFbo([&]() {
|
||||
if (fbo) {
|
||||
g.glBindFramebuffer(GL_FRAMEBUFFER, 0);
|
||||
g.glDeleteFramebuffers(1, &fbo);
|
||||
}
|
||||
});
|
||||
g.glBindFramebuffer(GL_FRAMEBUFFER, fbo);
|
||||
g.glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, rbo);
|
||||
if (g.glCheckFramebufferStatus(GL_FRAMEBUFFER) != GL_FRAMEBUFFER_COMPLETE) {
|
||||
fail("the probe framebuffer is incomplete");
|
||||
return;
|
||||
}
|
||||
|
||||
// Interleaved [vec4 clip-pos, float v]. Left w=1, right w=8; x/y pre-multiplied by w so the quad
|
||||
// still fills NDC after the perspective divide.
|
||||
const GLfloat verts[] = {
|
||||
-1.f, -1.f, 0.f, 1.f, 0.f, //
|
||||
8.f, -8.f, 0.f, 8.f, 1.f, //
|
||||
-1.f, 1.f, 0.f, 1.f, 0.f, //
|
||||
8.f, 8.f, 0.f, 8.f, 1.f, //
|
||||
};
|
||||
g.glGenBuffers(1, &vbo);
|
||||
const ScopeGuard delVbo([&]() { if (vbo) g.glDeleteBuffers(1, &vbo); });
|
||||
g.glBindBuffer(GL_ARRAY_BUFFER, vbo);
|
||||
g.glBufferData(GL_ARRAY_BUFFER, sizeof(verts), verts, GL_STATIC_DRAW);
|
||||
|
||||
g.glUseProgram(prog);
|
||||
const GLint posLoc = g.glGetAttribLocation(prog, "a_pos");
|
||||
const GLint vLoc = g.glGetAttribLocation(prog, "a_v");
|
||||
if (posLoc < 0 || vLoc < 0) {
|
||||
fail("the probe vertex attributes did not resolve");
|
||||
return;
|
||||
}
|
||||
g.glEnableVertexAttribArray(static_cast<GLuint>(posLoc));
|
||||
g.glVertexAttribPointer(static_cast<GLuint>(posLoc), 4, GL_FLOAT, GL_FALSE, 5 * sizeof(GLfloat),
|
||||
reinterpret_cast<const void*>(0));
|
||||
g.glEnableVertexAttribArray(static_cast<GLuint>(vLoc));
|
||||
g.glVertexAttribPointer(static_cast<GLuint>(vLoc), 1, GL_FLOAT, GL_FALSE, 5 * sizeof(GLfloat),
|
||||
reinterpret_cast<const void*>(4 * sizeof(GLfloat)));
|
||||
|
||||
g.glViewport(0, 0, kDim, kDim);
|
||||
g.glClearColor(0.f, 0.f, 0.f, 1.f);
|
||||
g.glClear(GL_COLOR_BUFFER_BIT);
|
||||
g.glDrawArrays(GL_TRIANGLE_STRIP, 0, 4);
|
||||
g.glFinish();
|
||||
|
||||
const GLenum drawError = g.glGetError();
|
||||
if (drawError != GL_NO_ERROR) {
|
||||
fail(format("GL error 0x{:x} while rendering the probe quad", drawError));
|
||||
return;
|
||||
}
|
||||
|
||||
GLubyte center[4] = {};
|
||||
g.glReadPixels(kDim / 2, kDim / 2, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, center);
|
||||
const GLenum readError = g.glGetError();
|
||||
if (readError != GL_NO_ERROR) {
|
||||
fail(format("GL error 0x{:x} while reading the probe pixel back", readError));
|
||||
return;
|
||||
}
|
||||
|
||||
// At the centre: screen-linear -> 0.5 (~128); perspective-correct -> 1/(8+1) ~= 0.111 (~28).
|
||||
const float observed = static_cast<float>(center[0]) / 255.0f;
|
||||
const int observedByte = center[0];
|
||||
constexpr float kScreenLinear = 0.5f;
|
||||
const bool screenLinear = observed > 0.5f * (kScreenLinear + 1.0f / 9.0f); // midpoint ~= 0.306
|
||||
if (!screenLinear) {
|
||||
fail(format("the centre texel read {} (~{:.3f}); expected the screen-linear ~0.5 - "
|
||||
"interpolation came out perspective-correct, so noperspective does not work here",
|
||||
observedByte, observed));
|
||||
return;
|
||||
}
|
||||
if (native) {
|
||||
builder.Pass("noperspective interpolation",
|
||||
pathNote + format("; native noperspective renders screen-linear (centre {} ~= 0.5)",
|
||||
observedByte));
|
||||
} else {
|
||||
builder.Warn("noperspective interpolation",
|
||||
pathNote +
|
||||
format("; the emulation renders screen-linear correctly (centre {} ~= 0.5), "
|
||||
"but this is the fallback path with less driver coverage",
|
||||
observedByte));
|
||||
}
|
||||
}
|
||||
|
||||
// Everything the "MobileGL reported ..." rows need from the GLES device probe.
|
||||
struct GlesProbeSummary {
|
||||
Bool capsValid = false;
|
||||
@@ -527,6 +751,7 @@ namespace MobileGL::MG_Util::SelfTest {
|
||||
builder.report.rendererInfo = format("{} ({})", caps.GLESRendererString, caps.GLESVersionString);
|
||||
EvaluateGlesChecklist(builder, caps, glesFuncs);
|
||||
ProbeGlesTimerQuery(builder, caps, glesFuncs);
|
||||
ProbeGlesNoperspective(builder, caps, glesFuncs);
|
||||
builder.report.formatCapabilities.emplace();
|
||||
MG_Backend::DirectGLES::PopulateFormatCapabilities(
|
||||
glesFuncs, caps, builder.report.formatCapabilities.value());
|
||||
|
||||
@@ -20,6 +20,8 @@
|
||||
#include "SpirvPasses/LowerDrawParametersPass.h"
|
||||
#include "SpirvPasses/RebaseInstanceIndexPass.h"
|
||||
#include "SpirvPasses/StripUboMemberRelaxedPrecisionPass.h"
|
||||
#include "SpirvPasses/StripNoPerspectivePass.h"
|
||||
#include "SpirvPasses/EmulateNoPerspectivePass.h"
|
||||
#include "spirv-tools/libspirv.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
@@ -334,6 +336,30 @@ namespace MobileGL {
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(StripNoPerspectivePass::CreateStripNoPerspectivePass());
|
||||
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(EmulateNoPerspectivePass::CreateEmulateNoPerspectivePass());
|
||||
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
|
||||
@@ -33,6 +33,16 @@ namespace MobileGL {
|
||||
// Only for the DirectGLES transpile path.
|
||||
static bool StripUboMemberRelaxedPrecisionForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Removes NoPerspective decorations so SPIRV-Cross emits plain (smooth) ESSL varyings.
|
||||
// DirectGLES fallback only, for devices lacking GL_NV_shader_noperspective_interpolation
|
||||
// (SPIRV-Cross would otherwise require that extension and the driver would reject it).
|
||||
static bool StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Emulates noperspective (screen-linear) interpolation via gl_Position.w / gl_FragCoord.w
|
||||
// so no NV extension is needed; strips what it cannot emulate. DirectGLES fallback for
|
||||
// devices lacking GL_NV_shader_noperspective_interpolation. See EmulateNoPerspectivePass.
|
||||
static bool EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Rebases loads of the InstanceIndex builtin to (InstanceIndex - BaseInstance) so
|
||||
// shaders see GL's zero-based gl_InstanceID. Vertex shaders only; DirectVulkan
|
||||
// backend only (glslang's relaxed mode aliases gl_InstanceID to gl_InstanceIndex,
|
||||
|
||||
@@ -96,6 +96,77 @@ namespace {
|
||||
return masked;
|
||||
}
|
||||
|
||||
// Blank out block comments in place, leaving line comments and every other byte where it is.
|
||||
//
|
||||
// The passes that follow scan the source as raw text, so block comments have to stop being
|
||||
// visible to them - but they must not be *deleted*: replacing the bytes with spaces keeps every
|
||||
// later offset valid and keeps newlines, so glslang's diagnostics still point at the line the
|
||||
// application wrote. It also has to be lexically aware. A banner line such as
|
||||
//
|
||||
// //*** lighting pass ***
|
||||
//
|
||||
// contains "/*" one byte in, and a naive search for that opener treats the rest of the file as
|
||||
// an unterminated comment.
|
||||
void BlankBlockComments(MobileGL::String& source) {
|
||||
enum class Region { Code, SingleLineComment, MultiLineComment, QuotedText };
|
||||
|
||||
Region region = Region::Code;
|
||||
char quote = '\0';
|
||||
bool escaped = false;
|
||||
|
||||
for (SizeT pos = 0; pos < source.size(); pos++) {
|
||||
const char ch = source[pos];
|
||||
const char next = pos + 1 < source.size() ? source[pos + 1] : '\0';
|
||||
|
||||
if (region == Region::Code) {
|
||||
if (ch == '/' && next == '/') {
|
||||
pos++;
|
||||
region = Region::SingleLineComment;
|
||||
} else if (ch == '/' && next == '*') {
|
||||
source[pos] = ' ';
|
||||
source[pos + 1] = ' ';
|
||||
pos++;
|
||||
region = Region::MultiLineComment;
|
||||
} else if (ch == '"' || ch == '\'') {
|
||||
quote = ch;
|
||||
escaped = false;
|
||||
region = Region::QuotedText;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (region == Region::SingleLineComment) {
|
||||
if (ch == '\n' || ch == '\r') region = Region::Code;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (region == Region::MultiLineComment) {
|
||||
if (ch == '*' && next == '/') {
|
||||
source[pos] = ' ';
|
||||
source[pos + 1] = ' ';
|
||||
pos++;
|
||||
region = Region::Code;
|
||||
} else if (ch != '\n' && ch != '\r') {
|
||||
source[pos] = ' ';
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// GLSL has no multi-line string literals, so a quote that reaches end of line was never
|
||||
// a literal to begin with - most likely an apostrophe in a #error or #pragma message.
|
||||
// Ending the region here keeps one stray apostrophe from swallowing the rest of the file.
|
||||
if (ch == '\n' || ch == '\r') {
|
||||
region = Region::Code;
|
||||
} else if (escaped) {
|
||||
escaped = false;
|
||||
} else if (ch == '\\') {
|
||||
escaped = true;
|
||||
} else if (ch == quote) {
|
||||
region = Region::Code;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct CodeToken {
|
||||
String text;
|
||||
SizeT begin = 0;
|
||||
@@ -502,6 +573,23 @@ namespace {
|
||||
static_cast<unsigned char>(source[1]) == 0xbb && static_cast<unsigned char>(source[2]) == 0xbf;
|
||||
}
|
||||
|
||||
// The GLSL versions MobileGL is willing to normalize. Anything else in a #version line - a number
|
||||
// that is not a real language version (329, 331), a bad profile keyword, a float/identifier where
|
||||
// the integer belongs, or trailing tokens - is left untouched so glslang rejects it, matching
|
||||
// KHR-GL33.shaders.preprocessor.directive.version_*. The set is deliberately generous (every real
|
||||
// desktop and ES version) so the normalizer never starts rejecting a form it used to accept.
|
||||
bool IsRecognizedGlslVersion(unsigned version) {
|
||||
switch (version) {
|
||||
case 100: case 110: case 120: case 130: case 140: case 150:
|
||||
case 300: case 310: case 320:
|
||||
case 330: case 400: case 410: case 420: case 430:
|
||||
case 440: case 450: case 460:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
struct ShaderLanguageInfo {
|
||||
unsigned version = 110;
|
||||
MobileGL::ShaderProfile profile = MobileGL::ShaderProfile::Core;
|
||||
@@ -509,6 +597,9 @@ namespace {
|
||||
SizeT versionDirectiveEnd = MobileGL::String::npos;
|
||||
bool hasUtf8Bom = false;
|
||||
bool enablesGpuShader5 = false;
|
||||
// Whether the parsed #version directive is a well-formed one MobileGL should rewrite. A
|
||||
// malformed directive (see IsRecognizedGlslVersion) is left alone for glslang to reject.
|
||||
bool hasValidVersionDirective = false;
|
||||
|
||||
bool HasVersionDirective() const { return versionDirectiveStart != MobileGL::String::npos; }
|
||||
};
|
||||
@@ -552,13 +643,25 @@ namespace {
|
||||
info.versionDirectiveEnd = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
SkipDirectiveWhitespace(code, probe, lineEnd);
|
||||
const MobileGL::String profile = ReadDirectiveIdentifier(code, probe, lineEnd);
|
||||
if (profile == "es" || profile == "ES") {
|
||||
bool profileTokenValid = true;
|
||||
if (profile.empty() || profile == "core") {
|
||||
info.profile = MobileGL::ShaderProfile::Core;
|
||||
} else if (profile == "es" || profile == "ES") {
|
||||
info.profile = MobileGL::ShaderProfile::ES;
|
||||
} else if (profile == "compatibility") {
|
||||
info.profile = MobileGL::ShaderProfile::Compatibility;
|
||||
} else {
|
||||
// "#version 330 foo": an unrecognized profile keyword. Keep Core for any
|
||||
// downstream routing, but mark the directive malformed.
|
||||
info.profile = MobileGL::ShaderProfile::Core;
|
||||
profileTokenValid = false;
|
||||
}
|
||||
// Comments are already masked to spaces, so anything non-blank left on the
|
||||
// line is real trailing garbage: "#version 330 foobar" / "#version 330.0".
|
||||
SkipDirectiveWhitespace(code, probe, lineEnd);
|
||||
const bool hasTrailingTokens = probe < lineEnd;
|
||||
info.hasValidVersionDirective =
|
||||
IsRecognizedGlslVersion(info.version) && profileTokenValid && !hasTrailingTokens;
|
||||
}
|
||||
} else if (directive == "extension") {
|
||||
SkipDirectiveWhitespace(code, probe, lineEnd);
|
||||
@@ -605,6 +708,17 @@ namespace {
|
||||
}
|
||||
|
||||
void NormalizeVersionDirective(MobileGL::String& source, const ShaderLanguageInfo& info) {
|
||||
// A malformed #version (329, 331, bad profile, float/trailing tokens) is left exactly as the
|
||||
// application wrote it so glslang rejects it - rewriting it to "#version 330 core" would
|
||||
// silently legalize the CTS directive.version_* rejection cases. Still drop a leading BOM so
|
||||
// the reported error is the bad version rather than a stray byte-order mark.
|
||||
if (info.HasVersionDirective() && !info.hasValidVersionDirective) {
|
||||
if (info.hasUtf8Bom) {
|
||||
source.erase(0, 3);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const MobileGL::String replacement = GetNormalizedVersionDirective(info);
|
||||
if (info.HasVersionDirective()) {
|
||||
source.replace(info.versionDirectiveStart, info.versionDirectiveEnd - info.versionDirectiveStart,
|
||||
@@ -686,7 +800,10 @@ namespace {
|
||||
|
||||
void RenameBuiltinShadowingFunction(MobileGL::String& source, const char* from, const char* to) {
|
||||
const MobileGL::String fromName = from;
|
||||
if (!HasSingleLineFunctionDefinition(source, fromName)) {
|
||||
// Decide from a comment-free view. A commented-out definition is not a definition, and
|
||||
// acting on one renames every genuine call to the builtin to a name nothing defines - which
|
||||
// then fails to resolve. Line comments survive BlankBlockComments, so this matters.
|
||||
if (!HasSingleLineFunctionDefinition(MaskCommentsAndQuotedText(source), fromName)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -759,6 +876,58 @@ namespace {
|
||||
return info.HasVersionDirective() ? info.versionDirectiveEnd : 0;
|
||||
}
|
||||
|
||||
// GLSL's #line takes integer expressions only, but plenty of shader-pack preprocessors emit the
|
||||
// C form with a quoted filename. Deleting every #line outright made those harmless - at the cost
|
||||
// of __LINE__ reporting the position in MobileGL's rewritten text rather than the one the pack
|
||||
// author wrote, and of every later diagnostic pointing at the wrong line. Dropping just the
|
||||
// quoted operand keeps the directive doing its job and still hands glslang something it accepts.
|
||||
void NormalizeLineDirectives(MobileGL::String& source) {
|
||||
const MobileGL::String masked = MaskCommentsAndQuotedText(source);
|
||||
const SizeT versionEnd = FindAfterVersionDirective(source);
|
||||
MobileGL::String result;
|
||||
result.reserve(source.size());
|
||||
|
||||
SizeT lineStart = 0;
|
||||
while (lineStart <= source.size()) {
|
||||
SizeT lineEnd = source.find('\n', lineStart);
|
||||
const bool lastLine = lineEnd == MobileGL::String::npos;
|
||||
if (lastLine) lineEnd = source.size();
|
||||
|
||||
SizeT probe = lineStart;
|
||||
while (probe < lineEnd && (source[probe] == ' ' || source[probe] == '\t')) probe++;
|
||||
|
||||
const bool isLineDirective = masked.compare(probe, 5, "#line") == 0 &&
|
||||
(probe + 5 >= lineEnd || !IsIdentifierChar(source[probe + 5]));
|
||||
if (isLineDirective && lineStart < versionEnd) {
|
||||
// #version has to be the first token in the shader, so a #line ahead of it could
|
||||
// never have taken effect. Drop it rather than hand glslang a source it must reject
|
||||
// - some pack preprocessors emit their directives before the version line.
|
||||
} else if (isLineDirective) {
|
||||
// Keep everything up to the first quote that the masker identified as string text.
|
||||
SizeT quotePos = MobileGL::String::npos;
|
||||
for (SizeT i = probe + 5; i < lineEnd; i++) {
|
||||
if (source[i] == '"' || source[i] == '\'') {
|
||||
quotePos = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (quotePos != MobileGL::String::npos) {
|
||||
result.append(source, lineStart, quotePos - lineStart);
|
||||
} else {
|
||||
result.append(source, lineStart, lineEnd - lineStart);
|
||||
}
|
||||
} else {
|
||||
result.append(source, lineStart, lineEnd - lineStart);
|
||||
}
|
||||
|
||||
if (lastLine) break;
|
||||
result.push_back('\n');
|
||||
lineStart = lineEnd + 1;
|
||||
}
|
||||
|
||||
source = std::move(result);
|
||||
}
|
||||
|
||||
bool IsExtensionAdvertised(MobileGL::GLExtension extension) {
|
||||
const auto& activeBackendObject = MobileGL::MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackendObject) {
|
||||
@@ -787,15 +956,28 @@ namespace {
|
||||
return;
|
||||
}
|
||||
|
||||
// Detect the directive on a comment/string-masked copy so a commented-out
|
||||
// "#extension GL_ARB_gpu_shader_int64" is never turned into a synthesized #error. Comments are
|
||||
// no longer blanked in the delivered source (glslang handles them), so this pass must mask
|
||||
// locally like its siblings. Masking preserves offsets, so edits collected against the scan
|
||||
// apply verbatim to `source`; they are applied back-to-front to keep earlier offsets valid.
|
||||
const MobileGL::String scan = MaskCommentsAndQuotedText(source);
|
||||
struct DirectiveEdit {
|
||||
SizeT pos;
|
||||
SizeT len;
|
||||
MobileGL::String replacement;
|
||||
};
|
||||
Vector<DirectiveEdit> edits;
|
||||
|
||||
SizeT lineStart = 0;
|
||||
while (lineStart < source.size()) {
|
||||
SizeT lineEnd = source.find('\n', lineStart);
|
||||
while (lineStart < scan.size()) {
|
||||
SizeT lineEnd = scan.find('\n', lineStart);
|
||||
const bool hasLineBreak = lineEnd != MobileGL::String::npos;
|
||||
if (!hasLineBreak) {
|
||||
lineEnd = source.size();
|
||||
lineEnd = scan.size();
|
||||
}
|
||||
|
||||
const MobileGL::String line = source.substr(lineStart, lineEnd - lineStart);
|
||||
const MobileGL::String line = scan.substr(lineStart, lineEnd - lineStart);
|
||||
SizeT probe = 0;
|
||||
while (probe < line.size() && std::isspace(static_cast<unsigned char>(line[probe]))) {
|
||||
probe++;
|
||||
@@ -837,16 +1019,12 @@ namespace {
|
||||
const MobileGL::String behavior = TrimDirectiveToken(line.substr(probe));
|
||||
const SizeT replaceLen = lineEnd - lineStart + (hasLineBreak ? 1 : 0);
|
||||
if (behavior == "require") {
|
||||
const MobileGL::String replacement =
|
||||
"#error GL_ARB_gpu_shader_int64 is not advertised by MobileGL\n";
|
||||
source.replace(lineStart, replaceLen, replacement);
|
||||
lineStart += replacement.size();
|
||||
edits.push_back({lineStart, replaceLen,
|
||||
"#error GL_ARB_gpu_shader_int64 is not advertised by MobileGL\n"});
|
||||
} else if (behavior == "enable" || behavior == "warn") {
|
||||
source.replace(lineStart, replaceLen, "\n");
|
||||
lineStart++;
|
||||
} else {
|
||||
lineStart = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
edits.push_back({lineStart, replaceLen, "\n"});
|
||||
}
|
||||
lineStart = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -856,6 +1034,10 @@ namespace {
|
||||
lineStart = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
}
|
||||
|
||||
for (auto it = edits.rbegin(); it != edits.rend(); ++it) {
|
||||
source.replace(it->pos, it->len, it->replacement);
|
||||
}
|
||||
|
||||
ReplaceIdentifier(source, "GL_ARB_gpu_shader_int64", "MG_DISABLED_GL_ARB_gpu_shader_int64");
|
||||
}
|
||||
|
||||
@@ -1078,42 +1260,22 @@ namespace MobileGL {
|
||||
const ShaderLanguageInfo originalLanguage = InspectShaderLanguage(source);
|
||||
NormalizeVersionDirective(source, originalLanguage);
|
||||
|
||||
// remove multi-line comment
|
||||
size_t commentStartPos = source.find("/*");
|
||||
while (commentStartPos != String::npos) {
|
||||
size_t commentEndPos = source.find("*/", commentStartPos);
|
||||
if (commentEndPos == String::npos) {
|
||||
source.erase(commentStartPos);
|
||||
break;
|
||||
}
|
||||
// + length of "*/"
|
||||
source = source.replace(commentStartPos, commentEndPos - commentStartPos + 2, "");
|
||||
commentStartPos = source.find("/*", commentStartPos);
|
||||
}
|
||||
// Comments are left intact for glslang's own preprocessor: a block comment is a single
|
||||
// preprocessing token that collapses to one space even across newlines and inside a
|
||||
// directive, so blanking it here (which preserved the interior newlines) truncated
|
||||
// multi-line #define bodies and broke otherwise-valid shaders (KHR-GL3x.shaders.
|
||||
// preprocessor multiline_comment_define / redefine_object / function_redefinition).
|
||||
// Every MobileGL pass that must ignore comment/string text already masks them locally
|
||||
// via MaskCommentsAndQuotedText/TokenizeCode, so the source we hand glslang keeps them.
|
||||
NormalizeLineDirectives(source);
|
||||
|
||||
// remove #line directives
|
||||
SizeT linedirPos = source.find("#line");
|
||||
while (linedirPos != String::npos) {
|
||||
SizeT newlinePos = source.find('\n', linedirPos);
|
||||
if (newlinePos == String::npos) {
|
||||
source.erase(linedirPos);
|
||||
break;
|
||||
}
|
||||
|
||||
// Preserve a line break so adjacent preprocessor directives do not merge.
|
||||
source = source.replace(linedirPos, newlinePos - linedirPos + 1, "\n");
|
||||
linedirPos = source.find("#line", linedirPos);
|
||||
}
|
||||
|
||||
// remove "noperspective"
|
||||
const char* str_np = "noperspective";
|
||||
const SizeT len_np = strlen(str_np);
|
||||
SizeT noperspectivePos = source.find(str_np);
|
||||
while (noperspectivePos != String::npos) {
|
||||
// + length of "\n"
|
||||
source = source.replace(noperspectivePos, len_np, "");
|
||||
noperspectivePos = source.find(str_np);
|
||||
}
|
||||
// noperspective is intentionally NOT touched here. It is core in desktop GLSL (1.30+)
|
||||
// and maps to the core SPIR-V NoPerspective decoration, which DirectVulkan renders
|
||||
// natively and SPIRV-Cross turns into ESSL `noperspective` + the
|
||||
// GL_NV_shader_noperspective_interpolation extension. The old naked substring erase
|
||||
// both discarded that interpolation (shader packs need it) and corrupted any
|
||||
// identifier that merely contained the word. The GLES fallback for devices without
|
||||
// the extension lives in the backend, where device capabilities are known.
|
||||
|
||||
FilterUnsupportedGpuShaderInt64(source);
|
||||
CoerceUniformBlockPackingToStd140(source);
|
||||
@@ -1137,6 +1299,10 @@ namespace MobileGL {
|
||||
// must not be mistaken for the real one.
|
||||
const ShaderLanguageInfo info = InspectShaderLanguage(source);
|
||||
if (!info.HasVersionDirective()) return false;
|
||||
// Never rescue a malformed directive to 460: that is precisely what re-legalized the
|
||||
// CTS directive.version_* rejection cases after the first compile failed. The shader-
|
||||
// pack retry this exists for only ever sees a valid low version (a real "#version 330").
|
||||
if (!info.hasValidVersionDirective) return false;
|
||||
// Only the set NormalizeVersionDirective downgraded: desktop core below 400. ES and
|
||||
// compatibility shaders keep whatever they declared.
|
||||
if (info.profile != ShaderProfile::Core || info.version >= 400) return false;
|
||||
|
||||
@@ -0,0 +1,407 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateNoPerspectivePass.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "EmulateNoPerspectivePass.h"
|
||||
|
||||
#include "spirv.hpp"
|
||||
#include "source/opt/constants.h"
|
||||
#include "source/opt/def_use_manager.h"
|
||||
#include "source/opt/instruction.h"
|
||||
#include "source/opt/ir_context.h"
|
||||
#include "source/opt/module.h"
|
||||
#include "source/opt/type_manager.h"
|
||||
#include "source/opt/types.h"
|
||||
#include "source/util/make_unique.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <vector>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
namespace {
|
||||
using spvtools::opt::Instruction;
|
||||
using spvtools::opt::IRContext;
|
||||
using spvtools::opt::Operand;
|
||||
namespace analysis = spvtools::opt::analysis;
|
||||
|
||||
spv::ExecutionModel EntryExecutionModel(IRContext* ctx) {
|
||||
for (Instruction& ep : ctx->module()->entry_points()) {
|
||||
return static_cast<spv::ExecutionModel>(ep.GetSingleWordInOperand(0));
|
||||
}
|
||||
return spv::ExecutionModel::Max;
|
||||
}
|
||||
|
||||
uint32_t VariablePointeeType(IRContext* ctx, Instruction* var) {
|
||||
Instruction* ptrType = ctx->get_def_use_mgr()->GetDef(var->type_id());
|
||||
// OpTypePointer <storage-class> <pointee>
|
||||
return ptrType->GetSingleWordInOperand(1);
|
||||
}
|
||||
|
||||
// If |typeId| is float or a vector of float, returns true and reports the scalar float
|
||||
// type and whether it is a vector. Matrices, structs, ints etc. are not emulatable.
|
||||
bool IsFloatScalarOrVector(IRContext* ctx, uint32_t typeId, uint32_t& floatTypeId, bool& isVector) {
|
||||
Instruction* t = ctx->get_def_use_mgr()->GetDef(typeId);
|
||||
if (t == nullptr) return false;
|
||||
if (t->opcode() == spv::Op::OpTypeFloat) {
|
||||
floatTypeId = typeId;
|
||||
isVector = false;
|
||||
return true;
|
||||
}
|
||||
if (t->opcode() == spv::Op::OpTypeVector) {
|
||||
const uint32_t comp = t->GetSingleWordInOperand(0);
|
||||
Instruction* ct = ctx->get_def_use_mgr()->GetDef(comp);
|
||||
if (ct != nullptr && ct->opcode() == spv::Op::OpTypeFloat) {
|
||||
floatTypeId = comp;
|
||||
isVector = true;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t PointerTypeTo(IRContext* ctx, uint32_t pointeeId, spv::StorageClass sc) {
|
||||
analysis::Type* pointee = ctx->get_type_mgr()->GetType(pointeeId);
|
||||
analysis::Pointer ptr(pointee, sc);
|
||||
return ctx->get_type_mgr()->GetTypeInstruction(&ptr);
|
||||
}
|
||||
|
||||
uint32_t V4FloatType(IRContext* ctx) {
|
||||
analysis::Float f(32);
|
||||
analysis::Type* freg = ctx->get_type_mgr()->GetRegisteredType(&f);
|
||||
analysis::Vector v(freg, 4);
|
||||
return ctx->get_type_mgr()->GetTypeInstruction(&v);
|
||||
}
|
||||
|
||||
uint32_t FloatType(IRContext* ctx) {
|
||||
analysis::Float f(32);
|
||||
return ctx->get_type_mgr()->GetTypeInstruction(&f);
|
||||
}
|
||||
|
||||
uint32_t SignedIntConstant(IRContext* ctx, int32_t value) {
|
||||
analysis::Integer i(32, true);
|
||||
analysis::Type* reg = ctx->get_type_mgr()->GetRegisteredType(&i);
|
||||
const analysis::Constant* c =
|
||||
ctx->get_constant_mgr()->GetConstant(reg, {static_cast<uint32_t>(value)});
|
||||
return ctx->get_constant_mgr()->GetDefiningInstruction(c)->result_id();
|
||||
}
|
||||
|
||||
// Multiply |valueId| (of type |valueTypeId|) by the scalar |scalarId|, inserting the op
|
||||
// before |before|. Returns the product's id.
|
||||
uint32_t InsertScale(IRContext* ctx, Instruction* before, uint32_t valueTypeId,
|
||||
uint32_t valueId, uint32_t scalarId, bool isVector) {
|
||||
const uint32_t productId = ctx->TakeNextId();
|
||||
const spv::Op op = isVector ? spv::Op::OpVectorTimesScalar : spv::Op::OpFMul;
|
||||
before->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, op, valueTypeId, productId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {valueId}},
|
||||
{SPV_OPERAND_TYPE_ID, {scalarId}}}));
|
||||
return productId;
|
||||
}
|
||||
|
||||
// --- Vertex stage: gl_Position discovery ------------------------------------------
|
||||
|
||||
// Finds gl_Position as member |memberIndex| of a gl_PerVertex-style block whose Output
|
||||
// variable is |blockVarId|; |v4floatTypeId| is that member's (vec4) type. Returns false
|
||||
// if gl_Position is not a block member (older plain-variable form is left to the strip).
|
||||
bool FindPositionBlock(IRContext* ctx, uint32_t& blockVarId, uint32_t& memberIndex,
|
||||
uint32_t& v4floatTypeId) {
|
||||
uint32_t structId = 0;
|
||||
uint32_t member = 0;
|
||||
for (Instruction& ann : ctx->annotations()) {
|
||||
if (ann.opcode() == spv::Op::OpMemberDecorate && ann.NumInOperands() >= 4 &&
|
||||
static_cast<spv::Decoration>(ann.GetSingleWordInOperand(2)) ==
|
||||
spv::Decoration::BuiltIn &&
|
||||
static_cast<spv::BuiltIn>(ann.GetSingleWordInOperand(3)) ==
|
||||
spv::BuiltIn::Position) {
|
||||
structId = ann.GetSingleWordInOperand(0);
|
||||
member = ann.GetSingleWordInOperand(1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (structId == 0) return false;
|
||||
|
||||
Instruction* structType = ctx->get_def_use_mgr()->GetDef(structId);
|
||||
if (structType == nullptr || member >= structType->NumInOperands()) return false;
|
||||
v4floatTypeId = structType->GetSingleWordInOperand(member);
|
||||
|
||||
for (Instruction& inst : ctx->module()->types_values()) {
|
||||
if (inst.opcode() == spv::Op::OpVariable &&
|
||||
static_cast<spv::StorageClass>(inst.GetSingleWordInOperand(0)) ==
|
||||
spv::StorageClass::Output &&
|
||||
VariablePointeeType(ctx, &inst) == structId) {
|
||||
blockVarId = inst.result_id();
|
||||
memberIndex = member;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// --- Fragment stage: gl_FragCoord discovery/synthesis -----------------------------
|
||||
|
||||
Instruction* FindBuiltinInput(IRContext* ctx, spv::BuiltIn builtin) {
|
||||
for (Instruction& ann : ctx->annotations()) {
|
||||
if (ann.opcode() != spv::Op::OpDecorate || ann.NumInOperands() < 3) continue;
|
||||
if (static_cast<spv::Decoration>(ann.GetSingleWordInOperand(1)) !=
|
||||
spv::Decoration::BuiltIn)
|
||||
continue;
|
||||
if (static_cast<spv::BuiltIn>(ann.GetSingleWordInOperand(2)) != builtin) continue;
|
||||
Instruction* var = ctx->get_def_use_mgr()->GetDef(ann.GetSingleWordInOperand(0));
|
||||
if (var != nullptr && var->opcode() == spv::Op::OpVariable &&
|
||||
static_cast<spv::StorageClass>(var->GetSingleWordInOperand(0)) ==
|
||||
spv::StorageClass::Input) {
|
||||
return var;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
uint32_t SynthesizeFragCoord(IRContext* ctx, uint32_t v4floatTypeId) {
|
||||
const uint32_t ptrType = PointerTypeTo(ctx, v4floatTypeId, spv::StorageClass::Input);
|
||||
const uint32_t varId = ctx->TakeNextId();
|
||||
ctx->AddGlobalValue(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpVariable, ptrType, varId,
|
||||
std::initializer_list<Operand>{
|
||||
{SPV_OPERAND_TYPE_STORAGE_CLASS,
|
||||
{static_cast<uint32_t>(spv::StorageClass::Input)}}}));
|
||||
ctx->AddAnnotationInst(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpDecorate, 0, 0,
|
||||
std::initializer_list<Operand>{
|
||||
{SPV_OPERAND_TYPE_ID, {varId}},
|
||||
{SPV_OPERAND_TYPE_DECORATION,
|
||||
{static_cast<uint32_t>(spv::Decoration::BuiltIn)}},
|
||||
{SPV_OPERAND_TYPE_LITERAL_INTEGER,
|
||||
{static_cast<uint32_t>(spv::BuiltIn::FragCoord)}}}));
|
||||
for (Instruction& ep : ctx->module()->entry_points()) {
|
||||
ep.AddOperand({SPV_OPERAND_TYPE_ID, {varId}});
|
||||
}
|
||||
return varId;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
spvtools::opt::Pass::Status EmulateNoPerspectivePass::Process() {
|
||||
auto* ctx = context();
|
||||
const spv::ExecutionModel model = EntryExecutionModel(ctx);
|
||||
const bool isVertex = model == spv::ExecutionModel::Vertex;
|
||||
const bool isFragment = model == spv::ExecutionModel::Fragment;
|
||||
|
||||
// Collect NoPerspective-decorated plain variables and every NoPerspective annotation.
|
||||
std::vector<uint32_t> plainVarIds;
|
||||
std::vector<Instruction*> decorationsToKill;
|
||||
for (Instruction& ann : ctx->annotations()) {
|
||||
if (ann.opcode() == spv::Op::OpDecorate && ann.NumInOperands() >= 2 &&
|
||||
static_cast<spv::Decoration>(ann.GetSingleWordInOperand(1)) ==
|
||||
spv::Decoration::NoPerspective) {
|
||||
plainVarIds.push_back(ann.GetSingleWordInOperand(0));
|
||||
decorationsToKill.push_back(&ann);
|
||||
} else if (ann.opcode() == spv::Op::OpMemberDecorate && ann.NumInOperands() >= 3 &&
|
||||
static_cast<spv::Decoration>(ann.GetSingleWordInOperand(2)) ==
|
||||
spv::Decoration::NoPerspective) {
|
||||
// Block-member noperspective is not emulated here; the decoration is stripped
|
||||
// (smooth fallback) so SPIRV-Cross does not require the NV extension.
|
||||
decorationsToKill.push_back(&ann);
|
||||
}
|
||||
}
|
||||
|
||||
if (decorationsToKill.empty()) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
|
||||
const spv::StorageClass wantStorage =
|
||||
isVertex ? spv::StorageClass::Output : spv::StorageClass::Input;
|
||||
|
||||
// Emulatable = plain variable of the stage's interface direction, float or floatN.
|
||||
struct Target {
|
||||
Instruction* var;
|
||||
uint32_t typeId;
|
||||
uint32_t floatTypeId;
|
||||
bool isVector;
|
||||
};
|
||||
std::vector<Target> targets;
|
||||
if (isVertex || isFragment) {
|
||||
for (const uint32_t id : plainVarIds) {
|
||||
Instruction* var = ctx->get_def_use_mgr()->GetDef(id);
|
||||
if (var == nullptr || var->opcode() != spv::Op::OpVariable) continue;
|
||||
if (static_cast<spv::StorageClass>(var->GetSingleWordInOperand(0)) != wantStorage)
|
||||
continue;
|
||||
const uint32_t pointee = VariablePointeeType(ctx, var);
|
||||
uint32_t floatTypeId = 0;
|
||||
bool isVector = false;
|
||||
if (IsFloatScalarOrVector(ctx, pointee, floatTypeId, isVector)) {
|
||||
targets.push_back({var, pointee, floatTypeId, isVector});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Force highp on the varyings we emulate: the a*w round-trip overflows a mediump (fp16)
|
||||
// varying at large clip-space w. Dropping RelaxedPrecision makes SPIRV-Cross emit them
|
||||
// highp on both stages, keeping the emulation exact. Only touches emulated variables.
|
||||
if (!targets.empty()) {
|
||||
std::vector<uint32_t> targetIds;
|
||||
targetIds.reserve(targets.size());
|
||||
for (const Target& t : targets) targetIds.push_back(t.var->result_id());
|
||||
for (Instruction& ann : ctx->annotations()) {
|
||||
if (ann.opcode() == spv::Op::OpDecorate && ann.NumInOperands() >= 2 &&
|
||||
static_cast<spv::Decoration>(ann.GetSingleWordInOperand(1)) ==
|
||||
spv::Decoration::RelaxedPrecision &&
|
||||
std::find(targetIds.begin(), targetIds.end(),
|
||||
ann.GetSingleWordInOperand(0)) != targetIds.end()) {
|
||||
decorationsToKill.push_back(&ann);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (isVertex && !targets.empty()) {
|
||||
uint32_t blockVarId = 0;
|
||||
uint32_t memberIndex = 0;
|
||||
uint32_t v4floatTypeId = 0;
|
||||
if (FindPositionBlock(ctx, blockVarId, memberIndex, v4floatTypeId)) {
|
||||
const uint32_t ptrOutV4 =
|
||||
PointerTypeTo(ctx, v4floatTypeId, spv::StorageClass::Output);
|
||||
const uint32_t memberConst = SignedIntConstant(ctx, static_cast<int32_t>(memberIndex));
|
||||
const uint32_t floatTy = FloatType(ctx);
|
||||
|
||||
uint32_t entryFuncId = 0;
|
||||
for (Instruction& ep : ctx->module()->entry_points()) {
|
||||
// OpEntryPoint <model> <function> "name" <interface...>
|
||||
entryFuncId = ep.GetSingleWordInOperand(1);
|
||||
break;
|
||||
}
|
||||
|
||||
// Pre-multiply every target output by gl_Position.w before each return of the
|
||||
// ENTRY function only. glslang does not inline, so a called helper survives as
|
||||
// its own OpFunction; instrumenting its returns too would scale the varying
|
||||
// more than once (w^2), breaking the identity.
|
||||
for (auto funcIt = ctx->module()->begin(); funcIt != ctx->module()->end(); ++funcIt) {
|
||||
if (funcIt->result_id() != entryFuncId) continue;
|
||||
funcIt->ForEachInst([&](Instruction* inst) {
|
||||
if (inst->opcode() != spv::Op::OpReturn &&
|
||||
inst->opcode() != spv::Op::OpReturnValue) {
|
||||
return;
|
||||
}
|
||||
const uint32_t posPtrId = ctx->TakeNextId();
|
||||
inst->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpAccessChain, ptrOutV4, posPtrId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {blockVarId}},
|
||||
{SPV_OPERAND_TYPE_ID, {memberConst}}}));
|
||||
const uint32_t posId = ctx->TakeNextId();
|
||||
inst->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpLoad, v4floatTypeId, posId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {posPtrId}}}));
|
||||
const uint32_t wId = ctx->TakeNextId();
|
||||
inst->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpCompositeExtract, floatTy, wId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {posId}},
|
||||
{SPV_OPERAND_TYPE_LITERAL_INTEGER, {3u}}}));
|
||||
for (const Target& t : targets) {
|
||||
const uint32_t valId = ctx->TakeNextId();
|
||||
inst->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpLoad, t.typeId, valId,
|
||||
std::initializer_list<Operand>{
|
||||
{SPV_OPERAND_TYPE_ID, {t.var->result_id()}}}));
|
||||
const uint32_t scaledId =
|
||||
InsertScale(ctx, inst, t.typeId, valId, wId, t.isVector);
|
||||
inst->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpStore, 0, 0,
|
||||
std::initializer_list<Operand>{
|
||||
{SPV_OPERAND_TYPE_ID, {t.var->result_id()}},
|
||||
{SPV_OPERAND_TYPE_ID, {scaledId}}}));
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (isFragment && !targets.empty()) {
|
||||
Instruction* fragCoord = FindBuiltinInput(ctx, spv::BuiltIn::FragCoord);
|
||||
uint32_t fragCoordId = 0;
|
||||
uint32_t v4floatTypeId = 0;
|
||||
if (fragCoord != nullptr) {
|
||||
fragCoordId = fragCoord->result_id();
|
||||
v4floatTypeId = VariablePointeeType(ctx, fragCoord);
|
||||
} else {
|
||||
v4floatTypeId = V4FloatType(ctx);
|
||||
fragCoordId = SynthesizeFragCoord(ctx, v4floatTypeId);
|
||||
}
|
||||
const uint32_t floatTy = FloatType(ctx);
|
||||
|
||||
auto* defUse = ctx->get_def_use_mgr();
|
||||
for (const Target& t : targets) {
|
||||
// Collect every load that reads the varying. glslang lowers a whole-variable
|
||||
// read to OpLoad(var), but a single-component read (v.x) to
|
||||
// OpAccessChain(var) + OpLoad(chain). Both must be scaled; the identity is
|
||||
// per-component, so scaling one loaded component by gl_FragCoord.w is valid.
|
||||
std::vector<Instruction*> loads;
|
||||
defUse->ForEachUser(t.var, [&](Instruction* user) {
|
||||
if (user->opcode() == spv::Op::OpLoad &&
|
||||
user->GetSingleWordInOperand(0) == t.var->result_id()) {
|
||||
loads.push_back(user);
|
||||
} else if (user->opcode() == spv::Op::OpAccessChain &&
|
||||
user->GetSingleWordInOperand(0) == t.var->result_id()) {
|
||||
const uint32_t chainId = user->result_id();
|
||||
defUse->ForEachUser(user, [&](Instruction* chainUser) {
|
||||
if (chainUser->opcode() == spv::Op::OpLoad &&
|
||||
chainUser->GetSingleWordInOperand(0) == chainId) {
|
||||
loads.push_back(chainUser);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// Rewrite `%r = OpLoad %ty %ptr` into
|
||||
// %orig = OpLoad %ty %ptr
|
||||
// %fc = OpLoad %v4float %fragCoord
|
||||
// %w = OpCompositeExtract %float %fc 3
|
||||
// %r = OpVectorTimesScalar/OpFMul %ty %orig %w (reuse %r: uses stay intact)
|
||||
// The op is chosen from the LOAD's own result type: a whole-vector load scales
|
||||
// with OpVectorTimesScalar, a scalar component load with OpFMul.
|
||||
for (Instruction* load : loads) {
|
||||
const uint32_t loadType = load->type_id();
|
||||
uint32_t componentFloat = 0;
|
||||
bool loadIsVector = false;
|
||||
if (!IsFloatScalarOrVector(ctx, loadType, componentFloat, loadIsVector)) {
|
||||
continue;
|
||||
}
|
||||
const uint32_t ptrId = load->GetSingleWordInOperand(0);
|
||||
const uint32_t origId = ctx->TakeNextId();
|
||||
load->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpLoad, loadType, origId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {ptrId}}}));
|
||||
const uint32_t fcId = ctx->TakeNextId();
|
||||
load->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpLoad, v4floatTypeId, fcId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {fragCoordId}}}));
|
||||
const uint32_t wId = ctx->TakeNextId();
|
||||
load->InsertBefore(spvtools::MakeUnique<Instruction>(
|
||||
ctx, spv::Op::OpCompositeExtract, floatTy, wId,
|
||||
std::initializer_list<Operand>{{SPV_OPERAND_TYPE_ID, {fcId}},
|
||||
{SPV_OPERAND_TYPE_LITERAL_INTEGER, {3u}}}));
|
||||
load->SetOpcode(loadIsVector ? spv::Op::OpVectorTimesScalar : spv::Op::OpFMul);
|
||||
load->SetInOperands(Instruction::OperandList{
|
||||
{SPV_OPERAND_TYPE_ID, {origId}}, {SPV_OPERAND_TYPE_ID, {wId}}});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Strip every NoPerspective decoration: emulated varyings now transport smooth, and
|
||||
// non-emulatable ones fall back to smooth.
|
||||
for (Instruction* dec : decorationsToKill) {
|
||||
ctx->KillInst(dec);
|
||||
}
|
||||
|
||||
ctx->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
spvtools::Optimizer::PassToken EmulateNoPerspectivePass::CreateEmulateNoPerspectivePass() {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<EmulateNoPerspectivePass>());
|
||||
}
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
@@ -0,0 +1,41 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateNoPerspectivePass.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include "source/opt/pass.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
// Emulates 'noperspective' (screen-linear) interpolation on GLES devices that lack
|
||||
// GL_NV_shader_noperspective_interpolation, so no NV extension is required. The hardware
|
||||
// interpolates perspective-correct; screen-linear L(a) is recovered from the identity
|
||||
// L(a) = P(a * w) * gl_FragCoord.w
|
||||
// where P is perspective-correct interpolation and w is the vertex clip-space w. So each
|
||||
// NoPerspective-decorated output is pre-multiplied by gl_Position.w in the vertex stage
|
||||
// and each NoPerspective-decorated input is multiplied by gl_FragCoord.w in the fragment
|
||||
// stage; the decoration is then removed so the varying transports smooth. This is exact
|
||||
// (modulo float precision - the emulated varyings want highp).
|
||||
//
|
||||
// Scope: plain interface variables of float or floatN type. Anything it cannot emulate
|
||||
// (interface-block members, matrices, or a stage lacking the needed builtin) has its
|
||||
// NoPerspective decoration stripped instead, degrading to smooth - the same result the
|
||||
// extension-less fallback produced before, and never invalid SPIR-V. DirectGLES only.
|
||||
class EmulateNoPerspectivePass : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "emulate-noperspective"; }
|
||||
Status Process() override;
|
||||
|
||||
static spvtools::Optimizer::PassToken CreateEmulateNoPerspectivePass();
|
||||
};
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
@@ -0,0 +1,72 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripNoPerspectivePass.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "StripNoPerspectivePass.h"
|
||||
|
||||
#include "spirv.hpp"
|
||||
#include "source/opt/instruction.h"
|
||||
#include "source/opt/ir_context.h"
|
||||
#include "source/util/make_unique.h"
|
||||
|
||||
#include <vector>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
namespace {
|
||||
using spvtools::opt::Instruction;
|
||||
using spvtools::opt::IRContext;
|
||||
|
||||
// OpDecorate <target-id> <decoration> [literals...]
|
||||
// OpMemberDecorate <struct-id> <member> <decoration> [literals...]
|
||||
constexpr uint32_t kDecorateDecorationOperand = 1;
|
||||
constexpr uint32_t kMemberDecorateDecorationOperand = 2;
|
||||
} // namespace
|
||||
|
||||
spvtools::opt::Pass::Status StripNoPerspectivePass::Process() {
|
||||
auto* irContext = context();
|
||||
|
||||
// Collect first: KillInst mutates the annotation list being walked.
|
||||
std::vector<Instruction*> toKill;
|
||||
for (Instruction& annotation : irContext->annotations()) {
|
||||
uint32_t decorationOperand = 0;
|
||||
if (annotation.opcode() == spv::Op::OpDecorate) {
|
||||
decorationOperand = kDecorateDecorationOperand;
|
||||
} else if (annotation.opcode() == spv::Op::OpMemberDecorate) {
|
||||
decorationOperand = kMemberDecorateDecorationOperand;
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (annotation.NumInOperands() <= decorationOperand) {
|
||||
continue;
|
||||
}
|
||||
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(decorationOperand)) ==
|
||||
spv::Decoration::NoPerspective) {
|
||||
toKill.push_back(&annotation);
|
||||
}
|
||||
}
|
||||
|
||||
if (toKill.empty()) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
|
||||
for (Instruction* inst : toKill) {
|
||||
irContext->KillInst(inst);
|
||||
}
|
||||
|
||||
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
spvtools::Optimizer::PassToken StripNoPerspectivePass::CreateStripNoPerspectivePass() {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<StripNoPerspectivePass>());
|
||||
}
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
@@ -0,0 +1,35 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripNoPerspectivePass.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include "source/opt/pass.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
// Removes the NoPerspective decoration from every interface variable and block member.
|
||||
// DirectGLES fallback only, for devices that lack GL_NV_shader_noperspective_interpolation:
|
||||
// SPIRV-Cross renders a NoPerspective-decorated varying as ESSL `noperspective` plus
|
||||
// `#extension GL_NV_shader_noperspective_interpolation : require`, which such a driver
|
||||
// rejects. Dropping the decoration falls the varying back to smooth (perspective-correct)
|
||||
// interpolation - the same visible result the old text-level strip produced, but without
|
||||
// corrupting identifiers and without touching DirectVulkan, where NoPerspective is native.
|
||||
// (The exact screen-linear emulation via gl_Position.w / gl_FragCoord.w is a later step.)
|
||||
class StripNoPerspectivePass : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "strip-noperspective"; }
|
||||
Status Process() override;
|
||||
|
||||
static spvtools::Optimizer::PassToken CreateStripNoPerspectivePass();
|
||||
};
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
+1
-1
Submodule include/FastSTL updated: 34f55f9df2...022211c998
@@ -0,0 +1,85 @@
|
||||
# Running the OpenGL CTS (VK-GL-CTS / KHR-GL33) against MobileGL on Android
|
||||
|
||||
Goal: measure how much of the OpenGL 3.3 core-profile conformance suite MobileGL
|
||||
passes, separately for each backend (`DirectGLES`, `DirectVulkan`).
|
||||
|
||||
## How MobileGL is reached from a test binary
|
||||
|
||||
MobileGL ships its own EGL implementation alongside its desktop-GL implementation
|
||||
in a single `libMobileGL.so`. A plain arm64 ELF in `/data/local/tmp` can therefore
|
||||
drive it with no APK and no Activity:
|
||||
|
||||
1. `setenv("MOBILEGL_BACKEND_TYPE", "DirectGLES"|"DirectVulkan")` **before** the
|
||||
library is mapped — MobileGL parses its configuration from an ELF constructor.
|
||||
2. `dlopen("libMobileGL.so")`, then `dlsym` the `egl*` and `gl*` entry points.
|
||||
MobileGL exports 45 EGL symbols and the desktop GL functions directly;
|
||||
`eglGetProcAddress` resolves the same set.
|
||||
3. `eglBindAPI(EGL_OPENGL_API)`, choose a config with `EGL_RENDERABLE_TYPE =
|
||||
EGL_OPENGL_BIT`, then `eglCreateContext` with
|
||||
`EGL_CONTEXT_OPENGL_PROFILE_MASK = EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT` and
|
||||
major/minor `3`/`3`.
|
||||
|
||||
This yields a genuine GL 3.3 core context (`GL_CONTEXT_PROFILE_MASK == 0x1`).
|
||||
|
||||
## Surface type, per backend
|
||||
|
||||
| backend | pbuffer (headless) | window |
|
||||
|---|---|---|
|
||||
| `DirectGLES` | works | works |
|
||||
| `DirectVulkan` | **unusable** | works |
|
||||
|
||||
`DirectVulkan`'s pbuffer path builds a headless `VkSurfaceKHR` and so requires the
|
||||
`VK_EXT_headless_surface` instance extension, which Adreno's Android driver does
|
||||
not expose. It fails inside `eglMakeCurrent`, not at surface creation.
|
||||
|
||||
The workaround that keeps everything in a shell process: obtain a real
|
||||
`ANativeWindow` from **`AImageReader`** (`AImageReader_newWithUsage` +
|
||||
`AImageReader_getWindow`). It is an ordinary BufferQueue producer, so
|
||||
`vkCreateAndroidSurfaceKHR` accepts it, and no Activity is involved. Register an
|
||||
`onImageAvailable` listener that acquires and deletes each image — otherwise the
|
||||
producer blocks once `maxImages` buffers are in flight and the next swap hangs.
|
||||
|
||||
## Why the suite must render into an FBO
|
||||
|
||||
On a window surface, `DirectVulkan`'s `glReadPixels` from the **default
|
||||
framebuffer** returns all zeros, with no GL error, both before and after
|
||||
`eglSwapBuffers`. `DirectGLES` on the identical window is correct, and readback
|
||||
from a **user FBO is correct on both backends**.
|
||||
|
||||
Verified on two SoCs and two drivers, so this is MobileGL's behaviour rather than
|
||||
a driver quirk:
|
||||
|
||||
| device | GPU | driver | default-FB | user FBO |
|
||||
|---|---|---|---|---|
|
||||
| Xiaomi 24129PN74C | Adreno 830 | Vulkan 1.3.284 / 512.800.46 | zeros | ok |
|
||||
| Lenovo TB321FU | Adreno 750 | Vulkan 1.3.128 / 512.762.28 | zeros | ok |
|
||||
|
||||
dEQP verifies nearly every case through `glReadPixels`, so running it against the
|
||||
default framebuffer would score `DirectVulkan` near zero for a reason unrelated to
|
||||
conformance. The runs therefore use `--deqp-surface-type=fbo`, uniformly for both
|
||||
backends so the two numbers stay comparable.
|
||||
|
||||
## Other constraints the harness must respect
|
||||
|
||||
- `eglMakeCurrent` requires **draw == read** and rejects `EGL_NO_SURFACE` with
|
||||
`EGL_BAD_MATCH`. dEQP's `surfaceless` platform is therefore unusable, which is
|
||||
why this port supplies its own `tcu::Platform`.
|
||||
- MobileGL aborts during static teardown (`FORTIFY: pthread_mutex_lock called on a
|
||||
destroyed mutex`) *after* all work completes. Flush and `_exit()` so the exit
|
||||
code and the `.qpa` log survive.
|
||||
|
||||
## Contents
|
||||
|
||||
probe/mgprobe.c preflight gate: one backend x one surface type, checks
|
||||
context version/profile and both readback paths
|
||||
scripts/qpa_report.py .qpa -> pass rate, status histogram, worst groups
|
||||
|
||||
### Preflight
|
||||
|
||||
aarch64-linux-android26-clang -O1 -o mgprobe mgprobe.c -ldl -llog -landroid -lmediandk
|
||||
adb push mgprobe libMobileGL.so /data/local/tmp/mgcts/
|
||||
adb shell 'cd /data/local/tmp/mgcts && LD_LIBRARY_PATH=. ./mgprobe \
|
||||
--backend DirectVulkan --surface imagereader --lib ./libMobileGL.so'
|
||||
|
||||
Exit status is 0 when a 3.3 core context came up and FBO readback is correct.
|
||||
Default-framebuffer readback is reported but deliberately does not gate.
|
||||
@@ -0,0 +1,109 @@
|
||||
diff --git a/framework/opengl/gluFboRenderContext.cpp b/framework/opengl/gluFboRenderContext.cpp
|
||||
index 588cf7d2a..0721ffee7 100644
|
||||
--- a/framework/opengl/gluFboRenderContext.cpp
|
||||
+++ b/framework/opengl/gluFboRenderContext.cpp
|
||||
@@ -132,6 +132,7 @@ FboRenderContext::FboRenderContext(RenderContext *context, const RenderConfig &c
|
||||
: m_context(context)
|
||||
, m_framebuffer(0)
|
||||
, m_colorBuffer(0)
|
||||
+ , m_colorIsTexture(false)
|
||||
, m_depthStencilBuffer(0)
|
||||
, m_renderTarget()
|
||||
{
|
||||
@@ -151,6 +152,7 @@ FboRenderContext::FboRenderContext(const ContextFactory &factory, const RenderCo
|
||||
: m_context(nullptr)
|
||||
, m_framebuffer(0)
|
||||
, m_colorBuffer(0)
|
||||
+ , m_colorIsTexture(false)
|
||||
, m_depthStencilBuffer(0)
|
||||
, m_renderTarget()
|
||||
{
|
||||
@@ -215,19 +217,41 @@ void FboRenderContext::createFramebuffer(const RenderConfig &config)
|
||||
height = (height == glu::RenderConfig::DONT_CARE) ? maxSize : height;
|
||||
}
|
||||
|
||||
+ // MOBILEGL: allow the colour attachment to be a texture instead of a
|
||||
+ // renderbuffer. MobileGL's DirectVulkan backend returns zeros when reading
|
||||
+ // back a renderbuffer-attached FBO, which makes every image comparison fail
|
||||
+ // for one reason and hides everything else. Setting
|
||||
+ // MOBILEGL_CTS_FBO_COLOR_TEXTURE=1 isolates that single defect so the rest
|
||||
+ // of the suite can be measured. Off by default: stock behaviour.
|
||||
{
|
||||
- pixelFormat = getPixelFormat(colorFormat);
|
||||
+ const char *useTexEnv = getenv("MOBILEGL_CTS_FBO_COLOR_TEXTURE");
|
||||
+ m_colorIsTexture = (useTexEnv && useTexEnv[0] == '1' && config.numSamples <= 0);
|
||||
|
||||
- gl.genRenderbuffers(1, &m_colorBuffer);
|
||||
- gl.bindRenderbuffer(GL_RENDERBUFFER, m_colorBuffer);
|
||||
+ pixelFormat = getPixelFormat(colorFormat);
|
||||
|
||||
- if (config.numSamples > 0)
|
||||
- gl.renderbufferStorageMultisample(GL_RENDERBUFFER, config.numSamples, colorFormat, width, height);
|
||||
+ if (m_colorIsTexture)
|
||||
+ {
|
||||
+ gl.genTextures(1, &m_colorBuffer);
|
||||
+ gl.bindTexture(GL_TEXTURE_2D, m_colorBuffer);
|
||||
+ gl.texStorage2D(GL_TEXTURE_2D, 1, colorFormat, width, height);
|
||||
+ gl.texParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
|
||||
+ gl.texParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
+ gl.bindTexture(GL_TEXTURE_2D, 0);
|
||||
+ GLU_EXPECT_NO_ERROR(gl.getError(), "Creating color texture");
|
||||
+ }
|
||||
else
|
||||
- gl.renderbufferStorage(GL_RENDERBUFFER, colorFormat, width, height);
|
||||
-
|
||||
- gl.bindRenderbuffer(GL_RENDERBUFFER, 0);
|
||||
- GLU_EXPECT_NO_ERROR(gl.getError(), "Creating color renderbuffer");
|
||||
+ {
|
||||
+ gl.genRenderbuffers(1, &m_colorBuffer);
|
||||
+ gl.bindRenderbuffer(GL_RENDERBUFFER, m_colorBuffer);
|
||||
+
|
||||
+ if (config.numSamples > 0)
|
||||
+ gl.renderbufferStorageMultisample(GL_RENDERBUFFER, config.numSamples, colorFormat, width, height);
|
||||
+ else
|
||||
+ gl.renderbufferStorage(GL_RENDERBUFFER, colorFormat, width, height);
|
||||
+
|
||||
+ gl.bindRenderbuffer(GL_RENDERBUFFER, 0);
|
||||
+ GLU_EXPECT_NO_ERROR(gl.getError(), "Creating color renderbuffer");
|
||||
+ }
|
||||
}
|
||||
|
||||
if (depthStencilFormat != GL_NONE)
|
||||
@@ -250,7 +274,12 @@ void FboRenderContext::createFramebuffer(const RenderConfig &config)
|
||||
gl.bindFramebuffer(GL_FRAMEBUFFER, m_framebuffer);
|
||||
|
||||
if (m_colorBuffer)
|
||||
- gl.framebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, m_colorBuffer);
|
||||
+ {
|
||||
+ if (m_colorIsTexture)
|
||||
+ gl.framebufferTexture2D(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, m_colorBuffer, 0);
|
||||
+ else
|
||||
+ gl.framebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, m_colorBuffer);
|
||||
+ }
|
||||
|
||||
if (m_depthStencilBuffer)
|
||||
{
|
||||
@@ -290,7 +319,10 @@ void FboRenderContext::destroyFramebuffer(void)
|
||||
|
||||
if (m_colorBuffer)
|
||||
{
|
||||
- gl.deleteRenderbuffers(1, &m_colorBuffer);
|
||||
+ if (m_colorIsTexture)
|
||||
+ gl.deleteTextures(1, &m_colorBuffer);
|
||||
+ else
|
||||
+ gl.deleteRenderbuffers(1, &m_colorBuffer);
|
||||
m_colorBuffer = 0;
|
||||
}
|
||||
}
|
||||
diff --git a/framework/opengl/gluFboRenderContext.hpp b/framework/opengl/gluFboRenderContext.hpp
|
||||
index 75a0ff6b7..09ff1e7a9 100644
|
||||
--- a/framework/opengl/gluFboRenderContext.hpp
|
||||
+++ b/framework/opengl/gluFboRenderContext.hpp
|
||||
@@ -80,6 +80,7 @@ private:
|
||||
RenderContext *m_context;
|
||||
uint32_t m_framebuffer;
|
||||
uint32_t m_colorBuffer;
|
||||
+ bool m_colorIsTexture;
|
||||
uint32_t m_depthStencilBuffer;
|
||||
tcu::RenderTarget m_renderTarget;
|
||||
};
|
||||
@@ -0,0 +1,473 @@
|
||||
/*-------------------------------------------------------------------------
|
||||
* dEQP platform port for MobileGL on Android
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*
|
||||
*//*!
|
||||
* \file
|
||||
* \brief MobileGL platform.
|
||||
*
|
||||
* Modelled on the surfaceless platform, but adapted to MobileGL, which ships
|
||||
* its own EGL implementation inside libMobileGL.so:
|
||||
*
|
||||
* - Every EGL call goes through the dynamically loaded library. The
|
||||
* surfaceless port mixes wrapper calls with globally linked egl* symbols;
|
||||
* doing that here would silently reach Android's system EGL instead.
|
||||
* - Desktop-GL configs are selected with EGL_OPENGL_BIT. The surfaceless port
|
||||
* always asks for an ES bit, which cannot satisfy a GL 3.3 core context.
|
||||
* - A real surface is always created. MobileGL rejects EGL_NO_SURFACE with
|
||||
* EGL_BAD_MATCH, and --deqp-surface-type=fbo asks the platform for
|
||||
* SURFACETYPE_DONT_CARE, so "no surface" is not an option.
|
||||
* - Window surfaces are backed by an AImageReader rather than an Activity,
|
||||
* which is what lets the suite run as a plain adb-shell binary. DirectVulkan
|
||||
* needs this: its pbuffer path requires VK_EXT_headless_surface, which
|
||||
* Adreno's Android driver does not expose.
|
||||
*
|
||||
* Environment:
|
||||
* MOBILEGL_CTS_LIB path/soname of the MobileGL library (default libMobileGL.so)
|
||||
* MOBILEGL_CTS_SURFACE "window" (default) or "pbuffer"
|
||||
* MOBILEGL_BACKEND_TYPE read by MobileGL itself; set it before launching
|
||||
*//*--------------------------------------------------------------------*/
|
||||
|
||||
#include "tcuMobileGLPlatform.hpp"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "deDynamicLibrary.hpp"
|
||||
#include "egluUtil.hpp"
|
||||
#include "eglwEnums.hpp"
|
||||
#include "eglwLibrary.hpp"
|
||||
#include "gluPlatform.hpp"
|
||||
#include "gluRenderConfig.hpp"
|
||||
#include "gluRenderContext.hpp"
|
||||
#include "glwInitFunctions.hpp"
|
||||
#include "tcuCommandLine.hpp"
|
||||
#include "tcuPixelFormat.hpp"
|
||||
#include "tcuPlatform.hpp"
|
||||
#include "tcuRenderTarget.hpp"
|
||||
|
||||
#include <android/hardware_buffer.h>
|
||||
#include <android/native_window.h>
|
||||
#include <media/NdkImageReader.h>
|
||||
|
||||
using std::string;
|
||||
using std::vector;
|
||||
|
||||
#if !defined(EGL_CONTEXT_OPENGL_PROFILE_MASK_KHR)
|
||||
#define EGL_CONTEXT_FLAGS_KHR 0x30FC
|
||||
#define EGL_CONTEXT_MAJOR_VERSION_KHR 0x3098
|
||||
#define EGL_CONTEXT_MINOR_VERSION_KHR 0x30FB
|
||||
#define EGL_CONTEXT_OPENGL_COMPATIBILITY_PROFILE_BIT_KHR 0x00000002
|
||||
#define EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT_KHR 0x00000001
|
||||
#define EGL_CONTEXT_OPENGL_DEBUG_BIT_KHR 0x00000001
|
||||
#define EGL_CONTEXT_OPENGL_FORWARD_COMPATIBLE_BIT_KHR 0x00000002
|
||||
#define EGL_CONTEXT_OPENGL_PROFILE_MASK_KHR 0x30FD
|
||||
#define EGL_CONTEXT_OPENGL_ROBUST_ACCESS_BIT_KHR 0x00000004
|
||||
#endif
|
||||
|
||||
namespace tcu
|
||||
{
|
||||
namespace mobilegl
|
||||
{
|
||||
|
||||
static string getLibraryName(void)
|
||||
{
|
||||
const char *env = std::getenv("MOBILEGL_CTS_LIB");
|
||||
return (env && env[0]) ? string(env) : string("libMobileGL.so");
|
||||
}
|
||||
|
||||
//! Window surfaces default on: they are the only kind DirectVulkan can use.
|
||||
static bool useWindowSurface(void)
|
||||
{
|
||||
const char *env = std::getenv("MOBILEGL_CTS_SURFACE");
|
||||
return !(env && string(env) == "pbuffer");
|
||||
}
|
||||
|
||||
/*--------------------------------------------------------------------*//*!
|
||||
* \brief A real ANativeWindow with no Activity behind it.
|
||||
*
|
||||
* AImageReader's window is an ordinary BufferQueue producer, so both
|
||||
* eglCreateWindowSurface and vkCreateAndroidSurfaceKHR accept it. The image
|
||||
* listener must drain the queue: without it the producer blocks once maxImages
|
||||
* buffers are in flight and the next swap deadlocks.
|
||||
*//*--------------------------------------------------------------------*/
|
||||
class ImageReaderWindow
|
||||
{
|
||||
public:
|
||||
ImageReaderWindow(int width, int height) : m_reader(nullptr), m_window(nullptr)
|
||||
{
|
||||
const media_status_t status =
|
||||
AImageReader_newWithUsage(width, height, AIMAGE_FORMAT_RGBA_8888,
|
||||
AHARDWAREBUFFER_USAGE_GPU_SAMPLED_IMAGE |
|
||||
AHARDWAREBUFFER_USAGE_GPU_COLOR_OUTPUT,
|
||||
kMaxImages, &m_reader);
|
||||
if (status != AMEDIA_OK || m_reader == nullptr)
|
||||
throw tcu::ResourceError("AImageReader_newWithUsage() failed");
|
||||
|
||||
AImageReader_ImageListener listener = {this, onImageAvailable};
|
||||
AImageReader_setImageListener(m_reader, &listener);
|
||||
|
||||
if (AImageReader_getWindow(m_reader, &m_window) != AMEDIA_OK || m_window == nullptr)
|
||||
{
|
||||
AImageReader_delete(m_reader);
|
||||
m_reader = nullptr;
|
||||
throw tcu::ResourceError("AImageReader_getWindow() failed");
|
||||
}
|
||||
ANativeWindow_acquire(m_window);
|
||||
}
|
||||
|
||||
~ImageReaderWindow(void)
|
||||
{
|
||||
if (m_window != nullptr)
|
||||
ANativeWindow_release(m_window);
|
||||
if (m_reader != nullptr)
|
||||
{
|
||||
AImageReader_setImageListener(m_reader, nullptr);
|
||||
AImageReader_delete(m_reader);
|
||||
}
|
||||
}
|
||||
|
||||
ANativeWindow *getWindow(void) const
|
||||
{
|
||||
return m_window;
|
||||
}
|
||||
|
||||
private:
|
||||
static const int kMaxImages = 4;
|
||||
|
||||
static void onImageAvailable(void *, AImageReader *reader)
|
||||
{
|
||||
AImage *image = nullptr;
|
||||
if (AImageReader_acquireNextImage(reader, &image) == AMEDIA_OK && image != nullptr)
|
||||
AImage_delete(image);
|
||||
}
|
||||
|
||||
ImageReaderWindow(const ImageReaderWindow &);
|
||||
ImageReaderWindow &operator=(const ImageReaderWindow &);
|
||||
|
||||
AImageReader *m_reader;
|
||||
ANativeWindow *m_window;
|
||||
};
|
||||
|
||||
class GetProcFuncLoader : public glw::FunctionLoader
|
||||
{
|
||||
public:
|
||||
GetProcFuncLoader(const eglw::Library &egl) : m_egl(egl)
|
||||
{
|
||||
}
|
||||
|
||||
glw::GenericFuncType get(const char *name) const
|
||||
{
|
||||
return (glw::GenericFuncType)m_egl.getProcAddress(name);
|
||||
}
|
||||
|
||||
protected:
|
||||
const eglw::Library &m_egl;
|
||||
};
|
||||
|
||||
class EglRenderContext : public glu::RenderContext
|
||||
{
|
||||
public:
|
||||
EglRenderContext(const glu::RenderConfig &config, const tcu::CommandLine &cmdLine,
|
||||
const glu::RenderContext *sharedContext);
|
||||
~EglRenderContext(void);
|
||||
|
||||
glu::ContextType getType(void) const
|
||||
{
|
||||
return m_contextType;
|
||||
}
|
||||
eglw::EGLContext getEglContext(void) const
|
||||
{
|
||||
return m_eglContext;
|
||||
}
|
||||
const glw::Functions &getFunctions(void) const
|
||||
{
|
||||
return m_glFunctions;
|
||||
}
|
||||
const tcu::RenderTarget &getRenderTarget(void) const
|
||||
{
|
||||
return m_renderTarget;
|
||||
}
|
||||
void postIterate(void);
|
||||
void makeCurrent(void);
|
||||
|
||||
glw::GenericFuncType getProcAddress(const char *name) const
|
||||
{
|
||||
return (glw::GenericFuncType)m_egl.getProcAddress(name);
|
||||
}
|
||||
|
||||
private:
|
||||
const eglw::DefaultLibrary m_egl;
|
||||
const glu::ContextType m_contextType;
|
||||
eglw::EGLDisplay m_eglDisplay;
|
||||
eglw::EGLContext m_eglContext;
|
||||
eglw::EGLSurface m_eglSurface;
|
||||
ImageReaderWindow *m_window;
|
||||
glw::Functions m_glFunctions;
|
||||
tcu::RenderTarget m_renderTarget;
|
||||
eglw::EGLContext m_sharedEglContext;
|
||||
};
|
||||
|
||||
class ContextFactory : public glu::ContextFactory
|
||||
{
|
||||
public:
|
||||
ContextFactory(void) : glu::ContextFactory("default", "MobileGL EGL context")
|
||||
{
|
||||
}
|
||||
|
||||
glu::RenderContext *createContext(const glu::RenderConfig &config, const tcu::CommandLine &cmdLine,
|
||||
const glu::RenderContext *sharedContext) const
|
||||
{
|
||||
return new EglRenderContext(config, cmdLine, sharedContext);
|
||||
}
|
||||
};
|
||||
|
||||
class Platform : public tcu::Platform, public glu::Platform
|
||||
{
|
||||
public:
|
||||
Platform(void)
|
||||
{
|
||||
m_contextFactoryRegistry.registerFactory(new ContextFactory());
|
||||
}
|
||||
|
||||
const glu::Platform &getGLPlatform(void) const
|
||||
{
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
EglRenderContext::EglRenderContext(const glu::RenderConfig &config, const tcu::CommandLine &cmdLine,
|
||||
const glu::RenderContext *sharedContext)
|
||||
: m_egl(getLibraryName().c_str())
|
||||
, m_contextType(config.type)
|
||||
, m_eglDisplay(EGL_NO_DISPLAY)
|
||||
, m_eglContext(EGL_NO_CONTEXT)
|
||||
, m_eglSurface(EGL_NO_SURFACE)
|
||||
, m_window(nullptr)
|
||||
, m_renderTarget(config.width, config.height,
|
||||
tcu::PixelFormat(config.redBits, config.greenBits, config.blueBits, config.alphaBits),
|
||||
config.depthBits, config.stencilBits, config.numSamples)
|
||||
, m_sharedEglContext(EGL_NO_CONTEXT)
|
||||
{
|
||||
DE_UNREF(cmdLine);
|
||||
|
||||
const glu::ContextType &contextType = config.type;
|
||||
const bool isES = glu::isContextTypeES(contextType);
|
||||
eglw::EGLint eglMajorVersion = 0;
|
||||
eglw::EGLint eglMinorVersion = 0;
|
||||
|
||||
m_eglDisplay = m_egl.getDisplay(EGL_DEFAULT_DISPLAY);
|
||||
EGLU_CHECK_MSG(m_egl, "eglGetDisplay()");
|
||||
if (m_eglDisplay == EGL_NO_DISPLAY)
|
||||
throw tcu::ResourceError("eglGetDisplay() failed");
|
||||
|
||||
EGLU_CHECK_CALL(m_egl, initialize(m_eglDisplay, &eglMajorVersion, &eglMinorVersion));
|
||||
|
||||
// MobileGL cannot make a context current without a surface, so
|
||||
// SURFACETYPE_DONT_CARE (which is what --deqp-surface-type=fbo requests)
|
||||
// still gets a real one.
|
||||
bool wantWindow = false;
|
||||
switch (config.surfaceType)
|
||||
{
|
||||
case glu::RenderConfig::SURFACETYPE_WINDOW:
|
||||
wantWindow = true;
|
||||
break;
|
||||
case glu::RenderConfig::SURFACETYPE_OFFSCREEN_NATIVE:
|
||||
case glu::RenderConfig::SURFACETYPE_OFFSCREEN_GENERIC:
|
||||
wantWindow = false;
|
||||
break;
|
||||
case glu::RenderConfig::SURFACETYPE_DONT_CARE:
|
||||
wantWindow = useWindowSurface();
|
||||
break;
|
||||
default:
|
||||
TCU_CHECK_INTERNAL(false);
|
||||
}
|
||||
|
||||
const int width = (config.width == glu::RenderConfig::DONT_CARE) ? 256 : config.width;
|
||||
const int height = (config.height == glu::RenderConfig::DONT_CARE) ? 256 : config.height;
|
||||
|
||||
vector<eglw::EGLint> cfgAttribs;
|
||||
cfgAttribs.push_back(EGL_RENDERABLE_TYPE);
|
||||
if (isES)
|
||||
{
|
||||
switch (contextType.getMajorVersion())
|
||||
{
|
||||
case 3:
|
||||
cfgAttribs.push_back(EGL_OPENGL_ES3_BIT);
|
||||
break;
|
||||
case 2:
|
||||
cfgAttribs.push_back(EGL_OPENGL_ES2_BIT);
|
||||
break;
|
||||
default:
|
||||
cfgAttribs.push_back(EGL_OPENGL_ES_BIT);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Desktop GL, which is the whole point of this port.
|
||||
cfgAttribs.push_back(EGL_OPENGL_BIT);
|
||||
}
|
||||
|
||||
cfgAttribs.push_back(EGL_SURFACE_TYPE);
|
||||
cfgAttribs.push_back(wantWindow ? EGL_WINDOW_BIT : EGL_PBUFFER_BIT);
|
||||
|
||||
static const struct
|
||||
{
|
||||
eglw::EGLint attrib;
|
||||
int glu::RenderConfig::*field;
|
||||
} s_sizeAttribs[] = {
|
||||
{EGL_RED_SIZE, &glu::RenderConfig::redBits}, {EGL_GREEN_SIZE, &glu::RenderConfig::greenBits},
|
||||
{EGL_BLUE_SIZE, &glu::RenderConfig::blueBits}, {EGL_ALPHA_SIZE, &glu::RenderConfig::alphaBits},
|
||||
{EGL_DEPTH_SIZE, &glu::RenderConfig::depthBits}, {EGL_STENCIL_SIZE, &glu::RenderConfig::stencilBits},
|
||||
{EGL_SAMPLES, &glu::RenderConfig::numSamples},
|
||||
};
|
||||
for (size_t ndx = 0; ndx < DE_LENGTH_OF_ARRAY(s_sizeAttribs); ndx++)
|
||||
{
|
||||
const int value = config.*(s_sizeAttribs[ndx].field);
|
||||
if (value != glu::RenderConfig::DONT_CARE)
|
||||
{
|
||||
cfgAttribs.push_back(s_sizeAttribs[ndx].attrib);
|
||||
cfgAttribs.push_back(value);
|
||||
}
|
||||
}
|
||||
cfgAttribs.push_back(EGL_NONE);
|
||||
|
||||
eglw::EGLConfig eglConfig = nullptr;
|
||||
eglw::EGLint numConfigs = 0;
|
||||
EGLU_CHECK_CALL(m_egl, chooseConfig(m_eglDisplay, &cfgAttribs[0], &eglConfig, 1, &numConfigs));
|
||||
if (numConfigs < 1)
|
||||
throw tcu::NotSupportedError("No matching EGL config for the requested context");
|
||||
|
||||
if (wantWindow)
|
||||
{
|
||||
m_window = new ImageReaderWindow(width, height);
|
||||
|
||||
eglw::EGLint visualId = 0;
|
||||
if (m_egl.getConfigAttrib(m_eglDisplay, eglConfig, EGL_NATIVE_VISUAL_ID, &visualId) && visualId != 0)
|
||||
ANativeWindow_setBuffersGeometry(m_window->getWindow(), width, height, visualId);
|
||||
|
||||
m_eglSurface = m_egl.createWindowSurface(m_eglDisplay, eglConfig,
|
||||
(eglw::EGLNativeWindowType)m_window->getWindow(), nullptr);
|
||||
EGLU_CHECK_MSG(m_egl, "eglCreateWindowSurface()");
|
||||
}
|
||||
else
|
||||
{
|
||||
const eglw::EGLint surfaceAttribs[] = {EGL_WIDTH, width, EGL_HEIGHT, height, EGL_NONE};
|
||||
m_eglSurface = m_egl.createPbufferSurface(m_eglDisplay, eglConfig, surfaceAttribs);
|
||||
EGLU_CHECK_MSG(m_egl, "eglCreatePbufferSurface()");
|
||||
}
|
||||
|
||||
if (m_eglSurface == EGL_NO_SURFACE)
|
||||
throw tcu::ResourceError("Failed to create EGL surface");
|
||||
|
||||
vector<eglw::EGLint> ctxAttribs;
|
||||
ctxAttribs.push_back(EGL_CONTEXT_MAJOR_VERSION_KHR);
|
||||
ctxAttribs.push_back(contextType.getMajorVersion());
|
||||
ctxAttribs.push_back(EGL_CONTEXT_MINOR_VERSION_KHR);
|
||||
ctxAttribs.push_back(contextType.getMinorVersion());
|
||||
|
||||
switch (contextType.getProfile())
|
||||
{
|
||||
case glu::PROFILE_ES:
|
||||
EGLU_CHECK_CALL(m_egl, bindAPI(EGL_OPENGL_ES_API));
|
||||
break;
|
||||
case glu::PROFILE_CORE:
|
||||
EGLU_CHECK_CALL(m_egl, bindAPI(EGL_OPENGL_API));
|
||||
ctxAttribs.push_back(EGL_CONTEXT_OPENGL_PROFILE_MASK_KHR);
|
||||
ctxAttribs.push_back(EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT_KHR);
|
||||
break;
|
||||
case glu::PROFILE_COMPATIBILITY:
|
||||
EGLU_CHECK_CALL(m_egl, bindAPI(EGL_OPENGL_API));
|
||||
ctxAttribs.push_back(EGL_CONTEXT_OPENGL_PROFILE_MASK_KHR);
|
||||
ctxAttribs.push_back(EGL_CONTEXT_OPENGL_COMPATIBILITY_PROFILE_BIT_KHR);
|
||||
break;
|
||||
default:
|
||||
TCU_CHECK_INTERNAL(false);
|
||||
}
|
||||
|
||||
eglw::EGLint flags = 0;
|
||||
if ((contextType.getFlags() & glu::CONTEXT_DEBUG) != 0)
|
||||
flags |= EGL_CONTEXT_OPENGL_DEBUG_BIT_KHR;
|
||||
if ((contextType.getFlags() & glu::CONTEXT_ROBUST) != 0)
|
||||
flags |= EGL_CONTEXT_OPENGL_ROBUST_ACCESS_BIT_KHR;
|
||||
if ((contextType.getFlags() & glu::CONTEXT_FORWARD_COMPATIBLE) != 0)
|
||||
flags |= EGL_CONTEXT_OPENGL_FORWARD_COMPATIBLE_BIT_KHR;
|
||||
if (flags != 0)
|
||||
{
|
||||
ctxAttribs.push_back(EGL_CONTEXT_FLAGS_KHR);
|
||||
ctxAttribs.push_back(flags);
|
||||
}
|
||||
ctxAttribs.push_back(EGL_NONE);
|
||||
|
||||
const EglRenderContext *sharedEglRenderContext = dynamic_cast<const EglRenderContext *>(sharedContext);
|
||||
m_sharedEglContext = sharedEglRenderContext ? sharedEglRenderContext->getEglContext() : EGL_NO_CONTEXT;
|
||||
|
||||
m_eglContext = m_egl.createContext(m_eglDisplay, eglConfig, m_sharedEglContext, &ctxAttribs[0]);
|
||||
EGLU_CHECK_MSG(m_egl, "eglCreateContext()");
|
||||
if (!m_eglContext)
|
||||
throw tcu::ResourceError("eglCreateContext() failed");
|
||||
|
||||
// MobileGL requires draw == read.
|
||||
EGLU_CHECK_CALL(m_egl, makeCurrent(m_eglDisplay, m_eglSurface, m_eglSurface, m_eglContext));
|
||||
|
||||
// MobileGL advertises EGL 1.5, so eglGetProcAddress resolves core entry
|
||||
// points too; there is no separate GL library to dlopen.
|
||||
GetProcFuncLoader funcLoader(m_egl);
|
||||
glu::initCoreFunctions(&m_glFunctions, &funcLoader, contextType.getAPI());
|
||||
glu::initExtensionFunctions(&m_glFunctions, &funcLoader, contextType.getAPI());
|
||||
}
|
||||
|
||||
EglRenderContext::~EglRenderContext(void)
|
||||
{
|
||||
try
|
||||
{
|
||||
if (m_eglDisplay != EGL_NO_DISPLAY)
|
||||
{
|
||||
m_egl.makeCurrent(m_eglDisplay, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT);
|
||||
|
||||
if (m_eglContext != EGL_NO_CONTEXT)
|
||||
m_egl.destroyContext(m_eglDisplay, m_eglContext);
|
||||
|
||||
if (m_eglSurface != EGL_NO_SURFACE)
|
||||
m_egl.destroySurface(m_eglDisplay, m_eglSurface);
|
||||
|
||||
if (m_sharedEglContext == EGL_NO_CONTEXT)
|
||||
m_egl.terminate(m_eglDisplay);
|
||||
}
|
||||
}
|
||||
catch (...)
|
||||
{
|
||||
}
|
||||
|
||||
delete m_window;
|
||||
}
|
||||
|
||||
void EglRenderContext::makeCurrent(void)
|
||||
{
|
||||
EGLU_CHECK_CALL(m_egl, makeCurrent(m_eglDisplay, m_eglSurface, m_eglSurface, m_eglContext));
|
||||
}
|
||||
|
||||
void EglRenderContext::postIterate(void)
|
||||
{
|
||||
m_glFunctions.finish();
|
||||
}
|
||||
|
||||
} // namespace mobilegl
|
||||
} // namespace tcu
|
||||
|
||||
tcu::Platform *createPlatform(void)
|
||||
{
|
||||
return new tcu::mobilegl::Platform();
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
#ifndef _TCUMOBILEGLPLATFORM_HPP
|
||||
#define _TCUMOBILEGLPLATFORM_HPP
|
||||
/*-------------------------------------------------------------------------
|
||||
* dEQP platform port for MobileGL on Android
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*
|
||||
*//*!
|
||||
* \file
|
||||
* \brief MobileGL platform - drives libMobileGL.so's own EGL from a bare
|
||||
* Android process, with no Activity and no system EGL involved.
|
||||
*//*--------------------------------------------------------------------*/
|
||||
|
||||
#include "tcuDefs.hpp"
|
||||
|
||||
namespace tcu
|
||||
{
|
||||
class Platform;
|
||||
}
|
||||
|
||||
tcu::Platform *createPlatform(void);
|
||||
|
||||
#endif // _TCUMOBILEGLPLATFORM_HPP
|
||||
@@ -0,0 +1,2 @@
|
||||
mgprobe
|
||||
*.o
|
||||
@@ -0,0 +1,359 @@
|
||||
/* mgprobe - preflight gate for running a GL conformance suite against MobileGL
|
||||
* from a bare adb-shell process (no APK, no Activity).
|
||||
*
|
||||
* Verifies, for one backend and one surface type, that MobileGL can hand out a
|
||||
* GL 3.3 core context and that pixels read back correctly - both from the
|
||||
* default framebuffer and from a user FBO. Run this before burning hours on a
|
||||
* CTS run; it catches a broken device/library pairing in about a second.
|
||||
*
|
||||
* mgprobe --backend DirectGLES|DirectVulkan --surface pbuffer|imagereader
|
||||
* [--lib /path/to/libMobileGL.so]
|
||||
*
|
||||
* Exit status: 0 if a context came up and FBO readback is correct, non-zero
|
||||
* otherwise. Default-framebuffer readback is reported but does NOT gate, because
|
||||
* DirectVulkan is known to return zeros there while FBO readback is sound.
|
||||
*
|
||||
* Build (NDK, arm64):
|
||||
* $NDK/toolchains/llvm/prebuilt/<host>/bin/aarch64-linux-android26-clang \
|
||||
* -O1 -o mgprobe mgprobe.c -ldl -llog -landroid -lmediandk
|
||||
*/
|
||||
#include <android/native_window.h>
|
||||
#include <dlfcn.h>
|
||||
#include <media/NdkImageReader.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
typedef void *EGLDisplay;
|
||||
typedef void *EGLConfig;
|
||||
typedef void *EGLSurface;
|
||||
typedef void *EGLContext;
|
||||
typedef int EGLint;
|
||||
typedef unsigned int EGLBoolean;
|
||||
typedef unsigned int EGLenum;
|
||||
typedef void *EGLNativeDisplayType;
|
||||
typedef void *EGLNativeWindowType;
|
||||
|
||||
#define EGL_DEFAULT_DISPLAY ((EGLNativeDisplayType)0)
|
||||
#define EGL_NO_CONTEXT ((EGLContext)0)
|
||||
#define EGL_NO_SURFACE ((EGLSurface)0)
|
||||
#define EGL_NONE 0x3038
|
||||
#define EGL_WIDTH 0x3057
|
||||
#define EGL_HEIGHT 0x3056
|
||||
#define EGL_RENDERABLE_TYPE 0x3040
|
||||
#define EGL_SURFACE_TYPE 0x3033
|
||||
#define EGL_WINDOW_BIT 0x0004
|
||||
#define EGL_PBUFFER_BIT 0x0001
|
||||
#define EGL_OPENGL_BIT 0x0008
|
||||
#define EGL_OPENGL_API 0x30A2
|
||||
#define EGL_RED_SIZE 0x3024
|
||||
#define EGL_GREEN_SIZE 0x3023
|
||||
#define EGL_BLUE_SIZE 0x3022
|
||||
#define EGL_ALPHA_SIZE 0x3021
|
||||
#define EGL_DEPTH_SIZE 0x3025
|
||||
#define EGL_STENCIL_SIZE 0x3026
|
||||
#define EGL_NATIVE_VISUAL_ID 0x302E
|
||||
#define EGL_CONTEXT_MAJOR_VERSION 0x3098
|
||||
#define EGL_CONTEXT_MINOR_VERSION 0x30FB
|
||||
#define EGL_CONTEXT_OPENGL_PROFILE_MASK 0x30FD
|
||||
#define EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT 0x00000001
|
||||
|
||||
#define GL_VENDOR 0x1F00
|
||||
#define GL_RENDERER 0x1F01
|
||||
#define GL_VERSION 0x1F02
|
||||
#define GL_SHADING_LANGUAGE_VERSION 0x8B8C
|
||||
#define GL_CONTEXT_PROFILE_MASK 0x9126
|
||||
#define GL_MAJOR_VERSION 0x821B
|
||||
#define GL_MINOR_VERSION 0x821C
|
||||
#define GL_COLOR_BUFFER_BIT 0x00004000
|
||||
#define GL_RGBA 0x1908
|
||||
#define GL_RGBA8 0x8058
|
||||
#define GL_UNSIGNED_BYTE 0x1401
|
||||
#define GL_TEXTURE_2D 0x0DE1
|
||||
#define GL_FRAMEBUFFER 0x8D40
|
||||
#define GL_COLOR_ATTACHMENT0 0x8CE0
|
||||
#define GL_FRAMEBUFFER_COMPLETE 0x8CD5
|
||||
#define GL_TEXTURE_MIN_FILTER 0x2801
|
||||
#define GL_TEXTURE_MAG_FILTER 0x2800
|
||||
#define GL_NEAREST 0x2600
|
||||
#define GL_RENDERBUFFER 0x8D41
|
||||
|
||||
typedef EGLDisplay (*P_getDisplay)(EGLNativeDisplayType);
|
||||
typedef EGLBoolean (*P_initialize)(EGLDisplay, EGLint *, EGLint *);
|
||||
typedef EGLBoolean (*P_bindAPI)(EGLenum);
|
||||
typedef EGLBoolean (*P_chooseConfig)(EGLDisplay, const EGLint *, EGLConfig *, EGLint, EGLint *);
|
||||
typedef EGLBoolean (*P_getConfigAttrib)(EGLDisplay, EGLConfig, EGLint, EGLint *);
|
||||
typedef EGLSurface (*P_createWindowSurface)(EGLDisplay, EGLConfig, EGLNativeWindowType, const EGLint *);
|
||||
typedef EGLSurface (*P_createPbufferSurface)(EGLDisplay, EGLConfig, const EGLint *);
|
||||
typedef EGLContext (*P_createContext)(EGLDisplay, EGLConfig, EGLContext, const EGLint *);
|
||||
typedef EGLBoolean (*P_makeCurrent)(EGLDisplay, EGLSurface, EGLSurface, EGLContext);
|
||||
typedef EGLint (*P_getError)(void);
|
||||
|
||||
typedef const unsigned char *(*P_glGetString)(unsigned int);
|
||||
typedef void (*P_glGetIntegerv)(unsigned int, int *);
|
||||
typedef void (*P_glClearColor)(float, float, float, float);
|
||||
typedef void (*P_glClear)(unsigned int);
|
||||
typedef void (*P_glFinish)(void);
|
||||
typedef void (*P_glReadPixels)(int, int, int, int, unsigned int, unsigned int, void *);
|
||||
typedef unsigned int (*P_glGetError)(void);
|
||||
typedef void (*P_glGenTextures)(int, unsigned int *);
|
||||
typedef void (*P_glBindTexture)(unsigned int, unsigned int);
|
||||
typedef void (*P_glTexImage2D)(unsigned int, int, int, int, int, int, unsigned int, unsigned int, const void *);
|
||||
typedef void (*P_glTexParameteri)(unsigned int, unsigned int, int);
|
||||
typedef void (*P_glGenFramebuffers)(int, unsigned int *);
|
||||
typedef void (*P_glBindFramebuffer)(unsigned int, unsigned int);
|
||||
typedef void (*P_glFramebufferTexture2D)(unsigned int, unsigned int, unsigned int, unsigned int, int);
|
||||
typedef unsigned int (*P_glCheckFramebufferStatus)(unsigned int);
|
||||
typedef void (*P_glViewport)(int, int, int, int);
|
||||
typedef void (*P_glGenRenderbuffers)(int, unsigned int *);
|
||||
typedef void (*P_glBindRenderbuffer)(unsigned int, unsigned int);
|
||||
typedef void (*P_glRenderbufferStorage)(unsigned int, unsigned int, int, int);
|
||||
typedef void (*P_glFramebufferRenderbuffer)(unsigned int, unsigned int, unsigned int, unsigned int);
|
||||
|
||||
static void *g_lib;
|
||||
static void *S(const char *n) { return dlsym(g_lib, n); }
|
||||
|
||||
static void on_image(void *ctx, AImageReader *r) {
|
||||
(void)ctx;
|
||||
AImage *img = NULL;
|
||||
/* Drain the queue, or the producer blocks once maxImages are in flight. */
|
||||
if (AImageReader_acquireNextImage(r, &img) == AMEDIA_OK && img) AImage_delete(img);
|
||||
}
|
||||
|
||||
#define DIM 256
|
||||
|
||||
static int near8(unsigned got, int want, int tol) {
|
||||
int d = (int)got - want;
|
||||
return d <= tol && d >= -tol;
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
const char *backend = "DirectGLES";
|
||||
const char *surface = "pbuffer";
|
||||
const char *libpath = "libMobileGL.so";
|
||||
|
||||
for (int i = 1; i < argc; ++i) {
|
||||
if (!strcmp(argv[i], "--backend") && i + 1 < argc) backend = argv[++i];
|
||||
else if (!strcmp(argv[i], "--surface") && i + 1 < argc) surface = argv[++i];
|
||||
else if (!strcmp(argv[i], "--lib") && i + 1 < argc) libpath = argv[++i];
|
||||
else {
|
||||
fprintf(stderr, "usage: %s [--backend DirectGLES|DirectVulkan]"
|
||||
" [--surface pbuffer|imagereader] [--lib path]\n", argv[0]);
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
setvbuf(stdout, NULL, _IONBF, 0);
|
||||
|
||||
/* MobileGL parses its config from an ELF constructor, so the backend must be
|
||||
* selected before the library is mapped. */
|
||||
setenv("MOBILEGL_BACKEND_TYPE", backend, 1);
|
||||
printf("mgprobe backend=%s surface=%s lib=%s\n", backend, surface, libpath);
|
||||
|
||||
int useWindow = !strcmp(surface, "imagereader");
|
||||
ANativeWindow *win = NULL;
|
||||
AImageReader *reader = NULL;
|
||||
if (useWindow) {
|
||||
if (AImageReader_newWithUsage(DIM, DIM, AIMAGE_FORMAT_RGBA_8888,
|
||||
AHARDWAREBUFFER_USAGE_GPU_SAMPLED_IMAGE |
|
||||
AHARDWAREBUFFER_USAGE_GPU_COLOR_OUTPUT,
|
||||
4, &reader) != AMEDIA_OK || !reader) {
|
||||
printf("FAIL AImageReader_newWithUsage\n");
|
||||
return 3;
|
||||
}
|
||||
AImageReader_ImageListener l = {NULL, on_image};
|
||||
AImageReader_setImageListener(reader, &l);
|
||||
if (AImageReader_getWindow(reader, &win) != AMEDIA_OK || !win) {
|
||||
printf("FAIL AImageReader_getWindow\n");
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
|
||||
g_lib = dlopen(libpath, RTLD_NOW | RTLD_LOCAL);
|
||||
if (!g_lib) {
|
||||
printf("FAIL dlopen: %s\n", dlerror());
|
||||
return 4;
|
||||
}
|
||||
|
||||
P_getDisplay eglGetDisplay_ = (P_getDisplay)S("eglGetDisplay");
|
||||
P_initialize eglInitialize_ = (P_initialize)S("eglInitialize");
|
||||
P_bindAPI eglBindAPI_ = (P_bindAPI)S("eglBindAPI");
|
||||
P_chooseConfig eglChooseConfig_ = (P_chooseConfig)S("eglChooseConfig");
|
||||
P_getConfigAttrib eglGetConfigAttrib_ = (P_getConfigAttrib)S("eglGetConfigAttrib");
|
||||
P_createWindowSurface eglCreateWindowSurface_ = (P_createWindowSurface)S("eglCreateWindowSurface");
|
||||
P_createPbufferSurface eglCreatePbufferSurface_ = (P_createPbufferSurface)S("eglCreatePbufferSurface");
|
||||
P_createContext eglCreateContext_ = (P_createContext)S("eglCreateContext");
|
||||
P_makeCurrent eglMakeCurrent_ = (P_makeCurrent)S("eglMakeCurrent");
|
||||
P_getError eglGetError_ = (P_getError)S("eglGetError");
|
||||
|
||||
if (!eglGetDisplay_ || !eglInitialize_ || !eglChooseConfig_ || !eglCreateContext_ || !eglMakeCurrent_) {
|
||||
printf("FAIL missing core EGL exports\n");
|
||||
return 5;
|
||||
}
|
||||
|
||||
EGLDisplay dpy = eglGetDisplay_(EGL_DEFAULT_DISPLAY);
|
||||
EGLint vmaj = 0, vmin = 0;
|
||||
if (!eglInitialize_(dpy, &vmaj, &vmin)) {
|
||||
printf("FAIL eglInitialize err=0x%x\n", eglGetError_ ? eglGetError_() : 0);
|
||||
return 6;
|
||||
}
|
||||
if (eglBindAPI_ && !eglBindAPI_(EGL_OPENGL_API)) {
|
||||
printf("FAIL eglBindAPI(EGL_OPENGL_API) err=0x%x\n", eglGetError_ ? eglGetError_() : 0);
|
||||
return 7;
|
||||
}
|
||||
|
||||
const EGLint cfgAttribs[] = {
|
||||
EGL_SURFACE_TYPE, useWindow ? EGL_WINDOW_BIT : EGL_PBUFFER_BIT,
|
||||
EGL_RENDERABLE_TYPE, EGL_OPENGL_BIT,
|
||||
EGL_RED_SIZE, 8, EGL_GREEN_SIZE, 8, EGL_BLUE_SIZE, 8, EGL_ALPHA_SIZE, 8,
|
||||
EGL_DEPTH_SIZE, 24, EGL_STENCIL_SIZE, 8,
|
||||
EGL_NONE};
|
||||
EGLConfig cfg = 0;
|
||||
EGLint ncfg = 0;
|
||||
if (!eglChooseConfig_(dpy, cfgAttribs, &cfg, 1, &ncfg) || ncfg < 1) {
|
||||
printf("FAIL eglChooseConfig n=%d err=0x%x\n", ncfg, eglGetError_ ? eglGetError_() : 0);
|
||||
return 8;
|
||||
}
|
||||
|
||||
EGLSurface surf;
|
||||
if (useWindow) {
|
||||
EGLint vis = 0;
|
||||
if (eglGetConfigAttrib_ && eglGetConfigAttrib_(dpy, cfg, EGL_NATIVE_VISUAL_ID, &vis) && vis)
|
||||
ANativeWindow_setBuffersGeometry(win, DIM, DIM, vis);
|
||||
surf = eglCreateWindowSurface_(dpy, cfg, (EGLNativeWindowType)win, NULL);
|
||||
} else {
|
||||
const EGLint sa[] = {EGL_WIDTH, DIM, EGL_HEIGHT, DIM, EGL_NONE};
|
||||
surf = eglCreatePbufferSurface_(dpy, cfg, sa);
|
||||
}
|
||||
if (surf == EGL_NO_SURFACE) {
|
||||
printf("FAIL create%sSurface err=0x%x\n", useWindow ? "Window" : "Pbuffer",
|
||||
eglGetError_ ? eglGetError_() : 0);
|
||||
return 9;
|
||||
}
|
||||
|
||||
const EGLint ctxAttribs[] = {
|
||||
EGL_CONTEXT_MAJOR_VERSION, 3, EGL_CONTEXT_MINOR_VERSION, 3,
|
||||
EGL_CONTEXT_OPENGL_PROFILE_MASK, EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT, EGL_NONE};
|
||||
EGLContext ctx = eglCreateContext_(dpy, cfg, EGL_NO_CONTEXT, ctxAttribs);
|
||||
if (ctx == EGL_NO_CONTEXT) {
|
||||
printf("FAIL eglCreateContext(3.3 core) err=0x%x\n", eglGetError_ ? eglGetError_() : 0);
|
||||
return 10;
|
||||
}
|
||||
/* MobileGL requires draw == read and rejects EGL_NO_SURFACE. */
|
||||
if (!eglMakeCurrent_(dpy, surf, surf, ctx)) {
|
||||
printf("FAIL eglMakeCurrent err=0x%x\n", eglGetError_ ? eglGetError_() : 0);
|
||||
return 11;
|
||||
}
|
||||
|
||||
P_glGetString glGetString_ = (P_glGetString)S("glGetString");
|
||||
P_glGetIntegerv glGetIntegerv_ = (P_glGetIntegerv)S("glGetIntegerv");
|
||||
P_glClearColor glClearColor_ = (P_glClearColor)S("glClearColor");
|
||||
P_glClear glClear_ = (P_glClear)S("glClear");
|
||||
P_glFinish glFinish_ = (P_glFinish)S("glFinish");
|
||||
P_glReadPixels glReadPixels_ = (P_glReadPixels)S("glReadPixels");
|
||||
P_glGetError glGetError_ = (P_glGetError)S("glGetError");
|
||||
P_glGenTextures glGenTextures_ = (P_glGenTextures)S("glGenTextures");
|
||||
P_glBindTexture glBindTexture_ = (P_glBindTexture)S("glBindTexture");
|
||||
P_glTexImage2D glTexImage2D_ = (P_glTexImage2D)S("glTexImage2D");
|
||||
P_glTexParameteri glTexParameteri_ = (P_glTexParameteri)S("glTexParameteri");
|
||||
P_glGenFramebuffers glGenFramebuffers_ = (P_glGenFramebuffers)S("glGenFramebuffers");
|
||||
P_glBindFramebuffer glBindFramebuffer_ = (P_glBindFramebuffer)S("glBindFramebuffer");
|
||||
P_glFramebufferTexture2D glFramebufferTexture2D_ = (P_glFramebufferTexture2D)S("glFramebufferTexture2D");
|
||||
P_glCheckFramebufferStatus glCheckFramebufferStatus_ = (P_glCheckFramebufferStatus)S("glCheckFramebufferStatus");
|
||||
P_glViewport glViewport_ = (P_glViewport)S("glViewport");
|
||||
P_glGenRenderbuffers glGenRenderbuffers_ = (P_glGenRenderbuffers)S("glGenRenderbuffers");
|
||||
P_glBindRenderbuffer glBindRenderbuffer_ = (P_glBindRenderbuffer)S("glBindRenderbuffer");
|
||||
P_glRenderbufferStorage glRenderbufferStorage_ = (P_glRenderbufferStorage)S("glRenderbufferStorage");
|
||||
P_glFramebufferRenderbuffer glFramebufferRenderbuffer_ = (P_glFramebufferRenderbuffer)S("glFramebufferRenderbuffer");
|
||||
|
||||
int major = -1, minor = -1, profile = -1;
|
||||
glGetIntegerv_(GL_MAJOR_VERSION, &major);
|
||||
glGetIntegerv_(GL_MINOR_VERSION, &minor);
|
||||
glGetIntegerv_(GL_CONTEXT_PROFILE_MASK, &profile);
|
||||
printf(" GL_VENDOR %s\n", (const char *)glGetString_(GL_VENDOR));
|
||||
printf(" GL_RENDERER %s\n", (const char *)glGetString_(GL_RENDERER));
|
||||
printf(" GL_VERSION %s\n", (const char *)glGetString_(GL_VERSION));
|
||||
printf(" GLSL %s\n", (const char *)glGetString_(GL_SHADING_LANGUAGE_VERSION));
|
||||
printf(" version %d.%d profile_mask 0x%x %s\n", major, minor, profile,
|
||||
(profile & 1) ? "(core)" : "(NOT CORE)");
|
||||
|
||||
unsigned char px[4];
|
||||
|
||||
/* Default framebuffer. */
|
||||
glClearColor_(0.25f, 0.5f, 0.75f, 1.0f);
|
||||
glClear_(GL_COLOR_BUFFER_BIT);
|
||||
if (glFinish_) glFinish_();
|
||||
memset(px, 0, sizeof px);
|
||||
glReadPixels_(DIM / 2, DIM / 2, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, px);
|
||||
int defOk = near8(px[0], 64, 10) && near8(px[1], 128, 10) && near8(px[2], 191, 10);
|
||||
printf(" default-FB readback (%u,%u,%u,%u) %s\n", px[0], px[1], px[2], px[3],
|
||||
defOk ? "ok" : "BROKEN");
|
||||
|
||||
/* User FBO - this is what dEQP uses with --deqp-surface-type=fbo. */
|
||||
unsigned int tex = 0, fbo = 0;
|
||||
glGenTextures_(1, &tex);
|
||||
glBindTexture_(GL_TEXTURE_2D, tex);
|
||||
glTexImage2D_(GL_TEXTURE_2D, 0, GL_RGBA8, DIM, DIM, 0, GL_RGBA, GL_UNSIGNED_BYTE, NULL);
|
||||
glTexParameteri_(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
|
||||
glTexParameteri_(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
glGenFramebuffers_(1, &fbo);
|
||||
glBindFramebuffer_(GL_FRAMEBUFFER, fbo);
|
||||
glFramebufferTexture2D_(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, tex, 0);
|
||||
unsigned int fbst = glCheckFramebufferStatus_(GL_FRAMEBUFFER);
|
||||
int fboOk = 0;
|
||||
if (fbst == GL_FRAMEBUFFER_COMPLETE) {
|
||||
glViewport_(0, 0, DIM, DIM);
|
||||
glClearColor_(0.9f, 0.2f, 0.4f, 1.0f);
|
||||
glClear_(GL_COLOR_BUFFER_BIT);
|
||||
if (glFinish_) glFinish_();
|
||||
memset(px, 0, sizeof px);
|
||||
glReadPixels_(DIM / 2, DIM / 2, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, px);
|
||||
fboOk = near8(px[0], 230, 10) && near8(px[1], 51, 10) && near8(px[2], 102, 10);
|
||||
printf(" user-FBO readback (%u,%u,%u,%u) %s\n", px[0], px[1], px[2], px[3],
|
||||
fboOk ? "ok" : "BROKEN");
|
||||
} else {
|
||||
printf(" user-FBO incomplete status=0x%x\n", fbst);
|
||||
}
|
||||
|
||||
/* FBO with a RENDERBUFFER colour attachment. This is what dEQP's
|
||||
* FboRenderContext allocates for --deqp-surface-type=fbo, so it is the path
|
||||
* that actually decides a conformance run - a texture-attached FBO working
|
||||
* says nothing about it. */
|
||||
unsigned int rbo = 0, rfbo = 0;
|
||||
int rboOk = 0;
|
||||
if (glGenRenderbuffers_ && glBindRenderbuffer_ && glRenderbufferStorage_ && glFramebufferRenderbuffer_) {
|
||||
glGenRenderbuffers_(1, &rbo);
|
||||
glBindRenderbuffer_(GL_RENDERBUFFER, rbo);
|
||||
glRenderbufferStorage_(GL_RENDERBUFFER, GL_RGBA8, DIM, DIM);
|
||||
glBindRenderbuffer_(GL_RENDERBUFFER, 0);
|
||||
glGenFramebuffers_(1, &rfbo);
|
||||
glBindFramebuffer_(GL_FRAMEBUFFER, rfbo);
|
||||
glFramebufferRenderbuffer_(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, rbo);
|
||||
unsigned int rst = glCheckFramebufferStatus_(GL_FRAMEBUFFER);
|
||||
if (rst == GL_FRAMEBUFFER_COMPLETE) {
|
||||
glViewport_(0, 0, DIM, DIM);
|
||||
glClearColor_(0.1f, 0.7f, 0.3f, 1.0f);
|
||||
glClear_(GL_COLOR_BUFFER_BIT);
|
||||
if (glFinish_) glFinish_();
|
||||
memset(px, 0, sizeof px);
|
||||
glReadPixels_(DIM / 2, DIM / 2, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, px);
|
||||
rboOk = near8(px[0], 26, 10) && near8(px[1], 179, 10) && near8(px[2], 77, 10);
|
||||
printf(" rbo-FBO readback (%u,%u,%u,%u) %s\n", px[0], px[1], px[2], px[3],
|
||||
rboOk ? "ok" : "BROKEN");
|
||||
} else {
|
||||
printf(" rbo-FBO incomplete status=0x%x\n", rst);
|
||||
}
|
||||
} else {
|
||||
printf(" rbo-FBO skipped (renderbuffer entry points unavailable)\n");
|
||||
}
|
||||
|
||||
unsigned glerr = glGetError_ ? glGetError_() : 0;
|
||||
int ok = fboOk && rboOk && (major > 3 || (major == 3 && minor >= 3)) && (profile & 1) && glerr == 0;
|
||||
printf("%s backend=%s surface=%s default_fb=%s user_fbo=%s rbo_fbo=%s glerr=0x%x\n",
|
||||
ok ? "PASS" : "FAIL", backend, surface, defOk ? "ok" : "broken",
|
||||
fboOk ? "ok" : "broken", rboOk ? "ok" : "broken", glerr);
|
||||
|
||||
fflush(stdout);
|
||||
/* MobileGL aborts in static teardown; leave before that runs. */
|
||||
_exit(ok ? 0 : 1);
|
||||
}
|
||||
@@ -0,0 +1,204 @@
|
||||
#!/usr/bin/env python
|
||||
"""Summarise dEQP/glcts .qpa logs into a conformance pass rate.
|
||||
|
||||
Handles the two ways a case can end in a .qpa: a normal
|
||||
``#beginTestCaseResult``/``#endTestCaseResult`` pair carrying a
|
||||
``<Result StatusCode="...">`` element, and ``#terminateTestCaseResult <reason>``,
|
||||
which is what the log contains when the process died partway through a case.
|
||||
Cases that were started but never terminated (the run was killed) are reported
|
||||
separately so a truncated chunk is never silently scored as a pass.
|
||||
|
||||
Usage:
|
||||
python qpa_report.py <file-or-dir> [<file-or-dir> ...] [--json out.json] [--top N]
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import Counter, defaultdict
|
||||
|
||||
# Khronos conformance treats these as non-failures: the test either passed or
|
||||
# the implementation legitimately does not expose the feature under test.
|
||||
NON_FAILURE = {
|
||||
"Pass",
|
||||
"NotSupported",
|
||||
"QualityWarning",
|
||||
"CompatibilityWarning",
|
||||
"Waiver",
|
||||
}
|
||||
|
||||
# Statuses that indicate the case did not merely fail but destabilised the run.
|
||||
HARD = {"Crash", "Timeout", "InternalError", "ResourceError", "DeviceHang"}
|
||||
|
||||
CASE_START = re.compile(r"^#beginTestCaseResult\s+(\S+)")
|
||||
CASE_END = re.compile(r"^#endTestCaseResult")
|
||||
CASE_TERM = re.compile(r"^#terminateTestCaseResult\s+(.*)")
|
||||
RESULT = re.compile(r'<Result\s+StatusCode="([^"]+)"')
|
||||
|
||||
|
||||
def parse_qpa(path):
|
||||
"""Yield (case_name, status) for every case recorded in one .qpa file."""
|
||||
current = None
|
||||
status = None
|
||||
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
||||
for line in fh:
|
||||
m = CASE_START.match(line)
|
||||
if m:
|
||||
if current is not None:
|
||||
# A new case started before the previous one closed.
|
||||
yield current, status or "Incomplete"
|
||||
current, status = m.group(1), None
|
||||
continue
|
||||
if current is None:
|
||||
continue
|
||||
m = RESULT.search(line)
|
||||
if m:
|
||||
status = m.group(1)
|
||||
continue
|
||||
m = CASE_TERM.match(line)
|
||||
if m:
|
||||
reason = m.group(1).strip() or "Terminated"
|
||||
# dEQP writes e.g. "Crash" / "Timeout" here.
|
||||
yield current, reason if reason in HARD else "Crash"
|
||||
current, status = None, None
|
||||
continue
|
||||
if CASE_END.match(line):
|
||||
yield current, status or "Incomplete"
|
||||
current, status = None, None
|
||||
if current is not None:
|
||||
# File ended mid-case: the runner was killed.
|
||||
yield current, "Incomplete"
|
||||
|
||||
|
||||
def collect(paths):
|
||||
files = []
|
||||
for p in paths:
|
||||
if os.path.isdir(p):
|
||||
for root, _dirs, names in os.walk(p):
|
||||
files.extend(os.path.join(root, n) for n in sorted(names) if n.endswith(".qpa"))
|
||||
else:
|
||||
files.append(p)
|
||||
return files
|
||||
|
||||
|
||||
def group_of(case):
|
||||
"""The case's parent group, e.g. KHR-GL33.shaders.arrays for ...arrays.foo."""
|
||||
parts = case.split(".")
|
||||
return ".".join(parts[:-1]) if len(parts) > 1 else case
|
||||
|
||||
|
||||
def load_sidecar(paths, name):
|
||||
"""Case names run_cts.py recorded in one of its sidecar lists."""
|
||||
out = set()
|
||||
for p in paths:
|
||||
d = p if os.path.isdir(p) else os.path.dirname(p)
|
||||
f = os.path.join(d, name)
|
||||
if os.path.isfile(f):
|
||||
with open(f, "r", encoding="utf-8") as fh:
|
||||
out.update(l.strip() for l in fh if l.strip() and not l.strip().startswith("#"))
|
||||
return out
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("paths", nargs="+")
|
||||
ap.add_argument("--json", dest="json_out")
|
||||
ap.add_argument("--top", type=int, default=25)
|
||||
ap.add_argument("--label", default="")
|
||||
args = ap.parse_args()
|
||||
|
||||
files = collect(args.paths)
|
||||
if not files:
|
||||
print("no .qpa files found", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
# Later chunks may re-run a case; last result wins.
|
||||
results = {}
|
||||
for f in files:
|
||||
for case, status in parse_qpa(f):
|
||||
results[case] = status
|
||||
|
||||
# A case the runner saw take the process down is a Crash, not merely an
|
||||
# unterminated log entry - but a real result from a later retry wins.
|
||||
for case in load_sidecar(args.paths, "crashed.txt"):
|
||||
if results.get(case, "Incomplete") == "Incomplete":
|
||||
results[case] = "Crash"
|
||||
# Worse than a crash: these rebooted the device.
|
||||
for case in load_sidecar(args.paths, "hung.txt"):
|
||||
if results.get(case, "Incomplete") in ("Incomplete", "Crash"):
|
||||
results[case] = "DeviceHang"
|
||||
|
||||
# Cases excluded up front, and cases the run never reached, are not results.
|
||||
# Report them separately so a partial run is never read as a complete one.
|
||||
skipped = load_sidecar(args.paths, "skipped.txt")
|
||||
unrun = load_sidecar(args.paths, "unrun.txt") - set(results)
|
||||
|
||||
counts = Counter(results.values())
|
||||
total = len(results)
|
||||
non_fail = sum(counts[s] for s in NON_FAILURE)
|
||||
strict_pass = counts["Pass"]
|
||||
failures = total - non_fail
|
||||
|
||||
by_group_fail = defaultdict(int)
|
||||
by_group_total = defaultdict(int)
|
||||
for case, status in results.items():
|
||||
g = group_of(case)
|
||||
by_group_total[g] += 1
|
||||
if status not in NON_FAILURE:
|
||||
by_group_fail[g] += 1
|
||||
|
||||
label = f" [{args.label}]" if args.label else ""
|
||||
print(f"=== glcts conformance summary{label} ===")
|
||||
print(f"files parsed : {len(files)}")
|
||||
print(f"cases with result : {total}")
|
||||
print()
|
||||
for status, n in counts.most_common():
|
||||
mark = " " if status in NON_FAILURE else " ! "
|
||||
print(f"{mark}{status:<22} {n:>7} {100.0 * n / total:6.2f}%")
|
||||
print()
|
||||
if total:
|
||||
print(f"conformance pass rate (Pass+NotSupported+warnings) : {100.0 * non_fail / total:6.2f}% ({non_fail}/{total})")
|
||||
print(f"strict pass rate (Pass only) : {100.0 * strict_pass / total:6.2f}% ({strict_pass}/{total})")
|
||||
print(f"failures : {failures}")
|
||||
|
||||
if skipped or unrun:
|
||||
print("\n--- NOT MEASURED (excluded from the rates above) ---")
|
||||
if skipped:
|
||||
print(f" quarantined up front : {len(skipped)}")
|
||||
if unrun:
|
||||
print(f" never reached : {len(unrun)}")
|
||||
print(" The rates above cover only cases that produced a result.")
|
||||
|
||||
if failures:
|
||||
print(f"\n--- worst groups (of {len(by_group_total)}) ---")
|
||||
worst = sorted(by_group_fail.items(), key=lambda kv: -kv[1])[: args.top]
|
||||
for g, nf in worst:
|
||||
nt = by_group_total[g]
|
||||
print(f" {g:<52} {nf:>6}/{nt:<6} fail ({100.0 * nf / nt:5.1f}%)")
|
||||
|
||||
if args.json_out:
|
||||
with open(args.json_out, "w", encoding="utf-8") as fh:
|
||||
json.dump(
|
||||
{
|
||||
"label": args.label,
|
||||
"files": len(files),
|
||||
"total": total,
|
||||
"counts": dict(counts),
|
||||
"non_failure": non_fail,
|
||||
"strict_pass": strict_pass,
|
||||
"failures": failures,
|
||||
"pass_rate": (non_fail / total) if total else 0.0,
|
||||
"strict_pass_rate": (strict_pass / total) if total else 0.0,
|
||||
"results": results,
|
||||
},
|
||||
fh,
|
||||
indent=1,
|
||||
)
|
||||
print(f"\nwrote {args.json_out}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,269 @@
|
||||
#!/usr/bin/env python
|
||||
"""Drive a glcts run on a device, resuming across crashes.
|
||||
|
||||
MobileGL crashes on some cases, and glcts takes the whole process down with it.
|
||||
A single invocation would therefore stop at the first crash and leave most of
|
||||
the suite unmeasured. This runner re-invokes glcts with only the cases that have
|
||||
not produced a result yet, records each crashed case as "Crash", and repeats
|
||||
until the list is exhausted, so one bad case costs one case rather than the run.
|
||||
|
||||
Usage:
|
||||
python run_cts.py --serial <adb-serial> --backend DirectGLES|DirectVulkan \\
|
||||
--caselist <host-path-to-mustpass.txt> --outdir <host-dir> [--device-dir /data/local/tmp/mgcts]
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
CASE_START = re.compile(r"^#beginTestCaseResult\s+(\S+)")
|
||||
CASE_END = re.compile(r"^#endTestCaseResult")
|
||||
CASE_TERM = re.compile(r"^#terminateTestCaseResult\s+(.*)")
|
||||
|
||||
|
||||
def adb(serial, *args, timeout=None):
|
||||
try:
|
||||
return subprocess.run(["adb", "-s", serial, *args], capture_output=True, text=True, timeout=timeout)
|
||||
except subprocess.TimeoutExpired:
|
||||
return subprocess.CompletedProcess(args, returncode=124, stdout="", stderr="adb timeout")
|
||||
|
||||
|
||||
def device_alive(serial, timeout=30):
|
||||
"""True only if the device answers a trivial shell command.
|
||||
|
||||
Distinguishes "glcts crashed" from "the device fell over". Without this a
|
||||
dead device looks like every remaining case crashing, which silently turns a
|
||||
broken run into a plausible-looking conformance number.
|
||||
"""
|
||||
r = adb(serial, "shell", "echo alive", timeout=timeout)
|
||||
return r.returncode == 0 and "alive" in (r.stdout or "")
|
||||
|
||||
|
||||
def wait_for_device(serial, attempts=20, delay=15):
|
||||
for i in range(attempts):
|
||||
if device_alive(serial):
|
||||
return True
|
||||
print(f"[run_cts] device {serial} unresponsive, waiting ({i + 1}/{attempts})")
|
||||
time.sleep(delay)
|
||||
return False
|
||||
|
||||
|
||||
def mem_available_kb(serial):
|
||||
r = adb(serial, "shell", "grep MemAvailable /proc/meminfo", timeout=30)
|
||||
m = re.search(r"(\d+)", r.stdout or "")
|
||||
return int(m.group(1)) if m else None
|
||||
|
||||
|
||||
def completed_cases(qpa_path):
|
||||
"""Return (finished_case_names, last_started_case_or_None).
|
||||
|
||||
A case that was started but never closed is the one the process died in.
|
||||
"""
|
||||
finished = []
|
||||
current = None
|
||||
if not os.path.exists(qpa_path):
|
||||
return finished, None
|
||||
with open(qpa_path, "r", encoding="utf-8", errors="replace") as fh:
|
||||
for line in fh:
|
||||
m = CASE_START.match(line)
|
||||
if m:
|
||||
current = m.group(1)
|
||||
continue
|
||||
if current is not None and (CASE_END.match(line) or CASE_TERM.match(line)):
|
||||
finished.append(current)
|
||||
current = None
|
||||
return finished, current
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--serial", required=True)
|
||||
ap.add_argument("--backend", required=True, choices=["DirectGLES", "DirectVulkan"])
|
||||
ap.add_argument("--caselist", required=True)
|
||||
ap.add_argument("--outdir", required=True)
|
||||
ap.add_argument("--device-dir", default="/data/local/tmp/mgcts")
|
||||
ap.add_argument("--surface", default="fbo", help="--deqp-surface-type value")
|
||||
ap.add_argument("--max-rounds", type=int, default=4000)
|
||||
ap.add_argument("--max-empty-streak", type=int, default=64,
|
||||
help="abort after this many consecutive chunks that produce no log at all")
|
||||
ap.add_argument("--min-mem-kb", type=int, default=400000,
|
||||
help="pause when the device drops below this much available memory")
|
||||
ap.add_argument("--chunk-timeout", type=int, default=900,
|
||||
help="seconds before giving up on one glcts invocation (a GPU hang never returns)")
|
||||
ap.add_argument("--skip-file", default=None,
|
||||
help="file of case names to exclude, e.g. cases known to hang the device")
|
||||
ap.add_argument("--env", action="append", default=[], metavar="K=V",
|
||||
help="extra environment variable for glcts (repeatable)")
|
||||
args = ap.parse_args()
|
||||
|
||||
os.makedirs(args.outdir, exist_ok=True)
|
||||
|
||||
with open(args.caselist, "r", encoding="utf-8") as fh:
|
||||
remaining = [l.strip() for l in fh if l.strip() and not l.strip().startswith("#")]
|
||||
|
||||
skipped = []
|
||||
if args.skip_file and os.path.isfile(args.skip_file):
|
||||
with open(args.skip_file, "r", encoding="utf-8") as fh:
|
||||
skip = {l.strip() for l in fh if l.strip() and not l.strip().startswith("#")}
|
||||
skipped = [c for c in remaining if c in skip]
|
||||
remaining = [c for c in remaining if c not in skip]
|
||||
print(f"[run_cts] skipping {len(skipped)} case(s) from {args.skip_file}")
|
||||
|
||||
total = len(remaining)
|
||||
print(f"[run_cts] {args.backend} on {args.serial}: {total} cases")
|
||||
|
||||
crashed = []
|
||||
hung = []
|
||||
done = set()
|
||||
chunk = 0
|
||||
started = time.time()
|
||||
empty_streak = 0
|
||||
|
||||
if not wait_for_device(args.serial):
|
||||
print("[run_cts] device not responding before start; aborting", file=sys.stderr)
|
||||
return 3
|
||||
|
||||
while remaining and chunk < args.max_rounds:
|
||||
listfile = os.path.join(args.outdir, "remaining.txt")
|
||||
with open(listfile, "w", encoding="utf-8", newline="\n") as fh:
|
||||
fh.write("\n".join(remaining) + "\n")
|
||||
|
||||
# Repeated process launches plus crash tombstones can drive the device
|
||||
# into memory pressure; give it room rather than pushing it over.
|
||||
mem = mem_available_kb(args.serial)
|
||||
if mem is not None and mem < args.min_mem_kb:
|
||||
print(f"[run_cts] low memory ({mem} kB available); pausing 30 s")
|
||||
time.sleep(30)
|
||||
|
||||
dev_list = f"{args.device_dir}/remaining.txt"
|
||||
dev_qpa = f"{args.device_dir}/chunk.qpa"
|
||||
push = adb(args.serial, "push", listfile, dev_list, timeout=120)
|
||||
if push.returncode != 0:
|
||||
print(f"[run_cts] push failed ({push.stderr.strip()}); treating as device trouble",
|
||||
file=sys.stderr)
|
||||
if not wait_for_device(args.serial):
|
||||
print("[run_cts] ABORTING: device unreachable.", file=sys.stderr)
|
||||
break
|
||||
continue
|
||||
adb(args.serial, "shell", f"rm -f {dev_qpa}", timeout=60)
|
||||
|
||||
extra_env = "".join(f"{kv} " for kv in args.env)
|
||||
cmd = (
|
||||
f"cd {args.device_dir} && "
|
||||
f"MOBILEGL_BACKEND_TYPE={args.backend} LD_LIBRARY_PATH=. {extra_env}"
|
||||
f"./glcts --deqp-caselist-file={dev_list} "
|
||||
f"--deqp-surface-type={args.surface} "
|
||||
f"--deqp-terminate-on-device-lost=disable "
|
||||
f"--deqp-log-images=disable --deqp-log-shader-sources=disable "
|
||||
f"--deqp-log-filename={dev_qpa} > /dev/null 2>&1; echo RC=$?"
|
||||
)
|
||||
run = adb(args.serial, "shell", cmd, timeout=args.chunk_timeout)
|
||||
if run.returncode == 124:
|
||||
print(f"[run_cts] chunk {chunk:04d} timed out after {args.chunk_timeout}s "
|
||||
f"(likely a GPU hang)", file=sys.stderr)
|
||||
|
||||
# Some cases hang the GPU hard enough to reboot the device. The log on
|
||||
# /data/local/tmp survives that, so wait for the device to come back and
|
||||
# pull it anyway rather than losing the whole chunk.
|
||||
rebooted = False
|
||||
if not device_alive(args.serial, timeout=30):
|
||||
print(f"[run_cts] device went away during chunk {chunk:04d}; waiting for it",
|
||||
file=sys.stderr)
|
||||
if not wait_for_device(args.serial, attempts=40, delay=15):
|
||||
print("[run_cts] ABORTING: device never came back. Results are incomplete; "
|
||||
"do NOT treat the remaining cases as failures.", file=sys.stderr)
|
||||
break
|
||||
rebooted = True
|
||||
print("[run_cts] device is back")
|
||||
|
||||
local_qpa = os.path.join(args.outdir, f"chunk{chunk:04d}.qpa")
|
||||
pull = adb(args.serial, "pull", dev_qpa, local_qpa, timeout=300)
|
||||
if pull.returncode != 0 and rebooted:
|
||||
time.sleep(10)
|
||||
adb(args.serial, "pull", dev_qpa, local_qpa, timeout=300)
|
||||
|
||||
finished, in_flight = completed_cases(local_qpa)
|
||||
for c in finished:
|
||||
done.add(c)
|
||||
|
||||
progressed = len(finished)
|
||||
if progressed > 0:
|
||||
empty_streak = 0
|
||||
if in_flight is not None:
|
||||
# The case that was open when the process (or the device) died.
|
||||
if rebooted:
|
||||
# It took the whole device down: quarantine it, or the next
|
||||
# invocation walks straight back into it.
|
||||
print(f"[run_cts] DEVICE HANG in {in_flight} - quarantining it")
|
||||
hung.append(in_flight)
|
||||
else:
|
||||
crashed.append(in_flight)
|
||||
done.add(in_flight)
|
||||
progressed += 1
|
||||
elif progressed == 0:
|
||||
# Nothing at all came back. Either the first remaining case takes
|
||||
# the process down before the log is flushed, or the device died.
|
||||
# Those look identical from here, so confirm the device is alive
|
||||
# before blaming the test.
|
||||
if not device_alive(args.serial):
|
||||
print(f"[run_cts] device went away during chunk {chunk:04d}", file=sys.stderr)
|
||||
if not wait_for_device(args.serial):
|
||||
print("[run_cts] ABORTING: device never came back. Results are "
|
||||
"incomplete; do NOT treat the remaining cases as crashes.", file=sys.stderr)
|
||||
break
|
||||
print("[run_cts] device recovered; retrying the same chunk")
|
||||
continue
|
||||
|
||||
empty_streak += 1
|
||||
if empty_streak >= args.max_empty_streak:
|
||||
print(f"[run_cts] ABORTING: {empty_streak} consecutive chunks produced no output "
|
||||
f"while the device stayed reachable. Something systemic is wrong; refusing "
|
||||
f"to label the rest of the suite as crashes.", file=sys.stderr)
|
||||
break
|
||||
|
||||
victim = remaining[0]
|
||||
print(f"[run_cts] no output at all; recording {victim} as Crash")
|
||||
crashed.append(victim)
|
||||
done.add(victim)
|
||||
progressed = 1
|
||||
|
||||
remaining = [c for c in remaining if c not in done]
|
||||
elapsed = time.time() - started
|
||||
print(
|
||||
f"[run_cts] chunk {chunk:04d}: +{progressed} (done {len(done)}/{total}, "
|
||||
f"crashes {len(crashed)}, {elapsed / 60:.1f} min)"
|
||||
)
|
||||
chunk += 1
|
||||
|
||||
with open(os.path.join(args.outdir, "crashed.txt"), "w", encoding="utf-8", newline="\n") as fh:
|
||||
fh.write("\n".join(crashed) + ("\n" if crashed else ""))
|
||||
|
||||
# Cases that rebooted the device. Feed this back in via --skip-file to avoid
|
||||
# paying for the same reboot on the next run.
|
||||
with open(os.path.join(args.outdir, "hung.txt"), "w", encoding="utf-8", newline="\n") as fh:
|
||||
fh.write("\n".join(hung) + ("\n" if hung else ""))
|
||||
if hung:
|
||||
print(f"[run_cts] {len(hung)} case(s) hung the device (see hung.txt):")
|
||||
for c in hung:
|
||||
print(f" {c}")
|
||||
|
||||
# Anything still in `remaining` was never measured. Record it so the report
|
||||
# cannot quietly present a partial run as a complete one.
|
||||
with open(os.path.join(args.outdir, "unrun.txt"), "w", encoding="utf-8", newline="\n") as fh:
|
||||
fh.write("\n".join(remaining) + ("\n" if remaining else ""))
|
||||
if skipped:
|
||||
with open(os.path.join(args.outdir, "skipped.txt"), "w", encoding="utf-8", newline="\n") as fh:
|
||||
fh.write("\n".join(skipped) + "\n")
|
||||
|
||||
if remaining:
|
||||
print(f"[run_cts] WARNING: {len(remaining)} cases were never run (see unrun.txt)", file=sys.stderr)
|
||||
print(f"[run_cts] finished: {len(done)}/{total} cases, {len(crashed)} crashes, {chunk} invocations")
|
||||
print(f"[run_cts] qpa chunks in {args.outdir}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,53 @@
|
||||
#!/usr/bin/env python
|
||||
"""Copy the MobileGL dEQP platform port into a VK-GL-CTS checkout.
|
||||
|
||||
The port is version-controlled here, in the MobileGL repo, so it survives a
|
||||
throwaway CTS clone. This drops it into the places VK-GL-CTS expects:
|
||||
|
||||
framework/platform/mobilegl/ <- platform sources
|
||||
targets/mobilegl/mobilegl.cmake <- target definition (-DDEQP_TARGET=mobilegl)
|
||||
|
||||
Usage:
|
||||
python sync_to_cts.py <path-to-VK-GL-CTS>
|
||||
"""
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
CTS_TOOLS = os.path.dirname(HERE)
|
||||
|
||||
COPIES = [
|
||||
(os.path.join(CTS_TOOLS, "platform"), "framework/platform/mobilegl", None),
|
||||
(os.path.join(CTS_TOOLS, "targets"), "targets/mobilegl", ["mobilegl.cmake", "ndk-modern.cmake"]),
|
||||
]
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) != 2:
|
||||
print(__doc__)
|
||||
return 2
|
||||
cts = sys.argv[1]
|
||||
if not os.path.isfile(os.path.join(cts, "CMakeLists.txt")):
|
||||
print(f"error: {cts} does not look like a VK-GL-CTS checkout", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
for src, reldst, only in COPIES:
|
||||
dst = os.path.join(cts, reldst)
|
||||
os.makedirs(dst, exist_ok=True)
|
||||
for name in sorted(os.listdir(src)):
|
||||
if only is not None and name not in only:
|
||||
continue
|
||||
s = os.path.join(src, name)
|
||||
if not os.path.isfile(s):
|
||||
continue
|
||||
shutil.copy2(s, os.path.join(dst, name))
|
||||
print(f" {reldst}/{name}")
|
||||
|
||||
print("\nsynced. configure with -DDEQP_TARGET=mobilegl")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,18 @@
|
||||
# MobileGL conformance-suite skills
|
||||
|
||||
Task-focused skills for running Khronos conformance suites against MobileGL.
|
||||
Each skill is a self-contained package, matching the layout used by
|
||||
`tools/trace_replay/skills/`:
|
||||
|
||||
- `SKILL.md` — the skill (frontmatter `name` + `description`, then the body). The
|
||||
directory name equals the frontmatter `name`.
|
||||
- `agents/openai.yaml` — OpenAI agent descriptor (`display_name`,
|
||||
`short_description`, `default_prompt`).
|
||||
- `scripts/` and/or `references/` — bundled tooling and supporting docs, when the
|
||||
skill has them.
|
||||
|
||||
## Skills
|
||||
|
||||
| Skill | What it does |
|
||||
| --- | --- |
|
||||
| [gl-cts-on-mobilegl](gl-cts-on-mobilegl/SKILL.md) | Build VK-GL-CTS `glcts` as a standalone Android arm64 binary against MobileGL's own EGL, run KHR-GL33, and report a per-backend OpenGL 3.3 core conformance rate. |
|
||||
@@ -0,0 +1,214 @@
|
||||
---
|
||||
name: gl-cts-on-mobilegl
|
||||
description: Run the Khronos OpenGL CTS (VK-GL-CTS glcts, KHR-GL33) against MobileGL on an Android device and compute a per-backend conformance rate. Use when measuring OpenGL 3.3 core conformance for DirectGLES or DirectVulkan, building glcts for Android arm64, porting a dEQP tcu::Platform onto MobileGL, or triaging CTS failures, crashes, and cases that hang the device.
|
||||
---
|
||||
|
||||
# OpenGL CTS on MobileGL (Android)
|
||||
|
||||
## Overview
|
||||
|
||||
`glcts` from VK-GL-CTS is built as a **standalone arm64 executable** and run from
|
||||
`adb shell`. It reaches OpenGL only through `libMobileGL.so`, which supplies both
|
||||
EGL and desktop GL, so a result is unambiguously MobileGL's and never the system
|
||||
GL stack's. No APK and no Activity are involved.
|
||||
|
||||
The port lives in this repository under `MobileGL/tools/cts/` and is copied into
|
||||
a VK-GL-CTS checkout by `scripts/sync_to_cts.py`, so it survives a throwaway CTS
|
||||
clone.
|
||||
|
||||
Set up paths first:
|
||||
|
||||
```sh
|
||||
export MG=<path-to-MobileGL-worktree> # do builds in a worktree, not the shared tree
|
||||
export CTS=<path-to-VK-GL-CTS-checkout>
|
||||
export NDK="$ANDROID_HOME/ndk/27.3.13750724"
|
||||
export SERIAL=<adb-device-serial>
|
||||
```
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Android NDK r27 (the repo builds MobileGL with 27.3.13750724), CMake, Ninja, Python 3.
|
||||
- A rooted-or-not Android device with `adb`; ~600 MB free under `/data/local/tmp`.
|
||||
- **A device you can physically power-cycle.** Some cases hang the GPU hard
|
||||
enough to reboot it — see "Cases that take the device down".
|
||||
- On Windows, invoke `python`, not `python3`: the latter resolves to the
|
||||
Microsoft Store alias stub and exits 49.
|
||||
|
||||
## Step 1 — build libMobileGL.so
|
||||
|
||||
Build in a git worktree (other agents share the main tree). A fresh worktree is
|
||||
missing glslang's bundled SPIR-V Tools, which is a hard configure blocker
|
||||
because `ENABLE_OPT` is forced on:
|
||||
|
||||
```sh
|
||||
cp -r <main-tree>/3rdparty/glslang/External/* "$MG/3rdparty/glslang/External/"
|
||||
./gradlew -p "$MG/android-plugin" :app:assembleTraceRelease
|
||||
```
|
||||
|
||||
The stripped library lands in
|
||||
`android-plugin/app/build/intermediates/stripped_native_libs/traceRelease/.../arm64-v8a/libMobileGL.so`.
|
||||
|
||||
## Step 2 — get VK-GL-CTS and its externals
|
||||
|
||||
Use a **release tag**, not `main`, so the mustpass list — and therefore the
|
||||
reported rate — is citable:
|
||||
|
||||
```sh
|
||||
git -C "$CTS" checkout opengl-cts-4.6.8.1
|
||||
cd "$CTS" && python external/fetch_sources.py
|
||||
```
|
||||
|
||||
## Step 3 — build glcts for Android arm64
|
||||
|
||||
```sh
|
||||
python "$MG/tools/cts/scripts/sync_to_cts.py" "$CTS"
|
||||
|
||||
cmake -S "$CTS" -B build-cts-a64 -G Ninja \
|
||||
-DDEQP_TARGET=mobilegl -DDEQP_TARGET_TOOLCHAIN=ndk-modern \
|
||||
-DANDROID_NDK_PATH="$NDK" -DDE_ANDROID_API=26 -DANDROID_ABI=arm64-v8a \
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
ninja -C build-cts-a64 glcts
|
||||
"$NDK"/toolchains/llvm/prebuilt/*/bin/llvm-strip build-cts-a64/external/openglcts/modules/glcts
|
||||
```
|
||||
|
||||
Confirm the configure output says `DE_OS = DE_OS_ANDROID`, `DE_CPU =
|
||||
DE_CPU_ARM_64` and `DEQP_ANDROID_BUILD = EXE`. Two things make that work and
|
||||
both are easy to get wrong:
|
||||
|
||||
- `DEQP_TARGET_TOOLCHAIN=ndk-modern` is required. dEQP includes `Defs.cmake`
|
||||
*before* the target file, so a target cannot set `DE_OS` itself. Without the
|
||||
toolchain hook the build mis-detects as `DE_OS_UNIX`/`x86_64` and dies on
|
||||
`__assert_fail` (bionic has `__assert2`).
|
||||
- The target sets `DEQP_ANDROID_EXE ON`. Otherwise dEQP builds the modules into
|
||||
the `libdeqp.so` an APK would load and no `glcts` executable exists.
|
||||
|
||||
`KHR-GL33` needs no ungating — the package registry registers it unconditionally;
|
||||
only the `dEQP-*` packages are `#if DE_OS != DE_OS_ANDROID`.
|
||||
|
||||
## Step 4 — deploy
|
||||
|
||||
```sh
|
||||
adb -s $SERIAL shell mkdir -p /data/local/tmp/mgcts
|
||||
adb -s $SERIAL push build-cts-a64/external/openglcts/modules/glcts /data/local/tmp/mgcts/
|
||||
adb -s $SERIAL push build-cts-a64/external/openglcts/modules/gl_cts /data/local/tmp/mgcts/
|
||||
adb -s $SERIAL push <libMobileGL.so> /data/local/tmp/mgcts/
|
||||
adb -s $SERIAL shell chmod 755 /data/local/tmp/mgcts/glcts
|
||||
```
|
||||
|
||||
## Step 5 — preflight
|
||||
|
||||
Never start a multi-hour run without this. It proves the device/library pair
|
||||
yields a 3.3 core context and that FBO readback is correct, in about a second:
|
||||
|
||||
```sh
|
||||
adb -s $SERIAL shell 'cd /data/local/tmp/mgcts && LD_LIBRARY_PATH=. ./mgprobe \
|
||||
--backend DirectVulkan --surface imagereader --lib ./libMobileGL.so'
|
||||
```
|
||||
|
||||
Expect `PASS ... user_fbo=ok`. `default_fb=broken` on DirectVulkan is expected
|
||||
and does not gate — see below.
|
||||
|
||||
## Step 6 — run
|
||||
|
||||
```sh
|
||||
python "$MG/tools/cts/scripts/run_cts.py" \
|
||||
--serial $SERIAL --backend DirectGLES \
|
||||
--caselist .../mustpass/gl/khronos_mustpass/main/gl33-main.txt \
|
||||
--outdir runs/gles --skip-file runs/skip.txt
|
||||
```
|
||||
|
||||
The runner re-invokes `glcts` with only the cases that have no result yet, so a
|
||||
crash costs one case rather than the run. It distinguishes a crashed *case* from
|
||||
a dead *device* by checking the device still answers a shell command — without
|
||||
that check a dead device looks like every remaining case crashing, which yields
|
||||
a completely bogus but plausible-looking conformance number. On a device reboot
|
||||
it waits, re-pulls the partial `.qpa` (which survives on `/data/local/tmp`),
|
||||
records the case that was open as `DeviceHang`, and quarantines it.
|
||||
|
||||
## Step 7 — report
|
||||
|
||||
```sh
|
||||
python "$MG/tools/cts/scripts/qpa_report.py" runs/gles --label DirectGLES
|
||||
```
|
||||
|
||||
Pass rate counts `Pass`, `NotSupported`, `QualityWarning`, `CompatibilityWarning`
|
||||
and `Waiver` as non-failures, matching how Khronos scores a submission; the
|
||||
strict rate counts only `Pass`. Quarantined and never-reached cases are reported
|
||||
separately and excluded from the rates, so a partial run cannot read as a
|
||||
complete one.
|
||||
|
||||
## Required flags, and why
|
||||
|
||||
| Flag | Why it is not optional |
|
||||
| --- | --- |
|
||||
| `--deqp-surface-type=fbo` | On DirectVulkan, `glReadPixels` from the **default framebuffer returns all zeros** with no GL error. dEQP verifies nearly everything through `glReadPixels`, so rendering to the surface scores DirectVulkan near zero for a reason unrelated to conformance. Use it for **both** backends so the two numbers stay comparable. |
|
||||
| `MOBILEGL_CTS_FBO_COLOR_TEXTURE=1` | **`--deqp-surface-type=fbo` alone is not enough.** dEQP's `FboRenderContext` allocates a *renderbuffer* colour attachment, and DirectVulkan returns zeros from a renderbuffer-attached FBO too — only a *texture*-attached FBO reads back correctly. This env var (a patch to `framework/opengl/gluFboRenderContext.cpp`, off by default) switches the attachment to a texture and isolates that single defect. Measured effect: `KHR-GL33.shaders.loops.for_constant_iterations.*` goes 0/62 → 62/62, and the whole-suite DirectVulkan conformance rate goes 46.15% → 72.74%. DirectGLES is bit-identical either way (93.05%), which is the control proving the switch is neutral where readback works. |
|
||||
| `--deqp-terminate-on-device-lost=disable` | Defaults to *enable*, which calls `glGetGraphicsResetStatus()` after every case. That is GL 4.5 / `KHR_robustness`, absent from GL 3.3 core, so the pointer is null and the process segfaults on the first case. Desktop drivers expose the extension, which is why upstream never trips on it. |
|
||||
|
||||
## Cases that take the device down
|
||||
|
||||
Some cases hang the GPU hard enough that the device reboots or stops answering
|
||||
adb entirely. Keep them in a `--skip-file`, and expect to find more:
|
||||
|
||||
- `KHR-GL33.clip_distance.functional` — wedged an Adreno 750 tablet; it rebooted
|
||||
and then stopped responding to adb altogether.
|
||||
- `KHR-GL33.framebuffer_blit.multisampled_to_singlesampled_blit_color_config_test`
|
||||
— rebooted an Adreno 830 phone after 862 cases, on DirectGLES.
|
||||
- `KHR-GL33.framebuffer_blit.multisampled_to_singlesampled_blit_depth_config_test`
|
||||
— same, on both backends (found and quarantined automatically by the runner).
|
||||
- `KHR-GL33.texture_repeat_mode.rgb565_11x131_0_clamp_to_edge` — on DirectVulkan.
|
||||
|
||||
The whole `framebuffer_blit.multisampled_to_singlesampled_*` family is suspect;
|
||||
treat a new variant as a device-hang candidate rather than a normal failure.
|
||||
|
||||
When a run dies, pull `/data/local/tmp/mgcts/chunk.qpa` — it survives the reboot,
|
||||
and the last `#beginTestCaseResult` with no matching `#endTestCaseResult` names
|
||||
the case that did it.
|
||||
|
||||
## Reference results
|
||||
|
||||
`opengl-cts-4.6.8.1`, KHR-GL33 mustpass (`gl33-main.txt`, 9886 cases), Adreno 830
|
||||
/ Android 15, MobileGL `dev`@199164c2, 9884 measured / 0 unrun / 2 quarantined.
|
||||
Conformance rate = Pass + NotSupported, as Khronos scores a submission.
|
||||
|
||||
| backend | conformance | strict Pass | Fail | Crash | InternalError | DeviceHang |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
||||
| DirectGLES | **93.05%** | 85.94% | 679 | 1 | 6 | 1 |
|
||||
| DirectVulkan (texture FBO) | **72.74%** | 65.71% | 2394 | 294 | 5 | 1 |
|
||||
| DirectVulkan (stock renderbuffer FBO) | 46.15% | 39.11% | 5024 | 292 | 5 | 2 |
|
||||
|
||||
The third row is what stock dEQP reports; the gap to the second row is entirely
|
||||
the renderbuffer-FBO readback defect.
|
||||
|
||||
## MobileGL constraints the port works around
|
||||
|
||||
- **DirectVulkan cannot use an EGL pbuffer.** That path needs
|
||||
`VK_EXT_headless_surface`, which Adreno's Android driver does not expose; it
|
||||
fails inside `eglMakeCurrent`. The platform therefore gets a real
|
||||
`ANativeWindow` from **`AImageReader`** — an ordinary BufferQueue producer that
|
||||
`vkCreateAndroidSurfaceKHR` accepts, with no Activity. An `onImageAvailable`
|
||||
listener must drain the queue or the producer blocks once `maxImages` buffers
|
||||
are in flight and the next swap deadlocks.
|
||||
- **`eglMakeCurrent` requires draw == read** and rejects `EGL_NO_SURFACE` with
|
||||
`EGL_BAD_MATCH`, so dEQP's `surfaceless` platform cannot be used at all, and
|
||||
`--deqp-surface-type=fbo` (which asks the platform for `SURFACETYPE_DONT_CARE`)
|
||||
must still be given a real surface.
|
||||
- **Every EGL call must go through the dynamically loaded library.** dEQP's
|
||||
`surfaceless` platform mixes wrapper calls with globally linked `egl*` symbols;
|
||||
copying that on Android silently reaches the system EGL and invalidates the
|
||||
measurement. The `mobilegl` target links no `libEGL`/`libGLESv*` at all.
|
||||
- **Desktop-GL configs need `EGL_OPENGL_BIT`.** The surfaceless port always asks
|
||||
for an ES bit, which can never satisfy a GL 3.3 core context.
|
||||
- MobileGL aborts during static teardown (`FORTIFY: pthread_mutex_lock called on
|
||||
a destroyed mutex`) *after* the work is done; flush and `_exit()` in any small
|
||||
tool, or its exit code and output are lost.
|
||||
|
||||
## Contents
|
||||
|
||||
platform/tcuMobileGLPlatform.{cpp,hpp} dEQP tcu::Platform for MobileGL
|
||||
targets/mobilegl.cmake VK-GL-CTS target (-DDEQP_TARGET=mobilegl)
|
||||
targets/ndk-modern.cmake NDK toolchain hook (sets DE_OS/DE_CPU)
|
||||
probe/mgprobe.c preflight gate
|
||||
scripts/sync_to_cts.py inject the port into a CTS checkout
|
||||
scripts/run_cts.py crash- and reboot-resuming runner
|
||||
scripts/qpa_report.py .qpa -> conformance rate
|
||||
@@ -0,0 +1,4 @@
|
||||
interface:
|
||||
display_name: "OpenGL CTS on MobileGL (Android)"
|
||||
short_description: "Build and run VK-GL-CTS KHR-GL33 against MobileGL and report per-backend conformance"
|
||||
default_prompt: "Use $gl-cts-on-mobilegl to run the OpenGL 3.3 core CTS against MobileGL on my Android device and report the conformance rate for DirectGLES and DirectVulkan."
|
||||
@@ -0,0 +1,36 @@
|
||||
#-------------------------------------------------------------------------
|
||||
# VK-GL-CTS target: MobileGL on Android
|
||||
#
|
||||
# Builds a standalone arm64 ELF that reaches OpenGL exclusively through
|
||||
# libMobileGL.so, loaded at runtime. Nothing here links libEGL or libGLESv*:
|
||||
# the whole point is that the system GL stack must not be reachable, so that a
|
||||
# conformance result is unambiguously MobileGL's.
|
||||
#-------------------------------------------------------------------------
|
||||
|
||||
message("*** Using MobileGL target")
|
||||
|
||||
set(DEQP_TARGET_NAME "MobileGL")
|
||||
|
||||
# Build the modules as standalone executables instead of the libdeqp.so an APK
|
||||
# would load. The suite runs from adb shell, with no Activity.
|
||||
set(DEQP_ANDROID_EXE ON)
|
||||
|
||||
# EGL comes from libMobileGL.so via the eglw dynamic wrapper, so the support
|
||||
# flag is on but no import library is supplied.
|
||||
set(DEQP_SUPPORT_EGL ON)
|
||||
set(DEQP_EGL_LIBRARIES)
|
||||
set(DEQP_GLES2_LIBRARIES)
|
||||
set(DEQP_GLES3_LIBRARIES)
|
||||
|
||||
set(TCUTIL_PLATFORM_SRCS
|
||||
mobilegl/tcuMobileGLPlatform.cpp
|
||||
mobilegl/tcuMobileGLPlatform.hpp
|
||||
)
|
||||
|
||||
find_library(LOG_LIBRARY NAMES log)
|
||||
find_library(ANDROID_LIBRARY NAMES android)
|
||||
find_library(MEDIANDK_LIBRARY NAMES mediandk)
|
||||
|
||||
# libmediandk supplies AImageReader, which is how a process with no Activity
|
||||
# gets a real ANativeWindow.
|
||||
list(APPEND TCUTIL_PLATFORM_LIBS ${ANDROID_LIBRARY} ${MEDIANDK_LIBRARY} ${LOG_LIBRARY})
|
||||
@@ -0,0 +1,61 @@
|
||||
#-------------------------------------------------------------------------
|
||||
# drawElements CMake utilities
|
||||
# ----------------------------
|
||||
#
|
||||
# Copyright 2016 The Android Open Source Project
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
#-------------------------------------------------------------------------
|
||||
|
||||
# Delegate most things to the NDK's cmake toolchain script
|
||||
|
||||
if (NOT DEFINED ANDROID_NDK_PATH)
|
||||
message(FATAL_ERROR "Please provide ANDROID_NDK_PATH")
|
||||
endif ()
|
||||
|
||||
set(ANDROID_PLATFORM "android-${DE_ANDROID_API}")
|
||||
set(ANDROID_STL c++_static)
|
||||
set(ANDROID_CPP_FEATURES "rtti exceptions")
|
||||
|
||||
include(${ANDROID_NDK_PATH}/build/cmake/android.toolchain.cmake)
|
||||
|
||||
# The try_compile() used to verify the C/C++ compilers are sane tries to
|
||||
# generate an executable, but doesn't seem to use the right compiler/linker
|
||||
# options when cross-compiling, so it fails even when building an actual
|
||||
# shared library or executable succeeds.
|
||||
#
|
||||
# I don't know why this doesn't affect simpler projects that use the NDK
|
||||
# toolchain.
|
||||
set(CMAKE_TRY_COMPILE_TARGET_TYPE STATIC_LIBRARY)
|
||||
|
||||
# Set variables used by other parts of dEQP's build scripts
|
||||
|
||||
set(DE_OS "DE_OS_ANDROID")
|
||||
|
||||
if (NOT DEFINED DE_COMPILER)
|
||||
set(DE_COMPILER "DE_COMPILER_CLANG")
|
||||
endif ()
|
||||
|
||||
if (ANDROID_ABI STREQUAL "x86")
|
||||
set(DE_CPU "DE_CPU_X86")
|
||||
elseif (ANDROID_ABI STREQUAL "armeabi" OR
|
||||
ANDROID_ABI STREQUAL "armeabi-v7a")
|
||||
set(DE_CPU "DE_CPU_ARM")
|
||||
elseif (ANDROID_ABI STREQUAL "arm64-v8a")
|
||||
set(DE_CPU "DE_CPU_ARM_64")
|
||||
elseif (ANDROID_ABI STREQUAL "x86_64")
|
||||
set(DE_CPU "DE_CPU_X86_64")
|
||||
else ()
|
||||
message(FATAL_ERROR "Unknown ABI \"${ANDROID_ABI}\"")
|
||||
endif ()
|
||||
@@ -0,0 +1,157 @@
|
||||
# piglit on Android against MobileGL
|
||||
|
||||
Run [piglit](https://gitlab.freedesktop.org/mesa/piglit) desktop-GL tests on a
|
||||
connected Android device with **MobileGL as the OpenGL implementation**, for
|
||||
both backends:
|
||||
|
||||
- `DirectGLES` — MobileGL over the system GLES driver (or ANGLE with
|
||||
`--use-angle`)
|
||||
- `DirectVulkan` — MobileGL over the system Vulkan driver
|
||||
|
||||
No APK and no on-device python: piglit test binaries run as the adb shell user
|
||||
from `/data/local/tmp`, create contexts through waffle's `surfaceless_egl`
|
||||
platform, and waffle is pointed at `libMobileGL.so` (which exports the full
|
||||
`egl*`/`gl*` API under real names).
|
||||
|
||||
## How it fits together
|
||||
|
||||
```
|
||||
piglit test binary (aarch64, bionic)
|
||||
└─ waffle surfaceless_egl (patched)
|
||||
├─ WAFFLE_EGL_LIBRARY=libMobileGL.so → dlopen MobileGL as the EGL impl
|
||||
├─ WAFFLE_GL_LIBRARY=libMobileGL.so → waffle_dl_sym resolves gl* here
|
||||
├─ WAFFLE_FORCE_GL_CONTEXT_VERSION=33core
|
||||
│ upgrades low compat context requests to GL 3.3 core (never
|
||||
│ downgrades) so piglit's supports_gl_compat_version=10 tests run
|
||||
└─ WAFFLE_ANDROID_WINDOW=imagereader (DirectVulkan only)
|
||||
windows are AImageReader ANativeWindows instead of EGL pbuffers,
|
||||
because Android ICDs lack VK_EXT_headless_surface which the
|
||||
MobileGL pbuffer path needs
|
||||
└─ libMobileGL.so
|
||||
├─ DirectGLES: dlopens the real system libEGL.so internally
|
||||
└─ DirectVulkan: links libvulkan.so
|
||||
```
|
||||
|
||||
Key rule: **never name MobileGL `libEGL.so`** anywhere on `LD_LIBRARY_PATH` —
|
||||
the DirectGLES backend loads the system driver with a bare-soname
|
||||
`dlopen("libEGL.so")` and would recursively pick itself up.
|
||||
|
||||
## One-time setup (host: macOS/Linux with the Android NDK)
|
||||
|
||||
```sh
|
||||
WORK=path/to/workdir && cd $WORK
|
||||
git clone --depth 1 https://gitlab.freedesktop.org/mesa/piglit.git
|
||||
git clone --depth 1 https://gitlab.freedesktop.org/mesa/waffle.git
|
||||
git -C waffle apply $MOBILEGL/tools/piglit-android/patches/waffle-mobilegl-android.patch
|
||||
git -C piglit apply $MOBILEGL/tools/piglit-android/patches/piglit-mobilegl-android.patch
|
||||
python3 -m venv venv && ./venv/bin/pip install mako numpy packaging
|
||||
```
|
||||
|
||||
The piglit patch matters beyond build fixes: upstream's
|
||||
`piglit_dispatch_default_init` runs while the waffle framework is still being
|
||||
constructed (`gl_fw` is NULL), so the waffle resolvers were never installed and
|
||||
GL functions bound through the **system** libEGL's `eglGetProcAddress` — every
|
||||
test silently ran on the raw GLES driver instead of MobileGL.
|
||||
|
||||
Build MobileGL for Android:
|
||||
|
||||
```sh
|
||||
cmake -S $MOBILEGL -B $MOBILEGL/build-android-arm64 -G Ninja \
|
||||
-DCMAKE_TOOLCHAIN_FILE=$NDK/build/cmake/android.toolchain.cmake \
|
||||
-DANDROID_ABI=arm64-v8a -DANDROID_PLATFORM=android-26 \
|
||||
-DCMAKE_BUILD_TYPE=RelWithDebInfo
|
||||
cmake --build $MOBILEGL/build-android-arm64 --target MobileGL -j
|
||||
$NDK/toolchains/llvm/prebuilt/*/bin/llvm-strip --strip-unneeded \
|
||||
-o $WORK/libMobileGL-stripped.so $MOBILEGL/build-android-arm64/libMobileGL.so
|
||||
```
|
||||
|
||||
Cross-build waffle (meson; a cross file and a stub `egl.pc` pointing at
|
||||
MobileGL's bundled EGL 1.5 headers are needed — see `cross-example/`):
|
||||
|
||||
```sh
|
||||
cd $WORK/waffle
|
||||
meson setup build-android --cross-file $WORK/cross/android-arm64.ini \
|
||||
-Dbuildtype=release -Dsurfaceless_egl=enabled \
|
||||
-Dglx=disabled -Dx11_egl=disabled -Dgbm=disabled -Dwayland=disabled \
|
||||
-Dbuild-tests=false -Dbuild-examples=false -Dprefix=$WORK/prefix
|
||||
ninja -C build-android && meson install -C build-android
|
||||
```
|
||||
|
||||
Cross-build piglit (needs `PKG_CONFIG_LIBDIR` with the installed `waffle-1.pc`
|
||||
plus the stub `egl.pc`):
|
||||
|
||||
```sh
|
||||
cd $WORK/piglit && export PKG_CONFIG_LIBDIR=$WORK/prefix/lib/pkgconfig:$WORK/cross/pkgconfig
|
||||
cmake -S . -B build-android -G Ninja \
|
||||
-DCMAKE_TOOLCHAIN_FILE=$NDK/build/cmake/android.toolchain.cmake \
|
||||
-DANDROID_ABI=arm64-v8a -DANDROID_PLATFORM=android-26 \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DPIGLIT_USE_WAFFLE=ON -DPIGLIT_BUILD_GL_TESTS=ON \
|
||||
-DPIGLIT_BUILD_GLES1_TESTS=OFF -DPIGLIT_BUILD_GLES2_TESTS=OFF \
|
||||
-DPIGLIT_BUILD_GLES3_TESTS=OFF -DPIGLIT_BUILD_EGL_TESTS=OFF \
|
||||
-DPIGLIT_BUILD_GLX_TESTS=OFF -DPIGLIT_BUILD_WGL_TESTS=OFF \
|
||||
-DPIGLIT_BUILD_CL_TESTS=OFF -DPIGLIT_BUILD_VK_TESTS=OFF \
|
||||
-DPIGLIT_BUILD_DMA_BUF_TESTS=OFF -DPIGLIT_USE_GBM=OFF \
|
||||
-DPIGLIT_USE_WAYLAND=OFF -DPIGLIT_USE_X11=OFF \
|
||||
-DPYTHON_EXECUTABLE=$WORK/venv/bin/python \
|
||||
-DOPENGL_INCLUDE_DIR=$MOBILEGL/include \
|
||||
-DOPENGL_gl_LIBRARY=$SYSROOT/usr/lib/aarch64-linux-android/26/libEGL.so \
|
||||
-DGLEXT_INCLUDE_DIR=$MOBILEGL/include
|
||||
ninja -C build-android
|
||||
```
|
||||
|
||||
## Selecting tests
|
||||
|
||||
Enumerate on the host with piglit's own profiles (no device needed):
|
||||
|
||||
```sh
|
||||
cd $WORK/piglit
|
||||
for prof in opengl shader glslparser; do
|
||||
PIGLIT_BUILD_DIR=$PWD/build-android ./venv/bin/python ./piglit print-cmd \
|
||||
-t "spec@!opengl 3[.]" -t "spec@glsl-3[.]30" $prof
|
||||
done > /tmp/gl33.list
|
||||
```
|
||||
|
||||
Group names use `@` separators (`spec@!opengl 3.3@minmax`). The version groups
|
||||
(`spec@!opengl 1.x…3.3`), GLSL groups (`spec@glsl-1.10…3.30`) plus the ARB
|
||||
extension groups folded into GL 3.1–3.3 core give a comprehensive "GL 3.3
|
||||
core" suite (~15k tests).
|
||||
|
||||
## Running
|
||||
|
||||
```sh
|
||||
python3 $MOBILEGL/tools/piglit-android/run_piglit_android.py \
|
||||
--piglit-root $WORK/piglit --list /tmp/gl33.list \
|
||||
--backend DirectGLES \
|
||||
--mobilegl-lib $WORK/libMobileGL-stripped.so \
|
||||
--waffle-lib $WORK/waffle/build-android/src/waffle/libwaffle-1.so \
|
||||
--out results-gles
|
||||
# then the same with --backend DirectVulkan --out results-vk
|
||||
```
|
||||
|
||||
The runner pushes binaries/libs/data (incremental; `--repush` forces), executes
|
||||
tests serially in chunked on-device shell scripts under `timeout`, parses the
|
||||
`PIGLIT: {...}` result lines, and writes `results.json` + `summary.txt` +
|
||||
`raw.log`. Exit-code semantics: parsed result wins; nonzero exit without a
|
||||
result line = `crash`; toybox timeout exits = `timeout`.
|
||||
|
||||
Quick sanity check for the whole stack (waffle build also produces `wflinfo`):
|
||||
|
||||
```sh
|
||||
adb shell 'cd /data/local/tmp/piglit-mgl && env LD_LIBRARY_PATH=$PWD/lib \
|
||||
WAFFLE_EGL_LIBRARY=libMobileGL.so WAFFLE_GL_LIBRARY=libMobileGL.so \
|
||||
MOBILEGL_BACKEND_TYPE=DirectVulkan WAFFLE_ANDROID_WINDOW=imagereader \
|
||||
./wflinfo --platform surfaceless_egl --api gl --version 3.3 --profile core'
|
||||
```
|
||||
|
||||
Expect `OpenGL version string: 3.3.0 MobileGL …, Direct (Vulkan) Backend`.
|
||||
|
||||
## Known caveats
|
||||
|
||||
- MSAA winsys configs never match (MobileGL exposes two RGBA8888 configs,
|
||||
samples=0); MSAA FBO tests are unaffected.
|
||||
- Tests that genuinely require compatibility-profile features will fail on the
|
||||
forced 3.3 core context; that is honest for a core-only implementation.
|
||||
- `eglTerminate` at test exit now tears MobileGL down deterministically (see
|
||||
the EGL-lifecycle refactor); a device-side `mobilegl.log` is written per run
|
||||
directory for debugging.
|
||||
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env python3
|
||||
# MobileGL - tools/piglit-android/compare_results.py
|
||||
# Copyright (c) 2025-2026 MobileGL-Dev
|
||||
# Licensed under the GNU Lesser General Public License v3.0:
|
||||
# https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
# https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
# SPDX-License-Identifier: LGPL-3.0-only
|
||||
# End of Source File Header
|
||||
"""Compare two run_piglit_android.py results.json files (e.g. DirectGLES vs
|
||||
DirectVulkan) and write a markdown report of totals plus categorized diffs."""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
BAD = ('crash', 'timeout', 'fail', 'missing', 'notrun', 'warn')
|
||||
|
||||
|
||||
def load(path):
|
||||
d = json.loads(Path(path).read_text())
|
||||
return d
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument('a', help='first results.json')
|
||||
ap.add_argument('b', help='second results.json')
|
||||
ap.add_argument('-o', '--out', help='markdown output path')
|
||||
args = ap.parse_args()
|
||||
|
||||
da, db = load(args.a), load(args.b)
|
||||
na, nb = da['backend'], db['backend']
|
||||
ta, tb = da['tests'], db['tests']
|
||||
names = sorted(set(ta) | set(tb))
|
||||
|
||||
lines = [f'# piglit: {na} vs {nb}', '']
|
||||
lines.append(f'| | {na} | {nb} |')
|
||||
lines.append('|---|---|---|')
|
||||
ca = Counter(o["status"] for o in ta.values())
|
||||
cb = Counter(o["status"] for o in tb.values())
|
||||
for k in sorted(set(ca) | set(cb)):
|
||||
lines.append(f'| {k} | {ca.get(k, 0)} | {cb.get(k, 0)} |')
|
||||
lines.append(f'| total | {len(ta)} | {len(tb)} |')
|
||||
lines.append(f'| elapsed | {da.get("elapsed_sec")}s | {db.get("elapsed_sec")}s |')
|
||||
lines.append('')
|
||||
|
||||
def bucket(pred, title):
|
||||
rows = [n for n in names
|
||||
if pred(ta.get(n, {}).get('status', 'absent'),
|
||||
tb.get(n, {}).get('status', 'absent'))]
|
||||
if rows:
|
||||
lines.append(f'## {title} ({len(rows)})')
|
||||
lines.append('')
|
||||
for n in rows:
|
||||
sa = ta.get(n, {}).get('status', 'absent')
|
||||
sb = tb.get(n, {}).get('status', 'absent')
|
||||
lines.append(f'- `{n}` — {na}: {sa}, {nb}: {sb}')
|
||||
lines.append('')
|
||||
|
||||
bucket(lambda a, b: a in BAD and b in BAD, 'Bad on both (likely frontend/state-tracker)')
|
||||
bucket(lambda a, b: a in BAD and b == 'pass', f'Bad only on {na}')
|
||||
bucket(lambda a, b: a == 'pass' and b in BAD, f'Bad only on {nb}')
|
||||
bucket(lambda a, b: a == 'skip' and b == 'pass' or a == 'pass' and b == 'skip',
|
||||
'Skip on one side only')
|
||||
|
||||
text = '\n'.join(lines) + '\n'
|
||||
if args.out:
|
||||
Path(args.out).write_text(text)
|
||||
print(f'wrote {args.out}')
|
||||
else:
|
||||
print(text)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,19 @@
|
||||
; meson cross file for waffle -> aarch64 Android
|
||||
; Replace NDK_TOOLCHAIN with e.g.
|
||||
; $HOME/Library/Android/sdk/ndk/27.3.13750724/toolchains/llvm/prebuilt/darwin-x86_64
|
||||
; and PKGCONFIG_DIR with the directory holding the stub egl.pc.
|
||||
[binaries]
|
||||
c = 'NDK_TOOLCHAIN/bin/aarch64-linux-android26-clang'
|
||||
cpp = 'NDK_TOOLCHAIN/bin/aarch64-linux-android26-clang++'
|
||||
ar = 'NDK_TOOLCHAIN/bin/llvm-ar'
|
||||
strip = 'NDK_TOOLCHAIN/bin/llvm-strip'
|
||||
pkg-config = '/usr/bin/pkg-config'
|
||||
|
||||
[host_machine]
|
||||
system = 'android'
|
||||
cpu_family = 'aarch64'
|
||||
cpu = 'aarch64'
|
||||
endian = 'little'
|
||||
|
||||
[properties]
|
||||
pkg_config_libdir = 'PKGCONFIG_DIR'
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user