mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-09 12:48:32 +09:00
Compare commits
67
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3223ecb14e | ||
|
|
fc4cd980f2 | ||
|
|
992d16267c | ||
|
|
0ea9e6de5f | ||
|
|
241ed377b4 | ||
|
|
bf312a4b67 | ||
|
|
56b31a9587 | ||
|
|
8a0a8a0274 | ||
|
|
d076c29146 | ||
|
|
930a607bdf | ||
|
|
34685b4bb0 | ||
|
|
c540fb88ee | ||
|
|
6ae3245a0d | ||
|
|
7e048fc2bf | ||
|
|
83cdfd6bdd | ||
|
|
1c76f886cf | ||
|
|
a2e109beff | ||
|
|
63f0756644 | ||
|
|
450215d12c | ||
|
|
3a9e520170 | ||
|
|
d2996ba1cf | ||
|
|
c8632dfefe | ||
|
|
b8a8a660e1 | ||
|
|
7ab83861ca | ||
|
|
eaeba556a3 | ||
|
|
72fa1221a5 | ||
|
|
c4254c4bbd | ||
|
|
199164c2e0 | ||
|
|
e87063e90c | ||
|
|
122da27249 | ||
|
|
bc2d698b3e | ||
|
|
3049c4b82b | ||
|
|
2b3850b76b | ||
|
|
2a0ae743a0 | ||
|
|
6839219c10 | ||
|
|
52ddb440ca | ||
|
|
79feeffd25 | ||
|
|
202037b5a3 | ||
|
|
bce9c48c8e | ||
|
|
b6a7807a3a | ||
|
|
48ba622387 | ||
|
|
05260d1262 | ||
|
|
6eb5ff51c5 | ||
|
|
e526f8e8ac | ||
|
|
e724e88eec | ||
|
|
3b175fb88a | ||
|
|
b5a4e7075a | ||
|
|
520c2b6750 | ||
|
|
57cc652b1d | ||
|
|
65ea54da9e | ||
|
|
c81dd04f08 | ||
|
|
c158bfa584 | ||
|
|
f9f455144c | ||
|
|
64e4840de2 | ||
|
|
bf7b5755cc | ||
|
|
293f64b3c2 | ||
|
|
f0cc07c937 | ||
|
|
4658536652 | ||
|
|
04b4627c65 | ||
|
|
68e13705c8 | ||
|
|
e5388c0e7e | ||
|
|
8bc4808b1a | ||
|
|
5d8a5387e2 | ||
|
|
aa2184e47a | ||
|
|
9152a4a4bc | ||
|
|
1963b427db | ||
|
|
f3def150e7 |
@@ -9,7 +9,23 @@ fi
|
||||
case_name="$1"
|
||||
fixture_dir="${2:-tools/trace_replay/fixtures}"
|
||||
python_bin="${PYTHON:-python3}"
|
||||
mirror_base="${MOBILEGL_TRACE_FIXTURE_MIRROR_BASE:-https://repo.miawa.cn/mgl/tools/trace_replay/fixtures}"
|
||||
# Fixture mirrors, tried in order before falling back to Git LFS. Override the
|
||||
# whole list with MOBILEGL_TRACE_FIXTURE_MIRROR_BASES (whitespace separated);
|
||||
# MOBILEGL_TRACE_FIXTURE_MIRROR_BASE still works and is tried first.
|
||||
default_mirror_bases=(
|
||||
"https://git.hit.moe/swung0x48/MobileGL/media/branch/dev/tools/trace_replay/fixtures"
|
||||
"https://repo.miawa.cn/mgl/tools/trace_replay/fixtures"
|
||||
)
|
||||
if [ -n "${MOBILEGL_TRACE_FIXTURE_MIRROR_BASES:-}" ]; then
|
||||
read -r -a mirror_bases <<< "${MOBILEGL_TRACE_FIXTURE_MIRROR_BASES}"
|
||||
else
|
||||
mirror_bases=("${default_mirror_bases[@]}")
|
||||
fi
|
||||
if [ -n "${MOBILEGL_TRACE_FIXTURE_MIRROR_BASE:-}" ]; then
|
||||
mirror_bases=("${MOBILEGL_TRACE_FIXTURE_MIRROR_BASE}" "${mirror_bases[@]}")
|
||||
fi
|
||||
# Optional bearer token for mirrors that require authentication (private Gitea).
|
||||
mirror_token="${MOBILEGL_TRACE_FIXTURE_MIRROR_TOKEN:-}"
|
||||
download_attempts="${MOBILEGL_TRACE_FIXTURE_DOWNLOAD_ATTEMPTS:-5}"
|
||||
retry_delay="${MOBILEGL_TRACE_FIXTURE_RETRY_DELAY:-2}"
|
||||
|
||||
@@ -30,7 +46,8 @@ fixture_list="$("${python_bin}" tools/trace_replay/trace_cases.py \
|
||||
--format fixture-files \
|
||||
--case "${case_name}" \
|
||||
--fixture-root "${fixture_dir}")"
|
||||
mapfile -t files <<< "${fixture_list}"
|
||||
# Strip CR so the script also works when python emits CRLF (Git Bash on Windows).
|
||||
mapfile -t files < <(printf '%s\n' "${fixture_list}" | tr -d '\r')
|
||||
|
||||
include="$(IFS=,; echo "${files[*]}")"
|
||||
if [ "${case_name}" = "OpenRA" ]; then
|
||||
@@ -106,6 +123,7 @@ fetch_file_from_mirror() {
|
||||
local attempt
|
||||
local partial_size
|
||||
local curl_status
|
||||
local curl_auth
|
||||
|
||||
metadata="$(get_lfs_metadata "${file}")" || return 1
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
@@ -136,7 +154,11 @@ fetch_file_from_mirror() {
|
||||
echo "Starting mirror download for ${file} (attempt ${attempt}/${download_attempts})"
|
||||
fi
|
||||
|
||||
if curl -L --fail --show-error --continue-at - --output "${tmp_file}" "${url}"; then
|
||||
curl_auth=()
|
||||
if [ -n "${mirror_token}" ]; then
|
||||
curl_auth=(--header "Authorization: token ${mirror_token}")
|
||||
fi
|
||||
if curl -L --fail --show-error --continue-at - "${curl_auth[@]}" --output "${tmp_file}" "${url}"; then
|
||||
if verify_fixture_file "${tmp_file}" "${file}" "${expected_oid}" "${expected_size}"; then
|
||||
mv "${tmp_file}" "${file}"
|
||||
return 0
|
||||
@@ -184,10 +206,19 @@ fetch_from_mirror() {
|
||||
for file in "${files[@]}"; do
|
||||
local name
|
||||
local url
|
||||
local base
|
||||
local fetched=0
|
||||
name="$(basename "${file}")"
|
||||
url="${mirror_base%/}/${name}"
|
||||
echo "Fetching trace fixture from mirror: ${url}"
|
||||
if ! fetch_file_from_mirror "${file}" "${url}"; then
|
||||
for base in "${mirror_bases[@]}"; do
|
||||
url="${base%/}/${name}"
|
||||
echo "Fetching trace fixture from mirror: ${url}"
|
||||
if fetch_file_from_mirror "${file}" "${url}"; then
|
||||
fetched=1
|
||||
break
|
||||
fi
|
||||
echo "Mirror did not serve ${name}; trying the next mirror" >&2
|
||||
done
|
||||
if [ "${fetched}" -ne 1 ]; then
|
||||
return 1
|
||||
fi
|
||||
done
|
||||
@@ -196,7 +227,7 @@ fetch_from_mirror() {
|
||||
if fetch_from_mirror; then
|
||||
echo "Fetched trace fixture files for ${case_name} from mirror: ${include}"
|
||||
else
|
||||
echo "Mirror fetch failed for ${case_name}; falling back to Git LFS: ${include}"
|
||||
echo "All mirrors failed for ${case_name}; falling back to Git LFS: ${include}"
|
||||
git lfs install --local
|
||||
git lfs pull --include="${include}" --exclude=""
|
||||
fi
|
||||
|
||||
@@ -455,6 +455,15 @@ jobs:
|
||||
if [ '${{ matrix.backend }}' = 'DirectVulkan' ]; then
|
||||
export MOBILEGL_MAGMA_R11G11B10F_FALLBACK=1
|
||||
fi
|
||||
# The blended depth-write quirk auto-enables only on Qualcomm, which no CI
|
||||
# runner has, so force it on for the OIT case it exists to fix. ForceOn
|
||||
# bypasses only the vendor gate, so this exercises the real strip on
|
||||
# lavapipe. The Android AVD lane deliberately leaves it off, keeping the
|
||||
# unstripped path covered for the same trace.
|
||||
if [ '${{ matrix.backend }}' = 'DirectVulkan' ] \
|
||||
&& [ '${{ matrix.case }}' = 'improved-transparency-minecraft-26.3' ]; then
|
||||
export MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE=1
|
||||
fi
|
||||
ctest -V --no-tests=error -R '^MobileGLTraceReplay\.${{ matrix.case }}\.${{ matrix.backend }}$'
|
||||
|
||||
- name: Upload actual image
|
||||
|
||||
+45
-1
@@ -188,9 +188,12 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EliminateFloatEqualsZeroPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RenameSamplerFunctionParameterPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecomposeWorkgroupVec3Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecoratePositionInvariantPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerDrawParametersPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUboMemberRelaxedPrecisionPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripNoPerspectivePass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateNoPerspectivePass.cpp
|
||||
|
||||
MobileGL/MG_Util/BackendLoaders/OpenGL/Loader.cpp
|
||||
MobileGL/MG_Util/BackendLoaders/Vulkan/Loader.cpp
|
||||
@@ -300,6 +303,13 @@ if (ANDROID)
|
||||
)
|
||||
endif()
|
||||
|
||||
if (WIN32)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Impl/WGLImpl/WGLImpl.cpp
|
||||
MobileGL/MG_Impl/WGLImpl/Exporting/Definitions.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
set(MOBILEGL_LINK_LIBRARIES
|
||||
glslang::glslang
|
||||
spirv-cross-c
|
||||
@@ -328,10 +338,18 @@ set(MOBILEGL_INCLUDE_DIR
|
||||
${SPIRV-Headers_SOURCE_DIR}/include
|
||||
)
|
||||
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
${SOURCE_FILES}
|
||||
)
|
||||
|
||||
if (WIN32)
|
||||
# The wgl* entry points are exported via .def (see the comment in wgl.def);
|
||||
# only the shared library links it.
|
||||
target_sources(${CMAKE_PROJECT_NAME} PRIVATE
|
||||
MobileGL/MG_Impl/WGLImpl/Exporting/wgl.def
|
||||
)
|
||||
endif()
|
||||
|
||||
if (CMAKE_BUILD_TYPE STREQUAL "Debug")
|
||||
set_target_properties(${CMAKE_PROJECT_NAME} PROPERTIES
|
||||
C_VISIBILITY_PRESET default
|
||||
@@ -374,6 +392,18 @@ if(UNIX AND NOT APPLE AND NOT ANDROID)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if(WIN32)
|
||||
# Drop-in for the classic GL loader path: a copy named opengl32.dll placed
|
||||
# next to a host executable is what LoadLibrary("opengl32.dll") and gdi32's
|
||||
# pixel-format forwarding will resolve.
|
||||
add_custom_command(TARGET ${CMAKE_PROJECT_NAME} POST_BUILD
|
||||
COMMAND ${CMAKE_COMMAND} -E copy_if_different
|
||||
"$<TARGET_FILE:${CMAKE_PROJECT_NAME}>"
|
||||
"$<TARGET_FILE_DIR:${CMAKE_PROJECT_NAME}>/opengl32.dll"
|
||||
COMMENT "Creating opengl32.dll drop-in copy"
|
||||
)
|
||||
endif()
|
||||
|
||||
if(NOT ANDROID)
|
||||
add_library(${CMAKE_PROJECT_NAME}_s STATIC
|
||||
${SOURCE_FILES}
|
||||
@@ -425,8 +455,21 @@ if (ANDROID)
|
||||
endif()
|
||||
|
||||
if (APPLE AND NOT MOBILEGL_IOS)
|
||||
# MobileGL statically embeds glslang, SPIRV-Tools, and SPIRV-Cross. When
|
||||
# this dylib is injected with DYLD_INSERT_LIBRARIES, exporting those C++
|
||||
# symbols interposes incompatible copies embedded by host libraries such
|
||||
# as shaderc. Keep only the public GL/EGL/CGL loader surface globally
|
||||
# visible; GetProcAddress can still return pointers to hidden internals.
|
||||
set(MOBILEGL_MACOS_EXPORTED_SYMBOLS
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/MobileGL/MG_Impl/DyldInterpose/ExportedSymbols.txt")
|
||||
target_link_options(${CMAKE_PROJECT_NAME} PRIVATE
|
||||
"LINKER:-exported_symbols_list,${MOBILEGL_MACOS_EXPORTED_SYMBOLS}")
|
||||
set_property(TARGET ${CMAKE_PROJECT_NAME} APPEND PROPERTY
|
||||
LINK_DEPENDS "${MOBILEGL_MACOS_EXPORTED_SYMBOLS}")
|
||||
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME} PUBLIC
|
||||
"-framework Cocoa"
|
||||
"-framework CoreVideo"
|
||||
"-framework QuartzCore"
|
||||
"-framework Foundation"
|
||||
"-framework OpenGL"
|
||||
@@ -434,6 +477,7 @@ if (APPLE AND NOT MOBILEGL_IOS)
|
||||
if(TARGET ${CMAKE_PROJECT_NAME}_s)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME}_s PUBLIC
|
||||
"-framework Cocoa"
|
||||
"-framework CoreVideo"
|
||||
"-framework QuartzCore"
|
||||
"-framework Foundation"
|
||||
"-framework OpenGL"
|
||||
|
||||
@@ -20,6 +20,15 @@ namespace MobileGL::MG_Config {
|
||||
|
||||
extern BackendType ActiveBackendType;
|
||||
|
||||
// Tri-state override for device-specific quirks: Auto lets the detected device decide,
|
||||
// ForceOn/ForceOff bypass the detection in either direction. ForceOn only bypasses the
|
||||
// device gate - each quirk keeps its structural safety checks.
|
||||
enum class QuirkOverride : Uint8 {
|
||||
Auto = 0,
|
||||
ForceOn,
|
||||
ForceOff,
|
||||
};
|
||||
|
||||
// Feature toggles parsed once from environment variables in MG_ConfigLoader::Init()
|
||||
// (ConfigLoader.cpp), before the accepted-env map is destroyed. All Bool fields share
|
||||
// one truthy rule: the variable is set, non-empty, not "0", and not "false"
|
||||
@@ -67,6 +76,21 @@ namespace MobileGL::MG_Config {
|
||||
// explicitly request a core profile via EGL_CONTEXT_OPENGL_PROFILE_MASK / a >=3.1
|
||||
// version request.
|
||||
Bool RelaxedSemantics = false;
|
||||
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that
|
||||
// rewrites the recognized workgroup prefix-scan template on Qualcomm devices with
|
||||
// subgroups wider than 32 lanes (see ShaderSourceProcessor's quirk registry).
|
||||
QuirkOverride SubgroupPrefixScanQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE: overrides the DirectVulkan quirk that
|
||||
// strips depth writes from accumulation-blended pipelines (MIN/MAX or additive
|
||||
// ONE+ONE - the multi-pass depth-equality signature) on drivers without
|
||||
// cross-pipeline vertex position invariance. Sorted-transparency "over" blends,
|
||||
// gl_FragDepth writers, and fully color-masked attachments are exempt (see
|
||||
// PipelineFactory::ShouldSuppressDepthWrite). Auto detects Qualcomm.
|
||||
QuirkOverride MagmaDisableBlendedDepthWriteQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS: leave the Vulkan robustBufferAccess device
|
||||
// feature off. It is enabled by default to match GL's defined out-of-range fetch
|
||||
// behavior; this escape hatch exists to measure or dodge its GPU cost on a device.
|
||||
Bool DisableRobustBufferAccess = false;
|
||||
};
|
||||
extern FeaturesTable Features;
|
||||
} // namespace MobileGL::MG_Config
|
||||
|
||||
@@ -86,6 +86,17 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
return it != acceptedEnvVariablesMap->end() && IsTruthyValue(it->second);
|
||||
}
|
||||
|
||||
// Quirk overrides are tri-state: an unset variable keeps device auto-detection, a truthy
|
||||
// value forces the quirk on, anything else set ("0", "false", "") forces it off.
|
||||
inline MG_Config::QuirkOverride QueryEnvQuirkOverride(const String& key) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
return MG_Config::QuirkOverride::Auto;
|
||||
}
|
||||
return IsTruthyValue(it->second) ? MG_Config::QuirkOverride::ForceOn
|
||||
: MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
|
||||
inline Uint32 QueryEnvUint32(const String& key, Uint32 defaultValue, Uint32 minValue, Uint32 maxValue) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
@@ -123,6 +134,10 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
||||
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||
features.SubgroupPrefixScanQuirk = QueryEnvQuirkOverride("MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN");
|
||||
features.MagmaDisableBlendedDepthWriteQuirk =
|
||||
QueryEnvQuirkOverride("MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE");
|
||||
features.DisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS");
|
||||
}
|
||||
|
||||
inline void InitBackendType() {
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
#define MOBILEGL_EGL_API MOBILEGL_API
|
||||
#define MOBILEGL_CGL_API MOBILEGL_API
|
||||
#define MOBILEGL_NSOPENGL_API MOBILEGL_API
|
||||
#define MOBILEGL_WGL_API MOBILEGL_API
|
||||
|
||||
// ====================== MobileGL configurations ======================= //
|
||||
#ifndef MOBILEGL_LOG_ACTIVE_LEVEL
|
||||
|
||||
@@ -14,7 +14,13 @@ namespace MobileGL {
|
||||
} // namespace MG_Config
|
||||
|
||||
namespace MG_Backend {
|
||||
UniquePtr<BackendObject> pActiveBackendObject;
|
||||
// Leak-at-exit storage: the UniquePtr itself lives on the heap and is
|
||||
// never destroyed by the runtime, so process exit runs no backend
|
||||
// destructors (static destruction order across TUs is undefined).
|
||||
// Deterministic teardown happens inside the EGL lifecycle instead:
|
||||
// the last eglTerminate calls MobileGL::Destroy(), which .reset()s
|
||||
// these singletons while the process is still healthy.
|
||||
UniquePtr<BackendObject>& pActiveBackendObject = *new UniquePtr<BackendObject>();
|
||||
GlobalBackendFunctionsTable gBackendFunctionsTable;
|
||||
} // namespace MG_Backend
|
||||
} // namespace MobileGL
|
||||
|
||||
+47
-33
@@ -9,14 +9,25 @@
|
||||
#include "Init.h"
|
||||
#include "Config.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Backend/DirectVulkan/DirectVulkan.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <MG_Impl/GLImpl/Texture/ProxyTexture.h>
|
||||
#include <MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h>
|
||||
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace {
|
||||
Bool g_isInitialized = false;
|
||||
std::atomic<Bool> g_isInitialized = false;
|
||||
thread_local Bool tl_initializing = false;
|
||||
|
||||
std::mutex& InitMutex() {
|
||||
static std::mutex mutex;
|
||||
return mutex;
|
||||
}
|
||||
|
||||
void DestroyImpl(Bool logLifecycle) {
|
||||
if (!g_isInitialized) {
|
||||
@@ -27,6 +38,12 @@ namespace MobileGL {
|
||||
MGLOG_I("MobileGL closing...");
|
||||
}
|
||||
glslang::FinalizeProcess();
|
||||
// GL syncs die with their contexts, and every context is gone by the
|
||||
// time full teardown runs: drain the live-sync registry while the
|
||||
// backend function table can still release the backend handles (and
|
||||
// before a re-initialized library could pair them with the wrong
|
||||
// backend's DeleteSync).
|
||||
MG_Impl::GLImpl::DestroyAllSyncObjects();
|
||||
MG_Backend::pActiveBackendObject.reset();
|
||||
MG_State::pGLContext.reset();
|
||||
MG_State::pEGLContext.reset();
|
||||
@@ -64,40 +81,37 @@ namespace MobileGL {
|
||||
MGLOG_I("MobileGL initialized");
|
||||
}
|
||||
|
||||
void EnsureInitialized() {
|
||||
if (g_isInitialized.load(std::memory_order_acquire)) {
|
||||
return;
|
||||
}
|
||||
// Re-entrant call while this thread is already inside Initialize()
|
||||
// (e.g. an init step routing back through a public entry point).
|
||||
if (tl_initializing) {
|
||||
return;
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(InitMutex());
|
||||
if (g_isInitialized.load(std::memory_order_acquire)) {
|
||||
return;
|
||||
}
|
||||
tl_initializing = true;
|
||||
Initialize();
|
||||
tl_initializing = false;
|
||||
}
|
||||
|
||||
void Destroy() {
|
||||
DestroyImpl(true);
|
||||
}
|
||||
|
||||
#if defined(__linux__) || defined(__APPLE__)
|
||||
__attribute__((constructor)) static void AutoInit() {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
__attribute__((destructor)) static void AutoDestroy() {
|
||||
if (MG_Config::Features.TraceSkipAutodestroy) {
|
||||
return;
|
||||
}
|
||||
#if defined(__APPLE__)
|
||||
// macOS injected dylibs can run destructors after logging/backend static state is already torn down.
|
||||
return;
|
||||
#else
|
||||
DestroyImpl(false);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef _WIN32
|
||||
BOOL WINAPI DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved) {
|
||||
switch (ul_reason_for_call) {
|
||||
case DLL_PROCESS_ATTACH:
|
||||
Initialize();
|
||||
break;
|
||||
|
||||
case DLL_PROCESS_DETACH:
|
||||
Destroy();
|
||||
break;
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
#endif
|
||||
// MobileGL's lifecycle is owned entirely by the host-API layers
|
||||
// (EGL/WGL/CGL): initialization happens lazily on the first entry point
|
||||
// via EnsureInitialized(), and full teardown happens deterministically
|
||||
// when the last EGL display is terminated with nothing current (EGLImpl
|
||||
// calls Destroy()). There is intentionally no backend-initializing static
|
||||
// constructor, no static destructor, and no DllMain: the global singletons
|
||||
// use leak-at-exit storage (see GlobalObjects.cpp), so a process that exits
|
||||
// without eglTerminate simply leaks them to the OS instead of running
|
||||
// backend destructors during static teardown. macOS has a lightweight
|
||||
// dyld constructor that installs NSOpenGL dispatch hooks only; full backend
|
||||
// initialization still enters here from the first hooked CGL context.
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -11,6 +11,13 @@
|
||||
|
||||
namespace MobileGL {
|
||||
void Initialize();
|
||||
// Thread-safe, idempotent, and re-entrant wrapper around Initialize().
|
||||
// Host layers (EGL/WGL/CGL entry points) call this lazily on first use so
|
||||
// full backend initialization never depends on ELF/DLL static constructors,
|
||||
// and so a fresh init can follow a full Destroy() (e.g. after the last
|
||||
// eglTerminate). The macOS dyld bootstrap installs only lightweight
|
||||
// NSOpenGL method hooks.
|
||||
void EnsureInitialized();
|
||||
void Destroy();
|
||||
|
||||
namespace MG_Util::Debug {
|
||||
|
||||
@@ -230,6 +230,21 @@ namespace MobileGL {
|
||||
void (*SetSwapInterval)(Int interval);
|
||||
};
|
||||
|
||||
// Coarse GPU vendor identity for gating device-specific quirks. Detected from the
|
||||
// Vulkan physical-device vendorID or the GLES GL_VENDOR/GL_RENDERER strings; stays
|
||||
// Unknown when detection is inconclusive, in which case auto-gated quirks stay off.
|
||||
enum class GpuVendorKind : Uint8 {
|
||||
Unknown = 0,
|
||||
Qualcomm,
|
||||
Arm,
|
||||
Nvidia,
|
||||
Amd,
|
||||
Intel,
|
||||
ImgTec,
|
||||
// Software rasterizers (llvmpipe/lavapipe, SwiftShader).
|
||||
Software,
|
||||
};
|
||||
|
||||
struct DynamicBackendParameters {
|
||||
SizeT UniformBufferOffsetAlignment = 256;
|
||||
// GL_MAX_TEXTURE_MAX_ANISOTROPY_EXT. 1.0 means the backend cannot filter anisotropically,
|
||||
@@ -272,6 +287,9 @@ namespace MobileGL {
|
||||
Int MaxUniformBlockSize = 16384;
|
||||
Int MaxImageUnits = 8;
|
||||
Int MaxCombinedImageUniforms = 8;
|
||||
Int MaxVertexImageUniforms = 0;
|
||||
Int MaxGeometryImageUniforms = 0;
|
||||
Int MaxFragmentImageUniforms = 8;
|
||||
Int MaxComputeImageUniforms = 8;
|
||||
Int MaxDrawBuffers = 8;
|
||||
Int MaxColorAttachments = 8;
|
||||
@@ -288,13 +306,15 @@ namespace MobileGL {
|
||||
Uint32 SubgroupSupportedStages = 0;
|
||||
Uint32 SubgroupSupportedFeatures = 0;
|
||||
Bool SubgroupQuadOperationsInAllStages = false;
|
||||
GpuVendorKind GpuVendor = GpuVendorKind::Unknown;
|
||||
};
|
||||
|
||||
enum class WindowBackend {
|
||||
Android,
|
||||
X11,
|
||||
MetalLayer,
|
||||
// TODO: Wayland, Windows, etc.
|
||||
Win32, // Handle is an HWND
|
||||
// TODO: Wayland, etc.
|
||||
WindowBackendCount,
|
||||
Unknown = -1
|
||||
};
|
||||
|
||||
@@ -13,6 +13,6 @@
|
||||
#include "DirectVulkan/BackendObject_DirectVulkan.h"
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
extern UniquePtr<BackendObject> pActiveBackendObject;
|
||||
extern UniquePtr<BackendObject>& pActiveBackendObject;
|
||||
extern GlobalBackendFunctionsTable gBackendFunctionsTable;
|
||||
} // namespace MobileGL::MG_Backend
|
||||
|
||||
@@ -701,9 +701,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
if ((handle.Backend != WindowBackend::Android &&
|
||||
handle.Backend != WindowBackend::X11 &&
|
||||
handle.Backend != WindowBackend::MetalLayer) ||
|
||||
handle.Backend != WindowBackend::MetalLayer &&
|
||||
handle.Backend != WindowBackend::Win32) ||
|
||||
!handle.Handle) {
|
||||
MGLOG_E("DirectGLES backend only supports Android, X11, and CAMetalLayer native windows");
|
||||
MGLOG_E("DirectGLES backend only supports Android, X11, CAMetalLayer, and Win32 native windows");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -826,7 +827,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object,
|
||||
E_GL_EXT_framebuffer_object, E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage,
|
||||
E_GL_ARB_texture_storage, E_GL_ARB_texture_storage_multisample,
|
||||
E_GL_ARB_direct_state_access,
|
||||
E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access,
|
||||
E_GL_ARB_multi_draw_indirect, E_GL_ARB_indirect_parameters,
|
||||
E_GL_ARB_shader_draw_parameters, E_GL_ARB_gpu_shader5, E_GL_ARB_multi_bind,
|
||||
E_GL_ARB_shading_language_420pack, E_GL_ARB_vertex_attrib_binding,
|
||||
@@ -947,6 +948,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return m_dynamicParameters;
|
||||
}
|
||||
|
||||
void BackendObject_DirectGLES::ApplyGLESCapabilitiesForTesting(
|
||||
const MG_External::GLESCapabilities& capabilities) {
|
||||
m_GLESCapabilities = capabilities;
|
||||
UpdateDynamicBackendParameters();
|
||||
}
|
||||
|
||||
void BackendObject_DirectGLES::UpdateDynamicBackendParameters() {
|
||||
m_dynamicParameters.UniformBufferOffsetAlignment = m_GLESCapabilities.UniformBufferOffsetAlignment;
|
||||
m_dynamicParameters.MaxTextureMaxAnisotropy = m_GLESCapabilities.MaxTextureMaxAnisotropy;
|
||||
@@ -1003,9 +1010,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
m_dynamicParameters.MaxUniformBlockSize = m_GLESCapabilities.MaxUniformBlockSize;
|
||||
const Int maxSupportedTextureUnits =
|
||||
static_cast<Int>(MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS);
|
||||
m_dynamicParameters.MaxImageUnits = std::min(m_GLESCapabilities.MaxImageUnits, maxSupportedTextureUnits);
|
||||
m_dynamicParameters.MaxCombinedImageUniforms = m_GLESCapabilities.MaxCombinedImageUniforms;
|
||||
m_dynamicParameters.MaxComputeImageUniforms = m_GLESCapabilities.MaxComputeImageUniforms;
|
||||
m_dynamicParameters.MaxImageUnits =
|
||||
std::max(std::min(m_GLESCapabilities.MaxImageUnits, maxSupportedTextureUnits), 0);
|
||||
m_dynamicParameters.MaxCombinedImageUniforms = std::max(m_GLESCapabilities.MaxCombinedImageUniforms, 0);
|
||||
const auto clampStageImageUniforms = [this](Int stageLimit) {
|
||||
return std::min({std::max(stageLimit, 0), m_dynamicParameters.MaxImageUnits,
|
||||
m_dynamicParameters.MaxCombinedImageUniforms});
|
||||
};
|
||||
m_dynamicParameters.MaxVertexImageUniforms =
|
||||
clampStageImageUniforms(m_GLESCapabilities.MaxVertexImageUniforms);
|
||||
m_dynamicParameters.MaxGeometryImageUniforms =
|
||||
clampStageImageUniforms(m_GLESCapabilities.MaxGeometryImageUniforms);
|
||||
m_dynamicParameters.MaxFragmentImageUniforms =
|
||||
clampStageImageUniforms(m_GLESCapabilities.MaxFragmentImageUniforms);
|
||||
m_dynamicParameters.MaxComputeImageUniforms =
|
||||
clampStageImageUniforms(m_GLESCapabilities.MaxComputeImageUniforms);
|
||||
m_dynamicParameters.MaxDrawBuffers = m_GLESCapabilities.MaxDrawBuffers;
|
||||
m_dynamicParameters.MaxColorAttachments = m_GLESCapabilities.MaxColorAttachments;
|
||||
m_dynamicParameters.MaxClipDistances = m_GLESCapabilities.MaxClipDistances;
|
||||
@@ -1017,6 +1036,32 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
m_dynamicParameters.ViewportSubpixelBits = m_GLESCapabilities.ViewportSubpixelBits;
|
||||
m_dynamicParameters.SupportsWideLines =
|
||||
m_GLESCapabilities.AliasedLineWidthRangeMax > 1.0f || m_GLESCapabilities.SmoothLineWidthRangeMax > 1.0f;
|
||||
|
||||
const auto containsAny = [](const String& haystack, std::initializer_list<const char*> needles) {
|
||||
return std::any_of(needles.begin(), needles.end(), [&](const char* needle) {
|
||||
return haystack.find(needle) != String::npos;
|
||||
});
|
||||
};
|
||||
const String vendorAndRenderer =
|
||||
m_GLESCapabilities.GLESVendorString + " " + m_GLESCapabilities.GLESRendererString;
|
||||
if (containsAny(vendorAndRenderer, {"llvmpipe", "SwiftShader", "softpipe"})) {
|
||||
// Check software rasterizers first: ANGLE-on-llvmpipe reports both.
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Software;
|
||||
} else if (containsAny(vendorAndRenderer, {"Qualcomm", "Adreno"})) {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Qualcomm;
|
||||
} else if (containsAny(vendorAndRenderer, {"Mali", "ARM"})) {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Arm;
|
||||
} else if (containsAny(vendorAndRenderer, {"NVIDIA"})) {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Nvidia;
|
||||
} else if (containsAny(vendorAndRenderer, {"AMD", "Radeon"})) {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Amd;
|
||||
} else if (containsAny(vendorAndRenderer, {"Intel"})) {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Intel;
|
||||
} else if (containsAny(vendorAndRenderer, {"Imagination", "PowerVR"})) {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::ImgTec;
|
||||
} else {
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Unknown;
|
||||
}
|
||||
}
|
||||
|
||||
const MG_External::GLESFunctionsTable& BackendObject_DirectGLES::GetGLESFunctions() const {
|
||||
|
||||
@@ -41,6 +41,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
const MG_External::GLESFunctionsTable& GetGLESFunctions() const;
|
||||
const MG_External::EGLFunctionsTable& GetEGLFunctions() const;
|
||||
void ApplyGLESCapabilitiesForTesting(const MG_External::GLESCapabilities& capabilities);
|
||||
|
||||
private:
|
||||
void UpdateDynamicBackendParameters();
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -267,16 +267,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Uint g_boundArrayBufferId = 0;
|
||||
Bool g_boundArrayBufferKnown = false;
|
||||
|
||||
// Driver-level GL_PIXEL_PACK/UNPACK_BUFFER binding shadows (see
|
||||
// Managers.h). Resting state between operations is 0; scopes in the
|
||||
// readback/upload paths bind what they need through the cache and
|
||||
// return to 0, so a stale user PBO can never capture a later
|
||||
// readback that meant to target client memory.
|
||||
Uint g_boundPixelPackBufferId = 0;
|
||||
Bool g_boundPixelPackBufferKnown = false;
|
||||
Uint g_boundPixelUnpackBufferId = 0;
|
||||
Bool g_boundPixelUnpackBufferKnown = false;
|
||||
|
||||
// Bumped whenever the backend ES context is destroyed; resources with
|
||||
// an older generation hold ids from a dead context.
|
||||
Uint g_bufferContextGeneration = 1;
|
||||
|
||||
// Defined next to the indexed-binding shadow below; forward-declared so
|
||||
// every glDeleteBuffers site in this namespace can scrub stale shadow
|
||||
// entries (GL resets a deleted buffer's indexed bindings to 0, and a
|
||||
// recycled name matching a stale shadow entry would otherwise
|
||||
// false-skip the rebind).
|
||||
void ScrubIndexedBufferBindingShadowForId(Uint id);
|
||||
// entries (GL resets a deleted buffer's bindings - indexed and pixel
|
||||
// pack/unpack alike - to 0, and a recycled name matching a stale shadow
|
||||
// entry would otherwise false-skip the rebind).
|
||||
void ScrubBufferBindingShadowsForId(Uint id);
|
||||
|
||||
// Resources whose owning BufferObject died; ids deleted at the next
|
||||
// sync point with a current ES context.
|
||||
@@ -318,10 +328,16 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == r.id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
// Pooling keeps the id alive (and thus any driver binding of it);
|
||||
// drop to unknown rather than claiming the post-delete 0 state.
|
||||
if ((g_boundPixelPackBufferKnown && g_boundPixelPackBufferId == r.id) ||
|
||||
(g_boundPixelUnpackBufferKnown && g_boundPixelUnpackBufferId == r.id)) {
|
||||
InvalidatePixelBufferBindingCaches();
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(g_poolMutex);
|
||||
auto& bucket = g_bufferPool[r.storageSize];
|
||||
if (bucket.size() >= kMaxEntriesPerBucket || g_pooledBytes + r.storageSize > kMaxPoolBytes) {
|
||||
ScrubIndexedBufferBindingShadowForId(r.id);
|
||||
ScrubBufferBindingShadowsForId(r.id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &r.id); // over budget: don't pool
|
||||
r.id = 0;
|
||||
return;
|
||||
@@ -502,7 +518,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Need a fresh id: glBufferStorage fails on a buffer that already has
|
||||
// immutable storage, and any prior mutable store is replaced anyway.
|
||||
if (resource->id != 0) {
|
||||
ScrubIndexedBufferBindingShadowForId(resource->id);
|
||||
ScrubBufferBindingShadowsForId(resource->id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &resource->id);
|
||||
resource->id = 0;
|
||||
}
|
||||
@@ -630,7 +646,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == glesResource->id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
ScrubIndexedBufferBindingShadowForId(glesResource->id);
|
||||
ScrubBufferBindingShadowsForId(glesResource->id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &glesResource->id);
|
||||
glesResource->id = 0;
|
||||
}
|
||||
@@ -668,6 +684,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void OnBackendContextDestroyed() {
|
||||
UnregisterBufferBackendOps();
|
||||
++g_bufferContextGeneration;
|
||||
InvalidateArrayBufferBindingCache();
|
||||
InvalidateIndexedBufferBindingCache();
|
||||
InvalidatePixelBufferBindingCaches();
|
||||
// The global-UBO ring's id and persistent map died with the context;
|
||||
// drop the handles (no GL) and let the next draw recreate the ring.
|
||||
ResetUboRingForNewContext();
|
||||
@@ -694,7 +713,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == glesResource->id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
ScrubIndexedBufferBindingShadowForId(glesResource->id);
|
||||
ScrubBufferBindingShadowsForId(glesResource->id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &glesResource->id);
|
||||
glesResource->id = 0;
|
||||
}
|
||||
@@ -819,6 +838,41 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
g_boundArrayBufferKnown = false;
|
||||
}
|
||||
|
||||
void BindPixelPackBufferId(Uint id) {
|
||||
if (g_boundPixelPackBufferKnown && g_boundPixelPackBufferId == id) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_PACK_BUFFER, id);
|
||||
g_boundPixelPackBufferId = id;
|
||||
g_boundPixelPackBufferKnown = true;
|
||||
}
|
||||
|
||||
void BindPixelUnpackBufferId(Uint id) {
|
||||
if (g_boundPixelUnpackBufferKnown && g_boundPixelUnpackBufferId == id) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, id);
|
||||
g_boundPixelUnpackBufferId = id;
|
||||
g_boundPixelUnpackBufferKnown = true;
|
||||
}
|
||||
|
||||
void InvalidatePixelBufferBindingCaches() {
|
||||
g_boundPixelPackBufferId = 0;
|
||||
g_boundPixelPackBufferKnown = false;
|
||||
g_boundPixelUnpackBufferId = 0;
|
||||
g_boundPixelUnpackBufferKnown = false;
|
||||
}
|
||||
|
||||
void NoteBufferIdDeleted(Uint id) {
|
||||
if (id == 0) {
|
||||
return;
|
||||
}
|
||||
if (g_boundArrayBufferKnown && g_boundArrayBufferId == id) {
|
||||
InvalidateArrayBufferBindingCache();
|
||||
}
|
||||
ScrubBufferBindingShadowsForId(id);
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Shadow of the GL indexed buffer bindings so redundant glBindBufferBase/Range
|
||||
// (same index + id + range) are skipped. isBase distinguishes a whole-buffer
|
||||
@@ -840,12 +894,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// glDeleteBuffers resets the deleted buffer's bindings (indexed ones
|
||||
// included) to 0 in the current context; mirror that in the shadow, or a
|
||||
// later buffer recycling the same name with a matching range would
|
||||
// false-skip its rebind. Default IndexedBufferBinding{} == base(0) ==
|
||||
// the post-delete GL state.
|
||||
void ScrubIndexedBufferBindingShadowForId(Uint id) {
|
||||
// glDeleteBuffers resets the deleted buffer's bindings (indexed and
|
||||
// pixel pack/unpack ones included) to 0 in the current context; mirror
|
||||
// that in the shadows, or a later buffer recycling the same name with a
|
||||
// matching shadow entry would false-skip its rebind. Default
|
||||
// IndexedBufferBinding{} == base(0) == the post-delete GL state.
|
||||
void ScrubBufferBindingShadowsForId(Uint id) {
|
||||
if (id == 0) return;
|
||||
for (auto& binding : g_indexedUBOBindings) {
|
||||
if (binding.id == id) binding = {};
|
||||
@@ -853,6 +907,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
for (auto& binding : g_indexedSSBOBindings) {
|
||||
if (binding.id == id) binding = {};
|
||||
}
|
||||
if (g_boundPixelPackBufferKnown && g_boundPixelPackBufferId == id) {
|
||||
g_boundPixelPackBufferId = 0;
|
||||
}
|
||||
if (g_boundPixelUnpackBufferKnown && g_boundPixelUnpackBufferId == id) {
|
||||
g_boundPixelUnpackBufferId = 0;
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
@@ -897,7 +957,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
auto& bucket = g_bufferPool[oldestKey];
|
||||
PooledBuffer& e = bucket[oldestIdx];
|
||||
if (e.contextGeneration == g_bufferContextGeneration && e.id != 0) {
|
||||
ScrubIndexedBufferBindingShadowForId(e.id);
|
||||
ScrubBufferBindingShadowsForId(e.id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &e.id);
|
||||
}
|
||||
g_pooledBytes -= e.size;
|
||||
@@ -1064,7 +1124,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const Bool staleContext = entry.contextGeneration != g_bufferContextGeneration;
|
||||
if (!staleContext && entry.retireSerial > completed) continue;
|
||||
if (!staleContext && entry.id != 0) {
|
||||
ScrubIndexedBufferBindingShadowForId(entry.id);
|
||||
ScrubBufferBindingShadowsForId(entry.id);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &entry.id);
|
||||
}
|
||||
g_retiredUboRings[i] = g_retiredUboRings.back();
|
||||
@@ -1152,6 +1212,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
for (auto& bufferId : m_clientAttributeBufferIds) {
|
||||
if (bufferId != 0) {
|
||||
BufferImpl::NoteBufferIdDeleted(bufferId);
|
||||
g_GLESFuncs.glDeleteBuffers(1, &bufferId);
|
||||
bufferId = 0;
|
||||
}
|
||||
@@ -1324,6 +1385,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
g_GLESFuncs.glGenTextures(1, &m_backendTextureId);
|
||||
m_contextGeneration = g_textureContextGeneration;
|
||||
if (m_backendTextureId == 0) {
|
||||
MGLOG_E("Failed to generate texture object.");
|
||||
MGLOG_E("ES glGetError(): %s", MG_Util::ConvertGLEnumToString(g_GLESFuncs.glGetError()).c_str());
|
||||
@@ -1332,6 +1394,27 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
}
|
||||
|
||||
BackendTextureObject::~BackendTextureObject() {
|
||||
if (m_backendTextureId == 0) {
|
||||
return;
|
||||
}
|
||||
// Scrub every driver-state shadow that could false-skip when the name
|
||||
// or this heap address is recycled - regardless of whether the id can
|
||||
// still be deleted.
|
||||
ScratchFBOImpl::NoteTextureIdDeleted(m_backendTextureId);
|
||||
for (auto& unitCache : g_boundTexturesCache) {
|
||||
for (auto& boundTexture : unitCache) {
|
||||
if (boundTexture == this) {
|
||||
boundTexture = nullptr;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (m_contextGeneration == g_textureContextGeneration && g_GLESFuncs.glDeleteTextures) {
|
||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||
}
|
||||
m_backendTextureId = 0;
|
||||
}
|
||||
|
||||
void BackendTextureObject::Bind(GLenum target, Uint unit) {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
@@ -1364,7 +1447,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
void BackendTextureObject::RecreateBackendTexture() {
|
||||
if (m_backendTextureId != 0) {
|
||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||
ScratchFBOImpl::NoteTextureIdDeleted(m_backendTextureId);
|
||||
if (m_contextGeneration == g_textureContextGeneration) {
|
||||
g_GLESFuncs.glDeleteTextures(1, &m_backendTextureId);
|
||||
}
|
||||
for (auto& unitCache : g_boundTexturesCache) {
|
||||
for (auto& boundTexture : unitCache) {
|
||||
if (boundTexture == this) {
|
||||
@@ -1375,6 +1461,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
|
||||
g_GLESFuncs.glGenTextures(1, &m_backendTextureId);
|
||||
m_contextGeneration = g_textureContextGeneration;
|
||||
if (m_backendTextureId == 0) {
|
||||
MGLOG_E("Failed to regenerate texture object.");
|
||||
MGLOG_E("ES glGetError(): %s", MG_Util::ConvertGLEnumToString(g_GLESFuncs.glGetError()).c_str());
|
||||
@@ -1391,7 +1478,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// glGetIntegerv - that query forces a driver pipeline sync and, because texture
|
||||
// uploads run it per dirty texture per frame, it dominated the DirectGLES draw
|
||||
// path. The backend unpack state is set ONLY by MobileGL's own save/restore
|
||||
// helpers (this class, TempPixelStoreParameterSync, the R32F copy path), all of
|
||||
// helpers (this class and, historically, the R32F copy path), all of
|
||||
// which restore to the resting default, so the shadow stays accurate; a one-time
|
||||
// forced sync pins the backend to that known default up front. Apply() is
|
||||
// compare-and-set, so the (now redundant) glPixelStorei calls also usually no-op.
|
||||
@@ -1563,6 +1650,56 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return convertedData.data();
|
||||
}
|
||||
|
||||
// RGB565/RGB5_A1 shadow data is stored as 8-bit unorm; uploading it as GL_UNSIGNED_BYTE
|
||||
// leaves the 8-bit -> 5/6-bit requantization to the driver, whose rounding direction is
|
||||
// implementation-defined: Adreno rounds to nearest (lossless round trip) but Mali floors,
|
||||
// drifting mid-range texels one 5-bit step down and failing the KHR-GL3x
|
||||
// pixelstoragemodes.teximage3d rgb565/rgb5a1 1/32-eps checks. Repack the shadow rows into
|
||||
// the packed 16-bit client type with round-to-nearest instead - that recovers the original
|
||||
// 5/6-bit values exactly (the shadow expansion round(v * 255 / max) is injective), so the
|
||||
// driver stores them verbatim with no requantization left to its discretion. 4-bit formats
|
||||
// (RGBA4) are exempt: their 8-bit expansion (v * 17) is exact under either rounding.
|
||||
// Always retargets *inOutType for these formats (even for null data) so every upload of a
|
||||
// level uses the same client type.
|
||||
static const void* PreparePackedNormUpload(TextureInternalFormat format, const IntVec3& texelSize,
|
||||
const void* data, SizeT byteSize, GLenum* inOutType,
|
||||
Vector<Uint8>& packedData) {
|
||||
if (format != TextureInternalFormat::RGB5 && format != TextureInternalFormat::RGB5A1) {
|
||||
return data;
|
||||
}
|
||||
const Bool hasAlpha = format == TextureInternalFormat::RGB5A1;
|
||||
const GLenum packedType = hasAlpha ? GL_UNSIGNED_SHORT_5_5_5_1 : GL_UNSIGNED_SHORT_5_6_5;
|
||||
// Idempotent across a region's level loop: glType is shared, so later levels arrive with
|
||||
// the already-retargeted packed type and must still be converted.
|
||||
if (*inOutType != GL_UNSIGNED_BYTE && *inOutType != packedType) {
|
||||
return data;
|
||||
}
|
||||
*inOutType = packedType;
|
||||
if (data == nullptr || byteSize == 0) {
|
||||
return data;
|
||||
}
|
||||
const SizeT srcPixelBytes = hasAlpha ? 4 : 3;
|
||||
const SizeT texelCount = std::min(static_cast<SizeT>(std::max(texelSize.x(), 0)) *
|
||||
static_cast<SizeT>(std::max(texelSize.y(), 0)) *
|
||||
static_cast<SizeT>(std::max(texelSize.z(), 1)),
|
||||
byteSize / srcPixelBytes);
|
||||
packedData.resize(texelCount * sizeof(Uint16));
|
||||
const Uint8* src = static_cast<const Uint8*>(data);
|
||||
auto* dst = reinterpret_cast<Uint16*>(packedData.data());
|
||||
for (SizeT i = 0; i < texelCount; ++i, src += srcPixelBytes) {
|
||||
const Uint32 r = (static_cast<Uint32>(src[0]) * 31u + 127u) / 255u;
|
||||
const Uint32 b = (static_cast<Uint32>(src[2]) * 31u + 127u) / 255u;
|
||||
if (hasAlpha) {
|
||||
const Uint32 g = (static_cast<Uint32>(src[1]) * 31u + 127u) / 255u;
|
||||
dst[i] = static_cast<Uint16>((r << 11) | (g << 6) | (b << 1) | (src[3] >= 128 ? 1u : 0u));
|
||||
} else {
|
||||
const Uint32 g = (static_cast<Uint32>(src[1]) * 63u + 127u) / 255u;
|
||||
dst[i] = static_cast<Uint16>((r << 11) | (g << 5) | b);
|
||||
}
|
||||
}
|
||||
return packedData.data();
|
||||
}
|
||||
|
||||
void BackendTextureObject::SyncMipmapsToBackend(
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject) {
|
||||
if (!stateTextureObject) {
|
||||
@@ -1706,9 +1843,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), levelTexelSize, pData, levelByteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData = PreparePackedNormUpload(textureMipmapObject->GetFormat(), levelTexelSize,
|
||||
uploadData, levelByteSize, &glType, packedUploadData);
|
||||
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
const IntVec3 uploadSize =
|
||||
GetBackendUploadSize(stateTextureObject->GetTarget(), levelTexelSize);
|
||||
switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) {
|
||||
@@ -1760,7 +1900,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const auto& uploadTargets = textureMipmapObject->GetUploadTargets();
|
||||
if (TextureImpl::IsMultisampleTextureTarget(targetInternal)) {
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
switch (targetInternal) {
|
||||
case TextureTarget::Texture2DMultisample:
|
||||
g_GLESFuncs.glTexStorage2DMultisample(
|
||||
@@ -1787,7 +1927,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
} else if (stateTextureObject->IsImmutable() || m_imageBindableStorageRequired) {
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
const IntVec3 storageSize = GetBackendUploadSize(targetInternal, baseSize);
|
||||
switch (MapToBackendTextureTarget(targetInternal)) {
|
||||
case TextureTarget::Texture2D:
|
||||
@@ -1831,9 +1971,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), levelTexelSize, pData, levelByteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData =
|
||||
PreparePackedNormUpload(textureMipmapObject->GetFormat(), levelTexelSize,
|
||||
uploadData, levelByteSize, &glType, packedUploadData);
|
||||
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
const IntVec3 uploadSize =
|
||||
GetBackendUploadSize(targetInternal, levelTexelSize);
|
||||
switch (MapToBackendTextureTarget(targetInternal)) {
|
||||
@@ -1885,6 +2029,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), levelTexelSize, pData, levelByteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData =
|
||||
PreparePackedNormUpload(textureMipmapObject->GetFormat(), levelTexelSize,
|
||||
uploadData, levelByteSize, &glType, packedUploadData);
|
||||
MGLOG_D("%s: target: %s: syncing mip %d: %dx%dx%d, byteSize = %d, pData = %p, "
|
||||
"levelDirty = %s",
|
||||
__func__, MG_Util::ConvertTextureUploadTargetToString(uploadTarget).c_str(),
|
||||
@@ -1892,7 +2040,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
levelByteSize, pData, levelDirty ? "true" : "false");
|
||||
|
||||
DebugImpl::ErrorLopper::Clear();
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
auto textureTarget = stateTextureObject->GetTarget();
|
||||
const IntVec3 uploadSize = GetBackendUploadSize(textureTarget, levelTexelSize);
|
||||
switch (MapToBackendTextureTarget(textureTarget)) {
|
||||
@@ -1978,7 +2126,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
textureMipmapObject->GetMipmapTexelSize(uploadTarget, level).y(), byteSize);
|
||||
|
||||
auto glUploadTarget = ConvertTextureUploadTargetToBackendGLEnum(uploadTarget);
|
||||
g_GLESFuncs.glBindBuffer(GL_PIXEL_UNPACK_BUFFER, 0);
|
||||
BufferImpl::BindPixelUnpackBufferId(0); // no-op once the resting 0 state is pinned
|
||||
DebugImpl::ErrorLopper::Loop(
|
||||
[file = __FILE__, line = __LINE__, func = __func__](GLenum err) {
|
||||
MGLOG_D("%s(%s:%d) ES error: %s", func, file, line,
|
||||
@@ -1990,6 +2138,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const void* uploadData = PrepareNormFloatFallbackUpload(
|
||||
textureMipmapObject->GetFormat(), texelSize, mipData, byteSize, glType,
|
||||
convertedUploadData);
|
||||
Vector<Uint8> packedUploadData;
|
||||
uploadData = PreparePackedNormUpload(textureMipmapObject->GetFormat(), texelSize,
|
||||
uploadData, byteSize, &glType, packedUploadData);
|
||||
const IntVec3 uploadSize =
|
||||
GetBackendUploadSize(stateTextureObject->GetTarget(), texelSize);
|
||||
switch (MapToBackendTextureTarget(stateTextureObject->GetTarget())) {
|
||||
@@ -2216,11 +2367,17 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return;
|
||||
}
|
||||
|
||||
if (TextureImpl::IsMultisampleTextureTarget(targetInternal)) {
|
||||
// Multisample targets reject the *sampler* parameters (LOD range, border color) but
|
||||
// GL_TEXTURE_SWIZZLE_* is texture state, not sampler state, and ES accepts it on them.
|
||||
// Bailing out entirely used to drop every swizzle write on the floor, which is what the
|
||||
// frontend already assumes is legal (see GL_Texture.cpp's MS-invalid pname list, which
|
||||
// deliberately omits the swizzle enums). Note the caches for the skipped parameters are
|
||||
// still refreshed so they never look stale, but m_cacheSwizzleParams must NOT be, or the
|
||||
// change detection below would swallow the very writes we came here to emit.
|
||||
const Bool isMultisampleTarget = TextureImpl::IsMultisampleTextureTarget(targetInternal);
|
||||
if (isMultisampleTarget) {
|
||||
m_cacheLodRange = stateTextureObject->GetLevelRange();
|
||||
m_cacheSwizzleParams = stateTextureObject->GetAllSwizzleParams();
|
||||
m_cacheBorderColor = stateTextureObject->GetBorderColor();
|
||||
return;
|
||||
}
|
||||
|
||||
Bind(target);
|
||||
@@ -2233,14 +2390,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
const auto& levelRange = stateTextureObject->GetLevelRange();
|
||||
|
||||
if (m_cacheLodRange.x() != levelRange.x()) {
|
||||
if (!isMultisampleTarget && m_cacheLodRange.x() != levelRange.x()) {
|
||||
g_GLESFuncs.glTexParameteri(target, GL_TEXTURE_BASE_LEVEL, static_cast<GLint>(levelRange.x()));
|
||||
m_cacheLodRange.x() = levelRange.x();
|
||||
}
|
||||
DebugImpl::ErrorLopper::Loop([file = __FILE__, line = __LINE__, func = __func__](GLenum err) {
|
||||
MGLOG_D("%s(%s:%d) ES error %s", func, file, line, MG_Util::ConvertGLEnumToString(err).c_str());
|
||||
});
|
||||
if (m_cacheLodRange.y() != levelRange.y()) {
|
||||
if (!isMultisampleTarget && m_cacheLodRange.y() != levelRange.y()) {
|
||||
g_GLESFuncs.glTexParameteri(target, GL_TEXTURE_MAX_LEVEL, static_cast<GLint>(levelRange.y()));
|
||||
m_cacheLodRange.y() = levelRange.y();
|
||||
}
|
||||
@@ -2266,7 +2423,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
});
|
||||
}
|
||||
|
||||
if (m_cacheBorderColor != stateTextureObject->GetBorderColor()) {
|
||||
if (!isMultisampleTarget && m_cacheBorderColor != stateTextureObject->GetBorderColor()) {
|
||||
const auto& borderColor = stateTextureObject->GetBorderColor();
|
||||
GLfloat borderColorArray[4] = {borderColor.x(), borderColor.y(), borderColor.z(), borderColor.w()};
|
||||
g_GLESFuncs.glTexParameterfv(target, GL_TEXTURE_BORDER_COLOR, borderColorArray);
|
||||
@@ -2295,6 +2452,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
|
||||
Uint g_activeTextureUnit = 0;
|
||||
Uint g_textureContextGeneration = 1;
|
||||
Array<Array<BackendTextureObject*, (SizeT)TextureTarget::TextureTargetCount>,
|
||||
MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS>
|
||||
g_boundTexturesCache;
|
||||
@@ -2320,9 +2478,59 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (target == FramebufferTarget::Read)
|
||||
g_GLESFuncs.glBindFramebuffer(GL_READ_FRAMEBUFFER, m_backendFBOId);
|
||||
BindFramebufferId(GL_READ_FRAMEBUFFER, m_backendFBOId);
|
||||
else
|
||||
g_GLESFuncs.glBindFramebuffer(GL_DRAW_FRAMEBUFFER, m_backendFBOId);
|
||||
BindFramebufferId(GL_DRAW_FRAMEBUFFER, m_backendFBOId);
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Driver-level framebuffer-binding shadow (see Managers.h). Indexed by
|
||||
// FramebufferTarget {Draw, Read}.
|
||||
Array<Uint, SizeT(FramebufferTarget::FramebufferTargetCount)> g_driverFBOBindings = {0, 0};
|
||||
Array<Bool, SizeT(FramebufferTarget::FramebufferTargetCount)> g_driverFBOBindingKnown = {false, false};
|
||||
} // namespace
|
||||
|
||||
void BindFramebufferId(GLenum fbTarget, Uint id) {
|
||||
const Bool bindsDraw = fbTarget == GL_DRAW_FRAMEBUFFER || fbTarget == GL_FRAMEBUFFER;
|
||||
const Bool bindsRead = fbTarget == GL_READ_FRAMEBUFFER || fbTarget == GL_FRAMEBUFFER;
|
||||
const SizeT drawIdx = SizeT(FramebufferTarget::Draw);
|
||||
const SizeT readIdx = SizeT(FramebufferTarget::Read);
|
||||
const Bool drawMatches =
|
||||
!bindsDraw || (g_driverFBOBindingKnown[drawIdx] && g_driverFBOBindings[drawIdx] == id);
|
||||
const Bool readMatches =
|
||||
!bindsRead || (g_driverFBOBindingKnown[readIdx] && g_driverFBOBindings[readIdx] == id);
|
||||
if (drawMatches && readMatches) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glBindFramebuffer(fbTarget, id);
|
||||
if (bindsDraw) {
|
||||
g_driverFBOBindings[drawIdx] = id;
|
||||
g_driverFBOBindingKnown[drawIdx] = true;
|
||||
}
|
||||
if (bindsRead) {
|
||||
g_driverFBOBindings[readIdx] = id;
|
||||
g_driverFBOBindingKnown[readIdx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
Uint CurrentFramebufferBinding(FramebufferTarget target) {
|
||||
const SizeT idx = SizeT(target);
|
||||
if (!g_driverFBOBindingKnown[idx]) {
|
||||
// Cold path: pin the shadow from the driver once (init probes and
|
||||
// pre-shadow code bind raw but restore what they found).
|
||||
GLint binding = 0;
|
||||
g_GLESFuncs.glGetIntegerv(
|
||||
target == FramebufferTarget::Read ? GL_READ_FRAMEBUFFER_BINDING : GL_DRAW_FRAMEBUFFER_BINDING,
|
||||
&binding);
|
||||
g_driverFBOBindings[idx] = static_cast<Uint>(binding);
|
||||
g_driverFBOBindingKnown[idx] = true;
|
||||
}
|
||||
return g_driverFBOBindings[idx];
|
||||
}
|
||||
|
||||
void InvalidateFramebufferBindingCache() {
|
||||
g_driverFBOBindings = {0, 0};
|
||||
g_driverFBOBindingKnown = {false, false};
|
||||
}
|
||||
|
||||
void BackendFramebufferObject::InvalidateSyncedState() {
|
||||
@@ -2376,7 +2584,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (glTextureTarget == GL_UNKNOWN_MGL) {
|
||||
glTextureTarget = TextureImpl::ConvertTextureTargetToBackendGLEnum(textureObject->GetTarget());
|
||||
}
|
||||
backendTextureObject->Bind(glTextureTarget);
|
||||
// glBindTexture rejects cube-face enums (INVALID_ENUM with no
|
||||
// bind, while Bind() would still record the cube-map cache slot
|
||||
// as bound): bind via the owning cube target; the attach below
|
||||
// keeps the face target.
|
||||
const Bool isCubeFace = glTextureTarget >= GL_TEXTURE_CUBE_MAP_POSITIVE_X &&
|
||||
glTextureTarget <= GL_TEXTURE_CUBE_MAP_NEGATIVE_Z;
|
||||
backendTextureObject->Bind(isCubeFace ? GL_TEXTURE_CUBE_MAP : glTextureTarget);
|
||||
g_GLESFuncs.glFramebufferTexture2D(glFBOTarget, glBackendAttachment, glTextureTarget,
|
||||
backendTextureObject->GetBackendTextureId(),
|
||||
static_cast<GLint>(attachmentObject.GetTextureLevel()));
|
||||
@@ -2465,6 +2679,28 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return false;
|
||||
}
|
||||
|
||||
void BackendFramebufferObject::SyncReadBufferToBackend(
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject) {
|
||||
if (!stateFBOObject) {
|
||||
return;
|
||||
}
|
||||
auto frontendReadBuf = stateFBOObject->GetReadBuffer();
|
||||
if (frontendReadBuf == m_frontendReadBuffer) {
|
||||
return;
|
||||
}
|
||||
m_frontendReadBuffer = frontendReadBuf;
|
||||
|
||||
GLenum glBackendReadBuffer = GetBackendAttachmentType(frontendReadBuf);
|
||||
if (m_backendReadBuffer != glBackendReadBuffer) {
|
||||
m_backendReadBuffer = glBackendReadBuffer;
|
||||
// glReadBuffer targets whatever FBO is bound to GL_READ_FRAMEBUFFER. When this is
|
||||
// reached from SyncCurrentFBO's "same FBO as draw" skip path the backend FBO was
|
||||
// only bound as DRAW, so bind it as READ first to route the read buffer correctly.
|
||||
Bind(FramebufferTarget::Read);
|
||||
g_GLESFuncs.glReadBuffer(glBackendReadBuffer);
|
||||
}
|
||||
}
|
||||
|
||||
void BackendFramebufferObject::SyncToBackend(
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject, FramebufferTarget asTarget) {
|
||||
#ifdef TRACY_ENABLE
|
||||
@@ -2546,16 +2782,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
// 2. Remap read buffer. glReadBuffer writes the READ-bound FBO's state, so
|
||||
// only apply (and stamp the memo) when this object is bound as READ.
|
||||
auto frontendReadBuf = stateFBOObject->GetReadBuffer();
|
||||
if (frontendReadBuf != m_frontendReadBuffer && asTarget == FramebufferTarget::Read) {
|
||||
m_frontendReadBuffer = frontendReadBuf;
|
||||
|
||||
GLenum glBackendReadBuffer = GetBackendAttachmentType(frontendReadBuf);
|
||||
|
||||
if (m_backendReadBuffer != glBackendReadBuffer) {
|
||||
m_backendReadBuffer = glBackendReadBuffer;
|
||||
g_GLESFuncs.glReadBuffer(glBackendReadBuffer);
|
||||
}
|
||||
if (asTarget == FramebufferTarget::Read) {
|
||||
SyncReadBufferToBackend(stateFBOObject);
|
||||
}
|
||||
|
||||
// -------------------- Attach texture to backend FBO -----------------------
|
||||
@@ -2662,6 +2890,295 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
g_fboSyncedObjects = {};
|
||||
} // namespace FramebufferImpl
|
||||
|
||||
namespace ScratchFBOImpl {
|
||||
namespace {
|
||||
ScratchFramebuffer g_tempFramebuffer;
|
||||
ScratchFramebuffer g_blitReadFramebuffer;
|
||||
ScratchFramebuffer g_blitDrawFramebuffer;
|
||||
Uint g_completeTinyFBOId = 0;
|
||||
Uint g_completeTinyRBOId = 0;
|
||||
|
||||
// Detach every point the shadow no longer vouches for. Used when the
|
||||
// shadow is unknown (context reset, texture id deleted while attached).
|
||||
void ScrubAllAttachments(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
fb.colorTex = 0;
|
||||
fb.colorTarget = 0;
|
||||
fb.colorLevel = 0;
|
||||
fb.colorLayer = -1;
|
||||
fb.depthTex = 0;
|
||||
fb.depthTarget = 0;
|
||||
fb.depthLevel = 0;
|
||||
fb.depthHasStencil = false;
|
||||
fb.attachmentsKnown = true;
|
||||
}
|
||||
|
||||
void PrepareForUse(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
if (!fb.attachmentsKnown) {
|
||||
ScrubAllAttachments(fb, fbTarget);
|
||||
}
|
||||
}
|
||||
|
||||
// The post-attach glGetError probe below must not misread an error some
|
||||
// earlier operation left queued; drain before attaching (rare path -
|
||||
// only runs when the attachment actually changes).
|
||||
void DrainPendingGLErrors() {
|
||||
while (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
}
|
||||
}
|
||||
|
||||
// Record the color point as detached when the shadow said something was
|
||||
// there; the actual detach call is the caller's (it may be replaced by
|
||||
// the new attach directly when the point is being overwritten).
|
||||
void RecordNoColor(ScratchFramebuffer& fb) {
|
||||
fb.colorTex = 0;
|
||||
fb.colorTarget = 0;
|
||||
fb.colorLevel = 0;
|
||||
fb.colorLayer = -1;
|
||||
}
|
||||
|
||||
void RecordNoDepth(ScratchFramebuffer& fb) {
|
||||
fb.depthTex = 0;
|
||||
fb.depthTarget = 0;
|
||||
fb.depthLevel = 0;
|
||||
fb.depthHasStencil = false;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
ScratchFramebuffer& TempFramebuffer() {
|
||||
return g_tempFramebuffer;
|
||||
}
|
||||
ScratchFramebuffer& BlitReadFramebuffer() {
|
||||
return g_blitReadFramebuffer;
|
||||
}
|
||||
ScratchFramebuffer& BlitDrawFramebuffer() {
|
||||
return g_blitDrawFramebuffer;
|
||||
}
|
||||
|
||||
Uint EnsureId(ScratchFramebuffer& fb) {
|
||||
if (fb.id == 0) {
|
||||
g_GLESFuncs.glGenFramebuffers(1, &fb.id);
|
||||
// A fresh FBO has nothing attached and COLOR_ATTACHMENT0 read/draw
|
||||
// buffers (the ES defaults for a non-default framebuffer).
|
||||
fb.attachmentsKnown = true;
|
||||
RecordNoColor(fb);
|
||||
RecordNoDepth(fb);
|
||||
fb.readBuffer = GL_COLOR_ATTACHMENT0;
|
||||
fb.drawBuffer = GL_COLOR_ATTACHMENT0;
|
||||
}
|
||||
return fb.id;
|
||||
}
|
||||
|
||||
void EnsureColorAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget,
|
||||
GLint level) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.depthTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoDepth(fb);
|
||||
}
|
||||
if (fb.colorTex == tex && fb.colorTarget == texTarget && fb.colorLevel == level && fb.colorLayer < 0) {
|
||||
return;
|
||||
}
|
||||
if (fb.colorTex != 0) {
|
||||
// Detach first: if the new attach fails, the point must read as
|
||||
// missing (incomplete FBO), not silently keep the old texture.
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
}
|
||||
DrainPendingGLErrors();
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, texTarget, tex, level);
|
||||
if (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
RecordNoColor(fb);
|
||||
return;
|
||||
}
|
||||
fb.colorTex = tex;
|
||||
fb.colorTarget = texTarget;
|
||||
fb.colorLevel = level;
|
||||
fb.colorLayer = -1;
|
||||
}
|
||||
|
||||
void EnsureColorAttachmentLayer(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLint level, GLint layer) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.depthTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoDepth(fb);
|
||||
}
|
||||
if (fb.colorTex == tex && fb.colorTarget == 0 && fb.colorLevel == level && fb.colorLayer == layer) {
|
||||
return;
|
||||
}
|
||||
if (fb.colorTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
}
|
||||
DrainPendingGLErrors();
|
||||
g_GLESFuncs.glFramebufferTextureLayer(fbTarget, GL_COLOR_ATTACHMENT0, tex, level, layer);
|
||||
if (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
RecordNoColor(fb);
|
||||
return;
|
||||
}
|
||||
fb.colorTex = tex;
|
||||
fb.colorTarget = 0;
|
||||
fb.colorLevel = level;
|
||||
fb.colorLayer = layer;
|
||||
}
|
||||
|
||||
void EnsureDepthAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget, GLint level,
|
||||
Bool withStencil) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.colorTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoColor(fb);
|
||||
}
|
||||
if (fb.depthTex == tex && fb.depthTarget == texTarget && fb.depthLevel == level &&
|
||||
fb.depthHasStencil == withStencil) {
|
||||
return;
|
||||
}
|
||||
if (fb.depthTex != 0) {
|
||||
// One call clears both depth and stencil points regardless of how
|
||||
// the previous attachment was made.
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
}
|
||||
DrainPendingGLErrors();
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget,
|
||||
withStencil ? GL_DEPTH_STENCIL_ATTACHMENT : GL_DEPTH_ATTACHMENT,
|
||||
texTarget, tex, level);
|
||||
if (g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
RecordNoDepth(fb);
|
||||
return;
|
||||
}
|
||||
fb.depthTex = tex;
|
||||
fb.depthTarget = texTarget;
|
||||
fb.depthLevel = level;
|
||||
fb.depthHasStencil = withStencil;
|
||||
}
|
||||
|
||||
void EnsureNoColorAttachment(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.colorTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoColor(fb);
|
||||
}
|
||||
}
|
||||
|
||||
void EnsureNoDepthAttachment(ScratchFramebuffer& fb, GLenum fbTarget) {
|
||||
PrepareForUse(fb, fbTarget);
|
||||
if (fb.depthTex != 0) {
|
||||
g_GLESFuncs.glFramebufferTexture2D(fbTarget, GL_DEPTH_STENCIL_ATTACHMENT, GL_TEXTURE_2D, 0, 0);
|
||||
RecordNoDepth(fb);
|
||||
}
|
||||
}
|
||||
|
||||
void EnsureReadBuffer(ScratchFramebuffer& fb, GLenum readBuffer) {
|
||||
if (fb.readBuffer == readBuffer) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glReadBuffer(readBuffer);
|
||||
fb.readBuffer = readBuffer;
|
||||
}
|
||||
|
||||
void EnsureDrawBuffer(ScratchFramebuffer& fb, GLenum drawBuffer) {
|
||||
if (fb.drawBuffer == drawBuffer) {
|
||||
return;
|
||||
}
|
||||
g_GLESFuncs.glDrawBuffers(1, &drawBuffer);
|
||||
fb.drawBuffer = drawBuffer;
|
||||
}
|
||||
|
||||
Uint EnsureCompleteTinyFramebufferId() {
|
||||
if (g_completeTinyFBOId != 0) {
|
||||
return g_completeTinyFBOId;
|
||||
}
|
||||
// One-time creation: the renderbuffer binding is context state with no
|
||||
// shadow, so save/restore it by query here (cold path only).
|
||||
GLint prevRenderbuffer = 0;
|
||||
g_GLESFuncs.glGetIntegerv(GL_RENDERBUFFER_BINDING, &prevRenderbuffer);
|
||||
g_GLESFuncs.glGenFramebuffers(1, &g_completeTinyFBOId);
|
||||
g_GLESFuncs.glGenRenderbuffers(1, &g_completeTinyRBOId);
|
||||
FramebufferImpl::BindFramebufferId(GL_FRAMEBUFFER, g_completeTinyFBOId);
|
||||
g_GLESFuncs.glBindRenderbuffer(GL_RENDERBUFFER, g_completeTinyRBOId);
|
||||
g_GLESFuncs.glRenderbufferStorage(GL_RENDERBUFFER, GL_RGBA8, 1, 1);
|
||||
g_GLESFuncs.glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER,
|
||||
g_completeTinyRBOId);
|
||||
const GLenum drawBuffer = GL_COLOR_ATTACHMENT0;
|
||||
g_GLESFuncs.glDrawBuffers(1, &drawBuffer);
|
||||
g_GLESFuncs.glReadBuffer(GL_COLOR_ATTACHMENT0);
|
||||
MOBILEGL_ASSERT(g_GLESFuncs.glCheckFramebufferStatus(GL_FRAMEBUFFER) == GL_FRAMEBUFFER_COMPLETE,
|
||||
"Scratch 1x1 framebuffer is incomplete.");
|
||||
g_GLESFuncs.glBindRenderbuffer(GL_RENDERBUFFER, static_cast<Uint>(prevRenderbuffer));
|
||||
return g_completeTinyFBOId;
|
||||
}
|
||||
|
||||
void NoteTextureIdDeleted(Uint textureId) {
|
||||
if (textureId == 0) {
|
||||
return;
|
||||
}
|
||||
for (ScratchFramebuffer* fb : {&g_tempFramebuffer, &g_blitReadFramebuffer, &g_blitDrawFramebuffer}) {
|
||||
if (fb->colorTex == textureId || fb->depthTex == textureId) {
|
||||
fb->attachmentsKnown = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OnBackendContextDestroyed() {
|
||||
g_tempFramebuffer = {};
|
||||
g_blitReadFramebuffer = {};
|
||||
g_blitDrawFramebuffer = {};
|
||||
g_completeTinyFBOId = 0;
|
||||
g_completeTinyRBOId = 0;
|
||||
}
|
||||
} // namespace ScratchFBOImpl
|
||||
|
||||
namespace PixelStoreImpl {
|
||||
namespace {
|
||||
PackState g_packState;
|
||||
Bool g_packStateKnown = false;
|
||||
|
||||
void PinPackState(const PackState& value) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ALIGNMENT, value.Alignment);
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ROW_LENGTH, value.RowLength);
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_ROWS, value.SkipRows);
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_PIXELS, value.SkipPixels);
|
||||
g_packState = value;
|
||||
g_packStateKnown = true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void ApplyPackState(const PackState& desired) {
|
||||
if (!g_packStateKnown) {
|
||||
PinPackState(desired);
|
||||
return;
|
||||
}
|
||||
if (desired.Alignment != g_packState.Alignment) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ALIGNMENT, desired.Alignment);
|
||||
g_packState.Alignment = desired.Alignment;
|
||||
}
|
||||
if (desired.RowLength != g_packState.RowLength) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_ROW_LENGTH, desired.RowLength);
|
||||
g_packState.RowLength = desired.RowLength;
|
||||
}
|
||||
if (desired.SkipRows != g_packState.SkipRows) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_ROWS, desired.SkipRows);
|
||||
g_packState.SkipRows = desired.SkipRows;
|
||||
}
|
||||
if (desired.SkipPixels != g_packState.SkipPixels) {
|
||||
g_GLESFuncs.glPixelStorei(GL_PACK_SKIP_PIXELS, desired.SkipPixels);
|
||||
g_packState.SkipPixels = desired.SkipPixels;
|
||||
}
|
||||
}
|
||||
|
||||
PackState CurrentPackState() {
|
||||
if (!g_packStateKnown) {
|
||||
// Fresh/unknown context: pin to the GL defaults (what a new context
|
||||
// starts with; writing them makes the shadow authoritative either way).
|
||||
PinPackState(PackState{});
|
||||
}
|
||||
return g_packState;
|
||||
}
|
||||
|
||||
void InvalidatePackStateCache() {
|
||||
g_packStateKnown = false;
|
||||
}
|
||||
} // namespace PixelStoreImpl
|
||||
|
||||
namespace PrgramImpl {
|
||||
Uint32 g_snormFallbackClampOutputMask = 0;
|
||||
Uint32 g_unormFallbackClampOutputMask = 0;
|
||||
@@ -2784,6 +3301,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
effectiveSpirv = &uboPrecisionSpirv;
|
||||
}
|
||||
|
||||
// noperspective is core desktop GLSL and reaches here as the SPIR-V NoPerspective
|
||||
// decoration. SPIRV-Cross renders it as ESSL `noperspective` + `#extension
|
||||
// GL_NV_shader_noperspective_interpolation : require`; a driver without that extension
|
||||
// rejects the require. So on such devices emulate screen-linear interpolation instead
|
||||
// (pre-multiply outputs by gl_Position.w, recover inputs via gl_FragCoord.w) and drop
|
||||
// the decoration - exact, extension-free. Devices that have the extension keep the
|
||||
// decoration and let the hardware do it natively.
|
||||
Vector<unsigned int> noperspectiveSpirv;
|
||||
if (!g_GLESCapabilities.SupportsNoperspectiveInterpolation &&
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::EmulateNoPerspectiveForEssl(
|
||||
*effectiveSpirv, noperspectiveSpirv) &&
|
||||
!noperspectiveSpirv.empty()) {
|
||||
effectiveSpirv = &noperspectiveSpirv;
|
||||
}
|
||||
|
||||
MG_Util::ShaderTranspiler::SpvcSession spvcSession(*effectiveSpirv,
|
||||
MG_Util::ShaderTranspiler::SessionUsageBit::Transpile);
|
||||
|
||||
|
||||
@@ -178,6 +178,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// glBindBuffer with a redundant-bind cache for GL_ARRAY_BUFFER.
|
||||
void BindBufferId(GLenum target, Uint id);
|
||||
void InvalidateArrayBufferBindingCache();
|
||||
// Redundant-bind caches for the driver-level GL_PIXEL_PACK/UNPACK_BUFFER
|
||||
// bindings. Every backend readback (glReadPixels / pack-PBO map) and pixel
|
||||
// upload site routes its binding through these so the shadow always matches
|
||||
// the driver; the resting state between operations is 0, which keeps any
|
||||
// path that implicitly assumes "no PBO bound" correct. Scrubbed when a
|
||||
// buffer id is deleted/pooled (GL resets a deleted buffer's bindings to 0,
|
||||
// and a recycled name matching the shadow would false-skip the rebind) and
|
||||
// invalidated on MakeCurrent (context may reset).
|
||||
void BindPixelPackBufferId(Uint id);
|
||||
void BindPixelUnpackBufferId(Uint id);
|
||||
void InvalidatePixelBufferBindingCaches();
|
||||
// A GL buffer id is being deleted by code outside BufferImpl (e.g. the VAO
|
||||
// client-attribute staging buffers): scrub every buffer-binding shadow that
|
||||
// could false-skip when the name is recycled.
|
||||
void NoteBufferIdDeleted(Uint id);
|
||||
// Redundant-bind cache for INDEXED buffer bindings (glBindBufferBase/Range on
|
||||
// GL_UNIFORM_BUFFER / GL_SHADER_STORAGE_BUFFER): skips the GL call when the
|
||||
// (id, range) already at that index matches, like the array-buffer/texture/
|
||||
@@ -332,6 +347,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
class BackendTextureObject {
|
||||
public:
|
||||
BackendTextureObject();
|
||||
// Deletes the GL texture (frontend glDeleteTextures used to leak every
|
||||
// backend id for the context lifetime) and scrubs the binding/scratch-FBO
|
||||
// shadows so a recycled name or heap address cannot false-skip a rebind.
|
||||
~BackendTextureObject();
|
||||
BackendTextureObject(const BackendTextureObject&) = delete;
|
||||
BackendTextureObject& operator=(const BackendTextureObject&) = delete;
|
||||
void SyncMipmapsToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void SyncBuiltinSamplerToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void SyncTextureParamsToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
@@ -343,6 +364,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void RecreateBackendTexture();
|
||||
|
||||
Uint m_backendTextureId = 0;
|
||||
// ES context generation the id was created under; a dtor running after
|
||||
// that context died must not delete a foreign (recycled) name.
|
||||
Uint m_contextGeneration = 0;
|
||||
Bool m_isInitialized = false;
|
||||
Bool m_imageBindableStorageRequired = false;
|
||||
Bool m_backendStorageImmutable = false;
|
||||
@@ -367,6 +391,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS>
|
||||
g_boundTexturesCache;
|
||||
extern Uint g_activeTextureUnit;
|
||||
// Bumped when the backend ES context is destroyed; texture ids stamped with
|
||||
// an older generation belong to a dead context and must not be deleted.
|
||||
extern Uint g_textureContextGeneration;
|
||||
} // namespace TextureImpl
|
||||
|
||||
namespace FramebufferImpl {
|
||||
@@ -375,6 +402,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
BackendFramebufferObject();
|
||||
void SyncToBackend(const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject,
|
||||
FramebufferTarget asTarget);
|
||||
// Apply only this FBO's read buffer (glReadBuffer) to the backend. Split out so it can
|
||||
// still run when SyncCurrentFBO skips the READ-target sync because the same GL FBO is
|
||||
// bound as both draw and read (otherwise glReadBuffer changes would be silently dropped).
|
||||
void SyncReadBufferToBackend(const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject);
|
||||
void InvalidateSyncedState();
|
||||
Uint GetBackendFramebufferId() const { return m_backendFBOId; }
|
||||
void Bind(FramebufferTarget target) const;
|
||||
@@ -414,8 +445,99 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
extern Array<Uint16, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedObjectVersions;
|
||||
extern Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
||||
g_fboSyncedObjects;
|
||||
|
||||
// Driver-level READ/DRAW framebuffer-binding shadow. Every backend
|
||||
// glBindFramebuffer routes through BindFramebufferId so scoped helpers can
|
||||
// save/restore the current binding without a glGetIntegerv round-trip (that
|
||||
// query forces a driver pipeline sync) and so redundant rebinds no-op.
|
||||
// Starts unknown; the first CurrentFramebufferBinding() query pins it from
|
||||
// the driver once. Invalidated on MakeCurrent (context may reset).
|
||||
// GL_FRAMEBUFFER binds both targets.
|
||||
void BindFramebufferId(GLenum fbTarget, Uint id);
|
||||
Uint CurrentFramebufferBinding(FramebufferTarget target);
|
||||
void InvalidateFramebufferBindingCache();
|
||||
} // namespace FramebufferImpl
|
||||
|
||||
// Shared scratch framebuffers for the readback/copy/blit emulation paths, with a
|
||||
// driver-side attachment shadow: repeated uses skip redundant detach/attach GL
|
||||
// calls, and an attachment left by one use (e.g. a depth copy's DEPTH_STENCIL
|
||||
// texture) is detached exactly when a later use of another aspect would
|
||||
// otherwise inherit it (stale cross-aspect attachments made the shared temp FBO
|
||||
// incomplete and silently degraded later readbacks).
|
||||
namespace ScratchFBOImpl {
|
||||
struct ScratchFramebuffer {
|
||||
Uint id = 0;
|
||||
// false => attachment state unknown; scrub every point on next use.
|
||||
// A fresh FBO starts with nothing attached, so creation sets it true.
|
||||
Bool attachmentsKnown = false;
|
||||
Uint colorTex = 0;
|
||||
GLenum colorTarget = 0;
|
||||
GLint colorLevel = 0;
|
||||
GLint colorLayer = -1; // >= 0 => attached via glFramebufferTextureLayer
|
||||
Uint depthTex = 0;
|
||||
GLenum depthTarget = 0;
|
||||
GLint depthLevel = 0;
|
||||
Bool depthHasStencil = false;
|
||||
// Per-FBO read/draw buffer state (0 = unknown, set on first use).
|
||||
GLenum readBuffer = 0;
|
||||
GLenum drawBuffer = 0;
|
||||
};
|
||||
ScratchFramebuffer& TempFramebuffer(); // GetTexImage READ / CopyTex*Image2D depth DRAW
|
||||
ScratchFramebuffer& BlitReadFramebuffer(); // texture-to-texture blit source
|
||||
ScratchFramebuffer& BlitDrawFramebuffer(); // texture-to-texture blit destination
|
||||
// Returns the GL id, generating it if needed (requires a current ES context).
|
||||
Uint EnsureId(ScratchFramebuffer& fb);
|
||||
// The fb must currently be bound at fbTarget (glReadBuffer/glDrawBuffers
|
||||
// target the READ/DRAW binding respectively). Each Ensure* performs the
|
||||
// minimal detach/attach set and keeps the shadow in sync; a failed attach
|
||||
// records the point as detached so the completeness check fails instead of
|
||||
// silently reading a stale attachment.
|
||||
void EnsureColorAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget, GLint level);
|
||||
void EnsureColorAttachmentLayer(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLint level, GLint layer);
|
||||
void EnsureDepthAttachment2D(ScratchFramebuffer& fb, GLenum fbTarget, Uint tex, GLenum texTarget, GLint level,
|
||||
Bool withStencil);
|
||||
void EnsureNoColorAttachment(ScratchFramebuffer& fb, GLenum fbTarget);
|
||||
void EnsureNoDepthAttachment(ScratchFramebuffer& fb, GLenum fbTarget);
|
||||
void EnsureReadBuffer(ScratchFramebuffer& fb, GLenum readBuffer);
|
||||
void EnsureDrawBuffer(ScratchFramebuffer& fb, GLenum drawBuffer);
|
||||
// A 1x1 RGBA8-renderbuffer-complete FBO (GenerateMipmap needs a complete
|
||||
// binding while respecifying texture storage). Attachment is set once at
|
||||
// creation and never changes.
|
||||
Uint EnsureCompleteTinyFramebufferId();
|
||||
// A backend texture id is being deleted or respecified: a scratch FBO still
|
||||
// referencing it would hold a dangling attachment (ES only auto-detaches
|
||||
// from the *bound* framebuffer), and a recycled name could false-skip a
|
||||
// re-attach; force a full scrub on next use.
|
||||
void NoteTextureIdDeleted(Uint textureId);
|
||||
// The ES context (and the scratch FBO ids with it) is going away.
|
||||
void OnBackendContextDestroyed();
|
||||
} // namespace ScratchFBOImpl
|
||||
|
||||
// Driver-level GL_PACK_* pixel-store shadow, the readback-side sibling of the
|
||||
// upload path's ScopedDefaultUnpackState (Managers.cpp): the backend PACK state
|
||||
// is written ONLY through ApplyPackState, so scoped helpers can save/restore it
|
||||
// from the shadow instead of glGetIntegerv (which forces a driver pipeline
|
||||
// sync), and redundant glPixelStorei calls no-op. The first Apply/Current call
|
||||
// pins the driver to the shadow by writing all fields once. Invalidated on
|
||||
// MakeCurrent (context may reset). PACK_IMAGE_HEIGHT/SKIP_IMAGES/SWAP_BYTES/
|
||||
// LSB_FIRST have no ES equivalents; readbacks honor them on the CPU from the
|
||||
// frontend context state instead.
|
||||
namespace PixelStoreImpl {
|
||||
struct PackState {
|
||||
GLint Alignment = 4;
|
||||
GLint RowLength = 0;
|
||||
GLint SkipRows = 0;
|
||||
GLint SkipPixels = 0;
|
||||
Bool operator==(const PackState& o) const {
|
||||
return Alignment == o.Alignment && RowLength == o.RowLength && SkipRows == o.SkipRows &&
|
||||
SkipPixels == o.SkipPixels;
|
||||
}
|
||||
};
|
||||
void ApplyPackState(const PackState& desired);
|
||||
PackState CurrentPackState();
|
||||
void InvalidatePackStateCache();
|
||||
} // namespace PixelStoreImpl
|
||||
|
||||
// Image uniforms take their unit from the layout(binding=N) qualifier baked into
|
||||
// the transpiled ESSL; unlike samplers they must not (and in ES cannot) be
|
||||
// assigned through glUniform1i.
|
||||
|
||||
@@ -764,5 +764,95 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static SizeT AlignReadbackRow(SizeT rowBytes, Int alignment) {
|
||||
const SizeT align = alignment > 0 ? static_cast<SizeT>(alignment) : 1;
|
||||
return (rowBytes + align - 1) / align * align;
|
||||
}
|
||||
|
||||
// Repacks wide RGBA(_INTEGER) rows into the client's (format, type) layout, honoring the
|
||||
// client-side PACK parameters and the bound pixel-pack buffer. `wide` holds
|
||||
// `sliceHeight * sliceCount` rows of `width` texels (slice-major, tightly stacked),
|
||||
// 4 components x GetReadbackComponentSize(wideType) bytes each.
|
||||
// applyPackImageParams: GL_PACK_IMAGE_HEIGHT / GL_PACK_SKIP_IMAGES apply only to GetTexImage
|
||||
// of 3D/array images; ReadPixels and 2D GetTexImage ignore them (GL 3.3 sections 4.3.1, 6.1.4).
|
||||
// Per the GL addressing rules, slice k row j lands at
|
||||
// SKIP_IMAGES*imageStride + SKIP_ROWS*rowStride + SKIP_PIXELS*pixelBytes
|
||||
// + k*imageStride + j*rowStride, with imageStride = max(IMAGE_HEIGHT, sliceHeight)*rowStride.
|
||||
Bool StoreWideRowsToClient(const Uint8* wide, GLenum wideType, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, const ReadbackChannelMapping& mapping, GLenum type,
|
||||
void* pixels, Bool applyPackImageParams) {
|
||||
const SizeT dstPixelBytes = GetReadbackDstPixelSize(mapping, type);
|
||||
if (dstPixelBytes == 0) {
|
||||
return false;
|
||||
}
|
||||
PackedReadbackLayout packedLayout{};
|
||||
const Bool isPackedType = GetPackedReadbackLayout(type, packedLayout);
|
||||
const SizeT dstComponentSize = GetReadbackComponentSize(type);
|
||||
|
||||
const auto& pixelPackBufferObject =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::PixelPack).GetBoundObject();
|
||||
|
||||
// Destination layout is computed from the client-side PACK parameters; only the actual pixel
|
||||
// rows are written so skip regions of the destination stay untouched.
|
||||
const auto packParams = MG_State::pGLContext->GetPixelStoreParameters(false);
|
||||
const SizeT rowPixels = static_cast<SizeT>(packParams.RowLength > 0 ? packParams.RowLength : width);
|
||||
const SizeT dstRowStride = AlignReadbackRow(rowPixels * dstPixelBytes, packParams.Alignment);
|
||||
const SizeT imageRows =
|
||||
applyPackImageParams && packParams.ImageHeight > 0
|
||||
? static_cast<SizeT>(packParams.ImageHeight)
|
||||
: static_cast<SizeT>(sliceHeight);
|
||||
const SizeT dstImageStride = imageRows * dstRowStride;
|
||||
const SizeT skipImages =
|
||||
applyPackImageParams ? static_cast<SizeT>(std::max(packParams.SkipImages, 0)) : SizeT{0};
|
||||
const SizeT dstSkipOffset = skipImages * dstImageStride +
|
||||
static_cast<SizeT>(std::max(packParams.SkipRows, 0)) * dstRowStride +
|
||||
static_cast<SizeT>(std::max(packParams.SkipPixels, 0)) * dstPixelBytes;
|
||||
const SizeT dstRowBytes = static_cast<SizeT>(width) * dstPixelBytes;
|
||||
|
||||
const SizeT pboBaseOffset = reinterpret_cast<SizeT>(pixels); // with a PBO, `pixels` is an offset
|
||||
if (pixelPackBufferObject) {
|
||||
const SizeT requiredSize = pboBaseOffset + dstSkipOffset +
|
||||
static_cast<SizeT>(sliceCount - 1) * dstImageStride +
|
||||
static_cast<SizeT>(sliceHeight - 1) * dstRowStride + dstRowBytes;
|
||||
if (requiredSize > pixelPackBufferObject->GetSize()) {
|
||||
MGLOG_E("Readback conversion: pixel pack buffer is too small");
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT srcComponentSize = GetReadbackComponentSize(wideType);
|
||||
const SizeT srcPixelBytes = 4 * srcComponentSize;
|
||||
Vector<Uint8> convertedRow(dstRowBytes);
|
||||
|
||||
for (GLsizei slice = 0; slice < sliceCount; ++slice) {
|
||||
for (GLsizei row = 0; row < sliceHeight; ++row) {
|
||||
const SizeT flatRow = static_cast<SizeT>(slice) * static_cast<SizeT>(sliceHeight) +
|
||||
static_cast<SizeT>(row);
|
||||
const Uint8* srcRow = wide + flatRow * static_cast<SizeT>(width) * srcPixelBytes;
|
||||
ConvertWideReadbackRow(srcRow, convertedRow.data(), static_cast<SizeT>(width), wideType,
|
||||
mapping, type);
|
||||
|
||||
if (packParams.SwapBytes) {
|
||||
const SizeT groupSize = isPackedType ? packedLayout.byteSize : dstComponentSize;
|
||||
if (groupSize > 1) {
|
||||
for (SizeT offset = 0; offset + groupSize <= dstRowBytes; offset += groupSize) {
|
||||
std::reverse(convertedRow.data() + offset, convertedRow.data() + offset + groupSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT dstOffset = dstSkipOffset + static_cast<SizeT>(slice) * dstImageStride +
|
||||
static_cast<SizeT>(row) * dstRowStride;
|
||||
if (pixelPackBufferObject) {
|
||||
pixelPackBufferObject->WritebackFromBackend({convertedRow.data(), dstRowBytes},
|
||||
pboBaseOffset + dstOffset);
|
||||
} else {
|
||||
Memcpy(static_cast<Uint8*>(pixels) + dstOffset, convertedRow.data(), dstRowBytes);
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
} // namespace ReadbackImpl
|
||||
} // namespace MobileGL::MG_Backend::DirectGLES
|
||||
|
||||
@@ -88,6 +88,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// bytes, dst receives width * GetReadbackDstPixelSize(mapping, type) bytes.
|
||||
void ConvertWideReadbackRow(const Uint8* src, Uint8* dst, SizeT width, GLenum wideType,
|
||||
const ReadbackChannelMapping& mapping, GLenum type);
|
||||
|
||||
// Stores wide RGBA(_INTEGER) rows into the client pointer or the bound PACK pixel buffer,
|
||||
// honoring the client-side PACK pixel-store parameters (row length, alignment, skips,
|
||||
// swap-bytes, and - when applyPackImageParams - image height/skip images). Shared by the
|
||||
// DirectGLES and DirectVulkan readback conversion paths.
|
||||
Bool StoreWideRowsToClient(const Uint8* wide, GLenum wideType, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, const ReadbackChannelMapping& mapping, GLenum type,
|
||||
void* pixels, Bool applyPackImageParams);
|
||||
} // namespace ReadbackImpl
|
||||
|
||||
namespace PrgramImpl {
|
||||
|
||||
@@ -140,6 +140,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
// Legacy low-bit-depth formats with no (or rarely supported) native Vulkan
|
||||
// encoding; a wider normalized fallback keeps at least the required precision.
|
||||
case TextureInternalFormat::R3G3B2:
|
||||
case TextureInternalFormat::RGB4:
|
||||
case TextureInternalFormat::RGB5:
|
||||
case TextureInternalFormat::RGBA2:
|
||||
case TextureInternalFormat::RGBA4:
|
||||
case TextureInternalFormat::RGB5A1:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
case TextureInternalFormat::RGB10:
|
||||
return TextureInternalFormat::RGB10A2;
|
||||
case TextureInternalFormat::RGB12:
|
||||
case TextureInternalFormat::RGBA12:
|
||||
return TextureInternalFormat::RGBA16;
|
||||
case TextureInternalFormat::SRGB8:
|
||||
return TextureInternalFormat::SRGB8Alpha8;
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
@@ -397,8 +411,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
if (!handle.Handle || (handle.Backend != WindowBackend::Android &&
|
||||
handle.Backend != WindowBackend::X11 &&
|
||||
handle.Backend != WindowBackend::MetalLayer)) {
|
||||
MGLOG_E("DirectVulkan backend only supports Android, X11, and CAMetalLayer native windows");
|
||||
handle.Backend != WindowBackend::MetalLayer &&
|
||||
handle.Backend != WindowBackend::Win32)) {
|
||||
MGLOG_E("DirectVulkan backend only supports Android, X11, CAMetalLayer, and Win32 native windows");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -454,6 +469,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// treat them as signaled/available with zero results from here on.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer.reset();
|
||||
// The reflection cache is file-scope, not renderer-owned; without this the
|
||||
// deleted programs' reflection strings survive full context teardown.
|
||||
ClearProgramResourceCaches();
|
||||
BackendObject::ReleaseEGLResources();
|
||||
}
|
||||
|
||||
@@ -463,6 +481,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// treat them as signaled/available with zero results from here on.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer.reset();
|
||||
// The reflection cache is file-scope, not renderer-owned; without this the
|
||||
// deleted programs' reflection strings survive full context teardown.
|
||||
ClearProgramResourceCaches();
|
||||
}
|
||||
|
||||
const RendererInfo& BackendObject_DirectVulkan::GetRendererInfo() const {
|
||||
@@ -504,7 +525,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
E_GL_ARB_multi_draw_indirect, E_GL_ARB_indirect_parameters,
|
||||
E_GL_EXT_framebuffer_object, E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage,
|
||||
E_GL_ARB_texture_storage, E_GL_ARB_texture_storage_multisample,
|
||||
E_GL_ARB_direct_state_access,
|
||||
E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access,
|
||||
E_GL_ARB_shader_draw_parameters, E_GL_ARB_gpu_shader_int64, E_GL_KHR_debug,
|
||||
E_GL_ARB_gpu_shader5, E_GL_ARB_multi_bind, E_GL_ARB_shading_language_420pack,
|
||||
E_GL_ARB_vertex_attrib_binding, E_GL_ARB_shader_image_size};
|
||||
@@ -746,9 +767,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_dynamicParameters.MaxTextureBufferSize = m_vulkanCaps.MaxTextureBufferSize;
|
||||
m_dynamicParameters.MaxUniformBufferBindings = m_vulkanCaps.MaxUniformBufferBindings;
|
||||
m_dynamicParameters.MaxUniformBlockSize = m_vulkanCaps.MaxUniformBlockSize;
|
||||
m_dynamicParameters.MaxImageUnits = std::min(m_vulkanCaps.MaxImageUnits, maxSupportedTextureUnits);
|
||||
m_dynamicParameters.MaxCombinedImageUniforms = m_vulkanCaps.MaxCombinedImageUniforms;
|
||||
m_dynamicParameters.MaxComputeImageUniforms = m_vulkanCaps.MaxComputeImageUniforms;
|
||||
m_dynamicParameters.MaxImageUnits =
|
||||
std::max(std::min(m_vulkanCaps.MaxImageUnits, maxSupportedTextureUnits), 0);
|
||||
m_dynamicParameters.MaxCombinedImageUniforms = std::max(m_vulkanCaps.MaxCombinedImageUniforms, 0);
|
||||
const Int maxPerStageImageUniforms =
|
||||
std::min(m_dynamicParameters.MaxImageUnits, m_dynamicParameters.MaxCombinedImageUniforms);
|
||||
// Vulkan uses one descriptor limit for every stage, but non-compute stores/atomics are
|
||||
// optional device features. VulkanRenderer enables each feature whenever the physical
|
||||
// device reports it, so these are the exact limits the logical device can compile and run.
|
||||
m_dynamicParameters.MaxVertexImageUniforms =
|
||||
m_vulkanCaps.SupportsVertexPipelineStoresAndAtomics ? maxPerStageImageUniforms : 0;
|
||||
m_dynamicParameters.MaxGeometryImageUniforms =
|
||||
m_vulkanCaps.SupportsVertexPipelineStoresAndAtomics && m_vulkanCaps.SupportsGeometryShader
|
||||
? maxPerStageImageUniforms
|
||||
: 0;
|
||||
m_dynamicParameters.MaxFragmentImageUniforms =
|
||||
m_vulkanCaps.SupportsFragmentStoresAndAtomics ? maxPerStageImageUniforms : 0;
|
||||
m_dynamicParameters.MaxComputeImageUniforms =
|
||||
std::min(std::max(m_vulkanCaps.MaxComputeImageUniforms, 0), maxPerStageImageUniforms);
|
||||
const Int maxSupportedDrawBuffers =
|
||||
static_cast<Int>(MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS);
|
||||
m_dynamicParameters.MaxDrawBuffers = std::min(m_vulkanCaps.MaxDrawBuffers, maxSupportedDrawBuffers);
|
||||
@@ -779,5 +815,32 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_vulkanCaps.MaxShaderStorageBlockSize,
|
||||
m_dynamicParameters.MaxShaderStorageBlockSize);
|
||||
}
|
||||
switch (m_vulkanCaps.VendorId) {
|
||||
case 0x5143u: // VK_VENDOR_ID: Qualcomm
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Qualcomm;
|
||||
break;
|
||||
case 0x13B5u: // ARM
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Arm;
|
||||
break;
|
||||
case 0x10DEu: // NVIDIA
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Nvidia;
|
||||
break;
|
||||
case 0x1002u: // AMD
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Amd;
|
||||
break;
|
||||
case 0x8086u: // Intel
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Intel;
|
||||
break;
|
||||
case 0x1010u: // Imagination
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::ImgTec;
|
||||
break;
|
||||
case 0x10005u: // Mesa software (lavapipe)
|
||||
case 0x1AE0u: // Google (SwiftShader)
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Software;
|
||||
break;
|
||||
default:
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Unknown;
|
||||
break;
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -20,7 +20,8 @@
|
||||
#include <spirv_reflect.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
UniquePtr<VulkanRenderer> pVulkanRenderer = nullptr;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<VulkanRenderer>& pVulkanRenderer = *new UniquePtr<VulkanRenderer>();
|
||||
|
||||
namespace {
|
||||
// Generation of the live VulkanRenderer instance, mirroring
|
||||
@@ -60,6 +61,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
|
||||
struct ProgramResourceCache {
|
||||
// Lifetime id of the program the cached reflection belongs to. GL names are
|
||||
// recycled (IndexGenerator hands freed indices straight back), and a
|
||||
// recreated program's backendStateVersion restarts at the same small values,
|
||||
// so the version alone can collide; the never-reused lifetime id makes the
|
||||
// slot's ownership unambiguous.
|
||||
Uint64 programLifetimeId = 0;
|
||||
Uint32 backendStateVersion = 0;
|
||||
Vector<StorageBlockResource> storageBlocks;
|
||||
Vector<BufferVariableResource> bufferVariables;
|
||||
@@ -81,6 +88,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 baseInstance = 0;
|
||||
};
|
||||
|
||||
// Keyed by GL program name so the freed-name reuse in IndexGenerator bounds the
|
||||
// map at the peak-simultaneous-program high-water mark; each slot's ownership is
|
||||
// checked against the program's lifetime id before it is served (see
|
||||
// GetProgramResourceCache). Cleared wholesale at EGL teardown via
|
||||
// ClearProgramResourceCaches.
|
||||
UnorderedMap<GLuint, ProgramResourceCache> g_programResourceCaches;
|
||||
|
||||
void ClearReadPixelsOutput(GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels) {
|
||||
@@ -141,13 +153,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
ProgramResourceCache& GetProgramResourceCache(const MG_State::GLState::ProgramObject& program) {
|
||||
auto& cache = g_programResourceCaches[program.GetExternalIndex()];
|
||||
const Uint64 programLifetimeId = program.GetLifetimeId();
|
||||
const Uint32 backendStateVersion = program.GetBackendStateVersion();
|
||||
if (cache.backendStateVersion == backendStateVersion &&
|
||||
// The lifetime id must match too: a new program that reuses a deleted
|
||||
// program's name and happens to land on the same backendStateVersion (both
|
||||
// count from zero) would otherwise be served the dead program's reflection.
|
||||
if (cache.programLifetimeId == programLifetimeId &&
|
||||
cache.backendStateVersion == backendStateVersion &&
|
||||
(!cache.storageBlocks.empty() || !cache.bufferVariables.empty())) {
|
||||
return cache;
|
||||
}
|
||||
|
||||
cache = {};
|
||||
cache.programLifetimeId = programLifetimeId;
|
||||
cache.backendStateVersion = backendStateVersion;
|
||||
|
||||
Vector<SpvReflectShaderModule> modules;
|
||||
@@ -365,6 +383,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void ClearProgramResourceCaches() {
|
||||
// Called from EGL teardown while the backend's m_eglStateMutex is held; GL
|
||||
// calls are serialized in this codebase (contexts migrate threads but never
|
||||
// run concurrently), so no other thread can be inside the unsynchronized map.
|
||||
// Live programs in another context self-heal: their entry rebuilds from the
|
||||
// retained generated SPIR-V on the next resource query.
|
||||
g_programResourceCaches.clear();
|
||||
}
|
||||
|
||||
GLuint GetShaderStorageBlockIndex(const MG_State::GLState::ProgramObject& program, const String& name) {
|
||||
auto& cache = GetProgramResourceCache(program);
|
||||
const auto it = std::find_if(cache.storageBlocks.begin(), cache.storageBlocks.end(),
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
#include "Renderer/VulkanRenderer.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
extern UniquePtr<VulkanRenderer> pVulkanRenderer;
|
||||
extern UniquePtr<VulkanRenderer>& pVulkanRenderer;
|
||||
|
||||
// Generation of the live VulkanRenderer instance, mirroring DirectGLES's
|
||||
// g_syncContextGeneration. BackendObject_DirectVulkan bumps it wherever
|
||||
@@ -23,6 +23,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 GetRendererGeneration();
|
||||
void BumpRendererGeneration();
|
||||
|
||||
// Drops every cached program-resource reflection entry (CPU-side strings/vectors
|
||||
// only, no Vulkan handles). Called at EGL teardown next to the renderer reset;
|
||||
// safe because GL calls are serialized in this codebase, and any still-live
|
||||
// program rebuilds its entry from the retained generated SPIR-V on demand.
|
||||
void ClearProgramResourceCaches();
|
||||
|
||||
void ClearBufferfi(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void ClearBufferfv(GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void ClearBufferuiv(GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
|
||||
@@ -150,12 +150,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
Bool FrameContext::TransitionToPresent(VkImage image, VkImageLayout oldLayout, VkImageLayout presentLayout) {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.hasCommandBufferRecorded || frame.isCommandRecording || oldLayout == presentLayout ||
|
||||
oldLayout == VK_IMAGE_LAYOUT_SHARED_PRESENT_KHR) {
|
||||
if (oldLayout == presentLayout || oldLayout == VK_IMAGE_LAYOUT_SHARED_PRESENT_KHR) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& commandBuffer = BeginCommandRecording();
|
||||
// The barrier belongs in the frame's own recording. Bailing out because
|
||||
// something was already recorded (the previous behaviour) dropped the
|
||||
// transition entirely for every frame that never ran a default-framebuffer
|
||||
// render pass - the only other thing that carries the image to
|
||||
// PRESENT_SRC_KHR, via that pass's finalLayout - so the swapchain image was
|
||||
// handed to the WSI still in the layout it was acquired in.
|
||||
// A closed-but-unsubmitted buffer can only come from a submit that already
|
||||
// failed (SubmitPendingCommandBuffer leaves the flag set on error), and
|
||||
// appending to it is illegal while reopening would reset the frame's own
|
||||
// commands away. The device is gone on that path anyway - stay silent-safe
|
||||
// rather than trade a lost device for a barrier into a closed buffer.
|
||||
if (frame.hasCommandBufferRecorded) {
|
||||
MGLOG_E("TransitionToPresent: command buffer already closed; skipping the present barrier");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Reopening a recording here would vkResetCommandBuffer this frame's own
|
||||
// commands away, so append to the open one and let the caller close it.
|
||||
const Bool openedRecording = !frame.isCommandRecording;
|
||||
VkCommandBuffer commandBuffer = openedRecording ? BeginCommandRecording() : frame.commandBuffer;
|
||||
|
||||
VkImageMemoryBarrier presentBarrier{};
|
||||
presentBarrier.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
@@ -174,7 +192,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0,
|
||||
nullptr, 0, nullptr, 1, &presentBarrier);
|
||||
|
||||
EndCommandRecording();
|
||||
if (openedRecording) {
|
||||
EndCommandRecording();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -227,12 +247,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
result = vkAcquireNextImageKHR(device, swapchain, timeout, frame.imageAvailableSemaphore, acquireFence,
|
||||
&outImageIndex);
|
||||
if (result != VK_SUCCESS) {
|
||||
// VK_SUBOPTIMAL_KHR is a success code: an image *was* acquired and
|
||||
// imageAvailableSemaphore *will* be signaled. Bailing out on it skipped both
|
||||
// the consumed-flag reset (leaving a stale "already consumed", so the next
|
||||
// submit never waited on the pending signal) and the fence reset (leaving
|
||||
// the slot's fence signaled for the next submit to reuse). Only a genuine
|
||||
// failure - VK_ERROR_OUT_OF_DATE_KHR and friends, where nothing is acquired
|
||||
// and nothing is signaled - skips the bookkeeping.
|
||||
if (result != VK_SUCCESS && result != VK_SUBOPTIMAL_KHR) {
|
||||
return result;
|
||||
}
|
||||
|
||||
frame.imageAvailableSemaphoreConsumed = false;
|
||||
return vkResetFences(device, 1, &frame.imageInFlightFence);
|
||||
const VkResult resetResult = vkResetFences(device, 1, &frame.imageInFlightFence);
|
||||
// Hand the acquire's own code back so the caller can schedule a rebuild.
|
||||
return resetResult == VK_SUCCESS ? result : resetResult;
|
||||
}
|
||||
|
||||
Uint32 FrameContext::GetCurrentFrameIndex() const {
|
||||
@@ -264,7 +293,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
frame.retiredCommandBuffers.push_back(frame.commandBuffer);
|
||||
// lastSubmitIndex was just written by the renderer for the submission
|
||||
// that carried this command buffer.
|
||||
frame.retiredCommandBuffers.push_back({frame.commandBuffer, frame.lastSubmitIndex});
|
||||
frame.commandBuffer = replacement;
|
||||
return VK_SUCCESS;
|
||||
}
|
||||
@@ -274,12 +305,40 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && m_commandPool != VK_NULL_HANDLE) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, static_cast<Uint32>(frame.retiredCommandBuffers.size()),
|
||||
frame.retiredCommandBuffers.data());
|
||||
for (const auto& retired : frame.retiredCommandBuffers) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &retired.commandBuffer);
|
||||
}
|
||||
}
|
||||
frame.retiredCommandBuffers.clear();
|
||||
}
|
||||
|
||||
void FrameContext::FreeRetiredCommandBuffersCompletedUpTo(Uint64 completedSubmitIndex) {
|
||||
if (m_device == VK_NULL_HANDLE || m_commandPool == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto& frame : m_frames) {
|
||||
// Retired buffers are appended in submit order, so the completed
|
||||
// ones form a prefix.
|
||||
SizeT completedCount = 0;
|
||||
while (completedCount < frame.retiredCommandBuffers.size() &&
|
||||
frame.retiredCommandBuffers[completedCount].submitIndex <= completedSubmitIndex) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1,
|
||||
&frame.retiredCommandBuffers[completedCount].commandBuffer);
|
||||
++completedCount;
|
||||
}
|
||||
if (completedCount > 0) {
|
||||
frame.retiredCommandBuffers.erase(frame.retiredCommandBuffers.begin(),
|
||||
frame.retiredCommandBuffers.begin() + completedCount);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void FrameContext::FreeAllRetiredCommandBuffers() {
|
||||
for (auto& frame : m_frames) {
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
}
|
||||
}
|
||||
|
||||
void FrameContext::AssertValidFrameIndex(Uint32 frameIndex) const {
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(), "FrameContext index out of range");
|
||||
}
|
||||
|
||||
@@ -40,6 +40,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPresentInfoKHR presentInfo{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR};
|
||||
};
|
||||
|
||||
// A command buffer submitted mid-frame (FlushPendingCommands), tagged
|
||||
// with the submit-tracker index it was submitted under so it can be
|
||||
// freed as soon as that submission is observed complete - without
|
||||
// waiting for the slot's fence to be waited again (present-less flush
|
||||
// loops never wait it).
|
||||
struct RetiredCommandBuffer {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
Uint64 submitIndex = 0;
|
||||
};
|
||||
|
||||
struct FrameData {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
VkSemaphore imageAvailableSemaphore = VK_NULL_HANDLE;
|
||||
@@ -47,10 +57,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool isCommandRecording = false;
|
||||
Bool hasCommandBufferRecorded = false;
|
||||
Bool imageAvailableSemaphoreConsumed = false;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands) whose
|
||||
// execution is only known complete once this slot's fence has been
|
||||
// waited again; freed at that point.
|
||||
Vector<VkCommandBuffer> retiredCommandBuffers;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands),
|
||||
// appended in submit order; freed once their submission is known
|
||||
// complete (fence wait or completion poll).
|
||||
Vector<RetiredCommandBuffer> retiredCommandBuffers;
|
||||
// Submit-tracker index of this slot's most recent queue submission
|
||||
// (written by the renderer at submit time).
|
||||
Uint64 lastSubmitIndex = 0;
|
||||
@@ -79,9 +89,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Parks the current (already ended and submitted) command buffer on the
|
||||
// slot's retired list and installs a freshly allocated one, so recording
|
||||
// can restart while the submitted buffer is still executing. Retired
|
||||
// buffers are freed after the slot's fence is next waited.
|
||||
// buffers are freed after the slot's fence is next waited, or as soon
|
||||
// as their submission is observed complete.
|
||||
VkResult RetireCurrentCommandBuffer();
|
||||
|
||||
// Frees every retired command buffer whose tagged submission index is
|
||||
// known complete. Driven by the renderer's submit tracker on completion
|
||||
// events (fence waits and non-blocking polls), so present-less flush
|
||||
// loops reclaim their buffers without any extra wait.
|
||||
void FreeRetiredCommandBuffersCompletedUpTo(Uint64 completedSubmitIndex);
|
||||
// Frees every slot's retired command buffers. Only valid when the
|
||||
// caller has proven every queue submission complete.
|
||||
void FreeAllRetiredCommandBuffers();
|
||||
|
||||
Uint32 GetCurrentFrameIndex() const;
|
||||
Uint32 GetFrameCount() const;
|
||||
|
||||
|
||||
@@ -8,6 +8,8 @@
|
||||
|
||||
#include "PipelineFactory.h"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static const char* PrimitiveTopologyToString(VkPrimitiveTopology topology) {
|
||||
switch (topology) {
|
||||
@@ -108,6 +110,81 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
"vkCreatePipelineCache");
|
||||
}
|
||||
|
||||
// Must be called once, before any pipeline is created: the flag is not part of the
|
||||
// pipeline hash, so flipping it mid-life would serve cached pipelines built under the
|
||||
// old value.
|
||||
void PipelineFactory::SetSuppressBlendedDepthWrite(Bool enabled) {
|
||||
s_suppressBlendedDepthWrite = enabled;
|
||||
}
|
||||
|
||||
Bool PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(MG_Config::QuirkOverride quirkOverride,
|
||||
Uint32 vendorId) {
|
||||
static constexpr Uint32 kVendorIdQualcomm = 0x5143;
|
||||
switch (quirkOverride) {
|
||||
case MG_Config::QuirkOverride::ForceOn:
|
||||
return true;
|
||||
case MG_Config::QuirkOverride::ForceOff:
|
||||
return false;
|
||||
case MG_Config::QuirkOverride::Auto:
|
||||
default:
|
||||
return vendorId == kVendorIdQualcomm;
|
||||
}
|
||||
}
|
||||
|
||||
namespace {
|
||||
// MIN/MAX extremum blending: the signature of a depth-bounds accumulation pass
|
||||
// (MC 26.3 OIT writes vec4(-linD, linD, deviceZ, 0) under GL_MAX while writing
|
||||
// depth for its equality chain). MIN/MAX ignore blend factors per the Vulkan spec.
|
||||
//
|
||||
// Deliberately the ONLY shape stripped. A quirk should touch as little unrelated
|
||||
// content as possible, and a trace sweep of every fixture showed the wider
|
||||
// alternatives all cost more than they fix:
|
||||
// - additive ONE+ONE with a depth write matched zero draws of the 26.3 chain
|
||||
// (its transmittance/accumulate passes disable depth writes themselves) - the
|
||||
// only real content it caught was harmless additive glow effects (Create);
|
||||
// - sorted-transparency "over" blends (SRC_ALPHA-style) are order-dependent,
|
||||
// drawn once per surface, and rely on their depth writes for occlusion;
|
||||
// - separate-alpha accumulation over an over-blending color channel has no
|
||||
// known pairing with a depth-equality chain (color channel only, see tests).
|
||||
// If a future workload pairs another blend shape with an equality chain, widen
|
||||
// this with that evidence in hand rather than pre-emptively.
|
||||
Bool IsAccumulationBlend(const VkPipelineColorBlendAttachmentState& attachment) {
|
||||
return attachment.colorBlendOp == VK_BLEND_OP_MIN ||
|
||||
attachment.colorBlendOp == VK_BLEND_OP_MAX;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool PipelineFactory::ShouldSuppressDepthWrite(const PipelineCreatePayload& payload) {
|
||||
if (!payload.depthWriteEnable) {
|
||||
return false;
|
||||
}
|
||||
// A shader that assigns gl_FragDepth supplies depth itself rather than taking the
|
||||
// pipeline's interpolated Z, so a driver that varies the vertex position math
|
||||
// between pipelines cannot desynchronize it. (A gl_FragDepth = gl_FragCoord.z
|
||||
// passthrough is the exception that stays exposed; no known content pairs one with
|
||||
// an equality chain, and 26.3's composite is a genuine computed-depth writer.)
|
||||
if (payload.fragmentReplacesDepth) {
|
||||
return false;
|
||||
}
|
||||
for (Uint32 i = 0; i < payload.colorAttachmentCount; ++i) {
|
||||
const VkPipelineColorBlendAttachmentState& attachment = payload.colorBlendAttachments[i];
|
||||
if (attachment.blendEnable != VK_TRUE) {
|
||||
continue;
|
||||
}
|
||||
// All color writes masked: blending is moot (depth-prepass pattern that left
|
||||
// GL_BLEND enabled); stripping the depth write would delete the whole prepass.
|
||||
if (attachment.colorWriteMask == 0) {
|
||||
continue;
|
||||
}
|
||||
// Any attachment qualifies, not just attachment 0: the 26.3 transmittance pass
|
||||
// accumulates into a 2-target MRT and must stay stripped.
|
||||
if (IsAccumulationBlend(attachment)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
PipelineFactory::~PipelineFactory() {
|
||||
DestroyAll();
|
||||
if (m_pipelineCache != VK_NULL_HANDLE) {
|
||||
@@ -152,6 +229,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXH64_update(m_hashState, &payload.backStencilDepthFailOp, sizeof(payload.backStencilDepthFailOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.backStencilCompareOp, sizeof(payload.backStencilCompareOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.fragmentReplacesDepth, sizeof(payload.fragmentReplacesDepth)));
|
||||
if (payload.colorAttachmentCount > 0) {
|
||||
XXHASH_VERIFY(XXH64_update(
|
||||
m_hashState,
|
||||
@@ -165,23 +244,108 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const HashType hash = ComputeHash(payload);
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
return it->second;
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
return it->second.pipeline;
|
||||
}
|
||||
|
||||
VkPipeline pipeline = CreatePipeline(payload);
|
||||
m_cache.emplace(hash, pipeline);
|
||||
m_cache.emplace(hash, PipelineCacheEntry{pipeline, payload.programHash, payload.renderPass,
|
||||
m_frameCounter});
|
||||
return pipeline;
|
||||
}
|
||||
|
||||
void PipelineFactory::DestroyAll() {
|
||||
for (auto& pair : m_cache) {
|
||||
if (pair.second != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, pair.second, nullptr);
|
||||
if (pair.second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, pair.second.pipeline, nullptr);
|
||||
}
|
||||
}
|
||||
m_cache.clear();
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::OnFrameBoundary() {
|
||||
++m_frameCounter;
|
||||
|
||||
// Sweep cadence and retire age mirror VkRenderPassManager::OnPresent: an entry
|
||||
// idle for more than kRetireAgeFrames frame boundaries cannot be referenced by
|
||||
// any in-flight command buffer (frames-in-flight <= MOBILEGL_MAGMA_FRAMESINFLIGHT),
|
||||
// so immediate vkDestroyPipeline is safe. The caller must drop its "last
|
||||
// pipeline" memo when this returns non-zero: the memo can return a cached
|
||||
// handle without touching this cache, so an evicted pipeline may still be
|
||||
// memoized (present-less flush loops never reset the memo per frame).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeFrames = 1024;
|
||||
if ((m_frameCounter % kSweepInterval) != 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameCounter - it->second.lastUsedFrame > kRetireAgeFrames) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::OnFrameBoundary: evicted %u idle pipelines (%zu remain)", evicted,
|
||||
m_cache.size());
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::EvictByRenderPasses(const Vector<VkRenderPass>& renderPasses) {
|
||||
if (renderPasses.empty() || m_cache.empty()) {
|
||||
return 0;
|
||||
}
|
||||
// Sorted-batch membership test keeps a mass eviction (shader-pack switch,
|
||||
// dimension exit) at one O(cache * log batch) scan instead of one full scan
|
||||
// per dying pass.
|
||||
Vector<VkRenderPass> sortedPasses = renderPasses;
|
||||
std::sort(sortedPasses.begin(), sortedPasses.end());
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (std::binary_search(sortedPasses.begin(), sortedPasses.end(), it->second.renderPass)) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::EvictByRenderPasses: evicted %u pipelines for %zu destroyed render passes",
|
||||
evicted, sortedPasses.size());
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::EvictByProgramHash(HashType programHash) {
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (it->second.programHash == programHash) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::EvictByProgramHash: evicted %u pipelines for program hash 0x%llx",
|
||||
evicted, static_cast<unsigned long long>(programHash));
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
VkPipeline PipelineFactory::CreatePipeline(const PipelineCreatePayload& payload) const {
|
||||
MOBILEGL_ASSERT(payload.stages != nullptr && !payload.stages->empty(), "PipelineFactory: stages are empty");
|
||||
MOBILEGL_ASSERT(payload.vertexInputState != nullptr, "PipelineFactory: vertexInputState is null");
|
||||
@@ -258,6 +422,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (Uint32 i = 0; i < payload.colorAttachmentCount; ++i) {
|
||||
colorAttachments[i] = payload.colorBlendAttachments[i];
|
||||
}
|
||||
// Suppress depth writes on accumulation-blended pipelines when the active driver
|
||||
// cannot keep vertex positions invariant across the pipelines of a multi-pass
|
||||
// depth-equality chain (see SetSuppressBlendedDepthWrite). The decision is narrowed
|
||||
// in ShouldSuppressDepthWrite: sorted-transparency "over" blends (vanilla MC water),
|
||||
// gl_FragDepth writers, and masked-out attachments keep their depth writes.
|
||||
// This bakes the decision into the pipeline, which only works because depth write is
|
||||
// static state here - adding VK_DYNAMIC_STATE_DEPTH_WRITE_ENABLE to kDynamicStates
|
||||
// would let the record-time value override it and silently disable the quirk.
|
||||
if (s_suppressBlendedDepthWrite && ShouldSuppressDepthWrite(payload)) {
|
||||
depthStencil.depthWriteEnable = VK_FALSE;
|
||||
}
|
||||
VkPipelineColorBlendStateCreateInfo blend{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO};
|
||||
blend.logicOpEnable = payload.logicOpEnable ? VK_TRUE : VK_FALSE;
|
||||
blend.logicOp = payload.logicOp;
|
||||
|
||||
@@ -49,6 +49,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkStencilOp backStencilPassOp = VK_STENCIL_OP_KEEP;
|
||||
VkStencilOp backStencilDepthFailOp = VK_STENCIL_OP_KEEP;
|
||||
VkCompareOp backStencilCompareOp = VK_COMPARE_OP_ALWAYS;
|
||||
// The fragment module writes gl_FragDepth (SPIR-V DepthReplacing); exempts the
|
||||
// pipeline from the blended depth-write quirk (see ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
Array<VkPipelineColorBlendAttachmentState, kMaxColorAttachments> colorBlendAttachments{};
|
||||
const Vector<VkPipelineShaderStageCreateInfo>* stages = nullptr;
|
||||
const VkPipelineVertexInputStateCreateInfo* vertexInputState = nullptr;
|
||||
@@ -62,13 +65,69 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPipeline GetOrCreatePipeline(const PipelineCreatePayload& payload);
|
||||
void DestroyAll();
|
||||
|
||||
// Frame boundary hook: ages the pipeline cache and destroys long-unused entries
|
||||
// (their command buffers retired many frames ago), mirroring
|
||||
// VkRenderPassManager::OnPresent's sweep. Returns the number of pipelines
|
||||
// destroyed so the caller can drop any memoized VkPipeline handle.
|
||||
Uint32 OnFrameBoundary();
|
||||
// Destroys every cached pipeline hashed on one of `renderPasses`. Only safe
|
||||
// when the caller guarantees GPU idleness for them - the render-pass manager
|
||||
// calls this (via the renderer) for passes its own >1024-boundary-idle sweep
|
||||
// just evicted, and a pipeline hashed on those handles is only ever bound by
|
||||
// draws that also hit the render-pass entries. Also closes the handle-recycling
|
||||
// hazard: a recycled VkRenderPass value must never serve a stale pipeline.
|
||||
// Batched: one cache scan regardless of how many passes died in the sweep.
|
||||
// Returns the number destroyed (callers invalidate memos when non-zero).
|
||||
Uint32 EvictByRenderPasses(const Vector<VkRenderPass>& renderPasses);
|
||||
// Destroys every cached pipeline built from the program with content hash
|
||||
// `programHash`. Called from the ProgramFactory eviction path, which proves the
|
||||
// same >1024-boundary idleness (the program's pipelines are only bound by draws
|
||||
// that stamp its factory entry). Returns the number destroyed.
|
||||
Uint32 EvictByProgramHash(HashType programHash);
|
||||
|
||||
// Driver quirk: suppress depth writes on accumulation-blended pipelines. Multi-pass
|
||||
// depth-equality rendering (a blended prepass writes depth that later passes re-test
|
||||
// with an equality-inclusive compare on the re-rasterized geometry) requires
|
||||
// cross-pipeline position invariance that some mobile compilers do not provide, even
|
||||
// with the SPIR-V Invariant decoration; whole primitives then drop out of the later
|
||||
// passes. Only MIN/MAX extremum blends are stripped - the signature of such a
|
||||
// chain's depth-bounds pass (MC 26.3 OIT), and per a fixture-wide trace sweep the
|
||||
// only depth-writing shape the chain actually uses - so every other blend
|
||||
// (sorted-transparency "over" like vanilla MC water, additive glows, ...) keeps
|
||||
// its depth writes. Set at renderer initialization based on the active driver.
|
||||
static void SetSuppressBlendedDepthWrite(Bool enabled);
|
||||
static Bool IsSuppressBlendedDepthWriteEnabled() { return s_suppressBlendedDepthWrite; }
|
||||
// Device gate for the quirk: ForceOn/ForceOff bypass detection, Auto enables it on
|
||||
// the known-affected vendor (Qualcomm).
|
||||
static Bool ShouldSuppressBlendedDepthWriteForDevice(MG_Config::QuirkOverride quirkOverride,
|
||||
Uint32 vendorId);
|
||||
// Pure per-pipeline strip decision (exempts gl_FragDepth writers, masked-out and
|
||||
// non-accumulation blends); combined with the device flag in CreatePipeline. Static
|
||||
// and payload-only so tests can pin the contract without a VkDevice.
|
||||
static Bool ShouldSuppressDepthWrite(const PipelineCreatePayload& payload);
|
||||
|
||||
private:
|
||||
struct PipelineCacheEntry {
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
// The hashed inputs the eviction paths key on: programHash ties the entry to
|
||||
// its ProgramFactory entry, renderPass records the exact handle the hash
|
||||
// folded in (the hash is one-way, so targeted eviction needs them verbatim).
|
||||
HashType programHash = 0;
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
// Frame-boundary counter value of the last GetOrCreatePipeline hit; drives
|
||||
// cache eviction (see OnFrameBoundary).
|
||||
Uint64 lastUsedFrame = 0;
|
||||
};
|
||||
|
||||
VkPipeline CreatePipeline(const PipelineCreatePayload& payload) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPipelineCache m_pipelineCache = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, VkPipeline> m_cache;
|
||||
UnorderedMap<HashType, PipelineCacheEntry> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
static inline Bool s_suppressBlendedDepthWrite = false;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -14,8 +14,16 @@
|
||||
#include "MG_State/GLState/TextureState/TextureEnum.h"
|
||||
|
||||
#include <Includes.h>
|
||||
#include <spirv_reflect.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class SamplerNumericDomain : Uint8 {
|
||||
Unknown = 0,
|
||||
Float,
|
||||
SignedInteger,
|
||||
UnsignedInteger,
|
||||
};
|
||||
|
||||
class ProgramFactory {
|
||||
public:
|
||||
enum class DescriptorBindingKind : Uint8 {
|
||||
@@ -34,6 +42,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
SurfaceRotate90 = 1 << 2,
|
||||
SurfaceRotate180 = 1 << 3,
|
||||
SurfaceRotate270 = 1 << 4,
|
||||
// Rewrites the fragment stage's implicit-LOD image samples to explicit LOD 0.
|
||||
// Only ever set for a draw whose every sampler binding is clamped to a single mip
|
||||
// level, which makes the two forms produce identical texels (the implicit lambda is
|
||||
// clamped into [minLod, maxLod] = [0, 0] regardless of derivatives or bias).
|
||||
ExplicitLod0Sampling = 1 << 5,
|
||||
// Fragment arithmetic may run at relaxed (fp16) precision. Only requested for draws
|
||||
// where every sampled texture and every colour attachment is an 8-bit-or-less
|
||||
// normalized format, so nothing the shader reads or writes carries more precision
|
||||
// than fp16 already represents exactly.
|
||||
RelaxedFragmentPrecision = 1 << 6,
|
||||
};
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
@@ -51,11 +69,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<DescriptorBindingKind> bindingKinds;
|
||||
Vector<Uint32> dynamicBindings;
|
||||
Vector<Int> uniformBlockIndexByBinding;
|
||||
// Descriptor count per binding (1 except for UBO instance arrays, which occupy one
|
||||
// binding with descriptorCount = N).
|
||||
Vector<Uint16> bindingDescriptorCounts;
|
||||
// Per-element GL uniform block indices for arrayed UBO bindings (count > 1);
|
||||
// element 0 of a non-arrayed binding stays in uniformBlockIndexByBinding.
|
||||
UnorderedMap<Uint32, Vector<Int>> arrayedUniformBlockIndicesByBinding;
|
||||
Vector<String> samplerNameByBinding;
|
||||
Vector<Int> samplerUniformLocationByBinding;
|
||||
Vector<TextureTarget> samplerTextureTargetByBinding;
|
||||
Vector<SamplerNumericDomain> samplerNumericDomainByBinding;
|
||||
Vector<VkFormat> storageImageFormatByBinding;
|
||||
Vector<Bool> storageImageUsesBindingFormatByBinding;
|
||||
Vector<String> storageBlockNameByBinding;
|
||||
Vector<Int> storageBlockIndexByBinding;
|
||||
// Set once during ReflectLayout so the per-draw path can skip the whole
|
||||
// storage-image preparation for the overwhelming majority of programs.
|
||||
Bool hasStorageImages = false;
|
||||
Int globalUboBinding = -1;
|
||||
Uint32 activeVertexInputLocationMask = 0;
|
||||
Array<GLenum, kMaxVertexInputLocations> vertexInputTypes{};
|
||||
@@ -64,6 +94,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
ShaderStage rasterizationProducerStage = ShaderStage::Unknown;
|
||||
Uint32 producerOutputComponentCount = 0;
|
||||
Uint32 fragmentInputComponentCount = 0;
|
||||
// The fragment module declares the DepthReplacing execution mode (writes
|
||||
// gl_FragDepth); shader-computed depth is immune to the cross-pipeline
|
||||
// position-invariance quirk (see PipelineFactory::ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
// Frame-boundary counter value of the last GetOrCreateProgram hit; drives
|
||||
// cache eviction (see OnFrameBoundary).
|
||||
Uint64 lastUsedFrame = 0;
|
||||
|
||||
static inline VkDevice s_device = VK_NULL_HANDLE;
|
||||
|
||||
@@ -79,11 +116,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
dynamicBindings = std::move(other.dynamicBindings);
|
||||
uniformBlockIndexByBinding = std::move(other.uniformBlockIndexByBinding);
|
||||
bindingDescriptorCounts = std::move(other.bindingDescriptorCounts);
|
||||
arrayedUniformBlockIndicesByBinding = std::move(other.arrayedUniformBlockIndicesByBinding);
|
||||
samplerNameByBinding = std::move(other.samplerNameByBinding);
|
||||
samplerUniformLocationByBinding = std::move(other.samplerUniformLocationByBinding);
|
||||
samplerTextureTargetByBinding = std::move(other.samplerTextureTargetByBinding);
|
||||
samplerNumericDomainByBinding = std::move(other.samplerNumericDomainByBinding);
|
||||
storageImageFormatByBinding = std::move(other.storageImageFormatByBinding);
|
||||
storageImageUsesBindingFormatByBinding =
|
||||
std::move(other.storageImageUsesBindingFormatByBinding);
|
||||
storageBlockNameByBinding = std::move(other.storageBlockNameByBinding);
|
||||
storageBlockIndexByBinding = std::move(other.storageBlockIndexByBinding);
|
||||
hasStorageImages = other.hasStorageImages;
|
||||
globalUboBinding = other.globalUboBinding;
|
||||
activeVertexInputLocationMask = other.activeVertexInputLocationMask;
|
||||
vertexInputTypes = other.vertexInputTypes;
|
||||
@@ -92,15 +136,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
rasterizationProducerStage = other.rasterizationProducerStage;
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.globalUboBinding = -1;
|
||||
other.activeVertexInputLocationMask = 0;
|
||||
other.activeFragmentOutputLocationMask = 0;
|
||||
other.rasterizationProducerStage = ShaderStage::Unknown;
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.lastUsedFrame = 0;
|
||||
}
|
||||
VkProgramObject& operator=(VkProgramObject&& other) noexcept {
|
||||
if (this == &other) {
|
||||
@@ -115,11 +164,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
dynamicBindings = std::move(other.dynamicBindings);
|
||||
uniformBlockIndexByBinding = std::move(other.uniformBlockIndexByBinding);
|
||||
bindingDescriptorCounts = std::move(other.bindingDescriptorCounts);
|
||||
arrayedUniformBlockIndicesByBinding = std::move(other.arrayedUniformBlockIndicesByBinding);
|
||||
samplerNameByBinding = std::move(other.samplerNameByBinding);
|
||||
samplerUniformLocationByBinding = std::move(other.samplerUniformLocationByBinding);
|
||||
samplerTextureTargetByBinding = std::move(other.samplerTextureTargetByBinding);
|
||||
samplerNumericDomainByBinding = std::move(other.samplerNumericDomainByBinding);
|
||||
storageImageFormatByBinding = std::move(other.storageImageFormatByBinding);
|
||||
storageImageUsesBindingFormatByBinding =
|
||||
std::move(other.storageImageUsesBindingFormatByBinding);
|
||||
storageBlockNameByBinding = std::move(other.storageBlockNameByBinding);
|
||||
storageBlockIndexByBinding = std::move(other.storageBlockIndexByBinding);
|
||||
hasStorageImages = other.hasStorageImages;
|
||||
globalUboBinding = other.globalUboBinding;
|
||||
activeVertexInputLocationMask = other.activeVertexInputLocationMask;
|
||||
vertexInputTypes = other.vertexInputTypes;
|
||||
@@ -128,15 +184,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
rasterizationProducerStage = other.rasterizationProducerStage;
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.globalUboBinding = -1;
|
||||
other.activeVertexInputLocationMask = 0;
|
||||
other.activeFragmentOutputLocationMask = 0;
|
||||
other.rasterizationProducerStage = ShaderStage::Unknown;
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.lastUsedFrame = 0;
|
||||
return *this;
|
||||
}
|
||||
|
||||
@@ -166,10 +227,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
};
|
||||
|
||||
// Notified when the OnFrameBoundary sweep destroys an aged-out cache entry,
|
||||
// carrying the entry's content hash and the VkDescriptorSetLayout it owned.
|
||||
// Dependent caches (compute pipelines, PipelineFactory entries, UniformManager's
|
||||
// per-layout descriptor sets) must purge in the same step: after vkDestroy the
|
||||
// layout handle value may be recycled for an unrelated layout, and the program
|
||||
// hash may be re-inserted by a later rebuild of the same content.
|
||||
class IEvictionObserver {
|
||||
public:
|
||||
virtual ~IEvictionObserver() = default;
|
||||
virtual void OnProgramEvicted(HashType programHash, VkDescriptorSetLayout descriptorSetLayout) = 0;
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings = 16,
|
||||
Bool shaderDrawParametersEnabled = false)
|
||||
Bool shaderDrawParametersEnabled = false,
|
||||
Bool unformattedFloatStorageImagesEnabled = false)
|
||||
: m_device(device), m_maxBindings(maxBindings), m_config(config),
|
||||
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled) {
|
||||
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled),
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled) {
|
||||
VkProgramObject::s_device = device;
|
||||
}
|
||||
~ProgramFactory() = default;
|
||||
@@ -179,7 +254,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VkProgramObject& GetOrCreateProgram(
|
||||
const MG_State::GLState::ProgramObject& program, CompileOptionFlags flags);
|
||||
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetEvictionObserver(IEvictionObserver* observer) { m_evictionObserver = observer; }
|
||||
// Frame boundary hook: ages the program cache and evicts long-unused entries
|
||||
// (their command buffers retired many frames ago), mirroring
|
||||
// VkRenderPassManager::OnPresent's sweep.
|
||||
void OnFrameBoundary();
|
||||
|
||||
static VkShaderStageFlagBits ToVkStage(ShaderStage stage);
|
||||
static VkFormat ConvertSpirvImageFormatToVkFormat(SpvImageFormat format);
|
||||
static SamplerNumericDomain UniformTypeToSamplerNumericDomain(GLenum glType);
|
||||
// True when any entry point declares the DepthReplacing execution mode, i.e. the
|
||||
// shader assigns gl_FragDepth. Exposed so the blended depth-write quirk's exemption
|
||||
// can be pinned by tests. A false negative loses the exemption, so such a shader is
|
||||
// stripped conservatively and forfeits its depth write.
|
||||
static Bool ReflectedFragmentReplacesDepth(const SpvReflectShaderModule& reflectModule);
|
||||
// True when an entry point reads the InstanceIndex builtin. Only gates a diagnostic:
|
||||
// without shaderDrawParameters such a shader cannot have gl_InstanceID rebased.
|
||||
static Bool ReflectedReadsInstanceIndexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
|
||||
private:
|
||||
struct ProgramLookupCache {
|
||||
@@ -206,7 +298,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// True when the device enabled shaderDrawParameters; gates the InstanceIndex rebase pass
|
||||
// (which needs the DrawParameters capability / gl_BaseInstance builtin).
|
||||
Bool m_shaderDrawParametersEnabled = false;
|
||||
// True only when the logical device enabled both
|
||||
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
mutable ProgramLookupCache m_lastLookup;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -247,6 +247,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_surfaceFormat = {createInfo.imageFormat, createInfo.imageColorSpace};
|
||||
m_extent = createInfo.imageExtent;
|
||||
// The surface-space extent this swapchain was built from, i.e. before the
|
||||
// quarter-turn swap above. Out-of-date checks must compare in THIS space: comparing a
|
||||
// freshly queried currentExtent against the swapped m_extent flips axes every rotation
|
||||
// and makes the comparison alternate forever.
|
||||
m_surfaceExtent = defaultFramebufferExtent;
|
||||
m_preTransform = createInfo.preTransform;
|
||||
|
||||
VK_VERIFY(vkCreateSwapchainKHR(device, &createInfo, nullptr, &m_swapchain));
|
||||
|
||||
@@ -35,6 +35,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSwapchainKHR GetHandle() const { return m_swapchain; }
|
||||
const VkSurfaceFormatKHR& GetSurfaceFormat() const { return m_surfaceFormat; }
|
||||
VkExtent2D GetExtent() const { return m_extent; }
|
||||
// Surface-space extent (before the pre-rotation quarter-turn swap) this swapchain was
|
||||
// created from - the value to compare a freshly queried currentExtent against.
|
||||
VkExtent2D GetSurfaceExtent() const { return m_surfaceExtent; }
|
||||
VkSurfaceTransformFlagBitsKHR GetPreTransform() const { return m_preTransform; }
|
||||
const Vector<VkImage>& GetImages() const { return m_images; }
|
||||
const Vector<VkImageView>& GetImageViews() const { return m_imageViews; }
|
||||
@@ -63,6 +66,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSwapchainKHR m_swapchain = VK_NULL_HANDLE;
|
||||
VkSurfaceFormatKHR m_surfaceFormat{};
|
||||
VkExtent2D m_extent{};
|
||||
VkExtent2D m_surfaceExtent{};
|
||||
VkSurfaceTransformFlagBitsKHR m_preTransform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
|
||||
Vector<VkImage> m_images;
|
||||
Vector<VkImageView> m_imageViews;
|
||||
|
||||
@@ -13,8 +13,10 @@
|
||||
#include "MG_State/GLState/ProgramState/ProgramObject.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObject2D.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObjectBuffer.h"
|
||||
#include "MG_Util/Converters/GLToMG/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include <vulkan/utility/vk_format_utils.h>
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include <Config.h>
|
||||
#include <cstdio>
|
||||
@@ -78,6 +80,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return uniformUnit >= 0 ? uniformUnit : 0;
|
||||
}
|
||||
|
||||
VkFormat UniformManager::ResolveStorageImageViewFormat(VkFormat reflectedFormat, GLenum bindingFormat,
|
||||
VkFormat resourceFormat, Bool useBindingFormat) {
|
||||
if (useBindingFormat) {
|
||||
const TextureInternalFormat bindingInternalFormat =
|
||||
MG_Util::ConvertGLEnumToTextureInternalFormat(bindingFormat);
|
||||
return MG_Util::ConvertTextureInternalFormatToVkEnum(bindingInternalFormat);
|
||||
}
|
||||
return reflectedFormat != VK_FORMAT_UNDEFINED ? reflectedFormat : resourceFormat;
|
||||
}
|
||||
|
||||
Bool UniformManager::Initialize(VkDevice device, VkBufferManager* bufferManager,
|
||||
ProgramFactory* programFactory,
|
||||
VkDeviceSize minUniformBufferOffsetAlignment, Uint32 frameCount,
|
||||
@@ -200,6 +212,40 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
void UniformManager::OnDescriptorSetLayoutDestroyed(VkDescriptorSetLayout descriptorSetLayout) {
|
||||
SizeT purgedSets = 0;
|
||||
for (auto& frame : m_frames) {
|
||||
const auto it = frame.descriptorSetCacheByLayout.find(descriptorSetLayout);
|
||||
if (it == frame.descriptorSetCacheByLayout.end()) {
|
||||
continue;
|
||||
}
|
||||
// Free the sets back to their pools and credit the bucket accounting, so
|
||||
// program churn recycles pool capacity instead of abandoning the slots.
|
||||
// GPU-safe: the layout only dies after >1024 idle frame boundaries, so no
|
||||
// in-flight command buffer references these sets.
|
||||
for (const auto& cached : it->second.sets) {
|
||||
if (cached.set == VK_NULL_HANDLE) {
|
||||
continue;
|
||||
}
|
||||
vkFreeDescriptorSets(m_device, cached.pool, 1, &cached.set);
|
||||
const auto bucket = std::find_if(
|
||||
frame.descriptorPools.begin(), frame.descriptorPools.end(),
|
||||
[&cached](const DescriptorPoolBucket& candidate) { return candidate.handle == cached.pool; });
|
||||
if (bucket != frame.descriptorPools.end() && bucket->allocatedSets > 0) {
|
||||
--bucket->allocatedSets;
|
||||
}
|
||||
}
|
||||
purgedSets += it->second.sets.size();
|
||||
frame.descriptorSetCacheByLayout.erase(it);
|
||||
}
|
||||
if (purgedSets > 0) {
|
||||
// The per-draw reuse memo folds the layout handle into its signature; drop
|
||||
// it so a recycled handle value cannot revive a purged set mid-frame.
|
||||
m_hasLastDescriptor = false;
|
||||
MGLOG_D("UniformDescriptorBinder: freed %zu descriptor sets for destroyed layout", purgedSets);
|
||||
}
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveSamplerDescriptor(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
@@ -276,6 +322,55 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
"ResolveSamplerDescriptor: invalid sampled image layout=%d for textureId=%d, binding=%u",
|
||||
static_cast<Int>(resource->layout), texture->GetExternalIndex(), binding);
|
||||
}
|
||||
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerNumericDomainByBinding.size(),
|
||||
"ResolveSamplerDescriptor: sampler numeric-domain binding %u out of range", binding);
|
||||
const SamplerNumericDomain numericDomain = programObj.samplerNumericDomainByBinding[binding];
|
||||
// Vulkan forbids linear filtering and anisotropy for integer sampled-image formats.
|
||||
// Some desktop GL shader packs deliberately bit-read a mutable float texture through a
|
||||
// usampler and still leave the texture's ordinary linear parameters in place; texelFetch
|
||||
// ignores filtering, so a nearest VkSampler preserves the operation while keeping the
|
||||
// descriptor valid.
|
||||
const Bool forceNearestFiltering = numericDomain == SamplerNumericDomain::SignedInteger ||
|
||||
numericDomain == SamplerNumericDomain::UnsignedInteger;
|
||||
SamplerResolveMemo* viewFormatMemo =
|
||||
binding < m_samplerResolveMemo.size() ? &m_samplerResolveMemo[binding] : nullptr;
|
||||
VkFormat sampledViewFormat;
|
||||
if (viewFormatMemo != nullptr && viewFormatMemo->viewFormatValid &&
|
||||
viewFormatMemo->viewFormatSource == resource->format &&
|
||||
viewFormatMemo->viewFormatDomain == numericDomain) {
|
||||
sampledViewFormat = viewFormatMemo->viewFormat;
|
||||
} else {
|
||||
sampledViewFormat =
|
||||
VkTextureManager::ResolveSampledImageViewFormat(resource->format, numericDomain);
|
||||
if (viewFormatMemo != nullptr) {
|
||||
viewFormatMemo->viewFormatSource = resource->format;
|
||||
viewFormatMemo->viewFormatDomain = numericDomain;
|
||||
viewFormatMemo->viewFormat = sampledViewFormat;
|
||||
viewFormatMemo->viewFormatValid = true;
|
||||
}
|
||||
}
|
||||
if (sampledViewFormat == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_E("ResolveSamplerDescriptor: no compatible sampled view for binding=%u ('%s') "
|
||||
"textureId=%d imageFormat=%d numericDomain=%d",
|
||||
binding, programObj.samplerNameByBinding[binding].c_str(), texture->GetExternalIndex(),
|
||||
static_cast<Int>(resource->format), static_cast<Int>(numericDomain));
|
||||
return false;
|
||||
}
|
||||
// No reinterpretation requested: bind the depth-or-color aspect view the sync above
|
||||
// already produced instead of re-entering GetOrCreateSampledImageView's sync path.
|
||||
const VkImageView sampledImageView =
|
||||
sampledViewFormat == resource->format
|
||||
? resource->sampledView
|
||||
: m_textureManager->GetOrCreateSampledImageView(*texture, sampledViewFormat);
|
||||
if (sampledImageView == VK_NULL_HANDLE) {
|
||||
MGLOG_E("ResolveSamplerDescriptor: failed to resolve sampled view for binding=%u ('%s') "
|
||||
"textureId=%d imageFormat=%d viewFormat=%d numericDomain=%d",
|
||||
binding, programObj.samplerNameByBinding[binding].c_str(), texture->GetExternalIndex(),
|
||||
static_cast<Int>(resource->format), static_cast<Int>(sampledViewFormat),
|
||||
static_cast<Int>(numericDomain));
|
||||
return false;
|
||||
}
|
||||
// Skip GetOrCreateSampler's per-draw key hash + map lookup when this binding's
|
||||
// sampler object and texture (both by lifetime id + version) are unchanged from the
|
||||
// last draw that resolved it: the resulting sampler key, and therefore the VkSampler
|
||||
@@ -290,24 +385,32 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const Uint16 samplerVersion = samplerToUse->GetVersion();
|
||||
const Uint64 textureLifetimeId = texture->GetLifetimeId();
|
||||
const Uint16 textureParamsVersion = texture->GetTextureParamsVersion();
|
||||
// The sampler's LOD clamp depends on how many levels the sampled view exposes, and that
|
||||
// follows uploads as well as GL parameters - so it belongs in the memo key too.
|
||||
const Uint32 viewLevelCount = resource->sampledLevelCount;
|
||||
if (memo.valid && memo.samplerLifetimeId == samplerLifetimeId && memo.samplerVersion == samplerVersion &&
|
||||
memo.textureLifetimeId == textureLifetimeId && memo.textureParamsVersion == textureParamsVersion) {
|
||||
memo.textureLifetimeId == textureLifetimeId && memo.textureParamsVersion == textureParamsVersion &&
|
||||
memo.forceNearestFiltering == forceNearestFiltering && memo.viewLevelCount == viewLevelCount) {
|
||||
resolvedSampler = memo.sampler;
|
||||
} else {
|
||||
resolvedSampler = m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture);
|
||||
resolvedSampler = m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture,
|
||||
forceNearestFiltering, viewLevelCount);
|
||||
memo.samplerLifetimeId = samplerLifetimeId;
|
||||
memo.samplerVersion = samplerVersion;
|
||||
memo.textureLifetimeId = textureLifetimeId;
|
||||
memo.textureParamsVersion = textureParamsVersion;
|
||||
memo.forceNearestFiltering = forceNearestFiltering;
|
||||
memo.viewLevelCount = viewLevelCount;
|
||||
memo.sampler = resolvedSampler;
|
||||
memo.valid = true;
|
||||
}
|
||||
} else {
|
||||
resolvedSampler = m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture);
|
||||
resolvedSampler = m_samplerManager->GetOrCreateSampler(*samplerToUse, *texture, forceNearestFiltering,
|
||||
resource->sampledLevelCount);
|
||||
}
|
||||
outImageInfo = {
|
||||
.sampler = resolvedSampler,
|
||||
.imageView = resource->sampledView != VK_NULL_HANDLE ? resource->sampledView : resource->fullView,
|
||||
.imageView = sampledImageView,
|
||||
.imageLayout = resource->layout,
|
||||
};
|
||||
return outImageInfo.sampler != VK_NULL_HANDLE;
|
||||
@@ -344,6 +447,106 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return outImageInfo.sampler != VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
namespace {
|
||||
// fp16 carries an 11-bit mantissa, so an 8-bit normalized channel round-trips exactly.
|
||||
// Anything wider - 16-bit normalized, half float, full float, and every packed HDR
|
||||
// encoding - holds precision or range that relaxing the arithmetic would throw away.
|
||||
Bool IsLowPrecisionNormalizedFormat(VkFormat format) {
|
||||
if (format == VK_FORMAT_UNDEFINED) return false;
|
||||
if (!vkuFormatIsUNORM(format) && !vkuFormatIsSNORM(format) && !vkuFormatIsSRGB(format)) {
|
||||
return false;
|
||||
}
|
||||
const struct VKU_FORMAT_INFO info = vkuGetFormatInfo(format);
|
||||
for (Uint32 i = 0; i < info.component_count; ++i) {
|
||||
if (info.components[i].size > 8) return false;
|
||||
}
|
||||
return info.component_count > 0;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool UniformManager::DrawTargetIsLowPrecision(const MG_State::GLState::FramebufferObject* drawFramebuffer) {
|
||||
// Default framebuffer: the swapchain is an 8-bit normalized surface.
|
||||
if (drawFramebuffer == nullptr) return true;
|
||||
|
||||
Bool sawColour = false;
|
||||
for (Int i = static_cast<Int>(FramebufferAttachmentType::Color0);
|
||||
i < static_cast<Int>(FramebufferAttachmentType::FramebufferAttachmentTypeCount);
|
||||
++i) {
|
||||
const auto& attachment =
|
||||
drawFramebuffer->GetAttachment(static_cast<FramebufferAttachmentType>(i));
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
if (const auto& texture = attachment.GetTexture()) {
|
||||
format = MG_Util::ConvertTextureInternalFormatToVkEnum(texture->GetFormat());
|
||||
} else if (const auto& renderbuffer = attachment.GetRenderbuffer()) {
|
||||
format = MG_Util::ConvertTextureInternalFormatToVkEnum(
|
||||
renderbuffer->GetInternalFormat());
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
if (!IsLowPrecisionNormalizedFormat(format)) return false;
|
||||
sawColour = true;
|
||||
}
|
||||
return sawColour;
|
||||
}
|
||||
|
||||
Bool UniformManager::ProgramSamplesOnlyLowPrecisionTextures(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj) {
|
||||
for (Uint32 binding = 0; binding < programObj.bindingKinds.size(); ++binding) {
|
||||
if (programObj.bindingKinds[binding] != ProgramFactory::DescriptorBindingKind::CombinedImageSampler) {
|
||||
continue;
|
||||
}
|
||||
const auto* texture = ResolveSamplerTextureRaw(program, programObj, binding);
|
||||
// An unresolvable binding is unknown territory, not licence to relax.
|
||||
if (texture == nullptr) return false;
|
||||
const VkFormat format =
|
||||
MG_Util::ConvertTextureInternalFormatToVkEnum(texture->GetFormat());
|
||||
if (!IsLowPrecisionNormalizedFormat(format)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::ProgramSamplesOnlySingleLevelTextures(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj) {
|
||||
Bool sawSampler = false;
|
||||
for (Uint32 binding = 0; binding < programObj.bindingKinds.size(); ++binding) {
|
||||
if (programObj.bindingKinds[binding] != ProgramFactory::DescriptorBindingKind::CombinedImageSampler) {
|
||||
continue;
|
||||
}
|
||||
const auto* texture = ResolveSamplerTextureRaw(program, programObj, binding);
|
||||
if (texture == nullptr) return false;
|
||||
const auto& levelRange = texture->GetLevelRange();
|
||||
if (levelRange.x() != levelRange.y()) return false;
|
||||
|
||||
// An explicit-LOD sample is a single filtered tap, so it also gives up anisotropic
|
||||
// filtering - which a single-level view can still have. Resolve the sampler exactly
|
||||
// the way ResolveSamplerDescriptor does and bail if anisotropy would apply.
|
||||
const Int location = programObj.samplerUniformLocationByBinding[binding];
|
||||
const Int unit = ResolveSamplerUnitIndex(program, location, binding);
|
||||
const auto& samplerOverride = MG_State::pGLContext->GetTextureUnitObject(unit).GetSamplerObject();
|
||||
const auto* effectiveSampler =
|
||||
samplerOverride ? samplerOverride.get() : texture->GetSamplerObject().get();
|
||||
if (effectiveSampler == nullptr) return false;
|
||||
if (effectiveSampler->GetMaxAnisotropy() > 1.0f &&
|
||||
effectiveSampler->GetMinFilter() == SamplerFilterMode::Linear &&
|
||||
effectiveSampler->GetMagFilter() == SamplerFilterMode::Linear) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// An explicit LOD 0 makes lambda exactly 0, which is the magnification side of the
|
||||
// min/mag decision. That only matches the implicit form when lambda could not have been
|
||||
// positive anyway (the LOD clamp already pins it at or below 0), or when the two
|
||||
// filters are the same and the choice cannot be observed.
|
||||
const Float effectiveMaxLod = effectiveSampler->GetMipmapMode() == SamplerMipmapMode::None
|
||||
? 0.0f
|
||||
: effectiveSampler->GetMaxLod();
|
||||
if (effectiveMaxLod > 0.0f && effectiveSampler->GetMinFilter() != effectiveSampler->GetMagFilter()) {
|
||||
return false;
|
||||
}
|
||||
sawSampler = true;
|
||||
}
|
||||
return sawSampler;
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveSamplerTexture(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture) {
|
||||
@@ -572,9 +775,32 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
const Uint32 mipLevel = static_cast<Uint32>(std::max<GLint>(0, imageBinding.Level));
|
||||
VkImageView view = m_textureManager->GetOrCreateViewAtMipLevel(*imageBinding.Texture, mipLevel);
|
||||
MOBILEGL_ASSERT(binding < programObj.storageImageFormatByBinding.size(),
|
||||
"ResolveStorageImageDescriptor: storage image format binding %u out of range", binding);
|
||||
MOBILEGL_ASSERT(binding < programObj.storageImageUsesBindingFormatByBinding.size(),
|
||||
"ResolveStorageImageDescriptor: storage image format policy binding %u out of range",
|
||||
binding);
|
||||
const VkFormat reflectedFormat = programObj.storageImageFormatByBinding[binding];
|
||||
const Bool useBindingFormat = programObj.storageImageUsesBindingFormatByBinding[binding];
|
||||
const VkFormat viewFormat = ResolveStorageImageViewFormat(
|
||||
reflectedFormat, imageBinding.Format, resource->format, useBindingFormat);
|
||||
if (viewFormat == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_E("ResolveStorageImageDescriptor: unsupported glBindImageTexture format=0x%x "
|
||||
"for binding=%u imageUnit=%d textureId=%d bindingPolicy=%s",
|
||||
imageBinding.Format, binding, imageUnit, imageBinding.Texture->GetExternalIndex(),
|
||||
useBindingFormat ? "true" : "false");
|
||||
return false;
|
||||
}
|
||||
const VkImageView view = m_textureManager->GetOrCreateStorageImageView(
|
||||
*imageBinding.Texture, mipLevel, viewFormat, imageBinding.Layered != GL_FALSE, imageBinding.Layer);
|
||||
if (view == VK_NULL_HANDLE) {
|
||||
view = resource->fullView;
|
||||
MGLOG_E("ResolveStorageImageDescriptor: failed to resolve storage view textureId=%d mip=%u "
|
||||
"bindingFormat=0x%x imageFormat=%d reflectedFormat=%d selectedFormat=%d bindingPolicy=%s",
|
||||
imageBinding.Texture->GetExternalIndex(), mipLevel, imageBinding.Format,
|
||||
static_cast<Int>(resource->format), static_cast<Int>(reflectedFormat),
|
||||
static_cast<Int>(viewFormat),
|
||||
useBindingFormat ? "true" : "false");
|
||||
return false;
|
||||
}
|
||||
outImageInfo.sampler = VK_NULL_HANDLE;
|
||||
outImageInfo.imageView = view;
|
||||
@@ -632,9 +858,53 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::CollectStorageImageTextures(
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures) const {
|
||||
outTextures.clear();
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr,
|
||||
"CollectStorageImageTextures: GL context is null");
|
||||
|
||||
const Uint32 bindingCount =
|
||||
std::min<Uint32>(m_maxBindings, static_cast<Uint32>(programObj.bindingKinds.size()));
|
||||
for (Uint32 binding = 0; binding < bindingCount; ++binding) {
|
||||
if (programObj.bindingKinds[binding] != ProgramFactory::DescriptorBindingKind::StorageImage) {
|
||||
continue;
|
||||
}
|
||||
if (binding >= programObj.samplerUniformLocationByBinding.size()) {
|
||||
MGLOG_E("CollectStorageImageTextures: binding %u has no uniform-location mapping", binding);
|
||||
return false;
|
||||
}
|
||||
|
||||
const Int location = programObj.samplerUniformLocationByBinding[binding];
|
||||
if (location < 0) {
|
||||
MGLOG_E("CollectStorageImageTextures: binding %u has no image uniform location", binding);
|
||||
return false;
|
||||
}
|
||||
const Int imageUnit = program.GetUniformSamplerOrImageUnitIndex(static_cast<Uint>(location));
|
||||
if (imageUnit < 0 || imageUnit >= MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS) {
|
||||
MGLOG_E("CollectStorageImageTextures: image unit %d is invalid for binding %u",
|
||||
imageUnit, binding);
|
||||
return false;
|
||||
}
|
||||
|
||||
auto* texture = MG_State::pGLContext->GetImageTextureBinding(imageUnit).Texture.get();
|
||||
if (texture == nullptr) {
|
||||
MGLOG_E("CollectStorageImageTextures: image unit %d is unbound for binding %u",
|
||||
imageUnit, binding);
|
||||
return false;
|
||||
}
|
||||
if (std::find(outTextures.begin(), outTextures.end(), texture) == outTextures.end()) {
|
||||
outTextures.push_back(texture);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveUniformBufferPayload(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
UboBindResult& out) const {
|
||||
Uint32 arrayElement, UboBindResult& out) const {
|
||||
const void* outData = nullptr;
|
||||
VkDeviceSize outSize = 0;
|
||||
|
||||
@@ -660,7 +930,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
MOBILEGL_ASSERT(binding < programObj.uniformBlockIndexByBinding.size(),
|
||||
"ResolveUniformBufferPayload: UBO mapping binding %u out of range", binding);
|
||||
const Int blockIndex = programObj.uniformBlockIndexByBinding[binding];
|
||||
Int blockIndex = programObj.uniformBlockIndexByBinding[binding];
|
||||
if (arrayElement > 0) {
|
||||
const auto arrayIt = programObj.arrayedUniformBlockIndicesByBinding.find(binding);
|
||||
const Bool elementValid = arrayIt != programObj.arrayedUniformBlockIndicesByBinding.end() &&
|
||||
arrayElement < arrayIt->second.size();
|
||||
MOBILEGL_ASSERT(elementValid,
|
||||
"ResolveUniformBufferPayload: UBO binding %u has no array element %u", binding,
|
||||
arrayElement);
|
||||
if (!elementValid) {
|
||||
return false;
|
||||
}
|
||||
blockIndex = arrayIt->second[arrayElement];
|
||||
}
|
||||
MOBILEGL_ASSERT(blockIndex >= 0,
|
||||
"ResolveUniformBufferPayload: no uniform block mapped to descriptor binding %u", binding);
|
||||
|
||||
@@ -768,6 +1050,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkDescriptorPoolCreateInfo poolInfo{};
|
||||
poolInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO;
|
||||
// FREE_DESCRIPTOR_SET_BIT lets a destroyed layout's cached sets be freed back
|
||||
// (OnDescriptorSetLayoutDestroyed) so program churn recycles pool capacity.
|
||||
// The cost is on set allocation only, which happens when a layout's per-frame
|
||||
// cache grows - never on the per-draw reuse path.
|
||||
poolInfo.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT;
|
||||
poolInfo.maxSets = maxSets;
|
||||
poolInfo.poolSizeCount = static_cast<Uint32>(std::size(poolSizes));
|
||||
poolInfo.pPoolSizes = poolSizes;
|
||||
@@ -847,7 +1134,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto& frame = m_frames[frameIndex];
|
||||
auto& cache = frame.descriptorSetCacheByLayout[programObj.descriptorSetLayout];
|
||||
if (cache.cursor < cache.sets.size()) {
|
||||
outDescriptorSet = cache.sets[cache.cursor++];
|
||||
outDescriptorSet = cache.sets[cache.cursor++].set;
|
||||
} else {
|
||||
VkResult allocResult = AllocateDescriptorSetsFromActivePool(frameIndex, programObj, outDescriptorSet);
|
||||
if (allocResult == VK_ERROR_OUT_OF_POOL_MEMORY || allocResult == VK_ERROR_FRAGMENTED_POOL) {
|
||||
@@ -861,7 +1148,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return allocResult;
|
||||
}
|
||||
|
||||
cache.sets.push_back(outDescriptorSet);
|
||||
// The successful allocation came from the bucket the alloc helper left
|
||||
// active; record it so a layout-destroyed purge can free the set back.
|
||||
cache.sets.push_back({outDescriptorSet, frame.descriptorPools[frame.activeDescriptorPoolIndex].handle});
|
||||
++cache.cursor;
|
||||
MGLOG_D("UniformDescriptorBinder: cached descriptor set count for frame=%u grew to %zu", frameIndex,
|
||||
cache.sets.size());
|
||||
@@ -907,11 +1196,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
imageInfos.clear();
|
||||
texelBufferViews.clear();
|
||||
dynamicOffsets.clear();
|
||||
// Arrayed UBO bindings contribute extra buffer infos and dynamic offsets; reserve for
|
||||
// the worst case so the pBufferInfo pointers taken below never dangle on reallocation.
|
||||
Uint32 uboArrayExtra = 0;
|
||||
for (const auto& arrayEntry : programObj.arrayedUniformBlockIndicesByBinding) {
|
||||
uboArrayExtra += static_cast<Uint32>(arrayEntry.second.size()) - 1u;
|
||||
}
|
||||
writes.reserve(m_maxBindings);
|
||||
bufferInfos.reserve(m_maxBindings);
|
||||
bufferInfos.reserve(m_maxBindings + uboArrayExtra);
|
||||
imageInfos.reserve(m_maxBindings);
|
||||
texelBufferViews.reserve(m_maxBindings);
|
||||
dynamicOffsets.reserve(programObj.dynamicBindings.size());
|
||||
dynamicOffsets.reserve(programObj.dynamicBindings.size() + uboArrayExtra);
|
||||
|
||||
const Uint32 bindingCount =
|
||||
std::min<Uint32>(m_maxBindings, static_cast<Uint32>(programObj.bindingKinds.size()));
|
||||
@@ -929,40 +1224,51 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
write.descriptorCount = 1;
|
||||
|
||||
if (kind == ProgramFactory::DescriptorBindingKind::UniformBufferDynamic) {
|
||||
UboBindResult ubo{};
|
||||
const Bool hasPayload = ResolveUniformBufferPayload(program, programObj, binding, ubo);
|
||||
MOBILEGL_ASSERT(hasPayload && ubo.payload != nullptr && ubo.payloadSize > 0,
|
||||
"UniformDescriptorBinder::BindProgramUniformBuffers failed: missing UBO payload on binding %u",
|
||||
binding);
|
||||
const Uint32 descriptorCount =
|
||||
binding < programObj.bindingDescriptorCounts.size()
|
||||
? std::max<Uint32>(1, programObj.bindingDescriptorCounts[binding])
|
||||
: 1u;
|
||||
const SizeT firstBufferInfoIndex = bufferInfos.size();
|
||||
for (Uint32 element = 0; element < descriptorCount; ++element) {
|
||||
UboBindResult ubo{};
|
||||
const Bool hasPayload =
|
||||
ResolveUniformBufferPayload(program, programObj, binding, element, ubo);
|
||||
MOBILEGL_ASSERT(hasPayload && ubo.payload != nullptr && ubo.payloadSize > 0,
|
||||
"UniformDescriptorBinder::BindProgramUniformBuffers failed: missing UBO payload on binding %u element %u",
|
||||
binding, element);
|
||||
|
||||
VkDescriptorBufferInfo bufferInfo{};
|
||||
// Keep offset 0 (sub-range selected via the dynamic offset) so the hashed bufferInfo
|
||||
// is stable across draws and the descriptor-set reuse cache keeps hitting.
|
||||
bufferInfo.offset = 0;
|
||||
Uint32 dynOffset;
|
||||
if (ubo.directBindable) {
|
||||
// Zero-copy: bind the app's resident VkBuffer directly, no per-draw memcpy.
|
||||
bufferInfo.buffer = ubo.buffer;
|
||||
bufferInfo.range = ubo.range;
|
||||
dynOffset = static_cast<Uint32>(ubo.dynamicOffset);
|
||||
} else {
|
||||
BufferSlice slice{};
|
||||
if (!m_bufferManager->UploadTransient(BufferKind::Uniform, frameIndex, ubo.payload,
|
||||
ubo.payloadSize, m_minDynamicOffsetAlignment, slice)) {
|
||||
MOBILEGL_ASSERT(false, "UniformDescriptorBinder::BindProgramUniformBuffers failed: UBO upload failed on binding %u",
|
||||
binding);
|
||||
return false;
|
||||
VkDescriptorBufferInfo bufferInfo{};
|
||||
// Keep offset 0 (sub-range selected via the dynamic offset) so the hashed bufferInfo
|
||||
// is stable across draws and the descriptor-set reuse cache keeps hitting.
|
||||
bufferInfo.offset = 0;
|
||||
Uint32 dynOffset;
|
||||
if (ubo.directBindable) {
|
||||
// Zero-copy: bind the app's resident VkBuffer directly, no per-draw memcpy.
|
||||
bufferInfo.buffer = ubo.buffer;
|
||||
bufferInfo.range = ubo.range;
|
||||
dynOffset = static_cast<Uint32>(ubo.dynamicOffset);
|
||||
} else {
|
||||
BufferSlice slice{};
|
||||
if (!m_bufferManager->UploadTransient(BufferKind::Uniform, frameIndex, ubo.payload,
|
||||
ubo.payloadSize, m_minDynamicOffsetAlignment, slice)) {
|
||||
MOBILEGL_ASSERT(false, "UniformDescriptorBinder::BindProgramUniformBuffers failed: UBO upload failed on binding %u element %u",
|
||||
binding, element);
|
||||
return false;
|
||||
}
|
||||
bufferInfo.buffer = slice.buffer;
|
||||
bufferInfo.range = ubo.payloadSize;
|
||||
dynOffset = static_cast<Uint32>(slice.offset);
|
||||
}
|
||||
bufferInfo.buffer = slice.buffer;
|
||||
bufferInfo.range = ubo.payloadSize;
|
||||
dynOffset = static_cast<Uint32>(slice.offset);
|
||||
bufferInfos.push_back(bufferInfo);
|
||||
// Dynamic offsets are consumed in binding order, then array element order,
|
||||
// matching Vulkan's dynamic-offset consumption rules.
|
||||
dynamicOffsets.push_back(dynOffset);
|
||||
}
|
||||
bufferInfos.push_back(bufferInfo);
|
||||
|
||||
write.descriptorType = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC;
|
||||
write.pBufferInfo = &bufferInfos.back();
|
||||
write.descriptorCount = descriptorCount;
|
||||
write.pBufferInfo = &bufferInfos[firstBufferInfoIndex];
|
||||
writes.push_back(write);
|
||||
dynamicOffsets.push_back(dynOffset);
|
||||
} else if (kind == ProgramFactory::DescriptorBindingKind::UniformTexelBuffer) {
|
||||
VkBufferView bufferView = VK_NULL_HANDLE;
|
||||
if (!ResolveTexelBufferDescriptor(program, programObj, binding, frameIndex, bufferView) ||
|
||||
|
||||
@@ -39,9 +39,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void Shutdown();
|
||||
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// A ProgramFactory eviction just destroyed this layout: purge every frame
|
||||
// slot's cached descriptor sets for it, so a recycled handle value can never
|
||||
// stale-hit sets written for the dead layout's bindings. The sets are
|
||||
// vkFreeDescriptorSets'd back to their pools (created with
|
||||
// FREE_DESCRIPTOR_SET_BIT) and the pool accounting is credited, so program
|
||||
// churn recycles pool capacity instead of abandoning it. GPU-safe: the layout
|
||||
// only dies after >1024 idle frame boundaries, so no in-flight command buffer
|
||||
// references its sets. This is the only eviction path for the per-layout
|
||||
// caches - a live layout's entry must never be purged (its sets would be
|
||||
// unreachable pool slots), so there is deliberately no age-based sweep here.
|
||||
void OnDescriptorSetLayoutDestroyed(VkDescriptorSetLayout descriptorSetLayout);
|
||||
Bool CollectSampledTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures);
|
||||
Bool CollectStorageImageTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures) const;
|
||||
Bool BindProgramUniformBuffers(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
@@ -49,6 +63,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPipelineBindPoint bindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS,
|
||||
const SamplerBindingOverride* samplerBindingOverride = nullptr);
|
||||
|
||||
// Pure format-policy helper kept public for host regression tests. Formatted storage
|
||||
// images use their shader qualifier; transformed float images use glBindImageTexture's
|
||||
// format and never silently fall back to the backing image format.
|
||||
static VkFormat ResolveStorageImageViewFormat(VkFormat reflectedFormat, GLenum bindingFormat,
|
||||
VkFormat resourceFormat, Bool useBindingFormat);
|
||||
|
||||
// True when the program reads at least one sampler and every one of them is bound to a
|
||||
// texture whose GL level range is a single level. Such a sampler resolves to
|
||||
// minLod = maxLod = 0 (see VkSamplerManager::GetOrCreateSampler), so an implicit-LOD sample
|
||||
// and an explicit LOD 0 sample must read the same texel - which is what makes the
|
||||
// ExplicitLod0Sampling SPIR-V rewrite safe to request. Deliberately conservative: it reads
|
||||
// only GL state, so a texture that ends up single-level for another reason (one uploaded
|
||||
// level under a wide level range) merely misses the rewrite.
|
||||
// True when every texture this program samples is an 8-bit-or-less normalized format, so
|
||||
// relaxing the fragment stage to fp16 cannot lose a bit the texel ever carried. Says
|
||||
// nothing about the render target - the caller must check that too.
|
||||
static Bool ProgramSamplesOnlyLowPrecisionTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
// True when every colour attachment the draw writes is an 8-bit-or-less normalized
|
||||
// format (nullptr = default framebuffer, which is). Blending happens at attachment
|
||||
// precision, so a wider target must keep the fragment stage at full precision.
|
||||
static Bool DrawTargetIsLowPrecision(const MG_State::GLState::FramebufferObject* drawFramebuffer);
|
||||
static Bool ProgramSamplesOnlySingleLevelTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
|
||||
private:
|
||||
struct DescriptorPoolBucket {
|
||||
VkDescriptorPool handle = VK_NULL_HANDLE;
|
||||
@@ -56,8 +95,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 allocatedSets = 0;
|
||||
};
|
||||
|
||||
// A cached descriptor set together with the pool it was allocated from, so a
|
||||
// layout-destroyed purge can vkFreeDescriptorSets it back and credit the
|
||||
// owning bucket's accounting.
|
||||
struct CachedDescriptorSet {
|
||||
VkDescriptorSet set = VK_NULL_HANDLE;
|
||||
VkDescriptorPool pool = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
struct DescriptorSetCacheEntry {
|
||||
Vector<VkDescriptorSet> sets;
|
||||
Vector<CachedDescriptorSet> sets;
|
||||
Uint32 cursor = 0;
|
||||
};
|
||||
|
||||
@@ -107,7 +154,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
Bool ResolveUniformBufferPayload(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
UboBindResult& out) const;
|
||||
Uint32 arrayElement, UboBindResult& out) const;
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
|
||||
VkResult AllocateDescriptorSetsFromActivePool(
|
||||
@@ -163,11 +210,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 samplerLifetimeId = 0;
|
||||
Uint64 textureLifetimeId = 0;
|
||||
VkSampler sampler = VK_NULL_HANDLE;
|
||||
Uint32 viewLevelCount = 0;
|
||||
Uint16 samplerVersion = 0;
|
||||
Uint16 textureParamsVersion = 0;
|
||||
Bool forceNearestFiltering = false;
|
||||
Bool valid = false;
|
||||
// ResolveSampledImageViewFormat is pure in (image format, numeric domain), but a
|
||||
// domain mismatch walks a ~184-entry format table. Memo the resolution per binding
|
||||
// so a reinterpreted sampler pays that scan once, not once per draw.
|
||||
VkFormat viewFormatSource = VK_FORMAT_UNDEFINED;
|
||||
SamplerNumericDomain viewFormatDomain = SamplerNumericDomain::Unknown;
|
||||
VkFormat viewFormat = VK_FORMAT_UNDEFINED;
|
||||
Bool viewFormatValid = false;
|
||||
};
|
||||
mutable Vector<SamplerResolveMemo> m_samplerResolveMemo;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
|
||||
@@ -32,6 +32,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.IsBgra, sizeof(attr.IsBgra)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.Divisor, sizeof(attr.Divisor)));
|
||||
|
||||
// The buffer's heap address is an identity component of the key: a freed
|
||||
// buffer's reused address can alias an old cache entry, but only under a
|
||||
// byte-identical attribute layout - and the entry payload is a pure function
|
||||
// of the hashed inputs, with the draw path re-resolving bindingBufferKeys
|
||||
// against the live VAO attribute pointers, so an aliased hit returns exactly
|
||||
// what a rebuild would. Address drift only grows the map; the OnFrameBoundary
|
||||
// aging sweep bounds that.
|
||||
const SizeT bufferKey = reinterpret_cast<SizeT>(attr.Buffer.get());
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &bufferKey, sizeof(bufferKey)));
|
||||
}
|
||||
@@ -58,6 +65,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash) {
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return it->second;
|
||||
}
|
||||
|
||||
@@ -66,6 +74,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<SizeT> bindingBaseOffsets;
|
||||
Vector<Uint32> bindingAttributeLocations;
|
||||
Vector<Bool> bindingUsesClientMemory;
|
||||
Vector<VertexStreamConversion> bindingConversions;
|
||||
Uint32 unsupportedAttribMask = 0;
|
||||
|
||||
for (Uint32 location = 0; location < MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS; ++location) {
|
||||
@@ -74,8 +83,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto vkFormat = ToVkVertexFormat(attr.Type, attr.Size, attr.Normalized, attr.IsInteger, attr.IsBgra);
|
||||
if (vkFormat == VK_FORMAT_UNDEFINED) {
|
||||
const VkFormat sourceVkFormat =
|
||||
ToVkVertexFormat(attr.Type, attr.Size, attr.Normalized, attr.IsInteger, attr.IsBgra);
|
||||
if (sourceVkFormat == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_E("Unsupported vertex attribute layout (location=%u, type=%s, size=%d): the array is "
|
||||
"enabled but cannot be mapped to a VkFormat",
|
||||
location, MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size);
|
||||
@@ -83,6 +93,33 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
}
|
||||
|
||||
VkFormat vkFormat = sourceVkFormat;
|
||||
VertexStreamConversion conversion = VertexStreamConversion::None;
|
||||
if (!SupportsVertexBufferFormat(vkFormat)) {
|
||||
if (IsScaledIntegerVertexFormat(vkFormat)) {
|
||||
const VkFormat fallbackFormat = ToFloat32VertexFormat(attr.Size);
|
||||
if (fallbackFormat != VK_FORMAT_UNDEFINED && SupportsVertexBufferFormat(fallbackFormat)) {
|
||||
vkFormat = fallbackFormat;
|
||||
conversion = VertexStreamConversion::ScaledIntegerToFloat32;
|
||||
MGLOG_W("Vertex attribute location=%u format=%d lacks "
|
||||
"VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT; using float32 stream format=%d "
|
||||
"(type=%s size=%d normalized=%s integer=%s)",
|
||||
location, static_cast<Int>(sourceVkFormat), static_cast<Int>(vkFormat),
|
||||
MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size,
|
||||
attr.Normalized ? "true" : "false", attr.IsInteger ? "true" : "false");
|
||||
}
|
||||
}
|
||||
|
||||
if (conversion == VertexStreamConversion::None) {
|
||||
MGLOG_E("Unsupported Vulkan vertex format (location=%u, format=%d, type=%s, size=%d): "
|
||||
"VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT is unavailable and no semantic fallback exists",
|
||||
location, static_cast<Int>(sourceVkFormat),
|
||||
MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size);
|
||||
unsupportedAttribMask |= (1u << location);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT attribByteSize = GetAttributeByteSize(attr.Type, attr.Size, attr.IsBgra);
|
||||
if (attribByteSize == 0) {
|
||||
MGLOG_E("Vertex attribute with unknown component size (location=%u, type=%s): the array is "
|
||||
@@ -92,8 +129,33 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
}
|
||||
|
||||
const Uint32 stride =
|
||||
const Uint32 sourceStride =
|
||||
attr.Stride > 0 ? static_cast<Uint32>(attr.Stride) : static_cast<Uint32>(attribByteSize);
|
||||
const Bool packedAttribute = attr.Type == DataType::Int2101010Rev ||
|
||||
attr.Type == DataType::Uint2101010Rev;
|
||||
const SizeT requiredAlignment = packedAttribute ? attribByteSize : GetComponentSize(attr.Type);
|
||||
// For a client-memory array attr.Offset holds the raw client pointer, and the
|
||||
// draw path re-uploads the data to a 16-aligned transient slice with attribute
|
||||
// offset 0, so only the stride can violate Vulkan's fetch alignment there.
|
||||
const Bool clientMemoryAttribute = attr.Buffer == nullptr;
|
||||
if (conversion == VertexStreamConversion::None && requiredAlignment > 1 &&
|
||||
((sourceStride % requiredAlignment) != 0 ||
|
||||
(!clientMemoryAttribute && (attr.Offset % requiredAlignment) != 0))) {
|
||||
// GL accepts arbitrary byte strides and offsets. Core Vulkan vertex fetches do not
|
||||
// unless VK_EXT_legacy_vertex_attributes is available, so deinterleave this one
|
||||
// attribute into a tightly packed transient stream without changing its format.
|
||||
conversion = VertexStreamConversion::Repack;
|
||||
MGLOG_W("Vertex attribute location=%u uses Vulkan-incompatible alignment "
|
||||
"(offset=%zu stride=%u required=%zu); using a tightly packed stream",
|
||||
location, attr.Offset, sourceStride, requiredAlignment);
|
||||
}
|
||||
|
||||
Uint32 stride = sourceStride;
|
||||
if (conversion == VertexStreamConversion::Repack) {
|
||||
stride = static_cast<Uint32>(attribByteSize);
|
||||
} else if (conversion == VertexStreamConversion::ScaledIntegerToFloat32) {
|
||||
stride = static_cast<Uint32>(attr.Size * static_cast<Int>(sizeof(Float)));
|
||||
}
|
||||
const VkVertexInputRate inputRate =
|
||||
(attr.Divisor == 0) ? VK_VERTEX_INPUT_RATE_VERTEX : VK_VERTEX_INPUT_RATE_INSTANCE;
|
||||
|
||||
@@ -103,6 +165,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
bindingBaseOffsets.push_back(attr.Buffer ? attr.Offset : 0);
|
||||
bindingAttributeLocations.push_back(location);
|
||||
bindingUsesClientMemory.push_back(attr.Buffer == nullptr);
|
||||
bindingConversions.push_back(conversion);
|
||||
builder.AddBinding(binding, stride, inputRate);
|
||||
builder.AddAttribute(location, binding, vkFormat, 0);
|
||||
}
|
||||
@@ -111,12 +174,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
auto& entry = m_cache[hash];
|
||||
entry.hash = hash;
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
entry.bindings = builder.GetBindings();
|
||||
entry.attributes = builder.GetAttributes();
|
||||
entry.bindingBufferKeys = std::move(bindingBufferKeys);
|
||||
entry.bindingBaseOffsets = std::move(bindingBaseOffsets);
|
||||
entry.bindingAttributeLocations = std::move(bindingAttributeLocations);
|
||||
entry.bindingUsesClientMemory = std::move(bindingUsesClientMemory);
|
||||
entry.bindingConversions = std::move(bindingConversions);
|
||||
entry.unsupportedAttribMask = unsupportedAttribMask;
|
||||
entry.state = state;
|
||||
entry.state.pVertexBindingDescriptions = entry.bindings.empty() ? nullptr : entry.bindings.data();
|
||||
@@ -124,6 +189,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return entry;
|
||||
}
|
||||
|
||||
void VertexInputStateFactory::OnFrameBoundary() {
|
||||
++m_frameBoundaryCounter;
|
||||
|
||||
// Sweep occasionally; evict entries whose last hit is far in the past.
|
||||
// Erasure happens only here, never mid-frame: the draw path holds a
|
||||
// reference into the current entry across its setup, and unordered_map
|
||||
// erase would invalidate it. Entries are CPU-side only, so no GPU-idle
|
||||
// proof is needed; an evicted entry that is used again is simply rebuilt
|
||||
// from the VAO state (same hash, same content).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
if ((m_frameBoundaryCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameBoundaryCounter - it->second.lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
it = m_cache.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VertexInputStateFactory::ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger,
|
||||
Bool isBgra) {
|
||||
if (isBgra) {
|
||||
@@ -282,4 +371,47 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const SizeT componentSize = GetComponentSize(type);
|
||||
return componentSize == 0 ? 0 : componentSize * static_cast<SizeT>(size);
|
||||
}
|
||||
|
||||
Bool VertexInputStateFactory::IsScaledIntegerVertexFormat(VkFormat format) {
|
||||
switch (format) {
|
||||
case VK_FORMAT_R8_USCALED:
|
||||
case VK_FORMAT_R8_SSCALED:
|
||||
case VK_FORMAT_R8G8_USCALED:
|
||||
case VK_FORMAT_R8G8_SSCALED:
|
||||
case VK_FORMAT_R8G8B8_USCALED:
|
||||
case VK_FORMAT_R8G8B8_SSCALED:
|
||||
case VK_FORMAT_R8G8B8A8_USCALED:
|
||||
case VK_FORMAT_R8G8B8A8_SSCALED:
|
||||
case VK_FORMAT_R16_USCALED:
|
||||
case VK_FORMAT_R16_SSCALED:
|
||||
case VK_FORMAT_R16G16_USCALED:
|
||||
case VK_FORMAT_R16G16_SSCALED:
|
||||
case VK_FORMAT_R16G16B16_USCALED:
|
||||
case VK_FORMAT_R16G16B16_SSCALED:
|
||||
case VK_FORMAT_R16G16B16A16_USCALED:
|
||||
case VK_FORMAT_R16G16B16A16_SSCALED:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VertexInputStateFactory::ToFloat32VertexFormat(Int componentCount) {
|
||||
switch (componentCount) {
|
||||
case 1: return VK_FORMAT_R32_SFLOAT;
|
||||
case 2: return VK_FORMAT_R32G32_SFLOAT;
|
||||
case 3: return VK_FORMAT_R32G32B32_SFLOAT;
|
||||
case 4: return VK_FORMAT_R32G32B32A32_SFLOAT;
|
||||
default: return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
Bool VertexInputStateFactory::SupportsVertexBufferFormat(VkFormat format) const {
|
||||
if (m_physicalDevice == VK_NULL_HANDLE || format == VK_FORMAT_UNDEFINED) {
|
||||
return false;
|
||||
}
|
||||
VkFormatProperties properties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &properties);
|
||||
return (properties.bufferFeatures & VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT) != 0;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -19,14 +19,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
public:
|
||||
using HashType = Uint64;
|
||||
|
||||
enum class VertexStreamConversion : Uint8 {
|
||||
None = 0,
|
||||
Repack,
|
||||
ScaledIntegerToFloat32,
|
||||
};
|
||||
|
||||
struct BackendVertexInputState {
|
||||
HashType hash = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age are evicted (CPU heap only).
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
Vector<VkVertexInputBindingDescription> bindings;
|
||||
Vector<VkVertexInputAttributeDescription> attributes;
|
||||
Vector<SizeT> bindingBufferKeys;
|
||||
Vector<SizeT> bindingBaseOffsets;
|
||||
Vector<Uint32> bindingAttributeLocations;
|
||||
Vector<Bool> bindingUsesClientMemory;
|
||||
Vector<VertexStreamConversion> bindingConversions;
|
||||
// Locations whose array is ENABLED but whose GL format has no VkFormat mapping. They are
|
||||
// absent from `attributes`, so without this mask the draw path cannot tell them apart from
|
||||
// a genuinely disabled array and would silently feed the shader the current attribute value.
|
||||
@@ -36,8 +46,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
};
|
||||
|
||||
explicit VertexInputStateFactory(const VulkanRendererConfig& config):
|
||||
m_config(config) {}
|
||||
VertexInputStateFactory(const VulkanRendererConfig& config, VkPhysicalDevice physicalDevice):
|
||||
m_config(config), m_physicalDevice(physicalDevice) {}
|
||||
~VertexInputStateFactory() = default;
|
||||
VertexInputStateFactory(const VertexInputStateFactory&) = delete;
|
||||
|
||||
@@ -48,6 +58,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash);
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(const MG_State::GLState::VertexArrayObject& vao);
|
||||
// Frame boundary hook: ages the cache and evicts entries not hit for many
|
||||
// frames. The key mixes buffer heap addresses, so buffer/VAO churn keeps
|
||||
// minting fresh keys; without eviction the map grows for the whole session.
|
||||
// Entries hold no Vulkan handles (pipeline creation copies the descriptions)
|
||||
// and the draw path's entry reference never spans a frame boundary, so
|
||||
// eviction here needs no GPU-idle proof. Self-gated: one counter bump and
|
||||
// compare except on sweep boundaries.
|
||||
void OnFrameBoundary();
|
||||
static SizeT GetComponentSize(DataType type);
|
||||
// Tightly-packed byte size of one vertex element for this attribute: componentSize * size for
|
||||
// normal types, and 4 (one packed word) for the 2_10_10_10 types and GL_BGRA. Returns 0 for
|
||||
@@ -56,9 +74,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
private:
|
||||
static VkFormat ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger, Bool isBgra = false);
|
||||
static Bool IsScaledIntegerVertexFormat(VkFormat format);
|
||||
static VkFormat ToFloat32VertexFormat(Int componentCount);
|
||||
Bool SupportsVertexBufferFormat(VkFormat format) const;
|
||||
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, BackendVertexInputState> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -141,6 +141,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_transientUploadArena.BeginFrame(frameIndex);
|
||||
}
|
||||
|
||||
void VkBufferManager::CollectAllDeferredReleases() {
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_deferredBufferReleases.size(); ++frameIndex) {
|
||||
CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_transientUploadArena.GetFrameCount(); ++frameIndex) {
|
||||
m_transientUploadArena.CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::NotifyDeviceIdle() {
|
||||
// Everything submitted so far has completed. Work recorded for the
|
||||
// current frame has not been submitted yet, so the current serial
|
||||
|
||||
@@ -77,6 +77,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Recreate all per-frame transient arenas
|
||||
Bool RecreateTransientArenas(Uint32 frameCount);
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Drains every frame slot's deferred buffer/resource releases (and the
|
||||
// transient arena's parked superseded blocks). Only valid when the
|
||||
// caller has proven every queue submission complete; used by the
|
||||
// present-less frame-boundary drain.
|
||||
void CollectAllDeferredReleases();
|
||||
// All previously submitted GPU work has completed (vkDeviceWaitIdle).
|
||||
void NotifyDeviceIdle();
|
||||
// A frame slot's submission fence has been waited: every serial up to
|
||||
|
||||
@@ -157,6 +157,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkBufferObject::Invalidate(VkDeviceSize size, VkDeviceSize offset) {
|
||||
MOBILEGL_ASSERT(IsValid(), "VkBufferObject::Invalidate called on invalid buffer");
|
||||
MOBILEGL_ASSERT(IsMapped(), "VkBufferObject::Invalidate requires mapped memory");
|
||||
MOBILEGL_ASSERT(offset <= m_size, "VkBufferObject::Invalidate offset out of range");
|
||||
|
||||
const VkDeviceSize resolvedSize = size == VK_WHOLE_SIZE ? m_size - offset : size;
|
||||
MOBILEGL_ASSERT(offset + resolvedSize <= m_size, "VkBufferObject::Invalidate range out of bounds");
|
||||
if (resolvedSize == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const VkResult result = vmaInvalidateAllocation(m_allocator, m_allocation, offset, resolvedSize);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("VkBufferObject::Invalidate failed: vmaInvalidateAllocation returned %d", result);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
BufferSlice VkBufferObject::GetSlice(VkDeviceSize offset, VkDeviceSize size) const {
|
||||
MOBILEGL_ASSERT(offset <= m_size, "VkBufferObject::GetSlice offset out of range");
|
||||
const VkDeviceSize resolvedSize = (size == VK_WHOLE_SIZE) ? (m_size - offset) : size;
|
||||
|
||||
@@ -44,6 +44,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void* Map();
|
||||
void Unmap();
|
||||
Bool Upload(const void* data, VkDeviceSize size, VkDeviceSize offset = 0);
|
||||
Bool Invalidate(VkDeviceSize size = VK_WHOLE_SIZE, VkDeviceSize offset = 0);
|
||||
|
||||
VkBuffer GetHandle() const { return m_buffer; }
|
||||
VkDeviceSize GetSize() const { return m_size; }
|
||||
|
||||
@@ -180,6 +180,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
internalFormat = TextureInternalFormat::Unknown;
|
||||
samples = 0;
|
||||
deadSinceFrame = kNeverObservedDead;
|
||||
}
|
||||
|
||||
VkRenderPassManager::VkRenderPassManager(VkDevice device,
|
||||
@@ -206,6 +207,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.Destroy(m_device, m_allocator);
|
||||
}
|
||||
m_renderbufferResources.clear();
|
||||
CollectDeferredRenderbufferReleases(/*destroyAll=*/true); // caller guarantees device idle
|
||||
m_pendingRenderbufferClears.clear();
|
||||
RenderPassEntry::s_textureResourcesScratch.clear();
|
||||
s_activeRenderPass = {};
|
||||
@@ -213,22 +215,75 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastValid = false;
|
||||
}
|
||||
|
||||
void VkRenderPassManager::CollectRenderbufferGarbage() {
|
||||
Vector<MG_State::GLState::RenderbufferObject*> deadRenderbuffers;
|
||||
deadRenderbuffers.reserve(m_renderbufferResources.size());
|
||||
for (auto& [renderbuffer, resource] : m_renderbufferResources) {
|
||||
const auto liveRenderbuffer = resource.renderbuffer.lock();
|
||||
if (!liveRenderbuffer || liveRenderbuffer.get() != renderbuffer) {
|
||||
deadRenderbuffers.emplace_back(renderbuffer);
|
||||
}
|
||||
Uint64 VkRenderPassManager::RetireAgeFrames() const {
|
||||
// MaxFramesInFlight + 2 covers the frame ring plus one boundary for the
|
||||
// recording-to-submit gap and one because OnPresent runs ahead of Present's
|
||||
// fence wait; the floor of 8 keeps a margin over the default ring of 3 while
|
||||
// still releasing multi-MB attachment memory promptly (the render-pass cache's
|
||||
// 1024-frame retirement would pin it for no additional safety).
|
||||
return std::max<Uint64>(8, static_cast<Uint64>(m_config.MaxFramesInFlight) + 2);
|
||||
}
|
||||
|
||||
void VkRenderPassManager::DeferRenderbufferBackingRelease(RenderbufferResource& resource) {
|
||||
// The superseded backing may still be referenced by in-flight command buffers
|
||||
// (glRenderbufferStorage can respecify a renderbuffer drawn this very frame),
|
||||
// so it is parked and destroyed only after RetireAgeFrames() boundaries.
|
||||
if (resource.image == VK_NULL_HANDLE && resource.view == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto* renderbuffer : deadRenderbuffers) {
|
||||
auto resourceIt = m_renderbufferResources.find(renderbuffer);
|
||||
if (resourceIt != m_renderbufferResources.end()) {
|
||||
resourceIt->second.Destroy(m_device, m_allocator);
|
||||
m_renderbufferResources.erase(resourceIt);
|
||||
m_deferredRenderbufferReleases.push_back({resource.image, resource.allocation, resource.view, m_frameCounter});
|
||||
resource.image = VK_NULL_HANDLE;
|
||||
resource.allocation = nullptr;
|
||||
resource.view = VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
void VkRenderPassManager::CollectDeferredRenderbufferReleases(Bool destroyAll) {
|
||||
if (m_deferredRenderbufferReleases.empty()) {
|
||||
return;
|
||||
}
|
||||
const Uint64 retireAgeFrames = RetireAgeFrames();
|
||||
std::erase_if(m_deferredRenderbufferReleases, [&](DeferredRenderbufferRelease& release) {
|
||||
if (!destroyAll && m_frameCounter - release.deferredAtFrame < retireAgeFrames) {
|
||||
return false;
|
||||
}
|
||||
m_pendingRenderbufferClears.erase(renderbuffer);
|
||||
if (release.view != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(m_device, release.view, nullptr);
|
||||
}
|
||||
if (release.image != VK_NULL_HANDLE) {
|
||||
vmaDestroyImage(m_allocator, release.image, release.allocation);
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
void VkRenderPassManager::CollectRenderbufferGarbage() {
|
||||
// Two-phase reclamation: a dead renderbuffer's VkImage may still be referenced by
|
||||
// command buffers submitted up to frames-in-flight frames ago (it was legally
|
||||
// attached and drawn right up to its deletion), so the first observation of an
|
||||
// expired weak reference only stamps the current frame counter; Destroy runs once
|
||||
// enough frame boundaries have passed that the stamping frame's submission fence
|
||||
// has provably been waited (see RetireAgeFrames).
|
||||
const Uint64 retireAgeFrames = RetireAgeFrames();
|
||||
for (auto it = m_renderbufferResources.begin(); it != m_renderbufferResources.end();) {
|
||||
auto& resource = it->second;
|
||||
const auto liveRenderbuffer = resource.renderbuffer.lock();
|
||||
if (liveRenderbuffer && liveRenderbuffer.get() == it->first) {
|
||||
resource.deadSinceFrame = RenderbufferResource::kNeverObservedDead;
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
if (resource.deadSinceFrame == RenderbufferResource::kNeverObservedDead) {
|
||||
resource.deadSinceFrame = m_frameCounter;
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
if (m_frameCounter - resource.deadSinceFrame < retireAgeFrames) {
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
m_pendingRenderbufferClears.erase(it->first);
|
||||
resource.Destroy(m_device, m_allocator);
|
||||
it = m_renderbufferResources.erase(it);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -251,11 +306,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const auto internalFormat = renderbuffer->GetInternalFormat();
|
||||
const VkFormat format = MG_Util::ConvertTextureInternalFormatToVkEnum(internalFormat);
|
||||
const VkImageAspectFlags aspect = ResolveImageAspectMaskForFormat(format);
|
||||
if ((aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0) {
|
||||
MGLOG_E("GetOrCreateRenderbufferResource: color renderbuffer %u is not supported by DirectVulkan render passes yet",
|
||||
renderbuffer->GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
// Renderbuffers are never sampled (GL has no way to bind one to a sampler), so the
|
||||
// usage set is attachment + transfer: transfer covers readback (vkCmdCopyImageToBuffer),
|
||||
// BlitFramebuffer, CopyTexImage sources, and out-of-render-pass clear materialization.
|
||||
const VkImageUsageFlags imageUsage =
|
||||
((aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0 ? VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT
|
||||
: VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT) |
|
||||
VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT;
|
||||
|
||||
auto& resource = m_renderbufferResources[renderbuffer.get()];
|
||||
const Bool needsCreate =
|
||||
@@ -268,9 +325,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.samples != renderbuffer->GetSamples();
|
||||
if (!needsCreate) {
|
||||
resource.renderbuffer = renderbuffer;
|
||||
// A new renderbuffer at a recycled address may adopt a compatible entry that
|
||||
// was already stamped dead; it is alive again, so cancel the aging.
|
||||
resource.deadSinceFrame = RenderbufferResource::kNeverObservedDead;
|
||||
return &resource;
|
||||
}
|
||||
|
||||
// Respecify: park the old backing for aged destruction instead of destroying
|
||||
// inline - it may still be referenced by in-flight command buffers.
|
||||
DeferRenderbufferBackingRelease(resource);
|
||||
resource.Destroy(m_device, m_allocator);
|
||||
resource.renderbuffer = renderbuffer;
|
||||
|
||||
@@ -285,7 +348,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
imageInfo.format = format;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
|
||||
imageInfo.usage = imageUsage;
|
||||
imageInfo.samples = sampleCount;
|
||||
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
|
||||
@@ -386,6 +449,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void VkRenderPassManager::QueueRenderbufferClear(
|
||||
GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo) {
|
||||
if ((mask & GL_COLOR_BUFFER_BIT) != 0) {
|
||||
// Color renderbuffer draw buffers take the framebuffer-level clear too; texture
|
||||
// attachments are skipped by the per-attachment overload's IsRenderbuffer guard.
|
||||
for (const auto attachmentType : drawFbo.GetDrawBuffers()) {
|
||||
if (attachmentType == FramebufferAttachmentType::None) {
|
||||
continue;
|
||||
}
|
||||
QueueRenderbufferClear(
|
||||
ClearAttachmentPayload{.mask = GL_COLOR_BUFFER_BIT, .color = clearPayload.color},
|
||||
drawFbo.GetAttachment(attachmentType));
|
||||
}
|
||||
}
|
||||
if ((mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
||||
QueueRenderbufferClear(
|
||||
ClearAttachmentPayload{.mask = GL_DEPTH_BUFFER_BIT, .depth = clearPayload.depth},
|
||||
@@ -682,6 +757,83 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// assuming default FBO has the right param
|
||||
for (Uint32 i = 0; i < colorAttachmentSlotCount; ++i) {
|
||||
auto drawbuf = drawbufs[i];
|
||||
|
||||
// Renderbuffer color attachments mirror the texture path below, with the
|
||||
// resource (image/view/format/layout) coming from the render-pass manager's
|
||||
// renderbuffer store instead of the texture manager.
|
||||
if (drawbuf != FramebufferAttachmentType::None && !isDefaultFbo) {
|
||||
const auto& rbAtt = fbo.GetAttachment(drawbuf);
|
||||
if (rbAtt.IsRenderbuffer() && rbAtt.IsComplete()) {
|
||||
const auto& renderbuffer = rbAtt.GetRenderbuffer();
|
||||
auto* rbResource = GetOrCreateRenderbufferResource(renderbuffer);
|
||||
if (rbResource == nullptr || (rbResource->aspect & VK_IMAGE_ASPECT_COLOR_BIT) == 0) {
|
||||
MGLOG_E("GetOrCreateRenderPass: draw buffer slot %u on FBO %u has an unsupported color "
|
||||
"renderbuffer %u; using VK_ATTACHMENT_UNUSED",
|
||||
i, fbo.GetExternalIndex(), renderbuffer->GetExternalIndex());
|
||||
continue;
|
||||
}
|
||||
|
||||
const Uint32 rbAttachmentIndex = static_cast<Uint32>(attachmentDescriptions.size());
|
||||
attachmentDescriptions.emplace_back();
|
||||
VkAttachmentDescription& rbDesc = attachmentDescriptions.back();
|
||||
|
||||
ClearAttachmentPayload rbClearPayload{};
|
||||
Bool rbHasClear = GetPendingRenderbufferClear(renderbuffer.get(), rbClearPayload) &&
|
||||
(rbClearPayload.mask & GL_COLOR_BUFFER_BIT) != 0;
|
||||
if (rbHasClear &&
|
||||
MG_Util::GetBaseInternalFormatComponentCount(renderbuffer->GetInternalFormat()) == 3) {
|
||||
// RGB renderbuffers are backed by an RGBA image; the missing alpha reads as 1.
|
||||
rbClearPayload.color =
|
||||
FloatVec4(rbClearPayload.color.x(), rbClearPayload.color.y(),
|
||||
rbClearPayload.color.z(), 1.0f);
|
||||
}
|
||||
|
||||
const VkImageLayout trackedRbLayout = rbResource->layout;
|
||||
rbDesc.flags = 0;
|
||||
rbDesc.format = rbResource->format;
|
||||
rbDesc.samples = rbResource->sampleCount;
|
||||
rbDesc.loadOp = rbHasClear ? VK_ATTACHMENT_LOAD_OP_CLEAR :
|
||||
(trackedRbLayout == VK_IMAGE_LAYOUT_UNDEFINED ? VK_ATTACHMENT_LOAD_OP_DONT_CARE
|
||||
: VK_ATTACHMENT_LOAD_OP_LOAD);
|
||||
rbDesc.storeOp = VK_ATTACHMENT_STORE_OP_STORE;
|
||||
rbDesc.stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
|
||||
rbDesc.stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
|
||||
rbDesc.initialLayout = (rbHasClear || trackedRbLayout == VK_IMAGE_LAYOUT_UNDEFINED) ?
|
||||
VK_IMAGE_LAYOUT_UNDEFINED : trackedRbLayout;
|
||||
rbDesc.finalLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL;
|
||||
adoptRenderPassSampleCount(rbResource->sampleCount, "color",
|
||||
static_cast<Int>(renderbuffer->GetExternalIndex()));
|
||||
|
||||
if (rbHasClear) {
|
||||
pendingClearAttachments.emplace_back(PendingClearAttachmentInfo {
|
||||
.attachmentIndex = rbAttachmentIndex,
|
||||
.colorAttachmentSlot = i,
|
||||
.renderbuffer = renderbuffer.get(),
|
||||
.hasInlinePayload = true,
|
||||
.inlinePayload = rbClearPayload,
|
||||
});
|
||||
}
|
||||
|
||||
if (width == 0)
|
||||
width = static_cast<Int>(rbResource->extent.width);
|
||||
if (height == 0)
|
||||
height = static_cast<Int>(rbResource->extent.height);
|
||||
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
.target = TrackedAttachmentTarget::Renderbuffer,
|
||||
.renderbuffer = renderbuffer,
|
||||
.finalLayout = rbDesc.finalLayout,
|
||||
});
|
||||
textureResources.emplace_back(nullptr);
|
||||
attachmentViews.emplace_back(rbResource->view);
|
||||
MOBILEGL_ASSERT(attachmentViews.back() != VK_NULL_HANDLE,
|
||||
"GetOrCreateRenderPass: renderbuffer view missing at color attachment %d", i);
|
||||
|
||||
colorAttachmentRefs[i].attachment = rbAttachmentIndex;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
auto* texture = ResolveCompleteColorAttachmentTexture(fbo, drawbuf, i);
|
||||
if (texture == nullptr)
|
||||
continue;
|
||||
@@ -700,6 +852,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case TextureTarget::Texture2D:
|
||||
case TextureTarget::Texture2DArray:
|
||||
case TextureTarget::Texture2DMultisample:
|
||||
case TextureTarget::Texture2DMultisampleArray:
|
||||
case TextureTarget::Texture3D:
|
||||
case TextureTarget::TextureCubeMap:
|
||||
case TextureTarget::TextureCubeMapArray:
|
||||
case TextureTarget::TextureRectangle: {
|
||||
desc.flags = 0;
|
||||
desc.format = isDefaultFbo ?
|
||||
@@ -1083,6 +1239,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void VkRenderPassManager::OnPresent() {
|
||||
++m_frameCounter;
|
||||
|
||||
// Runs every frame boundary, ahead of the render-pass sweep gate below: the walk
|
||||
// is O(#renderbuffer resources) — single digits in practice — and per-frame
|
||||
// invocation keeps dead-resource reclaim latency at the aging bound instead of
|
||||
// coupling it to renderbuffer *use* (the GetOrCreateRenderbufferResource call
|
||||
// site never runs again once an app stops using renderbuffers).
|
||||
CollectRenderbufferGarbage();
|
||||
CollectDeferredRenderbufferReleases(/*destroyAll=*/false);
|
||||
|
||||
// Sweep occasionally; evict entries whose last use is far past every
|
||||
// in-flight frame so their VkRenderPass/VkFramebuffer can be destroyed
|
||||
// safely (RenderPassEntry's destructor releases the handles).
|
||||
@@ -1092,6 +1256,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
|
||||
// Collect the dying handles and notify once after the loop: pipelines hashed
|
||||
// on them share the entries' >kRetireAgeFrames idleness (they are only bound
|
||||
// by draws that hit those entries), so the observer may destroy them
|
||||
// immediately - and a single batched notification costs one pipeline-cache
|
||||
// scan instead of one per evicted pass.
|
||||
Vector<VkRenderPass> destroyedRenderPasses;
|
||||
const Uint64 activeHash = s_hasActiveRenderPass ? s_activeRenderPass.hash : 0;
|
||||
for (auto it = m_renderPasses.begin(); it != m_renderPasses.end();) {
|
||||
const Bool isActive = s_hasActiveRenderPass && it->first == activeHash;
|
||||
@@ -1099,11 +1269,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (m_rpFastValid && m_rpFastRenderPassHash == it->first) {
|
||||
m_rpFastValid = false;
|
||||
}
|
||||
destroyedRenderPasses.push_back(it->second.renderPass);
|
||||
it = m_renderPasses.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (!destroyedRenderPasses.empty() && m_evictionObserver != nullptr) {
|
||||
m_evictionObserver->OnRenderPassesDestroyed(destroyedRenderPasses);
|
||||
}
|
||||
}
|
||||
|
||||
Bool VkRenderPassManager::BeginRenderPass(VkCommandBuffer commandBuffer, RenderPassEntry& renderPassEntry) {
|
||||
|
||||
@@ -157,11 +157,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkRenderPassManager {
|
||||
public:
|
||||
using HashType = Uint64;
|
||||
|
||||
// Notified once per OnPresent sweep with every aged-out entry's VkRenderPass
|
||||
// value: pipelines are hashed on the raw handle, and once destroyed the value
|
||||
// may be recycled for an incompatible pass, so dependent caches must purge
|
||||
// everything keyed on them before any new pass can be created (the sweep and
|
||||
// the notification run back-to-back with no creation in between; observers
|
||||
// compare the values, never dereference them). Batched so a mass-idle cohort
|
||||
// (shader-pack switch, dimension exit) costs the observer one pipeline-cache
|
||||
// scan, not one per dying pass. The wholesale paths
|
||||
// (Shutdown/RecreateSwapchain) do not notify - their callers already drop
|
||||
// every pipeline outright.
|
||||
class IEvictionObserver {
|
||||
public:
|
||||
virtual ~IEvictionObserver() = default;
|
||||
virtual void OnRenderPassesDestroyed(const Vector<VkRenderPass>& renderPasses) = 0;
|
||||
};
|
||||
|
||||
VkRenderPassManager(VkDevice device,
|
||||
VkPhysicalDevice physicalDevice, VmaAllocator allocator, const VulkanRendererConfig& config,
|
||||
VkClearManager& clearManager, VkTextureManager& textureManager, SwapchainObject& swapchainObject);
|
||||
~VkRenderPassManager();
|
||||
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetEvictionObserver(IEvictionObserver* observer) { m_evictionObserver = observer; }
|
||||
|
||||
Bool Initialize();
|
||||
void Shutdown();
|
||||
|
||||
@@ -192,6 +212,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
UnorderedMap<Uint64, RenderPassEntry> m_renderPasses;
|
||||
// Monotonic frame counter (bumped in OnPresent) for render-pass cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
|
||||
// Bumped whenever a renderbuffer VkImage is (re)created; together with the texture
|
||||
// manager's image epoch this invalidates the render-pass fast path on any attachment
|
||||
@@ -211,7 +232,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 m_rpFastRbEpoch = 0;
|
||||
Uint64 m_rpFastRenderPassHash = 0;
|
||||
|
||||
public:
|
||||
struct RenderbufferResource {
|
||||
// deadSinceFrame sentinel: the owning weak reference has not been observed
|
||||
// expired. Dead resources age past every in-flight frame before Destroy
|
||||
// (see CollectRenderbufferGarbage); the GPU may still reference the image
|
||||
// for frames-in-flight frames after the GL object dies.
|
||||
static constexpr Uint64 kNeverObservedDead = UINT64_MAX;
|
||||
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
@@ -223,25 +251,47 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::Unknown;
|
||||
Int samples = 0;
|
||||
// m_frameCounter value at which the weak reference was first seen expired.
|
||||
Uint64 deadSinceFrame = kNeverObservedDead;
|
||||
|
||||
void Destroy(VkDevice device, VmaAllocator allocator);
|
||||
};
|
||||
|
||||
// Public so the renderer's blit/copy/readback bindings can source renderbuffer
|
||||
// attachments the same way texture attachments go through the texture manager.
|
||||
RenderbufferResource* GetOrCreateRenderbufferResource(
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
Bool GetPendingRenderbufferClear(MG_State::GLState::RenderbufferObject* renderbuffer,
|
||||
ClearAttachmentPayload& outPayload) const;
|
||||
|
||||
private:
|
||||
struct PendingRenderbufferClear {
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
ClearAttachmentPayload payload{};
|
||||
};
|
||||
|
||||
// A superseded renderbuffer backing (glRenderbufferStorage respecify) parked
|
||||
// until enough frame boundaries have passed that no in-flight command buffer
|
||||
// can still reference it; destroyed in OnPresent (see RetireAgeFrames).
|
||||
struct DeferredRenderbufferRelease {
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
VkImageView view = VK_NULL_HANDLE;
|
||||
Uint64 deferredAtFrame = 0;
|
||||
};
|
||||
|
||||
UnorderedMap<MG_State::GLState::RenderbufferObject*, RenderbufferResource> m_renderbufferResources;
|
||||
UnorderedMap<MG_State::GLState::RenderbufferObject*, PendingRenderbufferClear> m_pendingRenderbufferClears;
|
||||
Vector<DeferredRenderbufferRelease> m_deferredRenderbufferReleases;
|
||||
|
||||
RenderbufferResource* GetOrCreateRenderbufferResource(
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
Bool GetPendingRenderbufferClear(MG_State::GLState::RenderbufferObject* renderbuffer,
|
||||
ClearAttachmentPayload& outPayload) const;
|
||||
Bool HasPendingRenderbufferClear(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) const;
|
||||
void CollectRenderbufferGarbage();
|
||||
// Frame-boundary margin after which a resource last referenced by a retired
|
||||
// GL object (or superseded backing) is provably past every in-flight frame.
|
||||
Uint64 RetireAgeFrames() const;
|
||||
void DeferRenderbufferBackingRelease(RenderbufferResource& resource);
|
||||
void CollectDeferredRenderbufferReleases(Bool destroyAll);
|
||||
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
static inline ActiveRenderPassInfo s_activeRenderPass{};
|
||||
|
||||
@@ -51,6 +51,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Float ResolveEffectiveMinLod(const MG_State::GLState::SamplerObject& sampler, Float effectiveMaxLod) {
|
||||
return std::min(sampler.GetMinLod(), effectiveMaxLod);
|
||||
}
|
||||
|
||||
// A single-level view can only ever deliver the base level, but the LOD clamp must not be
|
||||
// collapsed to exactly 0: both GL and Vulkan pick magFilter over minFilter from the
|
||||
// *clamped* lambda, so maxLod = 0 would make every fragment magnify and quietly retire the
|
||||
// min filter. 0.25 is the value VkSamplerCreateInfo's own note prescribes for emulating
|
||||
// GL's non-mipmapped minification - large enough for lambda to stay positive, small enough
|
||||
// that a NEAREST mip mode still rounds down to level 0. Clamped rather than assigned, so a
|
||||
// texture whose GL_TEXTURE_MAX_LOD really is 0 keeps magnifying as GL says it must.
|
||||
Float ResolveSingleLevelMaxLod(const MG_State::GLState::SamplerObject& sampler, Bool singleLevelView) {
|
||||
const Float maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
return singleLevelView ? std::min(maxLod, 0.25f) : maxLod;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool VkSamplerManager::Initialize(const InitInfo& initInfo) {
|
||||
@@ -65,8 +77,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
Float VkSamplerManager::ResolveEffectiveMaxAnisotropy(const MG_State::GLState::SamplerObject& sampler) const {
|
||||
Float VkSamplerManager::ResolveEffectiveMaxAnisotropy(const MG_State::GLState::SamplerObject& sampler,
|
||||
Bool forceNearestFiltering) const {
|
||||
if (!m_samplerAnisotropySupported) return 1.0f;
|
||||
if (forceNearestFiltering) return 1.0f;
|
||||
// VUID-VkSamplerCreateInfo-anisotropyEnable-01071/01072: anisotropy requires both filters to
|
||||
// be LINEAR and the value to sit within [1, limits.maxSamplerAnisotropy].
|
||||
if (sampler.GetMinFilter() != SamplerFilterMode::Linear ||
|
||||
@@ -87,13 +101,44 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_config = nullptr;
|
||||
m_frameBoundaryCounter = 0;
|
||||
}
|
||||
|
||||
void VkSamplerManager::OnFrameBoundary() {
|
||||
++m_frameBoundaryCounter;
|
||||
|
||||
// Sweep occasionally; destroy samplers whose last use is far past every
|
||||
// in-flight frame. Destroy and erase must stay atomic, or Shutdown would
|
||||
// double-free the handle; an evicted key that recurs simply re-creates
|
||||
// its sampler on the next miss.
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
if ((m_frameBoundaryCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_samplers.begin(); it != m_samplers.end();) {
|
||||
auto& entry = it->second;
|
||||
if (m_frameBoundaryCounter - entry.lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
if (m_device != VK_NULL_HANDLE && entry.handle != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, entry.handle, nullptr);
|
||||
}
|
||||
it = m_samplers.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Uint64 VkSamplerManager::BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture) const {
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const {
|
||||
MOBILEGL_ASSERT(m_config != nullptr, "VkSamplerManager::BuildSamplerKey: m_config is null");
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, m_config->CacheVersion));
|
||||
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &forceNearestFiltering, sizeof(forceNearestFiltering)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &singleLevelView, sizeof(singleLevelView)));
|
||||
|
||||
const auto minFilter = sampler.GetMinFilter();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &minFilter, sizeof(minFilter)));
|
||||
const auto magFilter = sampler.GetMagFilter();
|
||||
@@ -106,7 +151,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapT, sizeof(wrapT)));
|
||||
const auto wrapR = sampler.GetWrapR();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapR, sizeof(wrapR)));
|
||||
const auto maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
const auto maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
const auto minLod = ResolveEffectiveMinLod(sampler, maxLod);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &minLod, sizeof(minLod)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &maxLod, sizeof(maxLod)));
|
||||
@@ -115,7 +160,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// The RESOLVED value, not the GL request: samplers that only differ in an anisotropy Vulkan
|
||||
// will not apply (NEAREST filtering, or requests past the device limit) must still share one
|
||||
// VkSampler, while two samplers that really do differ must not collide onto the first one's.
|
||||
const auto maxAnisotropy = ResolveEffectiveMaxAnisotropy(sampler);
|
||||
const auto maxAnisotropy = ResolveEffectiveMaxAnisotropy(sampler, forceNearestFiltering);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &maxAnisotropy, sizeof(maxAnisotropy)));
|
||||
const auto compareMode = sampler.GetCompareMode();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &compareMode, sizeof(compareMode)));
|
||||
@@ -127,30 +172,44 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
VkSampler VkSamplerManager::GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture) {
|
||||
const Uint64 key = BuildSamplerKey(sampler, texture);
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Uint32 viewLevelCount) {
|
||||
// A view that exposes a single mip level has no second level to blend with, so GL's
|
||||
// *_MIPMAP_* minification filters degenerate to plain filtering on the base level -
|
||||
// sampling is unchanged by pinning the Vulkan sampler to NEAREST mip mode at LOD 0.
|
||||
// It is not cosmetic: MobileGL backs such a view with a fully allocated mip chain whose
|
||||
// tail is never written, and a LINEAR mip mode lets the texture unit issue the level+1
|
||||
// fetch anyway. On Adreno that fetch lands in uninitialized UBWC pages (or past the
|
||||
// allocation for a genuinely single-level image) and faults the GPU - the same failure
|
||||
// the default-framebuffer blit shader had to work around with an explicit-LOD sample.
|
||||
const Bool singleLevelView = viewLevelCount == 1;
|
||||
const Uint64 key = BuildSamplerKey(sampler, texture, forceNearestFiltering, singleLevelView);
|
||||
auto it = m_samplers.find(key);
|
||||
if (it != m_samplers.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return it->second.handle;
|
||||
}
|
||||
|
||||
VkSamplerCreateInfo samplerInfo{};
|
||||
samplerInfo.sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO;
|
||||
samplerInfo.magFilter = ToVkFilter(sampler.GetMagFilter());
|
||||
samplerInfo.minFilter = ToVkFilter(sampler.GetMinFilter());
|
||||
samplerInfo.mipmapMode = ToVkMipmapMode(sampler.GetMipmapMode());
|
||||
samplerInfo.magFilter = forceNearestFiltering ? VK_FILTER_NEAREST : ToVkFilter(sampler.GetMagFilter());
|
||||
samplerInfo.minFilter = forceNearestFiltering ? VK_FILTER_NEAREST : ToVkFilter(sampler.GetMinFilter());
|
||||
samplerInfo.mipmapMode = (forceNearestFiltering || singleLevelView)
|
||||
? VK_SAMPLER_MIPMAP_MODE_NEAREST
|
||||
: ToVkMipmapMode(sampler.GetMipmapMode());
|
||||
samplerInfo.addressModeU = ToVkAddressMode(sampler.GetWrapS());
|
||||
samplerInfo.addressModeV = ToVkAddressMode(sampler.GetWrapT());
|
||||
samplerInfo.addressModeW = ToVkAddressMode(sampler.GetWrapR());
|
||||
samplerInfo.mipLodBias = sampler.GetLodBias();
|
||||
// Must use the same resolver as BuildSamplerKey - a divergence would either collide two
|
||||
// different samplers or silently create duplicates.
|
||||
const Float maxAnisotropy = ResolveEffectiveMaxAnisotropy(sampler);
|
||||
const Float maxAnisotropy = ResolveEffectiveMaxAnisotropy(sampler, forceNearestFiltering);
|
||||
samplerInfo.anisotropyEnable = maxAnisotropy > 1.0f ? VK_TRUE : VK_FALSE;
|
||||
samplerInfo.maxAnisotropy = maxAnisotropy;
|
||||
samplerInfo.compareEnable = sampler.GetCompareMode() == SamplerCompareMode::CompareToTexture ? VK_TRUE : VK_FALSE;
|
||||
samplerInfo.compareOp = ToVkCompareOp(ResolveCompareFunc(sampler, texture));
|
||||
samplerInfo.maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
// Must match BuildSamplerKey's resolution exactly.
|
||||
samplerInfo.maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
samplerInfo.minLod = ResolveEffectiveMinLod(sampler, samplerInfo.maxLod);
|
||||
samplerInfo.borderColor = ResolveVkBorderColor(sampler, texture);
|
||||
samplerInfo.unnormalizedCoordinates = VK_FALSE;
|
||||
@@ -162,6 +221,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
entry.handle = vkSampler;
|
||||
entry.externalIndex = sampler.GetExternalIndex();
|
||||
entry.version = sampler.GetVersion();
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
m_samplers[key] = entry;
|
||||
return vkSampler;
|
||||
}
|
||||
|
||||
@@ -33,18 +33,38 @@ public:
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
|
||||
// viewLevelCount is the mip-level count of the image view this sampler will be paired
|
||||
// with; 0 means "unknown, do not narrow". See GetOrCreateSampler for why it matters.
|
||||
VkSampler GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture);
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering = false,
|
||||
Uint32 viewLevelCount = 0);
|
||||
// Frame boundary hook: ages the sampler cache and destroys samplers not used
|
||||
// for many frames. The key hashes continuous float state (lodBias, LOD clamps,
|
||||
// anisotropy), so an app animating those would otherwise mint an unbounded
|
||||
// stream of never-destroyed VkSamplers and eventually exhaust the device's
|
||||
// maxSamplerAllocationCount. A sampler idle for over a thousand frame
|
||||
// boundaries cannot be referenced by any in-flight command buffer (frames in
|
||||
// flight are single digits), and every descriptor set the GPU consumes is
|
||||
// written that same frame with live handles (the per-binding resolve memo and
|
||||
// descriptor-set reuse are both frame-reset), so destruction here needs no
|
||||
// fence wait. Self-gated: one counter bump and compare except on sweep
|
||||
// boundaries.
|
||||
void OnFrameBoundary();
|
||||
|
||||
private:
|
||||
struct SamplerCacheEntry {
|
||||
VkSampler handle = VK_NULL_HANDLE;
|
||||
Uint externalIndex = 0;
|
||||
Uint16 version = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age have their VkSampler destroyed.
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
};
|
||||
|
||||
Uint64 BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture) const;
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const;
|
||||
static VkFilter ToVkFilter(SamplerFilterMode mode);
|
||||
static VkSamplerMipmapMode ToVkMipmapMode(SamplerMipmapMode mode);
|
||||
static VkSamplerAddressMode ToVkAddressMode(SamplerWrapMode mode);
|
||||
@@ -57,13 +77,16 @@ private:
|
||||
// the sampler filters linearly both ways, otherwise the GL request clamped to the device limit.
|
||||
// GL happily carries GL_TEXTURE_MAX_ANISOTROPY on a NEAREST sampler (Blaze3D's blocks do exactly
|
||||
// that) while Vulkan forbids anisotropyEnable there, so the GL value must never be forwarded raw.
|
||||
Float ResolveEffectiveMaxAnisotropy(const MG_State::GLState::SamplerObject& sampler) const;
|
||||
Float ResolveEffectiveMaxAnisotropy(const MG_State::GLState::SamplerObject& sampler,
|
||||
Bool forceNearestFiltering) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
const VulkanRendererConfig* m_config = nullptr;
|
||||
Bool m_samplerAnisotropySupported = false;
|
||||
Float m_maxSamplerAnisotropy = 1.0f;
|
||||
UnorderedMap<Uint64, SamplerCacheEntry> m_samplers;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -8,6 +8,8 @@
|
||||
|
||||
#include "VkTextureManager.h"
|
||||
|
||||
#include "ProgramFactory.h"
|
||||
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include "MG_Util/Converters/MGToStr/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
@@ -17,6 +19,7 @@
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <memory>
|
||||
#include <vulkan/utility/vk_format_utils.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Compute shaders may legally sample framebuffer-attached textures (the GL feedback-loop rule
|
||||
@@ -63,6 +66,59 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
target == TextureUploadTarget::ProxyTexture2DMultisampleArray;
|
||||
}
|
||||
|
||||
static Bool IsMutableStorageImageFormat(VkFormat format) {
|
||||
if (!vkuFormatIsColor(format) || vkuFormatIsCompressed(format)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// These are the uncompressed color compatibility classes covered by the core GLSL/SPIR-V
|
||||
// storage-image formats. OpenGL mutable texture storage uses image-format compatibility by
|
||||
// size, so a shader may legally reinterpret (for example) RGBA16_UNORM storage as rgba16f. Vulkan
|
||||
// requires the image to be mutable and the view formats to share this exact compatibility
|
||||
// class for the equivalent operation.
|
||||
switch (vkuFormatCompatibilityClass(format)) {
|
||||
case VKU_FORMAT_COMPATIBILITY_CLASS_8BIT:
|
||||
case VKU_FORMAT_COMPATIBILITY_CLASS_16BIT:
|
||||
case VKU_FORMAT_COMPATIBILITY_CLASS_32BIT:
|
||||
case VKU_FORMAT_COMPATIBILITY_CLASS_64BIT:
|
||||
case VKU_FORMAT_COMPATIBILITY_CLASS_128BIT:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
static Bool HasMatchingColorComponentLayout(VkFormat lhs, VkFormat rhs) {
|
||||
const VKU_FORMAT_INFO lhsInfo = vkuGetFormatInfo(lhs);
|
||||
const VKU_FORMAT_INFO rhsInfo = vkuGetFormatInfo(rhs);
|
||||
if (lhsInfo.component_count == 0 || lhsInfo.component_count != rhsInfo.component_count ||
|
||||
lhsInfo.texel_block_size != rhsInfo.texel_block_size ||
|
||||
lhsInfo.texels_per_block != 1 || rhsInfo.texels_per_block != 1) {
|
||||
return false;
|
||||
}
|
||||
for (Uint32 component = 0; component < lhsInfo.component_count; ++component) {
|
||||
if (lhsInfo.components[component].type != rhsInfo.components[component].type ||
|
||||
lhsInfo.components[component].size != rhsInfo.components[component].size) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static Bool FormatMatchesSamplerNumericDomain(VkFormat format, SamplerNumericDomain numericDomain) {
|
||||
switch (numericDomain) {
|
||||
case SamplerNumericDomain::Float:
|
||||
return vkuFormatIsSampledFloat(format);
|
||||
case SamplerNumericDomain::SignedInteger:
|
||||
return vkuFormatIsSINT(format);
|
||||
case SamplerNumericDomain::UnsignedInteger:
|
||||
return vkuFormatIsUINT(format);
|
||||
case SamplerNumericDomain::Unknown:
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static Bool TryResolveSampleCountFlagBits(Int requestedSamples, VkSampleCountFlagBits& outSampleCount) {
|
||||
switch (requestedSamples) {
|
||||
case 1:
|
||||
@@ -319,7 +375,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
// Legacy low-bit RGB formats share the UNorm8 canonical shadow layout (see
|
||||
// TextureFormatProcessor), so they upload exactly like RGB8 with an alpha expand.
|
||||
case TextureInternalFormat::R3G3B2:
|
||||
case TextureInternalFormat::RGB4:
|
||||
case TextureInternalFormat::RGB5:
|
||||
return {VK_FORMAT_R8G8B8A8_UNORM, true, 1, {0xFF, 0x00, 0x00, 0x00}};
|
||||
// Low-bit RGBA formats: UNorm8x4 canonical shadow, no expansion needed.
|
||||
case TextureInternalFormat::RGBA2:
|
||||
case TextureInternalFormat::RGBA4:
|
||||
case TextureInternalFormat::RGB5A1:
|
||||
return {VK_FORMAT_R8G8B8A8_UNORM, false, 0, {0, 0, 0, 0}};
|
||||
// 10/12-bit RGB(A): UNorm16 canonical shadow.
|
||||
case TextureInternalFormat::RGB10:
|
||||
case TextureInternalFormat::RGB12:
|
||||
return {VK_FORMAT_R16G16B16A16_UNORM, true, 2, {0xFF, 0xFF, 0x00, 0x00}};
|
||||
case TextureInternalFormat::RGBA12:
|
||||
return {VK_FORMAT_R16G16B16A16_UNORM, false, 0, {0, 0, 0, 0}};
|
||||
case TextureInternalFormat::SRGB8:
|
||||
return {VK_FORMAT_R8G8B8A8_SRGB, true, 1, {0xFF, 0x00, 0x00, 0x00}};
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
@@ -515,6 +587,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_allocator = initInfo.allocator;
|
||||
m_commandPool = initInfo.commandPool;
|
||||
m_graphicsQueue = initInfo.graphicsQueue;
|
||||
m_imageFormatListSupported = initInfo.imageFormatListSupported;
|
||||
m_currentFrameIndex = 0;
|
||||
m_deferredReleases.clear();
|
||||
m_deferredReleases.resize(initInfo.frameCount);
|
||||
@@ -537,6 +610,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
DestroyDeferredReleases();
|
||||
m_textureResources.clear();
|
||||
m_aliveObjects.clear();
|
||||
m_storageImageTextures.clear();
|
||||
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_physicalDevice = VK_NULL_HANDLE;
|
||||
@@ -555,6 +629,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frameIndex, m_deferredViewReleases.size());
|
||||
m_currentFrameIndex = frameIndex;
|
||||
CollectDeferredReleases(frameIndex);
|
||||
|
||||
// Frame-boundary GC: every 64 frame boundaries (~1 s at 60 fps) bounds the reclaim
|
||||
// latency for dead textures regardless of draw traffic — workloads that churn
|
||||
// textures through clears/readbacks alone never reach the draw-gated
|
||||
// CollectGarbage. Must run after CollectDeferredReleases above: the prune defers
|
||||
// its releases into this frame's slot, which was just drained, so they are
|
||||
// destroyed only after the slot's fence has been waited again one full frame-ring
|
||||
// cycle from now (never while an in-flight frame may still reference them).
|
||||
constexpr Uint32 kGcFrameInterval = 64;
|
||||
++m_gcFrameCounter;
|
||||
if (m_gcFrameCounter % kGcFrameInterval == 0) {
|
||||
PruneDeadTextures();
|
||||
}
|
||||
}
|
||||
|
||||
void VkTextureManager::CollectAllDeferredReleases() {
|
||||
const SizeT frameCount = std::min(m_deferredReleases.size(), m_deferredViewReleases.size());
|
||||
for (SizeT frameIndex = 0; frameIndex < frameCount; ++frameIndex) {
|
||||
CollectDeferredReleases(static_cast<Uint32>(frameIndex));
|
||||
}
|
||||
}
|
||||
|
||||
void VkTextureManager::EraseTrackedTexture(const TextureIdentity& identity) {
|
||||
@@ -564,6 +658,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_textureResources.erase(resourceIt);
|
||||
}
|
||||
m_aliveObjects.erase(identity);
|
||||
m_storageImageTextures.erase(identity);
|
||||
}
|
||||
|
||||
void VkTextureManager::PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture) {
|
||||
@@ -630,9 +725,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// construction introduces a new identity. Doing this unconditionally made every
|
||||
// sampled-texture sync scan the entire alive-texture map per draw.
|
||||
if (aliveIt == m_aliveObjects.end()) {
|
||||
WeakPtr<MG_State::GLState::ITextureObject> aliveTexture;
|
||||
const auto& liveTexture = MG_State::pGLContext->GetTextureObject(texture.GetExternalIndex());
|
||||
if (liveTexture && liveTexture.get() == &texture) {
|
||||
m_aliveObjects[identity] = WeakPtr<MG_State::GLState::ITextureObject>(liveTexture);
|
||||
aliveTexture = liveTexture;
|
||||
} else {
|
||||
// The name lookup legally fails while the object is alive: the name was
|
||||
// deleted with the texture still attached to an FBO (the attachment's
|
||||
// SharedPtr keeps it alive), or the name was reused by a new texture, or
|
||||
// this is a default texture object (name 0 lives outside the name map).
|
||||
// Register through the object's own control block so the resource created
|
||||
// below still participates in weak-expiry GC instead of becoming an
|
||||
// orphan no reclamation path can reach until Shutdown.
|
||||
aliveTexture = texture.weak_from_this();
|
||||
}
|
||||
if (!aliveTexture.expired()) {
|
||||
m_aliveObjects[identity] = Move(aliveTexture);
|
||||
PruneStaleTextureAliases(&texture);
|
||||
}
|
||||
}
|
||||
@@ -767,6 +875,180 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return perMipSampledView;
|
||||
}
|
||||
|
||||
VkImageView VkTextureManager::GetOrCreateSampledImageView(MG_State::GLState::ITextureObject& texture,
|
||||
VkFormat format) {
|
||||
TextureResource* resource = SyncTextureAndGetDescriptor(texture);
|
||||
if (resource == nullptr || resource->image == VK_NULL_HANDLE ||
|
||||
resource->sampledView == VK_NULL_HANDLE) {
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
if (format == VK_FORMAT_UNDEFINED || format == resource->format) {
|
||||
return resource->sampledView;
|
||||
}
|
||||
if (!AreSampledImageViewFormatsCompatible(resource->format, format)) {
|
||||
MGLOG_E("%s: incompatible sampled image view format=%d for textureId=%d imageFormat=%d",
|
||||
__func__, static_cast<Int>(format), texture.GetExternalIndex(),
|
||||
static_cast<Int>(resource->format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
if ((resource->imageCreateFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) == 0) {
|
||||
MGLOG_E("%s: textureId=%d needs mutable image format=%d for sampled view format=%d",
|
||||
__func__, texture.GetExternalIndex(), static_cast<Int>(resource->format),
|
||||
static_cast<Int>(format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
const TextureResource::SampledImageViewKey key{
|
||||
.baseMipLevel = resource->sampledBaseMipLevel,
|
||||
.levelCount = resource->sampledLevelCount,
|
||||
.viewType = resource->viewType,
|
||||
.format = format,
|
||||
};
|
||||
const auto existing = resource->alternateSampledViews.find(key);
|
||||
if (existing != resource->alternateSampledViews.end()) {
|
||||
return existing->second;
|
||||
}
|
||||
|
||||
VkFormatProperties formatProperties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &formatProperties);
|
||||
if ((formatProperties.optimalTilingFeatures & VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT) == 0) {
|
||||
MGLOG_E("%s: sampled image view format=%d lacks VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT "
|
||||
"for textureId=%d (available=0x%x)",
|
||||
__func__, static_cast<Int>(format), texture.GetExternalIndex(),
|
||||
static_cast<Uint32>(formatProperties.optimalTilingFeatures));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
const TextureFormatInfo formatInfo = ResolveTextureFormatInfo(texture.GetFormat());
|
||||
const VkComponentMapping sampledComponents = ResolveSampledViewComponents(texture, formatInfo);
|
||||
const VkImageView view = CreateImageView(
|
||||
resource->image, format, VK_IMAGE_ASPECT_COLOR_BIT, resource->viewType,
|
||||
resource->sampledBaseMipLevel, resource->sampledLevelCount, 0, resource->arrayLayers,
|
||||
&sampledComponents, VK_IMAGE_USAGE_SAMPLED_BIT);
|
||||
if (view == VK_NULL_HANDLE) {
|
||||
MGLOG_E("%s: failed to create sampled image view textureId=%d imageFormat=%d viewFormat=%d",
|
||||
__func__, texture.GetExternalIndex(), static_cast<Int>(resource->format),
|
||||
static_cast<Int>(format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
resource->alternateSampledViews.emplace(key, view);
|
||||
MGLOG_D("%s: created sampled image view textureId=%d imageFormat=%d viewFormat=%d mip=[%u,%u)",
|
||||
__func__, texture.GetExternalIndex(), static_cast<Int>(resource->format),
|
||||
static_cast<Int>(format), resource->sampledBaseMipLevel,
|
||||
resource->sampledBaseMipLevel + resource->sampledLevelCount);
|
||||
return view;
|
||||
}
|
||||
|
||||
VkImageView VkTextureManager::GetOrCreateStorageImageView(MG_State::GLState::ITextureObject& texture,
|
||||
Uint32 mipLevel, VkFormat format,
|
||||
Bool layered, Int32 layer) {
|
||||
TextureResource* resource = SyncTextureAndGetDescriptor(texture);
|
||||
if (resource == nullptr || resource->image == VK_NULL_HANDLE || mipLevel >= resource->mipLevels ||
|
||||
resource->sampleCount != VK_SAMPLE_COUNT_1_BIT ||
|
||||
(resource->aspect & VK_IMAGE_ASPECT_COLOR_BIT) == 0) {
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
if (format == VK_FORMAT_UNDEFINED) {
|
||||
format = resource->format;
|
||||
}
|
||||
if (!AreStorageImageViewFormatsCompatible(resource->format, format)) {
|
||||
MGLOG_E("%s: incompatible storage image view format=%d for textureId=%d imageFormat=%d",
|
||||
__func__, static_cast<Int>(format), texture.GetExternalIndex(),
|
||||
static_cast<Int>(resource->format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
if (format != resource->format &&
|
||||
(resource->imageCreateFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) == 0) {
|
||||
MGLOG_E("%s: textureId=%d needs mutable image format=%d for storage view format=%d",
|
||||
__func__, texture.GetExternalIndex(), static_cast<Int>(resource->format),
|
||||
static_cast<Int>(format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = resource->arrayLayers;
|
||||
VkImageViewType viewType = resource->viewType;
|
||||
if (!layered) {
|
||||
switch (resource->viewType) {
|
||||
case VK_IMAGE_VIEW_TYPE_1D_ARRAY:
|
||||
viewType = VK_IMAGE_VIEW_TYPE_1D;
|
||||
break;
|
||||
case VK_IMAGE_VIEW_TYPE_2D_ARRAY:
|
||||
case VK_IMAGE_VIEW_TYPE_CUBE:
|
||||
case VK_IMAGE_VIEW_TYPE_CUBE_ARRAY:
|
||||
viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
break;
|
||||
case VK_IMAGE_VIEW_TYPE_3D:
|
||||
MGLOG_E("%s: non-layered 3D storage views are unsupported for textureId=%d",
|
||||
__func__, texture.GetExternalIndex());
|
||||
return VK_NULL_HANDLE;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
if (viewType != resource->viewType) {
|
||||
if (layer < 0 || static_cast<Uint32>(layer) >= resource->arrayLayers) {
|
||||
MGLOG_E("%s: storage image layer=%d is out of range for textureId=%d arrayLayers=%u",
|
||||
__func__, layer, texture.GetExternalIndex(), resource->arrayLayers);
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
baseArrayLayer = static_cast<Uint32>(layer);
|
||||
layerCount = 1;
|
||||
}
|
||||
}
|
||||
|
||||
const Bool isFullResourceView = baseArrayLayer == 0 && layerCount == resource->arrayLayers &&
|
||||
viewType == resource->viewType;
|
||||
if (format == resource->format && isFullResourceView) {
|
||||
return GetOrCreateViewAtMipLevel(texture, mipLevel);
|
||||
}
|
||||
|
||||
const TextureResource::StorageImageViewKey key{
|
||||
.mipLevel = mipLevel,
|
||||
.baseArrayLayer = baseArrayLayer,
|
||||
.layerCount = layerCount,
|
||||
.viewType = viewType,
|
||||
.format = format,
|
||||
};
|
||||
auto it = resource->storageImageViews.find(key);
|
||||
if (it != resource->storageImageViews.end()) {
|
||||
return it->second;
|
||||
}
|
||||
|
||||
VkFormatFeatureFlags requiredFormatFeatures = VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT;
|
||||
if (format != resource->format &&
|
||||
(format == VK_FORMAT_R32_UINT || format == VK_FORMAT_R32_SINT)) {
|
||||
requiredFormatFeatures |= VK_FORMAT_FEATURE_STORAGE_IMAGE_ATOMIC_BIT;
|
||||
}
|
||||
VkFormatProperties formatProperties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &formatProperties);
|
||||
if ((formatProperties.optimalTilingFeatures & requiredFormatFeatures) != requiredFormatFeatures) {
|
||||
MGLOG_E("%s: storage image view format=%d lacks required features=0x%x for textureId=%d "
|
||||
"(available=0x%x)",
|
||||
__func__, static_cast<Int>(format), static_cast<Uint32>(requiredFormatFeatures),
|
||||
texture.GetExternalIndex(), static_cast<Uint32>(formatProperties.optimalTilingFeatures));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
const VkImageView view = CreateImageView(resource->image, format, VK_IMAGE_ASPECT_COLOR_BIT, viewType,
|
||||
mipLevel, 1, baseArrayLayer, layerCount, nullptr,
|
||||
VK_IMAGE_USAGE_STORAGE_BIT);
|
||||
if (view == VK_NULL_HANDLE) {
|
||||
MGLOG_E("%s: failed to create storage image view for textureId=%d mip=%u imageFormat=%d viewFormat=%d",
|
||||
__func__, texture.GetExternalIndex(), mipLevel, static_cast<Int>(resource->format),
|
||||
static_cast<Int>(format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
resource->storageImageViews.emplace(key, view);
|
||||
MGLOG_D("%s: created storage image view textureId=%d mip=%u imageFormat=%d viewFormat=%d",
|
||||
__func__, texture.GetExternalIndex(), mipLevel, static_cast<Int>(resource->format),
|
||||
static_cast<Int>(format));
|
||||
return view;
|
||||
}
|
||||
|
||||
void VkTextureManager::UpdateTrackedImageLayout(MG_State::GLState::ITextureObject* texture, VkImageLayout newLayout) {
|
||||
MOBILEGL_ASSERT(texture != nullptr, "UpdateTrackedImageLayout: texture is null");
|
||||
auto it = m_textureResources.find(MakeTextureIdentity(texture));
|
||||
@@ -910,6 +1192,47 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return ok;
|
||||
}
|
||||
|
||||
void VkTextureManager::MarkStorageImageTexture(MG_State::GLState::ITextureObject& texture) {
|
||||
m_storageImageTextures.insert(MakeTextureIdentity(&texture));
|
||||
}
|
||||
|
||||
Bool VkTextureManager::NeedsStorageUsageUpgrade(MG_State::GLState::ITextureObject& texture) const {
|
||||
const TextureIdentity identity = MakeTextureIdentity(&texture);
|
||||
if (m_storageImageTextures.find(identity) == m_storageImageTextures.end()) {
|
||||
return false;
|
||||
}
|
||||
const auto it = m_textureResources.find(identity);
|
||||
// No image yet: the first sync creates it with STORAGE straight away, so there is nothing
|
||||
// to preserve and nothing to order against.
|
||||
return it != m_textureResources.end() && it->second.image != VK_NULL_HANDLE &&
|
||||
!it->second.storageUsageResolved;
|
||||
}
|
||||
|
||||
Bool VkTextureManager::NeedsStorageImagePreparation(MG_State::GLState::ITextureObject& texture) const {
|
||||
const TextureIdentity identity = MakeTextureIdentity(&texture);
|
||||
const auto it = m_textureResources.find(identity);
|
||||
if (it == m_textureResources.end()) {
|
||||
return true;
|
||||
}
|
||||
const TextureResource& resource = it->second;
|
||||
if (resource.image == VK_NULL_HANDLE || resource.layout != VK_IMAGE_LAYOUT_GENERAL) {
|
||||
return true;
|
||||
}
|
||||
// The image predates this texture's first image-unit binding, so it was created without
|
||||
// STORAGE usage and has to be recreated - which is illegal inside a render pass.
|
||||
if (!resource.storageUsageResolved &&
|
||||
m_storageImageTextures.find(identity) != m_storageImageTextures.end()) {
|
||||
return true;
|
||||
}
|
||||
// Mirror SyncTexture's cross-draw skip condition: any version drift means the sync
|
||||
// path may upload or rebuild, both of which need the render pass ended first.
|
||||
const auto* mipTexture = MG_State::GLState::AsMipmapTexture(&texture);
|
||||
const Uint32 mipLevelCount = mipTexture != nullptr ? mipTexture->GetMipmapLevelCount() : 0u;
|
||||
return resource.syncedContentVersion != texture.GetContentVersion() ||
|
||||
resource.syncedTextureParamsVersion != texture.GetTextureParamsVersion() ||
|
||||
resource.syncedMipLevelCount != mipLevelCount;
|
||||
}
|
||||
|
||||
Bool VkTextureManager::TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image,
|
||||
VkImageLayout& trackedLayout, VkImageLayout newLayout,
|
||||
VkPipelineStageFlags srcStageMask, VkPipelineStageFlags dstStageMask,
|
||||
@@ -948,10 +1271,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
SizeT VkTextureManager::CollectGarbage() {
|
||||
// Draw-gated stagger (1 in 256 calls): keeps the per-draw cost at one counter
|
||||
// bump. The guaranteed reclaim path is the frame-boundary prune in BeginFrame;
|
||||
// this remains as a cheap assist so draw-heavy workloads reclaim sooner.
|
||||
m_gcCounter++;
|
||||
if (m_gcCounter != 0) {
|
||||
return 0;
|
||||
}
|
||||
return PruneDeadTextures();
|
||||
}
|
||||
|
||||
SizeT VkTextureManager::PruneDeadTextures() {
|
||||
// Erasing entries would dangle the raw TextureResource pointers memoized for the
|
||||
// current draw; every call path (BeginFrame, and CollectGarbage at the top of a
|
||||
// freshly opened draw-sync scope) runs before any memo entry is recorded.
|
||||
MOBILEGL_ASSERT(m_drawSyncedThisDraw.empty(),
|
||||
"PruneDeadTextures: draw-sync memo holds raw resource pointers an erase would dangle");
|
||||
|
||||
Vector<MG_State::GLState::ITextureObject*> expiredTextures;
|
||||
expiredTextures.reserve(m_aliveObjects.size());
|
||||
@@ -963,7 +1298,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (auto* texture : expiredTextures) {
|
||||
PruneStaleTextureAliases(texture);
|
||||
}
|
||||
return expiredTextures.size();
|
||||
SizeT prunedCount = expiredTextures.size();
|
||||
|
||||
// Orphan sweep: after the pass above, m_aliveObjects holds only live entries.
|
||||
// Registration in SyncTextureAndGetDescriptor cannot fail for a SharedPtr-owned
|
||||
// texture (weak_from_this fallback), so a resource whose identity has no alive
|
||||
// entry has no trackable owner: its GL-side object is gone, or was never
|
||||
// shared-owned, in which case recreation on a later sync is the safe fallback.
|
||||
// Destruction goes through the per-frame deferred queues, never immediate.
|
||||
Vector<TextureIdentity> orphanIdentities;
|
||||
for (auto it = m_textureResources.begin(); it != m_textureResources.end(); ++it) {
|
||||
if (m_aliveObjects.find(it->first) == m_aliveObjects.end()) {
|
||||
orphanIdentities.emplace_back(it->first);
|
||||
}
|
||||
}
|
||||
for (const auto& identity : orphanIdentities) {
|
||||
EraseTrackedTexture(identity);
|
||||
}
|
||||
prunedCount += orphanIdentities.size();
|
||||
return prunedCount;
|
||||
}
|
||||
|
||||
Bool VkTextureManager::SyncTexture(MG_State::GLState::ITextureObject &texture,
|
||||
@@ -977,7 +1330,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const auto* syncingMipTexture = MG_State::GLState::AsMipmapTexture(&texture);
|
||||
const Uint32 syncingMipLevelCount =
|
||||
syncingMipTexture != nullptr ? syncingMipTexture->GetMipmapLevelCount() : 0u;
|
||||
if (outResource.image != VK_NULL_HANDLE &&
|
||||
// A pending storage-usage upgrade also has to bust the skip: nothing about the texture's
|
||||
// content or params changed, but the image itself must be recreated with STORAGE usage
|
||||
// before it can back an image-unit descriptor.
|
||||
const Bool storageUpgradePending =
|
||||
!outResource.storageUsageResolved &&
|
||||
m_storageImageTextures.find(MakeTextureIdentity(&texture)) != m_storageImageTextures.end();
|
||||
if (outResource.image != VK_NULL_HANDLE && !storageUpgradePending &&
|
||||
outResource.syncedContentVersion == syncingContentVersion &&
|
||||
outResource.syncedTextureParamsVersion == texture.GetTextureParamsVersion() &&
|
||||
outResource.syncedMipLevelCount == syncingMipLevelCount) {
|
||||
@@ -1085,6 +1444,47 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
|
||||
const VkImageAspectFlags aspect = GetAspectMaskForFormat(format);
|
||||
VkFormatProperties formatProperties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &formatProperties);
|
||||
// Only textures that have actually been bound to a GL image unit get STORAGE usage (and
|
||||
// the MUTABLE_FORMAT it drags in for format-reinterpreting image views). Requesting it
|
||||
// for every storage-capable colour texture costs real bandwidth: Adreno cannot keep UBWC
|
||||
// compression on an image that may be written through a storage descriptor, so the whole
|
||||
// render target - MC's included - runs uncompressed. MarkStorageImageTexture upgrades a
|
||||
// texture before its first image-unit draw, and the usage below feeds the compatibility
|
||||
// check so the upgrade recreates the image.
|
||||
const Bool markedAsStorageImage =
|
||||
m_storageImageTextures.find(MakeTextureIdentity(
|
||||
const_cast<MG_State::GLState::ITextureObject*>(&texture))) != m_storageImageTextures.end();
|
||||
// Storage-image CAPABILITY (does the format allow it at all) is deliberately separate from
|
||||
// whether this texture actually needs the usage. MUTABLE_FORMAT keys off capability, as
|
||||
// before: format-reinterpreting views are not a storage-only concern - the SAMPLED path
|
||||
// needs them too (GetOrCreateSampledImageView bails out without it, see ~line 892), so
|
||||
// tying MUTABLE_FORMAT to the image-unit mark would break sampled format reinterpretation
|
||||
// for every texture that never becomes a storage image.
|
||||
const Bool storageImageCapable =
|
||||
!isMultisampleTexture &&
|
||||
(aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0 &&
|
||||
(formatProperties.optimalTilingFeatures & VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0;
|
||||
const Bool supportsStorageImage = storageImageCapable && markedAsStorageImage;
|
||||
VkImageCreateFlags imageCreateFlags = shapeInfo.imageFlags;
|
||||
if (storageImageCapable && IsMutableStorageImageFormat(format) &&
|
||||
m_mutableFormatUnsupported.find(format) == m_mutableFormatUnsupported.end()) {
|
||||
imageCreateFlags |= VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT;
|
||||
}
|
||||
|
||||
VkImageUsageFlags desiredUsage =
|
||||
VK_IMAGE_USAGE_SAMPLED_BIT |
|
||||
(supportsStorageImage ? VK_IMAGE_USAGE_STORAGE_BIT : 0) |
|
||||
((aspect & VK_IMAGE_ASPECT_COLOR_BIT) ? VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT : 0) |
|
||||
(((aspect & VK_IMAGE_ASPECT_DEPTH_BIT) || (aspect & VK_IMAGE_ASPECT_STENCIL_BIT)) ?
|
||||
VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT :
|
||||
0);
|
||||
if (!isMultisampleTexture) {
|
||||
desiredUsage |= VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
|
||||
}
|
||||
|
||||
const Bool compatible = resource.image != VK_NULL_HANDLE && resource.format == format &&
|
||||
resource.extent.width == static_cast<Uint32>(texelSize.x()) &&
|
||||
resource.extent.height == static_cast<Uint32>(texelSize.y()) &&
|
||||
@@ -1092,6 +1492,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.arrayLayers == shapeInfo.arrayLayers &&
|
||||
resource.viewType == shapeInfo.viewType &&
|
||||
resource.sampleCount == resolvedSampleCount &&
|
||||
resource.imageCreateFlags == imageCreateFlags &&
|
||||
resource.usageFlags == desiredUsage &&
|
||||
resource.mipLevels == backingMipLevels;
|
||||
if (compatible) {
|
||||
if (resource.perMipViews.size() != backingMipLevels) {
|
||||
@@ -1100,6 +1502,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (resource.perMipSampledViews.size() != backingMipLevels) {
|
||||
resource.perMipSampledViews.resize(backingMipLevels, VK_NULL_HANDLE);
|
||||
}
|
||||
// Keeping the image is itself the answer to the mark: either it already carries
|
||||
// STORAGE, or this format can never carry it. Either way there is nothing left to
|
||||
// recreate, so stop reporting the texture as needing preparation.
|
||||
resource.storageUsageResolved = markedAsStorageImage;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -1112,8 +1518,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.arrayLayers == shapeInfo.arrayLayers &&
|
||||
resource.viewType == shapeInfo.viewType &&
|
||||
resource.sampleCount == resolvedSampleCount &&
|
||||
resource.imageCreateFlags == imageCreateFlags &&
|
||||
resolvedSampleCount == VK_SAMPLE_COUNT_1_BIT &&
|
||||
resource.mipLevels < backingMipLevels &&
|
||||
// '<=' rather than '<': a storage-usage upgrade recreates the image with an
|
||||
// unchanged mip count, and its contents (a render target's pixels live only on the
|
||||
// GPU) still have to survive. The vkCmdCopyImage below copies min(mipLevels).
|
||||
resource.mipLevels <= backingMipLevels &&
|
||||
resource.layout != VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
|
||||
std::unique_ptr<TextureResource> preservedResource;
|
||||
@@ -1123,45 +1533,78 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
DeferResourceRelease(Move(resource));
|
||||
}
|
||||
|
||||
auto aspect = GetAspectMaskForFormat(format);
|
||||
|
||||
VkImageCreateInfo imageInfo{};
|
||||
imageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
|
||||
imageInfo.flags = shapeInfo.imageFlags;
|
||||
imageInfo.imageType = shapeInfo.imageType;
|
||||
imageInfo.flags = imageCreateFlags;
|
||||
imageInfo.imageType = shapeInfo.imageType;
|
||||
imageInfo.extent.width = static_cast<Uint32>(texelSize.x());
|
||||
imageInfo.extent.height = static_cast<Uint32>(texelSize.y());
|
||||
imageInfo.extent.depth = shapeInfo.depth;
|
||||
imageInfo.extent.depth = shapeInfo.depth;
|
||||
imageInfo.mipLevels = backingMipLevels;
|
||||
imageInfo.arrayLayers = shapeInfo.arrayLayers;
|
||||
imageInfo.arrayLayers = shapeInfo.arrayLayers;
|
||||
imageInfo.format = format;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkFormatProperties formatProperties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &formatProperties);
|
||||
const Bool supportsStorageImage =
|
||||
!isMultisampleTexture &&
|
||||
(aspect & VK_IMAGE_ASPECT_COLOR_BIT) != 0 &&
|
||||
(formatProperties.optimalTilingFeatures & VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_SAMPLED_BIT |
|
||||
(supportsStorageImage ? VK_IMAGE_USAGE_STORAGE_BIT : 0) |
|
||||
((aspect & VK_IMAGE_ASPECT_COLOR_BIT) ? VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT : 0) |
|
||||
(((aspect & VK_IMAGE_ASPECT_DEPTH_BIT) || (aspect & VK_IMAGE_ASPECT_STENCIL_BIT)) ?
|
||||
VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT :
|
||||
0);
|
||||
if (!isMultisampleTexture) {
|
||||
imageInfo.usage |= VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
|
||||
}
|
||||
imageInfo.usage = desiredUsage;
|
||||
imageInfo.samples = resolvedSampleCount;
|
||||
if (isMultisampleTexture) {
|
||||
|
||||
// Bound the mutability. A blindly-mutable image has to be laid out so that ANY format in
|
||||
// its compatibility class can be viewed, which costs bandwidth compression on tilers;
|
||||
// naming the exact set instead lets the driver keep it. Only safe when that set really is
|
||||
// exhaustive, so it is restricted to textures that are not image-unit bound: sampled views
|
||||
// can only ever ask for ResolveSampledImageViewFormat's output, whereas glBindImageTexture
|
||||
// may name any compatible format, which nothing here can enumerate ahead of time.
|
||||
Vector<VkFormat> viewFormats;
|
||||
VkImageFormatListCreateInfo formatListInfo{};
|
||||
if (m_imageFormatListSupported && !supportsStorageImage &&
|
||||
(imageInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) {
|
||||
viewFormats.push_back(format);
|
||||
for (const SamplerNumericDomain domain : {SamplerNumericDomain::Float,
|
||||
SamplerNumericDomain::SignedInteger,
|
||||
SamplerNumericDomain::UnsignedInteger}) {
|
||||
const VkFormat viewFormat = ResolveSampledImageViewFormat(format, domain);
|
||||
if (viewFormat == VK_FORMAT_UNDEFINED) {
|
||||
continue;
|
||||
}
|
||||
if (std::find(viewFormats.begin(), viewFormats.end(), viewFormat) == viewFormats.end()) {
|
||||
viewFormats.push_back(viewFormat);
|
||||
}
|
||||
}
|
||||
formatListInfo.sType = VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO;
|
||||
formatListInfo.viewFormatCount = static_cast<Uint32>(viewFormats.size());
|
||||
formatListInfo.pViewFormats = viewFormats.data();
|
||||
imageInfo.pNext = &formatListInfo;
|
||||
}
|
||||
|
||||
if (isMultisampleTexture || (imageInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) {
|
||||
VkImageFormatProperties imageFormatProperties{};
|
||||
const VkResult imageFormatResult = vkGetPhysicalDeviceImageFormatProperties(
|
||||
VkResult imageFormatResult = vkGetPhysicalDeviceImageFormatProperties(
|
||||
m_physicalDevice, format, imageInfo.imageType, imageInfo.tiling, imageInfo.usage,
|
||||
imageInfo.flags, &imageFormatProperties);
|
||||
if (imageFormatResult != VK_SUCCESS && !isMultisampleTexture &&
|
||||
(imageInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) {
|
||||
// Losing reinterpreted views only degrades the formatless-image feature for
|
||||
// this texture; failing creation would lose the texture entirely, so retry
|
||||
// as a plain immutable-format image.
|
||||
MGLOG_W("%s: mutable image format=%d is unsupported for textureId=%d; creating "
|
||||
"without VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT (format reinterpretation "
|
||||
"will be unavailable for it)",
|
||||
__func__, static_cast<Int>(format), texture.GetExternalIndex());
|
||||
// Remember the verdict so later syncs of same-format textures neither retry
|
||||
// the probe nor flag-mismatch against this image and recreate it.
|
||||
m_mutableFormatUnsupported.insert(format);
|
||||
imageInfo.flags &= ~VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT;
|
||||
imageCreateFlags = imageInfo.flags;
|
||||
imageFormatResult = vkGetPhysicalDeviceImageFormatProperties(
|
||||
m_physicalDevice, format, imageInfo.imageType, imageInfo.tiling, imageInfo.usage,
|
||||
imageInfo.flags, &imageFormatProperties);
|
||||
}
|
||||
if (imageFormatResult != VK_SUCCESS ||
|
||||
(imageFormatProperties.sampleCounts & resolvedSampleCount) == 0) {
|
||||
MGLOG_D("%s: sampleCount=%d is unsupported for textureId=%d target=%s format=%d usage=0x%x",
|
||||
__func__, texture.GetSamples(), texture.GetExternalIndex(),
|
||||
(isMultisampleTexture && (imageFormatProperties.sampleCounts & resolvedSampleCount) == 0)) {
|
||||
MGLOG_D("%s: image flags=0x%x sampleCount=%d are unsupported for textureId=%d target=%s "
|
||||
"format=%d usage=0x%x",
|
||||
__func__, static_cast<Uint32>(imageInfo.flags), texture.GetSamples(),
|
||||
texture.GetExternalIndex(),
|
||||
MG_Util::ConvertTextureUploadTargetToString(uploadTarget).c_str(),
|
||||
static_cast<Int>(format), static_cast<Uint32>(imageInfo.usage));
|
||||
return false;
|
||||
@@ -1188,6 +1631,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.aspect = aspect;
|
||||
resource.viewType = shapeInfo.viewType;
|
||||
resource.sampleCount = resolvedSampleCount;
|
||||
resource.imageCreateFlags = imageCreateFlags;
|
||||
resource.usageFlags = imageInfo.usage;
|
||||
resource.storageUsageResolved = markedAsStorageImage;
|
||||
resource.syncedTextureParamsVersion = 0;
|
||||
|
||||
if (preservedResource) {
|
||||
@@ -1203,7 +1649,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void VkTextureManager::DeferResourceRelease(TextureResource&& resource) {
|
||||
if (resource.image == VK_NULL_HANDLE && resource.fullView == VK_NULL_HANDLE &&
|
||||
resource.sampledView == VK_NULL_HANDLE &&
|
||||
resource.perMipViews.empty() && resource.perMipSampledViews.empty()) {
|
||||
resource.perMipViews.empty() && resource.perMipSampledViews.empty() &&
|
||||
resource.attachmentViews.empty() && resource.alternateSampledViews.empty() &&
|
||||
resource.storageImageViews.empty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1299,6 +1747,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
sampledView = VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
for (const auto& [_, sampledView] : resource.alternateSampledViews) {
|
||||
DeferViewRelease(sampledView);
|
||||
}
|
||||
resource.alternateSampledViews.clear();
|
||||
|
||||
const TextureFormatInfo formatInfo = ResolveTextureFormatInfo(texture.GetFormat());
|
||||
const VkComponentMapping sampledComponents = ResolveSampledViewComponents(texture, formatInfo);
|
||||
@@ -1324,7 +1776,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkImageViewType viewType, Uint32 baseMipLevel, Uint32 levelCount,
|
||||
Uint32 baseArrayLayer,
|
||||
Uint32 layerCount,
|
||||
const VkComponentMapping* components) const {
|
||||
const VkComponentMapping* components,
|
||||
VkImageUsageFlags viewUsage) const {
|
||||
VkImageViewCreateInfo viewInfo{};
|
||||
viewInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
|
||||
viewInfo.image = image;
|
||||
@@ -1340,6 +1793,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
viewInfo.subresourceRange.baseArrayLayer = baseArrayLayer;
|
||||
viewInfo.subresourceRange.layerCount = layerCount;
|
||||
|
||||
VkImageViewUsageCreateInfo usageInfo{};
|
||||
if (viewUsage != 0) {
|
||||
usageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_USAGE_CREATE_INFO;
|
||||
usageInfo.usage = viewUsage;
|
||||
viewInfo.pNext = &usageInfo;
|
||||
}
|
||||
|
||||
VkImageView view = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateImageView(m_device, &viewInfo, nullptr, &view), "vkCreateImageView(texture)");
|
||||
return view;
|
||||
@@ -1665,4 +2125,64 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
return imageAspect;
|
||||
}
|
||||
|
||||
VkFormat VkTextureManager::ResolveSampledImageViewFormat(VkFormat imageFormat,
|
||||
SamplerNumericDomain numericDomain) {
|
||||
// Depth/stencil images always sample through the existing depth-aspect sampledView.
|
||||
// Combined formats (D24S8, D32FS8) are multi-numeric, so vkuFormatIsSampledFloat is
|
||||
// false for them by design, yet their depth aspect reads as float in every GL depth
|
||||
// texture mode; Vulkan also forbids reinterpreting them through color-class views.
|
||||
// Integer domains keep the same view (pre-reinterpretation behavior for stencil-index
|
||||
// style access) rather than failing the draw.
|
||||
if (vkuFormatIsDepthOrStencil(imageFormat)) {
|
||||
return imageFormat;
|
||||
}
|
||||
if (imageFormat == VK_FORMAT_UNDEFINED || numericDomain == SamplerNumericDomain::Unknown ||
|
||||
FormatMatchesSamplerNumericDomain(imageFormat, numericDomain)) {
|
||||
return imageFormat;
|
||||
}
|
||||
if (!IsMutableStorageImageFormat(imageFormat)) {
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
|
||||
// Preserve component ordering and bit widths. This selects R32_UINT for an R32_SFLOAT
|
||||
// texture sampled by a usampler rather than an arbitrary member (such as
|
||||
// R8G8B8A8_UINT) of Vulkan's broad 32-bit compatibility class.
|
||||
for (Int candidateValue = static_cast<Int>(VK_FORMAT_R4G4_UNORM_PACK8);
|
||||
candidateValue <= static_cast<Int>(VK_FORMAT_ASTC_12x12_SRGB_BLOCK);
|
||||
++candidateValue) {
|
||||
const VkFormat candidate = static_cast<VkFormat>(candidateValue);
|
||||
if (!IsMutableStorageImageFormat(candidate) ||
|
||||
!FormatMatchesSamplerNumericDomain(candidate, numericDomain) ||
|
||||
!HasMatchingColorComponentLayout(imageFormat, candidate) ||
|
||||
!AreSampledImageViewFormatsCompatible(imageFormat, candidate)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// If an integer backing is intentionally bit-read through a float sampler, require
|
||||
// a true floating-point view. Normalized/scaled views satisfy OpTypeFloat but apply
|
||||
// an unrelated numeric conversion to those bits.
|
||||
if (numericDomain == SamplerNumericDomain::Float && !vkuFormatIsSFLOAT(candidate)) {
|
||||
continue;
|
||||
}
|
||||
return candidate;
|
||||
}
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
|
||||
Bool VkTextureManager::AreSampledImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat) {
|
||||
if (imageFormat == viewFormat) {
|
||||
return true;
|
||||
}
|
||||
return IsMutableStorageImageFormat(imageFormat) && IsMutableStorageImageFormat(viewFormat) &&
|
||||
vkuFormatCompatibilityClass(imageFormat) == vkuFormatCompatibilityClass(viewFormat);
|
||||
}
|
||||
|
||||
Bool VkTextureManager::AreStorageImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat) {
|
||||
if (imageFormat == viewFormat) {
|
||||
return true;
|
||||
}
|
||||
return IsMutableStorageImageFormat(imageFormat) && IsMutableStorageImageFormat(viewFormat) &&
|
||||
vkuFormatCompatibilityClass(imageFormat) == vkuFormatCompatibilityClass(viewFormat);
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -13,12 +13,15 @@
|
||||
#include <MG_State/GLState/TextureState/TextureObject.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class SamplerNumericDomain : Uint8;
|
||||
|
||||
class VkTextureManager {
|
||||
public:
|
||||
// Monotonic epoch bumped whenever a texture VkImage is (re)created. The render-pass
|
||||
@@ -50,6 +53,9 @@ public:
|
||||
VkCommandPool commandPool = VK_NULL_HANDLE;
|
||||
VkQueue graphicsQueue = VK_NULL_HANDLE;
|
||||
Uint32 frameCount = 0;
|
||||
// VK_KHR_image_format_list is enabled: MUTABLE_FORMAT images can name the exact set of
|
||||
// formats they will be viewed as, which is what lets a tiler keep them compressed.
|
||||
Bool imageFormatListSupported = false;
|
||||
};
|
||||
|
||||
struct TextureResource {
|
||||
@@ -78,6 +84,61 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
struct StorageImageViewKey {
|
||||
Uint32 mipLevel = 0;
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = 1;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
|
||||
Bool operator==(const StorageImageViewKey& other) const {
|
||||
return mipLevel == other.mipLevel &&
|
||||
baseArrayLayer == other.baseArrayLayer &&
|
||||
layerCount == other.layerCount &&
|
||||
viewType == other.viewType &&
|
||||
format == other.format;
|
||||
}
|
||||
};
|
||||
|
||||
struct SampledImageViewKey {
|
||||
Uint32 baseMipLevel = 0;
|
||||
Uint32 levelCount = 1;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
|
||||
Bool operator==(const SampledImageViewKey& other) const {
|
||||
return baseMipLevel == other.baseMipLevel &&
|
||||
levelCount == other.levelCount &&
|
||||
viewType == other.viewType &&
|
||||
format == other.format;
|
||||
}
|
||||
};
|
||||
|
||||
struct SampledImageViewKeyHash {
|
||||
SizeT operator()(const SampledImageViewKey& key) const {
|
||||
SizeT hash = std::hash<Uint32>{}(key.baseMipLevel);
|
||||
hash ^= std::hash<Uint32>{}(key.levelCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewType)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.format)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
struct StorageImageViewKeyHash {
|
||||
SizeT operator()(const StorageImageViewKey& key) const {
|
||||
SizeT hash = std::hash<Uint32>{}(key.mipLevel);
|
||||
hash ^= std::hash<Uint32>{}(key.baseArrayLayer) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(key.layerCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewType)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.format)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
VkImageView fullView = VK_NULL_HANDLE;
|
||||
@@ -85,6 +146,8 @@ public:
|
||||
Vector<VkImageView> perMipViews;
|
||||
Vector<VkImageView> perMipSampledViews;
|
||||
UnorderedMap<AttachmentViewKey, VkImageView, AttachmentViewKeyHash> attachmentViews;
|
||||
UnorderedMap<SampledImageViewKey, VkImageView, SampledImageViewKeyHash> alternateSampledViews;
|
||||
UnorderedMap<StorageImageViewKey, VkImageView, StorageImageViewKeyHash> storageImageViews;
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkExtent2D extent = {0, 0};
|
||||
Uint32 depth = 1;
|
||||
@@ -96,6 +159,18 @@ public:
|
||||
VkImageAspectFlags aspect = VK_IMAGE_ASPECT_NONE;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
VkImageCreateFlags imageCreateFlags = 0;
|
||||
// Usage the live image was created with. STORAGE is only requested for textures that
|
||||
// have actually been bound to a GL image unit, because on Adreno a storage-capable
|
||||
// image loses UBWC bandwidth compression; a later image binding upgrades the usage
|
||||
// and recreates the image, so the resolved usage has to be part of the compatibility
|
||||
// check that decides whether the existing image can be kept.
|
||||
VkImageUsageFlags usageFlags = 0;
|
||||
// True once this image was (re)resolved while the texture was already marked as an
|
||||
// image-unit texture. Distinguishes "not upgraded yet" from "cannot be upgraded"
|
||||
// (a format whose optimalTilingFeatures lack STORAGE_IMAGE never gains the bit), so
|
||||
// NeedsStorageImagePreparation cannot ask for a recreate that will never happen.
|
||||
Bool storageUsageResolved = false;
|
||||
Uint16 syncedTextureParamsVersion = 0;
|
||||
// Snapshot of ITextureObject::GetContentVersion() at the last successful sync;
|
||||
// lets SyncTexture skip the whole re-check/re-upload when content is unchanged.
|
||||
@@ -115,6 +190,8 @@ public:
|
||||
std::swap(this->perMipViews, that.perMipViews);
|
||||
std::swap(this->perMipSampledViews, that.perMipSampledViews);
|
||||
std::swap(this->attachmentViews, that.attachmentViews);
|
||||
std::swap(this->alternateSampledViews, that.alternateSampledViews);
|
||||
std::swap(this->storageImageViews, that.storageImageViews);
|
||||
std::swap(this->layout, that.layout);
|
||||
std::swap(this->extent, that.extent);
|
||||
std::swap(this->depth, that.depth);
|
||||
@@ -126,6 +203,9 @@ public:
|
||||
std::swap(this->aspect, that.aspect);
|
||||
std::swap(this->viewType, that.viewType);
|
||||
std::swap(this->sampleCount, that.sampleCount);
|
||||
std::swap(this->imageCreateFlags, that.imageCreateFlags);
|
||||
std::swap(this->usageFlags, that.usageFlags);
|
||||
std::swap(this->storageUsageResolved, that.storageUsageResolved);
|
||||
std::swap(this->syncedTextureParamsVersion, that.syncedTextureParamsVersion);
|
||||
std::swap(this->syncedContentVersion, that.syncedContentVersion);
|
||||
std::swap(this->syncedMipLevelCount, that.syncedMipLevelCount);
|
||||
@@ -153,6 +233,16 @@ public:
|
||||
vkDestroyImageView(s_device, attachmentView, nullptr);
|
||||
}
|
||||
}
|
||||
for (const auto& [_, sampledView] : alternateSampledViews) {
|
||||
if (sampledView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, sampledView, nullptr);
|
||||
}
|
||||
}
|
||||
for (const auto& [_, storageImageView] : storageImageViews) {
|
||||
if (storageImageView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, storageImageView, nullptr);
|
||||
}
|
||||
}
|
||||
if (image != VK_NULL_HANDLE && allocation != nullptr) {
|
||||
vmaDestroyImage(s_allocator, image, allocation);
|
||||
}
|
||||
@@ -161,6 +251,8 @@ public:
|
||||
perMipViews.clear();
|
||||
perMipSampledViews.clear();
|
||||
attachmentViews.clear();
|
||||
alternateSampledViews.clear();
|
||||
storageImageViews.clear();
|
||||
image = VK_NULL_HANDLE;
|
||||
allocation = nullptr;
|
||||
layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
@@ -174,6 +266,9 @@ public:
|
||||
aspect = VK_IMAGE_ASPECT_NONE;
|
||||
viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageCreateFlags = 0;
|
||||
usageFlags = 0;
|
||||
storageUsageResolved = false;
|
||||
syncedTextureParamsVersion = 0;
|
||||
syncedContentVersion = 0;
|
||||
syncedMipLevelCount = 0;
|
||||
@@ -190,6 +285,10 @@ public:
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Drains every frame slot's deferred image/view releases. Only valid when
|
||||
// the caller has proven every queue submission complete; used by the
|
||||
// present-less frame-boundary drain.
|
||||
void CollectAllDeferredReleases();
|
||||
|
||||
TextureResource* SyncTextureAndGetDescriptor(
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
@@ -198,6 +297,9 @@ public:
|
||||
Uint32 baseArrayLayer, Uint32 layerCount,
|
||||
VkImageViewType viewType);
|
||||
VkImageView GetOrCreateSampledViewAtMipLevel(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel);
|
||||
VkImageView GetOrCreateSampledImageView(MG_State::GLState::ITextureObject& texture, VkFormat format);
|
||||
VkImageView GetOrCreateStorageImageView(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel,
|
||||
VkFormat format, Bool layered, Int32 layer);
|
||||
void UpdateTrackedImageLayout(MG_State::GLState::ITextureObject* texture, VkImageLayout newLayout);
|
||||
void UpdateTrackedImageLayoutAfterAttachmentWrite(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject* texture,
|
||||
@@ -205,8 +307,27 @@ public:
|
||||
VkImageLayout newLayout);
|
||||
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
// Records that this texture is bound to a GL image unit, so its image must carry
|
||||
// VK_IMAGE_USAGE_STORAGE_BIT. Must be called before NeedsStorageImagePreparation, and
|
||||
// therefore before the render pass is committed: an image that has to be upgraded is
|
||||
// recreated, which is illegal inside a render pass. Sticky for the texture's lifetime -
|
||||
// GL lets an image binding come and go, and re-creating the image every time it does
|
||||
// would cost far more than the compression it wins back.
|
||||
void MarkStorageImageTexture(MG_State::GLState::ITextureObject& texture);
|
||||
// True when this texture is marked but its live image predates the mark, i.e. the next sync
|
||||
// will recreate it with STORAGE usage and copy the old contents forward. Callers use this to
|
||||
// submit their pending recording first, so that copy cannot read pre-flush content.
|
||||
Bool NeedsStorageUsageUpgrade(MG_State::GLState::ITextureObject& texture) const;
|
||||
// Non-mutating probe for the per-draw storage-image fast path: true when preparing this
|
||||
// texture as a storage image may need work that is illegal inside a render pass (resource
|
||||
// creation, dirty-content upload, or a layout transition to GENERAL). Unknown state reports
|
||||
// true - a false positive merely ends the render pass, a false negative would skip a barrier.
|
||||
Bool NeedsStorageImagePreparation(MG_State::GLState::ITextureObject& texture) const;
|
||||
|
||||
static VkImageAspectFlags ResolveSampledImageViewAspectMask(VkImageAspectFlags imageAspect);
|
||||
static VkFormat ResolveSampledImageViewFormat(VkFormat imageFormat, SamplerNumericDomain numericDomain);
|
||||
static Bool AreSampledImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
|
||||
static Bool AreStorageImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
|
||||
|
||||
static Bool TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image, VkImageLayout& trackedLayout,
|
||||
VkImageLayout newLayout, VkPipelineStageFlags srcStageMask,
|
||||
@@ -255,7 +376,8 @@ private:
|
||||
VkImageViewType viewType, Uint32 baseMipLevel, Uint32 levelCount,
|
||||
Uint32 baseArrayLayer,
|
||||
Uint32 layerCount,
|
||||
const VkComponentMapping* components = nullptr) const;
|
||||
const VkComponentMapping* components = nullptr,
|
||||
VkImageUsageFlags viewUsage = 0) const;
|
||||
Bool UploadDirtyMipLevels(MG_State::GLState::TextureObjectMipmap &mipmapTexture,
|
||||
TextureUploadTarget uploadTarget,
|
||||
TextureResource &outResource);
|
||||
@@ -275,15 +397,20 @@ private:
|
||||
static TextureIdentity MakeTextureIdentity(MG_State::GLState::ITextureObject* texture);
|
||||
void EraseTrackedTexture(const TextureIdentity& identity);
|
||||
void PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture);
|
||||
SizeT PruneDeadTextures();
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
VkCommandPool m_commandPool = VK_NULL_HANDLE;
|
||||
VkQueue m_graphicsQueue = VK_NULL_HANDLE;
|
||||
Bool m_imageFormatListSupported = false;
|
||||
Uint32 m_currentFrameIndex = 0;
|
||||
|
||||
Uint8 m_gcCounter = 0;
|
||||
// Frame-boundary GC gate: counts BeginFrame calls, not draws, so texture churn
|
||||
// through non-draw paths (FBO clears, readbacks) still reaches the prune.
|
||||
Uint32 m_gcFrameCounter = 0;
|
||||
// Active only between BeginDrawSyncScope/EndDrawSyncScope; identities of
|
||||
// textures already fully synced in the current draw (small N -> flat scan).
|
||||
Bool m_drawSyncScopeActive = false;
|
||||
@@ -296,8 +423,13 @@ private:
|
||||
TextureResource* resource = nullptr;
|
||||
};
|
||||
Vector<DrawSyncedTexture> m_drawSyncedThisDraw;
|
||||
// Formats whose mutable-image probe failed on this device; their images are created
|
||||
// without MUTABLE_FORMAT_BIT so repeat syncs neither re-probe nor flag-mismatch.
|
||||
std::unordered_set<VkFormat> m_mutableFormatUnsupported;
|
||||
std::unordered_map<TextureIdentity, WeakPtr<MG_State::GLState::ITextureObject>, TextureIdentityHash> m_aliveObjects;
|
||||
std::unordered_map<TextureIdentity, TextureResource, TextureIdentityHash> m_textureResources;
|
||||
// Textures that have been bound to a GL image unit (see MarkStorageImageTexture).
|
||||
std::unordered_set<TextureIdentity, TextureIdentityHash> m_storageImageTextures;
|
||||
Vector<Vector<TextureResource>> m_deferredReleases;
|
||||
Vector<Vector<VkImageView>> m_deferredViewReleases;
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -51,6 +51,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 instanceCount = 1;
|
||||
Uint32 firstVertex = 0;
|
||||
Uint32 firstInstance = 0;
|
||||
// Indexed-draw metadata for bounding vertex-stream conversion. baseVertex is the
|
||||
// draw's base-vertex offset; indexRangeIsExactView is true only when the draw
|
||||
// fetches exactly the indices its IndexBufferView describes (direct DrawElements;
|
||||
// multi/indirect forms leave it false because the CPU cannot bound their ranges).
|
||||
Int32 baseVertex = 0;
|
||||
Bool indexRangeIsExactView = false;
|
||||
};
|
||||
|
||||
struct DrawIndexedCmdParam {
|
||||
@@ -108,7 +114,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
};
|
||||
|
||||
class VulkanRenderer : public IBufferCopyCommandProvider, public FrameContext::IRecordingObserver {
|
||||
class VulkanRenderer : public IBufferCopyCommandProvider,
|
||||
public FrameContext::IRecordingObserver,
|
||||
public VkRenderPassManager::IEvictionObserver,
|
||||
public ProgramFactory::IEvictionObserver {
|
||||
public:
|
||||
VulkanRenderer(NativeWindowType window, const VulkanRendererConfig& cfg = {});
|
||||
~VulkanRenderer();
|
||||
@@ -125,6 +134,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// recording, before any render pass.
|
||||
void OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer) override;
|
||||
|
||||
// VkRenderPassManager::IEvictionObserver: the render-pass aging sweep just
|
||||
// destroyed these VkRenderPasses; evict every graphics pipeline hashed on a
|
||||
// dying handle (they share its >1024-boundary idleness, so immediate
|
||||
// destruction is safe) and drop the last-pipeline memo if any went.
|
||||
void OnRenderPassesDestroyed(const Vector<VkRenderPass>& renderPasses) override;
|
||||
|
||||
// ProgramFactory::IEvictionObserver: an aged-out program entry was
|
||||
// destroyed; evict its compute pipeline and graphics pipelines (same
|
||||
// idleness guarantee - they are only bound through draws/dispatches that
|
||||
// stamp the program entry) and purge the descriptor-set cache entries
|
||||
// keyed by its now-recyclable VkDescriptorSetLayout handle.
|
||||
void OnProgramEvicted(ProgramFactory::HashType programHash,
|
||||
VkDescriptorSetLayout descriptorSetLayout) override;
|
||||
|
||||
Bool SetupDraw(FrameContext::FrameData& frame, GLenum mode, Flags<DrawSetupAspect> aspects,
|
||||
const DrawCmdParam& drawParams,
|
||||
const IndexBufferView* pIndexBufferView = nullptr);
|
||||
@@ -165,6 +188,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void GenerateMipmap(GLenum target);
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels);
|
||||
static SizeT GetReadbackTexelSize(VkFormat sourceFormat);
|
||||
static Bool ConvertReadbackPixels(const Uint8* sourcePixels, VkFormat sourceFormat,
|
||||
GLsizei width, GLsizei height, GLenum destinationFormat,
|
||||
GLenum destinationType, SizeT destinationRowStride,
|
||||
Uint8* destinationPixels);
|
||||
void GetTexImage(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels);
|
||||
void GetTextureImage(const SharedPtr<MG_State::GLState::ITextureObject>& texture,
|
||||
TextureUploadTarget uploadTarget, GLint level, GLenum format, GLenum type,
|
||||
@@ -246,7 +274,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 GetTimerQueryTimestampNs(const VkTimerQueryManager::TimestampRecord& record) const;
|
||||
|
||||
void RequestSwapchainResize(Uint32 width, Uint32 height);
|
||||
void RecreateSwapchain();
|
||||
// Re-query the surface and report whether the live swapchain no longer matches it
|
||||
// (size or orientation). This - not a VK_SUBOPTIMAL_KHR result - is what decides a
|
||||
// rebuild, so a surface the driver merely considers suboptimal cannot thrash.
|
||||
Bool SwapchainIsOutOfDate();
|
||||
// Returns false when the surface is zero-area (minimized/hidden window):
|
||||
// no new swapchain is installed and presentation must stay suspended.
|
||||
Bool RecreateSwapchain();
|
||||
|
||||
private:
|
||||
struct BlitUniformData {
|
||||
@@ -332,11 +366,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkFence AcquirePooledSubmitFence();
|
||||
void DestroySubmitFencePool();
|
||||
Bool HasPendingRecordedWork() const;
|
||||
// Frame-boundary housekeeping for paths that never reach Present's
|
||||
// tail (present-less readback loops, suspended presentation, blocking
|
||||
// sync waits): runs the same per-frame drains Present performs, but
|
||||
// only when every queue submission has been observed complete AND no
|
||||
// recorded-but-unsubmitted commands exist - i.e. when CPU-GPU overlap
|
||||
// is provably already zero. Never blocks (non-blocking fence poll
|
||||
// only), so the presenting path's frames-in-flight pipelining is
|
||||
// untouched. Returns true when the drain ran.
|
||||
Bool TryDrainFrameTransients();
|
||||
|
||||
Vector<SubmitRecord> m_inFlightSubmits;
|
||||
Vector<VkFence> m_freeSubmitFences;
|
||||
Uint64 m_submitCounter = 0;
|
||||
Uint64 m_completedSubmitCounter = 0;
|
||||
// Drains since the last Present, gating the drain's frame-boundary-equivalent
|
||||
// work (arena rewind + cache aging): a presenting app's mid-frame
|
||||
// readbacks/waits must neither churn the transient caches nor accelerate the
|
||||
// aging clocks, while present-less loops still cross a boundary every few
|
||||
// iterations. Reset in Present.
|
||||
Uint32 m_drainsSinceLastPresent = 0;
|
||||
|
||||
NativeWindowType m_window = 0;
|
||||
void* m_platformDisplay = nullptr;
|
||||
@@ -344,12 +393,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void* m_platformCloseDisplay = nullptr;
|
||||
VulkanRendererConfig m_config;
|
||||
Bool m_swapchainResizeRequested = false;
|
||||
// Presentation is suspended while the window is zero-area (minimized): the
|
||||
// swapchain is unusable/out of date, so Present drops frames instead of
|
||||
// submitting on a signaled fence / presenting never-acquired images.
|
||||
Bool m_presentSuspended = false;
|
||||
|
||||
// Vulkan objects
|
||||
Bool m_validationLayersEnabled = false;
|
||||
Vector<VkExtensionProperties> m_extensions;
|
||||
VkInstance m_instance = VK_NULL_HANDLE;
|
||||
VkDebugUtilsMessengerEXT m_debugMessenger = VK_NULL_HANDLE;
|
||||
// Fallback reporting channel for drivers that ship the validation layers but
|
||||
// only expose the older VK_EXT_debug_report (Adreno 650 / Vulkan 1.1.128).
|
||||
VkDebugReportCallbackEXT m_debugReportCallback = VK_NULL_HANDLE;
|
||||
PhysicalDevice m_physicalDevice;
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
@@ -365,6 +421,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool m_samplerAnisotropyFeatureEnabled = false;
|
||||
Bool m_shaderDrawParametersExtensionEnabled = false;
|
||||
Bool m_shaderDrawParametersFeatureEnabled = false;
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// fillModeNonSolid gates VK_POLYGON_MODE_LINE/_POINT (glPolygonMode); independentBlend gates
|
||||
// per-draw-buffer color write masks (glColorMaski). Both are cached at device creation and
|
||||
// drive a runtime fallback when the device lacks them.
|
||||
@@ -437,13 +494,70 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Per-draw scratch buffers (clear keeps capacity) — these paths run for every
|
||||
// draw call and must not allocate.
|
||||
Vector<MG_State::GLState::ITextureObject*> m_sampledTexturesScratch;
|
||||
Vector<MG_State::GLState::ITextureObject*> m_storageImageTexturesScratch;
|
||||
Vector<VkBuffer> m_vertexBuffersScratch;
|
||||
Vector<VkDeviceSize> m_vertexOffsetsScratch;
|
||||
Vector<VkVertexInputAttributeDescription> m_patchedAttributesScratch;
|
||||
Vector<Float> m_vertexConversionScratch;
|
||||
Vector<Uint8> m_vertexRepackScratch;
|
||||
|
||||
struct ConvertedVertexStreamKey {
|
||||
const MG_State::GLState::BufferObject* buffer = nullptr;
|
||||
Uint64 changeSerial = 0;
|
||||
SizeT baseOffset = 0;
|
||||
Uint32 sourceStride = 0;
|
||||
DataType type = DataType::Float32;
|
||||
Int size = 0;
|
||||
Bool normalized = false;
|
||||
Bool isInteger = false;
|
||||
VertexInputStateFactory::VertexStreamConversion conversion =
|
||||
VertexInputStateFactory::VertexStreamConversion::None;
|
||||
|
||||
Bool operator==(const ConvertedVertexStreamKey& other) const {
|
||||
return buffer == other.buffer && changeSerial == other.changeSerial &&
|
||||
baseOffset == other.baseOffset && sourceStride == other.sourceStride &&
|
||||
type == other.type && size == other.size && normalized == other.normalized &&
|
||||
isInteger == other.isInteger && conversion == other.conversion;
|
||||
}
|
||||
};
|
||||
|
||||
struct ConvertedVertexStreamKeyHash {
|
||||
SizeT operator()(const ConvertedVertexStreamKey& key) const {
|
||||
SizeT hash = std::hash<const void*>{}(key.buffer);
|
||||
auto combine = [&hash](SizeT value) {
|
||||
hash ^= value + static_cast<SizeT>(0x9e3779b97f4a7c15ull) + (hash << 6) + (hash >> 2);
|
||||
};
|
||||
combine(std::hash<Uint64>{}(key.changeSerial));
|
||||
combine(std::hash<SizeT>{}(key.baseOffset));
|
||||
combine(std::hash<Uint32>{}(key.sourceStride));
|
||||
combine(std::hash<Uint32>{}(static_cast<Uint32>(key.type)));
|
||||
combine(std::hash<Int>{}(key.size));
|
||||
combine(std::hash<Bool>{}(key.normalized));
|
||||
combine(std::hash<Bool>{}(key.isInteger));
|
||||
combine(std::hash<Uint32>{}(static_cast<Uint32>(key.conversion)));
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
struct ConvertedVertexStream {
|
||||
BufferSlice slice;
|
||||
// Number of source elements the cached slice covers. A draw needing a prefix of
|
||||
// this range reuses the slice (converted streams are tightly packed); a draw
|
||||
// needing more reconverts and replaces the entry, so per (buffer, layout) a
|
||||
// frame converts at most the largest range any draw asked for.
|
||||
SizeT elementCount = 0;
|
||||
// Pins the source buffer for the frame so its heap address cannot be reused by
|
||||
// a new BufferObject while this pointer-keyed entry is alive.
|
||||
SharedPtr<const MG_State::GLState::BufferObject> sourcePin;
|
||||
};
|
||||
UnorderedMap<ConvertedVertexStreamKey, ConvertedVertexStream, ConvertedVertexStreamKeyHash>
|
||||
m_convertedVertexStreams;
|
||||
|
||||
void CreateInstance();
|
||||
VkResult SetupDebugMessenger();
|
||||
VkResult DestroyDebugMessenger();
|
||||
VkResult SetupDebugReportCallback();
|
||||
void DestroyDebugReportCallback();
|
||||
VkDebugUtilsMessengerCreateInfoEXT PopulateDebugMessengerCreateInfo();
|
||||
void CreateSurface();
|
||||
void PickPhysicalDevice();
|
||||
@@ -462,10 +576,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const RenderPassEntry& renderPassEntry);
|
||||
VkPipeline GetOrCreateComputePipeline(const ProgramFactory::VkProgramObject& programObj);
|
||||
void DestroyComputePipelines();
|
||||
// Takes the frame rather than a command buffer: a first-time storage-usage upgrade has to
|
||||
// flush the pending recording (see the body), which retires the current command buffer.
|
||||
Bool PrepareStorageImageTextures(
|
||||
FrameContext::FrameData& frame,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
|
||||
Bool UploadAndBindVertexBuffers(VkCommandBuffer commandBuffer, const MG_State::GLState::VertexArrayObject& vao,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
const DrawCmdParam& drawParams);
|
||||
const DrawCmdParam& drawParams,
|
||||
const IndexBufferView* pIndexBufferView);
|
||||
Bool UploadAndBindIndexBuffer(FrameContext::FrameData& frame,
|
||||
const MG_State::GLState::VertexArrayObject& vao,
|
||||
const IndexBufferView* pIndexBufferView = nullptr);
|
||||
@@ -483,6 +604,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLenum filter);
|
||||
Bool MaterializePendingClearForTexture(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
Bool MaterializePendingClearForRenderbuffer(
|
||||
VkCommandBuffer commandBuffer,
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
VkPipeline GetOrCreateBlitPipeline(const RenderPassEntry& renderPassEntry);
|
||||
Bool GenerateDepthMipmapWithShader(FrameContext::FrameData& frame,
|
||||
MG_State::GLState::ITextureObject& texture,
|
||||
@@ -517,6 +641,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const PhysicalDevice& compareWithDevice,
|
||||
PhysicalDevice& outBetterDevice);
|
||||
static constexpr const char* s_validationLayerNames[] = {"VK_LAYER_KHRONOS_validation"};
|
||||
// VK_KHR_image_format_list: lets MUTABLE_FORMAT images declare their exact view-format
|
||||
// set so the driver can keep bandwidth compression (see CreateLogicalDeviceAndQueues).
|
||||
Bool m_imageFormatListExtensionEnabled = false;
|
||||
|
||||
static constexpr const char* s_deviceExtensionNames[] = {VK_KHR_SWAPCHAIN_EXTENSION_NAME};
|
||||
static Bool CheckValidationLayerSupport();
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
GLint Samples = 0;
|
||||
GLint Profile = kCGLOGLPVersion_3_2_Core;
|
||||
GLint RendererId = 0x4d474c;
|
||||
GLint DisplayMask = 0;
|
||||
};
|
||||
|
||||
struct ContextObject {
|
||||
@@ -134,6 +135,9 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
case kCGLPFARendererID:
|
||||
pixelFormat.RendererId = value;
|
||||
break;
|
||||
case kCGLPFADisplayMask:
|
||||
pixelFormat.DisplayMask = value;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -343,6 +347,9 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
case kCGLPFARendererID:
|
||||
*value = pixelFormat->RendererId;
|
||||
return kCGLNoError;
|
||||
case kCGLPFADisplayMask:
|
||||
*value = pixelFormat->DisplayMask;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAOpenGLProfile:
|
||||
*value = pixelFormat->Profile;
|
||||
return kCGLNoError;
|
||||
@@ -481,6 +488,32 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
return it == currentContexts.end() ? nullptr : it->second;
|
||||
}
|
||||
|
||||
CGLError SetVirtualScreen(CGLContextObj ctx, GLint screen) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (screen != 0) {
|
||||
return kCGLBadValue;
|
||||
}
|
||||
object->VirtualScreen = screen;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError GetVirtualScreen(CGLContextObj ctx, GLint* screen) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!screen) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
*screen = object->VirtualScreen;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError SetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
|
||||
@@ -32,6 +32,8 @@ namespace MobileGL::MG_Impl::CGLImpl {
|
||||
|
||||
CGLError SetCurrentContext(CGLContextObj ctx);
|
||||
CGLContextObj GetCurrentContext();
|
||||
CGLError SetVirtualScreen(CGLContextObj ctx, GLint screen);
|
||||
CGLError GetVirtualScreen(CGLContextObj ctx, GLint* screen);
|
||||
CGLError SetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params);
|
||||
CGLError GetParameter(CGLContextObj ctx, CGLContextParameter pname, GLint* params);
|
||||
CGLError UpdateContext(CGLContextObj ctx);
|
||||
|
||||
@@ -71,6 +71,14 @@ MOBILEGL_CGL_API CGLContextObj CGLGetCurrentContext(void) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetCurrentContext();
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetVirtualScreen(CGLContextObj ctx, GLint screen) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetVirtualScreen(ctx, screen);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLGetVirtualScreen(CGLContextObj ctx, GLint* screen) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetVirtualScreen(ctx, screen);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetParameter(ctx, pname, params);
|
||||
}
|
||||
|
||||
@@ -10,8 +10,12 @@
|
||||
|
||||
#if defined(__APPLE__)
|
||||
|
||||
#include "MG_Impl/CGLImpl/CGLImpl.h"
|
||||
#include "MG_Impl/GetProcAddress.h"
|
||||
|
||||
#include <CoreGraphics/CoreGraphics.h>
|
||||
#include <CoreVideo/CVDisplayLink.h>
|
||||
#include <cstdint>
|
||||
#include <dlfcn.h>
|
||||
|
||||
namespace {
|
||||
@@ -47,10 +51,52 @@ namespace {
|
||||
return dlsym(handle, symbol);
|
||||
}
|
||||
|
||||
CGDirectDisplayID DisplayForMask(GLint displayMask) {
|
||||
constexpr std::uint32_t MaxDisplays = sizeof(CGOpenGLDisplayMask) * 8;
|
||||
CGDirectDisplayID displays[MaxDisplays] = {};
|
||||
std::uint32_t displayCount = 0;
|
||||
if (displayMask != 0 &&
|
||||
CGGetActiveDisplayList(MaxDisplays, displays, &displayCount) == kCGErrorSuccess) {
|
||||
const auto mask = static_cast<CGOpenGLDisplayMask>(displayMask);
|
||||
for (std::uint32_t i = 0; i < displayCount; ++i) {
|
||||
if ((CGDisplayIDToOpenGLDisplayMask(displays[i]) & mask) != 0) {
|
||||
return displays[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
return CGMainDisplayID();
|
||||
}
|
||||
|
||||
#pragma clang diagnostic push
|
||||
#pragma clang diagnostic ignored "-Wdeprecated-declarations"
|
||||
CVReturn MobileGLCVDisplayLinkSetCurrentCGDisplayFromOpenGLContext(
|
||||
CVDisplayLinkRef displayLink,
|
||||
CGLContextObj context,
|
||||
CGLPixelFormatObj pixelFormat) {
|
||||
GLint virtualScreen = 0;
|
||||
if (MobileGL::MG_Impl::CGLImpl::GetVirtualScreen(context, &virtualScreen) == kCGLNoError) {
|
||||
GLint displayMask = 0;
|
||||
if (!displayLink ||
|
||||
MobileGL::MG_Impl::CGLImpl::DescribePixelFormat(
|
||||
pixelFormat, virtualScreen, kCGLPFADisplayMask, &displayMask) != kCGLNoError) {
|
||||
return kCVReturnInvalidArgument;
|
||||
}
|
||||
return CVDisplayLinkSetCurrentCGDisplay(displayLink, DisplayForMask(displayMask));
|
||||
}
|
||||
|
||||
using OriginalFunction = CVReturn (*)(CVDisplayLinkRef, CGLContextObj, CGLPixelFormatObj);
|
||||
static const auto original = reinterpret_cast<OriginalFunction>(
|
||||
dlsym(RTLD_NEXT, "CVDisplayLinkSetCurrentCGDisplayFromOpenGLContext"));
|
||||
return original ? original(displayLink, context, pixelFormat) : kCVReturnError;
|
||||
}
|
||||
|
||||
__attribute__((used)) static const DyldInterposeEntry kMobileGLDyldInterpose[]
|
||||
__attribute__((section("__DATA,__interpose"))) = {
|
||||
{reinterpret_cast<const void*>(MobileGLDlsym), reinterpret_cast<const void*>(dlsym)},
|
||||
{reinterpret_cast<const void*>(MobileGLCVDisplayLinkSetCurrentCGDisplayFromOpenGLContext),
|
||||
reinterpret_cast<const void*>(CVDisplayLinkSetCurrentCGDisplayFromOpenGLContext)},
|
||||
};
|
||||
#pragma clang diagnostic pop
|
||||
} // namespace
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
# Public CGL entry points.
|
||||
_CGL*
|
||||
|
||||
# Public EGL entry points.
|
||||
_egl*
|
||||
|
||||
# Public OpenGL and GLX entry points. OpenGL function names always use an
|
||||
# uppercase letter or digit after the "gl" prefix; excluding lowercase here
|
||||
# deliberately prevents glslang_* from matching this pattern.
|
||||
_gl[A-Z0-9]*
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "EGLImpl.h"
|
||||
#include "../GetProcAddress.h"
|
||||
#include <Init.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <mutex>
|
||||
@@ -25,6 +26,17 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return MG_State::pEGLContext.get();
|
||||
}
|
||||
|
||||
// Entry points that can legitimately be an application's FIRST EGL
|
||||
// call (display/proc-address/string queries) lazily bring MobileGL
|
||||
// up here, so the library needs no static constructor and can
|
||||
// re-initialize after the last eglTerminate tore everything down.
|
||||
// Teardown-ish entry points keep using GetState() and fail benignly
|
||||
// when MobileGL is not initialized.
|
||||
EGLStateContext* GetStateEnsureInitialized() {
|
||||
MobileGL::EnsureInitialized();
|
||||
return GetState();
|
||||
}
|
||||
|
||||
MG_Backend::BackendObject* GetBackendObject(EGLStateContext* state) {
|
||||
auto* backendObject = MG_Backend::pActiveBackendObject.get();
|
||||
if (!backendObject && state) {
|
||||
@@ -49,6 +61,8 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return MG_Backend::WindowBackend::Android;
|
||||
#elif defined(__APPLE__)
|
||||
return MG_Backend::WindowBackend::MetalLayer;
|
||||
#elif defined(_WIN32)
|
||||
return MG_Backend::WindowBackend::Win32;
|
||||
#elif defined(__linux__)
|
||||
return MG_Backend::WindowBackend::X11;
|
||||
#else
|
||||
@@ -187,7 +201,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean Initialize(EGLDisplay dpy, EGLint* major, EGLint* minor) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -208,7 +222,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLDisplay GetDisplay(NativeDisplayType display) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
@@ -313,6 +327,14 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
backendObject->ReleaseEGLResources();
|
||||
}
|
||||
// The last initialized display is gone and nothing is current on any
|
||||
// thread: tear the whole library down deterministically inside the
|
||||
// EGL lifecycle (backend, GL/EGL state, glslang). A later EGL call
|
||||
// re-initializes lazily via GetStateEnsureInitialized(); process exit
|
||||
// then has nothing left to destroy.
|
||||
if (!state->HasAnyInitializedDisplay() && !state->HasAnyCurrentContext()) {
|
||||
MobileGL::Destroy();
|
||||
}
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
@@ -345,7 +367,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean BindAPI(EGLenum api) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -378,7 +400,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
char const* QueryString(EGLDisplay display, EGLint name) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return nullptr;
|
||||
}
|
||||
@@ -641,7 +663,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
EGLDisplay GetPlatformDisplay(EGLenum platform, void* native_display, const EGLAttrib* attrib_list) {
|
||||
(void)attrib_list;
|
||||
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
@@ -737,6 +759,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!name) {
|
||||
return nullptr;
|
||||
}
|
||||
MobileGL::EnsureInitialized();
|
||||
|
||||
MGLOG_D("eglGetProcAddress(%s)", name);
|
||||
void* proc = MG_Impl::GetProcAddress(name);
|
||||
|
||||
@@ -295,7 +295,13 @@ DECLARE_GL_FUNCTION_HEAD(void, VertexAttribDivisor, GLuint index, GLuint divisor
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BindTransformFeedback, GLenum target, GLuint id) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BindTransformFeedback, target, id)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DeleteTransformFeedbacks, GLsizei n, const GLuint* ids) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DeleteTransformFeedbacks, n, ids)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GenTransformFeedbacks, GLsizei n, GLuint* ids) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GenTransformFeedbacks, n, ids)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLboolean, IsTransformFeedback, GLuint id) DECLARE_GL_FUNCTION_STUB_END(GLboolean, IsTransformFeedback, id)
|
||||
// Transform feedback objects are not implemented, so no name is ever a live object. The shared
|
||||
// stub returns (type)1, telling a probing caller that every id it invents already exists; GL_FALSE
|
||||
// is both truthful and what the spec requires for a name that was never generated.
|
||||
MOBILEGL_GL_API GLboolean glIsTransformFeedback(GLuint id) {
|
||||
MGLOG_W("Stub function: %s(...)", __FUNCTION__);
|
||||
return GL_FALSE;
|
||||
}
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PauseTransformFeedback) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PauseTransformFeedback)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ResumeTransformFeedback) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ResumeTransformFeedback)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetProgramBinary, GLuint program, GLsizei bufSize, GLsizei* length, GLenum* binaryFormat, void* binary) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetProgramBinary, program, bufSize, length, binaryFormat, binary)
|
||||
@@ -418,7 +424,7 @@ DECLARE_GL_FUNCTION_HEAD(void, DrawRangeElementsBaseVertex, GLenum mode, GLuint
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawElementsInstancedBaseVertex, GLenum mode, GLsizei count, GLenum type, const void* indices, GLsizei instancecount, GLint basevertex) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawElementsInstancedBaseVertex, mode, count, type, indices, instancecount, basevertex)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, FramebufferTexture, GLenum target, GLenum attachment, GLuint texture, GLint level) DECLARE_GL_FUNCTION_END_NO_RETURN(void, FramebufferTexture, target, attachment, texture, level)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PrimitiveBoundingBox, GLfloat minX, GLfloat minY, GLfloat minZ, GLfloat minW, GLfloat maxX, GLfloat maxY, GLfloat maxZ, GLfloat maxW) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PrimitiveBoundingBox, minX, minY, minZ, minW, maxX, maxY, maxZ, maxW)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLenum, GetGraphicsResetStatus) DECLARE_GL_FUNCTION_STUB_END(GLenum, GetGraphicsResetStatus)
|
||||
DECLARE_GL_FUNCTION_HEAD(GLenum, GetGraphicsResetStatus) DECLARE_GL_FUNCTION_END(GLenum, GetGraphicsResetStatus)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ReadnPixels, GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, GLsizei bufSize, void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ReadnPixels, x, y, width, height, format, type, bufSize, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformfv, GLuint program, GLint location, GLsizei bufSize, GLfloat* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformfv, program, location, bufSize, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformiv, GLuint program, GLint location, GLsizei bufSize, GLint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformiv, program, location, bufSize, params)
|
||||
@@ -998,8 +1004,8 @@ DECLARE_GL_FUNCTION_HEAD(void, ShaderStorageBlockBinding, GLuint program, GLuint
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TextureView, GLuint texture, GLenum target, GLuint origtexture, GLenum internalformat, GLuint minlevel, GLuint numlevels, GLuint minlayer, GLuint numlayers) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TextureView, texture, target, origtexture, internalformat, minlevel, numlevels, minlayer, numlayers)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribLFormat, GLuint attribindex, GLint size, GLenum type, GLuint relativeoffset) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, VertexAttribLFormat, attribindex, size, type, relativeoffset)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BufferStorage, GLenum target, GLsizeiptr size, const void* data, GLbitfield flags) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BufferStorage, target, size, data, flags)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearTexImage, GLuint texture, GLint level, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearTexImage, texture, level, format, type, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearTexSubImage, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearTexSubImage, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearTexImage, GLuint texture, GLint level, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearTexImage, texture, level, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearTexSubImage, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearTexSubImage, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindBuffersBase, GLenum target, GLuint first, GLsizei count, const GLuint* buffers) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindBuffersBase, target, first, count, buffers)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindBuffersRange, GLenum target, GLuint first, GLsizei count, const GLuint* buffers, const GLintptr* offsets, const GLsizeiptr* sizes) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindBuffersRange, target, first, count, buffers, offsets, sizes)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BindTextures, GLuint first, GLsizei count, const GLuint* textures) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BindTextures, first, count, textures)
|
||||
@@ -1063,7 +1069,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage1D, GLuint texture,
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLsizei width, GLsizei height, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage2D, texture, level, xoffset, yoffset, width, height, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage3D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage3D, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CopyTextureSubImage1D, GLuint texture, GLint level, GLint xoffset, GLint x, GLint y, GLsizei width) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CopyTextureSubImage1D, texture, level, xoffset, x, y, width)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CopyTextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CopyTextureSubImage2D, texture, level, xoffset, yoffset, x, y, width, height)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CopyTextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CopyTextureSubImage2D, texture, level, xoffset, yoffset, x, y, width, height)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CopyTextureSubImage3D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLint x, GLint y, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CopyTextureSubImage3D, texture, level, xoffset, yoffset, zoffset, x, y, width, height)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureParameterf, GLuint texture, GLenum pname, GLfloat param) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureParameterf, texture, pname, param)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureParameterfv, GLuint texture, GLenum pname, const GLfloat* param) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureParameterfv, texture, pname, param)
|
||||
@@ -2583,7 +2589,10 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, TransformFeedbackStreamAttribsNV, GLsizei co
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BindTransformFeedbackNV, GLenum target, GLuint id) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BindTransformFeedbackNV, target, id)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DeleteTransformFeedbacksNV, GLsizei n, const GLuint* ids) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DeleteTransformFeedbacksNV, n, ids)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GenTransformFeedbacksNV, GLsizei n, GLuint* ids) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GenTransformFeedbacksNV, n, ids)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLboolean, IsTransformFeedbackNV, GLuint id) DECLARE_GL_FUNCTION_STUB_END(GLboolean, IsTransformFeedbackNV, id)
|
||||
MOBILEGL_GL_API GLboolean glIsTransformFeedbackNV(GLuint id) {
|
||||
MGLOG_W("Stub function: %s(...)", __FUNCTION__);
|
||||
return GL_FALSE;
|
||||
}
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PauseTransformFeedbackNV, void) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PauseTransformFeedbackNV, )
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ResumeTransformFeedbackNV, void) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ResumeTransformFeedbackNV, )
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DrawTransformFeedbackNV, GLenum mode, GLuint id) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DrawTransformFeedbackNV, mode, id)
|
||||
|
||||
@@ -2144,6 +2144,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
namespace FramebufferImpl {
|
||||
UniquePtr<DefaultFramebufferInfo> pDefaultFramebufferInfo;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<DefaultFramebufferInfo>& pDefaultFramebufferInfo = *new UniquePtr<DefaultFramebufferInfo>();
|
||||
} // namespace FramebufferImpl
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -78,6 +78,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
SharedPtr<MG_State::GLState::ITextureObject> stencilAttachment;
|
||||
};
|
||||
|
||||
extern UniquePtr<DefaultFramebufferInfo> pDefaultFramebufferInfo;
|
||||
extern UniquePtr<DefaultFramebufferInfo>& pDefaultFramebufferInfo;
|
||||
} // namespace FramebufferImpl
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -1175,10 +1175,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxFragmentInputComponents;
|
||||
return;
|
||||
case GL_MAX_FRAGMENT_IMAGE_UNIFORMS:
|
||||
// TODO: Track per-stage image uniform limits separately instead of reusing the compute/backend stage cap.
|
||||
*params = MG_Backend::pActiveBackendObject
|
||||
? MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxComputeImageUniforms
|
||||
: MG_Backend::DynamicBackendParameters{}.MaxComputeImageUniforms;
|
||||
? MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxFragmentImageUniforms
|
||||
: MG_Backend::DynamicBackendParameters{}.MaxFragmentImageUniforms;
|
||||
return;
|
||||
case GL_MAX_FRAGMENT_UNIFORM_COMPONENTS:
|
||||
*params = kFrontendMaxFragmentUniformComponents;
|
||||
@@ -1208,7 +1207,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxGeometryTextureImageUnits;
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_IMAGE_UNIFORMS:
|
||||
*params = 0;
|
||||
*params = MG_Backend::pActiveBackendObject
|
||||
? MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxGeometryImageUniforms
|
||||
: MG_Backend::DynamicBackendParameters{}.MaxGeometryImageUniforms;
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_TOTAL_OUTPUT_COMPONENTS:
|
||||
*params = kFrontendMaxGeometryTotalOutputComponents;
|
||||
@@ -1277,7 +1278,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxVertexAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_VERTEX_IMAGE_UNIFORMS:
|
||||
*params = 0;
|
||||
*params = MG_Backend::pActiveBackendObject
|
||||
? MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxVertexImageUniforms
|
||||
: MG_Backend::DynamicBackendParameters{}.MaxVertexImageUniforms;
|
||||
return;
|
||||
case GL_MAX_VERTEX_SHADER_STORAGE_BLOCKS:
|
||||
*params = 16; // TODO
|
||||
@@ -1993,4 +1996,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
return MG_Util::ConvertErrorCodeToGLEnum(error->get()->code);
|
||||
}
|
||||
|
||||
GLenum GetGraphicsResetStatus() {
|
||||
// MobileGL does not implement robustness reset notification, so report GL_NO_ERROR
|
||||
// ("no reset detected"). Returning the generic stub's (GLenum)1 makes dEQP read a lost
|
||||
// device after every case (gl3cTestPackages.cpp:121) and, under the default
|
||||
// --deqp-terminate-on-device-lost=enable, tear the whole CTS run down.
|
||||
return GL_NO_ERROR;
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -21,4 +21,5 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data);
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
GLenum GetError();
|
||||
GLenum GetGraphicsResetStatus();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -133,4 +133,33 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
values[0] = value;
|
||||
}
|
||||
}
|
||||
|
||||
void DestroyAllSyncObjects() {
|
||||
// Detach the registry under the lock, release outside it. Entries the app
|
||||
// already deleted were erased by DeleteSync, so nothing here double-frees;
|
||||
// a DeleteSync racing this sweep finds an empty registry and returns. A
|
||||
// thread still blocked inside ClientWaitSync/GetSynciv during teardown
|
||||
// holds a raw SyncObject* these deletes invalidate - the same undefined
|
||||
// race an app-driven DeleteSync already has.
|
||||
UnorderedMap<GLsync, SyncObject*> orphans;
|
||||
{
|
||||
const std::lock_guard<std::mutex> lock(g_syncObjectsMutex);
|
||||
orphans.swap(g_liveSyncObjects);
|
||||
}
|
||||
if (orphans.empty()) {
|
||||
return;
|
||||
}
|
||||
// Both backends' DeleteSync only free the heap wrapper once their GL
|
||||
// context/renderer is gone (generation/current-thread guards), so this is
|
||||
// safe after the backend has released its EGL resources - but not after
|
||||
// the function table itself is cleared.
|
||||
const auto backendDeleteSync = MG_Backend::gBackendFunctionsTable.GL.DeleteSync;
|
||||
for (const auto& [_, syncObject] : orphans) {
|
||||
if (backendDeleteSync && syncObject->backendHandle) {
|
||||
backendDeleteSync(syncObject->backendHandle);
|
||||
}
|
||||
delete syncObject;
|
||||
}
|
||||
MGLOG_D("DestroyAllSyncObjects: reclaimed %zu sync object(s) the app left undeleted", orphans.size());
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -16,4 +16,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void WaitSync(GLsync sync, GLbitfield flags, GLuint64 timeout);
|
||||
void DeleteSync(GLsync sync);
|
||||
void GetSynciv(GLsync sync, GLenum pname, GLsizei bufSize, GLsizei* length, GLint* values);
|
||||
// Destroys every still-registered sync object exactly as DeleteSync would.
|
||||
// GL requires syncs to die with their context; called only from full library
|
||||
// teardown (DestroyImpl), where no context survives on any thread, so the
|
||||
// process-global registry can be drained wholesale. Must run while the
|
||||
// backend function table is still populated: each backend handle has to be
|
||||
// released by the backend that created it, never by a later re-initialized
|
||||
// one.
|
||||
void DestroyAllSyncObjects();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -350,6 +350,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
texture.AllocateStorage(uploadTarget, level, {levelTexelSize, levelByteSize});
|
||||
texture.MarkStorageDirty(uploadTarget, level, false);
|
||||
}
|
||||
// glGenerateMipmap defines exactly levels 0..requiredLevelCount-1. AllocateStorage only
|
||||
// grows, so a previously longer chain (a bigger base image before respecification) would
|
||||
// otherwise keep a tail of stale levels here and read as incomplete.
|
||||
texture.TruncateMipmapLevels(uploadTarget, requiredLevelCount);
|
||||
// Mip generation grows/regenerates the level set on the GPU without marking any CPU
|
||||
// level dirty (MarkStorageDirty(...,false) above). Bump the content version so the
|
||||
// backend re-syncs: a cached sampled VkImageView built for the pre-generate level
|
||||
@@ -429,8 +433,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
const Int maxSamples = GetMaxSupportedTextureSamples(textureInternalFormat);
|
||||
if (samples > maxSamples) {
|
||||
// GL specifies INVALID_OPERATION - not INVALID_VALUE - when the sample count
|
||||
// exceeds what the format supports, and the native Adreno driver agrees.
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", caller,
|
||||
std::format("Sample count {} exceeds the supported maximum {} for this texture format.",
|
||||
@@ -454,8 +460,56 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
textureObject->SetSamples(samples);
|
||||
textureObject->SetFixedSampleLocations(fixedsamplelocations == GL_TRUE);
|
||||
textureMipmapObject->AllocateStorage(textureUploadTarget, 0, {{width, height, depth}, 0});
|
||||
// Multisample textures are single-level by definition, so a name that previously held a
|
||||
// mip chain must not keep its tail now that AllocateStorage only grows.
|
||||
textureMipmapObject->TruncateMipmapLevels(textureUploadTarget, 1);
|
||||
textureMipmapObject->MarkStorageDirty(textureUploadTarget, 0, false);
|
||||
}
|
||||
|
||||
// Redefining level 0 of a texture that already had a base image drops the rest of the chain,
|
||||
// which is exactly what AllocateLevel used to do implicitly for every level. Keeping that
|
||||
// behaviour for level 0 - and only for level 0 - is what makes the grow-only change safe:
|
||||
// any level-0 respecification leaves the chain in precisely the state it would have had
|
||||
// before, while an upload to level N no longer destroys the levels beneath it.
|
||||
//
|
||||
// Why it has to be *every* level-0 respecification and not just a size change: Minecraft's
|
||||
// Mipmap Levels setting rebuilds the block atlas at the SAME dimensions with a different
|
||||
// level count. A size-only test would leave the old tail in place, and because Mojang
|
||||
// terminates its chains with a 0x0 level the result is the zero-then-nonzero pattern that
|
||||
// IsComplete() rejects (TextureObject.cpp) - whereupon DirectGLES skips syncing the texture
|
||||
// entirely (Managers.cpp) and the atlas samples black.
|
||||
//
|
||||
// The "already has a base image" test is what lets the fix work at all: a level that was
|
||||
// never written reads back as {0,0,0}, so building a chain top-down - upload level N first,
|
||||
// then level 0 - must not discard the levels just uploaded. That ordering is what
|
||||
// KHR-GL33.texture_repeat_mode does.
|
||||
// Scoped to the respecified upload target only, which is what AllocateLevel already did.
|
||||
// Cube maps keep six independent chains while reporting a single level count (face +X), so
|
||||
// respecifying a face other than +X can leave the count longer than that face - but that
|
||||
// asymmetry predates this change and widening the truncation to all six faces would destroy
|
||||
// mip data for faces the application never touched. Left alone deliberately.
|
||||
void DiscardMipmapChainOnBaseRespecification(MG_State::GLState::TextureObjectMipmap* texture,
|
||||
TextureUploadTarget uploadTarget, Uint level) {
|
||||
if (level != 0) return;
|
||||
|
||||
const IntVec3 existingBaseSize = texture->GetMipmapTexelSize(uploadTarget, 0);
|
||||
const Bool hasExistingBaseImage =
|
||||
existingBaseSize.x() > 0 && existingBaseSize.y() > 0 && existingBaseSize.z() > 0;
|
||||
if (!hasExistingBaseImage) return;
|
||||
|
||||
texture->TruncateMipmapLevels(uploadTarget, 1);
|
||||
}
|
||||
|
||||
// Compressed texture upload is not implemented yet. GL_NUM_COMPRESSED_TEXTURE_FORMATS
|
||||
// reports 0, so every compressed internalformat is by definition unsupported and
|
||||
// GL_INVALID_ENUM is the specified error - unlike THROW_UNIMPL_EXCEPTION, which unwinds
|
||||
// a C++ exception through the C GL ABI and takes the process down.
|
||||
void RecordUnsupportedCompressedFormat(const char* caller) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller,
|
||||
"Compressed texture formats are not supported."));
|
||||
}
|
||||
} // namespace
|
||||
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& GetTextureObjectByName(GLuint texture, const char* caller) {
|
||||
@@ -470,6 +524,223 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return textureObject;
|
||||
}
|
||||
|
||||
namespace {
|
||||
void RecordClearTextureError(const char* caller, ErrorCode code, const String& message) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
code, MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller, message));
|
||||
}
|
||||
|
||||
SharedPtr<MG_State::GLState::TextureObjectMipmap> GetClearTextureObject(GLuint texture, GLint level,
|
||||
const char* caller) {
|
||||
if (texture == 0) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidOperation,
|
||||
"Clear texture operations require a non-zero texture name.");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto textureObject = GetTextureObjectByName(texture, caller);
|
||||
if (!textureObject) return nullptr;
|
||||
if (textureObject->GetStorageType() != TextureStorageType::Mipmap) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidOperation,
|
||||
"Buffer textures cannot be cleared with glClearTexImage.");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto mipmapTexture = std::static_pointer_cast<MG_State::GLState::TextureObjectMipmap>(textureObject);
|
||||
if (level < 0) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidValue,
|
||||
std::format("Texture level {} is negative.", level));
|
||||
return nullptr;
|
||||
}
|
||||
// ARB_clear_texture: clearing an image that was never defined by TexImage*/
|
||||
// TexStorage* is INVALID_OPERATION, not INVALID_VALUE.
|
||||
if (static_cast<Uint>(level) >= mipmapTexture->GetMipmapLevelCount()) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidOperation,
|
||||
std::format("Texture level {} is not defined.", level));
|
||||
return nullptr;
|
||||
}
|
||||
return mipmapTexture;
|
||||
}
|
||||
|
||||
Bool BuildClearPixel(const SharedPtr<MG_State::GLState::TextureObjectMipmap>& textureObject,
|
||||
GLenum format, GLenum type, const void* data, Vector<Uint8>& clearPixel) {
|
||||
const TextureInputFormat inputFormat = MG_Util::ConvertGLEnumToTextureInputFormat(format);
|
||||
const TexturePixelDataType inputType = MG_Util::ConvertGLEnumToTexturePixelDataType(type);
|
||||
if (!TextureImpl::ValidateTextureInputFormat(inputFormat) ||
|
||||
!TextureImpl::ValidateTexturePixelDataType(inputType) ||
|
||||
!TextureImpl::ValidateTextureInternalFormatCompatibleWithInput(
|
||||
inputFormat, textureObject->GetFormat(), inputType)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
clearPixel.clear();
|
||||
if (data == nullptr) {
|
||||
// ARB_clear_texture defines a null clear value as all zeroes. Keeping the
|
||||
// pattern empty lets the region writer use a fast memset path.
|
||||
return true;
|
||||
}
|
||||
|
||||
PixelStoreParameters clearPixelStore{};
|
||||
clearPixelStore.Alignment = 1;
|
||||
SizeT clearPixelSize = 0;
|
||||
void* converted = MG_Util::PixelStoreProcessor::ProcessTexturePixelsDataUnpack(
|
||||
data, clearPixelStore, textureObject->GetFormat(), inputFormat, inputType,
|
||||
{1, 1, 1}, false, clearPixelSize);
|
||||
if (!converted || clearPixelSize == 0) {
|
||||
if (converted) free(converted);
|
||||
return false;
|
||||
}
|
||||
|
||||
clearPixel.resize(clearPixelSize);
|
||||
Memcpy(clearPixel.data(), converted, clearPixelSize);
|
||||
free(converted);
|
||||
return true;
|
||||
}
|
||||
|
||||
// Writes the clear into the CPU shadow and marks the whole level dirty, exactly like
|
||||
// TexSubImage*_State does. Shared limitation of the level-granular shadow sync: the
|
||||
// shadow does not reflect GPU-side writes (FBO rendering, imageStore), so a PARTIAL
|
||||
// clear of a GPU-written level re-uploads stale shadow bytes outside the region on
|
||||
// the next sync. Full-level clears (glClearTexImage, or a sub-clear covering the
|
||||
// level) rewrite the entire shadow and are always correct.
|
||||
Bool ClearMipmapRegion(const SharedPtr<MG_State::GLState::TextureObjectMipmap>& textureObject,
|
||||
TextureUploadTarget uploadTarget, GLint level,
|
||||
GLint xoffset, GLint yoffset, GLint zoffset,
|
||||
GLsizei width, GLsizei height, GLsizei depth,
|
||||
const Vector<Uint8>& clearPixel, const char* caller) {
|
||||
const IntVec3 texelSize = textureObject->GetMipmapTexelSize(uploadTarget, static_cast<Uint>(level));
|
||||
if (texelSize.x() <= 0 || texelSize.y() <= 0 || texelSize.z() <= 0) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidOperation,
|
||||
"The requested texture level has no storage.");
|
||||
return false;
|
||||
}
|
||||
if (xoffset < 0 || yoffset < 0 || zoffset < 0 ||
|
||||
width < 0 || height < 0 || depth < 0 ||
|
||||
width > texelSize.x() - xoffset ||
|
||||
height > texelSize.y() - yoffset ||
|
||||
depth > texelSize.z() - zoffset) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidValue,
|
||||
"The clear region lies outside the requested texture level.");
|
||||
return false;
|
||||
}
|
||||
if (width == 0 || height == 0 || depth == 0) return true;
|
||||
|
||||
const SizeT texelCount = static_cast<SizeT>(texelSize.x()) *
|
||||
static_cast<SizeT>(texelSize.y()) *
|
||||
static_cast<SizeT>(texelSize.z());
|
||||
const SizeT byteSize = textureObject->GetMipmapByteSize(uploadTarget, static_cast<Uint>(level));
|
||||
if (byteSize == 0 || byteSize % texelCount != 0) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidOperation,
|
||||
"The requested texture storage cannot be cleared.");
|
||||
return false;
|
||||
}
|
||||
|
||||
const SizeT bytesPerTexel = byteSize / texelCount;
|
||||
if (!clearPixel.empty() && clearPixel.size() != bytesPerTexel) {
|
||||
RecordClearTextureError(
|
||||
caller, ErrorCode::InvalidOperation,
|
||||
std::format("Converted clear value is {} bytes, but the texture stores {} bytes per texel.",
|
||||
clearPixel.size(), bytesPerTexel));
|
||||
return false;
|
||||
}
|
||||
|
||||
auto* destination = static_cast<Uint8*>(
|
||||
textureObject->MapMipmapData(uploadTarget, static_cast<Uint>(level)));
|
||||
if (!destination) {
|
||||
RecordClearTextureError(caller, ErrorCode::InvalidOperation,
|
||||
"The requested texture level could not be mapped.");
|
||||
return false;
|
||||
}
|
||||
|
||||
const SizeT fullRowBytes = static_cast<SizeT>(texelSize.x()) * bytesPerTexel;
|
||||
const SizeT fullSliceBytes = static_cast<SizeT>(texelSize.y()) * fullRowBytes;
|
||||
const SizeT clearRowBytes = static_cast<SizeT>(width) * bytesPerTexel;
|
||||
Uint8* firstClearRow = nullptr;
|
||||
|
||||
for (GLsizei z = 0; z < depth; ++z) {
|
||||
for (GLsizei y = 0; y < height; ++y) {
|
||||
Uint8* row = destination +
|
||||
static_cast<SizeT>(zoffset + z) * fullSliceBytes +
|
||||
static_cast<SizeT>(yoffset + y) * fullRowBytes +
|
||||
static_cast<SizeT>(xoffset) * bytesPerTexel;
|
||||
if (firstClearRow) {
|
||||
Memcpy(row, firstClearRow, clearRowBytes);
|
||||
continue;
|
||||
}
|
||||
|
||||
firstClearRow = row;
|
||||
if (clearPixel.empty()) {
|
||||
Memset(row, 0, clearRowBytes);
|
||||
continue;
|
||||
}
|
||||
|
||||
Memcpy(row, clearPixel.data(), bytesPerTexel);
|
||||
SizeT filled = bytesPerTexel;
|
||||
while (filled < clearRowBytes) {
|
||||
const SizeT copySize = std::min(filled, clearRowBytes - filled);
|
||||
Memcpy(row + filled, row, copySize);
|
||||
filled += copySize;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
textureObject->MarkStorageDirty(uploadTarget, static_cast<Uint>(level), true);
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void ClearTexImage(GLuint texture, GLint level, GLenum format, GLenum type, const void* data) {
|
||||
auto textureObject = GetClearTextureObject(texture, level, __func__);
|
||||
if (!textureObject) return;
|
||||
|
||||
Vector<Uint8> clearPixel;
|
||||
if (!BuildClearPixel(textureObject, format, type, data, clearPixel)) return;
|
||||
|
||||
for (TextureUploadTarget uploadTarget : textureObject->GetUploadTargets()) {
|
||||
const IntVec3 size = textureObject->GetMipmapTexelSize(uploadTarget, static_cast<Uint>(level));
|
||||
if (!ClearMipmapRegion(textureObject, uploadTarget, level, 0, 0, 0,
|
||||
size.x(), size.y(), size.z(), clearPixel, __func__)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ClearTexSubImage(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset,
|
||||
GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLenum type,
|
||||
const void* data) {
|
||||
auto textureObject = GetClearTextureObject(texture, level, __func__);
|
||||
if (!textureObject) return;
|
||||
|
||||
Vector<Uint8> clearPixel;
|
||||
if (!BuildClearPixel(textureObject, format, type, data, clearPixel)) return;
|
||||
|
||||
const auto& uploadTargets = textureObject->GetUploadTargets();
|
||||
if (textureObject->GetTarget() == TextureTarget::TextureCubeMap) {
|
||||
if (zoffset < 0 || depth < 0 ||
|
||||
static_cast<SizeT>(zoffset) > uploadTargets.size() ||
|
||||
static_cast<SizeT>(depth) > uploadTargets.size() - static_cast<SizeT>(zoffset)) {
|
||||
RecordClearTextureError(__func__, ErrorCode::InvalidValue,
|
||||
"The cube-map clear region selects invalid faces.");
|
||||
return;
|
||||
}
|
||||
for (GLsizei face = 0; face < depth; ++face) {
|
||||
if (!ClearMipmapRegion(textureObject, uploadTargets[static_cast<SizeT>(zoffset + face)], level,
|
||||
xoffset, yoffset, 0, width, height, 1, clearPixel, __func__)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (uploadTargets.empty()) {
|
||||
RecordClearTextureError(__func__, ErrorCode::InvalidOperation,
|
||||
"The requested texture has no upload target.");
|
||||
return;
|
||||
}
|
||||
ClearMipmapRegion(textureObject, uploadTargets.front(), level, xoffset, yoffset, zoffset,
|
||||
width, height, depth, clearPixel, __func__);
|
||||
}
|
||||
|
||||
Bool ValidateTextureParameterForTarget(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject,
|
||||
GLenum pname, GLint param, const char* caller) {
|
||||
const auto target = textureObject->GetTarget();
|
||||
@@ -1453,6 +1724,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
// RGTC is a 2D-only compression scheme, so a 3D target rejects it. This has to be tested on
|
||||
// the raw enum: the RGTC formats resolve to plain R8/RG8/SNORM storage on the way in (see
|
||||
// GLToMG's TextureEnumConverter), so once the internal format is converted there is nothing
|
||||
// left to distinguish them from an ordinary one- or two-channel upload.
|
||||
if ((textureUploadTarget == TextureUploadTarget::Texture3D ||
|
||||
textureUploadTarget == TextureUploadTarget::ProxyTexture3D) &&
|
||||
(internalformat == GL_COMPRESSED_RED_RGTC1 || internalformat == GL_COMPRESSED_SIGNED_RED_RGTC1 ||
|
||||
internalformat == GL_COMPRESSED_RG_RGTC2 || internalformat == GL_COMPRESSED_SIGNED_RG_RGTC2)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"RGTC compressed formats are invalid for 3D texture targets"));
|
||||
return;
|
||||
}
|
||||
|
||||
// TODO: GL_INVALID_OPERATION is generated if a non-zero buffer object name is bound to the
|
||||
// GL_PIXEL_UNPACK_BUFFER target and the buffer object's data store is currently mapped.
|
||||
// GL_INVALID_OPERATION is generated if a non-zero buffer object name is bound to the GL_PIXEL_UNPACK_BUFFER
|
||||
@@ -1510,6 +1796,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (isProxy) {
|
||||
MGLOG_D("%s: isProxy = true, not allocating", __func__);
|
||||
} else {
|
||||
DiscardMipmapChainOnBaseRespecification(textureMipmapObject, textureUploadTarget, level);
|
||||
textureMipmapObject->AllocateStorage(textureUploadTarget, level, {{width, height, depth}, internalBytes});
|
||||
}
|
||||
|
||||
@@ -1637,6 +1924,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_D("%s: isProxy = true, not allocating", __func__);
|
||||
} else {
|
||||
MGLOG_D("%s: Allocating %d bytes at mip %d", __func__, internalBytes, level);
|
||||
DiscardMipmapChainOnBaseRespecification(textureMipmapObject, textureUploadTarget, level);
|
||||
textureMipmapObject->AllocateStorage(textureUploadTarget, level,
|
||||
{{width, height, 1}, internalBytes});
|
||||
}
|
||||
@@ -1725,6 +2013,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Texture object here should always be an object with mipmap");
|
||||
auto textureMipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
if (!isProxy) {
|
||||
DiscardMipmapChainOnBaseRespecification(textureMipmapObject, textureUploadTarget, level);
|
||||
textureMipmapObject->AllocateStorage(textureUploadTarget, level, {{width, 1, 1}, internalBytes});
|
||||
}
|
||||
|
||||
@@ -2377,7 +2666,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void GetCompressedTexImage_State(GLenum target, GLint level, void* img) {
|
||||
// TODO: implement
|
||||
// TODO: implement compressed readback. Reporting success while writing nothing hands
|
||||
// the caller stale memory with GL_NO_ERROR; no texture can be compressed yet, and GL
|
||||
// specifies GL_INVALID_OPERATION when the bound level is not compressed.
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"Texture level is not stored in a compressed format."));
|
||||
}
|
||||
|
||||
void GenTextures_State(GLsizei n, GLuint* textures) {
|
||||
@@ -2564,20 +2859,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void CompressedTexSubImage3D_State(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint zoffset,
|
||||
GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize,
|
||||
const void* data) {
|
||||
// TODO: implement
|
||||
THROW_UNIMPL_EXCEPTION;
|
||||
// TODO: implement compressed upload - see CompressedTexImage2D_State.
|
||||
RecordUnsupportedCompressedFormat(__func__);
|
||||
}
|
||||
|
||||
void CompressedTexSubImage2D_State(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLsizei width,
|
||||
GLsizei height, GLenum format, GLsizei imageSize, const void* data) {
|
||||
// TODO: implement
|
||||
THROW_UNIMPL_EXCEPTION;
|
||||
// TODO: implement compressed upload - see CompressedTexImage2D_State.
|
||||
RecordUnsupportedCompressedFormat(__func__);
|
||||
}
|
||||
|
||||
void CompressedTexSubImage1D_State(GLenum target, GLint level, GLint xoffset, GLsizei width, GLenum format,
|
||||
GLsizei imageSize, const void* data) {
|
||||
// TODO: implement
|
||||
THROW_UNIMPL_EXCEPTION;
|
||||
// TODO: implement compressed upload - see CompressedTexImage2D_State.
|
||||
RecordUnsupportedCompressedFormat(__func__);
|
||||
}
|
||||
|
||||
void CompressedTexImage3D_State(GLenum target, GLint level, GLenum internalformat, GLsizei width, GLsizei height,
|
||||
@@ -2587,8 +2882,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto& textureObject = GetTextureObjectByTarget(textureUploadTarget, textureTarget);
|
||||
if (!ValidateTextureMutable(textureObject, __func__)) return;
|
||||
|
||||
// TODO: implement
|
||||
THROW_UNIMPL_EXCEPTION;
|
||||
// TODO: implement compressed upload - see CompressedTexImage2D_State.
|
||||
RecordUnsupportedCompressedFormat(__func__);
|
||||
}
|
||||
|
||||
void CompressedTexImage2D_State(GLenum target, GLint level, GLenum internalformat, GLsizei width, GLsizei height,
|
||||
@@ -2598,8 +2893,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto& textureObject = GetTextureObjectByTarget(textureUploadTarget, textureTarget);
|
||||
if (!ValidateTextureMutable(textureObject, __func__)) return;
|
||||
|
||||
// TODO: implement
|
||||
THROW_UNIMPL_EXCEPTION;
|
||||
// TODO: implement compressed upload. Until then report the spec error for an
|
||||
// unsupported compressed format rather than throwing - a C++ exception unwinding
|
||||
// through the C GL ABI is a hard crash for the caller, while GL_INVALID_ENUM is
|
||||
// exactly what GL_NUM_COMPRESSED_TEXTURE_FORMATS == 0 promises.
|
||||
RecordUnsupportedCompressedFormat(__func__);
|
||||
}
|
||||
|
||||
void CompressedTexImage1D_State(GLenum target, GLint level, GLenum internalformat, GLsizei width, GLint border,
|
||||
@@ -2609,8 +2907,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto& textureObject = GetTextureObjectByTarget(textureUploadTarget, textureTarget);
|
||||
if (!ValidateTextureMutable(textureObject, __func__)) return;
|
||||
|
||||
// TODO: implement
|
||||
THROW_UNIMPL_EXCEPTION;
|
||||
// TODO: implement compressed upload - see CompressedTexImage2D_State.
|
||||
RecordUnsupportedCompressedFormat(__func__);
|
||||
}
|
||||
|
||||
void BindTexture_State(GLenum target, GLuint texture) {
|
||||
@@ -2956,6 +3254,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
textureMipmapObject->AllocateStorage(textureUploadTarget, level, {{levelWidth, 1, 1}, byteSize});
|
||||
textureMipmapObject->MarkStorageDirty(textureUploadTarget, level, false);
|
||||
}
|
||||
// Immutable storage defines exactly `levels` levels; AllocateStorage only grows, so a
|
||||
// longer pre-existing chain has to be dropped explicitly.
|
||||
textureMipmapObject->TruncateMipmapLevels(textureUploadTarget, static_cast<Uint>(levels));
|
||||
textureObject->SetImmutableLevels(static_cast<Uint>(levels));
|
||||
}
|
||||
|
||||
@@ -3008,6 +3309,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
textureMipmapObject->AllocateStorage(textureUploadTarget, level, {{levelWidth, levelHeight, 1}, byteSize});
|
||||
textureMipmapObject->MarkStorageDirty(textureUploadTarget, level, false);
|
||||
}
|
||||
// See TextureStorage1D.
|
||||
textureMipmapObject->TruncateMipmapLevels(textureUploadTarget, static_cast<Uint>(levels));
|
||||
textureObject->SetImmutableLevels(static_cast<Uint>(levels));
|
||||
}
|
||||
|
||||
@@ -3060,6 +3363,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
{{levelWidth, levelHeight, levelDepth}, byteSize});
|
||||
textureMipmapObject->MarkStorageDirty(textureUploadTarget, level, false);
|
||||
}
|
||||
// See TextureStorage1D.
|
||||
textureMipmapObject->TruncateMipmapLevels(textureUploadTarget, static_cast<Uint>(levels));
|
||||
textureObject->SetImmutableLevels(static_cast<Uint>(levels));
|
||||
}
|
||||
|
||||
@@ -3818,6 +4123,27 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
CopyTexSubImage2D_Backend(target, level, xoffset, yoffset, x, y, width, height);
|
||||
}
|
||||
|
||||
void CopyTextureSubImage2D(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height) {
|
||||
auto textureObject = GetTextureObjectByName(texture, __func__);
|
||||
if (!textureObject) return;
|
||||
// GL 4.6 sec. 8.8: the 2D form only accepts these effective targets; cube maps must
|
||||
// go through CopyTextureSubImage3D with the face as a layer.
|
||||
const auto target = textureObject->GetTarget();
|
||||
if (target != TextureTarget::Texture2D && target != TextureTarget::Texture1DArray &&
|
||||
target != TextureTarget::TextureRectangle) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"CopyTextureSubImage2D requires a 2D, 1D-array, or "
|
||||
"rectangle texture."));
|
||||
return;
|
||||
}
|
||||
WithTemporarilyBoundNamedTexture(textureObject, [&](GLenum glTarget) {
|
||||
CopyTexSubImage2D_Backend(glTarget, level, xoffset, yoffset, x, y, width, height);
|
||||
});
|
||||
}
|
||||
|
||||
void CopyTexSubImage1D(GLenum target, GLint level, GLint xoffset, GLint x, GLint y, GLsizei width) {
|
||||
CopyTexSubImage1D_State(target, level, xoffset, x, y, width);
|
||||
}
|
||||
|
||||
@@ -11,6 +11,9 @@
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
/* @INSERTION_POINT:FUNCTION_DECLARATION@ */
|
||||
void ClearTexImage(GLuint texture, GLint level, GLenum format, GLenum type, const void* data);
|
||||
void ClearTexSubImage(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width,
|
||||
GLsizei height, GLsizei depth, GLenum format, GLenum type, const void* data);
|
||||
void BindImageTexture(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer, GLenum access,
|
||||
GLenum format);
|
||||
void GenerateMipmap(GLenum target);
|
||||
@@ -95,6 +98,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLsizei width, GLsizei height);
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height);
|
||||
void CopyTextureSubImage2D(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height);
|
||||
void CopyTexSubImage1D(GLenum target, GLint level, GLint xoffset, GLint x, GLint y, GLsizei width);
|
||||
void CopyTexImage2D(GLenum target, GLint level, GLenum internalformat, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height, GLint border);
|
||||
|
||||
@@ -14,7 +14,8 @@
|
||||
#include <MG_State/GLState/TextureState/TextureObjectStubs.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
UniquePtr<ProxyTextureManager> pProxyTextureManager;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<ProxyTextureManager>& pProxyTextureManager = *new UniquePtr<ProxyTextureManager>();
|
||||
|
||||
Bool IsProxyTextureTarget(TextureUploadTarget target) {
|
||||
switch (target) {
|
||||
|
||||
@@ -23,5 +23,5 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
UnorderedMap<TextureUploadTarget, SharedPtr<MG_State::GLState::ITextureObject>> m_proxyTexturesMap;
|
||||
};
|
||||
|
||||
extern UniquePtr<ProxyTextureManager> pProxyTextureManager;
|
||||
extern UniquePtr<ProxyTextureManager>& pProxyTextureManager;
|
||||
} // namespace MobileGL::MG_Impl::GLImpl::TextureImpl
|
||||
|
||||
@@ -85,6 +85,8 @@ namespace MobileGL::MG_Impl {
|
||||
GETPROC(CGLGetPixelFormat, name);
|
||||
GETPROC(CGLSetCurrentContext, name);
|
||||
GETPROC(CGLGetCurrentContext, name);
|
||||
GETPROC(CGLSetVirtualScreen, name);
|
||||
GETPROC(CGLGetVirtualScreen, name);
|
||||
GETPROC(CGLSetParameter, name);
|
||||
GETPROC(CGLGetParameter, name);
|
||||
GETPROC(CGLUpdateContext, name);
|
||||
|
||||
@@ -29,10 +29,19 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
char kContextViewKey;
|
||||
char kContextLayerKey;
|
||||
|
||||
std::once_flag g_installOnce;
|
||||
IMP g_pixelFormatDealloc = nullptr;
|
||||
IMP g_contextDealloc = nullptr;
|
||||
|
||||
std::mutex& HookInstallMutex() {
|
||||
static auto* mutex = new std::mutex();
|
||||
return *mutex;
|
||||
}
|
||||
|
||||
Bool& HooksInstalled() {
|
||||
static auto* installed = new Bool(false);
|
||||
return *installed;
|
||||
}
|
||||
|
||||
template <typename Fn>
|
||||
Fn ObjcMsgSend() {
|
||||
return reinterpret_cast<Fn>(objc_msgSend);
|
||||
@@ -431,12 +440,12 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
method_setImplementation(method, replacement);
|
||||
}
|
||||
|
||||
void InstallHooksOnce() {
|
||||
Bool InstallHooksOnce() {
|
||||
Class pixelFormatClass = objc_getClass("NSOpenGLPixelFormat");
|
||||
Class contextClass = objc_getClass("NSOpenGLContext");
|
||||
if (!pixelFormatClass || !contextClass) {
|
||||
MGLOG_W("NSOpenGLImpl: NSOpenGL classes are not loaded; hooks not installed");
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
|
||||
ReplaceInstanceMethod(pixelFormatClass, "initWithAttributes:",
|
||||
@@ -471,11 +480,34 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
ReplaceInstanceMethod(contextClass, "dealloc", reinterpret_cast<IMP>(ContextDealloc), &g_contextDealloc);
|
||||
|
||||
MGLOG_I("NSOpenGLImpl hooks installed");
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void InstallHooks() {
|
||||
std::call_once(g_installOnce, InstallHooksOnce);
|
||||
const std::lock_guard<std::mutex> lock(HookInstallMutex());
|
||||
if (!HooksInstalled()) {
|
||||
// Do not permanently consume the install attempt when the OpenGL
|
||||
// framework has not registered its Objective-C classes yet. The
|
||||
// dyld bootstrap normally runs after framework dependencies, but
|
||||
// an explicitly loaded/static-linked MobileGL can arrive earlier.
|
||||
HooksInstalled() = InstallHooksOnce();
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::NSOpenGLImpl
|
||||
|
||||
namespace {
|
||||
// SDL's Cocoa backend creates NSOpenGLPixelFormat/NSOpenGLContext before
|
||||
// its first dlsym("glGetString") or other MobileGL host-API call. Install
|
||||
// only the lightweight Objective-C dispatch hooks while the injected dylib
|
||||
// is loading so those first Cocoa objects are routed through CGLImpl. The
|
||||
// hooked context constructor reaches EGLImpl::GetDisplay(), which performs
|
||||
// the full, thread-safe MobileGL initialization outside this bootstrap.
|
||||
//
|
||||
// There is intentionally no matching destructor: backend teardown remains
|
||||
// owned by the EGL lifecycle and process-exit globals remain leak-at-exit.
|
||||
__attribute__((constructor)) void BootstrapNSOpenGLHooks() {
|
||||
MobileGL::MG_Impl::NSOpenGLImpl::InstallHooks();
|
||||
}
|
||||
} // namespace
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
// MobileGL - MobileGL/MG_Impl/WGLImpl/Exporting/Definitions.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
// wingdi.h declares most wgl* entry points as WINGDIAPI (__declspec(dllimport)),
|
||||
// which would reject our definitions. _GDI32_ is the SDK's "I am the module that
|
||||
// implements these" switch: it turns WINGDIAPI into a plain declaration. It must
|
||||
// be defined before the first windows.h inclusion in this translation unit.
|
||||
#if defined(_WIN32) && !defined(_GDI32_)
|
||||
#define _GDI32_ 1
|
||||
#endif
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include "../WGLImpl.h"
|
||||
|
||||
namespace WGL = MobileGL::MG_Impl::WGLImpl;
|
||||
|
||||
// ---- Pixel-format entry points (gdi32 forwards ChoosePixelFormat/SetPixelFormat/
|
||||
// ---- DescribePixelFormat/GetPixelFormat/SwapBuffers into these exports) ----
|
||||
|
||||
extern "C" int WINAPI wglChoosePixelFormat(HDC hdc, CONST PIXELFORMATDESCRIPTOR* ppfd) {
|
||||
return WGL::ChoosePixelFormat(hdc, ppfd);
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglDescribePixelFormat(HDC hdc, int iPixelFormat, UINT nBytes,
|
||||
LPPIXELFORMATDESCRIPTOR ppfd) {
|
||||
return WGL::DescribePixelFormat(hdc, iPixelFormat, nBytes, ppfd);
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglGetPixelFormat(HDC hdc) {
|
||||
return WGL::GetPixelFormat(hdc);
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglSetPixelFormat(HDC hdc, int iPixelFormat, CONST PIXELFORMATDESCRIPTOR* ppfd) {
|
||||
return WGL::SetPixelFormat(hdc, iPixelFormat, ppfd);
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglSwapBuffers(HDC hdc) {
|
||||
return WGL::SwapBuffers(hdc);
|
||||
}
|
||||
|
||||
// ---- Context management ----
|
||||
|
||||
extern "C" HGLRC WINAPI wglCreateContext(HDC hdc) {
|
||||
return WGL::CreateContext(hdc);
|
||||
}
|
||||
|
||||
extern "C" HGLRC WINAPI wglCreateLayerContext(HDC hdc, int iLayerPlane) {
|
||||
return iLayerPlane == 0 ? WGL::CreateContext(hdc) : nullptr;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglCopyContext(HGLRC, HGLRC, UINT) {
|
||||
MGLOG_W("wglCopyContext is not supported");
|
||||
SetLastError(ERROR_NOT_SUPPORTED);
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglDeleteContext(HGLRC hglrc) {
|
||||
return WGL::DeleteContext(hglrc);
|
||||
}
|
||||
|
||||
extern "C" HGLRC WINAPI wglGetCurrentContext(VOID) {
|
||||
return WGL::GetCurrentContext();
|
||||
}
|
||||
|
||||
extern "C" HDC WINAPI wglGetCurrentDC(VOID) {
|
||||
return WGL::GetCurrentDC();
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglMakeCurrent(HDC hdc, HGLRC hglrc) {
|
||||
return WGL::MakeCurrent(hdc, hglrc);
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglShareLists(HGLRC hglrcShare, HGLRC hglrcDest) {
|
||||
return WGL::ShareLists(hglrcShare, hglrcDest);
|
||||
}
|
||||
|
||||
// ---- Proc address ----
|
||||
|
||||
extern "C" PROC WINAPI wglGetProcAddress(LPCSTR lpszProc) {
|
||||
return WGL::GetProcAddress(lpszProc);
|
||||
}
|
||||
|
||||
extern "C" PROC WINAPI wglGetDefaultProcAddress(LPCSTR lpszProc) {
|
||||
return WGL::GetProcAddress(lpszProc);
|
||||
}
|
||||
|
||||
// ---- Layer planes and palettes (unsupported; overlay planes do not exist here) ----
|
||||
|
||||
extern "C" BOOL WINAPI wglDescribeLayerPlane(HDC, int, int, UINT, LPLAYERPLANEDESCRIPTOR) {
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglSetLayerPaletteEntries(HDC, int, int, int, CONST COLORREF*) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
extern "C" int WINAPI wglGetLayerPaletteEntries(HDC, int, int, int, COLORREF*) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglRealizeLayerPalette(HDC, int, BOOL) {
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglSwapLayerBuffers(HDC hdc, UINT fuPlanes) {
|
||||
if (fuPlanes & WGL_SWAP_MAIN_PLANE) {
|
||||
return WGL::SwapBuffers(hdc);
|
||||
}
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" DWORD WINAPI wglSwapMultipleBuffers(UINT n, CONST WGLSWAP* ps) {
|
||||
if (!ps) {
|
||||
return 0;
|
||||
}
|
||||
DWORD swapped = 0;
|
||||
for (UINT i = 0; i < n; ++i) {
|
||||
if (WGL::SwapBuffers(ps[i].hdc)) {
|
||||
++swapped;
|
||||
}
|
||||
}
|
||||
return swapped;
|
||||
}
|
||||
|
||||
// ---- Font rendering (legacy immediate-mode feature; not supported) ----
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontBitmapsA(HDC, DWORD, DWORD, DWORD) {
|
||||
MGLOG_W("wglUseFontBitmapsA is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontBitmapsW(HDC, DWORD, DWORD, DWORD) {
|
||||
MGLOG_W("wglUseFontBitmapsW is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontOutlinesA(HDC, DWORD, DWORD, DWORD, FLOAT, FLOAT, int,
|
||||
LPGLYPHMETRICSFLOAT) {
|
||||
MGLOG_W("wglUseFontOutlinesA is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontOutlinesW(HDC, DWORD, DWORD, DWORD, FLOAT, FLOAT, int,
|
||||
LPGLYPHMETRICSFLOAT) {
|
||||
MGLOG_W("wglUseFontOutlinesW is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
#endif // _WIN32
|
||||
@@ -0,0 +1,30 @@
|
||||
; MobileGL WGL exports. The wgl* entry points are defined without
|
||||
; __declspec(dllexport) because wingdi.h pre-declares them (with _GDI32_ they
|
||||
; become plain declarations, and MSVC rejects adding dllexport afterwards),
|
||||
; so this .def file is what actually exports them from the DLL.
|
||||
EXPORTS
|
||||
wglChoosePixelFormat
|
||||
wglCopyContext
|
||||
wglCreateContext
|
||||
wglCreateLayerContext
|
||||
wglDeleteContext
|
||||
wglDescribeLayerPlane
|
||||
wglDescribePixelFormat
|
||||
wglGetCurrentContext
|
||||
wglGetCurrentDC
|
||||
wglGetDefaultProcAddress
|
||||
wglGetLayerPaletteEntries
|
||||
wglGetPixelFormat
|
||||
wglGetProcAddress
|
||||
wglMakeCurrent
|
||||
wglRealizeLayerPalette
|
||||
wglSetLayerPaletteEntries
|
||||
wglSetPixelFormat
|
||||
wglShareLists
|
||||
wglSwapBuffers
|
||||
wglSwapLayerBuffers
|
||||
wglSwapMultipleBuffers
|
||||
wglUseFontBitmapsA
|
||||
wglUseFontBitmapsW
|
||||
wglUseFontOutlinesA
|
||||
wglUseFontOutlinesW
|
||||
@@ -0,0 +1,732 @@
|
||||
// MobileGL - MobileGL/MG_Impl/WGLImpl/WGLImpl.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "WGLImpl.h"
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include "../EGLImpl/EGLImpl.h"
|
||||
#include "../GetProcAddress.h"
|
||||
#include <Init.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::WGLImpl {
|
||||
namespace {
|
||||
// WGL_ARB_pixel_format
|
||||
constexpr int WGL_NUMBER_PIXEL_FORMATS_ARB = 0x2000;
|
||||
constexpr int WGL_DRAW_TO_WINDOW_ARB = 0x2001;
|
||||
constexpr int WGL_DRAW_TO_BITMAP_ARB = 0x2002;
|
||||
constexpr int WGL_ACCELERATION_ARB = 0x2003;
|
||||
constexpr int WGL_NEED_PALETTE_ARB = 0x2004;
|
||||
constexpr int WGL_NEED_SYSTEM_PALETTE_ARB = 0x2005;
|
||||
constexpr int WGL_SWAP_LAYER_BUFFERS_ARB = 0x2006;
|
||||
constexpr int WGL_SWAP_METHOD_ARB = 0x2007;
|
||||
constexpr int WGL_NUMBER_OVERLAYS_ARB = 0x2008;
|
||||
constexpr int WGL_NUMBER_UNDERLAYS_ARB = 0x2009;
|
||||
constexpr int WGL_TRANSPARENT_ARB = 0x200A;
|
||||
constexpr int WGL_SHARE_DEPTH_ARB = 0x200C;
|
||||
constexpr int WGL_SHARE_STENCIL_ARB = 0x200D;
|
||||
constexpr int WGL_SHARE_ACCUM_ARB = 0x200E;
|
||||
constexpr int WGL_SUPPORT_GDI_ARB = 0x200F;
|
||||
constexpr int WGL_SUPPORT_OPENGL_ARB = 0x2010;
|
||||
constexpr int WGL_DOUBLE_BUFFER_ARB = 0x2011;
|
||||
constexpr int WGL_STEREO_ARB = 0x2012;
|
||||
constexpr int WGL_PIXEL_TYPE_ARB = 0x2013;
|
||||
constexpr int WGL_COLOR_BITS_ARB = 0x2014;
|
||||
constexpr int WGL_RED_BITS_ARB = 0x2015;
|
||||
constexpr int WGL_RED_SHIFT_ARB = 0x2016;
|
||||
constexpr int WGL_GREEN_BITS_ARB = 0x2017;
|
||||
constexpr int WGL_GREEN_SHIFT_ARB = 0x2018;
|
||||
constexpr int WGL_BLUE_BITS_ARB = 0x2019;
|
||||
constexpr int WGL_BLUE_SHIFT_ARB = 0x201A;
|
||||
constexpr int WGL_ALPHA_BITS_ARB = 0x201B;
|
||||
constexpr int WGL_ALPHA_SHIFT_ARB = 0x201C;
|
||||
constexpr int WGL_ACCUM_BITS_ARB = 0x201D;
|
||||
constexpr int WGL_ACCUM_RED_BITS_ARB = 0x201E;
|
||||
constexpr int WGL_ACCUM_GREEN_BITS_ARB = 0x201F;
|
||||
constexpr int WGL_ACCUM_BLUE_BITS_ARB = 0x2020;
|
||||
constexpr int WGL_ACCUM_ALPHA_BITS_ARB = 0x2021;
|
||||
constexpr int WGL_DEPTH_BITS_ARB = 0x2022;
|
||||
constexpr int WGL_STENCIL_BITS_ARB = 0x2023;
|
||||
constexpr int WGL_AUX_BUFFERS_ARB = 0x2024;
|
||||
constexpr int WGL_NO_ACCELERATION_ARB = 0x2025;
|
||||
constexpr int WGL_FULL_ACCELERATION_ARB = 0x2027;
|
||||
constexpr int WGL_SWAP_EXCHANGE_ARB = 0x2028;
|
||||
constexpr int WGL_TYPE_RGBA_ARB = 0x202B;
|
||||
// WGL_ARB_multisample
|
||||
constexpr int WGL_SAMPLE_BUFFERS_ARB = 0x2041;
|
||||
constexpr int WGL_SAMPLES_ARB = 0x2042;
|
||||
// WGL_ARB_create_context / _profile / _no_error
|
||||
constexpr int WGL_CONTEXT_MAJOR_VERSION_ARB = 0x2091;
|
||||
constexpr int WGL_CONTEXT_MINOR_VERSION_ARB = 0x2092;
|
||||
constexpr int WGL_CONTEXT_LAYER_PLANE_ARB = 0x2093;
|
||||
constexpr int WGL_CONTEXT_FLAGS_ARB = 0x2094;
|
||||
constexpr int WGL_CONTEXT_PROFILE_MASK_ARB = 0x9126;
|
||||
constexpr int WGL_CONTEXT_DEBUG_BIT_ARB = 0x0001;
|
||||
constexpr int WGL_CONTEXT_FORWARD_COMPATIBLE_BIT_ARB = 0x0002;
|
||||
constexpr int WGL_CONTEXT_CORE_PROFILE_BIT_ARB = 0x00000001;
|
||||
constexpr int WGL_CONTEXT_COMPATIBILITY_PROFILE_BIT_ARB = 0x00000002;
|
||||
constexpr int WGL_CONTEXT_OPENGL_NO_ERROR_ARB = 0x31B3;
|
||||
constexpr DWORD ERROR_INVALID_VERSION_ARB = 0x2095;
|
||||
constexpr DWORD ERROR_INVALID_PROFILE_ARB = 0x2096;
|
||||
|
||||
struct PixelFormatInfo {
|
||||
GLint AlphaBits;
|
||||
GLint DepthBits;
|
||||
GLint StencilBits;
|
||||
};
|
||||
|
||||
// Mirrors the two EGLState configs (RGBA8 + depth24, stencil 8 / stencil 0).
|
||||
constexpr PixelFormatInfo kPixelFormats[] = {
|
||||
{8, 24, 8},
|
||||
{8, 24, 0},
|
||||
};
|
||||
constexpr int kPixelFormatCount = static_cast<int>(std::size(kPixelFormats));
|
||||
|
||||
struct ContextObject {
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLConfig Config = nullptr;
|
||||
EGLContext Context = EGL_NO_CONTEXT;
|
||||
};
|
||||
|
||||
struct WindowSurface {
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLSurface Surface = EGL_NO_SURFACE;
|
||||
Uint32 Width = 0;
|
||||
Uint32 Height = 0;
|
||||
};
|
||||
|
||||
std::recursive_mutex& RegistryMutex() {
|
||||
static auto* mutex = new std::recursive_mutex();
|
||||
return *mutex;
|
||||
}
|
||||
|
||||
UnorderedMap<HGLRC, ContextObject>& Contexts() {
|
||||
static auto* contexts = new UnorderedMap<HGLRC, ContextObject>();
|
||||
return *contexts;
|
||||
}
|
||||
|
||||
UnorderedMap<HWND, WindowSurface>& WindowSurfaces() {
|
||||
static auto* surfaces = new UnorderedMap<HWND, WindowSurface>();
|
||||
return *surfaces;
|
||||
}
|
||||
|
||||
UnorderedMap<HWND, int>& WindowPixelFormats() {
|
||||
static auto* formats = new UnorderedMap<HWND, int>();
|
||||
return *formats;
|
||||
}
|
||||
|
||||
Uint64& NextContextHandle() {
|
||||
static auto* handle = new Uint64(0x10000);
|
||||
return *handle;
|
||||
}
|
||||
|
||||
struct ThreadCurrent {
|
||||
HDC DC = nullptr;
|
||||
HGLRC Context = nullptr;
|
||||
};
|
||||
thread_local ThreadCurrent t_current;
|
||||
|
||||
Int& SwapIntervalShadow() {
|
||||
static auto* interval = new Int(1);
|
||||
return *interval;
|
||||
}
|
||||
|
||||
void EnsureInitialized() {
|
||||
// Initialize() loads backend libraries and glslang, which must not run
|
||||
// under the loader lock; first WGL call is the earliest safe moment.
|
||||
// MobileGL::EnsureInitialized (not a local once_flag) so a fresh init
|
||||
// can follow a full teardown from the last eglTerminate.
|
||||
MobileGL::EnsureInitialized();
|
||||
}
|
||||
|
||||
EGLDisplay EnsureDisplay() {
|
||||
EnsureInitialized();
|
||||
EGLDisplay display = EGLImpl::GetDisplay(EGL_DEFAULT_DISPLAY);
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
if (!EGLImpl::Initialize(display, nullptr, nullptr)) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
return display;
|
||||
}
|
||||
|
||||
HGLRC EncodeContext(Uint64 handle) {
|
||||
return reinterpret_cast<HGLRC>(static_cast<SizeT>(handle));
|
||||
}
|
||||
|
||||
ContextObject* TryGetContext(HGLRC hglrc) {
|
||||
auto& contexts = Contexts();
|
||||
auto it = contexts.find(hglrc);
|
||||
return it == contexts.end() ? nullptr : &it->second;
|
||||
}
|
||||
|
||||
const PixelFormatInfo& PixelFormatForWindow(HWND hwnd) {
|
||||
auto& formats = WindowPixelFormats();
|
||||
auto it = formats.find(hwnd);
|
||||
int index = it == formats.end() ? 1 : it->second;
|
||||
if (index < 1 || index > kPixelFormatCount) {
|
||||
index = 1;
|
||||
}
|
||||
return kPixelFormats[index - 1];
|
||||
}
|
||||
|
||||
Bool QueryClientSize(HWND hwnd, Uint32& width, Uint32& height) {
|
||||
RECT rect{};
|
||||
if (!GetClientRect(hwnd, &rect)) {
|
||||
return false;
|
||||
}
|
||||
width = static_cast<Uint32>(std::max<LONG>(rect.right - rect.left, 1));
|
||||
height = static_cast<Uint32>(std::max<LONG>(rect.bottom - rect.top, 1));
|
||||
return true;
|
||||
}
|
||||
|
||||
// The backends never query the HWND client size themselves; the WGL layer
|
||||
// owns size discovery and pushes changes through the internal resize hook
|
||||
// (same contract as the macOS CGL layer).
|
||||
void SyncSurfaceSize(HWND hwnd, WindowSurface& surface) {
|
||||
Uint32 width = 0;
|
||||
Uint32 height = 0;
|
||||
if (!QueryClientSize(hwnd, width, height)) {
|
||||
return;
|
||||
}
|
||||
if (width == surface.Width && height == surface.Height) {
|
||||
return;
|
||||
}
|
||||
if (EGLImpl::ResizePlatformWindowSurface(surface.Display, surface.Surface,
|
||||
static_cast<EGLint>(width), static_cast<EGLint>(height))) {
|
||||
surface.Width = width;
|
||||
surface.Height = height;
|
||||
}
|
||||
}
|
||||
|
||||
WindowSurface* EnsureWindowSurface(HWND hwnd, const ContextObject& context) {
|
||||
auto& surfaces = WindowSurfaces();
|
||||
auto it = surfaces.find(hwnd);
|
||||
if (it != surfaces.end()) {
|
||||
SyncSurfaceSize(hwnd, it->second);
|
||||
return &it->second;
|
||||
}
|
||||
|
||||
Uint32 width = 0;
|
||||
Uint32 height = 0;
|
||||
if (!QueryClientSize(hwnd, width, height)) {
|
||||
MGLOG_E("wgl: GetClientRect failed for HWND %p", hwnd);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const EGLAttrib attribs[] = {
|
||||
EGL_WIDTH, static_cast<EGLAttrib>(width),
|
||||
EGL_HEIGHT, static_cast<EGLAttrib>(height),
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLSurface surface =
|
||||
EGLImpl::CreatePlatformWindowSurface(context.Display, context.Config, hwnd, attribs);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E("wgl: failed to create window surface for HWND %p (%ux%u)", hwnd, width, height);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
WindowSurface record;
|
||||
record.Display = context.Display;
|
||||
record.Surface = surface;
|
||||
record.Width = width;
|
||||
record.Height = height;
|
||||
auto [inserted, _] = surfaces.emplace(hwnd, record);
|
||||
return &inserted->second;
|
||||
}
|
||||
|
||||
HGLRC CreateContextFromEGLAttribs(HDC hdc, HGLRC share, const EGLint* contextAttribs) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
EGLDisplay display = EnsureDisplay();
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
MGLOG_E("wgl: no EGL display");
|
||||
return nullptr;
|
||||
}
|
||||
EGLImpl::BindAPI(EGL_OPENGL_API);
|
||||
|
||||
EGLContext shareContext = EGL_NO_CONTEXT;
|
||||
if (share) {
|
||||
auto* shareObject = TryGetContext(share);
|
||||
if (!shareObject) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return nullptr;
|
||||
}
|
||||
shareContext = shareObject->Context;
|
||||
}
|
||||
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
const PixelFormatInfo& pixelFormat = PixelFormatForWindow(hwnd);
|
||||
const EGLint configAttribs[] = {
|
||||
EGL_RED_SIZE, 8,
|
||||
EGL_GREEN_SIZE, 8,
|
||||
EGL_BLUE_SIZE, 8,
|
||||
EGL_ALPHA_SIZE, pixelFormat.AlphaBits,
|
||||
EGL_DEPTH_SIZE, pixelFormat.DepthBits,
|
||||
EGL_STENCIL_SIZE, pixelFormat.StencilBits,
|
||||
EGL_SURFACE_TYPE, EGL_WINDOW_BIT | EGL_PBUFFER_BIT,
|
||||
EGL_RENDERABLE_TYPE, EGL_OPENGL_BIT,
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLConfig config = nullptr;
|
||||
EGLint configCount = 0;
|
||||
if (!EGLImpl::ChooseConfig(display, configAttribs, &config, 1, &configCount) || configCount <= 0) {
|
||||
MGLOG_E("wgl: eglChooseConfig failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
EGLContext eglContext = EGLImpl::CreateContext(display, config, shareContext, contextAttribs);
|
||||
if (eglContext == EGL_NO_CONTEXT) {
|
||||
MGLOG_E("wgl: eglCreateContext failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ContextObject object;
|
||||
object.Display = display;
|
||||
object.Config = config;
|
||||
object.Context = eglContext;
|
||||
const auto handle = EncodeContext(NextContextHandle()++);
|
||||
Contexts()[handle] = object;
|
||||
MGLOG_I("wgl: created context %p (EGL context %p)", handle, eglContext);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// ---- WGL extension entry points (resolved via wglGetProcAddress only) ----
|
||||
|
||||
const char* WINAPI Ext_GetExtensionsStringARB(HDC) {
|
||||
return "WGL_ARB_create_context WGL_ARB_create_context_no_error WGL_ARB_create_context_profile "
|
||||
"WGL_ARB_extensions_string WGL_ARB_pixel_format WGL_EXT_extensions_string WGL_EXT_swap_control";
|
||||
}
|
||||
|
||||
const char* WINAPI Ext_GetExtensionsStringEXT() {
|
||||
return Ext_GetExtensionsStringARB(nullptr);
|
||||
}
|
||||
|
||||
HGLRC WINAPI Ext_CreateContextAttribsARB(HDC hdc, HGLRC hShareContext, const int* attribList) {
|
||||
EnsureInitialized();
|
||||
int major = 1;
|
||||
int minor = 0;
|
||||
int profileMask = 0;
|
||||
int flags = 0;
|
||||
if (attribList) {
|
||||
for (SizeT i = 0; attribList[i] != 0; i += 2) {
|
||||
const int attrib = attribList[i];
|
||||
const int value = attribList[i + 1];
|
||||
switch (attrib) {
|
||||
case WGL_CONTEXT_MAJOR_VERSION_ARB:
|
||||
major = value;
|
||||
break;
|
||||
case WGL_CONTEXT_MINOR_VERSION_ARB:
|
||||
minor = value;
|
||||
break;
|
||||
case WGL_CONTEXT_PROFILE_MASK_ARB:
|
||||
profileMask = value;
|
||||
break;
|
||||
case WGL_CONTEXT_FLAGS_ARB:
|
||||
flags = value;
|
||||
break;
|
||||
case WGL_CONTEXT_LAYER_PLANE_ARB:
|
||||
if (value != 0) {
|
||||
SetLastError(ERROR_INVALID_PARAMETER);
|
||||
return nullptr;
|
||||
}
|
||||
break;
|
||||
case WGL_CONTEXT_OPENGL_NO_ERROR_ARB:
|
||||
// Accepted and ignored: MobileGL always validates.
|
||||
break;
|
||||
default:
|
||||
MGLOG_D("wglCreateContextAttribsARB: ignoring attrib 0x%04x = 0x%x", attrib, value);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (major < 1 || (profileMask & ~(WGL_CONTEXT_CORE_PROFILE_BIT_ARB |
|
||||
WGL_CONTEXT_COMPATIBILITY_PROFILE_BIT_ARB))) {
|
||||
SetLastError(profileMask ? ERROR_INVALID_PROFILE_ARB : ERROR_INVALID_VERSION_ARB);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
Vector<EGLint> attribs = {
|
||||
EGL_CONTEXT_MAJOR_VERSION, major,
|
||||
EGL_CONTEXT_MINOR_VERSION, minor,
|
||||
};
|
||||
const Bool wantsCompat = (profileMask & WGL_CONTEXT_COMPATIBILITY_PROFILE_BIT_ARB) != 0;
|
||||
if (major > 3 || (major == 3 && minor >= 2) || profileMask != 0) {
|
||||
attribs.push_back(EGL_CONTEXT_OPENGL_PROFILE_MASK);
|
||||
attribs.push_back(wantsCompat ? EGL_CONTEXT_OPENGL_COMPATIBILITY_PROFILE_BIT
|
||||
: EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT);
|
||||
}
|
||||
if (flags & WGL_CONTEXT_FORWARD_COMPATIBLE_BIT_ARB) {
|
||||
attribs.push_back(EGL_CONTEXT_OPENGL_FORWARD_COMPATIBLE);
|
||||
attribs.push_back(EGL_TRUE);
|
||||
}
|
||||
if (flags & WGL_CONTEXT_DEBUG_BIT_ARB) {
|
||||
attribs.push_back(EGL_CONTEXT_OPENGL_DEBUG);
|
||||
attribs.push_back(EGL_TRUE);
|
||||
}
|
||||
attribs.push_back(EGL_NONE);
|
||||
|
||||
return CreateContextFromEGLAttribs(hdc, hShareContext, attribs.data());
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_SwapIntervalEXT(int interval) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
EGLDisplay display = EnsureDisplay();
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
return FALSE;
|
||||
}
|
||||
if (interval < 0) {
|
||||
// Adaptive vsync is not supported; clamp to regular vsync.
|
||||
interval = 1;
|
||||
}
|
||||
EGLImpl::SwapInterval(display, interval);
|
||||
SwapIntervalShadow() = interval;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
int WINAPI Ext_GetSwapIntervalEXT() {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
return SwapIntervalShadow();
|
||||
}
|
||||
|
||||
int PixelFormatAttribValue(int format, int attrib) {
|
||||
const PixelFormatInfo& info = kPixelFormats[format - 1];
|
||||
switch (attrib) {
|
||||
case WGL_NUMBER_PIXEL_FORMATS_ARB:
|
||||
return kPixelFormatCount;
|
||||
case WGL_SUPPORT_OPENGL_ARB:
|
||||
case WGL_DRAW_TO_WINDOW_ARB:
|
||||
case WGL_DOUBLE_BUFFER_ARB:
|
||||
return 1;
|
||||
case WGL_ACCELERATION_ARB:
|
||||
return WGL_FULL_ACCELERATION_ARB;
|
||||
case WGL_PIXEL_TYPE_ARB:
|
||||
return WGL_TYPE_RGBA_ARB;
|
||||
case WGL_COLOR_BITS_ARB:
|
||||
return 32;
|
||||
case WGL_RED_BITS_ARB:
|
||||
case WGL_GREEN_BITS_ARB:
|
||||
case WGL_BLUE_BITS_ARB:
|
||||
return 8;
|
||||
case WGL_RED_SHIFT_ARB:
|
||||
return 16;
|
||||
case WGL_GREEN_SHIFT_ARB:
|
||||
return 8;
|
||||
case WGL_BLUE_SHIFT_ARB:
|
||||
return 0;
|
||||
case WGL_ALPHA_BITS_ARB:
|
||||
return info.AlphaBits;
|
||||
case WGL_ALPHA_SHIFT_ARB:
|
||||
return 24;
|
||||
case WGL_DEPTH_BITS_ARB:
|
||||
return info.DepthBits;
|
||||
case WGL_STENCIL_BITS_ARB:
|
||||
return info.StencilBits;
|
||||
case WGL_SWAP_METHOD_ARB:
|
||||
return WGL_SWAP_EXCHANGE_ARB;
|
||||
case WGL_DRAW_TO_BITMAP_ARB:
|
||||
case WGL_NEED_PALETTE_ARB:
|
||||
case WGL_NEED_SYSTEM_PALETTE_ARB:
|
||||
case WGL_SWAP_LAYER_BUFFERS_ARB:
|
||||
case WGL_NUMBER_OVERLAYS_ARB:
|
||||
case WGL_NUMBER_UNDERLAYS_ARB:
|
||||
case WGL_TRANSPARENT_ARB:
|
||||
case WGL_SHARE_DEPTH_ARB:
|
||||
case WGL_SHARE_STENCIL_ARB:
|
||||
case WGL_SHARE_ACCUM_ARB:
|
||||
case WGL_SUPPORT_GDI_ARB:
|
||||
case WGL_STEREO_ARB:
|
||||
case WGL_ACCUM_BITS_ARB:
|
||||
case WGL_ACCUM_RED_BITS_ARB:
|
||||
case WGL_ACCUM_GREEN_BITS_ARB:
|
||||
case WGL_ACCUM_BLUE_BITS_ARB:
|
||||
case WGL_ACCUM_ALPHA_BITS_ARB:
|
||||
case WGL_AUX_BUFFERS_ARB:
|
||||
case WGL_SAMPLE_BUFFERS_ARB:
|
||||
case WGL_SAMPLES_ARB:
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_GetPixelFormatAttribivARB(HDC, int iPixelFormat, int iLayerPlane, UINT nAttributes,
|
||||
const int* piAttributes, int* piValues) {
|
||||
if (iLayerPlane != 0 || !piAttributes || !piValues) {
|
||||
return FALSE;
|
||||
}
|
||||
// Format 0 is only valid for WGL_NUMBER_PIXEL_FORMATS_ARB queries.
|
||||
if (iPixelFormat < 0 || iPixelFormat > kPixelFormatCount) {
|
||||
return FALSE;
|
||||
}
|
||||
const int format = iPixelFormat == 0 ? 1 : iPixelFormat;
|
||||
for (UINT i = 0; i < nAttributes; ++i) {
|
||||
piValues[i] = PixelFormatAttribValue(format, piAttributes[i]);
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_GetPixelFormatAttribfvARB(HDC hdc, int iPixelFormat, int iLayerPlane, UINT nAttributes,
|
||||
const int* piAttributes, FLOAT* pfValues) {
|
||||
if (!pfValues) {
|
||||
return FALSE;
|
||||
}
|
||||
Vector<int> values(nAttributes);
|
||||
if (!Ext_GetPixelFormatAttribivARB(hdc, iPixelFormat, iLayerPlane, nAttributes, piAttributes,
|
||||
values.data())) {
|
||||
return FALSE;
|
||||
}
|
||||
for (UINT i = 0; i < nAttributes; ++i) {
|
||||
pfValues[i] = static_cast<FLOAT>(values[i]);
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL WINAPI Ext_ChoosePixelFormatARB(HDC, const int* piAttribIList, const FLOAT*, UINT nMaxFormats,
|
||||
int* piFormats, UINT* nNumFormats) {
|
||||
if (!piFormats || !nNumFormats) {
|
||||
return FALSE;
|
||||
}
|
||||
int wantedStencil = 0;
|
||||
if (piAttribIList) {
|
||||
for (SizeT i = 0; piAttribIList[i] != 0; i += 2) {
|
||||
if (piAttribIList[i] == WGL_STENCIL_BITS_ARB) {
|
||||
wantedStencil = piAttribIList[i + 1];
|
||||
}
|
||||
}
|
||||
}
|
||||
UINT count = 0;
|
||||
const int preferred = wantedStencil > 0 ? 1 : 2;
|
||||
const int fallback = wantedStencil > 0 ? 2 : 1;
|
||||
if (count < nMaxFormats) {
|
||||
piFormats[count++] = preferred;
|
||||
}
|
||||
if (count < nMaxFormats) {
|
||||
piFormats[count++] = fallback;
|
||||
}
|
||||
*nNumFormats = count;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
struct WGLExtensionProc {
|
||||
const char* Name;
|
||||
PROC Proc;
|
||||
};
|
||||
|
||||
const WGLExtensionProc kWGLExtensionProcs[] = {
|
||||
{"wglGetExtensionsStringARB", reinterpret_cast<PROC>(Ext_GetExtensionsStringARB)},
|
||||
{"wglGetExtensionsStringEXT", reinterpret_cast<PROC>(Ext_GetExtensionsStringEXT)},
|
||||
{"wglCreateContextAttribsARB", reinterpret_cast<PROC>(Ext_CreateContextAttribsARB)},
|
||||
{"wglSwapIntervalEXT", reinterpret_cast<PROC>(Ext_SwapIntervalEXT)},
|
||||
{"wglGetSwapIntervalEXT", reinterpret_cast<PROC>(Ext_GetSwapIntervalEXT)},
|
||||
{"wglGetPixelFormatAttribivARB", reinterpret_cast<PROC>(Ext_GetPixelFormatAttribivARB)},
|
||||
{"wglGetPixelFormatAttribfvARB", reinterpret_cast<PROC>(Ext_GetPixelFormatAttribfvARB)},
|
||||
{"wglChoosePixelFormatARB", reinterpret_cast<PROC>(Ext_ChoosePixelFormatARB)},
|
||||
};
|
||||
} // namespace
|
||||
|
||||
int ChoosePixelFormat(HDC hdc, const PIXELFORMATDESCRIPTOR* pfd) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglChoosePixelFormat(hdc=%p)", hdc);
|
||||
// Format 1 (RGBA8 + depth24/stencil8) satisfies every request; a format
|
||||
// exceeding the asked-for capabilities is a legal ChoosePixelFormat answer.
|
||||
(void)pfd;
|
||||
return 1;
|
||||
}
|
||||
|
||||
int DescribePixelFormat(HDC hdc, int format, UINT size, PIXELFORMATDESCRIPTOR* pfd) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglDescribePixelFormat(hdc=%p, format=%d)", hdc, format);
|
||||
if (!pfd) {
|
||||
return kPixelFormatCount;
|
||||
}
|
||||
if (size < sizeof(PIXELFORMATDESCRIPTOR) || format < 1 || format > kPixelFormatCount) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const PixelFormatInfo& info = kPixelFormats[format - 1];
|
||||
std::memset(pfd, 0, sizeof(PIXELFORMATDESCRIPTOR));
|
||||
pfd->nSize = sizeof(PIXELFORMATDESCRIPTOR);
|
||||
pfd->nVersion = 1;
|
||||
pfd->dwFlags = PFD_DRAW_TO_WINDOW | PFD_SUPPORT_OPENGL | PFD_DOUBLEBUFFER | PFD_SWAP_EXCHANGE
|
||||
#if defined(PFD_SUPPORT_COMPOSITION)
|
||||
| PFD_SUPPORT_COMPOSITION
|
||||
#endif
|
||||
;
|
||||
pfd->iPixelType = PFD_TYPE_RGBA;
|
||||
pfd->cColorBits = 32;
|
||||
pfd->cRedBits = 8;
|
||||
pfd->cRedShift = 16;
|
||||
pfd->cGreenBits = 8;
|
||||
pfd->cGreenShift = 8;
|
||||
pfd->cBlueBits = 8;
|
||||
pfd->cBlueShift = 0;
|
||||
pfd->cAlphaBits = static_cast<BYTE>(info.AlphaBits);
|
||||
pfd->cAlphaShift = 24;
|
||||
pfd->cDepthBits = static_cast<BYTE>(info.DepthBits);
|
||||
pfd->cStencilBits = static_cast<BYTE>(info.StencilBits);
|
||||
pfd->iLayerType = PFD_MAIN_PLANE;
|
||||
return kPixelFormatCount;
|
||||
}
|
||||
|
||||
int GetPixelFormat(HDC hdc) {
|
||||
EnsureInitialized();
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
return 0;
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto& formats = WindowPixelFormats();
|
||||
auto it = formats.find(hwnd);
|
||||
return it == formats.end() ? 0 : it->second;
|
||||
}
|
||||
|
||||
BOOL SetPixelFormat(HDC hdc, int format, const PIXELFORMATDESCRIPTOR*) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglSetPixelFormat(hdc=%p, format=%d)", hdc, format);
|
||||
if (format < 1 || format > kPixelFormatCount) {
|
||||
SetLastError(ERROR_INVALID_PARAMETER);
|
||||
return FALSE;
|
||||
}
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
WindowPixelFormats()[hwnd] = format;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL SwapBuffers(HDC hdc) {
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto& surfaces = WindowSurfaces();
|
||||
auto it = surfaces.find(hwnd);
|
||||
if (it == surfaces.end()) {
|
||||
MGLOG_W("wglSwapBuffers: no surface for HWND %p", hwnd);
|
||||
return FALSE;
|
||||
}
|
||||
SyncSurfaceSize(hwnd, it->second);
|
||||
return EGLImpl::SwapBuffers(it->second.Display, it->second.Surface) == EGL_TRUE ? TRUE : FALSE;
|
||||
}
|
||||
|
||||
HGLRC CreateContext(HDC hdc) {
|
||||
EnsureInitialized();
|
||||
MGLOG_I("wglCreateContext(hdc=%p)", hdc);
|
||||
// A legacy WGL context is a compatibility-profile context; MobileGL keys
|
||||
// its relaxed-semantics mode off the explicit compatibility bit.
|
||||
const EGLint attribs[] = {
|
||||
EGL_CONTEXT_MAJOR_VERSION, 3,
|
||||
EGL_CONTEXT_MINOR_VERSION, 3,
|
||||
EGL_CONTEXT_OPENGL_PROFILE_MASK, EGL_CONTEXT_OPENGL_COMPATIBILITY_PROFILE_BIT,
|
||||
EGL_NONE,
|
||||
};
|
||||
return CreateContextFromEGLAttribs(hdc, nullptr, attribs);
|
||||
}
|
||||
|
||||
BOOL DeleteContext(HGLRC hglrc) {
|
||||
EnsureInitialized();
|
||||
MGLOG_I("wglDeleteContext(%p)", hglrc);
|
||||
if (t_current.Context == hglrc) {
|
||||
MakeCurrent(nullptr, nullptr);
|
||||
}
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(hglrc);
|
||||
if (!object) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
if (object->Context != EGL_NO_CONTEXT) {
|
||||
EGLImpl::DestroyContext(object->Display, object->Context);
|
||||
}
|
||||
Contexts().erase(hglrc);
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
BOOL MakeCurrent(HDC hdc, HGLRC hglrc) {
|
||||
EnsureInitialized();
|
||||
MGLOG_D("wglMakeCurrent(hdc=%p, hglrc=%p)", hdc, hglrc);
|
||||
if (!hglrc) {
|
||||
if (!t_current.Context) {
|
||||
t_current = {};
|
||||
return TRUE;
|
||||
}
|
||||
const EGLBoolean released =
|
||||
EGLImpl::MakeCurrent(EGL_NO_DISPLAY, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT);
|
||||
t_current = {};
|
||||
return released == EGL_TRUE ? TRUE : FALSE;
|
||||
}
|
||||
|
||||
HWND hwnd = WindowFromDC(hdc);
|
||||
if (!hwnd) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(hglrc);
|
||||
if (!object) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
WindowSurface* surface = EnsureWindowSurface(hwnd, *object);
|
||||
if (!surface) {
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
if (!EGLImpl::MakeCurrent(object->Display, surface->Surface, surface->Surface, object->Context)) {
|
||||
MGLOG_E("wglMakeCurrent: eglMakeCurrent failed (hdc=%p, hglrc=%p)", hdc, hglrc);
|
||||
return FALSE;
|
||||
}
|
||||
t_current = {hdc, hglrc};
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
HGLRC GetCurrentContext() {
|
||||
return t_current.Context;
|
||||
}
|
||||
|
||||
HDC GetCurrentDC() {
|
||||
return t_current.DC;
|
||||
}
|
||||
|
||||
BOOL ShareLists(HGLRC hglrcShare, HGLRC hglrcDest) {
|
||||
// All MobileGL contexts alias one global GL object namespace, so every
|
||||
// pair of contexts already shares.
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
if (!TryGetContext(hglrcShare) || !TryGetContext(hglrcDest)) {
|
||||
SetLastError(ERROR_INVALID_HANDLE);
|
||||
return FALSE;
|
||||
}
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
PROC GetProcAddress(const char* name) {
|
||||
EnsureInitialized();
|
||||
if (!name) {
|
||||
return nullptr;
|
||||
}
|
||||
if (name[0] == 'w' && name[1] == 'g' && name[2] == 'l') {
|
||||
for (const auto& entry : kWGLExtensionProcs) {
|
||||
if (std::strcmp(entry.Name, name) == 0) {
|
||||
return entry.Proc;
|
||||
}
|
||||
}
|
||||
MGLOG_D("wglGetProcAddress: unknown wgl entry point %s", name);
|
||||
return nullptr;
|
||||
}
|
||||
return reinterpret_cast<PROC>(MG_Impl::GetProcAddress(name));
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::WGLImpl
|
||||
|
||||
#endif // _WIN32
|
||||
@@ -0,0 +1,34 @@
|
||||
// MobileGL - MobileGL/MG_Impl/WGLImpl/WGLImpl.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
#if defined(_WIN32)
|
||||
|
||||
namespace MobileGL::MG_Impl::WGLImpl {
|
||||
// Classic opengl32.dll surface. gdi32's ChoosePixelFormat/SetPixelFormat/
|
||||
// DescribePixelFormat/GetPixelFormat/SwapBuffers forward into the loaded
|
||||
// opengl32.dll's wgl* exports, so these back both call paths.
|
||||
int ChoosePixelFormat(HDC hdc, const PIXELFORMATDESCRIPTOR* pfd);
|
||||
int DescribePixelFormat(HDC hdc, int format, UINT size, PIXELFORMATDESCRIPTOR* pfd);
|
||||
int GetPixelFormat(HDC hdc);
|
||||
BOOL SetPixelFormat(HDC hdc, int format, const PIXELFORMATDESCRIPTOR* pfd);
|
||||
BOOL SwapBuffers(HDC hdc);
|
||||
|
||||
HGLRC CreateContext(HDC hdc);
|
||||
BOOL DeleteContext(HGLRC hglrc);
|
||||
BOOL MakeCurrent(HDC hdc, HGLRC hglrc);
|
||||
HGLRC GetCurrentContext();
|
||||
HDC GetCurrentDC();
|
||||
BOOL ShareLists(HGLRC hglrcShare, HGLRC hglrcDest);
|
||||
|
||||
PROC GetProcAddress(const char* name);
|
||||
} // namespace MobileGL::MG_Impl::WGLImpl
|
||||
|
||||
#endif // _WIN32
|
||||
@@ -368,6 +368,26 @@ namespace MobileGL {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool EGLContext::HasAnyInitializedDisplay() const {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_mutex);
|
||||
for (const auto& [handle, displayObject] : m_displays) {
|
||||
if (displayObject.Initialized) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool EGLContext::HasAnyCurrentContext() const {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_mutex);
|
||||
for (const auto& [threadId, current] : m_threadCurrents) {
|
||||
if (current.Context != nullptr) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool EGLContext::ChooseConfig(EGLDisplayHandle display, const EGLint* attribList, EGLConfigHandle* configs,
|
||||
EGLint configSize, EGLint* numConfig) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_mutex);
|
||||
@@ -1415,6 +1435,7 @@ namespace MobileGL {
|
||||
}
|
||||
} // namespace EGLState
|
||||
|
||||
UniquePtr<EGLState::EGLContext> pEGLContext;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<EGLState::EGLContext>& pEGLContext = *new UniquePtr<EGLState::EGLContext>();
|
||||
} // namespace MG_State
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -39,6 +39,10 @@ namespace MobileGL {
|
||||
Bool IsDisplayInitialized(EGLDisplayHandle display) const;
|
||||
Bool InitializeDisplay(EGLDisplayHandle display, EGLint* major, EGLint* minor);
|
||||
Bool TerminateDisplay(EGLDisplayHandle display);
|
||||
// Whole-library idle checks used by EGLImpl::Terminate to decide
|
||||
// when the last eglTerminate may tear MobileGL down entirely.
|
||||
Bool HasAnyInitializedDisplay() const;
|
||||
Bool HasAnyCurrentContext() const;
|
||||
|
||||
// Config
|
||||
Bool ChooseConfig(EGLDisplayHandle display, const EGLint* attribList, EGLConfigHandle* configs,
|
||||
@@ -262,6 +266,6 @@ namespace MobileGL {
|
||||
};
|
||||
} // namespace EGLState
|
||||
|
||||
extern UniquePtr<EGLState::EGLContext> pEGLContext;
|
||||
extern UniquePtr<EGLState::EGLContext>& pEGLContext;
|
||||
} // namespace MG_State
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -721,5 +721,6 @@ namespace MobileGL::MG_State {
|
||||
}
|
||||
} // namespace GLState
|
||||
|
||||
UniquePtr<GLState::GLContext> pGLContext;
|
||||
// Leak-at-exit storage; see GlobalObjects.cpp.
|
||||
UniquePtr<GLState::GLContext>& pGLContext = *new UniquePtr<GLState::GLContext>();
|
||||
} // namespace MobileGL::MG_State
|
||||
|
||||
@@ -252,7 +252,7 @@ namespace MobileGL {
|
||||
};
|
||||
} // namespace GLState
|
||||
|
||||
extern UniquePtr<GLState::GLContext> pGLContext;
|
||||
extern UniquePtr<GLState::GLContext>& pGLContext;
|
||||
|
||||
// True when relaxed GL semantics apply. Strict core rules are enforced only when the
|
||||
// current EGL context explicitly requested a core profile (core bit in
|
||||
|
||||
@@ -334,16 +334,32 @@ namespace MobileGL::MG_State::GLState {
|
||||
// draw. The memo is keyed by (backendStateVersion, flags); ResetLinkArtifacts and
|
||||
// the binding setters below invalidate it by bumping m_backendStateVersion.
|
||||
Bool GetBackendHashMemo(Uint flags, Uint64& outHash) const {
|
||||
if (m_backendHashMemoVersion != m_backendStateVersion || m_backendHashMemoFlags != flags) {
|
||||
return false;
|
||||
if (m_backendHashMemoVersion != m_backendStateVersion) return false;
|
||||
for (const auto& slot : m_backendHashMemoSlots) {
|
||||
if (slot.valid && slot.flags == flags) {
|
||||
outHash = slot.hash;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
outHash = m_backendHashMemo;
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
void SetBackendHashMemo(Uint flags, Uint64 hash) const {
|
||||
m_backendHashMemo = hash;
|
||||
m_backendHashMemoVersion = m_backendStateVersion;
|
||||
m_backendHashMemoFlags = flags;
|
||||
if (m_backendHashMemoVersion != m_backendStateVersion) {
|
||||
for (auto& slot : m_backendHashMemoSlots) slot.valid = false;
|
||||
m_backendHashMemoVersion = m_backendStateVersion;
|
||||
m_backendHashMemoNextSlot = 0;
|
||||
}
|
||||
for (auto& slot : m_backendHashMemoSlots) {
|
||||
if (slot.valid && slot.flags == flags) {
|
||||
slot.hash = hash;
|
||||
return;
|
||||
}
|
||||
}
|
||||
auto& slot = m_backendHashMemoSlots[m_backendHashMemoNextSlot];
|
||||
slot.flags = flags;
|
||||
slot.hash = hash;
|
||||
slot.valid = true;
|
||||
m_backendHashMemoNextSlot = (m_backendHashMemoNextSlot + 1) % kBackendHashMemoSlotCount;
|
||||
}
|
||||
|
||||
void SetUniformSamplerOrImageUnitIndex(Uint location, Int unit) {
|
||||
@@ -527,10 +543,19 @@ namespace MobileGL::MG_State::GLState {
|
||||
Uint32 m_backendStateVersion = 0;
|
||||
|
||||
// Backend-owned content-hash memo (see GetBackendHashMemo): valid only while
|
||||
// m_backendStateVersion and the compile flags match the recorded values.
|
||||
mutable Uint64 m_backendHashMemo = 0;
|
||||
// m_backendStateVersion matches. Several slots, not one: a backend may resolve the same
|
||||
// program under more than one compile-flag set within a frame (surface rotation, and the
|
||||
// explicit-LOD sampling variant), and a single slot would then miss on every lookup and
|
||||
// re-hash the program's whole SPIR-V once per draw.
|
||||
static constexpr SizeT kBackendHashMemoSlotCount = 4;
|
||||
struct BackendHashMemoSlot {
|
||||
Uint64 hash = 0;
|
||||
Uint flags = 0;
|
||||
Bool valid = false;
|
||||
};
|
||||
mutable Array<BackendHashMemoSlot, kBackendHashMemoSlotCount> m_backendHashMemoSlots{};
|
||||
mutable SizeT m_backendHashMemoNextSlot = 0;
|
||||
mutable Uint32 m_backendHashMemoVersion = ~0u;
|
||||
mutable Uint m_backendHashMemoFlags = 0;
|
||||
Uint32 m_uboContentVersion = 0;
|
||||
Uint32 m_linkVersion = 0;
|
||||
};
|
||||
|
||||
@@ -16,17 +16,32 @@ namespace MobileGL {
|
||||
}
|
||||
|
||||
void MipmapStorage::AllocateLevel(Uint level, MipmapInput input) {
|
||||
m_data.reserve(std::bit_ceil(level + 1));
|
||||
m_data.resize(level + 1);
|
||||
m_texelSizes.reserve(std::bit_ceil(level + 1));
|
||||
m_texelSizes.resize(level + 1);
|
||||
m_texelSizes[level] = input.texelSize;
|
||||
m_isDirty.resize(level + 1, false);
|
||||
// Grow only. GL respecifies exactly the level it is handed, so allocating level 0
|
||||
// must not disturb the levels above it - but resize() shrinks as readily as it
|
||||
// grows, so this used to truncate the whole chain to a single level. Callers that
|
||||
// genuinely redefine the complete level set say so with TruncateToLevelCount.
|
||||
const SizeT requiredLevelCount = static_cast<SizeT>(level) + 1;
|
||||
if (m_data.size() < requiredLevelCount) {
|
||||
m_data.reserve(std::bit_ceil(requiredLevelCount));
|
||||
m_data.resize(requiredLevelCount);
|
||||
m_texelSizes.reserve(std::bit_ceil(requiredLevelCount));
|
||||
m_texelSizes.resize(requiredLevelCount);
|
||||
m_isDirty.resize(requiredLevelCount, false);
|
||||
}
|
||||
|
||||
m_texelSizes[level] = input.texelSize;
|
||||
auto& data = m_data[level];
|
||||
data.resize(input.byteSize, 0);
|
||||
}
|
||||
|
||||
void MipmapStorage::TruncateToLevelCount(SizeT levelCount) {
|
||||
if (levelCount >= m_data.size()) return;
|
||||
|
||||
m_data.resize(levelCount);
|
||||
m_texelSizes.resize(levelCount);
|
||||
m_isDirty.resize(levelCount);
|
||||
}
|
||||
|
||||
void MipmapStorage::UpdateSubData(Uint level, DataPtr input) {
|
||||
auto& targetData = m_data;
|
||||
MOBILEGL_ASSERT(level < targetData.size(), "UpdateSubData: level out of range");
|
||||
@@ -55,6 +70,7 @@ namespace MobileGL {
|
||||
}
|
||||
|
||||
SizeT MipmapStorage::GetByteSize(Uint level) const {
|
||||
if (level >= m_data.size()) return 0;
|
||||
return m_data[level].size();
|
||||
}
|
||||
|
||||
|
||||
@@ -19,6 +19,10 @@ namespace MobileGL {
|
||||
public:
|
||||
SizeT GetLevelCount() const;
|
||||
void AllocateLevel(Uint level, MipmapInput input);
|
||||
// Discard every level at or above levelCount. AllocateLevel never shrinks, so this
|
||||
// is the only way a chain gets shorter - use it where the caller defines the whole
|
||||
// level set (glTexStorage*, mip regeneration, atlas respecification).
|
||||
void TruncateToLevelCount(SizeT levelCount);
|
||||
void UpdateSubData(Uint level, DataPtr input);
|
||||
void* MapData(Uint level);
|
||||
IntVec3 GetTexelSize(Uint level) const;
|
||||
|
||||
@@ -29,6 +29,14 @@ namespace MobileGL {
|
||||
m_storage[targetIndex].AllocateLevel(level, input);
|
||||
}
|
||||
|
||||
// Per-target, like AllocateLevel: cube-map faces are respecified independently, so
|
||||
// truncating one face must not disturb the others.
|
||||
void TruncateToLevelCount(Uint targetIndex, SizeT levelCount) {
|
||||
MOBILEGL_ASSERT(targetIndex < TargetCount, "TruncateToLevelCount: target invalid");
|
||||
|
||||
m_storage[targetIndex].TruncateToLevelCount(levelCount);
|
||||
}
|
||||
|
||||
void UpdateSubData(Uint targetIndex, Uint level, DataPtr input) {
|
||||
MOBILEGL_ASSERT(targetIndex < TargetCount, "UpdateSubData: target invalid");
|
||||
m_storage[targetIndex].UpdateSubData(level, input);
|
||||
|
||||
@@ -271,6 +271,10 @@ namespace MobileGL {
|
||||
m_textureStorage.AllocateLevel(GetIndexOfTextureUploadTarget(uploadTarget), mipmapLevel, input);
|
||||
}
|
||||
|
||||
void TextureObjectWithOneMipmap::TruncateMipmapLevels(TextureUploadTarget uploadTarget, Uint levelCount) {
|
||||
m_textureStorage.TruncateToLevelCount(GetIndexOfTextureUploadTarget(uploadTarget), levelCount);
|
||||
}
|
||||
|
||||
void TextureObjectWithOneMipmap::UpdateMipmapSubData(TextureUploadTarget uploadTarget, Uint mipmapLevel,
|
||||
DataPtr input) {
|
||||
m_textureStorage.UpdateSubData(GetIndexOfTextureUploadTarget(uploadTarget), mipmapLevel, input);
|
||||
|
||||
@@ -15,7 +15,11 @@
|
||||
#include <MG_Util/Math/VectorTypes.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject {
|
||||
// Texture objects are always SharedPtr-owned (TextureState creates every instance via
|
||||
// MakeShared, including the per-target default objects). enable_shared_from_this lets
|
||||
// backends that only receive a reference (e.g. syncing a name-deleted texture kept
|
||||
// alive by an FBO attachment) still register a weak liveness reference for GC.
|
||||
class ITextureObject : public std::enable_shared_from_this<ITextureObject> {
|
||||
public:
|
||||
using TargetEnum = TextureTarget;
|
||||
virtual ~ITextureObject() = default;
|
||||
@@ -134,6 +138,10 @@ namespace MobileGL::MG_State::GLState {
|
||||
virtual const IntVec3 GetMipmapTexelSize(TextureUploadTarget target, Uint mipmapLevel) const = 0;
|
||||
virtual const SizeT GetMipmapByteSize(TextureUploadTarget target, Uint mipmapLevel) const = 0;
|
||||
virtual void AllocateStorage(TextureUploadTarget uploadTarget, Uint mipmapLevel, MipmapInput input) = 0;
|
||||
// AllocateStorage only ever grows the chain. Callers that define the complete level set -
|
||||
// glTexStorage*, mip regeneration, or a level-0 respecification at a new size - drop the
|
||||
// leftovers explicitly, so a stale tail can never make the texture silently incomplete.
|
||||
virtual void TruncateMipmapLevels(TextureUploadTarget uploadTarget, Uint levelCount) = 0;
|
||||
virtual void UpdateMipmapSubData(TextureUploadTarget uploadTarget, Uint mipmapLevel, DataPtr input) = 0;
|
||||
virtual void* MapMipmapData(TextureUploadTarget uploadTarget, Uint mipmapLevel) = 0;
|
||||
virtual void MarkStorageDirty(TextureUploadTarget uploadTarget, Uint mipmapLevel, Bool dirty = true) = 0;
|
||||
@@ -175,6 +183,7 @@ namespace MobileGL::MG_State::GLState {
|
||||
const IntVec3 GetMipmapTexelSize(TextureUploadTarget target, Uint mipmapLevel) const override;
|
||||
const SizeT GetMipmapByteSize(TextureUploadTarget target, Uint mipmapLevel) const override;
|
||||
void AllocateStorage(TextureUploadTarget uploadTarget, Uint mipmapLevel, MipmapInput input) override;
|
||||
void TruncateMipmapLevels(TextureUploadTarget uploadTarget, Uint levelCount) override;
|
||||
void UpdateMipmapSubData(TextureUploadTarget uploadTarget, Uint mipmapLevel, DataPtr input) override;
|
||||
void* MapMipmapData(TextureUploadTarget uploadTarget, Uint mipmapLevel) override;
|
||||
void MarkStorageDirty(TextureUploadTarget uploadTarget, Uint mipmapLevel, Bool dirty) override;
|
||||
|
||||
@@ -31,6 +31,10 @@ namespace MobileGL {
|
||||
m_textureStorage.AllocateLevel(GetIndexOfTextureUploadTarget(uploadTarget), mipmapLevel, input);
|
||||
}
|
||||
|
||||
void TextureObject2DCube::TruncateMipmapLevels(TextureUploadTarget uploadTarget, Uint levelCount) {
|
||||
m_textureStorage.TruncateToLevelCount(GetIndexOfTextureUploadTarget(uploadTarget), levelCount);
|
||||
}
|
||||
|
||||
void TextureObject2DCube::UpdateMipmapSubData(TextureUploadTarget uploadTarget, Uint mipmapLevel,
|
||||
DataPtr input) {
|
||||
m_textureStorage.UpdateSubData(GetIndexOfTextureUploadTarget(uploadTarget), mipmapLevel, input);
|
||||
|
||||
@@ -22,6 +22,7 @@ namespace MobileGL {
|
||||
const IntVec3 GetMipmapTexelSize(TextureUploadTarget target, Uint mipmapLevel) const override;
|
||||
const SizeT GetMipmapByteSize(TextureUploadTarget target, Uint mipmapLevel) const override;
|
||||
void AllocateStorage(TextureUploadTarget uploadTarget, Uint mipmapLevel, MipmapInput input) override;
|
||||
void TruncateMipmapLevels(TextureUploadTarget uploadTarget, Uint levelCount) override;
|
||||
void UpdateMipmapSubData(TextureUploadTarget uploadTarget, Uint mipmapLevel, DataPtr input) override;
|
||||
void* MapMipmapData(TextureUploadTarget uploadTarget, Uint mipmapLevel) override;
|
||||
void MarkStorageDirty(TextureUploadTarget uploadTarget, Uint mipmapLevel, bool dirty) override;
|
||||
|
||||
@@ -27,6 +27,13 @@ namespace {
|
||||
struct FakeDriverState {
|
||||
// Behavior knobs, configured per test before running the probe.
|
||||
GLint maxVertexSsboBlocks = 4;
|
||||
GLint glesMajorVersion = 3;
|
||||
GLint glesMinorVersion = 1;
|
||||
GLint maxVertexImageUniforms = 2;
|
||||
GLint maxGeometryImageUniforms = 3;
|
||||
GLint maxFragmentImageUniforms = 4;
|
||||
GLint maxComputeImageUniforms = 5;
|
||||
bool maxGeometryImageUniformsQueried = false;
|
||||
// Emulates ANGLE-on-Vulkan: the draw reads the indirect command's
|
||||
// baseInstance word and exposes it through gl_InstanceID.
|
||||
bool drawLeaksBaseInstanceWord = false;
|
||||
@@ -88,13 +95,26 @@ namespace {
|
||||
case GL_MAX_VERTEX_SHADER_STORAGE_BLOCKS:
|
||||
*data = g_fake.maxVertexSsboBlocks;
|
||||
break;
|
||||
case GL_MAX_VERTEX_IMAGE_UNIFORMS:
|
||||
*data = g_fake.maxVertexImageUniforms;
|
||||
break;
|
||||
case GL_MAX_GEOMETRY_IMAGE_UNIFORMS:
|
||||
g_fake.maxGeometryImageUniformsQueried = true;
|
||||
*data = g_fake.maxGeometryImageUniforms;
|
||||
break;
|
||||
case GL_MAX_FRAGMENT_IMAGE_UNIFORMS:
|
||||
*data = g_fake.maxFragmentImageUniforms;
|
||||
break;
|
||||
case GL_MAX_COMPUTE_IMAGE_UNIFORMS:
|
||||
*data = g_fake.maxComputeImageUniforms;
|
||||
break;
|
||||
// FillInGLESCapabilities reads the context version before running the
|
||||
// baseInstance probe, which requires ES >= 3.1.
|
||||
case GL_MAJOR_VERSION:
|
||||
*data = 3;
|
||||
*data = g_fake.glesMajorVersion;
|
||||
break;
|
||||
case GL_MINOR_VERSION:
|
||||
*data = 1;
|
||||
*data = g_fake.glesMinorVersion;
|
||||
break;
|
||||
case GL_NUM_EXTENSIONS:
|
||||
*data = static_cast<GLint>(g_fake.extensions.size());
|
||||
@@ -417,6 +437,31 @@ TEST(IndirectInstanceIdProbe, FillInCapabilitiesWiresProbeResult) {
|
||||
ExpectProbeReleasedAllObjects();
|
||||
}
|
||||
|
||||
TEST(ImageUniformCapabilities, QueriesRealPerStageLimitsAndConservativelyGatesGeometry) {
|
||||
const auto funcs = MakeFakeGLESFunctions();
|
||||
|
||||
ResetFakeDriver();
|
||||
g_fake.maxVertexSsboBlocks = 0;
|
||||
MobileGL::MG_External::GLESCapabilities es31Caps;
|
||||
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(es31Caps, funcs));
|
||||
EXPECT_EQ(es31Caps.MaxVertexImageUniforms, g_fake.maxVertexImageUniforms);
|
||||
EXPECT_EQ(es31Caps.MaxGeometryImageUniforms, 0);
|
||||
EXPECT_EQ(es31Caps.MaxFragmentImageUniforms, g_fake.maxFragmentImageUniforms);
|
||||
EXPECT_EQ(es31Caps.MaxComputeImageUniforms, g_fake.maxComputeImageUniforms);
|
||||
EXPECT_FALSE(g_fake.maxGeometryImageUniformsQueried);
|
||||
|
||||
ResetFakeDriver();
|
||||
g_fake.maxVertexSsboBlocks = 0;
|
||||
g_fake.glesMinorVersion = 2;
|
||||
MobileGL::MG_External::GLESCapabilities es32Caps;
|
||||
ASSERT_TRUE(MobileGL::MG_Util::BackendLoader::FillInGLESCapabilities(es32Caps, funcs));
|
||||
EXPECT_EQ(es32Caps.MaxVertexImageUniforms, g_fake.maxVertexImageUniforms);
|
||||
EXPECT_EQ(es32Caps.MaxGeometryImageUniforms, g_fake.maxGeometryImageUniforms);
|
||||
EXPECT_EQ(es32Caps.MaxFragmentImageUniforms, g_fake.maxFragmentImageUniforms);
|
||||
EXPECT_EQ(es32Caps.MaxComputeImageUniforms, g_fake.maxComputeImageUniforms);
|
||||
EXPECT_TRUE(g_fake.maxGeometryImageUniformsQueried);
|
||||
}
|
||||
|
||||
// The extension string is what apps gate on (LWJGL builds GLCapabilities from it), so advertising
|
||||
// it on a driver that cannot filter anisotropically would leave them silently on trilinear.
|
||||
TEST(TextureAnisotropyCapabilities, ExtensionIsAdvertisedOnlyWhenTheHostDriverSupportsIt) {
|
||||
|
||||
@@ -72,6 +72,8 @@ add_subdirectory(Texture)
|
||||
add_subdirectory(VertexArray)
|
||||
add_subdirectory(Program)
|
||||
add_subdirectory(Query)
|
||||
add_subdirectory(Pipeline)
|
||||
add_subdirectory(ShaderTranspiler)
|
||||
if (ENABLE_INTEGRATION_TESTS)
|
||||
add_subdirectory(Backend/DirectVulkan)
|
||||
endif()
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
|
||||
add_executable(
|
||||
PipelineQuirkTest
|
||||
PipelineQuirkTest.cpp
|
||||
)
|
||||
|
||||
target_include_directories(PipelineQuirkTest PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
${MGL_ROOT}/3rdparty/xxHash
|
||||
${MGL_ROOT}/3rdparty/Vulkan-Headers/include
|
||||
${MGL_ROOT}/3rdparty/SPIRV-Reflect
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
PipelineQuirkTest PRIVATE
|
||||
GTest::gtest_main
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
if (MSVC)
|
||||
target_compile_options(PipelineQuirkTest PRIVATE /Zc:preprocessor)
|
||||
endif()
|
||||
|
||||
include(GoogleTest)
|
||||
gtest_discover_tests(PipelineQuirkTest DISCOVERY_TIMEOUT 30 PROPERTIES LABELS unit)
|
||||
@@ -0,0 +1,459 @@
|
||||
// MobileGL - MobileGL/MG_Test/Pipeline/PipelineQuirkTest.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <Config.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/PipelineFactory.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/ProgramFactory.h>
|
||||
|
||||
using namespace MobileGL;
|
||||
using MobileGL::MG_Backend::DirectVulkan::PipelineFactory;
|
||||
using MobileGL::MG_Backend::DirectVulkan::ProgramFactory;
|
||||
using MobileGL::MG_Config::QuirkOverride;
|
||||
|
||||
namespace {
|
||||
constexpr Uint32 kVendorIdQualcomm = 0x5143;
|
||||
constexpr Uint32 kVendorIdArm = 0x13B5;
|
||||
|
||||
constexpr VkColorComponentFlags kFullColorWriteMask =
|
||||
VK_COLOR_COMPONENT_R_BIT | VK_COLOR_COMPONENT_G_BIT |
|
||||
VK_COLOR_COMPONENT_B_BIT | VK_COLOR_COMPONENT_A_BIT;
|
||||
|
||||
// Builds non-separate blend state: the alpha channel repeats the color factors/op, which
|
||||
// is what glBlendFunc/glBlendEquation (as opposed to their *Separate forms) produce.
|
||||
// ShouldSuppressDepthWrite deliberately decides on the color channel alone, so these
|
||||
// cases cover its whole input space; SeparateAlphaAccumulationIsNotStripped below pins
|
||||
// the separate-alpha contract.
|
||||
VkPipelineColorBlendAttachmentState MakeBlendAttachment(Bool blendEnable,
|
||||
VkBlendFactor srcColor,
|
||||
VkBlendFactor dstColor,
|
||||
VkBlendOp colorOp,
|
||||
VkColorComponentFlags colorWriteMask) {
|
||||
VkPipelineColorBlendAttachmentState attachment{};
|
||||
attachment.blendEnable = blendEnable ? VK_TRUE : VK_FALSE;
|
||||
attachment.srcColorBlendFactor = srcColor;
|
||||
attachment.dstColorBlendFactor = dstColor;
|
||||
attachment.colorBlendOp = colorOp;
|
||||
attachment.srcAlphaBlendFactor = srcColor;
|
||||
attachment.dstAlphaBlendFactor = dstColor;
|
||||
attachment.alphaBlendOp = colorOp;
|
||||
attachment.colorWriteMask = colorWriteMask;
|
||||
return attachment;
|
||||
}
|
||||
|
||||
// glslangValidator -V output for:
|
||||
// #version 450
|
||||
// layout(location = 0) out vec4 outColor;
|
||||
// void main() { outColor = vec4(1.0); gl_FragDepth = 0.5; }
|
||||
// Assigning gl_FragDepth makes glslang emit OpExecutionMode ... DepthReplacing.
|
||||
constexpr Uint32 kFragDepthWriterSpirv[] = {
|
||||
0x07230203u, 0x00010000u, 0x0008000bu, 0x0000000fu, 0x00000000u, 0x00020011u,
|
||||
0x00000001u, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
|
||||
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x0007000fu, 0x00000004u,
|
||||
0x00000004u, 0x6e69616du, 0x00000000u, 0x00000009u, 0x0000000du, 0x00030010u,
|
||||
0x00000004u, 0x00000007u, 0x00030010u, 0x00000004u, 0x0000000cu, 0x00030003u,
|
||||
0x00000002u, 0x000001c2u, 0x00040005u, 0x00000004u, 0x6e69616du, 0x00000000u,
|
||||
0x00050005u, 0x00000009u, 0x4374756fu, 0x726f6c6fu, 0x00000000u, 0x00060005u,
|
||||
0x0000000du, 0x465f6c67u, 0x44676172u, 0x68747065u, 0x00000000u, 0x00040047u,
|
||||
0x00000009u, 0x0000001eu, 0x00000000u, 0x00040047u, 0x0000000du, 0x0000000bu,
|
||||
0x00000016u, 0x00020013u, 0x00000002u, 0x00030021u, 0x00000003u, 0x00000002u,
|
||||
0x00030016u, 0x00000006u, 0x00000020u, 0x00040017u, 0x00000007u, 0x00000006u,
|
||||
0x00000004u, 0x00040020u, 0x00000008u, 0x00000003u, 0x00000007u, 0x0004003bu,
|
||||
0x00000008u, 0x00000009u, 0x00000003u, 0x0004002bu, 0x00000006u, 0x0000000au,
|
||||
0x3f800000u, 0x0007002cu, 0x00000007u, 0x0000000bu, 0x0000000au, 0x0000000au,
|
||||
0x0000000au, 0x0000000au, 0x00040020u, 0x0000000cu, 0x00000003u, 0x00000006u,
|
||||
0x0004003bu, 0x0000000cu, 0x0000000du, 0x00000003u, 0x0004002bu, 0x00000006u,
|
||||
0x0000000eu, 0x3f000000u, 0x00050036u, 0x00000002u, 0x00000004u, 0x00000000u,
|
||||
0x00000003u, 0x000200f8u, 0x00000005u, 0x0003003eu, 0x00000009u, 0x0000000bu,
|
||||
0x0003003eu, 0x0000000du, 0x0000000eu, 0x000100fdu, 0x00010038u,
|
||||
};
|
||||
|
||||
// Same shader without the gl_FragDepth assignment.
|
||||
constexpr Uint32 kPlainFragmentSpirv[] = {
|
||||
0x07230203u, 0x00010000u, 0x0008000bu, 0x0000000cu, 0x00000000u, 0x00020011u,
|
||||
0x00000001u, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
|
||||
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x0006000fu, 0x00000004u,
|
||||
0x00000004u, 0x6e69616du, 0x00000000u, 0x00000009u, 0x00030010u, 0x00000004u,
|
||||
0x00000007u, 0x00030003u, 0x00000002u, 0x000001c2u, 0x00040005u, 0x00000004u,
|
||||
0x6e69616du, 0x00000000u, 0x00050005u, 0x00000009u, 0x4374756fu, 0x726f6c6fu,
|
||||
0x00000000u, 0x00040047u, 0x00000009u, 0x0000001eu, 0x00000000u, 0x00020013u,
|
||||
0x00000002u, 0x00030021u, 0x00000003u, 0x00000002u, 0x00030016u, 0x00000006u,
|
||||
0x00000020u, 0x00040017u, 0x00000007u, 0x00000006u, 0x00000004u, 0x00040020u,
|
||||
0x00000008u, 0x00000003u, 0x00000007u, 0x0004003bu, 0x00000008u, 0x00000009u,
|
||||
0x00000003u, 0x0004002bu, 0x00000006u, 0x0000000au, 0x3f800000u, 0x0007002cu,
|
||||
0x00000007u, 0x0000000bu, 0x0000000au, 0x0000000au, 0x0000000au, 0x0000000au,
|
||||
0x00050036u, 0x00000002u, 0x00000004u, 0x00000000u, 0x00000003u, 0x000200f8u,
|
||||
0x00000005u, 0x0003003eu, 0x00000009u, 0x0000000bu, 0x000100fdu, 0x00010038u,
|
||||
};
|
||||
|
||||
|
||||
// glslangValidator -V output for a vertex shader reading gl_InstanceIndex:
|
||||
// #version 450
|
||||
// layout(location = 0) in vec4 inPos;
|
||||
// void main() { gl_Position = inPos + vec4(float(gl_InstanceIndex)); }
|
||||
constexpr Uint32 kInstanceIndexVertexSpirv[] = {
|
||||
0x07230203u, 0x00010000u, 0x0008000bu, 0x0000001bu, 0x00000000u, 0x00020011u,
|
||||
0x00000001u, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
|
||||
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x0008000fu, 0x00000000u,
|
||||
0x00000004u, 0x6e69616du, 0x00000000u, 0x0000000du, 0x00000011u, 0x00000014u,
|
||||
0x00030003u, 0x00000002u, 0x000001c2u, 0x00040005u, 0x00000004u, 0x6e69616du,
|
||||
0x00000000u, 0x00060005u, 0x0000000bu, 0x505f6c67u, 0x65567265u, 0x78657472u,
|
||||
0x00000000u, 0x00060006u, 0x0000000bu, 0x00000000u, 0x505f6c67u, 0x7469736fu,
|
||||
0x006e6f69u, 0x00070006u, 0x0000000bu, 0x00000001u, 0x505f6c67u, 0x746e696fu,
|
||||
0x657a6953u, 0x00000000u, 0x00070006u, 0x0000000bu, 0x00000002u, 0x435f6c67u,
|
||||
0x4470696cu, 0x61747369u, 0x0065636eu, 0x00070006u, 0x0000000bu, 0x00000003u,
|
||||
0x435f6c67u, 0x446c6c75u, 0x61747369u, 0x0065636eu, 0x00030005u, 0x0000000du,
|
||||
0x00000000u, 0x00040005u, 0x00000011u, 0x6f506e69u, 0x00000073u, 0x00070005u,
|
||||
0x00000014u, 0x495f6c67u, 0x6174736eu, 0x4965636eu, 0x7865646eu, 0x00000000u,
|
||||
0x00030047u, 0x0000000bu, 0x00000002u, 0x00050048u, 0x0000000bu, 0x00000000u,
|
||||
0x0000000bu, 0x00000000u, 0x00050048u, 0x0000000bu, 0x00000001u, 0x0000000bu,
|
||||
0x00000001u, 0x00050048u, 0x0000000bu, 0x00000002u, 0x0000000bu, 0x00000003u,
|
||||
0x00050048u, 0x0000000bu, 0x00000003u, 0x0000000bu, 0x00000004u, 0x00040047u,
|
||||
0x00000011u, 0x0000001eu, 0x00000000u, 0x00040047u, 0x00000014u, 0x0000000bu,
|
||||
0x0000002bu, 0x00020013u, 0x00000002u, 0x00030021u, 0x00000003u, 0x00000002u,
|
||||
0x00030016u, 0x00000006u, 0x00000020u, 0x00040017u, 0x00000007u, 0x00000006u,
|
||||
0x00000004u, 0x00040015u, 0x00000008u, 0x00000020u, 0x00000000u, 0x0004002bu,
|
||||
0x00000008u, 0x00000009u, 0x00000001u, 0x0004001cu, 0x0000000au, 0x00000006u,
|
||||
0x00000009u, 0x0006001eu, 0x0000000bu, 0x00000007u, 0x00000006u, 0x0000000au,
|
||||
0x0000000au, 0x00040020u, 0x0000000cu, 0x00000003u, 0x0000000bu, 0x0004003bu,
|
||||
0x0000000cu, 0x0000000du, 0x00000003u, 0x00040015u, 0x0000000eu, 0x00000020u,
|
||||
0x00000001u, 0x0004002bu, 0x0000000eu, 0x0000000fu, 0x00000000u, 0x00040020u,
|
||||
0x00000010u, 0x00000001u, 0x00000007u, 0x0004003bu, 0x00000010u, 0x00000011u,
|
||||
0x00000001u, 0x00040020u, 0x00000013u, 0x00000001u, 0x0000000eu, 0x0004003bu,
|
||||
0x00000013u, 0x00000014u, 0x00000001u, 0x00040020u, 0x00000019u, 0x00000003u,
|
||||
0x00000007u, 0x00050036u, 0x00000002u, 0x00000004u, 0x00000000u, 0x00000003u,
|
||||
0x000200f8u, 0x00000005u, 0x0004003du, 0x00000007u, 0x00000012u, 0x00000011u,
|
||||
0x0004003du, 0x0000000eu, 0x00000015u, 0x00000014u, 0x0004006fu, 0x00000006u,
|
||||
0x00000016u, 0x00000015u, 0x00070050u, 0x00000007u, 0x00000017u, 0x00000016u,
|
||||
0x00000016u, 0x00000016u, 0x00000016u, 0x00050081u, 0x00000007u, 0x00000018u,
|
||||
0x00000012u, 0x00000017u, 0x00050041u, 0x00000019u, 0x0000001au, 0x0000000du,
|
||||
0x0000000fu, 0x0003003eu, 0x0000001au, 0x00000018u, 0x000100fdu, 0x00010038u,
|
||||
};
|
||||
|
||||
|
||||
// Same, but reading gl_VertexIndex instead: a DIFFERENT input builtin. glslang emits
|
||||
// this for GL's gl_VertexID, so nearly every real vertex shader has one - it is what
|
||||
// separates "declares some builtin" from "declares the InstanceIndex builtin".
|
||||
constexpr Uint32 kVertexIndexVertexSpirv[] = {
|
||||
0x07230203u, 0x00010000u, 0x0008000bu, 0x0000001bu, 0x00000000u, 0x00020011u,
|
||||
0x00000001u, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
|
||||
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x0008000fu, 0x00000000u,
|
||||
0x00000004u, 0x6e69616du, 0x00000000u, 0x0000000du, 0x00000011u, 0x00000014u,
|
||||
0x00030003u, 0x00000002u, 0x000001c2u, 0x00040005u, 0x00000004u, 0x6e69616du,
|
||||
0x00000000u, 0x00060005u, 0x0000000bu, 0x505f6c67u, 0x65567265u, 0x78657472u,
|
||||
0x00000000u, 0x00060006u, 0x0000000bu, 0x00000000u, 0x505f6c67u, 0x7469736fu,
|
||||
0x006e6f69u, 0x00070006u, 0x0000000bu, 0x00000001u, 0x505f6c67u, 0x746e696fu,
|
||||
0x657a6953u, 0x00000000u, 0x00070006u, 0x0000000bu, 0x00000002u, 0x435f6c67u,
|
||||
0x4470696cu, 0x61747369u, 0x0065636eu, 0x00070006u, 0x0000000bu, 0x00000003u,
|
||||
0x435f6c67u, 0x446c6c75u, 0x61747369u, 0x0065636eu, 0x00030005u, 0x0000000du,
|
||||
0x00000000u, 0x00040005u, 0x00000011u, 0x6f506e69u, 0x00000073u, 0x00060005u,
|
||||
0x00000014u, 0x565f6c67u, 0x65747265u, 0x646e4978u, 0x00007865u, 0x00030047u,
|
||||
0x0000000bu, 0x00000002u, 0x00050048u, 0x0000000bu, 0x00000000u, 0x0000000bu,
|
||||
0x00000000u, 0x00050048u, 0x0000000bu, 0x00000001u, 0x0000000bu, 0x00000001u,
|
||||
0x00050048u, 0x0000000bu, 0x00000002u, 0x0000000bu, 0x00000003u, 0x00050048u,
|
||||
0x0000000bu, 0x00000003u, 0x0000000bu, 0x00000004u, 0x00040047u, 0x00000011u,
|
||||
0x0000001eu, 0x00000000u, 0x00040047u, 0x00000014u, 0x0000000bu, 0x0000002au,
|
||||
0x00020013u, 0x00000002u, 0x00030021u, 0x00000003u, 0x00000002u, 0x00030016u,
|
||||
0x00000006u, 0x00000020u, 0x00040017u, 0x00000007u, 0x00000006u, 0x00000004u,
|
||||
0x00040015u, 0x00000008u, 0x00000020u, 0x00000000u, 0x0004002bu, 0x00000008u,
|
||||
0x00000009u, 0x00000001u, 0x0004001cu, 0x0000000au, 0x00000006u, 0x00000009u,
|
||||
0x0006001eu, 0x0000000bu, 0x00000007u, 0x00000006u, 0x0000000au, 0x0000000au,
|
||||
0x00040020u, 0x0000000cu, 0x00000003u, 0x0000000bu, 0x0004003bu, 0x0000000cu,
|
||||
0x0000000du, 0x00000003u, 0x00040015u, 0x0000000eu, 0x00000020u, 0x00000001u,
|
||||
0x0004002bu, 0x0000000eu, 0x0000000fu, 0x00000000u, 0x00040020u, 0x00000010u,
|
||||
0x00000001u, 0x00000007u, 0x0004003bu, 0x00000010u, 0x00000011u, 0x00000001u,
|
||||
0x00040020u, 0x00000013u, 0x00000001u, 0x0000000eu, 0x0004003bu, 0x00000013u,
|
||||
0x00000014u, 0x00000001u, 0x00040020u, 0x00000019u, 0x00000003u, 0x00000007u,
|
||||
0x00050036u, 0x00000002u, 0x00000004u, 0x00000000u, 0x00000003u, 0x000200f8u,
|
||||
0x00000005u, 0x0004003du, 0x00000007u, 0x00000012u, 0x00000011u, 0x0004003du,
|
||||
0x0000000eu, 0x00000015u, 0x00000014u, 0x0004006fu, 0x00000006u, 0x00000016u,
|
||||
0x00000015u, 0x00070050u, 0x00000007u, 0x00000017u, 0x00000016u, 0x00000016u,
|
||||
0x00000016u, 0x00000016u, 0x00050081u, 0x00000007u, 0x00000018u, 0x00000012u,
|
||||
0x00000017u, 0x00050041u, 0x00000019u, 0x0000001au, 0x0000000du, 0x0000000fu,
|
||||
0x0003003eu, 0x0000001au, 0x00000018u, 0x000100fdu, 0x00010038u,
|
||||
};
|
||||
|
||||
// Owns the reflection module so each test case cleans up after itself.
|
||||
class ReflectModule {
|
||||
public:
|
||||
template <SizeT WordCount>
|
||||
explicit ReflectModule(const Uint32 (&spirv)[WordCount]) {
|
||||
m_created = spvReflectCreateShaderModule(sizeof(spirv), spirv, &m_module) ==
|
||||
SPV_REFLECT_RESULT_SUCCESS;
|
||||
}
|
||||
~ReflectModule() {
|
||||
if (m_created) {
|
||||
spvReflectDestroyShaderModule(&m_module);
|
||||
}
|
||||
}
|
||||
ReflectModule(const ReflectModule&) = delete;
|
||||
ReflectModule& operator=(const ReflectModule&) = delete;
|
||||
|
||||
Bool Created() const { return m_created; }
|
||||
const SpvReflectShaderModule& Get() const { return m_module; }
|
||||
|
||||
private:
|
||||
SpvReflectShaderModule m_module{};
|
||||
Bool m_created = false;
|
||||
};
|
||||
|
||||
PipelineFactory::PipelineCreatePayload MakeDepthWritingPayload(
|
||||
const VkPipelineColorBlendAttachmentState& attachment0) {
|
||||
PipelineFactory::PipelineCreatePayload payload{};
|
||||
payload.colorAttachmentCount = 1;
|
||||
payload.depthTestEnable = true;
|
||||
payload.depthWriteEnable = true;
|
||||
payload.colorBlendAttachments[0] = attachment0;
|
||||
return payload;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// --- Device gate: MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE tri-state ---
|
||||
|
||||
TEST(PipelineQuirkDeviceGate, ForceOnEnablesOnAnyVendor) {
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(QuirkOverride::ForceOn,
|
||||
kVendorIdArm));
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(QuirkOverride::ForceOn,
|
||||
kVendorIdQualcomm));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkDeviceGate, ForceOffDisablesEvenOnQualcomm) {
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(QuirkOverride::ForceOff,
|
||||
kVendorIdQualcomm));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkDeviceGate, AutoDetectsQualcommOnly) {
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(QuirkOverride::Auto,
|
||||
kVendorIdQualcomm));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(QuirkOverride::Auto,
|
||||
kVendorIdArm));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkDeviceGate, ForceOnRoundTripsThroughTheFactoryFlag) {
|
||||
const Bool previous = PipelineFactory::IsSuppressBlendedDepthWriteEnabled();
|
||||
PipelineFactory::SetSuppressBlendedDepthWrite(
|
||||
PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(QuirkOverride::ForceOn, kVendorIdArm));
|
||||
EXPECT_TRUE(PipelineFactory::IsSuppressBlendedDepthWriteEnabled());
|
||||
PipelineFactory::SetSuppressBlendedDepthWrite(previous);
|
||||
}
|
||||
|
||||
// --- Per-pipeline strip decision against the pipeline create-info payload ---
|
||||
|
||||
TEST(PipelineQuirkStripDecision, MaxBlendIsStripped) {
|
||||
// MC 26.3 OIT depth_bounds: GL_MAX accumulation writing depth - the case the quirk fixes.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ZERO, VK_BLEND_OP_MAX, kFullColorWriteMask));
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, MinBlendIsStripped) {
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ZERO, VK_BLEND_OP_MIN, kFullColorWriteMask));
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, AdditiveOnePlusOneIsNotStripped) {
|
||||
// ONE+ONE additive with a depth write matched zero draws of the 26.3 chain in the
|
||||
// fixture sweep (transmittance/accumulate disable depth writes themselves); the only
|
||||
// real content with this shape was harmless additive glow effects (Create). A quirk
|
||||
// touches as little unrelated content as possible, so the shape stays exempt.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, SortedTransparencyOverBlendIsNotStripped) {
|
||||
// Vanilla MC translucent layer (water, stained glass): SRC_ALPHA "over" compositing
|
||||
// draws each surface once and depends on its depth writes to occlude particles, rain,
|
||||
// and clouds drawn later - it must keep them.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_SRC_ALPHA, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA, VK_BLEND_OP_ADD,
|
||||
kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, EffectivelyOpaqueBlendIsNotStripped) {
|
||||
// GL_BLEND left enabled with ONE/ZERO+ADD factors is opaque in effect; stripping its
|
||||
// depth write would break occlusion for plainly opaque geometry.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ZERO, VK_BLEND_OP_ADD, kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, FullyMaskedAccumulationBlendIsNotStripped) {
|
||||
// Depth-prepass pattern: colorMask(0,0,0,0) with blending left enabled - blending is
|
||||
// moot, and stripping would delete the entire prepass. MAX so the exemption, not the
|
||||
// blend-op filter, is what keeps the depth write.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, 0));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, DisabledBlendIsNotStripped) {
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
false, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, NoDepthWriteMeansNoStrip) {
|
||||
auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask));
|
||||
payload.depthWriteEnable = false;
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, FragDepthWriterIsExempt) {
|
||||
// gl_FragDepth output does not go through per-pipeline vertex position math, so the
|
||||
// cross-pipeline invariance hazard cannot affect it (e.g. the 26.3 OIT composite).
|
||||
auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask));
|
||||
payload.fragmentReplacesDepth = true;
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, AccumulationOnSecondaryAttachmentIsStripped) {
|
||||
// The scan is not limited to attachment 0: an extremum accumulation on any live
|
||||
// attachment marks the pipeline.
|
||||
PipelineFactory::PipelineCreatePayload payload{};
|
||||
payload.colorAttachmentCount = 2;
|
||||
payload.depthTestEnable = true;
|
||||
payload.depthWriteEnable = true;
|
||||
payload.colorBlendAttachments[0] = MakeBlendAttachment(
|
||||
false, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ZERO, VK_BLEND_OP_ADD, kFullColorWriteMask);
|
||||
payload.colorBlendAttachments[1] = MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask);
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, AlphaWeightedAdditiveIsNotStripped) {
|
||||
// SRC_ALPHA,ONE additive: the classic *sorted* particle/glow blend. Kept exempt like
|
||||
// every other ADD-op shape now that the strip is extremum-only.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_SRC_ALPHA, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_ADD, kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, ReverseSubtractIsNotStripped) {
|
||||
// Deliberate narrowing: only the MIN/MAX extremum ops carry the depth-bounds
|
||||
// signature. SUBTRACT-class ops stay outside the quirk until content demands them.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_REVERSE_SUBTRACT,
|
||||
kFullColorWriteMask));
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, PartiallyMaskedAccumulationIsStripped) {
|
||||
// Only a fully masked attachment is exempt; a live alpha channel still accumulates.
|
||||
const auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, VK_COLOR_COMPONENT_A_BIT));
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, NoColorAttachmentsMeansNoStrip) {
|
||||
// Depth-only FBO: the loop must not read the (stale) attachment array at all.
|
||||
PipelineFactory::PipelineCreatePayload payload{};
|
||||
payload.colorAttachmentCount = 0;
|
||||
payload.depthTestEnable = true;
|
||||
payload.depthWriteEnable = true;
|
||||
payload.colorBlendAttachments[0] = MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask);
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, SeparateAlphaAccumulationIsNotStripped) {
|
||||
// glBlendEquationSeparate(GL_FUNC_ADD, GL_MAX) over an ordinary color over-blend: the
|
||||
// alpha channel accumulates but the color channel does not. Pins that the decision is
|
||||
// color-channel only - widening it to alpha would re-capture sorted transparency.
|
||||
auto attachment = MakeBlendAttachment(true, VK_BLEND_FACTOR_SRC_ALPHA,
|
||||
VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA, VK_BLEND_OP_ADD,
|
||||
kFullColorWriteMask);
|
||||
attachment.srcAlphaBlendFactor = VK_BLEND_FACTOR_ONE;
|
||||
attachment.dstAlphaBlendFactor = VK_BLEND_FACTOR_ONE;
|
||||
attachment.alphaBlendOp = VK_BLEND_OP_MAX;
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(MakeDepthWritingPayload(attachment)));
|
||||
}
|
||||
|
||||
TEST(PipelineQuirkStripDecision, MixedOverAndMaskedAttachmentsAreNotStripped) {
|
||||
PipelineFactory::PipelineCreatePayload payload{};
|
||||
payload.colorAttachmentCount = 2;
|
||||
payload.depthTestEnable = true;
|
||||
payload.depthWriteEnable = true;
|
||||
payload.colorBlendAttachments[0] = MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_SRC_ALPHA, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA, VK_BLEND_OP_ADD,
|
||||
kFullColorWriteMask);
|
||||
payload.colorBlendAttachments[1] = MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, 0);
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
// --- DepthReplacing reflection feeding the gl_FragDepth exemption ---
|
||||
|
||||
TEST(ReflectedFragmentReplacesDepth, TrueForAShaderThatAssignsFragDepth) {
|
||||
const ReflectModule module(kFragDepthWriterSpirv);
|
||||
ASSERT_TRUE(module.Created());
|
||||
EXPECT_TRUE(ProgramFactory::ReflectedFragmentReplacesDepth(module.Get()));
|
||||
}
|
||||
|
||||
TEST(ReflectedFragmentReplacesDepth, FalseForAPlainFragmentShader) {
|
||||
const ReflectModule module(kPlainFragmentSpirv);
|
||||
ASSERT_TRUE(module.Created());
|
||||
EXPECT_FALSE(ProgramFactory::ReflectedFragmentReplacesDepth(module.Get()));
|
||||
}
|
||||
|
||||
TEST(ReflectedFragmentReplacesDepth, FalseForAnEmptyModule) {
|
||||
// A default-constructed module has no entry points; the scan must not dereference.
|
||||
SpvReflectShaderModule emptyModule{};
|
||||
EXPECT_FALSE(ProgramFactory::ReflectedFragmentReplacesDepth(emptyModule));
|
||||
}
|
||||
|
||||
TEST(ReflectedFragmentReplacesDepth, ReflectedFlagFlipsTheStripDecision) {
|
||||
// The two fixtures differ only by the gl_FragDepth assignment, so they pin that the
|
||||
// reflected flag is what flips the strip decision for an otherwise identical pipeline.
|
||||
const ReflectModule depthWriter(kFragDepthWriterSpirv);
|
||||
const ReflectModule plain(kPlainFragmentSpirv);
|
||||
ASSERT_TRUE(depthWriter.Created());
|
||||
ASSERT_TRUE(plain.Created());
|
||||
|
||||
auto payload = MakeDepthWritingPayload(MakeBlendAttachment(
|
||||
true, VK_BLEND_FACTOR_ONE, VK_BLEND_FACTOR_ONE, VK_BLEND_OP_MAX, kFullColorWriteMask));
|
||||
|
||||
payload.fragmentReplacesDepth = ProgramFactory::ReflectedFragmentReplacesDepth(plain.Get());
|
||||
EXPECT_TRUE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
|
||||
payload.fragmentReplacesDepth = ProgramFactory::ReflectedFragmentReplacesDepth(depthWriter.Get());
|
||||
EXPECT_FALSE(PipelineFactory::ShouldSuppressDepthWrite(payload));
|
||||
}
|
||||
|
||||
|
||||
// --- InstanceIndex reflection feeding the shaderDrawParameters diagnostic ---
|
||||
|
||||
TEST(ReflectedReadsInstanceIndexBuiltin, TrueForAShaderReadingInstanceIndex) {
|
||||
const ReflectModule module(kInstanceIndexVertexSpirv);
|
||||
ASSERT_TRUE(module.Created());
|
||||
EXPECT_TRUE(ProgramFactory::ReflectedReadsInstanceIndexBuiltin(module.Get()));
|
||||
}
|
||||
|
||||
TEST(ReflectedReadsInstanceIndexBuiltin, FalseForAShaderReadingADifferentBuiltin) {
|
||||
// Discriminates the builtin's identity, not merely its presence: weakening the check to
|
||||
// "has any BuiltIn decoration" would fire the diagnostic on every real vertex shader.
|
||||
const ReflectModule module(kVertexIndexVertexSpirv);
|
||||
ASSERT_TRUE(module.Created());
|
||||
EXPECT_FALSE(ProgramFactory::ReflectedReadsInstanceIndexBuiltin(module.Get()));
|
||||
}
|
||||
|
||||
TEST(ReflectedReadsInstanceIndexBuiltin, FalseForAShaderWithNoInputBuiltins) {
|
||||
const ReflectModule module(kPlainFragmentSpirv);
|
||||
ASSERT_TRUE(module.Created());
|
||||
EXPECT_FALSE(ProgramFactory::ReflectedReadsInstanceIndexBuiltin(module.Get()));
|
||||
}
|
||||
|
||||
TEST(ReflectedReadsInstanceIndexBuiltin, FalseForAnEmptyModule) {
|
||||
SpvReflectShaderModule emptyModule{};
|
||||
EXPECT_FALSE(ProgramFactory::ReflectedReadsInstanceIndexBuiltin(emptyModule));
|
||||
}
|
||||
@@ -7,6 +7,7 @@
|
||||
// End of Source File Header
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
#include <spirv_reflect.h>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
|
||||
@@ -1325,6 +1326,88 @@ TEST_F(ProgramTest, CompileAndLinkWithExplicitVertexIn) {
|
||||
<< "\")";
|
||||
}
|
||||
|
||||
TEST_F(ProgramTest, InactiveExplicitVertexBindingsDoNotReserveLocations) {
|
||||
const char* vertexSource = R"(#version 430 compatibility
|
||||
|
||||
in vec3 Position;
|
||||
in vec2 UV0;
|
||||
in vec3 vaPosition;
|
||||
|
||||
void main() {
|
||||
gl_Position = vec4(vaPosition, 1.0);
|
||||
}
|
||||
)";
|
||||
const char* fragmentSource = R"(#version 430 compatibility
|
||||
|
||||
out vec4 fragColor;
|
||||
|
||||
void main() {
|
||||
fragColor = vec4(1.0);
|
||||
}
|
||||
)";
|
||||
|
||||
GLuint vertexShader = CreateShader(GL_VERTEX_SHADER);
|
||||
ShaderSource(vertexShader, 1, &vertexSource, nullptr);
|
||||
CompileShader(vertexShader);
|
||||
GLint compileStatus = GL_FALSE;
|
||||
GetShaderiv(vertexShader, GL_COMPILE_STATUS, &compileStatus);
|
||||
ASSERT_EQ(compileStatus, GL_TRUE);
|
||||
|
||||
GLuint fragmentShader = CreateShader(GL_FRAGMENT_SHADER);
|
||||
ShaderSource(fragmentShader, 1, &fragmentSource, nullptr);
|
||||
CompileShader(fragmentShader);
|
||||
GetShaderiv(fragmentShader, GL_COMPILE_STATUS, &compileStatus);
|
||||
ASSERT_EQ(compileStatus, GL_TRUE);
|
||||
|
||||
GLuint program = CreateProgram();
|
||||
AttachShader(program, vertexShader);
|
||||
AttachShader(program, fragmentShader);
|
||||
|
||||
// Iris binds these canonical names before linking every program. Its compatibility
|
||||
// transformer can inject both declarations even when the shader pack instead reads
|
||||
// vaPosition. Inactive API bindings must not consume locations during the link.
|
||||
BindAttribLocation(program, 0, "Position");
|
||||
BindAttribLocation(program, 1, "UV0");
|
||||
LinkProgram(program);
|
||||
|
||||
GLint linkStatus = GL_FALSE;
|
||||
GetProgramiv(program, GL_LINK_STATUS, &linkStatus);
|
||||
ASSERT_EQ(linkStatus, GL_TRUE);
|
||||
|
||||
EXPECT_EQ(GetAttribLocation(program, "Position"), -1);
|
||||
EXPECT_EQ(GetAttribLocation(program, "UV0"), -1);
|
||||
EXPECT_EQ(GetAttribLocation(program, "vaPosition"), 0);
|
||||
|
||||
auto programObject = MG_State::pGLContext->GetProgramObject(program);
|
||||
ASSERT_NE(programObject, nullptr);
|
||||
const Int vertexIndex = programObject->GetShaderIndexByStage(ShaderStage::Vertex);
|
||||
ASSERT_GE(vertexIndex, 0);
|
||||
const auto& spirvs = programObject->GetGeneratedSpirv();
|
||||
ASSERT_LT(static_cast<SizeT>(vertexIndex), spirvs.size());
|
||||
|
||||
const auto& vertexSpirv = spirvs[vertexIndex];
|
||||
spv_reflect::ShaderModule reflection(vertexSpirv.size() * sizeof(Uint), vertexSpirv.data());
|
||||
ASSERT_EQ(reflection.GetResult(), SPV_REFLECT_RESULT_SUCCESS);
|
||||
|
||||
uint32_t inputCount = 0;
|
||||
ASSERT_EQ(reflection.EnumerateInputVariables(&inputCount, nullptr), SPV_REFLECT_RESULT_SUCCESS);
|
||||
Vector<SpvReflectInterfaceVariable*> inputs(inputCount);
|
||||
ASSERT_EQ(reflection.EnumerateInputVariables(&inputCount, inputs.data()), SPV_REFLECT_RESULT_SUCCESS);
|
||||
|
||||
Uint32 userInputCount = 0;
|
||||
Uint32 locationMask = 0;
|
||||
for (const auto* input : inputs) {
|
||||
if (input == nullptr || (input->decoration_flags & SPV_REFLECT_DECORATION_BUILT_IN) != 0) {
|
||||
continue;
|
||||
}
|
||||
ASSERT_LT(input->location, 32u);
|
||||
locationMask |= 1u << input->location;
|
||||
++userInputCount;
|
||||
}
|
||||
EXPECT_EQ(userInputCount, 1u);
|
||||
EXPECT_EQ(locationMask, 0x1u);
|
||||
}
|
||||
|
||||
TEST_F(ProgramTest, CompileAndLinkWithExplicitFragmentOut) {
|
||||
char infoLog[1024] = "";
|
||||
|
||||
|
||||
@@ -8,7 +8,9 @@
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
#include "Includes.h"
|
||||
#include "Init.h"
|
||||
@@ -92,6 +94,120 @@ TEST_F(ProgramUtilTest, RenameSamplerFunctionParameterInSpirvPass) {
|
||||
EXPECT_EQ(exactSamplerNameCount, 1u);
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, UnformattedFloatStorageImagesKeepIntegerAtomicImagesTyped) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const String source = R"(#version 430 core
|
||||
layout(local_size_x = 1, local_size_y = 1, local_size_z = 1) in;
|
||||
layout(rgba16, binding = 0) uniform image2D floatImage;
|
||||
layout(r32ui, binding = 1) uniform uimage2D atomicImage;
|
||||
|
||||
void main() {
|
||||
ivec2 coordinate = ivec2(gl_GlobalInvocationID.xy);
|
||||
imageStore(floatImage, coordinate, imageLoad(floatImage, coordinate));
|
||||
imageAtomicAdd(atomicImage, coordinate, 1u);
|
||||
}
|
||||
)";
|
||||
|
||||
ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source};
|
||||
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
|
||||
ASSERT_TRUE(shaderResult) << shaderResult.error().log;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
|
||||
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
|
||||
ASSERT_TRUE(programResult) << programResult.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()};
|
||||
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
ASSERT_TRUE(binaryResult) << binaryResult.error().log;
|
||||
ASSERT_EQ(binaryResult->size(), 1u);
|
||||
const auto& inputBinary = binaryResult->front();
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String inputText;
|
||||
ASSERT_TRUE(tools.Disassemble(inputBinary, &inputText));
|
||||
EXPECT_NE(inputText.find("2D 0 0 0 2 Rgba16"), String::npos) << inputText;
|
||||
EXPECT_NE(inputText.find("2D 0 0 0 2 R32ui"), String::npos) << inputText;
|
||||
EXPECT_EQ(inputText.find("StorageImageReadWithoutFormat"), String::npos) << inputText;
|
||||
EXPECT_EQ(inputText.find("StorageImageWriteWithoutFormat"), String::npos) << inputText;
|
||||
|
||||
Vector<Uint32> outputBinary;
|
||||
ASSERT_TRUE(ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(inputBinary, outputBinary));
|
||||
|
||||
String outputText;
|
||||
ASSERT_TRUE(tools.Disassemble(outputBinary, &outputText));
|
||||
EXPECT_EQ(outputText.find("2D 0 0 0 2 Rgba16"), String::npos) << outputText;
|
||||
EXPECT_NE(outputText.find("2D 0 0 0 2 Unknown"), String::npos) << outputText;
|
||||
EXPECT_NE(outputText.find("2D 0 0 0 2 R32ui"), String::npos) << outputText;
|
||||
|
||||
const auto countOccurrences = [](const String& text, const String& needle) {
|
||||
SizeT count = 0;
|
||||
for (SizeT offset = 0; (offset = text.find(needle, offset)) != String::npos;
|
||||
offset += needle.size()) {
|
||||
++count;
|
||||
}
|
||||
return count;
|
||||
};
|
||||
EXPECT_EQ(countOccurrences(outputText, "OpCapability StorageImageReadWithoutFormat"), 1u)
|
||||
<< outputText;
|
||||
EXPECT_EQ(countOccurrences(outputText, "OpCapability StorageImageWriteWithoutFormat"), 1u)
|
||||
<< outputText;
|
||||
EXPECT_TRUE(tools.Validate(outputBinary));
|
||||
|
||||
Vector<Uint32> secondOutputBinary;
|
||||
ASSERT_TRUE(ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(outputBinary, secondOutputBinary));
|
||||
EXPECT_EQ(secondOutputBinary, outputBinary);
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, UnformattedFloatStorageImagesKeepFloatAtomicImageTypesTyped) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const String spirvText = R"(
|
||||
OpCapability Shader
|
||||
OpCapability StorageImageExtendedFormats
|
||||
OpMemoryModel Logical GLSL450
|
||||
OpEntryPoint GLCompute %main "main"
|
||||
OpExecutionMode %main LocalSize 1 1 1
|
||||
OpDecorate %target DescriptorSet 0
|
||||
OpDecorate %target Binding 0
|
||||
%void = OpTypeVoid
|
||||
%float = OpTypeFloat 32
|
||||
%int = OpTypeInt 32 1
|
||||
%v2int = OpTypeVector %int 2
|
||||
%image = OpTypeImage %float 2D 0 0 0 2 R32f
|
||||
%imageUniformPtr = OpTypePointer UniformConstant %image
|
||||
%imageTexelPtr = OpTypePointer Image %float
|
||||
%mainType = OpTypeFunction %void
|
||||
%zero = OpConstant %int 0
|
||||
%coordinate = OpConstantComposite %v2int %zero %zero
|
||||
%target = OpVariable %imageUniformPtr UniformConstant
|
||||
%main = OpFunction %void None %mainType
|
||||
%entry = OpLabel
|
||||
%texelPtr = OpImageTexelPointer %imageTexelPtr %target %coordinate %zero
|
||||
OpReturn
|
||||
OpFunctionEnd
|
||||
)";
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
Vector<Uint32> inputBinary;
|
||||
ASSERT_TRUE(tools.Assemble(spirvText, &inputBinary));
|
||||
|
||||
Vector<Uint32> outputBinary;
|
||||
ASSERT_TRUE(ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(inputBinary, outputBinary));
|
||||
|
||||
String outputText;
|
||||
ASSERT_TRUE(tools.Disassemble(outputBinary, &outputText));
|
||||
EXPECT_NE(outputText.find("2D 0 0 0 2 R32f"), String::npos) << outputText;
|
||||
EXPECT_EQ(outputText.find("StorageImageReadWithoutFormat"), String::npos) << outputText;
|
||||
EXPECT_EQ(outputText.find("StorageImageWriteWithoutFormat"), String::npos) << outputText;
|
||||
String validationDiagnostics;
|
||||
tools.SetMessageConsumer([&validationDiagnostics](spv_message_level_t, const char*,
|
||||
const spv_position_t&, const char* message) {
|
||||
validationDiagnostics += message;
|
||||
});
|
||||
EXPECT_TRUE(tools.Validate(outputBinary)) << validationDiagnostics;
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessLegacyVertexShaderModernizesGlmarkStyleSource) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -122,6 +238,77 @@ void main() {
|
||||
}
|
||||
}
|
||||
|
||||
// KHR-GL33.shaders.preprocessor.* — a block comment is one preprocessing token that the C/GLSL
|
||||
// preprocessor replaces with a single space, even when it spans newlines inside a directive. glslang
|
||||
// handles this natively, so MobileGL must not mangle it. These reproduce the CTS cases that failed
|
||||
// because comment blanking preserved the interior newline, truncating multi-line #define bodies.
|
||||
static void ExpectCompiles(MobileGL::ShaderStage stage, GLenum glStage, MobileGL::String source) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
PreprocessShaderSource(stage, source);
|
||||
ShaderAttrib attrib{.shaderType = glStage, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessMultilineCommentInDefineBodyCompiles) {
|
||||
ExpectCompiles(ShaderStage::Fragment, GL_FRAGMENT_SHADER,
|
||||
R"(#version 330
|
||||
precision mediump float;
|
||||
out float out0;
|
||||
#define VALUE /* current
|
||||
value */ 4.2
|
||||
|
||||
void main()
|
||||
{
|
||||
out0 = VALUE;
|
||||
})");
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessRedefineObjectMultilineCommentCompiles) {
|
||||
ExpectCompiles(ShaderStage::Fragment, GL_FRAGMENT_SHADER,
|
||||
R"(#version 330
|
||||
precision mediump float;
|
||||
out float out0;
|
||||
# define VAL1 1.0
|
||||
#define VAL2 2.0
|
||||
|
||||
#define RES2 /* fdsjklfdsjkl
|
||||
dsfjkhfdsjkh
|
||||
fdsjklhfdsjkh */ (RES1 * VAL2)
|
||||
#define RES1 (VAL2 / VAL1)
|
||||
#define RES2 /* ewrlkjhsadf */ (RES1 * VAL2)
|
||||
#define VALUE (RES2 + RES1)
|
||||
|
||||
void main()
|
||||
{
|
||||
out0 = VALUE;
|
||||
})");
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessFunctionMacroRedefinitionMultilineCommentCompiles) {
|
||||
ExpectCompiles(ShaderStage::Fragment, GL_FRAGMENT_SHADER,
|
||||
R"(#version 330
|
||||
precision mediump float;
|
||||
out float out0;
|
||||
# define FUNC(a,b) (a +b)
|
||||
# define FUNC(a,b)(a /* comment
|
||||
*/ +b)
|
||||
|
||||
void main()
|
||||
{
|
||||
out0 = FUNC(1.0, 2.0);
|
||||
})");
|
||||
}
|
||||
|
||||
// Note: KHR-GL3x.shaders.preprocessor.conditional_inclusion.basic_2 (`#define AAA defined(BBB)` used
|
||||
// in `#if !AAA`) is intentionally NOT handled here. Generating the `defined` operator via macro
|
||||
// expansion is undefined per the C/GLSL preprocessor spec, and glslang deliberately rejects it
|
||||
// ("'defined' : cannot use in preprocessor expression when expanded from macros"). Making it pass
|
||||
// would require MobileGL to run its own macro expansion ahead of glslang, which is exactly the
|
||||
// preprocessing we defer to glslang; the two cases stay failing by design.
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessLegacyFragmentShaderModernizesGlmarkStyleSource) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -310,6 +497,62 @@ void main() {
|
||||
verifyVersion("#version 460 core");
|
||||
}
|
||||
|
||||
// KHR-GL33.shaders.preprocessor.directive.version_* (also re-run verbatim under GL40-GL44): the
|
||||
// compiler must REJECT a malformed #version line. MobileGL used to rewrite the whole line to
|
||||
// "#version 330 core" whenever it could scrape a leading integer - or treat an unknown profile token
|
||||
// as core - which silently legalized every form below. CTS compiles the shader's own #version
|
||||
// verbatim, so the rejection has to survive preprocessing (and the 460 retry).
|
||||
TEST_F(ProgramUtilTest, PreprocessRejectsMalformedVersionDirectives) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const char* body = "\nout vec4 fragColor;\nvoid main() { fragColor = vec4(1.0); }\n";
|
||||
const auto rejects = [](const String& fullSource) {
|
||||
String src = fullSource;
|
||||
PreprocessShaderSource(ShaderStage::Fragment, src);
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = src};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
return res ? false : true; // "rejects" == compile failed
|
||||
};
|
||||
|
||||
// Silently legalized today - the five this fix must flip to rejection:
|
||||
EXPECT_TRUE(rejects(String("#version 329") + body)) << "329 is not a real version";
|
||||
EXPECT_TRUE(rejects(String("#version 331") + body)) << "331 is not a real version";
|
||||
EXPECT_TRUE(rejects(String("#version 330 foo") + body)) << "unknown profile keyword";
|
||||
EXPECT_TRUE(rejects(String("#version 330.0") + body)) << "float literal, not an int token";
|
||||
EXPECT_TRUE(rejects(String("#version 330 foobar") + body)) << "trailing tokens after a valid decl";
|
||||
|
||||
// Already rejected (no leading integer, or #version is not the first token) - pinned so a future
|
||||
// change to the normalizer cannot start legalizing them either:
|
||||
EXPECT_TRUE(rejects(String("#version") + body)) << "missing version number";
|
||||
EXPECT_TRUE(rejects(String("#version foobar") + body)) << "identifier where the int belongs";
|
||||
EXPECT_TRUE(rejects(String("#version AAA") + body)) << "identifier where the int belongs";
|
||||
EXPECT_TRUE(rejects(String("precision mediump float;\n#version 330") + body))
|
||||
<< "#version must be the first statement";
|
||||
EXPECT_TRUE(rejects(String("#define FOO BAR\n#version 330") + body))
|
||||
<< "#version must precede a #define";
|
||||
}
|
||||
|
||||
// The PASS half of the same CTS group: a valid decl, and #version preceded only by whitespace or a
|
||||
// comment, must still compile. Guards the fix above from over-rejecting.
|
||||
TEST_F(ProgramUtilTest, PreprocessKeepsValidVersionDirectivesCompiling) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const char* body = "\nout vec4 fragColor;\nvoid main() { fragColor = vec4(1.0); }\n";
|
||||
const auto compiles = [](const String& fullSource) {
|
||||
String src = fullSource;
|
||||
PreprocessShaderSource(ShaderStage::Fragment, src);
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = src};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
return res ? true : false;
|
||||
};
|
||||
|
||||
EXPECT_TRUE(compiles(String("#version 330 core") + body));
|
||||
EXPECT_TRUE(compiles(String("\n#version 330 core") + body))
|
||||
<< "leading whitespace is legal before #version";
|
||||
EXPECT_TRUE(compiles(String("// test\n#version 330 core") + body))
|
||||
<< "a leading comment is legal before #version";
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessUsesRealSpacedVersionDirectiveForInjectedOutput) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -330,6 +573,9 @@ void main() {
|
||||
EXPECT_NE(versionPos, String::npos);
|
||||
EXPECT_EQ(outputPos, versionPos + std::strlen("#version 330 core\n"));
|
||||
EXPECT_NE(source.find("// #version 460 core"), String::npos);
|
||||
// This #line sits ahead of the version directive, where GLSL would never have honoured it, so
|
||||
// it is still dropped. Directives that follow the version line are kept - see
|
||||
// PreprocessKeepsPlainLineDirectivesAndSparesLookalikeIdentifiers.
|
||||
EXPECT_EQ(source.find("#line"), String::npos);
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
@@ -339,6 +585,105 @@ void main() {
|
||||
}
|
||||
}
|
||||
|
||||
// A banner line like "//*** NOTE ***" contains "/*" at offset 1 and no "*/" anywhere after it. The
|
||||
// old hand-rolled comment stripper searched for "/*" with no lexical state, found that, failed to
|
||||
// find a terminator, and erased everything from there to the end of the file - deleting the entire
|
||||
// shader. Banner comments in that exact shape are common in Iris and OptiFine packs.
|
||||
TEST_F(ProgramUtilTest, PreprocessKeepsShaderBodyAfterAStarredLineComment) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
//*** lighting pass ***
|
||||
out vec4 fragColor;
|
||||
void main() {
|
||||
fragColor = vec4(1.0);
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_NE(source.find("void main()"), String::npos) << "shader body was truncated:\n" << source;
|
||||
EXPECT_NE(source.find("fragColor = vec4(1.0);"), String::npos);
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
// The builtin-shadowing rename only fires when the shader really defines its own round/tanh/etc.
|
||||
// Deciding that from a commented-out definition renames every genuine call to the builtin to a
|
||||
// mg_ name that nothing defines, which fails to link.
|
||||
TEST_F(ProgramUtilTest, PreprocessIgnoresCommentedOutBuiltinShadowingDefinition) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
// float round(float x) { return floor(x + 0.5); }
|
||||
out vec4 fragColor;
|
||||
void main() {
|
||||
fragColor = vec4(round(1.25));
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_NE(source.find("round(1.25)"), String::npos) << "call was renamed from a comment:\n" << source;
|
||||
EXPECT_EQ(source.find("mg_round"), String::npos);
|
||||
}
|
||||
|
||||
// A block-commented extension directive must not be treated as a real one - the int64 filter turns
|
||||
// unsupported directives into #error, so reading one out of a comment manufactures a compile
|
||||
// failure for a shader that never asked for the extension.
|
||||
TEST_F(ProgramUtilTest, PreprocessIgnoresBlockCommentedExtensionDirectives) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
/*
|
||||
#extension GL_ARB_gpu_shader_int64 : require
|
||||
*/
|
||||
out vec4 fragColor;
|
||||
void main() {
|
||||
fragColor = vec4(1.0);
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_EQ(source.find("#error"), String::npos) << "#error synthesized from a comment:\n" << source;
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
// KHR-GL33.shaders.preprocessor.builtin.line_* checks that __LINE__ follows #line. That only works
|
||||
// if the directive reaches glslang, so a plain integer form must pass through untouched - while
|
||||
// "#linear" and friends must not be mistaken for it.
|
||||
TEST_F(ProgramUtilTest, PreprocessKeepsPlainLineDirectivesAndSparesLookalikeIdentifiers) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
out vec4 fragColor;
|
||||
#line 42
|
||||
float linear(float x) { return x; }
|
||||
void main() {
|
||||
#line 100
|
||||
fragColor = vec4(linear(float(__LINE__)));
|
||||
}
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
|
||||
EXPECT_NE(source.find("#line 42"), String::npos) << source;
|
||||
EXPECT_NE(source.find("#line 100"), String::npos) << source;
|
||||
EXPECT_NE(source.find("float linear(float x)"), String::npos) << "identifier lookalike was eaten:\n" << source;
|
||||
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = source};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) {
|
||||
FAIL() << "errc: " << res.error().errc << "\nlog: " << res.error().log << "\nsource:\n" << source;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, PreprocessModernSampleQualifierStaysAtVersion460) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
@@ -629,6 +974,16 @@ TEST_F(ProgramUtilTest, RetargetLegacyVersionDirectiveOnlyTouchesNormalizedDeskt
|
||||
String commented = "// #version 330 core\nvoid main() {}\n";
|
||||
EXPECT_FALSE(RetargetLegacyVersionDirectiveTo460(commented));
|
||||
EXPECT_EQ(commented.find("#version 460"), String::npos);
|
||||
|
||||
// A malformed directive must NOT be rescued to 460 - that is what silently legalized the CTS
|
||||
// directive.version_* rejection cases. The bad version stays put so glslang keeps rejecting it.
|
||||
String badNumber = "#version 331\nvoid main() {}\n";
|
||||
EXPECT_FALSE(RetargetLegacyVersionDirectiveTo460(badNumber));
|
||||
EXPECT_EQ(badNumber.find("#version 460"), String::npos);
|
||||
|
||||
String badProfile = "#version 330 foo\nvoid main() {}\n";
|
||||
EXPECT_FALSE(RetargetLegacyVersionDirectiveTo460(badProfile));
|
||||
EXPECT_EQ(badProfile.find("#version 460"), String::npos);
|
||||
}
|
||||
|
||||
const char* fs = R"(#version 150
|
||||
@@ -805,6 +1160,353 @@ TEST_F(ProgramUtilTest, CompileFragmentShaderWithDiscard) {
|
||||
}
|
||||
}
|
||||
|
||||
// noperspective is core desktop GLSL (1.30+) and maps to the SPIR-V NoPerspective decoration. It must
|
||||
// reach glslang (not be stripped as text) so the SPIR-V carries the decoration; SPIRV-Cross then emits
|
||||
// ESSL `noperspective` + the GL_NV_shader_noperspective_interpolation extension. Shader packs
|
||||
// (Iris/Complementary) depend on it, and KHR-GL33.glsl_noperspective fails if the result matches
|
||||
// smooth. This is the DirectGLES path with the NV extension available (SPIRV-Cross's default).
|
||||
TEST_F(ProgramUtilTest, NoperspectiveInterpolationSurvivesToEssl) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String fs = R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 fragColor;
|
||||
void main() { fragColor = vColor; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile errc: " << res.error().errc << "\nlog: " << res.error().log;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {res.value()}};
|
||||
auto program_res = ShaderCompiler::LinkProgram(programAttrib);
|
||||
if (!program_res) FAIL() << "link errc: " << program_res.error().errc << "\nlog: " << program_res.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *program_res.value()};
|
||||
auto bin_res = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
if (!bin_res) FAIL() << "spirv errc: " << bin_res.error().errc << "\nlog: " << bin_res.error().log;
|
||||
ASSERT_EQ(bin_res.value().size(), 1u);
|
||||
|
||||
SpvcSession session(bin_res.value()[0], SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile errc: " << essl.error().errc << "\nlog: " << essl.error().log;
|
||||
|
||||
EXPECT_NE(essl.value().find("noperspective"), String::npos)
|
||||
<< "noperspective was lost before it reached SPIR-V:\n" << essl.value();
|
||||
EXPECT_NE(essl.value().find("GL_NV_shader_noperspective_interpolation"), String::npos)
|
||||
<< "SPIRV-Cross must require the NV extension for ES noperspective:\n" << essl.value();
|
||||
}
|
||||
|
||||
// The old handling was a naked substring erase of "noperspective", so any identifier that merely
|
||||
// contained those characters (a uniform named noperspectiveBlend, say) got mangled. Removing the
|
||||
// strip fixes it - glslang, which is identifier-aware, is the only thing that should see the keyword.
|
||||
TEST_F(ProgramUtilTest, PreprocessDoesNotCorruptIdentifiersContainingNoperspective) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = R"(#version 330 core
|
||||
uniform float noperspectiveBlend;
|
||||
out vec4 fragColor;
|
||||
void main() { fragColor = vec4(noperspectiveBlend); }
|
||||
)";
|
||||
PreprocessShaderSource(ShaderStage::Fragment, source);
|
||||
EXPECT_NE(source.find("noperspectiveBlend"), String::npos)
|
||||
<< "identifier was corrupted by substring stripping:\n" << source;
|
||||
}
|
||||
|
||||
// The DirectGLES fallback for devices without GL_NV_shader_noperspective_interpolation: stripping the
|
||||
// NoPerspective decoration makes SPIRV-Cross emit a plain smooth varying with no `#extension … :
|
||||
// require`, so the shader still compiles (rendering as smooth) instead of being rejected by the driver.
|
||||
TEST_F(ProgramUtilTest, StripNoPerspectiveFallbackProducesPlainEssl) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String fs = R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 fragColor;
|
||||
void main() { fragColor = vColor; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile errc: " << res.error().errc << "\nlog: " << res.error().log;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {res.value()}};
|
||||
auto program_res = ShaderCompiler::LinkProgram(programAttrib);
|
||||
if (!program_res) FAIL() << "link errc: " << program_res.error().errc << "\nlog: " << program_res.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *program_res.value()};
|
||||
auto bin_res = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
if (!bin_res) FAIL() << "spirv errc: " << bin_res.error().errc << "\nlog: " << bin_res.error().log;
|
||||
ASSERT_EQ(bin_res.value().size(), 1u);
|
||||
|
||||
// Precondition: with the decoration present the default decompile requires the NV extension.
|
||||
{
|
||||
SpvcSession session(bin_res.value()[0], SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile errc: " << essl.error().errc;
|
||||
ASSERT_NE(essl.value().find("noperspective"), String::npos) << essl.value();
|
||||
}
|
||||
|
||||
// The fallback strips the decoration -> plain smooth ESSL, no extension require.
|
||||
Vector<Uint32> stripped;
|
||||
ASSERT_TRUE(ShaderCompiler::StripNoPerspectiveForEssl(bin_res.value()[0], stripped));
|
||||
ASSERT_FALSE(stripped.empty());
|
||||
|
||||
SpvcSession session(stripped, SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile errc: " << essl.error().errc << "\nlog: " << essl.error().log;
|
||||
EXPECT_EQ(essl.value().find("noperspective"), String::npos)
|
||||
<< "the decoration should be gone:\n" << essl.value();
|
||||
EXPECT_EQ(essl.value().find("GL_NV_shader_noperspective_interpolation"), String::npos)
|
||||
<< "no extension require without the decoration:\n" << essl.value();
|
||||
}
|
||||
|
||||
// Directly exercises BOTH decoration forms StripNoPerspectivePass handles: a plain-variable
|
||||
// OpDecorate NoPerspective (in-operand 1) and an interface-block-member OpMemberDecorate NoPerspective
|
||||
// (in-operand 2). The ESSL round-trip tests above use only a scalar input, so they never reach the
|
||||
// member-decorate branch, which a block varying like `in Block { noperspective vec4 c; }` (common in
|
||||
// shader packs) produces. Unrelated decorations (Flat, Location) must survive untouched.
|
||||
TEST_F(ProgramUtilTest, StripNoPerspectivePassRemovesBothDecorateForms) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const String spirvText = R"(
|
||||
OpCapability Shader
|
||||
OpMemoryModel Logical GLSL450
|
||||
OpEntryPoint Fragment %main "main" %plainVar %blockVar %flatVar
|
||||
OpExecutionMode %main OriginUpperLeft
|
||||
OpName %main "main"
|
||||
OpDecorate %plainVar Location 0
|
||||
OpDecorate %plainVar NoPerspective
|
||||
OpMemberDecorate %Block 0 NoPerspective
|
||||
OpDecorate %blockVar Location 1
|
||||
OpDecorate %flatVar Location 2
|
||||
OpDecorate %flatVar Flat
|
||||
%void = OpTypeVoid
|
||||
%mainFn = OpTypeFunction %void
|
||||
%float = OpTypeFloat 32
|
||||
%v4float = OpTypeVector %float 4
|
||||
%int = OpTypeInt 32 1
|
||||
%inV4Ptr = OpTypePointer Input %v4float
|
||||
%plainVar = OpVariable %inV4Ptr Input
|
||||
%Block = OpTypeStruct %v4float
|
||||
%inBlockPtr = OpTypePointer Input %Block
|
||||
%blockVar = OpVariable %inBlockPtr Input
|
||||
%inIntPtr = OpTypePointer Input %int
|
||||
%flatVar = OpVariable %inIntPtr Input
|
||||
%main = OpFunction %void None %mainFn
|
||||
%mainBody = OpLabel
|
||||
OpReturn
|
||||
OpFunctionEnd
|
||||
)";
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
Vector<uint32_t> inputBinary;
|
||||
ASSERT_TRUE(tools.Assemble(spirvText, &inputBinary));
|
||||
|
||||
const auto countNoPerspective = [](const String& text) {
|
||||
SizeT count = 0, offset = 0;
|
||||
while ((offset = text.find("NoPerspective", offset)) != String::npos) {
|
||||
++count;
|
||||
offset += std::strlen("NoPerspective");
|
||||
}
|
||||
return count;
|
||||
};
|
||||
|
||||
String inputText;
|
||||
ASSERT_TRUE(tools.Disassemble(inputBinary, &inputText));
|
||||
ASSERT_EQ(countNoPerspective(inputText), 2u)
|
||||
<< "fixture must carry both a plain and a member NoPerspective:\n" << inputText;
|
||||
|
||||
Vector<uint32_t> outputBinary;
|
||||
ASSERT_TRUE(ShaderCompiler::StripNoPerspectiveForEssl(inputBinary, outputBinary));
|
||||
ASSERT_FALSE(outputBinary.empty());
|
||||
|
||||
String outputText;
|
||||
ASSERT_TRUE(tools.Disassemble(outputBinary, &outputText));
|
||||
EXPECT_EQ(countNoPerspective(outputText), 0u)
|
||||
<< "both NoPerspective decorations (OpDecorate and OpMemberDecorate) must be stripped:\n" << outputText;
|
||||
EXPECT_NE(outputText.find("Flat"), String::npos)
|
||||
<< "the unrelated Flat decoration must survive:\n" << outputText;
|
||||
EXPECT_NE(outputText.find("Location"), String::npos)
|
||||
<< "Location decorations must survive:\n" << outputText;
|
||||
}
|
||||
|
||||
// Phase 2 emulation - fragment side. On a device without the NV extension the NoPerspective input is
|
||||
// recovered as `load * gl_FragCoord.w` and the decoration removed; gl_FragCoord is synthesized because
|
||||
// the shader did not otherwise use it. The emulated SPIR-V must validate and decompile without the
|
||||
// extension require.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveFragmentRecoversWithFragCoordW) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String fs = R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 f;
|
||||
void main() { f = vColor; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile: " << res.error().log;
|
||||
ProgramAttrib pa{.shaders = {res.value()}};
|
||||
auto pr = ShaderCompiler::LinkProgram(pa);
|
||||
if (!pr) FAIL() << "link: " << pr.error().log;
|
||||
ProgramBinaryAttrib ba{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *pr.value()};
|
||||
auto br = ShaderCompiler::GetSpirvBinaryFromProgram(ba);
|
||||
if (!br) FAIL() << "spirv: " << br.error().log;
|
||||
ASSERT_EQ(br.value().size(), 1u);
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(br.value()[0], emulated));
|
||||
ASSERT_FALSE(emulated.empty());
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << "emulated SPIR-V must be valid:\n" << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << "decoration must be stripped:\n" << dis;
|
||||
EXPECT_NE(dis.find("FragCoord"), String::npos) << "gl_FragCoord must be synthesized:\n" << dis;
|
||||
EXPECT_NE(dis.find("OpVectorTimesScalar"), String::npos) << "the recovery multiply must be present:\n" << dis;
|
||||
|
||||
SpvcSession session(emulated, SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile: " << essl.error().log;
|
||||
EXPECT_EQ(essl.value().find("noperspective"), String::npos) << essl.value();
|
||||
EXPECT_EQ(essl.value().find("GL_NV_shader_noperspective_interpolation"), String::npos) << essl.value();
|
||||
EXPECT_NE(essl.value().find("gl_FragCoord"), String::npos) << "recovery must reference gl_FragCoord:\n" << essl.value();
|
||||
}
|
||||
|
||||
// Phase 2 emulation - vertex side. The NoPerspective output is pre-multiplied by gl_Position.w before
|
||||
// return and the decoration removed. Emulated SPIR-V must validate and decompile without the extension.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveVertexPreMultipliesByPositionW) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String vs = R"(#version 330 core
|
||||
in vec4 pos;
|
||||
noperspective out vec4 vColor;
|
||||
void main() { gl_Position = pos; vColor = pos; }
|
||||
)";
|
||||
ShaderAttrib attrib{.shaderType = GL_VERTEX_SHADER, .sourceStr = vs};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
if (!res) FAIL() << "compile: " << res.error().log;
|
||||
ProgramAttrib pa{.shaders = {res.value()}};
|
||||
auto pr = ShaderCompiler::LinkProgram(pa);
|
||||
if (!pr) FAIL() << "link: " << pr.error().log;
|
||||
ProgramBinaryAttrib ba{.shaderTypes = {GL_VERTEX_SHADER}, .program = *pr.value()};
|
||||
auto br = ShaderCompiler::GetSpirvBinaryFromProgram(ba);
|
||||
if (!br) FAIL() << "spirv: " << br.error().log;
|
||||
ASSERT_EQ(br.value().size(), 1u);
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(br.value()[0], emulated));
|
||||
ASSERT_FALSE(emulated.empty());
|
||||
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << "emulated SPIR-V must be valid:\n" << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << "decoration must be stripped:\n" << dis;
|
||||
EXPECT_NE(dis.find("OpVectorTimesScalar"), String::npos) << "the pre-multiply must be present:\n" << dis;
|
||||
|
||||
SpvcSession session(emulated, SessionUsageBit::Transpile);
|
||||
auto essl = ShaderCompiler::DecompileShader(session);
|
||||
if (!essl) FAIL() << "decompile: " << essl.error().log;
|
||||
EXPECT_EQ(essl.value().find("noperspective"), String::npos) << essl.value();
|
||||
EXPECT_NE(essl.value().find("gl_Position"), String::npos) << "pre-multiply must reference gl_Position:\n" << essl.value();
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Compiles one shader stage through the full pipeline and returns its SPIR-V, or fails the test.
|
||||
MobileGL::Vector<uint32_t> CompileStageSpirv(GLenum type, const char* src) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
ShaderAttrib attrib{.shaderType = type, .sourceStr = src};
|
||||
auto res = ShaderCompiler::CompileShader(attrib);
|
||||
EXPECT_TRUE(static_cast<bool>(res)) << (res ? "" : res.error().log);
|
||||
if (!res) return {};
|
||||
ProgramAttrib pa{.shaders = {res.value()}};
|
||||
auto pr = ShaderCompiler::LinkProgram(pa);
|
||||
EXPECT_TRUE(static_cast<bool>(pr)) << (pr ? "" : pr.error().log);
|
||||
if (!pr) return {};
|
||||
ProgramBinaryAttrib ba{.shaderTypes = {type}, .program = *pr.value()};
|
||||
auto br = ShaderCompiler::GetSpirvBinaryFromProgram(ba);
|
||||
EXPECT_TRUE(static_cast<bool>(br)) << (br ? "" : br.error().log);
|
||||
if (!br || br.value().empty()) return {};
|
||||
return br.value()[0];
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// Regression: the vertex pre-multiply must be applied exactly once (in main), not once per function.
|
||||
// glslang does not inline, so a helper function survives as its own OpFunction; instrumenting its
|
||||
// return too would scale the varying by gl_Position.w twice (w^2).
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveVertexWithHelperScalesExactlyOnce) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
// helper() returns via OpReturnValue and adds (no vector*scalar), so the ONLY OpVectorTimesScalar
|
||||
// in the module is the emulation's pre-multiply. The old all-functions code injected it at both
|
||||
// helper's and main's return -> count 2; restricted to the entry function it is 1.
|
||||
auto spirv = CompileStageSpirv(GL_VERTEX_SHADER, R"(#version 330 core
|
||||
in vec4 pos;
|
||||
noperspective out vec4 vColor;
|
||||
vec4 helper(vec4 x) { return x + vec4(1.0); }
|
||||
void main() { gl_Position = pos; vColor = helper(pos); }
|
||||
)");
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(spirv, emulated));
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << dis;
|
||||
|
||||
SizeT count = 0, off = 0;
|
||||
while ((off = dis.find("OpVectorTimesScalar", off)) != String::npos) {
|
||||
++count;
|
||||
off += std::strlen("OpVectorTimesScalar");
|
||||
}
|
||||
EXPECT_EQ(count, 1u) << "the gl_Position.w pre-multiply must happen exactly once, not per function:\n" << dis;
|
||||
}
|
||||
|
||||
// Regression: a single-component read (vColor.x), which glslang lowers via OpAccessChain, must still be
|
||||
// recovered with gl_FragCoord.w - not silently left un-scaled.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveFragmentComponentReadIsRecovered) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
auto spirv = CompileStageSpirv(GL_FRAGMENT_SHADER, R"(#version 330 core
|
||||
noperspective in vec4 vColor;
|
||||
out vec4 f;
|
||||
void main() { f = vec4(vColor.x); }
|
||||
)");
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(spirv, emulated));
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << dis;
|
||||
EXPECT_NE(dis.find("FragCoord"), String::npos)
|
||||
<< "the component read must still be recovered via gl_FragCoord.w:\n" << dis;
|
||||
}
|
||||
|
||||
// Coverage: a scalar float varying exercises the OpFMul path; a vector varying the OpVectorTimesScalar
|
||||
// path; multiple noperspective varyings in one stage are all handled.
|
||||
TEST_F(ProgramUtilTest, EmulateNoperspectiveHandlesScalarAndMultipleVaryings) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
auto spirv = CompileStageSpirv(GL_FRAGMENT_SHADER, R"(#version 330 core
|
||||
noperspective in float a;
|
||||
noperspective in vec2 b;
|
||||
out vec4 f;
|
||||
void main() { f = vec4(a, b, 1.0); }
|
||||
)");
|
||||
ASSERT_FALSE(spirv.empty());
|
||||
|
||||
Vector<uint32_t> emulated;
|
||||
ASSERT_TRUE(ShaderCompiler::EmulateNoPerspectiveForEssl(spirv, emulated));
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
String dis;
|
||||
ASSERT_TRUE(tools.Disassemble(emulated, &dis));
|
||||
ASSERT_TRUE(tools.Validate(emulated)) << dis;
|
||||
EXPECT_EQ(dis.find("NoPerspective"), String::npos) << dis;
|
||||
EXPECT_NE(dis.find("OpFMul"), String::npos) << "the scalar varying must scale with OpFMul:\n" << dis;
|
||||
EXPECT_NE(dis.find("OpVectorTimesScalar"), String::npos)
|
||||
<< "the vector varying must scale with OpVectorTimesScalar:\n" << dis;
|
||||
}
|
||||
|
||||
const char* vs_location = R"(#version 460
|
||||
|
||||
in vec4 Position;
|
||||
@@ -1383,3 +2085,150 @@ void main() {
|
||||
EXPECT_NE(source.find("shared float sharedScratch[8];"), String::npos);
|
||||
EXPECT_NE(source.find("layout(std140) uniform Blk"), String::npos);
|
||||
}
|
||||
|
||||
namespace {
|
||||
String MakeLinearSubgroupPrefixScanShader() {
|
||||
return R"(#version 460 core
|
||||
#extension GL_KHR_shader_subgroup_arithmetic : enable
|
||||
layout(local_size_x = 1024) in;
|
||||
shared float prefixSumCache[64];
|
||||
|
||||
layout(std430, binding = 0) writeonly buffer OutputBuffer {
|
||||
float outputValues[];
|
||||
};
|
||||
|
||||
void main() {
|
||||
float importance = 1.0f;
|
||||
float prefixSum = subgroupInclusiveAdd(importance);
|
||||
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||
barrier();
|
||||
uint loopLength = uint(findMSB(gl_NumSubgroups));
|
||||
loopLength += uint(gl_NumSubgroups - (1u << (loopLength - 1u)) > 0u);
|
||||
for (uint i = 0; i < loopLength; i++) {
|
||||
if ((gl_SubgroupID & (1u << i)) > 0u) {
|
||||
prefixSum += prefixSumCache[(gl_SubgroupID >> i << i) - 1u];
|
||||
if (gl_SubgroupInvocationID == gl_SubgroupSize - 1u) prefixSumCache[gl_SubgroupID] = prefixSum;
|
||||
}
|
||||
barrier();
|
||||
}
|
||||
if (gl_LocalInvocationID.x == uint(1024 - 1)) prefixSumCache[0] = prefixSum;
|
||||
barrier();
|
||||
float sum = prefixSumCache[0];
|
||||
float warp = (prefixSum - importance) / sum - float(gl_LocalInvocationID.x + 1u) / float(1024);
|
||||
outputValues[gl_GlobalInvocationID.x] = warp;
|
||||
}
|
||||
)";
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanUsesSharedMemoryAndProducesValidSpirv) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
String source = MakeLinearSubgroupPrefixScanShader();
|
||||
ASSERT_TRUE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
|
||||
|
||||
EXPECT_NE(source.find("shared float prefixSumCache[1024]"), String::npos) << source;
|
||||
EXPECT_NE(source.find("mglVirtualSubgroupInvocation"), String::npos) << source;
|
||||
EXPECT_NE(source.find("for (uint mglPrefixLane"), String::npos) << source;
|
||||
EXPECT_EQ(source.find("subgroupInclusiveAdd"), String::npos) << source;
|
||||
EXPECT_EQ(source.find("gl_Subgroup"), String::npos) << source;
|
||||
|
||||
const String onceRewritten = source;
|
||||
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
|
||||
EXPECT_EQ(source, onceRewritten);
|
||||
|
||||
ShaderAttrib shaderAttrib{.shaderType = GL_COMPUTE_SHADER, .sourceStr = source};
|
||||
auto shaderResult = ShaderCompiler::CompileShader(shaderAttrib);
|
||||
ASSERT_TRUE(shaderResult) << shaderResult.error().log << "\nsource:\n" << source;
|
||||
|
||||
ProgramAttrib programAttrib{.shaders = {shaderResult.value()}};
|
||||
auto programResult = ShaderCompiler::LinkProgram(programAttrib);
|
||||
ASSERT_TRUE(programResult) << programResult.error().log;
|
||||
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_COMPUTE_SHADER}, .program = *programResult.value()};
|
||||
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
ASSERT_TRUE(binaryResult) << binaryResult.error().log;
|
||||
ASSERT_EQ(binaryResult->size(), 1u);
|
||||
|
||||
String validationDiagnostics;
|
||||
spvtools::SpirvTools tools(SPV_ENV_VULKAN_1_1);
|
||||
tools.SetMessageConsumer([&](spv_message_level_t, const char*, const spv_position_t&, const char* message) {
|
||||
validationDiagnostics += message;
|
||||
validationDiagnostics += '\n';
|
||||
});
|
||||
EXPECT_TRUE(tools.Validate(binaryResult->front())) << validationDiagnostics;
|
||||
|
||||
String spirvText;
|
||||
ASSERT_TRUE(tools.Disassemble(binaryResult->front(), &spirvText));
|
||||
EXPECT_EQ(spirvText.find("OpGroupNonUniform"), String::npos) << spirvText;
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsOtherStagesAndSubgroupWidths) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const String original = MakeLinearSubgroupPrefixScanShader();
|
||||
for (const auto& [stage, subgroupSize] :
|
||||
{std::pair{ShaderStage::Compute, Uint32{32}}, std::pair{ShaderStage::Fragment, Uint32{64}},
|
||||
std::pair{ShaderStage::Compute, Uint32{96}}}) {
|
||||
String source = original;
|
||||
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(stage, subgroupSize, source));
|
||||
EXPECT_EQ(source, original);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(ProgramUtilTest, RewriteLinearSubgroupPrefixScanRejectsPartialOrUnsafeTemplateMatches) {
|
||||
using namespace MG_Util::ShaderTranspiler;
|
||||
|
||||
const auto expectUnchanged = [](String source) {
|
||||
const String original = source;
|
||||
EXPECT_FALSE(RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage::Compute, 64, source));
|
||||
EXPECT_EQ(source, original);
|
||||
};
|
||||
|
||||
String wrongLocalSize = MakeLinearSubgroupPrefixScanShader();
|
||||
wrongLocalSize.replace(wrongLocalSize.find("local_size_x = 1024"), std::strlen("local_size_x = 1024"),
|
||||
"local_size_x = 512");
|
||||
expectUnchanged(std::move(wrongLocalSize));
|
||||
|
||||
String cacheHasAnotherUse = MakeLinearSubgroupPrefixScanShader();
|
||||
cacheHasAnotherUse.insert(cacheHasAnotherUse.find("float importance"), "prefixSumCache[0] = 0.0f;\n ");
|
||||
expectUnchanged(std::move(cacheHasAnotherUse));
|
||||
|
||||
String extraSubgroupBuiltin = MakeLinearSubgroupPrefixScanShader();
|
||||
extraSubgroupBuiltin.insert(extraSubgroupBuiltin.find("float importance"),
|
||||
"uvec4 extraMask = gl_SubgroupEqMask;\n ");
|
||||
expectUnchanged(std::move(extraSubgroupBuiltin));
|
||||
|
||||
String alteredBarrier = MakeLinearSubgroupPrefixScanShader();
|
||||
alteredBarrier.replace(alteredBarrier.find("barrier();"), std::strlen("barrier();"), "memoryBarrierShared();");
|
||||
expectUnchanged(std::move(alteredBarrier));
|
||||
|
||||
String nestedScan = MakeLinearSubgroupPrefixScanShader();
|
||||
nestedScan.insert(nestedScan.find("float prefixSum ="), "if (importance > 0.0f) {\n ");
|
||||
const SizeT consumerEnd = nestedScan.find(';', nestedScan.find("float warp ="));
|
||||
ASSERT_NE(consumerEnd, String::npos);
|
||||
nestedScan.insert(consumerEnd + 1, "\n }");
|
||||
expectUnchanged(std::move(nestedScan));
|
||||
|
||||
// ARB/NV spellings of lane-width-sensitive builtins must block the rewrite exactly
|
||||
// like their KHR counterparts.
|
||||
String arbSubgroupBuiltin = MakeLinearSubgroupPrefixScanShader();
|
||||
arbSubgroupBuiltin.insert(arbSubgroupBuiltin.find("float importance"),
|
||||
"uint arbLane = gl_SubGroupInvocationARB;\n ");
|
||||
expectUnchanged(std::move(arbSubgroupBuiltin));
|
||||
|
||||
String arbBallotCall = MakeLinearSubgroupPrefixScanShader();
|
||||
arbBallotCall.insert(arbBallotCall.find("float importance"),
|
||||
"uint64_t arbMask = ballotARB(true);\n ");
|
||||
expectUnchanged(std::move(arbBallotCall));
|
||||
|
||||
String nvWarpBuiltin = MakeLinearSubgroupPrefixScanShader();
|
||||
nvWarpBuiltin.insert(nvWarpBuiltin.find("float importance"),
|
||||
"uint warpSize = gl_WarpSizeNV;\n ");
|
||||
expectUnchanged(std::move(nvWarpBuiltin));
|
||||
|
||||
String nvShuffleCall = MakeLinearSubgroupPrefixScanShader();
|
||||
nvShuffleCall.insert(nvShuffleCall.find("float importance"),
|
||||
"float other = shuffleNV(1.0f, 0u, 32u);\n ");
|
||||
expectUnchanged(std::move(nvShuffleCall));
|
||||
}
|
||||
|
||||
@@ -24,12 +24,16 @@
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/GLState/FramebufferState/FramebufferObject.h>
|
||||
#include <MG_State/GLState/TextureState/TextureState.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/ProgramFactory.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/UniformManager.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/VkRenderPassManager.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/VkTextureManager.h>
|
||||
#include <MG_Backend/DirectVulkan/Renderer/VulkanRenderer.h>
|
||||
#include <MG_Util/Math/HalfFloat.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderSourceProcessor.h>
|
||||
#include <MG_Util/Debug/Log.h>
|
||||
#include <FastSTL/UnorderedMap.h>
|
||||
|
||||
namespace {
|
||||
class DynamicParameterBackend final : public MobileGL::MG_Backend::BackendObject {
|
||||
@@ -460,6 +464,58 @@ TEST(DirectVulkanSanity, ClampsAdvertisedTextureAndDrawBufferLimitsToFrontendSta
|
||||
EXPECT_EQ(lowParams.MaxColorAttachments, 6);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, GatesPerStageImageUniformLimitsOnPhysicalDeviceFeatures) {
|
||||
using namespace MobileGL;
|
||||
|
||||
MG_Backend::DirectVulkan::BackendObject_DirectVulkan backend;
|
||||
MG_External::VulkanCapabilities caps;
|
||||
caps.MaxImageUnits = 12;
|
||||
caps.MaxCombinedImageUniforms = 10;
|
||||
caps.MaxComputeImageUniforms = 9;
|
||||
caps.SupportsVertexPipelineStoresAndAtomics = true;
|
||||
caps.SupportsFragmentStoresAndAtomics = true;
|
||||
caps.SupportsGeometryShader = false;
|
||||
backend.ApplyVulkanCapabilitiesForTesting(caps);
|
||||
|
||||
const auto& withoutGeometry = backend.GetDynamicParameters();
|
||||
EXPECT_EQ(withoutGeometry.MaxVertexImageUniforms, 10);
|
||||
EXPECT_EQ(withoutGeometry.MaxGeometryImageUniforms, 0);
|
||||
EXPECT_EQ(withoutGeometry.MaxFragmentImageUniforms, 10);
|
||||
EXPECT_EQ(withoutGeometry.MaxComputeImageUniforms, 9);
|
||||
|
||||
caps.SupportsGeometryShader = true;
|
||||
backend.ApplyVulkanCapabilitiesForTesting(caps);
|
||||
EXPECT_EQ(backend.GetDynamicParameters().MaxGeometryImageUniforms, 10);
|
||||
|
||||
caps.SupportsVertexPipelineStoresAndAtomics = false;
|
||||
caps.SupportsFragmentStoresAndAtomics = false;
|
||||
backend.ApplyVulkanCapabilitiesForTesting(caps);
|
||||
EXPECT_EQ(backend.GetDynamicParameters().MaxVertexImageUniforms, 0);
|
||||
EXPECT_EQ(backend.GetDynamicParameters().MaxGeometryImageUniforms, 0);
|
||||
EXPECT_EQ(backend.GetDynamicParameters().MaxFragmentImageUniforms, 0);
|
||||
EXPECT_EQ(backend.GetDynamicParameters().MaxComputeImageUniforms, 9);
|
||||
}
|
||||
|
||||
TEST(DirectGLESSanity, PreservesHostPerStageImageUniformLimits) {
|
||||
using namespace MobileGL;
|
||||
|
||||
MG_Backend::DirectGLES::BackendObject_DirectGLES backend;
|
||||
MG_External::GLESCapabilities caps;
|
||||
caps.MaxImageUnits = 8;
|
||||
caps.MaxCombinedImageUniforms = 16;
|
||||
caps.MaxVertexImageUniforms = 2;
|
||||
caps.MaxGeometryImageUniforms = 3;
|
||||
caps.MaxFragmentImageUniforms = 4;
|
||||
caps.MaxComputeImageUniforms = 5;
|
||||
backend.ApplyGLESCapabilitiesForTesting(caps);
|
||||
|
||||
const auto& params = backend.GetDynamicParameters();
|
||||
EXPECT_EQ(params.MaxVertexImageUniforms, 2);
|
||||
EXPECT_EQ(params.MaxGeometryImageUniforms, 3);
|
||||
EXPECT_EQ(params.MaxFragmentImageUniforms, 4);
|
||||
EXPECT_EQ(params.MaxComputeImageUniforms, 5);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, AdvertisesSubgroupOnlyWhenVulkanReportsUsableSupport) {
|
||||
using namespace MobileGL;
|
||||
|
||||
@@ -582,6 +638,54 @@ TEST(GetterSanity, ClampsMaxVertexAttribsToCurrentValueStorageCapacity) {
|
||||
MG_State::pGLContext.reset();
|
||||
}
|
||||
|
||||
TEST(GetterSanity, PerStageImageUniformQueriesMatchShaderCompilerLimits) {
|
||||
using namespace MobileGL;
|
||||
|
||||
MG_Backend::DynamicBackendParameters params;
|
||||
params.MaxImageUnits = 8;
|
||||
params.MaxCombinedImageUniforms = 8;
|
||||
params.MaxVertexImageUniforms = 1;
|
||||
params.MaxGeometryImageUniforms = 2;
|
||||
params.MaxFragmentImageUniforms = 3;
|
||||
params.MaxComputeImageUniforms = 4;
|
||||
MG_Backend::pActiveBackendObject = MakeUnique<DynamicParameterBackend>(params);
|
||||
|
||||
GLint reported = -1;
|
||||
MG_Impl::GLImpl::GetIntegerv(GL_MAX_VERTEX_IMAGE_UNIFORMS, &reported);
|
||||
EXPECT_EQ(reported, 1);
|
||||
MG_Impl::GLImpl::GetIntegerv(GL_MAX_GEOMETRY_IMAGE_UNIFORMS, &reported);
|
||||
EXPECT_EQ(reported, 2);
|
||||
MG_Impl::GLImpl::GetIntegerv(GL_MAX_FRAGMENT_IMAGE_UNIFORMS, &reported);
|
||||
EXPECT_EQ(reported, 3);
|
||||
MG_Impl::GLImpl::GetIntegerv(GL_MAX_COMPUTE_IMAGE_UNIFORMS, &reported);
|
||||
EXPECT_EQ(reported, 4);
|
||||
|
||||
const String vertexImageStore = R"(#version 430 core
|
||||
layout(r32ui, binding = 0) uniform uimage2D targetImages[gl_MaxVertexImageUniforms];
|
||||
void main() {
|
||||
imageStore(targetImages[0], ivec2(0), uvec4(1));
|
||||
gl_Position = vec4(0.0, 0.0, 0.0, 1.0);
|
||||
}
|
||||
)";
|
||||
auto supported = MG_Util::ShaderTranspiler::ShaderCompiler::CompileShader({
|
||||
.shaderType = GL_VERTEX_SHADER,
|
||||
.sourceStr = vertexImageStore,
|
||||
});
|
||||
EXPECT_TRUE(supported) << (supported ? "" : supported.error().log);
|
||||
|
||||
params.MaxVertexImageUniforms = 0;
|
||||
MG_Backend::pActiveBackendObject = MakeUnique<DynamicParameterBackend>(params);
|
||||
MG_Impl::GLImpl::GetIntegerv(GL_MAX_VERTEX_IMAGE_UNIFORMS, &reported);
|
||||
EXPECT_EQ(reported, 0);
|
||||
auto unsupported = MG_Util::ShaderTranspiler::ShaderCompiler::CompileShader({
|
||||
.shaderType = GL_VERTEX_SHADER,
|
||||
.sourceStr = vertexImageStore,
|
||||
});
|
||||
EXPECT_FALSE(unsupported);
|
||||
|
||||
MG_Backend::pActiveBackendObject.reset();
|
||||
}
|
||||
|
||||
TEST(GetterSanity, ReportsKhrSubgroupDynamicParameters) {
|
||||
using namespace MobileGL;
|
||||
|
||||
@@ -629,6 +733,95 @@ TEST(DirectVulkanSanity, CommandMemoryBarrierMakesIndirectDrawCommandsVisible) {
|
||||
EXPECT_EQ(storageOnlyBarrier.dstAccessMask & VK_ACCESS_INDIRECT_COMMAND_READ_BIT, 0u);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, ReadbackUsesTheSourceFormatTexelSize) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::VulkanRenderer;
|
||||
|
||||
EXPECT_EQ(VulkanRenderer::GetReadbackTexelSize(VK_FORMAT_R8G8B8A8_UNORM), 4u);
|
||||
EXPECT_EQ(VulkanRenderer::GetReadbackTexelSize(VK_FORMAT_R16G16B16A16_SFLOAT), 8u);
|
||||
EXPECT_EQ(VulkanRenderer::GetReadbackTexelSize(VK_FORMAT_R32G32B32A32_SFLOAT), 16u);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, ReadbackConvertsRgba8AndRgba16fPixels) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::VulkanRenderer;
|
||||
using MobileGL::MG_Util::EncodeFloatToHalfBits;
|
||||
|
||||
const MobileGL::Uint8 rgba8[] = {17, 34, 51, 68, 85, 102, 119, 136};
|
||||
MobileGL::Uint8 rgba8Result[sizeof(rgba8)]{};
|
||||
ASSERT_TRUE(VulkanRenderer::ConvertReadbackPixels(
|
||||
rgba8, VK_FORMAT_R8G8B8A8_UNORM, 2, 1, GL_RGBA, GL_UNSIGNED_BYTE,
|
||||
sizeof(rgba8Result), rgba8Result));
|
||||
EXPECT_TRUE(std::equal(std::begin(rgba8), std::end(rgba8), std::begin(rgba8Result)));
|
||||
|
||||
const MobileGL::Uint8 bgra8[] = {51, 34, 17, 68};
|
||||
MobileGL::Uint8 bgra8Result[4]{};
|
||||
ASSERT_TRUE(VulkanRenderer::ConvertReadbackPixels(
|
||||
bgra8, VK_FORMAT_B8G8R8A8_UNORM, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE,
|
||||
sizeof(bgra8Result), bgra8Result));
|
||||
const MobileGL::Uint8 expectedBgra8[] = {17, 34, 51, 68};
|
||||
EXPECT_TRUE(std::equal(std::begin(expectedBgra8), std::end(expectedBgra8), std::begin(bgra8Result)));
|
||||
|
||||
const MobileGL::Uint16 rgba16f[] = {
|
||||
EncodeFloatToHalfBits(-0.25f), EncodeFloatToHalfBits(0.5f), EncodeFloatToHalfBits(1.5f),
|
||||
EncodeFloatToHalfBits(1.0f), EncodeFloatToHalfBits(0.25f), EncodeFloatToHalfBits(0.0f),
|
||||
EncodeFloatToHalfBits(1.0f), EncodeFloatToHalfBits(0.5f),
|
||||
EncodeFloatToHalfBits(0.75f), EncodeFloatToHalfBits(0.125f), EncodeFloatToHalfBits(-1.0f),
|
||||
EncodeFloatToHalfBits(2.0f), EncodeFloatToHalfBits(1.0f), EncodeFloatToHalfBits(0.75f),
|
||||
EncodeFloatToHalfBits(0.25f), EncodeFloatToHalfBits(0.0f),
|
||||
};
|
||||
constexpr MobileGL::SizeT kDestinationRowStride = 12;
|
||||
MobileGL::Uint8 rgba16fResult[kDestinationRowStride * 2];
|
||||
std::fill(std::begin(rgba16fResult), std::end(rgba16fResult), 0xCD);
|
||||
ASSERT_TRUE(VulkanRenderer::ConvertReadbackPixels(
|
||||
reinterpret_cast<const MobileGL::Uint8*>(rgba16f), VK_FORMAT_R16G16B16A16_SFLOAT,
|
||||
2, 2, GL_RGBA, GL_UNSIGNED_BYTE, kDestinationRowStride, rgba16fResult));
|
||||
const MobileGL::Uint8 expectedRgba16fRow0[] = {0, 128, 255, 255, 64, 0, 255, 128};
|
||||
const MobileGL::Uint8 expectedRgba16fRow1[] = {191, 32, 0, 255, 255, 191, 64, 0};
|
||||
EXPECT_TRUE(std::equal(std::begin(expectedRgba16fRow0), std::end(expectedRgba16fRow0),
|
||||
std::begin(rgba16fResult)));
|
||||
EXPECT_TRUE(std::equal(std::begin(expectedRgba16fRow1), std::end(expectedRgba16fRow1),
|
||||
std::begin(rgba16fResult) + kDestinationRowStride));
|
||||
EXPECT_TRUE(std::all_of(std::begin(rgba16fResult) + 8,
|
||||
std::begin(rgba16fResult) + kDestinationRowStride,
|
||||
[](MobileGL::Uint8 value) { return value == 0xCD; }));
|
||||
|
||||
MobileGL::Float rgba16fFloatResult[16]{};
|
||||
ASSERT_TRUE(VulkanRenderer::ConvertReadbackPixels(
|
||||
reinterpret_cast<const MobileGL::Uint8*>(rgba16f), VK_FORMAT_R16G16B16A16_SFLOAT,
|
||||
2, 2, GL_RGBA, GL_FLOAT, sizeof(MobileGL::Float) * 8,
|
||||
reinterpret_cast<MobileGL::Uint8*>(rgba16fFloatResult)));
|
||||
EXPECT_FLOAT_EQ(rgba16fFloatResult[0], -0.25f);
|
||||
EXPECT_FLOAT_EQ(rgba16fFloatResult[1], 0.5f);
|
||||
EXPECT_FLOAT_EQ(rgba16fFloatResult[2], 1.5f);
|
||||
EXPECT_FLOAT_EQ(rgba16fFloatResult[3], 1.0f);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, ReadbackDecodesSingleChannel32BitFormats) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::VulkanRenderer;
|
||||
|
||||
// The reinterpretation feature makes R32F/R32UI-class images common readback sources
|
||||
// (iterationRP custom images). Missing channels take GL defaults: 0 for GB, 1 for alpha.
|
||||
const MobileGL::Float r32f[] = {0.75f, -2.0f};
|
||||
MobileGL::Float r32fResult[8]{};
|
||||
ASSERT_TRUE(VulkanRenderer::ConvertReadbackPixels(
|
||||
reinterpret_cast<const MobileGL::Uint8*>(r32f), VK_FORMAT_R32_SFLOAT,
|
||||
2, 1, GL_RGBA, GL_FLOAT, sizeof(MobileGL::Float) * 8,
|
||||
reinterpret_cast<MobileGL::Uint8*>(r32fResult)));
|
||||
EXPECT_FLOAT_EQ(r32fResult[0], 0.75f);
|
||||
EXPECT_FLOAT_EQ(r32fResult[1], 0.0f);
|
||||
EXPECT_FLOAT_EQ(r32fResult[2], 0.0f);
|
||||
EXPECT_FLOAT_EQ(r32fResult[3], 1.0f);
|
||||
EXPECT_FLOAT_EQ(r32fResult[4], -2.0f);
|
||||
|
||||
const MobileGL::Uint32 r32ui[] = {12345u};
|
||||
MobileGL::Float r32uiResult[4]{};
|
||||
ASSERT_TRUE(VulkanRenderer::ConvertReadbackPixels(
|
||||
reinterpret_cast<const MobileGL::Uint8*>(r32ui), VK_FORMAT_R32_UINT,
|
||||
1, 1, GL_RGBA, GL_FLOAT, sizeof(MobileGL::Float) * 4,
|
||||
reinterpret_cast<MobileGL::Uint8*>(r32uiResult)));
|
||||
EXPECT_FLOAT_EQ(r32uiResult[0], 12345.0f);
|
||||
EXPECT_FLOAT_EQ(r32uiResult[3], 1.0f);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, DrawIndexedIndirectCommandMatchesGlAndVulkanLayout) {
|
||||
using namespace MobileGL::MG_Backend::DirectVulkan;
|
||||
|
||||
@@ -675,6 +868,193 @@ TEST(DirectVulkanSanity, SampledDepthStencilViewUsesSingleDepthAspect) {
|
||||
VK_IMAGE_ASPECT_DEPTH_BIT);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, SpirvStorageImageFormatsMapToVulkanFormats) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::ProgramFactory;
|
||||
|
||||
struct FormatCase {
|
||||
SpvImageFormat spirv;
|
||||
VkFormat vulkan;
|
||||
};
|
||||
const FormatCase cases[] = {
|
||||
{SpvImageFormatUnknown, VK_FORMAT_UNDEFINED},
|
||||
{SpvImageFormatRgba32f, VK_FORMAT_R32G32B32A32_SFLOAT},
|
||||
{SpvImageFormatRgba16f, VK_FORMAT_R16G16B16A16_SFLOAT},
|
||||
{SpvImageFormatR32f, VK_FORMAT_R32_SFLOAT},
|
||||
{SpvImageFormatRgba8, VK_FORMAT_R8G8B8A8_UNORM},
|
||||
{SpvImageFormatRgba8Snorm, VK_FORMAT_R8G8B8A8_SNORM},
|
||||
{SpvImageFormatRg32f, VK_FORMAT_R32G32_SFLOAT},
|
||||
{SpvImageFormatRg16f, VK_FORMAT_R16G16_SFLOAT},
|
||||
{SpvImageFormatR11fG11fB10f, VK_FORMAT_B10G11R11_UFLOAT_PACK32},
|
||||
{SpvImageFormatR16f, VK_FORMAT_R16_SFLOAT},
|
||||
{SpvImageFormatRgba16, VK_FORMAT_R16G16B16A16_UNORM},
|
||||
{SpvImageFormatRgb10A2, VK_FORMAT_A2R10G10B10_UNORM_PACK32},
|
||||
{SpvImageFormatRg16, VK_FORMAT_R16G16_UNORM},
|
||||
{SpvImageFormatRg8, VK_FORMAT_R8G8_UNORM},
|
||||
{SpvImageFormatR16, VK_FORMAT_R16_UNORM},
|
||||
{SpvImageFormatR8, VK_FORMAT_R8_UNORM},
|
||||
{SpvImageFormatRgba16Snorm, VK_FORMAT_R16G16B16A16_SNORM},
|
||||
{SpvImageFormatRg16Snorm, VK_FORMAT_R16G16_SNORM},
|
||||
{SpvImageFormatRg8Snorm, VK_FORMAT_R8G8_SNORM},
|
||||
{SpvImageFormatR16Snorm, VK_FORMAT_R16_SNORM},
|
||||
{SpvImageFormatR8Snorm, VK_FORMAT_R8_SNORM},
|
||||
{SpvImageFormatRgba32i, VK_FORMAT_R32G32B32A32_SINT},
|
||||
{SpvImageFormatRgba16i, VK_FORMAT_R16G16B16A16_SINT},
|
||||
{SpvImageFormatRgba8i, VK_FORMAT_R8G8B8A8_SINT},
|
||||
{SpvImageFormatR32i, VK_FORMAT_R32_SINT},
|
||||
{SpvImageFormatRg32i, VK_FORMAT_R32G32_SINT},
|
||||
{SpvImageFormatRg16i, VK_FORMAT_R16G16_SINT},
|
||||
{SpvImageFormatRg8i, VK_FORMAT_R8G8_SINT},
|
||||
{SpvImageFormatR16i, VK_FORMAT_R16_SINT},
|
||||
{SpvImageFormatR8i, VK_FORMAT_R8_SINT},
|
||||
{SpvImageFormatRgba32ui, VK_FORMAT_R32G32B32A32_UINT},
|
||||
{SpvImageFormatRgba16ui, VK_FORMAT_R16G16B16A16_UINT},
|
||||
{SpvImageFormatRgba8ui, VK_FORMAT_R8G8B8A8_UINT},
|
||||
{SpvImageFormatR32ui, VK_FORMAT_R32_UINT},
|
||||
{SpvImageFormatRgb10a2ui, VK_FORMAT_A2R10G10B10_UINT_PACK32},
|
||||
{SpvImageFormatRg32ui, VK_FORMAT_R32G32_UINT},
|
||||
{SpvImageFormatRg16ui, VK_FORMAT_R16G16_UINT},
|
||||
{SpvImageFormatRg8ui, VK_FORMAT_R8G8_UINT},
|
||||
{SpvImageFormatR16ui, VK_FORMAT_R16_UINT},
|
||||
{SpvImageFormatR8ui, VK_FORMAT_R8_UINT},
|
||||
{SpvImageFormatR64ui, VK_FORMAT_R64_UINT},
|
||||
{SpvImageFormatR64i, VK_FORMAT_R64_SINT},
|
||||
};
|
||||
|
||||
for (const auto& testCase : cases) {
|
||||
EXPECT_EQ(ProgramFactory::ConvertSpirvImageFormatToVkFormat(testCase.spirv), testCase.vulkan)
|
||||
<< "SpvImageFormat=" << static_cast<int>(testCase.spirv);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, MutableStorageImageViewsUseVulkanCompatibilityClasses) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::VkTextureManager;
|
||||
|
||||
EXPECT_TRUE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_R32_UINT));
|
||||
EXPECT_TRUE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_UINT, VK_FORMAT_R32_SINT));
|
||||
EXPECT_TRUE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R16G16B16A16_UNORM, VK_FORMAT_R16G16B16A16_SFLOAT));
|
||||
EXPECT_TRUE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_R8G8B8A8_UINT));
|
||||
EXPECT_TRUE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_R32_SFLOAT));
|
||||
EXPECT_FALSE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_R16G16B16A16_SFLOAT));
|
||||
EXPECT_FALSE(VkTextureManager::AreStorageImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_D32_SFLOAT));
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, StorageImageViewFormatUsesBindingOnlyForFormatlessFloatPolicy) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::UniformManager;
|
||||
|
||||
EXPECT_EQ(UniformManager::ResolveStorageImageViewFormat(
|
||||
VK_FORMAT_UNDEFINED, GL_RGBA16F, VK_FORMAT_R16G16B16A16_UNORM, true),
|
||||
VK_FORMAT_R16G16B16A16_SFLOAT);
|
||||
EXPECT_EQ(UniformManager::ResolveStorageImageViewFormat(
|
||||
VK_FORMAT_UNDEFINED, GL_RGBA16, VK_FORMAT_R16G16B16A16_SFLOAT, true),
|
||||
VK_FORMAT_R16G16B16A16_UNORM);
|
||||
EXPECT_EQ(UniformManager::ResolveStorageImageViewFormat(
|
||||
VK_FORMAT_R32_UINT, GL_RGBA16F, VK_FORMAT_R32_SFLOAT, false),
|
||||
VK_FORMAT_R32_UINT);
|
||||
EXPECT_EQ(UniformManager::ResolveStorageImageViewFormat(
|
||||
VK_FORMAT_UNDEFINED, GL_RGBA16F, VK_FORMAT_R32_SFLOAT, false),
|
||||
VK_FORMAT_R32_SFLOAT);
|
||||
EXPECT_EQ(UniformManager::ResolveStorageImageViewFormat(
|
||||
VK_FORMAT_UNDEFINED, GL_NONE, VK_FORMAT_R16G16B16A16_SFLOAT, true),
|
||||
VK_FORMAT_UNDEFINED);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, ProgramObjectMovePreservesStorageImageFormatPolicy) {
|
||||
using MobileGL::MG_Backend::DirectVulkan::ProgramFactory;
|
||||
|
||||
ProgramFactory::VkProgramObject source;
|
||||
source.storageImageFormatByBinding = {VK_FORMAT_UNDEFINED, VK_FORMAT_R32_UINT};
|
||||
source.storageImageUsesBindingFormatByBinding = {true, false};
|
||||
|
||||
ProgramFactory::VkProgramObject moved(std::move(source));
|
||||
ASSERT_EQ(moved.storageImageFormatByBinding.size(), 2u);
|
||||
ASSERT_EQ(moved.storageImageUsesBindingFormatByBinding.size(), 2u);
|
||||
EXPECT_EQ(moved.storageImageFormatByBinding[0], VK_FORMAT_UNDEFINED);
|
||||
EXPECT_EQ(moved.storageImageFormatByBinding[1], VK_FORMAT_R32_UINT);
|
||||
EXPECT_TRUE(moved.storageImageUsesBindingFormatByBinding[0]);
|
||||
EXPECT_FALSE(moved.storageImageUsesBindingFormatByBinding[1]);
|
||||
|
||||
ProgramFactory::VkProgramObject assigned;
|
||||
assigned = std::move(moved);
|
||||
ASSERT_EQ(assigned.storageImageFormatByBinding.size(), 2u);
|
||||
ASSERT_EQ(assigned.storageImageUsesBindingFormatByBinding.size(), 2u);
|
||||
EXPECT_TRUE(assigned.storageImageUsesBindingFormatByBinding[0]);
|
||||
EXPECT_FALSE(assigned.storageImageUsesBindingFormatByBinding[1]);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, SamplerUniformTypesPreserveTheirNumericDomain) {
|
||||
using namespace MobileGL::MG_Backend::DirectVulkan;
|
||||
|
||||
EXPECT_EQ(ProgramFactory::UniformTypeToSamplerNumericDomain(GL_SAMPLER_2D),
|
||||
SamplerNumericDomain::Float);
|
||||
EXPECT_EQ(ProgramFactory::UniformTypeToSamplerNumericDomain(GL_SAMPLER_CUBE_MAP_ARRAY_SHADOW),
|
||||
SamplerNumericDomain::Float);
|
||||
EXPECT_EQ(ProgramFactory::UniformTypeToSamplerNumericDomain(GL_INT_SAMPLER_2D_ARRAY),
|
||||
SamplerNumericDomain::SignedInteger);
|
||||
EXPECT_EQ(ProgramFactory::UniformTypeToSamplerNumericDomain(GL_UNSIGNED_INT_SAMPLER_2D),
|
||||
SamplerNumericDomain::UnsignedInteger);
|
||||
EXPECT_EQ(ProgramFactory::UniformTypeToSamplerNumericDomain(GL_IMAGE_2D),
|
||||
SamplerNumericDomain::Unknown);
|
||||
}
|
||||
|
||||
TEST(DirectVulkanSanity, SampledViewFormatMatchesSamplerNumericDomainWithoutChangingComponentLayout) {
|
||||
using namespace MobileGL::MG_Backend::DirectVulkan;
|
||||
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_R32_SFLOAT, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_R32_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_R32_SFLOAT, SamplerNumericDomain::SignedInteger),
|
||||
VK_FORMAT_R32_SINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_R32_UINT, SamplerNumericDomain::Float),
|
||||
VK_FORMAT_R32_SFLOAT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_R16G16B16A16_SFLOAT, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_R16G16B16A16_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_R8G8B8A8_UNORM, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_R8G8B8A8_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_R32_UINT, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_R32_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_B10G11R11_UFLOAT_PACK32, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_UNDEFINED);
|
||||
|
||||
// Depth/stencil formats never resolve through color-class reinterpretation; they pass
|
||||
// through unchanged so the existing depth-aspect sampled view is used. Combined
|
||||
// depth-stencil formats are multi-numeric (vkuFormatIsSampledFloat is false for them),
|
||||
// so without the passthrough a plain sampler2D/sampler2DShadow on GL_DEPTH24_STENCIL8
|
||||
// would resolve to UNDEFINED and the draw would be dropped.
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_D24_UNORM_S8_UINT, SamplerNumericDomain::Float),
|
||||
VK_FORMAT_D24_UNORM_S8_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_D32_SFLOAT_S8_UINT, SamplerNumericDomain::Float),
|
||||
VK_FORMAT_D32_SFLOAT_S8_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_D32_SFLOAT, SamplerNumericDomain::Float),
|
||||
VK_FORMAT_D32_SFLOAT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_D24_UNORM_S8_UINT, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_D24_UNORM_S8_UINT);
|
||||
EXPECT_EQ(VkTextureManager::ResolveSampledImageViewFormat(
|
||||
VK_FORMAT_D32_SFLOAT, SamplerNumericDomain::UnsignedInteger),
|
||||
VK_FORMAT_D32_SFLOAT);
|
||||
|
||||
EXPECT_TRUE(VkTextureManager::AreSampledImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_R32_UINT));
|
||||
EXPECT_FALSE(VkTextureManager::AreSampledImageViewFormatsCompatible(
|
||||
VK_FORMAT_R32_SFLOAT, VK_FORMAT_R16G16B16A16_UINT));
|
||||
}
|
||||
|
||||
TEST(RenderStateSanity, ProvokingVertexUpdatesStateAndValidatesEnum) {
|
||||
using namespace MobileGL;
|
||||
|
||||
@@ -1052,3 +1432,366 @@ TEST(RenderStateSanity, PrimitiveRestartIndexStoresAndReadsBack) {
|
||||
|
||||
MG_State::pGLContext.reset();
|
||||
}
|
||||
|
||||
|
||||
// ---- DirectGLES readback driver-state shadows ----------------------------------------------------
|
||||
// Regression coverage for the readback-path state-leak overhaul: the pixel-PBO
|
||||
// binding cache, the framebuffer-binding shadow, the PACK pixel-store shadow and
|
||||
// the scratch-FBO attachment shadow must (a) leave the driver in the documented
|
||||
// resting state, (b) skip redundant GL calls, and (c) scrub correctly on
|
||||
// deletion. All drive the real Managers.cpp implementations against a recording
|
||||
// mock GLES table.
|
||||
namespace {
|
||||
struct StateGuardCallLog {
|
||||
MobileGL::Vector<MobileGL::String> calls;
|
||||
|
||||
MobileGL::SizeT Count(const MobileGL::String& prefix) const {
|
||||
MobileGL::SizeT n = 0;
|
||||
for (const auto& c : calls) {
|
||||
if (c.compare(0, prefix.size(), prefix) == 0) ++n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
};
|
||||
|
||||
StateGuardCallLog* g_stateGuardLog = nullptr;
|
||||
GLuint g_nextStateGuardFBOId = 201;
|
||||
|
||||
void SG_Log(MobileGL::String entry) {
|
||||
if (g_stateGuardLog) g_stateGuardLog->calls.push_back(MobileGL::Move(entry));
|
||||
}
|
||||
void SG_BindBuffer(GLenum target, GLuint buffer) {
|
||||
SG_Log("BindBuffer:" + std::to_string(target) + ":" + std::to_string(buffer));
|
||||
}
|
||||
void SG_BindFramebuffer(GLenum target, GLuint framebuffer) {
|
||||
SG_Log("BindFramebuffer:" + std::to_string(target) + ":" + std::to_string(framebuffer));
|
||||
}
|
||||
void SG_GetIntegerv(GLenum pname, GLint* data) {
|
||||
SG_Log("GetIntegerv:" + std::to_string(pname));
|
||||
if (data) *data = 0;
|
||||
}
|
||||
void SG_PixelStorei(GLenum pname, GLint param) {
|
||||
SG_Log("PixelStorei:" + std::to_string(pname) + ":" + std::to_string(param));
|
||||
}
|
||||
void SG_GenFramebuffers(GLsizei count, GLuint* framebuffers) {
|
||||
for (GLsizei i = 0; i < count; ++i) framebuffers[i] = g_nextStateGuardFBOId++;
|
||||
}
|
||||
void SG_FramebufferTexture2D(GLenum target, GLenum attachment, GLenum textarget, GLuint texture, GLint level) {
|
||||
SG_Log("FramebufferTexture2D:" + std::to_string(target) + ":" + std::to_string(attachment) + ":" +
|
||||
std::to_string(textarget) + ":" + std::to_string(texture) + ":" + std::to_string(level));
|
||||
}
|
||||
void SG_FramebufferTextureLayer(GLenum target, GLenum attachment, GLuint texture, GLint level, GLint layer) {
|
||||
SG_Log("FramebufferTextureLayer:" + std::to_string(target) + ":" + std::to_string(attachment) + ":" +
|
||||
std::to_string(texture) + ":" + std::to_string(level) + ":" + std::to_string(layer));
|
||||
}
|
||||
void SG_ReadBuffer(GLenum src) {
|
||||
SG_Log("ReadBuffer:" + std::to_string(src));
|
||||
}
|
||||
void SG_DrawBuffers(GLsizei n, const GLenum* bufs) {
|
||||
SG_Log("DrawBuffers:" + std::to_string(n) + ":" + std::to_string(n > 0 && bufs ? bufs[0] : 0));
|
||||
}
|
||||
GLenum SG_NoError() {
|
||||
return GL_NO_ERROR;
|
||||
}
|
||||
|
||||
// Installs the recording table and resets every readback driver-state shadow on
|
||||
// both ends, so these tests cannot bleed into (or inherit from) other tests.
|
||||
struct ScopedStateGuardMocks {
|
||||
ScopedStateGuardMocks(): previousFunctions(MobileGL::MG_Backend::DirectGLES::g_GLESFuncs) {
|
||||
ResetShadows();
|
||||
MobileGL::MG_External::GLESFunctionsTable functions{};
|
||||
functions.glBindBuffer = SG_BindBuffer;
|
||||
functions.glBindFramebuffer = SG_BindFramebuffer;
|
||||
functions.glGetIntegerv = SG_GetIntegerv;
|
||||
functions.glPixelStorei = SG_PixelStorei;
|
||||
functions.glGenFramebuffers = SG_GenFramebuffers;
|
||||
functions.glFramebufferTexture2D = SG_FramebufferTexture2D;
|
||||
functions.glFramebufferTextureLayer = SG_FramebufferTextureLayer;
|
||||
functions.glReadBuffer = SG_ReadBuffer;
|
||||
functions.glDrawBuffers = SG_DrawBuffers;
|
||||
functions.glGetError = SG_NoError;
|
||||
MobileGL::MG_Backend::DirectGLES::SetGLESFuncsTable(functions);
|
||||
g_stateGuardLog = &log;
|
||||
}
|
||||
|
||||
~ScopedStateGuardMocks() {
|
||||
g_stateGuardLog = nullptr;
|
||||
MobileGL::MG_Backend::DirectGLES::SetGLESFuncsTable(previousFunctions);
|
||||
ResetShadows();
|
||||
}
|
||||
|
||||
ScopedStateGuardMocks(const ScopedStateGuardMocks&) = delete;
|
||||
ScopedStateGuardMocks& operator=(const ScopedStateGuardMocks&) = delete;
|
||||
|
||||
static void ResetShadows() {
|
||||
MobileGL::MG_Backend::DirectGLES::BufferImpl::InvalidatePixelBufferBindingCaches();
|
||||
MobileGL::MG_Backend::DirectGLES::FramebufferImpl::InvalidateFramebufferBindingCache();
|
||||
MobileGL::MG_Backend::DirectGLES::PixelStoreImpl::InvalidatePackStateCache();
|
||||
MobileGL::MG_Backend::DirectGLES::ScratchFBOImpl::OnBackendContextDestroyed();
|
||||
}
|
||||
|
||||
StateGuardCallLog log;
|
||||
MobileGL::MG_External::GLESFunctionsTable previousFunctions;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST(DirectGLESStateGuards, PixelPackBindingCacheSkipsRedundantBindsAndRestsAtZero) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
BufferImpl::BindPixelPackBufferId(5);
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 1u);
|
||||
BufferImpl::BindPixelPackBufferId(5); // redundant: must not reach the driver
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 1u);
|
||||
BufferImpl::BindPixelPackBufferId(0); // scope exit: resting state
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 2u);
|
||||
BufferImpl::BindPixelPackBufferId(0);
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 2u);
|
||||
|
||||
// After invalidation (MakeCurrent / context reset) the first bind must reach
|
||||
// the driver again even for the same value.
|
||||
BufferImpl::InvalidatePixelBufferBindingCaches();
|
||||
BufferImpl::BindPixelPackBufferId(0);
|
||||
EXPECT_EQ(mocks.log.Count("BindBuffer:"), 3u);
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, FramebufferBindingShadowPinsOnceThenSkips) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
// Cold path: one driver query pins the shadow; further reads are free.
|
||||
(void)FramebufferImpl::CurrentFramebufferBinding(MobileGL::FramebufferTarget::Read);
|
||||
EXPECT_EQ(mocks.log.Count("GetIntegerv:"), 1u);
|
||||
(void)FramebufferImpl::CurrentFramebufferBinding(MobileGL::FramebufferTarget::Read);
|
||||
EXPECT_EQ(mocks.log.Count("GetIntegerv:"), 1u);
|
||||
|
||||
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 1u);
|
||||
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 1u);
|
||||
// GL_FRAMEBUFFER touches both targets; DRAW is still unknown so it must bind.
|
||||
FramebufferImpl::BindFramebufferId(GL_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 2u);
|
||||
// Both halves now match: no further calls for either single target.
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 7);
|
||||
FramebufferImpl::BindFramebufferId(GL_READ_FRAMEBUFFER, 7);
|
||||
FramebufferImpl::BindFramebufferId(GL_FRAMEBUFFER, 7);
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 2u);
|
||||
EXPECT_EQ(FramebufferImpl::CurrentFramebufferBinding(MobileGL::FramebufferTarget::Draw), 7u);
|
||||
EXPECT_EQ(mocks.log.Count("GetIntegerv:"), 1u); // shadow answered, no new query
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, PackStateShadowAppliesMinimalDeltas) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
// First application pins all four parameters.
|
||||
PixelStoreImpl::ApplyPackState(PixelStoreImpl::PackState{4, 0, 0, 0});
|
||||
EXPECT_EQ(mocks.log.Count("PixelStorei:"), 4u);
|
||||
// Identical state: zero driver calls.
|
||||
PixelStoreImpl::ApplyPackState(PixelStoreImpl::PackState{4, 0, 0, 0});
|
||||
EXPECT_EQ(mocks.log.Count("PixelStorei:"), 4u);
|
||||
// One field changed: exactly one driver call.
|
||||
PixelStoreImpl::ApplyPackState(PixelStoreImpl::PackState{1, 0, 0, 0});
|
||||
EXPECT_EQ(mocks.log.Count("PixelStorei:"), 5u);
|
||||
|
||||
const auto current = PixelStoreImpl::CurrentPackState();
|
||||
EXPECT_EQ(current.Alignment, 1);
|
||||
EXPECT_EQ(current.RowLength, 0);
|
||||
EXPECT_EQ(current.SkipRows, 0);
|
||||
EXPECT_EQ(current.SkipPixels, 0);
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, ScratchFBODetachesCrossAspectResidue) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
auto& fb = ScratchFBOImpl::TempFramebuffer();
|
||||
EXPECT_NE(ScratchFBOImpl::EnsureId(fb), 0u);
|
||||
|
||||
// A depth copy leaves a DEPTH_STENCIL attachment (the pre-fix code never
|
||||
// detached it, wedging every later color readback through this FBO).
|
||||
ScratchFBOImpl::EnsureDepthAttachment2D(fb, GL_DRAW_FRAMEBUFFER, 11, GL_TEXTURE_2D, 0, /*withStencil=*/true);
|
||||
const MobileGL::String dsAttach = "FramebufferTexture2D:" + std::to_string(GL_DRAW_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_DEPTH_STENCIL_ATTACHMENT);
|
||||
EXPECT_EQ(mocks.log.Count(dsAttach), 1u);
|
||||
|
||||
// The next color use must detach the stale depth-stencil attachment exactly once.
|
||||
mocks.log.calls.clear();
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
const MobileGL::String dsDetach = "FramebufferTexture2D:" + std::to_string(GL_READ_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_DEPTH_STENCIL_ATTACHMENT) + ":" +
|
||||
std::to_string(GL_TEXTURE_2D) + ":0:0";
|
||||
const MobileGL::String colorAttach = "FramebufferTexture2D:" + std::to_string(GL_READ_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_COLOR_ATTACHMENT0) + ":" +
|
||||
std::to_string(GL_TEXTURE_2D) + ":22:0";
|
||||
EXPECT_EQ(mocks.log.Count(dsDetach), 1u);
|
||||
EXPECT_EQ(mocks.log.Count(colorAttach), 1u);
|
||||
|
||||
// Back-to-back identical color use: no driver traffic at all.
|
||||
mocks.log.calls.clear();
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
EXPECT_EQ(mocks.log.Count("FramebufferTexture2D:"), 0u);
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, ScratchFBOTextureDeletionForcesFullScrub) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
auto& fb = ScratchFBOImpl::TempFramebuffer();
|
||||
ScratchFBOImpl::EnsureId(fb);
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
|
||||
// The attached texture id dies: the shadow can no longer vouch for the FBO
|
||||
// (ES does not auto-detach from unbound FBOs, and the name may be recycled),
|
||||
// so the next use must scrub and re-attach instead of skipping.
|
||||
ScratchFBOImpl::NoteTextureIdDeleted(22);
|
||||
mocks.log.calls.clear();
|
||||
ScratchFBOImpl::EnsureColorAttachment2D(fb, GL_READ_FRAMEBUFFER, 22, GL_TEXTURE_2D, 0);
|
||||
EXPECT_GE(mocks.log.Count("FramebufferTexture2D:"), 2u); // scrub (color + depth) ...
|
||||
const MobileGL::String colorAttach = "FramebufferTexture2D:" + std::to_string(GL_READ_FRAMEBUFFER) + ":" +
|
||||
std::to_string(GL_COLOR_ATTACHMENT0) + ":" +
|
||||
std::to_string(GL_TEXTURE_2D) + ":22:0";
|
||||
EXPECT_EQ(mocks.log.Count(colorAttach), 1u); // ... then the real re-attach
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, ScratchFBOReadDrawBufferStateCached) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
auto& fb = ScratchFBOImpl::BlitReadFramebuffer();
|
||||
ScratchFBOImpl::EnsureId(fb);
|
||||
|
||||
// Fresh FBOs default to COLOR_ATTACHMENT0 for both buffers: no call needed.
|
||||
ScratchFBOImpl::EnsureReadBuffer(fb, GL_COLOR_ATTACHMENT0);
|
||||
EXPECT_EQ(mocks.log.Count("ReadBuffer:"), 0u);
|
||||
// Depth blits want GL_NONE; the transition costs one call, repeats are free.
|
||||
ScratchFBOImpl::EnsureReadBuffer(fb, GL_NONE);
|
||||
ScratchFBOImpl::EnsureReadBuffer(fb, GL_NONE);
|
||||
EXPECT_EQ(mocks.log.Count("ReadBuffer:"), 1u);
|
||||
ScratchFBOImpl::EnsureDrawBuffer(fb, GL_NONE);
|
||||
ScratchFBOImpl::EnsureDrawBuffer(fb, GL_NONE);
|
||||
EXPECT_EQ(mocks.log.Count("DrawBuffers:"), 1u);
|
||||
}
|
||||
|
||||
namespace {
|
||||
MobileGL::Vector<GLuint>* g_deletedTextureIds = nullptr;
|
||||
|
||||
void SG_DeleteTextures(GLsizei count, const GLuint* textures) {
|
||||
if (!g_deletedTextureIds) return;
|
||||
for (GLsizei i = 0; i < count; ++i) g_deletedTextureIds->push_back(textures[i]);
|
||||
}
|
||||
|
||||
// Clears the recording hook even when a gtest assertion unwinds the test body
|
||||
// (a dangling pointer to the dead stack vector would corrupt later tests).
|
||||
struct ScopedDeletedTextureRecording {
|
||||
explicit ScopedDeletedTextureRecording(MobileGL::Vector<GLuint>& sink) { g_deletedTextureIds = &sink; }
|
||||
~ScopedDeletedTextureRecording() { g_deletedTextureIds = nullptr; }
|
||||
ScopedDeletedTextureRecording(const ScopedDeletedTextureRecording&) = delete;
|
||||
ScopedDeletedTextureRecording& operator=(const ScopedDeletedTextureRecording&) = delete;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST(DirectGLESBackendTexture, DestructorDeletesIdAndScrubsBindingCache) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedDirectGLESTextureBindings scoped; // installs glGenTextures/glBindTexture mocks + resets caches
|
||||
MobileGL::Vector<GLuint> deleted;
|
||||
ScopedDeletedTextureRecording recording(deleted);
|
||||
auto functions = g_GLESFuncs;
|
||||
functions.glDeleteTextures = SG_DeleteTextures;
|
||||
SetGLESFuncsTable(functions);
|
||||
|
||||
const auto texture2DSlot = static_cast<MobileGL::SizeT>(MobileGL::TextureTarget::Texture2D);
|
||||
GLuint id = 0;
|
||||
{
|
||||
auto backendTexture = MobileGL::MakeShared<TextureImpl::BackendTextureObject>();
|
||||
id = backendTexture->GetBackendTextureId();
|
||||
ASSERT_NE(id, 0u);
|
||||
backendTexture->Bind(GL_TEXTURE_2D, 0);
|
||||
ASSERT_EQ(TextureImpl::g_boundTexturesCache[0][texture2DSlot], backendTexture.get());
|
||||
}
|
||||
// Frontend glDeleteTextures used to leak the backend id forever and leave the
|
||||
// cache pointer dangling (heap-address reuse then false-skips a later Bind).
|
||||
ASSERT_EQ(deleted.size(), 1u);
|
||||
EXPECT_EQ(deleted[0], id);
|
||||
EXPECT_EQ(TextureImpl::g_boundTexturesCache[0][texture2DSlot], nullptr);
|
||||
|
||||
// A wrapper whose context died must NOT delete a foreign (recycled) name.
|
||||
{
|
||||
auto backendTexture = MobileGL::MakeShared<TextureImpl::BackendTextureObject>();
|
||||
++TextureImpl::g_textureContextGeneration;
|
||||
backendTexture.reset();
|
||||
--TextureImpl::g_textureContextGeneration; // restore for later tests
|
||||
EXPECT_EQ(deleted.size(), 1u);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(DirectGLESStateGuards, DefaultFramebufferBindGoesThroughShadow) {
|
||||
using namespace MobileGL::MG_Backend::DirectGLES;
|
||||
ScopedStateGuardMocks mocks;
|
||||
|
||||
// The regression this guards against: binding framebuffer 0 raw while the
|
||||
// shadow keeps a user-FBO id makes the next re-bind of that FBO false-skip.
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 7);
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 0); // default-FBO path must use this API
|
||||
FramebufferImpl::BindFramebufferId(GL_DRAW_FRAMEBUFFER, 7); // must reach the driver again
|
||||
EXPECT_EQ(mocks.log.Count("BindFramebuffer:"), 3u);
|
||||
}
|
||||
|
||||
// FastSTL::unordered_map::erase(iterator) regression coverage. The open-addressing
|
||||
// iterator constructor snaps forward from a tombstoned slot to the successor, so
|
||||
// erase must NOT advance the rebuilt iterator again: the old double-advance skipped
|
||||
// one live element per erase, and erasing the element in the highest occupied
|
||||
// bucket pushed the returned index past bucket_count where it never compared equal
|
||||
// to end() again - erase-while-iterating sweeps (pipeline/program cache eviction)
|
||||
// then ran off the bucket array and fed garbage handles to vkDestroyPipeline
|
||||
// (device crash on first mass eviction during world load).
|
||||
TEST(FastSTLSanity, EraseWhileIteratingVisitsEveryElementExactlyOnce) {
|
||||
FastSTL::unordered_map<MobileGL::Uint64, MobileGL::Uint64> map;
|
||||
constexpr MobileGL::Uint64 kCount = 1000;
|
||||
for (MobileGL::Uint64 key = 0; key < kCount; ++key) {
|
||||
map.emplace(key * 0x9e3779b97f4a7c15ull, key);
|
||||
}
|
||||
ASSERT_EQ(map.size(), kCount);
|
||||
|
||||
MobileGL::SizeT visited = 0;
|
||||
for (auto it = map.begin(); it != map.end();) {
|
||||
it = map.erase(it);
|
||||
++visited;
|
||||
ASSERT_LE(visited, kCount); // old code: runaway past end / skipped entries
|
||||
}
|
||||
EXPECT_EQ(visited, kCount);
|
||||
EXPECT_EQ(map.size(), 0u);
|
||||
}
|
||||
|
||||
TEST(FastSTLSanity, EraseReturnsTheSuccessorElement) {
|
||||
FastSTL::unordered_map<MobileGL::Uint32, MobileGL::Uint32> map;
|
||||
for (MobileGL::Uint32 key = 1; key <= 64; ++key) {
|
||||
map.emplace(key, key);
|
||||
}
|
||||
|
||||
// Erasing every other visited element must still visit all 64 exactly once:
|
||||
// the iterator returned by erase names the very next element, not one past it.
|
||||
MobileGL::SizeT visited = 0;
|
||||
MobileGL::SizeT erased = 0;
|
||||
for (auto it = map.begin(); it != map.end();) {
|
||||
++visited;
|
||||
if ((visited & 1) != 0) {
|
||||
it = map.erase(it);
|
||||
++erased;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
ASSERT_LE(visited, 64u);
|
||||
}
|
||||
EXPECT_EQ(visited, 64u);
|
||||
EXPECT_EQ(map.size(), 64u - erased);
|
||||
}
|
||||
|
||||
TEST(FastSTLSanity, ErasingTheOnlyElementReturnsEnd) {
|
||||
FastSTL::unordered_map<MobileGL::Uint32, MobileGL::Uint32> map;
|
||||
map.emplace(42u, 1u);
|
||||
auto next = map.erase(map.begin());
|
||||
EXPECT_EQ(next, map.end());
|
||||
EXPECT_TRUE(map.empty());
|
||||
}
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
|
||||
add_executable(
|
||||
SpirvPassTest
|
||||
SpirvPassTest.cpp
|
||||
)
|
||||
|
||||
target_include_directories(SpirvPassTest PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
${MGL_ROOT}/3rdparty/SPIRV-Reflect
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
SpirvPassTest PRIVATE
|
||||
GTest::gtest_main
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
if (MSVC)
|
||||
target_compile_options(SpirvPassTest PRIVATE /Zc:preprocessor)
|
||||
endif()
|
||||
|
||||
include(GoogleTest)
|
||||
gtest_discover_tests(SpirvPassTest DISCOVERY_TIMEOUT 30 PROPERTIES LABELS unit)
|
||||
@@ -0,0 +1,170 @@
|
||||
// MobileGL - MobileGL/MG_Test/ShaderTranspiler/SpirvPassTest.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
|
||||
#include <spirv_reflect.h>
|
||||
|
||||
using namespace MobileGL;
|
||||
using MobileGL::MG_Util::ShaderTranspiler::ShaderCompiler;
|
||||
|
||||
namespace {
|
||||
// glslangValidator -V output. Both are vertex shaders writing gl_Position through
|
||||
// the gl_PerVertex block, i.e. the Position builtin arrives as OpMemberDecorate rather
|
||||
// than a plain OpDecorate - the shape real glslang output actually takes.
|
||||
|
||||
// #version 450
|
||||
// layout(location = 0) in vec4 inPos;
|
||||
// void main() { gl_Position = inPos; }
|
||||
constexpr Uint32 kPlainVertexSpirv[] = {
|
||||
0x07230203u, 0x00010000u, 0x0008000bu, 0x00000015u, 0x00000000u, 0x00020011u,
|
||||
0x00000001u, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
|
||||
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x0007000fu, 0x00000000u,
|
||||
0x00000004u, 0x6e69616du, 0x00000000u, 0x0000000du, 0x00000011u, 0x00030003u,
|
||||
0x00000002u, 0x000001c2u, 0x00040005u, 0x00000004u, 0x6e69616du, 0x00000000u,
|
||||
0x00060005u, 0x0000000bu, 0x505f6c67u, 0x65567265u, 0x78657472u, 0x00000000u,
|
||||
0x00060006u, 0x0000000bu, 0x00000000u, 0x505f6c67u, 0x7469736fu, 0x006e6f69u,
|
||||
0x00070006u, 0x0000000bu, 0x00000001u, 0x505f6c67u, 0x746e696fu, 0x657a6953u,
|
||||
0x00000000u, 0x00070006u, 0x0000000bu, 0x00000002u, 0x435f6c67u, 0x4470696cu,
|
||||
0x61747369u, 0x0065636eu, 0x00070006u, 0x0000000bu, 0x00000003u, 0x435f6c67u,
|
||||
0x446c6c75u, 0x61747369u, 0x0065636eu, 0x00030005u, 0x0000000du, 0x00000000u,
|
||||
0x00040005u, 0x00000011u, 0x6f506e69u, 0x00000073u, 0x00030047u, 0x0000000bu,
|
||||
0x00000002u, 0x00050048u, 0x0000000bu, 0x00000000u, 0x0000000bu, 0x00000000u,
|
||||
0x00050048u, 0x0000000bu, 0x00000001u, 0x0000000bu, 0x00000001u, 0x00050048u,
|
||||
0x0000000bu, 0x00000002u, 0x0000000bu, 0x00000003u, 0x00050048u, 0x0000000bu,
|
||||
0x00000003u, 0x0000000bu, 0x00000004u, 0x00040047u, 0x00000011u, 0x0000001eu,
|
||||
0x00000000u, 0x00020013u, 0x00000002u, 0x00030021u, 0x00000003u, 0x00000002u,
|
||||
0x00030016u, 0x00000006u, 0x00000020u, 0x00040017u, 0x00000007u, 0x00000006u,
|
||||
0x00000004u, 0x00040015u, 0x00000008u, 0x00000020u, 0x00000000u, 0x0004002bu,
|
||||
0x00000008u, 0x00000009u, 0x00000001u, 0x0004001cu, 0x0000000au, 0x00000006u,
|
||||
0x00000009u, 0x0006001eu, 0x0000000bu, 0x00000007u, 0x00000006u, 0x0000000au,
|
||||
0x0000000au, 0x00040020u, 0x0000000cu, 0x00000003u, 0x0000000bu, 0x0004003bu,
|
||||
0x0000000cu, 0x0000000du, 0x00000003u, 0x00040015u, 0x0000000eu, 0x00000020u,
|
||||
0x00000001u, 0x0004002bu, 0x0000000eu, 0x0000000fu, 0x00000000u, 0x00040020u,
|
||||
0x00000010u, 0x00000001u, 0x00000007u, 0x0004003bu, 0x00000010u, 0x00000011u,
|
||||
0x00000001u, 0x00040020u, 0x00000013u, 0x00000003u, 0x00000007u, 0x00050036u,
|
||||
0x00000002u, 0x00000004u, 0x00000000u, 0x00000003u, 0x000200f8u, 0x00000005u,
|
||||
0x0004003du, 0x00000007u, 0x00000012u, 0x00000011u, 0x00050041u, 0x00000013u,
|
||||
0x00000014u, 0x0000000du, 0x0000000fu, 0x0003003eu, 0x00000014u, 0x00000012u,
|
||||
0x000100fdu, 0x00010038u,
|
||||
};
|
||||
|
||||
// ... plus `invariant gl_Position;` - already carries OpMemberDecorate %gl_PerVertex 0
|
||||
// Invariant, so the pass must not add a duplicate.
|
||||
constexpr Uint32 kAlreadyInvariantVertexSpirv[] = {
|
||||
0x07230203u, 0x00010000u, 0x0008000bu, 0x00000015u, 0x00000000u, 0x00020011u,
|
||||
0x00000001u, 0x0006000bu, 0x00000001u, 0x4c534c47u, 0x6474732eu, 0x3035342eu,
|
||||
0x00000000u, 0x0003000eu, 0x00000000u, 0x00000001u, 0x0007000fu, 0x00000000u,
|
||||
0x00000004u, 0x6e69616du, 0x00000000u, 0x0000000du, 0x00000011u, 0x00030003u,
|
||||
0x00000002u, 0x000001c2u, 0x00040005u, 0x00000004u, 0x6e69616du, 0x00000000u,
|
||||
0x00060005u, 0x0000000bu, 0x505f6c67u, 0x65567265u, 0x78657472u, 0x00000000u,
|
||||
0x00060006u, 0x0000000bu, 0x00000000u, 0x505f6c67u, 0x7469736fu, 0x006e6f69u,
|
||||
0x00070006u, 0x0000000bu, 0x00000001u, 0x505f6c67u, 0x746e696fu, 0x657a6953u,
|
||||
0x00000000u, 0x00070006u, 0x0000000bu, 0x00000002u, 0x435f6c67u, 0x4470696cu,
|
||||
0x61747369u, 0x0065636eu, 0x00070006u, 0x0000000bu, 0x00000003u, 0x435f6c67u,
|
||||
0x446c6c75u, 0x61747369u, 0x0065636eu, 0x00030005u, 0x0000000du, 0x00000000u,
|
||||
0x00040005u, 0x00000011u, 0x6f506e69u, 0x00000073u, 0x00030047u, 0x0000000bu,
|
||||
0x00000002u, 0x00050048u, 0x0000000bu, 0x00000000u, 0x0000000bu, 0x00000000u,
|
||||
0x00040048u, 0x0000000bu, 0x00000000u, 0x00000012u, 0x00050048u, 0x0000000bu,
|
||||
0x00000001u, 0x0000000bu, 0x00000001u, 0x00050048u, 0x0000000bu, 0x00000002u,
|
||||
0x0000000bu, 0x00000003u, 0x00050048u, 0x0000000bu, 0x00000003u, 0x0000000bu,
|
||||
0x00000004u, 0x00040047u, 0x00000011u, 0x0000001eu, 0x00000000u, 0x00020013u,
|
||||
0x00000002u, 0x00030021u, 0x00000003u, 0x00000002u, 0x00030016u, 0x00000006u,
|
||||
0x00000020u, 0x00040017u, 0x00000007u, 0x00000006u, 0x00000004u, 0x00040015u,
|
||||
0x00000008u, 0x00000020u, 0x00000000u, 0x0004002bu, 0x00000008u, 0x00000009u,
|
||||
0x00000001u, 0x0004001cu, 0x0000000au, 0x00000006u, 0x00000009u, 0x0006001eu,
|
||||
0x0000000bu, 0x00000007u, 0x00000006u, 0x0000000au, 0x0000000au, 0x00040020u,
|
||||
0x0000000cu, 0x00000003u, 0x0000000bu, 0x0004003bu, 0x0000000cu, 0x0000000du,
|
||||
0x00000003u, 0x00040015u, 0x0000000eu, 0x00000020u, 0x00000001u, 0x0004002bu,
|
||||
0x0000000eu, 0x0000000fu, 0x00000000u, 0x00040020u, 0x00000010u, 0x00000001u,
|
||||
0x00000007u, 0x0004003bu, 0x00000010u, 0x00000011u, 0x00000001u, 0x00040020u,
|
||||
0x00000013u, 0x00000003u, 0x00000007u, 0x00050036u, 0x00000002u, 0x00000004u,
|
||||
0x00000000u, 0x00000003u, 0x000200f8u, 0x00000005u, 0x0004003du, 0x00000007u,
|
||||
0x00000012u, 0x00000011u, 0x00050041u, 0x00000013u, 0x00000014u, 0x0000000du,
|
||||
0x0000000fu, 0x0003003eu, 0x00000014u, 0x00000012u, 0x000100fdu, 0x00010038u,
|
||||
};
|
||||
|
||||
// OpMemberDecorate <struct-id> <member> <decoration>
|
||||
constexpr Uint32 kOpMemberDecorate = 72;
|
||||
constexpr Uint32 kDecorationInvariant = 18;
|
||||
constexpr Uint32 kSpirvHeaderWordCount = 5;
|
||||
|
||||
// Test-side reference walker. Deliberately independent of the production code so a bug in
|
||||
// the pass cannot hide behind the same helper; only used to count what the pass emitted.
|
||||
Uint32 CountInvariantMemberDecorations(const Vector<Uint32>& spirv) {
|
||||
Uint32 count = 0;
|
||||
for (SizeT i = kSpirvHeaderWordCount; i < spirv.size();) {
|
||||
const Uint32 wordCount = spirv[i] >> 16;
|
||||
const Uint32 opcode = spirv[i] & 0xFFFFu;
|
||||
if (wordCount == 0 || i + wordCount > spirv.size()) {
|
||||
break;
|
||||
}
|
||||
if (opcode == kOpMemberDecorate && wordCount >= 4 && spirv[i + 3] == kDecorationInvariant) {
|
||||
++count;
|
||||
}
|
||||
i += wordCount;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
template <SizeT WordCount>
|
||||
Vector<Uint32> ToVector(const Uint32 (&words)[WordCount]) {
|
||||
return Vector<Uint32>(words, words + WordCount);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// --- DecoratePositionInvariantPass ---
|
||||
|
||||
TEST(DecoratePositionInvariant, AddsInvariantToThePositionMember) {
|
||||
const Vector<Uint32> input = ToVector(kPlainVertexSpirv);
|
||||
ASSERT_EQ(CountInvariantMemberDecorations(input), 0u);
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DecoratePositionInvariantForVulkan(input, output));
|
||||
EXPECT_EQ(CountInvariantMemberDecorations(output), 1u);
|
||||
}
|
||||
|
||||
TEST(DecoratePositionInvariant, DoesNotDuplicateAnExistingInvariant) {
|
||||
const Vector<Uint32> input = ToVector(kAlreadyInvariantVertexSpirv);
|
||||
ASSERT_EQ(CountInvariantMemberDecorations(input), 1u);
|
||||
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DecoratePositionInvariantForVulkan(input, output));
|
||||
EXPECT_EQ(CountInvariantMemberDecorations(output), 1u);
|
||||
// The pass reports SuccessWithoutChange here, and SPIRV-Tools asserts (in assert-enabled
|
||||
// builds) that such a run round-trips byte-identically. Pin that from the outside so an
|
||||
// assert-enabled CI build cannot be the first thing to discover a violation.
|
||||
EXPECT_EQ(output, input);
|
||||
}
|
||||
|
||||
TEST(DecoratePositionInvariant, IsIdempotent) {
|
||||
Vector<Uint32> once;
|
||||
ASSERT_TRUE(ShaderCompiler::DecoratePositionInvariantForVulkan(ToVector(kPlainVertexSpirv), once));
|
||||
Vector<Uint32> twice;
|
||||
ASSERT_TRUE(ShaderCompiler::DecoratePositionInvariantForVulkan(once, twice));
|
||||
EXPECT_EQ(CountInvariantMemberDecorations(twice), 1u);
|
||||
}
|
||||
|
||||
TEST(DecoratePositionInvariant, OutputStaysAReflectableModule) {
|
||||
Vector<Uint32> output;
|
||||
ASSERT_TRUE(ShaderCompiler::DecoratePositionInvariantForVulkan(ToVector(kPlainVertexSpirv), output));
|
||||
|
||||
SpvReflectShaderModule module{};
|
||||
ASSERT_EQ(spvReflectCreateShaderModule(output.size() * sizeof(Uint32), output.data(), &module),
|
||||
SPV_REFLECT_RESULT_SUCCESS);
|
||||
EXPECT_EQ(module.entry_point_count, 1u);
|
||||
spvReflectDestroyShaderModule(&module);
|
||||
}
|
||||
|
||||
TEST(DecoratePositionInvariant, RejectsGarbageInput) {
|
||||
const Vector<Uint32> notSpirv{0xdeadbeefu, 0u, 0u, 0u, 0u};
|
||||
Vector<Uint32> output;
|
||||
EXPECT_FALSE(ShaderCompiler::DecoratePositionInvariantForVulkan(notSpirv, output));
|
||||
}
|
||||
@@ -158,6 +158,39 @@ namespace {
|
||||
auto* mipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
return static_cast<const Uint8*>(mipmapObject->MapMipmapData(TextureUploadTarget::Texture2D, level));
|
||||
}
|
||||
|
||||
class ScopedTextureBackendFunctionsOverride {
|
||||
public:
|
||||
ScopedTextureBackendFunctionsOverride(): m_snapshot(MG_Backend::gBackendFunctionsTable) {}
|
||||
~ScopedTextureBackendFunctionsOverride() { MG_Backend::gBackendFunctionsTable = m_snapshot; }
|
||||
|
||||
private:
|
||||
MG_Backend::GlobalBackendFunctionsTable m_snapshot;
|
||||
};
|
||||
|
||||
struct CopyTexSubImage2DCall {
|
||||
Bool Called = false;
|
||||
GLenum Target = GL_NONE;
|
||||
GLint Level = -1;
|
||||
GLint XOffset = -1;
|
||||
GLint YOffset = -1;
|
||||
GLint X = -1;
|
||||
GLint Y = -1;
|
||||
GLsizei Width = -1;
|
||||
GLsizei Height = -1;
|
||||
GLuint BoundTexture = 0;
|
||||
} g_copyTexSubImage2DCall;
|
||||
|
||||
void RecordCopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height) {
|
||||
g_copyTexSubImage2DCall = {
|
||||
true, target, level, xoffset, yoffset, x, y, width, height,
|
||||
MG_State::pGLContext->GetTextureUnitObject(MG_State::pGLContext->GetActiveTextureUnit())
|
||||
.GetBindingSlot(TextureTarget::Texture2D)
|
||||
.GetBoundObject()
|
||||
->GetExternalIndex(),
|
||||
};
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_F(TextureTest, CreateTexturesCreatesObjectsWithoutBinding) {
|
||||
@@ -178,6 +211,156 @@ TEST_F(TextureTest, CreateTexturesCreatesObjectsWithoutBinding) {
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, ClearTexImageNullClearsWholeNamedTextureAndMarksStorageDirty) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
|
||||
const Uint8 initialPixels[] = {
|
||||
1, 2, 3, 4,
|
||||
5, 6, 7, 8,
|
||||
9, 10, 11, 12,
|
||||
13, 14, 15, 16,
|
||||
};
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 2, 2, 0,
|
||||
GL_RGBA, GL_UNSIGNED_BYTE, initialPixels);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
auto* mipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
mipmapObject->MarkStorageDirty(TextureUploadTarget::Texture2D, 0, false);
|
||||
|
||||
MG_Impl::GLImpl::ClearTexImage(texture, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
const Uint8* stored = GetBoundTexture2DLevelBytes(texture);
|
||||
ASSERT_NE(stored, nullptr);
|
||||
const Uint8 zeros[sizeof(initialPixels)] = {};
|
||||
EXPECT_EQ(std::memcmp(stored, zeros, sizeof(zeros)), 0);
|
||||
EXPECT_TRUE(mipmapObject->IsStorageDirty(TextureUploadTarget::Texture2D, 0));
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, ClearTexImageRepeatsConvertedClearPixel) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 2, 2, 0,
|
||||
GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
const Uint8 clearPixel[] = {17, 34, 51, 68};
|
||||
MG_Impl::GLImpl::ClearTexImage(texture, 0, GL_RGBA, GL_UNSIGNED_BYTE, clearPixel);
|
||||
|
||||
const Uint8* stored = GetBoundTexture2DLevelBytes(texture);
|
||||
ASSERT_NE(stored, nullptr);
|
||||
const Uint8 expected[] = {
|
||||
17, 34, 51, 68,
|
||||
17, 34, 51, 68,
|
||||
17, 34, 51, 68,
|
||||
17, 34, 51, 68,
|
||||
};
|
||||
EXPECT_EQ(std::memcmp(stored, expected, sizeof(expected)), 0);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, ClearTexSubImageClearsOnlyRequestedRectangle) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
Uint8 initialPixels[3 * 2 * 4];
|
||||
std::memset(initialPixels, 0x7f, sizeof(initialPixels));
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 3, 2, 0,
|
||||
GL_RGBA, GL_UNSIGNED_BYTE, initialPixels);
|
||||
|
||||
MG_Impl::GLImpl::ClearTexSubImage(texture, 0, 1, 0, 0, 1, 2, 1,
|
||||
GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
const Uint8* stored = GetBoundTexture2DLevelBytes(texture);
|
||||
ASSERT_NE(stored, nullptr);
|
||||
for (Int y = 0; y < 2; ++y) {
|
||||
for (Int x = 0; x < 3; ++x) {
|
||||
for (Int channel = 0; channel < 4; ++channel) {
|
||||
EXPECT_EQ(stored[(y * 3 + x) * 4 + channel], x == 1 ? 0 : 0x7f);
|
||||
}
|
||||
}
|
||||
}
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, CopyTextureSubImage2DUsesNamedObjectAndRestoresBinding) {
|
||||
const ScopedTextureBackendFunctionsOverride backendGuard;
|
||||
MG_Backend::gBackendFunctionsTable.GL.CopyTexSubImage2D = RecordCopyTexSubImage2D;
|
||||
g_copyTexSubImage2DCall = {};
|
||||
|
||||
GLuint namedTexture = 0;
|
||||
GLuint boundTexture = 0;
|
||||
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_2D, 1, &namedTexture);
|
||||
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_2D, 1, &boundTexture);
|
||||
MG_Impl::GLImpl::BindTextureUnit(0, boundTexture);
|
||||
|
||||
const auto boundBefore = MG_State::pGLContext->GetTextureUnitObject(0)
|
||||
.GetBindingSlot(TextureTarget::Texture2D)
|
||||
.GetBoundObject();
|
||||
MG_Impl::GLImpl::CopyTextureSubImage2D(namedTexture, 2, 3, 4, 5, 6, 7, 8);
|
||||
|
||||
EXPECT_TRUE(g_copyTexSubImage2DCall.Called);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.Target, GL_TEXTURE_2D);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.Level, 2);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.XOffset, 3);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.YOffset, 4);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.X, 5);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.Y, 6);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.Width, 7);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.Height, 8);
|
||||
EXPECT_EQ(g_copyTexSubImage2DCall.BoundTexture, namedTexture);
|
||||
EXPECT_EQ(MG_State::pGLContext->GetTextureUnitObject(0)
|
||||
.GetBindingSlot(TextureTarget::Texture2D)
|
||||
.GetBoundObject(),
|
||||
boundBefore);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, CopyTextureSubImage2DRejectsCubeMapTargets) {
|
||||
const ScopedTextureBackendFunctionsOverride backendGuard;
|
||||
MG_Backend::gBackendFunctionsTable.GL.CopyTexSubImage2D = RecordCopyTexSubImage2D;
|
||||
g_copyTexSubImage2DCall = {};
|
||||
|
||||
GLuint cubeTexture = 0;
|
||||
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_CUBE_MAP, 1, &cubeTexture);
|
||||
MG_Impl::GLImpl::CopyTextureSubImage2D(cubeTexture, 0, 0, 0, 0, 0, 1, 1);
|
||||
|
||||
// GL 4.6 sec. 8.8: the 2D form only accepts 2D/1D-array/rectangle effective targets.
|
||||
EXPECT_FALSE(g_copyTexSubImage2DCall.Called);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), static_cast<GLenum>(GL_INVALID_OPERATION));
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, ClearTexImageErrorContracts) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 2, 2, 0,
|
||||
GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
// Zero texture name is INVALID_OPERATION (ARB_clear_texture).
|
||||
MG_Impl::GLImpl::ClearTexImage(0, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), static_cast<GLenum>(GL_INVALID_OPERATION));
|
||||
|
||||
// A negative level is INVALID_VALUE...
|
||||
MG_Impl::GLImpl::ClearTexImage(texture, -1, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), static_cast<GLenum>(GL_INVALID_VALUE));
|
||||
|
||||
// ...but clearing a level that was never defined is INVALID_OPERATION.
|
||||
MG_Impl::GLImpl::ClearTexImage(texture, 5, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), static_cast<GLenum>(GL_INVALID_OPERATION));
|
||||
|
||||
// A clear region outside the level is INVALID_VALUE.
|
||||
MG_Impl::GLImpl::ClearTexSubImage(texture, 0, 1, 1, 0, 4, 4, 1,
|
||||
GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), static_cast<GLenum>(GL_INVALID_VALUE));
|
||||
|
||||
// An invalid pixel-transfer format is INVALID_ENUM from the shared validators.
|
||||
MG_Impl::GLImpl::ClearTexImage(texture, 0, GL_NONE, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), static_cast<GLenum>(GL_INVALID_ENUM));
|
||||
}
|
||||
|
||||
// GL_MAX_TEXTURE_MAX_ANISOTROPY_EXT is float state that must answer every numeric query: GetFloatv
|
||||
// is authoritative and GetIntegerv would otherwise fall through to its INVALID_ENUM default.
|
||||
TEST_F(TextureTest, MaxTextureMaxAnisotropyIsAnsweredFromTheBackendLimit) {
|
||||
@@ -1244,6 +1427,158 @@ TEST_F(TextureTest, TextureStorage1DAndSubImageModifyNamedObjectOnly) {
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
// Building a mip chain top-down - upload level N, then level 0 - must not destroy the levels
|
||||
// already uploaded. AllocateLevel used to resize() the storage down to level+1 on every call, so
|
||||
// the level-0 upload truncated the chain to a single level; the higher level then read back as
|
||||
// {0,0,0}, IsComplete() rejected the zero-then-nonzero pattern, and DirectGLES answered that by
|
||||
// skipping the texture's sync entirely. This is the shape KHR-GL33.texture_repeat_mode uses, and
|
||||
// it accounted for 108 CTS failures in every GL version.
|
||||
TEST_F(TextureTest, TexImage2DOnLevelZeroKeepsAnAlreadyUploadedHigherLevel) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 1, GL_RGBA8, 49, 23, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 98, 46, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
auto* mipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
ASSERT_NE(mipmapObject, nullptr);
|
||||
EXPECT_EQ(mipmapObject->GetMipmapLevelCount(), 2u);
|
||||
EXPECT_EQ(mipmapObject->GetMipmapTexelSize(TextureUploadTarget::Texture2D, 0), IntVec3(98, 46, 1));
|
||||
EXPECT_EQ(mipmapObject->GetMipmapTexelSize(TextureUploadTarget::Texture2D, 1), IntVec3(49, 23, 1));
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
// The other half of the contract: respecifying a level 0 that already held an image still drops
|
||||
// the chain, exactly as before. Minecraft rebinds the block-atlas name and calls glTexImage2D on
|
||||
// level 0 before uploading the new levels; leaving the previous chain in place would strand a tail
|
||||
// at the wrong sizes and - because Mojang terminates its chains with a 0x0 level - reproduce the
|
||||
// same incomplete-texture black atlas the fix above exists to prevent.
|
||||
TEST_F(TextureTest, TexImage2DRespecifyingAnExistingLevelZeroDropsTheStaleChain) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 8, 8, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 1, GL_RGBA8, 4, 4, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 2, GL_RGBA8, 2, 2, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
auto* mipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
ASSERT_NE(mipmapObject, nullptr);
|
||||
ASSERT_EQ(mipmapObject->GetMipmapLevelCount(), 3u);
|
||||
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 16, 16, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
EXPECT_EQ(mipmapObject->GetMipmapLevelCount(), 1u);
|
||||
EXPECT_EQ(mipmapObject->GetMipmapTexelSize(TextureUploadTarget::Texture2D, 0), IntVec3(16, 16, 1));
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
// Same-size respecification has to drop the chain too. The Mipmap Levels video setting rebuilds
|
||||
// the atlas at identical dimensions with a different level count, so a size-change-only test would
|
||||
// let the old tail survive.
|
||||
TEST_F(TextureTest, TexImage2DRespecifyingLevelZeroAtTheSameSizeStillDropsTheChain) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 8, 8, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 1, GL_RGBA8, 4, 4, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 8, 8, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
auto* mipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
ASSERT_NE(mipmapObject, nullptr);
|
||||
EXPECT_EQ(mipmapObject->GetMipmapLevelCount(), 1u);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
// glTexStorage2D defines exactly `levels` levels. AllocateStorage only grows now, so the immutable
|
||||
// path has to drop a longer pre-existing chain explicitly.
|
||||
TEST_F(TextureTest, TexStorage2DTrimsALongerPreExistingMipChain) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 8, 8, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 1, GL_RGBA8, 4, 4, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 2, GL_RGBA8, 2, 2, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 3, GL_RGBA8, 1, 1, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
|
||||
MG_Impl::GLImpl::TexStorage2D(GL_TEXTURE_2D, 2, GL_RGBA8, 8, 8);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
auto* mipmapObject = static_cast<MG_State::GLState::TextureObjectMipmap*>(textureObject.get());
|
||||
ASSERT_NE(mipmapObject, nullptr);
|
||||
EXPECT_EQ(mipmapObject->GetMipmapLevelCount(), 2u);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
// glTexImage2D used to reject every GL_COMPRESSED_* internal format with GL_INVALID_ENUM, because
|
||||
// none of them mapped to a TextureInternalFormat and the "unknown format" gate fired. They now
|
||||
// resolve to the uncompressed storage that backs them - what GL prescribes for the generic formats,
|
||||
// and a deliberate deviation for RGTC, which ES cannot compress. The (format, type) pairs below are
|
||||
// the ones KHR-GL33.packed_pixels uploads with, so this table doubles as a pin for those 480 cases.
|
||||
TEST_F(TextureTest, CompressedInternalFormatsResolveToTheirUncompressedStorage) {
|
||||
struct Case {
|
||||
GLenum internalFormat;
|
||||
GLenum format;
|
||||
GLenum type;
|
||||
TextureInternalFormat expected;
|
||||
};
|
||||
const Case cases[] = {
|
||||
{GL_COMPRESSED_RED, GL_RED, GL_UNSIGNED_BYTE, TextureInternalFormat::R8},
|
||||
{GL_COMPRESSED_RG, GL_RG, GL_UNSIGNED_BYTE, TextureInternalFormat::RG8},
|
||||
{GL_COMPRESSED_RGB, GL_RGB, GL_UNSIGNED_BYTE, TextureInternalFormat::RGB8},
|
||||
{GL_COMPRESSED_RGBA, GL_RGBA, GL_UNSIGNED_BYTE, TextureInternalFormat::RGBA8},
|
||||
{GL_COMPRESSED_SRGB, GL_RGB, GL_UNSIGNED_BYTE, TextureInternalFormat::SRGB8},
|
||||
{GL_COMPRESSED_SRGB_ALPHA, GL_RGBA, GL_UNSIGNED_BYTE, TextureInternalFormat::SRGB8Alpha8},
|
||||
{GL_COMPRESSED_RED_RGTC1, GL_RED, GL_UNSIGNED_BYTE, TextureInternalFormat::R8},
|
||||
{GL_COMPRESSED_RG_RGTC2, GL_RG, GL_UNSIGNED_BYTE, TextureInternalFormat::RG8},
|
||||
// The signed RGTC pair is uploaded as GL_BYTE and must land on SNORM storage - resolving
|
||||
// them to plain R8/RG8 would silently reinterpret negative texels.
|
||||
{GL_COMPRESSED_SIGNED_RED_RGTC1, GL_RED, GL_BYTE, TextureInternalFormat::R8Snorm},
|
||||
{GL_COMPRESSED_SIGNED_RG_RGTC2, GL_RG, GL_BYTE, TextureInternalFormat::RG8Snorm},
|
||||
};
|
||||
|
||||
MG_Impl::GLImpl::PixelStorei(GL_UNPACK_ALIGNMENT, 1);
|
||||
for (const auto& c : cases) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_2D, texture);
|
||||
MG_Impl::GLImpl::TexImage2D(GL_TEXTURE_2D, 0, c.internalFormat, 4, 4, 0, c.format, c.type, nullptr);
|
||||
|
||||
const auto textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
ASSERT_NE(textureObject, nullptr) << "internalFormat 0x" << std::hex << c.internalFormat;
|
||||
EXPECT_EQ(textureObject->GetFormat(), c.expected) << "internalFormat 0x" << std::hex << c.internalFormat;
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR) << "internalFormat 0x" << std::hex << c.internalFormat;
|
||||
}
|
||||
}
|
||||
|
||||
// RGTC compresses 4x4 blocks of a 2D image and has no 3D form, so glTexImage3D must reject it even
|
||||
// though the same enum is accepted on a 2D target. The generic compressed formats carry no such
|
||||
// restriction and stay legal in 3D.
|
||||
TEST_F(TextureTest, RgtcInternalFormatsAreRejectedOnThreeDimensionalTargets) {
|
||||
const GLenum rgtc[] = {GL_COMPRESSED_RED_RGTC1, GL_COMPRESSED_SIGNED_RED_RGTC1, GL_COMPRESSED_RG_RGTC2,
|
||||
GL_COMPRESSED_SIGNED_RG_RGTC2};
|
||||
for (const GLenum internalFormat : rgtc) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &texture);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_3D, texture);
|
||||
MG_Impl::GLImpl::TexImage3D(GL_TEXTURE_3D, 0, internalFormat, 4, 4, 4, 0, GL_RED, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_INVALID_OPERATION)
|
||||
<< "internalFormat 0x" << std::hex << internalFormat;
|
||||
}
|
||||
|
||||
GLuint generic = 0;
|
||||
MG_Impl::GLImpl::GenTextures(1, &generic);
|
||||
MG_Impl::GLImpl::BindTexture(GL_TEXTURE_3D, generic);
|
||||
MG_Impl::GLImpl::TexImage3D(GL_TEXTURE_3D, 0, GL_COMPRESSED_RGBA, 4, 4, 4, 0, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);
|
||||
EXPECT_EQ(MG_Impl::GLImpl::GetError(), GL_NO_ERROR);
|
||||
}
|
||||
|
||||
TEST_F(TextureTest, TextureStorage3DAndSubImageModifyNamedObjectOnly) {
|
||||
GLuint texture = 0;
|
||||
MG_Impl::GLImpl::CreateTextures(GL_TEXTURE_3D, 1, &texture);
|
||||
|
||||
@@ -9,7 +9,12 @@
|
||||
#include "Loader.h"
|
||||
#include "MG_Util/Types.h"
|
||||
#include <Config.h>
|
||||
#if !defined(__WIN32) && !defined(_WIN32)
|
||||
#if defined(_WIN32)
|
||||
#ifndef WIN32_LEAN_AND_MEAN
|
||||
#define WIN32_LEAN_AND_MEAN 1
|
||||
#endif
|
||||
#include <windows.h>
|
||||
#else
|
||||
#include <dlfcn.h>
|
||||
#endif
|
||||
|
||||
@@ -60,7 +65,14 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
#endif
|
||||
|
||||
static void* OpenLib(const Vector<String>& names) {
|
||||
#if !defined(__WIN32) && !defined(_WIN32) && (!defined(__APPLE__) || defined(MOBILEGL_IOS))
|
||||
#if defined(_WIN32)
|
||||
for (const auto& name : names) {
|
||||
if (HMODULE lib = LoadLibraryA(name.c_str())) {
|
||||
MGLOG_I("Loaded GL backend library: %s", name.c_str());
|
||||
return reinterpret_cast<void*>(lib);
|
||||
}
|
||||
}
|
||||
#elif !defined(__APPLE__) || defined(MOBILEGL_IOS)
|
||||
static const String LibPathPrefixes[] = {
|
||||
#if defined(MOBILEGL_IOS)
|
||||
"@rpath/", "@executable_path/Frameworks/", "@loader_path/Frameworks/",
|
||||
@@ -104,7 +116,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
}
|
||||
|
||||
inline void* ProcAddress(void* lib, const char* name) {
|
||||
#if !defined(__WIN32) && !defined(_WIN32) && (!defined(__APPLE__) || defined(MOBILEGL_IOS))
|
||||
#if defined(_WIN32)
|
||||
return reinterpret_cast<void*>(::GetProcAddress(reinterpret_cast<HMODULE>(lib), name));
|
||||
#elif !defined(__APPLE__) || defined(MOBILEGL_IOS)
|
||||
return dlsym(lib, name);
|
||||
#else
|
||||
return nullptr;
|
||||
@@ -520,6 +534,15 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS) && defined(__ANDROID__)
|
||||
void* angleGlesLib = nullptr;
|
||||
#endif
|
||||
#if defined(_WIN32)
|
||||
// ANGLE is the GLES provider on Windows regardless of UseAngle(). Preload
|
||||
// libGLESv2.dll so libEGL.dll resolves its dependency from the same directory.
|
||||
if (!OpenLib({"libGLESv2.dll"})) {
|
||||
MGLOG_E("Failed to open ANGLE libGLESv2.dll");
|
||||
return;
|
||||
}
|
||||
eglLib = OpenLib({"libEGL.dll"});
|
||||
#else
|
||||
if (UseAngle()) {
|
||||
void* glesLib = OpenLib({"libGLESv2_angle.so"});
|
||||
if (!glesLib) {
|
||||
@@ -541,6 +564,7 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
eglLib = OpenLib({"libEGL.so"});
|
||||
#endif
|
||||
}
|
||||
#endif // !_WIN32
|
||||
|
||||
if (!eglLib) {
|
||||
MGLOG_E("Failed to open EGL library");
|
||||
@@ -811,6 +835,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
if (std::strcmp(extension, "GL_EXT_blend_func_extended") == 0) {
|
||||
caps.SupportsDualSourceBlend = true;
|
||||
}
|
||||
if (std::strcmp(extension, "GL_NV_shader_noperspective_interpolation") == 0) {
|
||||
caps.SupportsNoperspectiveInterpolation = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -865,6 +892,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
GLint maxUniformBlockSize = 16384;
|
||||
GLint maxImageUnits = 8;
|
||||
GLint maxCombinedImageUniforms = 8;
|
||||
GLint maxVertexImageUniforms = 0;
|
||||
GLint maxGeometryImageUniforms = 0;
|
||||
GLint maxFragmentImageUniforms = 8;
|
||||
GLint maxComputeImageUniforms = 8;
|
||||
GLint maxDrawBuffers = 8;
|
||||
GLint maxColorAttachments = 8;
|
||||
@@ -904,7 +934,16 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
glesFuncs.glGetIntegerv(GL_MAX_UNIFORM_BLOCK_SIZE, &maxUniformBlockSize);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_IMAGE_UNITS, &maxImageUnits);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_COMBINED_IMAGE_UNIFORMS, &maxCombinedImageUniforms);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_VERTEX_IMAGE_UNIFORMS, &maxVertexImageUniforms);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_FRAGMENT_IMAGE_UNIFORMS, &maxFragmentImageUniforms);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_COMPUTE_IMAGE_UNIFORMS, &maxComputeImageUniforms);
|
||||
// Geometry shaders and their image-uniform query are core only in ES 3.2. DirectGLES
|
||||
// emits ESSL 3.10 on an ES 3.1 context, so reporting zero there is both legal and an
|
||||
// accurate description of what the backend compiler can consume.
|
||||
if (caps.GLESVersion.Major > 3 ||
|
||||
(caps.GLESVersion.Major == 3 && caps.GLESVersion.Minor >= 2)) {
|
||||
glesFuncs.glGetIntegerv(GL_MAX_GEOMETRY_IMAGE_UNIFORMS, &maxGeometryImageUniforms);
|
||||
}
|
||||
glesFuncs.glGetIntegerv(GL_MAX_DRAW_BUFFERS, &maxDrawBuffers);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_COLOR_ATTACHMENTS, &maxColorAttachments);
|
||||
glesFuncs.glGetIntegerv(GL_MAX_CLIP_DISTANCES, &maxClipDistances);
|
||||
@@ -955,6 +994,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
caps.MaxUniformBlockSize = maxUniformBlockSize;
|
||||
caps.MaxImageUnits = maxImageUnits;
|
||||
caps.MaxCombinedImageUniforms = maxCombinedImageUniforms;
|
||||
caps.MaxVertexImageUniforms = maxVertexImageUniforms;
|
||||
caps.MaxGeometryImageUniforms = maxGeometryImageUniforms;
|
||||
caps.MaxFragmentImageUniforms = maxFragmentImageUniforms;
|
||||
caps.MaxComputeImageUniforms = maxComputeImageUniforms;
|
||||
caps.MaxDrawBuffers = maxDrawBuffers;
|
||||
caps.MaxColorAttachments = maxColorAttachments;
|
||||
@@ -1000,6 +1042,9 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
MGLOG_I(" GL_MAX_UNIFORM_BLOCK_SIZE: %d", caps.MaxUniformBlockSize);
|
||||
MGLOG_I(" GL_MAX_IMAGE_UNITS: %d", caps.MaxImageUnits);
|
||||
MGLOG_I(" GL_MAX_COMBINED_IMAGE_UNIFORMS: %d", caps.MaxCombinedImageUniforms);
|
||||
MGLOG_I(" GL_MAX_VERTEX_IMAGE_UNIFORMS: %d", caps.MaxVertexImageUniforms);
|
||||
MGLOG_I(" GL_MAX_GEOMETRY_IMAGE_UNIFORMS: %d", caps.MaxGeometryImageUniforms);
|
||||
MGLOG_I(" GL_MAX_FRAGMENT_IMAGE_UNIFORMS: %d", caps.MaxFragmentImageUniforms);
|
||||
MGLOG_I(" GL_MAX_COMPUTE_IMAGE_UNIFORMS: %d", caps.MaxComputeImageUniforms);
|
||||
MGLOG_I(" GL_MAX_DRAW_BUFFERS: %d", caps.MaxDrawBuffers);
|
||||
MGLOG_I(" GL_MAX_COLOR_ATTACHMENTS: %d", caps.MaxColorAttachments);
|
||||
|
||||
@@ -1050,6 +1050,11 @@ namespace MobileGL {
|
||||
// factors and layout(index = 1) fragment outputs. GLES core has no dual-source blending,
|
||||
// so without this a draw using a SRC1 factor cannot proceed.
|
||||
Bool SupportsDualSourceBlend = false;
|
||||
// GL_NV_shader_noperspective_interpolation is present: the driver accepts the
|
||||
// `noperspective` interpolation qualifier in ESSL. GLES core has none, so without this
|
||||
// SPIRV-Cross's `#extension ... : require` would fail to compile and MobileGL falls back
|
||||
// to stripping the NoPerspective decoration (smooth interpolation) via StripNoPerspectivePass.
|
||||
Bool SupportsNoperspectiveInterpolation = false;
|
||||
// GL_RENDERER contains "ANGLE".
|
||||
Bool IsAngleRenderer = false;
|
||||
// GL_RENDERER contains both "ANGLE" and "llvmpipe".
|
||||
@@ -1102,6 +1107,9 @@ namespace MobileGL {
|
||||
Int MaxUniformBlockSize = 16384;
|
||||
Int MaxImageUnits = 8;
|
||||
Int MaxCombinedImageUniforms = 8;
|
||||
Int MaxVertexImageUniforms = 0;
|
||||
Int MaxGeometryImageUniforms = 0;
|
||||
Int MaxFragmentImageUniforms = 8;
|
||||
Int MaxComputeImageUniforms = 8;
|
||||
Int MaxDrawBuffers = 8;
|
||||
Int MaxColorAttachments = 8;
|
||||
|
||||
@@ -121,6 +121,7 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
caps.VulkanAPIVersion = DecodeApiVersion(p.apiVersion);
|
||||
caps.DeviceName = p.deviceName;
|
||||
caps.DriverVersionString = DecodeDriverVersion(p.driverVersion);
|
||||
caps.VendorId = p.vendorID;
|
||||
caps.UniformBufferOffsetAlignment = static_cast<int>(p.limits.minUniformBufferOffsetAlignment);
|
||||
caps.AliasedLineWidthRangeMin = p.limits.lineWidthRange[0];
|
||||
caps.AliasedLineWidthRangeMax = p.limits.lineWidthRange[1];
|
||||
@@ -174,6 +175,10 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
VkPhysicalDeviceFeatures supportedFeatures{};
|
||||
vkGetPhysicalDeviceFeatures(physicalDevice, &supportedFeatures);
|
||||
caps.SupportsWideLines = supportedFeatures.wideLines == VK_TRUE;
|
||||
caps.SupportsVertexPipelineStoresAndAtomics =
|
||||
supportedFeatures.vertexPipelineStoresAndAtomics == VK_TRUE;
|
||||
caps.SupportsFragmentStoresAndAtomics = supportedFeatures.fragmentStoresAndAtomics == VK_TRUE;
|
||||
caps.SupportsGeometryShader = supportedFeatures.geometryShader == VK_TRUE;
|
||||
caps.MaxShaderStorageBlockSize = static_cast<SizeT>(p.limits.maxStorageBufferRange);
|
||||
const Bool supportsShaderSubgroup = vk.vkGetPhysicalDeviceProperties2 &&
|
||||
HasUsableShaderSubgroupSupport(subgroupProps);
|
||||
@@ -206,6 +211,7 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
caps.VulkanAPIVersion = DecodeApiVersion(properties.apiVersion);
|
||||
caps.DeviceName = properties.deviceName;
|
||||
caps.DriverVersionString = DecodeDriverVersion(properties.driverVersion);
|
||||
caps.VendorId = properties.vendorID;
|
||||
caps.UniformBufferOffsetAlignment = static_cast<int>(properties.limits.minUniformBufferOffsetAlignment);
|
||||
caps.AliasedLineWidthRangeMin = properties.limits.lineWidthRange[0];
|
||||
caps.AliasedLineWidthRangeMax = properties.limits.lineWidthRange[1];
|
||||
@@ -256,6 +262,11 @@ namespace MobileGL::MG_Util::BackendLoader {
|
||||
caps.ViewportBoundsRangeMax = properties.limits.viewportBoundsRange[1];
|
||||
caps.ViewportSubpixelBits = static_cast<Int>(properties.limits.viewportSubPixelBits);
|
||||
caps.SupportsWideLines = false;
|
||||
// This helper only receives properties, not VkPhysicalDeviceFeatures. Leave optional
|
||||
// stage writes disabled rather than inferring them from descriptor limits alone.
|
||||
caps.SupportsVertexPipelineStoresAndAtomics = false;
|
||||
caps.SupportsFragmentStoresAndAtomics = false;
|
||||
caps.SupportsGeometryShader = false;
|
||||
caps.MaxShaderStorageBlockSize = static_cast<SizeT>(properties.limits.maxStorageBufferRange);
|
||||
caps.SupportsShaderSubgroup = false;
|
||||
caps.SubgroupSize = 0;
|
||||
|
||||
@@ -15,6 +15,8 @@ namespace MobileGL {
|
||||
Version VulkanAPIVersion{1, 0, 0};
|
||||
String DeviceName;
|
||||
String DriverVersionString;
|
||||
// VkPhysicalDeviceProperties::vendorID, for device-quirk vendor gating.
|
||||
Uint32 VendorId = 0;
|
||||
Int UniformBufferOffsetAlignment = 256;
|
||||
Float AliasedLineWidthRangeMin = 1.0f;
|
||||
Float AliasedLineWidthRangeMax = 1.0f;
|
||||
@@ -67,6 +69,12 @@ namespace MobileGL {
|
||||
Float ViewportBoundsRangeMax = 0.0f;
|
||||
Int ViewportSubpixelBits = 0;
|
||||
Bool SupportsWideLines = false;
|
||||
// Storage-image descriptors are limited per stage by
|
||||
// maxPerStageDescriptorStorageImages, but writes/atomics outside compute additionally
|
||||
// require these core Vulkan features to be enabled on the logical device.
|
||||
Bool SupportsVertexPipelineStoresAndAtomics = false;
|
||||
Bool SupportsFragmentStoresAndAtomics = false;
|
||||
Bool SupportsGeometryShader = false;
|
||||
SizeT MaxShaderStorageBlockSize = 128 * 1024 * 1024;
|
||||
Bool SupportsShaderSubgroup = false;
|
||||
Uint32 SubgroupSize = 0;
|
||||
|
||||
@@ -255,6 +255,37 @@ namespace MobileGL {
|
||||
return TextureInternalFormat::DepthComponent;
|
||||
case GL_DEPTH_STENCIL:
|
||||
return TextureInternalFormat::DepthStencil;
|
||||
// Compressed internal formats resolve to the uncompressed storage that backs them.
|
||||
//
|
||||
// For the six generic formats this is exactly what GL prescribes: the implementation
|
||||
// picks a specific compressed format, and when none is available it falls back to the
|
||||
// corresponding base format. Nothing downstream ever sees a compressed enum, so the
|
||||
// metrics, pixel-store and backend tables keep their "one format, N bytes per texel"
|
||||
// invariant instead of each needing a compressed-aware arm.
|
||||
//
|
||||
// The four RGTC formats are a deliberate deviation: they are specific formats that GL
|
||||
// 3.3 requires, but ES exposes no RGTC compressor to hand the data to. Storing the
|
||||
// texels uncompressed keeps them renderable at the cost of the memory saving, which is
|
||||
// strictly better than the INVALID_ENUM the application used to get. Note the signed
|
||||
// variants must land on SNORM storage - CTS uploads them as GL_BYTE.
|
||||
case GL_COMPRESSED_RED:
|
||||
case GL_COMPRESSED_RED_RGTC1:
|
||||
return TextureInternalFormat::R8;
|
||||
case GL_COMPRESSED_SIGNED_RED_RGTC1:
|
||||
return TextureInternalFormat::R8Snorm;
|
||||
case GL_COMPRESSED_RG:
|
||||
case GL_COMPRESSED_RG_RGTC2:
|
||||
return TextureInternalFormat::RG8;
|
||||
case GL_COMPRESSED_SIGNED_RG_RGTC2:
|
||||
return TextureInternalFormat::RG8Snorm;
|
||||
case GL_COMPRESSED_RGB:
|
||||
return TextureInternalFormat::RGB8;
|
||||
case GL_COMPRESSED_RGBA:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
case GL_COMPRESSED_SRGB:
|
||||
return TextureInternalFormat::SRGB8;
|
||||
case GL_COMPRESSED_SRGB_ALPHA:
|
||||
return TextureInternalFormat::SRGB8Alpha8;
|
||||
case GL_ALPHA:
|
||||
case GL_RED:
|
||||
return TextureInternalFormat::Red;
|
||||
|
||||
@@ -133,9 +133,11 @@ namespace MobileGL {
|
||||
case TextureInternalFormat::RGBA8Snorm:
|
||||
return VK_FORMAT_R8G8B8A8_SNORM;
|
||||
case TextureInternalFormat::RGB10A2:
|
||||
return VK_FORMAT_A2R10G10B10_UNORM_PACK32;
|
||||
// GL_UNSIGNED_INT_2_10_10_10_REV puts R in bits 0-9, which is Vulkan's
|
||||
// A2B10G10R10 layout - A2R10G10B10 silently swaps R and B on upload.
|
||||
return VK_FORMAT_A2B10G10R10_UNORM_PACK32;
|
||||
case TextureInternalFormat::RGB10A2UI:
|
||||
return VK_FORMAT_A2R10G10B10_UINT_PACK32;
|
||||
return VK_FORMAT_A2B10G10R10_UINT_PACK32;
|
||||
case TextureInternalFormat::RGBA16:
|
||||
return VK_FORMAT_R16G16B16A16_UNORM;
|
||||
case TextureInternalFormat::RGBA16Snorm:
|
||||
|
||||
@@ -401,6 +401,230 @@ namespace MobileGL::MG_Util::SelfTest {
|
||||
disabledNote);
|
||||
}
|
||||
|
||||
// Compiles + links a two-stage program on the probe context. Returns 0 on failure and writes a
|
||||
// human-readable reason into |detail|.
|
||||
GLuint CompileLinkProgram(const MG_External::GLESFunctionsTable& g, const char* vs, const char* fs,
|
||||
String& detail) {
|
||||
const auto compile = [&](GLenum stage, const char* src, GLuint& out) -> bool {
|
||||
out = g.glCreateShader(stage);
|
||||
if (out == 0) {
|
||||
detail = "glCreateShader returned 0";
|
||||
return false;
|
||||
}
|
||||
g.glShaderSource(out, 1, &src, nullptr);
|
||||
g.glCompileShader(out);
|
||||
GLint ok = GL_FALSE;
|
||||
g.glGetShaderiv(out, GL_COMPILE_STATUS, &ok);
|
||||
if (ok != GL_TRUE) {
|
||||
GLchar log[512] = {};
|
||||
GLsizei len = 0;
|
||||
g.glGetShaderInfoLog(out, static_cast<GLsizei>(sizeof(log) - 1), &len, log);
|
||||
detail = format("{} shader compile failed: {}",
|
||||
stage == GL_VERTEX_SHADER ? "vertex" : "fragment",
|
||||
len > 0 ? log : "(no info log)");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
};
|
||||
GLuint v = 0, f = 0;
|
||||
const ScopeGuard delV([&]() { if (v) g.glDeleteShader(v); });
|
||||
const ScopeGuard delF([&]() { if (f) g.glDeleteShader(f); });
|
||||
if (!compile(GL_VERTEX_SHADER, vs, v) || !compile(GL_FRAGMENT_SHADER, fs, f)) {
|
||||
return 0;
|
||||
}
|
||||
const GLuint prog = g.glCreateProgram();
|
||||
if (prog == 0) {
|
||||
detail = "glCreateProgram returned 0";
|
||||
return 0;
|
||||
}
|
||||
g.glAttachShader(prog, v);
|
||||
g.glAttachShader(prog, f);
|
||||
g.glLinkProgram(prog);
|
||||
GLint linked = GL_FALSE;
|
||||
g.glGetProgramiv(prog, GL_LINK_STATUS, &linked);
|
||||
if (linked != GL_TRUE) {
|
||||
detail = "program link failed";
|
||||
g.glDeleteProgram(prog);
|
||||
return 0;
|
||||
}
|
||||
return prog;
|
||||
}
|
||||
|
||||
// "noperspective interpolation" row - a real correctness render, not just a compile. A viewport-
|
||||
// filling quad is drawn with strong perspective (left clip-w 1, right clip-w 8) and a varying that
|
||||
// runs 0..1 across it. At the screen centre screen-linear interpolation gives 0.5 while perspective-
|
||||
// correct gives 1/(w+1) ~= 0.11, so reading the centre texel tells the two apart. The varying is
|
||||
// carried either through the native `noperspective` qualifier (extension present) or through the
|
||||
// exact gl_Position.w / gl_FragCoord.w rewrite MobileGL applies when it is absent. Verdict:
|
||||
// PASS - extension present and the native noperspective result is screen-linear;
|
||||
// WARN - extension absent but the gl_Position.w/gl_FragCoord.w emulation renders screen-linear
|
||||
// (correct, just the fallback path shipping shader packs hit on such devices);
|
||||
// FAIL - either path renders perspective-correct / wrong (noperspective does not actually work),
|
||||
// or the program will not compile/link, or the render errors.
|
||||
// Requires the probe context to still be current.
|
||||
void ProbeGlesNoperspective(ReportBuilder& builder, const MG_External::GLESCapabilities& caps,
|
||||
const MG_External::GLESFunctionsTable& g) {
|
||||
const Bool native = caps.SupportsNoperspectiveInterpolation;
|
||||
const String pathNote = native ? "GL_NV_shader_noperspective_interpolation present (native path)"
|
||||
: "GL_NV_shader_noperspective_interpolation absent (gl_Position.w / "
|
||||
"gl_FragCoord.w emulation path)";
|
||||
const auto fail = [&](const String& detail) {
|
||||
builder.Fail("noperspective interpolation", pathNote + "; " + detail);
|
||||
};
|
||||
|
||||
if (!g.glCreateShader || !g.glShaderSource || !g.glCompileShader || !g.glGetShaderiv ||
|
||||
!g.glGetShaderInfoLog || !g.glDeleteShader || !g.glCreateProgram || !g.glAttachShader ||
|
||||
!g.glLinkProgram || !g.glGetProgramiv || !g.glUseProgram || !g.glDeleteProgram ||
|
||||
!g.glGenFramebuffers || !g.glBindFramebuffer || !g.glDeleteFramebuffers ||
|
||||
!g.glGenRenderbuffers || !g.glBindRenderbuffer || !g.glRenderbufferStorage ||
|
||||
!g.glFramebufferRenderbuffer || !g.glDeleteRenderbuffers || !g.glCheckFramebufferStatus ||
|
||||
!g.glGenBuffers || !g.glBindBuffer || !g.glBufferData || !g.glDeleteBuffers ||
|
||||
!g.glGetAttribLocation || !g.glVertexAttribPointer || !g.glEnableVertexAttribArray ||
|
||||
!g.glViewport || !g.glClearColor || !g.glClear || !g.glDrawArrays || !g.glReadPixels ||
|
||||
!g.glFinish || !g.glGetError) {
|
||||
fail("the render entry points did not resolve through eglGetProcAddress");
|
||||
return;
|
||||
}
|
||||
|
||||
// Match MobileGL's own ESSL target (the device's version). At #version 300 es some drivers
|
||||
// (Adreno) still treat `noperspective` as reserved even with the extension enabled; the ES 3.2
|
||||
// form the backend actually emits compiles. Emulated shaders are version-agnostic but use the
|
||||
// same header for consistency.
|
||||
const Int esslVer = caps.GLESVersion.Major * 100 + caps.GLESVersion.Minor * 10;
|
||||
const String header = format("#version {} es\n", esslVer >= 300 ? esslVer : 300);
|
||||
static const char* const kVsNativeBody =
|
||||
"#extension GL_NV_shader_noperspective_interpolation : require\n"
|
||||
"in vec4 a_pos;\n"
|
||||
"in float a_v;\n"
|
||||
"noperspective out highp float v_out;\n"
|
||||
"void main() { gl_Position = a_pos; v_out = a_v; }\n";
|
||||
static const char* const kFsNativeBody =
|
||||
"#extension GL_NV_shader_noperspective_interpolation : require\n"
|
||||
"precision highp float;\n"
|
||||
"noperspective in highp float v_out;\n"
|
||||
"out vec4 fragColor;\n"
|
||||
"void main() { fragColor = vec4(v_out, 0.0, 0.0, 1.0); }\n";
|
||||
// Exactly MobileGL's emulation (verified against EmulateNoPerspectivePass output): pre-multiply
|
||||
// the varying by clip-w in the vertex stage, recover with gl_FragCoord.w in the fragment stage,
|
||||
// no noperspective qualifier (so the driver interpolates it perspective-correct).
|
||||
static const char* const kVsEmuBody =
|
||||
"in vec4 a_pos;\n"
|
||||
"in float a_v;\n"
|
||||
"out highp float v_out;\n"
|
||||
"void main() { gl_Position = a_pos; v_out = a_v * gl_Position.w; }\n";
|
||||
static const char* const kFsEmuBody =
|
||||
"precision highp float;\n"
|
||||
"in highp float v_out;\n"
|
||||
"out vec4 fragColor;\n"
|
||||
"void main() { fragColor = vec4(v_out * gl_FragCoord.w, 0.0, 0.0, 1.0); }\n";
|
||||
|
||||
while (g.glGetError() != GL_NO_ERROR) {
|
||||
}
|
||||
|
||||
const String vsSrc = header + (native ? kVsNativeBody : kVsEmuBody);
|
||||
const String fsSrc = header + (native ? kFsNativeBody : kFsEmuBody);
|
||||
String linkDetail;
|
||||
const GLuint prog = CompileLinkProgram(g, vsSrc.c_str(), fsSrc.c_str(), linkDetail);
|
||||
if (prog == 0) {
|
||||
fail(native ? "a noperspective program failed to build though the extension is advertised: " +
|
||||
linkDetail
|
||||
: "the emulation program failed to build: " + linkDetail);
|
||||
return;
|
||||
}
|
||||
const ScopeGuard delProg([&]() { g.glDeleteProgram(prog); });
|
||||
|
||||
// 9x9 so the centre texel (4,4) sits exactly at NDC (0,0).
|
||||
constexpr GLsizei kDim = 9;
|
||||
GLuint rbo = 0, fbo = 0, vbo = 0;
|
||||
g.glGenRenderbuffers(1, &rbo);
|
||||
const ScopeGuard delRbo([&]() { if (rbo) g.glDeleteRenderbuffers(1, &rbo); });
|
||||
g.glBindRenderbuffer(GL_RENDERBUFFER, rbo);
|
||||
g.glRenderbufferStorage(GL_RENDERBUFFER, GL_RGBA8, kDim, kDim);
|
||||
g.glGenFramebuffers(1, &fbo);
|
||||
const ScopeGuard delFbo([&]() {
|
||||
if (fbo) {
|
||||
g.glBindFramebuffer(GL_FRAMEBUFFER, 0);
|
||||
g.glDeleteFramebuffers(1, &fbo);
|
||||
}
|
||||
});
|
||||
g.glBindFramebuffer(GL_FRAMEBUFFER, fbo);
|
||||
g.glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, rbo);
|
||||
if (g.glCheckFramebufferStatus(GL_FRAMEBUFFER) != GL_FRAMEBUFFER_COMPLETE) {
|
||||
fail("the probe framebuffer is incomplete");
|
||||
return;
|
||||
}
|
||||
|
||||
// Interleaved [vec4 clip-pos, float v]. Left w=1, right w=8; x/y pre-multiplied by w so the quad
|
||||
// still fills NDC after the perspective divide.
|
||||
const GLfloat verts[] = {
|
||||
-1.f, -1.f, 0.f, 1.f, 0.f, //
|
||||
8.f, -8.f, 0.f, 8.f, 1.f, //
|
||||
-1.f, 1.f, 0.f, 1.f, 0.f, //
|
||||
8.f, 8.f, 0.f, 8.f, 1.f, //
|
||||
};
|
||||
g.glGenBuffers(1, &vbo);
|
||||
const ScopeGuard delVbo([&]() { if (vbo) g.glDeleteBuffers(1, &vbo); });
|
||||
g.glBindBuffer(GL_ARRAY_BUFFER, vbo);
|
||||
g.glBufferData(GL_ARRAY_BUFFER, sizeof(verts), verts, GL_STATIC_DRAW);
|
||||
|
||||
g.glUseProgram(prog);
|
||||
const GLint posLoc = g.glGetAttribLocation(prog, "a_pos");
|
||||
const GLint vLoc = g.glGetAttribLocation(prog, "a_v");
|
||||
if (posLoc < 0 || vLoc < 0) {
|
||||
fail("the probe vertex attributes did not resolve");
|
||||
return;
|
||||
}
|
||||
g.glEnableVertexAttribArray(static_cast<GLuint>(posLoc));
|
||||
g.glVertexAttribPointer(static_cast<GLuint>(posLoc), 4, GL_FLOAT, GL_FALSE, 5 * sizeof(GLfloat),
|
||||
reinterpret_cast<const void*>(0));
|
||||
g.glEnableVertexAttribArray(static_cast<GLuint>(vLoc));
|
||||
g.glVertexAttribPointer(static_cast<GLuint>(vLoc), 1, GL_FLOAT, GL_FALSE, 5 * sizeof(GLfloat),
|
||||
reinterpret_cast<const void*>(4 * sizeof(GLfloat)));
|
||||
|
||||
g.glViewport(0, 0, kDim, kDim);
|
||||
g.glClearColor(0.f, 0.f, 0.f, 1.f);
|
||||
g.glClear(GL_COLOR_BUFFER_BIT);
|
||||
g.glDrawArrays(GL_TRIANGLE_STRIP, 0, 4);
|
||||
g.glFinish();
|
||||
|
||||
const GLenum drawError = g.glGetError();
|
||||
if (drawError != GL_NO_ERROR) {
|
||||
fail(format("GL error 0x{:x} while rendering the probe quad", drawError));
|
||||
return;
|
||||
}
|
||||
|
||||
GLubyte center[4] = {};
|
||||
g.glReadPixels(kDim / 2, kDim / 2, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, center);
|
||||
const GLenum readError = g.glGetError();
|
||||
if (readError != GL_NO_ERROR) {
|
||||
fail(format("GL error 0x{:x} while reading the probe pixel back", readError));
|
||||
return;
|
||||
}
|
||||
|
||||
// At the centre: screen-linear -> 0.5 (~128); perspective-correct -> 1/(8+1) ~= 0.111 (~28).
|
||||
const float observed = static_cast<float>(center[0]) / 255.0f;
|
||||
const int observedByte = center[0];
|
||||
constexpr float kScreenLinear = 0.5f;
|
||||
const bool screenLinear = observed > 0.5f * (kScreenLinear + 1.0f / 9.0f); // midpoint ~= 0.306
|
||||
if (!screenLinear) {
|
||||
fail(format("the centre texel read {} (~{:.3f}); expected the screen-linear ~0.5 - "
|
||||
"interpolation came out perspective-correct, so noperspective does not work here",
|
||||
observedByte, observed));
|
||||
return;
|
||||
}
|
||||
if (native) {
|
||||
builder.Pass("noperspective interpolation",
|
||||
pathNote + format("; native noperspective renders screen-linear (centre {} ~= 0.5)",
|
||||
observedByte));
|
||||
} else {
|
||||
builder.Warn("noperspective interpolation",
|
||||
pathNote +
|
||||
format("; the emulation renders screen-linear correctly (centre {} ~= 0.5), "
|
||||
"but this is the fallback path with less driver coverage",
|
||||
observedByte));
|
||||
}
|
||||
}
|
||||
|
||||
// Everything the "MobileGL reported ..." rows need from the GLES device probe.
|
||||
struct GlesProbeSummary {
|
||||
Bool capsValid = false;
|
||||
@@ -527,6 +751,7 @@ namespace MobileGL::MG_Util::SelfTest {
|
||||
builder.report.rendererInfo = format("{} ({})", caps.GLESRendererString, caps.GLESVersionString);
|
||||
EvaluateGlesChecklist(builder, caps, glesFuncs);
|
||||
ProbeGlesTimerQuery(builder, caps, glesFuncs);
|
||||
ProbeGlesNoperspective(builder, caps, glesFuncs);
|
||||
builder.report.formatCapabilities.emplace();
|
||||
MG_Backend::DirectGLES::PopulateFormatCapabilities(
|
||||
glesFuncs, caps, builder.report.formatCapabilities.value());
|
||||
|
||||
@@ -6,27 +6,36 @@
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#define SPV_ENABLE_UTILITY_CODE
|
||||
#include "glslang/SPIRV/spirv.hpp11"
|
||||
#undef SPV_ENABLE_UTILITY_CODE
|
||||
|
||||
#include "ShaderCompiler.h"
|
||||
|
||||
#include "SpirvPasses/EliminateFloatEqualsZeroPass.h"
|
||||
#include "SpirvPasses/FlattenInterfaceStructPass.h"
|
||||
#include "SpirvPasses/RenameSamplerFunctionParameterPass.h"
|
||||
#include "SpirvPasses/DecomposeWorkgroupVec3Pass.h"
|
||||
#include "SpirvPasses/DecoratePositionInvariantPass.h"
|
||||
#include "SpirvPasses/LowerDrawParametersPass.h"
|
||||
#include "SpirvPasses/RebaseInstanceIndexPass.h"
|
||||
#include "SpirvPasses/StripUboMemberRelaxedPrecisionPass.h"
|
||||
#include "SpirvPasses/StripNoPerspectivePass.h"
|
||||
#include "SpirvPasses/EmulateNoPerspectivePass.h"
|
||||
#include "spirv-tools/libspirv.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
#include "ShaderSourceProcessor.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToGlslang/ProgramEnumConverter.h>
|
||||
#include <cstdlib>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
TBuiltInResource& GetTBuiltInResourceInstance() {
|
||||
static TBuiltInResource Resources{};
|
||||
TBuiltInResource BuildTBuiltInResource() {
|
||||
TBuiltInResource Resources{};
|
||||
Resources.maxLights = 32;
|
||||
Resources.maxClipPlanes = 6;
|
||||
Resources.maxTextureUnits = 32;
|
||||
@@ -121,6 +130,22 @@ namespace MobileGL {
|
||||
Resources.maxTaskWorkGroupSizeZ_NV = 1;
|
||||
Resources.maxMeshViewCountNV = 4;
|
||||
|
||||
// Resource checking must describe the same backend contract exposed through
|
||||
// glGetIntegerv. Keeping this copy local also avoids racing on a process-global
|
||||
// TBuiltInResource when Iris compiles shaders concurrently.
|
||||
const MG_Backend::DynamicBackendParameters fallbackParameters{};
|
||||
const auto& activeBackend = MG_Backend::pActiveBackendObject;
|
||||
const auto& dynamicParameters =
|
||||
activeBackend ? activeBackend->GetDynamicParameters() : fallbackParameters;
|
||||
Resources.maxImageUnits = dynamicParameters.MaxImageUnits;
|
||||
Resources.maxCombinedImageUnitsAndFragmentOutputs =
|
||||
dynamicParameters.MaxImageUnits + dynamicParameters.MaxDrawBuffers;
|
||||
Resources.maxVertexImageUniforms = dynamicParameters.MaxVertexImageUniforms;
|
||||
Resources.maxGeometryImageUniforms = dynamicParameters.MaxGeometryImageUniforms;
|
||||
Resources.maxFragmentImageUniforms = dynamicParameters.MaxFragmentImageUniforms;
|
||||
Resources.maxComputeImageUniforms = dynamicParameters.MaxComputeImageUniforms;
|
||||
Resources.maxCombinedImageUniforms = dynamicParameters.MaxCombinedImageUniforms;
|
||||
|
||||
Resources.limits.nonInductiveForLoops = true;
|
||||
Resources.limits.whileLoops = true;
|
||||
Resources.limits.doWhileLoops = true;
|
||||
@@ -166,7 +191,8 @@ namespace MobileGL {
|
||||
tshader->setAutoMapLocations(true);
|
||||
tshader->setAutoMapBindings(true);
|
||||
tshader->setGlobalUniformBlockName(GLOBAL_UBO_NAME);
|
||||
if (!tshader->parse(&GetTBuiltInResourceInstance(), 460, ECoreProfile,
|
||||
auto resources = BuildTBuiltInResource();
|
||||
if (!tshader->parse(&resources, 460, ECoreProfile,
|
||||
/*forceDefaultVersionAndProfile: */ false,
|
||||
/*forwardCompatible: */ true, EShMsgDefault)) {
|
||||
ResultInfo r;
|
||||
@@ -310,6 +336,30 @@ namespace MobileGL {
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(StripNoPerspectivePass::CreateStripNoPerspectivePass());
|
||||
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(EmulateNoPerspectivePass::CreateEmulateNoPerspectivePass());
|
||||
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
@@ -322,6 +372,148 @@ namespace MobileGL {
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(DecoratePositionInvariantPass::CreateDecoratePositionInvariantPass());
|
||||
|
||||
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::UseUnformattedFloatStorageImagesForVulkan(
|
||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary) {
|
||||
constexpr SizeT kSpirvHeaderWordCount = 5;
|
||||
outputBinary.clear();
|
||||
if (inputBinary.size() < kSpirvHeaderWordCount || inputBinary[0] != spv::MagicNumber) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Vector<Uint32> floatTypeIds;
|
||||
Vector<Uint32> resultTypeById(inputBinary[3], 0);
|
||||
Vector<Uint32> pointerPointeeTypeById(inputBinary[3], 0);
|
||||
Bool hasReadWithoutFormatCapability = false;
|
||||
Bool hasWriteWithoutFormatCapability = false;
|
||||
SizeT capabilityInsertOffset = kSpirvHeaderWordCount;
|
||||
|
||||
for (SizeT offset = kSpirvHeaderWordCount; offset < inputBinary.size();) {
|
||||
const Uint32 instructionWord = inputBinary[offset];
|
||||
const Uint32 wordCount = instructionWord >> 16u;
|
||||
const auto opcode = static_cast<spv::Op>(instructionWord & 0xffffu);
|
||||
if (wordCount == 0 || offset + wordCount > inputBinary.size()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (opcode == spv::Op::OpCapability && wordCount >= 2) {
|
||||
capabilityInsertOffset = offset + wordCount;
|
||||
const auto capability = static_cast<spv::Capability>(inputBinary[offset + 1]);
|
||||
hasReadWithoutFormatCapability |=
|
||||
capability == spv::Capability::StorageImageReadWithoutFormat;
|
||||
hasWriteWithoutFormatCapability |=
|
||||
capability == spv::Capability::StorageImageWriteWithoutFormat;
|
||||
} else if (opcode == spv::Op::OpTypeFloat && wordCount >= 3) {
|
||||
floatTypeIds.push_back(inputBinary[offset + 1]);
|
||||
} else if (opcode == spv::Op::OpTypePointer && wordCount >= 4) {
|
||||
const Uint32 pointerTypeId = inputBinary[offset + 1];
|
||||
if (pointerTypeId >= pointerPointeeTypeById.size()) {
|
||||
return false;
|
||||
}
|
||||
pointerPointeeTypeById[pointerTypeId] = inputBinary[offset + 3];
|
||||
}
|
||||
|
||||
bool hasResult = false;
|
||||
bool hasResultType = false;
|
||||
spv::HasResultAndType(opcode, &hasResult, &hasResultType);
|
||||
if (hasResult && hasResultType && wordCount >= 3) {
|
||||
const Uint32 resultTypeId = inputBinary[offset + 1];
|
||||
const Uint32 resultId = inputBinary[offset + 2];
|
||||
if (resultId >= resultTypeById.size()) {
|
||||
return false;
|
||||
}
|
||||
resultTypeById[resultId] = resultTypeId;
|
||||
}
|
||||
offset += wordCount;
|
||||
}
|
||||
|
||||
// OpImageTexelPointer is the bridge to image atomic instructions. Vulkan requires
|
||||
// those image types to retain an atomic-compatible declared format, so exclude only
|
||||
// the exact image types used by an atomic path rather than disabling formatless
|
||||
// access for unrelated float images in the same module.
|
||||
Vector<Uint32> atomicImageTypeIds;
|
||||
for (SizeT offset = kSpirvHeaderWordCount; offset < inputBinary.size();) {
|
||||
const Uint32 instructionWord = inputBinary[offset];
|
||||
const Uint32 wordCount = instructionWord >> 16u;
|
||||
const auto opcode = static_cast<spv::Op>(instructionWord & 0xffffu);
|
||||
if (opcode == spv::Op::OpImageTexelPointer && wordCount >= 6) {
|
||||
const Uint32 imageId = inputBinary[offset + 3];
|
||||
if (imageId >= resultTypeById.size()) {
|
||||
return false;
|
||||
}
|
||||
Uint32 imageTypeId = resultTypeById[imageId];
|
||||
if (imageTypeId < pointerPointeeTypeById.size() &&
|
||||
pointerPointeeTypeById[imageTypeId] != 0) {
|
||||
imageTypeId = pointerPointeeTypeById[imageTypeId];
|
||||
}
|
||||
if (imageTypeId != 0 &&
|
||||
std::find(atomicImageTypeIds.begin(), atomicImageTypeIds.end(), imageTypeId) ==
|
||||
atomicImageTypeIds.end()) {
|
||||
atomicImageTypeIds.push_back(imageTypeId);
|
||||
}
|
||||
}
|
||||
offset += wordCount;
|
||||
}
|
||||
|
||||
outputBinary = inputBinary;
|
||||
Bool hasFloatStorageImage = false;
|
||||
for (SizeT offset = kSpirvHeaderWordCount; offset < outputBinary.size();) {
|
||||
const Uint32 instructionWord = outputBinary[offset];
|
||||
const Uint32 wordCount = instructionWord >> 16u;
|
||||
const auto opcode = static_cast<spv::Op>(instructionWord & 0xffffu);
|
||||
|
||||
// OpTypeImage operands are: result id, sampled type, dim, depth, arrayed,
|
||||
// multisampled, sampled, image format, and an optional access qualifier.
|
||||
if (opcode == spv::Op::OpTypeImage && wordCount >= 9) {
|
||||
const Uint32 imageTypeId = outputBinary[offset + 1];
|
||||
const Uint32 sampledTypeId = outputBinary[offset + 2];
|
||||
const Uint32 sampled = outputBinary[offset + 7];
|
||||
const Bool hasFloatSampledType =
|
||||
std::find(floatTypeIds.begin(), floatTypeIds.end(), sampledTypeId) != floatTypeIds.end();
|
||||
const Bool usedByAtomic =
|
||||
std::find(atomicImageTypeIds.begin(), atomicImageTypeIds.end(), imageTypeId) !=
|
||||
atomicImageTypeIds.end();
|
||||
if (sampled == 2 && hasFloatSampledType && !usedByAtomic) {
|
||||
outputBinary[offset + 8] = static_cast<Uint32>(spv::ImageFormat::Unknown);
|
||||
hasFloatStorageImage = true;
|
||||
}
|
||||
}
|
||||
offset += wordCount;
|
||||
}
|
||||
|
||||
if (!hasFloatStorageImage) {
|
||||
return true;
|
||||
}
|
||||
|
||||
Vector<Uint32> addedCapabilities;
|
||||
const Uint32 capabilityInstruction =
|
||||
(2u << 16u) | static_cast<Uint32>(spv::Op::OpCapability);
|
||||
if (!hasReadWithoutFormatCapability) {
|
||||
addedCapabilities.push_back(capabilityInstruction);
|
||||
addedCapabilities.push_back(
|
||||
static_cast<Uint32>(spv::Capability::StorageImageReadWithoutFormat));
|
||||
}
|
||||
if (!hasWriteWithoutFormatCapability) {
|
||||
addedCapabilities.push_back(capabilityInstruction);
|
||||
addedCapabilities.push_back(
|
||||
static_cast<Uint32>(spv::Capability::StorageImageWriteWithoutFormat));
|
||||
}
|
||||
outputBinary.insert(outputBinary.begin() + static_cast<std::ptrdiff_t>(capabilityInsertOffset),
|
||||
addedCapabilities.begin(), addedCapabilities.end());
|
||||
return true;
|
||||
}
|
||||
|
||||
Result<String> ShaderCompiler::DecompileShader(SpvcSession& session) {
|
||||
spvc_compiler_options options;
|
||||
session.CreateOptions(&options);
|
||||
|
||||
@@ -33,12 +33,37 @@ namespace MobileGL {
|
||||
// Only for the DirectGLES transpile path.
|
||||
static bool StripUboMemberRelaxedPrecisionForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Removes NoPerspective decorations so SPIRV-Cross emits plain (smooth) ESSL varyings.
|
||||
// DirectGLES fallback only, for devices lacking GL_NV_shader_noperspective_interpolation
|
||||
// (SPIRV-Cross would otherwise require that extension and the driver would reject it).
|
||||
static bool StripNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Emulates noperspective (screen-linear) interpolation via gl_Position.w / gl_FragCoord.w
|
||||
// so no NV extension is needed; strips what it cannot emulate. DirectGLES fallback for
|
||||
// devices lacking GL_NV_shader_noperspective_interpolation. See EmulateNoPerspectivePass.
|
||||
static bool EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Rebases loads of the InstanceIndex builtin to (InstanceIndex - BaseInstance) so
|
||||
// shaders see GL's zero-based gl_InstanceID. Vertex shaders only; DirectVulkan
|
||||
// backend only (glslang's relaxed mode aliases gl_InstanceID to gl_InstanceIndex,
|
||||
// which wrongly includes baseInstance).
|
||||
static bool RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Adds the Invariant decoration to every Position builtin output. GL apps
|
||||
// routinely rely on cross-program position invariance for multi-pass
|
||||
// equality depth tests (e.g. GEQUAL re-draws of the same geometry), and
|
||||
// mobile drivers that optimize per-pipeline break that without the
|
||||
// decoration. DirectVulkan only.
|
||||
static bool DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Replaces the declared format of float storage images with Unknown and adds the
|
||||
// matching SPIR-V capabilities. DirectVulkan uses this only when both Vulkan
|
||||
// shaderStorageImage*WithoutFormat features are enabled, allowing the
|
||||
// glBindImageTexture format to select the descriptor view at runtime. Integer
|
||||
// storage images deliberately keep their declared format for GL-compatible bit
|
||||
// reinterpretation paths (for example, R32F storage accessed as r32ui).
|
||||
static bool UseUnformattedFloatStorageImagesForVulkan(
|
||||
const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary);
|
||||
static Result<String> DecompileShader(SpvcSession& session);
|
||||
};
|
||||
} // namespace ShaderTranspiler
|
||||
|
||||
@@ -10,10 +10,15 @@
|
||||
|
||||
#include <algorithm>
|
||||
#include <cctype>
|
||||
#include <initializer_list>
|
||||
#include <utility>
|
||||
#include <Config.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
|
||||
namespace {
|
||||
using MobileGL::SizeT;
|
||||
using MobileGL::String;
|
||||
using MobileGL::Vector;
|
||||
|
||||
bool IsIdentifierChar(char ch) {
|
||||
return (ch >= '0' && ch <= '9') || (ch >= 'A' && ch <= 'Z') || (ch >= 'a' && ch <= 'z') || ch == '_';
|
||||
@@ -91,6 +96,460 @@ namespace {
|
||||
return masked;
|
||||
}
|
||||
|
||||
// Blank out block comments in place, leaving line comments and every other byte where it is.
|
||||
//
|
||||
// The passes that follow scan the source as raw text, so block comments have to stop being
|
||||
// visible to them - but they must not be *deleted*: replacing the bytes with spaces keeps every
|
||||
// later offset valid and keeps newlines, so glslang's diagnostics still point at the line the
|
||||
// application wrote. It also has to be lexically aware. A banner line such as
|
||||
//
|
||||
// //*** lighting pass ***
|
||||
//
|
||||
// contains "/*" one byte in, and a naive search for that opener treats the rest of the file as
|
||||
// an unterminated comment.
|
||||
void BlankBlockComments(MobileGL::String& source) {
|
||||
enum class Region { Code, SingleLineComment, MultiLineComment, QuotedText };
|
||||
|
||||
Region region = Region::Code;
|
||||
char quote = '\0';
|
||||
bool escaped = false;
|
||||
|
||||
for (SizeT pos = 0; pos < source.size(); pos++) {
|
||||
const char ch = source[pos];
|
||||
const char next = pos + 1 < source.size() ? source[pos + 1] : '\0';
|
||||
|
||||
if (region == Region::Code) {
|
||||
if (ch == '/' && next == '/') {
|
||||
pos++;
|
||||
region = Region::SingleLineComment;
|
||||
} else if (ch == '/' && next == '*') {
|
||||
source[pos] = ' ';
|
||||
source[pos + 1] = ' ';
|
||||
pos++;
|
||||
region = Region::MultiLineComment;
|
||||
} else if (ch == '"' || ch == '\'') {
|
||||
quote = ch;
|
||||
escaped = false;
|
||||
region = Region::QuotedText;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (region == Region::SingleLineComment) {
|
||||
if (ch == '\n' || ch == '\r') region = Region::Code;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (region == Region::MultiLineComment) {
|
||||
if (ch == '*' && next == '/') {
|
||||
source[pos] = ' ';
|
||||
source[pos + 1] = ' ';
|
||||
pos++;
|
||||
region = Region::Code;
|
||||
} else if (ch != '\n' && ch != '\r') {
|
||||
source[pos] = ' ';
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// GLSL has no multi-line string literals, so a quote that reaches end of line was never
|
||||
// a literal to begin with - most likely an apostrophe in a #error or #pragma message.
|
||||
// Ending the region here keeps one stray apostrophe from swallowing the rest of the file.
|
||||
if (ch == '\n' || ch == '\r') {
|
||||
region = Region::Code;
|
||||
} else if (escaped) {
|
||||
escaped = false;
|
||||
} else if (ch == '\\') {
|
||||
escaped = true;
|
||||
} else if (ch == quote) {
|
||||
region = Region::Code;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct CodeToken {
|
||||
String text;
|
||||
SizeT begin = 0;
|
||||
SizeT end = 0;
|
||||
};
|
||||
|
||||
Vector<CodeToken> TokenizeCode(const String& source) {
|
||||
const String masked = MaskCommentsAndQuotedText(source);
|
||||
Vector<CodeToken> tokens;
|
||||
tokens.reserve(source.size() / 4);
|
||||
|
||||
SizeT pos = 0;
|
||||
while (pos < masked.size()) {
|
||||
const char ch = masked[pos];
|
||||
if (std::isspace(static_cast<unsigned char>(ch))) {
|
||||
++pos;
|
||||
continue;
|
||||
}
|
||||
|
||||
const SizeT begin = pos;
|
||||
if (IsIdentifierStart(ch)) {
|
||||
++pos;
|
||||
while (pos < masked.size() && IsIdentifierChar(masked[pos])) {
|
||||
++pos;
|
||||
}
|
||||
} else if (std::isdigit(static_cast<unsigned char>(ch))) {
|
||||
++pos;
|
||||
while (pos < masked.size()) {
|
||||
const char numberChar = masked[pos];
|
||||
if (!IsIdentifierChar(numberChar) && numberChar != '.') {
|
||||
break;
|
||||
}
|
||||
++pos;
|
||||
}
|
||||
} else {
|
||||
++pos;
|
||||
if (pos < masked.size()) {
|
||||
const String twoChars = masked.substr(begin, 2);
|
||||
if (twoChars == "==" || twoChars == "!=" || twoChars == "<=" || twoChars == ">=" ||
|
||||
twoChars == "+=" || twoChars == "-=" || twoChars == "<<" || twoChars == ">>" ||
|
||||
twoChars == "++" || twoChars == "--" || twoChars == "&&" || twoChars == "||") {
|
||||
++pos;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
tokens.push_back(CodeToken{source.substr(begin, pos - begin), begin, pos});
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
|
||||
bool IsIdentifierToken(const CodeToken& token) {
|
||||
if (token.text.empty() || !IsIdentifierStart(token.text.front())) {
|
||||
return false;
|
||||
}
|
||||
return std::all_of(token.text.begin() + 1, token.text.end(), IsIdentifierChar);
|
||||
}
|
||||
|
||||
class TokenCursor {
|
||||
public:
|
||||
TokenCursor(const Vector<CodeToken>& tokens, SizeT position) : m_tokens(tokens), m_position(position) {}
|
||||
|
||||
bool Consume(const char* expected) {
|
||||
if (m_position >= m_tokens.size() || m_tokens[m_position].text != expected) {
|
||||
return false;
|
||||
}
|
||||
++m_position;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ConsumeAnyIdentifier(String& identifier) {
|
||||
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position])) {
|
||||
return false;
|
||||
}
|
||||
identifier = m_tokens[m_position++].text;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ConsumeAnyIdentifier() {
|
||||
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position])) {
|
||||
return false;
|
||||
}
|
||||
++m_position;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ConsumeIdentifier(const String& expected) {
|
||||
if (m_position >= m_tokens.size() || !IsIdentifierToken(m_tokens[m_position]) ||
|
||||
m_tokens[m_position].text != expected) {
|
||||
return false;
|
||||
}
|
||||
++m_position;
|
||||
return true;
|
||||
}
|
||||
|
||||
SizeT Position() const { return m_position; }
|
||||
|
||||
private:
|
||||
const Vector<CodeToken>& m_tokens;
|
||||
SizeT m_position;
|
||||
};
|
||||
|
||||
SizeT CountToken(const Vector<CodeToken>& tokens, const String& tokenText) {
|
||||
return static_cast<SizeT>(std::count_if(tokens.begin(), tokens.end(),
|
||||
[&](const CodeToken& token) { return token.text == tokenText; }));
|
||||
}
|
||||
|
||||
bool HasIdentifierWithPrefixOutsideAllowed(const Vector<CodeToken>& tokens, const String& prefix,
|
||||
std::initializer_list<const char*> allowedIdentifiers) {
|
||||
return std::any_of(tokens.begin(), tokens.end(), [&](const CodeToken& token) {
|
||||
if (!IsIdentifierToken(token) || !token.text.starts_with(prefix)) {
|
||||
return false;
|
||||
}
|
||||
return std::none_of(allowedIdentifiers.begin(), allowedIdentifiers.end(),
|
||||
[&](const char* allowed) { return token.text == allowed; });
|
||||
});
|
||||
}
|
||||
|
||||
bool MatchTokenSequence(const Vector<CodeToken>& tokens, SizeT position,
|
||||
std::initializer_list<const char*> expected) {
|
||||
if (position + expected.size() > tokens.size()) {
|
||||
return false;
|
||||
}
|
||||
for (const char* token : expected) {
|
||||
if (tokens[position++].text != token) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
struct LinearPrefixScanMatch {
|
||||
SizeT sharedArraySizeBegin = 0;
|
||||
SizeT sharedArraySizeEnd = 0;
|
||||
SizeT scanBegin = 0;
|
||||
SizeT scanEnd = 0;
|
||||
String cache;
|
||||
String importance;
|
||||
String prefixSum;
|
||||
String loopLength;
|
||||
String loopIndex;
|
||||
String sum;
|
||||
};
|
||||
|
||||
bool ParseLinearPrefixScanTemplate(const Vector<CodeToken>& tokens, LinearPrefixScanMatch& match) {
|
||||
// The workaround deliberately recognizes one complete algorithm, not merely the
|
||||
// subgroupInclusiveAdd token. Changing scratch storage is only safe when that storage is
|
||||
// private to this scan and the workgroup has exactly 1024 X invocations.
|
||||
SizeT localSizeDeclarationCount = 0;
|
||||
for (SizeT i = 0; i < tokens.size(); ++i) {
|
||||
if (MatchTokenSequence(tokens, i, {"layout", "(", "local_size_x", "=", "1024", ")", "in", ";"})) {
|
||||
++localSizeDeclarationCount;
|
||||
}
|
||||
}
|
||||
if (localSizeDeclarationCount != 1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
SizeT sharedDeclarationIndex = String::npos;
|
||||
SizeT sharedDeclarationCount = 0;
|
||||
String cacheName;
|
||||
for (SizeT i = 0; i + 6 < tokens.size(); ++i) {
|
||||
if (tokens[i].text != "shared" || tokens[i + 1].text != "float" || !IsIdentifierToken(tokens[i + 2]) ||
|
||||
tokens[i + 3].text != "[" || tokens[i + 4].text != "64" || tokens[i + 5].text != "]" ||
|
||||
tokens[i + 6].text != ";") {
|
||||
continue;
|
||||
}
|
||||
++sharedDeclarationCount;
|
||||
sharedDeclarationIndex = i;
|
||||
cacheName = tokens[i + 2].text;
|
||||
}
|
||||
if (sharedDeclarationCount != 1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
SizeT scanTokenIndex = String::npos;
|
||||
SizeT scanCount = 0;
|
||||
for (SizeT i = 0; i + 7 < tokens.size(); ++i) {
|
||||
if (tokens[i].text == "float" && IsIdentifierToken(tokens[i + 1]) && tokens[i + 2].text == "=" &&
|
||||
tokens[i + 3].text == "subgroupInclusiveAdd" && tokens[i + 4].text == "(" &&
|
||||
IsIdentifierToken(tokens[i + 5]) && tokens[i + 6].text == ")" && tokens[i + 7].text == ";") {
|
||||
++scanCount;
|
||||
scanTokenIndex = i;
|
||||
}
|
||||
}
|
||||
if (scanCount != 1 || sharedDeclarationIndex >= scanTokenIndex) {
|
||||
return false;
|
||||
}
|
||||
|
||||
TokenCursor cursor(tokens, scanTokenIndex);
|
||||
String prefixSum;
|
||||
String importance;
|
||||
String loopLength;
|
||||
String loopIndex;
|
||||
String sum;
|
||||
if (!cursor.Consume("float") || !cursor.ConsumeAnyIdentifier(prefixSum) || !cursor.Consume("=") ||
|
||||
!cursor.Consume("subgroupInclusiveAdd") || !cursor.Consume("(") ||
|
||||
!cursor.ConsumeAnyIdentifier(importance) || !cursor.Consume(")") || !cursor.Consume(";") ||
|
||||
!cursor.Consume("if") || !cursor.Consume("(") || !cursor.Consume("gl_SubgroupInvocationID") ||
|
||||
!cursor.Consume("==") || !cursor.Consume("gl_SubgroupSize") || !cursor.Consume("-") ||
|
||||
!cursor.Consume("1u") || !cursor.Consume(")") || !cursor.ConsumeIdentifier(cacheName) ||
|
||||
!cursor.Consume("[") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume("]") || !cursor.Consume("=") ||
|
||||
!cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume(";") || !cursor.Consume("barrier") ||
|
||||
!cursor.Consume("(") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("uint") ||
|
||||
!cursor.ConsumeAnyIdentifier(loopLength) || !cursor.Consume("=") || !cursor.Consume("uint") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("findMSB") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("gl_NumSubgroups") || !cursor.Consume(")") || !cursor.Consume(")") ||
|
||||
!cursor.Consume(";") || !cursor.ConsumeIdentifier(loopLength) || !cursor.Consume("+=") ||
|
||||
!cursor.Consume("uint") || !cursor.Consume("(") || !cursor.Consume("gl_NumSubgroups") ||
|
||||
!cursor.Consume("-") || !cursor.Consume("(") || !cursor.Consume("1u") || !cursor.Consume("<<") ||
|
||||
!cursor.Consume("(") || !cursor.ConsumeIdentifier(loopLength) || !cursor.Consume("-") ||
|
||||
!cursor.Consume("1u") || !cursor.Consume(")") || !cursor.Consume(")") || !cursor.Consume(">") ||
|
||||
!cursor.Consume("0u") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("for") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("uint") || !cursor.ConsumeAnyIdentifier(loopIndex) ||
|
||||
!cursor.Consume("=") || !cursor.Consume("0") || !cursor.Consume(";") ||
|
||||
!cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("<") || !cursor.ConsumeIdentifier(loopLength) ||
|
||||
!cursor.Consume(";") || !cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("++") ||
|
||||
!cursor.Consume(")") || !cursor.Consume("{") || !cursor.Consume("if") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume("&") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("1u") || !cursor.Consume("<<") || !cursor.ConsumeIdentifier(loopIndex) ||
|
||||
!cursor.Consume(")") || !cursor.Consume(")") || !cursor.Consume(">") || !cursor.Consume("0u") ||
|
||||
!cursor.Consume(")") || !cursor.Consume("{") || !cursor.ConsumeIdentifier(prefixSum) ||
|
||||
!cursor.Consume("+=") || !cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_SubgroupID") || !cursor.Consume(">>") ||
|
||||
!cursor.ConsumeIdentifier(loopIndex) || !cursor.Consume("<<") || !cursor.ConsumeIdentifier(loopIndex) ||
|
||||
!cursor.Consume(")") || !cursor.Consume("-") || !cursor.Consume("1u") || !cursor.Consume("]") ||
|
||||
!cursor.Consume(";") || !cursor.Consume("if") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("gl_SubgroupInvocationID") || !cursor.Consume("==") || !cursor.Consume("gl_SubgroupSize") ||
|
||||
!cursor.Consume("-") || !cursor.Consume("1u") || !cursor.Consume(")") ||
|
||||
!cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") || !cursor.Consume("gl_SubgroupID") ||
|
||||
!cursor.Consume("]") || !cursor.Consume("=") || !cursor.ConsumeIdentifier(prefixSum) ||
|
||||
!cursor.Consume(";") || !cursor.Consume("}") || !cursor.Consume("barrier") || !cursor.Consume("(") ||
|
||||
!cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("}") || !cursor.Consume("if") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_LocalInvocationID") || !cursor.Consume(".") ||
|
||||
!cursor.Consume("x") || !cursor.Consume("==") || !cursor.Consume("uint") || !cursor.Consume("(") ||
|
||||
!cursor.Consume("1024") || !cursor.Consume("-") || !cursor.Consume("1") || !cursor.Consume(")") ||
|
||||
!cursor.Consume(")") || !cursor.ConsumeIdentifier(cacheName) || !cursor.Consume("[") ||
|
||||
!cursor.Consume("0") || !cursor.Consume("]") || !cursor.Consume("=") ||
|
||||
!cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume(";") || !cursor.Consume("barrier") ||
|
||||
!cursor.Consume("(") || !cursor.Consume(")") || !cursor.Consume(";") || !cursor.Consume("float") ||
|
||||
!cursor.ConsumeAnyIdentifier(sum) || !cursor.Consume("=") || !cursor.ConsumeIdentifier(cacheName) ||
|
||||
!cursor.Consume("[") || !cursor.Consume("0") || !cursor.Consume("]") || !cursor.Consume(";")) {
|
||||
return false;
|
||||
}
|
||||
const SizeT scanEndToken = cursor.Position() - 1;
|
||||
|
||||
// Require the scan's immediate consumer as well. This makes the match specific to a
|
||||
// linear distribution warp, and avoids changing unrelated prefix scans which may rely on
|
||||
// the implementation's native subgroup partitioning.
|
||||
if (!cursor.Consume("float") || !cursor.ConsumeAnyIdentifier() || !cursor.Consume("=") ||
|
||||
!cursor.Consume("(") || !cursor.ConsumeIdentifier(prefixSum) || !cursor.Consume("-") ||
|
||||
!cursor.ConsumeIdentifier(importance) || !cursor.Consume(")") || !cursor.Consume("/") ||
|
||||
!cursor.ConsumeIdentifier(sum) || !cursor.Consume("-") || !cursor.Consume("float") ||
|
||||
!cursor.Consume("(") || !cursor.Consume("gl_LocalInvocationID") || !cursor.Consume(".") ||
|
||||
!cursor.Consume("x") || !cursor.Consume("+") || !cursor.Consume("1u") || !cursor.Consume(")") ||
|
||||
!cursor.Consume("/") || !cursor.Consume("float") || !cursor.Consume("(") || !cursor.Consume("1024") ||
|
||||
!cursor.Consume(")") || !cursor.Consume(";")) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// No other use may share the scratch array, and no additional subgroup operation or
|
||||
// builtin may silently retain native-64 semantics after this module becomes virtual-32.
|
||||
if (CountToken(tokens, cacheName) != 6 || CountToken(tokens, "subgroupInclusiveAdd") != 1 ||
|
||||
CountToken(tokens, "gl_SubgroupInvocationID") != 2 || CountToken(tokens, "gl_SubgroupSize") != 2 ||
|
||||
CountToken(tokens, "gl_SubgroupID") != 4 || CountToken(tokens, "gl_NumSubgroups") != 2 ||
|
||||
CountToken(tokens, "gl_LocalInvocationID") != 2 || CountToken(tokens, "barrier") != 3 ||
|
||||
CountToken(tokens, "findMSB") != 1 ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "subgroup", {"subgroupInclusiveAdd"}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(
|
||||
tokens, "gl_Subgroup",
|
||||
{"gl_SubgroupInvocationID", "gl_SubgroupSize", "gl_SubgroupID", "gl_NumSubgroups"}) ||
|
||||
// ARB/NV spellings of lane-width-sensitive builtins and functions
|
||||
// (gl_SubGroupSizeARB, ballotARB, gl_WarpSizeNV, shuffleNV, ...) must block the
|
||||
// rewrite just like their KHR counterparts: they would silently keep native-width
|
||||
// semantics in a module rewritten to the virtual 32-lane model.
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SubGroup", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Warp", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_Thread", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "gl_SMID", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "ballot", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "shuffle", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "readInvocation", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "readFirstInvocation", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "anyInvocation", {}) ||
|
||||
HasIdentifierWithPrefixOutsideAllowed(tokens, "allInvocations", {})) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// The scan must be at the top level of the sole main() body. Its existing barriers already
|
||||
// require uniform control flow; this check prevents us from introducing extra barriers in
|
||||
// a nested branch or loop.
|
||||
SizeT mainOpenBrace = String::npos;
|
||||
SizeT mainCloseBrace = String::npos;
|
||||
SizeT mainCount = 0;
|
||||
for (SizeT i = 0; i + 4 < tokens.size(); ++i) {
|
||||
if (!MatchTokenSequence(tokens, i, {"void", "main", "(", ")", "{"})) {
|
||||
continue;
|
||||
}
|
||||
++mainCount;
|
||||
mainOpenBrace = i + 4;
|
||||
int depth = 1;
|
||||
for (SizeT j = mainOpenBrace + 1; j < tokens.size(); ++j) {
|
||||
if (tokens[j].text == "{")
|
||||
++depth;
|
||||
else if (tokens[j].text == "}" && --depth == 0) {
|
||||
mainCloseBrace = j;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (mainCount != 1 || mainCloseBrace == String::npos || scanTokenIndex <= mainOpenBrace ||
|
||||
scanEndToken >= mainCloseBrace) {
|
||||
return false;
|
||||
}
|
||||
int depthAtScan = 1;
|
||||
for (SizeT i = mainOpenBrace + 1; i < scanTokenIndex; ++i) {
|
||||
if (tokens[i].text == "{")
|
||||
++depthAtScan;
|
||||
else if (tokens[i].text == "}")
|
||||
--depthAtScan;
|
||||
}
|
||||
if (depthAtScan != 1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
constexpr const char* injectedNames[] = {"mglPrefixScanLane", "mglVirtualSubgroupInvocation",
|
||||
"mglVirtualSubgroup", "mglVirtualSubgroupBase",
|
||||
"mglPrefixLane", "mglVirtualSubgroupCount"};
|
||||
for (const char* injectedName : injectedNames) {
|
||||
if (CountToken(tokens, injectedName) != 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
match.sharedArraySizeBegin = tokens[sharedDeclarationIndex + 4].begin;
|
||||
match.sharedArraySizeEnd = tokens[sharedDeclarationIndex + 4].end;
|
||||
match.scanBegin = tokens[scanTokenIndex].begin;
|
||||
match.scanEnd = tokens[scanEndToken].end;
|
||||
match.cache = std::move(cacheName);
|
||||
match.importance = std::move(importance);
|
||||
match.prefixSum = std::move(prefixSum);
|
||||
match.loopLength = std::move(loopLength);
|
||||
match.loopIndex = std::move(loopIndex);
|
||||
match.sum = std::move(sum);
|
||||
return true;
|
||||
}
|
||||
|
||||
String BuildLinearPrefixScanReplacement(const LinearPrefixScanMatch& match) {
|
||||
String replacement;
|
||||
replacement.reserve(1800);
|
||||
replacement += "uint mglPrefixScanLane = gl_LocalInvocationID.x;\n";
|
||||
replacement += "uint mglVirtualSubgroupInvocation = mglPrefixScanLane & 31u;\n";
|
||||
replacement += "uint mglVirtualSubgroup = mglPrefixScanLane >> 5u;\n";
|
||||
replacement += "const uint mglVirtualSubgroupCount = 32u;\n";
|
||||
replacement += match.cache + "[mglPrefixScanLane] = " + match.importance + ";\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "float " + match.prefixSum + " = 0.0f;\n";
|
||||
replacement += "uint mglVirtualSubgroupBase = mglVirtualSubgroup << 5u;\n";
|
||||
replacement += "for (uint mglPrefixLane = mglVirtualSubgroupBase; "
|
||||
"mglPrefixLane <= mglPrefixScanLane; ++mglPrefixLane) {\n";
|
||||
replacement += match.prefixSum + " += " + match.cache + "[mglPrefixLane];\n";
|
||||
replacement += "}\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "if (mglVirtualSubgroupInvocation == 31u) " + match.cache +
|
||||
"[mglVirtualSubgroup] = " + match.prefixSum + ";\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "uint " + match.loopLength + " = uint(findMSB(mglVirtualSubgroupCount));\n";
|
||||
replacement +=
|
||||
match.loopLength + " += uint(mglVirtualSubgroupCount - (1u << (" + match.loopLength + " - 1u)) > 0u);\n";
|
||||
replacement += "for (uint " + match.loopIndex + " = 0u; " + match.loopIndex + " < " + match.loopLength +
|
||||
"; ++" + match.loopIndex + ") {\n";
|
||||
replacement += "if ((mglVirtualSubgroup & (1u << " + match.loopIndex + ")) > 0u) {\n";
|
||||
replacement += match.prefixSum + " += " + match.cache + "[(mglVirtualSubgroup >> " + match.loopIndex + " << " +
|
||||
match.loopIndex + ") - 1u];\n";
|
||||
replacement += "if (mglVirtualSubgroupInvocation == 31u) " + match.cache +
|
||||
"[mglVirtualSubgroup] = " + match.prefixSum + ";\n";
|
||||
replacement += "}\nbarrier();\n}\n";
|
||||
replacement += "if (mglPrefixScanLane == 1023u) " + match.cache + "[0] = " + match.prefixSum + ";\n";
|
||||
replacement += "barrier();\n";
|
||||
replacement += "float " + match.sum + " = " + match.cache + "[0];";
|
||||
return replacement;
|
||||
}
|
||||
|
||||
void SkipDirectiveWhitespace(const MobileGL::String& source, SizeT& pos, SizeT lineEnd) {
|
||||
while (pos < lineEnd && std::isspace(static_cast<unsigned char>(source[pos]))) {
|
||||
pos++;
|
||||
@@ -114,6 +573,23 @@ namespace {
|
||||
static_cast<unsigned char>(source[1]) == 0xbb && static_cast<unsigned char>(source[2]) == 0xbf;
|
||||
}
|
||||
|
||||
// The GLSL versions MobileGL is willing to normalize. Anything else in a #version line - a number
|
||||
// that is not a real language version (329, 331), a bad profile keyword, a float/identifier where
|
||||
// the integer belongs, or trailing tokens - is left untouched so glslang rejects it, matching
|
||||
// KHR-GL33.shaders.preprocessor.directive.version_*. The set is deliberately generous (every real
|
||||
// desktop and ES version) so the normalizer never starts rejecting a form it used to accept.
|
||||
bool IsRecognizedGlslVersion(unsigned version) {
|
||||
switch (version) {
|
||||
case 100: case 110: case 120: case 130: case 140: case 150:
|
||||
case 300: case 310: case 320:
|
||||
case 330: case 400: case 410: case 420: case 430:
|
||||
case 440: case 450: case 460:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
struct ShaderLanguageInfo {
|
||||
unsigned version = 110;
|
||||
MobileGL::ShaderProfile profile = MobileGL::ShaderProfile::Core;
|
||||
@@ -121,6 +597,9 @@ namespace {
|
||||
SizeT versionDirectiveEnd = MobileGL::String::npos;
|
||||
bool hasUtf8Bom = false;
|
||||
bool enablesGpuShader5 = false;
|
||||
// Whether the parsed #version directive is a well-formed one MobileGL should rewrite. A
|
||||
// malformed directive (see IsRecognizedGlslVersion) is left alone for glslang to reject.
|
||||
bool hasValidVersionDirective = false;
|
||||
|
||||
bool HasVersionDirective() const { return versionDirectiveStart != MobileGL::String::npos; }
|
||||
};
|
||||
@@ -164,13 +643,25 @@ namespace {
|
||||
info.versionDirectiveEnd = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
SkipDirectiveWhitespace(code, probe, lineEnd);
|
||||
const MobileGL::String profile = ReadDirectiveIdentifier(code, probe, lineEnd);
|
||||
if (profile == "es" || profile == "ES") {
|
||||
bool profileTokenValid = true;
|
||||
if (profile.empty() || profile == "core") {
|
||||
info.profile = MobileGL::ShaderProfile::Core;
|
||||
} else if (profile == "es" || profile == "ES") {
|
||||
info.profile = MobileGL::ShaderProfile::ES;
|
||||
} else if (profile == "compatibility") {
|
||||
info.profile = MobileGL::ShaderProfile::Compatibility;
|
||||
} else {
|
||||
// "#version 330 foo": an unrecognized profile keyword. Keep Core for any
|
||||
// downstream routing, but mark the directive malformed.
|
||||
info.profile = MobileGL::ShaderProfile::Core;
|
||||
profileTokenValid = false;
|
||||
}
|
||||
// Comments are already masked to spaces, so anything non-blank left on the
|
||||
// line is real trailing garbage: "#version 330 foobar" / "#version 330.0".
|
||||
SkipDirectiveWhitespace(code, probe, lineEnd);
|
||||
const bool hasTrailingTokens = probe < lineEnd;
|
||||
info.hasValidVersionDirective =
|
||||
IsRecognizedGlslVersion(info.version) && profileTokenValid && !hasTrailingTokens;
|
||||
}
|
||||
} else if (directive == "extension") {
|
||||
SkipDirectiveWhitespace(code, probe, lineEnd);
|
||||
@@ -217,6 +708,17 @@ namespace {
|
||||
}
|
||||
|
||||
void NormalizeVersionDirective(MobileGL::String& source, const ShaderLanguageInfo& info) {
|
||||
// A malformed #version (329, 331, bad profile, float/trailing tokens) is left exactly as the
|
||||
// application wrote it so glslang rejects it - rewriting it to "#version 330 core" would
|
||||
// silently legalize the CTS directive.version_* rejection cases. Still drop a leading BOM so
|
||||
// the reported error is the bad version rather than a stray byte-order mark.
|
||||
if (info.HasVersionDirective() && !info.hasValidVersionDirective) {
|
||||
if (info.hasUtf8Bom) {
|
||||
source.erase(0, 3);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const MobileGL::String replacement = GetNormalizedVersionDirective(info);
|
||||
if (info.HasVersionDirective()) {
|
||||
source.replace(info.versionDirectiveStart, info.versionDirectiveEnd - info.versionDirectiveStart,
|
||||
@@ -298,7 +800,10 @@ namespace {
|
||||
|
||||
void RenameBuiltinShadowingFunction(MobileGL::String& source, const char* from, const char* to) {
|
||||
const MobileGL::String fromName = from;
|
||||
if (!HasSingleLineFunctionDefinition(source, fromName)) {
|
||||
// Decide from a comment-free view. A commented-out definition is not a definition, and
|
||||
// acting on one renames every genuine call to the builtin to a name nothing defines - which
|
||||
// then fails to resolve. Line comments survive BlankBlockComments, so this matters.
|
||||
if (!HasSingleLineFunctionDefinition(MaskCommentsAndQuotedText(source), fromName)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -371,6 +876,58 @@ namespace {
|
||||
return info.HasVersionDirective() ? info.versionDirectiveEnd : 0;
|
||||
}
|
||||
|
||||
// GLSL's #line takes integer expressions only, but plenty of shader-pack preprocessors emit the
|
||||
// C form with a quoted filename. Deleting every #line outright made those harmless - at the cost
|
||||
// of __LINE__ reporting the position in MobileGL's rewritten text rather than the one the pack
|
||||
// author wrote, and of every later diagnostic pointing at the wrong line. Dropping just the
|
||||
// quoted operand keeps the directive doing its job and still hands glslang something it accepts.
|
||||
void NormalizeLineDirectives(MobileGL::String& source) {
|
||||
const MobileGL::String masked = MaskCommentsAndQuotedText(source);
|
||||
const SizeT versionEnd = FindAfterVersionDirective(source);
|
||||
MobileGL::String result;
|
||||
result.reserve(source.size());
|
||||
|
||||
SizeT lineStart = 0;
|
||||
while (lineStart <= source.size()) {
|
||||
SizeT lineEnd = source.find('\n', lineStart);
|
||||
const bool lastLine = lineEnd == MobileGL::String::npos;
|
||||
if (lastLine) lineEnd = source.size();
|
||||
|
||||
SizeT probe = lineStart;
|
||||
while (probe < lineEnd && (source[probe] == ' ' || source[probe] == '\t')) probe++;
|
||||
|
||||
const bool isLineDirective = masked.compare(probe, 5, "#line") == 0 &&
|
||||
(probe + 5 >= lineEnd || !IsIdentifierChar(source[probe + 5]));
|
||||
if (isLineDirective && lineStart < versionEnd) {
|
||||
// #version has to be the first token in the shader, so a #line ahead of it could
|
||||
// never have taken effect. Drop it rather than hand glslang a source it must reject
|
||||
// - some pack preprocessors emit their directives before the version line.
|
||||
} else if (isLineDirective) {
|
||||
// Keep everything up to the first quote that the masker identified as string text.
|
||||
SizeT quotePos = MobileGL::String::npos;
|
||||
for (SizeT i = probe + 5; i < lineEnd; i++) {
|
||||
if (source[i] == '"' || source[i] == '\'') {
|
||||
quotePos = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (quotePos != MobileGL::String::npos) {
|
||||
result.append(source, lineStart, quotePos - lineStart);
|
||||
} else {
|
||||
result.append(source, lineStart, lineEnd - lineStart);
|
||||
}
|
||||
} else {
|
||||
result.append(source, lineStart, lineEnd - lineStart);
|
||||
}
|
||||
|
||||
if (lastLine) break;
|
||||
result.push_back('\n');
|
||||
lineStart = lineEnd + 1;
|
||||
}
|
||||
|
||||
source = std::move(result);
|
||||
}
|
||||
|
||||
bool IsExtensionAdvertised(MobileGL::GLExtension extension) {
|
||||
const auto& activeBackendObject = MobileGL::MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackendObject) {
|
||||
@@ -399,15 +956,28 @@ namespace {
|
||||
return;
|
||||
}
|
||||
|
||||
// Detect the directive on a comment/string-masked copy so a commented-out
|
||||
// "#extension GL_ARB_gpu_shader_int64" is never turned into a synthesized #error. Comments are
|
||||
// no longer blanked in the delivered source (glslang handles them), so this pass must mask
|
||||
// locally like its siblings. Masking preserves offsets, so edits collected against the scan
|
||||
// apply verbatim to `source`; they are applied back-to-front to keep earlier offsets valid.
|
||||
const MobileGL::String scan = MaskCommentsAndQuotedText(source);
|
||||
struct DirectiveEdit {
|
||||
SizeT pos;
|
||||
SizeT len;
|
||||
MobileGL::String replacement;
|
||||
};
|
||||
Vector<DirectiveEdit> edits;
|
||||
|
||||
SizeT lineStart = 0;
|
||||
while (lineStart < source.size()) {
|
||||
SizeT lineEnd = source.find('\n', lineStart);
|
||||
while (lineStart < scan.size()) {
|
||||
SizeT lineEnd = scan.find('\n', lineStart);
|
||||
const bool hasLineBreak = lineEnd != MobileGL::String::npos;
|
||||
if (!hasLineBreak) {
|
||||
lineEnd = source.size();
|
||||
lineEnd = scan.size();
|
||||
}
|
||||
|
||||
const MobileGL::String line = source.substr(lineStart, lineEnd - lineStart);
|
||||
const MobileGL::String line = scan.substr(lineStart, lineEnd - lineStart);
|
||||
SizeT probe = 0;
|
||||
while (probe < line.size() && std::isspace(static_cast<unsigned char>(line[probe]))) {
|
||||
probe++;
|
||||
@@ -449,16 +1019,12 @@ namespace {
|
||||
const MobileGL::String behavior = TrimDirectiveToken(line.substr(probe));
|
||||
const SizeT replaceLen = lineEnd - lineStart + (hasLineBreak ? 1 : 0);
|
||||
if (behavior == "require") {
|
||||
const MobileGL::String replacement =
|
||||
"#error GL_ARB_gpu_shader_int64 is not advertised by MobileGL\n";
|
||||
source.replace(lineStart, replaceLen, replacement);
|
||||
lineStart += replacement.size();
|
||||
edits.push_back({lineStart, replaceLen,
|
||||
"#error GL_ARB_gpu_shader_int64 is not advertised by MobileGL\n"});
|
||||
} else if (behavior == "enable" || behavior == "warn") {
|
||||
source.replace(lineStart, replaceLen, "\n");
|
||||
lineStart++;
|
||||
} else {
|
||||
lineStart = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
edits.push_back({lineStart, replaceLen, "\n"});
|
||||
}
|
||||
lineStart = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -468,6 +1034,10 @@ namespace {
|
||||
lineStart = lineEnd + (hasLineBreak ? 1 : 0);
|
||||
}
|
||||
|
||||
for (auto it = edits.rbegin(); it != edits.rend(); ++it) {
|
||||
source.replace(it->pos, it->len, it->replacement);
|
||||
}
|
||||
|
||||
ReplaceIdentifier(source, "GL_ARB_gpu_shader_int64", "MG_DISABLED_GL_ARB_gpu_shader_int64");
|
||||
}
|
||||
|
||||
@@ -576,48 +1146,136 @@ namespace {
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
Bool RewriteLinearSubgroupPrefixScanForVulkan(ShaderStage stage, Uint32 nativeSubgroupSize,
|
||||
String& source) {
|
||||
constexpr Uint32 capturedSubgroupSize = 32;
|
||||
if (stage != ShaderStage::Compute || nativeSubgroupSize <= capturedSubgroupSize ||
|
||||
nativeSubgroupSize % capturedSubgroupSize != 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Vulkan subgroup widths are powers of two. Keep the workaround restricted to
|
||||
// wider widths which are a power-of-two multiple of the captured 32-lane model.
|
||||
const Uint32 subgroupScale = nativeSubgroupSize / capturedSubgroupSize;
|
||||
if ((subgroupScale & (subgroupScale - 1u)) != 0u) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Vector<CodeToken> tokens = TokenizeCode(source);
|
||||
LinearPrefixScanMatch match;
|
||||
if (!ParseLinearPrefixScanTemplate(tokens, match)) {
|
||||
// Diagnosability: when the trigger op is present but the template no longer
|
||||
// matches (e.g. the pack shipped a new shader revision), the affected device
|
||||
// silently falls back to the driver's miscompiled path. Make that visible.
|
||||
if (CountToken(tokens, "subgroupInclusiveAdd") > 0) {
|
||||
MGLOG_W("%s: subgroupInclusiveAdd present but the linear prefix-scan template "
|
||||
"did not match; the wide-subgroup rewrite was NOT applied",
|
||||
__func__);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
const String replacement = BuildLinearPrefixScanReplacement(match);
|
||||
source.replace(match.scanBegin, match.scanEnd - match.scanBegin, replacement);
|
||||
// The declaration occurs before the replaced scan, so its original offsets remain
|
||||
// valid after the first replacement.
|
||||
source.replace(match.sharedArraySizeBegin, match.sharedArraySizeEnd - match.sharedArraySizeBegin,
|
||||
"1024");
|
||||
return true;
|
||||
}
|
||||
|
||||
namespace {
|
||||
struct ShaderSourceQuirkContext {
|
||||
ShaderStage stage = ShaderStage::Unknown;
|
||||
BackendType backend = BackendType::Unknown;
|
||||
MG_Backend::GpuVendorKind vendor = MG_Backend::GpuVendorKind::Unknown;
|
||||
Uint32 subgroupSize = 0;
|
||||
};
|
||||
|
||||
// Device-quirk registry. Every entry is a narrowly scoped source rewrite that
|
||||
// works around a specific driver defect. A quirk runs when its env override
|
||||
// forces it on, or when the override is Auto and DeviceApplies matches the
|
||||
// detected device. ForceOn bypasses only the device gate - each Apply keeps
|
||||
// its own structural safety checks. Add new per-device workarounds here
|
||||
// instead of open-coding them in PreprocessShaderSource.
|
||||
struct ShaderSourceQuirk {
|
||||
const char* name;
|
||||
MG_Config::QuirkOverride (*GetOverride)();
|
||||
Bool (*DeviceApplies)(const ShaderSourceQuirkContext&);
|
||||
Bool (*Apply)(const ShaderSourceQuirkContext&, String&);
|
||||
};
|
||||
|
||||
constexpr ShaderSourceQuirk kShaderSourceQuirks[] = {
|
||||
{
|
||||
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN
|
||||
"subgroup-prefix-scan-rewrite",
|
||||
[] { return MG_Config::Features.SubgroupPrefixScanQuirk; },
|
||||
[](const ShaderSourceQuirkContext& ctx) {
|
||||
// Qualcomm's Vulkan driver miscompiles the recognized float
|
||||
// InclusiveScan pattern for native subgroups wider than the
|
||||
// captured 32 lanes; other vendors compile it correctly and
|
||||
// should keep their native scan.
|
||||
return ctx.backend == BackendType::DirectVulkan &&
|
||||
ctx.vendor == MG_Backend::GpuVendorKind::Qualcomm;
|
||||
},
|
||||
[](const ShaderSourceQuirkContext& ctx, String& source) {
|
||||
return RewriteLinearSubgroupPrefixScanForVulkan(ctx.stage, ctx.subgroupSize,
|
||||
source);
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
void ApplyShaderSourceQuirks(ShaderStage stage, String& source) {
|
||||
const auto& activeBackend = MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackend) {
|
||||
return;
|
||||
}
|
||||
const auto& dynamicParameters = activeBackend->GetDynamicParameters();
|
||||
const ShaderSourceQuirkContext quirkContext{
|
||||
stage,
|
||||
activeBackend->GetBackendType(),
|
||||
dynamicParameters.GpuVendor,
|
||||
dynamicParameters.SubgroupSize,
|
||||
};
|
||||
for (const ShaderSourceQuirk& quirk : kShaderSourceQuirks) {
|
||||
const MG_Config::QuirkOverride quirkOverride = quirk.GetOverride();
|
||||
if (quirkOverride == MG_Config::QuirkOverride::ForceOff) {
|
||||
continue;
|
||||
}
|
||||
if (quirkOverride == MG_Config::QuirkOverride::Auto &&
|
||||
!quirk.DeviceApplies(quirkContext)) {
|
||||
continue;
|
||||
}
|
||||
if (quirk.Apply(quirkContext, source)) {
|
||||
MGLOG_I("ApplyShaderSourceQuirks: applied '%s'%s", quirk.name,
|
||||
quirkOverride == MG_Config::QuirkOverride::ForceOn ? " (forced on)" : "");
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void PreprocessShaderSource(ShaderStage stage, String& source) {
|
||||
// Normalize while the inspector's source span still refers to the untouched input. Later passes
|
||||
// remove comments and directives, so any subsequent insertion re-inspects the current source.
|
||||
const ShaderLanguageInfo originalLanguage = InspectShaderLanguage(source);
|
||||
NormalizeVersionDirective(source, originalLanguage);
|
||||
|
||||
// remove multi-line comment
|
||||
size_t commentStartPos = source.find("/*");
|
||||
while (commentStartPos != String::npos) {
|
||||
size_t commentEndPos = source.find("*/", commentStartPos);
|
||||
if (commentEndPos == String::npos) {
|
||||
source.erase(commentStartPos);
|
||||
break;
|
||||
}
|
||||
// + length of "*/"
|
||||
source = source.replace(commentStartPos, commentEndPos - commentStartPos + 2, "");
|
||||
commentStartPos = source.find("/*", commentStartPos);
|
||||
}
|
||||
// Comments are left intact for glslang's own preprocessor: a block comment is a single
|
||||
// preprocessing token that collapses to one space even across newlines and inside a
|
||||
// directive, so blanking it here (which preserved the interior newlines) truncated
|
||||
// multi-line #define bodies and broke otherwise-valid shaders (KHR-GL3x.shaders.
|
||||
// preprocessor multiline_comment_define / redefine_object / function_redefinition).
|
||||
// Every MobileGL pass that must ignore comment/string text already masks them locally
|
||||
// via MaskCommentsAndQuotedText/TokenizeCode, so the source we hand glslang keeps them.
|
||||
NormalizeLineDirectives(source);
|
||||
|
||||
// remove #line directives
|
||||
SizeT linedirPos = source.find("#line");
|
||||
while (linedirPos != String::npos) {
|
||||
SizeT newlinePos = source.find('\n', linedirPos);
|
||||
if (newlinePos == String::npos) {
|
||||
source.erase(linedirPos);
|
||||
break;
|
||||
}
|
||||
|
||||
// Preserve a line break so adjacent preprocessor directives do not merge.
|
||||
source = source.replace(linedirPos, newlinePos - linedirPos + 1, "\n");
|
||||
linedirPos = source.find("#line", linedirPos);
|
||||
}
|
||||
|
||||
// remove "noperspective"
|
||||
const char* str_np = "noperspective";
|
||||
const SizeT len_np = strlen(str_np);
|
||||
SizeT noperspectivePos = source.find(str_np);
|
||||
while (noperspectivePos != String::npos) {
|
||||
// + length of "\n"
|
||||
source = source.replace(noperspectivePos, len_np, "");
|
||||
noperspectivePos = source.find(str_np);
|
||||
}
|
||||
// noperspective is intentionally NOT touched here. It is core in desktop GLSL (1.30+)
|
||||
// and maps to the core SPIR-V NoPerspective decoration, which DirectVulkan renders
|
||||
// natively and SPIRV-Cross turns into ESSL `noperspective` + the
|
||||
// GL_NV_shader_noperspective_interpolation extension. The old naked substring erase
|
||||
// both discarded that interpolation (shader packs need it) and corrupted any
|
||||
// identifier that merely contained the word. The GLES fallback for devices without
|
||||
// the extension lives in the backend, where device capabilities are known.
|
||||
|
||||
FilterUnsupportedGpuShaderInt64(source);
|
||||
CoerceUniformBlockPackingToStd140(source);
|
||||
@@ -631,6 +1289,8 @@ namespace MobileGL {
|
||||
RenameBuiltinShadowingFunction(source, "max3", "mg_max3");
|
||||
ModernizeLegacyGLSL(stage, source);
|
||||
InjectDepthRangeBuiltinShim(stage, source);
|
||||
|
||||
ApplyShaderSourceQuirks(stage, source);
|
||||
}
|
||||
|
||||
Bool RetargetLegacyVersionDirectiveTo460(String& source) {
|
||||
@@ -639,6 +1299,10 @@ namespace MobileGL {
|
||||
// must not be mistaken for the real one.
|
||||
const ShaderLanguageInfo info = InspectShaderLanguage(source);
|
||||
if (!info.HasVersionDirective()) return false;
|
||||
// Never rescue a malformed directive to 460: that is precisely what re-legalized the
|
||||
// CTS directive.version_* rejection cases after the first compile failed. The shader-
|
||||
// pack retry this exists for only ever sees a valid low version (a real "#version 330").
|
||||
if (!info.hasValidVersionDirective) return false;
|
||||
// Only the set NormalizeVersionDirective downgraded: desktop core below 400. ES and
|
||||
// compatibility shaders keep whatever they declared.
|
||||
if (info.profile != ShaderProfile::Core || info.version >= 400) return false;
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user