mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-09 20:58:31 +09:00
Compare commits
455
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8c458cd594 | ||
|
|
c2c6a655ea | ||
|
|
9f60aadc1d | ||
|
|
a690032f85 | ||
|
|
173f1dd273 | ||
|
|
a38bdab4e2 | ||
|
|
6035c9d7f8 | ||
|
|
419f9941b5 | ||
|
|
72b4c91a45 | ||
|
|
ccb7b6b331 | ||
|
|
e4579ee80e | ||
|
|
0d01405cf8 | ||
|
|
dcc31e95ca | ||
|
|
4678519f99 | ||
|
|
4614abb9c6 | ||
|
|
7107d0f47c | ||
|
|
972dd811d7 | ||
|
|
f9c0e7ec43 | ||
|
|
b938c83b6f | ||
|
|
e62abff6b6 | ||
|
|
9a0a7f7608 | ||
|
|
993ce0fb76 | ||
|
|
ea38cfcb99 | ||
|
|
9c3e8ffd32 | ||
|
|
2f35014d73 | ||
|
|
8c8d460e14 | ||
|
|
b6e5c2bb39 | ||
|
|
246434bb35 | ||
|
|
41b7905000 | ||
|
|
2e03b82a67 | ||
|
|
2bebbbdcb7 | ||
|
|
500a8e462a | ||
|
|
be939e2cfe | ||
|
|
68d09bb585 | ||
|
|
5e7fb18c4a | ||
|
|
74d2daea1b | ||
|
|
7afa2f18d3 | ||
|
|
64332060fd | ||
|
|
96dca5a8ba | ||
|
|
e24e30bbec | ||
|
|
929d162594 | ||
|
|
0b12f89f3b | ||
|
|
31e6a7d6d8 | ||
|
|
29bd4d2c94 | ||
|
|
588b2277c5 | ||
|
|
1e8cdc5931 | ||
|
|
a632e26e47 | ||
|
|
ee31944bce | ||
|
|
245daca06a | ||
|
|
29d51ab98b | ||
|
|
17396216e0 | ||
|
|
a666201708 | ||
|
|
71a174a531 | ||
|
|
83a209bbd4 | ||
|
|
832620ddeb | ||
|
|
25e1a8483b | ||
|
|
2bd8f3c82d | ||
|
|
e021b4f62b | ||
|
|
a3bd7f0d7d | ||
|
|
5046b3fc89 | ||
|
|
5e5dc36e57 | ||
|
|
2bd19d5ee2 | ||
|
|
33f5e878e7 | ||
|
|
568d587ae7 | ||
|
|
602326f379 | ||
|
|
0338812cf3 | ||
|
|
09e0a83cee | ||
|
|
771e8e06a1 | ||
|
|
2d2090cf1c | ||
|
|
a11a5eb2af | ||
|
|
612dbac44f | ||
|
|
92dffb81b4 | ||
|
|
79b581984b | ||
|
|
d6e52f75c3 | ||
|
|
ac09e5b37a | ||
|
|
820d1a60b9 | ||
|
|
f0f8cab682 | ||
|
|
b2edca2728 | ||
|
|
ccde29064a | ||
|
|
bc2aad2176 | ||
|
|
5a3e9f0857 | ||
|
|
1534cf3784 | ||
|
|
712c946744 | ||
|
|
8f1eaafa79 | ||
|
|
08922b6f78 | ||
|
|
711d5c61ba | ||
|
|
5ab90dec24 | ||
|
|
4299acd9c1 | ||
|
|
953d73f111 | ||
|
|
d01439d846 | ||
|
|
c9958075e8 | ||
|
|
ae1a1c503f | ||
|
|
a02f1571f5 | ||
|
|
dca3eb868e | ||
|
|
7408bad9fd | ||
|
|
e8502a6100 | ||
|
|
9ea44389e7 | ||
|
|
17db759891 | ||
|
|
2cb44039b5 | ||
|
|
32033d6993 | ||
|
|
08192d7266 | ||
|
|
37da3c3a07 | ||
|
|
fde5fda3b5 | ||
|
|
6515c8e6ae | ||
|
|
d54ec57a5d | ||
|
|
c20e2f2b67 | ||
|
|
680ea63360 | ||
|
|
959ca30810 | ||
|
|
0bff6875b3 | ||
|
|
707bced438 | ||
|
|
83c8101a9b | ||
|
|
2a5e0195b8 | ||
|
|
afda613bd9 | ||
|
|
433f51a065 | ||
|
|
9ba5d7ba1e | ||
|
|
f0beefa85c | ||
|
|
78ff014547 | ||
|
|
3e298c9ad1 | ||
|
|
fd6b5bdbf2 | ||
|
|
8d3cd67b94 | ||
|
|
f6c7dcd1b6 | ||
|
|
4ad88214cd | ||
|
|
4412cef00c | ||
|
|
6e9943b36e | ||
|
|
0cb46fe2bd | ||
|
|
83302ca247 | ||
|
|
3c55e02797 | ||
|
|
75ea7ee2df | ||
|
|
13b380feda | ||
|
|
ed83424c75 | ||
|
|
df055eab13 | ||
|
|
511b3752c0 | ||
|
|
b28058f37c | ||
|
|
14137bc9a6 | ||
|
|
39933613ae | ||
|
|
816373ffd9 | ||
|
|
7047331a41 | ||
|
|
7ef8cb93fc | ||
|
|
b97a228cea | ||
|
|
3840cf734b | ||
|
|
68db6db7f8 | ||
|
|
cd05de504e | ||
|
|
ce24e2a734 | ||
|
|
4e44650199 | ||
|
|
9951961d9c | ||
|
|
31e370bed5 | ||
|
|
0c55560510 | ||
|
|
f11e78b0a4 | ||
|
|
d9bde13127 | ||
|
|
cc427ec4de | ||
|
|
b11bb9650a | ||
|
|
14efd6eb24 | ||
|
|
874d1ee77d | ||
|
|
b9eaa47480 | ||
|
|
42d43af25b | ||
|
|
12e6bfcf14 | ||
|
|
e6452ce948 | ||
|
|
56366331dc | ||
|
|
355c60b901 | ||
|
|
6eb0e675ad | ||
|
|
45c8f1a8be | ||
|
|
5cb826b01e | ||
|
|
e01c0ccc53 | ||
|
|
c036900d72 | ||
|
|
44c2b5cf3a | ||
|
|
738b289df8 | ||
|
|
a9778eaabe | ||
|
|
c73ae7d443 | ||
|
|
55d2af9bd1 | ||
|
|
2d690754dd | ||
|
|
7a2e256133 | ||
|
|
bb2a236d5f | ||
|
|
13d7e32b7b | ||
|
|
b1c37699b1 | ||
|
|
e5603f9a46 | ||
|
|
08d14d85ef | ||
|
|
af20dba6db | ||
|
|
a5d1136c02 | ||
|
|
d704401a56 | ||
|
|
ce9f44a24c | ||
|
|
1a012f2820 | ||
|
|
b9c137e146 | ||
|
|
e9499d38bd | ||
|
|
f1780b9000 | ||
|
|
e5c032c89e | ||
|
|
a174a06c79 | ||
|
|
46841ac706 | ||
|
|
01179c54d2 | ||
|
|
3594f03c4e | ||
|
|
43f8b47088 | ||
|
|
c74c4819fb | ||
|
|
96c544514e | ||
|
|
59191cd296 | ||
|
|
8d0ed5b82c | ||
|
|
f15b0fdf4b | ||
|
|
8f66c374aa | ||
|
|
067b186677 | ||
|
|
5d4d91fe7e | ||
|
|
bb781df527 | ||
|
|
c574043c13 | ||
|
|
3ab394e2b8 | ||
|
|
43bf97cc87 | ||
|
|
3302ee82b5 | ||
|
|
7dec32a574 | ||
|
|
dcfa5ad311 | ||
|
|
aa64c91052 | ||
|
|
fcd4ad3799 | ||
|
|
de532f55a9 | ||
|
|
3d1a866e82 | ||
|
|
df784c6752 | ||
|
|
c9dd173201 | ||
|
|
f5bd1a0412 | ||
|
|
6cb7d1b83b | ||
|
|
7c97fcfee3 | ||
|
|
caa0a7221b | ||
|
|
eb81705130 | ||
|
|
e10f5d6750 | ||
|
|
5a3c0616b7 | ||
|
|
9a8369296e | ||
|
|
149e26a79a | ||
|
|
3160c4b85b | ||
|
|
bd2092f4eb | ||
|
|
f4dbea2300 | ||
|
|
9eae98581f | ||
|
|
d7655247f7 | ||
|
|
842af23331 | ||
|
|
d1a7c5f159 | ||
|
|
ce370a3e84 | ||
|
|
bee07c3273 | ||
|
|
7c2c1456f8 | ||
|
|
ad1238bd6f | ||
|
|
a9bb99a46a | ||
|
|
02b970e9c1 | ||
|
|
eec92cd221 | ||
|
|
810850b13a | ||
|
|
9c6a8a25d8 | ||
|
|
4826806881 | ||
|
|
e7a6a72f6a | ||
|
|
62a7786184 | ||
|
|
ef6227e19b | ||
|
|
9bd6d39403 | ||
|
|
6b681c4a63 | ||
|
|
80a6b39003 | ||
|
|
97b997d5da | ||
|
|
7d80c9678e | ||
|
|
72aa9191b4 | ||
|
|
1e3a74686f | ||
|
|
0fe7bf82d2 | ||
|
|
416cd23c28 | ||
|
|
5f8e8db1b9 | ||
|
|
bdf05514c3 | ||
|
|
bf8b39a867 | ||
|
|
d4504e30f8 | ||
|
|
9087f13308 | ||
|
|
44ffafb2dc | ||
|
|
12b57055b7 | ||
|
|
d0ff647581 | ||
|
|
30d72c5b4e | ||
|
|
d9f4698d98 | ||
|
|
878db2c405 | ||
|
|
77ecde1524 | ||
|
|
510ecd9293 | ||
|
|
440d3c5253 | ||
|
|
83b16561c2 | ||
|
|
275dd3edb4 | ||
|
|
a196ada4c1 | ||
|
|
3aa4d8af1f | ||
|
|
bf86b1ede6 | ||
|
|
087685d19b | ||
|
|
5635e33ffe | ||
|
|
5d99ee435f | ||
|
|
b566bf4db9 | ||
|
|
09d2bb11b7 | ||
|
|
fee3902472 | ||
|
|
da249f30e2 | ||
|
|
8566a288f8 | ||
|
|
2318f6ae44 | ||
|
|
fe3dc1dde8 | ||
|
|
6672778b80 | ||
|
|
6e0e3df372 | ||
|
|
bee22f9d26 | ||
|
|
8952b14024 | ||
|
|
458ccde176 | ||
|
|
901d48a678 | ||
|
|
e8ee7b1a88 | ||
|
|
1154f9a00d | ||
|
|
7ef7c7e543 | ||
|
|
6c7ad0a1bf | ||
|
|
38d4c2372c | ||
|
|
8a239177ac | ||
|
|
87ee17c68c | ||
|
|
aa005720d0 | ||
|
|
c1a7ffac94 | ||
|
|
bdd4bed431 | ||
|
|
10315e71f3 | ||
|
|
bfa087d0f7 | ||
|
|
a1e22c26ab | ||
|
|
bd2b4158e0 | ||
|
|
9c7339b214 | ||
|
|
50815a232e | ||
|
|
d380a01f32 | ||
|
|
7566a0b002 | ||
|
|
42e0f47ebb | ||
|
|
9bbf71990c | ||
|
|
2f8d0f0d51 | ||
|
|
3363258908 | ||
|
|
9c773182bb | ||
|
|
8349babe90 | ||
|
|
1794ac94b1 | ||
|
|
8b31de2f8d | ||
|
|
50fb13430f | ||
|
|
d4f8adcf6d | ||
|
|
795e08f7e6 | ||
|
|
1e7ecab4db | ||
|
|
81b17c0b75 | ||
|
|
d1edf765f5 | ||
|
|
97e07190ac | ||
|
|
a4dcdf989e | ||
|
|
1c113e4b26 | ||
|
|
bf9cfb3079 | ||
|
|
e1818d497a | ||
|
|
d7f66722d1 | ||
|
|
92dc41ebf9 | ||
|
|
feea131d8b | ||
|
|
19f4402fbf | ||
|
|
1350031368 | ||
|
|
0ee3384b22 | ||
|
|
ba3f8d6774 | ||
|
|
3327784fd0 | ||
|
|
ff426da3a9 | ||
|
|
08419a1fe6 | ||
|
|
5d51372c44 | ||
|
|
7fd4550968 | ||
|
|
971537058e | ||
|
|
dd98c450ad | ||
|
|
5dbbbbd7eb | ||
|
|
5d140a41ce | ||
|
|
bf376b230f | ||
|
|
29599dcf90 | ||
|
|
734fab9f90 | ||
|
|
2cd1809c29 | ||
|
|
faed498476 | ||
|
|
8282eecbfa | ||
|
|
f4f3afb0b6 | ||
|
|
a28da07641 | ||
|
|
200c21336f | ||
|
|
2c3fc583d5 | ||
|
|
645a12d8bc | ||
|
|
8e6acc5528 | ||
|
|
525ffe0f14 | ||
|
|
ad28d2b744 | ||
|
|
eab622388f | ||
|
|
75e573c923 | ||
|
|
0d0ef13619 | ||
|
|
23565fcacd | ||
|
|
5dc26e3e2c | ||
|
|
90564aa82e | ||
|
|
7520607d47 | ||
|
|
57635a9198 | ||
|
|
c19d0f0b75 | ||
|
|
e42e7d00f5 | ||
|
|
62695ee3c2 | ||
|
|
02cc0ce83c | ||
|
|
9e52a0b23e | ||
|
|
9dee53337f | ||
|
|
28c5badf8f | ||
|
|
01116f7b41 | ||
|
|
05d627ba2d | ||
|
|
ea5d52f126 | ||
|
|
3c70b4fc0f | ||
|
|
b6d6316333 | ||
|
|
3c9ab5a68f | ||
|
|
532b5e9cc5 | ||
|
|
06605ed0ea | ||
|
|
e1d5bdc4a5 | ||
|
|
66867a41ba | ||
|
|
ebff4b21f7 | ||
|
|
d4e7378868 | ||
|
|
01d20e5c96 | ||
|
|
b04c67d9a8 | ||
|
|
685d83e3ec | ||
|
|
009b140691 | ||
|
|
1f44e5bc1d | ||
|
|
d9aebcba26 | ||
|
|
0db666897e | ||
|
|
5f445e499f | ||
|
|
c52ebd5bf6 | ||
|
|
8e072bc793 | ||
|
|
88ee75be0e | ||
|
|
0dbb4ceba8 | ||
|
|
a94b3e0bd5 | ||
|
|
764b6e044d | ||
|
|
c1d89de729 | ||
|
|
c136384f97 | ||
|
|
1920a3d16f | ||
|
|
11f4b4bd3b | ||
|
|
9ef33f4274 | ||
|
|
e430e1b3be | ||
|
|
e315d9e798 | ||
|
|
6f299372c6 | ||
|
|
be7bf21eb8 | ||
|
|
0e4302b399 | ||
|
|
c9c2dcb42a | ||
|
|
2d938971b9 | ||
|
|
f7e23d5d83 | ||
|
|
02fbb816e9 | ||
|
|
0e0882cfc6 | ||
|
|
de09646d5e | ||
|
|
747864777e | ||
|
|
6fc3504bd9 | ||
|
|
df1bcdba09 | ||
|
|
843c61dee1 | ||
|
|
c0a3f4cc50 | ||
|
|
06744fde7f | ||
|
|
6cc9faf772 | ||
|
|
7168f2ef77 | ||
|
|
07669aacd4 | ||
|
|
d52a3b2196 | ||
|
|
9e23016dd4 | ||
|
|
f1354dc25e | ||
|
|
b7a694711a | ||
|
|
2ae848ca19 | ||
|
|
acf86d1fb6 | ||
|
|
07a0408a28 | ||
|
|
8cf2e2aea9 | ||
|
|
31252cf0da | ||
|
|
2635fe84b6 | ||
|
|
e3163233a5 | ||
|
|
eb9e4fdac1 | ||
|
|
90b7a689c5 | ||
|
|
e69e939d1a | ||
|
|
b675e2a0b0 | ||
|
|
6979926a6f | ||
|
|
d5286e69b6 | ||
|
|
0f2fcbc469 | ||
|
|
18fccdd796 | ||
|
|
0d2fceab0e | ||
|
|
82decded58 | ||
|
|
d04a3394de | ||
|
|
33eabfc2ff | ||
|
|
56e5d13dc9 | ||
|
|
34b7cc772f | ||
|
|
c1d6a3c908 | ||
|
|
386bd7e461 | ||
|
|
12df061e0b | ||
|
|
d83a48da5c | ||
|
|
b3100b0de5 | ||
|
|
1cde801a01 | ||
|
|
52c050131e | ||
|
|
ebb8a4cebf | ||
|
|
5398fb4289 | ||
|
|
7f2ca68615 | ||
|
|
b12ef4d717 | ||
|
|
724755d9df | ||
|
|
59f7059bf4 |
@@ -44,12 +44,12 @@ require 'key:MOBILEGL_BACKEND_TYPE' "$plugin_resource_text" 'V2 backend variable
|
||||
require 'defaultValue:DirectGLES' "$plugin_resource_text" 'V2 DirectGLES default'
|
||||
require 'DirectVulkan' "$plugin_resource_text" 'V2 DirectVulkan option'
|
||||
require 'key:MOBILEGL_DISABLE_TIMERQUERY' "$plugin_resource_text" 'V2 timer-query toggle'
|
||||
require 'key:MOBILEGL_DISABLE_SUBGROUP' "$plugin_resource_text" 'V2 Vulkan subgroup toggle'
|
||||
require 'key:MOBILEGL_MAGMA_DISABLE_SUBGROUP' "$plugin_resource_text" 'V2 Vulkan subgroup toggle'
|
||||
require 'key:MOBILEGL_MAGMA_R11G11B10F_FALLBACK' "$plugin_resource_text" 'V2 Magma format fallback toggle'
|
||||
require 'key:MOBILEGL_MAGMA_FRAMESINFLIGHT' "$plugin_resource_text" 'V2 Magma frames-in-flight setting'
|
||||
require 'key:MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER' "$plugin_resource_text" 'V2 sampler workaround toggle'
|
||||
require 'key:MOBILEGL_ESPRYT_AVOID_SAMPLER_MIPMAP_MIN_FILTER' "$plugin_resource_text" 'V2 sampler workaround toggle'
|
||||
require 'key:MOBILEGL_COHERENT_AS_FLUSH' "$plugin_resource_text" 'V2 coherent-as-flush toggle'
|
||||
require 'key:MOBILEGL_USE_ANGLE' "$plugin_resource_text" 'V2 ANGLE toggle'
|
||||
require 'key:MOBILEGL_ESPRYT_USE_ANGLE' "$plugin_resource_text" 'V2 ANGLE toggle'
|
||||
|
||||
if [[ $(grep -Fc 'fclPlugin_V2' <<<"$plugin_manifest") -ne 1 ]]; then
|
||||
echo '::error::Plugin manifest must expose exactly one V2 descriptor' >&2
|
||||
|
||||
@@ -6,6 +6,10 @@ on:
|
||||
- dev
|
||||
- Feat/Backend-Direct-GLES
|
||||
- Feat/Backend-Direct-Vulkan
|
||||
# TEMPORARY, remove before merging the MGPipe work into dev: the disaggregation
|
||||
# branch runs the full lane on every push so a phase's landing is not gated on
|
||||
# someone remembering to dispatch the workflow by hand.
|
||||
- feat/disaggregated
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
@@ -417,12 +421,12 @@ jobs:
|
||||
|
||||
- name: Retrace and validate
|
||||
env:
|
||||
MOBILEGL_USE_ANGLE: ${{ matrix.backend.name == 'DirectGLES' && '1' || '0' }}
|
||||
MOBILEGL_ESPRYT_USE_ANGLE: ${{ matrix.backend.name == 'DirectGLES' && '1' || '0' }}
|
||||
MOBILEGL_TRACE_ANGLE_VARIANT: ${{ matrix.case.name == 'minecraft-1.21.4-fabric-iris-bliss-in-world' && '90a62123d794' || 'ec889e6ea831' }}
|
||||
MOBILEGL_MAGMA_R11G11B10F_FALLBACK: ${{ matrix.backend.name == 'DirectVulkan' && '1' || '0' }}
|
||||
MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_DERIVE_NUM_SUBGROUPS: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_ITERATIONRP_FIX_BARRIER: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_MAGMA_FIX_ITERATIONRP_SUBGROUP_SCRATCH: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_MAGMA_DERIVE_NUM_SUBGROUPS: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_MAGMA_ITERATIONRP_FIX_BARRIER: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
run: |
|
||||
apk_file="android-retrace-apks/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk"
|
||||
test -f "${apk_file}"
|
||||
|
||||
+984
-9
File diff suppressed because it is too large
Load Diff
@@ -34,3 +34,6 @@
|
||||
[submodule "include/ska"]
|
||||
path = include/ska
|
||||
url = https://github.com/MobileGL-Dev/flat_hash_map.git
|
||||
[submodule "3rdparty/flatbuffers"]
|
||||
path = 3rdparty/flatbuffers
|
||||
url = https://github.com/google/flatbuffers.git
|
||||
|
||||
+1
Submodule 3rdparty/flatbuffers added at 7e163021e5
Vendored
+1
-1
Submodule 3rdparty/glslang updated: fa562bb911...d89cf443bc
+170
@@ -14,6 +14,26 @@ option(MOBILEGL_ENABLE_TRACY "Enable tracy for profiling"
|
||||
option(MOBILEGL_BUILD_TRACE_REPLAY "Build desktop apitrace replay runner" OFF)
|
||||
option(MOBILEGL_TRACE_ANGLE_VARIANTS "Enable signed trace-APK ANGLE variant loading" OFF)
|
||||
option(MOBILEGL_IOS "Build MobileGL for iOS instead of macOS when APPLE is set" OFF)
|
||||
# The disaggregated (two-process) shape. OFF is the shipping default and OFF
|
||||
# must stay byte-comparable to a tree without MG_Remote at all: nothing under
|
||||
# MobileGL/MG_Remote/ is compiled, no include path is added, and no library is
|
||||
# linked, so `nm --defined-only libMobileGL.so | grep -i MG_Remote` is empty.
|
||||
# That emptiness is one of the two byte-level equalities the plan's validation
|
||||
# gates keep (section 10.3).
|
||||
option(MOBILEGL_BUILD_DISAGGREGATED "Build the MG_Remote transport layer (two-process shape)" OFF)
|
||||
option(MOBILEGL_BUILD_SERVER_SPIKE "Build the P0 spike-A MobileGLServer delivery-chain executable (Android only)" OFF)
|
||||
# The PipeInputs strangler (ARCHITECTURE.md 9.2). OFF is the pull build and must stay
|
||||
# byte-identical to a tree without either option: MGB_CTX is the live GLContext, no
|
||||
# MGPipe/PipeInputs source is compiled, every MGP_FILL is ((void)0).
|
||||
option(MOBILEGL_PIPE_PUSH "Backends read frontend state through the MGPipe PipeInputs block instead of MG_State::pGLContext (ARCHITECTURE.md 9.2 phase A)" OFF)
|
||||
option(MOBILEGL_PIPE_VERIFY "Compile SnapshotFromGLContext() and the G4 per-verb shadow comparator; implies MOBILEGL_PIPE_PUSH; never shipped" OFF)
|
||||
# Track H's old-versus-new arm (ARCHITECTURE.md 9.6). With a MOBILEGL_PIPE_PUSH bit clear
|
||||
# the backend would still run the RE-KEYED memo code, so the bitmask alone stops being a
|
||||
# valid A/B the moment a handle wave lands: this option compiles the pre-handle arm - the
|
||||
# registries, OwnerEquals, the TwinLookupMemos, g_fbSlotCache, ComputePipelineStateHash,
|
||||
# the address-keyed VaoDrawMemo - beside it, behind the same PipeInputs interface. ON for
|
||||
# the whole migration window; it retires with the pull path itself at P13.
|
||||
option(MOBILEGL_PIPE_LEGACY_MEMOS "Compile the pre-handle memo arm beside the {slot, gen} arm so Track H has a real A/B (ARCHITECTURE.md 9.6)" ON)
|
||||
set(MOBILEGL_LOG_ACTIVE_LEVEL "MOBILEGL_LOG_LEVEL_INFO" CACHE STRING "MobileGL active log level macro")
|
||||
set(MOBILEGL_VULKAN_LIBRARY "" CACHE FILEPATH "Vulkan loader/MoltenVK library to link for iOS builds")
|
||||
|
||||
@@ -238,6 +258,8 @@ set(SOURCE_FILES
|
||||
|
||||
MobileGL/MG_Util/Metrics/BufferMetrics.cpp
|
||||
|
||||
MobileGL/MG_Util/Metrics/PipeStats.cpp
|
||||
|
||||
MobileGL/MG_Util/Converters/GLToStr/GLEnumConverter.cpp
|
||||
MobileGL/MG_Util/Converters/EGLToStr/EGLEnumConverter.cpp
|
||||
MobileGL/MG_Util/Converters/MGToStr/DataTypeConverter.cpp
|
||||
@@ -285,6 +307,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PackDoubleVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenXfbInterfaceBlocksPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/UniquifyIoBlockNamesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripIoBlockLocationsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/SplitArrayVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
|
||||
@@ -306,13 +329,16 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeFragmentOutputIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeResourceArrayIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenAtomicCounterBlockPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DemotePointSizePass.cpp
|
||||
|
||||
MobileGL/MG_Util/BackendLoaders/OpenGL/Loader.cpp
|
||||
MobileGL/MG_Util/BackendLoaders/Vulkan/Loader.cpp
|
||||
|
||||
MobileGL/MG_Util/SelfTest/DriverBugProbes.cpp
|
||||
MobileGL/MG_Util/SelfTest/PersistentBufferOrderingProbe.cpp
|
||||
MobileGL/MG_Util/SelfTest/DriverPost.cpp
|
||||
MobileGL/MG_Util/SelfTest/DriverPostIterationRPWitness.cpp
|
||||
MobileGL/MG_Util/SelfTest/PrimitivesGeneratedNoXfbProbe.cpp
|
||||
|
||||
MobileGL/MG_Util/Texture/PixelStoreProcessor.cpp
|
||||
MobileGL/MG_Util/Texture/TextureFormatProcessor.cpp
|
||||
@@ -414,6 +440,85 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_State/GLState/RenderbufferState/RenderbufferState.cpp
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# MG_Remote (disaggregated transport). Everything below is gated: with the
|
||||
# option OFF not one file here is compiled and no include path is added.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# FlatBuffers is a submodule and its runtime is header-only. Guard both ways:
|
||||
# a checkout without the submodule must configure and build, just without the
|
||||
# disaggregated shape, rather than fail with a missing-header error a hundred
|
||||
# lines later. Note this only checks for the RUNTIME headers - flatc is never
|
||||
# built here (see scripts/gen_protocol.py).
|
||||
if (MOBILEGL_BUILD_DISAGGREGATED AND
|
||||
NOT EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/3rdparty/flatbuffers/include/flatbuffers/flatbuffers.h")
|
||||
message(WARNING
|
||||
"MOBILEGL_BUILD_DISAGGREGATED=ON but 3rdparty/flatbuffers/include is missing. "
|
||||
"Run `git submodule update --init 3rdparty/flatbuffers`. Building without the "
|
||||
"disaggregated shape for this configure; the cached ON takes effect once the "
|
||||
"submodule is present.")
|
||||
# A NORMAL variable, deliberately not `CACHE BOOL ... FORCE`: forcing OFF into the cache
|
||||
# made the plain re-configure after `git submodule update` stay OFF with no message at
|
||||
# all. Shadowing the cache entry for this configure only keeps the operator's ON where it
|
||||
# was, so the next configure - with the submodule there - honours it.
|
||||
set(MOBILEGL_BUILD_DISAGGREGATED OFF)
|
||||
endif()
|
||||
|
||||
# MOBILEGL_PIPE_VERIFY implies MOBILEGL_PIPE_PUSH: the comparator compares the pushed block
|
||||
# against a snapshot, so there has to be a pushed block. A normal variable, not a forced
|
||||
# cache write, for the same reason as the disaggregated fallback above.
|
||||
if (MOBILEGL_PIPE_VERIFY AND NOT MOBILEGL_PIPE_PUSH)
|
||||
message(STATUS "MobileGL: MOBILEGL_PIPE_VERIFY=ON forces MOBILEGL_PIPE_PUSH ON for this configure")
|
||||
set(MOBILEGL_PIPE_PUSH ON)
|
||||
endif()
|
||||
|
||||
# In a pull build the legacy arm is the ONLY arm, so the option cannot be off there.
|
||||
# A normal variable, not a forced cache write, for the same reason as the two above.
|
||||
if (NOT MOBILEGL_PIPE_PUSH AND NOT MOBILEGL_PIPE_LEGACY_MEMOS)
|
||||
message(STATUS "MobileGL: MOBILEGL_PIPE_PUSH=OFF forces MOBILEGL_PIPE_LEGACY_MEMOS ON for this "
|
||||
"configure: with nothing pushed it is the only arm there is")
|
||||
set(MOBILEGL_PIPE_LEGACY_MEMOS ON)
|
||||
endif()
|
||||
|
||||
if (MOBILEGL_PIPE_PUSH)
|
||||
message(STATUS "MobileGL: PipeInputs push ON, appending the MGPipe fill sources")
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Backend/MGPipe/PipeInputs.cpp
|
||||
MobileGL/MG_Impl/Pipe/PipeFill.cpp
|
||||
# P2's contract: the chunk table and its subset hash, the in-process applier, and
|
||||
# the client's {slot, gen} allocator. All three are push-only, which is how the
|
||||
# pull build gains no symbol from P2 (G1) - a declaration emits nothing.
|
||||
MobileGL/MG_Pipe/MGPipeRenderStateSpans.cpp
|
||||
MobileGL/MG_Pipe/PipeApply.cpp
|
||||
MobileGL/MG_Impl/Pipe/SlotAllocator.cpp
|
||||
# P4a's contract: the reflection-archive serializer over ProgramArtifacts.h's
|
||||
# VisitFields tables. Push-only for the same G1 reason as the three above - in
|
||||
# monolith the archive never crosses (create_shader_state hands the two structs over
|
||||
# by pointer beside the record), so the codec is live code only in the VERIFY lane,
|
||||
# where the applier serialises, deserialises and field-compares before storing.
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramArtifactsCodec.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
if (MOBILEGL_BUILD_DISAGGREGATED)
|
||||
message(STATUS "MobileGL: disaggregated transport ON, appending MG_Remote sources")
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Remote/Transport/Ring.cpp
|
||||
MobileGL/MG_Remote/Transport/Doorbell.cpp
|
||||
MobileGL/MG_Remote/Transport/ShmSegment.cpp
|
||||
# Both platform halves are listed unconditionally and each is empty on
|
||||
# the other OS, so neither can rot behind an `if (WIN32)` nobody
|
||||
# configures.
|
||||
MobileGL/MG_Remote/Transport/ShmSegmentPosix.cpp
|
||||
MobileGL/MG_Remote/Transport/ShmSegmentWin32.cpp
|
||||
MobileGL/MG_Remote/Transport/FdPassing.cpp
|
||||
MobileGL/MG_Remote/Transport/InProcessTransport.cpp
|
||||
# Keeps MG_Util/Debug/Log.h - and through it the GL frontend's
|
||||
# umbrella header - out of the header-only wire code (WireLog.h).
|
||||
MobileGL/MG_Remote/Transport/WireLog.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
if (APPLE AND NOT MOBILEGL_IOS)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Impl/CGLImpl/CGLImpl.cpp
|
||||
@@ -464,11 +569,29 @@ set(MOBILEGL_COMPILE_DEF
|
||||
-DASIO_NO_DEPRECATED
|
||||
)
|
||||
|
||||
if (MOBILEGL_BUILD_DISAGGREGATED)
|
||||
list(APPEND MOBILEGL_COMPILE_DEF -DMOBILEGL_BUILD_DISAGGREGATED=1)
|
||||
endif()
|
||||
|
||||
if (MOBILEGL_PIPE_PUSH)
|
||||
list(APPEND MOBILEGL_COMPILE_DEF -DMOBILEGL_PIPE_PUSH=1)
|
||||
endif()
|
||||
if (MOBILEGL_PIPE_VERIFY)
|
||||
list(APPEND MOBILEGL_COMPILE_DEF -DMOBILEGL_PIPE_VERIFY=1)
|
||||
endif()
|
||||
if (MOBILEGL_PIPE_LEGACY_MEMOS)
|
||||
list(APPEND MOBILEGL_COMPILE_DEF -DMOBILEGL_PIPE_LEGACY_MEMOS=1)
|
||||
endif()
|
||||
|
||||
message(STATUS "MOBILEGL_COMPILE_DEF=${MOBILEGL_COMPILE_DEF}")
|
||||
|
||||
set(MOBILEGL_INCLUDE_DIR
|
||||
${CMAKE_SOURCE_DIR}/include
|
||||
${CMAKE_SOURCE_DIR}/MobileGL
|
||||
# The MGPipe boundary headers. They are reachable as <MG_Pipe/MGPipe.h> through the
|
||||
# line above too; this entry lets the client, the backends and MG_Remote spell them
|
||||
# as <MGPipe.h> once MG_Pipe stops being a leaf of the frontend tree.
|
||||
${CMAKE_SOURCE_DIR}/MobileGL/MG_Pipe
|
||||
${spirv-tools_SOURCE_DIR}
|
||||
${spirv-tools_SOURCE_DIR}/include
|
||||
${spirv-tools_BINARY_DIR}
|
||||
@@ -479,6 +602,13 @@ set(MOBILEGL_INCLUDE_DIR
|
||||
${CMAKE_SOURCE_DIR}/3rdparty/asio/include
|
||||
)
|
||||
|
||||
if (MOBILEGL_BUILD_DISAGGREGATED)
|
||||
# Header-only runtime: an include path, no add_subdirectory, no link
|
||||
# target, and above all no flatc in the build graph. protocol_generated.h
|
||||
# is committed and regenerated by scripts/gen_protocol.py.
|
||||
list(APPEND MOBILEGL_INCLUDE_DIR ${CMAKE_SOURCE_DIR}/3rdparty/flatbuffers/include)
|
||||
endif()
|
||||
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
${SOURCE_FILES}
|
||||
)
|
||||
@@ -695,3 +825,43 @@ endif()
|
||||
if (ANDROID AND MOBILEGL_BUILD_INTEGRATION_TEST)
|
||||
add_subdirectory(MobileGL/MG_IntegrationTest)
|
||||
endif()
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# P0 spike A: the Android delivery chain for a second native executable.
|
||||
#
|
||||
# The disaggregated design needs a server process on Android (PLAN-B.md §8.1,
|
||||
# inheriting PLAN.md §11.1-§11.6). An APK's only exec-able install location is
|
||||
# lib/<abi>/, and the packager only puts a file there if it is named lib*.so -
|
||||
# so a second executable has to be built with an .so name and exec'd out of
|
||||
# getApplicationInfo().nativeLibraryDir. This target is the stub that proves the
|
||||
# chain end to end: it is packaged like a library, exec'd from the app's own
|
||||
# untrusted_app process, and writes a marker the parent reads back.
|
||||
#
|
||||
# Off by default and ANDROID-only, so no shipping configuration builds it. The
|
||||
# trace flavour of the plugin APK turns it on (android-plugin/build.gradle).
|
||||
# ---------------------------------------------------------------------------
|
||||
if (ANDROID AND MOBILEGL_BUILD_SERVER_SPIKE)
|
||||
add_executable(MobileGLServer
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/tools/spikes/server_stub/main.cpp)
|
||||
|
||||
# An executable that is named like a shared library still has to be a real
|
||||
# PIE executable: Android has refused non-PIE executables since API 21, and
|
||||
# the name alone does not change what the loader demands of the file.
|
||||
set_target_properties(MobileGLServer PROPERTIES
|
||||
PREFIX "lib"
|
||||
SUFFIX ".so"
|
||||
OUTPUT_NAME "MobileGLServer"
|
||||
POSITION_INDEPENDENT_CODE ON)
|
||||
target_compile_options(MobileGLServer PRIVATE -fPIE)
|
||||
target_link_options(MobileGLServer PRIVATE -pie)
|
||||
|
||||
# AGP packages what the external native build drops into the per-ABI output
|
||||
# directory, and it selects by the .so extension. CMake puts executables in
|
||||
# CMAKE_RUNTIME_OUTPUT_DIRECTORY, which is not the directory AGP hands to
|
||||
# CMAKE_LIBRARY_OUTPUT_DIRECTORY, so point this target's runtime output at
|
||||
# the library directory when the generator gave us one.
|
||||
if (CMAKE_LIBRARY_OUTPUT_DIRECTORY)
|
||||
set_target_properties(MobileGLServer PROPERTIES
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_LIBRARY_OUTPUT_DIRECTORY}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
+204
-24
@@ -14,7 +14,7 @@ namespace MobileGL::MG_Config {
|
||||
inline const String ProjectName = "MobileGL";
|
||||
inline const String CoreName = "MobileGL Core";
|
||||
inline const String CoreVendor = "MobileGL-Dev (BZLZHH, Swung0x48, Tungsten)";
|
||||
inline const Version CoreVersion = {26, 8, 0, "-dev", VersionType::Development};
|
||||
inline const Version CoreVersion = {26, 9, 0, "-dev", VersionType::Development};
|
||||
inline const VersionStringFormatAttrib DefaultVersionStringFormatAttrib = {2, 2, 0, true, true};
|
||||
inline const Uint64 CacheVersion = 0;
|
||||
|
||||
@@ -69,34 +69,34 @@ namespace MobileGL::MG_Config {
|
||||
struct FeaturesTable {
|
||||
// MOBILEGL_DISABLE_TIMERQUERY: do not advertise or use GPU timer queries.
|
||||
Bool DisableTimerQuery = false;
|
||||
// MOBILEGL_ENABLE_GLES_TEXTURE_VIEW: advertise GL_ARB_texture_view on DirectGLES when
|
||||
// MOBILEGL_ESPRYT_ENABLE_TEXTURE_VIEW: advertise GL_ARB_texture_view on DirectGLES when
|
||||
// the host ES driver has EXT/OES_texture_view. Off by default: the host extension is
|
||||
// present on Adreno 830 and the functional half of KHR-GL4{2,3}.texture_view still fails
|
||||
// there, because the view's ES internalformat is normalized independently of the storage
|
||||
// it aliases (see BackendObject_DirectGLES::BuildAdvertisedExtensions). The flag exists
|
||||
// so that work can be done without editing the gate.
|
||||
Bool EnableGlesTextureView = false;
|
||||
Bool EsprytEnableTextureView = false;
|
||||
// MOBILEGL_ENABLE_SPIRV_VALIDATION: validate generated and transformed SPIR-V.
|
||||
// Disabled by default because validation is a diagnostics-only cost.
|
||||
Bool EnableSpirvValidation = false;
|
||||
// MOBILEGL_USE_ANGLE: load ANGLE EGL/GLES libraries.
|
||||
Bool UseAngle = false;
|
||||
// MOBILEGL_ESPRYT_USE_ANGLE: load ANGLE EGL/GLES libraries.
|
||||
Bool EsprytUseAngle = false;
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
// MOBILEGL_TRACE_ANGLE_VARIANT: signed trace-APK ANGLE build short hash.
|
||||
String TraceAngleVariant;
|
||||
#endif
|
||||
// MOBILEGL_DISABLE_SUBGROUP: force-disable Vulkan shader subgroup support,
|
||||
// MOBILEGL_MAGMA_DISABLE_SUBGROUP: force-disable Vulkan shader subgroup support,
|
||||
// including the opt-in emulated compute path below.
|
||||
Bool DisableSubgroup = false;
|
||||
Bool MagmaDisableSubgroup = false;
|
||||
// MOBILEGL_MAGMA_EMULATE_SUBGROUP: implement GL_KHR_shader_subgroup's compute
|
||||
// stage on a 32-lane VIRTUAL subgroup lowered to workgroup-shared memory
|
||||
// (ShaderTranspiler::EmulateSubgroupsPass). Strictly a last resort: it only ever
|
||||
// engages when this flag is set AND the device has no native subgroup support at
|
||||
// all - a device with real subgroup operations always uses them natively,
|
||||
// whatever their width (the known iterationRP defect is patched by
|
||||
// FixIterationRPSubgroupScratch below instead). Off by default.
|
||||
// MagmaFixIterationRPSubgroupScratch below instead). Off by default.
|
||||
Bool MagmaEmulateSubgroup = false;
|
||||
// MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: patch iterationRP's own bug - the
|
||||
// MOBILEGL_MAGMA_FIX_ITERATIONRP_SUBGROUP_SCRATCH: patch iterationRP's own bug - the
|
||||
// pack declares `shared vec2 prefixSumCache[32]` for a 512-invocation exposure
|
||||
// reduction and indexes it by gl_SubgroupID, so any device with sub-16-lane
|
||||
// subgroups (8-lane lavapipe -> 64 subgroups) writes shared memory out of
|
||||
@@ -106,12 +106,12 @@ namespace MobileGL::MG_Config {
|
||||
// so every other shader passes through byte-identical - as does iterationRP
|
||||
// itself on >= 16-lane devices. Auto is ON; ForceOff replays the pack's bug
|
||||
// verbatim.
|
||||
QuirkOverride FixIterationRPSubgroupScratch = QuirkOverride::Auto;
|
||||
// MOBILEGL_ITERATIONRP_FIX_BARRIER: repair Program 203's missing workgroup
|
||||
QuirkOverride MagmaFixIterationRPSubgroupScratch = QuirkOverride::Auto;
|
||||
// MOBILEGL_MAGMA_ITERATIONRP_FIX_BARRIER: repair Program 203's missing workgroup
|
||||
// rendezvous between its two reductions over prefixSumCache. Off by default and
|
||||
// fingerprint-gated by FixIterationRPBarrierPass when enabled.
|
||||
Bool IterationRPFixBarrier = false;
|
||||
// MOBILEGL_DERIVE_NUM_SUBGROUPS: replace compute gl_NumSubgroups loads with
|
||||
Bool MagmaIterationRPFixBarrier = false;
|
||||
// MOBILEGL_MAGMA_DERIVE_NUM_SUBGROUPS: replace compute gl_NumSubgroups loads with
|
||||
// ceil(workgroup invocations / gl_SubgroupSize) on the NATIVE subgroup path
|
||||
// (ShaderTranspiler::DeriveNumSubgroupsPass). Auto is ON: GL requires
|
||||
// gl_SubgroupID < gl_NumSubgroups, Adreno's builtin reports 1 while the same
|
||||
@@ -119,7 +119,7 @@ namespace MobileGL::MG_Config {
|
||||
// whenever the pipeline can request REQUIRE_FULL_SUBGROUPS (which the renderer
|
||||
// does whenever local_size_x is a multiple of the native width). ForceOff returns
|
||||
// to the raw driver builtin.
|
||||
QuirkOverride DeriveNumSubgroups = QuirkOverride::Auto;
|
||||
QuirkOverride MagmaDeriveNumSubgroups = QuirkOverride::Auto;
|
||||
// MOBILEGL_ADVERTISE_FP64: add GL_ARB_gpu_shader_fp64 to the advertised extension
|
||||
// string. `double` in a shader always WORKS - it is narrowed to 32 bits before any
|
||||
// module reaches a backend (ShaderTranspiler::DemoteFloat64Pass) - but the extension
|
||||
@@ -132,16 +132,39 @@ namespace MobileGL::MG_Config {
|
||||
Bool MagmaR11G11B10FFallback = false;
|
||||
// MOBILEGL_MAGMA_FRAMESINFLIGHT: requested Magma frames in flight, defaulting to 3.
|
||||
Uint32 MagmaFramesInFlight = 3;
|
||||
// MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER: avoid mipmap min filters in samplers,
|
||||
// MOBILEGL_ESPRYT_AVOID_SAMPLER_MIPMAP_MIN_FILTER: avoid mipmap min filters in samplers,
|
||||
// resolves certain rendering bugs on ANGLE + llvmpipe.
|
||||
Bool AvoidSamplerMipmapMinFilter = false;
|
||||
// MOBILEGL_AVOID_EXPLICIT_LOD_BIAS: leave an already-explicit LOD argument alone when
|
||||
Bool EsprytAvoidSamplerMipmapMinFilter = false;
|
||||
// MOBILEGL_ESPRYT_AVOID_EXPLICIT_LOD_BIAS: leave an already-explicit LOD argument alone when
|
||||
// emulating GL_TEXTURE_LOD_BIAS, instead of adding the bias uniform to it. Injecting
|
||||
// the uniform turns a compile-time-constant LOD into a runtime expression, which
|
||||
// sends ANGLE + llvmpipe down a mip-selection path that dereferences a NULL
|
||||
// descriptor and kills the process. Deviates from spec (Vulkan adds the bias to
|
||||
// OpImageSampleExplicitLod), so it is an avoidance for that stack only.
|
||||
Bool AvoidExplicitLodBias = false;
|
||||
Bool EsprytAvoidExplicitLodBias = false;
|
||||
// MOBILEGL_ESPRYT_UNLOCATED_IO_BLOCKS: emit a tessellation/geometry program's
|
||||
// inter-stage interface blocks WITHOUT their layout(location=) qualifier, letting ES
|
||||
// match them by block name and member sequence instead. The Mali ES driver delivers
|
||||
// nothing at all through a located block once a tessellation or geometry stage is in
|
||||
// the pipeline; the driver POST measures that and turns this on by itself, so Auto is
|
||||
// the right setting everywhere. ForceOn exists so the emulation can be exercised on a
|
||||
// healthy driver - which is what the integration lane does, since llvmpipe and
|
||||
// lavapipe carry a located block correctly and would otherwise never run this code -
|
||||
// and ForceOff is the negative control. See StripIoBlockLocationsPass.
|
||||
QuirkOverride EsprytUnlocatedIoBlocks = QuirkOverride::Auto;
|
||||
// MOBILEGL_POINT_SIZE_DEMOTION: demote gl_PointSize out of tessellation/geometry
|
||||
// stages into an ordinary varying (ShaderCompiler::
|
||||
// DemoteTessellationGeometryPointSizeForProgram) instead of declining such programs
|
||||
// on a device that advertises neither EXT/OES_tessellation_point_size /
|
||||
// geometry_point_size (DirectGLES) nor shaderTessellationAndGeometryPointSize
|
||||
// (DirectVulkan). Auto arms it exactly where the detection says the capability is
|
||||
// absent, which is the right setting everywhere. ForceOn exists so the demotion can
|
||||
// be exercised on a healthy driver - llvmpipe and lavapipe host the built-in
|
||||
// natively and would otherwise never run this code, which is what the pinned
|
||||
// integration lane uses - and ForceOff restores the plain declines (escape hatch /
|
||||
// negative control). Cross-backend by design: the demotion runs in the shared
|
||||
// phase-B chain, so one switch covers both. See DemotePointSizePass.
|
||||
QuirkOverride PointSizeDemotion = QuirkOverride::Auto;
|
||||
// MOBILEGL_COHERENT_AS_FLUSH: app-compat for engines (e.g. Flywheel) that write
|
||||
// GPU-read data through persistent GL_MAP_FLUSH_EXPLICIT_BIT maps they never
|
||||
// flush. Persistent FLUSH_EXPLICIT map requests are rewritten to coherent
|
||||
@@ -151,10 +174,38 @@ namespace MobileGL::MG_Config {
|
||||
Bool CoherentAsFlush = false;
|
||||
// MOBILEGL_TRACE_SKIP_AUTODESTROY: skip teardown in the ELF destructor (Init.cpp).
|
||||
Bool TraceSkipAutodestroy = false;
|
||||
// MOBILEGL_DISABLE_UBO_RING: force the DirectGLES global-UBO upload back to the
|
||||
// MOBILEGL_ESPRYT_DISABLE_UBO_RING: force the DirectGLES global-UBO upload back to the
|
||||
// per-draw glBufferSubData path instead of the persistent-mapped ring allocator
|
||||
// (negative control / driver-bug escape hatch).
|
||||
Bool DisableUboRing = false;
|
||||
Bool EsprytDisableUboRing = false;
|
||||
// MOBILEGL_ESPRYT_DISABLE_UNPACK_RING: force DirectGLES texture uploads back to
|
||||
// glTexSubImage from the client pointer instead of staging them through the
|
||||
// persistent-mapped unpack-PBO ring (negative control / driver-bug escape
|
||||
// hatch).
|
||||
Bool EsprytDisableUnpackRing = false;
|
||||
// MOBILEGL_ESPRYT_DISABLE_UPLOAD_RING: force DirectGLES app buffer updates
|
||||
// (glBufferSubData / map flushes) back to the immediate driver upload instead
|
||||
// of queueing them for the staged-copy flush through the persistent-mapped
|
||||
// upload ring (negative control / driver-bug escape hatch; the immediate
|
||||
// upload stalls on drivers that resolve the WAR hazard on the CPU, e.g. Mali).
|
||||
Bool EsprytDisableUploadRing = false;
|
||||
// MOBILEGL_ESPRYT_DISABLE_INVALIDATE_FLUSH: skip the glMapBufferRange(WRITE |
|
||||
// INVALIDATE_RANGE) tier of the DirectGLES pending-range flush and go straight
|
||||
// to the upload ring's staged glCopyBufferSubData (negative control / escape
|
||||
// hatch for a driver whose range-invalidating map misbehaves). The map tier is
|
||||
// what keeps a partial write into a large in-flight buffer priced by the RANGE:
|
||||
// on Mali both the immediate glBufferSubData and a staged copy into a busy
|
||||
// mutable store ghost the whole destination on the CPU.
|
||||
Bool EsprytDisableInvalidateFlush = false;
|
||||
// MOBILEGL_DISABLE_LARGE_BUFFER_ADOPTION: keep mesh-arena-sized buffer stores
|
||||
// (>= 16MiB) on the CPU-shadow model instead of backing them with the backend's
|
||||
// persistently+coherently mapped storage at definition time (negative control /
|
||||
// escape hatch). Frontend-scoped: it engages only where the active backend
|
||||
// provides AcquirePersistentMap. With adoption on, an app SubData into a busy
|
||||
// 128MB arena is a plain memcpy into GPU-visible memory; every driver-mediated
|
||||
// route for the same write stalls the thread or ghost-copies the whole arena on
|
||||
// this class of Mali driver, and the arena stops costing its size again in RAM.
|
||||
Bool DisableLargeBufferAdoption = false;
|
||||
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
|
||||
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
|
||||
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
|
||||
@@ -175,10 +226,10 @@ namespace MobileGL::MG_Config {
|
||||
// gl_FragDepth writers, and fully color-masked attachments are exempt (see
|
||||
// PipelineFactory::ShouldSuppressDepthWrite). Auto detects Qualcomm.
|
||||
QuirkOverride MagmaDisableBlendedDepthWriteQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS: leave the Vulkan robustBufferAccess device
|
||||
// MOBILEGL_MAGMA_DISABLE_ROBUST_BUFFER_ACCESS: leave the Vulkan robustBufferAccess device
|
||||
// feature off. It is enabled by default to match GL's defined out-of-range fetch
|
||||
// behavior; this escape hatch exists to measure or dodge its GPU cost on a device.
|
||||
Bool DisableRobustBufferAccess = false;
|
||||
Bool MagmaDisableRobustBufferAccess = false;
|
||||
// MOBILEGL_MAGMA_MULTIDRAW_MODE: preferred DirectVulkan multi-draw dispatch tier
|
||||
// ("ext" | "indirect" | "unroll", see MultiDrawMode). Clamped to device support;
|
||||
// unset picks the best supported tier.
|
||||
@@ -219,7 +270,7 @@ namespace MobileGL::MG_Config {
|
||||
// miscompiled shader: if a device ever renders differently with the cache
|
||||
// on, one run with this falsy says so.
|
||||
QuirkOverride ShaderTranslationCache = QuirkOverride::Auto;
|
||||
// MOBILEGL_FORCE_VIEWPORT_ARRAY_EMULATION: DirectGLES' gl_ViewportIndex routing
|
||||
// MOBILEGL_ESPRYT_FORCE_VIEWPORT_ARRAY_EMULATION: DirectGLES' gl_ViewportIndex routing
|
||||
// emulation - the builtin becomes a flat varying, the fragment stage gets a
|
||||
// per-pass gate, and a routed draw is REPLAYED once per distinct viewport state
|
||||
// with the real glViewport/glScissor/glDepthRangef set for it. Auto is ON, and
|
||||
@@ -231,7 +282,136 @@ namespace MobileGL::MG_Config {
|
||||
// the pre-emulation path, extension passthrough where it exists and
|
||||
// LowerViewportIndexPass' demote-to-a-plain-global where it does not - and is
|
||||
// the negative control the emulation is measured against.
|
||||
QuirkOverride ViewportArrayEmulation = QuirkOverride::Auto;
|
||||
QuirkOverride EsprytViewportArrayEmulation = QuirkOverride::Auto;
|
||||
// MOBILEGL_ESPRYT_WIDEN_PACKED16_STORAGE: DirectGLES stores GL_RGB565/GL_RGB5(A1)/GL_RGBA4
|
||||
// images as 8-bit-per-channel ES storage (GL_RGB8/GL_RGBA8) instead of the driver's
|
||||
// native 16-bit packed formats. Auto defers to a POST driver-bug probe
|
||||
// (SelfTest::CopyImageMirrorsPacked16FieldOrder): some Mali drivers store SOME
|
||||
// packed16 allocations with a MIRRORED field order (allocation-scoped and
|
||||
// shape/context dependent - the failing 30x30x12 GL_TEXTURE_2D_ARRAYs are mirrored
|
||||
// at every level), so glCopyImageSubData - a raw texel-block move - lands R/G/B/A
|
||||
// reversed whenever exactly one endpoint sits in a mirrored allocation
|
||||
// (KHR-GL4x.copy_image.functional rgb5/rgb5_a1/rgba4 x every *2d_array* pair).
|
||||
// With no 16-bit packed ES image left there is no field order to disagree about; the
|
||||
// client word still round-trips exactly, because the canonical shadow is already
|
||||
// UNorm8 and an n-bit field encodes to UNorm8 and back losslessly for n <= 8.
|
||||
// ForceOn widens on any driver (the llvmpipe suites use it to exercise the widened
|
||||
// path); ForceOff keeps the native narrow storage even where the probe fires - the
|
||||
// negative control that replays the corruption. Costs 2x the memory of the affected
|
||||
// formats where it engages, which is why Auto is probe-gated rather than always-on.
|
||||
QuirkOverride EsprytWidenPacked16Storage = QuirkOverride::Auto;
|
||||
// MOBILEGL_MAGMA_PRIMGEN_QUERY_REROUTE: DirectVulkan's GL_PRIMITIVES_GENERATED
|
||||
// reroute for draws made while transform feedback is INACTIVE. The stream query
|
||||
// (VK_QUERY_TYPE_TRANSFORM_FEEDBACK_STREAM_EXT primitivesNeeded) is defined to count
|
||||
// them, but a Mali driver - and Mesa lavapipe - answers 0 unless a capture span is
|
||||
// open, which is exactly the shape the CTS uses to measure the tessellator, so ~29
|
||||
// tessellation tests per tree size a capture buffer from the 0 and die on the
|
||||
// zero-length map. Auto defers to a device probe at renderer bring-up
|
||||
// (SelfTest::RunPrimitivesGeneratedNoXfbProbe), which measures two substitutes on
|
||||
// the same capture-less draws and arms the best proven one: the dedicated
|
||||
// VK_EXT_primitives_generated_query (exact semantics by definition; lavapipe passes
|
||||
// it, rasterizer discard included), else a clipping-invocations pipeline-statistics
|
||||
// pool (see the verdict vocabulary for its rasterizer-discard split). ForceOn pins
|
||||
// the reroute structurally wherever a pool can exist (the arming-observable lane,
|
||||
// immune to the probe's verdict moving), and ForceOff is the negative control that
|
||||
// replays the driver's silence.
|
||||
QuirkOverride MagmaPrimGenQueryReroute = QuirkOverride::Auto;
|
||||
// --- MGPipe (the disaggregation plan's explicit frontend/backend boundary) ---
|
||||
// MOBILEGL_PIPE_PUSH: per-subsystem bitmask selecting which state the frontend
|
||||
// PUSHES over MGPipe instead of leaving the backend to pull it out of GLContext.
|
||||
// 0 - the only shipped value until the migration lands - is "pull everything",
|
||||
// i.e. exactly today's behaviour, and is the default of a PULL build, where the
|
||||
// knob is meaningless anyway. A PUSH build defaults to every subsystem migrated so
|
||||
// far (MG_Pipe::kMGPipeSubsystemsMigratedAtP4a), so MOBILEGL_PIPE_PUSH=0 in the
|
||||
// environment is the all-pull control and 0x1ff (kMGPipeSubsystemsMigratedAtP3a) is
|
||||
// the "everything before P4a" control P4a's A/B is run against - each phase's
|
||||
// constant survives as the next phase's control, which is why none of them is ever
|
||||
// edited. Accepts decimal or 0x-prefixed hex, and operators pass it as hex, so the
|
||||
// bits are listed here (MG_Pipe/MGPipe.h owns them):
|
||||
// 0x01 render state (create/bind_render_state + set_dynamic_state)
|
||||
// 0x02 pixel pack 0x04 patch state 0x08 vertex attrib defaults
|
||||
// 0x10 residual values 0x20 Espryt slots 0x40 Magma vertex input
|
||||
// 0x80 resources (the resource_* family: the seven BufferBackendOps hooks)
|
||||
// 0x100 vertex input (vertex elements / vertex buffers / index buffer)
|
||||
// 0x200 framebuffer (set_framebuffer_state) - requires 0x400
|
||||
// 0x400 texture resources (texture + renderbuffer resource_*,
|
||||
// set_texture_params) - requires 0x80 AND 0x800
|
||||
// (the built-in sampler CSO a set_texture_params record names is minted by
|
||||
// the sampler family alone, ID-15; the four rows are MG_Impl/Pipe/PipeFill.cpp's
|
||||
// kMGPipeP4aFamilyDependencies, mirrored bit for bit by Espryt's resolvers)
|
||||
// 0x800 samplers (sampler CSO, sampler view, set_sampler_views /
|
||||
// bind_sampler_states / set_shader_images) - requires 0x400
|
||||
// 0x1000 programs (shader CSO, set_draw/dispatch_program, global constants)
|
||||
// A dependency that is not met is REFUSED with one ERROR naming both bits and the
|
||||
// family runs its legacy arm; it is never half-run.
|
||||
// 1<<63 NOT a subsystem, a BEHAVIOUR: turn OFF client-side content addressing of
|
||||
// CSOs, so every pipeline-version change mints a fresh CSO and the map is
|
||||
// never probed. The negative control the CSO design is measured against.
|
||||
Uint64 PipePush = 0;
|
||||
// MOBILEGL_PIPE_VERIFY: per-draw, per-FIELD shadow comparison of the pushed state
|
||||
// against a snapshot taken from GLContext the old way, printing the first field
|
||||
// that differs and the draw serial. Roughly 5-10x slower and never shipped; it is
|
||||
// the semantic gate that replaces byte identity, and it catches the dangerous
|
||||
// direction - a dirty bit that fires too RARELY - which no purity gate can see.
|
||||
Bool PipeVerify = false;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// The three knobs of the MOBILEGL_PIPE_VERIFY build (P1 brief D2). Compiled only
|
||||
// under MOBILEGL_PIPE_PUSH so the pull build's FeaturesTable does not change size.
|
||||
// MOBILEGL_PIPE_VERIFY_FATAL: the first divergence aborts (default). 0 logs and
|
||||
// counts instead, for triage and for the lane that must survive to read its own
|
||||
// log. Tri-state parse like PipeLegacyMemos: only an explicit falsy value turns it
|
||||
// off.
|
||||
Bool PipeVerifyFatal = true;
|
||||
// MOBILEGL_PIPE_VERIFY_CORRUPT: a field name from kMGPipeInputFieldNames[]; the
|
||||
// comparator perturbs that field in the SNAPSHOT arm before the entry compare, so a
|
||||
// green verify run goes red naming it (negative control A). Unknown name is
|
||||
// Fatal{PipeVerifyBadKnob}.
|
||||
String PipeVerifyCorrupt;
|
||||
// MOBILEGL_PIPE_POISON_OMIT: <Verb>:<FieldName>; the filler skips the STAMP (not
|
||||
// the value) of that field for that verb, an omission indistinguishable from a
|
||||
// forgotten FillPoints.def row, so that verb's read of it is
|
||||
// Fatal{UnmigratedPipeInput} (negative control B). Unknown name is
|
||||
// Fatal{PipeVerifyBadKnob}.
|
||||
String PipePoisonOmit;
|
||||
// MOBILEGL_PIPE_HANDLE_ABA_CONTROL (negative control C, P2 brief D18): replace the
|
||||
// OBJECT IDENTITY in every DirectVulkan vertex-input memo key with a constant, on
|
||||
// whichever arm the run is on - the pre-handle (address, lifetime id) pair AND the
|
||||
// handle arm's {slot, gen} generation - so a replacement object inherits its dead
|
||||
// predecessor's resolved vertex bindings and HandleRecycleScenario.AbaControl asserts
|
||||
// the WRONG pixels. That is what proves the reproducer still reproduces. D18 wrote
|
||||
// this as "hash the raw BufferObject* instead of its lifetime id"; measured, the heap
|
||||
// block is never handed back, so that spelling collided with nothing and the control
|
||||
// went vacuous - see MagmaPipeArms.h's MagmaPipeAbaControlDefeatsIdentity for the
|
||||
// measurement and for what the control still leaves standing. Under
|
||||
// MOBILEGL_PIPE_PUSH only, so it cannot exist in a shipping pull build.
|
||||
Bool PipeHandleAbaControl = false;
|
||||
#endif
|
||||
// MOBILEGL_PIPE_STATS: dump the boundary counters (bytes, calls, roundtrips,
|
||||
// texture pulls, upload shapes, residual-block bytes, index mirror bytes).
|
||||
Bool PipeStats = false;
|
||||
// MOBILEGL_PIPE_LEGACY_MEMOS: keep the pre-handle registries and TwinLookupMemos
|
||||
// alive so the first handle waves have a real old-versus-new arm to be compared
|
||||
// against. ON by default for the whole migration window, deleted with the pull
|
||||
// path itself.
|
||||
Bool PipeLegacyMemos = true;
|
||||
// MOBILEGL_PIPE_TEXEL_RETAIN_MB: LRU budget for texels retained against a
|
||||
// server-initiated texture re-send. Default 0, i.e. OFF: MipmapStorage already
|
||||
// holds a complete CPU shadow, so this cache buys latency, never correctness.
|
||||
Uint32 PipeTexelRetainMb = 0;
|
||||
// MOBILEGL_PIPE_INDEX_MIRROR_MB: budget for the server-side index host mirror,
|
||||
// which is what lets primitive-restart rewriting and multi-draw flattening stay on
|
||||
// the server without shipping index bytes per draw. Over budget it degrades to
|
||||
// per-draw staging, counted separately in the stats.
|
||||
Uint32 PipeIndexMirrorMb = 64;
|
||||
// MOBILEGL_PIPE_STATS_PERIOD: frames per boundary-counter summary line. 120 is the
|
||||
// steady-state cadence; the device retrace harness never reaches the teardown dump
|
||||
// and a trimmed fixture (create-indirect) is shorter than 120 frames, so a run that
|
||||
// needs its numbers at all sets this low enough to land at least one window.
|
||||
Uint32 PipeStatsPeriod = 120;
|
||||
// MOBILEGL_PIPE_STATS_FILE: where the boundary counters' teardown JSON dump goes.
|
||||
// Empty (the default) means no dump; the per-120-frame summary line still goes to
|
||||
// the log whenever PipeStats is on, so a device run needs no writable path.
|
||||
String PipeStatsFile;
|
||||
};
|
||||
extern FeaturesTable Features;
|
||||
} // namespace MobileGL::MG_Config
|
||||
|
||||
+95
-14
@@ -7,6 +7,12 @@
|
||||
// End of Source File Header
|
||||
|
||||
#include "Config.h"
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// For kMGPipeSubsystemsMigratedAtP4a, the push build's PipePush default (the P2 and P3a
|
||||
// constants beside it are the phase-by-phase controls, not the default). Push-only, so the
|
||||
// pull build's translation unit is unchanged.
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#endif
|
||||
|
||||
#include <cerrno>
|
||||
#include <cstdlib>
|
||||
@@ -159,36 +165,74 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
return static_cast<Uint32>(parsedValue);
|
||||
}
|
||||
|
||||
// Same contract as QueryEnvUint32, over 64 bits and accepting an explicit 0x prefix: the
|
||||
// one consumer is a subsystem BITMASK, and a bitmask written in decimal is unreadable.
|
||||
// Decimal otherwise - never strtoull's base 0, whose "leading zero means octal" rule
|
||||
// silently read MOBILEGL_PIPE_PUSH=010 as 8 - and a '-' anywhere is rejected rather than
|
||||
// wrapped, which strtoull would otherwise do without complaint (-1 -> every bit set).
|
||||
inline Uint64 QueryEnvUint64(const String& key, Uint64 defaultValue) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
return defaultValue;
|
||||
}
|
||||
|
||||
const String& value = it->second;
|
||||
const char* text = value.c_str();
|
||||
int base = 10;
|
||||
if (value.size() > 2 && text[0] == '0' && (text[1] == 'x' || text[1] == 'X')) {
|
||||
text += 2;
|
||||
base = 16;
|
||||
}
|
||||
char* parseEnd = nullptr;
|
||||
errno = 0;
|
||||
const bool negative = value.find('-') != String::npos;
|
||||
const unsigned long long parsedValue = negative ? 0 : std::strtoull(text, &parseEnd, base);
|
||||
if (negative || parseEnd == text || *parseEnd != '\0' || errno == ERANGE) {
|
||||
MGLOG_W("Config: Ignoring invalid env variable %s='%s'; expected a non-negative integer "
|
||||
"(decimal, or 0x-prefixed hexadecimal), using default %llu",
|
||||
key.c_str(), value.c_str(), static_cast<unsigned long long>(defaultValue));
|
||||
return defaultValue;
|
||||
}
|
||||
|
||||
return static_cast<Uint64>(parsedValue);
|
||||
}
|
||||
|
||||
inline void InitFeatures() {
|
||||
auto& features = MG_Config::Features;
|
||||
features.DisableTimerQuery = QueryEnvFlag("MOBILEGL_DISABLE_TIMERQUERY");
|
||||
features.EnableGlesTextureView = QueryEnvFlag("MOBILEGL_ENABLE_GLES_TEXTURE_VIEW");
|
||||
features.EsprytEnableTextureView = QueryEnvFlag("MOBILEGL_ESPRYT_ENABLE_TEXTURE_VIEW");
|
||||
features.EnableSpirvValidation = QueryEnvFlag("MOBILEGL_ENABLE_SPIRV_VALIDATION");
|
||||
features.UseAngle = QueryEnvFlag("MOBILEGL_USE_ANGLE");
|
||||
features.EsprytUseAngle = QueryEnvFlag("MOBILEGL_ESPRYT_USE_ANGLE");
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
QueryEnvVariable("MOBILEGL_TRACE_ANGLE_VARIANT", features.TraceAngleVariant, "");
|
||||
#endif
|
||||
features.DisableSubgroup = QueryEnvFlag("MOBILEGL_DISABLE_SUBGROUP");
|
||||
features.MagmaDisableSubgroup = QueryEnvFlag("MOBILEGL_MAGMA_DISABLE_SUBGROUP");
|
||||
features.MagmaEmulateSubgroup = QueryEnvFlag("MOBILEGL_MAGMA_EMULATE_SUBGROUP");
|
||||
features.FixIterationRPSubgroupScratch =
|
||||
QueryEnvQuirkOverride("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
|
||||
features.IterationRPFixBarrier = QueryEnvFlag("MOBILEGL_ITERATIONRP_FIX_BARRIER");
|
||||
features.DeriveNumSubgroups = QueryEnvQuirkOverride("MOBILEGL_DERIVE_NUM_SUBGROUPS");
|
||||
features.MagmaFixIterationRPSubgroupScratch =
|
||||
QueryEnvQuirkOverride("MOBILEGL_MAGMA_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
|
||||
features.MagmaIterationRPFixBarrier = QueryEnvFlag("MOBILEGL_MAGMA_ITERATIONRP_FIX_BARRIER");
|
||||
features.MagmaDeriveNumSubgroups = QueryEnvQuirkOverride("MOBILEGL_MAGMA_DERIVE_NUM_SUBGROUPS");
|
||||
features.AdvertiseFp64 = QueryEnvFlag("MOBILEGL_ADVERTISE_FP64");
|
||||
features.MagmaR11G11B10FFallback = QueryEnvFlag("MOBILEGL_MAGMA_R11G11B10F_FALLBACK");
|
||||
features.MagmaFramesInFlight = QueryEnvUint32("MOBILEGL_MAGMA_FRAMESINFLIGHT", 3, 1, 64);
|
||||
features.AvoidSamplerMipmapMinFilter =
|
||||
QueryEnvFlag("MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER");
|
||||
features.AvoidExplicitLodBias = QueryEnvFlag("MOBILEGL_AVOID_EXPLICIT_LOD_BIAS");
|
||||
features.EsprytAvoidSamplerMipmapMinFilter =
|
||||
QueryEnvFlag("MOBILEGL_ESPRYT_AVOID_SAMPLER_MIPMAP_MIN_FILTER");
|
||||
features.EsprytAvoidExplicitLodBias = QueryEnvFlag("MOBILEGL_ESPRYT_AVOID_EXPLICIT_LOD_BIAS");
|
||||
features.EsprytUnlocatedIoBlocks = QueryEnvQuirkOverride("MOBILEGL_ESPRYT_UNLOCATED_IO_BLOCKS");
|
||||
features.PointSizeDemotion = QueryEnvQuirkOverride("MOBILEGL_POINT_SIZE_DEMOTION");
|
||||
features.CoherentAsFlush = QueryEnvFlag("MOBILEGL_COHERENT_AS_FLUSH");
|
||||
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
||||
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
||||
features.EsprytDisableUboRing = QueryEnvFlag("MOBILEGL_ESPRYT_DISABLE_UBO_RING");
|
||||
features.EsprytDisableUnpackRing = QueryEnvFlag("MOBILEGL_ESPRYT_DISABLE_UNPACK_RING");
|
||||
features.EsprytDisableUploadRing = QueryEnvFlag("MOBILEGL_ESPRYT_DISABLE_UPLOAD_RING");
|
||||
features.EsprytDisableInvalidateFlush = QueryEnvFlag("MOBILEGL_ESPRYT_DISABLE_INVALIDATE_FLUSH");
|
||||
features.DisableLargeBufferAdoption = QueryEnvFlag("MOBILEGL_DISABLE_LARGE_BUFFER_ADOPTION");
|
||||
features.EsprytForceDepthStencilReadbackEmulation =
|
||||
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||
features.MagmaDisableBlendedDepthWriteQuirk =
|
||||
QueryEnvQuirkOverride("MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE");
|
||||
features.DisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS");
|
||||
features.MagmaDisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_MAGMA_DISABLE_ROBUST_BUFFER_ACCESS");
|
||||
features.MagmaMultiDrawMode = QueryEnvMultiDrawMode("MOBILEGL_MAGMA_MULTIDRAW_MODE");
|
||||
features.EsprytMultiDrawMode = QueryEnvGLESMultiDrawMode("MOBILEGL_ESPRYT_MULTIDRAW_MODE");
|
||||
features.AsyncShaderCompile = QueryEnvQuirkOverride("MOBILEGL_ASYNC_SHADER_COMPILE");
|
||||
@@ -196,8 +240,45 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
features.AsyncOptimisticShaderStatus =
|
||||
QueryEnvQuirkOverride("MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS");
|
||||
features.ShaderTranslationCache = QueryEnvQuirkOverride("MOBILEGL_SHADER_CACHE");
|
||||
features.ViewportArrayEmulation =
|
||||
QueryEnvQuirkOverride("MOBILEGL_FORCE_VIEWPORT_ARRAY_EMULATION");
|
||||
features.EsprytViewportArrayEmulation =
|
||||
QueryEnvQuirkOverride("MOBILEGL_ESPRYT_FORCE_VIEWPORT_ARRAY_EMULATION");
|
||||
features.EsprytWidenPacked16Storage =
|
||||
QueryEnvQuirkOverride("MOBILEGL_ESPRYT_WIDEN_PACKED16_STORAGE");
|
||||
features.MagmaPrimGenQueryReroute = QueryEnvQuirkOverride("MOBILEGL_MAGMA_PRIMGEN_QUERY_REROUTE");
|
||||
// MGPipe. Nothing here needs adding to an allow-list: InitializeAcceptedEnvVariables
|
||||
// accepts every MOBILEGL_ / LIBGL_ prefixed variable in the environment, so a name
|
||||
// that starts with MOBILEGL_ is visible to these queries by construction.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// A push build with the knob unset runs every subsystem migrated so far, so the
|
||||
// shipped path is the one the gates measure; MOBILEGL_PIPE_PUSH=0 in the
|
||||
// environment is the all-subsystems-pull control that reproduces P1 exactly, and
|
||||
// kMGPipeSubsystemsMigratedAtP3a (0x1ff) is the phase-by-phase control - P4a's four
|
||||
// subsystems off, everything P3a landed still on.
|
||||
features.PipePush = QueryEnvUint64("MOBILEGL_PIPE_PUSH", MG_Pipe::kMGPipeSubsystemsMigratedAtP4a);
|
||||
#else
|
||||
// Meaningless in a pull build: there is nothing to push. Config.h documents 0 as
|
||||
// "pull everything" and that stays literally true.
|
||||
features.PipePush = QueryEnvUint64("MOBILEGL_PIPE_PUSH", 0);
|
||||
#endif
|
||||
features.PipeVerify = QueryEnvFlag("MOBILEGL_PIPE_VERIFY");
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// Defaults ON: read as a tri-state so only an explicitly falsy value turns it off.
|
||||
features.PipeVerifyFatal =
|
||||
QueryEnvQuirkOverride("MOBILEGL_PIPE_VERIFY_FATAL") != MG_Config::QuirkOverride::ForceOff;
|
||||
QueryEnvVariable("MOBILEGL_PIPE_VERIFY_CORRUPT", features.PipeVerifyCorrupt, "");
|
||||
QueryEnvVariable("MOBILEGL_PIPE_POISON_OMIT", features.PipePoisonOmit, "");
|
||||
features.PipeHandleAbaControl = QueryEnvFlag("MOBILEGL_PIPE_HANDLE_ABA_CONTROL");
|
||||
#endif
|
||||
features.PipeStats = QueryEnvFlag("MOBILEGL_PIPE_STATS");
|
||||
// Defaults ON, so the flag has to be read as a tri-state rather than as a plain
|
||||
// truthy check: unset must keep the memos, and only an explicitly falsy value may
|
||||
// drop them.
|
||||
features.PipeLegacyMemos =
|
||||
QueryEnvQuirkOverride("MOBILEGL_PIPE_LEGACY_MEMOS") != MG_Config::QuirkOverride::ForceOff;
|
||||
features.PipeTexelRetainMb = QueryEnvUint32("MOBILEGL_PIPE_TEXEL_RETAIN_MB", 0, 0, 4096);
|
||||
features.PipeIndexMirrorMb = QueryEnvUint32("MOBILEGL_PIPE_INDEX_MIRROR_MB", 64, 0, 4096);
|
||||
features.PipeStatsPeriod = QueryEnvUint32("MOBILEGL_PIPE_STATS_PERIOD", 120, 1, 1000000);
|
||||
QueryEnvVariable("MOBILEGL_PIPE_STATS_FILE", features.PipeStatsFile, "");
|
||||
}
|
||||
|
||||
inline void InitBackendType() {
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
||||
#include <MG_Impl/GLImpl/Query/GL_Query.h>
|
||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
#include <MG_State/GLState/ProgramState/ProgramTranslationCache.h>
|
||||
#include <MG_Util/ShaderTranspiler/TranslationCache.h>
|
||||
@@ -42,6 +43,11 @@ namespace MobileGL {
|
||||
if (logLifecycle) {
|
||||
MGLOG_I("MobileGL closing...");
|
||||
}
|
||||
// Before any subsystem the counters name goes away, and before the last frame's
|
||||
// numbers can be lost: emits the final summary line and, when
|
||||
// MOBILEGL_PIPE_STATS_FILE is set, the JSON dump. A no-op when the counters are
|
||||
// off, and idempotent.
|
||||
MG_Util::PipeStats::Shutdown();
|
||||
// First, before anything else is torn down. In-flight compile/link jobs own
|
||||
// their own inputs and are safe against everything below EXCEPT glslang's
|
||||
// process globals and the TShader/TProgram objects hanging off pGLContext,
|
||||
@@ -102,6 +108,10 @@ namespace MobileGL {
|
||||
MGLOG_I("Initializing MobileGL...");
|
||||
MG_ConfigLoader::Init();
|
||||
MGLOG_I("Config loaded");
|
||||
// Immediately after the config load and before anything can count: the MGPipe
|
||||
// boundary counters latch their enable flag here, so every counting site in the
|
||||
// two backends is a load of an already-settled global for the rest of the run.
|
||||
MG_Util::PipeStats::Init();
|
||||
MG_State::Init();
|
||||
MGLOG_D("MG_State initialized");
|
||||
MG_Backend::Init();
|
||||
|
||||
@@ -192,9 +192,23 @@ namespace MobileGL {
|
||||
void (*MemoryBarrierByRegion)(GLbitfield barriers);
|
||||
void (*BindImageTexture)(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer,
|
||||
GLenum access, GLenum format);
|
||||
// The ONLY indexed query that is genuinely a backend one, and only for the pnames
|
||||
// MG_Impl/GLImpl/Getter/GL_Getter.cpp does not already own. Every indexed pname that
|
||||
// names FRONTEND state - the indexed buffer bindings, the per-unit texture/sampler
|
||||
// bindings, the image-unit bindings, the viewport rectangles, the indexed capabilities
|
||||
// - is answered in GL_Getter::GetIntegeri_v and never reaches this entry; the
|
||||
// 64-bit and float/double widths are derived there from the same answer, which is why
|
||||
// no GetInteger64i_v/GetFloati_v/GetDoublei_v table entry exists. In practice this
|
||||
// leaves GL_MAX_COMPUTE_WORK_GROUP_COUNT / _SIZE (also asked directly by
|
||||
// MG_Util/ShaderTranspiler/CompileEnv.cpp) plus whatever pname the frontend has no
|
||||
// case for at all.
|
||||
void (*GetIntegeri_v)(GLenum target, GLuint index, GLint* data);
|
||||
void (*GetInteger64i_v)(GLenum target, GLuint index, GLint64* data);
|
||||
void (*GetProgramiv)(GLuint program, GLenum pname, GLint* params);
|
||||
// There is deliberately NO GetProgramiv entry: glGetProgramiv describes the program
|
||||
// the APPLICATION wrote - link status, the transform-feedback mode, the compute local
|
||||
// size - all of which are frontend link artifacts on ProgramObject, and
|
||||
// MG_Impl/GLImpl/Program/GL_Program.cpp answers every one of them from there. Asking a
|
||||
// backend would mean asking about a DIFFERENT program (a SPIRV-Cross-generated ESSL
|
||||
// one, or a SPIR-V module), in a namespace the application never sees.
|
||||
// The GL program interface (glGetProgramInterfaceiv / glGetProgramResource*) is NOT
|
||||
// a backend query: it describes the program the application wrote, in the
|
||||
// application's namespace, which neither backend program is in. It is answered
|
||||
@@ -301,6 +315,12 @@ namespace MobileGL {
|
||||
|
||||
struct DynamicBackendParameters {
|
||||
SizeT UniformBufferOffsetAlignment = 256;
|
||||
// GL_SHADER_STORAGE_BUFFER_OFFSET_ALIGNMENT, which is a SEPARATE limit from the
|
||||
// uniform one and is routinely larger: Adreno 830 reports 32 for uniform buffers and
|
||||
// 64 for storage buffers. Answering the storage query with the uniform value let an
|
||||
// application bind a storage range at an offset the driver cannot address, which it
|
||||
// accepted without error and then wrote somewhere else entirely.
|
||||
SizeT ShaderStorageBufferOffsetAlignment = 256;
|
||||
// GL_MAX_TEXTURE_MAX_ANISOTROPY_EXT. 1.0 means the backend cannot filter anisotropically,
|
||||
// which is also why the extension is not advertised in that case.
|
||||
Float MaxTextureMaxAnisotropy = 1.0f;
|
||||
@@ -358,6 +378,19 @@ namespace MobileGL {
|
||||
Int MaxFragmentShaderStorageBlocks = 8;
|
||||
Int MaxComputeUniformBlocks = 12;
|
||||
Int MaxComputeWorkGroupInvocations = 128;
|
||||
// GL_MAX_COMPUTE_WORK_GROUP_COUNT / GL_MAX_COMPUTE_WORK_GROUP_SIZE, one value per
|
||||
// axis. These six, with the invocations limit above, are the only indexed limits a
|
||||
// backend genuinely OWNS - the device answers them (glGetIntegeri_v on DirectGLES,
|
||||
// VkPhysicalDeviceLimits::maxComputeWorkGroupCount/Size on DirectVulkan) - and so
|
||||
// the only ones that survive the retirement of the GetIntegeri_v table entry: they
|
||||
// cross the MGPipe boundary inside MGPCaps, by inclusion of this struct (plan B
|
||||
// section 4.4.1). Every other indexed pname names frontend state. RAW driver
|
||||
// answers, like the invocations limit: GL_Getter and the compile environment floor
|
||||
// them at the shared MIN_COMPUTE_WORK_GROUP_* minimums themselves. The defaults are
|
||||
// the GL 4.3 core minimums (table 23.60) and describe the no-backend case, as
|
||||
// MaxClipDistances' does.
|
||||
Int MaxComputeWorkGroupCount[3] = {65535, 65535, 65535};
|
||||
Int MaxComputeWorkGroupSize[3] = {1024, 1024, 64};
|
||||
Int MaxShaderStorageBufferBindings = 8;
|
||||
Int MaxTextureBufferSize = 65536;
|
||||
// GL_TEXTURE_BUFFER_OFFSET_ALIGNMENT; 1 means the offset is unconstrained.
|
||||
@@ -383,6 +416,19 @@ namespace MobileGL {
|
||||
// where there is no device to be honest about and BuildTBuiltInResource still has to
|
||||
// hand glslang a workable gl_MaxClipDistances.
|
||||
Int MaxClipDistances = 8;
|
||||
// GL_MAX_CULL_DISTANCES and GL_MAX_COMBINED_CLIP_AND_CULL_DISTANCES, under exactly
|
||||
// the contract stated for MaxClipDistances above: ZERO IS A LEGAL ANSWER and a
|
||||
// backend that cannot host a cull distance MUST report it. The failure this prevents
|
||||
// is worse than the clip one, because cull distance discards the whole primitive:
|
||||
// glslang bounds gl_CullDistance[i] against maxCullDistances and expands
|
||||
// gl_MaxCullDistances from it, SPIRV-Cross then emits
|
||||
// `#extension GL_EXT_clip_cull_distance : require` into the ESSL, and a host driver
|
||||
// without that extension rejects the program in an info log nobody surfaces. These
|
||||
// used to be bare 8s inside BuildTBuiltInResource with no backend consulted at all.
|
||||
// The DEFAULTS are the GL 4.5 core minimums for the same reason MaxClipDistances'
|
||||
// is: they describe the no-backend case (standalone compiles, unit tests).
|
||||
Int MaxCullDistances = 8;
|
||||
Int MaxCombinedClipAndCullDistances = 8;
|
||||
Int MaxViewports = 16;
|
||||
// GL_LAYER_PROVOKING_VERTEX / GL_VIEWPORT_INDEX_PROVOKING_VERTEX: which vertex of a
|
||||
// primitive supplies gl_Layer and gl_ViewportIndex. GL 4.6 table 23.65 makes
|
||||
@@ -476,6 +522,24 @@ namespace MobileGL {
|
||||
// halves (PackDoubleVertexInputsPass and VertexInputStateFactory::ToVkVertexFormat)
|
||||
// still see one consistent world.
|
||||
Bool SupportsFloat64VertexAttributes = false;
|
||||
// Whether a TESSELLATION stage of this backend may access gl_PointSize - i.e.
|
||||
// whether a module declaring OpCapability TessellationPointSize can reach the
|
||||
// driver at all. DirectVulkan sets both this and the geometry twin from the one
|
||||
// shaderTessellationAndGeometryPointSize feature; DirectGLES sets them
|
||||
// independently from the EXT/OES_tessellation_point_size /
|
||||
// geometry_point_size extension pairs (PointSizeTier), which really do come
|
||||
// separately. When absent, ProgramSpirvTask demotes the built-in to an ordinary
|
||||
// varying program-wide (ShaderCompiler::
|
||||
// DemoteTessellationGeometryPointSizeForProgram); MOBILEGL_POINT_SIZE_DEMOTION
|
||||
// overrides the detection in either direction at backend init.
|
||||
//
|
||||
// Defaults TRUE, deliberately against the house "assume absent" rule: false
|
||||
// ARMS a rewrite, so the conservative no-backend answer (standalone compiles,
|
||||
// unit tests) is the one that leaves modules untouched. A backend that never
|
||||
// sets it gets standard modules and, at worst, the old honest declines.
|
||||
Bool SupportsTessellationPointSize = true;
|
||||
// The geometry-stage twin (OpCapability GeometryPointSize).
|
||||
Bool SupportsGeometryPointSize = true;
|
||||
SizeT MaxShaderStorageBlockSize = 128 * 1024 * 1024;
|
||||
Uint32 SubgroupSize = 0;
|
||||
Uint32 SubgroupSupportedStages = 0;
|
||||
|
||||
@@ -307,6 +307,23 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return capabilities.MaxColorTextureSamples;
|
||||
}
|
||||
|
||||
// The RENDERBUFFER twin, and it is a different set of pnames on purpose.
|
||||
// GL_MAX_{COLOR,DEPTH}_TEXTURE_SAMPLES bound multisample TEXTURES; a renderbuffer is
|
||||
// bounded by GL_MAX_SAMPLES (GL 4.6 core 9.2.4), with GL_MAX_INTEGER_SAMPLES for the
|
||||
// integer formats. Using the texture ceilings here - which is what the renderbuffer probe
|
||||
// did - is not merely untidy: the two texture pnames are ES 3.1 state, so on an ES 3.0
|
||||
// context the loader's rejected-probe clamp leaves them at 1 (see the multisample clamps
|
||||
// in the GLES loader) and the walk below would never run past one sample, recording {1}
|
||||
// for EVERY colour format while GL_MAX_SAMPLES - ES 3.0 core, so genuinely answered -
|
||||
// reports 4. Once the frontend validates against this list, that would reject every
|
||||
// multisample renderbuffer on such a context.
|
||||
Int GetGLESRenderbufferFormatMaxSamples(const MG_External::GLESCapabilities& capabilities,
|
||||
GLenum imageFormat) {
|
||||
const Bool isInteger = imageFormat == GL_RED_INTEGER || imageFormat == GL_RG_INTEGER ||
|
||||
imageFormat == GL_RGB_INTEGER || imageFormat == GL_RGBA_INTEGER;
|
||||
return isInteger ? capabilities.MaxIntegerSamples : capabilities.MaxSamples;
|
||||
}
|
||||
|
||||
Bool ProbeFramebufferCompletenessForTexture(const MG_External::GLESFunctionsTable& gl, TextureTarget target,
|
||||
GLuint texture, TextureInternalFormat format) {
|
||||
GLuint framebuffer = 0;
|
||||
@@ -717,7 +734,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
AddFullFormatCaps(cache, renderbufferTargetIndex, formatIndex,
|
||||
GetRenderbufferFeatureCaps(logicalFormat));
|
||||
const Int maxSamples =
|
||||
GetGLESFormatMaxSamples(capabilities, logicalFormat, nativeInfo.ImageFormat);
|
||||
GetGLESRenderbufferFormatMaxSamples(capabilities, nativeInfo.ImageFormat);
|
||||
cache.SampleCounts[renderbufferTargetIndex][formatIndex] =
|
||||
ProbeRenderbufferSampleCounts(gl, nativeInfo.InternalFormat, logicalFormat, maxSamples);
|
||||
} else {
|
||||
@@ -731,7 +748,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
LogGLESFormatCaveat(logicalFormat, renderbufferTargetIndex, renderbufferFallbackInfo);
|
||||
}
|
||||
const Int maxSamples =
|
||||
GetGLESFormatMaxSamples(capabilities, logicalFormat, renderbufferFallbackInfo.ImageFormat);
|
||||
GetGLESRenderbufferFormatMaxSamples(capabilities, renderbufferFallbackInfo.ImageFormat);
|
||||
cache.SampleCounts[renderbufferTargetIndex][formatIndex] = ProbeRenderbufferSampleCounts(
|
||||
gl, renderbufferFallbackInfo.InternalFormat, logicalFormat, maxSamples);
|
||||
}
|
||||
@@ -749,7 +766,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
.ExtraVendor = Nullopt, // Extra vendor
|
||||
.RendererGLInfo =
|
||||
{
|
||||
.TargetGLVersion = {4, 3, 0}, // GL target version
|
||||
.TargetGLVersion = {4, 6, 0}, // GL target version
|
||||
.TargetGLSLVersion = {4, 6, 0}, // Target Shading Language Version
|
||||
// Baseline advertisement (no runtime capabilities yet); reconciled once
|
||||
// the ES capabilities exist, see UpdateAdvertisedCapabilityExtensions.
|
||||
@@ -995,10 +1012,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Bool textureViewSupported, Bool cubeMapArraySupported) {
|
||||
Vector<GLExtension> extensions = {
|
||||
// The version tokens have to reach the version the backend actually claims:
|
||||
// TargetGLVersion is {4,3,0}, and a list that stopped at OpenGL40 told an
|
||||
// TargetGLVersion is {4,6,0}, and a list that stopped at OpenGL40 told an
|
||||
// application feature-detecting off these tokens the opposite of what
|
||||
// GL_MAJOR_VERSION / GL_MINOR_VERSION told it.
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, V_OpenGL41, V_OpenGL42, V_OpenGL43,
|
||||
V_OpenGL44, V_OpenGL45, V_OpenGL46,
|
||||
E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
|
||||
@@ -1180,8 +1198,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
//
|
||||
// Until that reconciliation exists, advertising here would be the same lie the comment
|
||||
// above refuses to tell, just with an extra prerequisite met. Set
|
||||
// MOBILEGL_ENABLE_GLES_TEXTURE_VIEW=1 to re-enable it for that work.
|
||||
if (textureViewSupported && MG_Config::Features.EnableGlesTextureView) {
|
||||
// MOBILEGL_ESPRYT_ENABLE_TEXTURE_VIEW=1 to re-enable it for that work.
|
||||
if (textureViewSupported && MG_Config::Features.EsprytEnableTextureView) {
|
||||
extensions.push_back(E_GL_ARB_texture_view);
|
||||
}
|
||||
// Only advertised when the host ES driver actually filters anisotropically: the sampler
|
||||
@@ -1237,8 +1255,6 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
funcsTable.GL.MemoryBarrierByRegion = MemoryBarrierByRegion;
|
||||
funcsTable.GL.BindImageTexture = BindImageTexture;
|
||||
funcsTable.GL.GetIntegeri_v = GetIntegeri_v;
|
||||
funcsTable.GL.GetInteger64i_v = GetInteger64i_v;
|
||||
funcsTable.GL.GetProgramiv = GetProgramiv;
|
||||
funcsTable.GL.ShaderStorageBlockBinding = ShaderStorageBlockBinding;
|
||||
funcsTable.GL.Clear = Clear;
|
||||
funcsTable.GL.ClearBufferfi = ClearBufferfi;
|
||||
@@ -1323,6 +1339,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
void BackendObject_DirectGLES::UpdateDynamicBackendParameters() {
|
||||
m_dynamicParameters.UniformBufferOffsetAlignment = m_GLESCapabilities.UniformBufferOffsetAlignment;
|
||||
m_dynamicParameters.ShaderStorageBufferOffsetAlignment =
|
||||
m_GLESCapabilities.ShaderStorageBufferOffsetAlignment;
|
||||
m_dynamicParameters.MaxTextureMaxAnisotropy = m_GLESCapabilities.MaxTextureMaxAnisotropy;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMin = m_GLESCapabilities.AliasedLineWidthRangeMin;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMax = m_GLESCapabilities.AliasedLineWidthRangeMax;
|
||||
@@ -1397,6 +1415,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
clampStageStorageBlocks(m_GLESCapabilities.MaxFragmentShaderStorageBlocks);
|
||||
m_dynamicParameters.MaxComputeUniformBlocks = m_GLESCapabilities.MaxComputeUniformBlocks;
|
||||
m_dynamicParameters.MaxComputeWorkGroupInvocations = m_GLESCapabilities.MaxComputeWorkGroupInvocations;
|
||||
// The six per-axis compute limits: the driver's raw glGetIntegeri_v answers, the same
|
||||
// numbers GLFunctionsTable::GetIntegeri_v forwards live. Carried here so that MGPCaps has
|
||||
// them once the table entry retires (plan B section 4.4.1); GL_Getter floors them.
|
||||
for (SizeT axis = 0; axis < 3; ++axis) {
|
||||
m_dynamicParameters.MaxComputeWorkGroupCount[axis] = m_GLESCapabilities.MaxComputeWorkGroupCount[axis];
|
||||
m_dynamicParameters.MaxComputeWorkGroupSize[axis] = m_GLESCapabilities.MaxComputeWorkGroupSize[axis];
|
||||
}
|
||||
// (MaxShaderStorageBufferBindings is assigned above, before the per-stage clamp reads it.)
|
||||
// This is the number glGetIntegerv(GL_MAX_TEXTURE_BUFFER_SIZE) hands the application, and
|
||||
// on a host without buffer textures it is knowingly a floor MobileGL cannot honour rather
|
||||
@@ -1459,9 +1484,44 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Follows the line above, and must: OpenGL ES has no double-precision vertex format and no
|
||||
// fp64 type to consume one with, so a 64-bit vertex attribute has nowhere to land here.
|
||||
m_dynamicParameters.SupportsFloat64VertexAttributes = false;
|
||||
// Whether a tessellation / geometry stage's ESSL may name gl_PointSize at all: the two
|
||||
// extension pairs the loader probed, independently, because they really do come
|
||||
// separately. False arms the shared phase-B demotion
|
||||
// (ShaderCompiler::DemoteTessellationGeometryPointSizeForProgram), whose ESSL then
|
||||
// never names the built-in in those stages and needs no extension.
|
||||
// MOBILEGL_POINT_SIZE_DEMOTION=1 pretends both are absent so the demotion can be
|
||||
// exercised on a healthy driver (the pinned integration lane); =0 restores the
|
||||
// detected answer's declines.
|
||||
m_dynamicParameters.SupportsTessellationPointSize =
|
||||
m_GLESCapabilities.TessellationPointSizeSupport !=
|
||||
MG_External::GLESCapabilities::PointSizeTier::None;
|
||||
m_dynamicParameters.SupportsGeometryPointSize =
|
||||
m_GLESCapabilities.GeometryPointSizeSupport !=
|
||||
MG_External::GLESCapabilities::PointSizeTier::None;
|
||||
switch (MG_Config::Features.PointSizeDemotion) {
|
||||
case MG_Config::QuirkOverride::ForceOn:
|
||||
MGLOG_I("DirectGLES: MOBILEGL_POINT_SIZE_DEMOTION=1 - treating tessellation/geometry "
|
||||
"gl_PointSize as unhosted so the demotion runs on this driver");
|
||||
m_dynamicParameters.SupportsTessellationPointSize = false;
|
||||
m_dynamicParameters.SupportsGeometryPointSize = false;
|
||||
break;
|
||||
case MG_Config::QuirkOverride::ForceOff:
|
||||
MGLOG_I("DirectGLES: MOBILEGL_POINT_SIZE_DEMOTION=0 - keeping the built-in and the "
|
||||
"plain declines regardless of the driver's extensions");
|
||||
m_dynamicParameters.SupportsTessellationPointSize = true;
|
||||
m_dynamicParameters.SupportsGeometryPointSize = true;
|
||||
break;
|
||||
case MG_Config::QuirkOverride::Auto:
|
||||
break;
|
||||
}
|
||||
m_dynamicParameters.MaxDrawBuffers = m_GLESCapabilities.MaxDrawBuffers;
|
||||
m_dynamicParameters.MaxColorAttachments = m_GLESCapabilities.MaxColorAttachments;
|
||||
m_dynamicParameters.MaxClipDistances = m_GLESCapabilities.MaxClipDistances;
|
||||
// The loader already gated both on GL_EXT_clip_cull_distance and left 0 without it, which
|
||||
// is the answer that keeps glslang from accepting a gl_CullDistance the ESSL compiler
|
||||
// would reject.
|
||||
m_dynamicParameters.MaxCullDistances = m_GLESCapabilities.MaxCullDistances;
|
||||
m_dynamicParameters.MaxCombinedClipAndCullDistances = m_GLESCapabilities.MaxCombinedClipAndCullDistances;
|
||||
m_dynamicParameters.MaxViewports = m_GLESCapabilities.MaxViewports;
|
||||
// Whatever the driver said about which vertex supplies gl_Layer, and GL_UNDEFINED_VERTEX
|
||||
// for gl_ViewportIndex on every driver without GL_OES_viewport_array - which is both test
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -92,8 +92,6 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void BindImageTexture(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer, GLenum access,
|
||||
GLenum format);
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data);
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
void GetProgramiv(GLuint program, GLenum pname, GLint* params);
|
||||
void ShaderStorageBlockBinding(GLuint program, const GLchar* storageBlockName, GLuint storageBlockBinding);
|
||||
Bool InitWindowSurface(NativeWindowType window);
|
||||
Bool InitPbufferSurface(EGLint width, EGLint height);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -9,6 +9,8 @@
|
||||
#include "MultiDraw.h"
|
||||
#include "Managers.h"
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
|
||||
@@ -29,21 +31,26 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// The all-ones value of an index type, which is what GL restarts on once
|
||||
// primitive restart is in play. CheckPrimitiveRestartSupported has already
|
||||
// rejected the arbitrary-index form of GL_PRIMITIVE_RESTART, so an enabled
|
||||
// restart always restarts here and nowhere else.
|
||||
// The index value this batch restarts on, compared at 32 bits against the zero-extended
|
||||
// source index. Normally the all-ones value of the source type, which is what
|
||||
// GL_PRIMITIVE_RESTART_FIXED_INDEX and GLES both restart on; with desktop
|
||||
// GL_PRIMITIVE_RESTART it is instead whatever glPrimitiveRestartIndex named. The rebased
|
||||
// tier turns whichever it is into 0xFFFFFFFF in its widened stream, which is what the
|
||||
// driver restarts on.
|
||||
//
|
||||
// No truncation, deliberately, and the same rule ResolveRestartSubstitution applies: a
|
||||
// restart index the source type cannot hold simply matches nothing, so returning it
|
||||
// verbatim is already "this batch restarts nowhere".
|
||||
Uint32 RestartSentinelFor(GLenum type) {
|
||||
switch (type) {
|
||||
case GL_UNSIGNED_BYTE: return 0xFFu;
|
||||
case GL_UNSIGNED_SHORT: return 0xFFFFu;
|
||||
default: return 0xFFFFFFFFu;
|
||||
if (ResolveRestartSubstitution(type) != RestartSubstitutionKind::None) {
|
||||
return MGB_CTX->GetPrimitiveRestartIndex();
|
||||
}
|
||||
return MG_Util::FixedRestartIndexForGLType(type);
|
||||
}
|
||||
|
||||
Bool RestartActive() {
|
||||
return MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::PrimitiveRestart) ||
|
||||
MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::PrimitiveRestartFixedIndex);
|
||||
return MGB_CTX->IsCapabilityEnabled(CapabilityInput::PrimitiveRestart) ||
|
||||
MGB_CTX->IsCapabilityEnabled(CapabilityInput::PrimitiveRestartFixedIndex);
|
||||
}
|
||||
|
||||
// Vertices per primitive for the modes whose sub-draws may be concatenated into a
|
||||
@@ -78,7 +85,7 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
|
||||
Uint BoundDrawIndirectBufferId() {
|
||||
const auto& indirect =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
MGB_CTX->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (!indirect) return 0;
|
||||
const auto* resource = BufferImpl::EnsureBufferResource(indirect);
|
||||
return resource ? resource->id : 0;
|
||||
@@ -86,7 +93,7 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
|
||||
const SharedPtr<MG_State::GLState::BufferObject>& BoundIndexBuffer() {
|
||||
static const SharedPtr<MG_State::GLState::BufferObject> none;
|
||||
const auto& vao = MG_State::pGLContext->GetBoundVertexArray();
|
||||
const auto& vao = MGB_CTX->GetBoundVertexArray();
|
||||
if (!vao) return none;
|
||||
return vao->GetIndexBufferBindingSlot().GetBoundObject();
|
||||
}
|
||||
@@ -151,7 +158,10 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// are bound as storage blocks. Respecifies rather than sub-updates: glBufferData
|
||||
// orphans the previous store, so the upload never waits on a dispatch still reading
|
||||
// the old contents out of the same name.
|
||||
Bool UploadScratch(ScratchBuffer& buffer, SizeT bytes, const void* data) {
|
||||
// statsClass: which MGPipe byte population these bytes belong to. Counted here
|
||||
// rather than at the four call sites so a new tier cannot forget it.
|
||||
Bool UploadScratch(ScratchBuffer& buffer, SizeT bytes, const void* data,
|
||||
MG_Util::PipeStats::ByteClass statsClass) {
|
||||
if (bytes == 0) return true;
|
||||
if (!EnsureScratchName(buffer)) return false;
|
||||
BufferImpl::BindBufferId(BufferImpl::TempBufferTarget, buffer.id);
|
||||
@@ -164,6 +174,9 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
buffer.cursor = 0;
|
||||
if (data) {
|
||||
g_GLESFuncs.glBufferSubData(BufferImpl::TempBufferTarget, 0, static_cast<GLsizeiptr>(bytes), data);
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(statsClass, static_cast<Uint64>(bytes));
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -178,7 +191,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
constexpr SizeT kRingAlignment = 16; // >= 4, so both command and uint32-index offsets stay legal
|
||||
constexpr SizeT kMinRingBytes = 1u << 16;
|
||||
|
||||
Bool UploadScratchRing(ScratchBuffer& buffer, SizeT bytes, const void* data, SizeT& outOffset) {
|
||||
Bool UploadScratchRing(ScratchBuffer& buffer, SizeT bytes, const void* data,
|
||||
MG_Util::PipeStats::ByteClass statsClass, SizeT& outOffset) {
|
||||
outOffset = 0;
|
||||
if (bytes == 0) return true;
|
||||
if (!EnsureScratchName(buffer)) return false;
|
||||
@@ -202,6 +216,9 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
if (data) {
|
||||
g_GLESFuncs.glBufferSubData(BufferImpl::TempBufferTarget, static_cast<GLintptr>(outOffset),
|
||||
static_cast<GLsizeiptr>(bytes), data);
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(statsClass, static_cast<Uint64>(bytes));
|
||||
}
|
||||
}
|
||||
buffer.cursor += aligned;
|
||||
return true;
|
||||
@@ -275,10 +292,20 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// its remaining feasibility checks inside its implementation, where the data it
|
||||
// has to walk is already in hand.
|
||||
GLESMultiDrawMode ResolveTierForBatch(Bool programReadsDrawID, Bool perSubDrawBaseVertex,
|
||||
Bool hasIndexBuffer) {
|
||||
Bool hasIndexBuffer, Bool arbitraryRestart) {
|
||||
ResolveTierOnce();
|
||||
GLESMultiDrawMode tier = g_resolvedTier;
|
||||
|
||||
// Desktop GL_PRIMITIVE_RESTART restarts on an application-chosen index; the driver
|
||||
// only ever restarts on the all-ones value. Every tier but the rebased one hands
|
||||
// the application's own index data to the driver, which would then see no restarts
|
||||
// at all and weld the primitives together. The rebased tier is the one that
|
||||
// REWRITES the stream, and RestartSentinelFor already tells it which value to
|
||||
// translate, so it is the only tier this batch can take.
|
||||
if (arbitraryRestart) {
|
||||
return GLESMultiDrawMode::DrawElements;
|
||||
}
|
||||
|
||||
// Batched tiers issue one driver entry for the whole batch, so the emulated
|
||||
// gl_DrawID uniform can only hold one value across every sub-draw. A program
|
||||
// that reads gl_DrawID gets an unrolled tier, which feeds each sub-draw its
|
||||
@@ -402,7 +429,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
|
||||
const SizeT commandBytes = g_commandStaging.size() * sizeof(DrawElementsIndirectCommand);
|
||||
SizeT commandBase = 0;
|
||||
if (!UploadScratchRing(g_indirectCommands, commandBytes, g_commandStaging.data(), commandBase)) {
|
||||
if (!UploadScratchRing(g_indirectCommands, commandBytes, g_commandStaging.data(),
|
||||
MG_Util::PipeStats::ByteClass::StageIndirectCmd, commandBase)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -488,6 +516,16 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
|
||||
const Bool restartActive = RestartActive();
|
||||
const Uint32 restartSentinel = RestartSentinelFor(type);
|
||||
// Widening to GL_UNSIGNED_INT gives a UBYTE/USHORT source a sentinel it can never
|
||||
// spell, so those batches are lossless. A UINT source that already uses 0xFFFFFFFF as
|
||||
// a real vertex index while restarting on a different one is the one shape 32 bits
|
||||
// cannot express - the same corner the single-draw substitution reports.
|
||||
if (restartActive && indexSize == 4 && restartSentinel != 0xFFFFFFFFu) {
|
||||
MGLOG_E_ONCE("GL_PRIMITIVE_RESTART with restart index %u over GL_UNSIGNED_INT multi-draw indices: "
|
||||
"any index that is already 0xFFFFFFFF will restart too, because the rewritten stream "
|
||||
"has no wider sentinel to move to.",
|
||||
restartSentinel);
|
||||
}
|
||||
g_indexStaging.resize(total);
|
||||
SizeT cursor = 0;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
@@ -507,7 +545,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
}
|
||||
|
||||
SizeT indexBase = 0;
|
||||
if (!UploadScratchRing(g_rebasedIndices, total * sizeof(Uint32), g_indexStaging.data(), indexBase)) {
|
||||
if (!UploadScratchRing(g_rebasedIndices, total * sizeof(Uint32), g_indexStaging.data(),
|
||||
MG_Util::PipeStats::ByteClass::StageIndexClient, indexBase)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -712,10 +751,16 @@ void main() {
|
||||
if (total == 0) return; // nothing to draw; the ordinary tiers no-op just as well
|
||||
|
||||
if (!EnsureComputeProgram()) return;
|
||||
if (!UploadScratch(g_drawInfo, g_drawInfoStaging.size() * sizeof(Uint32), g_drawInfoStaging.data())) {
|
||||
if (!UploadScratch(g_drawInfo, g_drawInfoStaging.size() * sizeof(Uint32), g_drawInfoStaging.data(),
|
||||
MG_Util::PipeStats::ByteClass::StageIndirectCmd)) {
|
||||
return;
|
||||
}
|
||||
// data == nullptr: pure respecify, the compute pass writes the contents, so no
|
||||
// host bytes cross here and nothing is counted.
|
||||
if (!UploadScratch(g_flattenedIndices, total * sizeof(Uint32), nullptr,
|
||||
MG_Util::PipeStats::ByteClass::StageIndexClient)) {
|
||||
return;
|
||||
}
|
||||
if (!UploadScratch(g_flattenedIndices, total * sizeof(Uint32), nullptr)) return;
|
||||
|
||||
BufferImpl::BindBufferBaseCached(GL_SHADER_STORAGE_BUFFER, 0, sourceResource->id);
|
||||
BufferImpl::BindBufferBaseCached(GL_SHADER_STORAGE_BUFFER, 1, g_drawInfo.id);
|
||||
@@ -852,8 +897,14 @@ void main() {
|
||||
void DrawElementsBatch(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex) {
|
||||
if (drawcount <= 0 || !count || !indices) return;
|
||||
// State-independent and possibly throwing, so it runs before any GL work.
|
||||
CheckPrimitiveRestartSupported(type);
|
||||
// Read before any GL work, because it decides the tier below: a desktop restart index
|
||||
// the driver does not know about can only be honoured by the tier that rewrites the
|
||||
// index stream (see ResolveTierForBatch). A restart index this index type cannot hold
|
||||
// needs no rewrite at all - nothing can match it - but it does need the driver's own
|
||||
// fixed-index restart held off for the batch, which is what the scope below does.
|
||||
const RestartSubstitutionKind restartKind = ResolveRestartSubstitution(type);
|
||||
const Bool arbitraryRestart = restartKind == RestartSubstitutionKind::RewriteIndices;
|
||||
const ScopedSuppressedPrimitiveRestart restartCapOverride(restartKind);
|
||||
|
||||
const Bool hasIndexBuffer = BoundIndexBuffer() != nullptr;
|
||||
|
||||
@@ -889,7 +940,8 @@ void main() {
|
||||
// the tier choice and the per-sub-draw feeds use those, not the guess above.
|
||||
const Bool feedDrawID = CurrentProgramReadsDrawID();
|
||||
const Bool feedBaseVertex = basevertex != nullptr && CurrentProgramReadsBaseVertex();
|
||||
const GLESMultiDrawMode tier = ResolveTierForBatch(feedDrawID, feedBaseVertex, hasIndexBuffer);
|
||||
const GLESMultiDrawMode tier =
|
||||
ResolveTierForBatch(feedDrawID, feedBaseVertex, hasIndexBuffer, arbitraryRestart);
|
||||
|
||||
Bool drawn = false;
|
||||
switch (tier) {
|
||||
@@ -921,8 +973,10 @@ void main() {
|
||||
// Every tier above may decline a batch whose shape it cannot express. The two
|
||||
// below are the floor: a base-vertex replay where the driver has one, and the
|
||||
// rewritten index stream where it does not. Both are safe for any batch these
|
||||
// entry points can receive.
|
||||
if (!drawn) {
|
||||
// entry points can receive - except that the base-vertex replay hands the
|
||||
// application's own indices to the driver, which cannot restart on a desktop
|
||||
// restart index, so that batch has only the rewriting floor.
|
||||
if (!drawn && !arbitraryRestart) {
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
|
||||
@@ -0,0 +1,577 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectGLES/SlotTables.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
#include <MG_Pipe/MGPipeHandles.h>
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#endif
|
||||
|
||||
// Espryt 0b, the first Track H slice: the DENSE, {slot, gen}-keyed twin table that replaces
|
||||
// StateBackendObjectRegistry's UnorderedMap<StateObject*, Entry>.
|
||||
//
|
||||
// What changes, and why each of them is the point:
|
||||
//
|
||||
// * The KEY stops being a frontend heap address. It is MGPipeHandle{Slot, Gen}, minted by the
|
||||
// client's MGPipeSlotAllocator off the frontend object's GetLifetimeId(). A recycled heap
|
||||
// address cannot reproduce a handle, so the weak_ptr the registry carried per entry purely
|
||||
// to catch that (its Entry::stateRef, used as an IDENTITY test) stops being an identity
|
||||
// mechanism, and OwnerEquals / TwinLookupMemo x3 / UnitSamplerLookupMemo's owner compare all
|
||||
// lose their reason to exist.
|
||||
// * The lookup stops being a hash probe into an open-addressed map and becomes one bounds
|
||||
// check plus one array index, so a returned BackendPtr* is NOT invalidated by the next Find
|
||||
// on the table. That kills the hazard Managers.h documents at length, and with it the
|
||||
// by-value copy plus second Find that SyncTextureObjectToBackend paid to survive it.
|
||||
// * Slots are dense per kind, which is what lets the server side (ARCHITECTURE.md 10.1,
|
||||
// MG_Remote/Server/PipeObjectTables) be an array rather than an object graph.
|
||||
//
|
||||
// Death is ANNOUNCED, and that is what lets this table have no garbage collector - the
|
||||
// deliverable ROADMAP.md:18 spells "GC" in and the one D13 makes a precondition of the switch-
|
||||
// over. All six re-keyed object classes raise MG_State::GLState::NotifyStateObjectDestroyed()
|
||||
// from their destructor (BufferBackendOps' shape, one entry point for six kinds), the backend
|
||||
// consumes it in Managers.cpp, and OnFrontendObjectDestroyed() below drops the twin in EVERY
|
||||
// table of the kind and returns the slot, at the moment the frontend object's last SharedPtr
|
||||
// goes. So:
|
||||
// * there is NO draw-path tick, NO creation tick and NO sweep of any kind on this arm. The
|
||||
// seven CollectGarbageIfNeeded call sites in DirectGLES.cpp drive the LEGACY registry only;
|
||||
// * a twin, and the driver storage it owns, is freed when the application lets go of the
|
||||
// object rather than up to 64 creations or 1024 draw ticks later. That is what
|
||||
// Managers.h's "dead gigabytes" note asked for.
|
||||
//
|
||||
// EVERY HOLDER OF THE KIND, not one. Two live tables of one kind is a real configuration - the
|
||||
// ScopedDirectGLESTextureBindings fixture keeps a by-value copy of the Texture registry for the
|
||||
// length of a test, and a context reset does the same in reverse - and the slot allocator
|
||||
// erases its lifetimeId -> slot mapping on Free, so a notice delivered to one holder and
|
||||
// resolved again by the next would find nothing to resolve. Every table therefore links itself
|
||||
// into a per-table-type list at construction and out at destruction, and one notice resolves
|
||||
// the handle ONCE, drops the twin in each holder BY HANDLE, and frees the slot once, last. No
|
||||
// holder can be left naming a live entry for a dead object, and there is nothing a sweep could
|
||||
// still find. (The list is per table TYPE; the kind is the type's template parameter, and each
|
||||
// of the six kinds has exactly one table type in this backend. Magma's subsystem-4 table mints
|
||||
// out of its own per-renderer allocator, not MGPipeSlots(), so it is not a holder here.)
|
||||
//
|
||||
// The weak_ptr per entry survives for exactly one reason: ForEachLive() hands the callee a
|
||||
// STRONG reference to the frontend object, which the one direct-iteration site
|
||||
// (ScopedDetachedTextureFramebufferAttachments) needs. It is never an identity test - that is
|
||||
// what Gen is for - and it is never read to decide whether an entry is dead: a destructor that
|
||||
// runs after exit() has begun has its notice dropped by InProcessTeardown(), and that twin is
|
||||
// then a DELIBERATE leak (the process is exiting, the driver reclaims the object, and a twin
|
||||
// destructor must not call into a driver that may already be unloaded), not something to be
|
||||
// collected later.
|
||||
//
|
||||
// P3+ DEBT, recorded rather than hidden: this header is under MG_Backend/ and it MINTS
|
||||
// handles (MGPipeSlots().Acquire below) off a frontend SharedPtr's GetLifetimeId().
|
||||
// MGPipeHandles.h:13-16 says a handle is minted by the CLIENT and never by the server, and
|
||||
// under a real split neither the frontend object nor its lifetime id exists on this side of
|
||||
// the wire. This is monolith glue: the minting and the lifetimeId -> handle resolution both
|
||||
// belong on the client, and the backend should receive the handle in the verb payload. It is
|
||||
// NOT part of "Track H done" and check_include_closure.py does not probe MG_Backend headers,
|
||||
// so nothing catches it automatically.
|
||||
namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
|
||||
// Declared in Managers.h as well; repeated here because this header is included from it
|
||||
// before that declaration, and the table below is the arming site on this arm (D13: "the
|
||||
// arming site moves to the slot table's first insertion").
|
||||
void EnsureProcessTeardownSentinel();
|
||||
|
||||
// What the two knobs add up to. Split out as a PURE function of them so a test can drive
|
||||
// every combination without needing a process per combination.
|
||||
enum class EsprytSlotArmVerdict {
|
||||
Handles, // kMGPipeSubsystemEsprytSlots is set: the {slot, gen} tables run.
|
||||
Legacy, // the bit is clear and the legacy address-keyed registry is reachable.
|
||||
NoArm, // the bit is clear AND MOBILEGL_PIPE_LEGACY_MEMOS=0 made the legacy arm
|
||||
// unreachable, so the operator asked for a configuration with no arm at all.
|
||||
};
|
||||
|
||||
EsprytSlotArmVerdict ClassifyEsprytSlotArm(Bool subsystemBitSet, Bool legacyMemosEnabled);
|
||||
|
||||
// This process's verdict, read off MG_Config::Features. Latches nothing and stops nothing.
|
||||
EsprytSlotArmVerdict CurrentEsprytSlotArmVerdict();
|
||||
|
||||
// Says, at backend bring-up, that the knobs leave no arm - and does NOT stop.
|
||||
//
|
||||
// The stop cannot live here, and that is the whole point of the split. Backend context
|
||||
// creation runs inside eglMakeCurrent, and the integration harness pre-flights exactly that
|
||||
// sequence in a FORKED CHILD (MG_IntegrationTest/Harness/HeadlessGL.cpp): a child that dies
|
||||
// on a signal is reported as "no usable GPU/display/ICD" and every scenario in the lane is
|
||||
// SKIPPED - i.e. the lane goes green having run nothing, on the very pair of env vars the
|
||||
// D14/D18 A/B is driven with, which is what ROADMAP.md:7 forbids. So bring-up only
|
||||
// DIAGNOSES; the stop is raised by ResolveEsprytSlotTablesArm() at the first twin lookup,
|
||||
// which happens in the test body where the harness reports it as a failure.
|
||||
//
|
||||
// The CALL SITE (InitDisplayAndContext in DirectGLES.cpp) is pinned by
|
||||
// DirectGLESSlotTable.EglBringUpUnderTheArmlessKnobPairReturnsInsteadOfStopping, which runs
|
||||
// the real bring-up entry point under the pair in a forked child: edit that site back to
|
||||
// ResolveEsprytSlotTablesArm() and the case fails naming both knobs.
|
||||
void DiagnoseEsprytSlotArm();
|
||||
|
||||
// Reads the config, logs, installs the death-notice consumer, and STOPS when the operator
|
||||
// left no arm at all. Cold: called exactly once per process, from the latch below - i.e. at
|
||||
// the first twin lookup, which is the first moment an arm is actually needed. A process
|
||||
// that never twins anything needs no arm and is not stopped.
|
||||
Bool ResolveEsprytSlotTablesArm();
|
||||
|
||||
// True when this process runs the {slot, gen} arm. Fixed for the life of the process: the
|
||||
// two arms hold their twins in different containers, so flipping mid-run would strand them.
|
||||
//
|
||||
// INLINE on purpose. Every Find / GetOrCreate / HandleOf / ForEachLive on the twin tables
|
||||
// consults it, i.e. it is on the per-draw path several times per draw. As an out-of-line
|
||||
// function in Managers.cpp (no LTO in any shipped configuration) that was a call through
|
||||
// the PLT per lookup; here the caller sees a guard-variable load and a perfectly-predicted
|
||||
// branch, and the arm dispatch folds into the caller.
|
||||
inline Bool EsprytSlotTablesEnabled() {
|
||||
static const Bool enabled = ResolveEsprytSlotTablesArm();
|
||||
return enabled;
|
||||
}
|
||||
|
||||
template <typename StateObject, typename BackendObject, MG_Pipe::MGPipeKind kKind>
|
||||
class BackendSlotTable {
|
||||
public:
|
||||
using StatePtr = SharedPtr<StateObject>;
|
||||
using StateWeakPtr = std::weak_ptr<StateObject>;
|
||||
using BackendPtr = SharedPtr<BackendObject>;
|
||||
|
||||
// The largest slot index this table will grow to for a handle that ARRIVED in a call's
|
||||
// payload. Slots are dense and allocated per kind, so a million of one kind is already
|
||||
// far past any application's live object count; the cap is here because the alternative
|
||||
// is letting a corrupt 32-bit slot decide a vector resize. See GetOrCreate(MGPipeHandle).
|
||||
static constexpr Uint32 kMaxHandleSlot = 1u << 20;
|
||||
|
||||
struct Entry {
|
||||
BackendPtr backend;
|
||||
// LIVENESS ONLY, and only for ForEachLive(), which locks it so the callee holds a
|
||||
// strong ref. Never compared against another object to decide identity - that is
|
||||
// what Gen is for - never dereferenced for its address, and never read to decide
|
||||
// whether the slot is dead: death is announced, not discovered.
|
||||
StateWeakPtr stateRef;
|
||||
// The generation this entry's twin was built for. An entry whose Gen no longer
|
||||
// matches the allocator's is a twin of the slot's PREVIOUS owner.
|
||||
Uint32 Gen = 0;
|
||||
Bool Live = false;
|
||||
};
|
||||
|
||||
// Every constructor links the table into the per-type holder list and the destructor
|
||||
// unlinks it, so a by-value copy (the ScopedDirectGLESTextureBindings fixture's saved
|
||||
// registry) is a holder for exactly as long as it exists. Copy and move carry the
|
||||
// ENTRIES and the memo; the links are the table's own and are never copied.
|
||||
BackendSlotTable() { LinkHolder(); }
|
||||
BackendSlotTable(const BackendSlotTable& other):
|
||||
m_slots(other.m_slots),
|
||||
m_nullTwin(other.m_nullTwin),
|
||||
m_memoLifetimeId(other.m_memoLifetimeId),
|
||||
m_memoHandle(other.m_memoHandle) {
|
||||
LinkHolder();
|
||||
}
|
||||
BackendSlotTable(BackendSlotTable&& other) noexcept:
|
||||
m_slots(std::move(other.m_slots)),
|
||||
m_nullTwin(std::move(other.m_nullTwin)),
|
||||
m_memoLifetimeId(other.m_memoLifetimeId),
|
||||
m_memoHandle(other.m_memoHandle) {
|
||||
other.m_slots.clear();
|
||||
other.ForgetHandle();
|
||||
LinkHolder();
|
||||
}
|
||||
BackendSlotTable& operator=(const BackendSlotTable& other) {
|
||||
if (this != &other) {
|
||||
m_slots = other.m_slots;
|
||||
m_nullTwin = other.m_nullTwin;
|
||||
m_memoLifetimeId = other.m_memoLifetimeId;
|
||||
m_memoHandle = other.m_memoHandle;
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
BackendSlotTable& operator=(BackendSlotTable&& other) noexcept {
|
||||
if (this != &other) {
|
||||
m_slots = std::move(other.m_slots);
|
||||
m_nullTwin = std::move(other.m_nullTwin);
|
||||
m_memoLifetimeId = other.m_memoLifetimeId;
|
||||
m_memoHandle = other.m_memoHandle;
|
||||
other.m_slots.clear();
|
||||
other.ForgetHandle();
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
~BackendSlotTable() { UnlinkHolder(); }
|
||||
|
||||
// Resolve-or-create. The handle comes from the client allocator keyed on the frontend
|
||||
// object's lifetime id, so two calls for the same live object always land on the same
|
||||
// slot, and a successor object at the same heap address never does.
|
||||
BackendPtr& GetOrCreate(const StatePtr& stateObj) {
|
||||
// No assert on null here, unlike the map arm: null is TOLERATED, so a DEBUG build
|
||||
// must not trap where the release build quietly does the documented thing.
|
||||
if (stateObj == nullptr) {
|
||||
// The registry this replaces inserted a null KEY and handed back that entry's
|
||||
// twin (DirectGLES.cpp's SyncTextureObjectToBackend documents relying on
|
||||
// exactly that tolerance), so a release build never dereferenced null here.
|
||||
// Keep the shape exactly, INCLUDING across calls: the map kept its null-keyed
|
||||
// entry, so a second null call was handed the same twin the first one got.
|
||||
// Resetting here instead would have destroyed it - an arm difference in the one
|
||||
// path that documents relying on this. One per-table parking slot, never live,
|
||||
// never handed a handle, because a null object has no identity and
|
||||
// therefore cannot have a {slot, gen}.
|
||||
return m_nullTwin;
|
||||
}
|
||||
|
||||
// D13: the teardown sentinel is armed by the slot table's first insertion. Twin
|
||||
// creation is the moment a driver-owned id starts needing a guarded destructor;
|
||||
// this is the cold path, so the once-guard costs nothing per draw. On the legacy
|
||||
// arm StateBackendObjectRegistry::GetOrCreate arms it itself.
|
||||
EnsureProcessTeardownSentinel();
|
||||
|
||||
const MG_Pipe::MGPipeHandle handle =
|
||||
MG_Pipe::MGPipeSlots().Acquire(kKind, stateObj->GetLifetimeId());
|
||||
MOBILEGL_ASSERT(!MG_Pipe::MGPipeHandleIsNull(handle),
|
||||
"MGPipe slot space of kind %u is exhausted",
|
||||
static_cast<Uint32>(kKind));
|
||||
Entry& entry = EntryAt(handle.Slot);
|
||||
if (entry.Live && entry.Gen != handle.Gen) {
|
||||
// The slot was reclaimed and handed to a new object: the twin at it describes
|
||||
// driver ids the new state object never made.
|
||||
entry.backend.reset();
|
||||
}
|
||||
entry.Gen = handle.Gen;
|
||||
entry.Live = true;
|
||||
entry.stateRef = stateObj;
|
||||
// No creation tick and no sweep here. The registry this replaces needed both,
|
||||
// because nothing told it a texture or a renderbuffer had been DELETED and object
|
||||
// CHURN rather than draw count is what made that urgent. Every one of the six kinds
|
||||
// now announces its own death from its destructor, so a dead twin's slot is already
|
||||
// back before the next creation asks for one.
|
||||
RememberHandle(stateObj->GetLifetimeId(), handle);
|
||||
return entry.backend;
|
||||
}
|
||||
|
||||
// P3a: resolve-or-create BY HANDLE, and it is the shape that discharges the debt this
|
||||
// header records against itself at the top of the file.
|
||||
//
|
||||
// The overload above mints - it calls MGPipeSlots().Acquire off a frontend object's
|
||||
// lifetime id, from inside MG_Backend - which is monolith glue: a handle is minted by
|
||||
// the CLIENT, and under a real split neither the object nor its lifetime id exists on
|
||||
// this side. This overload never touches the allocator at all. The handle ARRIVED, in
|
||||
// the call's payload, already minted by the side that owns minting; all this does is
|
||||
// index the slot, notice a generation that no longer matches (the slot was recycled,
|
||||
// so the twin at it describes driver ids the new resource never made) and hand back
|
||||
// the twin pointer. FindByHandle beside it is the same shape and already existed.
|
||||
//
|
||||
// No StatePtr, therefore no Entry::stateRef: the weak pointer is liveness for
|
||||
// ForEachLive() and a handle-keyed entry has no frontend object to weakly hold. Such
|
||||
// an entry is therefore invisible to ForEachLive, which is correct - the one direct
|
||||
// iteration site walks texture twins, and it is not one of these tables.
|
||||
//
|
||||
// Death stays ANNOUNCED, as it is on the other overload: for a handle-keyed kind the
|
||||
// announcement is the family's own destroy call, not the shared death notice, and the
|
||||
// slot is freed by the CLIENT after that call returns.
|
||||
//
|
||||
// UNUSED AT THE CONTRACT COMMIT, deliberately: it is a member of a class template, so
|
||||
// an uninstantiated one costs nothing anywhere, and the backend package is what gives
|
||||
// it its first caller.
|
||||
BackendPtr& GetOrCreate(MG_Pipe::MGPipeHandle handle) {
|
||||
MOBILEGL_ASSERT(!MG_Pipe::MGPipeHandleIsNull(handle),
|
||||
"GetOrCreate(handle) named the reserved null handle");
|
||||
if (MG_Pipe::MGPipeHandleIsNull(handle)) return m_nullTwin;
|
||||
|
||||
// A slot index that ARRIVED in a payload indexes a vector this call would RESIZE,
|
||||
// and nothing between the payload and here bounds it: the applier's blob gates sit
|
||||
// in front of the vertex-input family, not in front of the resource family, which
|
||||
// dispatches ops->Create(record.Res, ...) straight through. There is no allocator
|
||||
// constant to check against on this side - the allocator is the client's - so this
|
||||
// is a sanity cap and is documented as one: kMaxHandleSlot entries of one kind is
|
||||
// already orders of magnitude past any real GL object count, while a corrupt 32-bit
|
||||
// slot asks for a four-billion-entry resize.
|
||||
if (handle.Slot >= kMaxHandleSlot) {
|
||||
MOBILEGL_ASSERT(false, "GetOrCreate(handle) named slot %u, past this table's %u bound",
|
||||
handle.Slot, kMaxHandleSlot);
|
||||
return m_nullTwin;
|
||||
}
|
||||
|
||||
// Same arming as the minting overload, and for the same reason: twin creation is
|
||||
// the moment a driver-owned id starts needing a guarded destructor.
|
||||
EnsureProcessTeardownSentinel();
|
||||
|
||||
// THE TWO DIRECTIONS ARE NOT SYMMETRIC HERE, where they are on the minting overload.
|
||||
// There the handle comes straight out of MGPipeSlots().Acquire and can never be
|
||||
// BEHIND the entry, so a bare `!=` only ever means "the slot was recycled forward".
|
||||
// Here the handle arrived in a payload, so `handle.Gen < entry.Gen` is a reachable
|
||||
// input, and adopting it would destroy the INCUMBENT LIVE twin - a driver buffer id,
|
||||
// a persistent map, a pooled store, released by a defaulted destructor that issues
|
||||
// no glDeleteBuffers and no pool enrolment - and then stamp the slot back to the
|
||||
// dead resource's generation, after which the incumbent's own FindByHandle refuses
|
||||
// it and it is silently handed a fresh, empty twin. That is a leak AND a resource
|
||||
// that loses its storage with no diagnostic, i.e. the shape commit d7655247 fixed
|
||||
// and the thing MGPipeHandle::Gen exists to prevent. So: forward is a recycle and
|
||||
// resets the twin, BACKWARD is refused - which is the same answer FindByHandle
|
||||
// below already gives the same input.
|
||||
Entry& entry = EntryAt(handle.Slot);
|
||||
if (entry.Live && entry.Gen > handle.Gen) {
|
||||
MOBILEGL_ASSERT(false,
|
||||
"GetOrCreate(handle) named generation %u at slot %u, which is BEHIND "
|
||||
"the live entry's %u - refusing rather than destroying the incumbent",
|
||||
handle.Gen, handle.Slot, entry.Gen);
|
||||
return m_nullTwin;
|
||||
}
|
||||
if (entry.Live && entry.Gen != handle.Gen) entry.backend.reset();
|
||||
entry.Gen = handle.Gen;
|
||||
entry.Live = true;
|
||||
return entry.backend;
|
||||
}
|
||||
|
||||
// The generation of the LIVE entry at this slot, or 0 when the slot is out of range or
|
||||
// holds no live entry. It exists so a caller can DIAGNOSE - in a release build, where
|
||||
// MOBILEGL_ASSERT is inert - the refusal GetOrCreate(handle) above performs silently.
|
||||
Uint32 LiveGenAt(Uint32 slot) const {
|
||||
if (slot >= m_slots.size()) return 0;
|
||||
const Entry& entry = m_slots[slot];
|
||||
return entry.Live ? entry.Gen : 0;
|
||||
}
|
||||
|
||||
// P3a: the death half of the overload above, for a kind whose announcement is its own
|
||||
// destroy CALL rather than the shared death notice (D-L). Hands the twin OUT rather
|
||||
// than destroying it in place, because the caller may still have to decide what
|
||||
// happens to the driver id it owns - Espryt pools it, deletes it, or parks it on the
|
||||
// deferred-release list when no context is current on this thread - and every one of
|
||||
// those outcomes has to be reached with the entry already retired, so a re-entrant
|
||||
// GetOrCreate from a twin destructor cannot resurrect it.
|
||||
//
|
||||
// The slot itself is NOT freed here: it belongs to the kind, and for a handle-keyed
|
||||
// kind the CLIENT frees it after the destroy call returns (SlotAllocator.h:60 - the
|
||||
// Gen bump rides the next handout, so a double free cannot skip a generation). An
|
||||
// entry whose Gen no longer matches is a twin of the slot's previous owner and is
|
||||
// left alone: the successor's own GetOrCreate resets it.
|
||||
BackendPtr ReleaseByHandle(MG_Pipe::MGPipeHandle handle) {
|
||||
if (MG_Pipe::MGPipeHandleIsNull(handle)) return BackendPtr{};
|
||||
if (m_memoHandle.Slot == handle.Slot) ForgetHandle();
|
||||
if (handle.Slot >= m_slots.size()) return BackendPtr{};
|
||||
Entry& entry = m_slots[handle.Slot];
|
||||
if (!entry.Live || entry.Gen != handle.Gen) return BackendPtr{};
|
||||
BackendPtr dead = std::move(entry.backend);
|
||||
entry.backend.reset();
|
||||
entry.stateRef.reset();
|
||||
entry.Live = false;
|
||||
return dead;
|
||||
}
|
||||
|
||||
// Null when no live twin of this object exists. Unlike the registry's Find this NEVER
|
||||
// mutates the table, so the returned pointer survives any later Find on it; only a
|
||||
// GetOrCreate that grows the vector can move it, and callers that hold one across a
|
||||
// possible insertion still copy the BackendPtr out.
|
||||
BackendPtr* Find(StateObject* stateObj) {
|
||||
if (stateObj == nullptr) return nullptr;
|
||||
return FindByHandle(HandleOf(stateObj));
|
||||
}
|
||||
|
||||
const BackendPtr* Find(StateObject* stateObj) const {
|
||||
return const_cast<BackendSlotTable*>(this)->Find(stateObj);
|
||||
}
|
||||
|
||||
BackendPtr* FindByHandle(MG_Pipe::MGPipeHandle handle) {
|
||||
if (MG_Pipe::MGPipeHandleIsNull(handle)) return nullptr;
|
||||
if (handle.Slot >= m_slots.size()) return nullptr;
|
||||
Entry& entry = m_slots[handle.Slot];
|
||||
if (!entry.Live || entry.Gen != handle.Gen) return nullptr;
|
||||
return &entry.backend;
|
||||
}
|
||||
|
||||
// The handle this object's twin is keyed on, or the null handle. This is what a backend
|
||||
// memo stores instead of a raw pointer, a GL name or a bare lifetime id.
|
||||
//
|
||||
// A NULL answer is never memoised. The memo is per table and the allocator is per
|
||||
// kind, so with two holders of one kind the OTHER table can be the one that acquires;
|
||||
// a cached "no handle" here would then outlive the twin's creation over there, and
|
||||
// nothing on this table's own acquire path would ever refresh it. A miss costs the
|
||||
// allocator probe it always cost; a hit is refreshed the moment anyone acquires.
|
||||
MG_Pipe::MGPipeHandle HandleOf(const StateObject* stateObj) const {
|
||||
if (stateObj == nullptr) return MG_Pipe::kMGPipeNullHandle;
|
||||
const Uint64 lifetimeId = stateObj->GetLifetimeId();
|
||||
if (lifetimeId == m_memoLifetimeId) return m_memoHandle;
|
||||
const MG_Pipe::MGPipeHandle handle =
|
||||
MG_Pipe::MGPipeSlots().FindByLifetimeId(kKind, lifetimeId);
|
||||
if (!MG_Pipe::MGPipeHandleIsNull(handle)) RememberHandle(lifetimeId, handle);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// P2 step e2's backend half. The frontend object with this lifetime id has just been
|
||||
// DESTROYED: resolve its handle ONCE, drop its twin in EVERY table of this type, and
|
||||
// return the slot to the allocator - in that order, because the allocator forgets the
|
||||
// lifetime id on Free and a holder told second could no longer resolve it.
|
||||
//
|
||||
// The slot is returned whether or not any holder still had a twin at it: the lifetime
|
||||
// id is dead and MG_State never hands one out twice, so nothing can acquire it again,
|
||||
// and a slot minted for it that no table holds (a table reset with `= {}` drops its
|
||||
// entries without freeing) would otherwise stay allocated for the life of the process.
|
||||
//
|
||||
// STATIC, and deliberately so: a notice is about an object, not about a table, and
|
||||
// "which table holds it" is exactly the question that produced the two-holder leak.
|
||||
// Returns whether the object had a slot of this kind, i.e. whether anything was freed;
|
||||
// a second call for the same id answers false because the allocator no longer maps it.
|
||||
static Bool OnFrontendObjectDestroyed(Uint64 lifetimeId) {
|
||||
const MG_Pipe::MGPipeHandle handle =
|
||||
MG_Pipe::MGPipeSlots().FindByLifetimeId(kKind, lifetimeId);
|
||||
if (MG_Pipe::MGPipeHandleIsNull(handle)) return false;
|
||||
for (BackendSlotTable* holder = s_firstHolder; holder != nullptr;) {
|
||||
// The successor is read BEFORE the release: ReleaseTwinAt runs the twin's
|
||||
// destructor, which is a driver call, and nothing that outlives it may be a
|
||||
// reference into this holder.
|
||||
BackendSlotTable* const next = holder->m_nextHolder;
|
||||
holder->ReleaseTwinAt(handle);
|
||||
holder = next;
|
||||
}
|
||||
MG_Pipe::MGPipeSlots().Free(kKind, handle);
|
||||
return true;
|
||||
}
|
||||
|
||||
// How many tables of this type exist right now. For the tests that pin the holder
|
||||
// list; nothing on a shipping path asks.
|
||||
static Uint32 HolderCount() {
|
||||
Uint32 count = 0;
|
||||
for (const BackendSlotTable* holder = s_firstHolder; holder != nullptr;
|
||||
holder = holder->m_nextHolder) {
|
||||
++count;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
// fn(const StatePtr& state, const BackendPtr& twin) over every live, still-owned entry.
|
||||
// Replaces the registry's begin()/end(), whose iterator exposed the raw frontend
|
||||
// address as the map key - the one place the backend read an identity it must not have.
|
||||
// The state object is handed over as a STRONG reference, so the callee cannot be handed
|
||||
// a dangling key the way the old iteration could.
|
||||
template <typename Fn>
|
||||
void ForEachLive(Fn&& fn) const {
|
||||
// Index loop and a COPIED twin, not a range-for over references: fn is arbitrary
|
||||
// backend code, and a nested GetOrCreate on this table would resize m_slots and
|
||||
// invalidate both the iterator and any reference into the vector that outlives the
|
||||
// call. The one caller today happens not to insert; that is not a property the
|
||||
// walk should depend on.
|
||||
for (SizeT slot = 0; slot < m_slots.size(); ++slot) {
|
||||
const Entry& entry = m_slots[slot];
|
||||
if (!entry.Live || !entry.backend) continue;
|
||||
const StatePtr state = entry.stateRef.lock();
|
||||
if (!state) continue;
|
||||
const BackendPtr twin = entry.backend;
|
||||
fn(state, twin);
|
||||
}
|
||||
}
|
||||
|
||||
Uint32 LiveCount() const {
|
||||
Uint32 count = 0;
|
||||
for (const Entry& entry : m_slots) {
|
||||
if (entry.Live) ++count;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
private:
|
||||
// Drop the twin at `handle` if THIS table holds it. Frees nothing: the slot belongs to
|
||||
// the kind, not to the table, and OnFrontendObjectDestroyed returns it once, after
|
||||
// every holder has let go.
|
||||
Bool ReleaseTwinAt(MG_Pipe::MGPipeHandle handle) {
|
||||
// Forget the memo whenever it names this slot, even if this table has no entry
|
||||
// there: a memo can be a handle learned from the allocator for an object another
|
||||
// holder twinned, and it must not survive the slot's next handout.
|
||||
if (m_memoHandle.Slot == handle.Slot) ForgetHandle();
|
||||
if (handle.Slot >= m_slots.size()) return false;
|
||||
// The twin's destructor is a driver call and could, in principle, re-enter
|
||||
// GetOrCreate on this table and resize m_slots. So NOTHING that outlives the
|
||||
// destructor may be a reference into m_slots: the twin is moved out into a local,
|
||||
// the entry is finished with, and only then is the local released.
|
||||
BackendPtr dead;
|
||||
{
|
||||
Entry& entry = m_slots[handle.Slot];
|
||||
if (!entry.Live || entry.Gen != handle.Gen) return false;
|
||||
dead = std::move(entry.backend);
|
||||
entry.backend.reset();
|
||||
entry.stateRef.reset();
|
||||
entry.Live = false;
|
||||
}
|
||||
dead.reset();
|
||||
return true;
|
||||
}
|
||||
|
||||
// Grows the table to hold `slot`. Every caller bounds `slot` first - the minting
|
||||
// overload because the allocator produced it, the handle overload against
|
||||
// kMaxHandleSlot - because this is the one place a client-supplied number decides an
|
||||
// allocation size.
|
||||
Entry& EntryAt(Uint32 slot) {
|
||||
if (slot >= m_slots.size()) m_slots.resize(static_cast<SizeT>(slot) + 1);
|
||||
return m_slots[slot];
|
||||
}
|
||||
|
||||
void RememberHandle(Uint64 lifetimeId, MG_Pipe::MGPipeHandle handle) const {
|
||||
m_memoLifetimeId = lifetimeId;
|
||||
m_memoHandle = handle;
|
||||
}
|
||||
void ForgetHandle() const {
|
||||
m_memoLifetimeId = 0;
|
||||
m_memoHandle = MG_Pipe::kMGPipeNullHandle;
|
||||
}
|
||||
|
||||
// The holder list: intrusive and doubly linked, so registering and unregistering are
|
||||
// two pointer writes with no allocation, and its head is a constant-initialised
|
||||
// static - which is what lets the process-lifetime registry globals in Managers.cpp
|
||||
// link themselves in from their own constructors with no initialisation-order
|
||||
// question to answer. Single-threaded, like every table it links (the tables live and
|
||||
// die on the context thread, as the notice they answer does).
|
||||
void LinkHolder() {
|
||||
m_prevHolder = nullptr;
|
||||
m_nextHolder = s_firstHolder;
|
||||
if (s_firstHolder != nullptr) s_firstHolder->m_prevHolder = this;
|
||||
s_firstHolder = this;
|
||||
}
|
||||
void UnlinkHolder() {
|
||||
if (m_prevHolder != nullptr) {
|
||||
m_prevHolder->m_nextHolder = m_nextHolder;
|
||||
} else {
|
||||
s_firstHolder = m_nextHolder;
|
||||
}
|
||||
if (m_nextHolder != nullptr) m_nextHolder->m_prevHolder = m_prevHolder;
|
||||
m_prevHolder = nullptr;
|
||||
m_nextHolder = nullptr;
|
||||
}
|
||||
|
||||
static inline BackendSlotTable* s_firstHolder = nullptr;
|
||||
BackendSlotTable* m_prevHolder = nullptr;
|
||||
BackendSlotTable* m_nextHolder = nullptr;
|
||||
|
||||
// Indexed by MGPipeHandle::Slot; [0] is the reserved slot and is never live.
|
||||
Vector<Entry> m_slots;
|
||||
// Handed back by GetOrCreate for a null state object. Never live, never handed a handle.
|
||||
BackendPtr m_nullTwin;
|
||||
|
||||
// ONE-entry resolution memo, lifetimeId -> handle. It exists because without it every
|
||||
// resolution goes through the allocator's ByLifetimeId hash, which the deleted
|
||||
// TwinLookupMemos existed to avoid and which D13 promises to replace with "direct slot
|
||||
// indexing".
|
||||
//
|
||||
// It is one entry and therefore only helps a caller that asks for the SAME object twice
|
||||
// running - ResolveVaoTwin and SyncCurrentProgram do, once per draw each. Two callers
|
||||
// it does NOT help, recorded rather than claimed away: BindCurrentFBO resolves BOTH
|
||||
// targets in a frame, and ResolveUnitSamplerBackend asks for a different sampler per
|
||||
// texture unit, so both thrash a single-entry memo and pay the probe P1 did not (P1 had
|
||||
// a per-unit memo and a direct-mapped 6-slot array there). Making the memo per-unit /
|
||||
// per-target is the fix, and G11 - the device-side gate that would price it - is owed.
|
||||
//
|
||||
// It cannot serve a stale answer, by three independent arguments:
|
||||
// * the key is a lifetime id, which MG_State never hands out twice, so a recycled
|
||||
// heap address cannot hit this memo the way it could hit an address-keyed one;
|
||||
// * a null answer is never stored, so another holder's acquire cannot be hidden by
|
||||
// a "no handle" this table remembered earlier; and
|
||||
// * even a hit for a slot that has since been freed and re-handed is caught, because
|
||||
// the caller resolves the handle through FindByHandle, which compares Gen.
|
||||
// Cleared anyway when a death notice names the memoised slot. 0 is never a live
|
||||
// lifetime id (MG_State's counters start at 1), so a zeroed memo is a guaranteed miss.
|
||||
mutable Uint64 m_memoLifetimeId = 0;
|
||||
mutable MG_Pipe::MGPipeHandle m_memoHandle = MG_Pipe::kMGPipeNullHandle;
|
||||
};
|
||||
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
} // namespace MobileGL::MG_Backend::DirectGLES
|
||||
@@ -11,10 +11,13 @@
|
||||
#include "Managers.h"
|
||||
#include "MG_Backend/BackendObjects.h"
|
||||
#include "MG_Util/Converters/GLToMG/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/SelfTest/DriverBugProbes.h"
|
||||
#include "MG_Util/Texture/TextureFormatProcessor.h"
|
||||
#include "MG_Util/ShaderTranspiler/ShaderCompiler.h"
|
||||
#include <Config.h>
|
||||
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include <MG_Util/BackendLoaders/OpenGL/Loader.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
|
||||
@@ -26,6 +29,7 @@
|
||||
#include <cmath>
|
||||
#include <cctype>
|
||||
#include <cstring>
|
||||
#include <format>
|
||||
#include <regex>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectGLES {
|
||||
@@ -124,6 +128,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
requestedInternalFormat,
|
||||
TextureImpl::GetRenderTargetNormalizeOptions(g_GLESCapabilities, targetIndex));
|
||||
}
|
||||
// Outside the caveat branch on purpose: the driver CAN create the native narrow
|
||||
// storage - the capability probes say so - it just cannot be trusted as a raw-copy
|
||||
// endpoint. Texture and renderbuffer targets both come through here, which is what
|
||||
// keeps a renderbuffer -> texture copy of these formats same-ES-format when the
|
||||
// widening engages.
|
||||
if (TextureImpl::UsesWidenedPacked16NormStorage(internalFormat)) {
|
||||
options |= PixelFormatNormalizeOptionBit::WidenPacked16Norm;
|
||||
}
|
||||
NormalizePixelFormat(requestedInternalFormat, options, outInternalFormat, outFormat, outType);
|
||||
}
|
||||
} // namespace
|
||||
@@ -181,6 +193,36 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return options;
|
||||
}
|
||||
|
||||
Bool UsesWidenedPacked16NormStorage(TextureInternalFormat internalFormat) {
|
||||
switch (internalFormat) {
|
||||
// TextureInternalFormat::RGB5 is both GL_RGB5 and GL_RGB565 - the GL-to-MG
|
||||
// converter folds the two spellings onto one logical format.
|
||||
case TextureInternalFormat::RGB5:
|
||||
case TextureInternalFormat::RGB5A1:
|
||||
case TextureInternalFormat::RGBA4:
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
switch (MG_Config::Features.EsprytWidenPacked16Storage) {
|
||||
case MG_Config::QuirkOverride::ForceOn:
|
||||
return true;
|
||||
case MG_Config::QuirkOverride::ForceOff:
|
||||
return false;
|
||||
case MG_Config::QuirkOverride::Auto:
|
||||
break;
|
||||
}
|
||||
// Behind the backend gate on purpose: the memoized probe latches its first answer
|
||||
// for the whole process, and before the backend is up the GL function table may
|
||||
// not be resolved yet - a probe run then would latch "cannot tell" as "clean"
|
||||
// forever. Once the backend exists, the first narrow-format image this process
|
||||
// creates runs the probe on a live context.
|
||||
if (pActiveBackendObject == nullptr) {
|
||||
return false;
|
||||
}
|
||||
return MG_Util::SelfTest::CopyImageMirrorsPacked16FieldOrder(g_GLESFuncs);
|
||||
}
|
||||
|
||||
void GenerateTextureFormatInfo(TextureInternalFormat internalFormat, GLenum* outInternalFormat,
|
||||
GLenum* outFormat, GLenum* outType, TextureTarget target) {
|
||||
#ifdef TRACY_ENABLE
|
||||
@@ -712,6 +754,47 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return glslCode;
|
||||
}
|
||||
|
||||
const char* PointSizeExtensionName(MG_External::GLESCapabilities::PointSizeTier tier, Bool tessellation) {
|
||||
using Tier = MG_External::GLESCapabilities::PointSizeTier;
|
||||
switch (tier) {
|
||||
case Tier::ExtensionEXT:
|
||||
return tessellation ? "GL_EXT_tessellation_point_size" : "GL_EXT_geometry_point_size";
|
||||
case Tier::ExtensionOES:
|
||||
return tessellation ? "GL_OES_tessellation_point_size" : "GL_OES_geometry_point_size";
|
||||
default:
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
String RequestPointSizeExtension(String glslCode, const char* extensionName) {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
// The gl_ViewportIndex story, one built-in over: ESSL 320 makes the tessellation and
|
||||
// geometry STAGES core but leaves gl_PointSize out of their gl_PerVertex entirely,
|
||||
// and SPIRV-Cross - which only ever sees a SPIR-V BuiltIn PointSize decoration -
|
||||
// prints the identifier with no directive behind it. Same hard rule as the two
|
||||
// neighbours: never emitted speculatively, because `#extension` on a name the driver
|
||||
// does not advertise is a compile error of its own.
|
||||
if (extensionName == nullptr || glslCode.find(extensionName) != String::npos) {
|
||||
return glslCode;
|
||||
}
|
||||
const String directive = String("#extension ") + extensionName + " : require\n";
|
||||
// Right after the #version line, the one position that must stay first;
|
||||
// ForceSupporterOutput's scan for the LAST #extension directive still finds
|
||||
// whichever one that ends up being.
|
||||
const SizeT versionPos = glslCode.find("#version");
|
||||
if (versionPos == String::npos) {
|
||||
return directive + glslCode;
|
||||
}
|
||||
const SizeT lineEnd = glslCode.find('\n', versionPos);
|
||||
if (lineEnd == String::npos) {
|
||||
return glslCode + "\n" + directive;
|
||||
}
|
||||
glslCode.insert(lineEnd + 1, directive);
|
||||
return glslCode;
|
||||
}
|
||||
|
||||
String BakeImageFormatQualifiers(String glslCode,
|
||||
const UnorderedMap<String, String>& esslFormatByUniformName) {
|
||||
#ifdef TRACY_ENABLE
|
||||
@@ -836,7 +919,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
String BuildPassthroughTessControlEssl(const Uint esslVersion, const Uint patchVertices,
|
||||
const String& inPerVertexMembers,
|
||||
const String& outPerVertexMembers) {
|
||||
const String& outPerVertexMembers,
|
||||
const FloatVec4& defaultOuterLevel,
|
||||
const FloatVec2& defaultInnerLevel) {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
@@ -866,12 +951,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// was declined before this was ever called (ModuleReadsLocatedInput), and gl_PointSize
|
||||
// from a tessellation stage is a separate capability on both targets.
|
||||
source += " gl_out[gl_InvocationID].gl_Position = gl_in[gl_InvocationID].gl_Position;\n";
|
||||
source += " gl_TessLevelOuter[0] = 1.0;\n";
|
||||
source += " gl_TessLevelOuter[1] = 1.0;\n";
|
||||
source += " gl_TessLevelOuter[2] = 1.0;\n";
|
||||
source += " gl_TessLevelOuter[3] = 1.0;\n";
|
||||
source += " gl_TessLevelInner[0] = 1.0;\n";
|
||||
source += " gl_TessLevelInner[1] = 1.0;\n";
|
||||
for (Uint i = 0; i < 4; ++i) {
|
||||
source += " gl_TessLevelOuter[" + std::to_string(i) +
|
||||
"] = " + MG_Util::ShaderTranspiler::TessellationLevelLiteral(defaultOuterLevel[i]) + ";\n";
|
||||
}
|
||||
for (Uint i = 0; i < 2; ++i) {
|
||||
source += " gl_TessLevelInner[" + std::to_string(i) +
|
||||
"] = " + MG_Util::ShaderTranspiler::TessellationLevelLiteral(defaultInnerLevel[i]) + ";\n";
|
||||
}
|
||||
source += "}\n";
|
||||
return source;
|
||||
}
|
||||
@@ -2208,11 +2295,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
static Bool StoreClientRows(SizeT dstPixelBytes, SizeT swapGroupSize, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, void* pixels, Bool applyPackImageParams, FillRow&& fillRow) {
|
||||
const auto& pixelPackBufferObject =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::PixelPack).GetBoundObject();
|
||||
MGB_CTX->GetBufferBindingSlot(BufferTarget::PixelPack).GetBoundObject();
|
||||
|
||||
// Destination layout is computed from the client-side PACK parameters; only the actual pixel
|
||||
// rows are written so skip regions of the destination stay untouched.
|
||||
const auto packParams = MG_State::pGLContext->GetPixelStoreParameters(false);
|
||||
const auto packParams = MGB_CTX->GetPixelStoreParameters(false);
|
||||
const SizeT rowPixels = static_cast<SizeT>(packParams.RowLength > 0 ? packParams.RowLength : width);
|
||||
const SizeT dstRowStride = AlignReadbackRow(rowPixels * dstPixelBytes, packParams.Alignment);
|
||||
const SizeT imageRows =
|
||||
|
||||
@@ -46,6 +46,15 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Flags<PixelFormatNormalizeOptionBit> GetRenderTargetNormalizeOptions(
|
||||
const MG_External::GLESCapabilities& capabilities, SizeT targetIndex);
|
||||
|
||||
// Whether this format's ES storage is widened to 8-bit-per-channel because the
|
||||
// driver stores some packed16 allocations with a mirrored field order
|
||||
// (PixelFormatNormalizeOptionBit::WidenPacked16Norm). True only for
|
||||
// GL_RGB565/GL_RGB5(_A1)/GL_RGBA4, and only where the POST probe measured the
|
||||
// divergence (or MOBILEGL_ESPRYT_WIDEN_PACKED16_STORAGE forces it). The transfer paths
|
||||
// consult it too: the packed-norm re-upload leg must stand down when the ES storage
|
||||
// is no longer 16-bit packed.
|
||||
Bool UsesWidenedPacked16NormStorage(TextureInternalFormat internalFormat);
|
||||
|
||||
void GenerateTextureFormatInfo(TextureInternalFormat internalFormat, GLenum* outInternalFormat,
|
||||
GLenum* outFormat, GLenum* outType,
|
||||
TextureTarget target = TextureTarget::Unknown);
|
||||
@@ -273,6 +282,22 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// error, so this is never emitted speculatively. A no-op when not needed or already
|
||||
// present.
|
||||
String RequestViewportArrayExtension(String glslCode, Bool needed);
|
||||
// Adds `#extension <extensionName> : require` when a TESSELLATION or GEOMETRY stage's
|
||||
// emitted ESSL names gl_PointSize. Desktop GL has that built-in in gl_PerVertex for every
|
||||
// vertex-processing stage; ESSL does NOT have it in those two at any version - not even
|
||||
// 320, where the stages themselves are core - until EXT/OES_tessellation_point_size resp.
|
||||
// EXT/OES_geometry_point_size is requested. SPIRV-Cross prints the identifier bare and
|
||||
// asks for nothing, exactly as it does for gl_ViewportIndex, so without this the stage
|
||||
// fails to compile with "`gl_PointSize' undeclared" and the WHOLE program is replaced by
|
||||
// program 0 - the draw renders nothing and any transform-feedback capture it was carrying
|
||||
// is rejected outright. `extensionName` is the caller's answer, nullptr when the driver
|
||||
// advertises neither spelling, because requesting an unadvertised extension is itself a
|
||||
// compile error. A no-op when nullptr or already present.
|
||||
String RequestPointSizeExtension(String glslCode, const char* extensionName);
|
||||
// The extension name RequestPointSizeExtension should be given for `tier`, or nullptr for
|
||||
// PointSizeTier::None. `tessellation` picks the tessellation spellings over the geometry
|
||||
// ones; the two extensions are separate and neither implies the other.
|
||||
const char* PointSizeExtensionName(MG_External::GLESCapabilities::PointSizeTier tier, Bool tessellation);
|
||||
// Writes a format layout qualifier into the image declarations named in
|
||||
// `esslFormatByUniformName` that still have none. The completion half of the image-format
|
||||
// bake, and ONLY that: the SPIR-V pass (BakeImageFormatsPass) is what normally puts the
|
||||
@@ -368,11 +393,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
//
|
||||
// All four outer levels and both inner levels are written unconditionally: writing a
|
||||
// level the evaluation stage's domain does not use is legal and ignored, and it saves
|
||||
// this from having to know the domain. They are literal 1.0 because that is the GL
|
||||
// default and glPatchParameterfv - their only setter - is a stub in this frontend
|
||||
// (MG_Impl/GLImpl/Exporting/Definitions.cpp). Implementing that entry point means making
|
||||
// the levels a parameter here AND part of what makes a built program stale, exactly as
|
||||
// PATCH_VERTICES already is; the two must move together, so they are named together.
|
||||
// this from having to know the domain. They are the GL_PATCH_DEFAULT_OUTER_LEVEL /
|
||||
// GL_PATCH_DEFAULT_INNER_LEVEL state, baked in as literals - ES has no such state and no
|
||||
// glPatchParameterfv to forward to, so compiling them in is the only way to honour them.
|
||||
// That makes them part of what a built program is stale against, exactly as PATCH_VERTICES
|
||||
// is: see the staleness clause in DirectGLES.cpp's SyncCurrentProgram, which compares both.
|
||||
//
|
||||
// The same stage, for the same reason, that DirectVulkan synthesizes in
|
||||
// ProgramFactory::BuildPassthroughTessControlSource - Vulkan likewise requires both
|
||||
@@ -382,7 +407,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// VkShaderModule against a driver shader object.
|
||||
String BuildPassthroughTessControlEssl(Uint esslVersion, Uint patchVertices,
|
||||
const String& inPerVertexMembers,
|
||||
const String& outPerVertexMembers);
|
||||
const String& outPerVertexMembers,
|
||||
const FloatVec4& defaultOuterLevel,
|
||||
const FloatVec2& defaultInnerLevel);
|
||||
// Prefix of the writeonly half a read+write image uniform is split into (see
|
||||
// SplitReadWriteImageUniforms); the suffix is the image's own (already access-tagged) name.
|
||||
constexpr const char* IMAGE_WRITE_ALIAS_PREFIX = "mg_imageWrite_";
|
||||
@@ -505,7 +532,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// avoidExplicitLodBias leaves lookups that already carry an explicit LOD untouched,
|
||||
// so their constant level stays constant; only the implicit-LOD forms take the bias.
|
||||
// Off by default and only ever set on ANGLE + llvmpipe, where injecting the uniform
|
||||
// into a constant LOD crashes the driver (MOBILEGL_AVOID_EXPLICIT_LOD_BIAS).
|
||||
// into a constant LOD crashes the driver (MOBILEGL_ESPRYT_AVOID_EXPLICIT_LOD_BIAS).
|
||||
String EmulateTextureLodBias(const String& glslCode, Bool avoidExplicitLodBias = false);
|
||||
} // namespace PrgramImpl
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include "SubgroupSupportPolicy.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include "MG_State/GLState/TextureState/TextureState.h"
|
||||
#include "MG_Util/Classifiers/TextureEnumClassifier.h"
|
||||
#include "MG_Util/Converters/MGToGL/TextureEnumConverter.h"
|
||||
@@ -385,8 +386,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
if (MG_State::pGLContext) {
|
||||
MG_State::pGLContext->InvalidateCompileEnv();
|
||||
if (MGB_CTX_LIVE) {
|
||||
MGB_CTX->InvalidateCompileEnv();
|
||||
}
|
||||
PopulateFormatCapabilities(physicalDevice.handle, vkGetPhysicalDeviceFormatProperties, m_vulkanCaps,
|
||||
MutableFormatCapabilities());
|
||||
@@ -500,7 +501,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
.RendererName = "Magma",
|
||||
.BackendName = "Direct (Vulkan)",
|
||||
.ExtraVendor = Nullopt,
|
||||
.RendererGLInfo = {.TargetGLVersion = {4, 3, 0},
|
||||
.RendererGLInfo = {.TargetGLVersion = {4, 6, 0},
|
||||
.TargetGLSLVersion = {4, 6, 0},
|
||||
// Baseline advertisement (no runtime-gated capabilities); a live
|
||||
// backend reconciles its copy in UpdateAdvertisedExtensions.
|
||||
@@ -516,10 +517,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool cubeMapArraySupported) {
|
||||
Vector<GLExtension> extensions = {
|
||||
// The version tokens have to reach the version the backend actually claims:
|
||||
// TargetGLVersion is {4,3,0}, and a list that stopped at OpenGL40 told an
|
||||
// TargetGLVersion is {4,6,0}, and a list that stopped at OpenGL40 told an
|
||||
// application feature-detecting off these tokens the opposite of what
|
||||
// GL_MAJOR_VERSION / GL_MINOR_VERSION told it.
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, V_OpenGL41, V_OpenGL42, V_OpenGL43,
|
||||
V_OpenGL44, V_OpenGL45, V_OpenGL46,
|
||||
E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_draw_indirect,
|
||||
@@ -623,7 +625,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (nonZeroIndirectBaseInstanceSupported) {
|
||||
extensions.push_back(E_GL_ARB_base_instance);
|
||||
}
|
||||
if (shaderSubgroupSupported && !MG_Config::Features.DisableSubgroup) {
|
||||
if (shaderSubgroupSupported && !MG_Config::Features.MagmaDisableSubgroup) {
|
||||
extensions.push_back(E_GL_KHR_shader_subgroup);
|
||||
}
|
||||
// GL_KHR_parallel_shader_compile is MobileGL's own capability, not the Vulkan
|
||||
@@ -739,8 +741,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
funcsTable.GL.MemoryBarrierByRegion = MemoryBarrierByRegion;
|
||||
funcsTable.GL.BindImageTexture = BindImageTexture;
|
||||
funcsTable.GL.GetIntegeri_v = GetIntegeri_v;
|
||||
funcsTable.GL.GetInteger64i_v = GetInteger64i_v;
|
||||
funcsTable.GL.GetProgramiv = GetProgramiv;
|
||||
funcsTable.GL.ShaderStorageBlockBinding = ShaderStorageBlockBinding;
|
||||
funcsTable.GL.FenceSync = FenceSync;
|
||||
funcsTable.GL.ClientWaitSync = ClientWaitSync;
|
||||
@@ -784,8 +784,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_vulkanCaps = capabilities;
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
if (MG_State::pGLContext) {
|
||||
MG_State::pGLContext->InvalidateCompileEnv();
|
||||
if (MGB_CTX_LIVE) {
|
||||
MGB_CTX->InvalidateCompileEnv();
|
||||
}
|
||||
MutableFormatCapabilities().Clear();
|
||||
}
|
||||
@@ -855,6 +855,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
static constexpr SizeT kMaxAdvertisedShaderStorageBlockSize = 512ull * 1024ull * 1024ull;
|
||||
m_dynamicParameters.UniformBufferOffsetAlignment = m_vulkanCaps.UniformBufferOffsetAlignment;
|
||||
m_dynamicParameters.ShaderStorageBufferOffsetAlignment = m_vulkanCaps.ShaderStorageBufferOffsetAlignment;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMin = m_vulkanCaps.AliasedLineWidthRangeMin;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMax = m_vulkanCaps.AliasedLineWidthRangeMax;
|
||||
// Without the samplerAnisotropy feature the limit is unusable, so report 1.0 (no anisotropy)
|
||||
@@ -937,6 +938,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
clampLimit("GL_MAX_COMPUTE_UNIFORM_BLOCKS", m_vulkanCaps.MaxComputeUniformBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxComputeWorkGroupInvocations = m_vulkanCaps.MaxComputeWorkGroupInvocations;
|
||||
// The six per-axis compute limits, from the same VkPhysicalDeviceLimits fields
|
||||
// GLFunctionsTable::GetIntegeri_v (DirectVulkan.cpp) reads live. Carried here so that
|
||||
// MGPCaps has them once the table entry retires (plan B section 4.4.1); GL_Getter floors
|
||||
// them. Not clamped: unlike the block counts these are not amounts an application
|
||||
// allocates, and the frontend already raises them to the GL minimum.
|
||||
for (SizeT axis = 0; axis < 3; ++axis) {
|
||||
m_dynamicParameters.MaxComputeWorkGroupCount[axis] = m_vulkanCaps.MaxComputeWorkGroupCount[axis];
|
||||
m_dynamicParameters.MaxComputeWorkGroupSize[axis] = m_vulkanCaps.MaxComputeWorkGroupSize[axis];
|
||||
}
|
||||
m_dynamicParameters.MaxShaderStorageBufferBindings =
|
||||
clampLimit("GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS", m_vulkanCaps.MaxShaderStorageBufferBindings,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
@@ -1004,6 +1014,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// it the limit describes a capacity no shader may use, so report none.
|
||||
m_dynamicParameters.MaxClipDistances =
|
||||
m_vulkanCaps.SupportsShaderClipDistance ? std::max(m_vulkanCaps.MaxClipDistances, 0) : 0;
|
||||
// The cull pair, gated on its own feature. shaderCullDistance is separate from
|
||||
// shaderClipDistance and VulkanRenderer enables it independently, so it gets its own
|
||||
// gate rather than riding on the clip one.
|
||||
m_dynamicParameters.MaxCullDistances =
|
||||
m_vulkanCaps.SupportsShaderCullDistance ? std::max(m_vulkanCaps.MaxCullDistances, 0) : 0;
|
||||
// GL 4.6 core 11.1.3.10: the combined limit is at least as large as either half. A device
|
||||
// with only one of the two features must not report a combined capacity that implies the
|
||||
// other, so the gate is "either feature" and the value never drops below what is enabled.
|
||||
m_dynamicParameters.MaxCombinedClipAndCullDistances =
|
||||
(m_vulkanCaps.SupportsShaderClipDistance || m_vulkanCaps.SupportsShaderCullDistance)
|
||||
? std::max({m_vulkanCaps.MaxCombinedClipAndCullDistances, m_dynamicParameters.MaxClipDistances,
|
||||
m_dynamicParameters.MaxCullDistances})
|
||||
: 0;
|
||||
m_dynamicParameters.MaxViewports = m_vulkanCaps.MaxViewports;
|
||||
// Assigned explicitly rather than left to the struct's defaults, like every other
|
||||
// parameter here, so a second fill cannot inherit a stale value. GL_UNDEFINED_VERTEX is
|
||||
@@ -1066,6 +1089,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// report VK_FALSE, so on every real mobile device this is false and the demotion runs
|
||||
// exactly as it always has.
|
||||
m_dynamicParameters.SupportsShaderFloat64 = m_vulkanCaps.SupportsShaderFloat64;
|
||||
// shaderTessellationAndGeometryPointSize, both stage families from the one feature.
|
||||
// False arms the shared phase-B point-size demotion, whose modules then carry no
|
||||
// TessellationPointSize/GeometryPointSize capability and build without the feature.
|
||||
// MOBILEGL_POINT_SIZE_DEMOTION=1 pretends it is absent so the demotion can be
|
||||
// exercised on a healthy driver (lavapipe advertises the feature); =0 restores the
|
||||
// detected answer's declines.
|
||||
{
|
||||
Bool supportsStagePointSize = m_vulkanCaps.SupportsTessellationAndGeometryPointSize;
|
||||
switch (MG_Config::Features.PointSizeDemotion) {
|
||||
case MG_Config::QuirkOverride::ForceOn:
|
||||
MGLOG_I("DirectVulkan: MOBILEGL_POINT_SIZE_DEMOTION=1 - treating tessellation/geometry "
|
||||
"gl_PointSize as unhosted so the demotion runs on this driver");
|
||||
supportsStagePointSize = false;
|
||||
break;
|
||||
case MG_Config::QuirkOverride::ForceOff:
|
||||
MGLOG_I("DirectVulkan: MOBILEGL_POINT_SIZE_DEMOTION=0 - keeping the built-in and the "
|
||||
"plain declines regardless of the device feature");
|
||||
supportsStagePointSize = true;
|
||||
break;
|
||||
case MG_Config::QuirkOverride::Auto:
|
||||
break;
|
||||
}
|
||||
m_dynamicParameters.SupportsTessellationPointSize = supportsStagePointSize;
|
||||
m_dynamicParameters.SupportsGeometryPointSize = supportsStagePointSize;
|
||||
}
|
||||
// Never, on any device, and DELIBERATELY NOT COUPLED to the line above even though it
|
||||
// once tracked the same feature. It used to, because a `dvec` input needed Float64 to
|
||||
// exist in the module at all; a 64-bit vertex FETCH was already impossible
|
||||
|
||||
@@ -70,7 +70,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const RendererInfo& GetRendererIdentity();
|
||||
|
||||
// The full OpenGL extension list Magma advertises (glGetString(GL_EXTENSIONS)) for
|
||||
// a device with the given raw capabilities. The MOBILEGL_DISABLE_SUBGROUP and
|
||||
// a device with the given raw capabilities. The MOBILEGL_MAGMA_DISABLE_SUBGROUP and
|
||||
// MOBILEGL_DISABLE_TIMERQUERY escape hatches are applied inside, so callers pass
|
||||
// the detected device support (passing an already-gated value is harmless).
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
|
||||
|
||||
@@ -10,9 +10,11 @@
|
||||
#include "DirectVulkanResourceState.h"
|
||||
#include "MG_Backend/BackendObjects.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include "MG_State/GLState/ErrorState/ErrorInfo.h"
|
||||
#include "MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h"
|
||||
#include "MG_Util/Converters/GLToMG/TextureEnumConverter.h"
|
||||
#include "MG_Util/Metrics/PipeStats.h"
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include "MG_Util/Miscellany/IndexGenerator.h"
|
||||
#include <atomic>
|
||||
@@ -77,7 +79,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 blockBindingVersion = 0;
|
||||
Vector<StorageBlockResource> storageBlocks;
|
||||
Vector<BufferVariableResource> bufferVariables;
|
||||
GLint computeWorkGroupSize[3] = {1, 1, 1};
|
||||
};
|
||||
|
||||
struct DrawElementsIndirectCommand {
|
||||
@@ -208,16 +209,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
for (auto& module : modules) {
|
||||
for (Uint32 entryIndex = 0; entryIndex < module.entry_point_count; ++entryIndex) {
|
||||
const auto& entryPoint = module.entry_points[entryIndex];
|
||||
if ((entryPoint.shader_stage & SPV_REFLECT_SHADER_STAGE_COMPUTE_BIT) == 0) {
|
||||
continue;
|
||||
}
|
||||
cache.computeWorkGroupSize[0] = static_cast<GLint>(std::max<Uint32>(entryPoint.local_size.x, 1));
|
||||
cache.computeWorkGroupSize[1] = static_cast<GLint>(std::max<Uint32>(entryPoint.local_size.y, 1));
|
||||
cache.computeWorkGroupSize[2] = static_cast<GLint>(std::max<Uint32>(entryPoint.local_size.z, 1));
|
||||
}
|
||||
|
||||
uint32_t bindingCount = 0;
|
||||
SpvReflectResult result = spvReflectEnumerateDescriptorBindings(&module, &bindingCount, nullptr);
|
||||
if (result != SPV_REFLECT_RESULT_SUCCESS || bindingCount == 0) {
|
||||
@@ -277,15 +268,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
MG_State::GLState::ProgramObject* TryGetDirectVulkanProgram(GLuint program) {
|
||||
if (!MG_State::pGLContext->ValidateProgramName(program)) {
|
||||
if (!MGB_CTX->ValidateProgramName(program)) {
|
||||
return nullptr;
|
||||
}
|
||||
auto& programObject = MG_State::pGLContext->GetProgramObject(program);
|
||||
auto& programObject = MGB_CTX->GetProgramObject(program);
|
||||
return programObject.get();
|
||||
}
|
||||
|
||||
const Uint8* ResolveIndirectCommandBytes(const void* indirect, SizeT requiredBytes, const char* label) {
|
||||
auto drawBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
auto drawBuffer = MGB_CTX->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (drawBuffer) {
|
||||
drawBuffer->SyncPersistentMappedRange();
|
||||
const SizeT commandOffset = reinterpret_cast<SizeT>(indirect);
|
||||
@@ -344,64 +335,64 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void ClearBufferfi(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearBufferfi called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearBufferfi called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearBufferfi called with null GL context");
|
||||
pVulkanRenderer->ClearBufferfi(buffer, drawbuffer, depth, stencil);
|
||||
}
|
||||
|
||||
void ClearBufferfv(GLenum buffer, GLint drawbuffer, const GLfloat* value) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearBufferfv called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearBufferfv called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearBufferfv called with null GL context");
|
||||
pVulkanRenderer->ClearBufferfv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearBufferuiv(GLenum buffer, GLint drawbuffer, const GLuint* value) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearBufferuiv called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearBufferuiv called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearBufferuiv called with null GL context");
|
||||
pVulkanRenderer->ClearBufferuiv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearBufferiv(GLenum buffer, GLint drawbuffer, const GLint* value) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearBufferiv called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearBufferiv called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearBufferiv called with null GL context");
|
||||
pVulkanRenderer->ClearBufferiv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearNamedFramebufferfv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, const GLfloat* value) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearNamedFramebufferfv called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearNamedFramebufferfv called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearNamedFramebufferfv called with null GL context");
|
||||
pVulkanRenderer->ClearNamedFramebufferfv(framebuffer, buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearNamedFramebufferiv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, const GLint* value) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearNamedFramebufferiv called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearNamedFramebufferiv called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearNamedFramebufferiv called with null GL context");
|
||||
pVulkanRenderer->ClearNamedFramebufferiv(framebuffer, buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearNamedFramebufferuiv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, const GLuint* value) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearNamedFramebufferuiv called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearNamedFramebufferuiv called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearNamedFramebufferuiv called with null GL context");
|
||||
pVulkanRenderer->ClearNamedFramebufferuiv(framebuffer, buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearNamedFramebufferfi(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, GLfloat depth, GLint stencil) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ClearNamedFramebufferfi called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ClearNamedFramebufferfi called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ClearNamedFramebufferfi called with null GL context");
|
||||
pVulkanRenderer->ClearNamedFramebufferfi(framebuffer, buffer, drawbuffer, depth, stencil);
|
||||
}
|
||||
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawElementsIndirect called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawElementsIndirect called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawElementsIndirect called with null GL context");
|
||||
pVulkanRenderer->MultiDrawElementsIndirect(mode, type, indirect, drawcount, stride);
|
||||
}
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawArraysIndirect called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawArraysIndirect called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawArraysIndirect called with null GL context");
|
||||
|
||||
if (drawcount <= 0) {
|
||||
return;
|
||||
@@ -409,7 +400,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
// With a bound GL_DRAW_INDIRECT_BUFFER the command parameters may be GPU-written
|
||||
// (e.g. by a compute shader), so consume them natively on the GPU.
|
||||
auto drawBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
auto drawBuffer = MGB_CTX->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (drawBuffer) {
|
||||
pVulkanRenderer->MultiDrawArraysIndirect(mode, indirect, drawcount, stride);
|
||||
return;
|
||||
@@ -452,13 +443,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawElementsIndirectCount called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawElementsIndirectCount called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawElementsIndirectCount called with null GL context");
|
||||
pVulkanRenderer->MultiDrawElementsIndirectCount(mode, type, indirect, drawcount, maxdrawcount, stride);
|
||||
}
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawArraysIndirectCount called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawArraysIndirectCount called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawArraysIndirectCount called with null GL context");
|
||||
|
||||
if (maxdrawcount <= 0) {
|
||||
return;
|
||||
@@ -472,7 +463,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
|
||||
auto parameterBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
auto parameterBuffer = MGB_CTX->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
if (!parameterBuffer || drawcount < 0 || static_cast<SizeT>(drawcount) + sizeof(Uint32) > parameterBuffer->GetSize()) {
|
||||
MGLOG_E_ONCE("MultiDrawArraysIndirectCount skipped: invalid GL_PARAMETER_BUFFER binding or range");
|
||||
return;
|
||||
@@ -503,7 +494,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void DrawElementsInstancedBaseVertexBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLint basevertex, GLuint baseinstance) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawElementsInstancedBaseVertexBaseInstance called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawElementsInstancedBaseVertexBaseInstance called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawElementsInstancedBaseVertexBaseInstance called with null GL context");
|
||||
|
||||
DrawIndexedCmd payload{};
|
||||
payload.mode = mode;
|
||||
@@ -530,7 +521,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
void DrawElementsIndirect(GLenum mode, GLenum type, const void* indirect) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawElementsIndirect called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawElementsIndirect called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawElementsIndirect called with null GL context");
|
||||
|
||||
const SizeT indexSize = MG_Util::GetGLTypeSize(type);
|
||||
if (indexSize == 0) {
|
||||
@@ -540,7 +531,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
// With a bound GL_DRAW_INDIRECT_BUFFER the command parameters may be GPU-written
|
||||
// (e.g. by a compute shader), so consume them natively on the GPU.
|
||||
auto drawBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
auto drawBuffer = MGB_CTX->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (drawBuffer) {
|
||||
pVulkanRenderer->MultiDrawElementsIndirect(mode, type, indirect, 1, 0);
|
||||
return;
|
||||
@@ -574,7 +565,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void DrawArraysInstancedBaseInstance(GLenum mode, GLint first, GLsizei count, GLsizei instancecount,
|
||||
GLuint baseinstance) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawArraysInstancedBaseInstance called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawArraysInstancedBaseInstance called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawArraysInstancedBaseInstance called with null GL context");
|
||||
|
||||
DrawCmd payload{};
|
||||
payload.mode = mode;
|
||||
@@ -589,11 +580,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
void DrawArraysIndirect(GLenum mode, const void* indirect) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawArraysIndirect called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawArraysIndirect called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawArraysIndirect called with null GL context");
|
||||
|
||||
// With a bound GL_DRAW_INDIRECT_BUFFER the command parameters may be GPU-written
|
||||
// (e.g. by a compute shader), so consume them natively on the GPU.
|
||||
auto drawBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
auto drawBuffer = MGB_CTX->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (drawBuffer) {
|
||||
pVulkanRenderer->MultiDrawArraysIndirect(mode, indirect, 1, 0);
|
||||
return;
|
||||
@@ -623,13 +614,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void CopyTexImage2D(GLenum target, GLint level, GLenum internalformat, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height, GLint border) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::CopyTexImage2D called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::CopyTexImage2D called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::CopyTexImage2D called with null GL context");
|
||||
pVulkanRenderer->CopyTexSubImage2D(target, level, 0, 0, x, y, width, height);
|
||||
}
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::CopyTexSubImage2D called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::CopyTexSubImage2D called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::CopyTexSubImage2D called with null GL context");
|
||||
pVulkanRenderer->CopyTexSubImage2D(target, level, xoffset, yoffset, x, y, width, height);
|
||||
}
|
||||
void CopyImageSubData(const CopyImageEndpoint& src,
|
||||
@@ -638,32 +629,32 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::CopyImageSubData called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::CopyImageSubData called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::CopyImageSubData called with null GL context");
|
||||
pVulkanRenderer->CopyImageSubData(src, srcTarget, srcLevel, srcX, srcY, srcZ,
|
||||
dst, dstTarget, dstLevel, dstX, dstY, dstZ,
|
||||
srcWidth, srcHeight, srcDepth);
|
||||
}
|
||||
void GenerateMipmap(GLenum target) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::GenerateMipmap called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::GenerateMipmap called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::GenerateMipmap called with null GL context");
|
||||
pVulkanRenderer->GenerateMipmap(target);
|
||||
}
|
||||
|
||||
void DispatchCompute(GLuint numGroupsX, GLuint numGroupsY, GLuint numGroupsZ) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DispatchCompute called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DispatchCompute called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DispatchCompute called with null GL context");
|
||||
pVulkanRenderer->DispatchCompute(numGroupsX, numGroupsY, numGroupsZ);
|
||||
}
|
||||
|
||||
void DispatchComputeIndirect(GLintptr indirect) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DispatchComputeIndirect called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DispatchComputeIndirect called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DispatchComputeIndirect called with null GL context");
|
||||
pVulkanRenderer->DispatchComputeIndirect(indirect);
|
||||
}
|
||||
|
||||
void MemoryBarrier(GLbitfield barriers) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MemoryBarrier called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MemoryBarrier called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MemoryBarrier called with null GL context");
|
||||
pVulkanRenderer->MemoryBarrier(barriers);
|
||||
}
|
||||
|
||||
@@ -682,130 +673,40 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
(void)format;
|
||||
}
|
||||
|
||||
// The two compute limits are the only indexed pnames a backend genuinely owns: they come
|
||||
// from the physical device, and MG_Impl/GLImpl/Getter/GL_Getter.cpp asks for them here so it
|
||||
// can raise the answer to the GL required minimum. The same six numbers are carried in
|
||||
// DynamicBackendParameters::MaxComputeWorkGroupCount/Size (filled at capability init from
|
||||
// the same limits), which is their MGPCaps carrier once this entry retires - the
|
||||
// AdvertisedLimitsScenario pins the two against each other. Every other indexed pname names FRONTEND
|
||||
// state (the indexed buffer bindings, the per-unit texture/sampler bindings, the image-unit
|
||||
// bindings, the viewport rectangles, the indexed capabilities) and is answered there before
|
||||
// the table is consulted, so the arms this function used to carry for
|
||||
// GL_SHADER_STORAGE_BUFFER_* and GL_IMAGE_BINDING_* were unreachable duplicates - and not
|
||||
// even faithful ones: the frontend reports the range glBindBufferRange was ASKED for,
|
||||
// verbatim, while these clamped it to the buffer's current storage.
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data) {
|
||||
if (!data) return;
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::GetIntegeri_v called with null VulkanRenderer");
|
||||
if (index >= 3) {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
switch (target) {
|
||||
case GL_MAX_COMPUTE_WORK_GROUP_COUNT:
|
||||
if (index >= 3) {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
*data = static_cast<GLint>(
|
||||
pVulkanRenderer->GetPhysicalDevice().properties.limits.maxComputeWorkGroupCount[index]);
|
||||
return;
|
||||
case GL_MAX_COMPUTE_WORK_GROUP_SIZE:
|
||||
if (index >= 3) {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
*data = static_cast<GLint>(
|
||||
pVulkanRenderer->GetPhysicalDevice().properties.limits.maxComputeWorkGroupSize[index]);
|
||||
return;
|
||||
case GL_SHADER_STORAGE_BUFFER_BINDING: {
|
||||
auto& point = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::ShaderStorage, index);
|
||||
auto& obj = point.GetBoundObject();
|
||||
*data = obj ? static_cast<GLint>(obj->GetExternalIndex()) : 0;
|
||||
return;
|
||||
}
|
||||
case GL_SHADER_STORAGE_BUFFER_START: {
|
||||
auto& point = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::ShaderStorage, index);
|
||||
*data = static_cast<GLint>(point.GetRange().start);
|
||||
return;
|
||||
}
|
||||
case GL_SHADER_STORAGE_BUFFER_SIZE: {
|
||||
auto& point = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::ShaderStorage, index);
|
||||
auto& obj = point.GetBoundObject();
|
||||
if (!obj) {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
const auto& range = point.GetRange();
|
||||
const auto start = std::min(range.start, obj->GetSize());
|
||||
const auto end = std::min(range.end, obj->GetSize());
|
||||
*data = static_cast<GLint>(end - start);
|
||||
return;
|
||||
}
|
||||
case GL_IMAGE_BINDING_NAME:
|
||||
case GL_IMAGE_BINDING_LEVEL:
|
||||
case GL_IMAGE_BINDING_LAYERED:
|
||||
case GL_IMAGE_BINDING_LAYER:
|
||||
case GL_IMAGE_BINDING_ACCESS:
|
||||
case GL_IMAGE_BINDING_FORMAT: {
|
||||
if (index >= MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS) {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
auto& imageBinding = MG_State::pGLContext->GetImageTextureBinding(static_cast<Int>(index));
|
||||
if (target == GL_IMAGE_BINDING_NAME) {
|
||||
*data = imageBinding.Texture ? static_cast<GLint>(imageBinding.Texture->GetExternalIndex()) : 0;
|
||||
} else if (target == GL_IMAGE_BINDING_LEVEL) {
|
||||
*data = imageBinding.Level;
|
||||
} else if (target == GL_IMAGE_BINDING_LAYERED) {
|
||||
*data = imageBinding.Layered;
|
||||
} else if (target == GL_IMAGE_BINDING_LAYER) {
|
||||
*data = imageBinding.Layer;
|
||||
} else if (target == GL_IMAGE_BINDING_ACCESS) {
|
||||
*data = static_cast<GLint>(imageBinding.Access);
|
||||
} else {
|
||||
*data = static_cast<GLint>(imageBinding.Format);
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data) {
|
||||
if (!data) return;
|
||||
switch (target) {
|
||||
case GL_SHADER_STORAGE_BUFFER_START: {
|
||||
auto& point = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::ShaderStorage, index);
|
||||
*data = static_cast<GLint64>(point.GetRange().start);
|
||||
return;
|
||||
}
|
||||
case GL_SHADER_STORAGE_BUFFER_SIZE: {
|
||||
auto& point = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::ShaderStorage, index);
|
||||
auto& obj = point.GetBoundObject();
|
||||
if (!obj) {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
const auto& range = point.GetRange();
|
||||
const auto start = std::min(range.start, obj->GetSize());
|
||||
const auto end = std::min(range.end, obj->GetSize());
|
||||
*data = static_cast<GLint64>(end - start);
|
||||
return;
|
||||
}
|
||||
default:
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
void GetProgramiv(GLuint program, GLenum pname, GLint* params) {
|
||||
if (!params) return;
|
||||
auto* programObject = TryGetDirectVulkanProgram(program);
|
||||
if (!programObject) {
|
||||
params[0] = 0;
|
||||
return;
|
||||
}
|
||||
switch (pname) {
|
||||
case GL_COMPUTE_WORK_GROUP_SIZE: {
|
||||
auto& cache = GetProgramResourceCache(*programObject);
|
||||
params[0] = cache.computeWorkGroupSize[0];
|
||||
params[1] = cache.computeWorkGroupSize[1];
|
||||
params[2] = cache.computeWorkGroupSize[2];
|
||||
return;
|
||||
}
|
||||
default:
|
||||
params[0] = 0;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
void ShaderStorageBlockBinding(GLuint program, const GLchar* storageBlockName, GLuint storageBlockBinding) {
|
||||
auto* programObject = TryGetDirectVulkanProgram(program);
|
||||
if (!programObject || storageBlockName == nullptr) return;
|
||||
@@ -813,7 +714,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
? pActiveBackendObject->GetDynamicParameters().MaxShaderStorageBufferBindings
|
||||
: 0;
|
||||
if (storageBlockBinding >= static_cast<GLuint>(maxBindings)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
MGB_CTX->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("DirectVulkan", __func__, "Shader storage binding is out of range."));
|
||||
return;
|
||||
@@ -838,24 +739,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::ReadPixels called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::ReadPixels called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::ReadPixels called with null GL context");
|
||||
pVulkanRenderer->ReadPixels(x, y, width, height, format, type, pixels);
|
||||
}
|
||||
void GetTexImage(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::GetTexImage called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::GetTexImage called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::GetTexImage called with null GL context");
|
||||
pVulkanRenderer->GetTexImage(target, level, format, type, pixels);
|
||||
}
|
||||
void GetTextureImage(const SharedPtr<MG_State::GLState::ITextureObject>& texture, TextureUploadTarget uploadTarget,
|
||||
GLint level, GLenum format, GLenum type, GLsizei bufSize, GLvoid* pixels) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::GetTextureImage called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::GetTextureImage called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::GetTextureImage called with null GL context");
|
||||
pVulkanRenderer->GetTextureImage(texture, uploadTarget, level, format, type, bufSize, pixels);
|
||||
}
|
||||
|
||||
void Clear(GLbitfield mask) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::Clear called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::Clear called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::Clear called with null GL context");
|
||||
pVulkanRenderer->Clear(mask);
|
||||
}
|
||||
|
||||
@@ -883,7 +784,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
const Uint8* indexBytes = nullptr;
|
||||
const auto& vao = *MG_State::pGLContext->GetBoundVertexArray();
|
||||
const auto& vao = *MGB_CTX->GetBoundVertexArray();
|
||||
const auto& indexBufferShared = vao.GetIndexBufferBindingSlot().GetBoundObject();
|
||||
if (indexBufferShared != nullptr) {
|
||||
const SizeT offset = reinterpret_cast<SizeT>(indices);
|
||||
@@ -914,7 +815,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void DrawArrays(GLenum mode, GLint first, GLsizei count) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawArrays called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawArrays called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawArrays called with null GL context");
|
||||
|
||||
if (mode == GL_LINE_LOOP) {
|
||||
if (count < 2) {
|
||||
@@ -939,7 +840,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void DrawElements(GLenum mode, GLsizei count, GLenum type, const void* indices) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawElements called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawElements called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawElements called with null GL context");
|
||||
|
||||
if (mode == GL_LINE_LOOP) {
|
||||
Vector<Uint32> closedIndices;
|
||||
@@ -962,7 +863,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawArrays called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawArrays called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawArrays called with null GL context");
|
||||
if (drawcount <= 0) {
|
||||
return;
|
||||
}
|
||||
@@ -1007,7 +908,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// MultiDrawIndexedCmd left the client-memory shape addressing a view whose byte
|
||||
// offset is a hardcoded 0, so UploadAndBindIndexBuffer saw a null client pointer,
|
||||
// declined the whole batch and painted nothing.)
|
||||
const auto& vao = *MG_State::pGLContext->GetBoundVertexArray();
|
||||
const auto& vao = *MGB_CTX->GetBoundVertexArray();
|
||||
if (vao.GetIndexBufferBindingSlot().GetBoundObject() == nullptr) {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) {
|
||||
@@ -1068,13 +969,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void MultiDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawElements called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawElements called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawElements called with null GL context");
|
||||
MultiDrawElementsImpl(mode, count, type, indices, drawcount, nullptr);
|
||||
}
|
||||
|
||||
void DrawElementsBaseVertex(GLenum mode, GLsizei count, GLenum type, const GLvoid* indices, GLint basevertex) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::DrawElementsBaseVertex called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::DrawElementsBaseVertex called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::DrawElementsBaseVertex called with null GL context");
|
||||
if (mode == GL_LINE_LOOP) {
|
||||
Vector<Uint32> closedIndices;
|
||||
if (BuildClosedLineLoopIndices(count, type, indices, closedIndices)) {
|
||||
@@ -1098,14 +999,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void MultiDrawElementsBaseVertex(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::MultiDrawElementsBaseVertex called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::MultiDrawElementsBaseVertex called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::MultiDrawElementsBaseVertex called with null GL context");
|
||||
MultiDrawElementsImpl(mode, count, type, indices, drawcount, basevertex);
|
||||
}
|
||||
|
||||
void BlitFramebuffer(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1, GLint dstX0, GLint dstY0, GLint dstX1,
|
||||
GLint dstY1, GLbitfield mask, GLenum filter) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::BlitFramebuffer called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::BlitFramebuffer called with null GL context");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "DirectVulkan::BlitFramebuffer called with null GL context");
|
||||
pVulkanRenderer->BlitFramebuffer(srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1, dstY1, mask, filter);
|
||||
}
|
||||
|
||||
@@ -1206,6 +1107,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
SharedPtr<VkTimerQueryManager::TimestampRecord> end;
|
||||
// Kind::Occlusion - pool slots recorded between Begin/End; summed at result time.
|
||||
Vector<Uint32> occlusionSlots;
|
||||
// Kind::XfbGenerated - reroute-pool slots for the span's XFB-INACTIVE
|
||||
// draws, where the renderer's reroute is armed (the affected driver's
|
||||
// stream query counts nothing without an open capture; see
|
||||
// VulkanRenderer::BeginXfbQueryForDraw). Summed alongside the stream
|
||||
// slots above, which keep the span's XFB-active draws.
|
||||
Vector<Uint32> rerouteSlots;
|
||||
// Renderer generation the records were written under (see
|
||||
// g_rendererGeneration). A stale generation resolves as available
|
||||
// with a final zero result: the records' pool indices and frame
|
||||
@@ -1215,11 +1122,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// stale queries are always safe to delete.
|
||||
Uint64 rendererGeneration = 0;
|
||||
// Kind::XfbGenerated - the frontend's paused-draw primitive counter when the
|
||||
// query began. VK_QUERY_TYPE_TRANSFORM_FEEDBACK_STREAM_EXT counts only what the
|
||||
// capture saw, so a draw made while the span was paused is invisible to it -
|
||||
// but GL_PRIMITIVES_GENERATED counts what the last vertex processing stage
|
||||
// emitted regardless. The delta closes that gap at result time.
|
||||
// query began. On the affected drivers VK_QUERY_TYPE_TRANSFORM_FEEDBACK_STREAM_EXT
|
||||
// counts only what the capture saw, so a draw made while the span was paused is
|
||||
// invisible to it - but GL_PRIMITIVES_GENERATED counts what the last vertex
|
||||
// processing stage emitted regardless. The delta closes that gap at result time.
|
||||
Uint64 pausedPrimitiveSnapshot = 0;
|
||||
// ...unless the GPU already counted those paused draws when the span opened -
|
||||
// through the reroute pool (VulkanRenderer::BeginXfbQueryForDraw reroutes every
|
||||
// draw with no open capture, paused ones included) or, where the probe measured
|
||||
// the stream query as counting capture-less draws, through the stream slot the
|
||||
// paused draw still takes. Adding the CPU delta on top would count them twice,
|
||||
// and the CPU counter is the weaker source anyway: only 3 of the ~15 draw entry
|
||||
// points write it and it answers 0 for GL_PATCHES.
|
||||
Bool pausedPrimitivesCountedByGpu = false;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
@@ -1313,13 +1228,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (query->kind == VulkanTimerQuery::Kind::XfbWritten ||
|
||||
query->kind == VulkanTimerQuery::Kind::XfbGenerated) {
|
||||
Uint64 primitives = 0;
|
||||
if (!pVulkanRenderer->ResolveXfbQueryResult(query->occlusionSlots,
|
||||
if (!pVulkanRenderer->ResolveXfbQueryResult(query->occlusionSlots, query->rerouteSlots,
|
||||
query->kind == VulkanTimerQuery::Kind::XfbGenerated,
|
||||
primitives)) {
|
||||
return false;
|
||||
}
|
||||
if (query->kind == VulkanTimerQuery::Kind::XfbGenerated && MG_State::pGLContext != nullptr) {
|
||||
primitives += MG_State::pGLContext->GetTransformFeedbackPausedPrimitiveCounter() -
|
||||
if (query->kind == VulkanTimerQuery::Kind::XfbGenerated &&
|
||||
!query->pausedPrimitivesCountedByGpu && MGB_CTX_LIVE) {
|
||||
primitives += MGB_CTX->GetTransformFeedbackPausedPrimitiveCounter() -
|
||||
query->pausedPrimitiveSnapshot;
|
||||
}
|
||||
*outNanoseconds = primitives;
|
||||
@@ -1366,7 +1282,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
query->kind = generated ? VulkanTimerQuery::Kind::XfbGenerated : VulkanTimerQuery::Kind::XfbWritten;
|
||||
query->rendererGeneration = GetRendererGeneration();
|
||||
query->pausedPrimitiveSnapshot =
|
||||
MG_State::pGLContext ? MG_State::pGLContext->GetTransformFeedbackPausedPrimitiveCounter() : 0;
|
||||
MGB_CTX_LIVE ? MGB_CTX->GetTransformFeedbackPausedPrimitiveCounter() : 0;
|
||||
// Read AFTER StartXfbQueryCapture, which is where a failed reroute-pool creation
|
||||
// disarms: the answer is then what this span will actually do for every draw.
|
||||
query->pausedPrimitivesCountedByGpu = generated && pVulkanRenderer->ArePausedDrawsGpuCounted();
|
||||
return query;
|
||||
}
|
||||
|
||||
@@ -1377,7 +1296,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
pVulkanRenderer->StopXfbQueryCapture(
|
||||
query->kind == VulkanTimerQuery::Kind::XfbGenerated ? 1u : 0u, query->occlusionSlots);
|
||||
query->kind == VulkanTimerQuery::Kind::XfbGenerated ? 1u : 0u, query->occlusionSlots,
|
||||
query->rerouteSlots);
|
||||
}
|
||||
|
||||
BackendQueryHandle BeginOcclusionQuery() {
|
||||
@@ -1411,5 +1331,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void Present() {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::Present called with null VulkanRenderer");
|
||||
pVulkanRenderer->Present();
|
||||
// THE frame boundary for the MGPipe counters, at the backend entry point rather
|
||||
// than inside VulkanRenderer::Present: that function has an early return for the
|
||||
// no-usable-swapchain case, and a suspended frame is still a frame the counters
|
||||
// must close.
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::OnPresent();
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -95,8 +95,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void BindImageTexture(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer, GLenum access,
|
||||
GLenum format);
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data);
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
void GetProgramiv(GLuint program, GLenum pname, GLint* params);
|
||||
void ShaderStorageBlockBinding(GLuint program, const GLchar* storageBlockName, GLuint storageBlockBinding);
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels);
|
||||
void GetTexImage(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels);
|
||||
|
||||
@@ -0,0 +1,603 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/MagmaPipeArms.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
#include <Config.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// kMGPipeSubsystem* - the runtime bitmask's named bits - and MGPipeHandle itself. Both are
|
||||
// header-only constant/POD declarations, and both are push-only, so the pull build's include
|
||||
// graph is unchanged (G1).
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/MGPipeHandles.h>
|
||||
#endif
|
||||
|
||||
#include <cstdlib>
|
||||
|
||||
// Magma's arm selector for the P2 Track H / render-state re-keys (P2 brief D14), and the
|
||||
// {slot, gen} mint the re-keyed sites are written against.
|
||||
//
|
||||
// Two switches decide which arm a re-keyed site runs, and they are NOT the same switch:
|
||||
//
|
||||
// MOBILEGL_PIPE_PUSH (compile) - is the pushed state there to be keyed on at all
|
||||
// Features.PipePush (runtime bitmask) - is THIS subsystem migrated in THIS run
|
||||
// MOBILEGL_PIPE_LEGACY_MEMOS (compile) - is the pre-handle arm compiled beside it
|
||||
// Features.PipeLegacyMemos (runtime) - may the pre-handle arm be ENTERED in this run
|
||||
//
|
||||
// ARCHITECTURE.md 9.6's point: once a handle wave lands, a clear MOBILEGL_PIPE_PUSH bit is
|
||||
// only a valid A/B while the legacy arm is still compiled, because with the bit clear the
|
||||
// backend would otherwise still run the re-keyed code. So a clear bit selects the legacy
|
||||
// arm, and a run that has explicitly disabled the legacy arm may not fall into it.
|
||||
//
|
||||
// D14 spends that last sentence at STARTUP, not per draw: "a Track-H subsystem whose bit is
|
||||
// clear is a startup Fatal{PipeLegacyMemosDisabled}". Nothing in the draw path aborts, and
|
||||
// nothing outside Track H consults the legacy-memo lever at all - see
|
||||
// MagmaPipeValidateSubsystemConfiguration below for both halves of that rule.
|
||||
//
|
||||
// The whole header is inert in a pull build: MOBILEGL_PIPE_PUSH is 0 there, every helper
|
||||
// below is behind it, and the pull build's translation units are byte-identical (G1).
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// Is `subsystemBit` (MG_Pipe/MGPipe.h's kMGPipeSubsystem*) migrated in this run?
|
||||
inline Bool MagmaPipeSubsystemOn(Uint64 subsystemBit) {
|
||||
return (MG_Config::Features.PipePush & subsystemBit) != 0;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D14's startup gate
|
||||
// ---------------------------------------------------------------------------------
|
||||
//
|
||||
// Called once from VulkanRenderer::Initialize(), i.e. only when Magma is the backend
|
||||
// that is actually running. It answers exactly one question and it answers it before the
|
||||
// first draw: is there an arm for Magma's Track-H subsystem in this configuration?
|
||||
//
|
||||
// Three deliberate boundaries, each of which the per-draw shape this replaces got wrong:
|
||||
//
|
||||
// * ONLY Magma's own Track-H bit is checked. Espryt's bit 5 is Espryt's business (a
|
||||
// DirectVulkan run does not execute one line of DirectGLES' re-key), so
|
||||
// MOBILEGL_PIPE_PUSH=0x20 must not kill a Magma run, and MOBILEGL_PIPE_PUSH=0x40 must
|
||||
// not kill an Espryt one.
|
||||
// * bit 0 (kMGPipeSubsystemRenderState) is NOT Track H and is NOT fatal. It is not a
|
||||
// memo re-key at all: it decides where the pipeline memo's STATE KEY comes from, and
|
||||
// a clear bit there simply means the client is not pushing render-state CSOs in this
|
||||
// run, which GetOrCreatePipeline answers with its own state hash. D14 labels bits 5
|
||||
// and 6 "Track H" and labels bit 0 nothing of the sort.
|
||||
// * it is Fatal at STARTUP, once, not on a draw. A per-draw abort inside
|
||||
// GetOrCreatePipeline turns a configuration mistake into a mid-frame crash and puts a
|
||||
// branch nobody needs on the hottest path in the backend.
|
||||
//
|
||||
// [declared deviation from D14, review v2 minor 2] D14's runtime row reads "false: the
|
||||
// legacy arm is never entered", and D14's compile-switch row names ComputePipelineStateHash
|
||||
// as part of the pre-handle arm. Those two together would make MOBILEGL_PIPE_LEGACY_MEMOS=0
|
||||
// with bit 0 CLEAR a contradiction: the pipeline memo has no CSO handle to key on, so it
|
||||
// keys on a state hash, and in a build that compiles the pre-handle arm that hash IS
|
||||
// ComputePipelineStateHash. Magma does not make that fatal - bit 0 is not Track H, and
|
||||
// there is a correct answer (the state hash) where for bits 5/6 there is none - but it no
|
||||
// longer does it SILENTLY: the combination is named once, at startup, right here.
|
||||
inline void MagmaPipeValidateSubsystemConfiguration() {
|
||||
if (!MG_Config::Features.PipeLegacyMemos &&
|
||||
!MagmaPipeSubsystemOn(MG_Pipe::kMGPipeSubsystemRenderState)) {
|
||||
MGLOG_W("MGPipe: MOBILEGL_PIPE_LEGACY_MEMOS=0 with kMGPipeSubsystemRenderState (bit 0 "
|
||||
"of MOBILEGL_PIPE_PUSH) clear - Magma's pipeline memo has no CSO handle to key "
|
||||
"on, so every draw whose pipeline-state version moved runs the pre-handle STATE "
|
||||
"HASH instead. That is not a Track-H subsystem and not fatal, but it is not the "
|
||||
"handle arm either: set bit 0 (MOBILEGL_PIPE_PUSH=0x%llx) if this run was meant "
|
||||
"to measure it.",
|
||||
static_cast<unsigned long long>(MG_Config::Features.PipePush |
|
||||
MG_Pipe::kMGPipeSubsystemRenderState));
|
||||
}
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
// The pre-handle arm is compiled AND the operator has not forbidden entering it, so a
|
||||
// clear bit is an ordinary, valid A/B: the site takes the legacy arm.
|
||||
if (MG_Config::Features.PipeLegacyMemos) return;
|
||||
#endif
|
||||
if (MagmaPipeSubsystemOn(MG_Pipe::kMGPipeSubsystemMagmaVertexInput)) return;
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
const char* const why = "this run has MOBILEGL_PIPE_LEGACY_MEMOS=0";
|
||||
#else
|
||||
const char* const why =
|
||||
"this build has cmake -DMOBILEGL_PIPE_LEGACY_MEMOS=OFF, which compiles no such arm";
|
||||
#endif
|
||||
MGLOG_F("MGPipe: Fatal{PipeLegacyMemosDisabled} Magma's Track-H subsystem "
|
||||
"(kMGPipeSubsystemMagmaVertexInput, bit 6 of MOBILEGL_PIPE_PUSH) is clear, so the "
|
||||
"vertex-input cache and the VAO draw memo want the pre-handle arm - but %s. Set "
|
||||
"bit 6 (MOBILEGL_PIPE_PUSH=0x%llx, or the default 0x%llx), or allow the legacy arm.",
|
||||
why,
|
||||
static_cast<unsigned long long>(MG_Config::Features.PipePush |
|
||||
MG_Pipe::kMGPipeSubsystemMagmaVertexInput),
|
||||
static_cast<unsigned long long>(MG_Pipe::kMGPipeSubsystemsMigratedAtP2));
|
||||
std::abort();
|
||||
}
|
||||
|
||||
// "Does this Track-H site run the handle arm?" - the ONE question every re-keyed Track-H
|
||||
// site asks, so that they cannot disagree with each other or with the startup gate.
|
||||
inline Bool MagmaPipeTrackHArmIsHandles(Uint64 trackHBit) {
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
return MagmaPipeSubsystemOn(trackHBit);
|
||||
#else
|
||||
// No pre-handle arm exists in this build, and MagmaPipeValidateSubsystemConfiguration
|
||||
// has already made a clear bit a startup Fatal, so the handle arm is the only arm a
|
||||
// running process can be on.
|
||||
(void)trackHBit;
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// Negative control C (P2 brief D18): MOBILEGL_PIPE_HANDLE_ABA_CONTROL
|
||||
// ---------------------------------------------------------------------------------
|
||||
//
|
||||
// "Is the object-identity half of every vertex-input memo key deliberately defeated in
|
||||
// this run?" - the ONE question the control's sites ask, for the same reason
|
||||
// MagmaPipeTrackHArmIsHandles exists: three sites deciding separately could disagree,
|
||||
// and a control that defeats two of three guards proves nothing.
|
||||
//
|
||||
// WHAT IT DEFEATS, AND WHY IT IS SPELLED AS "REPLACE THE IDENTITY WITH A CONSTANT"
|
||||
// RATHER THAN "USE THE HEAP ADDRESS".
|
||||
//
|
||||
// D18 wrote the control as "hash attr.Buffer.get() instead of GetLifetimeId(), and skip
|
||||
// the vaoLifetimeId compare", on the theory that a deleted object's replacement lands at
|
||||
// the freed heap block and so reproduces the key. Measured, it does not: in
|
||||
// HandleRecycleScenario the GL NAMES come back (glGen* hands the deleted name straight
|
||||
// out) but the C++ heap blocks do not - a VertexArrayObject is 3920 bytes, too large for
|
||||
// glibc's tcache, so its chunk goes to the unsorted bin and is split by the very next
|
||||
// allocation the replacement path makes. Four create/delete cycles in one run produced
|
||||
// four distinct addresses, ~1 MiB apart. With no address reuse there is nothing for
|
||||
// "hash the address" to collide with: the replacement hashes differently, indexes a
|
||||
// different memo slot, and inherits nothing - so the arm asserted stale pixels and saw
|
||||
// fresh ones, which is a FAILING negative control that had stopped controlling anything.
|
||||
//
|
||||
// So the control no longer asks the allocator for the collision; it manufactures it. On
|
||||
// both arms the object identity is replaced by a constant, which is the strongest form of
|
||||
// "the allocator handed the block back" and is deterministic. That covers strictly more
|
||||
// than D18's spelling, and in particular it reaches the arm P2 SHIPS: on the handle arm
|
||||
// the constant defeats the OBJECT IDENTITY THAT SELECTS THE SLOT - the key the handle arm
|
||||
// ships - so the replacement VAO is handed the dead one's memo entry and its content hash.
|
||||
// Defeating only the retired lifetime-id/address guards would leave that key untested,
|
||||
// which is exactly the vacuity this control exists to catch.
|
||||
//
|
||||
// WHAT IT DOES NOT COVER, AND WHY NO REPRODUCER OF THIS SHAPE CAN [fix-aba review v1,
|
||||
// MAJOR 1]. It does NOT exercise the GENERATION half of {slot, gen}:
|
||||
//
|
||||
// * this mint has no death notification - nothing in MG_Backend/DirectVulkan consumes
|
||||
// NotifyStateObjectDestroyed - so a slot returns to the free list only through
|
||||
// OnFrameBoundary's age sweep (kSweepInterval 256, kRetireAgeBoundaries 1024, below);
|
||||
// * HandleRecycleScenario issues five frame boundaries, so the free list is empty when
|
||||
// the replacement VAO acquires and it gets a BRAND-NEW slot at Gen 1 (measured:
|
||||
// redVao slot=2 gen=1, greenVao slot=3 gen=1). The knob-off FRESH verdict there is
|
||||
// decided by the SLOT alone, and deleting the ++Gen below leaves all four arms green;
|
||||
// * a genuine slot REUSE needs >= 1024 idle boundaries after the dead object's last
|
||||
// draw, which necessarily puts the two draws in different frames - and the only memo
|
||||
// that carries a GPU slice rather than a layout, ResolvedVertexBindings, declines
|
||||
// across frames by design. The two requirements are mutually exclusive, so the
|
||||
// generation is out of reach of any same-frame pixel reproducer for this memo.
|
||||
//
|
||||
// The generation is covered where it IS expressible, over this mint and the claim rule
|
||||
// MagmaPipeClaimSlotMemos below: MG_Test/Pipe/MagmaPipeIdentityTest.cpp drives a real
|
||||
// retire -> reuse and asserts that a memo stamped at {slot, gen=N} is not served at
|
||||
// {slot, gen=N+1} with the knob off and IS served with it on. Deleting the ++Gen reds that
|
||||
// suite; it is the only place in the tree where that deletion is caught.
|
||||
//
|
||||
// Everything the control does NOT defeat is as load-bearing as what it does. It never
|
||||
// touches a guard that is not an IDENTITY guard: the resolved-bindings memo's frame
|
||||
// serial, its slice-epoch compares and its host-map check all stay in force, so a green
|
||||
// AbaControl arm still means "a replacement object was handed its dead predecessor's
|
||||
// resolved vertex bindings because the identity halves of the keys were defeated", not
|
||||
// "every safety net was switched off until something broke".
|
||||
//
|
||||
// Off by default (Config.h), set only by the HandleRecycle AbaControl ctest lanes, and
|
||||
// #if MOBILEGL_PIPE_PUSH throughout, so no shipping pull build can even parse it.
|
||||
// P4a (BRIEF-P4A.md D-I2, G8): WHICH KINDS THIS ANSWER COVERS, and it is not "all of them".
|
||||
//
|
||||
// P4a mints six more client-side kinds - Texture, Renderbuffer, Framebuffer, SamplerCso,
|
||||
// SamplerViewCso and ShaderCso - and requires the ABA control to defeat "the identity half
|
||||
// of P4a's memo keys as well", because a control that only defeats the guards a phase
|
||||
// RETIRED says nothing about the key that phase SHIPS.
|
||||
//
|
||||
// On Magma there is no such key to defeat, and that is a fact about the roadmap rather than
|
||||
// an omission here. MagmaPipeIdentityTables below mints exactly TWO kinds,
|
||||
// VertexElementsCso and Buffer; a texture, a framebuffer, a sampler, a view and a program
|
||||
// are all still reached from their frontend objects on this backend, and moving them onto
|
||||
// handles is P7's work (ROADMAP.md:24 - "Magma anything"; P4a leaves MG_Backend/DirectVulkan
|
||||
// untouched apart from this file). So the honest statement is per KIND, and it is spelled as
|
||||
// code rather than as a comment so that a caller cannot read the blanket answer above and
|
||||
// conclude the knob covers its kind:
|
||||
//
|
||||
// * for the two kinds this backend really keys on {slot, gen}, the knob defeats the
|
||||
// identity exactly as it always has (MagmaPipeClaimSlotMemos);
|
||||
// * for P4a's six there is nothing here to defeat, so the answer is FALSE - and
|
||||
// MG_IntegrationTest's HandleRecycleScenario reads that through its own build probe and
|
||||
// makes those cases' AbaControl arm assert the CORRECT pixels while SAYING that it is
|
||||
// not controlling anything for that kind. It does not assert a corruption that no code
|
||||
// on this tree can produce, which would be a permanently red always-on lane.
|
||||
//
|
||||
// WHAT MAKES IT TRUE LATER, in one sentence, so the next reader does not have to derive it:
|
||||
// when a backend grows a Features.PipeHandleAbaControl consumer over its P4a object slot
|
||||
// tables - one `if` in GetOrCreate / FindByHandle, the shape MagmaPipeClaimSlotMemos already
|
||||
// has for vertex input - this function's per-kind answer becomes that consumer's, the
|
||||
// integration probe finds the consumer, and the six cases flip to expecting the corruption.
|
||||
inline Bool MagmaPipeAbaControlDefeatsIdentity() {
|
||||
return MG_Config::Features.PipeHandleAbaControl;
|
||||
}
|
||||
|
||||
// WHICH KINDS THIS BACKEND ACTUALLY KEYS ON {slot, gen}, and therefore which kinds the knob
|
||||
// above has an identity to defeat at all. `kind` is MG_Pipe::MGPipeKind.
|
||||
//
|
||||
// EXHAUSTIVE, WITH NO `default:`, for MG_IntegrationTest/Harness/PipeSlotPeek.cpp's reason:
|
||||
// a kind added to MGPipeKind without a decision here must be a -Wswitch warning in this
|
||||
// file rather than a row that silently inherits somebody else's answer. Being wrong in the
|
||||
// "covered" direction is the expensive one - a control asserting a corruption nobody can
|
||||
// produce is a permanently red always-on lane - so an undecided kind must never read true,
|
||||
// and with no `default:` there is no arm for it to read true from.
|
||||
//
|
||||
// constexpr AND PINNED BY static_assert BELOW, which is what stops it rotting the way a
|
||||
// predicate with no caller does: MagmaPipeIdentityTables mints exactly two kinds, the
|
||||
// asserts say so in both directions, and the file no longer compiles if the tables and this
|
||||
// statement of them ever part company. (Review F-m5: the earlier form had no caller at all
|
||||
// and could not make anything red or green.)
|
||||
inline constexpr Bool MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind kind) {
|
||||
switch (kind) {
|
||||
// The two MagmaPipeIdentityTables really mints.
|
||||
case MG_Pipe::MGPipeKind::VertexElementsCso:
|
||||
case MG_Pipe::MGPipeKind::Buffer:
|
||||
return true;
|
||||
// P4a's six object classes: still reached from their frontend objects on this
|
||||
// backend (Magma's object paths are P7, ROADMAP.md:24), so there is no key here for
|
||||
// the knob to defeat.
|
||||
case MG_Pipe::MGPipeKind::Texture:
|
||||
case MG_Pipe::MGPipeKind::Renderbuffer:
|
||||
case MG_Pipe::MGPipeKind::Framebuffer:
|
||||
case MG_Pipe::MGPipeKind::SamplerCso:
|
||||
case MG_Pipe::MGPipeKind::SamplerViewCso:
|
||||
case MG_Pipe::MGPipeKind::ShaderCso:
|
||||
// ...and everything else this backend does not mint a handle for.
|
||||
case MG_Pipe::MGPipeKind::None:
|
||||
case MG_Pipe::MGPipeKind::Xfb:
|
||||
case MG_Pipe::MGPipeKind::RenderStateCso:
|
||||
case MG_Pipe::MGPipeKind::Fence:
|
||||
case MG_Pipe::MGPipeKind::Query:
|
||||
case MG_Pipe::MGPipeKind::Context:
|
||||
case MG_Pipe::MGPipeKind::KindCount:
|
||||
return false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static_assert(MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::VertexElementsCso),
|
||||
"MagmaPipeIdentityTables mints VertexElementsCso: the knob has an identity to "
|
||||
"defeat for it");
|
||||
static_assert(MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::Buffer),
|
||||
"MagmaPipeIdentityTables mints Buffer: the knob has an identity to defeat for it");
|
||||
static_assert(!MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::Texture) &&
|
||||
!MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::Renderbuffer) &&
|
||||
!MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::Framebuffer) &&
|
||||
!MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::SamplerCso) &&
|
||||
!MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::SamplerViewCso) &&
|
||||
!MagmaPipeAbaControlKindIsRekeyedHere(MG_Pipe::MGPipeKind::ShaderCso),
|
||||
"P4a's six object classes are not keyed on {slot, gen} on this backend, so "
|
||||
"HandleRecycleScenario's six AbaControl arms must NOT expect a corruption here. "
|
||||
"Wiring one of them is what flips this assert, this predicate and that arm - and "
|
||||
"MG_IntegrationTest's two-symbol probe over MG_Backend/DirectVulkan is what "
|
||||
"carries the answer into the lane");
|
||||
|
||||
// THERE IS DELIBERATELY NO PER-KIND WRAPPER HERE, and review F-v2-m3 is why. An earlier
|
||||
// round carried `MagmaPipeAbaControlCoversKind(kind)` - the conjunction of the two
|
||||
// statements above - and it had no caller anywhere in the tree: the knob's only two
|
||||
// consumers (VulkanRenderer.cpp's VAO draw memo and VertexInputStateFactory.cpp's pipeline
|
||||
// key) each hold ONE kind, VertexElementsCso, by construction, so the kind is not a
|
||||
// variable at either site. A conjunction no build ever evaluates cannot be pinned the way
|
||||
// the predicate above is pinned - it is not constexpr, because it reads MG_Config::Features,
|
||||
// so no static_assert can reach it - which makes it exactly the rot F-m5 was raised about,
|
||||
// one level up: an `&&` whose operands could be inverted or dropped with nothing to say so.
|
||||
//
|
||||
// The two pieces stand alone instead, and each is pinned by something that runs:
|
||||
// MagmaPipeAbaControlKindIsRekeyedHere is constexpr and asserted in BOTH directions by the
|
||||
// three static_asserts above, which compile in every Magma build; MagmaPipeAbaControlDefeats
|
||||
// Identity is the knob, and its two consumers are what make it true or false. A call site
|
||||
// that ever does hold a variable kind writes the `&&` there, where a build will run it.
|
||||
|
||||
// The single consumer-table entry every VAO collapses onto while the control is on. Slot
|
||||
// 0 is a real, ordinary entry of both tables (MagmaPipeSlotIndex maps the first allocatable
|
||||
// handle onto it), so nothing about the tables changes shape for the control's sake.
|
||||
inline constexpr Uint32 kMagmaPipeAbaControlSlotIndex = 0;
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// The {slot, gen} mint
|
||||
// ---------------------------------------------------------------------------------
|
||||
//
|
||||
// Maps a frontend object's never-reused lifetime id to a dense {slot, gen}. Three
|
||||
// properties, and the third is the one review v2 got wrong:
|
||||
//
|
||||
// 1. exact identity - Gen moves whenever a slot changes owner, so a stale handle can
|
||||
// never match a live object even if the allocator hands back the same heap address
|
||||
// (the ABA HandleRecycleScenario reproduces);
|
||||
// 2. dense slots - the slot IS an index, so a consumer's per-slot table needs no hash,
|
||||
// no probe and no mix;
|
||||
// 3. NO CAPACITY CLIFF. A live object's handle never changes while the object is being
|
||||
// drawn, whatever the working set size.
|
||||
//
|
||||
// Property 3 is why this is not the fixed 2-way set-associative LRU the previous round
|
||||
// shipped. That structure evicted a LIVE object once the working set passed its capacity,
|
||||
// and every consumer memo keyed on the handle died with it: measured on a verbatim
|
||||
// transcription, 54% of uses lost their handle at 2500 live VAOs against 2048 entries, and
|
||||
// 20% at 1024 live VAOs once the lifetime ids are sparse (an app that creates and destroys
|
||||
// VAOs, which is the Minecraft chunk shape this exists for). Two of the three memos it
|
||||
// fed - the content-hash memo and the resolved-state memo - had NO capacity before this
|
||||
// package: they were unbounded mutable fields on VertexArrayObject. Introducing eviction
|
||||
// there turns one ComputeHash per VAO reconfiguration into one per DRAW, and, once the
|
||||
// buffer table thrashes too, makes the vertex-input content hash a per-draw value that
|
||||
// inserts a fresh heap-allocated BackendVertexInputState into an unbounded map on every
|
||||
// draw. That is a worse leak than the one it was introduced to avoid.
|
||||
//
|
||||
// So: grow on demand, and reclaim by AGE instead of by capacity.
|
||||
//
|
||||
// * Acquire hits an UnorderedMap<lifetimeId, slotIndex>, in front of which sits a
|
||||
// one-entry memo. Every re-keyed site in a draw asks about the SAME VAO, so the memo
|
||||
// turns the five-or-six acquisitions a draw makes into one map probe plus five Uint64
|
||||
// compares - less than the address multiply plus two-way probe the pre-handle arm ran.
|
||||
// * OnFrameBoundary retires slots whose object has not been drawn for
|
||||
// kRetireAgeBoundaries boundaries and returns them to a free list, so the table's
|
||||
// footprint tracks the LIVE DRAWN working set, not objects ever created. That is the
|
||||
// property MG_Impl/Pipe/SlotAllocator cannot have here: nothing in P2 can call its
|
||||
// Free (the tracker emits no object-class state, BufferBackendOps::OnDestroy is handed
|
||||
// a BackendBufferResource rather than the BufferObject, and VertexArrayObject has no
|
||||
// death hook at all - adding one is D13's explicit-destroy work, which covers Espryt's
|
||||
// six kinds, not VertexElementsCso), so an allocator here would grow by one SlotState
|
||||
// plus one map node per object EVER created, for the life of the process, on a
|
||||
// platform with an LMK. Age-based reclamation is the stand-in for the death
|
||||
// notification, and it is exactly as ABA-proof, because reuse bumps Gen.
|
||||
// * A retire costs at most one memo recompute if the object is drawn again - the same
|
||||
// price a cache miss costs - and it is charged only to objects that went idle for
|
||||
// ~1024 frames, never to a hot one.
|
||||
//
|
||||
// Memory: one map node plus one 24-byte Entry per live object, i.e. tens of bytes against
|
||||
// the kilobyte a VertexArrayObject or a BufferObject already costs the frontend. There is
|
||||
// no capacity to size off a device measurement because there is no capacity; what the
|
||||
// device run in D.4.2 can still want is the number itself, so the high-water mark is
|
||||
// logged at MGLOG_D on the allocate-a-new-slot branch (once per new object, never on a
|
||||
// draw - ROADMAP.md:7).
|
||||
//
|
||||
// Single-threaded, like the rest of the renderer. Owned per VulkanRenderer (see
|
||||
// MagmaPipeIdentityTables): a process-global would share one table, and one reclamation
|
||||
// clock, across two live contexts.
|
||||
class MagmaPipeIdentityTable {
|
||||
public:
|
||||
explicit MagmaPipeIdentityTable(const char* kindName) : m_kindName(kindName) {}
|
||||
|
||||
// Slots ever minted. A consumer table indexed by MagmaPipeSlotIndex() needs this many
|
||||
// entries; MagmaPipeSlotTable below grows itself, so nobody has to ask.
|
||||
Uint32 Count() const { return static_cast<Uint32>(m_entries.size()); }
|
||||
// Objects currently holding a slot - the live working set this table tracks.
|
||||
Uint32 LiveCount() const { return static_cast<Uint32>(m_index.size()); }
|
||||
|
||||
MG_Pipe::MGPipeHandle Acquire(Uint64 lifetimeId) {
|
||||
// Unreachable: MG_State hands out lifetime ids from 1 precisely so that a
|
||||
// zero-initialised memo slot cannot carry a live object's id. Guarded anyway so
|
||||
// that a zero can never be minted into a slot and then indexed with.
|
||||
if (lifetimeId == 0) return MG_Pipe::kMGPipeNullHandle;
|
||||
// The one-entry front memo. Cleared by any retire, so it can never serve a slot
|
||||
// that has been handed back to the free list.
|
||||
if (lifetimeId == m_lastLifetimeId) {
|
||||
m_entries[m_lastIndex].LastUse = m_boundary;
|
||||
return m_lastHandle;
|
||||
}
|
||||
Uint32 index = 0;
|
||||
const auto it = m_index.find(lifetimeId);
|
||||
if (it != m_index.end()) {
|
||||
index = it->second;
|
||||
} else {
|
||||
index = ClaimSlot();
|
||||
m_entries[index].LifetimeId = lifetimeId;
|
||||
m_index.emplace(lifetimeId, index);
|
||||
}
|
||||
Entry& entry = m_entries[index];
|
||||
entry.LastUse = m_boundary;
|
||||
m_lastLifetimeId = lifetimeId;
|
||||
m_lastIndex = index;
|
||||
m_lastHandle = MG_Pipe::MGPipeHandle{index + MG_Pipe::kMGPipeFirstAllocatableSlot,
|
||||
entry.Gen};
|
||||
return m_lastHandle;
|
||||
}
|
||||
|
||||
// Ages the table and returns idle slots to the free list. Same shape and the same
|
||||
// self-gating as VertexInputStateFactory::OnFrameBoundary, which is what the reclaimed
|
||||
// slots' consumers use.
|
||||
void OnFrameBoundary() {
|
||||
++m_boundary;
|
||||
if ((m_boundary % kSweepInterval) != 0) return;
|
||||
SizeT retired = 0;
|
||||
for (auto it = m_index.begin(); it != m_index.end();) {
|
||||
Entry& entry = m_entries[it->second];
|
||||
if ((m_boundary - entry.LastUse) > kRetireAgeBoundaries) {
|
||||
entry.LifetimeId = 0;
|
||||
m_freeSlots.push_back(it->second);
|
||||
it = m_index.erase(it);
|
||||
++retired;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (retired != 0) {
|
||||
// A retired slot's Gen has not moved yet - it moves when the slot is reused -
|
||||
// so a front memo pointing at one would still hand out a handle the consumer
|
||||
// tables would accept. Drop it.
|
||||
m_lastLifetimeId = 0;
|
||||
m_lastHandle = MG_Pipe::kMGPipeNullHandle;
|
||||
MGLOG_D("MagmaPipeIdentityTable(%s): retired %zu idle slots, %u live of %u minted",
|
||||
m_kindName, retired, LiveCount(), Count());
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Sweep cadence and retirement age, deliberately the same numbers
|
||||
// VertexInputStateFactory::OnFrameBoundary uses for the entries these slots key: a slot
|
||||
// retired earlier than its cache entry would mint a new handle for an object whose
|
||||
// entry is still live and still correct, which is a pure waste.
|
||||
static constexpr Uint64 kSweepInterval = 256;
|
||||
static constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
|
||||
struct Entry {
|
||||
Uint64 LifetimeId = 0;
|
||||
Uint64 LastUse = 0;
|
||||
// Moves ONLY on slot reuse, never on respecify: an object that keeps its slot keeps
|
||||
// its generation, which is what makes a memo survive a reconfiguration.
|
||||
Uint32 Gen = 0;
|
||||
};
|
||||
|
||||
Uint32 ClaimSlot() {
|
||||
while (!m_freeSlots.empty()) {
|
||||
const Uint32 index = m_freeSlots.back();
|
||||
m_freeSlots.pop_back();
|
||||
// MGPipeHandles.h:52-58 defends the Gen wrap only in a debug allocator, and
|
||||
// MOBILEGL_ASSERT is compiled out of every build P2 runs (Defines.h: asserts are
|
||||
// live only at MOBILEGL_LOG_ACTIVE_LEVEL == DEBUG). So the wrap is handled on the
|
||||
// RELEASE path instead of asserted: a slot that has been reused 2^32 times is
|
||||
// permanently retired rather than wrapped, because a wrapped Gen would let a
|
||||
// stale handle match a live object. It costs one slot.
|
||||
if (m_entries[index].Gen == ~Uint32{0}) {
|
||||
MGLOG_W("MagmaPipeIdentityTable(%s): slot %u reached generation 2^32-1 and is "
|
||||
"retired for good; {slot, gen} stays unique",
|
||||
m_kindName, index + MG_Pipe::kMGPipeFirstAllocatableSlot);
|
||||
continue;
|
||||
}
|
||||
++m_entries[index].Gen;
|
||||
return index;
|
||||
}
|
||||
const Uint32 index = static_cast<Uint32>(m_entries.size());
|
||||
m_entries.push_back(Entry{});
|
||||
m_entries[index].Gen = 1;
|
||||
// The high-water mark, at powers of two from 1024 up: at most a handful of lines
|
||||
// for a whole session, emitted from the allocate-a-NEW-slot branch, i.e. once per
|
||||
// object this backend has ever seen and never on a draw (ROADMAP.md:7).
|
||||
//
|
||||
// [narrow, declared deviation from D20's "MGLOG_D for anything non-critical"] This
|
||||
// one is I, not D, because D is compiled out of every build that ships and of every
|
||||
// build P2 measures, and this line IS the measurement review v2's MAJOR 1 asks for:
|
||||
// the live-object high-water mark of minecraft-1.21.4-in-world and
|
||||
// ...-sodium-in-world, which nothing on desktop reaches and no gate here can see.
|
||||
// The structure no longer has a capacity to size off it, so the number is evidence
|
||||
// rather than a tuning input - but D.4.2 should still read it out of the device log,
|
||||
// and it cannot read a line that was compiled away.
|
||||
const SizeT minted = m_entries.size();
|
||||
if (minted >= 1024 && (minted & (minted - 1)) == 0) {
|
||||
MGLOG_I("MagmaPipeIdentityTable(%s): high-water %zu slots minted, %u live",
|
||||
m_kindName, minted, LiveCount());
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
const char* m_kindName = "";
|
||||
Uint64 m_boundary = 0;
|
||||
Vector<Entry> m_entries;
|
||||
Vector<Uint32> m_freeSlots;
|
||||
UnorderedMap<Uint64, Uint32> m_index;
|
||||
// One-entry front memo (see Acquire). m_lastLifetimeId == 0 means "empty": a live
|
||||
// object's lifetime id is never 0.
|
||||
Uint64 m_lastLifetimeId = 0;
|
||||
Uint32 m_lastIndex = 0;
|
||||
MG_Pipe::MGPipeHandle m_lastHandle = MG_Pipe::kMGPipeNullHandle;
|
||||
};
|
||||
|
||||
// The two mints one renderer owns. Per renderer, NOT process-global: two live contexts (or
|
||||
// a context recreation, which destroys and rebuilds the renderer) would otherwise share one
|
||||
// table and one reclamation clock, and both consumer tables are per-instance already.
|
||||
class MagmaPipeIdentityTables {
|
||||
public:
|
||||
// A VAO is kind VertexElementsCso: that is the gallium-shaped CSO a vertex array
|
||||
// resolves to, and the only kind in MGPipeKind that names vertex-input state.
|
||||
MG_Pipe::MGPipeHandle HandleOf(MG_Pipe::MGPipeKind kind, Uint64 lifetimeId) {
|
||||
return kind == MG_Pipe::MGPipeKind::Buffer ? m_buffers.Acquire(lifetimeId)
|
||||
: m_vaos.Acquire(lifetimeId);
|
||||
}
|
||||
void OnFrameBoundary() {
|
||||
m_vaos.OnFrameBoundary();
|
||||
m_buffers.OnFrameBoundary();
|
||||
}
|
||||
const MagmaPipeIdentityTable& Vaos() const { return m_vaos; }
|
||||
const MagmaPipeIdentityTable& Buffers() const { return m_buffers; }
|
||||
|
||||
private:
|
||||
MagmaPipeIdentityTable m_vaos{"VertexElementsCso"};
|
||||
MagmaPipeIdentityTable m_buffers{"Buffer"};
|
||||
};
|
||||
|
||||
// The table entry a handle names. Every per-slot table Magma keeps is indexed by this.
|
||||
//
|
||||
// A null handle has no slot, and it is unreachable here: both lifetime-id sources start at
|
||||
// 1 (VertexArrayObject.cpp, BufferObject.cpp), so Acquire's zero guard never fires. The
|
||||
// ternary, not the assertion, is what has effect in a shipped build (Defines.h compiles
|
||||
// MOBILEGL_ASSERT out at INFO), and slot 0 of a consumer table is a real entry that a null
|
||||
// handle can never match, because MGPipeHandleIsNull is also what the consumers compare.
|
||||
inline Uint32 MagmaPipeSlotIndex(const MG_Pipe::MGPipeHandle& handle) {
|
||||
MOBILEGL_ASSERT(!MG_Pipe::MGPipeHandleIsNull(handle),
|
||||
"a null MGPipeHandle has no slot to index a per-slot table with");
|
||||
return MG_Pipe::MGPipeHandleIsNull(handle)
|
||||
? 0u
|
||||
: handle.Slot - MG_Pipe::kMGPipeFirstAllocatableSlot;
|
||||
}
|
||||
|
||||
// A grow-on-demand per-slot table whose ENTRY ADDRESSES NEVER MOVE.
|
||||
//
|
||||
// D12.4 asks for a grow-on-demand Vector, and with an unbounded mint that is what a
|
||||
// consumer needs - but a Vector that grows relocates its elements, and the draw path holds
|
||||
// references into these entries across nested calls. Chunks of kChunkEntries are appended
|
||||
// instead: the Vector of owning pointers reallocates, the chunks never do, so an entry
|
||||
// reference is valid for the life of the table. That is the same guarantee the fixed table
|
||||
// it replaces gave, without the fixed capacity.
|
||||
template <typename T, Uint32 kChunkEntries = 256>
|
||||
class MagmaPipeSlotTable {
|
||||
public:
|
||||
T& operator[](Uint32 index) {
|
||||
const Uint32 chunk = index / kChunkEntries;
|
||||
while (m_chunks.size() <= chunk) {
|
||||
m_chunks.push_back(MakeUnique<Chunk>());
|
||||
}
|
||||
return m_chunks[chunk]->Entries[index % kChunkEntries];
|
||||
}
|
||||
SizeT Capacity() const { return m_chunks.size() * kChunkEntries; }
|
||||
|
||||
private:
|
||||
struct Chunk {
|
||||
T Entries[kChunkEntries] = {};
|
||||
};
|
||||
Vector<UniquePtr<Chunk>> m_chunks;
|
||||
};
|
||||
|
||||
// The claim rule every per-slot memo table uses, in one place so that the rule and the
|
||||
// negative control that defeats it cannot drift apart between consumers - and so that the
|
||||
// unit suite which drives a REAL slot reuse (MG_Test/Pipe/MagmaPipeIdentityTest.cpp) tests
|
||||
// this code rather than a copy of it.
|
||||
//
|
||||
// The SLOT picks the entry; the WHOLE handle - Gen included - decides whether the entry is
|
||||
// this object's. A slot the mint recycled for a different object comes back with a moved
|
||||
// Gen, so the compare fails and the entry is cleared rather than inherited. That is the
|
||||
// half HandleRecycleScenario cannot reach (see MagmaPipeAbaControlDefeatsIdentity).
|
||||
//
|
||||
// With negative control C on, every object collapses onto one entry and the entry is handed
|
||||
// back UNCLEARED and UNCLAIMED - at once "the replacement reproduced its predecessor's
|
||||
// slot" and "the slot was reused and Gen did not move".
|
||||
//
|
||||
// `Memos` needs a MG_Pipe::MGPipeHandle member named Owner and a default constructor that
|
||||
// means "empty"; VertexInputStateFactory::VaoBackendMemos is the one production instance.
|
||||
template <typename Memos, Uint32 kChunkEntries>
|
||||
inline Memos& MagmaPipeClaimSlotMemos(MagmaPipeSlotTable<Memos, kChunkEntries>& table,
|
||||
const MG_Pipe::MGPipeHandle& handle) {
|
||||
if (MagmaPipeAbaControlDefeatsIdentity()) {
|
||||
return table[kMagmaPipeAbaControlSlotIndex];
|
||||
}
|
||||
Memos& memos = table[MagmaPipeSlotIndex(handle)];
|
||||
if (!(memos.Owner == handle)) {
|
||||
memos = Memos{};
|
||||
memos.Owner = handle;
|
||||
}
|
||||
return memos;
|
||||
}
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -201,11 +201,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.renderPass, sizeof(payload.renderPass)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.colorAttachmentCount, sizeof(payload.colorAttachmentCount)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.rasterizationSamples, sizeof(payload.rasterizationSamples)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.sampleShadingEnable, sizeof(payload.sampleShadingEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.minSampleShading, sizeof(payload.minSampleShading)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.sampleMask, sizeof(payload.sampleMask)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.subpass, sizeof(payload.subpass)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.topology, sizeof(payload.topology)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.primitiveRestartEnable, sizeof(payload.primitiveRestartEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.patchControlPoints, sizeof(payload.patchControlPoints)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.passthroughTessControlKey,
|
||||
sizeof(payload.passthroughTessControlKey)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.viewportCount, sizeof(payload.viewportCount)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.polygonMode, sizeof(payload.polygonMode)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.cullMode, sizeof(payload.cullMode)));
|
||||
@@ -435,6 +440,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkPipelineMultisampleStateCreateInfo ms{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO};
|
||||
ms.rasterizationSamples = payload.rasterizationSamples;
|
||||
ms.sampleShadingEnable = payload.sampleShadingEnable ? VK_TRUE : VK_FALSE;
|
||||
// Ignored by Vulkan unless sampleShadingEnable is set, but written unconditionally so the
|
||||
// struct's bytes match the hash the payload was keyed by.
|
||||
ms.minSampleShading = payload.minSampleShading;
|
||||
// GL_SAMPLE_MASK / glSampleMaski. Left at nullptr - which Vulkan reads as all-ones - until
|
||||
// now, so glSampleMaski was a silent no-op on this backend while DirectGLES forwarded it.
|
||||
// The pointer has to outlive the vkCreateGraphicsPipelines call, which the payload does.
|
||||
ms.pSampleMask = payload.sampleMask;
|
||||
|
||||
VkPipelineDepthStencilStateCreateInfo depthStencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO};
|
||||
depthStencil.depthTestEnable = payload.depthTestEnable ? VK_TRUE : VK_FALSE;
|
||||
|
||||
@@ -37,11 +37,39 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
Uint32 colorAttachmentCount = 1;
|
||||
VkSampleCountFlagBits rasterizationSamples = VK_SAMPLE_COUNT_1_BIT;
|
||||
// glEnable(GL_SAMPLE_SHADING) + glMinSampleShading, which Vulkan bakes into the
|
||||
// pipeline rather than exposing as dynamic state - so both are part of the pipeline's
|
||||
// identity and both are hashed. The renderer leaves the enable false unless the
|
||||
// device's sampleRateShading feature was enabled
|
||||
// (VUID-VkPipelineMultisampleStateCreateInfo-sampleShadingEnable-00784).
|
||||
Bool sampleShadingEnable = false;
|
||||
Float minSampleShading = 0.0f;
|
||||
// glEnable(GL_SAMPLE_MASK) + glSampleMaski, the fixed-function coverage mask, already
|
||||
// reduced to what GL says this draw gets (VulkanRenderer::ResolveEffectiveSampleMask:
|
||||
// all-ones unless the target is genuinely multisampled). Pipeline state like the two
|
||||
// above - Vulkan has no dynamic sample mask before VK_EXT_extended_dynamic_state3 -
|
||||
// so it is hashed with them, and all-ones has to keep producing the pipeline a null
|
||||
// pSampleMask always did.
|
||||
//
|
||||
// TWO words, though GL only ever fills the first. GL_MAX_SAMPLE_MASK_WORDS is clamped
|
||||
// to 1 on both backends, so glSampleMaski writes index 0 and nothing else - but the
|
||||
// count Vulkan READS is ceil(rasterizationSamples / 32), which is 2 on a 64-sample
|
||||
// target, and GetAdvertisedMaxSamples does not cap the driver's sample count. A
|
||||
// single Uint32 here let such a pipeline read one word past the member (the next
|
||||
// struct field). The second word is all-ones: full coverage for samples 32..63, which
|
||||
// is the only honest answer when GL has no state describing them.
|
||||
Uint32 sampleMask[2] = {0xffffffffu, 0xffffffffu};
|
||||
Uint32 subpass = 0;
|
||||
VkPrimitiveTopology topology = VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST;
|
||||
Bool primitiveRestartEnable = false;
|
||||
// GL_PATCH_VERTICES; only read for a PATCH_LIST topology.
|
||||
Uint32 patchControlPoints = 3;
|
||||
// ProgramFactory::ComputePassthroughTessControlKey of the synthesized pass-through
|
||||
// tessellation control stage below, or 0 when this pipeline has none. Hashed, because
|
||||
// the levels glPatchParameterfv set are compiled INTO that module and are not a
|
||||
// function of the program or of patchControlPoints - see the note on
|
||||
// passthroughTessControlStage.
|
||||
Uint64 passthroughTessControlKey = 0;
|
||||
// How many of ARB_viewport_array's viewports this pipeline rasterizes into. 1 for
|
||||
// every program that never assigns gl_ViewportIndex, which is all of them outside the
|
||||
// conformance suite - the wide shape costs a longer vkCmdSetViewport/Scissor per state
|
||||
@@ -87,8 +115,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// renderer could not build one, and CreatePipeline refuses the pipeline - the same
|
||||
// refusal it applies when `stages` itself is half-tessellated.
|
||||
//
|
||||
// NOT hashed: it is a pure function of the program and of patchControlPoints, both
|
||||
// of which ComputeHash already mixes in.
|
||||
// NOT hashed directly: it is a pure function of the program, of patchControlPoints and
|
||||
// of the default tessellation levels - the first two of which ComputeHash already
|
||||
// mixes in, and the third of which arrives through passthroughTessControlKey above.
|
||||
VkPipelineShaderStageCreateInfo passthroughTessControlStage{};
|
||||
const VkPipelineVertexInputStateCreateInfo* vertexInputState = nullptr;
|
||||
// Diagnostic only; may be null. Read solely from the pipeline-creation failure path.
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -76,6 +76,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
|
||||
// The gl_PerVertex members a pass-through tessellation control stage may have to carry,
|
||||
// in the order glslang declares them - which is the order a redeclaration must use.
|
||||
// Which of them exist is a function of the neighbouring stage's GLSL VERSION
|
||||
// (gl_CullDistance joins the block at #version 450), so the mask is read off that
|
||||
// stage's SPIR-V rather than assumed. See ReflectPerVertexInputMembers.
|
||||
enum class PerVertexMemberBit : Uint32 {
|
||||
Position = 1u << 0,
|
||||
PointSize = 1u << 1,
|
||||
ClipDistance = 1u << 2,
|
||||
CullDistance = 1u << 3,
|
||||
};
|
||||
// What a program parsed below #version 450 carries, and the fallback when a module's
|
||||
// block cannot be read.
|
||||
static constexpr Uint32 kDefaultPerVertexMembers =
|
||||
static_cast<Uint32>(PerVertexMemberBit::Position) | static_cast<Uint32>(PerVertexMemberBit::PointSize) |
|
||||
static_cast<Uint32>(PerVertexMemberBit::ClipDistance);
|
||||
|
||||
struct UpdateAfterBindLimits {
|
||||
Bool enabled = false;
|
||||
Uint32 maxPerStageSamplers = 0;
|
||||
@@ -185,6 +202,32 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// tessellation stages are present or neither
|
||||
// (VUID-VkGraphicsPipelineCreateInfo-pStages-00730). So the draw path has to supply
|
||||
// the pass-through stage GL describes; see GetOrCreatePassthroughTessControlStage.
|
||||
// True when this program was built AS a transform-feedback capture variant but its
|
||||
// last pre-rasterization module does NOT carry the Xfb execution mode - so the
|
||||
// renderer must decline the capture span instead of issuing
|
||||
// vkCmdBeginTransformFeedbackEXT against it
|
||||
// (VUID-vkCmdBeginTransformFeedbackEXT-None-04128).
|
||||
//
|
||||
// Two ways to get here, and neither is visible from GL state, which is all
|
||||
// BeginXfbCaptureForDraw otherwise consults: the clip/XFB validation backstop had to
|
||||
// rewind past the capture decoration, or XfbCaptureDecoratePass resolved none of the
|
||||
// requested varyings and returned without changing anything (its own MGLOG_E path)
|
||||
// while its runner still reported success. Both used to ship a non-Xfb module under
|
||||
// an Xfb-flagged cache entry - the flag and the layout are part of the program cache
|
||||
// key, so it was sticky for every later captured draw of the program, not a glitch.
|
||||
Bool xfbCaptureDeclined = false;
|
||||
// The program has a tessellation or geometry module declaring TessellationPointSize /
|
||||
// GeometryPointSize on a device whose shaderTessellationAndGeometryPointSize feature
|
||||
// is off, so a pipeline built from it is invalid usage
|
||||
// (VUID-RuntimeSpirv-PointSize-06439). Its draws are refused in SetupDraw rather than
|
||||
// handed to the driver - the same contract PipelineFactory's half-tessellated refusal
|
||||
// implements one level up, and the counterpart of the DirectGLES arm that reports a
|
||||
// driver with neither point-size extension by name.
|
||||
//
|
||||
// Sticky by construction, which is what makes ONE log line honest: the flag lives on
|
||||
// the cache entry, so every later draw of the same program variant reads the same
|
||||
// answer instead of re-deciding it.
|
||||
Bool pointSizeCapabilityUnsupported = false;
|
||||
Bool needsPassthroughTessControl = false;
|
||||
// ...and the pass-through this renderer can synthesize carries gl_Position and
|
||||
// nothing else, so it is only correct when the evaluation stage's inputs are
|
||||
@@ -194,6 +237,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// instead (PipelineFactory::CreatePipeline refuses the pipeline and the draw is
|
||||
// skipped). See ReflectPassthroughTessControlNeed.
|
||||
Bool passthroughTessControlEmulatable = false;
|
||||
// Which gl_PerVertex members the evaluation stage's `in gl_PerVertex gl_in[]` block
|
||||
// actually carries, as a PerVertexMemberBit mask read off its SPIR-V. The synthesized
|
||||
// control stage has to redeclare the SAME shape: glslang appends gl_CullDistance to
|
||||
// that block from #version 450 upward, so a 450/460 program - and every ESSL program,
|
||||
// which the source processor rewrites to "#version 460 core" - carries four members
|
||||
// where a 430 program carries three. A fixed three-member pass-through fed the
|
||||
// evaluation stage a differently-shaped block, which is the black-frame-no-error case
|
||||
// this whole family is written around.
|
||||
Uint32 passthroughPerVertexMembers = 0;
|
||||
// Frame-boundary counter value of the last GetOrCreateProgram hit; drives
|
||||
// cache eviction (see OnFrameBoundary). Mutable: the draw snapshot's memoised
|
||||
// entry pointer re-stamps use through a const reference (StampProgramUse).
|
||||
@@ -249,6 +301,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
writesViewportIndexBuiltin = other.writesViewportIndexBuiltin;
|
||||
needsPassthroughTessControl = other.needsPassthroughTessControl;
|
||||
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
|
||||
passthroughPerVertexMembers = other.passthroughPerVertexMembers;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
@@ -267,6 +320,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.writesViewportIndexBuiltin = false;
|
||||
other.needsPassthroughTessControl = false;
|
||||
other.passthroughTessControlEmulatable = false;
|
||||
other.passthroughPerVertexMembers = 0;
|
||||
other.lastUsedFrame = 0;
|
||||
}
|
||||
VkProgramObject& operator=(VkProgramObject&& other) noexcept {
|
||||
@@ -311,6 +365,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
writesViewportIndexBuiltin = other.writesViewportIndexBuiltin;
|
||||
needsPassthroughTessControl = other.needsPassthroughTessControl;
|
||||
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
|
||||
passthroughPerVertexMembers = other.passthroughPerVertexMembers;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
@@ -329,6 +384,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.writesViewportIndexBuiltin = false;
|
||||
other.needsPassthroughTessControl = false;
|
||||
other.passthroughTessControlEmulatable = false;
|
||||
other.passthroughPerVertexMembers = 0;
|
||||
other.lastUsedFrame = 0;
|
||||
return *this;
|
||||
}
|
||||
@@ -396,12 +452,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings,
|
||||
Bool shaderDrawParametersEnabled,
|
||||
Bool unformattedFloatStorageImagesEnabled,
|
||||
Bool tessellationAndGeometryPointSizeEnabled,
|
||||
Bool enableSpirvValidation,
|
||||
UpdateAfterBindLimits updateAfterBindLimits,
|
||||
SubgroupLoweringPolicy subgroupPolicy)
|
||||
: m_device(device), m_maxBindings(maxBindings), m_config(config),
|
||||
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled),
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled),
|
||||
m_tessellationAndGeometryPointSizeEnabled(tessellationAndGeometryPointSizeEnabled),
|
||||
m_enableSpirvValidation(enableSpirvValidation),
|
||||
m_updateAfterBindLimits(updateAfterBindLimits),
|
||||
m_subgroupPolicy(subgroupPolicy) {
|
||||
@@ -448,6 +506,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static VkShaderStageFlagBits ToVkStage(ShaderStage stage);
|
||||
static VkFormat ConvertSpirvImageFormatToVkFormat(SpvImageFormat format);
|
||||
static SamplerNumericDomain UniformTypeToSamplerNumericDomain(GLenum glType);
|
||||
// The same question for an IMAGE uniform (`image2D`, `uimageBuffer`, ...), which the
|
||||
// sampler form above deliberately does not answer. Kept separate rather than folded in
|
||||
// because the two are asked in different places for different reasons: a sampler's domain
|
||||
// decides a sampled VIEW format, an image's decides what a placeholder descriptor for an
|
||||
// UNBOUND image unit must be (see UniformManager::AcquireUnboundTexelBufferView and
|
||||
// GetUnboundStorageImageTexture) - a formatless `writeonly` declaration reflects no
|
||||
// format at all, and the numeric domain is then the only thing that constrains it.
|
||||
static SamplerNumericDomain UniformTypeToImageNumericDomain(GLenum glType);
|
||||
// True when any entry point declares the DepthReplacing execution mode, i.e. the
|
||||
// shader assigns gl_FragDepth. Exposed so the blended depth-write quirk's exemption
|
||||
// can be pinned by tests. A false negative loses the exemption, so such a shader is
|
||||
@@ -477,18 +543,40 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// the caller then has no control stage to inject, and CreatePipeline refuses the
|
||||
// pipeline rather than handing the driver a half-tessellated one.
|
||||
//
|
||||
// Keyed on the patch size because GL takes the output patch size from PATCH_VERTICES,
|
||||
// which is draw state, not link state - the CTS case that motivated this links at the
|
||||
// default 3 and draws at 4. The pipeline cache already re-keys on patchControlPoints,
|
||||
// so the module a pipeline was built with is part of that pipeline's identity.
|
||||
// Compiling is bounded by the number of distinct patch sizes a program draws with
|
||||
// (MAX_PATCH_VERTICES = 32 in the worst case, one or two in practice) and only ever
|
||||
// happens for the rare program that has no control stage at all.
|
||||
VkPipelineShaderStageCreateInfo GetOrCreatePassthroughTessControlStage(Uint32 patchVertices);
|
||||
// Keyed on the patch size, the six default tessellation levels AND the gl_PerVertex
|
||||
// member set, because all three decide what the generator emits. The size comes from
|
||||
// PATCH_VERTICES and the levels from PATCH_DEFAULT_OUTER_LEVEL / PATCH_DEFAULT_INNER_LEVEL
|
||||
// - draw state rather than link state, and the CTS case that motivated this links at the
|
||||
// default 3 and draws at 4. The member set comes from the neighbouring evaluation stage's
|
||||
// own SPIR-V, so two programs at different GLSL versions need different modules. The
|
||||
// pipeline cache re-keys on the same inputs, so the module a pipeline was built with is
|
||||
// part of that pipeline's identity. Compiling is bounded by the number of distinct
|
||||
// (size, levels, members) combinations a program draws with - one or two in practice -
|
||||
// and only ever happens for the rare program that has no control stage at all.
|
||||
VkPipelineShaderStageCreateInfo GetOrCreatePassthroughTessControlStage(Uint32 patchVertices,
|
||||
const FloatVec4& defaultOuterLevel,
|
||||
const FloatVec2& defaultInnerLevel,
|
||||
Uint32 perVertexMembers);
|
||||
|
||||
// Source of the module above. Exposed for tests: the generated GLSL is the whole
|
||||
// contract with the evaluation stage, so it is worth pinning independently of a device.
|
||||
static String BuildPassthroughTessControlSource(Uint32 patchVertices);
|
||||
static String BuildPassthroughTessControlSource(Uint32 patchVertices, const FloatVec4& defaultOuterLevel,
|
||||
const FloatVec2& defaultInnerLevel, Uint32 perVertexMembers);
|
||||
|
||||
// The identity of one such module: everything the generator bakes in, folded into a
|
||||
// 64-bit key over the raw bits (so -0.0 and +0.0 key apart, which is harmless, and NaN
|
||||
// keys to itself, which is what matters). Shared with PipelineFactory, which mixes the
|
||||
// same value into the pipeline hash so a pipeline can never be handed a module built for
|
||||
// different levels or a different block shape.
|
||||
static Uint64 ComputePassthroughTessControlKey(Uint32 patchVertices, const FloatVec4& defaultOuterLevel,
|
||||
const FloatVec2& defaultInnerLevel, Uint32 perVertexMembers);
|
||||
|
||||
// The PerVertexMemberBit mask of the INPUT per-vertex block a module declares, read
|
||||
// straight out of its SPIR-V (OpMemberDecorate ... BuiltIn on the struct behind the one
|
||||
// Input variable that is an array of a Block-decorated struct). Zero when the module has
|
||||
// no such block. Exposed for tests, which is the only way to pin the shape agreement
|
||||
// without a device.
|
||||
static Uint32 ReflectPerVertexInputMembers(const Vector<Uint>& spirv);
|
||||
|
||||
private:
|
||||
struct ProgramLookupCache {
|
||||
@@ -531,6 +619,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// True only when the logical device enabled both
|
||||
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// True when the logical device enabled shaderTessellationAndGeometryPointSize. When it is
|
||||
// FALSE a program whose tessellation or geometry module declares TessellationPointSize /
|
||||
// GeometryPointSize is refused at build time (see VkProgramObject::
|
||||
// pointSizeCapabilityUnsupported) instead of being handed to the driver as invalid usage.
|
||||
Bool m_tessellationAndGeometryPointSizeEnabled = false;
|
||||
// Startup snapshot used only by internally synthesized shader modules, which do not
|
||||
// originate from a ProgramLinkTask.
|
||||
Bool m_enableSpirvValidation = false;
|
||||
@@ -548,11 +641,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// See GetCacheStructureEpoch(). Starts at 1 so a zero-initialized memo can never match.
|
||||
Uint64 m_cacheStructureEpoch = 1;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
// Pass-through tessellation control stages by input patch size. Never evicted: at most
|
||||
// MAX_PATCH_VERTICES entries exist for the lifetime of the device, and every pipeline
|
||||
// ever built from one keeps referencing its module. A failed build is cached as
|
||||
// Pass-through tessellation control stages by the identity of what was compiled into
|
||||
// them - the input patch size and the six default tessellation levels, folded into one
|
||||
// 64-bit key by ComputePassthroughTessControlKey (the levels are float state, so the map
|
||||
// cannot simply be keyed on the patch size any more). A failed build is cached as
|
||||
// VK_NULL_HANDLE so a broken generator costs one compile, not one per draw.
|
||||
UnorderedMap<Uint32, VkPipelineShaderStageCreateInfo> m_passthroughTessControlStages;
|
||||
//
|
||||
// Hard-capped, because the key is application-controlled: glPatchParameterfv clamps
|
||||
// nothing, so an application that recomputes a level per frame mints a new key per frame.
|
||||
// Reaching the cap destroys every module and starts over (see the flush in
|
||||
// GetOrCreatePassthroughTessControlStage); the cap is far above what any program that
|
||||
// holds its levels still will ever need. The gl_PerVertex member set is in the key too
|
||||
// and adds only a handful of values, so it does not move the cap in practice.
|
||||
static constexpr SizeT kMaxPassthroughTessControlStages = 64;
|
||||
UnorderedMap<Uint64, VkPipelineShaderStageCreateInfo> m_passthroughTessControlStages;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -10,15 +10,22 @@
|
||||
|
||||
#include "MG_Backend/DirectVulkan/DirectVulkanResourceState.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include "MG_State/GLState/ProgramState/ProgramObject.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObject1D.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObject2D.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObject2DCube.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObject3D.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObjectBuffer.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObjectStubs.h"
|
||||
#include "MG_Util/Converters/GLToMG/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include "MG_Util/Metrics/PipeStats.h"
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include "MG_Util/ShaderTranspiler/Types.h"
|
||||
#include <Config.h>
|
||||
#include <vulkan/utility/vk_format_utils.h>
|
||||
#include <algorithm>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
@@ -28,6 +35,154 @@
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
constexpr Uint kFallbackTexture2DExternalIndex = 0xFFFFFF00u;
|
||||
// One id for every storage-image placeholder. They are never reachable through GL - no
|
||||
// glGenTextures ever hands this out, and nothing looks a placeholder up by name - so the
|
||||
// id only has to stay clear of the application's, exactly like the sampled fallback's.
|
||||
constexpr Uint kUnboundStorageImageExternalIndex = 0xFFFFFF01u;
|
||||
// The multisample sampled fallbacks: one per (target, numeric domain), because unlike the
|
||||
// single-sampled fallback they cannot be reinterpreted into another domain at view time
|
||||
// (see GetFallbackMultisampleTexture). Six reserved ids, contiguous from this base for the
|
||||
// same reason as the two above - they must not collide with anything glGenTextures can
|
||||
// hand out.
|
||||
constexpr Uint kFallbackMultisampleExternalIndexBase = 0xFFFFFF02u;
|
||||
constexpr Uint kFallbackMultisampleExternalIndexCount = 6u;
|
||||
|
||||
// MobileGL's own stand-in textures, by the reserved ids above. Nothing an application can
|
||||
// do reaches one, so anything keyed on the GL object an application bound - image-unit
|
||||
// aliasing above all - has to leave them alone.
|
||||
Bool IsPlaceholderTexture(const MG_State::GLState::ITextureObject* texture) {
|
||||
if (texture == nullptr) return false;
|
||||
const Uint index = static_cast<Uint>(texture->GetExternalIndex());
|
||||
return index == kFallbackTexture2DExternalIndex || index == kUnboundStorageImageExternalIndex ||
|
||||
(index >= kFallbackMultisampleExternalIndexBase &&
|
||||
index < kFallbackMultisampleExternalIndexBase + kFallbackMultisampleExternalIndexCount);
|
||||
}
|
||||
|
||||
// The R32 member of each numeric class. Every one of the three is a MANDATORY-support
|
||||
// format for uniform texel buffers, storage texel buffers and storage images alike
|
||||
// (Vulkan 1.0, "Required Format Support"), which is what makes them a fallback that
|
||||
// cannot itself fail for want of device features.
|
||||
VkFormat PlaceholderFormatForNumericDomain(SamplerNumericDomain numericDomain) {
|
||||
switch (numericDomain) {
|
||||
case SamplerNumericDomain::Float:
|
||||
return VK_FORMAT_R32_SFLOAT;
|
||||
case SamplerNumericDomain::SignedInteger:
|
||||
return VK_FORMAT_R32_SINT;
|
||||
case SamplerNumericDomain::UnsignedInteger:
|
||||
return VK_FORMAT_R32_UINT;
|
||||
case SamplerNumericDomain::Unknown:
|
||||
break;
|
||||
}
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
|
||||
Bool BufferFormatSupportsFeature(VkPhysicalDevice physicalDevice, VkFormat format,
|
||||
VkFormatFeatureFlags requiredFeature) {
|
||||
if (physicalDevice == VK_NULL_HANDLE || format == VK_FORMAT_UNDEFINED) {
|
||||
return false;
|
||||
}
|
||||
VkFormatProperties properties{};
|
||||
vkGetPhysicalDeviceFormatProperties(physicalDevice, format, &properties);
|
||||
return (properties.bufferFeatures & requiredFeature) == requiredFeature;
|
||||
}
|
||||
|
||||
// Reverse of MG_Util::ConvertTextureInternalFormatToVkEnum. A placeholder texture is
|
||||
// built through the ordinary frontend texture object (that is what gets it an image with
|
||||
// STORAGE usage, a GENERAL transition and a view, for free), and that object is described
|
||||
// by a GL internal format - while everything upstream of here speaks VkFormat. Scanned
|
||||
// rather than tabulated: it runs once per (target, format) placeholder ever created, the
|
||||
// enum is ~70 entries, and a second hand-written table is a second thing to drift.
|
||||
// Ascending order matters: the sized formats precede the unsized aliases, so a scan
|
||||
// answers with the sized one.
|
||||
TextureInternalFormat InternalFormatForVkFormat(VkFormat format) {
|
||||
if (format == VK_FORMAT_UNDEFINED) {
|
||||
return TextureInternalFormat::Unknown;
|
||||
}
|
||||
for (Int index = 0; index < static_cast<Int>(TextureInternalFormat::TextureInternalFormatCount);
|
||||
++index) {
|
||||
const auto candidate = static_cast<TextureInternalFormat>(index);
|
||||
if (MG_Util::ConvertTextureInternalFormatToVkEnum(candidate) == format) {
|
||||
return candidate;
|
||||
}
|
||||
}
|
||||
return TextureInternalFormat::Unknown;
|
||||
}
|
||||
|
||||
// What a 1x1 placeholder of a given target has to allocate for the backend to give it the
|
||||
// Vulkan view type that target's image declaration demands (see
|
||||
// VkTextureManager's TryResolveTextureShapeInfo, which reads exactly these two things).
|
||||
struct PlaceholderShape {
|
||||
Array<TextureUploadTarget, 6> uploadTargets{};
|
||||
Uint32 uploadTargetCount = 0;
|
||||
// The GL depth of the single level: the array length for an array target, the depth
|
||||
// for a 3D one, and 6 for a cube map array (one whole cube).
|
||||
Int depth = 1;
|
||||
Bool valid = false;
|
||||
};
|
||||
|
||||
PlaceholderShape PlaceholderShapeForTarget(TextureTarget target) {
|
||||
PlaceholderShape shape{};
|
||||
switch (target) {
|
||||
case TextureTarget::Texture1D:
|
||||
shape = {{TextureUploadTarget::Texture1D}, 1, 1, true};
|
||||
break;
|
||||
case TextureTarget::Texture2D:
|
||||
shape = {{TextureUploadTarget::Texture2D}, 1, 1, true};
|
||||
break;
|
||||
case TextureTarget::TextureRectangle:
|
||||
shape = {{TextureUploadTarget::TextureRectangle}, 1, 1, true};
|
||||
break;
|
||||
case TextureTarget::Texture3D:
|
||||
shape = {{TextureUploadTarget::Texture3D}, 1, 1, true};
|
||||
break;
|
||||
case TextureTarget::Texture1DArray:
|
||||
shape = {{TextureUploadTarget::Texture1DArray}, 1, 1, true};
|
||||
break;
|
||||
case TextureTarget::Texture2DArray:
|
||||
shape = {{TextureUploadTarget::Texture2DArray}, 1, 1, true};
|
||||
break;
|
||||
case TextureTarget::TextureCubeMap:
|
||||
shape = {{TextureUploadTarget::CubeMapPositiveX, TextureUploadTarget::CubeMapNegativeX,
|
||||
TextureUploadTarget::CubeMapPositiveY, TextureUploadTarget::CubeMapNegativeY,
|
||||
TextureUploadTarget::CubeMapPositiveZ, TextureUploadTarget::CubeMapNegativeZ},
|
||||
6, 1, true};
|
||||
break;
|
||||
case TextureTarget::TextureCubeMapArray:
|
||||
// Layers are cube faces, so the count must be a whole number of cubes.
|
||||
shape = {{TextureUploadTarget::CubeMapArray}, 1, 6, true};
|
||||
break;
|
||||
default:
|
||||
// Multisample targets above all: their descriptor needs a multisample view.
|
||||
break;
|
||||
}
|
||||
return shape;
|
||||
}
|
||||
|
||||
// TextureObjectMipmap, not ITextureObject: AllocateStorage and MarkStorageDirty live
|
||||
// there, and every placeholder shape above is one of its subclasses.
|
||||
SharedPtr<MG_State::GLState::TextureObjectMipmap> MakePlaceholderTextureObject(TextureTarget target,
|
||||
Uint index) {
|
||||
switch (target) {
|
||||
case TextureTarget::Texture1D:
|
||||
return MakeShared<MG_State::GLState::TextureObject1D>(index);
|
||||
case TextureTarget::Texture2D:
|
||||
return MakeShared<MG_State::GLState::TextureObject2D>(index);
|
||||
case TextureTarget::TextureRectangle:
|
||||
return MakeShared<MG_State::GLState::TextureObjectRectangle>(index);
|
||||
case TextureTarget::Texture3D:
|
||||
return MakeShared<MG_State::GLState::TextureObject3D>(index);
|
||||
case TextureTarget::Texture1DArray:
|
||||
return MakeShared<MG_State::GLState::TextureObject1DArray>(index);
|
||||
case TextureTarget::Texture2DArray:
|
||||
return MakeShared<MG_State::GLState::TextureObject2DArray>(index);
|
||||
case TextureTarget::TextureCubeMap:
|
||||
return MakeShared<MG_State::GLState::TextureObject2DCube>(index);
|
||||
case TextureTarget::TextureCubeMapArray:
|
||||
return MakeShared<MG_State::GLState::TextureObjectCubeMapArray>(index);
|
||||
default:
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static Bool FindFramebufferAttachmentForTexture(const MG_State::GLState::FramebufferObject& framebuffer,
|
||||
@@ -115,7 +270,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return reflectedFormat != VK_FORMAT_UNDEFINED ? reflectedFormat : resourceFormat;
|
||||
}
|
||||
|
||||
Bool UniformManager::Initialize(VkDevice device, VkBufferManager* bufferManager,
|
||||
Bool UniformManager::Initialize(VkDevice device, VkPhysicalDevice physicalDevice,
|
||||
VkBufferManager* bufferManager,
|
||||
ProgramFactory* programFactory,
|
||||
VkDeviceSize minUniformBufferOffsetAlignment, Uint32 frameCount,
|
||||
Uint32 maxBindings, Uint32 setsPerFrame,
|
||||
@@ -123,6 +279,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Shutdown();
|
||||
|
||||
MOBILEGL_ASSERT(device != VK_NULL_HANDLE, "UniformDescriptorBinder::Initialize requires valid VkDevice");
|
||||
MOBILEGL_ASSERT(physicalDevice != VK_NULL_HANDLE,
|
||||
"UniformDescriptorBinder::Initialize requires valid VkPhysicalDevice");
|
||||
MOBILEGL_ASSERT(bufferManager != nullptr, "UniformDescriptorBinder::Initialize requires valid buffer manager");
|
||||
MOBILEGL_ASSERT(programFactory != nullptr,
|
||||
"UniformDescriptorBinder::Initialize requires valid program factory");
|
||||
@@ -135,6 +293,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
"UniformDescriptorBinder::Initialize requires valid sampler manager");
|
||||
|
||||
m_device = device;
|
||||
m_physicalDevice = physicalDevice;
|
||||
m_bufferManager = bufferManager;
|
||||
m_programFactory = programFactory;
|
||||
m_minDynamicOffsetAlignment = std::max<VkDeviceSize>(1, minUniformBufferOffsetAlignment);
|
||||
@@ -173,6 +332,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
void UniformManager::Shutdown() {
|
||||
// Before the per-frame loop, because these views are NOT owned by any frame slot (see
|
||||
// m_unboundTexelBufferViews) and the loop below is what clears m_device.
|
||||
if (m_device != VK_NULL_HANDLE) {
|
||||
for (const auto& viewEntry : m_unboundTexelBufferViews) {
|
||||
if (viewEntry.second != VK_NULL_HANDLE) {
|
||||
vkDestroyBufferView(m_device, viewEntry.second, nullptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
m_unboundTexelBufferViews.clear();
|
||||
m_unboundStorageImageTextures.clear();
|
||||
for (auto& frame : m_frames) {
|
||||
if (m_device != VK_NULL_HANDLE) {
|
||||
for (auto& view : frame.texelBufferViews) {
|
||||
@@ -199,6 +369,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_bufferManager = nullptr;
|
||||
m_programFactory = nullptr;
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_physicalDevice = VK_NULL_HANDLE;
|
||||
m_minDynamicOffsetAlignment = 1;
|
||||
m_frameCount = 0;
|
||||
m_maxBindings = 0;
|
||||
@@ -209,6 +380,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_textureManager = nullptr;
|
||||
m_samplerManager = nullptr;
|
||||
m_fallbackTexture2D.reset();
|
||||
m_fallbackMultisampleTextures.clear();
|
||||
}
|
||||
|
||||
void UniformManager::BeginFrame(Uint32 frameIndex) {
|
||||
@@ -332,7 +504,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// alive through the draw via GL binding state. Only the fallback path needs a SharedPtr to
|
||||
// keep the fallback texture alive for the rest of this call.
|
||||
MG_State::GLState::ITextureObject* texture = ResolveSamplerTextureRaw(program, programObj, binding, element);
|
||||
auto& textureUnit = MG_State::pGLContext->GetTextureUnitObject(unit);
|
||||
auto& textureUnit = MGB_CTX->GetTextureUnitObject(unit);
|
||||
const auto& samplerOverride = textureUnit.GetSamplerObject();
|
||||
const auto preferredTarget = programObj.samplerTextureTargetByBinding[binding];
|
||||
SharedPtr<MG_State::GLState::ITextureObject> fallbackHolder;
|
||||
@@ -345,7 +517,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
texture = nullptr;
|
||||
}
|
||||
if (texture == nullptr) {
|
||||
fallbackHolder = GetFallbackTexture(preferredTarget);
|
||||
// The binding's sampler class, read here rather than through the `numericDomain`
|
||||
// local further down (it is declared after this point): the multisample placeholder
|
||||
// has to be built in the class the shader will read it in.
|
||||
fallbackHolder = GetFallbackTexture(preferredTarget, programObj.samplerNumericDomainByBinding[binding]);
|
||||
texture = fallbackHolder.get();
|
||||
if (texture == nullptr) {
|
||||
MGLOG_E_ONCE("ResolveSamplerDescriptor: no fallback texture available for binding=%u ('%s') "
|
||||
@@ -378,7 +553,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
if (!IsValidSampledImageLayout(resource->layout)) {
|
||||
auto drawFbo = MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||
auto drawFbo = MGB_CTX->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||
FramebufferAttachmentType attachmentType = FramebufferAttachmentType::None;
|
||||
Int attachmentLevel = 0;
|
||||
if (drawFbo &&
|
||||
@@ -605,7 +780,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// filtering - which a single-level view can still have. Resolve the sampler exactly
|
||||
// the way ResolveSamplerDescriptor does and bail if anisotropy would apply.
|
||||
const Int unit = ResolveSamplerUnitIndex(program, location, binding);
|
||||
const auto& samplerOverride = MG_State::pGLContext->GetTextureUnitObject(unit).GetSamplerObject();
|
||||
const auto& samplerOverride = MGB_CTX->GetTextureUnitObject(unit).GetSamplerObject();
|
||||
const auto* effectiveSampler =
|
||||
samplerOverride ? samplerOverride.get() : texture->GetSamplerObject().get();
|
||||
if (effectiveSampler == nullptr) return false;
|
||||
@@ -635,7 +810,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture) {
|
||||
outTexture.reset();
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveSamplerTexture: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveSamplerTexture: GL context is null");
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerUniformLocationByBinding.size(),
|
||||
"ResolveSamplerTexture: sampler location binding %u out of range", binding);
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerTextureTargetByBinding.size(),
|
||||
@@ -644,7 +819,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const Int location = programObj.samplerUniformLocationByBinding[binding];
|
||||
const Int unit = ResolveSamplerUnitIndex(program, location, binding);
|
||||
|
||||
auto& textureUnit = MG_State::pGLContext->GetTextureUnitObject(unit);
|
||||
auto& textureUnit = MGB_CTX->GetTextureUnitObject(unit);
|
||||
const TextureTarget preferredTarget = programObj.samplerTextureTargetByBinding[binding];
|
||||
outTexture = textureUnit.GetBindingSlot(preferredTarget).GetBoundObject();
|
||||
// The slot always holds at least the target's default texture (name 0). While that
|
||||
@@ -660,7 +835,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MG_State::GLState::ITextureObject* UniformManager::ResolveSamplerTextureRaw(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj,
|
||||
Uint32 binding, Uint32 element) {
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveSamplerTextureRaw: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveSamplerTextureRaw: GL context is null");
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerUniformLocationByBinding.size(),
|
||||
"ResolveSamplerTextureRaw: sampler location binding %u out of range", binding);
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerTextureTargetByBinding.size(),
|
||||
@@ -670,7 +845,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
ResolveDescriptorElementLocation(program, programObj.samplerUniformLocationByBinding[binding], element);
|
||||
const Int unit = ResolveSamplerUnitIndex(program, location, binding);
|
||||
|
||||
auto& textureUnit = MG_State::pGLContext->GetTextureUnitObject(unit);
|
||||
auto& textureUnit = MGB_CTX->GetTextureUnitObject(unit);
|
||||
const TextureTarget preferredTarget = programObj.samplerTextureTargetByBinding[binding];
|
||||
// GetBoundObject() returns the SharedPtr by const ref; .get() reads the pointer without
|
||||
// touching the refcount (no atomic inc/dec per binding per draw).
|
||||
@@ -693,11 +868,29 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(m_bufferManager != nullptr, "ResolveTexelBufferDescriptor: buffer manager is null");
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(), "ResolveTexelBufferDescriptor: frame index out of range");
|
||||
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerNumericDomainByBinding.size(),
|
||||
"ResolveTexelBufferDescriptor: numeric domain binding %u out of range", binding);
|
||||
const SamplerNumericDomain numericDomain = programObj.samplerNumericDomainByBinding[binding];
|
||||
|
||||
SharedPtr<MG_State::GLState::ITextureObject> texture;
|
||||
if (!ResolveSamplerTexture(program, programObj, binding, texture) || texture == nullptr) {
|
||||
MGLOG_E_ONCE("ResolveTexelBufferDescriptor: texture buffer binding %u ('%s') is unbound", binding,
|
||||
programObj.samplerNameByBinding[binding].c_str());
|
||||
return false;
|
||||
// NOT an error, and not a reason to lose the draw. A texture unit with nothing on it
|
||||
// is a legal GL state (4.6 core 8.24): the sampler is incomplete, so a fetch through
|
||||
// it returns undefined values - the same answer the sampled path above gives with its
|
||||
// fallback texture, which a buffer texture simply cannot use because its descriptor is
|
||||
// a VkBufferView. A per-format placeholder view is the equivalent for this kind.
|
||||
const VkBufferView placeholder =
|
||||
AcquireUnboundTexelBufferView(VK_FORMAT_UNDEFINED, numericDomain, false);
|
||||
if (placeholder == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("ResolveTexelBufferDescriptor: texture buffer binding %u ('%s') is unbound, and the "
|
||||
"placeholder descriptor could not be created", binding,
|
||||
programObj.samplerNameByBinding[binding].c_str());
|
||||
return false;
|
||||
}
|
||||
MGLOG_D("ResolveTexelBufferDescriptor: binding %u ('%s') is unbound; using the placeholder descriptor",
|
||||
binding, programObj.samplerNameByBinding[binding].c_str());
|
||||
outBufferView = placeholder;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (texture->GetStorageType() != TextureStorageType::Buffer ||
|
||||
@@ -712,9 +905,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto* textureBuffer = static_cast<MG_State::GLState::TextureObjectBuffer*>(texture.get());
|
||||
const auto& bufferObject = textureBuffer->GetBufferBindingSlot().GetBoundObject();
|
||||
if (bufferObject == nullptr) {
|
||||
MGLOG_E_ONCE("ResolveTexelBufferDescriptor: texture buffer binding %u ('%s') has no GL buffer bound",
|
||||
binding, programObj.samplerNameByBinding[binding].c_str());
|
||||
return false;
|
||||
// A buffer texture with no buffer object attached is INCOMPLETE, not illegal (GL 4.6
|
||||
// core 8.9), and sampling an incomplete texture is undefined - so this too keeps the
|
||||
// draw on a placeholder rather than dropping it.
|
||||
const VkBufferView placeholder =
|
||||
AcquireUnboundTexelBufferView(VK_FORMAT_UNDEFINED, numericDomain, false);
|
||||
if (placeholder == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("ResolveTexelBufferDescriptor: texture buffer binding %u ('%s') has no GL buffer bound, "
|
||||
"and the placeholder descriptor could not be created", binding,
|
||||
programObj.samplerNameByBinding[binding].c_str());
|
||||
return false;
|
||||
}
|
||||
MGLOG_D("ResolveTexelBufferDescriptor: binding %u ('%s') has no attached GL buffer; using the "
|
||||
"placeholder descriptor", binding, programObj.samplerNameByBinding[binding].c_str());
|
||||
outBufferView = placeholder;
|
||||
return true;
|
||||
}
|
||||
|
||||
BufferSlice slice{};
|
||||
@@ -785,7 +990,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkBufferView& outBufferView) {
|
||||
outBufferView = VK_NULL_HANDLE;
|
||||
MOBILEGL_ASSERT(m_bufferManager != nullptr, "ResolveStorageTexelBufferDescriptor: buffer manager is null");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveStorageTexelBufferDescriptor: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveStorageTexelBufferDescriptor: GL context is null");
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(),
|
||||
"ResolveStorageTexelBufferDescriptor: frame index out of range");
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerUniformLocationByBinding.size(),
|
||||
@@ -804,12 +1009,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& imageBinding = MG_State::pGLContext->GetImageTextureBinding(imageUnit);
|
||||
MOBILEGL_ASSERT(binding < programObj.storageImageFormatByBinding.size(),
|
||||
"ResolveStorageTexelBufferDescriptor: binding %u has no reflected format slot", binding);
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerNumericDomainByBinding.size(),
|
||||
"ResolveStorageTexelBufferDescriptor: numeric domain binding %u out of range", binding);
|
||||
|
||||
auto& imageBinding = MGB_CTX->GetImageTextureBinding(imageUnit);
|
||||
const auto& texture = imageBinding.Texture;
|
||||
if (texture == nullptr) {
|
||||
MGLOG_E_ONCE("ResolveStorageTexelBufferDescriptor: image unit %d is unbound for binding %u", imageUnit,
|
||||
binding);
|
||||
return false;
|
||||
// An image unit with no texture on it is legal GL (4.6 core 8.26): loads return zero
|
||||
// and stores are discarded. Declining here took the whole draw or dispatch with it -
|
||||
// the same shape as the unbound storage block fixed alongside this. A placeholder view
|
||||
// in the shader's own declared format lets the work proceed with the stores landing
|
||||
// nowhere anyone can observe, which is what GL asks for.
|
||||
const VkBufferView placeholder =
|
||||
AcquireUnboundTexelBufferView(programObj.storageImageFormatByBinding[binding],
|
||||
programObj.samplerNumericDomainByBinding[binding], true);
|
||||
if (placeholder == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("ResolveStorageTexelBufferDescriptor: image unit %d is unbound for binding %u, and the "
|
||||
"placeholder descriptor could not be created", imageUnit, binding);
|
||||
return false;
|
||||
}
|
||||
MGLOG_D("ResolveStorageTexelBufferDescriptor: image unit %d (binding %u) is unbound; using the "
|
||||
"placeholder descriptor", imageUnit, binding);
|
||||
outBufferView = placeholder;
|
||||
return true;
|
||||
}
|
||||
if (texture->GetStorageType() != TextureStorageType::Buffer ||
|
||||
texture->GetTarget() != TextureTarget::TextureBuffer) {
|
||||
@@ -824,9 +1048,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto* textureBuffer = static_cast<MG_State::GLState::TextureObjectBuffer*>(texture.get());
|
||||
const auto& bufferObject = textureBuffer->GetBufferBindingSlot().GetBoundObject();
|
||||
if (bufferObject == nullptr) {
|
||||
MGLOG_E_ONCE("ResolveStorageTexelBufferDescriptor: texture buffer on image unit %d has no GL buffer bound",
|
||||
imageUnit);
|
||||
return false;
|
||||
// Incomplete buffer texture, same as the sampled path: legal state, undefined data,
|
||||
// and no reason to drop the work.
|
||||
const VkBufferView placeholder =
|
||||
AcquireUnboundTexelBufferView(programObj.storageImageFormatByBinding[binding],
|
||||
programObj.samplerNumericDomainByBinding[binding], true);
|
||||
if (placeholder == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("ResolveStorageTexelBufferDescriptor: texture buffer on image unit %d has no GL buffer "
|
||||
"bound, and the placeholder descriptor could not be created", imageUnit);
|
||||
return false;
|
||||
}
|
||||
MGLOG_D("ResolveStorageTexelBufferDescriptor: texture buffer on image unit %d has no attached GL buffer; "
|
||||
"using the placeholder descriptor", imageUnit);
|
||||
outBufferView = placeholder;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Unlike the sampled texel buffer, the shader MAY write this one, and those writes land
|
||||
@@ -852,8 +1087,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// policy as a storage image: a typed `layout(r32ui) uniform uimageBuffer` must be read as
|
||||
// r32ui whatever the texture's own attachment format says. Falling back, in order:
|
||||
// reflected format, then the bind format, then the texture's attached format.
|
||||
MOBILEGL_ASSERT(binding < programObj.storageImageFormatByBinding.size(),
|
||||
"ResolveStorageTexelBufferDescriptor: binding %u has no reflected format slot", binding);
|
||||
const auto internalFormat = textureBuffer->GetFormat();
|
||||
const VkFormat resourceFormat = MG_Util::ConvertTextureInternalFormatToVkEnum(internalFormat);
|
||||
const VkFormat reflectedFormat = programObj.storageImageFormatByBinding[binding];
|
||||
@@ -917,7 +1150,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDescriptorBufferInfo& outBufferInfo) const {
|
||||
outBufferInfo = {};
|
||||
MOBILEGL_ASSERT(m_bufferManager != nullptr, "ResolveStorageBufferDescriptor: buffer manager is null");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveStorageBufferDescriptor: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveStorageBufferDescriptor: GL context is null");
|
||||
MOBILEGL_ASSERT(binding < programObj.storageBlockIndexByBinding.size(),
|
||||
"ResolveStorageBufferDescriptor: binding %u out of range", binding);
|
||||
|
||||
@@ -954,12 +1187,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
? static_cast<GLuint>(atomicCounterBinding)
|
||||
: GetShaderStorageBlockBinding(program, static_cast<GLuint>(blockIndex)) + element;
|
||||
const Uint32 bindingPointCount =
|
||||
static_cast<Uint32>(MG_State::pGLContext->GetBufferBindingPointCount(bufferTarget));
|
||||
static_cast<Uint32>(MGB_CTX->GetBufferBindingPointCount(bufferTarget));
|
||||
MOBILEGL_ASSERT(frontendBinding < bindingPointCount,
|
||||
"ResolveStorageBufferDescriptor: frontend binding %u out of range for block '%s'",
|
||||
frontendBinding, blockName.c_str());
|
||||
|
||||
auto& bindingPoint = MG_State::pGLContext->GetBufferBindingPoint(bufferTarget, frontendBinding);
|
||||
auto& bindingPoint = MGB_CTX->GetBufferBindingPoint(bufferTarget, frontendBinding);
|
||||
const auto& bufferObject = bindingPoint.GetBoundObject();
|
||||
if (bufferObject == nullptr) {
|
||||
// NOT an error, and above all not a reason to lose the draw. GL 4.6 core 7.8 lets a
|
||||
@@ -1031,7 +1264,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDescriptorImageInfo& outImageInfo) const {
|
||||
outImageInfo = {};
|
||||
MOBILEGL_ASSERT(m_textureManager != nullptr, "ResolveStorageImageDescriptor: texture manager is null");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveStorageImageDescriptor: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveStorageImageDescriptor: GL context is null");
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerUniformLocationByBinding.size(),
|
||||
"ResolveStorageImageDescriptor: binding %u out of range", binding);
|
||||
|
||||
@@ -1060,10 +1293,42 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& imageBinding = MG_State::pGLContext->GetImageTextureBinding(imageUnit);
|
||||
auto& imageBinding = MGB_CTX->GetImageTextureBinding(imageUnit);
|
||||
if (imageBinding.Texture == nullptr) {
|
||||
MGLOG_E_ONCE("ResolveStorageImageDescriptor: image unit %d is unbound for binding %u", imageUnit, binding);
|
||||
return false;
|
||||
// Legal GL: an image unit with no texture bound makes loads return zero and discards
|
||||
// stores (4.6 core 8.26). It is not a reason to lose the draw, which is what returning
|
||||
// false here did - both SetupDraw and DispatchCompute skip everything on it. The
|
||||
// placeholder is a 1x1 image of the target and format the shader's declaration asks
|
||||
// for, so the descriptor is valid and the stores land where nobody can see them.
|
||||
TextureTarget placeholderTarget = TextureTarget::Unknown;
|
||||
VkFormat placeholderFormat = VK_FORMAT_UNDEFINED;
|
||||
SharedPtr<MG_State::GLState::ITextureObject> placeholder;
|
||||
if (ResolveUnboundStorageImagePlaceholder(programObj, binding, placeholderTarget, placeholderFormat)) {
|
||||
placeholder = GetUnboundStorageImageTexture(placeholderTarget, placeholderFormat);
|
||||
}
|
||||
VkImageView placeholderView = VK_NULL_HANDLE;
|
||||
if (placeholder != nullptr &&
|
||||
m_textureManager->TransitionTextureForStorageImage(commandBuffer, *placeholder)) {
|
||||
// layered=true, layer=0: the placeholder's own view type IS the one the shader's
|
||||
// image declaration demands, and that is exactly what the layered form asks for
|
||||
// (see GetOrCreateStorageImageView, which only narrows the view type when a
|
||||
// non-layered binding names a single layer).
|
||||
placeholderView =
|
||||
m_textureManager->GetOrCreateStorageImageView(*placeholder, 0, placeholderFormat, true, 0);
|
||||
}
|
||||
if (placeholderView == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("ResolveStorageImageDescriptor: image unit %d is unbound for binding %u, and no "
|
||||
"placeholder descriptor could be built (target=%d format=%d)",
|
||||
imageUnit, binding, static_cast<Int>(placeholderTarget),
|
||||
static_cast<Int>(placeholderFormat));
|
||||
return false;
|
||||
}
|
||||
MGLOG_D("ResolveStorageImageDescriptor: image unit %d (binding %u) is unbound; using the placeholder "
|
||||
"descriptor", imageUnit, binding);
|
||||
outImageInfo.sampler = VK_NULL_HANDLE;
|
||||
outImageInfo.imageView = placeholderView;
|
||||
outImageInfo.imageLayout = VK_IMAGE_LAYOUT_GENERAL;
|
||||
return true;
|
||||
}
|
||||
|
||||
const Bool ready = m_textureManager->TransitionTextureForStorageImage(commandBuffer, *imageBinding.Texture);
|
||||
@@ -1126,18 +1391,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return outImageInfo.imageView != VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
SharedPtr<MG_State::GLState::ITextureObject> UniformManager::GetFallbackTexture(TextureTarget target) const {
|
||||
// The fallback is a single-sampled 2D image, so it can only stand in for a sampler that
|
||||
// would accept one. A multisample sampler in particular cannot: its descriptor demands a
|
||||
// multisample view, and handing it this one is invalid Vulkan, not a degraded picture.
|
||||
// Report that there is no fallback and let the caller decline the draw - aborting the
|
||||
// process over an unbound sampler is never the right answer.
|
||||
SharedPtr<MG_State::GLState::ITextureObject> UniformManager::GetFallbackTexture(
|
||||
TextureTarget target, SamplerNumericDomain numericDomain) const {
|
||||
// A multisample sampler cannot be served by the single-sampled 2D image below - its
|
||||
// descriptor demands a multisample view - so it gets its own placeholder rather than no
|
||||
// placeholder at all. Without one, ResolveSamplerDescriptor declined and
|
||||
// BindProgramUniformBuffers dropped the WHOLE draw, which is how every
|
||||
// sample_variables.*.samples_0 body failed: the CTS's resolve program declares both a
|
||||
// sampler2D and a sampler2DMS and deliberately points the unused one at an empty texture
|
||||
// unit, and at samples_0 the unused one is the sampler2DMS. GL says sampling an
|
||||
// incomplete texture is undefined, not fatal, so the draw has to happen.
|
||||
if (target == TextureTarget::Texture2DMultisample ||
|
||||
target == TextureTarget::Texture2DMultisampleArray) {
|
||||
return GetFallbackMultisampleTexture(target, numericDomain);
|
||||
}
|
||||
if (target != TextureTarget::Texture2D && target != TextureTarget::TextureRectangle) {
|
||||
MGLOG_E_ONCE("UniformManager::GetFallbackTexture: no fallback exists for target=%d",
|
||||
static_cast<Int>(target));
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// The single-sampled fallback stays domain-agnostic: it is storage-image capable, so its
|
||||
// image carries VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT and ResolveSampledImageViewFormat can
|
||||
// hand an integer sampler an R8G8B8A8_UINT view of these same RGBA8 texels. A multisample
|
||||
// image can never carry that bit, which is why the arm above needs one object per domain.
|
||||
if (m_fallbackTexture2D == nullptr) {
|
||||
auto fallbackTexture = MakeShared<MG_State::GLState::TextureObject2D>(kFallbackTexture2DExternalIndex);
|
||||
fallbackTexture->SetInternalFormat(TextureInternalFormat::RGBA8);
|
||||
@@ -1155,6 +1432,215 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return m_fallbackTexture2D;
|
||||
}
|
||||
|
||||
SharedPtr<MG_State::GLState::ITextureObject> UniformManager::GetFallbackMultisampleTexture(
|
||||
TextureTarget target, SamplerNumericDomain numericDomain) const {
|
||||
// ONE PLACEHOLDER PER NUMERIC DOMAIN, unlike the single-sampled fallback.
|
||||
//
|
||||
// A descriptor whose image format is in a different numeric class than the sampler that
|
||||
// reads it needs a format-reinterpreting view, and building one needs
|
||||
// VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT on the image. A multisample image can never have it:
|
||||
// SyncTextureResource computes storageImageCapable as `!isMultisampleTexture && ...`, and
|
||||
// the only other source of the bit is the sRGB twin, which RGBA8 is not. So an RGBA8
|
||||
// placeholder handed to a usampler2DMS made GetOrCreateSampledImageView bail with "needs
|
||||
// mutable image format", ResolveSamplerDescriptor return false, and the draw be dropped -
|
||||
// the exact outcome the placeholder exists to prevent, just reached later. Matching the
|
||||
// image's own format to the sampler's class instead means no reinterpreting view is
|
||||
// needed at all.
|
||||
const Bool arrayed = target == TextureTarget::Texture2DMultisampleArray;
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::RGBA8;
|
||||
Uint32 domainSlot = 0;
|
||||
switch (numericDomain) {
|
||||
case SamplerNumericDomain::SignedInteger:
|
||||
internalFormat = TextureInternalFormat::RGBA8I;
|
||||
domainSlot = 1;
|
||||
break;
|
||||
case SamplerNumericDomain::UnsignedInteger:
|
||||
internalFormat = TextureInternalFormat::RGBA8UI;
|
||||
domainSlot = 2;
|
||||
break;
|
||||
case SamplerNumericDomain::Float:
|
||||
case SamplerNumericDomain::Unknown:
|
||||
default:
|
||||
// Unknown reads as float, matching PlaceholderFormatForNumericDomain's own default:
|
||||
// a shader whose sampler class could not be reflected is far likelier to be a plain
|
||||
// sampler2DMS than an integer one, and a float view is the only one buildable without
|
||||
// the mutable bit anyway.
|
||||
break;
|
||||
}
|
||||
const Uint32 key = (arrayed ? kFallbackMultisampleExternalIndexCount / 2 : 0u) + domainSlot;
|
||||
auto cached = m_fallbackMultisampleTextures.find(key);
|
||||
if (cached != m_fallbackMultisampleTextures.end()) {
|
||||
return cached->second;
|
||||
}
|
||||
|
||||
const TextureUploadTarget uploadTarget = arrayed ? TextureUploadTarget::Texture2DMultisampleArray
|
||||
: TextureUploadTarget::Texture2DMultisample;
|
||||
const Uint externalIndex = kFallbackMultisampleExternalIndexBase + key;
|
||||
SharedPtr<MG_State::GLState::TextureObjectMipmap> texture;
|
||||
if (arrayed) {
|
||||
texture = MakeShared<MG_State::GLState::TextureObject2DMultisampleArray>(externalIndex);
|
||||
} else {
|
||||
texture = MakeShared<MG_State::GLState::TextureObject2DMultisample>(externalIndex);
|
||||
}
|
||||
texture->SetInternalFormat(internalFormat);
|
||||
// TWO samples, never one. VUID-RuntimeSpirv-samples-08726 forbids an OpTypeImage with
|
||||
// MS = 1 from reading a VK_SAMPLE_COUNT_1_BIT image, which is exactly the hazard
|
||||
// VkTextureManager::SyncTextureResource's one-sample floor exists to avoid; a placeholder
|
||||
// that re-created it would be worse than none.
|
||||
texture->SetSamples(2);
|
||||
texture->SetFixedSampleLocations(true);
|
||||
// No upload, and MarkStorageDirty(dirty = false) to say so: a multisample image cannot be
|
||||
// written by a transfer at all - it deliberately carries no TRANSFER_DST usage - so unlike
|
||||
// the 2D fallback this one cannot be given (0, 0, 0, 1) content. Its texels are undefined,
|
||||
// which is precisely what GL 4.6 core 8.17 promises for a texelFetch on a multisample
|
||||
// texture that is not complete. The point of the placeholder is that the DRAW happens.
|
||||
texture->AllocateStorage(uploadTarget, 0, {.texelSize = {1, 1, 1}, .byteSize = 0});
|
||||
texture->TruncateMipmapLevels(uploadTarget, 1);
|
||||
texture->MarkStorageDirty(uploadTarget, 0, false);
|
||||
// Worth knowing if it ever fires: an integer multisample format can legitimately support
|
||||
// no count above one on a device (framebufferIntegerColorSampleCounts is allowed to be
|
||||
// VK_SAMPLE_COUNT_1_BIT), and SyncTextureResource's round-down would then hand this
|
||||
// placeholder a single-sampled image, which is the samples-08726 shape the SetSamples(2)
|
||||
// above exists to avoid. It already warns from there; nothing better is available - a
|
||||
// one-sample integer image is still a draw, and declining is the outcome this whole
|
||||
// placeholder replaced.
|
||||
MGLOG_D("UniformManager::GetFallbackMultisampleTexture: created placeholder target=%d domain=%d format=%d",
|
||||
static_cast<Int>(target), static_cast<Int>(numericDomain), static_cast<Int>(internalFormat));
|
||||
return m_fallbackMultisampleTextures.emplace(key, Move(texture)).first->second;
|
||||
}
|
||||
|
||||
VkBufferView UniformManager::AcquireUnboundTexelBufferView(VkFormat declaredFormat,
|
||||
SamplerNumericDomain numericDomain, Bool storage) {
|
||||
MOBILEGL_ASSERT(m_bufferManager != nullptr, "AcquireUnboundTexelBufferView: buffer manager is null");
|
||||
const VkFormatFeatureFlags requiredFeature = storage ? VK_FORMAT_FEATURE_STORAGE_TEXEL_BUFFER_BIT
|
||||
: VK_FORMAT_FEATURE_UNIFORM_TEXEL_BUFFER_BIT;
|
||||
const VkFormat fallbackFormat = PlaceholderFormatForNumericDomain(numericDomain);
|
||||
|
||||
VkFormat format = declaredFormat;
|
||||
if (format == VK_FORMAT_UNDEFINED || !BufferFormatSupportsFeature(m_physicalDevice, format, requiredFeature)) {
|
||||
// The declared format is what a shader that WRITES through this descriptor is
|
||||
// validated against, so it is tried first and kept whenever the device can use it.
|
||||
// Falling back is for the two cases where it cannot be: a sampled texel buffer, which
|
||||
// declares no format at all, and a device that does not list the declared one as a
|
||||
// texel buffer. The fallback stays inside the shader's numeric class, which is the
|
||||
// part the descriptor is checked on for a formatless declaration - and the R32
|
||||
// members of the three classes are mandatory-support formats, so this cannot fail for
|
||||
// want of device features.
|
||||
format = fallbackFormat;
|
||||
}
|
||||
if (format == VK_FORMAT_UNDEFINED || !BufferFormatSupportsFeature(m_physicalDevice, format, requiredFeature)) {
|
||||
MGLOG_E_ONCE("AcquireUnboundTexelBufferView: no usable placeholder format (declared=%d fallback=%d "
|
||||
"storage=%s)",
|
||||
static_cast<Int>(declaredFormat), static_cast<Int>(fallbackFormat),
|
||||
storage ? "true" : "false");
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
const Uint64 key = (static_cast<Uint64>(format) << 1) | (storage ? 1ull : 0ull);
|
||||
const auto cached = m_unboundTexelBufferViews.find(key);
|
||||
if (cached != m_unboundTexelBufferViews.end()) {
|
||||
return cached->second;
|
||||
}
|
||||
|
||||
const BufferSlice placeholder = m_bufferManager->AcquireUnboundTexelBufferDescriptor();
|
||||
if (!placeholder.IsValid()) {
|
||||
MGLOG_E_ONCE("AcquireUnboundTexelBufferView: placeholder buffer unavailable");
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
// A buffer view's range must be a whole number of texels of its own format, and the
|
||||
// placeholder is sized for the largest of them - so floor rather than assume.
|
||||
const VkDeviceSize texelSize = std::max<VkDeviceSize>(1, vkuFormatTexelBlockSize(format));
|
||||
const VkDeviceSize range = (placeholder.size / texelSize) * texelSize;
|
||||
if (range == 0) {
|
||||
MGLOG_E_ONCE("AcquireUnboundTexelBufferView: placeholder holds no whole texel of format=%d",
|
||||
static_cast<Int>(format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
VkBufferViewCreateInfo viewInfo{};
|
||||
viewInfo.sType = VK_STRUCTURE_TYPE_BUFFER_VIEW_CREATE_INFO;
|
||||
viewInfo.buffer = placeholder.buffer;
|
||||
viewInfo.format = format;
|
||||
viewInfo.offset = placeholder.offset;
|
||||
viewInfo.range = range;
|
||||
|
||||
VkBufferView view = VK_NULL_HANDLE;
|
||||
const VkResult result = vkCreateBufferView(m_device, &viewInfo, nullptr, &view);
|
||||
if (result != VK_SUCCESS || view == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("AcquireUnboundTexelBufferView: vkCreateBufferView failed result=%d format=%d", result,
|
||||
static_cast<Int>(format));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
m_unboundTexelBufferViews.emplace(key, view);
|
||||
MGLOG_D("AcquireUnboundTexelBufferView: created placeholder view format=%d storage=%s",
|
||||
static_cast<Int>(format), storage ? "true" : "false");
|
||||
return view;
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveUnboundStorageImagePlaceholder(const ProgramFactory::VkProgramObject& programObj,
|
||||
Uint32 binding, TextureTarget& outTarget,
|
||||
VkFormat& outFormat) const {
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerTextureTargetByBinding.size(),
|
||||
"ResolveUnboundStorageImagePlaceholder: binding %u out of range", binding);
|
||||
MOBILEGL_ASSERT(binding < programObj.storageImageFormatByBinding.size(),
|
||||
"ResolveUnboundStorageImagePlaceholder: format binding %u out of range", binding);
|
||||
outTarget = programObj.samplerTextureTargetByBinding[binding];
|
||||
// The shader's own format qualifier, exactly as the bound path prefers it over the one
|
||||
// glBindImageTexture named - there is no binding here to name one. A `writeonly` image
|
||||
// may carry no qualifier at all; its numeric class is then the only constraint, and the
|
||||
// R32 member of that class is what carries it (see AcquireUnboundTexelBufferView).
|
||||
outFormat = programObj.storageImageFormatByBinding[binding];
|
||||
if (outFormat == VK_FORMAT_UNDEFINED) {
|
||||
outFormat = PlaceholderFormatForNumericDomain(programObj.samplerNumericDomainByBinding[binding]);
|
||||
}
|
||||
return outFormat != VK_FORMAT_UNDEFINED && PlaceholderShapeForTarget(outTarget).valid;
|
||||
}
|
||||
|
||||
SharedPtr<MG_State::GLState::ITextureObject> UniformManager::GetUnboundStorageImageTexture(
|
||||
TextureTarget target, VkFormat format) const {
|
||||
const Uint64 key = (static_cast<Uint64>(target) << 32) | static_cast<Uint32>(format);
|
||||
const auto cached = m_unboundStorageImageTextures.find(key);
|
||||
if (cached != m_unboundStorageImageTextures.end()) {
|
||||
return cached->second;
|
||||
}
|
||||
|
||||
const PlaceholderShape shape = PlaceholderShapeForTarget(target);
|
||||
if (!shape.valid) {
|
||||
// A multisample image uniform is the case with no answer here: its descriptor demands
|
||||
// a multisample view, and a single-sampled 1x1 image is invalid Vulkan in that slot,
|
||||
// not a degraded picture. The caller declines the binding exactly as it did before.
|
||||
MGLOG_D("GetUnboundStorageImageTexture: no placeholder shape for target=%d", static_cast<Int>(target));
|
||||
return nullptr;
|
||||
}
|
||||
const TextureInternalFormat internalFormat = InternalFormatForVkFormat(format);
|
||||
if (internalFormat == TextureInternalFormat::Unknown) {
|
||||
MGLOG_E_ONCE("GetUnboundStorageImageTexture: no GL internal format matches VkFormat=%d",
|
||||
static_cast<Int>(format));
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto texture = MakePlaceholderTextureObject(target, kUnboundStorageImageExternalIndex);
|
||||
if (texture == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
texture->SetInternalFormat(internalFormat);
|
||||
const SizeT texelBytes = MG_Util::GetSizedInternalFormatSizeInBytes(internalFormat);
|
||||
for (Uint32 index = 0; index < shape.uploadTargetCount; ++index) {
|
||||
texture->AllocateStorage(shape.uploadTargets[index], 0,
|
||||
{.texelSize = {1, 1, shape.depth},
|
||||
.byteSize = texelBytes * static_cast<SizeT>(shape.depth)});
|
||||
// Not dirty: there is deliberately nothing to upload. The image is created and
|
||||
// transitioned to GENERAL by the storage-image preparation pass like any other, and
|
||||
// its contents are exactly as undefined as GL says a fetch through an unbound image
|
||||
// unit is.
|
||||
texture->MarkStorageDirty(shape.uploadTargets[index], 0, false);
|
||||
}
|
||||
m_unboundStorageImageTextures.emplace(key, texture);
|
||||
MGLOG_D("GetUnboundStorageImageTexture: created placeholder target=%d format=%d", static_cast<Int>(target),
|
||||
static_cast<Int>(format));
|
||||
return texture;
|
||||
}
|
||||
|
||||
Bool UniformManager::ResolveSampledBinding(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Uint32 binding, Uint32 element,
|
||||
@@ -1163,7 +1649,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Open-coded ResolveSamplerTextureRaw so the unit is resolved once for both the
|
||||
// texture and the sampler override - this runs per binding per full-path draw,
|
||||
// and program-alternating draw streams take the full path on every draw.
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveSampledBinding: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveSampledBinding: GL context is null");
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerUniformLocationByBinding.size(),
|
||||
"ResolveSampledBinding: sampler location binding %u out of range", binding);
|
||||
MOBILEGL_ASSERT(binding < programObj.samplerTextureTargetByBinding.size(),
|
||||
@@ -1174,29 +1660,53 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
const Int unit = ResolveSamplerUnitIndex(program, location, binding);
|
||||
auto& textureUnit = MG_State::pGLContext->GetTextureUnitObject(unit);
|
||||
auto& textureUnit = MGB_CTX->GetTextureUnitObject(unit);
|
||||
const TextureTarget preferredTarget = programObj.samplerTextureTargetByBinding[binding];
|
||||
MG_State::GLState::ITextureObject* texture =
|
||||
textureUnit.GetBindingSlot(preferredTarget).GetBoundObject().get();
|
||||
// The sampler in effect, resolved BEFORE the completeness test below rather than after:
|
||||
// GL's completeness rules are a property of (texture, sampler in effect), so the test
|
||||
// cannot be asked without it.
|
||||
const auto& samplerOverride = textureUnit.GetSamplerObject();
|
||||
const MG_State::GLState::SamplerObject* effectiveSampler =
|
||||
samplerOverride ? samplerOverride.get()
|
||||
: (texture != nullptr ? texture->GetSamplerObject().get() : nullptr);
|
||||
// Undefined default texture (name 0, no image) resolves as "unbound", exactly
|
||||
// like ResolveSamplerTextureRaw reports it.
|
||||
if (MG_State::GLState::IsUndefinedDefaultTexture(texture)) {
|
||||
texture = nullptr;
|
||||
}
|
||||
// ...and so does a texture that fails the completeness rules for the filter in effect,
|
||||
// because that is precisely what ResolveSamplerDescriptor does with it. The two used to
|
||||
// disagree: this one asked only whether the default texture was UNDEFINED, so a default
|
||||
// texture that had been given a base level but no mip chain - which is what the GL-CTS
|
||||
// state reset between test cases leaves behind, and what any application that uploads to
|
||||
// texture 0 has - stayed in the sampled set while the descriptor path swapped it for the
|
||||
// fallback. SetupDraw then synced a texture no descriptor would use, the sync declined
|
||||
// (GL calls it incomplete), and the null it returned was dereferenced one line later.
|
||||
// Keeping the two predicates identical is the invariant; CollectSampledTextures exists to
|
||||
// pre-sync exactly the textures the descriptors will hold.
|
||||
if (MG_State::GLState::SamplesAsIncompleteTexture(texture, effectiveSampler)) {
|
||||
texture = nullptr;
|
||||
}
|
||||
if (texture == nullptr) {
|
||||
// ResolveSamplerDescriptor will substitute the fallback texture for this binding;
|
||||
// include it in the sampled set so the pre-render-pass sync/transition pass covers
|
||||
// its first use instead of leaving that work to happen inside an active pass.
|
||||
if (preferredTarget != TextureTarget::Texture2D &&
|
||||
preferredTarget != TextureTarget::TextureRectangle) {
|
||||
// Ask GetFallbackTexture rather than re-listing the targets it serves: that list grew
|
||||
// a multisample arm and the two must not drift apart.
|
||||
texture = GetFallbackTexture(preferredTarget, programObj.samplerNumericDomainByBinding[binding]).get();
|
||||
if (texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
texture = GetFallbackTexture(preferredTarget).get();
|
||||
// The substitution changed the texture, so the "no override" arm of the effective
|
||||
// sampler has to follow it to the fallback's own.
|
||||
if (!samplerOverride) {
|
||||
effectiveSampler = texture != nullptr ? texture->GetSamplerObject().get() : nullptr;
|
||||
}
|
||||
}
|
||||
const auto& samplerOverride = textureUnit.GetSamplerObject();
|
||||
outTexture = texture;
|
||||
outSampler = samplerOverride ? samplerOverride.get()
|
||||
: (texture != nullptr ? texture->GetSamplerObject().get() : nullptr);
|
||||
outSampler = effectiveSampler;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -1294,7 +1804,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures) const {
|
||||
outTextures.clear();
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr,
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE,
|
||||
"CollectStorageImageTextures: GL context is null");
|
||||
// Same as the sampled walk: a declined program is refused at bind time, and its declined
|
||||
// binding has no uniform location to reach an image unit through.
|
||||
@@ -1338,11 +1848,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto* texture = MG_State::pGLContext->GetImageTextureBinding(imageUnit).Texture.get();
|
||||
auto* texture = MGB_CTX->GetImageTextureBinding(imageUnit).Texture.get();
|
||||
if (texture == nullptr) {
|
||||
MGLOG_E_ONCE("CollectStorageImageTextures: image unit %d is unbound for binding %u element %u",
|
||||
imageUnit, binding, element);
|
||||
return false;
|
||||
// ResolveStorageImageDescriptor will substitute the placeholder image for this
|
||||
// binding; include it here for the same reason the sampled walk includes the
|
||||
// fallback texture - this walk is what gets a storage image created,
|
||||
// STORAGE-usage-marked and transitioned to GENERAL BEFORE the render pass
|
||||
// opens, and all three of those are illegal once it has. A target with no
|
||||
// placeholder shape (multisample) contributes nothing and is declined at
|
||||
// resolve time exactly as it was.
|
||||
TextureTarget placeholderTarget = TextureTarget::Unknown;
|
||||
VkFormat placeholderFormat = VK_FORMAT_UNDEFINED;
|
||||
if (!ResolveUnboundStorageImagePlaceholder(programObj, binding, placeholderTarget,
|
||||
placeholderFormat)) {
|
||||
continue;
|
||||
}
|
||||
texture = GetUnboundStorageImageTexture(placeholderTarget, placeholderFormat).get();
|
||||
if (texture == nullptr) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (std::find(outTextures.begin(), outTextures.end(), texture) == outTextures.end()) {
|
||||
outTextures.push_back(texture);
|
||||
@@ -1362,7 +1886,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<SamplerImageFeedbackBinding>& outBindings) const {
|
||||
outBindings.clear();
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr,
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE,
|
||||
"CollectSamplerImageFeedback: GL context is null");
|
||||
if (programObj.declinedDescriptors) return true;
|
||||
|
||||
@@ -1378,9 +1902,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (!ResolveSampledBinding(program, programObj, samplerBinding, samplerElement,
|
||||
sampledTexture, sampledSampler) ||
|
||||
sampledTexture == nullptr || sampledSampler == nullptr ||
|
||||
MG_State::GLState::SamplesAsIncompleteTexture(sampledTexture, sampledSampler)) {
|
||||
IsPlaceholderTexture(sampledTexture)) {
|
||||
// ResolveSamplerDescriptor uses a fallback in these cases, which cannot
|
||||
// alias the image-unit binding of the original texture.
|
||||
// alias the image-unit binding of the original texture. The unbound and
|
||||
// incomplete cases both arrive here AS that fallback now that
|
||||
// ResolveSampledBinding applies the completeness rule itself, so the test is
|
||||
// "is this one of ours" rather than a second completeness check.
|
||||
continue;
|
||||
}
|
||||
// Multisample source images intentionally omit TRANSFER_SRC usage. Keep their existing
|
||||
@@ -1409,7 +1936,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (imageUnit < 0 || imageUnit >= MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS) {
|
||||
return false;
|
||||
}
|
||||
const auto& image = MG_State::pGLContext->GetImageTextureBinding(imageUnit);
|
||||
const auto& image = MGB_CTX->GetImageTextureBinding(imageUnit);
|
||||
// A sampler view exposes all layers of its target; equal texture plus an
|
||||
// overlapping mip therefore aliases the writable image subresource.
|
||||
if (image.Texture.get() == sampledTexture &&
|
||||
@@ -1439,7 +1966,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const void* outData = nullptr;
|
||||
VkDeviceSize outSize = 0;
|
||||
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext != nullptr, "ResolveUniformBufferPayload: GL context is null");
|
||||
MOBILEGL_ASSERT(MGB_CTX_LIVE, "ResolveUniformBufferPayload: GL context is null");
|
||||
MOBILEGL_ASSERT(binding < programObj.bindingKinds.size(),
|
||||
"ResolveUniformBufferPayload: binding %u out of range", binding);
|
||||
MOBILEGL_ASSERT(programObj.bindingKinds[binding] == ProgramFactory::DescriptorBindingKind::UniformBufferDynamic,
|
||||
@@ -1484,12 +2011,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const Uint32 frontendBinding = program.GetUniformBlockBinding(static_cast<Uint32>(blockIndex));
|
||||
const Uint32 uniformBindingPointCount =
|
||||
static_cast<Uint32>(MG_State::pGLContext->GetBufferBindingPointCount(BufferTarget::Uniform));
|
||||
static_cast<Uint32>(MGB_CTX->GetBufferBindingPointCount(BufferTarget::Uniform));
|
||||
MOBILEGL_ASSERT(frontendBinding < uniformBindingPointCount,
|
||||
"ResolveUniformBufferPayload: frontend UBO binding %u out of range for block '%s'",
|
||||
frontendBinding, program.GetUniformBlockName(static_cast<Uint32>(blockIndex)).c_str());
|
||||
|
||||
auto& bindingPoint = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::Uniform, frontendBinding);
|
||||
auto& bindingPoint = MGB_CTX->GetBufferBindingPoint(BufferTarget::Uniform, frontendBinding);
|
||||
const auto& bufferObject = bindingPoint.GetBoundObject();
|
||||
MOBILEGL_ASSERT(bufferObject != nullptr,
|
||||
"ResolveUniformBufferPayload: no UBO bound at frontend binding %u for block '%s'",
|
||||
@@ -1551,6 +2078,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
out.dynamicOffset = rangeStart;
|
||||
}
|
||||
}
|
||||
if (MG_Util::PipeStats::Enabled() && !out.directBindable) {
|
||||
// D-B8: the bytes Magma repacks into its own UBO ring, i.e. exactly the host
|
||||
// payload a split build would have to ship with set_shader_buffers. Espryt binds
|
||||
// the frontend buffer to the driver and contributes nothing here, which is why
|
||||
// the class is named for the payload and not for the call. Counted AFTER the
|
||||
// zero-copy direct-bind decision: a direct bind repacks nothing, and counting it
|
||||
// here reported a copy that never happened.
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageUboNamed,
|
||||
static_cast<Uint64>(outSize));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -1758,6 +2295,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
outBuffer = slice.buffer;
|
||||
outRange = ubo.payloadSize;
|
||||
outDynamicOffset = static_cast<Uint32>(slice.offset);
|
||||
if (isGlobalUbo && MG_Util::PipeStats::Enabled()) {
|
||||
// Magma's half of stage-ubo-global, so the class means the same on both
|
||||
// backends. The memo hit above returns before this, so a frame that reuses the
|
||||
// slice correctly contributes nothing.
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageUboGlobal,
|
||||
static_cast<Uint64>(ubo.payloadSize));
|
||||
}
|
||||
if (isGlobalUbo) {
|
||||
m_globalUboMemo[m_globalUboMemoNext] =
|
||||
GlobalUboSliceMemo{uboProgramLifetimeId, uboFrameSerial, uboContentVersion,
|
||||
|
||||
@@ -42,7 +42,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
SamplerNumericDomain numericDomain = SamplerNumericDomain::Unknown;
|
||||
};
|
||||
|
||||
Bool Initialize(VkDevice device, VkBufferManager* bufferManager,
|
||||
// `physicalDevice` is only ever asked for format properties: a placeholder descriptor for
|
||||
// an unbound texel-buffer binding has to be built from a format the DEVICE accepts as a
|
||||
// texel buffer, and there is no other route to that answer from here.
|
||||
Bool Initialize(VkDevice device, VkPhysicalDevice physicalDevice, VkBufferManager* bufferManager,
|
||||
ProgramFactory* programFactory,
|
||||
VkDeviceSize minUniformBufferOffsetAlignment, Uint32 frameCount,
|
||||
Uint32 maxBindings = 16, Uint32 setsPerFrame = 64,
|
||||
@@ -176,7 +179,43 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static MG_State::GLState::ITextureObject* ResolveSamplerTextureRaw(
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding, Uint32 element);
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetFallbackTexture(TextureTarget target) const;
|
||||
// `numericDomain` is the sampler's class, and it matters only for the multisample arm -
|
||||
// see GetFallbackMultisampleTexture for why the single-sampled fallback can ignore it.
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetFallbackTexture(
|
||||
TextureTarget target, SamplerNumericDomain numericDomain) const;
|
||||
// The multisample arm of GetFallbackTexture. One object per (target, numeric domain) and
|
||||
// no upload path: a multisample image cannot be written by a transfer, so its texels stay
|
||||
// undefined - which is what GL promises for a texelFetch on an incomplete multisample
|
||||
// texture - and it cannot carry MUTABLE_FORMAT, so its format has to match the sampler's
|
||||
// class outright rather than being reinterpreted at view time.
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetFallbackMultisampleTexture(
|
||||
TextureTarget target, SamplerNumericDomain numericDomain) const;
|
||||
// ---- placeholders for UNBOUND image-backed descriptors -------------------------
|
||||
// GL lets a program declare `samplerBuffer`, `imageBuffer` or `image2D` and bind nothing
|
||||
// to the unit it names: the fetch is then undefined (GL 4.6 core 8.9 for an incomplete
|
||||
// buffer texture, 8.26 for an image unit with no texture) - undefined VALUES, not a
|
||||
// dropped draw. Vulkan has no unwritten descriptor, so something valid has to sit in the
|
||||
// set or the whole draw or dispatch is lost, which is what these two build. Same shape as
|
||||
// VkBufferManager::AcquireUnboundStorageDescriptor, one level up: per FORMAT rather than
|
||||
// one shared object, because a descriptor whose format disagrees with the shader's
|
||||
// declaration is invalid Vulkan even when nothing ever reads it.
|
||||
//
|
||||
// `declaredFormat` is the format the SHADER declared (VK_FORMAT_UNDEFINED for a sampled
|
||||
// texel buffer, which never carries one, or for a formatless `writeonly` image);
|
||||
// `numericDomain` decides the format when there is no declaration and is the fallback
|
||||
// class when the device cannot use the declared one as a texel buffer.
|
||||
VkBufferView AcquireUnboundTexelBufferView(VkFormat declaredFormat, SamplerNumericDomain numericDomain,
|
||||
Bool storage);
|
||||
// A 1x1 (x1 layer, or 6 faces for a cube) texture of `format`, shaped for `target` so the
|
||||
// view the descriptor gets has the view type the shader's image declaration demands.
|
||||
// Null for a target with no single-sampled placeholder shape - multisample images, whose
|
||||
// descriptor needs a multisample view that this cannot stand in for.
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetUnboundStorageImageTexture(TextureTarget target,
|
||||
VkFormat format) const;
|
||||
// The (target, format) pair a storage-image binding's placeholder is keyed by, resolved
|
||||
// from reflection alone. False when the binding has no placeholder shape.
|
||||
Bool ResolveUnboundStorageImagePlaceholder(const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
TextureTarget& outTarget, VkFormat& outFormat) const;
|
||||
// `element` indexes a sampler ARRAY inside one binding; each element carries its own
|
||||
// independently assigned GL texture unit, so it selects the texture, the sampler
|
||||
// override and the fallback separately from its neighbours.
|
||||
@@ -251,6 +290,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDescriptorSet& outDescriptorSet);
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VkBufferManager* m_bufferManager = nullptr;
|
||||
ProgramFactory* m_programFactory = nullptr;
|
||||
Vector<FrameResources> m_frames;
|
||||
@@ -263,6 +303,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkTextureManager* m_textureManager = nullptr;
|
||||
VkSamplerManager* m_samplerManager = nullptr;
|
||||
mutable SharedPtr<MG_State::GLState::ITextureObject> m_fallbackTexture2D;
|
||||
// Keyed by (arrayed, numeric domain); see GetFallbackMultisampleTexture. Lazily populated,
|
||||
// never evicted - at most six tiny 1x1 images - and torn down with the manager.
|
||||
mutable UnorderedMap<Uint32, SharedPtr<MG_State::GLState::ITextureObject>> m_fallbackMultisampleTextures;
|
||||
// See AcquireUnboundTexelBufferView / GetUnboundStorageImageTexture. Both are lazily
|
||||
// populated, never evicted (a program's declared formats are a fixed, tiny set) and torn
|
||||
// down with the manager. The texel views are keyed by format AND by storage-vs-sampled
|
||||
// because the two descriptor kinds demand different format FEATURES of the device, so one
|
||||
// format can be usable for one and not the other. Deliberately NOT the per-frame
|
||||
// texelBufferViews list: those are destroyed at every frame boundary, and these must
|
||||
// outlive it or the placeholder would be rebuilt for every unbound binding every frame.
|
||||
UnorderedMap<Uint64, VkBufferView> m_unboundTexelBufferViews;
|
||||
mutable UnorderedMap<Uint64, SharedPtr<MG_State::GLState::ITextureObject>> m_unboundStorageImageTextures;
|
||||
|
||||
// Per-draw scratch buffers for BindProgramUniformBuffers: reused (clear keeps
|
||||
// capacity) so the descriptor-write path stops allocating on every draw.
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
// End of Source File Header
|
||||
|
||||
#include "VertexInputStateFactory.h"
|
||||
#include "MagmaPipeArms.h"
|
||||
#include "MG_Util/Converters/MGToStr/DataTypeConverter.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <utility>
|
||||
@@ -45,25 +46,149 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// capture came back holding a dead VAO's vertex data (0,0,0,1 - the previous
|
||||
// test's positions) instead of its own.
|
||||
// Zero for client memory (no buffer), which is a distinct identity of its own.
|
||||
const Uint64 bufferKey = attr.Buffer ? attr.Buffer->GetLifetimeId() : 0;
|
||||
//
|
||||
// P2 D12.4 / ARCHITECTURE.md 9.5: under the handle arm the identity is the
|
||||
// buffer's {slot, gen} rather than its lifetime id - "lifetimeId -> gen mixed
|
||||
// into every server-side content hash". The two are equally ABA-proof (the
|
||||
// allocator maps one onto the other and bumps Gen only on slot REUSE); what
|
||||
// changes is that the key is now the identity the SERVER will be handed once
|
||||
// buffers travel as handles, instead of a number only the client can mint.
|
||||
Uint64 bufferKey = attr.Buffer ? attr.Buffer->GetLifetimeId() : 0;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
if (attr.Buffer) {
|
||||
// The SAME arm question the other four re-keyed sites ask, through the same
|
||||
// helper: a site that decided for itself could silently key on the pre-handle
|
||||
// identity while its neighbours keyed on the handle.
|
||||
if (MagmaPipeTrackHArmIsHandles(MG_Pipe::kMGPipeSubsystemMagmaVertexInput)) {
|
||||
const MG_Pipe::MGPipeHandle handle =
|
||||
m_identity->HandleOf(MG_Pipe::MGPipeKind::Buffer, attr.Buffer->GetLifetimeId());
|
||||
bufferKey = static_cast<Uint64>(handle.Slot) | (static_cast<Uint64>(handle.Gen) << 32);
|
||||
}
|
||||
if (MagmaPipeAbaControlDefeatsIdentity()) {
|
||||
// Negative control C (P2 brief D18), on WHICHEVER arm this run is on - the
|
||||
// pre-handle lifetime id and the handle's {slot, gen} are the same guard
|
||||
// wearing two hats, and a control that defeated only the retired one would
|
||||
// say nothing about the key P2 ships.
|
||||
//
|
||||
// The identity is replaced by a constant rather than by the raw
|
||||
// BufferObject*, because the address is not recycled in practice and so
|
||||
// never collides (see MagmaPipeAbaControlDefeatsIdentity). Zero is what a
|
||||
// key with NO buffer identity in it looks like - the exact defect this
|
||||
// hash was fixed for: "the hash is what TryBindResolvedVertexBindings
|
||||
// accepts as proof that a memoised binding still reads the buffer it was
|
||||
// resolved from", and with the identity gone it accepts a binding resolved
|
||||
// from a different buffer. HandleRecycleScenario.AbaControl then draws a
|
||||
// replacement VAO and gets its dead predecessor's vertex data.
|
||||
bufferKey = 0;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &bufferKey, sizeof(bufferKey)));
|
||||
}
|
||||
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
VertexInputStateFactory::VaoBackendMemos& VertexInputStateFactory::MemosFor(
|
||||
const MG_State::GLState::VertexArrayObject& vao) const {
|
||||
const MG_Pipe::MGPipeHandle handle =
|
||||
m_identity->HandleOf(MG_Pipe::MGPipeKind::VertexElementsCso, vao.GetLifetimeId());
|
||||
// One entry per mintable slot, grown on demand: the mint has no capacity, so neither
|
||||
// does this, and no two live VAOs can share an entry however large the working set is.
|
||||
// There is no probe in front of it because the mint itself is one - a one-entry memo
|
||||
// hit for every acquisition after this draw's first, and a hash probe otherwise.
|
||||
//
|
||||
// The claim rule - the slot picks the entry, the whole handle (Gen included) decides
|
||||
// whose it is - and negative control C's defeat of it are MagmaPipeArms.h's
|
||||
// MagmaPipeClaimSlotMemos, so that the unit suite which drives a REAL slot reuse
|
||||
// (MG_Test/Pipe/MagmaPipeIdentityTest.cpp) exercises this code and not a copy of it.
|
||||
// What the control defeats HERE is the identity that SELECTS the entry: every VAO
|
||||
// collapses onto one, handed back uncleared, so the replacement inherits the dead
|
||||
// VAO's content hash and its resolved-entry pointer. The GENERATION half is the unit
|
||||
// suite's business, for the reason MagmaPipeAbaControlDefeatsIdentity spells out.
|
||||
return MagmaPipeClaimSlotMemos(m_vaoMemos, handle);
|
||||
}
|
||||
#endif
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
Bool VertexInputStateFactory::TryGetMemoizedHash(const MG_State::GLState::VertexArrayObject& vao,
|
||||
Uint64& outHash) const {
|
||||
if (MagmaPipeTrackHArmIsHandles(MG_Pipe::kMGPipeSubsystemMagmaVertexInput)) {
|
||||
const VaoBackendMemos& memos = MemosFor(vao);
|
||||
if (memos.HashConfigVersion != vao.GetConfigVersion()) return false;
|
||||
outHash = memos.Hash;
|
||||
return true;
|
||||
}
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
return vao.GetBackendHashMemo(outHash);
|
||||
#else
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
VertexInputStateFactory::HashType VertexInputStateFactory::GetOrComputeHash(
|
||||
const MG_State::GLState::VertexArrayObject& vao) const {
|
||||
HashType hash = 0;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P2 D12.5: the same memo, on the backend's side of the boundary.
|
||||
if (MagmaPipeTrackHArmIsHandles(MG_Pipe::kMGPipeSubsystemMagmaVertexInput)) {
|
||||
VaoBackendMemos& memos = MemosFor(vao);
|
||||
if (memos.HashConfigVersion == vao.GetConfigVersion()) {
|
||||
return memos.Hash;
|
||||
}
|
||||
hash = ComputeHash(vao);
|
||||
memos.Hash = hash;
|
||||
memos.HashConfigVersion = vao.GetConfigVersion();
|
||||
return hash;
|
||||
}
|
||||
#endif
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
if (!vao.GetBackendHashMemo(hash)) {
|
||||
hash = ComputeHash(vao);
|
||||
vao.SetBackendHashMemo(hash);
|
||||
}
|
||||
#endif
|
||||
return hash;
|
||||
}
|
||||
|
||||
const VertexInputStateFactory::BackendVertexInputState& VertexInputStateFactory::GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao) {
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P2 D12.5: the same per-draw fast path, but the resolved-entry pointer lives in this
|
||||
// factory's slot-indexed table instead of on the frontend VAO. The eviction epoch
|
||||
// survives the move and is still what stops a stale pointer being dereferenced: the
|
||||
// POINTEE is a cache entry this factory can erase at a frame boundary, and moving the
|
||||
// memo does not change that.
|
||||
if (MagmaPipeTrackHArmIsHandles(MG_Pipe::kMGPipeSubsystemMagmaVertexInput)) {
|
||||
VaoBackendMemos& memos = MemosFor(vao);
|
||||
if (memos.StateConfigVersion == vao.GetConfigVersion() && memos.State != nullptr &&
|
||||
memos.StateEpoch == m_evictionEpoch) {
|
||||
const auto* memoEntry = static_cast<const BackendVertexInputState*>(memos.State);
|
||||
memoEntry->lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return *memoEntry;
|
||||
}
|
||||
const BackendVertexInputState& resolved =
|
||||
GetOrCreateVertexInputState(vao, GetOrComputeHash(vao));
|
||||
// MemosFor is re-taken rather than kept live across GetOrCreateVertexInputState:
|
||||
// the reference is not worth holding across a call that can resize the table.
|
||||
VaoBackendMemos& stamp = MemosFor(vao);
|
||||
stamp.State = &resolved;
|
||||
stamp.StateEpoch = m_evictionEpoch;
|
||||
stamp.StateConfigVersion = vao.GetConfigVersion();
|
||||
// The AUX memo is deliberately NOT stamped here: its two words already live in
|
||||
// VulkanRenderer::VaoDrawMemo (layoutHash / layoutAuxMasks) and its getter has no
|
||||
// live reader anywhere, so the handle arm retires it rather than moving it.
|
||||
return resolved;
|
||||
}
|
||||
#endif
|
||||
#if !MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
// Unreachable: with no legacy arm compiled MagmaPipeTrackHArmIsHandles is a compile-
|
||||
// time true, so the handle arm above always returns. Written out rather than left to
|
||||
// fall off the end so the function still has a return on every path a compiler sees.
|
||||
return GetOrCreateVertexInputState(vao, GetOrComputeHash(vao));
|
||||
#else
|
||||
// Per-draw fast path: the VAO carries a pointer to its resolved entry,
|
||||
// valid while its config version and the cache's eviction epoch both
|
||||
// match - no re-hash, no map lookup.
|
||||
@@ -83,6 +208,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
vao.SetBackendAuxMemo(entry.layoutHash,
|
||||
PackVertexInputAuxMasks(entry.unsupportedAttribMask, entry.attributeLocationMask));
|
||||
return entry;
|
||||
#endif // MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
}
|
||||
|
||||
const VertexInputStateFactory::BackendVertexInputState& VertexInputStateFactory::GetOrCreateVertexInputState(
|
||||
@@ -316,8 +442,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Invalidate every VAO's state-pointer memo: the erased node's
|
||||
// address may be reused by a future insert. Advance through the
|
||||
// process-wide source so the value stays unique across factory
|
||||
// instances (see the member comment).
|
||||
// instances (see the member comment). With no legacy arm the memos
|
||||
// live in this factory and die with it, so a per-instance bump is
|
||||
// enough - P2 D12.5.
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
m_evictionEpoch = ++s_evictionEpochSource;
|
||||
#else
|
||||
++m_evictionEpoch;
|
||||
#endif
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
|
||||
@@ -7,8 +7,12 @@
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
// MG_Pipe::MGPipeHandle for the P2 D12.5 memo table below. A header of constexpr constants,
|
||||
// so the pull build gains nothing from it.
|
||||
#include <MG_Pipe/MGPipeHandles.h>
|
||||
|
||||
#include "Config.h"
|
||||
#include "MagmaPipeArms.h"
|
||||
#include "VertexInputStateBuilder.h"
|
||||
#include "MG_State/GLState/VertexArrayState/VertexArrayObject.h"
|
||||
#include <Includes.h>
|
||||
@@ -70,8 +74,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
};
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// The mint is the RENDERER's (MagmaPipeIdentityTables), not a process-global and not
|
||||
// this factory's: VulkanRenderer::LookupVaoDrawMemo has to derive the same {slot, gen}
|
||||
// for the same VAO, and a table that outlived the context it was minted for would share
|
||||
// one reclamation clock across two live contexts (review v2 minor 4).
|
||||
VertexInputStateFactory(const VulkanRendererConfig& config, VkPhysicalDevice physicalDevice,
|
||||
MagmaPipeIdentityTables& identity):
|
||||
m_config(config), m_physicalDevice(physicalDevice), m_identity(&identity) {}
|
||||
#else
|
||||
VertexInputStateFactory(const VulkanRendererConfig& config, VkPhysicalDevice physicalDevice):
|
||||
m_config(config), m_physicalDevice(physicalDevice) {}
|
||||
#endif
|
||||
~VertexInputStateFactory() = default;
|
||||
VertexInputStateFactory(const VertexInputStateFactory&) = delete;
|
||||
|
||||
@@ -86,6 +100,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Memoized ComputeHash: reuses the VAO's cached hash while its config version
|
||||
// is unchanged. Use this on per-draw paths.
|
||||
HashType GetOrComputeHash(const MG_State::GLState::VertexArrayObject& vao) const;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// The VAO's content hash IF it has already been memoized, without computing one.
|
||||
// P2 D12.5: the three draw-path readers that used to ask the VAO object this
|
||||
// question ask the factory instead, because that is where the memo lives once the
|
||||
// frontend object stops carrying the backend's state.
|
||||
Bool TryGetMemoizedHash(const MG_State::GLState::VertexArrayObject& vao, Uint64& outHash) const;
|
||||
#endif
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash);
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(const MG_State::GLState::VertexArrayObject& vao);
|
||||
@@ -112,6 +133,44 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static VkFormat ToFloat32VertexFormat(Int componentCount);
|
||||
Bool SupportsVertexBufferFormat(VkFormat format) const;
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// ---- P2 D12.5: the backend's memos, off the frontend VAO and into the backend ----
|
||||
//
|
||||
// The two facts that used to live as `mutable` fields on VertexArrayObject
|
||||
// (Get/SetBackendHashMemo and Get/SetBackendStateMemo), kept here instead, keyed on
|
||||
// the VAO's {slot, gen} and guarded by exactly the same config version. A frontend
|
||||
// state object holding the backend's raw pointer is what P2 retires: under split the
|
||||
// backend is in another process and its cache entry has no address a client could
|
||||
// store, so the memo has to live on the side that owns the pointee.
|
||||
//
|
||||
// The AUX memo is not carried over: its two words moved into VaoDrawMemo::layoutHash
|
||||
// and layoutAuxMasks long ago and its getter has no live reader anywhere in the tree,
|
||||
// so the handle arm simply stops writing it (D12.5 says delete rather than move).
|
||||
struct VaoBackendMemos {
|
||||
// Whose memos these are. The identity table can recycle a slot for a different
|
||||
// VAO under LRU pressure, and the handle compare - Gen included - is what says
|
||||
// the contents are this object's and not its predecessor's.
|
||||
MG_Pipe::MGPipeHandle Owner = MG_Pipe::kMGPipeNullHandle;
|
||||
Uint64 Hash = 0;
|
||||
Uint32 HashConfigVersion = ~0u;
|
||||
const void* State = nullptr;
|
||||
Uint64 StateEpoch = 0;
|
||||
Uint32 StateConfigVersion = ~0u;
|
||||
};
|
||||
// Grow-on-demand (D12.4), one entry per slot the renderer's mint has ever handed
|
||||
// out, and NO CAPACITY: these two memos had none before this package either - they
|
||||
// were unbounded mutable fields on the VertexArrayObject itself - and re-introducing
|
||||
// eviction here is what review v2 rejected. MagmaPipeSlotTable grows in chunks so an
|
||||
// entry reference stays valid across the nested GetOrCreateVertexInputState call.
|
||||
// 48 B per live VAO, reclaimed with the slot when the object goes idle.
|
||||
mutable MagmaPipeSlotTable<VaoBackendMemos> m_vaoMemos;
|
||||
// The renderer's {slot, gen} mint (see the constructor). Never null under push.
|
||||
MagmaPipeIdentityTables* m_identity = nullptr;
|
||||
// The entry belonging to `vao`, claimed (and cleared) if the slot currently holds
|
||||
// someone else's.
|
||||
VaoBackendMemos& MemosFor(const MG_State::GLState::VertexArrayObject& vao) const;
|
||||
#endif
|
||||
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
// Values are heap-allocated: UnorderedMap is open-addressing, so INSERT
|
||||
@@ -130,15 +189,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// stale memo.
|
||||
//
|
||||
// Drawn from a process-wide source, never a per-instance counter: the VAO
|
||||
// memos outlive this factory (they live on pGLContext's VAOs, the renderer
|
||||
// memos outlive this factory (they live on the frontend context's VAOs, the renderer
|
||||
// is destroyed and recreated on EGL surface release/re-create), so a fresh
|
||||
// factory restarting at a dead factory's epoch value would honor its
|
||||
// dangling entry pointers. The constructor takes a value strictly greater
|
||||
// than anything a predecessor ever stamped, so a dead factory's memo can
|
||||
// never compare equal here - the same never-reused idiom as the lifetime ids.
|
||||
// Single-threaded like the rest of the factory (renderer-thread only).
|
||||
//
|
||||
// P2 D12.5: the process-wide source is the LEGACY arm's need. It exists because the
|
||||
// memos live on the frontend VAOs and therefore outlive the factory. The handle arm's
|
||||
// memo table is owned by this factory and dies with it, so a per-instance counter is
|
||||
// enough there and the epoch shrinks back to what it looks like it should be.
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
static inline Uint64 s_evictionEpochSource = 0;
|
||||
Uint64 m_evictionEpoch = ++s_evictionEpochSource;
|
||||
#else
|
||||
Uint64 m_evictionEpoch = 1;
|
||||
#endif
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
#include "../DirectVulkan.h"
|
||||
#include "VulkanRenderer.h"
|
||||
|
||||
#include "MG_Util/Metrics/PipeStats.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
constexpr VmaAllocationCreateFlags kResidentBufferAllocationFlags =
|
||||
@@ -19,6 +21,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// See VkBufferManager::AcquireUnboundStorageDescriptor. 256 bytes: comfortably past
|
||||
// every minStorageBufferOffsetAlignment in the wild, and free.
|
||||
constexpr VkDeviceSize kUnboundStorageDescriptorBytes = 256;
|
||||
// See VkBufferManager::AcquireUnboundTexelBufferDescriptor. The same 256 bytes, for the
|
||||
// same reason plus one: a texel buffer view's range must be a whole number of texels of
|
||||
// whatever format the placeholder is asked for, and 256 divides by every texel size in
|
||||
// the GL image-format table (1, 2, 4, 8 and 16 bytes).
|
||||
constexpr VkDeviceSize kUnboundTexelBufferDescriptorBytes = 256;
|
||||
|
||||
// A zero-copy persistent buffer is created once and never recreated (the app holds
|
||||
// its mapped pointer), and may be bound to any role, so it carries every usage.
|
||||
@@ -135,6 +142,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
m_transientUploadArena.Shutdown();
|
||||
m_unboundStorageBuffer.Destroy();
|
||||
m_unboundTexelBuffer.Destroy();
|
||||
DestroyAllDeferredReleases();
|
||||
ReleaseAllLiveResources();
|
||||
m_copyProvider = nullptr;
|
||||
@@ -223,8 +231,38 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
Bool VkBufferManager::UploadTransient(BufferKind kind, Uint32 frameIndex, const void* data,
|
||||
VkDeviceSize size, VkDeviceSize alignment, BufferSlice& outSlice) {
|
||||
(void)kind;
|
||||
return m_transientUploadArena.Upload(frameIndex, data, size, alignment, outSlice);
|
||||
if (!m_transientUploadArena.Upload(frameIndex, data, size, alignment, outSlice)) {
|
||||
return false;
|
||||
}
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
// The single chokepoint for Magma's per-draw staging. Uniform is deliberately
|
||||
// absent: its bytes are counted by the caller, which is the only place that
|
||||
// knows whether the payload is the default block (stage-ubo-global) or a named
|
||||
// one repacked into the ring (stage-ubo-named), and counting here as well would
|
||||
// double every uniform byte.
|
||||
switch (kind) {
|
||||
case BufferKind::Vertex:
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageVertexClient,
|
||||
static_cast<Uint64>(size));
|
||||
break;
|
||||
case BufferKind::Index:
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageIndexClient,
|
||||
static_cast<Uint64>(size));
|
||||
break;
|
||||
case BufferKind::Indirect:
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageIndirectCmd,
|
||||
static_cast<Uint64>(size));
|
||||
break;
|
||||
case BufferKind::TextureBuffer:
|
||||
case BufferKind::ShaderStorage:
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer,
|
||||
static_cast<Uint64>(size));
|
||||
break;
|
||||
case BufferKind::Uniform:
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkBufferManager::InitializeTransientArenas() {
|
||||
@@ -333,6 +371,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource.pendingFullUpload = true;
|
||||
return false;
|
||||
}
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer, static_cast<Uint64>(size));
|
||||
}
|
||||
resource.pendingFullUpload = false;
|
||||
return true;
|
||||
}
|
||||
@@ -347,6 +388,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static_cast<VkDeviceSize>(size), 16, staging)) {
|
||||
return false;
|
||||
}
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
// The staging fill is the host copy; the vkCmdCopyBuffer below is the device
|
||||
// half of the same bytes and is not counted twice.
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer, static_cast<Uint64>(size));
|
||||
}
|
||||
VkCommandBuffer commandBuffer = m_copyProvider->AcquireBufferCopyCommandBuffer();
|
||||
if (commandBuffer == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
@@ -416,6 +462,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData(), size, 0)) {
|
||||
MGLOG_E_ONCE("VkBufferManager::OnRespecify: in-place upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
} else if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer, static_cast<Uint64>(size));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -441,6 +489,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static_cast<VkDeviceSize>(size), static_cast<VkDeviceSize>(offset))) {
|
||||
MGLOG_E_ONCE("VkBufferManager::OnSubData: host upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
} else if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer,
|
||||
static_cast<Uint64>(size));
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -478,6 +529,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static_cast<VkDeviceSize>(size), static_cast<VkDeviceSize>(offset))) {
|
||||
MGLOG_E_ONCE("VkBufferManager::OnFlushMappedRange: host upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
} else if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer,
|
||||
static_cast<Uint64>(size));
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -548,6 +602,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const Uint8* seed = bufferObject.MappedData();
|
||||
if (seed != nullptr) {
|
||||
resource->buffer.Upload(seed, size, 0);
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
// The one-time seed of a persistent map. Everything the app writes AFTER
|
||||
// this goes straight through the mapping and is persistent-map-push
|
||||
// territory (unwired, D4/D-B4), not this class.
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer,
|
||||
static_cast<Uint64>(size));
|
||||
}
|
||||
}
|
||||
resource->persistentMapped = true;
|
||||
resource->pendingFullUpload = false;
|
||||
@@ -596,6 +657,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource->usageFlags = 0;
|
||||
return false;
|
||||
}
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer,
|
||||
static_cast<Uint64>(size));
|
||||
}
|
||||
resource->pendingFullUpload = false;
|
||||
}
|
||||
|
||||
@@ -675,6 +740,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
outSlice)) {
|
||||
return false;
|
||||
}
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageBuffer, static_cast<Uint64>(size));
|
||||
}
|
||||
resource->transientSlice = outSlice;
|
||||
resource->transientFrameSerial = m_frameSerial;
|
||||
resource->transientChangeSerial = changeSerial;
|
||||
@@ -742,6 +810,39 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return m_unboundStorageBuffer.GetSlice();
|
||||
}
|
||||
|
||||
BufferSlice VkBufferManager::AcquireUnboundTexelBufferDescriptor() {
|
||||
if (!m_unboundTexelBuffer.IsValid()) {
|
||||
if (m_initInfo.allocator == nullptr) {
|
||||
return {};
|
||||
}
|
||||
// A SECOND placeholder rather than more usage bits on the storage-block one. The two
|
||||
// are independent failure domains: a device that refuses this allocation must not
|
||||
// take the storage-block placeholder - and with it the fix this one is a sibling of -
|
||||
// down with it. Host-visible and zero-filled for the same reason as that one: this is
|
||||
// reached from descriptor resolution, inside an already-open recording, which must
|
||||
// not start a copy of its own.
|
||||
const Bool created = m_unboundTexelBuffer.Create({
|
||||
.allocator = m_initInfo.allocator,
|
||||
.size = kUnboundTexelBufferDescriptorBytes,
|
||||
.usage = VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT,
|
||||
.memoryUsage = VMA_MEMORY_USAGE_AUTO,
|
||||
.allocationFlags = VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT |
|
||||
VMA_ALLOCATION_CREATE_MAPPED_BIT,
|
||||
.requiredFlags = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT,
|
||||
});
|
||||
if (!created) {
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireUnboundTexelBufferDescriptor: placeholder creation failed");
|
||||
m_unboundTexelBuffer.Destroy();
|
||||
return {};
|
||||
}
|
||||
if (void* mapped = m_unboundTexelBuffer.GetMappedData()) {
|
||||
Memset(mapped, 0, static_cast<SizeT>(kUnboundTexelBufferDescriptorBytes));
|
||||
}
|
||||
}
|
||||
return m_unboundTexelBuffer.GetSlice();
|
||||
}
|
||||
|
||||
VkBufferUsageFlags VkBufferManager::GetVkBufferUsage(BufferKind kind) {
|
||||
switch (kind) {
|
||||
case BufferKind::Vertex:
|
||||
|
||||
@@ -131,6 +131,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// that indexes past it.
|
||||
BufferSlice AcquireUnboundStorageDescriptor();
|
||||
|
||||
// The store a texel-buffer descriptor - `samplerBuffer` or `imageBuffer` - gets when the
|
||||
// unit the program's uniform names has no buffer texture on it, or the buffer texture on
|
||||
// it has no GL buffer attached. Both are legal GL states that make a fetch return
|
||||
// undefined values (GL 4.6 core 8.9: a buffer texture with no attached buffer object is
|
||||
// incomplete, and sampling an incomplete texture is undefined - not a lost draw), and both
|
||||
// used to take the whole draw or dispatch with them. The VIEW over this - one per format,
|
||||
// and the descriptor is a VkBufferView, not a buffer - is built by
|
||||
// UniformManager::AcquireUnboundTexelBufferView.
|
||||
BufferSlice AcquireUnboundTexelBufferDescriptor();
|
||||
|
||||
// Draw-time acquire for resident (device-storage) buffers: ensures the
|
||||
// resource exists and is fully uploaded, marks it used this frame.
|
||||
Bool AcquireResidentSlice(BufferKind kind, const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
@@ -194,6 +204,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// See AcquireUnboundStorageDescriptor. Lazily created, never re-created, torn down
|
||||
// with the manager.
|
||||
VkBufferObject m_unboundStorageBuffer;
|
||||
// See AcquireUnboundTexelBufferDescriptor. Same lifetime rules.
|
||||
VkBufferObject m_unboundTexelBuffer;
|
||||
IBufferCopyCommandProvider* m_copyProvider = nullptr;
|
||||
Vector<Vector<VkBufferObject>> m_deferredBufferReleases;
|
||||
Vector<Vector<SharedPtr<VkBufferResource>>> m_deferredResourceReleases;
|
||||
|
||||
@@ -8,7 +8,12 @@
|
||||
|
||||
#include "VkClearManager.h"
|
||||
|
||||
// For the shared ResolveAttachmentLayerCount (and the ToVulkanLevelExtent it is built on): the
|
||||
// clear key's layer span has to be the same one the render pass builds its attachment view from.
|
||||
#include "VkTextureManager.h"
|
||||
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/TextureEnumConverter.h"
|
||||
|
||||
@@ -50,7 +55,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (payload.colorEncoding != ClearColorEncoding::Float) return;
|
||||
// With GL_FRAMEBUFFER_SRGB enabled GL performs the encoding itself, so the driver doing it
|
||||
// is exactly right and there is nothing to undo.
|
||||
if (MG_State::pGLContext->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb)) return;
|
||||
if (MGB_CTX->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb)) return;
|
||||
if (ResolveSrgbAttachmentWriteFormat(destinationFormat, false) == destinationFormat) return;
|
||||
|
||||
// sRGB -> linear (GL 4.6 core 8.24), applied to the colour channels only: alpha is stored
|
||||
@@ -100,13 +105,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return ResolveAttachmentBaseArrayLayer(uploadTarget);
|
||||
}
|
||||
|
||||
static Uint32 ResolveAttachmentLayerCount(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (attachment.IsLayered()) {
|
||||
return static_cast<Uint32>(std::max(attachment.GetSize().z(), 1));
|
||||
}
|
||||
return 1u;
|
||||
}
|
||||
// ResolveAttachmentLayerCount used to be duplicated here, reading attachment.GetSize().z()
|
||||
// raw - no ToVulkanLevelExtent remap for a 1D array, no six-faces arm for a cube map. That is
|
||||
// not a cosmetic difference: the count below is not key-only, it is written straight into
|
||||
// VkImageSubresourceRange::layerCount by MaterializePendingClearForTexture, which then POPS
|
||||
// the entry - so a layered cube map's glClear reached one face and the other five were lost
|
||||
// for good, while the very same queued clear cleared all six through the render pass's
|
||||
// LOAD_OP_CLEAR. The helper now lives once, in VkTextureManager.h beside ToVulkanLevelExtent.
|
||||
|
||||
static const MG_State::GLState::FramebufferAttachmentObject* GetClearableAttachment(
|
||||
const MG_State::GLState::FramebufferObject& drawFbo, FramebufferAttachmentType attachmentType) {
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static Bool TryResolveSampleCountFlagBits(Int requestedSamples, VkSampleCountFlagBits& outSampleCount) {
|
||||
@@ -86,25 +87,39 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return ToStorageArrayLayer(texture, face);
|
||||
}
|
||||
|
||||
// The attachment's size is GL geometry, and GL_TEXTURE_1D_ARRAY keeps its layer count in the
|
||||
// state-side HEIGHT rather than in z (see ToVulkanLevelExtent, which exists for exactly this
|
||||
// remap). Reading z directly gave every layered 1D-array attachment layerCount = 1, so a
|
||||
// geometry shader writing gl_Layer = 1..n had its output silently dropped and the parent's
|
||||
// upper layers were never written at all.
|
||||
static Uint32 ResolveAttachmentLayerCount(const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (attachment.IsLayered()) {
|
||||
const auto& texture = attachment.GetTexture();
|
||||
const TextureTarget target = texture != nullptr ? texture->GetTarget() : TextureTarget::Unknown;
|
||||
return static_cast<Uint32>(std::max(ToVulkanLevelExtent(target, attachment.GetSize()).z(), 1));
|
||||
}
|
||||
return 1u;
|
||||
}
|
||||
// ResolveAttachmentLayerCount lives in VkTextureManager.h, beside ToVulkanLevelExtent, because
|
||||
// VkClearManager needs the SAME answer: its pending-clear key's layerCount becomes a real
|
||||
// VkImageSubresourceRange when a clear is materialised outside a render pass. See the header.
|
||||
|
||||
// VUID-VkFramebufferCreateInfo-flags-04113: every view handed to vkCreateFramebuffer must have
|
||||
// been created as VK_IMAGE_VIEW_TYPE_2D or VK_IMAGE_VIEW_TYPE_2D_ARRAY. The image's OWN view
|
||||
// type is not a legal answer for several of the targets GL can attach, and returning it
|
||||
// unchanged is what took the process down on every layered 3D / cube-map-array attachment:
|
||||
// a 3D view is refused outright by the layer-span guard in GetOrCreateAttachmentViewAtMipLevel
|
||||
// (3D images have arrayLayers == 1) and a CUBE_ARRAY view is built happily and then rejected -
|
||||
// or dereferenced - by the driver inside vkCreateFramebuffer.
|
||||
//
|
||||
// A 2D_ARRAY view is the legal spelling of all three: over a 2D-array-compatible 3D image its
|
||||
// "layers" are the mip's z slices (VUID-VkImageViewCreateInfo-image-04970), and over a
|
||||
// CUBE_COMPATIBLE 2D image - which is what both cube targets are - its layers are the faces.
|
||||
//
|
||||
// Knowingly NOT remapped: VK_IMAGE_VIEW_TYPE_1D / _1D_ARRAY, which 04113 also forbids. There is
|
||||
// no legal alternative for them (a VK_IMAGE_TYPE_1D image admits no 2D-family view at all), so
|
||||
// the only honest answer would be to decline the attachment - and every driver this has run on,
|
||||
// lavapipe included, accepts them. Declining would turn working GL_TEXTURE_1D[_ARRAY] render
|
||||
// targets into skipped draws to satisfy a VU nothing enforces. Left as-is, deliberately.
|
||||
static VkImageViewType ResolveAttachmentViewType(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment,
|
||||
const VkTextureManager::TextureResource& resource) {
|
||||
if (attachment.IsLayered()) {
|
||||
return resource.viewType;
|
||||
switch (resource.viewType) {
|
||||
case VK_IMAGE_VIEW_TYPE_3D:
|
||||
case VK_IMAGE_VIEW_TYPE_CUBE:
|
||||
case VK_IMAGE_VIEW_TYPE_CUBE_ARRAY:
|
||||
return VK_IMAGE_VIEW_TYPE_2D_ARRAY;
|
||||
default:
|
||||
return resource.viewType;
|
||||
}
|
||||
}
|
||||
// A non-layered attachment names ONE layer, so the view over it is a plain 2D view whatever
|
||||
// the image's own view type is. The cube-face upload targets always meant this; a cube map
|
||||
@@ -112,8 +127,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// a single layer is not a legal attachment. The CUBE arm is inert today - no frontend path
|
||||
// produces a non-layered cube attachment without a face upload target - and is kept for
|
||||
// symmetry with CUBE_ARRAY.
|
||||
//
|
||||
// 3D belongs in the same list and was missing from it, which is why the "per-slice
|
||||
// attachment view is a 2D view whose array layer is the slice" branch in
|
||||
// GetOrCreateAttachmentViewAtMipLevel was unreachable: glFramebufferTextureLayer on a
|
||||
// GL_TEXTURE_3D asked for a 3D view (illegal as an attachment) whose span was then checked
|
||||
// against arrayLayers == 1, so every slice above z = 0 came back VK_NULL_HANDLE.
|
||||
if (IsCubeMapFaceUploadTarget(attachment.GetTextureUploadTarget()) ||
|
||||
resource.viewType == VK_IMAGE_VIEW_TYPE_CUBE_ARRAY || resource.viewType == VK_IMAGE_VIEW_TYPE_CUBE) {
|
||||
resource.viewType == VK_IMAGE_VIEW_TYPE_CUBE_ARRAY || resource.viewType == VK_IMAGE_VIEW_TYPE_CUBE ||
|
||||
resource.viewType == VK_IMAGE_VIEW_TYPE_3D) {
|
||||
return VK_IMAGE_VIEW_TYPE_2D;
|
||||
}
|
||||
return resource.viewType;
|
||||
@@ -334,47 +356,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
const auto internalFormat = renderbuffer->GetInternalFormat();
|
||||
// Three-channel color formats widen to their RGBA twin exactly like textures do
|
||||
// (VkTextureManager::ResolveTextureFormatInfo): blits/resolves between a
|
||||
// renderbuffer and a texture of the same GL format then see one VkFormat.
|
||||
const VkFormat format = [&]() -> VkFormat {
|
||||
switch (internalFormat) {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
case TextureInternalFormat::R3G3B2:
|
||||
case TextureInternalFormat::RGB4:
|
||||
case TextureInternalFormat::RGB5:
|
||||
return VK_FORMAT_R8G8B8A8_UNORM;
|
||||
case TextureInternalFormat::SRGB8:
|
||||
return VK_FORMAT_R8G8B8A8_SRGB;
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
return VK_FORMAT_R8G8B8A8_SNORM;
|
||||
case TextureInternalFormat::RGB10:
|
||||
case TextureInternalFormat::RGB12:
|
||||
case TextureInternalFormat::RGB16:
|
||||
return VK_FORMAT_R16G16B16A16_UNORM;
|
||||
case TextureInternalFormat::RGB16Snorm:
|
||||
return VK_FORMAT_R16G16B16A16_SNORM;
|
||||
case TextureInternalFormat::RGB16F:
|
||||
return VK_FORMAT_R16G16B16A16_SFLOAT;
|
||||
case TextureInternalFormat::RGB32F:
|
||||
return VK_FORMAT_R32G32B32A32_SFLOAT;
|
||||
case TextureInternalFormat::RGB8I:
|
||||
return VK_FORMAT_R8G8B8A8_SINT;
|
||||
case TextureInternalFormat::RGB8UI:
|
||||
return VK_FORMAT_R8G8B8A8_UINT;
|
||||
case TextureInternalFormat::RGB16I:
|
||||
return VK_FORMAT_R16G16B16A16_SINT;
|
||||
case TextureInternalFormat::RGB16UI:
|
||||
return VK_FORMAT_R16G16B16A16_UINT;
|
||||
case TextureInternalFormat::RGB32I:
|
||||
return VK_FORMAT_R32G32B32A32_SINT;
|
||||
case TextureInternalFormat::RGB32UI:
|
||||
return VK_FORMAT_R32G32B32A32_UINT;
|
||||
default:
|
||||
return MG_Util::ConvertTextureInternalFormatToVkEnum(internalFormat);
|
||||
}
|
||||
}();
|
||||
// ONE resolver, shared with textures (VkTextureManager::ResolveTextureFormatInfo), so a
|
||||
// renderbuffer and a texture of the same GL format cannot disagree about their VkFormat.
|
||||
// `expandRgbToRgba` / `componentByteCount` / `alphaBytes` describe how to reshape a SHADOW
|
||||
// UPLOAD, and a renderbuffer has none, so only `.format` is taken.
|
||||
//
|
||||
// This used to be a hand-maintained second copy of that table, and it was missing exactly
|
||||
// four rows: RGBA2 and RGBA12 fell through to ConvertTextureInternalFormatToVkEnum's
|
||||
// VK_FORMAT_UNDEFINED (no image at all - bound as a draw buffer the attachment became
|
||||
// VK_ATTACHMENT_UNUSED and every draw into it was dropped), while RGBA4 and RGB5A1 fell
|
||||
// through to the 16-bit packed formats and then faced 32-bit R8G8B8A8_UNORM textures across
|
||||
// a size-incompatible vkCmdCopyImage.
|
||||
const VkFormat format = ResolveTextureFormatInfo(internalFormat).format;
|
||||
const VkImageAspectFlags aspect = ResolveImageAspectMaskForFormat(format);
|
||||
// Renderbuffers are never sampled (GL has no way to bind one to a sampler), so the
|
||||
// usage set is attachment + transfer: transfer covers readback (vkCmdCopyImageToBuffer),
|
||||
@@ -618,7 +611,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// sRGB attachments switch between their sRGB and UNORM-twin views with this
|
||||
// capability (ResolveSrgbAttachmentWriteFormat), changing the render pass formats.
|
||||
const Bool framebufferSrgbEnabled =
|
||||
MG_State::pGLContext->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb);
|
||||
MGB_CTX->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &framebufferSrgbEnabled, sizeof(framebufferSrgbEnabled)));
|
||||
auto& drawBuffers = fbo.GetDrawBuffers();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, drawBuffers.data(), drawBuffers.size() * sizeof(drawBuffers[0])));
|
||||
@@ -772,7 +765,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
RenderPassEntry& VkRenderPassManager::GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
RenderPassEntry* VkRenderPassManager::GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil) {
|
||||
// Resolve the default-FBO depth flavor (see the header comment): keep the
|
||||
@@ -858,7 +851,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto activeIt = m_renderPasses.find(activeRenderPass->hash);
|
||||
if (activeIt != m_renderPasses.end()) {
|
||||
activeIt->second.lastUsedFrame = m_frameCounter;
|
||||
return activeIt->second;
|
||||
return &activeIt->second;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -882,13 +875,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastRenderPassHash = activeRenderPass->hash;
|
||||
m_rpFastHadDepthStencil = activeIt->second.hasDepthStencilAttachment;
|
||||
activeIt->second.lastUsedFrame = m_frameCounter;
|
||||
return activeIt->second;
|
||||
return &activeIt->second;
|
||||
}
|
||||
auto hash = ComputeHash(fbo, swapchainImageIndex, true, includeDefaultFboDepthStencil);
|
||||
auto it = m_renderPasses.find(hash);
|
||||
if (it != m_renderPasses.end()) {
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
return it->second;
|
||||
return &it->second;
|
||||
}
|
||||
|
||||
Bool isDefaultFbo = fbo.IsDefaultFramebuffer();
|
||||
@@ -970,7 +963,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VkImageLayout trackedRbLayout = rbResource->layout;
|
||||
const Bool rbFramebufferSrgb =
|
||||
MG_State::pGLContext->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb);
|
||||
MGB_CTX->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb);
|
||||
const VkFormat rbAttachmentFormat =
|
||||
ResolveSrgbAttachmentWriteFormat(rbResource->format, rbFramebufferSrgb);
|
||||
rbDesc.flags = 0;
|
||||
@@ -1011,8 +1004,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
textureResources.emplace_back(nullptr);
|
||||
attachmentViews.emplace_back(rbAttachmentFormat != rbResource->format ? rbResource->unormTwinView
|
||||
: rbResource->view);
|
||||
MOBILEGL_ASSERT(attachmentViews.back() != VK_NULL_HANDLE,
|
||||
"GetOrCreateRenderPass: renderbuffer view missing at color attachment %d", i);
|
||||
if (attachmentViews.back() == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: renderbuffer %u has no usable view for color attachment "
|
||||
"%u on FBO %u; declining the render pass",
|
||||
renderbuffer->GetExternalIndex(), i, fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
colorAttachmentRefs[i].attachment = rbAttachmentIndex;
|
||||
continue;
|
||||
@@ -1100,12 +1097,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
attachmentViews.emplace_back(swapchainViews[swapchainImageIndex]);
|
||||
} else {
|
||||
auto* textureResource = m_textureManager.SyncTextureAndGetDescriptor(*texture);
|
||||
MOBILEGL_ASSERT(textureResource,
|
||||
"GetOrCreateRenderPass: SyncTextureAndGetDescriptor failed at color attachment %d", i);
|
||||
if (textureResource == nullptr) {
|
||||
// SyncTextureResource legitimately declines - an unsupported format,
|
||||
// sample count or image-flag combination, or a vkCreateImage the driver
|
||||
// refused. There is no image to attach, so there is no render pass.
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: textureId=%d could not be backed for color "
|
||||
"attachment %u on FBO %u; declining the render pass",
|
||||
texture->GetExternalIndex(), i, fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
textureResources.emplace_back(textureResource);
|
||||
desc.format = ResolveSrgbAttachmentWriteFormat(
|
||||
textureResource->format,
|
||||
MG_State::pGLContext->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb));
|
||||
MGB_CTX->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb));
|
||||
attachmentSampleCount = textureResource->sampleCount;
|
||||
trackedColorLayout = textureResource->layout;
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
@@ -1122,8 +1126,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
attachmentViews.emplace_back(
|
||||
m_textureManager.GetOrCreateAttachmentViewAtMipLevel(
|
||||
*texture, attachmentMipLevel, baseArrayLayer, layerCount, attachmentViewType));
|
||||
MOBILEGL_ASSERT(attachmentViews.back() != VK_NULL_HANDLE,
|
||||
"GetOrCreateRenderPass: GetOrCreateAttachmentView failed at color attachment %d", i);
|
||||
if (attachmentViews.back() == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: no attachment view for textureId=%d mip=%u layers "
|
||||
"[%u, %u) viewType=%d at color attachment %u on FBO %u; declining the "
|
||||
"render pass",
|
||||
texture->GetExternalIndex(), attachmentMipLevel, baseArrayLayer,
|
||||
baseArrayLayer + layerCount, static_cast<Int>(attachmentViewType), i,
|
||||
fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
desc.samples = attachmentSampleCount;
|
||||
adoptRenderPassSampleCount(attachmentSampleCount, "color", texture->GetExternalIndex());
|
||||
@@ -1216,8 +1227,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
} else if (selectedDepthStencilAttachment->IsTexture()) {
|
||||
auto& texture = *selectedDepthStencilAttachment->GetTexture();
|
||||
depthTextureResource = m_textureManager.SyncTextureAndGetDescriptor(texture);
|
||||
MOBILEGL_ASSERT(depthTextureResource,
|
||||
"GetOrCreateRenderPass: SyncTextureAndGetDescriptor failed at depth attachment");
|
||||
if (depthTextureResource == nullptr) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: textureId=%d could not be backed for the depth/stencil "
|
||||
"attachment of FBO %u; declining the render pass",
|
||||
texture.GetExternalIndex(), fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
trackedDepthLayout = depthTextureResource->layout;
|
||||
depthAttachmentDescription.format = depthTextureResource->format;
|
||||
depthAttachmentSampleCount = depthTextureResource->sampleCount;
|
||||
@@ -1229,8 +1244,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
} else {
|
||||
const auto& renderbuffer = selectedDepthStencilAttachment->GetRenderbuffer();
|
||||
depthRenderbufferResource = GetOrCreateRenderbufferResource(renderbuffer);
|
||||
MOBILEGL_ASSERT(depthRenderbufferResource,
|
||||
"GetOrCreateRenderPass: GetOrCreateRenderbufferResource failed at depth attachment");
|
||||
if (depthRenderbufferResource == nullptr) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: renderbuffer %u could not be backed for the depth/stencil "
|
||||
"attachment of FBO %u; declining the render pass",
|
||||
renderbuffer->GetExternalIndex(), fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
trackedDepthLayout = depthRenderbufferResource->layout;
|
||||
depthAttachmentDescription.format = depthRenderbufferResource->format;
|
||||
depthAttachmentSampleCount = depthRenderbufferResource->sampleCount;
|
||||
@@ -1304,8 +1323,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
attachmentViews.emplace_back(
|
||||
m_textureManager.GetOrCreateAttachmentViewAtMipLevel(
|
||||
texture, attachmentMipLevel, baseArrayLayer, layerCount, attachmentViewType));
|
||||
MOBILEGL_ASSERT(attachmentViews.back() != VK_NULL_HANDLE,
|
||||
"GetOrCreateRenderPass: GetOrCreateAttachmentView failed at depth attachment");
|
||||
if (attachmentViews.back() == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: no attachment view for textureId=%d mip=%u layers [%u, %u) "
|
||||
"viewType=%d at the depth/stencil attachment of FBO %u; declining the render pass",
|
||||
texture.GetExternalIndex(), attachmentMipLevel, baseArrayLayer,
|
||||
baseArrayLayer + layerCount, static_cast<Int>(attachmentViewType),
|
||||
fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
if (width == 0 || height == 0) {
|
||||
width = attachmentExtent.x();
|
||||
height = attachmentExtent.y();
|
||||
@@ -1327,6 +1352,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
});
|
||||
textureResources.emplace_back(nullptr);
|
||||
attachmentViews.emplace_back(depthRenderbufferResource->view);
|
||||
if (attachmentViews.back() == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: renderbuffer %u has no usable view for the depth/stencil "
|
||||
"attachment of FBO %u; declining the render pass",
|
||||
renderbuffer->GetExternalIndex(), fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
if (width == 0 || height == 0) {
|
||||
width = attachmentExtent.x();
|
||||
height = attachmentExtent.y();
|
||||
@@ -1424,8 +1455,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
renderPassCreateInfo.dependencyCount = 2;
|
||||
renderPassCreateInfo.pDependencies = subpassDependencies;
|
||||
|
||||
// NOT VK_VERIFY. VkIncludes.h states the rule this function now lives by: VK_VERIFY is the
|
||||
// INVARIANT check - a should-never-happen state, fatal-logged unlatched and trapped in a
|
||||
// DEBUG build - and "a soft, recoverable failure must therefore NOT be routed through
|
||||
// VK_VERIFY. Check the VkResult directly and report it with MGLOG_E_ONCE". A decline here
|
||||
// is recoverable by construction: the caller drops the draw. Routing it through VK_VERIFY
|
||||
// would have made the recovery dead code in a DEBUG build (the TRAP fires inside the macro,
|
||||
// before the handle is ever examined) and, in an INFO build, printed an UNLATCHED fatal
|
||||
// line on every draw for the life of the process - a decline caches nothing, so every
|
||||
// later draw to the same framebuffer re-enters this path and fails again.
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateRenderPass(m_device, &renderPassCreateInfo, nullptr, &renderPass));
|
||||
const VkResult renderPassResult =
|
||||
vkCreateRenderPass(m_device, &renderPassCreateInfo, nullptr, &renderPass);
|
||||
if (renderPassResult != VK_SUCCESS || renderPass == VK_NULL_HANDLE) {
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: vkCreateRenderPass failed (%s, %d) for FBO %u; declining the "
|
||||
"render pass",
|
||||
VkResultToString(renderPassResult), static_cast<Int>(renderPassResult),
|
||||
fbo.GetExternalIndex());
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// Framebuffer
|
||||
VkFramebufferCreateInfo framebufferCreateInfo;
|
||||
@@ -1438,8 +1486,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
framebufferCreateInfo.width = width;
|
||||
framebufferCreateInfo.height = height;
|
||||
framebufferCreateInfo.layers = framebufferLayers;
|
||||
// Direct VkResult check, for the same reason as vkCreateRenderPass above.
|
||||
VkFramebuffer framebuffer = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateFramebuffer(m_device, &framebufferCreateInfo, nullptr, &framebuffer));
|
||||
const VkResult framebufferResult =
|
||||
vkCreateFramebuffer(m_device, &framebufferCreateInfo, nullptr, &framebuffer);
|
||||
if (framebufferResult != VK_SUCCESS || framebuffer == VK_NULL_HANDLE) {
|
||||
// The render pass has no entry to own it yet, so it is destroyed here rather than
|
||||
// leaked - RenderPassEntry's destructor is the only other thing that would.
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: vkCreateFramebuffer failed (%s, %d) for FBO %u (%dx%d, "
|
||||
"%u attachments, %u layers); declining the render pass",
|
||||
VkResultToString(framebufferResult), static_cast<Int>(framebufferResult),
|
||||
fbo.GetExternalIndex(), width, height,
|
||||
static_cast<Uint32>(attachmentViews.size()), framebufferLayers);
|
||||
vkDestroyRenderPass(m_device, renderPass, nullptr);
|
||||
return nullptr;
|
||||
}
|
||||
IntVec2 extent = {width, height};
|
||||
RenderPassEntry renderPassEntry {
|
||||
hash,
|
||||
@@ -1464,7 +1525,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
extent.y());
|
||||
auto [insertedIt, _] = m_renderPasses.emplace(hash, Move(renderPassEntry));
|
||||
insertedIt->second.lastUsedFrame = m_frameCounter;
|
||||
return insertedIt->second;
|
||||
return &insertedIt->second;
|
||||
}
|
||||
|
||||
void VkRenderPassManager::OnPresent() {
|
||||
|
||||
@@ -243,9 +243,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// draw against a depth-less active pass resolves to a new (incompatible)
|
||||
// entry, which the caller's compatibility check turns into a pass split;
|
||||
// the new pass's depth loads DONT_CARE (content was undefined all along).
|
||||
RenderPassEntry& GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil = true);
|
||||
//
|
||||
// Returns NULLPTR when this framebuffer cannot be represented as a Vulkan render pass at
|
||||
// all - a texture the texture manager declined to back (an unsupported format or sample
|
||||
// count), or an attachment view it cannot construct (a layer span the image has no room
|
||||
// for, a 3D image whose format was refused 2D-array compatibility). This used to be
|
||||
// unrepresentable: the function returned a reference, so the only thing the two fallible
|
||||
// calls it builds on could do was trip a MOBILEGL_ASSERT - which is compiled out of every
|
||||
// INFO build - and then dereference the null resource, or hand VK_NULL_HANDLE to
|
||||
// vkCreateFramebuffer. That took the whole process down (51 lost CTS records over 21
|
||||
// bodies, one runner restart each) where a declined draw is merely a wrong picture.
|
||||
//
|
||||
// EVERY caller must handle nullptr by dropping the operation, exactly as the draw path
|
||||
// already drops a draw whose sampler descriptor could not be resolved
|
||||
// (UniformManager::BindProgramUniformBuffers). The failure paths log MGLOG_E_ONCE
|
||||
// themselves, so a caller needs no message of its own.
|
||||
[[nodiscard]] RenderPassEntry* GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil = true);
|
||||
void QueueRenderbufferClear(GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo);
|
||||
void QueueRenderbufferClear(const ClearAttachmentPayload& clearPayload,
|
||||
|
||||
@@ -21,6 +21,156 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
sampler.GetWrapR() == SamplerWrapMode::ClampToBorder;
|
||||
}
|
||||
|
||||
// The numeric domain the texture is SAMPLED in. Vulkan splits VkBorderColor into a float
|
||||
// family and an integer family and requires the sampler's choice to match the image view's
|
||||
// format (a float border on an integer view, or the reverse, is undefined) - so the domain
|
||||
// comes from the TEXTURE, while the value comes from whichever GL entry point wrote it.
|
||||
enum class BorderColorDomain {
|
||||
Float,
|
||||
SignedInteger,
|
||||
UnsignedInteger
|
||||
};
|
||||
|
||||
BorderColorDomain ResolveBorderColorDomain(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::R8I:
|
||||
case TextureInternalFormat::R16I:
|
||||
case TextureInternalFormat::R32I:
|
||||
case TextureInternalFormat::RG8I:
|
||||
case TextureInternalFormat::RG16I:
|
||||
case TextureInternalFormat::RG32I:
|
||||
case TextureInternalFormat::RGB8I:
|
||||
case TextureInternalFormat::RGB16I:
|
||||
case TextureInternalFormat::RGB32I:
|
||||
case TextureInternalFormat::RGBA8I:
|
||||
case TextureInternalFormat::RGBA16I:
|
||||
case TextureInternalFormat::RGBA32I:
|
||||
return BorderColorDomain::SignedInteger;
|
||||
case TextureInternalFormat::R8UI:
|
||||
case TextureInternalFormat::R16UI:
|
||||
case TextureInternalFormat::R32UI:
|
||||
case TextureInternalFormat::RG8UI:
|
||||
case TextureInternalFormat::RG16UI:
|
||||
case TextureInternalFormat::RG32UI:
|
||||
case TextureInternalFormat::RGB8UI:
|
||||
case TextureInternalFormat::RGB16UI:
|
||||
case TextureInternalFormat::RGB32UI:
|
||||
case TextureInternalFormat::RGBA8UI:
|
||||
case TextureInternalFormat::RGBA16UI:
|
||||
case TextureInternalFormat::RGBA32UI:
|
||||
case TextureInternalFormat::RGB10A2UI:
|
||||
return BorderColorDomain::UnsignedInteger;
|
||||
default:
|
||||
return BorderColorDomain::Float;
|
||||
}
|
||||
}
|
||||
|
||||
Bool IsSignedNormalizedFormat(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::R8Snorm:
|
||||
case TextureInternalFormat::R16Snorm:
|
||||
case TextureInternalFormat::RG8Snorm:
|
||||
case TextureInternalFormat::RG16Snorm:
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
case TextureInternalFormat::RGB16Snorm:
|
||||
case TextureInternalFormat::RGBA8Snorm:
|
||||
case TextureInternalFormat::RGBA16Snorm:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// GL 4.6 core 8.14.2: "The border values are clamped before they are used, according to the
|
||||
// format in which texture components are stored. For signed and unsigned normalized
|
||||
// fixed-point formats, border values are clamped to [-1,1] and [0,1] respectively. For
|
||||
// floating-point and integer formats, border values are clamped to the representable range of
|
||||
// the format." Every clause of that sentence is a real case here - the clamp is not just the
|
||||
// normalized one.
|
||||
//
|
||||
// Only the 32-bit float formats are genuinely unclamped: every finite float is representable
|
||||
// in them. Half-float has a finite maximum, and the two packed "float" formats are UNSIGNED,
|
||||
// so a negative border on them must come back as 0 rather than as a negative number the
|
||||
// driver delivers verbatim through VK_BORDER_COLOR_FLOAT_CUSTOM_EXT.
|
||||
struct FloatBorderRange {
|
||||
Bool clamped = true;
|
||||
Float minValue = 0.0f;
|
||||
Float maxValue = 1.0f;
|
||||
};
|
||||
|
||||
FloatBorderRange ResolveFloatBorderRange(TextureInternalFormat format, Bool isSignedNormalized) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::R32F:
|
||||
case TextureInternalFormat::RG32F:
|
||||
case TextureInternalFormat::RGB32F:
|
||||
case TextureInternalFormat::RGBA32F:
|
||||
return {false, 0.0f, 0.0f};
|
||||
case TextureInternalFormat::R16F:
|
||||
case TextureInternalFormat::RG16F:
|
||||
case TextureInternalFormat::RGB16F:
|
||||
case TextureInternalFormat::RGBA16F:
|
||||
return {true, -65504.0f, 65504.0f};
|
||||
// Unsigned packed floats: no sign bit at all. 65024 is the largest 11-bit float; the
|
||||
// 10-bit blue channel tops out lower (64512) and RGB9E5 higher (65408), but the bound
|
||||
// that matters for correctness is the lower one, and a single conservative upper bound
|
||||
// costs nothing a real border colour will ever notice.
|
||||
case TextureInternalFormat::R11FG11FB10F:
|
||||
return {true, 0.0f, 64512.0f};
|
||||
case TextureInternalFormat::RGB9E5:
|
||||
return {true, 0.0f, 65408.0f};
|
||||
default:
|
||||
return {true, isSignedNormalized ? -1.0f : 0.0f, 1.0f};
|
||||
}
|
||||
}
|
||||
|
||||
// Per-component representable range of an integer texture format, as Int64 so that the whole
|
||||
// signed and unsigned 32-bit ranges are expressible in one type and the clamp can be written
|
||||
// once for both domains. Alpha is carried separately because RGB10_A2UI is the one format
|
||||
// whose alpha is narrower than its colour channels.
|
||||
struct IntegerBorderRange {
|
||||
Int64 rgbMin = 0;
|
||||
Int64 rgbMax = 0;
|
||||
Int64 alphaMin = 0;
|
||||
Int64 alphaMax = 0;
|
||||
};
|
||||
|
||||
IntegerBorderRange ResolveIntegerBorderRange(TextureInternalFormat format) {
|
||||
const auto uniform = [](Int64 low, Int64 high) { return IntegerBorderRange{low, high, low, high}; };
|
||||
switch (format) {
|
||||
case TextureInternalFormat::R8I:
|
||||
case TextureInternalFormat::RG8I:
|
||||
case TextureInternalFormat::RGB8I:
|
||||
case TextureInternalFormat::RGBA8I:
|
||||
return uniform(-128, 127);
|
||||
case TextureInternalFormat::R16I:
|
||||
case TextureInternalFormat::RG16I:
|
||||
case TextureInternalFormat::RGB16I:
|
||||
case TextureInternalFormat::RGBA16I:
|
||||
return uniform(-32768, 32767);
|
||||
case TextureInternalFormat::R8UI:
|
||||
case TextureInternalFormat::RG8UI:
|
||||
case TextureInternalFormat::RGB8UI:
|
||||
case TextureInternalFormat::RGBA8UI:
|
||||
return uniform(0, 255);
|
||||
case TextureInternalFormat::R16UI:
|
||||
case TextureInternalFormat::RG16UI:
|
||||
case TextureInternalFormat::RGB16UI:
|
||||
case TextureInternalFormat::RGBA16UI:
|
||||
return uniform(0, 65535);
|
||||
case TextureInternalFormat::R32UI:
|
||||
case TextureInternalFormat::RG32UI:
|
||||
case TextureInternalFormat::RGB32UI:
|
||||
case TextureInternalFormat::RGBA32UI:
|
||||
return uniform(0, 4294967295LL);
|
||||
case TextureInternalFormat::RGB10A2UI:
|
||||
return {0, 1023, 0, 3};
|
||||
default:
|
||||
// The signed 32-bit formats, and anything unexpected: the full int32 range, i.e. a
|
||||
// clamp that cannot alter a value the GL entry points could have carried.
|
||||
return uniform(-2147483648LL, 2147483647LL);
|
||||
}
|
||||
}
|
||||
|
||||
Bool IsDepthTextureFormat(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::DepthComponent:
|
||||
@@ -72,6 +222,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_config = initInfo.config;
|
||||
m_samplerAnisotropySupported = initInfo.samplerAnisotropySupported;
|
||||
m_maxSamplerAnisotropy = std::max(initInfo.maxSamplerAnisotropy, 1.0f);
|
||||
m_customBorderColorSupported = initInfo.customBorderColorSupported;
|
||||
m_maxCustomBorderColorSamplers = initInfo.maxCustomBorderColorSamplers;
|
||||
m_customBorderColorSamplerCount = 0;
|
||||
MOBILEGL_ASSERT(m_device != VK_NULL_HANDLE && m_config != nullptr,
|
||||
"VkSamplerManager::Initialize failed: invalid initialization info");
|
||||
return true;
|
||||
@@ -102,6 +255,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_config = nullptr;
|
||||
m_frameBoundaryCounter = 0;
|
||||
m_customBorderColorSupported = false;
|
||||
m_maxCustomBorderColorSamplers = 0;
|
||||
m_customBorderColorSamplerCount = 0;
|
||||
}
|
||||
|
||||
void VkSamplerManager::OnFrameBoundary() {
|
||||
@@ -123,6 +279,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (m_device != VK_NULL_HANDLE && entry.handle != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, entry.handle, nullptr);
|
||||
}
|
||||
if (entry.usesCustomBorderColor && m_customBorderColorSamplerCount > 0) {
|
||||
--m_customBorderColorSamplerCount;
|
||||
}
|
||||
it = m_samplers.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
@@ -131,8 +290,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
Uint64 VkSamplerManager::BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const {
|
||||
Bool forceNearestFiltering, Bool singleLevelView,
|
||||
const ResolvedBorderColor& borderColor) const {
|
||||
MOBILEGL_ASSERT(m_config != nullptr, "VkSamplerManager::BuildSamplerKey: m_config is null");
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, m_config->CacheVersion));
|
||||
|
||||
@@ -166,8 +325,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &compareMode, sizeof(compareMode)));
|
||||
const auto compareFunc = sampler.GetSamplerCompareFunc();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &compareFunc, sizeof(compareFunc)));
|
||||
const auto borderColor = ResolveVkBorderColor(sampler, texture);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &borderColor, sizeof(borderColor)));
|
||||
// The resolved enum AND, when it is one of the *_CUSTOM_EXT values, the sixteen bytes of the
|
||||
// colour itself: two samplers that differ only in a custom border colour carry the same enum
|
||||
// and would otherwise collide onto whichever one was created first.
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &borderColor.color, sizeof(borderColor.color)));
|
||||
if (borderColor.isCustom) {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &borderColor.customValue, sizeof(borderColor.customValue)));
|
||||
}
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
@@ -183,7 +347,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// allocation for a genuinely single-level image) and faults the GPU - the same failure
|
||||
// the default-framebuffer blit shader had to work around with an explicit-LOD sample.
|
||||
const Bool singleLevelView = viewLevelCount == 1;
|
||||
const Uint64 key = BuildSamplerKey(sampler, texture, forceNearestFiltering, singleLevelView);
|
||||
// Resolved once and used for both the key and the create-info; see ResolvedBorderColor.
|
||||
const ResolvedBorderColor borderColor = ResolveBorderColor(sampler, texture);
|
||||
const Uint64 key = BuildSamplerKey(sampler, forceNearestFiltering, singleLevelView, borderColor);
|
||||
auto it = m_samplers.find(key);
|
||||
if (it != m_samplers.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
@@ -211,9 +377,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Must match BuildSamplerKey's resolution exactly.
|
||||
samplerInfo.maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
samplerInfo.minLod = ResolveEffectiveMinLod(sampler, samplerInfo.maxLod);
|
||||
samplerInfo.borderColor = ResolveVkBorderColor(sampler, texture);
|
||||
samplerInfo.borderColor = borderColor.color;
|
||||
samplerInfo.unnormalizedCoordinates = VK_FALSE;
|
||||
|
||||
// VK_EXT_custom_border_color. `format` stays UNDEFINED, which is legal only because
|
||||
// customBorderColorWithoutFormat was required alongside customBorderColors at device
|
||||
// creation - a GL sampler object has no idea which texture it will be paired with.
|
||||
VkSamplerCustomBorderColorCreateInfoEXT customBorderColorInfo{};
|
||||
if (borderColor.isCustom) {
|
||||
customBorderColorInfo.sType = VK_STRUCTURE_TYPE_SAMPLER_CUSTOM_BORDER_COLOR_CREATE_INFO_EXT;
|
||||
customBorderColorInfo.customBorderColor = borderColor.customValue;
|
||||
customBorderColorInfo.format = VK_FORMAT_UNDEFINED;
|
||||
customBorderColorInfo.pNext = samplerInfo.pNext;
|
||||
samplerInfo.pNext = &customBorderColorInfo;
|
||||
}
|
||||
|
||||
VkSampler vkSampler = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateSampler(m_device, &samplerInfo, nullptr, &vkSampler), "vkCreateSampler(texture)");
|
||||
|
||||
@@ -222,6 +400,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
entry.externalIndex = sampler.GetExternalIndex();
|
||||
entry.version = sampler.GetVersion();
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
entry.usesCustomBorderColor = borderColor.isCustom;
|
||||
if (entry.usesCustomBorderColor) {
|
||||
++m_customBorderColorSamplerCount;
|
||||
}
|
||||
m_samplers[key] = entry;
|
||||
return vkSampler;
|
||||
}
|
||||
@@ -281,39 +463,148 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
VkBorderColor VkSamplerManager::ResolveVkBorderColor(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture) {
|
||||
VkSamplerManager::ResolvedBorderColor VkSamplerManager::ResolveBorderColor(
|
||||
const MG_State::GLState::SamplerObject& sampler, const MG_State::GLState::ITextureObject& texture) const {
|
||||
ResolvedBorderColor resolved{};
|
||||
if (!UsesBorderColor(sampler)) {
|
||||
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
return resolved; // FLOAT_TRANSPARENT_BLACK, never sampled
|
||||
}
|
||||
|
||||
// Border colour is sampler state: a bound sampler object supplies its own, and a texture
|
||||
// with none reaches the very same value through the sampler object it owns.
|
||||
const auto& borderColor = sampler.GetBorderColor();
|
||||
const Bool isDepthTexture = IsDepthTextureFormat(texture.GetFormat());
|
||||
const auto format = texture.GetFormat();
|
||||
const auto domain = ResolveBorderColorDomain(format);
|
||||
const Bool canUseCustom = m_customBorderColorSupported && m_maxCustomBorderColorSamplers > 0 &&
|
||||
m_customBorderColorSamplerCount < m_maxCustomBorderColorSamplers;
|
||||
|
||||
if (isDepthTexture) {
|
||||
if (domain != BorderColorDomain::Float) {
|
||||
// An integer image view REQUIRES an integer border colour, whatever the value is - even
|
||||
// (0,0,0,1). The value itself is whichever integer form the application wrote; a float
|
||||
// border on an integer texture is nonsense GL leaves undefined, so the derived integer
|
||||
// representation (a plain cast) is as good an answer as any.
|
||||
//
|
||||
// Clamped to the format's representable range FIRST, per GL 4.6 core 8.14.2, and read
|
||||
// through Int64 so the whole signed and unsigned 32-bit ranges are expressible at once.
|
||||
//
|
||||
// Which representation to start from is the TEXTURE's domain, not the entry-point form
|
||||
// the application used. GL 4.6 core 8.10 stores an "I"-form border colour unmodified with
|
||||
// an integer internal data type and does not define a sign conversion between the two
|
||||
// integer forms, so the stored bits are reinterpreted in the sampled format's own
|
||||
// signedness. Measured, not assumed: a border of -1 written with glTexParameterIiv
|
||||
// against a GL_R8UI texture samples as 255 on the ES driver, i.e. as 0xFFFFFFFF clamped
|
||||
// to the format's maximum - see the IntegerBorderColorScenario case that pins it. Picking
|
||||
// the representation by the FORM instead would answer 0 here, which is a defensible
|
||||
// reading of the same spec text but puts DirectVulkan at odds with DirectGLES - and
|
||||
// DirectGLES cannot deviate, it forwards the value to the driver verbatim. Cross-backend
|
||||
// agreement decides it.
|
||||
const auto range = ResolveIntegerBorderRange(format);
|
||||
const auto& borderColorI = sampler.GetBorderColorI();
|
||||
const auto& borderColorUI = sampler.GetBorderColorUI();
|
||||
const Bool startFromUnsigned = domain == BorderColorDomain::UnsignedInteger;
|
||||
Int64 clamped[4];
|
||||
for (SizeT channel = 0; channel < 4; ++channel) {
|
||||
const Int64 raw = startFromUnsigned ? static_cast<Int64>(borderColorUI[channel])
|
||||
: static_cast<Int64>(borderColorI[channel]);
|
||||
const Int64 low = channel == 3 ? range.alphaMin : range.rgbMin;
|
||||
const Int64 high = channel == 3 ? range.alphaMax : range.rgbMax;
|
||||
clamped[channel] = std::clamp(raw, low, high);
|
||||
}
|
||||
|
||||
// Matched against the CLAMPED value, so a border the format cannot hold still lands on
|
||||
// the palette entry it clamps to rather than missing every one of them.
|
||||
const Bool allZeroRgb = clamped[0] == 0 && clamped[1] == 0 && clamped[2] == 0;
|
||||
if (allZeroRgb && clamped[3] == 0) {
|
||||
resolved.color = VK_BORDER_COLOR_INT_TRANSPARENT_BLACK;
|
||||
return resolved;
|
||||
}
|
||||
if (allZeroRgb && clamped[3] == 1) {
|
||||
resolved.color = VK_BORDER_COLOR_INT_OPAQUE_BLACK;
|
||||
return resolved;
|
||||
}
|
||||
if (clamped[0] == 1 && clamped[1] == 1 && clamped[2] == 1 && clamped[3] == 1) {
|
||||
resolved.color = VK_BORDER_COLOR_INT_OPAQUE_WHITE;
|
||||
return resolved;
|
||||
}
|
||||
if (canUseCustom) {
|
||||
resolved.color = VK_BORDER_COLOR_INT_CUSTOM_EXT;
|
||||
resolved.isCustom = true;
|
||||
for (SizeT channel = 0; channel < 4; ++channel) {
|
||||
if (domain == BorderColorDomain::UnsignedInteger) {
|
||||
resolved.customValue.uint32[channel] = static_cast<Uint32>(clamped[channel]);
|
||||
} else {
|
||||
resolved.customValue.int32[channel] = static_cast<Int32>(clamped[channel]);
|
||||
}
|
||||
}
|
||||
return resolved;
|
||||
}
|
||||
// No custom colour available: pick the nearest of the three integer palette entries
|
||||
// rather than always answering transparent black, which is what turned an integer border
|
||||
// of (-1,-1,-1,-1) into 0 and broke the CTS's clamped-texel detection outright.
|
||||
const Bool opaque = clamped[3] != 0;
|
||||
const Bool bright = clamped[0] != 0 || clamped[1] != 0 || clamped[2] != 0;
|
||||
resolved.color = !opaque ? VK_BORDER_COLOR_INT_TRANSPARENT_BLACK
|
||||
: (bright ? VK_BORDER_COLOR_INT_OPAQUE_WHITE : VK_BORDER_COLOR_INT_OPAQUE_BLACK);
|
||||
return resolved;
|
||||
}
|
||||
|
||||
// Float domain. GL 4.6 core 8.14.2/8.23: the border colour is interpreted in the texture's
|
||||
// format, so it is clamped to that format's representable range first. Without the clamp the
|
||||
// CTS's border of (255,255,255,255) on a GL_RGBA8 texture matched none of the palette entries
|
||||
// and fell through to transparent black - every border texel sampled 0 where the test wanted
|
||||
// 255. The range is per format class, not just the normalized [0,1] / [-1,1] pair: only the
|
||||
// 32-bit float formats are unclamped.
|
||||
FloatVec4 borderColor = sampler.GetBorderColor();
|
||||
if (const auto range = ResolveFloatBorderRange(format, IsSignedNormalizedFormat(format)); range.clamped) {
|
||||
borderColor = FloatVec4(std::clamp(borderColor.x(), range.minValue, range.maxValue),
|
||||
std::clamp(borderColor.y(), range.minValue, range.maxValue),
|
||||
std::clamp(borderColor.z(), range.minValue, range.maxValue),
|
||||
std::clamp(borderColor.w(), range.minValue, range.maxValue));
|
||||
}
|
||||
|
||||
// A depth texture samples one component, so only x decides - and its alpha reads as 1.
|
||||
if (IsDepthTextureFormat(format)) {
|
||||
if (NearlyEqual(borderColor.x(), 1.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
||||
resolved.color = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
||||
return resolved;
|
||||
}
|
||||
if (NearlyEqual(borderColor.x(), 0.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK;
|
||||
resolved.color = VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK;
|
||||
return resolved;
|
||||
}
|
||||
}
|
||||
|
||||
const Bool rgbZero = NearlyEqual(borderColor.x(), 0.0f) && NearlyEqual(borderColor.y(), 0.0f) &&
|
||||
NearlyEqual(borderColor.z(), 0.0f);
|
||||
if (rgbZero && NearlyEqual(borderColor.w(), 0.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
resolved.color = VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
return resolved;
|
||||
}
|
||||
if (rgbZero && NearlyEqual(borderColor.w(), 1.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK;
|
||||
resolved.color = VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK;
|
||||
return resolved;
|
||||
}
|
||||
if (NearlyEqual(borderColor.x(), 1.0f) && NearlyEqual(borderColor.y(), 1.0f) &&
|
||||
NearlyEqual(borderColor.z(), 1.0f) && NearlyEqual(borderColor.w(), 1.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
||||
resolved.color = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
||||
return resolved;
|
||||
}
|
||||
|
||||
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
if (canUseCustom) {
|
||||
resolved.color = VK_BORDER_COLOR_FLOAT_CUSTOM_EXT;
|
||||
resolved.isCustom = true;
|
||||
resolved.customValue.float32[0] = borderColor.x();
|
||||
resolved.customValue.float32[1] = borderColor.y();
|
||||
resolved.customValue.float32[2] = borderColor.z();
|
||||
resolved.customValue.float32[3] = borderColor.w();
|
||||
return resolved;
|
||||
}
|
||||
|
||||
// Nearest of the three float palette entries. Transparent black stays the answer for a
|
||||
// transparent border, which is what the old unconditional fallback got right by accident.
|
||||
const Bool opaque = borderColor.w() >= 0.5f;
|
||||
const Bool bright = (borderColor.x() + borderColor.y() + borderColor.z()) >= 1.5f;
|
||||
resolved.color = !opaque ? VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK
|
||||
: (bright ? VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE : VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK);
|
||||
return resolved;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -28,6 +28,13 @@ public:
|
||||
Bool samplerAnisotropySupported = false;
|
||||
// VkPhysicalDeviceLimits::maxSamplerAnisotropy.
|
||||
Float maxSamplerAnisotropy = 1.0f;
|
||||
// VK_EXT_custom_border_color was enabled with BOTH customBorderColors and
|
||||
// customBorderColorWithoutFormat; see VulkanRenderer::m_customBorderColorFeatureEnabled.
|
||||
Bool customBorderColorSupported = false;
|
||||
// VkPhysicalDeviceCustomBorderColorPropertiesEXT::maxCustomBorderColorSamplers. A hard device
|
||||
// limit on how many LIVE samplers may carry a custom border colour, so the cache counts them
|
||||
// and falls back to the snapped predefined value once it is reached.
|
||||
Uint32 maxCustomBorderColorSamplers = 0;
|
||||
};
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
@@ -52,6 +59,21 @@ public:
|
||||
// boundaries.
|
||||
void OnFrameBoundary();
|
||||
|
||||
// What GL_TEXTURE_BORDER_COLOR resolves to for one (sampler, texture) pair. `color` is always a
|
||||
// legal VkBorderColor; when `isCustom` it is one of the *_CUSTOM_EXT values and `customValue`
|
||||
// carries the actual components in a VkSamplerCustomBorderColorCreateInfoEXT.
|
||||
//
|
||||
// Resolved ONCE per GetOrCreateSampler call and threaded into both the cache key and the
|
||||
// create-info, so the two cannot disagree - the same discipline the resolved anisotropy needs,
|
||||
// and here it also makes the maxCustomBorderColorSamplers fallback deterministic: whether a
|
||||
// custom colour was affordable is decided before the key is built, not twice with a budget
|
||||
// change in between.
|
||||
struct ResolvedBorderColor {
|
||||
VkBorderColor color = VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
VkClearColorValue customValue{};
|
||||
Bool isCustom = false;
|
||||
};
|
||||
|
||||
private:
|
||||
struct SamplerCacheEntry {
|
||||
VkSampler handle = VK_NULL_HANDLE;
|
||||
@@ -60,17 +82,18 @@ private:
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age have their VkSampler destroyed.
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
// Counted against maxCustomBorderColorSamplers for as long as this entry lives.
|
||||
Bool usesCustomBorderColor = false;
|
||||
};
|
||||
|
||||
Uint64 BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const;
|
||||
Uint64 BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler, Bool forceNearestFiltering,
|
||||
Bool singleLevelView, const ResolvedBorderColor& borderColor) const;
|
||||
static VkFilter ToVkFilter(SamplerFilterMode mode);
|
||||
static VkSamplerMipmapMode ToVkMipmapMode(SamplerMipmapMode mode);
|
||||
static VkSamplerAddressMode ToVkAddressMode(SamplerWrapMode mode);
|
||||
static VkCompareOp ToVkCompareOp(SamplerCompareFunc func);
|
||||
static VkBorderColor ResolveVkBorderColor(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture);
|
||||
ResolvedBorderColor ResolveBorderColor(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture) const;
|
||||
// The anisotropy Vulkan will actually apply: 1.0 (i.e. disabled) unless the feature is on and
|
||||
// the sampler filters linearly both ways, otherwise the GL request clamped to the device limit.
|
||||
// GL happily carries GL_TEXTURE_MAX_ANISOTROPY on a NEAREST sampler (Blaze3D's blocks do exactly
|
||||
@@ -82,6 +105,12 @@ private:
|
||||
const VulkanRendererConfig* m_config = nullptr;
|
||||
Bool m_samplerAnisotropySupported = false;
|
||||
Float m_maxSamplerAnisotropy = 1.0f;
|
||||
Bool m_customBorderColorSupported = false;
|
||||
Uint32 m_maxCustomBorderColorSamplers = 0;
|
||||
// Live cache entries carrying a custom border colour. Kept in step with the entries themselves
|
||||
// in exactly the three places one can appear or disappear: creation, the OnFrameBoundary sweep,
|
||||
// and Shutdown.
|
||||
Uint32 m_customBorderColorSamplerCount = 0;
|
||||
UnorderedMap<Uint64, SamplerCacheEntry> m_samplers;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
|
||||
@@ -11,8 +11,10 @@
|
||||
#include "ProgramFactory.h"
|
||||
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include <MG_Pipe/PipeInputsSwitch.h>
|
||||
#include "MG_Util/Converters/MGToStr/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include "MG_Util/Metrics/PipeStats.h"
|
||||
|
||||
#include <Config.h>
|
||||
#include <algorithm>
|
||||
@@ -46,13 +48,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return mipLevelCount;
|
||||
}
|
||||
|
||||
struct TextureFormatInfo {
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
Bool expandRgbToRgba = false;
|
||||
Uint32 componentByteCount = 0;
|
||||
Array<Uint8, 4> alphaBytes = {0, 0, 0, 0};
|
||||
};
|
||||
|
||||
struct TextureShapeInfo {
|
||||
VkImageType imageType = VK_IMAGE_TYPE_2D;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
@@ -380,7 +375,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
static TextureFormatInfo ResolveTextureFormatInfo(TextureInternalFormat format) {
|
||||
TextureFormatInfo ResolveTextureFormatInfo(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
@@ -812,7 +807,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// sampled-texture sync scan the entire alive-texture map per draw.
|
||||
if (aliveIt == m_aliveObjects.end()) {
|
||||
WeakPtr<MG_State::GLState::ITextureObject> aliveTexture;
|
||||
const auto& liveTexture = MG_State::pGLContext->GetTextureObject(texture.GetExternalIndex());
|
||||
const auto& liveTexture = MGB_CTX->GetTextureObject(texture.GetExternalIndex());
|
||||
if (liveTexture && liveTexture.get() == &texture) {
|
||||
aliveTexture = liveTexture;
|
||||
} else {
|
||||
@@ -921,18 +916,28 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (mipLevel >= resource->mipLevels) {
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
// A 3D image has arrayLayers == 1 and keeps its GL layers on the z axis, so a per-slice
|
||||
// attachment view is a 2D view whose "array layer" is the slice - legal only on a
|
||||
// 2D-array-compatible image (VUID-VkImageViewCreateInfo-image-04970), which
|
||||
// SyncTextureResource asks for and may have had refused per format.
|
||||
if (resource->viewType == VK_IMAGE_VIEW_TYPE_3D && viewType == VK_IMAGE_VIEW_TYPE_2D) {
|
||||
// A 3D image has arrayLayers == 1 and keeps its GL layers on the z axis, so an attachment
|
||||
// view over it addresses SLICES through baseArrayLayer/layerCount: one slice for a
|
||||
// non-layered attachment (a 2D view) and the whole span for a layered one (a 2D_ARRAY view,
|
||||
// which is what a layered GL_TEXTURE_3D attachment plus a gl_Layer-writing geometry shader
|
||||
// means). BOTH spellings are legal only on a 2D-array-compatible image
|
||||
// (VUID-VkImageViewCreateInfo-image-04970 / -06723), which SyncTextureResource asks for and
|
||||
// may have had refused per format.
|
||||
//
|
||||
// The span is validated against the MIP's slice count, never against arrayLayers: a 3D
|
||||
// image's arrayLayers is 1 by construction, so measuring a layered span against it rejected
|
||||
// every layered 3D attachment - the null view that used to reach vkCreateFramebuffer.
|
||||
if (resource->viewType == VK_IMAGE_VIEW_TYPE_3D &&
|
||||
(viewType == VK_IMAGE_VIEW_TYPE_2D || viewType == VK_IMAGE_VIEW_TYPE_2D_ARRAY)) {
|
||||
const Uint32 sliceCount = std::max(resource->depth >> mipLevel, 1u);
|
||||
if ((resource->imageCreateFlags & VK_IMAGE_CREATE_2D_ARRAY_COMPATIBLE_BIT) == 0 ||
|
||||
layerCount == 0 || baseArrayLayer >= sliceCount || baseArrayLayer + layerCount > sliceCount) {
|
||||
MGLOG_D("%s: cannot name slice span [%u, %u) of 3D textureId=%d (mip %u has %u slices, "
|
||||
"2D-array-compatible=%d)",
|
||||
// Not an error line: the render-pass builder turns the null view into one
|
||||
// MGLOG_E_ONCE and a skipped draw, which is the level this belongs at.
|
||||
MGLOG_D("%s: cannot name slice span [%u, %u) of 3D textureId=%d as viewType=%d (mip %u has %u "
|
||||
"slices, 2D-array-compatible=%d)",
|
||||
__func__, baseArrayLayer, baseArrayLayer + layerCount, texture.GetExternalIndex(),
|
||||
mipLevel, sliceCount,
|
||||
static_cast<Int>(viewType), mipLevel, sliceCount,
|
||||
(int)((resource->imageCreateFlags & VK_IMAGE_CREATE_2D_ARRAY_COMPATIBLE_BIT) != 0));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
@@ -945,7 +950,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
const Bool framebufferSrgbEnabled =
|
||||
MG_State::pGLContext->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb);
|
||||
MGB_CTX->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb);
|
||||
const VkFormat baseAttachmentFormat =
|
||||
viewFormatOverride != VK_FORMAT_UNDEFINED ? viewFormatOverride : resource->format;
|
||||
const VkFormat attachmentFormat =
|
||||
@@ -2173,12 +2178,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
if (imageFormatResult != VK_SUCCESS && !isMultisampleTexture &&
|
||||
(imageInfo.flags & VK_IMAGE_CREATE_2D_ARRAY_COMPATIBLE_BIT) != 0) {
|
||||
// Losing 2D-array compatibility only costs per-slice framebuffer attachment for this
|
||||
// format; failing creation would lose the texture entirely. Remembered so later syncs
|
||||
// neither reprobe nor flag-mismatch against this image and recreate it.
|
||||
// Losing 2D-array compatibility only costs framebuffer attachment of this format's
|
||||
// 3D images - per-slice AND layered, since both are spelled as a 2D-family view over
|
||||
// the z axis; failing creation would lose the texture entirely. Recorded here (the
|
||||
// per-format set below) so later syncs neither reprobe nor flag-mismatch against this
|
||||
// image and recreate it, and so GetOrCreateAttachmentViewAtMipLevel declines rather
|
||||
// than handing back a view that cannot exist - the render-pass builder then turns
|
||||
// that decline into a skipped draw instead of a null VkImageView in pAttachments.
|
||||
MGLOG_W_ONCE("%s: VK_IMAGE_CREATE_2D_ARRAY_COMPATIBLE_BIT is unsupported for format=%d "
|
||||
"textureId=%d; creating without it (per-slice framebuffer attachment will be "
|
||||
"unavailable for it)",
|
||||
"textureId=%d; creating without it (per-slice and layered framebuffer "
|
||||
"attachment of 3D textures in this format will be unavailable)",
|
||||
__func__, static_cast<Int>(format), texture.GetExternalIndex());
|
||||
m_2dArrayCompatibleUnsupported.insert(format);
|
||||
imageInfo.flags &= ~VK_IMAGE_CREATE_2D_ARRAY_COMPATIBLE_BIT;
|
||||
@@ -3143,6 +3152,32 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
packBox(dst, item.regionLo, item.regionSize);
|
||||
}
|
||||
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
// Same shape split as Espryt's: one union box per item, or one job per rect of
|
||||
// a refined rect list. The box/rect decision is invisible to SSIM and is what
|
||||
// the +6 ms/frame Mali cliff of section 7.3 was, so it is counted apart from
|
||||
// the bytes.
|
||||
Uint64 boxEmissions = 0;
|
||||
Uint64 rectEmissions = 0;
|
||||
Uint64 jobs = 0;
|
||||
for (const auto& item : uploadItems) {
|
||||
if (item.rects.empty()) {
|
||||
++boxEmissions;
|
||||
jobs += isCombinedDepthStencil ? 2u : 1u;
|
||||
} else {
|
||||
++rectEmissions;
|
||||
jobs += static_cast<Uint64>(item.rects.size());
|
||||
}
|
||||
}
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::StageTexture,
|
||||
static_cast<Uint64>(stagingSize));
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::TextureUploadEmissions,
|
||||
static_cast<Uint64>(uploadItems.size()));
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::TextureUploadBoxEmissions, boxEmissions);
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::TextureUploadRectEmissions, rectEmissions);
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::TextureUploadJobs, jobs);
|
||||
}
|
||||
|
||||
const VkImageAspectFlags aspectMask = GetAspectMaskForFormat(outResource.format);
|
||||
VkPipelineStageFlags uploadSrcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
VkAccessFlags uploadSrcAccessMask = 0;
|
||||
|
||||
@@ -10,8 +10,10 @@
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/FramebufferState/FramebufferObject.h>
|
||||
#include <MG_State/GLState/TextureState/TextureObject.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
#include <algorithm>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
|
||||
@@ -22,6 +24,31 @@ class ITextureObject;
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class SamplerNumericDomain : Uint8;
|
||||
|
||||
// What VkFormat a GL internal format is BACKED with, and how a shadow upload has to be reshaped to
|
||||
// fit it. This is not the same question as "is there an exact VkFormat for this GL format", which is
|
||||
// what ConvertTextureInternalFormatToVkEnum answers: several GL formats have no Vulkan twin at all
|
||||
// (RGBA2, RGBA12) and several three-channel ones are deliberately widened to their four-channel twin
|
||||
// because Vulkan devices rarely support the 3-channel layouts.
|
||||
//
|
||||
// SHARED, and it must stay the only answer to that question. A renderbuffer and a texture of the
|
||||
// same GL format have to resolve to the SAME VkFormat or every blit, resolve and glCopyImageSubData
|
||||
// between them crosses a size-incompatible pair, which vkCmdCopyImage leaves undefined
|
||||
// (VUID-vkCmdCopyImage-srcImage-01548). The renderbuffer path used to carry a hand-maintained second
|
||||
// copy of this table that was missing four rows - RGBA2, RGBA4, RGB5A1 and RGBA12 - so those four
|
||||
// renderbuffer formats either got no image at all or a 16-bit-packed one facing a 32-bit texture.
|
||||
struct TextureFormatInfo {
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
// The GL format has three channels and is carried in a four-channel image; a shadow upload has
|
||||
// to be expanded, inserting `alphaBytes` after every `componentByteCount * 3` source bytes.
|
||||
Bool expandRgbToRgba = false;
|
||||
Uint32 componentByteCount = 0;
|
||||
Array<Uint8, 4> alphaBytes = {0, 0, 0, 0};
|
||||
};
|
||||
|
||||
// Callers that only need the backing VkFormat (a renderbuffer has no shadow upload to reshape) take
|
||||
// `.format` and ignore the rest.
|
||||
TextureFormatInfo ResolveTextureFormatInfo(TextureInternalFormat format);
|
||||
|
||||
// A GL 1D-ARRAY level keeps its LAYER COUNT in the state-side HEIGHT: that is what
|
||||
// glTexImage2D(GL_TEXTURE_1D_ARRAY, width, layers) means, and the frontend records the level
|
||||
// as {width, layers, 1} (see GL_Texture.cpp's AllocateStorage and the completeness walk in
|
||||
@@ -41,6 +68,37 @@ inline IntVec3 ToVulkanLevelExtent(TextureTarget stateTarget, const IntVec3& glT
|
||||
return glTexelSize;
|
||||
}
|
||||
|
||||
// How many Vulkan array layers (or, for a 3D image, z slices) a GL framebuffer attachment spans.
|
||||
//
|
||||
// THE ONE COPY, deliberately. This used to exist twice - privately in VkRenderPassManager.cpp and
|
||||
// again in VkClearManager.cpp - and the two are not independent: the render pass builds the
|
||||
// attachment view and VkFramebufferCreateInfo::layers from one, while the CLEAR key built from the
|
||||
// other is written verbatim into VkImageSubresourceRange::layerCount when a queued glClear is
|
||||
// materialised outside a render pass (MaterializePendingClearForTexture). They are two consumers
|
||||
// of the same GL clear, so any disagreement means the same glClear produces two different pictures
|
||||
// depending only on which path happens to consume it first - and the materialise path then POPS
|
||||
// the entry, so the other one never runs. Fixing one copy and leaving the other is exactly how
|
||||
// that split gets introduced; keep them the same function.
|
||||
//
|
||||
// Two shapes make this more than `size.z()`:
|
||||
// * GL_TEXTURE_1D_ARRAY keeps its layer count in the state-side HEIGHT (see ToVulkanLevelExtent
|
||||
// just above), so z reads 1 and every layer above the first was silently dropped.
|
||||
// * GL_TEXTURE_CUBE_MAP is attached layered as its REPRESENTATIVE upload target, the +X face
|
||||
// (ResolveRepresentableFramebufferTextureUploadTarget), and one face's level size has z = 1 -
|
||||
// but a layered cube attachment names all six faces (GL 4.6 core 9.2.8), which are the image's
|
||||
// six array layers. A cube ARRAY needs no such arm: its representative target carries 6n in z.
|
||||
inline Uint32 ResolveAttachmentLayerCount(const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (!attachment.IsLayered()) {
|
||||
return 1u;
|
||||
}
|
||||
const auto& texture = attachment.GetTexture();
|
||||
const TextureTarget target = texture != nullptr ? texture->GetTarget() : TextureTarget::Unknown;
|
||||
if (target == TextureTarget::TextureCubeMap) {
|
||||
return 6u;
|
||||
}
|
||||
return static_cast<Uint32>(std::max(ToVulkanLevelExtent(target, attachment.GetSize()).z(), 1));
|
||||
}
|
||||
|
||||
// A GL framebuffer attachment's level/layer, and a GL image unit's, are relative to the texture
|
||||
// the application NAMED. When that texture was created by glTextureView (ARB_texture_view) they
|
||||
// are relative to the VIEW, and have to be shifted into the storage image's numbering before they
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -9,6 +9,7 @@
|
||||
#pragma once
|
||||
#include "Config.h"
|
||||
#include "FrameContext.h"
|
||||
#include "MagmaPipeArms.h"
|
||||
#include "PipelineFactory.h"
|
||||
#include "ProgramFactory.h"
|
||||
#include "SwapchainObject.h"
|
||||
@@ -24,6 +25,13 @@
|
||||
#include "MG_Util/Math/VectorTypes.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
#include <MG_Pipe/MGPipeHandles.h>
|
||||
#include <MG_Util/SelfTest/PrimitivesGeneratedNoXfbProbe.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// The applier's CSO store: MGPipeApplier().BoundRenderStateCso is what the pipeline memo
|
||||
// keys on after P2 (D12.1). Push-only, so the pull build's include graph is unchanged.
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#endif
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
@@ -563,7 +571,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Native subgroup topology, queried at device creation for the compute-module
|
||||
// subgroup repairs (SubgroupSupportPolicy.h) and the REQUIRE_FULL_SUBGROUPS
|
||||
// stage flag; 0 / false when the device has no usable compute subgroups or
|
||||
// MOBILEGL_DISABLE_SUBGROUP forced them off.
|
||||
// MOBILEGL_MAGMA_DISABLE_SUBGROUP forced them off.
|
||||
Uint32 m_nativeSubgroupSize = 0;
|
||||
Bool m_nativeSubgroupSupported = false;
|
||||
Bool m_computeFullSubgroupsFeatureEnabled = false;
|
||||
@@ -584,6 +592,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// needs no feature). Both cached at device creation and drive a hard-fail-at-draw when absent.
|
||||
Bool m_dualSrcBlendFeatureEnabled = false;
|
||||
Bool m_primitiveTopologyListRestartFeatureEnabled = false;
|
||||
// shaderTessellationAndGeometryPointSize gates the PointSize built-in in a tessellation
|
||||
// or geometry stage, which desktop GL treats as an ordinary per-vertex output (writable,
|
||||
// and capturable by name through transform feedback). Cached at device creation and
|
||||
// handed to ProgramFactory, which refuses a program whose tessellation or geometry module
|
||||
// declares the matching SPIR-V capability while this is false - SetupDraw then skips its
|
||||
// draws (VkProgramObject::pointSizeCapabilityUnsupported) rather than building a pipeline
|
||||
// that is invalid usage.
|
||||
Bool m_tessellationAndGeometryPointSizeFeatureEnabled = false;
|
||||
// VK_EXT_custom_border_color. Vulkan's four predefined VkBorderColor values cover only
|
||||
// transparent/opaque black and opaque white; GL_TEXTURE_BORDER_COLOR is an arbitrary vec4 (or
|
||||
// an arbitrary ivec4/uvec4 through the "I" entry points). Without this extension a border
|
||||
// colour outside the palette has to be snapped to the nearest predefined one. Both features
|
||||
// are required together: customBorderColorWithoutFormat is what lets a sampler carry a custom
|
||||
// colour without naming the image format it will be paired with, which GL's sampler objects
|
||||
// cannot know. maxCustomBorderColorSamplers is a real device limit, so the sampler cache has
|
||||
// to be able to fall back to the snapped value once it is reached.
|
||||
Bool m_customBorderColorFeatureEnabled = false;
|
||||
Uint32 m_maxCustomBorderColorSamplers = 0;
|
||||
// sampleRateShading gates VkPipelineMultisampleStateCreateInfo::sampleShadingEnable, i.e.
|
||||
// glEnable(GL_SAMPLE_SHADING) + glMinSampleShading. Unlike dualSrcBlend this does NOT
|
||||
// hard-fail the draw when absent: sample shading is a rate hint, and every sample-rate
|
||||
// pipeline is still correct (just not per-sample) at the default rate - so the enable is
|
||||
// dropped and the draw proceeds, which is what a GL implementation with SAMPLES=1 does too.
|
||||
Bool m_sampleRateShadingFeatureEnabled = false;
|
||||
// multiViewport gates rasterizing into more than one of ARB_viewport_array's 16 viewports
|
||||
// (gl_ViewportIndex). m_maxRasterizableViewports is min(MAX_VIEWPORTS, device limit), or 1
|
||||
// when the feature is off, and is the viewportCount a gl_ViewportIndex-writing pipeline
|
||||
@@ -650,8 +682,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// per object: one group of four slots each, handed out on first use.
|
||||
static constexpr SizeT kXfbCounterObjectSlots = 16;
|
||||
VkBufferObject m_xfbCounterBuffer;
|
||||
UnorderedMap<Uint, Uint32> m_xfbCounterSlotByObject;
|
||||
Uint32 m_xfbNextCounterSlot = 0;
|
||||
// Which transform feedback object owns each slot group, by the frontend's never-reused
|
||||
// lifetime id (0 = the slot is free). This used to be an UnorderedMap keyed on the GL
|
||||
// NAME, which is recycled by glGenTransformFeedbacks: a deleted-and-recreated object
|
||||
// inherited the dead one's slot, and since nothing ever removed an entry the map also
|
||||
// grew for the life of the context. A fixed table cannot do either: a group is taken over
|
||||
// only from an owner with no OPEN span (see CurrentXfbCounterSlot), so an object whose
|
||||
// counters can still be resumed never loses them, and a dead object's group comes back.
|
||||
Array<Uint64, kXfbCounterObjectSlots> m_xfbCounterSlotOwner{};
|
||||
// Tie-break among reclaimable groups only; never on its own, because the paused span the
|
||||
// groups exist for is by construction the least recently used one.
|
||||
Array<Uint64, kXfbCounterObjectSlots> m_xfbCounterSlotLastUse{};
|
||||
Uint64 m_xfbCounterSlotUseSerial = 0;
|
||||
// Set for a slot once a captured draw has been recorded into its span; selects
|
||||
// counter-buffer resume on the next captured draw of the same span.
|
||||
Array<Bool, kXfbCounterObjectSlots> m_xfbCountersValid{};
|
||||
@@ -693,15 +735,74 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<Uint32> m_xfbQueryActiveSlots[2];
|
||||
Bool m_xfbQuerySlotOpen = false;
|
||||
Uint32 m_xfbQueryOpenSlot = 0;
|
||||
// GL_PRIMITIVES_GENERATED reroute for draws made while transform feedback is
|
||||
// INACTIVE. The stream pool's primitivesNeeded is defined to count those draws
|
||||
// too, but a Mali driver (and Mesa lavapipe) answers 0 unless a capture span
|
||||
// is open (the CTS's tessellator-measuring shape). Where the bring-up probe
|
||||
// finds that defect with a working control - or
|
||||
// MOBILEGL_MAGMA_PRIMGEN_QUERY_REROUTE forces it - such draws accumulate the
|
||||
// GENERATED count through this pool instead, whose type the arming picks:
|
||||
// VK_QUERY_TYPE_PRIMITIVES_GENERATED_EXT where the device hosts the dedicated
|
||||
// query with its rasterizer-discard feature (exact semantics by definition -
|
||||
// the extension exists because GL needs this count without a capture), else a
|
||||
// VK_QUERY_TYPE_PIPELINE_STATISTICS pool over clipping-stage invocations (one
|
||||
// per primitive reaching primitive clipping - after every vertex processing
|
||||
// stage, before rasterizer discard - which is the same set).
|
||||
// XFB-ACTIVE draws keep the stream slot (exact today, and WRITTEN needs it);
|
||||
// every draw with no open capture - a PAUSED span's draws included - takes a
|
||||
// reroute slot, and the span then ignores the frontend's CPU paused-primitive
|
||||
// counter rather than adding it on top (see IsPrimGenRerouteArmed): that
|
||||
// counter is written by only 3 of the ~15 draw entry points and answers 0 for
|
||||
// GL_PATCHES, so it cannot price the draws this reroute exists to repair. One
|
||||
// GL query span may therefore hold slots of both pools.
|
||||
Bool m_pipelineStatisticsQueryFeatureEnabled = false;
|
||||
// VK_EXT_primitives_generated_query: base feature, and the
|
||||
// ...WithRasterizerDiscard feature without which a discarding draw inside the
|
||||
// query is invalid usage (so the reroute never picks the dedicated pool on a
|
||||
// base-only device - GL applications toggle discard freely).
|
||||
Bool m_primitivesGeneratedQueryFeatureEnabled = false;
|
||||
Bool m_primitivesGeneratedQueryDiscardFeatureEnabled = false;
|
||||
// tessellationShader was enabled at device creation (it is taken whenever the
|
||||
// device advertises it); gates the probe's PATCHES shape.
|
||||
Bool m_tessellationShaderFeatureEnabled = false;
|
||||
MG_Util::SelfTest::PrimGenRerouteKind m_primGenRerouteKind =
|
||||
MG_Util::SelfTest::PrimGenRerouteKind::None;
|
||||
// The bring-up probe measured this device's stream query as counting draws made
|
||||
// with no capture span open (the StreamCounts verdict) - so it counts the
|
||||
// PAUSED-span ones too, through the stream slot they take when nothing is
|
||||
// rerouted. Only the probe can know this, so it stays false wherever the probe
|
||||
// is not consulted (the forced arms), which keeps those lanes' accounting as it
|
||||
// was.
|
||||
Bool m_primGenStreamCountsXfbInactiveDraws = false;
|
||||
VkQueryPool m_primGenReroutePool = VK_NULL_HANDLE;
|
||||
Uint32 m_primGenRerouteSlotCursor = 0;
|
||||
Vector<Uint32> m_primGenRerouteActiveSlots;
|
||||
Bool m_primGenRerouteSlotOpen = false;
|
||||
Uint32 m_primGenRerouteOpenSlot = 0;
|
||||
// Runs the bring-up probe (memoized per process) and decides
|
||||
// m_primGenRerouteKind. Called at the end of device creation: it records on
|
||||
// m_graphicsQueue, which nothing else is using yet.
|
||||
void ArmPrimGenReroute();
|
||||
|
||||
public:
|
||||
// Whether a GENERATED span opened now will have the draws made while the GL
|
||||
// span is PAUSED counted on the GPU - through the reroute pool, which takes
|
||||
// every draw with no open capture, or (where the reroute is not armed because
|
||||
// the stream query was measured to count capture-less draws) through the stream
|
||||
// slot such a draw still takes. The frontend's CPU paused-primitive counter
|
||||
// must not be added on top of either: it would double count, and it cannot
|
||||
// price the draws that matter anyway - only 3 of the ~15 draw entry points
|
||||
// write it and it answers 0 for GL_PATCHES. Read once per span, after
|
||||
// StartXfbQueryCapture (whose pool creation may disarm the reroute).
|
||||
Bool ArePausedDrawsGpuCounted() const;
|
||||
// kind: 0 = PRIMITIVES_WRITTEN, 1 = PRIMITIVES_GENERATED.
|
||||
Bool StartXfbQueryCapture(Uint32 kind);
|
||||
void StopXfbQueryCapture(Uint32 kind, Vector<Uint32>& outSlots);
|
||||
Bool ResolveXfbQueryResult(const Vector<Uint32>& slots, Bool wantGenerated, Uint64& outPrimitives);
|
||||
void StopXfbQueryCapture(Uint32 kind, Vector<Uint32>& outSlots, Vector<Uint32>& outRerouteSlots);
|
||||
Bool ResolveXfbQueryResult(const Vector<Uint32>& slots, const Vector<Uint32>& rerouteSlots,
|
||||
Bool wantGenerated, Uint64& outPrimitives);
|
||||
|
||||
private:
|
||||
void BeginXfbQueryForDraw(VkCommandBuffer commandBuffer);
|
||||
void BeginXfbQueryForDraw(VkCommandBuffer commandBuffer, Bool xfbActive);
|
||||
void EndXfbQueryForDraw(VkCommandBuffer commandBuffer);
|
||||
|
||||
VkCommandPool m_commandPool = VK_NULL_HANDLE;
|
||||
@@ -726,29 +827,186 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 programHash = 0;
|
||||
Uint64 vertexInputHash = 0;
|
||||
Uint64 renderPassHash = 0;
|
||||
// VALUE hash of the pipeline-relevant fixed-function state (see
|
||||
// ComputePipelineStateHash), not the monotonic pipeline-state version:
|
||||
// the version never repeats, so a per-draw GL_BLEND toggle would miss
|
||||
// all entries forever even though the state alternates between two
|
||||
// values the memo already holds.
|
||||
// The PRE-HANDLE arm's key component (P2 brief D12.1), and 0 in every entry the
|
||||
// handle arm mints. VALUE hash of the pipeline-relevant fixed-function state (see
|
||||
// ComputePipelineStateHash), not the monotonic pipeline-state version: the version
|
||||
// never repeats, so a per-draw GL_BLEND toggle would miss all entries forever even
|
||||
// though the state alternates between two values the memo already holds.
|
||||
Uint64 pipelineStateHash = 0;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// The HANDLE arm's key component, and the whole of D12.1: the CLIENT already
|
||||
// hashed the pipeline subset of RenderStateParameters and minted a content-
|
||||
// addressed CSO for it (MG_Pipe/MGPipeRenderStateSpans.h, MG_Impl/Pipe/CsoCache),
|
||||
// so re-hashing the same 396 bytes here was work the boundary had already done.
|
||||
// Two draws share a CSO handle exactly when their pipeline bytes are equal, and
|
||||
// the client's subset is a strict SUPERSET of what ComputePipelineStateHash read,
|
||||
// so the handle discriminates at least as finely as the hash it replaces.
|
||||
//
|
||||
// renderPassHash STAYS beside it and is what keeps this key complete: the CSO
|
||||
// carries GL state only, while colorAttachmentCount and the rasterization sample
|
||||
// count - which ComputePipelineStateHash folded in through its signature and
|
||||
// through ResolveEffectiveSampleMask - are render-pass facts that the render-pass
|
||||
// hash already separates.
|
||||
//
|
||||
// Null in an entry minted by the legacy arm, so entries of the two arms can never
|
||||
// match each other: the compare below tests BOTH components.
|
||||
MG_Pipe::MGPipeHandle renderStateCso = MG_Pipe::kMGPipeNullHandle;
|
||||
#endif
|
||||
ProgramFactory::CompileOptionFlags transformFlags = {};
|
||||
// Baked into the pipeline (PipelineFactory::ComputeHash mixes it), and NOT derivable
|
||||
// from anything else in this key: it depends on whether the draw is indexed and on the
|
||||
// index type, neither of which the mode/program/state hashes carry. Without it an
|
||||
// indexed and a non-indexed draw over the same program and state collide on one entry
|
||||
// and the second one gets the first one's restart setting.
|
||||
Bool primitiveRestartEnable = false;
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
};
|
||||
static constexpr Uint32 kPipelineMemoSize = 8;
|
||||
PipelineMemoEntry m_pipelineMemo[kPipelineMemoSize];
|
||||
Uint32 m_pipelineMemoCount = 0;
|
||||
Uint32 m_pipelineMemoNext = 0;
|
||||
// Hash of every fixed-function GL state the pipeline payload reads that the
|
||||
// memo key's other fields (mode / program / vertex input / render pass /
|
||||
// transform flags) do not already pin down. Equal hash under an equal rest
|
||||
// of key => byte-identical PipelineCreatePayload. Cached per pipeline-state
|
||||
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P2 D12.1's arm selector, and the whole of the pipeline memo's re-key. Returns the
|
||||
// render-state CSO this draw is keyed on, or the null handle when the pre-handle arm
|
||||
// is the one that runs.
|
||||
//
|
||||
// Under the handle arm the memo's state key IS this handle. The client hashed those
|
||||
// 396 pipeline bytes when it minted the CSO (MGPipeComputePipelineSubsetHash), so
|
||||
// recomputing an overlapping hash here was work the boundary had already done; the
|
||||
// client's pipeline subset is a strict SUPERSET of what ComputePipelineStateHash read,
|
||||
// so the handle discriminates at least as finely as the hash it replaces. What the
|
||||
// handle does NOT carry is the render-pass side - colorAttachmentCount and the
|
||||
// rasterization sample count, which ComputePipelineStateHash folded in through its
|
||||
// signature and through ResolveEffectiveSampleMask - and that is exactly why
|
||||
// entry.renderPassHash stays in the key beside it.
|
||||
//
|
||||
// The arm is live only when the render-state subsystem is migrated in this run AND the
|
||||
// client has actually bound a CSO. The second half is not belt and braces: a tree whose
|
||||
// tracker does not emit create/bind_render_state yet has no handle to key on, and
|
||||
// delete_render_state clears the binding (MG_Pipe/PipeApply.cpp), so the null handle is
|
||||
// reachable on any tree. Keying every draw on it would alias every render state onto
|
||||
// one memo entry, so a null handle means "fall back to a state hash" - never an abort,
|
||||
// and never a per-draw consultation of the legacy-memo lever: bit 0 is not a Track-H
|
||||
// subsystem (D14 labels only bits 5 and 6 that), and the lever's Fatal is a STARTUP
|
||||
// one, in MagmaPipeValidateSubsystemConfiguration.
|
||||
//
|
||||
// The fallback is warned ONCE rather than logged at debug, and that is deliberate: a
|
||||
// silent fallback is what makes "the CSO arm never ran" easy to miss. W is compiled in
|
||||
// at every shipped log level.
|
||||
//
|
||||
// The latch is a plain member bool, NOT MGLOG_W_ONCE. MOBILEGL_LOG_ONCE_INTERNAL
|
||||
// (MG_Util/Debug/Log.h) is an UNCONDITIONAL std::atomic_flag::test_and_set - a locked
|
||||
// xchg, executed on every evaluation, not "one static bool test" as an earlier round of
|
||||
// this comment claimed - and this site is on the per-draw pipeline path in the very
|
||||
// configuration that reaches it (no tracker: every draw). ROADMAP.md:7 forbids leaving
|
||||
// instrumentation on a hot path, so the once-ness is one non-atomic, always-predicted
|
||||
// load of a member that is false exactly once. Single-threaded like the rest of the
|
||||
// renderer, and per renderer rather than per process, which is also the right scope: a
|
||||
// second context that never binds a CSO deserves to say so.
|
||||
//
|
||||
// What the absence of this warning from a run's log proves, EXACTLY: that no draw took
|
||||
// the fallback WHILE bit 0 was set. With kMGPipeSubsystemRenderState clear the function
|
||||
// returns before the latch, so absence proves nothing at all - and no draw is keyed on a
|
||||
// handle either. Grep the mask out of the log beside it (review v2 minor 3).
|
||||
//
|
||||
// Push-only by construction: the pull build does not compile this function at all, so
|
||||
// its two callers are statement-for-statement what they were (G1).
|
||||
//
|
||||
// [routed to the integrator, review v2 minor 11] MG_Pipe::MGPipeApplier() is ONE
|
||||
// process-global applier (MG_Pipe/PipeApply.cpp), not the per-context CSO store D2
|
||||
// specifies. In a multi-context process this reads whatever CSO another context last
|
||||
// bound. The defect is package A's and the fix belongs there; Magma is its only P2
|
||||
// consumer, so it is named here rather than left for both reviews to assume the other
|
||||
// caught it.
|
||||
MG_Pipe::MGPipeHandle ResolveBoundRenderStateCso() const {
|
||||
if (!MagmaPipeSubsystemOn(MG_Pipe::kMGPipeSubsystemRenderState)) {
|
||||
return MG_Pipe::kMGPipeNullHandle;
|
||||
}
|
||||
const MG_Pipe::MGPipeHandle boundCso = MG_Pipe::MGPipeApplier().BoundRenderStateCso;
|
||||
if (MG_Pipe::MGPipeHandleIsNull(boundCso) && !m_pipelineCsoFallbackWarned) {
|
||||
m_pipelineCsoFallbackWarned = true;
|
||||
MGLOG_W("MGPipe: kMGPipeSubsystemRenderState is on but no render-state CSO is "
|
||||
"bound; the pipeline memo is running on a state hash, not on the CSO "
|
||||
"handle (no tracker on this build, or a draw between "
|
||||
"delete_render_state and the next bind)");
|
||||
}
|
||||
return boundCso;
|
||||
}
|
||||
// Latch for the warning above. Mutable because the resolve is const and the latch is
|
||||
// not part of the renderer's observable state.
|
||||
mutable Bool m_pipelineCsoFallbackWarned = false;
|
||||
// The memo key's STATE-HASH half, for a draw that has no CSO handle to key on: the
|
||||
// pre-handle arm, and the fallback of D12.1's handle arm. Cached on the pipeline-state
|
||||
// version plus the two render-pass facts the hash's inputs depend on, so an unchanged
|
||||
// (version, colorAttachmentCount, sampleCount) proves the bytes are unchanged.
|
||||
//
|
||||
// [deviation from D12.1] The brief deletes this gate and its cached fields outright.
|
||||
// They cannot go while a no-CSO draw is reachable - and it is, on any tree: a draw
|
||||
// between delete_render_state and the next bind has no handle. On a tree whose tracker
|
||||
// binds a CSO these five words are written once and never read again; they retire for
|
||||
// real when the pull path does, at P13.
|
||||
Uint64 ResolveFallbackPipelineStateHash(Uint renderStateVersion, Uint32 colorAttachmentCount,
|
||||
VkSampleCountFlagBits rasterizationSamples) {
|
||||
if (!m_pipelineStateHashValid || m_pipelineStateHashVersion != renderStateVersion ||
|
||||
m_pipelineStateHashColorCount != colorAttachmentCount ||
|
||||
m_pipelineStateHashSampleCount != rasterizationSamples) {
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
m_pipelineStateHash =
|
||||
ComputePipelineStateHash(colorAttachmentCount, rasterizationSamples);
|
||||
#else
|
||||
m_pipelineStateHash = ComputePipelineSubsetStateHashFallback();
|
||||
#endif
|
||||
m_pipelineStateHashVersion = renderStateVersion;
|
||||
m_pipelineStateHashColorCount = colorAttachmentCount;
|
||||
m_pipelineStateHashSampleCount = rasterizationSamples;
|
||||
m_pipelineStateHashValid = true;
|
||||
}
|
||||
return m_pipelineStateHash;
|
||||
}
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
#if MOBILEGL_PIPE_PUSH && !MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
// The same answer as ComputePipelineStateHash, computed from the P2 chunk table
|
||||
// instead of from a hand-written field list, for the build that compiles no
|
||||
// pre-handle arm (cmake -DMOBILEGL_PIPE_LEGACY_MEMOS=OFF). It is the CLIENT's own
|
||||
// hash function - MGPipeComputePipelineSubsetHash over the 396 pipeline bytes - so a
|
||||
// draw keyed on it and a draw keyed on a CSO handle are keyed on the same equivalence
|
||||
// class of state, and the render-pass facts stay separated by renderPassHash either
|
||||
// way. This is what makes the no-legacy build RUNNABLE rather than a configuration
|
||||
// that aborts on the first draw that arrives without a CSO.
|
||||
Uint64 ComputePipelineSubsetStateHashFallback() const;
|
||||
#endif
|
||||
#if MOBILEGL_PIPE_LEGACY_MEMOS
|
||||
// THE PRE-HANDLE ARM (P2 brief D12.1 / D14). Hash of every fixed-function GL state the
|
||||
// pipeline payload reads that the memo key's other fields (mode / program / vertex
|
||||
// input / render pass / transform flags) do not already pin down. Equal hash under an
|
||||
// equal rest of key => byte-identical PipelineCreatePayload. Cached per pipeline-state
|
||||
// version: the version is monotonic and bumps on every pipeline-state
|
||||
// change, so an unchanged (version, colorAttachmentCount) proves the state
|
||||
// bytes are unchanged and the hash can be reused without re-reading them.
|
||||
Uint64 ComputePipelineStateHash(Uint32 colorAttachmentCount) const;
|
||||
//
|
||||
// The handle arm computes none of this: the client hashed the same bytes when it
|
||||
// minted the CSO, so all five cached-hash members below exist only to avoid a
|
||||
// re-hash the handle arm never performs.
|
||||
Uint64 ComputePipelineStateHash(Uint32 colorAttachmentCount,
|
||||
VkSampleCountFlagBits rasterizationSamples) const;
|
||||
#endif
|
||||
// The effective GL_SAMPLE_MASK word for a draw at this rasterization sample count; see
|
||||
// the definition for the GL-vs-Vulkan rule it reconciles. Shared by the pipeline payload
|
||||
// and the pipeline-state memo word so the two cannot disagree. NOT part of the legacy
|
||||
// arm: it is a PAYLOAD computation that depends on rasterizationSamples, so it survives
|
||||
// the re-key and keeps reading Multisample / SampleMask / SampleMaskValue out of the
|
||||
// working block.
|
||||
Uint32 ResolveEffectiveSampleMask(VkSampleCountFlagBits rasterizationSamples) const;
|
||||
// ResolveFallbackPipelineStateHash's cache. Written once and never read again on a
|
||||
// build whose client binds a render-state CSO; see that function for why it survives
|
||||
// the re-key at all.
|
||||
Uint m_pipelineStateHashVersion = 0;
|
||||
Uint32 m_pipelineStateHashColorCount = 0;
|
||||
// The sample count the cached hash was computed at. A pipeline-state input now depends on
|
||||
// it (the effective sample mask), so a draw that changes only the target's sample count
|
||||
// has to recompute rather than reuse.
|
||||
VkSampleCountFlagBits m_pipelineStateHashSampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
Uint64 m_pipelineStateHash = 0;
|
||||
Bool m_pipelineStateHashValid = false;
|
||||
// GetShaderTransformFlags memo. NOT pure in the pre-transform alone: the
|
||||
@@ -768,7 +1026,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Drops every memoized pipeline handle. Required at command-buffer
|
||||
// boundaries and whenever any pipeline may have been destroyed. Also drops
|
||||
// the cached pipeline-state hash: the same boundaries can retire the GL
|
||||
// context whose monotonic version the cache is keyed on.
|
||||
// context whose monotonic version the cache is keyed on. The handle arm has no
|
||||
// such cache to drop - a CSO handle is not derived from a monotonic version.
|
||||
void InvalidatePipelineMemo() {
|
||||
m_pipelineMemoCount = 0;
|
||||
m_pipelineMemoNext = 0;
|
||||
@@ -790,7 +1049,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Skip the per-draw CollectSampledTextures walk (~5% of the render thread) when the sampled
|
||||
// texture SET is provably unchanged from the previous draw: same program (lifetime id +
|
||||
// backend-state version, which covers sampler-uniform reassignment / relink) and transform
|
||||
// flags, and no texture bind/unbind/delete since (GetTextureBindGeneration). On a hit,
|
||||
// flags, no texture bind/unbind/delete since (GetTextureBindGeneration), and nothing that
|
||||
// moves a texture's shape or a sampler's parameters since (GetSamplingResolutionGeneration
|
||||
// - membership depends on mipmap-completeness, which both of those decide). On a hit,
|
||||
// m_sampledTexturesScratch still holds the previous draw's list and steps 2-4 (feedback /
|
||||
// layout probe / transition) re-run on it, so layout correctness is unaffected - only the GL
|
||||
// walk is skipped. The program lifetime id (never reused, unlike the GL name) and the
|
||||
@@ -801,6 +1062,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 m_lastSampledSetProgramVersion = 0;
|
||||
ProgramFactory::CompileOptionFlags m_lastSampledSetTransformFlags = {};
|
||||
Uint64 m_lastSampledSetBindGeneration = 0;
|
||||
Uint64 m_lastSampledSetSamplingGeneration = 0;
|
||||
// Set from the draw's resolved VkProgramObject on both the full and the fast setup paths;
|
||||
// read by BeginXfbCaptureForDraw, which has only GL state otherwise. See
|
||||
// VkProgramObject::xfbCaptureDeclined.
|
||||
Bool m_currentDrawXfbCaptureDeclined = false;
|
||||
|
||||
// Memo for the per-draw explicit-LOD-0 eligibility probe
|
||||
// (ProgramSamplesOnlySingleLevelTextures): same key family as the
|
||||
@@ -852,6 +1118,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// common shape), and "the VAO did not move" would then skip the layout
|
||||
// re-resolve for a different VAO.
|
||||
Uint64 vaoLifetimeId = 0;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P2 D12.4: the handle arm's answer to the same question, and one compare rather
|
||||
// than the pair above. Kept BESIDE them rather than replacing them because the
|
||||
// pre-handle arm is still compiled (MOBILEGL_PIPE_LEGACY_MEMOS) and this snapshot
|
||||
// is a value struct, not a wire type.
|
||||
MG_Pipe::MGPipeHandle vaoHandle = MG_Pipe::kMGPipeNullHandle;
|
||||
#endif
|
||||
Uint32 vaoConfigVersion = 0;
|
||||
const void* drawFbo = nullptr;
|
||||
// Never-reused lifetime id beside the raw pointer + Uint16 version: a
|
||||
@@ -865,6 +1138,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 bindGeneration = 0;
|
||||
Uint32 baseTransformFlags = 0;
|
||||
Uint32 resolvedTransformFlags = 0;
|
||||
// What ResolvePrimitiveRestartEnable answered for the draw this snapshot was taken
|
||||
// from, i.e. what its pipeline's primitiveRestartEnable was built with. `aspects`
|
||||
// already separates indexed from non-indexed draws, but not one index TYPE from
|
||||
// another, and a restart index that fits GL_UNSIGNED_INT but not GL_UNSIGNED_SHORT
|
||||
// makes those two draws want different pipelines.
|
||||
Bool primitiveRestartEnable = false;
|
||||
Uint64 renderPassHash = 0;
|
||||
Uint32 imageIndex = 0;
|
||||
Uint64 textureEraseEpoch = 0;
|
||||
@@ -893,6 +1172,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// probe the pipeline memo after a state change without re-fetching the
|
||||
// render-pass entry (the pass itself is pinned by renderPassHash above).
|
||||
Uint32 renderPassColorCount = 0;
|
||||
// Pinned with the colour count and for the same reason: the fast path recomputes the
|
||||
// pipeline-state value hash from the snapshot, and that hash reads the sample count.
|
||||
VkSampleCountFlagBits renderPassSampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
// layoutHash of the snapshotting draw's vertex-input state. The pipeline and
|
||||
// the vertex-input pre-flight depend on the VAO only through this (plus the
|
||||
@@ -1117,6 +1399,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// - bindings revalidates per draw exactly as before (frame serial, content
|
||||
// hash, per-binding live buffer pointers and slice epochs).
|
||||
struct alignas(64) VaoDrawMemo {
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P2 D12.4: the handle arm's key, and the ONLY key it needs. {slot, gen} is an
|
||||
// identity, so the pointer-plus-lifetime-id pair below stops being a key here;
|
||||
// the slot also picks the table entry, so the address hash and the two-way probe
|
||||
// go with it. Null in an entry that has never been claimed.
|
||||
MG_Pipe::MGPipeHandle vaoHandle = MG_Pipe::kMGPipeNullHandle;
|
||||
#endif
|
||||
const MG_State::GLState::VertexArrayObject* vaoKey = nullptr;
|
||||
// The VAO's never-reused lifetime id, checked alongside vaoKey. The pointer
|
||||
// ALONE is not an identity: a deleted VAO's heap address is handed straight
|
||||
@@ -1143,8 +1432,48 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// fixed table also makes every VaoDrawMemo/ResolvedVertexBindings pointer
|
||||
// stable for the duration of a draw, which the EBO memo handoff
|
||||
// (m_currentDrawResolvedEntry) relies on.
|
||||
//
|
||||
// [deviation from D12.4, deliberate and narrow] The brief asks for a grow-on-demand
|
||||
// Vector. This one stays FIXED at exactly the capacity and exactly the 2-way victim
|
||||
// rule it has on the base ref, and only its KEY changes (a {slot, gen} handle instead
|
||||
// of a hashed heap address plus a lifetime id). Two reasons, and the second is the
|
||||
// whole of review v2's MAJOR 1:
|
||||
// * a VaoDrawMemo is ~450 B (ResolvedVertexBindings dominates), so growing this
|
||||
// table with the live VAO set is megabytes on a platform with an LMK, where the
|
||||
// other two memos are 48 B and can afford it;
|
||||
// * this is the ONLY one of the three memos that had a capacity before this package.
|
||||
// Losing an entry here costs a vertex-binding re-resolve, exactly what losing it
|
||||
// cost on the base ref, so at any working-set size this table is no worse than what
|
||||
// it replaces - and strictly better below capacity, where the handle is a bijection
|
||||
// with the slot and the two-way probe never collides at all. The other two memos
|
||||
// (VertexInputStateFactory::m_vaoMemos) had NO capacity, so they keep having none.
|
||||
static constexpr Uint32 kVaoDrawMemoSlotCount = 2048; // power of two
|
||||
Vector<VaoDrawMemo> m_vaoDrawMemoTable;
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// The renderer's {slot, gen} mint, shared with its VertexInputStateFactory so both
|
||||
// derive the same handle for the same VAO. Per renderer, never a process-global: a
|
||||
// global would share one table and one reclamation clock across two live contexts and
|
||||
// outlive every one of them (review v2 minor 4).
|
||||
MagmaPipeIdentityTables m_pipeIdentity;
|
||||
// The VAO's {slot, gen}. A one-entry memo hit for every acquisition after a draw's
|
||||
// first, so there is no second memo in front of it here.
|
||||
MG_Pipe::MGPipeHandle ResolveVaoHandle(const MG_State::GLState::VertexArrayObject& vao) {
|
||||
return m_pipeIdentity.HandleOf(MG_Pipe::MGPipeKind::VertexElementsCso,
|
||||
vao.GetLifetimeId());
|
||||
}
|
||||
#endif
|
||||
// "Is this VAO's content hash already memoized?", asked of whichever side owns the
|
||||
// memo (P2 D12.5). Force-inlined and defined in the class body so that the PULL
|
||||
// build's three readers keep compiling to the very same two loads they always did -
|
||||
// G1 admits no resize, and an out-of-line call here would be one.
|
||||
[[gnu::always_inline]] inline Bool VaoContentHashIfKnown(
|
||||
const MG_State::GLState::VertexArrayObject& vao, Uint64& outHash) const {
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
return m_vertexInputStateFactory->TryGetMemoizedHash(vao, outHash);
|
||||
#else
|
||||
return vao.GetBackendHashMemo(outHash);
|
||||
#endif
|
||||
}
|
||||
// Finds the slot holding `vao`, or recycles the older of its two candidate
|
||||
// slots into an empty memo keyed on `vao`. Never returns null.
|
||||
VaoDrawMemo* LookupVaoDrawMemo(const MG_State::GLState::VertexArrayObject* vao);
|
||||
@@ -1168,13 +1497,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void CreateSwapchain();
|
||||
void CreateCommandPool();
|
||||
|
||||
// Whether THIS draw's primitive stream restarts, and therefore what
|
||||
// VkPipelineInputAssemblyStateCreateInfo::primitiveRestartEnable must be. Resolved by the
|
||||
// caller because it needs two facts a pipeline cannot see: whether the draw is indexed at
|
||||
// all (GL primitive restart acts on the index stream, so it is a no-op for glDrawArrays),
|
||||
// and the index TYPE (an application restart index that does not fit the type matches no
|
||||
// index, so that draw restarts nowhere - see UploadAndBindIndexBuffer).
|
||||
Bool ResolvePrimitiveRestartEnable(Flags<DrawSetupAspect> aspects,
|
||||
const IndexBufferView* pIndexBufferView) const;
|
||||
|
||||
VkPipeline GetOrCreatePipeline(
|
||||
GLenum mode,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
ProgramFactory::CompileOptionFlags transformFlags,
|
||||
const MG_State::GLState::VertexArrayObject& vao,
|
||||
const RenderPassEntry& renderPassEntry);
|
||||
const RenderPassEntry& renderPassEntry,
|
||||
Bool primitiveRestartEnable);
|
||||
VkPipeline GetOrCreateComputePipeline(const ProgramFactory::VkProgramObject& programObj);
|
||||
void DestroyComputePipelines();
|
||||
// Takes the frame rather than a command buffer: a first-time storage-usage upgrade has to
|
||||
@@ -1241,13 +1580,25 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLenum filter);
|
||||
// Clears one z slice of a VK_IMAGE_TYPE_3D colour image. See the call site in
|
||||
// MaterializePendingClearForTexture for why a transfer clear cannot do this.
|
||||
// Clears one layer of a colour image through a throwaway render pass whose entire content
|
||||
// is its LOAD_OP_CLEAR. Two callers, both of which a transfer clear cannot serve: a z
|
||||
// slice of a VK_IMAGE_TYPE_3D image (vkCmdClearColorImage cannot name one), and a
|
||||
// MULTISAMPLE image (which carries no TRANSFER_DST usage at all). `finalLayout` is the
|
||||
// layout the caller already tracks for the whole image, so this never has to touch
|
||||
// resource->layout.
|
||||
Bool ClearDepthSliceWithRenderPass(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject& texture, Uint32 mipLevel,
|
||||
Uint32 depthSlice, const VkClearValue& clearValue);
|
||||
Uint32 depthSlice, const VkClearValue& clearValue,
|
||||
VkImageLayout finalLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL);
|
||||
Bool MaterializePendingClearForTexture(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
// The multisample arm of the above. Split out rather than branched inline because it
|
||||
// shares none of the transfer path: a multisample image carries no TRANSFER_DST usage, so
|
||||
// neither the TRANSFER_DST transition nor vkCmdClearColorImage is legal on one.
|
||||
Bool MaterializeMultisamplePendingClear(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject& texture,
|
||||
VkTextureManager::TextureResource& resource,
|
||||
const Vector<PendingClearEntry>& pendingClears);
|
||||
Bool MaterializePendingClearForRenderbuffer(
|
||||
VkCommandBuffer commandBuffer,
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
|
||||
@@ -39,18 +39,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
inline Bool ShouldEmulateSubgroups(const Bool nativeSubgroupSupported) {
|
||||
return MG_Config::Features.MagmaEmulateSubgroup && !nativeSubgroupSupported &&
|
||||
!MG_Config::Features.DisableSubgroup;
|
||||
!MG_Config::Features.MagmaDisableSubgroup;
|
||||
}
|
||||
|
||||
inline Bool ShouldFixIterationRPSubgroupScratch() {
|
||||
// Auto is ON: the patch is fingerprint-gated to iterationRP's reduction and
|
||||
// grows one under-declared array; every other module passes through untouched.
|
||||
return MG_Config::Features.FixIterationRPSubgroupScratch !=
|
||||
return MG_Config::Features.MagmaFixIterationRPSubgroupScratch !=
|
||||
MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
|
||||
inline Bool ShouldFixIterationRPBarrier() {
|
||||
return MG_Config::Features.IterationRPFixBarrier;
|
||||
return MG_Config::Features.MagmaIterationRPFixBarrier;
|
||||
}
|
||||
|
||||
inline Bool ShouldDeriveNumSubgroups() {
|
||||
@@ -58,6 +58,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// contract to hold, and the derived ceil() value is the one the renderer can pin
|
||||
// with REQUIRE_FULL_SUBGROUPS - the driver builtin is the value with no
|
||||
// cross-driver guarantee (Adreno returns 1 for an 8-subgroup dispatch).
|
||||
return MG_Config::Features.DeriveNumSubgroups != MG_Config::QuirkOverride::ForceOff;
|
||||
return MG_Config::Features.MagmaDeriveNumSubgroups != MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -0,0 +1,179 @@
|
||||
// MobileGL - MobileGL/MG_Backend/MGPipe/PipeInputs.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
// The backend-side half of the PipeInputs block: the poison Fatal with its verb name, the
|
||||
// name lookups the runtime knobs need, and - in a verify build - the per-field equality,
|
||||
// the entry comparator and the corruption injector. Compiled only under MOBILEGL_PIPE_PUSH
|
||||
// (CMakeLists.txt appends it to SOURCE_FILES there), so the pull build never sees it. Spells
|
||||
// no MG_State global: everything that reads the live context lives in MG_Impl/Pipe/PipeFill.cpp.
|
||||
#include <MG_Backend/MGPipe/PipeInputs.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
const char* MGPipeVerbName(MGPipeVerb verb) {
|
||||
const auto index = static_cast<SizeT>(verb);
|
||||
return index < kMGPipeVerbCount ? kMGPipeVerbNames[index] : "<none>";
|
||||
}
|
||||
|
||||
[[noreturn]] void MGPipeInputPoisonFatalForVerb(MGPipeInputField field, MGPipeVerb verb) {
|
||||
MGPipeInputPoisonFatal(field, MGPipeVerbName(verb));
|
||||
}
|
||||
|
||||
Optional<MGPipeInputField> MGPipeFindInputField(const char* name) {
|
||||
if (name == nullptr) return std::nullopt;
|
||||
for (SizeT i = 0; i < kMGPipeInputFieldCount; ++i) {
|
||||
if (std::strcmp(kMGPipeInputFieldNames[i], name) == 0) return static_cast<MGPipeInputField>(i);
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
Optional<MGPipeVerb> MGPipeFindVerb(const char* name) {
|
||||
if (name == nullptr) return std::nullopt;
|
||||
for (SizeT i = 0; i < kMGPipeVerbCount; ++i) {
|
||||
if (std::strcmp(kMGPipeVerbNames[i], name) == 0) return static_cast<MGPipeVerb>(i);
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
#if MOBILEGL_PIPE_VERIFY
|
||||
namespace {
|
||||
using CurrentVertexAttributeValue = PipeInputs::CurrentVertexAttributeValue;
|
||||
|
||||
// Every overload is declared up front: the array overloads recurse into their element
|
||||
// type, and a call inside a template only sees what was declared before the template.
|
||||
template <class T>
|
||||
Bool StorageEqual(const T& a, const T& b);
|
||||
template <class T>
|
||||
Bool StorageEqual(T* const& a, T* const& b);
|
||||
template <class T>
|
||||
Bool StorageEqual(const SharedPtr<T>& a, const SharedPtr<T>& b);
|
||||
template <class T, SizeT N>
|
||||
Bool StorageEqual(const T (&a)[N], const T (&b)[N]);
|
||||
Bool StorageEqual(const PipeInputs::IndexedCapabilities& a, const PipeInputs::IndexedCapabilities& b);
|
||||
Bool StorageEqual(const CurrentVertexAttributeValue& a, const CurrentVertexAttributeValue& b);
|
||||
template <class T>
|
||||
void CorruptStorage(T& v);
|
||||
template <class T>
|
||||
void CorruptStorage(T*& p);
|
||||
template <class T>
|
||||
void CorruptStorage(SharedPtr<T>& p);
|
||||
template <class T, SizeT N>
|
||||
void CorruptStorage(T (&a)[N]);
|
||||
void CorruptStorage(PipeInputs::IndexedCapabilities& c);
|
||||
void CorruptStorage(CurrentVertexAttributeValue& v);
|
||||
|
||||
// ---- equality over one field's storage ----
|
||||
// O-class storage compares by identity: a raw pointer into the context, or the object a
|
||||
// SharedPtr owns. Everything else goes through G4's MGPipeFieldEqual, recursing through
|
||||
// C arrays element-wise.
|
||||
template <class T>
|
||||
Bool StorageEqual(T* const& a, T* const& b) {
|
||||
return a == b;
|
||||
}
|
||||
template <class T>
|
||||
Bool StorageEqual(const SharedPtr<T>& a, const SharedPtr<T>& b) {
|
||||
return a.get() == b.get();
|
||||
}
|
||||
template <class T, SizeT N>
|
||||
Bool StorageEqual(const T (&a)[N], const T (&b)[N]) {
|
||||
for (SizeT i = 0; i < N; ++i) {
|
||||
if (!StorageEqual(a[i], b[i])) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
Bool StorageEqual(const PipeInputs::IndexedCapabilities& a, const PipeInputs::IndexedCapabilities& b) {
|
||||
return StorageEqual(a.Blend, b.Blend) && StorageEqual(a.ScissorTest, b.ScissorTest);
|
||||
}
|
||||
// Three scalar arrays and nothing else (Core.h), so a bitwise compare has no padding to
|
||||
// false-differ on and keeps a NaN float attribute equal to itself. The size assertion is
|
||||
// what turns a fourth member into a build break rather than a blind spot.
|
||||
Bool StorageEqual(const CurrentVertexAttributeValue& a, const CurrentVertexAttributeValue& b) {
|
||||
static_assert(sizeof(CurrentVertexAttributeValue) == 3 * 4 * 4,
|
||||
"CurrentVertexAttributeValue grew a member; update the comparator");
|
||||
return std::memcmp(&a, &b, sizeof(CurrentVertexAttributeValue)) == 0;
|
||||
}
|
||||
template <class T>
|
||||
Bool StorageEqual(const T& a, const T& b) {
|
||||
return MGPipeFieldEqual(a, b);
|
||||
}
|
||||
|
||||
// ---- corruption of one field's storage ----
|
||||
// Every shape is perturbed in a way the comparator above must see: a Bool flips, a
|
||||
// scalar or enum moves by one, a pointer's low bits are flipped (never dereferenced:
|
||||
// the snapshot is only ever compared), a SharedPtr becomes an aliasing pointer to a
|
||||
// flipped address with no control block, an array corrupts its first element, and any
|
||||
// other struct has its first byte XOR'ed with 0x5A.
|
||||
template <class T>
|
||||
T* FlipPointer(T* p) {
|
||||
return reinterpret_cast<T*>(reinterpret_cast<std::uintptr_t>(p) ^ 0x5A);
|
||||
}
|
||||
template <class T>
|
||||
void CorruptStorage(T*& p) {
|
||||
p = FlipPointer(p);
|
||||
}
|
||||
template <class T>
|
||||
void CorruptStorage(SharedPtr<T>& p) {
|
||||
p = SharedPtr<T>(SharedPtr<T>(), FlipPointer(p.get()));
|
||||
}
|
||||
template <class T, SizeT N>
|
||||
void CorruptStorage(T (&a)[N]) {
|
||||
CorruptStorage(a[0]);
|
||||
}
|
||||
void CorruptStorage(PipeInputs::IndexedCapabilities& c) {
|
||||
CorruptStorage(c.Blend);
|
||||
}
|
||||
void CorruptStorage(CurrentVertexAttributeValue& v) {
|
||||
v.floatValue[0] += 1.f;
|
||||
}
|
||||
template <class T>
|
||||
void CorruptStorage(T& v) {
|
||||
if constexpr (std::is_same_v<T, Bool>) {
|
||||
v = !v;
|
||||
} else if constexpr (std::is_enum_v<T>) {
|
||||
v = static_cast<T>(static_cast<std::underlying_type_t<T>>(v) + 1);
|
||||
} else if constexpr (std::is_arithmetic_v<T>) {
|
||||
v = static_cast<T>(v + 1);
|
||||
} else {
|
||||
static_assert(std::is_trivially_copyable_v<T>, "PipeInputs storage must be trivially copyable");
|
||||
unsigned char first = 0;
|
||||
std::memcpy(&first, &v, 1);
|
||||
first ^= 0x5A;
|
||||
std::memcpy(&v, &first, 1);
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool MGPipeInputsFieldEqual(MGPipeInputField field, const PipeInputs& a, const PipeInputs& b) {
|
||||
// A forwarded field has no storage and is equal by definition; VisitStorage answers
|
||||
// false for it, hence the explicit sticky test first.
|
||||
if (kMGPipeInputFieldSticky[static_cast<SizeT>(field)]) return true;
|
||||
return PipeInputs::VisitStorage(field, a, b, [](const auto& x, const auto& y) { return StorageEqual(x, y); });
|
||||
}
|
||||
|
||||
Bool MGPipeVerifyInputs(const PipeInputs& pushed, const PipeInputs& snapshot, const MGPipeFieldMask& mask,
|
||||
MGPipeInputField* outField) {
|
||||
for (SizeT i = 0; i < kMGPipeInputFieldCount; ++i) {
|
||||
const auto field = static_cast<MGPipeInputField>(i);
|
||||
if (!MGPipeFieldMaskHas(mask, field)) continue;
|
||||
if (MGPipeInputsFieldEqual(field, pushed, snapshot)) continue;
|
||||
if (outField != nullptr) *outField = field;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool MGPipeApplyVerifyCorruption(PipeInputs& snapshot, MGPipeInputField field) {
|
||||
return PipeInputs::VisitStorage(field, snapshot, snapshot, [](auto& x, auto&) {
|
||||
CorruptStorage(x);
|
||||
return true;
|
||||
});
|
||||
}
|
||||
#endif // MOBILEGL_PIPE_VERIFY
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
@@ -0,0 +1,736 @@
|
||||
// MobileGL - MobileGL/MG_Backend/MGPipe/PipeInputs.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
// The frontend types the accessors return. Allowed here: P13 keeps this include for the
|
||||
// verify arm (ARCHITECTURE.md 9.5). This header spells no MG_State global - every read of
|
||||
// the live context happens on the client side, in MG_Impl/Pipe/PipeFill.cpp.
|
||||
#include <MG_State/GLState/Core.h>
|
||||
|
||||
// MOBILEGL_PIPE_POISON: the per-verb generation stamps and the read-side
|
||||
// Fatal{UnmigratedPipeInput} check. Derived here, once. The repository's debug gate is
|
||||
// MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG (Defines.h); the verify CI build is
|
||||
// Release/INFO with MOBILEGL_BUILD_DISAGGREGATED=OFF, so the third arm is what arms the poison
|
||||
// there without dragging MG_Remote in.
|
||||
#if MOBILEGL_PIPE_PUSH && (MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG || MOBILEGL_BUILD_DISAGGREGATED || \
|
||||
MOBILEGL_PIPE_VERIFY)
|
||||
#define MOBILEGL_PIPE_POISON 1
|
||||
#else
|
||||
#define MOBILEGL_PIPE_POISON 0
|
||||
#endif
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
// PipeInputs.cpp. The poison Fatal with the verb's name ("<none>" before the first
|
||||
// verb): MGLOG_F + std::abort(), live at every log level on purpose - this is not
|
||||
// MOBILEGL_ASSERT, which is inert in INFO builds.
|
||||
[[noreturn]] void MGPipeInputPoisonFatalForVerb(MGPipeInputField field, MGPipeVerb verb);
|
||||
// kMGPipeVerbNames[verb], or "<none>" for kVerbCount (no verb has been filled yet).
|
||||
const char* MGPipeVerbName(MGPipeVerb verb);
|
||||
// Name lookups for the runtime knobs (MOBILEGL_PIPE_VERIFY_CORRUPT names a field,
|
||||
// MOBILEGL_PIPE_POISON_OMIT a Verb:Field pair). Empty on an unknown name.
|
||||
Optional<MGPipeInputField> MGPipeFindInputField(const char* name);
|
||||
Optional<MGPipeVerb> MGPipeFindVerb(const char* name);
|
||||
|
||||
// The read-side poison check, on every non-forwarded accessor. Under MOBILEGL_PIPE_POISON
|
||||
// a read of a field whose stamp is older than the current verb serial is
|
||||
// Fatal{UnmigratedPipeInput, "Field@Verb"}; otherwise the accessor is a plain load.
|
||||
#if MOBILEGL_PIPE_POISON
|
||||
#define MGP_INPUT_CHECK(Field) \
|
||||
do { \
|
||||
if (!::MobileGL::MG_Pipe::MGPipeInputFieldIsFresh(m_filled, (Field))) { \
|
||||
::MobileGL::MG_Pipe::MGPipeInputPoisonFatalForVerb((Field), m_currentVerb); \
|
||||
} \
|
||||
} while (0)
|
||||
#else
|
||||
#define MGP_INPUT_CHECK(Field) ((void)0)
|
||||
#endif
|
||||
// The compare-at-read hook of the MOBILEGL_PIPE_VERIFY comparator (P1 brief D8), defined
|
||||
// in MG_Impl/Pipe/PipeFill.cpp: re-reads the field from the live context and compares it
|
||||
// against the stored value, and reports the FIRST divergence as
|
||||
// Fatal{PipeVerifyDiffer, "Field@Verb", verb=<serial>, where=read} (the indices go in a
|
||||
// preceding MGLOG_E). Only the live block (gPipeInputs) is verified; a snapshot's own
|
||||
// accessors are plain loads. Off in every other build.
|
||||
struct PipeInputs;
|
||||
#if MOBILEGL_PIPE_VERIFY
|
||||
void MGPipeVerifyReadHook(const PipeInputs& self, MGPipeInputField field, Uint index0, Uint index1);
|
||||
#define MGP_INPUT_VERIFY_READ(Field, Index0, Index1) \
|
||||
::MobileGL::MG_Pipe::MGPipeVerifyReadHook(*this, (Field), static_cast<Uint>(Index0), static_cast<Uint>(Index1))
|
||||
#else
|
||||
#define MGP_INPUT_VERIFY_READ(Field, Index0, Index1) ((void)0)
|
||||
#endif
|
||||
|
||||
// The V/O storage of every field that has storage, by field id. The seven F-class
|
||||
// (forwarded) fields have none. PipeInputs::VisitStorage dispatches on this list, which
|
||||
// is what keeps the comparator and the corruption injector one function each instead of
|
||||
// two sixty-way switches.
|
||||
// clang-format off
|
||||
#define MGP_INPUT_STORAGE_LIST(X) \
|
||||
X(GetActiveTextureUnit, m_activeTextureUnit) \
|
||||
X(GetBlendColor, m_blendColor) \
|
||||
X(GetBlendEquationIndexed, m_blendEquation) \
|
||||
X(GetBlendFuncIndexed, m_blendFunc) \
|
||||
X(GetBoundTransformFeedbackName, m_boundTransformFeedbackName) \
|
||||
X(GetBoundVertexArray, m_boundVertexArray) \
|
||||
X(GetBufferBindingSlot, m_bufferBindingSlot) \
|
||||
X(GetBufferBindingPoint, m_bufferBindingPointBase) \
|
||||
X(GetTouchedBufferBindingPointCount, m_touchedBindingPointCount) \
|
||||
X(GetClampReadColor, m_clampReadColor) \
|
||||
X(GetClearColor, m_clearColor) \
|
||||
X(GetClearDepth, m_clearDepth) \
|
||||
X(GetClearStencil, m_clearStencil) \
|
||||
X(GetColorMaskIndexed, m_colorMask) \
|
||||
X(GetCullFaceMode, m_cullFaceMode) \
|
||||
X(GetCurrentVertexAttribute, m_currentVertexAttribute) \
|
||||
X(GetDepthFunc, m_depthFunc) \
|
||||
X(GetDepthMask, m_depthMask) \
|
||||
X(GetDepthRangeIndexed, m_depthRange) \
|
||||
X(GetFramebufferBindingSlot, m_framebufferBindingSlot) \
|
||||
X(GetImageTextureBinding, m_imageTextureBindingBase) \
|
||||
X(GetLineWidth, m_lineWidth) \
|
||||
X(GetLogicOp, m_logicOp) \
|
||||
X(GetMaxTouchedTextureUnit, m_maxTouchedTextureUnit) \
|
||||
X(GetMinSampleShadingValue, m_minSampleShadingValue) \
|
||||
X(GetPatchDefaultInnerLevel, m_patchDefaultInnerLevel) \
|
||||
X(GetPatchDefaultOuterLevel, m_patchDefaultOuterLevel) \
|
||||
X(GetPatchVertices, m_patchVertices) \
|
||||
X(GetPipelineStateVersion, m_pipelineStateVersion) \
|
||||
X(GetPixelStoreParameters, m_pixelStore) \
|
||||
X(GetPolygonModeFront, m_polygonModeFront) \
|
||||
X(GetPolygonOffsetFactor, m_polygonOffsetFactor) \
|
||||
X(GetPolygonOffsetUnits, m_polygonOffsetUnits) \
|
||||
X(GetPrimitiveRestartIndex, m_primitiveRestartIndex) \
|
||||
X(GetProgramForDispatch, m_programForDispatch) \
|
||||
X(GetProgramForDraw, m_programForDraw) \
|
||||
X(GetProvokingVertexMode, m_provokingVertexMode) \
|
||||
X(GetRenderStateParameters, m_renderState) \
|
||||
X(GetRenderStateParametersVersion, m_renderStateParametersVersion) \
|
||||
X(GetSamplingResolutionGeneration, m_samplingResolutionGeneration) \
|
||||
X(GetScissorBox, m_scissorBox) \
|
||||
X(GetStencilState, m_stencil) \
|
||||
X(GetTextureBindGeneration, m_textureBindGeneration) \
|
||||
X(GetTextureContextId, m_textureContextId) \
|
||||
X(GetTextureUnitObject, m_textureUnitBase) \
|
||||
X(GetTransformFeedbackCapturedVertices, m_transformFeedbackCapturedVertices) \
|
||||
X(GetTransformFeedbackGeneration, m_transformFeedbackGeneration) \
|
||||
X(GetTransformFeedbackPausedPrimitiveCounter, m_transformFeedbackPausedPrimitiveCounter) \
|
||||
X(GetTransformFeedbackProgram, m_transformFeedbackProgram) \
|
||||
X(GetViewport, m_viewport) \
|
||||
X(GetViewportIndexed, m_viewportIndexed) \
|
||||
X(IsCapabilityEnabled, m_capability) \
|
||||
X(IsCapabilityEnabledIndexed, m_capabilityIndexed) \
|
||||
X(IsTransformFeedbackActive, m_transformFeedbackActive) \
|
||||
X(IsTransformFeedbackPaused, m_transformFeedbackPaused) \
|
||||
X(GetBoundTransformFeedbackLifetimeId, m_boundTransformFeedbackLifetimeId)
|
||||
// clang-format on
|
||||
|
||||
// The seven F-class fields, for the arithmetic below and for the sticky table's proof.
|
||||
// The forwarded set IS the sticky set (PipeFields.def marks the same seven rows F and
|
||||
// sticky), so an eighth sticky row without a forwarder is refused here, not by a test.
|
||||
inline constexpr SizeT kMGPipeForwardedFieldCount = 7;
|
||||
static_assert(kMGPipeForwardedFieldCount == kMGPipeInputStickyFieldCount,
|
||||
"the forwarded (F-class) fields and the sticky fields of PipeFields.def are the same seven rows");
|
||||
|
||||
// The block the backends read instead of GLContext (ARCHITECTURE.md 9.2 phase A, P1 brief
|
||||
// D4). One struct, three storage classes, and every accessor keeps the NAME, PARAMETERS
|
||||
// and RETURN TYPE of its GLContext counterpart (MG_State/GLState/Core.h) so the strangler
|
||||
// sed is type-neutral:
|
||||
//
|
||||
// V (value) copied out of GLContext at fill time by calling the same accessor;
|
||||
// no derivation logic is re-implemented here, which is what keeps the
|
||||
// copy semantically identical by construction.
|
||||
// O (object reference) a SharedPtr copy, or a raw pointer to the live GLContext-owned
|
||||
// slot/array for the accessors that return a non-const reference into
|
||||
// the context. Identity is what phase C turns into a handle.
|
||||
// F (forwarded) argument-keyed lookups and reverse-channel calls, defined out of
|
||||
// line in MG_Impl/Pipe/PipeFill.cpp (the client side, where the live
|
||||
// context may be spelled). Sticky: stamped once by the first fill that
|
||||
// sees a live context.
|
||||
//
|
||||
// Every non-forwarded accessor is MGP_INPUT_CHECK (poison) -> MGP_INPUT_VERIFY_READ
|
||||
// (compare-at-read) -> the storage. Both macros expand to nothing when their switch is
|
||||
// off, so a plain MOBILEGL_PIPE_PUSH build's accessor is a load.
|
||||
struct PipeInputs {
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
using BufferObject = MG_State::GLState::BufferObject;
|
||||
using BufferTarget = ::MobileGL::BufferTarget;
|
||||
using FramebufferObject = MG_State::GLState::FramebufferObject;
|
||||
using FramebufferTarget = ::MobileGL::FramebufferTarget;
|
||||
using VertexArrayObject = MG_State::GLState::VertexArrayObject;
|
||||
using ProgramObject = MG_State::GLState::ProgramObject;
|
||||
using ITextureObject = MG_State::GLState::ITextureObject;
|
||||
using TextureUnit = MG_State::GLState::TextureUnit;
|
||||
using ImageTextureBinding = MG_State::GLState::ImageTextureBinding;
|
||||
using CurrentVertexAttributeValue = MG_State::GLState::CurrentVertexAttributeValue;
|
||||
|
||||
static constexpr SizeT kBufferTargetCount = static_cast<SizeT>(BufferTarget::BufferTargetCount);
|
||||
static constexpr SizeT kFramebufferTargetCount = static_cast<SizeT>(FramebufferTarget::FramebufferTargetCount);
|
||||
static constexpr SizeT kCapabilityCount = static_cast<SizeT>(CapabilityInput::CapabilityInputCount);
|
||||
static constexpr SizeT kMaxViewports = RenderStateParameters::MAX_VIEWPORTS;
|
||||
static constexpr SizeT kMaxVertexAttribs = VertexArrayObject::MAX_VERTEX_ATTRIBS;
|
||||
static constexpr SizeT kStencilFaceCount = static_cast<SizeT>(StencilFace::StencilFaceCount);
|
||||
|
||||
// IsCapabilityEnabledIndexed's two indexed capabilities, the only ones GLContext keeps
|
||||
// indexed state for (RenderState::IsCapabilityEnabledIndexed).
|
||||
struct IndexedCapabilities {
|
||||
Bool Blend[kMGMaxDrawBuffers];
|
||||
Bool ScissorTest[kMaxViewports];
|
||||
};
|
||||
|
||||
// ---- identity / liveness (not fields) ----
|
||||
// Whether a live GLContext exists. Forwarded (PipeFill.cpp): under push MGB_CTX_LIVE
|
||||
// must be true as soon as a context exists, fill or no fill, which is what today's
|
||||
// null-context guards test.
|
||||
Bool IsLive() const;
|
||||
// The live GLContext's address at the last fill; serves MGB_CTX_IDENTITY.
|
||||
const void* ContextIdentity() const { return m_contextIdentity; }
|
||||
// The verb of the last fill, kVerbCount before the first one.
|
||||
MGPipeVerb CurrentVerb() const { return m_currentVerb; }
|
||||
#if MOBILEGL_PIPE_POISON
|
||||
const MGPipeFilledState& FilledState() const { return m_filled; }
|
||||
#endif
|
||||
|
||||
// ---- V: values ----
|
||||
Int GetActiveTextureUnit() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetActiveTextureUnit);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetActiveTextureUnit, 0, 0);
|
||||
return m_activeTextureUnit;
|
||||
}
|
||||
const FloatVec4& GetBlendColor() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBlendColor);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBlendColor, 0, 0);
|
||||
return m_blendColor;
|
||||
}
|
||||
void GetBlendEquationIndexed(Uint index, BlendEquation& color, BlendEquation& alpha) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBlendEquationIndexed);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBlendEquationIndexed, index, 0);
|
||||
if (index >= kMGMaxDrawBuffers) {
|
||||
MOBILEGL_ASSERT(false, "Blend equation index out of range: %u", index);
|
||||
return;
|
||||
}
|
||||
color = m_blendEquation[index][0];
|
||||
alpha = m_blendEquation[index][1];
|
||||
}
|
||||
void GetBlendFuncIndexed(Uint index, BlendFactor& srcRGB, BlendFactor& dstRGB, BlendFactor& srcAlpha,
|
||||
BlendFactor& dstAlpha) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBlendFuncIndexed);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBlendFuncIndexed, index, 0);
|
||||
if (index >= kMGMaxDrawBuffers) {
|
||||
MOBILEGL_ASSERT(false, "Blend func index out of range: %u", index);
|
||||
return;
|
||||
}
|
||||
srcRGB = m_blendFunc[index][0];
|
||||
dstRGB = m_blendFunc[index][1];
|
||||
srcAlpha = m_blendFunc[index][2];
|
||||
dstAlpha = m_blendFunc[index][3];
|
||||
}
|
||||
// Dead field: filled, read by no backend since the D21 XFB counter-slot rekey; kept so
|
||||
// the vendored inventory row keeps its mapping (Coverage.def).
|
||||
Uint GetBoundTransformFeedbackName() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBoundTransformFeedbackName);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBoundTransformFeedbackName, 0, 0);
|
||||
return m_boundTransformFeedbackName;
|
||||
}
|
||||
SizeT GetTouchedBufferBindingPointCount(BufferTarget target) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTouchedBufferBindingPointCount);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTouchedBufferBindingPointCount, static_cast<Uint>(target), 0);
|
||||
return m_touchedBindingPointCount[static_cast<SizeT>(target)];
|
||||
}
|
||||
GLenum GetClampReadColor() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetClampReadColor);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetClampReadColor, 0, 0);
|
||||
return m_clampReadColor;
|
||||
}
|
||||
const FloatVec4& GetClearColor() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetClearColor);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetClearColor, 0, 0);
|
||||
return m_clearColor;
|
||||
}
|
||||
Float GetClearDepth() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetClearDepth);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetClearDepth, 0, 0);
|
||||
return m_clearDepth;
|
||||
}
|
||||
Uint32 GetClearStencil() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetClearStencil);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetClearStencil, 0, 0);
|
||||
return m_clearStencil;
|
||||
}
|
||||
BoolVec4 GetColorMaskIndexed(Uint index) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetColorMaskIndexed);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetColorMaskIndexed, index, 0);
|
||||
return m_colorMask[index];
|
||||
}
|
||||
CullFaceMode GetCullFaceMode() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetCullFaceMode);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetCullFaceMode, 0, 0);
|
||||
return m_cullFaceMode;
|
||||
}
|
||||
const CurrentVertexAttributeValue& GetCurrentVertexAttribute(Uint index) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetCurrentVertexAttribute);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetCurrentVertexAttribute, index, 0);
|
||||
if (index >= kMaxVertexAttribs) {
|
||||
static const CurrentVertexAttributeValue defaultValue{};
|
||||
MGLOG_E_ONCE("PipeInputs::GetCurrentVertexAttribute: index %u is out of range", index);
|
||||
return defaultValue;
|
||||
}
|
||||
return m_currentVertexAttribute[index];
|
||||
}
|
||||
DepthTestFunc GetDepthFunc() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetDepthFunc);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetDepthFunc, 0, 0);
|
||||
return m_depthFunc;
|
||||
}
|
||||
Bool GetDepthMask() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetDepthMask);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetDepthMask, 0, 0);
|
||||
return m_depthMask;
|
||||
}
|
||||
const FloatVec2& GetDepthRangeIndexed(Uint index) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetDepthRangeIndexed);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetDepthRangeIndexed, index, 0);
|
||||
if (index >= kMaxViewports) {
|
||||
MOBILEGL_ASSERT(false, "Depth range index out of range: %u", index);
|
||||
return m_depthRange[0];
|
||||
}
|
||||
return m_depthRange[index];
|
||||
}
|
||||
Float GetLineWidth() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetLineWidth);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetLineWidth, 0, 0);
|
||||
return m_lineWidth;
|
||||
}
|
||||
LogicOperation GetLogicOp() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetLogicOp);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetLogicOp, 0, 0);
|
||||
return m_logicOp;
|
||||
}
|
||||
Int GetMaxTouchedTextureUnit() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetMaxTouchedTextureUnit);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetMaxTouchedTextureUnit, 0, 0);
|
||||
return m_maxTouchedTextureUnit;
|
||||
}
|
||||
Float GetMinSampleShadingValue() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetMinSampleShadingValue);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetMinSampleShadingValue, 0, 0);
|
||||
return m_minSampleShadingValue;
|
||||
}
|
||||
const FloatVec2& GetPatchDefaultInnerLevel() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPatchDefaultInnerLevel);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPatchDefaultInnerLevel, 0, 0);
|
||||
return m_patchDefaultInnerLevel;
|
||||
}
|
||||
const FloatVec4& GetPatchDefaultOuterLevel() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPatchDefaultOuterLevel);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPatchDefaultOuterLevel, 0, 0);
|
||||
return m_patchDefaultOuterLevel;
|
||||
}
|
||||
Uint GetPatchVertices() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPatchVertices);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPatchVertices, 0, 0);
|
||||
return m_patchVertices;
|
||||
}
|
||||
Uint GetPipelineStateVersion() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPipelineStateVersion);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPipelineStateVersion, 0, 0);
|
||||
return m_pipelineStateVersion;
|
||||
}
|
||||
Uint GetRenderStateParametersVersion() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetRenderStateParametersVersion);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetRenderStateParametersVersion, 0, 0);
|
||||
return m_renderStateParametersVersion;
|
||||
}
|
||||
PixelStoreParameters GetPixelStoreParameters(Bool isUnpack) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPixelStoreParameters);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPixelStoreParameters, isUnpack ? 1u : 0u, 0);
|
||||
return m_pixelStore[isUnpack ? 1 : 0];
|
||||
}
|
||||
GLenum GetPolygonModeFront() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPolygonModeFront);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPolygonModeFront, 0, 0);
|
||||
return m_polygonModeFront;
|
||||
}
|
||||
Float GetPolygonOffsetFactor() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPolygonOffsetFactor);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPolygonOffsetFactor, 0, 0);
|
||||
return m_polygonOffsetFactor;
|
||||
}
|
||||
Float GetPolygonOffsetUnits() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPolygonOffsetUnits);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPolygonOffsetUnits, 0, 0);
|
||||
return m_polygonOffsetUnits;
|
||||
}
|
||||
Uint32 GetPrimitiveRestartIndex() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetPrimitiveRestartIndex);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetPrimitiveRestartIndex, 0, 0);
|
||||
return m_primitiveRestartIndex;
|
||||
}
|
||||
ProvokingVertexMode GetProvokingVertexMode() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetProvokingVertexMode);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetProvokingVertexMode, 0, 0);
|
||||
return m_provokingVertexMode;
|
||||
}
|
||||
const RenderStateParameters& GetRenderStateParameters() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetRenderStateParameters);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetRenderStateParameters, 0, 0);
|
||||
return m_renderState;
|
||||
}
|
||||
Uint64 GetSamplingResolutionGeneration() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetSamplingResolutionGeneration);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetSamplingResolutionGeneration, 0, 0);
|
||||
return m_samplingResolutionGeneration;
|
||||
}
|
||||
const IntVec4& GetScissorBox() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetScissorBox);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetScissorBox, 0, 0);
|
||||
return m_scissorBox;
|
||||
}
|
||||
const StencilFaceState& GetStencilState(StencilFace face) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetStencilState);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetStencilState, static_cast<Uint>(face), 0);
|
||||
return m_stencil[face == StencilFace::Back ? 1 : 0];
|
||||
}
|
||||
Uint64 GetTextureBindGeneration() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTextureBindGeneration);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTextureBindGeneration, 0, 0);
|
||||
return m_textureBindGeneration;
|
||||
}
|
||||
Uint64 GetTextureContextId() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTextureContextId);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTextureContextId, 0, 0);
|
||||
return m_textureContextId;
|
||||
}
|
||||
Uint64 GetTransformFeedbackCapturedVertices() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTransformFeedbackCapturedVertices);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTransformFeedbackCapturedVertices, 0, 0);
|
||||
return m_transformFeedbackCapturedVertices;
|
||||
}
|
||||
Uint64 GetTransformFeedbackGeneration() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTransformFeedbackGeneration);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTransformFeedbackGeneration, 0, 0);
|
||||
return m_transformFeedbackGeneration;
|
||||
}
|
||||
Uint64 GetTransformFeedbackPausedPrimitiveCounter() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTransformFeedbackPausedPrimitiveCounter);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTransformFeedbackPausedPrimitiveCounter, 0, 0);
|
||||
return m_transformFeedbackPausedPrimitiveCounter;
|
||||
}
|
||||
Uint64 GetBoundTransformFeedbackLifetimeId() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBoundTransformFeedbackLifetimeId);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBoundTransformFeedbackLifetimeId, 0, 0);
|
||||
return m_boundTransformFeedbackLifetimeId;
|
||||
}
|
||||
IntVec4 GetViewport() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetViewport);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetViewport, 0, 0);
|
||||
return m_viewport;
|
||||
}
|
||||
const FloatVec4& GetViewportIndexed(Uint index) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetViewportIndexed);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetViewportIndexed, index, 0);
|
||||
if (index >= kMaxViewports) {
|
||||
MOBILEGL_ASSERT(false, "Viewport index out of range: %u", index);
|
||||
return m_viewportIndexed[0];
|
||||
}
|
||||
return m_viewportIndexed[index];
|
||||
}
|
||||
Bool IsCapabilityEnabled(CapabilityInput cap) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::IsCapabilityEnabled);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::IsCapabilityEnabled, static_cast<Uint>(cap), 0);
|
||||
const auto index = static_cast<SizeT>(cap);
|
||||
return index < kCapabilityCount ? m_capability[index] : false;
|
||||
}
|
||||
// Blend and ScissorTest are the only indexed capabilities GLContext keeps; no backend
|
||||
// asks for another (VulkanRenderer asks Blend). Any other cap is a read the fill cannot
|
||||
// have served: Fatal{UnmigratedPipeInput} naming the field and the verb, the cap in a
|
||||
// preceding MGLOG_E.
|
||||
Bool IsCapabilityEnabledIndexed(CapabilityInput cap, Uint index) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::IsCapabilityEnabledIndexed);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::IsCapabilityEnabledIndexed, static_cast<Uint>(cap), index);
|
||||
if (cap == CapabilityInput::Blend) {
|
||||
return index < kMGMaxDrawBuffers ? m_capabilityIndexed.Blend[index] : false;
|
||||
}
|
||||
if (cap == CapabilityInput::ScissorTest) {
|
||||
return index < kMaxViewports ? m_capabilityIndexed.ScissorTest[index] : false;
|
||||
}
|
||||
MGLOG_E("PipeInputs::IsCapabilityEnabledIndexed: no indexed storage for cap=%d (index=%u)",
|
||||
static_cast<int>(cap), index);
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::IsCapabilityEnabledIndexed, m_currentVerb);
|
||||
}
|
||||
Bool IsTransformFeedbackActive() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::IsTransformFeedbackActive);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::IsTransformFeedbackActive, 0, 0);
|
||||
return m_transformFeedbackActive;
|
||||
}
|
||||
Bool IsTransformFeedbackPaused() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::IsTransformFeedbackPaused);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::IsTransformFeedbackPaused, 0, 0);
|
||||
return m_transformFeedbackPaused;
|
||||
}
|
||||
|
||||
// ---- O: object references ----
|
||||
const SharedPtr<VertexArrayObject>& GetBoundVertexArray() {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBoundVertexArray);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBoundVertexArray, 0, 0);
|
||||
return m_boundVertexArray;
|
||||
}
|
||||
// A target the fill left null (one outside GlobalBufferTargets / BufferBindPointTargets,
|
||||
// or a read before any fill) is a read the fill cannot have served: the poison Fatal,
|
||||
// the target in a preceding MGLOG_E.
|
||||
BindingSlot<BufferObject>& GetBufferBindingSlot(BufferTarget target) {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBufferBindingSlot);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBufferBindingSlot, static_cast<Uint>(target), 0);
|
||||
const auto index = static_cast<SizeT>(target);
|
||||
if (index >= kBufferTargetCount || m_bufferBindingSlot[index] == nullptr) {
|
||||
MGLOG_E("PipeInputs::GetBufferBindingSlot: no slot for target=%d", static_cast<int>(target));
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::GetBufferBindingSlot, m_currentVerb);
|
||||
}
|
||||
return *m_bufferBindingSlot[index];
|
||||
}
|
||||
BindingSlotRange1D<BufferObject>& GetBufferBindingPoint(BufferTarget target, Uint index) {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetBufferBindingPoint);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetBufferBindingPoint, static_cast<Uint>(target), index);
|
||||
const auto targetIndex = static_cast<SizeT>(target);
|
||||
if (targetIndex >= kBufferTargetCount || m_bufferBindingPointBase[targetIndex] == nullptr) {
|
||||
MGLOG_E("PipeInputs::GetBufferBindingPoint: no binding points for target=%d (index=%u)",
|
||||
static_cast<int>(target), index);
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::GetBufferBindingPoint, m_currentVerb);
|
||||
}
|
||||
// The live storage is Array<Array<BindingSlotRange1D, BufferBindingPointCount>, N>
|
||||
// (BufferState.h), so base[index] is the live slot GLContext would hand out.
|
||||
return m_bufferBindingPointBase[targetIndex][index];
|
||||
}
|
||||
BindingSlot<FramebufferObject>& GetFramebufferBindingSlot(FramebufferTarget target) {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetFramebufferBindingSlot);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetFramebufferBindingSlot, static_cast<Uint>(target), 0);
|
||||
const auto index = static_cast<SizeT>(target);
|
||||
if (index >= kFramebufferTargetCount || m_framebufferBindingSlot[index] == nullptr) {
|
||||
MGLOG_E("PipeInputs::GetFramebufferBindingSlot: no slot for target=%d", static_cast<int>(target));
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::GetFramebufferBindingSlot, m_currentVerb);
|
||||
}
|
||||
return *m_framebufferBindingSlot[index];
|
||||
}
|
||||
ImageTextureBinding& GetImageTextureBinding(Int unit) {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetImageTextureBinding);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetImageTextureBinding, static_cast<Uint>(unit), 0);
|
||||
if (m_imageTextureBindingBase == nullptr) {
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::GetImageTextureBinding, m_currentVerb);
|
||||
}
|
||||
return m_imageTextureBindingBase[unit];
|
||||
}
|
||||
const ImageTextureBinding& GetImageTextureBinding(Int unit) const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetImageTextureBinding);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetImageTextureBinding, static_cast<Uint>(unit), 0);
|
||||
if (m_imageTextureBindingBase == nullptr) {
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::GetImageTextureBinding, m_currentVerb);
|
||||
}
|
||||
return m_imageTextureBindingBase[unit];
|
||||
}
|
||||
const SharedPtr<ProgramObject>& GetProgramForDispatch() {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetProgramForDispatch);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetProgramForDispatch, 0, 0);
|
||||
return m_programForDispatch;
|
||||
}
|
||||
const SharedPtr<ProgramObject>& GetProgramForDraw() {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetProgramForDraw);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetProgramForDraw, 0, 0);
|
||||
return m_programForDraw;
|
||||
}
|
||||
const SharedPtr<ProgramObject>& GetTransformFeedbackProgram() const {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTransformFeedbackProgram);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTransformFeedbackProgram, 0, 0);
|
||||
return m_transformFeedbackProgram;
|
||||
}
|
||||
TextureUnit& GetTextureUnitObject(Int unit) {
|
||||
MGP_INPUT_CHECK(MGPipeInputField::GetTextureUnitObject);
|
||||
MGP_INPUT_VERIFY_READ(MGPipeInputField::GetTextureUnitObject, static_cast<Uint>(unit), 0);
|
||||
if (m_textureUnitBase == nullptr) {
|
||||
MGPipeInputPoisonFatalForVerb(MGPipeInputField::GetTextureUnitObject, m_currentVerb);
|
||||
}
|
||||
return m_textureUnitBase[unit];
|
||||
}
|
||||
|
||||
// ---- F: forwarded to the live context (MG_Impl/Pipe/PipeFill.cpp); sticky ----
|
||||
// Each takes an argument that is not verb state - a GL name, a lifetime id, a target -
|
||||
// i.e. it is a lookup or a reverse-channel write, not a state read; there is no value
|
||||
// the filler could copy and no verb whose fill could make it stale. Phase C replaces
|
||||
// them with handle tables and callbacks.
|
||||
// They carry no MGP_INPUT_CHECK / MGP_INPUT_VERIFY_READ (the declared exception to
|
||||
// P1 brief D4's "every accessor body"): a forward is a live call, not a stored value,
|
||||
// and InvalidateCompileEnv is reached from backend initialisation before any verb has
|
||||
// filled, where a check would be Fatal{...@<none>} on every start. Their sticky stamp
|
||||
// is therefore consulted by no accessor; the tests pin it through
|
||||
// MGPipeInputFieldIsFresh directly.
|
||||
SizeT GetBufferBindingPointCount(BufferTarget target) const;
|
||||
const SharedPtr<ProgramObject>& GetProgramObject(Uint index);
|
||||
const SharedPtr<ITextureObject>& GetTextureObject(Uint index);
|
||||
Bool HasOpenTransformFeedbackSpan(Uint64 lifetimeId) const;
|
||||
void InvalidateCompileEnv();
|
||||
Bool ValidateProgramName(Uint index) const;
|
||||
// Dropped with an MGLOG_E_ONCE when no context is live; today's guarded sites never
|
||||
// reach it without one.
|
||||
void RecordError(ErrorCode code, UniquePtr<ErrorInfo> info);
|
||||
|
||||
// ---- the storage visitor ----
|
||||
// Calls fn(a.<member>, b.<member>) for the field's storage and returns its result; returns
|
||||
// false without calling fn for a forwarded field, which has none. The comparator's
|
||||
// per-field equality and the verify corruption injector are both one call of this.
|
||||
template <class Fn>
|
||||
static Bool VisitStorage(MGPipeInputField field, PipeInputs& a, PipeInputs& b, Fn&& fn) {
|
||||
switch (field) {
|
||||
#define MGP_INPUT_VISIT(Field, Member) \
|
||||
case MGPipeInputField::Field: \
|
||||
return fn(a.Member, b.Member);
|
||||
MGP_INPUT_STORAGE_LIST(MGP_INPUT_VISIT)
|
||||
#undef MGP_INPUT_VISIT
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
template <class Fn>
|
||||
static Bool VisitStorage(MGPipeInputField field, const PipeInputs& a, const PipeInputs& b, Fn&& fn) {
|
||||
switch (field) {
|
||||
#define MGP_INPUT_VISIT(Field, Member) \
|
||||
case MGPipeInputField::Field: \
|
||||
return fn(a.Member, b.Member);
|
||||
MGP_INPUT_STORAGE_LIST(MGP_INPUT_VISIT)
|
||||
#undef MGP_INPUT_VISIT
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// The one door into the storage from the client side (MG_Impl/Pipe/PipeFill.cpp):
|
||||
// the filler's per-field copies and stamps, and the verify snapshot.
|
||||
friend struct MGPipeFillAccess;
|
||||
// The other door, and the one that exists because of what this block IS after P2:
|
||||
// the server's working RenderStateParameters. MG_Pipe/PipeApply.cpp scatters
|
||||
// bind_render_state's and set_dynamic_state's chunks straight into m_renderState,
|
||||
// which is why DirectGLES' SyncRenderState is not one line changed. It deliberately
|
||||
// does NOT stamp the poison generations - a stamp says "the filler published this
|
||||
// for THIS verb", which is the walk's statement, not the applier's.
|
||||
friend struct MGPipeApplyAccess;
|
||||
|
||||
// ---- identity ----
|
||||
const void* m_contextIdentity = nullptr;
|
||||
Bool m_live = false;
|
||||
MGPipeVerb m_currentVerb = MGPipeVerb::kVerbCount;
|
||||
#if MOBILEGL_PIPE_POISON
|
||||
MGPipeFilledState m_filled{};
|
||||
#endif
|
||||
|
||||
// ---- V ----
|
||||
Int m_activeTextureUnit = 0;
|
||||
FloatVec4 m_blendColor{};
|
||||
BlendEquation m_blendEquation[kMGMaxDrawBuffers][2]{};
|
||||
BlendFactor m_blendFunc[kMGMaxDrawBuffers][4]{};
|
||||
Uint m_boundTransformFeedbackName = 0;
|
||||
SizeT m_touchedBindingPointCount[kBufferTargetCount]{};
|
||||
GLenum m_clampReadColor = 0;
|
||||
FloatVec4 m_clearColor{};
|
||||
Float m_clearDepth = 0.f;
|
||||
Uint32 m_clearStencil = 0;
|
||||
BoolVec4 m_colorMask[kMGMaxDrawBuffers]{};
|
||||
CullFaceMode m_cullFaceMode{};
|
||||
CurrentVertexAttributeValue m_currentVertexAttribute[kMaxVertexAttribs]{};
|
||||
DepthTestFunc m_depthFunc{};
|
||||
Bool m_depthMask = false;
|
||||
FloatVec2 m_depthRange[kMaxViewports]{};
|
||||
Float m_lineWidth = 0.f;
|
||||
LogicOperation m_logicOp{};
|
||||
Int m_maxTouchedTextureUnit = -1;
|
||||
Float m_minSampleShadingValue = 0.f;
|
||||
FloatVec2 m_patchDefaultInnerLevel{};
|
||||
FloatVec4 m_patchDefaultOuterLevel{};
|
||||
Uint m_patchVertices = 0;
|
||||
Uint m_pipelineStateVersion = 0;
|
||||
Uint m_renderStateParametersVersion = 0;
|
||||
PixelStoreParameters m_pixelStore[2]{}; // [0] = pack, [1] = unpack
|
||||
GLenum m_polygonModeFront = 0;
|
||||
Float m_polygonOffsetFactor = 0.f;
|
||||
Float m_polygonOffsetUnits = 0.f;
|
||||
Uint32 m_primitiveRestartIndex = 0;
|
||||
ProvokingVertexMode m_provokingVertexMode{};
|
||||
RenderStateParameters m_renderState{};
|
||||
Uint64 m_samplingResolutionGeneration = 0;
|
||||
Uint64 m_textureBindGeneration = 0;
|
||||
Uint64 m_textureContextId = 0;
|
||||
IntVec4 m_scissorBox{};
|
||||
StencilFaceState m_stencil[kStencilFaceCount]{};
|
||||
Uint64 m_transformFeedbackCapturedVertices = 0;
|
||||
Uint64 m_transformFeedbackGeneration = 0;
|
||||
Uint64 m_transformFeedbackPausedPrimitiveCounter = 0;
|
||||
Uint64 m_boundTransformFeedbackLifetimeId = 0;
|
||||
IntVec4 m_viewport{};
|
||||
FloatVec4 m_viewportIndexed[kMaxViewports]{};
|
||||
Bool m_capability[kCapabilityCount]{};
|
||||
IndexedCapabilities m_capabilityIndexed{};
|
||||
Bool m_transformFeedbackActive = false;
|
||||
Bool m_transformFeedbackPaused = false;
|
||||
|
||||
// ---- O ----
|
||||
SharedPtr<VertexArrayObject> m_boundVertexArray;
|
||||
BindingSlot<BufferObject>* m_bufferBindingSlot[kBufferTargetCount]{};
|
||||
BindingSlotRange1D<BufferObject>* m_bufferBindingPointBase[kBufferTargetCount]{};
|
||||
BindingSlot<FramebufferObject>* m_framebufferBindingSlot[kFramebufferTargetCount]{};
|
||||
ImageTextureBinding* m_imageTextureBindingBase = nullptr;
|
||||
SharedPtr<ProgramObject> m_programForDispatch;
|
||||
SharedPtr<ProgramObject> m_programForDraw;
|
||||
SharedPtr<ProgramObject> m_transformFeedbackProgram;
|
||||
TextureUnit* m_textureUnitBase = nullptr;
|
||||
};
|
||||
|
||||
// The single global the backends read through MGB_CTX (ARCHITECTURE.md 9.2). An inline
|
||||
// variable: no .cpp is needed for the definition.
|
||||
//
|
||||
// LEAK-AT-EXIT STORAGE, and it is the same rule Init.cpp and GlobalObjects.cpp state for
|
||||
// pGLContext and pActiveBackendObject: "a process that exits without eglTerminate simply
|
||||
// leaks the global singletons to the OS instead of running destructors during static
|
||||
// teardown". This block breaks that rule if it is a value, because its O-class members
|
||||
// are SharedPtrs to FRONTEND objects: a VertexArrayObject that the application deleted
|
||||
// while it was bound has its last reference here, and destroying this block from
|
||||
// __run_exit_handlers therefore runs ~VertexArrayObject -> ~BufferObject at exit. Those
|
||||
// destructors are not exit-safe and cannot be made so - they reach the client's slot
|
||||
// allocator, the resource tracker, the vertex-input emitter, the applier AND, through
|
||||
// MGPipeApplyResourceDestroy, the backend's own twin tables, deferred-release queue,
|
||||
// buffer pool and driver entry points, every one of which is either already destroyed or
|
||||
// about to be. So the reference is never dropped: nothing here can start such a chain.
|
||||
// A live context releases these SharedPtrs the ordinary way, at the fill point.
|
||||
// (P3a; the exit-time heap corruption this closes is p3a-results/exit-order-v1.md.)
|
||||
inline PipeInputs& gPipeInputs = *new PipeInputs();
|
||||
|
||||
// Every field has storage or is forwarded, and nothing else.
|
||||
#define MGP_INPUT_COUNT_ONE(Field, Member) +1
|
||||
static_assert(0 MGP_INPUT_STORAGE_LIST(MGP_INPUT_COUNT_ONE) + kMGPipeForwardedFieldCount == kMGPipeInputFieldCount,
|
||||
"MGP_INPUT_STORAGE_LIST plus the seven forwarded fields is not the PipeInputs field set");
|
||||
#undef MGP_INPUT_COUNT_ONE
|
||||
// The docs budget ~20 KB; the block is a few KB.
|
||||
static_assert(sizeof(PipeInputs) < 20 * 1024, "PipeInputs outgrew its budget");
|
||||
|
||||
#if MOBILEGL_PIPE_VERIFY
|
||||
// PipeInputs.cpp. Per-field equality for the entry compare (P1 brief D8): V by value
|
||||
// through G4's MGPipeFieldEqual (bitwise floats, field-wise structs), O by identity, F
|
||||
// always equal (no storage).
|
||||
Bool MGPipeInputsFieldEqual(MGPipeInputField field, const PipeInputs& a, const PipeInputs& b);
|
||||
// PipeInputs.cpp. The entry compare: every field in `mask` of the pushed block against the
|
||||
// snapshot, first differing field out. Exported from the shared library on purpose - the
|
||||
// retrace-verify CI job proves it swapped in a verify build by finding this symbol with
|
||||
// nm -D, so a "green" run against a library without the comparator cannot happen.
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
__attribute__((visibility("default")))
|
||||
#endif
|
||||
Bool MGPipeVerifyInputs(const PipeInputs& pushed, const PipeInputs& snapshot, const MGPipeFieldMask& mask,
|
||||
MGPipeInputField* outField);
|
||||
// PipeInputs.cpp. Negative control A: perturbs one field's storage (flip a Bool, +1 a
|
||||
// scalar, ^0x5A the first byte of a struct, flip a pointer's low bits - never
|
||||
// dereferenced, the snapshot is only ever compared). Returns false for a forwarded field,
|
||||
// which has nothing to corrupt.
|
||||
Bool MGPipeApplyVerifyCorruption(PipeInputs& snapshot, MGPipeInputField field);
|
||||
#endif
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
@@ -11,5 +11,51 @@ endif()
|
||||
add_executable(DriverBench DriverBench.c)
|
||||
target_link_libraries(DriverBench PRIVATE dl)
|
||||
|
||||
# WHY EVERY ENTRY HERE CARRIES A PASS_REGULAR_EXPRESSION.
|
||||
#
|
||||
# DriverBench prints one CSV row per case it ran and exits 0 whatever it ran. Before this, a ctest
|
||||
# entry naming a case therefore could not answer the only question it exists to ask: an argument
|
||||
# matching nothing in kBenchCases selected no case, printed only the header row, and still exited
|
||||
# 0. DriverBench.c now refuses an unknown case name (exit 2), which closes it at the source - but
|
||||
# the entry must be able to go red for the reason it exists WITHOUT depending on that check
|
||||
# staying in the binary, so each entry also requires the case's own output row to appear.
|
||||
#
|
||||
# The regex is what a healthy run of that case prints and nothing else does: the case name at the
|
||||
# start of a line, then the frames / ops-per-frame / median-ms / ns-per-op / fps columns
|
||||
# (run_case()). A rename, a drop from kBenchCases, a boot_egl() failure or
|
||||
# a crash part-way through the case all remove that row and turn the entry red.
|
||||
#
|
||||
# Note that a PASS_REGULAR_EXPRESSION makes ctest ignore the process exit code (cmCTestRunTest:
|
||||
# success is `retVal == 0 || !RequiredRegularExpressions.empty()`), which is why the row itself
|
||||
# has to be the evidence rather than a companion to the rc.
|
||||
add_test(NAME DriverBench COMMAND DriverBench draw_tiny)
|
||||
set_tests_properties(DriverBench PROPERTIES LABELS benchmark)
|
||||
# draw_tiny's a/ops scale with $DRIVERBENCH_DRAWS (main()), so only the shape of
|
||||
# the row is pinned here, not the column values.
|
||||
set_tests_properties(DriverBench PROPERTIES
|
||||
LABELS benchmark
|
||||
PASS_REGULAR_EXPRESSION "(^|\n)draw_tiny,[0-9]+,[0-9]+,[0-9.]+,[0-9.]+,[0-9.]+")
|
||||
|
||||
# The Blaze3D blend toggle, as its own entry.
|
||||
#
|
||||
# mc_state_toggle is glEnable(GL_BLEND) / glBlendFuncSeparate / glDrawElements /
|
||||
# glDisable(GL_BLEND) / glDrawElements, 46 times - the measured vanilla-frame rate, and the exact
|
||||
# shape ROADMAP.md writes down as the microbenchmark P2 owes the GO/NO-GO. It is the workload the
|
||||
# whole "push at validate, not in the setter" decision was made for: a per-setter design pays for
|
||||
# every toggle, and a CSO that is minted twice and then reused pays for none of them.
|
||||
#
|
||||
# The case has existed in kBenchCases since P0 and nothing ran it, so nothing noticed if it broke.
|
||||
# Exposing it costs about 1.2 s inside an existing three-minute job, and it means the number the
|
||||
# P2 report quotes comes from a case CI has been executing all along rather than from a code path
|
||||
# whose first run is the day it is measured.
|
||||
#
|
||||
# Like the entry above, this runs against whatever $DRIVERBENCH_EGL_LIB names (the system driver
|
||||
# when unset) - the ctest entry is a "does this case still run" gate, not the measurement. The
|
||||
# measurement is run_driver_bench.sh against each of {native, espryt, magma}.
|
||||
add_test(NAME DriverBenchStateToggle COMMAND DriverBench mc_state_toggle)
|
||||
# The ops-per-frame column is pinned to 46 here, unlike the entry above: the mc_* cases are
|
||||
# excluded from the $DRIVERBENCH_DRAWS scaling on purpose ("the mc_* rates are measured and must
|
||||
# not move, or the numbers stop being comparable", main()), so 46 toggles per frame
|
||||
# is part of what "this case still runs" means. Change the workload and this entry says so.
|
||||
set_tests_properties(DriverBenchStateToggle PROPERTIES
|
||||
LABELS benchmark
|
||||
PASS_REGULAR_EXPRESSION "(^|\n)mc_state_toggle,[0-9]+,46,[0-9.]+,[0-9.]+,[0-9.]+")
|
||||
|
||||
@@ -476,6 +476,28 @@ int main(int argc, char** argv) {
|
||||
if (getenv("DRIVERBENCH_FRAMES")) g_frames = atoi(getenv("DRIVERBENCH_FRAMES"));
|
||||
if (getenv("DRIVERBENCH_SPRITES")) g_mixSprites = atol(getenv("DRIVERBENCH_SPRITES"));
|
||||
|
||||
/* A requested case name that matches nothing used to select nothing, print the header row and
|
||||
* exit 0 - so a caller that names a case (run_driver_bench.sh, and the two ctest entries in
|
||||
* CMakeLists.txt) could not tell "the case ran" from "the case has been renamed or deleted".
|
||||
* Refuse it here, before any GL work, so the refusal reaches a caller that has no display
|
||||
* either, and name what does exist so the fix is obvious. */
|
||||
int unknownCases = 0;
|
||||
for (int j = 1; j < argc; ++j) {
|
||||
int known = 0;
|
||||
for (int i = 0; i < kBenchCaseCount; ++i)
|
||||
if (strcmp(argv[j], kBenchCases[i].name) == 0) known = 1;
|
||||
if (!known) {
|
||||
fprintf(stderr, "DriverBench: no case named '%s'\n", argv[j]);
|
||||
unknownCases = 1;
|
||||
}
|
||||
}
|
||||
if (unknownCases) {
|
||||
fprintf(stderr, "DriverBench: the %d cases in kBenchCases are:\n", kBenchCaseCount);
|
||||
for (int i = 0; i < kBenchCaseCount; ++i)
|
||||
fprintf(stderr, " %s\n", kBenchCases[i].name);
|
||||
return 2;
|
||||
}
|
||||
|
||||
if (boot_egl()) return 1;
|
||||
build_resources();
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Impl/Pipe/PipeFill.h>
|
||||
#include "../Getter/GL_Getter.h"
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
@@ -34,8 +35,81 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
static Bool ValidateCurrentProgramForExecution(const char* functionName) {
|
||||
return ValidateProgramForExecution(MG_State::pGLContext->GetProgramForDraw(), functionName);
|
||||
// Takes the ALREADY-RESOLVED draw program rather than looking it up: GLContext::GetProgramForDraw
|
||||
// is not a plain getter (it settles the program's link and SPIR-V jobs so every version a
|
||||
// backend samples during this draw describes the program it is drawing), so the draw funnel
|
||||
// below resolves it exactly once and hands it to both users.
|
||||
static Bool ValidateResolvedProgramForDraw(const SharedPtr<MG_State::GLState::ProgramObject>& currentProgram,
|
||||
const char* functionName) {
|
||||
// "If there is no current program object or bound program pipeline object, the results of
|
||||
// a draw are UNDEFINED" - and undefined is not an error (GL 4.6 core 7.3, ES 3.1 7.3).
|
||||
// The draw is dropped, silently, which is one of the shapes "undefined" is allowed to
|
||||
// take; recording INVALID_OPERATION here is not, and es31cSeparateShaderObjsTests'
|
||||
// StateInteraction reads exactly that error back after useProgram(0) + bindProgramPipeline(0).
|
||||
// A DISPATCH is the opposite rule ("INVALID_OPERATION if there is no active program for
|
||||
// the compute shader stage"), which is why this lives on the draw path and not in the
|
||||
// shared ValidateProgramForExecution below.
|
||||
if (!currentProgram) return false;
|
||||
if (!ValidateProgramForExecution(currentProgram, functionName)) return false;
|
||||
|
||||
// GL 4.6 core 7.4.1, the pipeline validation rule every vertex-transferring command
|
||||
// inherits: it is an INVALID_OPERATION when a tessellation control, tessellation
|
||||
// evaluation or geometry stage has an executable but no program supplies an executable
|
||||
// VERTEX shader. A non-separable program cannot reach this - the link rule forbids the
|
||||
// shape - so in practice it catches a program pipeline assembled out of stage programs,
|
||||
// which today draws happily and renders nothing.
|
||||
//
|
||||
// Asked of the EXECUTABLE, like the compute check below: for a pipeline the resolved
|
||||
// program is the graphics composite, whose linked-shader snapshot is built out of exactly
|
||||
// the pipeline's own graphics stage programs (GLContext::GetProgramForDraw), and the only
|
||||
// stage compositing ever invents is a default FRAGMENT shader. A fragment-only pipeline is
|
||||
// deliberately NOT rejected: the rule above names the three pre-rasterization stages, and
|
||||
// nothing else here should start refusing draws GL accepts.
|
||||
//
|
||||
// On the DRAW path only, never in ValidateProgramForExecution itself, so a dispatch -
|
||||
// which shares that helper and legitimately has no vertex stage - is untouched.
|
||||
const Bool hasPreRasterizationStage = currentProgram->HasLinkedShaderStage(ShaderStage::Geometry) ||
|
||||
currentProgram->HasLinkedShaderStage(ShaderStage::TessControl) ||
|
||||
currentProgram->HasLinkedShaderStage(ShaderStage::TessEval);
|
||||
if (hasPreRasterizationStage && !currentProgram->HasLinkedShaderStage(ShaderStage::Vertex)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", functionName,
|
||||
"The program in use runs a geometry or tessellation stage but has no vertex shader stage."));
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// gl_NumSamples has no SPIR-V built-in, so the source pipeline lowers it onto a reserved
|
||||
// default-block uniform (see InjectNumSamplesBuiltinShim). This is where that uniform is paid
|
||||
// for: the value is a property of the DRAW FRAMEBUFFER, not of the program, so one program
|
||||
// drawn into a 4x target and then into the default framebuffer must see 4 and then 1 - which
|
||||
// rules out baking it at link time.
|
||||
//
|
||||
// Per draw rather than on framebuffer changes because the pair (program, framebuffer) is what
|
||||
// decides the value and either half can move between draws. It costs a phase-A flag read for
|
||||
// every program that has no shim, and a 4-byte compare for the ones that do: the write only
|
||||
// bumps the UBO content version when the number actually changes, so a run of draws into one
|
||||
// framebuffer re-uploads nothing.
|
||||
static void PublishDrawFramebufferSampleCount(const SharedPtr<MG_State::GLState::ProgramObject>& program) {
|
||||
if (!program || !program->UsesReservedNumSamples()) return;
|
||||
// GL 4.6 core 15.2.2: gl_NumSamples is the number of samples in the framebuffer, or ONE
|
||||
// when the target is not multisampled - where glGetIntegerv(GL_SAMPLES) answers zero.
|
||||
program->WriteReservedNumSamples(static_cast<Int>(std::max<GLint>(ResolveDrawFramebufferSampleCount(), 1)));
|
||||
}
|
||||
|
||||
// The one funnel every drawing command passes through. Order is load-bearing: validate first
|
||||
// (a rejected draw must leave state alone), then publish the sample count - which reads the
|
||||
// DRAW FRAMEBUFFER binding, so it has to run after the caller's framebuffer state is settled
|
||||
// and before the backend consumes the program's UBO content version.
|
||||
static Bool PrepareCurrentProgramForDraw(const char* functionName) {
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDraw();
|
||||
if (!ValidateResolvedProgramForDraw(currentProgram, functionName)) return false;
|
||||
PublishDrawFramebufferSampleCount(currentProgram);
|
||||
return true;
|
||||
}
|
||||
|
||||
// A dispatch resolves its program through the DISPATCH accessor: with a pipeline bound
|
||||
@@ -73,6 +147,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_TRIANGLES: return static_cast<Uint64>(count / 3);
|
||||
case GL_TRIANGLE_STRIP:
|
||||
case GL_TRIANGLE_FAN: return count >= 3 ? static_cast<Uint64>(count - 2) : 0;
|
||||
// Adjacency primitives (GL 4.6 core table 10.1). Only a geometry stage can consume
|
||||
// them, and it is the ADJACENT-free primitive count that reaches it: 4 vertices per
|
||||
// line, 6 per triangle, one per step for the strips. Answering 0 here - which is what
|
||||
// the default arm did - made AccountTransformFeedbackPrimitives bail before it had
|
||||
// recorded anything, so an adjacency capture advanced neither the captured-vertex
|
||||
// counter the scattered-capture path is bounded by nor the geometry-capture-draw flag
|
||||
// that routes the transform feedback queries to the driver's own counter.
|
||||
case GL_LINES_ADJACENCY: return static_cast<Uint64>(count / 4);
|
||||
case GL_LINE_STRIP_ADJACENCY: return count >= 4 ? static_cast<Uint64>(count - 3) : 0;
|
||||
case GL_TRIANGLES_ADJACENCY: return static_cast<Uint64>(count / 6);
|
||||
case GL_TRIANGLE_STRIP_ADJACENCY: return count >= 6 ? static_cast<Uint64>((count - 4) / 2) : 0;
|
||||
// GL_PATCHES is deliberately absent: the tessellator's amplification is not knowable
|
||||
// on the CPU, and answering 0 is what defers GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN
|
||||
// to the driver's own counter, which is the only correct source for a patch capture.
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
@@ -99,11 +187,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_LINES:
|
||||
case GL_LINE_STRIP:
|
||||
case GL_LINE_LOOP:
|
||||
// An adjacency primitive delivers the same line/triangle to the geometry stage; the
|
||||
// adjacent vertices are context, not part of the primitive.
|
||||
case GL_LINES_ADJACENCY:
|
||||
case GL_LINE_STRIP_ADJACENCY:
|
||||
verticesPerPrimitive = 2;
|
||||
break;
|
||||
case GL_TRIANGLES:
|
||||
case GL_TRIANGLE_STRIP:
|
||||
case GL_TRIANGLE_FAN:
|
||||
case GL_TRIANGLES_ADJACENCY:
|
||||
case GL_TRIANGLE_STRIP_ADJACENCY:
|
||||
verticesPerPrimitive = 3;
|
||||
break;
|
||||
default:
|
||||
@@ -308,11 +402,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_POINTS:
|
||||
compatible = mode == GL_POINTS;
|
||||
break;
|
||||
// The adjacency modes belong here too (GL 4.6 core table 13.1, ES 3.2 table 12.1).
|
||||
// This arm is only reached when the program has NO geometry or tessellation
|
||||
// evaluation stage, and without a geometry stage the adjacent vertices are simply
|
||||
// ignored (GL 4.6 core 10.1) - the primitive assembled IS a plain line or triangle,
|
||||
// so the combination is legal and must capture. Omitting them raised a spurious
|
||||
// GL_INVALID_OPERATION and dropped the draw entirely, leaving the capture buffer
|
||||
// with its pre-draw bytes. The geometry-stage input table above already carries the
|
||||
// same four arms; this is the second table catching up with it.
|
||||
case GL_LINES:
|
||||
compatible = mode == GL_LINES || mode == GL_LINE_STRIP || mode == GL_LINE_LOOP;
|
||||
compatible = mode == GL_LINES || mode == GL_LINE_STRIP || mode == GL_LINE_LOOP ||
|
||||
mode == GL_LINES_ADJACENCY || mode == GL_LINE_STRIP_ADJACENCY;
|
||||
break;
|
||||
case GL_TRIANGLES:
|
||||
compatible = mode == GL_TRIANGLES || mode == GL_TRIANGLE_STRIP || mode == GL_TRIANGLE_FAN;
|
||||
compatible = mode == GL_TRIANGLES || mode == GL_TRIANGLE_STRIP || mode == GL_TRIANGLE_FAN ||
|
||||
mode == GL_TRIANGLES_ADJACENCY || mode == GL_TRIANGLE_STRIP_ADJACENCY;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
@@ -424,6 +528,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(Clear);
|
||||
MG_Backend::gBackendFunctionsTable.GL.Clear(mask);
|
||||
}
|
||||
|
||||
@@ -432,6 +537,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawElements);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElements(mode, count, type, indices);
|
||||
}
|
||||
|
||||
@@ -441,6 +547,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawElements);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElements(mode, count, type, indices, drawcount);
|
||||
}
|
||||
|
||||
@@ -450,6 +557,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawElementsBaseVertex);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsBaseVertex(mode, count, type, indices, drawcount,
|
||||
basevertex);
|
||||
}
|
||||
@@ -459,6 +567,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawArrays);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArrays(mode, first, count);
|
||||
}
|
||||
|
||||
@@ -467,6 +576,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawArrays);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawArrays(mode, first, count, drawcount);
|
||||
}
|
||||
|
||||
@@ -476,6 +586,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawElementsBaseVertex);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsBaseVertex(mode, count, type, indices, basevertex);
|
||||
}
|
||||
|
||||
@@ -485,6 +596,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawElementsIndirect);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsIndirect(mode, type, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
@@ -493,6 +605,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawArraysIndirect);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawArraysIndirect(mode, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
@@ -502,6 +615,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawElementsIndirectCount);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsIndirectCount(mode, type, indirect, drawcount,
|
||||
maxdrawcount, stride);
|
||||
}
|
||||
@@ -512,6 +626,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(MultiDrawArraysIndirectCount);
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawArraysIndirectCount(mode, indirect, drawcount, maxdrawcount,
|
||||
stride);
|
||||
}
|
||||
@@ -522,6 +637,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawRangeElementsBaseVertex);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawRangeElementsBaseVertex(mode, start, end, count, type, indices,
|
||||
basevertex);
|
||||
}
|
||||
@@ -532,6 +648,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawRangeElements);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawRangeElements(mode, start, end, count, type, indices);
|
||||
}
|
||||
|
||||
@@ -542,6 +659,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_SET_BASE_INSTANCE(baseinstance);
|
||||
MGP_FILL(DrawElementsInstancedBaseVertexBaseInstance);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstancedBaseVertexBaseInstance(
|
||||
mode, count, type, indices, instancecount, basevertex, baseinstance);
|
||||
}
|
||||
@@ -552,6 +671,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawElementsInstancedBaseVertex);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstancedBaseVertex(mode, count, type, indices, instancecount,
|
||||
basevertex);
|
||||
}
|
||||
@@ -562,6 +682,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_SET_BASE_INSTANCE(baseinstance);
|
||||
MGP_FILL(DrawElementsInstancedBaseInstance);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstancedBaseInstance(mode, count, type, indices,
|
||||
instancecount, baseinstance);
|
||||
}
|
||||
@@ -572,6 +694,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawElementsInstanced);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstanced(mode, count, type, indices, instancecount);
|
||||
}
|
||||
|
||||
@@ -580,6 +703,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawElementsIndirect);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsIndirect(mode, type, indirect);
|
||||
}
|
||||
void DrawArraysInstancedBaseInstance_Backend(GLenum mode, GLint first, GLsizei count, GLsizei instancecount,
|
||||
@@ -588,6 +712,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_SET_BASE_INSTANCE(baseinstance);
|
||||
MGP_FILL(DrawArraysInstancedBaseInstance);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArraysInstancedBaseInstance(mode, first, count, instancecount,
|
||||
baseinstance);
|
||||
}
|
||||
@@ -597,6 +723,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawArraysInstanced);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArraysInstanced(mode, first, count, instancecount);
|
||||
}
|
||||
|
||||
@@ -605,6 +732,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DrawArraysIndirect);
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArraysIndirect(mode, indirect);
|
||||
}
|
||||
|
||||
@@ -636,6 +764,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// GL 4.3 added both dispatches to the conditional-render set (GL 4.6 core 10.9), which is
|
||||
// exactly what KHR-GL43.compute_shader.conditional-dispatching checks.
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DispatchCompute);
|
||||
dispatchCompute(numGroupsX, numGroupsY, numGroupsZ);
|
||||
}
|
||||
|
||||
@@ -688,6 +817,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
if (!ValidateCurrentProgramForCompute(__func__)) return;
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MGP_FILL(DispatchComputeIndirect);
|
||||
dispatchComputeIndirect(indirect);
|
||||
}
|
||||
|
||||
@@ -709,10 +839,42 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
MG_State::pGLContext->SetPatchVertices(static_cast<Uint>(value));
|
||||
if (const auto patchParameteri = MG_Backend::gBackendFunctionsTable.GL.PatchParameteri) {
|
||||
MGP_FILL(PatchParameteri);
|
||||
patchParameteri(pname, value);
|
||||
}
|
||||
}
|
||||
|
||||
// GL 4.6 core 11.2.2. The default tessellation levels a program with an evaluation stage and
|
||||
// NO control stage tessellates at; both backends have to synthesize that control stage
|
||||
// themselves (ES 3.2 and Vulkan both require one), and they compile these numbers into it, so
|
||||
// there is no backend entry point to forward to - ES has none at all. INVALID_ENUM on a bad
|
||||
// pname is the only error the spec lists: any float values are accepted, negatives and NaN
|
||||
// included, and it is the tessellator that clamps them.
|
||||
//
|
||||
// This used to be a stub, which is why the two synthesizers hardcoded 1.0.
|
||||
void PatchParameterfv(GLenum pname, const GLfloat* values) {
|
||||
if (pname != GL_PATCH_DEFAULT_OUTER_LEVEL && pname != GL_PATCH_DEFAULT_INNER_LEVEL) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
"pname must be GL_PATCH_DEFAULT_OUTER_LEVEL or GL_PATCH_DEFAULT_INNER_LEVEL."));
|
||||
return;
|
||||
}
|
||||
if (!values) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "values pointer cannot be null"));
|
||||
return;
|
||||
}
|
||||
if (pname == GL_PATCH_DEFAULT_OUTER_LEVEL) {
|
||||
MG_State::pGLContext->SetPatchDefaultOuterLevel(
|
||||
FloatVec4(values[0], values[1], values[2], values[3]));
|
||||
} else {
|
||||
MG_State::pGLContext->SetPatchDefaultInnerLevel(FloatVec2(values[0], values[1]));
|
||||
}
|
||||
}
|
||||
|
||||
namespace {
|
||||
// GL 4.6 core 7.11.2 (and ARB_shader_image_load_store, which introduced the call): the
|
||||
// barrier bitfield is INVALID_VALUE unless every bit is one of the defined ones, with
|
||||
@@ -748,9 +910,32 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "Backend does not support memory barriers."));
|
||||
return;
|
||||
}
|
||||
MGP_FILL(MemoryBarrier);
|
||||
memoryBarrier(barriers);
|
||||
}
|
||||
|
||||
void TextureBarrier() {
|
||||
// GL 4.5 core 8.26 / GL_ARB_texture_barrier: order every write the fixed-function
|
||||
// framebuffer has already issued ahead of every subsequent texture fetch, so a shader may
|
||||
// read texels of a texture that is also attached to the current framebuffer.
|
||||
//
|
||||
// Both backends serve this through their existing memory-barrier hook rather than a new
|
||||
// entry point of their own: GL_FRAMEBUFFER_BARRIER_BIT is the source half (framebuffer
|
||||
// writes) and GL_TEXTURE_FETCH_BARRIER_BIT the destination half (texture fetches), which
|
||||
// is exactly the dependency ARB_texture_barrier defines - just expressed with the wider
|
||||
// scope glMemoryBarrier gives it. That is a superset of the required ordering, never a
|
||||
// subset, so it cannot under-synchronize.
|
||||
auto memoryBarrier = MG_Backend::gBackendFunctionsTable.GL.MemoryBarrier;
|
||||
if (!memoryBarrier) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "Backend does not support memory barriers."));
|
||||
return;
|
||||
}
|
||||
MGP_FILL(MemoryBarrier);
|
||||
memoryBarrier(GL_TEXTURE_FETCH_BARRIER_BIT | GL_FRAMEBUFFER_BARRIER_BIT);
|
||||
}
|
||||
|
||||
void MemoryBarrierByRegion(GLbitfield barriers) {
|
||||
if (!ValidateMemoryBarrierBits(__func__, barriers)) return;
|
||||
auto memoryBarrierByRegion = MG_Backend::gBackendFunctionsTable.GL.MemoryBarrierByRegion;
|
||||
@@ -761,19 +946,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Backend does not support regional memory barriers."));
|
||||
return;
|
||||
}
|
||||
MGP_FILL(MemoryBarrierByRegion);
|
||||
memoryBarrierByRegion(barriers);
|
||||
}
|
||||
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
MultiDrawElementsIndirect_Backend(mode, type, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
MultiDrawArraysIndirect_Backend(mode, indirect, drawcount, stride);
|
||||
}
|
||||
@@ -851,7 +1037,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// NegativeApiErrorsTest.IndirectParameterDrawsCheckBothBuffers pins the INVALID_VALUE
|
||||
// they produce for a call made with no program bound. Same precedence decision, and
|
||||
// the same reason, as DispatchComputeIndirect above.
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
auto multiDrawElementsIndirectCount = MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsIndirectCount;
|
||||
if (!multiDrawElementsIndirectCount) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -872,7 +1058,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
// See MultiDrawElementsIndirectCount, including why this one goes last.
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
auto multiDrawArraysIndirectCount = MG_Backend::gBackendFunctionsTable.GL.MultiDrawArraysIndirectCount;
|
||||
if (!multiDrawArraysIndirectCount) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -887,7 +1073,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "count", count)) return;
|
||||
@@ -897,7 +1083,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawRangeElements_Backend(mode, start, end, count, type, indices);
|
||||
}
|
||||
@@ -905,7 +1091,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void DrawElementsInstancedBaseVertexBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLint basevertex, GLuint baseinstance) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawElementsInstancedBaseVertexBaseInstance_Backend(mode, count, type, indices, instancecount, basevertex,
|
||||
baseinstance);
|
||||
@@ -914,7 +1100,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void DrawElementsInstancedBaseVertex(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLint basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "count", count)) return;
|
||||
@@ -925,21 +1111,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void DrawElementsInstancedBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLuint baseinstance) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawElementsInstancedBaseInstance_Backend(mode, count, type, indices, instancecount, baseinstance);
|
||||
}
|
||||
|
||||
void DrawElementsInstanced(GLenum mode, GLsizei count, GLenum type, const void* indices, GLsizei instancecount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawElementsInstanced_Backend(mode, count, type, indices, instancecount);
|
||||
}
|
||||
|
||||
void DrawElementsIndirect(GLenum mode, GLenum type, const void* indirect) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateIndirectDrawSource(__func__, indirect, kDrawElementsIndirectCommandBytes)) return;
|
||||
@@ -949,21 +1135,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void DrawArraysInstancedBaseInstance(GLenum mode, GLint first, GLsizei count, GLsizei instancecount,
|
||||
GLuint baseinstance) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawArraysInstancedBaseInstance_Backend(mode, first, count, instancecount, baseinstance);
|
||||
}
|
||||
|
||||
void DrawArraysInstanced(GLenum mode, GLint first, GLsizei count, GLsizei instancecount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawArraysInstanced_Backend(mode, first, count, instancecount);
|
||||
}
|
||||
|
||||
void DrawArraysIndirect(GLenum mode, const void* indirect) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateIndirectDrawSource(__func__, indirect, kDrawArraysIndirectCommandBytes)) return;
|
||||
DrawArraysIndirect_Backend(mode, indirect);
|
||||
@@ -971,7 +1157,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawElementsBaseVertex(GLenum mode, GLsizei count, GLenum type, const void* indices, GLint basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "count", count)) return;
|
||||
@@ -981,7 +1167,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawArrays(GLenum mode, GLint first, GLsizei count) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
AccountTransformFeedbackPrimitives(mode, count);
|
||||
DrawArrays_Backend(mode, first, count);
|
||||
@@ -989,7 +1175,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (drawcount < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -1003,7 +1189,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void MultiDrawElements(GLenum mode, const GLsizei* count, GLenum type, const void* const* indices,
|
||||
GLsizei drawcount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
MultiDrawElements_Backend(mode, count, type, indices, drawcount);
|
||||
}
|
||||
@@ -1011,7 +1197,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void MultiDrawElementsBaseVertex(GLenum mode, const GLsizei* count, GLenum type, const void* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "drawcount", drawcount)) return;
|
||||
@@ -1035,7 +1221,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawElements(GLenum mode, GLsizei count, GLenum type, const void* indices) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!PrepareCurrentProgramForDraw(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
AccountTransformFeedbackPrimitives(mode, count);
|
||||
DrawElements_Backend(mode, count, type, indices);
|
||||
@@ -1083,6 +1269,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
MG_State::pGLContext->BeginTransformFeedback(primitiveMode, program);
|
||||
if (const auto beginXfb = MG_Backend::gBackendFunctionsTable.GL.BeginTransformFeedback) {
|
||||
MGP_FILL(BeginTransformFeedback);
|
||||
beginXfb(primitiveMode);
|
||||
}
|
||||
}
|
||||
@@ -1165,6 +1352,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// Closed while the capture state is still active: a backend that captures
|
||||
// through its own driver reads the capture program and buffer bindings here.
|
||||
if (const auto endXfb = MG_Backend::gBackendFunctionsTable.GL.EndTransformFeedback) {
|
||||
MGP_FILL(EndTransformFeedback);
|
||||
endXfb();
|
||||
}
|
||||
MG_State::pGLContext->EndTransformFeedback();
|
||||
@@ -1173,9 +1361,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// the GPU work is all that is required.
|
||||
auto& backendGL = MG_Backend::gBackendFunctionsTable.GL;
|
||||
if (backendGL.FenceSync && backendGL.ClientWaitSync) {
|
||||
MGP_FILL(FenceSync);
|
||||
if (auto sync = backendGL.FenceSync()) {
|
||||
MGP_FILL(ClientWaitSync);
|
||||
backendGL.ClientWaitSync(sync, GL_SYNC_FLUSH_COMMANDS_BIT, ~0ull);
|
||||
if (backendGL.DeleteSync) {
|
||||
MGP_FILL(DeleteSync);
|
||||
backendGL.DeleteSync(sync);
|
||||
}
|
||||
}
|
||||
@@ -1194,6 +1385,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
MG_State::pGLContext->SetTransformFeedbackPaused(true);
|
||||
if (const auto pauseXfb = MG_Backend::gBackendFunctionsTable.GL.PauseTransformFeedback) {
|
||||
MGP_FILL(PauseTransformFeedback);
|
||||
pauseXfb();
|
||||
}
|
||||
}
|
||||
@@ -1208,6 +1400,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
MG_State::pGLContext->SetTransformFeedbackPaused(false);
|
||||
if (const auto resumeXfb = MG_Backend::gBackendFunctionsTable.GL.ResumeTransformFeedback) {
|
||||
MGP_FILL(ResumeTransformFeedback);
|
||||
resumeXfb();
|
||||
}
|
||||
}
|
||||
@@ -1413,6 +1606,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
continue;
|
||||
}
|
||||
if (const auto deleteXfb = MG_Backend::gBackendFunctionsTable.GL.DeleteTransformFeedback) {
|
||||
MGP_FILL(DeleteTransformFeedback);
|
||||
deleteXfb(id);
|
||||
}
|
||||
MG_State::pGLContext->MarkTransformFeedbackObjectForDeletion(id);
|
||||
@@ -1444,6 +1638,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
MG_State::pGLContext->BindTransformFeedbackObject(id);
|
||||
if (const auto bindXfb = MG_Backend::gBackendFunctionsTable.GL.BindTransformFeedback) {
|
||||
MGP_FILL(BindTransformFeedback);
|
||||
bindXfb(id);
|
||||
}
|
||||
}
|
||||
@@ -1459,7 +1654,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// (GL 4.6 core 10.3.7).
|
||||
static void DrawTransformFeedbackImpl(const char* functionName, GLenum mode, GLuint id, GLuint stream,
|
||||
GLsizei instancecount) {
|
||||
if (!ValidateCurrentProgramForExecution(functionName)) return;
|
||||
if (!PrepareCurrentProgramForDraw(functionName)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(functionName, mode)) return;
|
||||
if (instancecount < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -1482,8 +1677,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::to_string(id) + " is not a transform feedback object name."));
|
||||
return;
|
||||
}
|
||||
// GL_MAX_VERTEX_STREAMS is 1, so stream 0 is the only one that exists.
|
||||
if (stream != 0) {
|
||||
// GL 4.6 core 10.3.7 bounds `stream` by GL_MAX_VERTEX_STREAMS, which this implementation
|
||||
// answers as 1 - so stream 0 is the only one that exists and anything else is
|
||||
// INVALID_VALUE. Read from the getter rather than written as `stream != 0` so the two can
|
||||
// never drift: if vertex-stream support ever lands, this bound moves with the limit.
|
||||
GLint maxVertexStreams = 1;
|
||||
GetIntegerv(GL_MAX_VERTEX_STREAMS, &maxVertexStreams);
|
||||
if (stream >= static_cast<GLuint>(std::max(maxVertexStreams, 1))) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
@@ -1501,6 +1701,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
// `stream` is provably 0 here (the bound above is 1), so this is stream 0's record.
|
||||
const Uint64 vertices = MG_State::pGLContext->GetTransformFeedbackRecordedVertices(id);
|
||||
if (vertices == 0) return;
|
||||
const auto count = static_cast<GLsizei>(vertices);
|
||||
|
||||
@@ -32,8 +32,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void DispatchCompute(GLuint numGroupsX, GLuint numGroupsY, GLuint numGroupsZ);
|
||||
void DispatchComputeIndirect(GLintptr indirect);
|
||||
void PatchParameteri(GLenum pname, GLint value);
|
||||
void PatchParameterfv(GLenum pname, const GLfloat* values);
|
||||
void MemoryBarrier(GLbitfield barriers);
|
||||
void MemoryBarrierByRegion(GLbitfield barriers);
|
||||
void TextureBarrier();
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
|
||||
@@ -160,7 +160,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, ReleaseShaderCompiler) DECLARE_GL_FUNCTION_S
|
||||
DECLARE_GL_FUNCTION_HEAD(void, RenderbufferStorage, GLenum target, GLenum internalformat, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, RenderbufferStorage, target, internalformat, width, height)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, SampleCoverage, GLfloat value, GLboolean invert) DECLARE_GL_FUNCTION_END_NO_RETURN(void, SampleCoverage, value, invert)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, Scissor, GLint x, GLint y, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, Scissor, x, y, width, height)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ShaderBinary, GLsizei count, const GLuint* shaders, GLenum binaryformat, const void* binary, GLsizei length) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ShaderBinary, count, shaders, binaryformat, binary, length)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ShaderBinary, GLsizei count, const GLuint* shaders, GLenum binaryformat, const void* binary, GLsizei length) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ShaderBinary, count, shaders, binaryformat, binary, length)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ShaderSource, GLuint shader, GLsizei count, const GLchar* const* string, const GLint* length) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ShaderSource, shader, count, string, length)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, StencilFunc, GLenum func, GLint ref, GLuint mask) DECLARE_GL_FUNCTION_END_NO_RETURN(void, StencilFunc, func, ref, mask)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, StencilFuncSeparate, GLenum face, GLenum func, GLint ref, GLuint mask) DECLARE_GL_FUNCTION_END_NO_RETURN(void, StencilFuncSeparate, face, func, ref, mask)
|
||||
@@ -411,7 +411,7 @@ DECLARE_GL_FUNCTION_HEAD(void, ReadnPixels, GLint x, GLint y, GLsizei width, GLs
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformfv, GLuint program, GLint location, GLsizei bufSize, GLfloat* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformfv, program, location, bufSize, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformiv, GLuint program, GLint location, GLsizei bufSize, GLint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformiv, program, location, bufSize, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnUniformuiv, GLuint program, GLint location, GLsizei bufSize, GLuint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnUniformuiv, program, location, bufSize, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, MinSampleShading, GLfloat value) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, MinSampleShading, value)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, MinSampleShading, GLfloat value) DECLARE_GL_FUNCTION_END_NO_RETURN(void, MinSampleShading, value)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, PatchParameteri, GLenum pname, GLint value) DECLARE_GL_FUNCTION_END_NO_RETURN(void, PatchParameteri, pname, value)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TexParameterIiv, GLenum target, GLenum pname, const GLint* params) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TexParameterIiv, target, pname, params)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TexParameterIuiv, GLenum target, GLenum pname, const GLuint* params) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TexParameterIuiv, target, pname, params)
|
||||
@@ -923,7 +923,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, GetActiveSubroutineName, GLuint program, GLe
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, UniformSubroutinesuiv, GLenum shadertype, GLsizei count, const GLuint* indices) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, UniformSubroutinesuiv, shadertype, count, indices)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetUniformSubroutineuiv, GLenum shadertype, GLint location, GLuint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetUniformSubroutineuiv, shadertype, location, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetProgramStageiv, GLuint program, GLenum shadertype, GLenum pname, GLint* values) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetProgramStageiv, program, shadertype, pname, values)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PatchParameterfv, GLenum pname, const GLfloat* values) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PatchParameterfv, pname, values)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, PatchParameterfv, GLenum pname, const GLfloat* values) DECLARE_GL_FUNCTION_END_NO_RETURN(void, PatchParameterfv, pname, values)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedback, GLenum mode, GLuint id) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedback, mode, id)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackStream, GLenum mode, GLuint id, GLuint stream) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackStream, mode, id, stream)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BeginQueryIndexed, GLenum target, GLuint index, GLuint id) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BeginQueryIndexed, target, index, id)
|
||||
@@ -994,7 +994,7 @@ DECLARE_GL_FUNCTION_HEAD(void, BindTextures, GLuint first, GLsizei count, const
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindSamplers, GLuint first, GLsizei count, const GLuint* samplers) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindSamplers, first, count, samplers)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindImageTextures, GLuint first, GLsizei count, const GLuint* textures) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindImageTextures, first, count, textures)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindVertexBuffers, GLuint first, GLsizei count, const GLuint* buffers, const GLintptr* offsets, const GLsizei* strides) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindVertexBuffers, first, count, buffers, offsets, strides)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClipControl, GLenum origin, GLenum depth) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClipControl, origin, depth)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClipControl, GLenum origin, GLenum depth) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClipControl, origin, depth)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CreateTransformFeedbacks, GLsizei n, GLuint* ids) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CreateTransformFeedbacks, n, ids)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TransformFeedbackBufferBase, GLuint xfb, GLuint index, GLuint buffer) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TransformFeedbackBufferBase, xfb, index, buffer)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TransformFeedbackBufferRange, GLuint xfb, GLuint index, GLuint buffer, GLintptr offset, GLsizeiptr size) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TransformFeedbackBufferRange, xfb, index, buffer, offset, size)
|
||||
@@ -1107,11 +1107,11 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnConvolutionFilter, GLenum target, GLenum
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnSeparableFilter, GLenum target, GLenum format, GLenum type, GLsizei rowBufSize, void* row, GLsizei columnBufSize, void* column, void* span) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnSeparableFilter, target, format, type, rowBufSize, row, columnBufSize, column, span)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnHistogram, GLenum target, GLboolean reset, GLenum format, GLenum type, GLsizei bufSize, void* values) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnHistogram, target, reset, format, type, bufSize, values)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetnMinmax, GLenum target, GLboolean reset, GLenum format, GLenum type, GLsizei bufSize, void* values) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetnMinmax, target, reset, format, type, bufSize, values)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TextureBarrier, void) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TextureBarrier, )
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, SpecializeShader, GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants, const GLuint* pConstantIndex, const GLuint* pConstantValue) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, SpecializeShader, shader, pEntryPoint, numSpecializationConstants, pConstantIndex, pConstantValue)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureBarrier, void) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureBarrier, )
|
||||
DECLARE_GL_FUNCTION_HEAD(void, SpecializeShader, GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants, const GLuint* pConstantIndex, const GLuint* pConstantValue) DECLARE_GL_FUNCTION_END_NO_RETURN(void, SpecializeShader, shader, pEntryPoint, numSpecializationConstants, pConstantIndex, pConstantValue)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, MultiDrawArraysIndirectCount, GLenum mode, const void* indirect, GLintptr drawcount, GLsizei maxdrawcount, GLsizei stride) DECLARE_GL_FUNCTION_END_NO_RETURN(void, MultiDrawArraysIndirectCount, mode, indirect, drawcount, maxdrawcount, stride)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, MultiDrawElementsIndirectCount, GLenum mode, GLenum type, const void* indirect, GLintptr drawcount, GLsizei maxdrawcount, GLsizei stride) DECLARE_GL_FUNCTION_END_NO_RETURN(void, MultiDrawElementsIndirectCount, mode, type, indirect, drawcount, maxdrawcount, stride)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PolygonOffsetClamp, GLfloat factor, GLfloat units, GLfloat clamp) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PolygonOffsetClamp, factor, units, clamp)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, PolygonOffsetClamp, GLfloat factor, GLfloat units, GLfloat clamp) DECLARE_GL_FUNCTION_END_NO_RETURN(void, PolygonOffsetClamp, factor, units, clamp)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PrimitiveBoundingBoxARB, GLfloat minX, GLfloat minY, GLfloat minZ, GLfloat minW, GLfloat maxX, GLfloat maxY, GLfloat maxZ, GLfloat maxW) DECLARE_GL_FUNCTION_STUB_END(void, PrimitiveBoundingBoxARB, minX, minY, minZ, minW, maxX, maxY, maxZ, maxW)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLuint64, GetTextureHandleARB, GLuint texture) DECLARE_GL_FUNCTION_STUB_END(GLuint64, GetTextureHandleARB, texture)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLuint64, GetTextureSamplerHandleARB, GLuint texture, GLuint sampler) DECLARE_GL_FUNCTION_STUB_END(GLuint64, GetTextureSamplerHandleARB, texture, sampler)
|
||||
@@ -1150,7 +1150,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, GetProgramLocalParameterdvARB, GLenum target
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetProgramLocalParameterfvARB, GLenum target, GLuint index, GLfloat* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetProgramLocalParameterfvARB, target, index, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetProgramStringARB, GLenum target, GLenum pname, void* string) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetProgramStringARB, target, pname, string)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, FramebufferTextureFaceARB, GLenum target, GLenum attachment, GLuint texture, GLint level, GLenum face) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, FramebufferTextureFaceARB, target, attachment, texture, level, face)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, SpecializeShaderARB, GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants, const GLuint* pConstantIndex, const GLuint* pConstantValue) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, SpecializeShaderARB, shader, pEntryPoint, numSpecializationConstants, pConstantIndex, pConstantValue)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, SpecializeShaderARB, GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants, const GLuint* pConstantIndex, const GLuint* pConstantValue) DECLARE_GL_FUNCTION_END_NO_RETURN(void, SpecializeShader, shader, pEntryPoint, numSpecializationConstants, pConstantIndex, pConstantValue)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, Uniform1i64ARB, GLint location, GLint64 x) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, Uniform1i64ARB, location, x)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, Uniform2i64ARB, GLint location, GLint64 x, GLint64 y) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, Uniform2i64ARB, location, x, y)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, Uniform3i64ARB, GLint location, GLint64 x, GLint64 y, GLint64 z) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, Uniform3i64ARB, location, x, y, z)
|
||||
@@ -2049,7 +2049,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, GetPixelTransformParameterivEXT, GLenum targ
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetPixelTransformParameterfvEXT, GLenum target, GLenum pname, GLfloat* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetPixelTransformParameterfvEXT, target, pname, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PointParameterfEXT, GLenum pname, GLfloat param) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PointParameterfEXT, pname, param)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PointParameterfvEXT, GLenum pname, const GLfloat* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PointParameterfvEXT, pname, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PolygonOffsetClampEXT, GLfloat factor, GLfloat units, GLfloat clamp) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PolygonOffsetClampEXT, factor, units, clamp)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, PolygonOffsetClampEXT, GLfloat factor, GLfloat units, GLfloat clamp) DECLARE_GL_FUNCTION_END_NO_RETURN(void, PolygonOffsetClamp, factor, units, clamp)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ProvokingVertexEXT, GLenum mode) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ProvokingVertex, mode)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, RasterSamplesEXT, GLuint samples, GLboolean fixedsamplelocations) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, RasterSamplesEXT, samples, fixedsamplelocations)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, SecondaryColor3bEXT, GLbyte red, GLbyte green, GLbyte blue) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, SecondaryColor3bEXT, red, green, blue)
|
||||
@@ -2546,7 +2546,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, ShadingRateImageBarrierNV, GLboolean synchro
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ShadingRateImagePaletteNV, GLuint viewport, GLuint first, GLsizei count, const GLenum* rates) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ShadingRateImagePaletteNV, viewport, first, count, rates)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ShadingRateSampleOrderNV, GLenum order) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ShadingRateSampleOrderNV, order)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ShadingRateSampleOrderCustomNV, GLenum rate, GLuint samples, const GLint* locations) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ShadingRateSampleOrderCustomNV, rate, samples, locations)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TextureBarrierNV, void) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TextureBarrierNV, )
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureBarrierNV, void) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureBarrier, )
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TexImage2DMultisampleCoverageNV, GLenum target, GLsizei coverageSamples, GLsizei colorSamples, GLint internalFormat, GLsizei width, GLsizei height, GLboolean fixedSampleLocations) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TexImage2DMultisampleCoverageNV, target, coverageSamples, colorSamples, internalFormat, width, height, fixedSampleLocations)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TexImage3DMultisampleCoverageNV, GLenum target, GLsizei coverageSamples, GLsizei colorSamples, GLint internalFormat, GLsizei width, GLsizei height, GLsizei depth, GLboolean fixedSampleLocations) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TexImage3DMultisampleCoverageNV, target, coverageSamples, colorSamples, internalFormat, width, height, depth, fixedSampleLocations)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TextureImage2DMultisampleNV, GLuint texture, GLenum target, GLsizei samples, GLint internalFormat, GLsizei width, GLsizei height, GLboolean fixedSampleLocations) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TextureImage2DMultisampleNV, texture, target, samples, internalFormat, width, height, fixedSampleLocations)
|
||||
|
||||
@@ -15,6 +15,15 @@
|
||||
#include <MG_Impl/GLImpl/Texture/Validators.h>
|
||||
#include <MG_Impl/GLImpl/Getter/GL_Getter.h>
|
||||
#include <MG_State/GLState/ErrorState/Error.h>
|
||||
#include <MG_Impl/Pipe/PipeFill.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P4a, ID-19(c). This file is the ONLY place every DSA framebuffer entry point lives, and the
|
||||
// emitter it reaches is this package's own header rather than a declaration in one of the
|
||||
// contract's: MG_Pipe/PipeMutation.h is the door MG_State has into the client and carries no
|
||||
// framebuffer row, and MG_Impl/GLImpl and MG_Impl/Pipe are the same layer (this file already
|
||||
// includes MG_Impl/Pipe/PipeFill.h for MGP_FILL).
|
||||
#include <MG_Impl/Pipe/FramebufferEmit.h>
|
||||
#endif
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/TextureEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToMG/TextureEnumConverter.h>
|
||||
@@ -474,6 +483,75 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// GL 4.6 core 9.2.8 conditions that depend only on the framebuffer and the attachment
|
||||
// point. Shared, because glFramebufferTexture / 1D / 2D / 3D / TextureLayer are aliases of
|
||||
// one another in that section and a CTS case that walks the family must not get five
|
||||
// different answers - which is exactly what happened when these lived in one helper that
|
||||
// only two of the five went through.
|
||||
Bool ValidateFramebufferTextureAttachmentPoint(const char* functionName,
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>&
|
||||
framebufferObject,
|
||||
FramebufferAttachmentType attachmentType) {
|
||||
// "An INVALID_OPERATION error is generated if COLOR_ATTACHMENTm is used with m greater
|
||||
// than or equal to MAX_COLOR_ATTACHMENTS."
|
||||
if (!FramebufferImpl::ValidateColorAttachmentInRange(attachmentType, functionName)) return false;
|
||||
// "An INVALID_OPERATION error is generated if zero is bound to target." MobileGL keeps
|
||||
// a real FramebufferObject for framebuffer 0, so a null test can never see this - the
|
||||
// object is always there, and framebuffer 0 has to be recognised by identity instead,
|
||||
// the same comparison DrawBuffers_State makes. Without this an attach onto the default
|
||||
// framebuffer silently REPLACED its colour attachment, permanently desynchronising it
|
||||
// from what the swapchain keeps publishing.
|
||||
const auto& defaultFramebufferInfo = FramebufferImpl::pDefaultFramebufferInfo;
|
||||
if (!framebufferObject ||
|
||||
(defaultFramebufferInfo && framebufferObject == defaultFramebufferInfo->defaultFBO)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", functionName,
|
||||
"No framebuffer object is bound to the target; the default framebuffer's attachments "
|
||||
"cannot be named."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// The other half of 9.2.8: "level must be greater than or equal to zero", and for a
|
||||
// texture with immutable storage it "must be smaller than the number of levels the texture
|
||||
// has". Split from the attachment-point half because the caller only has a texture object
|
||||
// once the detach (texture == 0) case is behind it.
|
||||
Bool ValidateFramebufferTextureLevel(const char* functionName,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& textureObject,
|
||||
GLint level) {
|
||||
if (level < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"Texture level must be non-negative."));
|
||||
return false;
|
||||
}
|
||||
if (!textureObject || !textureObject->IsImmutable()) {
|
||||
// A mutable texture has no level bound here: a level it has not specified yet is
|
||||
// not an error, it just leaves the framebuffer incomplete.
|
||||
return true;
|
||||
}
|
||||
// GetAddressableLevelCount(), NOT GetImmutableLevels(): for a VIEW the latter is
|
||||
// deliberately the ORIGINAL texture's count (GL 4.6 core 8.18 defines
|
||||
// TEXTURE_IMMUTABLE_LEVELS on a view that way), which is far too large a bound - a
|
||||
// two-level view onto a ten-level texture would accept level 5 and attach an image
|
||||
// nothing can draw into.
|
||||
const Uint levelBound = textureObject->GetAddressableLevelCount();
|
||||
if (static_cast<Uint>(level) >= levelBound) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", functionName,
|
||||
std::format("Texture level {} is beyond the {} level(s) this texture has.", level,
|
||||
levelBound)));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void AttachFramebufferTextureWithUploadTarget(const char* functionName, GLenum target, GLenum attachment,
|
||||
GLuint texture, GLint level,
|
||||
TextureUploadTarget textureUploadTarget, Bool layered = false) {
|
||||
@@ -482,10 +560,24 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
if (attachment == GL_DEPTH_STENCIL_ATTACHMENT) {
|
||||
// `layered` has to travel with the split. GL_DEPTH_STENCIL_ATTACHMENT is only a
|
||||
// shorthand for attaching the same image to both halves (GL 4.6 core 9.2.6), so
|
||||
// whether glFramebufferTexture made it LAYERED is a property of the call, not of
|
||||
// which half is being recorded - and dropping it here (the parameter defaults to
|
||||
// false) recorded a non-layered depth/stencil attachment beside a layered colour
|
||||
// one for every layered target. That is an inconsistent framebuffer by 9.4.1's
|
||||
// own rule, and downstream it means the depth/stencil attachment covers layer 0
|
||||
// alone: DirectVulkan built its view with layerCount 1 under a framebuffer
|
||||
// declaring N layers (VUID-VkFramebufferCreateInfo-flags-04535), and DirectGLES
|
||||
// attached one layer of it beside a layered colour target, which the driver
|
||||
// answers with GL_FRAMEBUFFER_INCOMPLETE_LAYER_TARGETS - every draw silently
|
||||
// produced nothing. This is the shape
|
||||
// texture_cube_map_array.stencil_attachments_*_layered and
|
||||
// geometry_shader.layered_framebuffer.stencil_support are built on.
|
||||
AttachFramebufferTextureWithUploadTarget(functionName, target, GL_DEPTH_ATTACHMENT, texture, level,
|
||||
textureUploadTarget);
|
||||
textureUploadTarget, layered);
|
||||
AttachFramebufferTextureWithUploadTarget(functionName, target, GL_STENCIL_ATTACHMENT, texture, level,
|
||||
textureUploadTarget);
|
||||
textureUploadTarget, layered);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -497,13 +589,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
auto& bindingSlot = MG_State::pGLContext->GetFramebufferBindingSlot(framebufferTarget);
|
||||
auto& framebufferObject = bindingSlot.GetBoundObject();
|
||||
if (!framebufferObject) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"Framebuffer target is bound to no framebuffer object."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateFramebufferTextureAttachmentPoint(functionName, framebufferObject, attachmentType)) return;
|
||||
|
||||
if (texture == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
@@ -518,6 +604,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::format("Texture object {} is not valid.", texture)));
|
||||
return;
|
||||
}
|
||||
if (!ValidateFramebufferTextureLevel(functionName, textureObject, level)) return;
|
||||
|
||||
const auto expectedTextureTarget = MG_Util::ConvertTextureUploadTargetToTextureTarget(textureUploadTarget);
|
||||
if (expectedTextureTarget == TextureTarget::Unknown ||
|
||||
@@ -534,10 +621,40 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
framebufferObject->AttachTexture(attachmentType, textureObject, textureUploadTarget, level, 0, layered);
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
// P4a, ID-19(c): ANY FRAMEBUFFER THE SERVER IS ABOUT TO RECEIVE BY NAME HAS A RECORD.
|
||||
//
|
||||
// The applier keeps framebuffer records PER OBJECT, keyed by the handle - but before
|
||||
// ID-19 it held only the two BOUND-target records, and the emitter only ever built them
|
||||
// at the validate point out of the two bindings. So glClearNamedFramebufferfv(fbo) or
|
||||
// glBlitNamedFramebuffer(..., fbo, ...) on an fbo bound to NEITHER binding reached a
|
||||
// backend that minted a fresh driver framebuffer with no attachments, found no record
|
||||
// for it, declined, and issued the clear against it anyway: GL_INVALID_FRAMEBUFFER_-
|
||||
// OPERATION and nothing cleared, where the legacy arm cleared correctly.
|
||||
//
|
||||
// TWO CLASSES OF SITE call this, and both are "the point at which the object is final
|
||||
// for this call": the five CONSUMERS (blit and the four clears) publish immediately
|
||||
// before MGP_FILL, so the record precedes the verb that hands the object over and a
|
||||
// later bound-target record for the same object still wins; the ten MUTATORS (the DSA
|
||||
// attachment, draw-buffer and read-buffer setters) publish immediately after the
|
||||
// frontend mutation, because they have no validate point at all - FillPoints.def has no
|
||||
// verb for any of them, so there is no MGP_FILL to sit in front of.
|
||||
//
|
||||
// A CALL THAT MOVED NOTHING IS FREE: the record's ContentHash is the emitter's own
|
||||
// suppressor and it is keyed per framebuffer object, so a redundant publish emits zero
|
||||
// bytes. EmitFramebufferByName picks Draw/Read/Both over Named when the object IS
|
||||
// bound, so a Named record can never overwrite a bound record's Target underneath the
|
||||
// binding that resolves through it.
|
||||
void PipePublishFramebufferByName(const SharedPtr<MG_State::GLState::FramebufferObject>& fbo) {
|
||||
if (!fbo) return;
|
||||
MG_Pipe::MGPipeFramebufferEmitterInstance().EmitFramebufferByName(*fbo);
|
||||
}
|
||||
#endif
|
||||
} // namespace
|
||||
|
||||
void BlitFramebuffer_Backend(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1, GLint dstX0, GLint dstY0,
|
||||
GLint dstX1, GLint dstY1, GLbitfield mask, GLenum filter) {
|
||||
MGP_FILL(BlitFramebuffer);
|
||||
MG_Backend::gBackendFunctionsTable.GL.BlitFramebuffer(srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1, dstY1,
|
||||
mask, filter);
|
||||
}
|
||||
@@ -551,6 +668,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_E_ONCE("glBlitNamedFramebuffer skipped: backend does not implement explicit framebuffer blit.");
|
||||
return;
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(readFramebuffer);
|
||||
PipePublishFramebufferByName(drawFramebuffer);
|
||||
#endif
|
||||
MGP_FILL(BlitNamedFramebuffer);
|
||||
blitNamedFramebuffer(readFramebuffer, drawFramebuffer, srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1,
|
||||
dstY1, mask, filter);
|
||||
}
|
||||
@@ -562,6 +684,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferfv skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebuffer);
|
||||
#endif
|
||||
MGP_FILL(ClearNamedFramebufferfv);
|
||||
clearNamedFramebufferfv(framebuffer, buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
@@ -572,6 +698,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferfi skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebuffer);
|
||||
#endif
|
||||
MGP_FILL(ClearNamedFramebufferfi);
|
||||
clearNamedFramebufferfi(framebuffer, buffer, drawbuffer, depth, stencil);
|
||||
}
|
||||
|
||||
@@ -582,6 +712,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferiv skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebuffer);
|
||||
#endif
|
||||
MGP_FILL(ClearNamedFramebufferiv);
|
||||
clearNamedFramebufferiv(framebuffer, buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
@@ -592,6 +726,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferuiv skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebuffer);
|
||||
#endif
|
||||
MGP_FILL(ClearNamedFramebufferuiv);
|
||||
clearNamedFramebufferuiv(framebuffer, buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
@@ -624,16 +762,33 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// GL_MAX_SAMPLES is the ceiling over all formats; an integer format has its own
|
||||
// (GL_MAX_INTEGER_SAMPLES) and GL 4.6 core 9.2.4 makes exceeding it INVALID_OPERATION.
|
||||
// The multisample TEXTURE path resolves the limit per format the same way
|
||||
// (GL_Texture.cpp, GetMaxSupportedTextureSamples). Both are floored to the value MobileGL
|
||||
// advertises: on a driver where the two differ - Adreno reports GL_MAX_SAMPLES 4 and
|
||||
// GL_MAX_INTEGER_SAMPLES 1 - rejecting the advertised count here only moves the failure
|
||||
// from the driver into MobileGL, so the frontend accepts it and the backend clamps the
|
||||
// count it actually hands the driver.
|
||||
// (GL_Texture.cpp, GetMaxSupportedTextureSamples), and both now enforce exactly what their
|
||||
// pname advertises. The integer ceiling used to be floored at GL_MAX_SAMPLES so that the
|
||||
// frontend would accept a count it had advertised globally - but on Adreno and Mali the
|
||||
// integer path is genuinely one sample, and accepting four only moved the failure from an
|
||||
// honest INVALID_OPERATION here to a silently under-allocated renderbuffer.
|
||||
// The head of the per-format renderbuffer sample list the backend probed, or 0 when nothing
|
||||
// was probed for it. Same shape as GetProbedMaxTextureSamples in GL_Texture.cpp, and reads
|
||||
// the same cache glGetInternalformativ(GL_RENDERBUFFER, ..., GL_SAMPLES) answers from.
|
||||
static Int GetProbedMaxRenderbufferSamples(TextureInternalFormat format) {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
const SizeT targetIndex = MG_Backend::GetRenderbufferFormatCapabilityTargetIndex();
|
||||
const SizeT formatIndex = static_cast<SizeT>(format);
|
||||
if (targetIndex >= MG_Backend::kFormatCapabilityTargetCount ||
|
||||
formatIndex >= MG_Backend::kFormatCapabilityFormatCount) {
|
||||
return 0;
|
||||
}
|
||||
const auto& sampleCounts =
|
||||
MG_Backend::pActiveBackendObject->GetFormatCapabilities().SampleCounts[targetIndex][formatIndex];
|
||||
return sampleCounts.empty() ? 0 : sampleCounts.front();
|
||||
}
|
||||
|
||||
Int GetMaxRenderbufferSamplesForFormat_State(TextureInternalFormat format) {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return std::numeric_limits<Int>::max();
|
||||
}
|
||||
const auto& dynamicParameters = MG_Backend::pActiveBackendObject->GetDynamicParameters();
|
||||
|
||||
GLenum normalizedInternalFormat = MG_Util::ConvertTextureInternalFormatToGLEnum(format);
|
||||
GLenum normalizedFormat = GL_RGBA;
|
||||
@@ -644,13 +799,24 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
&normalizedType);
|
||||
const Bool isIntegerFormat = normalizedFormat == GL_RED_INTEGER || normalizedFormat == GL_RG_INTEGER ||
|
||||
normalizedFormat == GL_RGB_INTEGER || normalizedFormat == GL_RGBA_INTEGER;
|
||||
// The per-format probe first, for the same reason the texture path takes it first: GL 4.6
|
||||
// core 9.2.4 words the error as "samples is greater than the maximum number of samples
|
||||
// supported for internalformat (see GetInternalformativ)", and
|
||||
// glGetInternalformativ(GL_RENDERBUFFER, ..., GL_SAMPLES) is answered from exactly this
|
||||
// list. It was never consulted here - the TODO that deferred it was written before the
|
||||
// query was backed and had gone stale - so a format whose multisample probes fail inside
|
||||
// a category that allows four was accepted at four, quietly allocated at one by
|
||||
// ClampSamplesToBackendSupport, and then reported as four by
|
||||
// glGetRenderbufferParameteriv(GL_RENDERBUFFER_SAMPLES).
|
||||
const Int probedMaxSamples = GetProbedMaxRenderbufferSamples(format);
|
||||
if (probedMaxSamples > 0) {
|
||||
return probedMaxSamples;
|
||||
}
|
||||
if (!isIntegerFormat) {
|
||||
return GetMaxRenderbufferSamples_State();
|
||||
}
|
||||
// Per-format still, but never below the ceiling glGetIntegerv(GL_MAX_SAMPLES) promised:
|
||||
// the driver's raw GL_MAX_INTEGER_SAMPLES stays the *backend* limit and the backend
|
||||
// clamps to it, while the frontend honours what it advertised.
|
||||
return std::max(dynamicParameters.MaxIntegerSamples, GetAdvertisedMaxSamples());
|
||||
// Exactly what glGetIntegerv(GL_MAX_INTEGER_SAMPLES) reports.
|
||||
return GetAdvertisedIntegerMaxSamples();
|
||||
}
|
||||
|
||||
Bool ValidateRenderbufferStorageSize_State(GLsizei width, GLsizei height, const char* caller) {
|
||||
@@ -682,8 +848,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return false;
|
||||
}
|
||||
|
||||
// TODO: Resolve the remaining per-internalformat renderbuffer sample limits once
|
||||
// glGetInternalformativ is backed; integer formats are handled below.
|
||||
// Per-internalformat, from the probe list glGetInternalformativ answers with, falling back
|
||||
// to the format's category pname where nothing was probed. (This carried a TODO deferring
|
||||
// the per-format resolution "once glGetInternalformativ is backed"; it has been backed for
|
||||
// both renderbuffers and multisample textures since, so the deferral was collected.)
|
||||
const Int maxSamples = GetMaxRenderbufferSamplesForFormat_State(format);
|
||||
if (samples > maxSamples) {
|
||||
// GL 4.6 core 9.2.4 makes asking for more samples than the format supports
|
||||
@@ -1048,13 +1216,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
auto& bindingSlot = MG_State::pGLContext->GetFramebufferBindingSlot(framebufferTarget);
|
||||
auto& framebufferObject = bindingSlot.GetBoundObject();
|
||||
if (!framebufferObject) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"Framebuffer target is bound to no framebuffer object."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateFramebufferTextureAttachmentPoint(functionName, framebufferObject, attachmentType)) return;
|
||||
|
||||
if (texture == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
@@ -1069,6 +1231,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::format("Texture object {} is not valid.", texture)));
|
||||
return;
|
||||
}
|
||||
if (!ValidateFramebufferTextureLevel(functionName, textureObject, level)) return;
|
||||
if (layer < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
@@ -1191,6 +1354,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Framebuffer target is bound to no framebuffer object."));
|
||||
return;
|
||||
}
|
||||
// glFramebufferTexture2D is by far the most-used member of the family and the only one
|
||||
// that inlines its own logic instead of going through the shared helper, so the 9.2.8
|
||||
// conditions have to be asked here explicitly.
|
||||
if (!ValidateFramebufferTextureAttachmentPoint("FramebufferTexture2D_State", framebufferObject,
|
||||
attachmentType)) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (texture == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
@@ -1205,6 +1375,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::format("Texture object {} is not valid.", texture)));
|
||||
return;
|
||||
}
|
||||
if (!ValidateFramebufferTextureLevel("FramebufferTexture2D_State", textureObject, level)) return;
|
||||
|
||||
const auto expectedTextureTarget = MG_Util::ConvertTextureUploadTargetToTextureTarget(textureUploadTarget);
|
||||
if (expectedTextureTarget == TextureTarget::Unknown ||
|
||||
@@ -1241,6 +1412,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
// The name's validity is an INVALID_VALUE condition (GL 4.6 core 9.2.8), and it has to be
|
||||
// asked BEFORE the object is resolved: reporting the miss as the INVALID_OPERATION below
|
||||
// pre-empted the shared helper's ValidateTextureName and answered the wrong error code for
|
||||
// every texture name that was never generated.
|
||||
if (!TextureImpl::ValidateTextureName(texture, true)) return;
|
||||
|
||||
auto& textureObject = MG_State::pGLContext->GetTextureObject(texture);
|
||||
if (!textureObject) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -1280,6 +1457,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
if (texture == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1291,13 +1471,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::format("Texture object {} is not valid.", texture)));
|
||||
return;
|
||||
}
|
||||
if (level < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "NamedFramebufferTexture_State",
|
||||
"Texture level must be non-negative."));
|
||||
return;
|
||||
}
|
||||
// The whole level condition, not just its negative half: glNamedFramebufferTexture and
|
||||
// glFramebufferTexture are equivalent in 9.2.8, so an out-of-range immutable level has to
|
||||
// be rejected on both or a CTS case gets two answers for one rule.
|
||||
if (!ValidateFramebufferTextureLevel("NamedFramebufferTexture_State", textureObject, level)) return;
|
||||
|
||||
TextureUploadTarget textureUploadTarget = TextureUploadTarget::Unknown;
|
||||
Bool layered = false;
|
||||
@@ -1309,6 +1486,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
framebufferObject->AttachTexture(attachmentType, textureObject, textureUploadTarget, level, 0, layered);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
void NamedFramebufferTextureWithUploadTarget_State(const char* functionName, GLuint framebuffer, GLenum attachment,
|
||||
@@ -1332,6 +1512,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
if (texture == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1359,6 +1542,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
framebufferObject->AttachTexture(attachmentType, textureObject, textureUploadTarget, level);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
void NamedFramebufferTexture1D_State(GLuint framebuffer, GLenum attachment, GLenum textarget, GLuint texture,
|
||||
@@ -1413,6 +1599,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
if (texture == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1515,6 +1704,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
framebufferObject->AttachTexture(attachmentType, textureObject, textureUploadTarget, level, layer,
|
||||
/*layered=*/false);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
void FramebufferRenderbuffer_State(GLenum target, GLenum attachment, GLenum renderbuffertarget,
|
||||
@@ -1584,6 +1776,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
if (renderbuffer == 0) {
|
||||
framebufferObject->Detach(attachmentType);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1593,6 +1788,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!renderbufferObject) return;
|
||||
|
||||
framebufferObject->AttachRenderbuffer(attachmentType, renderbufferObject);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
void DrawBuffersForFramebuffer_State(const SharedPtr<MG_State::GLState::FramebufferObject>& fbo, Bool isDefaultFBO,
|
||||
@@ -1792,6 +1990,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
: GetNamedFramebufferObject_State(framebuffer, "NamedFramebufferDrawBuffers_State");
|
||||
if (!framebufferObject) return;
|
||||
DrawBuffersForFramebuffer_State(framebufferObject, framebuffer == 0, n, bufs, false);
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
void NamedFramebufferDrawBuffer_State(GLuint framebuffer, GLenum buf) {
|
||||
@@ -1806,6 +2007,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
const GLenum bufs[] = {buf};
|
||||
DrawBuffersForFramebuffer_State(framebufferObject, framebuffer == 0, 1, bufs, true);
|
||||
}
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
void NamedFramebufferReadBuffer_State(GLuint framebuffer, GLenum src) {
|
||||
@@ -1815,6 +2019,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!framebufferObject) return;
|
||||
ReadBufferForFramebuffer_State(framebufferObject, framebuffer == 0, src,
|
||||
"NamedFramebufferReadBuffer_State");
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
PipePublishFramebufferByName(framebufferObject);
|
||||
#endif
|
||||
}
|
||||
|
||||
SharedPtr<MG_State::GLState::FramebufferObject> GetFramebufferObjectForNamedClear(GLuint framebuffer,
|
||||
@@ -2615,24 +2822,28 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void ClearBufferfi_Backend(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MGP_FILL(ClearBufferfi);
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferfi(buffer, drawbuffer, depth, stencil);
|
||||
}
|
||||
|
||||
void ClearBufferfv_Backend(GLenum buffer, GLint drawbuffer, const GLfloat* value) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MGP_FILL(ClearBufferfv);
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferfv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearBufferuiv_Backend(GLenum buffer, GLint drawbuffer, const GLuint* value) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MGP_FILL(ClearBufferuiv);
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferuiv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearBufferiv_Backend(GLenum buffer, GLint drawbuffer, const GLint* value) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MGP_FILL(ClearBufferiv);
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferiv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
@@ -2880,6 +3091,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void ReadPixels_Backend(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels) {
|
||||
MGP_FILL(ReadPixels);
|
||||
MG_Backend::gBackendFunctionsTable.GL.ReadPixels(x, y, width, height, format, type, pixels);
|
||||
}
|
||||
|
||||
|
||||
@@ -7,7 +7,9 @@
|
||||
// End of Source File Header
|
||||
|
||||
#include "GL_Getter.h"
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
#include <Config.h>
|
||||
#include <MGGitHash.h>
|
||||
#include <MG_Impl/GLImpl/Debug/GL_Debug.h>
|
||||
@@ -28,6 +30,7 @@
|
||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||
#include <MG_Util/ShaderTranspiler/Types.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Impl/Pipe/PipeFill.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
// Declared rather than #included from GL_RenderState.h on purpose: that header also declares
|
||||
@@ -93,8 +96,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// limits they advertise still have to be legal.
|
||||
constexpr GLint kFrontendMaxDebugGroupStackDepth = 64;
|
||||
constexpr GLint kFrontendMaxDebugLoggedMessages = 1;
|
||||
constexpr GLint kFrontendMaxVertexUniformComponents = 4096;
|
||||
constexpr GLint kFrontendMaxVertexUniformVectors = 128;
|
||||
// The *_VECTORS answers are the *_COMPONENTS ones divided by four, never a second
|
||||
// literal: they used to be independent (4096 components against 128 vectors, 64 varying
|
||||
// components against 8 varying vectors) and could not both be describing the same
|
||||
// capacity. Both are shared with BuildTBuiltInResource through Types.h, because
|
||||
// gl_MaxVertexUniformVectors and gl_MaxVaryingVectors expand from the same numbers.
|
||||
constexpr GLint kFrontendMaxVertexUniformComponents =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_VERTEX_UNIFORM_COMPONENTS);
|
||||
constexpr GLint kFrontendMaxVertexUniformVectors =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_VERTEX_UNIFORM_VECTORS);
|
||||
constexpr GLint kFrontendMaxVertexUniformBlocks = 14;
|
||||
constexpr GLint kFrontendMaxVertexOutputComponents = 64;
|
||||
constexpr GLint kFrontendMaxFragmentInputComponents = 128;
|
||||
@@ -106,21 +116,61 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
constexpr GLint kFrontendMaxGeometryTextureImageUnits = 16;
|
||||
constexpr GLint kFrontendMaxGeometryUniformComponents = 1024;
|
||||
constexpr GLint kFrontendMaxGeometryUniformBlocks = 14;
|
||||
constexpr GLint kFrontendMaxCombinedUniformBlocks = kFrontendMaxVertexUniformBlocks +
|
||||
kFrontendMaxGeometryUniformBlocks +
|
||||
kFrontendMaxFragmentUniformBlocks;
|
||||
constexpr GLint kFrontendMaxVaryingComponents = 64;
|
||||
constexpr GLint kFrontendMaxVaryingVectors = 8;
|
||||
// ARB_geometry_shader4's per-invocation count. No TBuiltInResource field and no
|
||||
// gl_MaxGeometryShaderInvocations built-in exists to keep in step, so this is a getter
|
||||
// answer only; 32 is the GL 4.6 core minimum (table 23.57).
|
||||
constexpr GLint kFrontendMaxGeometryShaderInvocations = 32;
|
||||
constexpr GLint kFrontendMaxTessControlUniformBlocks = 14;
|
||||
constexpr GLint kFrontendMaxTessEvaluationUniformBlocks = 14;
|
||||
// The compute stage's share of the combined sum below. Compute's own per-stage answer is
|
||||
// backend-derived (GL_MAX_COMPUTE_UNIFORM_BLOCKS reads dynamicParameters), so this is not
|
||||
// what that query returns - it is the GL 4.3 core minimum, present here only so the
|
||||
// combined total covers all SIX stages.
|
||||
constexpr GLint kFrontendMaxComputeUniformBlocksShare = 14;
|
||||
// GL 4.6 table 23.64 orders MAX_UNIFORM_BUFFER_BINDINGS >= MAX_COMBINED_UNIFORM_BLOCKS >=
|
||||
// every per-stage count, and the sum has to run over SIX stages, not three and not five.
|
||||
// Three (42) was the original bug. Five (70) replaced it and broke the middle term the
|
||||
// other way: compute's per-stage count is backend-derived and clamps at the binding count,
|
||||
// so a device reporting descriptor-indexing-scale uniform buffers (Adreno reports
|
||||
// maxPerStageDescriptorUniformBuffers = 16777216) advertised 84 compute blocks against a
|
||||
// combined 70. Six stages x 14 = 84, which is also exactly the binding-point count and the
|
||||
// arithmetic the GL 4.5 minimum of 84 bindings is built from, so the ordering is now tight
|
||||
// rather than accidental.
|
||||
constexpr GLint kFrontendMaxCombinedUniformBlocks =
|
||||
kFrontendMaxVertexUniformBlocks + kFrontendMaxTessControlUniformBlocks +
|
||||
kFrontendMaxTessEvaluationUniformBlocks + kFrontendMaxGeometryUniformBlocks +
|
||||
kFrontendMaxFragmentUniformBlocks + kFrontendMaxComputeUniformBlocksShare;
|
||||
constexpr GLint kFrontendMaxVaryingComponents =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_VARYING_COMPONENTS);
|
||||
constexpr GLint kFrontendMaxVaryingVectors =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_VARYING_VECTORS);
|
||||
constexpr GLint kFrontendMaxProgramTexelOffset = 7;
|
||||
constexpr GLint kFrontendMinProgramTexelOffset = -8;
|
||||
constexpr GLint kFrontendMaxTransformFeedbackInterleavedComponents = 64;
|
||||
constexpr GLint kFrontendMaxTransformFeedbackSeparateAttribs = 4;
|
||||
constexpr GLint kFrontendMaxTransformFeedbackSeparateComponents = 4;
|
||||
// ARB_transform_feedback3's vertex-stream count. One is what this implementation can
|
||||
// actually emit to; see the GL_MAX_VERTEX_STREAMS case for why it is not four.
|
||||
constexpr GLint kFrontendMaxVertexStreams = 1;
|
||||
constexpr GLint kFrontendMaxGeometryOutputVertices = 256;
|
||||
constexpr GLint kFrontendMaxGeometryTotalOutputComponents = 1024;
|
||||
constexpr GLint kFrontendMinUniformBufferBindings = 36;
|
||||
// GL 4.5 core table 23.64 requires 84 indexed uniform binding points, and that is exactly
|
||||
// how wide the state layer's array is (BufferState::BufferBindingPointCount) - see the
|
||||
// GL_MAX_UNIFORM_BUFFER_BINDINGS case for why the ES driver's own, smaller count is not
|
||||
// the ceiling here.
|
||||
constexpr GLint kFrontendMinUniformBufferBindings = 84;
|
||||
constexpr GLint kFrontendSubpixelBits = 4;
|
||||
constexpr GLint kFrontendMaxSamples = 4;
|
||||
constexpr GLint kFrontendMaxSamples =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MIN_ADVERTISED_MAX_SAMPLES);
|
||||
// ARB_shader_subroutine's two limits. NOTHING IMPLEMENTS SUBROUTINES: there is no
|
||||
// glGetSubroutineIndex / glUniformSubroutinesuiv, only the program-interface enum
|
||||
// plumbing. These are answered - with the GL 4.5 core minimums - because the conformance
|
||||
// suite queries them before it checks for the feature and an INVALID_ENUM both leaves the
|
||||
// caller reading its own uninitialised stack slot and strands an error for the next
|
||||
// unrelated call to trip over. The extension is deliberately NOT advertised, so the
|
||||
// numbers are a table entry, not a capability claim.
|
||||
constexpr GLint kFrontendMaxSubroutines = 256;
|
||||
constexpr GLint kFrontendMaxSubroutineUniformLocations = 1024;
|
||||
|
||||
// The floors under GL_MAX_COMPUTE_WORK_GROUP_COUNT / _SIZE. Shared with the compile
|
||||
// pipeline (CaptureCompileEnv floors the same driver answers at them, and
|
||||
@@ -134,9 +184,19 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return index < 3 ? static_cast<GLint>(MG_Util::ShaderTranspiler::MIN_COMPUTE_WORK_GROUP_SIZE[index]) : 0;
|
||||
}
|
||||
|
||||
// GL 4.6 core table 23.64: components + blocks * (blockSize / 4). The product has to be
|
||||
// formed in 64 bits and saturated on the way out - it overflowed a signed 32-bit int on
|
||||
// every Vulkan host that reports a large maxUniformBufferRange. A Mali driver answering
|
||||
// 0xFFFFFFFF saturates to INT32_MAX in the loader, and 14 * (2147483647 / 4) + 4096 wraps
|
||||
// to -1073737742, which the conformance suite read back as a limit "smaller than 58368".
|
||||
// Saturating instead of wrapping is also the only honest answer: an implementation that
|
||||
// can serve more components than a GLint holds still has to report a GLint.
|
||||
GLint GetMaxCombinedUniformComponents(GLint maxDefaultUniformComponents, GLint maxUniformBlocks,
|
||||
GLint maxUniformBlockSizeBytes) {
|
||||
return maxDefaultUniformComponents + maxUniformBlocks * (maxUniformBlockSizeBytes / 4);
|
||||
const Int64 blocks = std::max<Int64>(static_cast<Int64>(maxUniformBlocks), 0);
|
||||
const Int64 componentsPerBlock = std::max<Int64>(static_cast<Int64>(maxUniformBlockSizeBytes), 0) / 4;
|
||||
const Int64 total = static_cast<Int64>(maxDefaultUniformComponents) + blocks * componentsPerBlock;
|
||||
return static_cast<GLint>(std::min<Int64>(total, std::numeric_limits<GLint>::max()));
|
||||
}
|
||||
|
||||
bool TryDecodeIndexedBufferQuery(GLenum pname, BufferTarget& bufferTarget, IndexedBufferQueryKind& queryKind) {
|
||||
@@ -304,24 +364,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
GLint ResolveDrawFramebufferSampleCount() {
|
||||
const auto& drawFbo =
|
||||
MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||
if (!drawFbo) return 0;
|
||||
|
||||
GLint maxSamples = 0;
|
||||
for (const auto& attachment : drawFbo->GetAllAttachmentObjects()) {
|
||||
if (attachment.IsRenderbuffer() && attachment.GetRenderbuffer()) {
|
||||
maxSamples = std::max(maxSamples, static_cast<GLint>(attachment.GetRenderbuffer()->GetSamples()));
|
||||
} else if (attachment.IsTexture() && attachment.GetTexture()) {
|
||||
// Multisample texture attachments count too (GL_SAMPLE_BUFFERS must
|
||||
// report 1 for any multisampled draw framebuffer).
|
||||
maxSamples = std::max(maxSamples, static_cast<GLint>(attachment.GetTexture()->GetSamples()));
|
||||
}
|
||||
}
|
||||
return maxSamples;
|
||||
}
|
||||
|
||||
void RecordIndexedOnlyGetterError(const char* functionName, GLenum pname) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
@@ -473,10 +515,18 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
} // namespace
|
||||
|
||||
// GL 4.6 core table 23.53 requires GL_MAX_SAMPLES >= 4, so the driver's value is floored
|
||||
// before it is advertised. Every other multisample ceiling MobileGL advertises has to be
|
||||
// floored the same way: promising 4 samples globally while answering GL_MAX_INTEGER_SAMPLES
|
||||
// 1 - which is exactly what Adreno reports - makes the frontend reject the very count it
|
||||
// just told the application to use. The backends clamp the realised count instead.
|
||||
// before it is advertised. gl_MaxSamples expands from the same floored number
|
||||
// (BuildTBuiltInResource), which is also what sizes gl_SampleMask[].
|
||||
//
|
||||
// THE FLOOR STOPS HERE, and that is the point. It used to be applied to
|
||||
// GL_MAX_INTEGER_SAMPLES, GL_MAX_COLOR_TEXTURE_SAMPLES and GL_MAX_DEPTH_TEXTURE_SAMPLES too,
|
||||
// on the reasoning that an application reads GL_MAX_SAMPLES once and hands that count to
|
||||
// every glTexStorage*Multisample. Table 23.53 gives those three a minimum of ONE, and the
|
||||
// reasoning had it backwards: Adreno and Mali back an integer multisample texture with a
|
||||
// single sample, so flooring the query at 4 did not make four samples exist - it made the
|
||||
// backend silently under-allocate (ClampSamplesToBackendSupport) while the application wrote
|
||||
// per-sample data it could never read back. Reporting what was probed turns that into an
|
||||
// honest "unsupported" the application can branch on.
|
||||
GLint GetAdvertisedMaxSamples() {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return kFrontendMaxSamples;
|
||||
@@ -484,6 +534,50 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return std::max(MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxSamples, kFrontendMaxSamples);
|
||||
}
|
||||
|
||||
// GL 4.6 core table 23.53 minimum for the per-category multisample ceilings. One, not four:
|
||||
// see the note on GetAdvertisedMaxSamples. A zero would be a probe that never ran, so it is
|
||||
// floored rather than trusted.
|
||||
namespace {
|
||||
GLint AdvertisedCategoryMaxSamples(Int MG_Backend::DynamicBackendParameters::*categoryLimit) {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return 1;
|
||||
}
|
||||
return std::max(MG_Backend::pActiveBackendObject->GetDynamicParameters().*categoryLimit, 1);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
GLint GetAdvertisedColorTextureMaxSamples() {
|
||||
return AdvertisedCategoryMaxSamples(&MG_Backend::DynamicBackendParameters::MaxColorTextureSamples);
|
||||
}
|
||||
|
||||
GLint GetAdvertisedDepthTextureMaxSamples() {
|
||||
return AdvertisedCategoryMaxSamples(&MG_Backend::DynamicBackendParameters::MaxDepthTextureSamples);
|
||||
}
|
||||
|
||||
GLint GetAdvertisedIntegerMaxSamples() {
|
||||
return AdvertisedCategoryMaxSamples(&MG_Backend::DynamicBackendParameters::MaxIntegerSamples);
|
||||
}
|
||||
|
||||
// Declared in GL_Getter.h, so that the draw path can feed the same number to the reserved
|
||||
// gl_NumSamples stand-in that glGetIntegerv(GL_SAMPLES) reports.
|
||||
GLint ResolveDrawFramebufferSampleCount() {
|
||||
const auto& drawFbo =
|
||||
MG_State::pGLContext->GetFramebufferBindingSlot(FramebufferTarget::Draw).GetBoundObject();
|
||||
if (!drawFbo) return 0;
|
||||
|
||||
GLint maxSamples = 0;
|
||||
for (const auto& attachment : drawFbo->GetAllAttachmentObjects()) {
|
||||
if (attachment.IsRenderbuffer() && attachment.GetRenderbuffer()) {
|
||||
maxSamples = std::max(maxSamples, static_cast<GLint>(attachment.GetRenderbuffer()->GetSamples()));
|
||||
} else if (attachment.IsTexture() && attachment.GetTexture()) {
|
||||
// Multisample texture attachments count too (GL_SAMPLE_BUFFERS must
|
||||
// report 1 for any multisampled draw framebuffer).
|
||||
maxSamples = std::max(maxSamples, static_cast<GLint>(attachment.GetTexture()->GetSamples()));
|
||||
}
|
||||
}
|
||||
return maxSamples;
|
||||
}
|
||||
|
||||
/* @INSERTION_POINT:FUNCTION_IMPLEMENTATION@ */
|
||||
const GLubyte* GetString(GLenum name) {
|
||||
static String vendorString;
|
||||
@@ -680,12 +774,30 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
case GL_MIN_FRAGMENT_INTERPOLATION_OFFSET:
|
||||
case GL_MAX_FRAGMENT_INTERPOLATION_OFFSET:
|
||||
case GL_FRAGMENT_INTERPOLATION_OFFSET_BITS: {
|
||||
case GL_FRAGMENT_INTERPOLATION_OFFSET_BITS:
|
||||
// Same reason as the three above: the integer fallback would round the fraction to 0
|
||||
// or 1 first, so a 0.25 sample-shading rate would answer GL_FALSE.
|
||||
case GL_MIN_SAMPLE_SHADING_VALUE: {
|
||||
GLfloat value = 0.0f;
|
||||
GetFloatv(pname, &value);
|
||||
*params = value != 0.0f ? GL_TRUE : GL_FALSE;
|
||||
return;
|
||||
}
|
||||
// Float-native state, so GL 4.6 core 2.2.2's "zero becomes FALSE, every other value
|
||||
// becomes TRUE" has to be applied to the VALUE. Answering these through the integer getter
|
||||
// below instead - which rounds - reported GL_FALSE for a perfectly non-zero level of 0.25,
|
||||
// and every other float state in this function already reads through GetFloatv for exactly
|
||||
// that reason.
|
||||
case GL_PATCH_DEFAULT_OUTER_LEVEL:
|
||||
case GL_PATCH_DEFAULT_INNER_LEVEL: {
|
||||
const GLsizei componentCount = pname == GL_PATCH_DEFAULT_OUTER_LEVEL ? 4 : 2;
|
||||
GLfloat levels[4] = {};
|
||||
GetFloatv(pname, levels);
|
||||
for (GLsizei i = 0; i < componentCount; ++i) {
|
||||
params[i] = levels[i] != 0.0f ? GL_TRUE : GL_FALSE;
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -735,6 +847,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
params[1] = depthRange.y();
|
||||
return;
|
||||
}
|
||||
// glPatchParameterfv's two states. Float-native, so they are answered here rather than
|
||||
// through the integer fallback below - which rounds, and would report 0 for a level of 0.5.
|
||||
case GL_PATCH_DEFAULT_OUTER_LEVEL: {
|
||||
const FloatVec4& outer = MG_State::pGLContext->GetPatchDefaultOuterLevel();
|
||||
params[0] = outer.x();
|
||||
params[1] = outer.y();
|
||||
params[2] = outer.z();
|
||||
params[3] = outer.w();
|
||||
return;
|
||||
}
|
||||
case GL_PATCH_DEFAULT_INNER_LEVEL: {
|
||||
const FloatVec2& inner = MG_State::pGLContext->GetPatchDefaultInnerLevel();
|
||||
params[0] = inner.x();
|
||||
params[1] = inner.y();
|
||||
return;
|
||||
}
|
||||
case GL_VIEWPORT_BOUNDS_RANGE: {
|
||||
const auto& dynamicParameters = MG_Backend::pActiveBackendObject->GetDynamicParameters();
|
||||
params[0] = dynamicParameters.ViewportBoundsRangeMin;
|
||||
@@ -800,6 +928,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_POLYGON_OFFSET_UNITS:
|
||||
params[0] = MG_State::pGLContext->GetPolygonOffsetUnits();
|
||||
return;
|
||||
case GL_POLYGON_OFFSET_CLAMP:
|
||||
// Float-native state, so it is answered here rather than through the integer
|
||||
// fallback: glPolygonOffsetClamp(1, 1, 0.5) must read back as 0.5, not as 0.
|
||||
params[0] = MG_State::pGLContext->GetPolygonOffsetClamp();
|
||||
return;
|
||||
case GL_SMOOTH_LINE_WIDTH_RANGE: {
|
||||
const auto& dynamicParameters = MG_Backend::pActiveBackendObject->GetDynamicParameters();
|
||||
params[0] = dynamicParameters.SmoothLineWidthRangeMin;
|
||||
@@ -815,6 +948,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_SAMPLE_COVERAGE_VALUE:
|
||||
params[0] = MG_State::pGLContext->GetSampleCoverageValue();
|
||||
return;
|
||||
case GL_MIN_SAMPLE_SHADING_VALUE:
|
||||
// Float state, so it has to be answered here rather than through the integer
|
||||
// fallback: glMinSampleShading(0.5) must read back as 0.5 and not as 0.
|
||||
params[0] = MG_State::pGLContext->GetMinSampleShadingValue();
|
||||
return;
|
||||
case GL_POINT_FADE_THRESHOLD_SIZE:
|
||||
// Float state: read it directly so the fractional part is not lost to the integer path.
|
||||
params[0] = MG_State::pGLContext->GetPointFadeThresholdSize();
|
||||
@@ -1036,6 +1174,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
: GetMinComputeWorkGroupSize(index);
|
||||
GLint backendValue = 0;
|
||||
if (getIntegeri) {
|
||||
MGP_FILL(GetIntegeri_v);
|
||||
getIntegeri(target, index, &backendValue);
|
||||
}
|
||||
*data = std::max(backendValue, minimum);
|
||||
@@ -1186,6 +1325,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
switch (pname) {
|
||||
case GL_MAX_ELEMENT_INDEX:
|
||||
// The largest value a GL_UNSIGNED_INT index may take. It has to be answered HERE and
|
||||
// not left to the 32-bit fallback below: the conformance suite reads it with
|
||||
// glGetInteger64v, and widening the saturated GLint would report INT32_MAX where the
|
||||
// spec requires 2^32-1.
|
||||
params[0] = 0xFFFFFFFFLL;
|
||||
return;
|
||||
case GL_MAX_SHADER_STORAGE_BLOCK_SIZE:
|
||||
if (MG_Backend::pActiveBackendObject) {
|
||||
params[0] = static_cast<GLint64>(
|
||||
@@ -1209,6 +1355,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
Int64 timestamp = 0;
|
||||
if (!MG_Config::Features.DisableTimerQuery) {
|
||||
if (const auto getGpuTimestampNs = MG_Backend::gBackendFunctionsTable.GL.GetGpuTimestampNs) {
|
||||
MGP_FILL(GetGpuTimestampNs);
|
||||
timestamp = getGpuTimestampNs();
|
||||
}
|
||||
}
|
||||
@@ -1222,12 +1369,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLint ints[4] = {};
|
||||
GetIntegerv(pname, ints);
|
||||
|
||||
// GL 4.6 core 22.1 gives glGetInteger64v the same accepted-pname set as glGetIntegerv, so
|
||||
// every pname the integer getter answers with several components owes them all here too.
|
||||
// A pname that reaches the `default:` arm writes params[0] and leaves the caller's other
|
||||
// components holding whatever they held, with no error to say so.
|
||||
switch (pname) {
|
||||
case GL_BLEND_COLOR:
|
||||
case GL_COLOR_CLEAR_VALUE:
|
||||
case GL_COLOR_WRITEMASK:
|
||||
case GL_SCISSOR_BOX:
|
||||
case GL_VIEWPORT:
|
||||
case GL_PATCH_DEFAULT_OUTER_LEVEL:
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
params[i] = static_cast<GLint64>(ints[i]);
|
||||
}
|
||||
@@ -1237,6 +1389,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_VIEWPORT_DIMS:
|
||||
case GL_POINT_SIZE_RANGE:
|
||||
case GL_VIEWPORT_BOUNDS_RANGE:
|
||||
case GL_PATCH_DEFAULT_INNER_LEVEL:
|
||||
params[0] = static_cast<GLint64>(ints[0]);
|
||||
params[1] = static_cast<GLint64>(ints[1]);
|
||||
return;
|
||||
@@ -1268,6 +1421,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_POINT_SIZE_RANGE:
|
||||
case GL_SMOOTH_LINE_WIDTH_RANGE:
|
||||
case GL_MAX_VIEWPORT_DIMS:
|
||||
case GL_PATCH_DEFAULT_INNER_LEVEL:
|
||||
count = 2;
|
||||
break;
|
||||
case GL_BLEND_COLOR:
|
||||
@@ -1275,6 +1429,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_VIEWPORT:
|
||||
case GL_SCISSOR_BOX:
|
||||
case GL_COLOR_WRITEMASK:
|
||||
case GL_PATCH_DEFAULT_OUTER_LEVEL:
|
||||
count = 4;
|
||||
break;
|
||||
default:
|
||||
@@ -1314,6 +1469,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = 0;
|
||||
return;
|
||||
}
|
||||
// GL_TEXTURE_BUFFER_BINDING and GL_TEXTURE_BUFFER are the same token (0x8C2A): as a
|
||||
// glGetIntegerv pname it asks which BUFFER object is bound to the buffer-texture target,
|
||||
// not which texture is (that one is GL_TEXTURE_BINDING_BUFFER, handled by the texture-unit
|
||||
// decoder above).
|
||||
case GL_TEXTURE_BUFFER_BINDING: {
|
||||
auto& obj = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Texture).GetBoundObject();
|
||||
*params = obj ? static_cast<GLint>(obj->GetExternalIndex()) : 0;
|
||||
return;
|
||||
}
|
||||
case GL_BLEND:
|
||||
*params = MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::Blend) ? GL_TRUE : GL_FALSE;
|
||||
return;
|
||||
@@ -1369,6 +1533,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// this single case serves every getter flavor.
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetClampReadColor());
|
||||
return;
|
||||
// glClipControl's two state variables (GL 4.5 core table 23.7). They answer from the
|
||||
// state the entry point records, which is what the conformance suite's initial-value and
|
||||
// set-then-get cases read - the RASTERIZATION half of clip control is a separate,
|
||||
// backend-side question and does not gate the query.
|
||||
case GL_CLIP_ORIGIN:
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetClipOrigin());
|
||||
return;
|
||||
case GL_CLIP_DEPTH_MODE:
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetClipDepthMode());
|
||||
return;
|
||||
case GL_COLOR_CLEAR_VALUE: {
|
||||
const FloatVec4& clearColor = MG_State::pGLContext->GetClearColor();
|
||||
params[0] = static_cast<GLint>(clearColor.x());
|
||||
@@ -1657,6 +1831,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_GEOMETRY_UNIFORM_COMPONENTS:
|
||||
*params = kFrontendMaxGeometryUniformComponents;
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_SHADER_INVOCATIONS:
|
||||
*params = kFrontendMaxGeometryShaderInvocations;
|
||||
return;
|
||||
case GL_MAX_IMAGE_SAMPLES:
|
||||
*params = 0; // multisampled image load/store is not exposed by the DirectGLES frontend
|
||||
return;
|
||||
@@ -1710,6 +1887,59 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params =
|
||||
StageStorageBlockCount(&MG_Backend::DynamicBackendParameters::MaxTessEvaluationShaderStorageBlocks);
|
||||
return;
|
||||
// The tessellation per-stage resource limits. Every one of these is ALSO a GLSL built-in
|
||||
// constant that BuildTBuiltInResource expands, and the two must report the same number
|
||||
// (KHR-GL45.limits.max_tess_* compares them directly) - which is why the values come from
|
||||
// the shared block in MG_Util/ShaderTranspiler/Types.h rather than from literals here.
|
||||
// They were the whole per-stage tess family: the table had been filled in only where the
|
||||
// honest answer was zero (the atomic counters, the image uniforms) or where a driver
|
||||
// query existed (GL_MAX_PATCH_VERTICES, GL_MAX_TESS_GEN_LEVEL), so every pname whose
|
||||
// answer is a real resource count fell through to GL_INVALID_ENUM.
|
||||
case GL_MAX_TESS_CONTROL_INPUT_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_CONTROL_INPUT_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_OUTPUT_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_CONTROL_OUTPUT_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_TOTAL_OUTPUT_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_CONTROL_TOTAL_OUTPUT_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_TEXTURE_IMAGE_UNITS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_CONTROL_TEXTURE_IMAGE_UNITS);
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_UNIFORM_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_CONTROL_UNIFORM_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_INPUT_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_EVALUATION_INPUT_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_OUTPUT_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_EVALUATION_OUTPUT_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_TEXTURE_IMAGE_UNITS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_EVALUATION_TEXTURE_IMAGE_UNITS);
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_UNIFORM_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_EVALUATION_UNIFORM_COMPONENTS);
|
||||
return;
|
||||
case GL_MAX_TESS_PATCH_COMPONENTS:
|
||||
*params = static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_PATCH_COMPONENTS);
|
||||
return;
|
||||
// Routed through the same clamp as every other per-stage block count so the
|
||||
// MAX_UNIFORM_BUFFER_BINDINGS >= MAX_COMBINED_UNIFORM_BLOCKS >= per-stage ordering of
|
||||
// GL 4.6 table 23.64 cannot be broken by the two families moving independently.
|
||||
case GL_MAX_TESS_CONTROL_UNIFORM_BLOCKS:
|
||||
*params = ClampUniformBlockCount(kFrontendMaxTessControlUniformBlocks);
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_UNIFORM_BLOCKS:
|
||||
*params = ClampUniformBlockCount(kFrontendMaxTessEvaluationUniformBlocks);
|
||||
return;
|
||||
case GL_MAX_SUBROUTINES:
|
||||
*params = kFrontendMaxSubroutines;
|
||||
return;
|
||||
case GL_MAX_SUBROUTINE_UNIFORM_LOCATIONS:
|
||||
*params = kFrontendMaxSubroutineUniformLocations;
|
||||
return;
|
||||
case GL_MAX_TEXTURE_LOD_BIAS:
|
||||
*params = 15; // TODO
|
||||
return;
|
||||
@@ -1755,8 +1985,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_NUM_PROGRAM_BINARY_FORMATS:
|
||||
*params = 0;
|
||||
return;
|
||||
// GL_ARB_spirv_extensions / GL 4.6 core 22.2. An implementation that advertises no
|
||||
// SPIR-V extension answers zero here, and glGetStringi(GL_SPIR_V_EXTENSIONS, i) is then
|
||||
// never legally called - MobileGL runs the module through its own translation pipeline
|
||||
// and relies on no SPIR-V extension to do it, so zero is the true answer rather than a
|
||||
// placeholder.
|
||||
case GL_NUM_SPIR_V_EXTENSIONS:
|
||||
*params = 0;
|
||||
return;
|
||||
// GL_ARB_gl_spirv, core since 4.6: exactly one shader binary format, and the pair has to
|
||||
// agree - an application sizes its GL_SHADER_BINARY_FORMATS array from the count.
|
||||
case GL_NUM_SHADER_BINARY_FORMATS:
|
||||
*params = 0; // ShaderBinary entrypoints are stubbed
|
||||
*params = 1;
|
||||
return;
|
||||
case GL_SHADER_BINARY_FORMATS:
|
||||
*params = static_cast<GLint>(GL_SHADER_BINARY_FORMAT_SPIR_V);
|
||||
return;
|
||||
case GL_PACK_ALIGNMENT:
|
||||
*params = MG_State::pGLContext->GetPixelStoreParam(PixelStoreParam::PackAlignment);
|
||||
@@ -1815,6 +2058,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_PRIMITIVE_RESTART_INDEX:
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetPrimitiveRestartIndex());
|
||||
return;
|
||||
case GL_POLYGON_OFFSET_CLAMP:
|
||||
// Float state (see GetFloatv); rounded to nearest for the integer query per GL 4.6
|
||||
// core 22.1's float-to-integer rule.
|
||||
*params = static_cast<GLint>(std::lround(MG_State::pGLContext->GetPolygonOffsetClamp()));
|
||||
return;
|
||||
case GL_PROGRAM_BINARY_FORMATS:
|
||||
*params = 0; // program-binary entrypoints are stubbed
|
||||
return;
|
||||
@@ -1900,6 +2148,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_SAMPLE_MASK:
|
||||
*params = MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::SampleMask) ? GL_TRUE : GL_FALSE;
|
||||
return;
|
||||
case GL_SAMPLE_SHADING:
|
||||
*params = MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::SampleShading) ? GL_TRUE : GL_FALSE;
|
||||
return;
|
||||
case GL_MIN_SAMPLE_SHADING_VALUE:
|
||||
// GL 4.6 core 22.2: a floating-point value queried as an integer rounds to nearest.
|
||||
*params = static_cast<GLint>(std::lround(MG_State::pGLContext->GetMinSampleShadingValue()));
|
||||
return;
|
||||
case GL_SAMPLE_MASK_VALUE:
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetSampleMaskValue());
|
||||
return;
|
||||
@@ -2011,6 +2266,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
Int64 timestamp = 0;
|
||||
if (!MG_Config::Features.DisableTimerQuery) {
|
||||
if (const auto getGpuTimestampNs = MG_Backend::gBackendFunctionsTable.GL.GetGpuTimestampNs) {
|
||||
MGP_FILL(GetGpuTimestampNs);
|
||||
timestamp = getGpuTimestampNs();
|
||||
}
|
||||
}
|
||||
@@ -2118,7 +2374,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
case GL_MAX_ELEMENT_INDEX:
|
||||
*params = 1024 * 1024; // TODO
|
||||
// 64-bit state (see GetInteger64v); the 32-bit query saturates, per the GL
|
||||
// state-query conversion rules - the same shape GL_MAX_SHADER_STORAGE_BLOCK_SIZE
|
||||
// uses. The real answer is 2^32-1 because both backends draw with GL_UNSIGNED_INT
|
||||
// indices and neither bounds an index value; the old `1024 * 1024` was a placeholder
|
||||
// that no draw path ever consulted.
|
||||
*params = INT32_MAX;
|
||||
return;
|
||||
case GL_CONTEXT_PROFILE_MASK:
|
||||
// Reports the requested context profile (EGL defaults 3.x contexts to core);
|
||||
@@ -2174,8 +2435,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = dynamicParameters.MaxComputeTextureImageUnits;
|
||||
break;
|
||||
case GL_MAX_COMBINED_COMPUTE_UNIFORM_COMPONENTS:
|
||||
// The CLAMPED block count, i.e. exactly what GL_MAX_COMPUTE_UNIFORM_BLOCKS answers.
|
||||
// GL 4.6 table 23.64 defines this as the components reachable through the blocks a
|
||||
// stage may declare, so deriving it from the raw backend number described 256 blocks
|
||||
// an application is only ever allowed 84 of.
|
||||
*params = GetMaxCombinedUniformComponents(kFrontendMaxComputeUniformComponents,
|
||||
dynamicParameters.MaxComputeUniformBlocks,
|
||||
ClampUniformBlockCount(dynamicParameters.MaxComputeUniformBlocks),
|
||||
dynamicParameters.MaxUniformBlockSize);
|
||||
break;
|
||||
case GL_MAX_COMPUTE_WORK_GROUP_INVOCATIONS:
|
||||
@@ -2219,16 +2484,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = static_cast<GLint>(dynamicParameters.ViewportIndexProvokingVertex);
|
||||
break;
|
||||
case GL_MAX_COLOR_TEXTURE_SAMPLES:
|
||||
*params = std::max(dynamicParameters.MaxColorTextureSamples, GetAdvertisedMaxSamples());
|
||||
*params = GetAdvertisedColorTextureMaxSamples();
|
||||
break;
|
||||
case GL_MAX_COMBINED_FRAGMENT_UNIFORM_COMPONENTS:
|
||||
*params = GetMaxCombinedUniformComponents(kFrontendMaxFragmentUniformComponents,
|
||||
kFrontendMaxFragmentUniformBlocks,
|
||||
ClampUniformBlockCount(kFrontendMaxFragmentUniformBlocks),
|
||||
dynamicParameters.MaxUniformBlockSize);
|
||||
break;
|
||||
case GL_MAX_COMBINED_GEOMETRY_UNIFORM_COMPONENTS:
|
||||
*params = GetMaxCombinedUniformComponents(kFrontendMaxGeometryUniformComponents,
|
||||
kFrontendMaxGeometryUniformBlocks,
|
||||
ClampUniformBlockCount(kFrontendMaxGeometryUniformBlocks),
|
||||
dynamicParameters.MaxUniformBlockSize);
|
||||
break;
|
||||
case GL_MAX_GEOMETRY_OUTPUT_VERTICES:
|
||||
@@ -2242,14 +2507,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
break;
|
||||
case GL_MAX_COMBINED_VERTEX_UNIFORM_COMPONENTS:
|
||||
*params = GetMaxCombinedUniformComponents(kFrontendMaxVertexUniformComponents,
|
||||
kFrontendMaxVertexUniformBlocks,
|
||||
ClampUniformBlockCount(kFrontendMaxVertexUniformBlocks),
|
||||
dynamicParameters.MaxUniformBlockSize);
|
||||
break;
|
||||
case GL_MAX_CUBE_MAP_TEXTURE_SIZE:
|
||||
*params = dynamicParameters.MaxCubeMapTextureSize;
|
||||
break;
|
||||
case GL_MAX_DEPTH_TEXTURE_SAMPLES:
|
||||
*params = std::max(dynamicParameters.MaxDepthTextureSamples, GetAdvertisedMaxSamples());
|
||||
*params = GetAdvertisedDepthTextureMaxSamples();
|
||||
break;
|
||||
case GL_MAX_FRAMEBUFFER_WIDTH:
|
||||
*params = dynamicParameters.MaxFramebufferWidth;
|
||||
@@ -2276,7 +2541,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = dynamicParameters.MaxComputeImageUniforms;
|
||||
break;
|
||||
case GL_MAX_INTEGER_SAMPLES:
|
||||
*params = std::max(dynamicParameters.MaxIntegerSamples, GetAdvertisedMaxSamples());
|
||||
*params = GetAdvertisedIntegerMaxSamples();
|
||||
break;
|
||||
case GL_MAX_RENDERBUFFER_SIZE:
|
||||
*params = dynamicParameters.MaxRenderbufferSize;
|
||||
@@ -2287,12 +2552,56 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_PATCH_VERTICES:
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetPatchVertices());
|
||||
break;
|
||||
// Float state, so glGetIntegerv rounds it (GL 4.6 core 2.2.2) - the exact values come back
|
||||
// through glGetFloatv. Answered here so glGetBooleanv, which delegates to this getter for
|
||||
// everything its own switch does not handle, does not report INVALID_ENUM for them.
|
||||
case GL_PATCH_DEFAULT_OUTER_LEVEL: {
|
||||
const FloatVec4& outer = MG_State::pGLContext->GetPatchDefaultOuterLevel();
|
||||
for (Uint i = 0; i < 4; ++i) params[i] = static_cast<GLint>(std::lround(outer[i]));
|
||||
break;
|
||||
}
|
||||
case GL_PATCH_DEFAULT_INNER_LEVEL: {
|
||||
const FloatVec2& inner = MG_State::pGLContext->GetPatchDefaultInnerLevel();
|
||||
for (Uint i = 0; i < 2; ++i) params[i] = static_cast<GLint>(std::lround(inner[i]));
|
||||
break;
|
||||
}
|
||||
// GL 4.6 core table 23.66: whether the primitive-restart index terminates a patch.
|
||||
// GL_FALSE is a legal answer and the true one - neither backend cuts a patch short, and
|
||||
// the DirectVulkan draw path relies on this staying false (it resolves primitive restart
|
||||
// to "never" for a PATCH_LIST topology on the strength of it).
|
||||
case GL_PRIMITIVE_RESTART_FOR_PATCHES_SUPPORTED:
|
||||
*params = GL_FALSE;
|
||||
break;
|
||||
case GL_MAX_PATCH_VERTICES:
|
||||
*params = dynamicParameters.MaxPatchVertices;
|
||||
break;
|
||||
case GL_MAX_TESS_GEN_LEVEL:
|
||||
*params = dynamicParameters.MaxTessGenLevel;
|
||||
break;
|
||||
// Same helper, and so the same arithmetic, as every other GL_MAX_COMBINED_*_UNIFORM_
|
||||
// COMPONENTS: default-block components + blocks * (block size / 4). It reproduces the
|
||||
// conformance suite's own formula exactly, so the two cannot drift.
|
||||
case GL_MAX_COMBINED_TESS_CONTROL_UNIFORM_COMPONENTS:
|
||||
*params = GetMaxCombinedUniformComponents(
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_CONTROL_UNIFORM_COMPONENTS),
|
||||
ClampUniformBlockCount(kFrontendMaxTessControlUniformBlocks), dynamicParameters.MaxUniformBlockSize);
|
||||
break;
|
||||
case GL_MAX_COMBINED_TESS_EVALUATION_UNIFORM_COMPONENTS:
|
||||
*params = GetMaxCombinedUniformComponents(
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_TESS_EVALUATION_UNIFORM_COMPONENTS),
|
||||
ClampUniformBlockCount(kFrontendMaxTessEvaluationUniformBlocks), dynamicParameters.MaxUniformBlockSize);
|
||||
break;
|
||||
// ARB_cull_distance. Backend-derived exactly like GL_MAX_CLIP_DISTANCES beside it, and
|
||||
// for a stronger reason: a cull distance discards the whole primitive, so advertising
|
||||
// eight the rasterizer cannot serve turns every culling draw into a silent no-op. Zero is
|
||||
// the honest answer on a host with no cull-distance route, and the conformance suite then
|
||||
// skips the functional cases instead of failing them deep inside a pixel comparison.
|
||||
case GL_MAX_CULL_DISTANCES:
|
||||
*params = dynamicParameters.MaxCullDistances;
|
||||
break;
|
||||
case GL_MAX_COMBINED_CLIP_AND_CULL_DISTANCES:
|
||||
*params = dynamicParameters.MaxCombinedClipAndCullDistances;
|
||||
break;
|
||||
case GL_MIN_PROGRAM_TEXTURE_GATHER_OFFSET:
|
||||
*params = dynamicParameters.MinProgramTextureGatherOffset;
|
||||
break;
|
||||
@@ -2343,7 +2652,25 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxTransformFeedbackSeparateAttribs;
|
||||
break;
|
||||
case GL_MAX_VERTEX_STREAMS:
|
||||
*params = 1;
|
||||
// ONE, which is under the GL 4.5 core table 23.62 minimum of four and is a known,
|
||||
// deliberate non-conformance. It was briefly raised to 4 on the theory that streams
|
||||
// 1..3 could exist and be permanently empty; measuring that decision refuted it.
|
||||
// Raising the limit un-gates two CTS cases per package across KHR-GL40..GL46 -
|
||||
// transform_feedback.draw_xfb_stream_test (which stops being skipped) and
|
||||
// transform_feedback3.multiple_streams (which stops reporting NotSupported) - and
|
||||
// both then fail, because nothing in the shader pipeline supports layout(stream = N),
|
||||
// EmitStreamVertex or EndStreamPrimitive, and because the query state machine tracks
|
||||
// one active query per TARGET rather than per (target, stream). That is 14 new
|
||||
// failures against 2 gained limits passes, and a 4 nothing can back is the
|
||||
// advertised-caps lie with the sign flipped.
|
||||
//
|
||||
// The real fix is the feature, not the number: per-stream capture needs
|
||||
// layout(stream = N) through the transpiler plus per-(target, stream) query slots,
|
||||
// which DirectVulkan could back with VK_EXT_transform_feedback's geometryStreams and
|
||||
// DirectGLES cannot back at all (ES has no vertex streams). Until that lands, one is
|
||||
// the honest count and every stream-addressing entry point bounds itself by THIS
|
||||
// query, so raising it later moves them all together.
|
||||
*params = kFrontendMaxVertexStreams;
|
||||
break;
|
||||
case GL_TRANSFORM_FEEDBACK_ACTIVE:
|
||||
*params = MG_State::pGLContext->IsTransformFeedbackActive() ? 1 : 0;
|
||||
@@ -2360,15 +2687,36 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_TEXTURE_SIZE:
|
||||
*params = dynamicParameters.MaxTextureSize;
|
||||
break;
|
||||
case GL_MAX_UNIFORM_BUFFER_BINDINGS:
|
||||
case GL_MAX_UNIFORM_BUFFER_BINDINGS: {
|
||||
// Never advertise more bindings than the state layer's indexed-binding array can track
|
||||
// (BufferState::BufferBindingPointCount): glBindBufferBase rejects indices past that
|
||||
// capacity, and the GL CTS per-case state reset calls glBindBufferBase on every
|
||||
// advertised index and expects no error. The floor equals the GL 3.3 core minimum
|
||||
// (36), so the clamp never under-advertises.
|
||||
// advertised index and expects no error. The floor is the GL 4.5 core minimum, and
|
||||
// the array was widened to exactly it, so the two coincide by construction.
|
||||
//
|
||||
// WHY THE BACKEND'S OWN COUNT IS NOT THE CEILING HERE, unlike the shader-storage
|
||||
// family. A GL uniform binding point is where an APPLICATION parks a buffer; it is
|
||||
// not a driver binding point. Neither backend forwards it as one on the draw path:
|
||||
// DirectGLES rebinds the blocks a program declares onto COMPACTED ES points
|
||||
// (BindCurrentProgramWithResources maps block i to ES point i+1) and DirectVulkan
|
||||
// resolves each block to a descriptor. So what the host driver's count bounds is how
|
||||
// many blocks ONE PROGRAM may use, not how many points an application may bind.
|
||||
//
|
||||
// That per-program number is NOT GL_MAX_COMBINED_UNIFORM_BLOCKS (84, the six-stage
|
||||
// sum): no single program can reach it. A graphics program is bounded by the five
|
||||
// graphics stages' per-stage counts, 14 each, so 70 blocks plus the global UBO at ES
|
||||
// point 0 = 71 - inside the ES 3.2 minimum of 72. A compute program is bounded by
|
||||
// GL_MAX_COMPUTE_UNIFORM_BLOCKS, which on DirectGLES is the ES driver's own count
|
||||
// (GL-scale, ~14) and on DirectVulkan is served from descriptors with no ES binding
|
||||
// points involved. Raising any per-stage graphics count past 14 is what would break
|
||||
// this, so that is the edit to check against the ES ceiling - not this one.
|
||||
static_assert(static_cast<GLint>(MG_State::GLState::BufferBindingPointCount) >=
|
||||
kFrontendMinUniformBufferBindings,
|
||||
"the indexed-binding array must be able to hold every advertised uniform binding point");
|
||||
*params = std::clamp(dynamicParameters.MaxUniformBufferBindings, kFrontendMinUniformBufferBindings,
|
||||
static_cast<GLint>(MG_State::GLState::BufferBindingPointCount));
|
||||
break;
|
||||
}
|
||||
case GL_MAX_UNIFORM_BLOCK_SIZE:
|
||||
*params = dynamicParameters.MaxUniformBlockSize;
|
||||
break;
|
||||
@@ -2406,7 +2754,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = static_cast<GLint>(dynamicParameters.PointSizeGranularity);
|
||||
break;
|
||||
case GL_SHADER_STORAGE_BUFFER_OFFSET_ALIGNMENT:
|
||||
*params = static_cast<GLint>(dynamicParameters.UniformBufferOffsetAlignment);
|
||||
// The STORAGE alignment, which is its own limit - this used to answer with the
|
||||
// uniform one. They differ on real hardware (Adreno 830: 32 uniform, 64 storage), and
|
||||
// under-reporting it is silent: ValidateBindBufferRange accepts the offset, the ES
|
||||
// driver accepts it too without raising an error, and the shader's writes then land
|
||||
// at an address the application never bound.
|
||||
*params = static_cast<GLint>(dynamicParameters.ShaderStorageBufferOffsetAlignment);
|
||||
break;
|
||||
case GL_SMOOTH_LINE_WIDTH_RANGE:
|
||||
params[0] = static_cast<GLint>(dynamicParameters.SmoothLineWidthRangeMin);
|
||||
|
||||
@@ -25,7 +25,24 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLenum GetError();
|
||||
GLenum GetGraphicsResetStatus();
|
||||
// The GL_MAX_SAMPLES value MobileGL advertises, i.e. the driver's value floored to the GL
|
||||
// core minimum. Frontend multisample validators have to honour this ceiling for every
|
||||
// format, otherwise MobileGL rejects a sample count it advertised itself.
|
||||
// core minimum of 4. This is the RENDERBUFFER ceiling; the three per-category texture
|
||||
// ceilings below have a minimum of one and are reported as probed.
|
||||
GLint GetAdvertisedMaxSamples();
|
||||
// Exactly what GL_MAX_COLOR_TEXTURE_SAMPLES / GL_MAX_DEPTH_TEXTURE_SAMPLES /
|
||||
// GL_MAX_INTEGER_SAMPLES report: the probed backend limit floored at the GL 4.6 core minimum
|
||||
// of ONE (table 23.53). Exported so the frontend's storage validation enforces exactly what
|
||||
// the query promised - it used to floor both at 4 and then let the backend quietly
|
||||
// under-allocate whatever the driver could not actually provide.
|
||||
GLint GetAdvertisedColorTextureMaxSamples();
|
||||
GLint GetAdvertisedDepthTextureMaxSamples();
|
||||
GLint GetAdvertisedIntegerMaxSamples();
|
||||
// What glGetIntegerv(GL_SAMPLES) answers for the CURRENT draw framebuffer: the largest sample
|
||||
// count over its attachments, and 0 for a single-sample or default framebuffer (GL 4.6 core
|
||||
// 9.2.3 / 22.2 - GL_SAMPLE_BUFFERS is 1 exactly when this is non-zero).
|
||||
//
|
||||
// Shared rather than duplicated because two callers need the identical number and disagreeing
|
||||
// would be a silent bug: the query itself, and the draw path's write of the reserved
|
||||
// gl_NumSamples stand-in - a shader comparing gl_NumSamples against glGetIntegerv(GL_SAMPLES)
|
||||
// is exactly what the sample_variables CTS does.
|
||||
GLint ResolveDrawFramebufferSampleCount();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -11,6 +11,8 @@
|
||||
#include "Config.h"
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
#include <set>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
#include <MG_Impl/GLImpl/VertexArray/Validators.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
@@ -19,6 +21,7 @@
|
||||
#include <MG_Util/Converters/SPIRVCrossToGL/SpvcTypeConverter.h>
|
||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Impl/Pipe/PipeFill.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
// The flattened uniform type these helpers used to take as a raw glslang::TType*
|
||||
@@ -30,10 +33,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
static bool CheckShaderNameValidity(Uint shader) {
|
||||
if (shader == 0 || !MG_State::pGLContext->ValidateShaderName(shader)) {
|
||||
// The mirror of CheckProgramNameValidity below, and for the same reason: programs and
|
||||
// shaders are drawn from ONE name space (ProgramState hands both out of a single
|
||||
// generator), so a name that exists but belongs to a PROGRAM is the wrong kind of
|
||||
// object - GL 3.3 core 2.11.x makes that INVALID_OPERATION - while a name GL never
|
||||
// handed out is INVALID_VALUE. This half of the split was missing, so every shader
|
||||
// entry point handed a program name reported INVALID_VALUE; the conformance suite
|
||||
// reads exactly that code back from glSpecializeShader.
|
||||
const ErrorCode error = (shader != 0 && MG_State::pGLContext->ValidateProgramName(shader))
|
||||
? ErrorCode::InvalidOperation
|
||||
: ErrorCode::InvalidValue;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
error,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
std::to_string(shader) + " is not a valid name."));
|
||||
std::to_string(shader) +
|
||||
(error == ErrorCode::InvalidOperation ? " is not a shader object."
|
||||
: " is not a valid name.")));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
@@ -245,6 +260,30 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
// GL 4.6 core 7.6.3: INVALID_VALUE when uniformBlockBinding >= MAX_UNIFORM_BUFFER_BINDINGS.
|
||||
// The storage-block twin below has always had this check; the uniform one never did, and the
|
||||
// value it stores is used as a RAW SUBSCRIPT into the state layer's fixed indexed-binding
|
||||
// array on every draw and dispatch (DirectGLES's per-program UBO rebind, DirectVulkan's
|
||||
// descriptor resolve, whose only guard is a MOBILEGL_ASSERT that compiles away in release).
|
||||
// An out-of-range binding therefore did not merely go unreported - it read past the array and
|
||||
// dereferenced whatever SharedPtr it found there.
|
||||
bool ValidateUniformBlockBinding(GLuint binding) {
|
||||
// Exactly what glGetIntegerv(GL_MAX_UNIFORM_BUFFER_BINDINGS) advertises: the state
|
||||
// layer's array width, which the getter clamps to as well.
|
||||
const SizeT maxBindingCount = MG_State::pGLContext->GetBufferBindingPointCount(BufferTarget::Uniform);
|
||||
if (binding < maxBindingCount) {
|
||||
return true;
|
||||
}
|
||||
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
std::format("Uniform block binding {} is not less than GL_MAX_UNIFORM_BUFFER_BINDINGS ({}).", binding,
|
||||
maxBindingCount)));
|
||||
return false;
|
||||
}
|
||||
|
||||
bool ValidateShaderStorageBlockBinding(GLuint binding) {
|
||||
SizeT maxBindingCount = MG_State::pGLContext->GetBufferBindingPointCount(BufferTarget::ShaderStorage);
|
||||
if (MG_Backend::pActiveBackendObject) {
|
||||
@@ -307,9 +346,195 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void CompileShader_State(GLuint shader) {
|
||||
auto& shaderObject = TryToGetShaderObject(shader);
|
||||
if (!shaderObject) return;
|
||||
// ARB_gl_spirv: "INVALID_OPERATION is generated by CompileShader if shader has been
|
||||
// associated with a SPIR-V binary". Such an object has no GLSL source to compile - it is
|
||||
// waiting for glSpecializeShader, which is the operation that compiles it.
|
||||
if (shaderObject->HasSpirvBinary()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
"shader " + std::to_string(shader) +
|
||||
" holds a SPIR-V binary; use glSpecializeShader instead of glCompileShader."));
|
||||
return;
|
||||
}
|
||||
shaderObject->Compile();
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// GL_ARB_gl_spirv
|
||||
// ---------------------------------------------------------------------------------------
|
||||
|
||||
void ShaderBinary_State(GLsizei count, const GLuint* shaders, GLenum binaryformat, const void* binary,
|
||||
GLsizei length) {
|
||||
if (count < 0 || length < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "count and length must be non-negative."));
|
||||
return;
|
||||
}
|
||||
// GL_NUM_SHADER_BINARY_FORMATS advertises exactly one format, so every other value is
|
||||
// INVALID_ENUM (GL 4.6 core 7.2). This is the check that used to be missing entirely -
|
||||
// the entry point was a silent stub, so an application handed a format nothing supports
|
||||
// and was told nothing.
|
||||
if (binaryformat != GL_SHADER_BINARY_FORMAT_SPIR_V) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"binaryformat must be GL_SHADER_BINARY_FORMAT_SPIR_V."));
|
||||
return;
|
||||
}
|
||||
if (count == 0) return;
|
||||
if (shaders == nullptr || (length > 0 && binary == nullptr)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "shaders and binary must not be null."));
|
||||
return;
|
||||
}
|
||||
// A SPIR-V module is a sequence of 32-bit words, so a length that is not a multiple of
|
||||
// four cannot be one (ARB_gl_spirv makes this INVALID_VALUE).
|
||||
if ((length % 4) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"length must be a multiple of four for a SPIR-V module."));
|
||||
return;
|
||||
}
|
||||
|
||||
// EVERY name is validated before ANY of them is written: the entry point is all-or-
|
||||
// nothing, and half-applying it would leave some objects holding a module the call was
|
||||
// rejected for. The duplicate check is the extension's own ("INVALID_VALUE ... if the
|
||||
// same shader object is specified more than once").
|
||||
std::set<GLuint> seen;
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
if (!seen.insert(shaders[i]).second) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"shader " + std::to_string(shaders[i]) +
|
||||
" appears more than once in `shaders`."));
|
||||
return;
|
||||
}
|
||||
if (!MG_State::pGLContext->ValidateShaderName(shaders[i])) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
std::to_string(shaders[i]) + " is not the name of a shader object."));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT wordCount = static_cast<SizeT>(length) / 4;
|
||||
Vector<Uint32> module(wordCount);
|
||||
if (wordCount != 0) {
|
||||
Memcpy(module.data(), binary, static_cast<SizeT>(length));
|
||||
}
|
||||
// spirv-val here, not at glSpecializeShader: this is where the words arrive, and past it
|
||||
// they reach SPIRV-Cross, which parses rather than validates. ARB_gl_spirv lets an
|
||||
// implementation reject an invalid module at either call; rejecting at the earlier one
|
||||
// means the application's error is reported next to the data that caused it.
|
||||
if (const auto validated = MG_Util::ShaderTranspiler::ShaderCompiler::ValidateSpirvModule(module);
|
||||
!validated) {
|
||||
MGLOG_D("%s: rejected SPIR-V module: %s", __func__, validated.error().log.c_str());
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, validated.error().log));
|
||||
return;
|
||||
}
|
||||
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
auto& shaderObject = TryToGetShaderObject(shaders[i]);
|
||||
if (!shaderObject) continue;
|
||||
// A copy per object, not a shared buffer: each shader object may be specialized with
|
||||
// different constants, and each specialization re-reads its own original words.
|
||||
Vector<Uint32> perObject = module;
|
||||
shaderObject->SetSpirvBinary(Move(perObject));
|
||||
}
|
||||
}
|
||||
|
||||
void SpecializeShader_State(GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants,
|
||||
const GLuint* pConstantIndex, const GLuint* pConstantValue) {
|
||||
auto& shaderObject = TryToGetShaderObject(shader);
|
||||
if (!shaderObject) return;
|
||||
if (!shaderObject->HasSpirvBinary()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"shader " + std::to_string(shader) +
|
||||
" has no SPIR-V binary; call glShaderBinary first."));
|
||||
return;
|
||||
}
|
||||
// ARB_gl_spirv: a shader that has already been specialized may not be specialized again
|
||||
// until glShaderBinary re-associates a module with it.
|
||||
if (shaderObject->HasBeenSpecialized()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"shader " + std::to_string(shader) +
|
||||
" has already been specialized; re-associate its module with "
|
||||
"glShaderBinary before specializing it again."));
|
||||
return;
|
||||
}
|
||||
// pEntryPoint names the entry point to specialize; there is no default. A null pointer
|
||||
// cannot name one, and neither can the empty string.
|
||||
if (pEntryPoint == nullptr || *pEntryPoint == '\0') {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "pEntryPoint must name an entry point."));
|
||||
return;
|
||||
}
|
||||
if (numSpecializationConstants > 0 && (pConstantIndex == nullptr || pConstantValue == nullptr)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"pConstantIndex and pConstantValue must not be null."));
|
||||
return;
|
||||
}
|
||||
// "INVALID_VALUE is generated if any value in pConstantIndex is repeated" - checked before
|
||||
// anything is applied, for the same all-or-nothing reason glShaderBinary checks its names
|
||||
// up front.
|
||||
Vector<Uint32> constantIds(pConstantIndex, pConstantIndex + numSpecializationConstants);
|
||||
Vector<Uint32> constantValues(pConstantValue, pConstantValue + numSpecializationConstants);
|
||||
{
|
||||
std::set<Uint32> seen;
|
||||
for (const Uint32 id : constantIds) {
|
||||
if (seen.insert(id).second) continue;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"constant index " + std::to_string(id) + " is repeated."));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
const String entryPoint(pEntryPoint);
|
||||
const GLenum shaderType = MG_Util::ConvertShaderStageToGLEnum(shaderObject->GetShaderStage());
|
||||
using SpecializationFailure = MG_Util::ShaderTranspiler::ShaderCompiler::SpecializationFailure;
|
||||
SpecializationFailure failure = SpecializationFailure::None;
|
||||
auto specialized = MG_Util::ShaderTranspiler::ShaderCompiler::SpecializeAndDecompileSpirvModule(
|
||||
shaderObject->GetSpirvBinary(), shaderType, entryPoint, constantIds, constantValues, failure);
|
||||
if (!specialized) {
|
||||
MGLOG_D("%s: specialization failed for shader %u: %s", __func__, shader,
|
||||
specialized.error().log.c_str());
|
||||
// The two conditions ARB_gl_spirv ENUMERATES are GL errors, and an erroring GL command
|
||||
// must have no other effect - so the shader object is left exactly as it was rather
|
||||
// than being pushed into a failed-compile state. Anything else is a genuine compile
|
||||
// failure of a well-formed request, which the extension routes through COMPILE_STATUS
|
||||
// and the info log exactly as glCompileShader does.
|
||||
if (failure == SpecializationFailure::UnknownEntryPoint ||
|
||||
failure == SpecializationFailure::UnknownConstantId) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, specialized.error().log));
|
||||
return;
|
||||
}
|
||||
shaderObject->RecordSpecializationFailure(String(specialized.error().log));
|
||||
return;
|
||||
}
|
||||
shaderObject->SpecializeFromSpirv(Move(specialized.value().glsl), Move(specialized.value().xfbVaryings),
|
||||
specialized.value().xfbBufferMode);
|
||||
}
|
||||
|
||||
// glMaxShaderCompilerThreadsKHR / glMaxShaderCompilerThreadsARB - one implementation,
|
||||
// because GL_KHR_parallel_shader_compile and GL_ARB_parallel_shader_compile define the
|
||||
// same entry point with the same semantics and GetProcAddress.cpp maps both spellings.
|
||||
@@ -744,12 +969,77 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = programObject->GetBinaryRetrievableHint() ? GL_TRUE : GL_FALSE;
|
||||
break;
|
||||
case GL_PROGRAM_SEPARABLE:
|
||||
*params = programObject->GetSeparable() ? GL_TRUE : GL_FALSE;
|
||||
// The LATCHED flag, not the live one: glProgramParameteri's write takes effect at the
|
||||
// next link (GL 4.6 core 7.3), so a program told to be separable and then never
|
||||
// linked still reports GL_FALSE.
|
||||
*params = programObject->GetLinkedSeparable() ? GL_TRUE : GL_FALSE;
|
||||
break;
|
||||
|
||||
// The geometry and tessellation link properties (GL 4.6 core table 23.35). Same shape as
|
||||
// GL_COMPUTE_WORK_GROUP_SIZE above, and for the same reason: "a linked program object
|
||||
// with a geometry shader" is one whose EXECUTABLE has the stage, so an
|
||||
// attached-but-not-yet-linked shader must give INVALID_OPERATION rather than the previous
|
||||
// link's value. The geometry three used to be listed here only to fall through into the
|
||||
// INVALID_ENUM default, and the tessellation five were not listed at all.
|
||||
case GL_GEOMETRY_VERTICES_OUT:
|
||||
case GL_GEOMETRY_INPUT_TYPE:
|
||||
case GL_GEOMETRY_OUTPUT_TYPE:
|
||||
case GL_GEOMETRY_SHADER_INVOCATIONS: {
|
||||
if (!programObject->GetLinkStatus() || !programObject->HasLinkedShaderStage(ShaderStage::Geometry)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
std::to_string(program) +
|
||||
" is not a linked program object with a geometry shader."));
|
||||
return;
|
||||
}
|
||||
switch (pname) {
|
||||
case GL_GEOMETRY_VERTICES_OUT: *params = programObject->GetGeometryVerticesOut(); break;
|
||||
case GL_GEOMETRY_INPUT_TYPE: *params = static_cast<GLint>(programObject->GetGeometryInputType()); break;
|
||||
case GL_GEOMETRY_OUTPUT_TYPE: *params = static_cast<GLint>(programObject->GetGeometryOutputType()); break;
|
||||
default: *params = programObject->GetGeometryShaderInvocations(); break;
|
||||
}
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
}
|
||||
case GL_TESS_CONTROL_OUTPUT_VERTICES: {
|
||||
if (!programObject->GetLinkStatus() || !programObject->HasLinkedShaderStage(ShaderStage::TessControl)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
std::to_string(program) +
|
||||
" is not a linked program object with a tessellation control shader."));
|
||||
return;
|
||||
}
|
||||
*params = programObject->GetTessControlOutputVertices();
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
}
|
||||
case GL_TESS_GEN_MODE:
|
||||
case GL_TESS_GEN_SPACING:
|
||||
case GL_TESS_GEN_VERTEX_ORDER:
|
||||
case GL_TESS_GEN_POINT_MODE: {
|
||||
if (!programObject->GetLinkStatus() || !programObject->HasLinkedShaderStage(ShaderStage::TessEval)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
std::to_string(program) +
|
||||
" is not a linked program object with a tessellation evaluation shader."));
|
||||
return;
|
||||
}
|
||||
switch (pname) {
|
||||
case GL_TESS_GEN_MODE: *params = static_cast<GLint>(programObject->GetTessGenMode()); break;
|
||||
case GL_TESS_GEN_SPACING: *params = static_cast<GLint>(programObject->GetTessGenSpacing()); break;
|
||||
case GL_TESS_GEN_VERTEX_ORDER:
|
||||
*params = static_cast<GLint>(programObject->GetTessGenVertexOrder());
|
||||
break;
|
||||
default: *params = programObject->GetTessGenPointMode() ? GL_TRUE : GL_FALSE; break;
|
||||
}
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
MGLOG_D("%s: %s", __func__, MG_Util::ConvertGLEnumToString(pname).c_str());
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -811,8 +1101,19 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
*params = shaderObject->GetInfoLog().empty() ? 0 : (GLint)shaderObject->GetInfoLog().length() + 1;
|
||||
break;
|
||||
case GL_SHADER_SOURCE_LENGTH:
|
||||
*params = shaderObject->GetShaderSource().empty() ? 0 : (GLint)shaderObject->GetShaderSource().length() + 1;
|
||||
case GL_SHADER_SOURCE_LENGTH: {
|
||||
// The APPLICATION's source, which is empty for a shader that came from glShaderBinary -
|
||||
// see ShaderObject::GetApplicationShaderSource.
|
||||
const auto& source = shaderObject->GetApplicationShaderSource();
|
||||
*params = source.empty() ? 0 : (GLint)source.length() + 1;
|
||||
break;
|
||||
}
|
||||
// GL_ARB_gl_spirv. GL_SPIR_V_BINARY and GL_SPIR_V_BINARY_ARB are the same token: TRUE
|
||||
// while the object stands for an application-supplied module. It is the FIRST thing the
|
||||
// conformance suite asks after glShaderBinary, and it used to fall into the terminal
|
||||
// default arm below and take the whole test with it.
|
||||
case GL_SPIR_V_BINARY:
|
||||
*params = shaderObject->HasSpirvBinary() ? GL_TRUE : GL_FALSE;
|
||||
break;
|
||||
// GL_KHR_parallel_shader_compile. THIS CASE MUST NOT JOIN - see the identical case in
|
||||
// GetProgramiv_State. GL_COMPILE_STATUS two cases up deliberately DOES join (it has
|
||||
@@ -858,13 +1159,23 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto& shaderObject = TryToGetShaderObject(shader);
|
||||
if (!shaderObject) return;
|
||||
|
||||
auto& src = shaderObject->GetShaderSource();
|
||||
auto& src = shaderObject->GetApplicationShaderSource();
|
||||
CopyStr(bufSize, length, source, src.c_str(), (GLsizei)src.length());
|
||||
}
|
||||
|
||||
GLint GetUniformLocation_State(GLuint program, const GLchar* name) {
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return -1;
|
||||
// GL 4.6 core 7.6: "INVALID_OPERATION is generated if program has not been successfully
|
||||
// linked". Answering -1 silently is not the same thing - the conformance suite reads the
|
||||
// error, not the location.
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return -1;
|
||||
}
|
||||
auto loc = programObject->GetUniformLocation(name);
|
||||
MGLOG_D("%s: loc %02d = %s", __func__, loc, name);
|
||||
return loc;
|
||||
@@ -1277,11 +1588,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
template <GLsizei ItemCount, typename T>
|
||||
void ProgramUniformv_State(GLuint program, GLint location, GLsizei count, T* value) {
|
||||
if (location == -1) return;
|
||||
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
|
||||
// The link check comes BEFORE the location == -1 early-out, not after. GL 4.6 core 7.6
|
||||
// makes an unlinked program INVALID_OPERATION regardless of the location, and -1 is
|
||||
// exactly the location an application holds after glGetUniformLocation on such a program -
|
||||
// so checking -1 first swallowed the very case the rule exists for.
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
@@ -1289,6 +1602,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
// "If location is equal to -1, the data passed in will be silently ignored and the
|
||||
// specified uniform variable will not be changed" - after the program itself has been
|
||||
// found acceptable.
|
||||
if (location == -1) return;
|
||||
|
||||
for (GLint offset = 0; offset < count; offset++) {
|
||||
if (offset > 0 && !programObject->UniformLocationsAliasSameUniform(location, location + offset)) {
|
||||
@@ -1699,8 +2016,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix2fv_State(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLfloat* value) {
|
||||
if (location == -1) return;
|
||||
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
|
||||
@@ -1712,14 +2027,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
if (location == -1) return;
|
||||
|
||||
UniformMatrixfv_Object(*programObject, __func__, location, count, transpose, value, 2, 2,
|
||||
"program " + std::to_string(program));
|
||||
}
|
||||
|
||||
void ProgramUniformMatrix3fv_State(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLfloat* value) {
|
||||
if (location == -1) return;
|
||||
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
|
||||
@@ -1731,6 +2046,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
if (location == -1) return;
|
||||
|
||||
for (GLint i = 0; i < count; i++) {
|
||||
if (i > 0 && !programObject->UniformLocationsAliasSameUniform(location, location + i)) {
|
||||
// Values for elements beyond the end of the uniform array are ignored.
|
||||
@@ -1756,8 +2073,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix4fv_State(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLfloat* value) {
|
||||
if (location == -1) return;
|
||||
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
|
||||
@@ -1769,6 +2084,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
if (location == -1) return;
|
||||
|
||||
for (GLint i = 0; i < count; i++) {
|
||||
if (i > 0 && !programObject->UniformLocationsAliasSameUniform(location, location + i)) {
|
||||
// Values for elements beyond the end of the uniform array are ignored.
|
||||
@@ -1790,8 +2107,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrixNonSquarefv_State(const char* caller, GLuint program, GLint location, GLsizei count,
|
||||
GLboolean transpose, const GLfloat* value, Int columns, Int rows) {
|
||||
if (location == -1) return;
|
||||
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
|
||||
@@ -1803,6 +2118,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
if (location == -1) return;
|
||||
|
||||
UniformMatrixfv_Object(*programObject, caller, location, count, transpose, value, columns, rows,
|
||||
"program " + std::to_string(program));
|
||||
}
|
||||
@@ -1836,6 +2153,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Program object" + std::to_string(program) + " that has been linked."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateUniformBlockBinding(uniformBlockBinding)) return;
|
||||
if (!programObject->IsActiveGlUniformBlock(uniformBlockIndex)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
@@ -2083,6 +2401,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
BindAttribLocation_State(program, index, name);
|
||||
}
|
||||
|
||||
void ShaderBinary(GLsizei count, const GLuint* shaders, GLenum binaryformat, const void* binary, GLsizei length) {
|
||||
ShaderBinary_State(count, shaders, binaryformat, binary, length);
|
||||
}
|
||||
|
||||
void SpecializeShader(GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants,
|
||||
const GLuint* pConstantIndex, const GLuint* pConstantValue) {
|
||||
SpecializeShader_State(shader, pEntryPoint, numSpecializationConstants, pConstantIndex, pConstantValue);
|
||||
}
|
||||
|
||||
void CompileShader(GLuint shader) {
|
||||
CompileShader_State(shader);
|
||||
}
|
||||
@@ -2342,7 +2669,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix2dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2352,6 +2678,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 2, 2);
|
||||
}
|
||||
void UniformMatrix3dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2368,7 +2695,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix3dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2378,6 +2704,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 3, 3);
|
||||
}
|
||||
void UniformMatrix4dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2394,7 +2721,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix4dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2404,6 +2730,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 4, 4);
|
||||
}
|
||||
void UniformMatrix2x3dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2420,7 +2747,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix2x3dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2430,6 +2756,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 2, 3);
|
||||
}
|
||||
void UniformMatrix2x4dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2446,7 +2773,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix2x4dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2456,6 +2782,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 2, 4);
|
||||
}
|
||||
void UniformMatrix3x2dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2472,7 +2799,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix3x2dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2482,6 +2808,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 3, 2);
|
||||
}
|
||||
void UniformMatrix3x4dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2498,7 +2825,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix3x4dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2508,6 +2834,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 3, 4);
|
||||
}
|
||||
void UniformMatrix4x2dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2524,7 +2851,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix4x2dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2534,6 +2860,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 4, 2);
|
||||
}
|
||||
void UniformMatrix4x3dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
@@ -2550,7 +2877,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void ProgramUniformMatrix4x3dv(GLuint program, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
auto& programObject = TryToGetProgramObject(program);
|
||||
if (!programObject) return;
|
||||
if (!programObject->GetLinkStatus()) {
|
||||
@@ -2560,6 +2886,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"program " + std::to_string(program) + " is not linked."));
|
||||
return;
|
||||
}
|
||||
if (location == -1) return;
|
||||
UniformMatrixdv_Object(*programObject, location, count, transpose, value, 4, 3);
|
||||
}
|
||||
void GetUniformdv(GLuint program, GLint location, GLdouble* params) {
|
||||
@@ -3072,6 +3399,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Backend does not support shader storage block binding."));
|
||||
return;
|
||||
}
|
||||
MGP_FILL(ShaderStorageBlockBinding);
|
||||
shaderStorageBlockBinding(program, blockName.c_str(), storageBlockBinding);
|
||||
}
|
||||
|
||||
|
||||
@@ -13,6 +13,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void AttachShader(GLuint program, GLuint shader);
|
||||
void BindAttribLocation(GLuint program, GLuint index, const GLchar* name);
|
||||
void CompileShader(GLuint shader);
|
||||
// GL_ARB_gl_spirv, core since 4.6. The pair is a two-step operation: glShaderBinary attaches
|
||||
// the module to one or more shader objects, glSpecializeShader names its entry point and
|
||||
// supplies its specialization constants and is what actually compiles them.
|
||||
void ShaderBinary(GLsizei count, const GLuint* shaders, GLenum binaryformat, const void* binary, GLsizei length);
|
||||
void SpecializeShader(GLuint shader, const GLchar* pEntryPoint, GLuint numSpecializationConstants,
|
||||
const GLuint* pConstantIndex, const GLuint* pConstantValue);
|
||||
GLuint CreateProgram(void);
|
||||
GLuint CreateShader(GLenum type);
|
||||
void DeleteProgram(GLuint program);
|
||||
|
||||
@@ -192,6 +192,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::format("Program {} has not been linked successfully.", program));
|
||||
return;
|
||||
}
|
||||
// GL 4.6 core 7.4: "INVALID_OPERATION is generated if program was not linked with its
|
||||
// PROGRAM_SEPARABLE status set". The LATCHED flag is the one that decides - a program
|
||||
// whose live flag was cleared after a separable link is still a legal stage, and a
|
||||
// program whose live flag was set after a non-separable link is not.
|
||||
if (!programObject->GetLinkedSeparable()) {
|
||||
RecordPipelineError(ErrorCode::InvalidOperation, __func__,
|
||||
std::format("Program {} was not linked as a separable program.", program));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
const GLbitfield selected = stages == GL_ALL_SHADER_BITS ? kAllStageBits : stages;
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/GLState/ErrorState/ErrorInfo.h>
|
||||
#include <MG_Impl/Pipe/PipeFill.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
@@ -59,6 +60,62 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLuint g_activePrimitivesGeneratedQueryId = 0;
|
||||
// Id of the query active on GL_SAMPLES_PASSED (0 = none).
|
||||
GLuint g_activeSamplesPassedQueryId = 0;
|
||||
// Ids of the queries active on the GL_ARB_pipeline_statistics_query targets, one slot per
|
||||
// target (0 = none). A map rather than a field per target: the eleven behave identically
|
||||
// and none of them has any state beyond "which object is counting".
|
||||
UnorderedMap<GLenum, GLuint> g_activePipelineStatisticsQueryIds;
|
||||
|
||||
// Whether MobileGL puts GL_ARB_tessellation_shader in its extension string. Read from the
|
||||
// ADVERTISED list rather than from a capability bit for the same reason
|
||||
// BackendSupportsTextureViews does (GL_Texture.cpp): it makes "MobileGL claims tessellation
|
||||
// support" and "the tessellation-conditional API surface is open" the same fact by
|
||||
// construction, so the day a backend starts advertising the string the surface below opens
|
||||
// with it and no second edit is owed.
|
||||
Bool AdvertisesTessellationShaderExtension() {
|
||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackendObject) return false;
|
||||
const auto& extensions = activeBackendObject->GetRendererInfo().RendererGLInfo.Extensions;
|
||||
return std::find(extensions.begin(), extensions.end(), E_GL_ARB_tessellation_shader) != extensions.end();
|
||||
}
|
||||
|
||||
// The eleven pipeline-statistics counters (GL 4.6 core table 4.3 / ARB_pipeline_statistics_query).
|
||||
// A 4.6 core context ACCEPTS the nine unconditional ones at glBeginQuery - there is no query
|
||||
// by which an application could learn otherwise before calling. MobileGL instruments none of
|
||||
// them, and says so the way GL 4.6 core 4.2.1 provides for: GL_QUERY_COUNTER_BITS answers
|
||||
// zero for these targets, which is the spec's own signal that the counter is unsupported and
|
||||
// its results indeterminate. That is an honest zero, not an advertised capability - the
|
||||
// alternative, GL_INVALID_ENUM on a core entry point, is both non-conformant AND less
|
||||
// informative.
|
||||
//
|
||||
// The two TESSELLATION targets are the exception, because ARB_pipeline_statistics_query
|
||||
// makes them conditional on tessellation support rather than unconditional, and the only
|
||||
// thing an application (or the conformance suite) can read to decide whether an
|
||||
// implementation has it is the GL_ARB_tessellation_shader string. MobileGL does not emit it
|
||||
// today, so these two answer GL_INVALID_ENUM: an API surface that accepts a
|
||||
// tessellation-conditional token while withholding the string that announces the condition
|
||||
// is self-contradictory, and it is the contradiction the suite catches
|
||||
// (KHR-GL46.pipeline_statistics_query_tests_ARB.api_coverage_unsupported_calls, whose
|
||||
// support probe is gl4cPipelineStatisticsQueryTests.cpp:1166-1176). The gate is the
|
||||
// advertisement itself, not a hardcoded "no", so this is one switch and not two.
|
||||
Bool IsPipelineStatisticsQueryTarget(GLenum target) {
|
||||
switch (target) {
|
||||
case GL_VERTICES_SUBMITTED:
|
||||
case GL_PRIMITIVES_SUBMITTED:
|
||||
case GL_VERTEX_SHADER_INVOCATIONS:
|
||||
case GL_GEOMETRY_SHADER_INVOCATIONS:
|
||||
case GL_GEOMETRY_SHADER_PRIMITIVES_EMITTED:
|
||||
case GL_FRAGMENT_SHADER_INVOCATIONS:
|
||||
case GL_COMPUTE_SHADER_INVOCATIONS:
|
||||
case GL_CLIPPING_INPUT_PRIMITIVES:
|
||||
case GL_CLIPPING_OUTPUT_PRIMITIVES:
|
||||
return true;
|
||||
case GL_TESS_CONTROL_SHADER_PATCHES:
|
||||
case GL_TESS_EVALUATION_SHADER_INVOCATIONS:
|
||||
return AdvertisesTessellationShaderExtension();
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
Bool TimerQueryDisabled() {
|
||||
return MG_Config::Features.DisableTimerQuery;
|
||||
@@ -108,6 +165,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void ResetQueryObjectLocked(QueryObject* queryObject) {
|
||||
if (queryObject->backendHandle) {
|
||||
if (const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery) {
|
||||
MGP_FILL(DeleteBackendQuery);
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->backendHandle = nullptr;
|
||||
@@ -122,6 +180,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void EndTimeElapsedQueryLocked(QueryObject* queryObject) {
|
||||
const auto endTimeElapsedQuery = MG_Backend::gBackendFunctionsTable.GL.EndTimeElapsedQuery;
|
||||
if (endTimeElapsedQuery && queryObject->backendHandle) {
|
||||
MGP_FILL(EndTimeElapsedQuery);
|
||||
endTimeElapsedQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->active = false;
|
||||
@@ -201,6 +260,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
Uint64 result = 0;
|
||||
const auto getQueryResult64 = MG_Backend::gBackendFunctionsTable.GL.GetQueryResult64;
|
||||
MGP_FILL(GetQueryResult64);
|
||||
if (queryObject->backendHandle && getQueryResult64 &&
|
||||
!getQueryResult64(queryObject->backendHandle, /*wait=*/false, &result)) {
|
||||
// Not ready. The whole point of the no-wait form is that the caller's
|
||||
@@ -215,6 +275,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
if (queryObject->backendHandle) {
|
||||
if (const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery) {
|
||||
MGP_FILL(DeleteBackendQuery);
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->backendHandle = nullptr;
|
||||
@@ -230,6 +291,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
const auto isQueryResultAvailable = MG_Backend::gBackendFunctionsTable.GL.IsQueryResultAvailable;
|
||||
MGP_FILL(IsQueryResultAvailable);
|
||||
outValue = (!isQueryResultAvailable || isQueryResultAvailable(queryObject->backendHandle)) ? 1 : 0;
|
||||
return true;
|
||||
}
|
||||
@@ -241,6 +303,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
Uint64 result = 0;
|
||||
if (queryObject->backendHandle) {
|
||||
const auto getQueryResult64 = MG_Backend::gBackendFunctionsTable.GL.GetQueryResult64;
|
||||
MGP_FILL(GetQueryResult64);
|
||||
if (getQueryResult64 &&
|
||||
!getQueryResult64(queryObject->backendHandle, /*wait=*/true, &result)) {
|
||||
// The backend could not produce the result YET (e.g. a
|
||||
@@ -261,6 +324,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// query degrades to a zero result); the backend handle is
|
||||
// consumed and the value cached for later reads.
|
||||
if (const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery) {
|
||||
MGP_FILL(DeleteBackendQuery);
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->backendHandle = nullptr;
|
||||
@@ -366,10 +430,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
queryObject->target == GL_ANY_SAMPLES_PASSED_CONSERVATIVE) {
|
||||
if (const auto endOcclusionQuery = MG_Backend::gBackendFunctionsTable.GL.EndOcclusionQuery;
|
||||
endOcclusionQuery && queryObject->backendHandle) {
|
||||
MGP_FILL(EndOcclusionQuery);
|
||||
endOcclusionQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->active = false;
|
||||
g_activeSamplesPassedQueryId = 0;
|
||||
} else if (IsPipelineStatisticsQueryTarget(queryObject->target)) {
|
||||
queryObject->active = false;
|
||||
g_activePipelineStatisticsQueryIds[queryObject->target] = 0;
|
||||
} else if (queryObject->target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN ||
|
||||
queryObject->target == GL_PRIMITIVES_GENERATED) {
|
||||
queryObject->active = false;
|
||||
@@ -382,6 +450,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
if (queryObject->backendHandle) {
|
||||
if (const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery) {
|
||||
MGP_FILL(DeleteBackendQuery);
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->backendHandle = nullptr;
|
||||
@@ -410,7 +479,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
(target == GL_SAMPLES_PASSED || target == GL_ANY_SAMPLES_PASSED ||
|
||||
target == GL_ANY_SAMPLES_PASSED_CONSERVATIVE) &&
|
||||
MG_Backend::gBackendFunctionsTable.GL.BeginOcclusionQuery != nullptr;
|
||||
if (target != GL_TIME_ELAPSED && !isTransformFeedbackQuery && !isOcclusionQuery) {
|
||||
const Bool isPipelineStatisticsQuery = IsPipelineStatisticsQueryTarget(target);
|
||||
if (target != GL_TIME_ELAPSED && !isTransformFeedbackQuery && !isOcclusionQuery &&
|
||||
!isPipelineStatisticsQuery) {
|
||||
// GL_TIMESTAMP is not a valid BeginQuery target; the occlusion targets
|
||||
// need backend support.
|
||||
RecordQueryError(ErrorCode::InvalidEnum, __FUNCTION__, "Query target is not supported.");
|
||||
@@ -426,10 +497,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__, "Query object does not exist.");
|
||||
return;
|
||||
}
|
||||
GLuint& activeQueryId = isTransformFeedbackQuery
|
||||
? (target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN ? g_activePrimitivesWrittenQueryId
|
||||
: g_activePrimitivesGeneratedQueryId)
|
||||
: (isOcclusionQuery ? g_activeSamplesPassedQueryId : g_activeTimeElapsedQueryId);
|
||||
GLuint& activeQueryId = isPipelineStatisticsQuery
|
||||
? g_activePipelineStatisticsQueryIds[target]
|
||||
: (isTransformFeedbackQuery
|
||||
? (target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN ? g_activePrimitivesWrittenQueryId
|
||||
: g_activePrimitivesGeneratedQueryId)
|
||||
: (isOcclusionQuery ? g_activeSamplesPassedQueryId : g_activeTimeElapsedQueryId));
|
||||
if (activeQueryId != 0) {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__,
|
||||
"A query is already active on this target.");
|
||||
@@ -448,10 +521,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ResetQueryObjectLocked(queryObject); // discard any previous result
|
||||
queryObject->target = target;
|
||||
queryObject->active = true;
|
||||
if (isTransformFeedbackQuery) {
|
||||
if (isPipelineStatisticsQuery) {
|
||||
// Nothing to start: the counter is uninstrumented and GL_QUERY_COUNTER_BITS says so.
|
||||
// The object still becomes a real, target-latched query so every other rule about it
|
||||
// (re-use with another target, double-begin, EndQuery pairing) keeps holding.
|
||||
} else if (isTransformFeedbackQuery) {
|
||||
// Prefer real GPU transform-feedback queries (exact with geometry shaders);
|
||||
// the CPU accounting delta stays as the fallback when the backend lacks them.
|
||||
const auto beginXfbPrimitivesQuery = MG_Backend::gBackendFunctionsTable.GL.BeginXfbPrimitivesQuery;
|
||||
MGP_FILL(BeginXfbPrimitivesQuery);
|
||||
queryObject->backendHandle =
|
||||
beginXfbPrimitivesQuery ? beginXfbPrimitivesQuery(target == GL_PRIMITIVES_GENERATED) : nullptr;
|
||||
queryObject->counterSnapshot = TransformFeedbackCounterForTarget(target);
|
||||
@@ -460,9 +538,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
queryObject->geometryCaptureDrawSnapshot =
|
||||
MG_State::pGLContext->GetTransformFeedbackGeometryCaptureDraws();
|
||||
} else if (isOcclusionQuery) {
|
||||
MGP_FILL(BeginOcclusionQuery);
|
||||
queryObject->backendHandle = MG_Backend::gBackendFunctionsTable.GL.BeginOcclusionQuery();
|
||||
} else {
|
||||
const auto beginTimeElapsedQuery = MG_Backend::gBackendFunctionsTable.GL.BeginTimeElapsedQuery;
|
||||
MGP_FILL(BeginTimeElapsedQuery);
|
||||
queryObject->backendHandle =
|
||||
(!TimerQueryDisabled() && beginTimeElapsedQuery) ? beginTimeElapsedQuery() : nullptr;
|
||||
}
|
||||
@@ -476,15 +556,19 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
(target == GL_SAMPLES_PASSED || target == GL_ANY_SAMPLES_PASSED ||
|
||||
target == GL_ANY_SAMPLES_PASSED_CONSERVATIVE) &&
|
||||
MG_Backend::gBackendFunctionsTable.GL.BeginOcclusionQuery != nullptr;
|
||||
if (target != GL_TIME_ELAPSED && !isTransformFeedbackQuery && !isOcclusionQuery) {
|
||||
const Bool isPipelineStatisticsQuery = IsPipelineStatisticsQueryTarget(target);
|
||||
if (target != GL_TIME_ELAPSED && !isTransformFeedbackQuery && !isOcclusionQuery &&
|
||||
!isPipelineStatisticsQuery) {
|
||||
RecordQueryError(ErrorCode::InvalidEnum, __FUNCTION__, "Query target is not supported.");
|
||||
return;
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(g_queryObjectsMutex);
|
||||
GLuint& activeQueryId = isTransformFeedbackQuery
|
||||
? (target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN ? g_activePrimitivesWrittenQueryId
|
||||
: g_activePrimitivesGeneratedQueryId)
|
||||
: (isOcclusionQuery ? g_activeSamplesPassedQueryId : g_activeTimeElapsedQueryId);
|
||||
GLuint& activeQueryId = isPipelineStatisticsQuery
|
||||
? g_activePipelineStatisticsQueryIds[target]
|
||||
: (isTransformFeedbackQuery
|
||||
? (target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN ? g_activePrimitivesWrittenQueryId
|
||||
: g_activePrimitivesGeneratedQueryId)
|
||||
: (isOcclusionQuery ? g_activeSamplesPassedQueryId : g_activeTimeElapsedQueryId));
|
||||
if (activeQueryId == 0) {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__, "No query is active on this target.");
|
||||
return;
|
||||
@@ -494,9 +578,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
activeQueryId = 0; // should not happen; keep state consistent
|
||||
return;
|
||||
}
|
||||
if (isPipelineStatisticsQuery) {
|
||||
// The result is a definite zero rather than an unread backend handle, so a later
|
||||
// GetQueryObject* answers immediately and never waits on something that was never
|
||||
// started. GL_QUERY_COUNTER_BITS = 0 is what marks that zero indeterminate.
|
||||
queryObject->cachedResult = 0;
|
||||
queryObject->resultCached = true;
|
||||
queryObject->active = false;
|
||||
queryObject->ended = true;
|
||||
activeQueryId = 0;
|
||||
return;
|
||||
}
|
||||
if (isTransformFeedbackQuery) {
|
||||
if (queryObject->backendHandle) {
|
||||
if (const auto endXfbPrimitivesQuery = MG_Backend::gBackendFunctionsTable.GL.EndXfbPrimitivesQuery) {
|
||||
MGP_FILL(EndXfbPrimitivesQuery);
|
||||
endXfbPrimitivesQuery(queryObject->backendHandle);
|
||||
}
|
||||
}
|
||||
@@ -506,6 +602,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!queryObject->backendHandle || PrefersCpuTransformFeedbackResult(queryObject)) {
|
||||
if (queryObject->backendHandle) {
|
||||
if (const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery) {
|
||||
MGP_FILL(DeleteBackendQuery);
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->backendHandle = nullptr;
|
||||
@@ -522,6 +619,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (isOcclusionQuery) {
|
||||
if (const auto endOcclusionQuery = MG_Backend::gBackendFunctionsTable.GL.EndOcclusionQuery;
|
||||
endOcclusionQuery && queryObject->backendHandle) {
|
||||
MGP_FILL(EndOcclusionQuery);
|
||||
endOcclusionQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->active = false;
|
||||
@@ -560,6 +658,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ResetQueryObjectLocked(queryObject); // discard any previous result
|
||||
queryObject->target = target;
|
||||
const auto queryCounterTimestamp = MG_Backend::gBackendFunctionsTable.GL.QueryCounterTimestamp;
|
||||
MGP_FILL(QueryCounterTimestamp);
|
||||
queryObject->backendHandle =
|
||||
(!TimerQueryDisabled() && queryCounterTimestamp) ? queryCounterTimestamp() : nullptr;
|
||||
queryObject->ended = true;
|
||||
@@ -657,7 +756,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = static_cast<GLint>(g_activePrimitivesGeneratedQueryId);
|
||||
break;
|
||||
default:
|
||||
*params = 0;
|
||||
if (IsPipelineStatisticsQueryTarget(target)) {
|
||||
const auto it = g_activePipelineStatisticsQueryIds.find(target);
|
||||
*params = it != g_activePipelineStatisticsQueryIds.end() ? static_cast<GLint>(it->second) : 0;
|
||||
} else {
|
||||
*params = 0;
|
||||
}
|
||||
break;
|
||||
}
|
||||
return;
|
||||
@@ -668,6 +772,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// entry points / timestamp valid bits at call time, not at table
|
||||
// init), and the MOBILEGL_DISABLE_TIMERQUERY kill switch always
|
||||
// wins.
|
||||
if (IsPipelineStatisticsQueryTarget(target)) {
|
||||
// Zero: GL 4.6 core 4.2.1's way of saying the counter is not implemented and its
|
||||
// results are indeterminate. The conformance suite reads exactly this and skips
|
||||
// the functional half of each such target, which is the outcome an uninstrumented
|
||||
// counter should produce.
|
||||
*params = 0;
|
||||
return;
|
||||
}
|
||||
if (target == GL_SAMPLES_PASSED || target == GL_ANY_SAMPLES_PASSED ||
|
||||
target == GL_ANY_SAMPLES_PASSED_CONSERVATIVE) {
|
||||
const Bool occlusionSupported = MG_Backend::gBackendFunctionsTable.GL.BeginOcclusionQuery != nullptr;
|
||||
@@ -676,6 +788,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
const Bool timerTarget = target == GL_TIME_ELAPSED || target == GL_TIMESTAMP;
|
||||
const auto isTimerQuerySupported = MG_Backend::gBackendFunctionsTable.GL.IsTimerQuerySupported;
|
||||
MGP_FILL(IsTimerQuerySupported);
|
||||
const Bool supported =
|
||||
timerTarget && !TimerQueryDisabled() && isTimerQuerySupported && isTimerQuerySupported();
|
||||
*params = supported ? 64 : 0;
|
||||
@@ -741,14 +854,24 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
namespace {
|
||||
Bool IsPerVertexStreamQueryTarget(GLenum target) {
|
||||
return target == GL_PRIMITIVES_GENERATED || target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN;
|
||||
}
|
||||
|
||||
// The indexed query entry points differ from the plain ones only in the vertex
|
||||
// stream they address (GL 4.6 core 4.2.1): index must be below GL_MAX_VERTEX_STREAMS
|
||||
// for the two transform feedback targets and zero for every other target. With a
|
||||
// single vertex stream both bounds are 1, so a valid call is always index 0 and
|
||||
// forwards to the unindexed implementation.
|
||||
// for the two transform feedback targets and zero for every other target. MobileGL
|
||||
// implements ONE vertex stream, so both bounds are 1 and a valid call is always index 0 -
|
||||
// which is what makes the three forwards below equivalent to the unindexed entry points.
|
||||
//
|
||||
// THAT EQUIVALENCE IS THE WHOLE JUSTIFICATION, and it is read out of the getter rather
|
||||
// than assumed: the moment GL_MAX_VERTEX_STREAMS answers more than one, index 1..3 starts
|
||||
// reaching EndQueryIndexed and GetQueryIndexediv, which resolve the active query from
|
||||
// per-TARGET globals and would end - or report - a query begun on a different stream.
|
||||
// Raising that limit therefore means giving each active query a stream index and
|
||||
// comparing it here, not just changing the number.
|
||||
Bool ValidateQueryStreamIndex(const char* function, GLenum target, GLuint index) {
|
||||
const Bool perStreamTarget =
|
||||
target == GL_PRIMITIVES_GENERATED || target == GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN;
|
||||
const Bool perStreamTarget = IsPerVertexStreamQueryTarget(target);
|
||||
GLint maxVertexStreams = 1;
|
||||
if (perStreamTarget) {
|
||||
GetIntegerv(GL_MAX_VERTEX_STREAMS, &maxVertexStreams);
|
||||
@@ -761,6 +884,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
: "index must be zero for this query target.");
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void BeginQueryIndexed(GLenum target, GLuint index, GLuint id) {
|
||||
@@ -806,6 +930,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery;
|
||||
for (const auto& [_, queryObject] : orphans) {
|
||||
if (deleteBackendQuery && queryObject->backendHandle) {
|
||||
MGP_FILL(DeleteBackendQuery);
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
delete queryObject;
|
||||
|
||||
@@ -328,10 +328,50 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MG_State::pGLContext->SetSampleCoverage(std::clamp(static_cast<Float>(value), 0.0f, 1.0f), invert == GL_TRUE);
|
||||
}
|
||||
|
||||
// ARB_sample_shading / GL 4.6 core 14.3.1: "value is clamped to [0, 1] when specified", so
|
||||
// there is no error to raise - a caller that asks for 2.0 gets 1.0 and GL_MIN_SAMPLE_SHADING_-
|
||||
// VALUE reads back 1.0. Was a logging no-op while ARB_sample_shading was advertised, which
|
||||
// let an application enable GL_SAMPLE_SHADING and then quietly get the driver's default rate.
|
||||
void MinSampleShading_State(GLfloat value) {
|
||||
MG_State::pGLContext->SetMinSampleShadingValue(std::clamp(static_cast<Float>(value), 0.0f, 1.0f));
|
||||
}
|
||||
|
||||
void PolygonOffset_State(GLfloat factor, GLfloat units) {
|
||||
MG_State::pGLContext->SetPolygonOffset(static_cast<Float>(factor), static_cast<Float>(units));
|
||||
}
|
||||
|
||||
void PolygonOffsetClamp_State(GLfloat factor, GLfloat units, GLfloat clamp) {
|
||||
// GL 4.6 core 14.6.5 / GL_EXT_polygon_offset_clamp. No error cases: any three floats are
|
||||
// legal, and clamp = 0 is exactly glPolygonOffset. Whether the backend can APPLY the clamp
|
||||
// is a separate question (see the DirectGLES/DirectVulkan forwarding); the state is
|
||||
// recorded either way, because GL_POLYGON_OFFSET_CLAMP has to read back what was written.
|
||||
MG_State::pGLContext->SetPolygonOffsetClamped(static_cast<Float>(factor), static_cast<Float>(units),
|
||||
static_cast<Float>(clamp));
|
||||
}
|
||||
|
||||
void ClipControl_State(GLenum origin, GLenum depth) {
|
||||
// GL 4.5 core 13.5: both arguments are strict enums, and either being wrong is
|
||||
// GL_INVALID_ENUM with the state left untouched.
|
||||
if (origin != GL_LOWER_LEFT && origin != GL_UPPER_LEFT) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"glClipControl origin must be GL_LOWER_LEFT or GL_UPPER_LEFT; got " +
|
||||
MG_Util::ConvertGLEnumToString(origin) + "."));
|
||||
return;
|
||||
}
|
||||
if (depth != GL_NEGATIVE_ONE_TO_ONE && depth != GL_ZERO_TO_ONE) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
"glClipControl depth must be GL_NEGATIVE_ONE_TO_ONE or GL_ZERO_TO_ONE; got " +
|
||||
MG_Util::ConvertGLEnumToString(depth) + "."));
|
||||
return;
|
||||
}
|
||||
MG_State::pGLContext->SetClipControl(origin, depth);
|
||||
}
|
||||
|
||||
void PolygonMode_State(GLenum face, GLenum mode) {
|
||||
// GL 3.3 core: separate front/back polygon modes were removed in 3.1, so the only legal
|
||||
// face is GL_FRONT_AND_BACK. GL_FRONT / GL_BACK must be rejected (some desktop drivers
|
||||
@@ -1013,10 +1053,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
SampleCoverage_State(value, invert);
|
||||
}
|
||||
|
||||
void MinSampleShading(GLfloat value) {
|
||||
MinSampleShading_State(value);
|
||||
}
|
||||
|
||||
void PolygonOffset(GLfloat factor, GLfloat units) {
|
||||
PolygonOffset_State(factor, units);
|
||||
}
|
||||
|
||||
void PolygonOffsetClamp(GLfloat factor, GLfloat units, GLfloat clamp) {
|
||||
PolygonOffsetClamp_State(factor, units, clamp);
|
||||
}
|
||||
|
||||
void ClipControl(GLenum origin, GLenum depth) {
|
||||
ClipControl_State(origin, depth);
|
||||
}
|
||||
|
||||
void PolygonMode(GLenum face, GLenum mode) {
|
||||
PolygonMode_State(face, mode);
|
||||
}
|
||||
|
||||
@@ -38,7 +38,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void StencilFunc(GLenum func, GLint ref, GLuint mask);
|
||||
void Scissor(GLint x, GLint y, GLsizei width, GLsizei height);
|
||||
void SampleCoverage(GLfloat value, GLboolean invert);
|
||||
void MinSampleShading(GLfloat value);
|
||||
void PolygonOffset(GLfloat factor, GLfloat units);
|
||||
void PolygonOffsetClamp(GLfloat factor, GLfloat units, GLfloat clamp);
|
||||
void ClipControl(GLenum origin, GLenum depth);
|
||||
void PolygonMode(GLenum face, GLenum mode);
|
||||
void PointSize(GLfloat size);
|
||||
void PointParameterf(GLenum pname, GLfloat param);
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Converters/GLToMG/TextureEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
|
||||
#include <MG_Util/Math/FixedPointConversion.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
@@ -22,6 +23,50 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return static_cast<Float>(*(const GLint*)param);
|
||||
}
|
||||
|
||||
// GL_TEXTURE_BORDER_COLOR is the only sampler parameter with more than one component, and it
|
||||
// is also the only one whose meaning depends on WHICH entry point wrote it. Everything else
|
||||
// reads exactly one component and does not care.
|
||||
Bool IsVectorOnlySamplerPname(GLenum pname) {
|
||||
return pname == GL_TEXTURE_BORDER_COLOR;
|
||||
}
|
||||
|
||||
// A state query returns the value CONVERTED to the type the caller asked for (GL 4.6 core
|
||||
// 2.2.2 / 6.1), never the other type's bits. These two are the sampler side of the numeric
|
||||
// casts GetTexParameterfv_State/GetTexParameteriv_State already do on the texture side; the
|
||||
// sampler path funnels all three spellings through one void* function, which is precisely how
|
||||
// it came to write a fixed type regardless of the caller.
|
||||
//
|
||||
// Truncation rather than rounding for the float -> integer direction, matching the texture
|
||||
// twin (GetTexParameteriv_State's static_cast<GLint> on MIN_LOD/MAX_LOD/LOD_BIAS): the two
|
||||
// spellings of the same state disagreeing is the bug being fixed here, and a texture and a
|
||||
// sampler queried the same way must answer the same number.
|
||||
void StoreSamplerScalar(void* params, Bool isFloat, Bool isUnsignedInteger, Float value) {
|
||||
if (isFloat) {
|
||||
*(GLfloat*)params = value;
|
||||
return;
|
||||
}
|
||||
// Via GLint in both integer spellings: a direct float -> GLuint cast of a negative value
|
||||
// (GL_TEXTURE_MIN_LOD defaults to -1000) is undefined behaviour, while the two-step
|
||||
// conversion is the well-defined modular one, and it is what the texture-side
|
||||
// GetTexParameterIuiv fallback does.
|
||||
const GLint asInt = static_cast<GLint>(value);
|
||||
if (isUnsignedInteger) {
|
||||
*(GLuint*)params = static_cast<GLuint>(asInt);
|
||||
} else {
|
||||
*(GLint*)params = asInt;
|
||||
}
|
||||
}
|
||||
|
||||
void StoreSamplerEnum(void* params, Bool isFloat, Bool isUnsignedInteger, GLenum value) {
|
||||
if (isFloat) {
|
||||
*(GLfloat*)params = static_cast<GLfloat>(value);
|
||||
} else if (isUnsignedInteger) {
|
||||
*(GLuint*)params = value;
|
||||
} else {
|
||||
*(GLint*)params = static_cast<GLint>(value);
|
||||
}
|
||||
}
|
||||
|
||||
Bool ValidateSamplerParameterValue(GLenum pname, const void* param, Bool isFloat, Bool isUnsignedInteger) {
|
||||
if (param == nullptr) return false;
|
||||
|
||||
@@ -56,8 +101,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// `isIntegerCommand` distinguishes the "I" spellings (glSamplerParameterIiv / Iuiv) from the
|
||||
// plain ones. It only matters for GL_TEXTURE_BORDER_COLOR, and there it decides everything:
|
||||
// GL 4.6 core 8.10 says the I forms store the components unmodified with an integer internal
|
||||
// type, while glSamplerParameteriv converts them to floating point with equation 2.2. Routing
|
||||
// both to the same setter - which is what this file used to do - meant glSamplerParameteriv
|
||||
// stored raw integers (so a border of 255 became float 255.0 instead of the spec's ~1.19e-7)
|
||||
// and glSamplerParameterIiv lost the fact that it was ever an integer at all.
|
||||
void SetSamplerParam_State(GLuint sampler, GLenum pname, const void* param, bool isFloat,
|
||||
bool isUnsignedInteger) {
|
||||
bool isUnsignedInteger, bool isIntegerCommand) {
|
||||
if (param == nullptr) return;
|
||||
if (!SamplerImpl::ValidateSamplerName(sampler)) return;
|
||||
|
||||
@@ -112,6 +164,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (isFloat) {
|
||||
const auto* values = (const GLfloat*)param;
|
||||
samplerObj->SetBorderColor(FloatVec4(values[0], values[1], values[2], values[3]));
|
||||
} else if (!isIntegerCommand) {
|
||||
// glSamplerParameteriv: GL 4.6 core equation 2.2 into the FLOAT border colour.
|
||||
const auto* values = (const GLint*)param;
|
||||
samplerObj->SetBorderColor(FloatVec4(MG_Util::SignedNormalizedInt32ToFloat(values[0]),
|
||||
MG_Util::SignedNormalizedInt32ToFloat(values[1]),
|
||||
MG_Util::SignedNormalizedInt32ToFloat(values[2]),
|
||||
MG_Util::SignedNormalizedInt32ToFloat(values[3])));
|
||||
} else if (isUnsignedInteger) {
|
||||
const auto* values = (const GLuint*)param;
|
||||
samplerObj->SetBorderColorUI(UintVec4(values[0], values[1], values[2], values[3]));
|
||||
@@ -128,7 +187,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void GetSamplerParam_State(GLuint sampler, GLenum pname, void* params, bool isFloat,
|
||||
bool isUnsignedInteger) {
|
||||
bool isUnsignedInteger, bool isIntegerCommand) {
|
||||
if (params == nullptr) return;
|
||||
if (!SamplerImpl::ValidateSamplerName(sampler)) return;
|
||||
|
||||
@@ -141,47 +200,56 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!SamplerImpl::ValidateSamplerObject(sampler)) return;
|
||||
|
||||
using namespace MG_Util;
|
||||
// Every scalar pname goes through StoreSamplerScalar/StoreSamplerEnum so the CALLER'S form
|
||||
// decides the destination type. Writing a fixed type regardless - which is what these case
|
||||
// labels used to do - hands back the other type's bit pattern rather than a converted value:
|
||||
// glGetSamplerParameterfv(GL_TEXTURE_WRAP_S) deposited the integer 10497 into a GLfloat and
|
||||
// the caller read 1.47e-41, and glGetSamplerParameteriv(GL_TEXTURE_MIN_LOD) deposited the
|
||||
// IEEE bits of -1000.0f and the caller read -998637568. Sixteen (pname, entry-point) pairs
|
||||
// were broken this way; only MAX_ANISOTROPY_EXT and BORDER_COLOR branched correctly, which is
|
||||
// how the same bug class was already found and fixed once for a single pname.
|
||||
switch (pname) {
|
||||
case GL_TEXTURE_WRAP_S:
|
||||
*(GLuint*)params = MG_Util::ConvertSamplerWrapModeToGLEnum(samplerObj->GetWrapS());
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerWrapModeToGLEnum(samplerObj->GetWrapS()));
|
||||
break;
|
||||
case GL_TEXTURE_WRAP_T:
|
||||
*(GLuint*)params = MG_Util::ConvertSamplerWrapModeToGLEnum(samplerObj->GetWrapT());
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerWrapModeToGLEnum(samplerObj->GetWrapT()));
|
||||
break;
|
||||
case GL_TEXTURE_WRAP_R:
|
||||
*(GLuint*)params = MG_Util::ConvertSamplerWrapModeToGLEnum(samplerObj->GetWrapR());
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerWrapModeToGLEnum(samplerObj->GetWrapR()));
|
||||
break;
|
||||
case GL_TEXTURE_MIN_FILTER:
|
||||
*(GLuint*)params =
|
||||
MG_Util::ConvertSamplerFilterModeToGLEnum(samplerObj->GetMinFilter(), samplerObj->GetMipmapMode());
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerFilterModeToGLEnum(samplerObj->GetMinFilter(),
|
||||
samplerObj->GetMipmapMode()));
|
||||
break;
|
||||
case GL_TEXTURE_MAG_FILTER:
|
||||
*(GLuint*)params =
|
||||
MG_Util::ConvertSamplerFilterModeToGLEnum(samplerObj->GetMagFilter(), SamplerMipmapMode::None);
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerFilterModeToGLEnum(samplerObj->GetMagFilter(),
|
||||
SamplerMipmapMode::None));
|
||||
break;
|
||||
case GL_TEXTURE_MIN_LOD:
|
||||
*(GLfloat*)params = samplerObj->GetMinLod();
|
||||
StoreSamplerScalar(params, isFloat, isUnsignedInteger, samplerObj->GetMinLod());
|
||||
break;
|
||||
case GL_TEXTURE_MAX_LOD:
|
||||
*(GLfloat*)params = samplerObj->GetMaxLod();
|
||||
StoreSamplerScalar(params, isFloat, isUnsignedInteger, samplerObj->GetMaxLod());
|
||||
break;
|
||||
case GL_TEXTURE_LOD_BIAS:
|
||||
*(GLfloat*)params = samplerObj->GetLodBias();
|
||||
StoreSamplerScalar(params, isFloat, isUnsignedInteger, samplerObj->GetLodBias());
|
||||
break;
|
||||
case GL_TEXTURE_MAX_ANISOTROPY_EXT:
|
||||
if (isFloat) {
|
||||
*(GLfloat*)params = samplerObj->GetMaxAnisotropy();
|
||||
} else if (isUnsignedInteger) {
|
||||
*(GLuint*)params = static_cast<GLuint>(samplerObj->GetMaxAnisotropy());
|
||||
} else {
|
||||
*(GLint*)params = static_cast<GLint>(samplerObj->GetMaxAnisotropy());
|
||||
}
|
||||
StoreSamplerScalar(params, isFloat, isUnsignedInteger, samplerObj->GetMaxAnisotropy());
|
||||
break;
|
||||
case GL_TEXTURE_COMPARE_MODE:
|
||||
*(GLuint*)params = MG_Util::ConvertSamplerCompareModeToGLEnum(samplerObj->GetCompareMode());
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerCompareModeToGLEnum(samplerObj->GetCompareMode()));
|
||||
break;
|
||||
case GL_TEXTURE_COMPARE_FUNC:
|
||||
*(GLuint*)params = MG_Util::ConvertSamplerCompareFuncToGLEnum(samplerObj->GetSamplerCompareFunc());
|
||||
StoreSamplerEnum(params, isFloat, isUnsignedInteger,
|
||||
MG_Util::ConvertSamplerCompareFuncToGLEnum(samplerObj->GetSamplerCompareFunc()));
|
||||
break;
|
||||
case GL_TEXTURE_BORDER_COLOR: {
|
||||
if (isFloat) {
|
||||
@@ -191,6 +259,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
out[1] = color.y();
|
||||
out[2] = color.z();
|
||||
out[3] = color.w();
|
||||
} else if (!isIntegerCommand) {
|
||||
// glGetSamplerParameteriv: the inverse of the write side, GL 4.6 core equation 2.3.
|
||||
// Exactly inverse, so a {0,1,2,4} written with glSamplerParameteriv reads back as
|
||||
// {0,1,2,4}; a bare truncating cast answered {0,0,0,0}.
|
||||
const auto& color = samplerObj->GetBorderColor();
|
||||
auto* out = (GLint*)params;
|
||||
out[0] = MG_Util::FloatToSignedNormalizedInt32(color.x());
|
||||
out[1] = MG_Util::FloatToSignedNormalizedInt32(color.y());
|
||||
out[2] = MG_Util::FloatToSignedNormalizedInt32(color.z());
|
||||
out[3] = MG_Util::FloatToSignedNormalizedInt32(color.w());
|
||||
} else if (isUnsignedInteger) {
|
||||
const auto& color = samplerObj->GetBorderColorUI();
|
||||
auto* out = (GLuint*)params;
|
||||
@@ -293,16 +371,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (sampler == 0) {
|
||||
textureUnit.SetSamplerObject(nullptr);
|
||||
} else {
|
||||
// GL 3.3 core 3.8.2: BindSampler on a name GenSamplers never returned - or one already
|
||||
// deleted - is INVALID_OPERATION. SamplerParameter* raises INVALID_VALUE for the same
|
||||
// name, which is why this cannot go through the shared SamplerImpl validator.
|
||||
if (!MG_State::pGLContext->ValidateSamplerName(sampler)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "BindSampler_State",
|
||||
std::format("Invalid sampler name {}", sampler)));
|
||||
return;
|
||||
}
|
||||
// GL 4.6 core 8.2: BindSampler on a name GenSamplers never returned - or one already
|
||||
// deleted - is INVALID_OPERATION, and so is every other sampler entry point on such a
|
||||
// name, so the shared validator answers for all of them.
|
||||
if (!SamplerImpl::ValidateSamplerName(sampler)) return;
|
||||
Bool doesSamplerObjectCreated = MG_State::pGLContext->ValidateSamplerObject(sampler);
|
||||
if (!doesSamplerObjectCreated) {
|
||||
MG_State::pGLContext->CreateSamplerObject(sampler);
|
||||
@@ -356,30 +428,50 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
/* @INSERTION_POINT:FUNCTION_IMPLEMENTATION@ */
|
||||
void GetSamplerParameteriv(GLuint sampler, GLenum pname, GLint* params) {
|
||||
GetSamplerParam_State(sampler, pname, params, false, false);
|
||||
GetSamplerParam_State(sampler, pname, params, false, false, false);
|
||||
}
|
||||
|
||||
void SamplerParameterIuiv(GLuint sampler, GLenum pname, const GLuint* param) {
|
||||
SetSamplerParam_State(sampler, pname, param, false, true);
|
||||
SetSamplerParam_State(sampler, pname, param, false, true, true);
|
||||
}
|
||||
|
||||
void SamplerParameterIiv(GLuint sampler, GLenum pname, const GLint* param) {
|
||||
SetSamplerParam_State(sampler, pname, param, false, false);
|
||||
SetSamplerParam_State(sampler, pname, param, false, false, true);
|
||||
}
|
||||
|
||||
void SamplerParameteriv(GLuint sampler, GLenum pname, const GLint* param) {
|
||||
SetSamplerParam_State(sampler, pname, param, false, false);
|
||||
SetSamplerParam_State(sampler, pname, param, false, false, false);
|
||||
}
|
||||
|
||||
void SamplerParameterfv(GLuint sampler, GLenum pname, const GLfloat* param) {
|
||||
SetSamplerParam_State(sampler, pname, param, true, false);
|
||||
SetSamplerParam_State(sampler, pname, param, true, false, false);
|
||||
}
|
||||
|
||||
// GL 4.6 core 8.10: the scalar spellings take "the value of pname", so a pname with more than one
|
||||
// component is INVALID_ENUM here rather than something to read four components of. Guarding at
|
||||
// the entry point rather than downstream is also what stops the vector path reading twelve bytes
|
||||
// past the caller's single stack scalar - taking the address of a by-value argument and handing
|
||||
// it to a four-component reader is what these used to do. The texture-side twins already answer
|
||||
// INVALID_ENUM for GL_TEXTURE_BORDER_COLOR (TexParameteri/f name it as unsupported outright).
|
||||
void SamplerParameteri(GLuint sampler, GLenum pname, GLint param) {
|
||||
if (IsVectorOnlySamplerPname(pname)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "SamplerParameteri",
|
||||
"pname has more than one component and needs a vector form."));
|
||||
return;
|
||||
}
|
||||
SamplerParameteriv(sampler, pname, ¶m);
|
||||
}
|
||||
|
||||
void SamplerParameterf(GLuint sampler, GLenum pname, GLfloat param) {
|
||||
if (IsVectorOnlySamplerPname(pname)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "SamplerParameterf",
|
||||
"pname has more than one component and needs a vector form."));
|
||||
return;
|
||||
}
|
||||
SamplerParameterfv(sampler, pname, ¶m);
|
||||
}
|
||||
|
||||
@@ -388,15 +480,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void GetSamplerParameterIuiv(GLuint sampler, GLenum pname, GLuint* params) {
|
||||
GetSamplerParam_State(sampler, pname, params, false, true);
|
||||
GetSamplerParam_State(sampler, pname, params, false, true, true);
|
||||
}
|
||||
|
||||
void GetSamplerParameterIiv(GLuint sampler, GLenum pname, GLint* params) {
|
||||
GetSamplerParam_State(sampler, pname, params, false, false);
|
||||
GetSamplerParam_State(sampler, pname, params, false, false, true);
|
||||
}
|
||||
|
||||
void GetSamplerParameterfv(GLuint sampler, GLenum pname, GLfloat* params) {
|
||||
GetSamplerParam_State(sampler, pname, params, true, false);
|
||||
GetSamplerParam_State(sampler, pname, params, true, false, false);
|
||||
}
|
||||
|
||||
void GenSamplers(GLsizei count, GLuint* samplers) {
|
||||
|
||||
@@ -12,11 +12,17 @@
|
||||
#include <MG_Util/Converters/GLToMG/TextureEnumConverter.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl::SamplerImpl {
|
||||
// GL 4.6 core 8.2: "An INVALID_OPERATION error is generated if sampler is not the name of a
|
||||
// sampler object previously returned from a call to GenSamplers." That class is shared by every
|
||||
// sampler entry point - BindSampler, SamplerParameter*, GetSamplerParameter* - so this one gate
|
||||
// answers for all of them. It used to report INVALID_VALUE (the GL 3.3 wording), which forced
|
||||
// BindSampler to carry a bespoke duplicate of the same check just to get the class right.
|
||||
Bool ValidateSamplerName(GLuint sampler) {
|
||||
if (!MG_State::pGLContext->ValidateSamplerName(sampler)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue, MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "ValidateSamplerName",
|
||||
std::format("Invalid sampler name {}", sampler)));
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "ValidateSamplerName",
|
||||
std::format("Invalid sampler name {}", sampler)));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "GL_Sync.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Impl/Pipe/PipeFill.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
@@ -56,6 +57,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
syncObject->condition = condition;
|
||||
syncObject->flags = flags;
|
||||
if (const auto backendFenceSync = MG_Backend::gBackendFunctionsTable.GL.FenceSync) {
|
||||
MGP_FILL(FenceSync);
|
||||
syncObject->backendHandle = backendFenceSync();
|
||||
}
|
||||
const GLsync handle = reinterpret_cast<GLsync>(syncObject);
|
||||
@@ -94,6 +96,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!backendClientWaitSync || !syncObject->backendHandle) {
|
||||
return GL_ALREADY_SIGNALED; // legacy always-signaled fallback
|
||||
}
|
||||
MGP_FILL(ClientWaitSync);
|
||||
return backendClientWaitSync(syncObject->backendHandle, flags, timeout);
|
||||
}
|
||||
|
||||
@@ -119,6 +122,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
const auto backendWaitSync = MG_Backend::gBackendFunctionsTable.GL.WaitSync;
|
||||
if (backendWaitSync && syncObject->backendHandle) {
|
||||
MGP_FILL(WaitSync);
|
||||
backendWaitSync(syncObject->backendHandle, flags, timeout);
|
||||
}
|
||||
}
|
||||
@@ -139,6 +143,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
const auto backendDeleteSync = MG_Backend::gBackendFunctionsTable.GL.DeleteSync;
|
||||
if (backendDeleteSync && syncObject->backendHandle) {
|
||||
MGP_FILL(DeleteSync);
|
||||
backendDeleteSync(syncObject->backendHandle);
|
||||
}
|
||||
delete syncObject;
|
||||
@@ -174,6 +179,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
break;
|
||||
case GL_SYNC_STATUS: {
|
||||
const auto backendGetSyncStatus = MG_Backend::gBackendFunctionsTable.GL.GetSyncStatus;
|
||||
MGP_FILL(GetSyncStatus);
|
||||
const Bool signaled = !backendGetSyncStatus || !syncObject->backendHandle ||
|
||||
backendGetSyncStatus(syncObject->backendHandle);
|
||||
value = signaled ? GL_SIGNALED : GL_UNSIGNALED;
|
||||
@@ -227,6 +233,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
const auto backendDeleteSync = MG_Backend::gBackendFunctionsTable.GL.DeleteSync;
|
||||
for (const auto& [_, syncObject] : orphans) {
|
||||
if (backendDeleteSync && syncObject->backendHandle) {
|
||||
MGP_FILL(DeleteSync);
|
||||
backendDeleteSync(syncObject->backendHandle);
|
||||
}
|
||||
delete syncObject;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -8,9 +8,24 @@
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/TextureState/TextureObject.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
/* @INSERTION_POINT:FUNCTION_DECLARATION@ */
|
||||
// Answers a texture-image query straight out of the CPU shadow, into client memory or a bound
|
||||
// PIXEL_PACK_BUFFER. This is the whole of glGetTexImage on a build with no backend readback, and
|
||||
// it is also the sound fallback for a backend that has no GPU image to read: with no image,
|
||||
// nothing GPU-side can ever have written the texture, so the shadow IS its content.
|
||||
//
|
||||
// It answers a NARROWER contract than glGetTexImage's, and refuses what it cannot do rather than
|
||||
// answering wrongly. The copy is verbatim: it performs no format or type conversion, and it packs
|
||||
// rows tightly, honouring only GL_PACK_SWAP_BYTES and the bitmap GL_PACK_LSB_FIRST path. A
|
||||
// request whose (format, type) texel size differs from the texture's own, or a pixel-store state
|
||||
// that adds row padding / a row-length override / a skip offset, is rejected with
|
||||
// GL_INVALID_OPERATION (see ValidateShadowReadbackLayout, which spells out why each is unsafe).
|
||||
void CopyTextureImageToClientOrPBO_State(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject,
|
||||
TextureUploadTarget textureUploadTarget, GLint level, GLenum format,
|
||||
GLenum type, GLsizei bufSize, void* pixels, const char* caller);
|
||||
// The sized internal formats a buffer texture accepts (GL 4.6 core table 8.16). The buffer
|
||||
// clears take the same list, so it is shared rather than written out twice.
|
||||
Bool IsBufferTextureInternalFormat(GLenum internalformat);
|
||||
|
||||
@@ -103,6 +103,28 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool ValidateCubeMapArrayShape(TextureUploadTarget target, GLsizei width, GLsizei height, GLsizei depth,
|
||||
const char* caller) {
|
||||
if (target != TextureUploadTarget::CubeMapArray && target != TextureUploadTarget::ProxyCubeMapArray) {
|
||||
return true;
|
||||
}
|
||||
if (width != height) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller,
|
||||
"Cube map array levels must be square (width == height)"));
|
||||
return false;
|
||||
}
|
||||
if (depth % 6 != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller,
|
||||
"Cube map array depth must be a multiple of six"));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool ValidateTextureSizeWithTextureUploadTarget(TextureUploadTarget target, GLsizei width, GLsizei height) {
|
||||
if (target == TextureUploadTarget::CubeMapPositiveX || target == TextureUploadTarget::CubeMapNegativeX ||
|
||||
target == TextureUploadTarget::CubeMapPositiveY || target == TextureUploadTarget::CubeMapNegativeY ||
|
||||
|
||||
@@ -20,6 +20,13 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
Bool ValidateTexturePixelDataType(TexturePixelDataType texturePixelDataType);
|
||||
Bool ValidateTextureLevelNumber(Int level);
|
||||
Bool ValidateTextureSizeWithTextureUploadTarget(TextureUploadTarget target, GLsizei width, GLsizei height);
|
||||
// The two shape rules a cube-map-array level owes (GL 4.6 core 8.5): its faces are square, and
|
||||
// its depth counts whole cubes. Both are GL_INVALID_VALUE. This used to be spelled inline in
|
||||
// glTexStorage3D only, which is why glTexImage3D let both violations through - every entry
|
||||
// point that DEFINES a cube-array level calls this now, so the two cannot drift again. A
|
||||
// non-cube-array upload target answers true untouched.
|
||||
Bool ValidateCubeMapArrayShape(TextureUploadTarget target, GLsizei width, GLsizei height, GLsizei depth,
|
||||
const char* caller);
|
||||
Bool ValidateTextureSizeRange(Int width, Int height, Int depth);
|
||||
Bool ValidateTextureInternalFormat(TextureInternalFormat format);
|
||||
Bool ValidateTextureBorderNumber(Int border);
|
||||
|
||||
@@ -12,15 +12,15 @@
|
||||
#include <MG_State/GLState/ErrorState/Error.h>
|
||||
#include <MG_Util/Converters/MGToGL/DataTypeConverter.h>
|
||||
#include <MG_Util/Converters/MGToStr/DataTypeConverter.h>
|
||||
#include <MG_Util/ShaderTranspiler/CompileEnv.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl::VertexArrayImpl {
|
||||
Uint GetMaxVertexAttribs() {
|
||||
constexpr Uint capacity = static_cast<Uint>(MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS);
|
||||
if (!MG_Backend::pActiveBackendObject) return capacity;
|
||||
|
||||
const Int backendLimit = MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxVertexAttribs;
|
||||
if (backendLimit <= 0) return capacity;
|
||||
return std::min(static_cast<Uint>(backendLimit), capacity);
|
||||
// Shared with reflection's limit and with gl_MaxVertexAttribs; see ResolveMaxVertexAttribs.
|
||||
const Bool hasBackend = MG_Backend::pActiveBackendObject != nullptr;
|
||||
const Int backendLimit =
|
||||
hasBackend ? MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxVertexAttribs : 0;
|
||||
return static_cast<Uint>(MG_Util::ShaderTranspiler::ResolveMaxVertexAttribs(hasBackend, backendLimit));
|
||||
}
|
||||
|
||||
Uint GetMaxVertexAttribBindings() {
|
||||
|
||||
@@ -0,0 +1,270 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/CompositeResolver.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// P4a's PROGRAM-PIPELINE COMPOSITE, on the client side.
|
||||
//
|
||||
// GLContext::GetProgramForDraw() already flattens a bound pipeline into one hidden composite
|
||||
// ProgramObject entirely in the frontend - it joins every graphics stage, computes the
|
||||
// pipeline's draw-program signature, looks it up in the pipeline's own cache and, on a miss,
|
||||
// attaches each stage's LINKED SNAPSHOT into a fresh ProgramObject and links it. All of that
|
||||
// is frontend work and none of it moves. What this file adds is the one thing the wire needs:
|
||||
// the composite gets ONE handle, out of the ShaderCso reserved high band, and
|
||||
// create_shader_state goes out for it exactly as for an ordinary program. THE SERVER NEVER
|
||||
// LEARNS IT IS A COMPOSITE and needs no "resolved draw program" hook at all.
|
||||
//
|
||||
// WHY A BAND RATHER THAN A FLAG ON THE HANDLE: a flag would have to be carried, honoured and
|
||||
// masked off by every consumer of a ShaderCso handle, on both sides; a reserved slot range is
|
||||
// a property of the allocator instead, so "an ordinary program can never be handed a composite
|
||||
// slot" is true by construction. MGPipeSlotAllocator::Allocate refuses the band outright and
|
||||
// AllocateComposite is the only door in.
|
||||
//
|
||||
// WHAT THIS FILE IS ACTUALLY FOR: the composite's slot has TWO INDEPENDENT RELEASE PATHS and
|
||||
// either order has to free it exactly once.
|
||||
// * the pipeline cache drops the composite when the draw-program signature moves. In the
|
||||
// frontend that overwrite drops the last SharedPtr, so the composite's own destructor
|
||||
// usually runs first; the resolver still speaks the release, because "usually" is not a
|
||||
// contract and a client that only reacted to destructors would leak a slot the moment the
|
||||
// frontend started holding a second reference.
|
||||
// * the composite ProgramObject's own ~ProgramObject, which is an ordinary program's death
|
||||
// path and takes the same helper.
|
||||
// Both go through MGPipeEmitShaderCsoDestroyAndFree, and whichever runs second is a PROVEN
|
||||
// no-op: MGPipeSlotAllocator::Free refuses a slot that is not live at that generation and
|
||||
// bumps no generation of its own, so a double release cannot skip a generation either.
|
||||
//
|
||||
// THE MEMO's KEY IS (CONTEXT ID, PIPELINE GL NAME) AND THE CONTEXT HALF IS NOT OPTIONAL.
|
||||
// This resolver is a PROCESS singleton while a pipeline's GL name is per context: GLContext
|
||||
// owns m_programPipelines AND its own name generator m_programPipelineNames (Core.h), so name
|
||||
// N names two different ProgramPipelineObjects in two contexts, each with its own composite
|
||||
// and its own handle. Keyed on the name alone, the first emission after a make-current found
|
||||
// the OTHER context's entry, matched nothing - two composites are two ProgramObjects with two
|
||||
// lifetime ids, so the handles differ even when the stage set and the signature are identical
|
||||
// - and released it: a delete_shader_state and a cleared publication latch for a composite
|
||||
// whose frontend ProgramObject is alive, its band slot handed back and re-issued at gen + 1,
|
||||
// and the server rebuilding that program (glslang + SPIR-V + spirv-opt, the very cost the
|
||||
// signature below exists to avoid) once per context switch.
|
||||
//
|
||||
// THE CONTEXT ID IS GLContext::GetTextureContextId() AND NOTHING ELSE - the tree's existing
|
||||
// never-reused per-context id (TextureState::AllocateContextId; PipeInputs carries it as
|
||||
// m_textureContextId at seven fill points and the backends' own per-context memos key on it).
|
||||
// Deliberately NOT the GLContext ADDRESS that MGB_CTX_IDENTITY and MGPipeTracker::m_context
|
||||
// compare, because Core.h states the reason that id exists at all: a context freed and remade
|
||||
// lands on the old heap address, which would put this same defect back one context recreation
|
||||
// later.
|
||||
//
|
||||
// WHAT RELEASES A DESTROYED CONTEXT's ENTRIES: nothing in this file, and that is the correct
|
||||
// answer rather than an omission. Destroying a context drops m_programPipelines, which drops
|
||||
// each ProgramPipelineObject, which drops the composite it cached; ~ProgramObject then runs
|
||||
// MGPipeEmitShaderCsoDestroyAndFree - the composite's OWN release path, the second of the two
|
||||
// above - and the slot goes back exactly once. The entries those composites leave behind can
|
||||
// never be found again (no future Observe can carry a dead context id) and could not release
|
||||
// anything if they were (the allocator erases the lifetime-id mapping on Free), so Reset()
|
||||
// DROPS them instead of releasing them. That is also what bounds the vector; see Reset().
|
||||
//
|
||||
// THE SIGNATURE IS ComputeDrawProgramSignature(), the per-graphics-stage {lifetimeId,
|
||||
// GetLinkVersion()} array - and DELIBERATELY NOT GetBackendStateVersion(), which is what made
|
||||
// the SSO conformance loop rebuild the composite (glslang + SPIR-V + spirv-opt) on every draw,
|
||||
// because a glUniform1i to a sampler moves it.
|
||||
//
|
||||
// HEADER-ONLY, for the ownership reason Tracker.h states: a new .cpp would need the root
|
||||
// CMakeLists.txt, which is the contract package's.
|
||||
//
|
||||
// IT IS INCLUDED BY ProgramEmit.h AND NOT THE OTHER WAY ROUND, deliberately: the composite is
|
||||
// a special case of the program family's own emission, so the family header depends on this
|
||||
// one and this one depends on nothing of the family's. The reverse arrangement would make the
|
||||
// resolver reachable only from a translation unit that had already decided to use it, i.e.
|
||||
// dead in the build that matters and live only in the tests.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/PipeMutation.h>
|
||||
#include <MG_State/GLState/ProgramState/ProgramObject.h>
|
||||
#include <MG_State/GLState/ProgramState/ProgramPipelineObject.h>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// IS THIS PROGRAM A PIPELINE COMPOSITE? A composite is the one ProgramObject in the system
|
||||
// constructed with external index 0 (Core.cpp's MakeShared<ProgramObject>(0u)), and that is
|
||||
// not an accident of implementation: it is deliberately not a named program, so it must not
|
||||
// answer glIsProgram and must not consume a GL name, and glCreateProgram never returns 0.
|
||||
//
|
||||
// ASKED THIS WAY RATHER THAN CARRIED ON THE OBJECT because a Bool member on ProgramObject
|
||||
// would resize the pull build's object and break G1 outright - the phase's admitted-resize
|
||||
// set is empty - and a hook in Core.cpp would have to be maintained on a path that already
|
||||
// states the invariant in its own comment.
|
||||
inline Bool MGPipeProgramIsPipelineComposite(const MG_State::GLState::ProgramObject& program) {
|
||||
return program.GetExternalIndex() == 0;
|
||||
}
|
||||
|
||||
class MGPipeCompositeResolver {
|
||||
public:
|
||||
using ProgramObject = MG_State::GLState::ProgramObject;
|
||||
using ProgramPipelineObject = MG_State::GLState::ProgramPipelineObject;
|
||||
using DrawProgramSignature = ProgramPipelineObject::DrawProgramSignature;
|
||||
|
||||
struct Counters {
|
||||
Uint64 Mints = 0; // signatures this resolver has seen minted
|
||||
Uint64 Reuses = 0; // a signature that had not moved
|
||||
Uint64 Releases = 0; // signature-move releases, i.e. the pipeline-cache path
|
||||
// Entries dropped by Reset() because the composite's slot was already gone - the
|
||||
// shape every entry of a DESTROYED CONTEXT ends in. A dropped entry is not a
|
||||
// release: nothing is emitted and nothing is freed, the obligation having been
|
||||
// discharged by the composite's own ~ProgramObject.
|
||||
Uint64 Sweeps = 0;
|
||||
};
|
||||
|
||||
// Told, at every emission, which composite the frontend handed out for which pipeline.
|
||||
// Returns the handle the emitter should use, which is always the one already minted off
|
||||
// the composite's own lifetime id - the resolver never mints a second identity for an
|
||||
// object that has one.
|
||||
//
|
||||
// WHEN THE SIGNATURE MOVES the previous composite's slot is released here, through the
|
||||
// one death helper and in its fixed order. That is the pipeline-cache release path; the
|
||||
// composite's own destructor is the other one and the second of the two is the proven
|
||||
// no-op.
|
||||
MGPipeHandle Observe(Uint64 contextId, const ProgramPipelineObject& pipeline,
|
||||
const ProgramObject& composite, MGPipeHandle handle) {
|
||||
const DrawProgramSignature signature = pipeline.ComputeDrawProgramSignature();
|
||||
const Uint pipelineName = pipeline.GetExternalIndex();
|
||||
Entry* entry = Find(contextId, pipelineName);
|
||||
if (entry != nullptr) {
|
||||
if (entry->Signature == signature && entry->Handle == handle) {
|
||||
// THE SAME COMPOSITE. Not merely "the same signature": the handle is minted
|
||||
// off the composite ProgramObject's own lifetime id, so an identical handle
|
||||
// IS an identical object and there is nothing to release. Live is
|
||||
// deliberately NOT touched - it is the release obligation and it is still
|
||||
// owed for exactly this handle.
|
||||
++m_counters.Reuses;
|
||||
return handle;
|
||||
}
|
||||
// A MOVED SIGNATURE ON THIS CONTEXT's OWN ENTRY, which is the only thing that
|
||||
// can reach here now: another context's pipeline of the same name is not found
|
||||
// above and therefore not released, its obligation staying owed to the context
|
||||
// that took it.
|
||||
ReleaseEntry(*entry);
|
||||
} else {
|
||||
m_entries.push_back(Entry{});
|
||||
entry = &m_entries.back();
|
||||
entry->ContextId = contextId;
|
||||
entry->PipelineName = pipelineName;
|
||||
}
|
||||
entry->Signature = signature;
|
||||
entry->Handle = handle;
|
||||
entry->CompositeLifetimeId = composite.GetLifetimeId();
|
||||
entry->Live = true;
|
||||
++m_counters.Mints;
|
||||
return handle;
|
||||
}
|
||||
|
||||
// A make-current, and it RELEASES NOTHING. The entries name composites that belong to
|
||||
// the frontend objects of the context being left, those objects outlive the switch, and
|
||||
// releasing them would emit a delete for a live program.
|
||||
//
|
||||
// NOR IS ANY MEMO INVALIDATED, and that is what the context key bought. This used to
|
||||
// clear a per-entry `Fresh` flag beside `Live`, because with a name-only key an entry
|
||||
// could not say whether it described "my own pipeline before the switch" or "another
|
||||
// context's pipeline of the same name" - and exactly one of those two properties could
|
||||
// hold at a time. The key answers the question directly now, so the freshness flag and
|
||||
// its one reader (a HandleFor() accessor that had no caller anywhere in the tree) are
|
||||
// both gone rather than left as scaffolding: `Live`, the release obligation, is the
|
||||
// entry's only state and nothing but ReleaseEntry may clear it.
|
||||
//
|
||||
// WHAT IS LEFT TO DO HERE IS RECLAMATION, and this is the one moment the client is told
|
||||
// that a context boundary was crossed. An entry whose composite slot is no longer live
|
||||
// has had its obligation discharged elsewhere - by that composite's own ~ProgramObject,
|
||||
// which is precisely what happened to EVERY entry of a context that has just been
|
||||
// destroyed - so it is DROPPED rather than released: a release would resolve nothing
|
||||
// anyway (the allocator erases the lifetime-id mapping on Free) and no reader is left.
|
||||
// Without this the vector would grow by one per (context, pipeline name) pair the
|
||||
// process ever used, where the name-only key bounded it by the highest pipeline name;
|
||||
// with it, it is bounded by the pairs whose composite slot is actually live.
|
||||
void Reset() {
|
||||
SizeT kept = 0;
|
||||
for (SizeT i = 0; i < m_entries.size(); ++i) {
|
||||
if (!m_entries[i].Live || !MGPipeSlots().IsLive(MGPipeKind::ShaderCso, m_entries[i].Handle)) {
|
||||
++m_counters.Sweeps;
|
||||
continue;
|
||||
}
|
||||
if (kept != i) m_entries[kept] = m_entries[i];
|
||||
++kept;
|
||||
}
|
||||
m_entries.resize(kept);
|
||||
}
|
||||
|
||||
void ResetCounters() { m_counters = Counters{}; }
|
||||
|
||||
// Diagnostics and unit cases only; nothing on the emission path asks. There is no
|
||||
// HandleFor(name) accessor and there must not be one: the emitter takes the handle from
|
||||
// the composite ProgramObject it already holds, so a lookup by name would be a second
|
||||
// authority on an identity the allocator already owns.
|
||||
SizeT Size() const { return m_entries.size(); }
|
||||
const Counters& GetCounters() const { return m_counters; }
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
// NO FRONTEND SharedPtr, and that is the exit-order rule rather than a style
|
||||
// choice: a static that held one would put a frontend destructor on an exit
|
||||
// handler's path into a torn-down pipe. A GL name, a signature of plain integers,
|
||||
// a handle and a lifetime id are all this needs.
|
||||
// KEYED ON (CONTEXT ID, GL NAME), and the name half is the GL name because a
|
||||
// ProgramPipelineObject has no lifetime id - ComputeDrawProgramSignature reads the
|
||||
// STAGE programs' ids and the pipeline itself carries none. The context half is
|
||||
// GLContext::GetTextureContextId(); see the file header for why the name alone was
|
||||
// wrong and why the context ADDRESS would be too.
|
||||
//
|
||||
// WITHIN ONE CONTEXT glGenProgramPipelines recycles names, so a deleted-and-
|
||||
// recreated pipeline can still inherit its predecessor's entry; that is bounded and
|
||||
// self-correcting rather than a hazard. The first Observe on the new object finds a
|
||||
// signature and a handle that do not match and releases the old entry, and that
|
||||
// release resolves NOTHING - the allocator erases the lifetime-id mapping on Free,
|
||||
// so a stale CompositeLifetimeId emits no delete and frees no slot; all it costs is
|
||||
// one redundant, idempotent death notice, which is the same shape the composite's
|
||||
// own second release path already has.
|
||||
Uint64 ContextId = 0;
|
||||
Uint PipelineName = 0;
|
||||
DrawProgramSignature Signature{};
|
||||
MGPipeHandle Handle = kMGPipeNullHandle;
|
||||
Uint64 CompositeLifetimeId = 0;
|
||||
// THE RELEASE OBLIGATION. Set when this entry takes responsibility for a composite's
|
||||
// slot, cleared ONLY by ReleaseEntry when that responsibility is discharged.
|
||||
Bool Live = false;
|
||||
};
|
||||
|
||||
// BOTH HALVES OF THE KEY, always. An entry of another context is not this pipeline's
|
||||
// entry: not found, not matched, not released.
|
||||
Entry* Find(Uint64 contextId, Uint pipelineName) {
|
||||
for (Entry& entry : m_entries) {
|
||||
if (entry.ContextId == contextId && entry.PipelineName == pipelineName) return &entry;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void ReleaseEntry(Entry& entry) {
|
||||
if (!entry.Live || entry.CompositeLifetimeId == 0) return;
|
||||
entry.Live = false;
|
||||
MGPipeEmitShaderCsoDestroyAndFree(entry.CompositeLifetimeId);
|
||||
entry.Handle = kMGPipeNullHandle;
|
||||
entry.CompositeLifetimeId = 0;
|
||||
++m_counters.Releases;
|
||||
}
|
||||
|
||||
Vector<Entry> m_entries;
|
||||
Counters m_counters;
|
||||
};
|
||||
|
||||
inline MGPipeCompositeResolver& MGPipeCompositeResolverInstance() {
|
||||
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason, and named in the phase's risk
|
||||
// list beside the other three new client singletons: heap-constructed and intentionally
|
||||
// leaked at exit, holding no frontend SharedPtr.
|
||||
static MGPipeCompositeResolver* resolver = new MGPipeCompositeResolver();
|
||||
return *resolver;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
@@ -0,0 +1,205 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/CsoCache.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The render-state CSO cache (ARCHITECTURE.md 4.5.2 / 5.3, P2 brief D7).
|
||||
//
|
||||
// THE LOOKUP, and the first step is the whole point:
|
||||
// 1. m_pipelineStateVersion (widened) did not move -> reuse the last handle. ZERO hashing,
|
||||
// zero probing, and nothing is emitted unless m_version also moved. That is the steady
|
||||
// state of every frame, and it is why the tracker asks the cache at all only when the
|
||||
// dirty walk says the pipeline version moved.
|
||||
// 2. moved -> hash the 396 pipeline bytes, probe, and on a hit CONFIRM WITH A MEMCMP
|
||||
// before reusing the handle. ARCHITECTURE.md 4.1 says content addressing on an
|
||||
// xxHash; a bare 64-bit equality would let a collision alias two different render
|
||||
// states onto one CSO, which is silent wrong pixels with no gate that can see it.
|
||||
// Mesa's cso_cache memcmps for the same reason. The memcmp only ever runs on a
|
||||
// pipeline-version change, i.e. never in the steady state.
|
||||
// 3. miss -> mint a slot, emit create_render_state with every pipeline chunk, then bind.
|
||||
//
|
||||
// CAPACITY 64 (ROADMAP.md P2). 64 x (8 + 8 + 396 + 8) = about 26 KB per context. ROADMAP.md
|
||||
// open question 4 says 64 is provisional and the counters retune it at P13; this ships 64
|
||||
// and publishes the mint / bind / evict counters that retune reads.
|
||||
//
|
||||
// THE NEGATIVE CONTROL. kMGPipeBehaviourNoCsoContentAddressing (bit 63 of the runtime
|
||||
// MOBILEGL_PIPE_PUSH bitmask) turns off the PROBE and the handle reuse, not the records:
|
||||
// every pipeline-version change then mints a fresh CSO, binds it and evicts, which is
|
||||
// precisely "whole-block content addressing" and reproduces the regression
|
||||
// RenderState.h records. It is what separates "push is slower" from "the CSO design is
|
||||
// slower", and CsoContentAddressingScenario (package E) is the always-on ctest that stops
|
||||
// the switch from rotting.
|
||||
//
|
||||
// Header-only for the same ownership reason as Tracker.h: the root CMakeLists.txt that
|
||||
// would name a new .cpp is package A's and is frozen behind the p2/contract tag.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <Config.h>
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/MGPipeRenderStateSpans.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
inline constexpr SizeT kMGPipeCsoCacheCapacity = 64;
|
||||
|
||||
class MGPipeCsoCache {
|
||||
public:
|
||||
struct Counters {
|
||||
Uint64 Mints = 0; // create_render_state emissions
|
||||
// bind_render_state emissions, mint or reuse. Counted in Acquire because Acquire
|
||||
// has exactly ONE caller (PipeFill.cpp's EmitRenderState) and that caller binds
|
||||
// immediately after every call - so "acquisitions" and "binds" are the same
|
||||
// number, and counting it here keeps the count from depending on an emitter
|
||||
// remembering to tick it. mints/binds is the cache's hit rate and it is the
|
||||
// number the CSO content-addressing negative control moves.
|
||||
Uint64 Binds = 0;
|
||||
Uint64 Hits = 0; // a probe that found a live entry and passed the memcmp
|
||||
Uint64 Collisions = 0; // a hash hit the memcmp REJECTED - the reason it exists
|
||||
Uint64 Evictions = 0; // LRU evictions, each one a delete_render_state
|
||||
};
|
||||
|
||||
// The handle for `params`' pipeline subset. Mints and emits create_render_state on a
|
||||
// miss; emits delete_render_state for whatever it evicts to make room. `payloadBytes`
|
||||
// accumulates what went on the wire, for PipeStats::RecordDrawPayloadBytes.
|
||||
MGPipeHandle Acquire(const RenderStateParameters& params, Uint64& payloadBytes) {
|
||||
Array<Uint8, kMGPipePipelineChunkBytes> bytes;
|
||||
MGPipeGatherPipelineBytes(params, bytes.data());
|
||||
++m_counters.Binds;
|
||||
|
||||
const Bool contentAddressed =
|
||||
(MG_Config::Features.PipePush & kMGPipeBehaviourNoCsoContentAddressing) == 0;
|
||||
if (contentAddressed) {
|
||||
const Uint64 hash = s_hashForTest != nullptr ? s_hashForTest(bytes.data())
|
||||
: MGPipeHashPipelineBytes(bytes.data());
|
||||
for (SizeT i = 0; i < m_entries.size(); ++i) {
|
||||
if (m_entries[i].Hash != hash) continue;
|
||||
if (std::memcmp(m_entries[i].Bytes.data(), bytes.data(), bytes.size()) != 0) {
|
||||
// A 64-bit collision between two DIFFERENT render states. Reusing the
|
||||
// handle here would render one state with the other's pipeline, so the
|
||||
// entry is dropped and the caller mints - correctness first, and the
|
||||
// counter says how often it happened.
|
||||
++m_counters.Collisions;
|
||||
Evict(i);
|
||||
break;
|
||||
}
|
||||
m_entries[i].LastUsed = ++m_clock;
|
||||
++m_counters.Hits;
|
||||
return m_entries[i].Cso;
|
||||
}
|
||||
return Mint(hash, bytes, payloadBytes);
|
||||
}
|
||||
// Content addressing OFF: never probe, always mint. The records still exist, so
|
||||
// the arm differs from the default one in exactly one thing - whether a handle is
|
||||
// reused - which is what makes it a control rather than a different design.
|
||||
return Mint(0, bytes, payloadBytes);
|
||||
}
|
||||
|
||||
// Context teardown, a server reset, a unit test's fixture. Emits nothing: the applier
|
||||
// is reset alongside, and a delete for a record that is about to be dropped anyway
|
||||
// would be a wire message with no reader.
|
||||
void Reset() {
|
||||
for (auto& entry : m_entries) MGPipeSlots().Free(MGPipeKind::RenderStateCso, entry.Cso);
|
||||
m_entries.clear();
|
||||
m_clock = 0;
|
||||
}
|
||||
|
||||
void ResetCounters() { m_counters = Counters{}; }
|
||||
|
||||
SizeT Size() const { return m_entries.size(); }
|
||||
const Counters& GetCounters() const { return m_counters; }
|
||||
|
||||
// TEST SEAM, and it is here because the thing it tests cannot be reached any other
|
||||
// way. A 64-bit collision between two DIFFERENT render states is silent wrong pixels
|
||||
// and it is exactly what the memcmp confirm above exists to stop, so
|
||||
// CsoCacheTest.HashCollisionDoesNotAliasTwoStates has to be able to make one happen.
|
||||
// Null in every real build - one never-taken, perfectly-predicted branch on a path
|
||||
// that runs only when the pipeline version moved, i.e. never in the steady state.
|
||||
using HashForTestFn = Uint64 (*)(const void* pipelineBytes);
|
||||
inline static HashForTestFn s_hashForTest = nullptr;
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
Uint64 Hash = 0;
|
||||
Uint64 LastUsed = 0;
|
||||
MGPipeHandle Cso = kMGPipeNullHandle;
|
||||
Array<Uint8, kMGPipePipelineChunkBytes> Bytes{};
|
||||
};
|
||||
|
||||
MGPipeHandle Mint(Uint64 hash, const Array<Uint8, kMGPipePipelineChunkBytes>& bytes,
|
||||
Uint64& payloadBytes) {
|
||||
if (m_entries.size() >= kMGPipeCsoCacheCapacity) {
|
||||
SizeT victim = 0;
|
||||
for (SizeT i = 1; i < m_entries.size(); ++i) {
|
||||
if (m_entries[i].LastUsed < m_entries[victim].LastUsed) victim = i;
|
||||
}
|
||||
Evict(victim);
|
||||
}
|
||||
|
||||
const MGPipeHandle cso = MGPipeSlots().Allocate(MGPipeKind::RenderStateCso);
|
||||
MGPRenderStateDesc desc{};
|
||||
desc.Cso = cso;
|
||||
desc.BaseCso = kMGPipeNullHandle;
|
||||
// A brand-new CSO names every pipeline chunk; the incremental form against a
|
||||
// BaseCso is what the applier's assertion allows and P3 will use once a CSO is
|
||||
// minted from a neighbour rather than from nothing.
|
||||
desc.ChunkMask = kAllPipelineChunks;
|
||||
desc.Blob.Size = kMGPipePipelineChunkBytes;
|
||||
MGPipeApplyCreateRenderState(desc, bytes.data());
|
||||
payloadBytes += sizeof(MGPRenderStateDesc) + kMGPipePipelineChunkBytes;
|
||||
|
||||
Entry entry;
|
||||
entry.Hash = hash;
|
||||
entry.LastUsed = ++m_clock;
|
||||
entry.Cso = cso;
|
||||
entry.Bytes = bytes;
|
||||
m_entries.push_back(entry);
|
||||
|
||||
++m_counters.Mints;
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::RenderStateCsoMints, 1);
|
||||
}
|
||||
return cso;
|
||||
}
|
||||
|
||||
void Evict(SizeT index) {
|
||||
MGPHandleOnly handle{};
|
||||
handle.Handle = m_entries[index].Cso;
|
||||
handle.Kind = static_cast<Uint32>(MGPipeKind::RenderStateCso);
|
||||
MGPipeApplyDeleteRenderState(handle);
|
||||
MGPipeSlots().Free(MGPipeKind::RenderStateCso, m_entries[index].Cso);
|
||||
m_entries[index] = m_entries.back();
|
||||
m_entries.pop_back();
|
||||
++m_counters.Evictions;
|
||||
}
|
||||
|
||||
static constexpr Uint32 kAllPipelineChunks =
|
||||
static_cast<Uint32>((Uint64{1} << kMGPipePipelineChunkCount) - 1);
|
||||
|
||||
Vector<Entry> m_entries;
|
||||
Uint64 m_clock = 0;
|
||||
Counters m_counters;
|
||||
};
|
||||
|
||||
// The monolith's one cache, held beside the tracker. A Vector scan rather than a hash
|
||||
// map on purpose: 64 entries of Uint64 is a handful of cache lines, it is probed only
|
||||
// when the pipeline version moved, and it keeps the eviction order in the same array as
|
||||
// the content - a map would need a second structure to answer "which is oldest".
|
||||
inline MGPipeCsoCache& MGPipeCsoCacheInstance() {
|
||||
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason (MG_Impl/Pipe/Tracker.h): the
|
||||
// rule covers every MGPipe process singleton, not only the ones on today's death
|
||||
// paths.
|
||||
static MGPipeCsoCache* cache = new MGPipeCsoCache();
|
||||
return *cache;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
@@ -0,0 +1,659 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/FramebufferEmit.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The CLIENT side of P4a's framebuffer family: set_framebuffer_state, emitted at the validate
|
||||
// point once per bound TARGET that moved, or once with Target = Both when the two bindings
|
||||
// name the same object.
|
||||
//
|
||||
// THIS FILE IS CREATED BY THE CONTRACT COMMIT AND FILLED BY THE PACKAGE THAT OWNS IT, and the
|
||||
// split is the whole reason it exists this early. MG_Impl/Pipe/PipeFill.cpp is the contract
|
||||
// package's for the entire phase - it carries Coverage.def's enum-coupled block, the validate
|
||||
// point and the death helpers - so the emitter package must not edit it. What it edits instead
|
||||
// is this header: the emitter's BODY, and the value of kMGPipeWiredFramebufferSubsystem below.
|
||||
// That is what makes "no file is touched twice by two packages" structural rather than a
|
||||
// convention, and it is what the bb2a236d semantic-merge trap taught (two branches green
|
||||
// separately, the integrated tree not compiling).
|
||||
//
|
||||
// HEADER-ONLY, for the ownership reason Tracker.h and ResourceTracker.h both state: the root
|
||||
// CMakeLists.txt that would name a new .cpp is the contract package's and is frozen behind the
|
||||
// tag. MG_Impl/Pipe/PipeFill.cpp is the one translation unit that includes it in the library.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/SetHashSuppressor.h>
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Impl/Pipe/TextureEmit.h>
|
||||
#include <MG_Impl/Pipe/Tracker.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
#include <xxhash.h>
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// WHICH SUBSYSTEM BIT THIS BUILD ACTUALLY EMITS FOR. PipeFill.cpp ORs the four per-family
|
||||
// constants into kMGPipeWiredSubsystems, so the bit is added by the commit that gives the
|
||||
// emitters their bodies, with no file touched twice - and a Coverage.def row can never
|
||||
// silently drop a field on the floor before the call that carries it exists.
|
||||
//
|
||||
// TURNING IT ON RETIRES NO PULL. GetFramebufferBindingSlot is the family's one
|
||||
// Coverage.def emitted row and PipeFill.cpp's EmittedCallSuppliesTheWholeField answers
|
||||
// FALSE for it, with the reason: the field's storage is a BindingSlot<FramebufferObject> -
|
||||
// a frontend heap reference - and the call that supplies it carries eight-byte {slot, gen}
|
||||
// handles and a fully resolved descriptor. So this bit switches the EMISSION on and the
|
||||
// residual fill keeps writing the mirror, which is what keeps the verify lane at zero
|
||||
// divergence.
|
||||
inline constexpr Uint64 kMGPipeWiredFramebufferSubsystem = kMGPipeSubsystemFramebuffer;
|
||||
|
||||
inline Bool MGPipeFramebufferSubsystemEnabled() {
|
||||
return (kMGPipeWiredFramebufferSubsystem & kMGPipeSubsystemFramebuffer) != 0 &&
|
||||
(MG_Config::Features.PipePush & kMGPipeSubsystemFramebuffer) != 0;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-C1: the MGPSurface builder, one pure function, one statement per field
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
// MGPSurface::Kind's three constants ARE THE CONTRACT'S (ID-12 DV-4, c0c):
|
||||
// kMGPipeSurfaceKindNone / ...Texture / ...Renderbuffer live in MG_Pipe/MGPipeTypes.h under
|
||||
// exactly these names with the same MGPipeKind derivation and the same static_assert. This
|
||||
// package's copies were a redefinition in the same namespace and are deleted.
|
||||
|
||||
// The upload target an attachment names, RESOLVED: an attachment made through an entry
|
||||
// point that carries no face token stores TextureUploadTarget::Unknown, and the record goes
|
||||
// out fully resolved - nothing in it may require a lookup on the far side.
|
||||
//
|
||||
// THE FALLBACK IS ONLY LEGAL FOR A SINGLE-TARGET TEXTURE (m1), and v1's was not. The
|
||||
// precedent it copied - FramebufferAttachmentObject::GetSize - needs an EXTENT, which is
|
||||
// identical across a cube map's six faces; face IDENTITY is not, so
|
||||
// `glFramebufferTexture(GL_COLOR_ATTACHMENT0, cube, 0)` resolved to targets[0] and the
|
||||
// record ASSERTED CubeMapPositiveX for a layered attachment that names all six. A texture
|
||||
// with exactly one upload target has a [0] that IS the truth; anything else keeps Unknown,
|
||||
// which is the value the field already carries for "this attachment names no single face"
|
||||
// and which Layered = 1 tells the reader to ignore.
|
||||
inline MobileGL::TextureUploadTarget MGPipeResolveAttachmentUploadTarget(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
MobileGL::TextureUploadTarget resolved = attachment.GetTextureUploadTarget();
|
||||
if (resolved != MobileGL::TextureUploadTarget::Unknown) return resolved;
|
||||
const auto& texture = attachment.GetTexture();
|
||||
if (!texture) return MobileGL::TextureUploadTarget::Unknown;
|
||||
const auto& targets = texture->GetUploadTargets();
|
||||
return targets.size() == 1 ? targets[0] : MobileGL::TextureUploadTarget::Unknown;
|
||||
}
|
||||
|
||||
// ONE PURE FUNCTION, ONE STATEMENT PER FIELD, and that shape is a gate requirement rather
|
||||
// than taste: G7's scripted control stops this conversion copying exactly one member
|
||||
// (MGPSurface::Layered) and expects the framebuffer suite to go red NAMING that field. A
|
||||
// loop or a memcpy would make the control unanswerable.
|
||||
//
|
||||
// `res` is handed in because resolving it needs the slot allocator and this function stays
|
||||
// pure; `internalFormat` is INLINE in the record on purpose, so the four cross-object masks
|
||||
// fall out at push time with no lookup on the far side.
|
||||
// THE EMPTY POINT IS THE ZERO-INITIALISED RECORD EXCEPT FOR ITS TWO TARGET FIELDS. Both
|
||||
// are Uint16 enumerations whose zero is a REAL value - TextureTarget::Texture1D and
|
||||
// TextureUploadTarget::Texture1D - so a reader that forgot to gate on Kind would read a
|
||||
// plausible wrong answer rather than a nonsense one. Unknown (0xFFFF) is what the contract
|
||||
// spells for TextureTarget (kMGPipeSurfaceNoTextureTarget) and m6 applies the same rule to
|
||||
// UploadTarget, which shares the collision ID-12 DV-3 ruled on for MGPSubData::Target.
|
||||
inline MGPSurface MGPipeEmptySurface() {
|
||||
MGPSurface surface{};
|
||||
surface.UploadTarget = static_cast<Uint16>(MobileGL::TextureUploadTarget::Unknown);
|
||||
surface.TextureTarget = kMGPipeSurfaceNoTextureTarget;
|
||||
return surface;
|
||||
}
|
||||
|
||||
inline MGPSurface MGPipeBuildSurface(const MG_State::GLState::FramebufferAttachmentObject& attachment,
|
||||
MGPipeHandle res) {
|
||||
MGPSurface surface = MGPipeEmptySurface();
|
||||
if (attachment.IsEmpty()) return surface;
|
||||
surface.Res = res;
|
||||
if (attachment.IsTexture()) {
|
||||
const auto& texture = attachment.GetTexture();
|
||||
surface.Kind = kMGPipeSurfaceKindTexture;
|
||||
surface.InternalFormat = static_cast<Uint32>(texture->GetFormat());
|
||||
surface.Layered = attachment.IsLayered() ? 1 : 0;
|
||||
surface.Level = static_cast<Uint16>(std::max<Int>(attachment.GetTextureLevel(), 0));
|
||||
surface.Layer = static_cast<Uint32>(std::max<Int>(attachment.GetTextureLayer(), 0));
|
||||
surface.UploadTarget = static_cast<Uint16>(MGPipeResolveAttachmentUploadTarget(attachment));
|
||||
// ID-12 DV-5: the field that WAS Pad0, and the size did not move. The four
|
||||
// cross-object masks all reduce to (format, TEXTURE TARGET) -
|
||||
// ShouldUseCaveatTextureFormat / BackendTextureFormatAddsAlpha - and no
|
||||
// TextureUploadTarget -> TextureTarget inverse exists anywhere in the tree, so
|
||||
// without this the inline InternalFormat cannot make them fall out at push time and
|
||||
// the backend keeps reading the frontend attachment objects.
|
||||
surface.TextureTarget = static_cast<Uint16>(texture->GetTarget());
|
||||
return surface;
|
||||
}
|
||||
const auto& renderbuffer = attachment.GetRenderbuffer();
|
||||
surface.Kind = kMGPipeSurfaceKindRenderbuffer;
|
||||
surface.InternalFormat = static_cast<Uint32>(renderbuffer->GetInternalFormat());
|
||||
surface.Layered = 0;
|
||||
surface.Level = 0;
|
||||
surface.Layer = 0;
|
||||
return surface;
|
||||
}
|
||||
|
||||
// MGPFramebufferState::DrawBuffers[i]: an index INTO THIS RECORD'S OWN Color[] array, and
|
||||
// -1 for NONE, which is the field's documented convention read literally.
|
||||
//
|
||||
// THE FOUR DEFAULT-FRAMEBUFFER TOKENS map to 0, and that is a deliberate narrowing rather
|
||||
// than an oversight: a default framebuffer has one colour surface, this record carries it
|
||||
// in Color[0] (see MGPipeBuildFramebufferState), and IsDefault is what tells the server
|
||||
// which framebuffer it is looking at. The distinction the narrowing loses is FRONT versus
|
||||
// BACK and LEFT versus RIGHT, which MobileGL's frontend never gives a default framebuffer
|
||||
// in the first place - FramebufferObject's constructor seeds BackLeft and nothing writes
|
||||
// another. A phase that needs stereo has to widen the field, not re-encode this one.
|
||||
inline Int8 MGPipeDrawBufferIndex(MobileGL::FramebufferAttachmentType buffer) {
|
||||
using MobileGL::FramebufferAttachmentType;
|
||||
if (buffer == FramebufferAttachmentType::None) return -1;
|
||||
if (buffer >= FramebufferAttachmentType::Color0 && buffer <= FramebufferAttachmentType::ColorMax) {
|
||||
return static_cast<Int8>(static_cast<Int>(buffer) - static_cast<Int>(FramebufferAttachmentType::Color0));
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
// m2: A DRAW-BUFFER TOKEN CAN NAME A COLOUR POINT THE RECORD CANNOT CARRY, and D-C3's
|
||||
// refusal loop only ever scanned ATTACHMENTS. `glDrawBuffers(1, {GL_COLOR_ATTACHMENT10})`
|
||||
// with nothing attached at 10 is legal state - draw-incomplete, but legal - and the index
|
||||
// above would have written 10 into a record whose Color[] is 8 wide, so the server would
|
||||
// index out of its own storage or invent a bound the record does not carry. Truncating
|
||||
// silently is the bug class this phase is closing, so the record is refused exactly as an
|
||||
// over-wide attachment is.
|
||||
inline Bool MGPipeDrawBufferIsInsideTheWireWidth(MobileGL::FramebufferAttachmentType buffer) {
|
||||
using MobileGL::FramebufferAttachmentType;
|
||||
if (buffer < FramebufferAttachmentType::Color0 || buffer > FramebufferAttachmentType::ColorMax) {
|
||||
return true; // None and the four default-framebuffer tokens; neither indexes Color[]
|
||||
}
|
||||
return static_cast<Int>(buffer) - static_cast<Int>(FramebufferAttachmentType::Color0) <
|
||||
static_cast<Int>(kMGPipeMaxColorAttachments);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-C4: ContentHash, and the one input it must not swallow
|
||||
// ---------------------------------------------------------------------------------
|
||||
//
|
||||
// XXH64 over the WHOLE record with ContentHash itself zeroed, computed field-wise into a
|
||||
// zero-initialised staging copy so that no padding byte can enter the hash. Two jobs: the
|
||||
// server's render-pass memo key, and this client's emission suppressor.
|
||||
//
|
||||
// IT MUST COVER Fbo. A recycled framebuffer handle whose successor happens to carry an
|
||||
// identical attachment set would otherwise be suppressed against its predecessor; Fbo
|
||||
// carries Gen, so it cannot be.
|
||||
//
|
||||
// IT MUST COVER DrawBuffers[8], and this is the trap worth naming. The backend derives the
|
||||
// fragColor BROADCAST COUNT from the draw-buffer array, and it does that at the verb, from
|
||||
// the framebuffer state it then holds, precisely so a program can relink inside the same
|
||||
// draw. A hash that did not cover the array would let a suppressed set_framebuffer_state
|
||||
// mean "the draw buffers did not move" when they had, and the shader would be specialised
|
||||
// for the previous output shape. With the array in the hash, a suppression provably means
|
||||
// the array did not move, which provably means the broadcast count did not move.
|
||||
inline void MGPipeCopySurfaceForHash(MGPSurface& dst, const MGPSurface& src) {
|
||||
dst.Res = src.Res;
|
||||
dst.InternalFormat = src.InternalFormat;
|
||||
dst.Kind = src.Kind;
|
||||
dst.Layered = src.Layered;
|
||||
dst.Level = src.Level;
|
||||
dst.Layer = src.Layer;
|
||||
dst.UploadTarget = src.UploadTarget;
|
||||
// MANDATORY, not optional: TextureTarget is a PipeFields.def row now, so a
|
||||
// field-wise copy that skipped it would suppress a record whose only moved field is
|
||||
// the attachment's texture target - and that field decides three of the four
|
||||
// cross-object masks.
|
||||
dst.TextureTarget = src.TextureTarget;
|
||||
}
|
||||
|
||||
inline Uint64 MGPipeFramebufferStateContentHash(const MGPFramebufferState& state) {
|
||||
MGPFramebufferState staging{};
|
||||
staging.Fbo = state.Fbo;
|
||||
for (SizeT i = 0; i < kMGPipeMaxColorAttachments; ++i) {
|
||||
MGPipeCopySurfaceForHash(staging.Color[i], state.Color[i]);
|
||||
}
|
||||
MGPipeCopySurfaceForHash(staging.Depth, state.Depth);
|
||||
MGPipeCopySurfaceForHash(staging.Stencil, state.Stencil);
|
||||
MGPipeCopySurfaceForHash(staging.ReadSurface, state.ReadSurface);
|
||||
for (SizeT i = 0; i < kMGPipeMaxColorAttachments; ++i) {
|
||||
staging.DrawBuffers[i] = state.DrawBuffers[i];
|
||||
}
|
||||
staging.Width = state.Width;
|
||||
staging.Height = state.Height;
|
||||
staging.Layers = state.Layers;
|
||||
staging.Samples = state.Samples;
|
||||
staging.FixedSampleLocations = state.FixedSampleLocations;
|
||||
staging.IsDefault = state.IsDefault;
|
||||
staging.Complete = state.Complete;
|
||||
staging.Target = state.Target;
|
||||
// staging.ContentHash stays 0 - that is the whole point.
|
||||
return XXH64(&staging, sizeof(staging), 0);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// The emitter
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
class MGPipeFramebufferEmitter {
|
||||
public:
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
using FramebufferObject = MG_State::GLState::FramebufferObject;
|
||||
using FramebufferAttachmentType = MobileGL::FramebufferAttachmentType;
|
||||
|
||||
// The handle for `fbo`. kMGPipeDefaultFramebuffer ({0,1}) for the default framebuffer,
|
||||
// which is what retires the four pDefaultFramebufferInfo->defaultFBO identity
|
||||
// comparisons into an ordinary handle compare; a client-minted {slot, gen} otherwise.
|
||||
//
|
||||
// Minted, never gated: a framebuffer handle is CLIENT state and costs one free-list pop.
|
||||
static MGPipeHandle HandleFor(const FramebufferObject& fbo) {
|
||||
if (fbo.IsDefaultFramebuffer()) return kMGPipeDefaultFramebuffer;
|
||||
return MGPipeSlots().Acquire(MGPipeKind::Framebuffer, fbo.GetLifetimeId());
|
||||
}
|
||||
|
||||
// Returns the bytes that went on the wire, for the per-draw payload histogram.
|
||||
Uint64 EmitFramebufferState(GLContext& ctx) {
|
||||
if (!MGPipeFramebufferSubsystemEnabled()) return 0;
|
||||
const auto& drawFbo = ctx.GetFramebufferBindingSlot(MobileGL::FramebufferTarget::Draw).GetBoundObject();
|
||||
const auto& readFbo = ctx.GetFramebufferBindingSlot(MobileGL::FramebufferTarget::Read).GetBoundObject();
|
||||
if (!drawFbo && !readFbo) return 0;
|
||||
|
||||
// ONE OBJECT BOUND TO BOTH TARGETS IS ONE RECORD WITH Target = Both, and that is
|
||||
// not an optimisation: Espryt's "same FBO as draw" skip is the habitat of the
|
||||
// read-buffer defect class, and a record that says which target it describes turns
|
||||
// "apply the draw buffers only for the draw target" from call-site discipline into
|
||||
// a one-line test on the far side.
|
||||
const Bool shared = drawFbo && readFbo && drawFbo.get() == readFbo.get();
|
||||
|
||||
MGPFramebufferState drawState{};
|
||||
MGPFramebufferState readState{};
|
||||
Bool drawOk = false;
|
||||
Bool readOk = false;
|
||||
if (shared) {
|
||||
drawOk = BuildFramebufferState(*drawFbo, MGPipeFramebufferTarget::Both, drawState);
|
||||
} else {
|
||||
if (drawFbo) {
|
||||
drawOk = BuildFramebufferState(*drawFbo, MGPipeFramebufferTarget::Draw, drawState);
|
||||
}
|
||||
if (readFbo) {
|
||||
readOk = BuildFramebufferState(*readFbo, MGPipeFramebufferTarget::Read, readState);
|
||||
}
|
||||
}
|
||||
if (!drawOk && !readOk) return 0;
|
||||
|
||||
// THE SUPPRESSOR SLOT IS FED THE COMBINED ANSWER and the per-target latches decide
|
||||
// which of the two records actually goes out. The slot exists so that
|
||||
// InvalidateAll() on a fresh context reaches this family like every other, and so
|
||||
// that "nothing moved" costs one compare rather than two.
|
||||
const Uint64 drawHash = drawOk ? drawState.ContentHash : 0;
|
||||
const Uint64 readHash = readOk ? readState.ContentHash : 0;
|
||||
const Uint64 combined =
|
||||
MGPipeMixShutter(MGPipeMixShutter(drawHash, readHash), shared ? 1u : 0u);
|
||||
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetFramebufferState,
|
||||
combined)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Uint64 bytes = 0;
|
||||
if (shared) {
|
||||
if (drawOk && (drawHash != m_lastEmitted[kDraw] || drawHash != m_lastEmitted[kRead])) {
|
||||
bytes += Emit(drawState);
|
||||
m_lastEmitted[kDraw] = drawHash;
|
||||
m_lastEmitted[kRead] = drawHash;
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
if (drawOk && drawHash != m_lastEmitted[kDraw]) {
|
||||
bytes += Emit(drawState);
|
||||
m_lastEmitted[kDraw] = drawHash;
|
||||
}
|
||||
if (readOk && readHash != m_lastEmitted[kRead]) {
|
||||
bytes += Emit(readState);
|
||||
m_lastEmitted[kRead] = readHash;
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
|
||||
// ID-19(c): EVERY DSA ENTRY POINT THAT HANDS A FRAMEBUFFER TO THE SERVER BY NAME IS
|
||||
// PRECEDED BY A RECORD FOR IT, and that is the phase's main correction rather than a
|
||||
// nicety. With only the two BOUND-target records, glClearNamedFramebufferfv(fbo) on an
|
||||
// unbound fbo made the backend mint a fresh driver framebuffer with NO ATTACHMENTS,
|
||||
// find no record for it, decline, and issue the clear against it anyway -
|
||||
// GL_INVALID_FRAMEBUFFER_OPERATION and nothing cleared, where the legacy arm cleared
|
||||
// correctly (esprytobj C-1).
|
||||
//
|
||||
// THE TARGET IS Named ONLY WHEN THE OBJECT IS BOUND TO NEITHER BINDING. A record always
|
||||
// writes FramebufferRecords[Fbo.Slot]; Draw/Read/Both ADDITIONALLY set the bound
|
||||
// handle(s). So handing a currently-bound framebuffer a Named record would overwrite
|
||||
// the bound record's Target with one that says "no binding" while BoundFramebuffer
|
||||
// still names it, and the server would read a record whose Target contradicts the
|
||||
// binding it is resolved through. Re-asserting the binding the object already has is
|
||||
// free (the content hash suppresses it) and keeps the two consistent.
|
||||
//
|
||||
// Returns the bytes that went on the wire.
|
||||
Uint64 EmitFramebufferByName(const FramebufferObject& fbo) {
|
||||
if (!MGPipeFramebufferSubsystemEnabled()) return 0;
|
||||
MGPipeFramebufferTarget target = MGPipeFramebufferTarget::Named;
|
||||
const Bool boundToDraw = IsBoundTo(fbo, MobileGL::FramebufferTarget::Draw);
|
||||
const Bool boundToRead = IsBoundTo(fbo, MobileGL::FramebufferTarget::Read);
|
||||
if (boundToDraw && boundToRead) {
|
||||
target = MGPipeFramebufferTarget::Both;
|
||||
} else if (boundToDraw) {
|
||||
target = MGPipeFramebufferTarget::Draw;
|
||||
} else if (boundToRead) {
|
||||
target = MGPipeFramebufferTarget::Read;
|
||||
}
|
||||
|
||||
MGPFramebufferState state{};
|
||||
if (!BuildFramebufferState(fbo, target, state)) return 0;
|
||||
|
||||
// THE SUPPRESSOR IS KEYED BY THE FRAMEBUFFER THE RECORD NAMES, never by one global
|
||||
// slot (MGPipeTypes.h states the rule): two different objects' Named records in a
|
||||
// row must both go out, and a Named record must never be suppressed against the
|
||||
// same object's bound record or the reverse. Target is a ContentHash input, so the
|
||||
// second half holds by construction; the per-object table is what buys the first.
|
||||
// The two BOUND latches stay what they are - "does the server's draw/read binding
|
||||
// already hold this record" - and a bound-target emission from here consults them,
|
||||
// because a rebind of an unchanged object must still move the binding.
|
||||
if (target == MGPipeFramebufferTarget::Named) {
|
||||
NamedEntry& entry = NamedEntryFor(state.Fbo);
|
||||
if (entry.Has && entry.Gen == state.Fbo.Gen && entry.LastHash == state.ContentHash) {
|
||||
return 0;
|
||||
}
|
||||
const Uint64 bytes = Emit(state);
|
||||
entry.Has = true;
|
||||
entry.Gen = state.Fbo.Gen;
|
||||
entry.LastHash = state.ContentHash;
|
||||
return bytes;
|
||||
}
|
||||
if (target == MGPipeFramebufferTarget::Both) {
|
||||
if (state.ContentHash == m_lastEmitted[kDraw] && state.ContentHash == m_lastEmitted[kRead]) {
|
||||
return 0;
|
||||
}
|
||||
const Uint64 bytes = Emit(state);
|
||||
m_lastEmitted[kDraw] = state.ContentHash;
|
||||
m_lastEmitted[kRead] = state.ContentHash;
|
||||
return bytes;
|
||||
}
|
||||
const SizeT slot = target == MGPipeFramebufferTarget::Read ? kRead : kDraw;
|
||||
if (state.ContentHash == m_lastEmitted[slot]) return 0;
|
||||
const Uint64 bytes = Emit(state);
|
||||
m_lastEmitted[slot] = state.ContentHash;
|
||||
return bytes;
|
||||
}
|
||||
|
||||
// ---- what a unit case reads. The emitter builds INTO these and hands the applier the
|
||||
// same objects, so "what was emitted" costs no copy. ----
|
||||
// ---- the death half (P4a final review C-2) ----
|
||||
//
|
||||
// Called by the contract's death helper before the slot is freed (there is no wire
|
||||
// delete for this kind, D-I2, so this is the only client-side thing a framebuffer's
|
||||
// death has to do). The per-object Named latch is the entry: a recycled handle's Gen
|
||||
// already refuses the stale latch, so this is hygiene rather than a fix - the rule
|
||||
// (ID-8) is that whatever mints a handle retires everything it keeps under it at the
|
||||
// death, and every P4a kind takes the same shape. Gen-keyed for a late notice.
|
||||
void NoteFramebufferDied(MGPipeHandle handle) {
|
||||
const SizeT slot = handle.Slot;
|
||||
if (MGPipeHandleIsNull(handle) || slot >= m_named.size()) return;
|
||||
if (m_named[slot].Gen == handle.Gen) m_named[slot] = NamedEntry{};
|
||||
}
|
||||
// "Does this emitter hold a Named-record latch for this handle at its generation."
|
||||
Bool NamedRecordIsLatched(MGPipeHandle handle) const {
|
||||
const SizeT slot = handle.Slot;
|
||||
if (MGPipeHandleIsNull(handle) || slot >= m_named.size()) return false;
|
||||
return m_named[slot].Has && m_named[slot].Gen == handle.Gen;
|
||||
}
|
||||
|
||||
const MGPFramebufferState& LastDraw() const { return m_lastDraw; }
|
||||
const MGPFramebufferState& LastRead() const { return m_lastRead; }
|
||||
const MGPFramebufferState& LastNamed() const { return m_lastNamed; }
|
||||
Uint64 EmissionCount() const { return m_emissions; }
|
||||
Uint64 RefusedCount() const { return m_refusals; }
|
||||
|
||||
// A fresh context: what the server has is no longer what this emitter last sent. Only
|
||||
// LATCHES reset here - MGPipeApplierReset clears the applier's DrawFramebuffer and
|
||||
// ReadFramebuffer working state, so these mirrors have to go with them or the first
|
||||
// emission after a make-current would be suppressed as unchanged and the server would
|
||||
// draw into the previous context's framebuffer. The suppressor slot is invalidated by
|
||||
// the validate point's own InvalidateAll(), beside this call.
|
||||
void Reset() {
|
||||
m_lastEmitted[kDraw] = 0;
|
||||
m_lastEmitted[kRead] = 0;
|
||||
// The per-object latch goes too, and the safe direction is why: MGPipeApplierReset
|
||||
// keeps FramebufferRecords standing (they are object state, ID-19(b)) but
|
||||
// ReleaseObjectRecords clears the whole table, and this emitter cannot tell the two
|
||||
// scopes apart from here. Keeping a latch across a table that may have been dropped
|
||||
// would suppress the one record that had to go out; dropping it costs one extra
|
||||
// 304-byte record per named framebuffer after a context switch.
|
||||
m_named.clear();
|
||||
}
|
||||
|
||||
void ResetCounters() { m_emissions = m_refusals = 0; }
|
||||
|
||||
void ResetForTest() {
|
||||
Reset();
|
||||
ResetCounters();
|
||||
m_lastDraw = MGPFramebufferState{};
|
||||
m_lastRead = MGPFramebufferState{};
|
||||
m_lastNamed = MGPFramebufferState{};
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr SizeT kDraw = 0;
|
||||
static constexpr SizeT kRead = 1;
|
||||
|
||||
Uint64 Emit(const MGPFramebufferState& state) {
|
||||
if (state.Target == static_cast<Uint8>(MGPipeFramebufferTarget::Named)) {
|
||||
m_lastNamed = state;
|
||||
} else if (state.Target == static_cast<Uint8>(MGPipeFramebufferTarget::Read)) {
|
||||
m_lastRead = state;
|
||||
} else {
|
||||
m_lastDraw = state;
|
||||
if (state.Target == static_cast<Uint8>(MGPipeFramebufferTarget::Both)) m_lastRead = state;
|
||||
}
|
||||
MGPipeApplySetFramebufferState(state);
|
||||
++m_emissions;
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::FramebufferEmissions, 1);
|
||||
}
|
||||
return sizeof(MGPFramebufferState);
|
||||
}
|
||||
|
||||
// ONE RECORD DESCRIBES ONE FRAMEBUFFER OBJECT - the one named by `fbo` - and every
|
||||
// field in it is a property of THAT object. Target is the only binding-specific one.
|
||||
//
|
||||
// ReadSurface IS RESOLVED FROM THIS FRAMEBUFFER'S OWN READ BUFFER UNDER EVERY TARGET,
|
||||
// Named included (c0e / MGPipeTypes.h). v1 resolved a Draw record's ReadSurface from
|
||||
// the READ-bound object, which was D-C2's letter and muddled in substance: the record
|
||||
// then described a surface that is not part of the framebuffer its own Fbo names, and a
|
||||
// glReadBuffer on the read FBO moved the DRAW record's ContentHash and forced a
|
||||
// redundant draw emission. Resolving it per object is what makes the
|
||||
// read-buffer-shared-FBO defect class unrepresentable rather than merely fixed - the
|
||||
// record carries a surface, not an index, and no field of it refers to "whatever is
|
||||
// bound".
|
||||
Bool BuildFramebufferState(const FramebufferObject& fbo, MGPipeFramebufferTarget target,
|
||||
MGPFramebufferState& out) {
|
||||
// D-C3, THE CLIENT HALF OF THE BRING-UP REFUSAL. The wire array is 8 wide and
|
||||
// GetDynamicParameters().MaxColorAttachments is the driver's raw ES cap, not
|
||||
// clamped to 8 on the GLES path. An attachment point at or above the wire width
|
||||
// cannot be carried at all, so the record is REFUSED and the legacy arm runs -
|
||||
// truncating it silently is exactly the bug class this phase is closing. The
|
||||
// backend half of the same refusal (bit 9 declined at its first lookup, with one
|
||||
// ERROR naming the cap) rides ResolveFramebufferSubsystemArm.
|
||||
for (Int point = static_cast<Int>(FramebufferAttachmentType::Color0) +
|
||||
static_cast<Int>(kMGPipeMaxColorAttachments);
|
||||
point <= static_cast<Int>(FramebufferAttachmentType::ColorMax); ++point) {
|
||||
if (fbo.GetAttachment(static_cast<FramebufferAttachmentType>(point)).IsEmpty()) continue;
|
||||
MGLOG_E_ONCE("MGPipe: framebuffer %u has an attachment at colour point %d, which is at or "
|
||||
"above the wire width of %u - set_framebuffer_state is refused rather than "
|
||||
"truncated and the legacy arm runs",
|
||||
fbo.GetExternalIndex(),
|
||||
point - static_cast<Int>(FramebufferAttachmentType::Color0),
|
||||
static_cast<Uint>(kMGPipeMaxColorAttachments));
|
||||
++m_refusals;
|
||||
return false;
|
||||
}
|
||||
|
||||
// m2, THE SAME REFUSAL ONE FIELD OVER. A draw-buffer token may name a colour point
|
||||
// at or above the wire width with nothing attached there, which the loop above
|
||||
// cannot see; MGPipeDrawBufferIndex would then write 8..31 into an 8-wide array.
|
||||
{
|
||||
const auto& tokens = fbo.GetDrawBuffers();
|
||||
for (SizeT i = 0; i < kMGPipeMaxColorAttachments; ++i) {
|
||||
if (MGPipeDrawBufferIsInsideTheWireWidth(tokens[i])) continue;
|
||||
MGLOG_E_ONCE("MGPipe: framebuffer %u names colour point %d in draw buffer %u, which is "
|
||||
"at or above the wire width of %u - set_framebuffer_state is refused "
|
||||
"rather than truncated and the legacy arm runs",
|
||||
fbo.GetExternalIndex(),
|
||||
static_cast<Int>(tokens[i]) -
|
||||
static_cast<Int>(FramebufferAttachmentType::Color0),
|
||||
static_cast<Uint>(i), static_cast<Uint>(kMGPipeMaxColorAttachments));
|
||||
++m_refusals;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
out = MGPFramebufferState{};
|
||||
out.Fbo = HandleFor(fbo);
|
||||
out.Target = static_cast<Uint8>(target);
|
||||
out.IsDefault = fbo.IsDefaultFramebuffer() ? 1 : 0;
|
||||
|
||||
// THE COLOUR POINTS. A default framebuffer keeps its one colour surface under
|
||||
// BackLeft rather than under Color0, and the record has exactly one place to put
|
||||
// it: Color[0], which is also the index MGPipeDrawBufferIndex maps that token to,
|
||||
// so the array and the draw-buffer indices agree by construction.
|
||||
if (out.IsDefault != 0) {
|
||||
out.Color[0] = SurfaceOf(fbo, FramebufferAttachmentType::BackLeft);
|
||||
} else {
|
||||
for (SizeT i = 0; i < kMGPipeMaxColorAttachments; ++i) {
|
||||
out.Color[i] = SurfaceOf(fbo, static_cast<FramebufferAttachmentType>(
|
||||
static_cast<Int>(FramebufferAttachmentType::Color0) +
|
||||
static_cast<Int>(i)));
|
||||
}
|
||||
}
|
||||
out.Depth = SurfaceOf(fbo, FramebufferAttachmentType::Depth);
|
||||
out.Stencil = SurfaceOf(fbo, FramebufferAttachmentType::Stencil);
|
||||
out.ReadSurface = SurfaceOf(fbo, fbo.GetReadBuffer());
|
||||
|
||||
const auto& drawBuffers = fbo.GetDrawBuffers();
|
||||
for (SizeT i = 0; i < kMGPipeMaxColorAttachments; ++i) {
|
||||
out.DrawBuffers[i] = MGPipeDrawBufferIndex(drawBuffers[i]);
|
||||
}
|
||||
|
||||
FillGeometry(fbo, out);
|
||||
// Complete is FramebufferObject::CheckCompleteness(), the FRONTEND-ONLY answer, and
|
||||
// never glCheckFramebufferStatus's: that entry point additionally consults the
|
||||
// backend's probed format-capability cache, and a client emitting it would be
|
||||
// reading the backend from the client side - the exact coupling this boundary
|
||||
// exists to remove. glCheckFramebufferStatus keeps answering from the frontend
|
||||
// exactly as it does today.
|
||||
out.Complete = fbo.CheckCompleteness() ? 1 : 0;
|
||||
out.ContentHash = MGPipeFramebufferStateContentHash(out);
|
||||
return true;
|
||||
}
|
||||
|
||||
static Bool IsBoundTo(const FramebufferObject& fbo, MobileGL::FramebufferTarget target) {
|
||||
if (MG_State::pGLContext == nullptr) return false;
|
||||
const auto& bound = MG_State::pGLContext->GetFramebufferBindingSlot(target).GetBoundObject();
|
||||
return bound && bound.get() == &fbo;
|
||||
}
|
||||
|
||||
struct NamedEntry {
|
||||
Uint32 Gen = 0;
|
||||
Uint64 LastHash = 0;
|
||||
Bool Has = false;
|
||||
};
|
||||
|
||||
NamedEntry& NamedEntryFor(MGPipeHandle fbo) {
|
||||
const SizeT slot = fbo.Slot;
|
||||
if (slot >= m_named.size()) m_named.resize(slot + 1);
|
||||
return m_named[slot];
|
||||
}
|
||||
|
||||
MGPSurface SurfaceOf(const FramebufferObject& fbo, FramebufferAttachmentType type) {
|
||||
if (type == FramebufferAttachmentType::None || type == FramebufferAttachmentType::Unknown) {
|
||||
return MGPipeEmptySurface();
|
||||
}
|
||||
const auto& attachment = fbo.GetAttachment(type);
|
||||
if (attachment.IsEmpty()) return MGPipeEmptySurface();
|
||||
MGPipeTextureEmitter& textures = MGPipeTextureEmitterInstance();
|
||||
// D-A4's two producers: an attachment point is what sets RENDER_TARGET and
|
||||
// DEPTH_STENCIL, the two sticky bind bits nothing set before P4a. Sticky and ORed,
|
||||
// so a texture that was ever a colour attachment keeps saying so, and the mask is
|
||||
// republished on the resource's next respecify.
|
||||
const Uint16 bit = (type == FramebufferAttachmentType::Depth ||
|
||||
type == FramebufferAttachmentType::Stencil)
|
||||
? static_cast<Uint16>(kMGPipeBindDepthStencil)
|
||||
: static_cast<Uint16>(kMGPipeBindRenderTarget);
|
||||
MGPipeHandle res = kMGPipeNullHandle;
|
||||
if (attachment.IsTexture()) {
|
||||
const auto& texture = attachment.GetTexture();
|
||||
res = textures.AcquireTexture(texture->GetLifetimeId(), texture.get());
|
||||
textures.NoteTextureBoundAs(res, bit);
|
||||
} else if (attachment.IsRenderbuffer()) {
|
||||
const auto& renderbuffer = attachment.GetRenderbuffer();
|
||||
res = textures.AcquireRenderbuffer(renderbuffer->GetLifetimeId());
|
||||
textures.NoteRenderbufferBoundAs(res, bit);
|
||||
}
|
||||
return MGPipeBuildSurface(attachment, res);
|
||||
}
|
||||
|
||||
// The attachments' common extent, and the ARB_framebuffer_no_attachments defaults when
|
||||
// there is no attachment at all (GL 4.6 core table 23.24 - the shape a framebuffer with
|
||||
// no attachments rasterizes at).
|
||||
static void FillGeometry(const FramebufferObject& fbo, MGPFramebufferState& out) {
|
||||
Bool found = false;
|
||||
for (const auto& attachment : fbo.GetAllAttachmentObjects()) {
|
||||
if (attachment.IsEmpty()) continue;
|
||||
const IntVec3 size = attachment.GetSize();
|
||||
if (!found) {
|
||||
out.Width = static_cast<Uint16>(std::clamp<Int>(size.x(), 0, 0xFFFF));
|
||||
out.Height = static_cast<Uint16>(std::clamp<Int>(size.y(), 0, 0xFFFF));
|
||||
out.Layers = static_cast<Uint16>(
|
||||
attachment.IsLayered() ? std::clamp<Int>(size.z(), 1, 0xFFFF) : 1);
|
||||
if (attachment.IsTexture()) {
|
||||
const auto& texture = attachment.GetTexture();
|
||||
out.Samples = static_cast<Uint16>(std::max<Int>(texture->GetSamples(), 0));
|
||||
out.FixedSampleLocations = texture->HasFixedSampleLocations() ? 1 : 0;
|
||||
} else {
|
||||
out.Samples = static_cast<Uint16>(
|
||||
std::max<Int>(attachment.GetRenderbuffer()->GetSamples(), 0));
|
||||
out.FixedSampleLocations = 1;
|
||||
}
|
||||
found = true;
|
||||
}
|
||||
}
|
||||
if (found) return;
|
||||
out.Width = static_cast<Uint16>(std::clamp<Int>(fbo.GetDefaultWidth(), 0, 0xFFFF));
|
||||
out.Height = static_cast<Uint16>(std::clamp<Int>(fbo.GetDefaultHeight(), 0, 0xFFFF));
|
||||
out.Layers = static_cast<Uint16>(std::clamp<Int>(fbo.GetDefaultLayers(), 0, 0xFFFF));
|
||||
out.Samples = static_cast<Uint16>(std::clamp<Int>(fbo.GetDefaultSamples(), 0, 0xFFFF));
|
||||
out.FixedSampleLocations = fbo.GetDefaultFixedSampleLocations() ? 1 : 0;
|
||||
}
|
||||
|
||||
Array<Uint64, 2> m_lastEmitted{};
|
||||
// The per-FRAMEBUFFER suppressor for Named records, slot-indexed with the generation
|
||||
// checked, exactly as the applier's own table is. A framebuffer has no wire lifetime
|
||||
// (D-I2), so a successor simply overwrites its predecessor's entry.
|
||||
Vector<NamedEntry> m_named;
|
||||
MGPFramebufferState m_lastDraw{};
|
||||
MGPFramebufferState m_lastRead{};
|
||||
MGPFramebufferState m_lastNamed{};
|
||||
Uint64 m_emissions = 0;
|
||||
Uint64 m_refusals = 0;
|
||||
};
|
||||
|
||||
inline MGPipeFramebufferEmitter& MGPipeFramebufferEmitterInstance() {
|
||||
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason (MG_Impl/Pipe/Tracker.h): the
|
||||
// rule covers every MGPipe process singleton, not only the ones a frontend destructor
|
||||
// reaches today, and it is what keeps exit() out of a torn-down pipe.
|
||||
static MGPipeFramebufferEmitter* emitter = new MGPipeFramebufferEmitter();
|
||||
return *emitter;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
@@ -0,0 +1,182 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/ImageEmit.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The CLIENT side of set_shader_images, the third of P4a's kVarTail unit sets. It rides
|
||||
// SamplerEmit.h's subsystem bit (kMGPipeWiredSamplerSubsystem): one family, one A/B.
|
||||
//
|
||||
// TWO INVARIANTS THAT MUST SURVIVE INTO THE BODY, and they are the kind an optimisation
|
||||
// deletes:
|
||||
// 1. THE HIGH-WATER-ZERO EARLY-OUT. An image high-water mark of 0 emits nothing, BEFORE any
|
||||
// hash - that is what makes every Minecraft draw pay one integer test for a feature it
|
||||
// does not use.
|
||||
// 2. THE SWEEP'S GATE IS KEYED ON FRONTEND GENERATIONS AND DELIBERATELY NOT ON A BACKEND
|
||||
// RE-MINT COUNTER. A texture bound ONLY to an image unit is re-minted INSIDE the sweep,
|
||||
// so a server-side epoch would be bumped after the gate had already declined. The
|
||||
// client's bit-14 shutter is Mix(Mix(textureContent, textureParams), programImageUnitVersion)
|
||||
// - all three FRONTEND counters - so the property is preserved by construction, and it is
|
||||
// written here because it is invisible from the shutter itself.
|
||||
//
|
||||
// The record carries the APPLICATION's format and access; the bind-format recast (a GL_RG32F
|
||||
// bind is INVALID_VALUE on 19 of 26 non-core formats on Adreno) and the buffer-texture split
|
||||
// view stay SERVER-side and unchanged. ContentHash therefore has to cover InternalFormat and
|
||||
// Access as well as the binding, because the format the shader was built against is live
|
||||
// glBindImageTexture state and the format-less image bake keys on it.
|
||||
//
|
||||
// THIS FILE IS CREATED BY THE CONTRACT COMMIT AND FILLED BY THE PACKAGE THAT OWNS IT - see
|
||||
// FramebufferEmit.h for why, in full.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/SamplerEmit.h>
|
||||
#include <MG_Impl/Pipe/SetHashSuppressor.h>
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_Pipe/PipeMutation.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/GLState/TextureState/TextureState.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// D-G3. Over the tail with Start and Count mixed in, the same shape the two sampler sets
|
||||
// use - and it covers InternalFormat and Access because those are live glBindImageTexture
|
||||
// state that the format-less image bake keys on, not decoration.
|
||||
inline Uint64 MGPipeShaderImageSetContentHash(const MGPImageView* entries, Uint32 start, Uint32 count) {
|
||||
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPImageView), 0);
|
||||
hash = MGPipeMixShutter(hash, start);
|
||||
hash = MGPipeMixShutter(hash, count);
|
||||
return hash;
|
||||
}
|
||||
|
||||
class MGPipeImageEmitter {
|
||||
public:
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
|
||||
// set_shader_images. Start is 0 and Count is the image-unit window described below.
|
||||
//
|
||||
// WHERE THE HIGH-WATER MARK COMES FROM, because the frontend has none and this is the
|
||||
// one place a reader will look for it. DirectGLES keeps g_imageUnitHighWaterMark, but
|
||||
// that is written from inside its own per-unit sync and lives on the far side of the
|
||||
// boundary; TextureState::NoteUnitTouched is the TEXTURE-unit path and
|
||||
// glBindImageTexture does not reach it. Adding a counter to TextureState would edit
|
||||
// another package's file and resize the pull build's object, which G1 forbids outright.
|
||||
//
|
||||
// So the window is derived instead, from the one thing that decides whether an image
|
||||
// unit can matter at all: the highest image unit the CURRENT PROGRAM names, memoised
|
||||
// per program state in SamplerEmit.h's shared inversion, UNIONED with a sticky mark of
|
||||
// every unit this emitter has already described. A program with no image uniforms
|
||||
// gives MaxImageUnit == -1 and, with nothing sticky yet, a window of 0 - which is the
|
||||
// zero early-out, taken BEFORE any hash and before any 192-entry walk, exactly as
|
||||
// property 1 requires. The mark is sticky so that a program which stops naming a unit
|
||||
// does not silently stop describing it: the window only grows, and shrinking it is how
|
||||
// a stale binding would become invisible to the server.
|
||||
Uint64 EmitShaderImages(GLContext& ctx) {
|
||||
const auto& program = ctx.GetProgramForDraw();
|
||||
const auto& resolution = MGPipeProgramOpaqueUnitsShared().For(program.get());
|
||||
const Uint32 programWindow =
|
||||
resolution.MaxImageUnit < 0 ? 0u : static_cast<Uint32>(resolution.MaxImageUnit) + 1u;
|
||||
if (programWindow > m_window) m_window = programWindow;
|
||||
const Uint32 count = m_window < kMGPipeMaxImageUnits ? m_window : kMGPipeMaxImageUnits;
|
||||
// PROPERTY 1, and it is one integer test on every draw of every application that
|
||||
// never binds an image.
|
||||
if (count == 0) return 0;
|
||||
|
||||
for (Uint32 unit = 0; unit < count; ++unit) {
|
||||
const auto& binding = ctx.GetImageTextureBinding(static_cast<Int>(unit));
|
||||
MGPImageView& entry = m_entries[unit];
|
||||
entry = MGPImageView{};
|
||||
entry.Unit = unit;
|
||||
entry.Res = binding.Texture ? MGPipeSlots().Acquire(MGPipeKind::Texture,
|
||||
binding.Texture->GetLifetimeId())
|
||||
: kMGPipeNullHandle;
|
||||
// D-A4: a texture named in an emitted MGPImageView is SHADER-IMAGE-bound from
|
||||
// then on - the bit ImageBindableHint is derived from. The bind itself noted it
|
||||
// first (TextureState.h, so the hint precedes the first sync); this is the
|
||||
// letter of the rule and a one-compare early-out once the bit is set.
|
||||
if (!MGPipeHandleIsNull(entry.Res)) {
|
||||
MGPipeNoteTextureBoundAs(entry.Res, static_cast<Uint32>(kMGPipeBindShaderImage));
|
||||
}
|
||||
// THE APPLICATION's format and access, verbatim. The bind-format recast and the
|
||||
// buffer-texture split view are server-side and stay there; so does
|
||||
// SupportsLayeredImageBinding's rule, which asks the BACKEND target after
|
||||
// MapToBackendTextureTarget and forces layer to 0 for a non-layerable one -
|
||||
// Adreno took a stray layer index literally. A client that pre-applied any of
|
||||
// that would be answering a driver question from the wrong side.
|
||||
entry.InternalFormat = static_cast<Uint32>(binding.Format);
|
||||
entry.Layer = static_cast<Uint32>(binding.Layer);
|
||||
entry.Level = static_cast<Uint16>(binding.Level);
|
||||
entry.Layered = binding.Layered != GL_FALSE ? 1 : 0;
|
||||
entry.Access = static_cast<Uint8>(MGPipeEncodeImageAccess(binding.Access));
|
||||
}
|
||||
|
||||
const Uint64 hash = MGPipeShaderImageSetContentHash(m_entries.data(), 0, count);
|
||||
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetShaderImages, hash)) {
|
||||
return 0;
|
||||
}
|
||||
m_lastImages = MGPShaderImages{};
|
||||
m_lastImages.Start = 0;
|
||||
m_lastImages.Count = count;
|
||||
m_lastImages.ContentHash = hash;
|
||||
MGPipeApplySetShaderImages(m_lastImages, m_entries.data());
|
||||
++m_imageSets;
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::ShaderImageEmissions, 1);
|
||||
}
|
||||
return sizeof(MGPShaderImages) + static_cast<Uint64>(count) * sizeof(MGPImageView);
|
||||
}
|
||||
|
||||
// The validate point's FreshlyPrimed arm. A fresh context is a fresh set of image
|
||||
// bindings, so the sticky window starts over; the suppressor slot this set latches is
|
||||
// invalidated beside this call. There is no record half here at all - set_shader_images
|
||||
// is pure working state and mints no object of its own.
|
||||
void Reset() { m_window = 0; }
|
||||
|
||||
void ResetCounters() { m_imageSets = 0; }
|
||||
|
||||
const MGPShaderImages& LastShaderImages() const { return m_lastImages; }
|
||||
const Array<MGPImageView, kMGPipeMaxImageUnits>& LastImageViews() const { return m_entries; }
|
||||
Uint64 ImageSetCount() const { return m_imageSets; }
|
||||
Uint32 Window() const { return m_window; }
|
||||
|
||||
private:
|
||||
// GL_READ_ONLY / GL_WRITE_ONLY / GL_READ_WRITE folded into the one byte the wire
|
||||
// carries. A value the enum does not name would otherwise truncate silently into a
|
||||
// Uint8, which is the class of bug the descriptors exist to close.
|
||||
static Uint32 MGPipeEncodeImageAccess(GLenum access) {
|
||||
switch (access) {
|
||||
case GL_READ_ONLY:
|
||||
return 0;
|
||||
case GL_WRITE_ONLY:
|
||||
return 1;
|
||||
case GL_READ_WRITE:
|
||||
return 2;
|
||||
default:
|
||||
MOBILEGL_ASSERT(false, "glBindImageTexture access 0x%x is not one of the three GL names",
|
||||
static_cast<Uint>(access));
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
Array<MGPImageView, kMGPipeMaxImageUnits> m_entries{};
|
||||
MGPShaderImages m_lastImages{};
|
||||
Uint32 m_window = 0;
|
||||
Uint64 m_imageSets = 0;
|
||||
};
|
||||
|
||||
inline MGPipeImageEmitter& MGPipeImageEmitterInstance() {
|
||||
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason; heap-constructed and
|
||||
// intentionally leaked at exit, like every other MGPipe process singleton.
|
||||
static MGPipeImageEmitter* emitter = new MGPipeImageEmitter();
|
||||
return *emitter;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,151 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/PipeFill.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
// The fill point (ARCHITECTURE.md 9.2, P1 brief D7). MG_Impl spells MGP_FILL(Verb); as the
|
||||
// statement immediately before every call through gBackendFunctionsTable.GL - after every
|
||||
// early return the call is behind, inside the loop body for a call made in a loop - so the
|
||||
// frontend fills the PipeInputs block for exactly the verbs that reach a backend. In the
|
||||
// pull build the macro is ((void)0) and the pull build is byte-identical to a tree without
|
||||
// it.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
namespace MobileGL::MG_Pipe {
|
||||
struct PipeInputs;
|
||||
|
||||
// PipeFill.cpp. THE VALIDATE POINT (ARCHITECTURE.md 5.1, P2 brief D1). In order:
|
||||
// 1. bump the per-verb serial, record the verb and the context identity;
|
||||
// 2. run the tracker's DIRTY WALK for this verb's class (MG_Impl/Pipe/Tracker.h);
|
||||
// 3. EMIT, for each set dirty bit whose subsystem bit is on in the runtime
|
||||
// MOBILEGL_PIPE_PUSH bitmask, the P2 call that carries it;
|
||||
// 4. run the P1 residual fill for every field an emitted call did NOT supply,
|
||||
// stamping each with the new serial exactly as before;
|
||||
// 5. in a verify build, the entry compare against a second snapshot (P1 brief D8) -
|
||||
// which stops being a tautology the moment step 3 supplies a field step 4 skips.
|
||||
//
|
||||
// It was MGPipeFillForVerb through P1, when steps 2 and 3 did not exist. The macro
|
||||
// spelling, the 83 call sites and the verb enum are unchanged: the dispatch is
|
||||
// kMGPipeVerbClass's nine classes, which is the same code as nine named ValidateFor*
|
||||
// entry points with one call site per verb instead of nine.
|
||||
void MGPipeValidateForVerb(MGPipeVerb verb);
|
||||
|
||||
// Ends the verb in flight without starting another: bumps the serial, so every field the
|
||||
// verb stamped goes stale, and puts the current verb back to "none", so a read made after
|
||||
// it aborts as Fatal{UnmigratedPipeInput, "<Field>@<none>"} - which is what such a read
|
||||
// is - instead of naming whichever verb happened to be filled last. Nothing in the GL
|
||||
// entry points calls this: a real verb is always followed by the next verb's fill. It
|
||||
// exists for a caller that drives a backend helper directly and wants its declaration to
|
||||
// stop where it says it stops (MG_Test/ScopedPipeVerb.h).
|
||||
void MGPipeLeaveVerb();
|
||||
|
||||
// PipeFill.cpp. DOES THIS BUILD, ON THIS BACKEND, AT THIS MASK, EMIT FOR THIS P4a FAMILY?
|
||||
// (ID-39, widened by S-3 / ID-41.) The four conjuncts are the operator's per-subsystem bit
|
||||
// in MOBILEGL_PIPE_PUSH, the family's own kMGPipeWired*Subsystem constant (`wired`, which
|
||||
// the caller passes because it lives in the family's emit header and this header may not
|
||||
// include one), and - for the four families P4a migrates - a backend having registered
|
||||
// MGPipeResourceOps (the same per-backend signal `MGPipeResourceSubsystemEnabled()` has
|
||||
// applied to P3a's buffers since the phase began) and every D-K2 dependency bit of the
|
||||
// family being set in the same mask.
|
||||
//
|
||||
// THE LAST TWO CONJUNCTS ARE THE ONES THIS DECLARATION EXISTS FOR, and they are the same
|
||||
// defect twice. Magma (DirectVulkan) registers no table and has no P4a twins; at a mask like
|
||||
// 0x7ff Espryt REFUSES the texture family server-side because D-K2's fourth row says bit 10
|
||||
// requires bit 11. In both cases the client emitted anyway, the applier accepted, the
|
||||
// emitters cleared their per-level dirty flags on that acceptance, and the legacy upload
|
||||
// path that still owed those texels found nothing to upload (66 DirectVulkan cases at ID-39,
|
||||
// 47 DirectGLES cases at ID-41). With them the four families emit NOTHING in that state and
|
||||
// the legacy pull path runs exactly as it does on a pull build.
|
||||
//
|
||||
// D-K2's TABLE IS IN PipeFill.cpp, ONCE: bit 9 requires bit 10, bit 10 requires bits 7 and
|
||||
// 11, bit 11 requires bit 10, bit 12 depends on nothing - the client mirror, bit for bit, of
|
||||
// the four `Resolve<Family>SubsystemArm()` refusals in DirectGLES/Managers.cpp.
|
||||
//
|
||||
// It is exported for the unit gate and for no other caller: the gate itself is
|
||||
// FamilyIsLive() inside PipeFill.cpp, every birth hook and every `wants()` row resolves
|
||||
// through it, and this returns that same expression rather than a second copy of it.
|
||||
Bool MGPipeP4aFamilyEmits(Uint64 subsystem, Uint64 wired);
|
||||
|
||||
// PipeFill.cpp. P3a D-H2.1: the DRAW's raw vertex-fetch base instance, which
|
||||
// set_vertex_buffers now carries as an explicit field.
|
||||
//
|
||||
// It replaces an ambient process global the backend read at VAO sync time, which is a
|
||||
// shape that cannot cross a pushed boundary. The client sends the raw value and never a
|
||||
// pre-shifted offset: whether to emulate the fetch shift or let GL_EXT_base_instance do
|
||||
// it is the SERVER's decision. It is also an input to set_vertex_buffers' content hash
|
||||
// and to the tracker's bit-9 shutter, so a draw whose only change is its base instance
|
||||
// still reaches the emitter and still goes out.
|
||||
//
|
||||
// DO NOT CALL IT DIRECTLY FROM A GL ENTRY POINT - use MGP_SET_BASE_INSTANCE below. This
|
||||
// whole declaration block is inside #if MOBILEGL_PIPE_PUSH, so a bare call would not even
|
||||
// compile in a pull build, and the three call sites are in a file that is compiled in
|
||||
// both. The macro is the same shape MGP_FILL already has, for the same reason.
|
||||
//
|
||||
// The validate point consumes and clears it - on both of its exits - and MGPipeLeaveVerb
|
||||
// clears it too, so a plain draw that follows a base-instanced one sees 0 again. The
|
||||
// tracker's Reset() deliberately does NOT clear it (Tracker.h): a make-current happens
|
||||
// BETWEEN the setter and the fill that reads it.
|
||||
//
|
||||
// The three GL entry points that make this call (ID-10's grant) are
|
||||
// MG_Impl/GLImpl/Drawing/GL_Drawing.cpp's DrawElementsInstancedBaseVertexBaseInstance,
|
||||
// DrawElementsInstancedBaseInstance and DrawArraysInstancedBaseInstance - one line each,
|
||||
// immediately above the MGP_FILL, carrying the RAW baseinstance argument.
|
||||
void MGPipeSetPendingBaseInstance(Uint32 baseInstance);
|
||||
// What the next set_vertex_buffers will carry. The unit gate reads it to pin that a
|
||||
// make-current between the setter and the fill does not eat it
|
||||
// (TrackerWalk.ABaseInstanceSurvivesTheFirstWalkOnAFreshContext).
|
||||
Uint32 MGPipePendingBaseInstance();
|
||||
|
||||
// PipeFill.cpp. Negative control B (P1 brief D6): the filler withholds the STAMP - never
|
||||
// the value - of `field` at `verb`, so that verb's read of it is
|
||||
// Fatal{UnmigratedPipeInput, "Field@Verb"} while every other verb is unaffected. The
|
||||
// MOBILEGL_PIPE_POISON_OMIT knob ("<Verb>:<FieldName>") calls this once, on the first
|
||||
// fill; tests call it directly. Both null clears the omission. An unknown name is
|
||||
// Fatal{PipeVerifyBadKnob}.
|
||||
void MGPipeSetPoisonOmission(const char* verb, const char* field);
|
||||
|
||||
// PipeFill.cpp. How many times set_vertex_attrib_defaults' applier failed to reproduce
|
||||
// the value the call carried, so the client wrote the mirror itself
|
||||
// (EmitVertexAttribDefaults). It is the ONE observable of that repair: the window it
|
||||
// covers is a verb whose class does not read m_currentVertexAttribute, where reading the
|
||||
// storage to check it would be the poison violation the fill table exists to forbid. So
|
||||
// TrackerShippedEmitter asserts on this counter instead, and the day package A's applier
|
||||
// switches on MGPAttribValue::ValueClass the counter stops moving.
|
||||
//
|
||||
// Not hot-path instrumentation: it is incremented only inside the repair branch, which
|
||||
// runs only when the call actually went out, which is only when an attribute default
|
||||
// moved.
|
||||
Uint64 MGPipeVertexAttribDefaultRepairCount();
|
||||
|
||||
// PipeFill.cpp. The header of the last set_vertex_attrib_defaults that actually went out
|
||||
// - Mask, and Count == 0 for "none ever did", since a call naming no attribute is not
|
||||
// emitted. Two properties of this call have no other observable, because reading
|
||||
// m_currentVertexAttribute back at a verb whose class does not carry it is the poison
|
||||
// violation the fill table exists to forbid: that a FRESH CONTEXT republishes all 32
|
||||
// (the server's mirror still holds the previous context's defaults), and that one moved
|
||||
// attribute publishes exactly one. Eight bytes, written only when a call goes out.
|
||||
MGPVertexAttribDefaults MGPipeVertexAttribDefaultsLastHeader();
|
||||
|
||||
#if MOBILEGL_PIPE_VERIFY
|
||||
// PipeFill.cpp. The second arm of the comparator (P1 brief D8, ARCHITECTURE.md 13.2-2):
|
||||
// fills `snapshot` from the live GLContext the old way, for every field in `mask`. This
|
||||
// is the branch that survives P13, which is why it is its own function rather than the
|
||||
// filler's loop.
|
||||
void SnapshotFromGLContext(PipeInputs& snapshot, const MGPipeFieldMask& mask);
|
||||
#endif
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#define MGP_FILL(Verb) ::MobileGL::MG_Pipe::MGPipeValidateForVerb(::MobileGL::MG_Pipe::MGPipeVerb::Verb)
|
||||
// P3a D-H2.1. One line immediately ABOVE the MGP_FILL of a draw entry point that takes a
|
||||
// baseinstance, carrying the argument RAW. It has to be a macro for MGP_FILL's reason: the
|
||||
// three call sites are compiled in the pull build too, where MGPipeSetPendingBaseInstance is
|
||||
// neither declared nor defined.
|
||||
#define MGP_SET_BASE_INSTANCE(BaseInstance) \
|
||||
::MobileGL::MG_Pipe::MGPipeSetPendingBaseInstance(static_cast<::MobileGL::Uint32>(BaseInstance))
|
||||
#else
|
||||
#define MGP_FILL(Verb) ((void)0)
|
||||
#define MGP_SET_BASE_INSTANCE(BaseInstance) ((void)0)
|
||||
#endif
|
||||
@@ -0,0 +1,484 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/ProgramEmit.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The CLIENT side of P4a's program family: create/bind/delete_shader_state,
|
||||
// set_draw_program, set_dispatch_program and set_global_constants.
|
||||
//
|
||||
// WHERE create_shader_state IS EMITTED FROM, and why it is not the tracker's business: the
|
||||
// tracker's bit-6 shutter reads GetCurrentProgram() and DELIBERATELY NOT GetProgramForDraw(),
|
||||
// because the tracker must not force a compile just to answer "did the shader move". So the
|
||||
// tracker keeps its shutter and the EMITTER joins - from the same GetProgramForDraw() /
|
||||
// GetProgramForDispatch() call the verb is about to make anyway, so no join happens that would
|
||||
// not have happened. Emitting from the compile pool's terminal continuation is a real
|
||||
// asynchronous win and is a LATER phase's: in monolith the applier is one function call away,
|
||||
// so it is unmeasurable here.
|
||||
//
|
||||
// WHAT THE SERVER STILL SPECIALISES, so nobody reads create_shader_state as self-contained
|
||||
// and produces a per-draw rebuild: the draw-FBO clamp masks, the fragColor broadcast count,
|
||||
// the storage-block binding signature, the atomic-counter set, the live image formats and the
|
||||
// patch parameters are all inputs a backend program depends on BEYOND the artefacts. This call
|
||||
// publishes the ARTEFACTS; the server specialises at the verb from the state it holds. The
|
||||
// clause count does not shrink - its inputs move.
|
||||
//
|
||||
// THE ARTEFACTS DO NOT TRAVEL IN MONOLITH. All seven of MGPProgramDesc's blob refs are
|
||||
// declared with Size 0 and the LinkArtifacts / SpirvArtifacts ride beside the record through
|
||||
// MGPipeApplyCreateShaderState's companion pointers, so the codec is never called on the hot
|
||||
// path; the verify build is where it is exercised.
|
||||
//
|
||||
// THIS FILE IS CREATED BY THE CONTRACT COMMIT AND FILLED BY THE PACKAGE THAT OWNS IT - see
|
||||
// FramebufferEmit.h for why, in full.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/CompositeResolver.h>
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/MGPipeHostSpan.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_Pipe/PipeMutation.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/GLState/ProgramState/ProgramObject.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// WIRED. create/bind/delete_shader_state, set_draw_program, set_dispatch_program and
|
||||
// set_global_constants all have bodies, so this family contributes its bit to
|
||||
// kMGPipeWiredSubsystems.
|
||||
//
|
||||
// AND SINCE c0b THAT CONSTANT REALLY IS PART OF THE EMISSION GATE, so the note that used to
|
||||
// say otherwise here was true only against the contract commit: the validate point's
|
||||
// `wants()` asks the subsystem mapping, the operator's MOBILEGL_PIPE_PUSH mask, THIS
|
||||
// CONSTANT and the dirty bit, and the birth hooks' `FamilyIsLive` asks the same pair one
|
||||
// level in. It is also a compile-time contract - while it is non-zero PipeFill.cpp's
|
||||
// `if constexpr` seam instantiates the forward to EmitShaderCso below, so a missing entry
|
||||
// point is a build error here rather than at the merge. The RUNTIME A/B that switches the
|
||||
// family off is still the mask. See SamplerEmit.h's twin note.
|
||||
inline constexpr Uint64 kMGPipeWiredProgramSubsystem = kMGPipeSubsystemPrograms;
|
||||
|
||||
// D-H6. ~0u is the BACKENDS' "never uploaded" sentinel for a global-constants version, and
|
||||
// ProgramObject::MarkUBOContentDirty skips it on the wrap for exactly that reason. The
|
||||
// client must never put it on the wire either: a server that received it would read its own
|
||||
// record as "nothing has ever been uploaded here" and re-upload for ever.
|
||||
inline constexpr Uint32 kMGPipeGlobalConstantsNeverUploaded = ~Uint32{0};
|
||||
|
||||
inline constexpr Bool MGPipeGlobalConstantsVersionIsEmittable(Uint32 version) {
|
||||
return version != kMGPipeGlobalConstantsNeverUploaded;
|
||||
}
|
||||
|
||||
// A program's identity for the wire, out of the SNAPSHOT the last link consumed and never
|
||||
// out of the live attach list: glAttachShader and glCompileShader take effect only at the
|
||||
// NEXT link and neither moves m_linkVersion, so a stage mask built from GetAttachedShaders
|
||||
// would describe a program that does not exist yet. GetLinkedShaderStages() is also what
|
||||
// indexes GetGeneratedSpirv(), so the two halves of this descriptor are guaranteed to agree
|
||||
// by construction rather than by care.
|
||||
inline Uint32 MGPipeStageMaskOf(const MG_State::GLState::ProgramObject& program) {
|
||||
Uint32 mask = 0;
|
||||
for (const ShaderStage stage : program.GetLinkedShaderStages()) {
|
||||
if (stage == ShaderStage::Unknown) continue;
|
||||
mask |= Uint32{1} << static_cast<Uint32>(stage);
|
||||
}
|
||||
return mask;
|
||||
}
|
||||
|
||||
class MGPipeProgramEmitter {
|
||||
public:
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
using ProgramObject = MG_State::GLState::ProgramObject;
|
||||
|
||||
// create_shader_state (re-issued on the SAME handle whenever the link version moves -
|
||||
// Gen moves only on slot reuse), then bind_shader_state and set_draw_program /
|
||||
// set_dispatch_program. Two program calls because the frontend has two joins and two
|
||||
// PipeInputs slots.
|
||||
//
|
||||
// BOTH JOINS HAPPEN HERE and both are the verb's own: GetProgramForDraw flattens a
|
||||
// bound pipeline into its composite and GetProgramForDispatch answers the compute
|
||||
// question, and with a plain glUseProgram they are the same object, so the ordinary
|
||||
// frame pays one join it was going to pay anyway.
|
||||
Uint64 EmitShaderState(GLContext& ctx) {
|
||||
Uint64 bytes = 0;
|
||||
const auto& drawProgram = ctx.GetProgramForDraw();
|
||||
const auto& dispatchProgram = ctx.GetProgramForDispatch();
|
||||
|
||||
const MGPipeHandle drawCso =
|
||||
drawProgram ? AcquireShaderCso(*drawProgram, bytes) : kMGPipeNullHandle;
|
||||
// THE COMPOSITE'S SECOND RELEASE PATH is spoken here, not in a destructor: when the
|
||||
// bound pipeline's draw-program signature moves, the resolver releases the slot the
|
||||
// previous composite held. Whichever of the two paths runs second - this one or the
|
||||
// composite ProgramObject's own ~ProgramObject - is a proven no-op, because the slot
|
||||
// allocator refuses a slot that is not live at that generation.
|
||||
if (drawProgram && MGPipeProgramIsPipelineComposite(*drawProgram)) {
|
||||
if (const auto& pipeline = ctx.GetBoundProgramPipeline()) {
|
||||
// THE CONTEXT IS PART OF THE RESOLVER's KEY and this is the only place that
|
||||
// supplies it: the resolver is a process singleton and a pipeline's GL name
|
||||
// is per context, so without it a make-current between two contexts holding
|
||||
// one pipeline name released the other context's LIVE composite.
|
||||
// GetTextureContextId() is the tree's never-reused per-context id, the same
|
||||
// one PipeInputs carries and the backends' per-context memos key on.
|
||||
MGPipeCompositeResolverInstance().Observe(ctx.GetTextureContextId(), *pipeline,
|
||||
*drawProgram, drawCso);
|
||||
}
|
||||
}
|
||||
const MGPipeHandle dispatchCso =
|
||||
dispatchProgram ? (dispatchProgram == drawProgram ? drawCso
|
||||
: AcquireShaderCso(*dispatchProgram, bytes))
|
||||
: kMGPipeNullHandle;
|
||||
|
||||
// THE BOUND CSO IS THE DRAW ONE WHEN THERE IS ONE. bind_shader_state names what
|
||||
// glUseProgram selected, and when a program pipeline is bound instead that is the
|
||||
// composite; a compute-only pipeline has no draw program at all, and then the
|
||||
// dispatch program is the only thing bound. A null handle is legal here and means
|
||||
// exactly "nothing bound".
|
||||
const MGPipeHandle boundCso = !MGPipeHandleIsNull(drawCso) ? drawCso : dispatchCso;
|
||||
if (boundCso != m_boundCso) {
|
||||
MGPipeApplyBindShaderState(HandleOnly(boundCso));
|
||||
m_boundCso = boundCso;
|
||||
++m_binds;
|
||||
bytes += sizeof(MGPHandleOnly);
|
||||
}
|
||||
if (drawCso != m_drawCso) {
|
||||
MGPipeApplySetDrawProgram(HandleOnly(drawCso));
|
||||
m_drawCso = drawCso;
|
||||
++m_drawSets;
|
||||
bytes += sizeof(MGPHandleOnly);
|
||||
}
|
||||
if (dispatchCso != m_dispatchCso) {
|
||||
MGPipeApplySetDispatchProgram(HandleOnly(dispatchCso));
|
||||
m_dispatchCso = dispatchCso;
|
||||
++m_dispatchSets;
|
||||
bytes += sizeof(MGPHandleOnly);
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
|
||||
// set_global_constants: the DEFAULT UNIFORM BLOCK only, keyed (ShaderCso, Version) and
|
||||
// at most once per program per frame. Version is GetUBOContentVersion() and must never
|
||||
// be ~0u, which is the backends' "never uploaded" sentinel - the wrap skips it.
|
||||
//
|
||||
// NAMED uniform blocks are NOT this call's: set_shader_buffers(Uniform) is a later
|
||||
// phase's and BindCurrentProgramWithResources' named-UBO block is untouched. What
|
||||
// travels here is globalUboScratch, the link phase's CPU array, which has no GL name
|
||||
// and no BufferObject behind it.
|
||||
Uint64 EmitGlobalConstants(GLContext& ctx) {
|
||||
const auto& program = ctx.GetProgramForDraw();
|
||||
if (!program) return 0;
|
||||
const Uint32 version = program->GetUBOContentVersion();
|
||||
// THE SENTINEL IS NEVER EMITTED. A server that received ~0u would read its own
|
||||
// record as "never uploaded" and re-upload every frame for ever.
|
||||
if (!MGPipeGlobalConstantsVersionIsEmittable(version)) return 0;
|
||||
const Uint size = program->GetUBOSize();
|
||||
if (size == 0) return 0;
|
||||
|
||||
Uint64 bytes = 0;
|
||||
const MGPipeHandle cso = AcquireShaderCso(*program, bytes);
|
||||
// (ShaderCso, Version) IS the key, so the latch is the key: an unchanged pair means
|
||||
// the server already holds these bytes and re-sending them would move the record's
|
||||
// serial for nothing.
|
||||
if (cso == m_constantsCso && version == m_constantsVersion) return bytes;
|
||||
|
||||
m_lastConstants = MGPGlobalConstants{};
|
||||
m_lastConstants.ShaderCso = cso;
|
||||
m_lastConstants.Version = version;
|
||||
// THE ONE BLOB RULE: Size 0 means "this record does not declare its blob" - which
|
||||
// is what a monolith emission is - and the bytes ride beside it as a companion
|
||||
// pointer. Offset carries the staging address for diagnostics only; nothing reads
|
||||
// it as a length.
|
||||
m_lastConstants.Blob.Seg = kMGHostSpanSegNone;
|
||||
m_lastConstants.Blob.Offset = reinterpret_cast<Uint64>(program->GetUBOData());
|
||||
m_lastConstants.Blob.Size = 0;
|
||||
MGPipeApplySetGlobalConstants(m_lastConstants, program->GetUBOData());
|
||||
m_constantsCso = cso;
|
||||
m_constantsVersion = version;
|
||||
++m_constantSets;
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::CsoBlobBytes, size);
|
||||
}
|
||||
return bytes + sizeof(MGPGlobalConstants) + size;
|
||||
}
|
||||
|
||||
// D-H4's re-issue rule, and it is the CreateVertexElements shape one for one: the
|
||||
// record goes out again on the SAME handle whenever the link version moves, which is
|
||||
// legal because MGPipeHandle::Gen increments only on slot reuse and never on a
|
||||
// respecify. A program that relinks is the same GL object and the server's twin table
|
||||
// must not be asked to mint a second one.
|
||||
MGPipeHandle AcquireShaderCso(const ProgramObject& program, Uint64& payloadBytes) {
|
||||
const MGPipeHandle handle = AcquireShaderCsoHandle(program);
|
||||
if (MGPipeHandleIsNull(handle)) return handle;
|
||||
Latch& latch = LatchFor(handle);
|
||||
|
||||
const Uint32 linkVersion = program.GetLinkVersion();
|
||||
if (latch.RecordLive && latch.RecordGen == handle.Gen && latch.LinkVersion == linkVersion) {
|
||||
return handle;
|
||||
}
|
||||
|
||||
const auto& link = program.GetLinkReflection();
|
||||
const auto& spirv = program.GetSpirvReflection();
|
||||
|
||||
m_lastDesc = MGPProgramDesc{};
|
||||
m_lastDesc.Cso = handle;
|
||||
m_lastDesc.StageMask = MGPipeStageMaskOf(program);
|
||||
m_lastDesc.GlobalUboSize = static_cast<Uint32>(program.GetUBOSize());
|
||||
m_lastDesc.ReservedNumSamplesOffset = static_cast<Uint32>(spirv.reservedNumSamplesOffset);
|
||||
m_lastDesc.SpirvStatus = spirv.spirvStatus ? 1 : 0;
|
||||
m_lastDesc.NativeFloat64 = spirv.nativeFloat64 ? 1 : 0;
|
||||
m_lastDesc.PointSizeDemoted = spirv.pointSizeDemoted ? 1 : 0;
|
||||
m_lastDesc.EnableSpirvValidation = spirv.enableSpirvValidation ? 1 : 0;
|
||||
|
||||
// ONE BLOB REF PER MODULE, IN THE LINKED-SHADER-SNAPSHOT'S ORDER, which is the
|
||||
// order GetGeneratedSpirv() is indexed in - so Spirv[i] and StageMask agree because
|
||||
// they came out of the same snapshot. Every one of them declares Size 0 (the one
|
||||
// Blob rule); Offset carries the module's staging address so a reader can see which
|
||||
// slots are occupied without the record pretending to declare a length it does not
|
||||
// own.
|
||||
//
|
||||
// A COUNTED REFUSAL AND NOT AN ASSERTION (D-J3). MOBILEGL_ASSERT compiles out at
|
||||
// INFO, which is all three gate builds and every shipped build, so an assert here
|
||||
// would leave the truncation below completely silent in exactly the builds that
|
||||
// run - which is the idiom D-J3 exists to forbid. generatedSpirv cannot exceed six
|
||||
// stages today, so this is a guard against a seventh; truncation is the safe
|
||||
// direction and the counter is what makes it visible.
|
||||
const SizeT moduleCount = spirv.generatedSpirv.size();
|
||||
if (moduleCount > 6) ++m_moduleTruncations;
|
||||
for (SizeT i = 0; i < moduleCount && i < 6; ++i) {
|
||||
m_lastDesc.Spirv[i].Seg = kMGHostSpanSegNone;
|
||||
m_lastDesc.Spirv[i].Offset = reinterpret_cast<Uint64>(spirv.generatedSpirv[i].data());
|
||||
m_lastDesc.Spirv[i].Size = 0;
|
||||
}
|
||||
m_lastDesc.Reflection.Seg = kMGHostSpanSegNone;
|
||||
m_lastDesc.Reflection.Offset = reinterpret_cast<Uint64>(&link);
|
||||
m_lastDesc.Reflection.Size = 0;
|
||||
|
||||
MGPipeApplyCreateShaderState(m_lastDesc, &link, &spirv);
|
||||
// THE CREATE WENT OUT, so the publication latch is taken here and nowhere else
|
||||
// (contract-v2 §3.1). MGPipeEmitShaderCsoDestroyAndFree reads it, and without it
|
||||
// delete_shader_state can never go out - for an ordinary program or for a
|
||||
// composite, both of which take that one helper.
|
||||
MGPipeNoteHandlePublished(MGPipeKind::ShaderCso, handle);
|
||||
++m_creates;
|
||||
payloadBytes += sizeof(MGPProgramDesc);
|
||||
|
||||
// A RE-ISSUED create_shader_state CLEARS THE APPLIER's DEFAULT UNIFORM BLOCK (wire
|
||||
// W6), so the (Cso, Version) latch that suppresses set_global_constants has to go
|
||||
// with it or the block is never re-sent. The case the design worries about is a
|
||||
// FAILED relink of a bound program - GL keeps the previous executable and its
|
||||
// uniforms running - and the general one is any future re-issue trigger that does
|
||||
// not happen to move the content version, of which a recycled slot is one.
|
||||
// Invalidated rather than re-emitted here, because this function has no business
|
||||
// deciding when the constants go out: the next EmitGlobalConstants sees an
|
||||
// unlatched key and sends them.
|
||||
if (m_constantsCso == handle) {
|
||||
m_constantsCso = kMGPipeNullHandle;
|
||||
m_constantsVersion = kMGPipeGlobalConstantsNeverUploaded;
|
||||
}
|
||||
|
||||
latch.RecordLive = true;
|
||||
latch.RecordGen = handle.Gen;
|
||||
latch.LinkVersion = linkVersion;
|
||||
return handle;
|
||||
}
|
||||
|
||||
// ---- THE CONTRACT ENTRY POINT THIS FAMILY OWES (contract-v2 §3.4) ----
|
||||
//
|
||||
// PipeFill.cpp's MGPipeEmitShaderCsoCreate forwards here through the `if constexpr`
|
||||
// seam keyed on kMGPipeWiredProgramSubsystem, so while that constant is non-zero this
|
||||
// must exist and be spelled exactly like this. A thin wrapper on purpose:
|
||||
// AcquireShaderCso above IS this family's handle rule - identity-addressed per
|
||||
// ProgramObject, the composite band entered through the one door, the re-issue on the
|
||||
// same handle and the publication - and a second copy of any of it here would be a
|
||||
// second authority.
|
||||
//
|
||||
// THE HOOK HAS ALREADY APPLIED BOTH GATES (the operator's mask and the wired constant),
|
||||
// so this body applies none of its own. The byte count is discarded: a birth is not a
|
||||
// validate-point emission and has no payload budget to report into.
|
||||
void EmitShaderCso(ProgramObject& program) {
|
||||
Uint64 bytes = 0;
|
||||
AcquireShaderCso(program, bytes);
|
||||
}
|
||||
|
||||
// The emitter's OWN record memo - "have I already published a create_shader_state at
|
||||
// this slot, for this generation, at this link version".
|
||||
//
|
||||
// IT IS NOT WHAT THE DEATH PATH ASKS, and that changed at c0b (contract-v2 §3.1/D17):
|
||||
// MGPipeEmitShaderCsoDestroyAndFree reads A's publication latch, which is one answer
|
||||
// per {kind, slot, gen} that all six death helpers share. This stays because the
|
||||
// VERSION-FIRST SKIP needs it - it is the same latch AcquireShaderCso consults before
|
||||
// it builds a descriptor - and because a unit case reads it.
|
||||
//
|
||||
// THE COMPOSITE BAND IS INDEXED SEPARATELY, for the allocator's own reason: the band
|
||||
// base is 983040, so a slot-indexed vector would allocate ~983k latches for one program
|
||||
// pipeline. Both spaces stay dense against their own high-water mark.
|
||||
Bool RecordIsPublished(MGPipeHandle handle) const {
|
||||
if (MGPipeHandleIsNull(handle)) return false;
|
||||
const Vector<Latch>& table = TableOf(handle);
|
||||
const SizeT slot = SlotIndexOf(handle);
|
||||
if (slot >= table.size()) return false;
|
||||
const Latch& latch = table[slot];
|
||||
return latch.RecordLive && latch.RecordGen == handle.Gen;
|
||||
}
|
||||
|
||||
// The memo's other half, and the bound-mirror clearing beside it.
|
||||
//
|
||||
// THE CALLER IS THE CONTRACT's DEATH HELPER (P4a final review C-2): the death path
|
||||
// reads the contract's latch for the wire delete and then forwards here, before the
|
||||
// slot is freed, so a dead handle no longer reads as published in this memo between
|
||||
// the death and the recycle and the three bound mirrors never name a dead program.
|
||||
// Gen-keyed, so a late notice for a slot already handed out again clears nothing of
|
||||
// the successor's.
|
||||
void NoteRecordDestroyed(MGPipeHandle handle) {
|
||||
if (MGPipeHandleIsNull(handle)) return;
|
||||
Vector<Latch>& table = TableOf(handle);
|
||||
const SizeT slot = SlotIndexOf(handle);
|
||||
if (slot < table.size() && table[slot].RecordGen == handle.Gen) {
|
||||
table[slot] = Latch{};
|
||||
}
|
||||
if (m_boundCso == handle) m_boundCso = kMGPipeNullHandle;
|
||||
if (m_drawCso == handle) m_drawCso = kMGPipeNullHandle;
|
||||
if (m_dispatchCso == handle) m_dispatchCso = kMGPipeNullHandle;
|
||||
if (m_constantsCso == handle) {
|
||||
m_constantsCso = kMGPipeNullHandle;
|
||||
m_constantsVersion = kMGPipeGlobalConstantsNeverUploaded;
|
||||
}
|
||||
}
|
||||
|
||||
// The validate point's FreshlyPrimed arm. MGPipeApplierReset clears DrawProgram,
|
||||
// DispatchProgram and BoundShaderCso - all three are per-context WORKING STATE - so
|
||||
// the three mirrors here go with them, or the first emission after a make-current
|
||||
// would be suppressed as unchanged and the server would draw with the previous
|
||||
// context's program bound.
|
||||
//
|
||||
// The RECORD half stays, and that is the rule rather than an oversight: the applier
|
||||
// keeps its shader-CSO records across a make-current because a program lives in a share
|
||||
// group, and re-publishing one would move its Serial for nothing. The global-constants
|
||||
// key goes with the working state because its record's bytes are per (Cso, Version) and
|
||||
// a fresh server has not been told them.
|
||||
void Reset() {
|
||||
m_boundCso = kMGPipeNullHandle;
|
||||
m_drawCso = kMGPipeNullHandle;
|
||||
m_dispatchCso = kMGPipeNullHandle;
|
||||
m_constantsCso = kMGPipeNullHandle;
|
||||
m_constantsVersion = kMGPipeGlobalConstantsNeverUploaded;
|
||||
// The composite memo's freshness goes with them - and only its freshness. Its
|
||||
// ENTRIES name composites whose frontend objects outlive the context switch, so
|
||||
// releasing them here would emit a delete for a live program.
|
||||
MGPipeCompositeResolverInstance().Reset();
|
||||
}
|
||||
|
||||
void ResetCounters() {
|
||||
m_creates = m_binds = m_drawSets = m_dispatchSets = m_constantSets = 0;
|
||||
m_moduleTruncations = 0;
|
||||
}
|
||||
|
||||
// ---- what a unit case reads ----
|
||||
const MGPProgramDesc& LastProgramDesc() const { return m_lastDesc; }
|
||||
const MGPGlobalConstants& LastGlobalConstants() const { return m_lastConstants; }
|
||||
// THE (Cso, Version) KEY set_global_constants is suppressed against. Exposed so a case
|
||||
// can pin that a re-issued create_shader_state invalidates it - the applier clears the
|
||||
// block on the re-issue (wire W6), so a latch that survived it would never re-send.
|
||||
MGPipeHandle GlobalConstantsCso() const { return m_constantsCso; }
|
||||
Uint32 GlobalConstantsVersion() const { return m_constantsVersion; }
|
||||
// D-J3's counted refusal: programs whose linked snapshot carried more modules than
|
||||
// MGPProgramDesc::Spirv[] can name, and whose tail was therefore dropped.
|
||||
Uint64 TruncatedModuleCount() const { return m_moduleTruncations; }
|
||||
MGPipeHandle BoundCso() const { return m_boundCso; }
|
||||
MGPipeHandle DrawCso() const { return m_drawCso; }
|
||||
MGPipeHandle DispatchCso() const { return m_dispatchCso; }
|
||||
Uint64 CreateCount() const { return m_creates; }
|
||||
Uint64 BindCount() const { return m_binds; }
|
||||
Uint64 DrawProgramSetCount() const { return m_drawSets; }
|
||||
Uint64 DispatchProgramSetCount() const { return m_dispatchSets; }
|
||||
Uint64 GlobalConstantsSetCount() const { return m_constantSets; }
|
||||
|
||||
private:
|
||||
// THE ONE PLACE THE BAND CAN ENTER. An ordinary program's slot comes from the ordinary
|
||||
// allocator door keyed on its lifetime id. CompositeResolver.h widens this to send a
|
||||
// pipeline composite through MGPipeSlotAllocator::AllocateComposite instead, and
|
||||
// nothing else about the emission changes - the server never learns a composite is a
|
||||
// composite.
|
||||
MGPipeHandle AcquireShaderCsoHandle(const ProgramObject& program) {
|
||||
const Uint64 lifetimeId = program.GetLifetimeId();
|
||||
const MGPipeHandle existing = MGPipeSlots().FindByLifetimeId(MGPipeKind::ShaderCso, lifetimeId);
|
||||
if (!MGPipeHandleIsNull(existing)) return existing;
|
||||
// A composite is minted off ITS OWN lifetime id, out of the reserved band, and is
|
||||
// an ordinary ShaderCso handle in every other respect - the same kind, the same
|
||||
// {slot, gen} rules, the same Free, the same death helper. Keying it on its own
|
||||
// lifetime id rather than on the pipeline's signature is what makes ~ProgramObject
|
||||
// able to release it at all, and it is why two pipelines that happen to have the
|
||||
// same signature keep their own composite: sharing one handle between two frontend
|
||||
// objects would let the first one's death free a slot the second still names.
|
||||
return MGPipeProgramIsPipelineComposite(program)
|
||||
? MGPipeSlots().AllocateComposite(lifetimeId)
|
||||
: MGPipeSlots().AllocateFor(MGPipeKind::ShaderCso, lifetimeId);
|
||||
}
|
||||
|
||||
struct Latch {
|
||||
Bool RecordLive = false;
|
||||
Uint32 RecordGen = 0;
|
||||
Uint32 LinkVersion = 0;
|
||||
};
|
||||
|
||||
// TWO TABLES, NOT A WIDER ONE, and it is the allocator's own reason repeated where it
|
||||
// bites a second time: the composite band starts at slot 983040, so folding a composite
|
||||
// into the ordinary slot-indexed vector would allocate ~983k latches - and grow them
|
||||
// again on every future push_back - for a single program pipeline. Both spaces stay
|
||||
// dense against their own high-water mark, which is exactly what the allocator does one
|
||||
// level down.
|
||||
Vector<Latch>& TableOf(MGPipeHandle handle) {
|
||||
return MGPipeIsCompositeShaderSlot(handle.Slot) ? m_compositeLatch : m_latch;
|
||||
}
|
||||
const Vector<Latch>& TableOf(MGPipeHandle handle) const {
|
||||
return MGPipeIsCompositeShaderSlot(handle.Slot) ? m_compositeLatch : m_latch;
|
||||
}
|
||||
static SizeT SlotIndexOf(MGPipeHandle handle) {
|
||||
return MGPipeIsCompositeShaderSlot(handle.Slot)
|
||||
? static_cast<SizeT>(handle.Slot - kMGPipeShaderCsoCompositeSlotBase)
|
||||
: static_cast<SizeT>(handle.Slot);
|
||||
}
|
||||
Latch& LatchFor(MGPipeHandle handle) {
|
||||
Vector<Latch>& table = TableOf(handle);
|
||||
const SizeT slot = SlotIndexOf(handle);
|
||||
if (slot >= table.size()) table.resize(slot + 1);
|
||||
return table[slot];
|
||||
}
|
||||
|
||||
static MGPHandleOnly HandleOnly(MGPipeHandle handle) {
|
||||
MGPHandleOnly only{};
|
||||
only.Handle = handle;
|
||||
only.Kind = static_cast<Uint32>(MGPipeKind::ShaderCso);
|
||||
return only;
|
||||
}
|
||||
|
||||
MGPProgramDesc m_lastDesc{};
|
||||
MGPGlobalConstants m_lastConstants{};
|
||||
|
||||
Vector<Latch> m_latch;
|
||||
Vector<Latch> m_compositeLatch;
|
||||
MGPipeHandle m_boundCso = kMGPipeNullHandle;
|
||||
MGPipeHandle m_drawCso = kMGPipeNullHandle;
|
||||
MGPipeHandle m_dispatchCso = kMGPipeNullHandle;
|
||||
MGPipeHandle m_constantsCso = kMGPipeNullHandle;
|
||||
Uint32 m_constantsVersion = kMGPipeGlobalConstantsNeverUploaded;
|
||||
|
||||
Uint64 m_creates = 0;
|
||||
Uint64 m_binds = 0;
|
||||
Uint64 m_drawSets = 0;
|
||||
Uint64 m_dispatchSets = 0;
|
||||
Uint64 m_constantSets = 0;
|
||||
Uint64 m_moduleTruncations = 0;
|
||||
};
|
||||
|
||||
inline MGPipeProgramEmitter& MGPipeProgramEmitterInstance() {
|
||||
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason; heap-constructed and
|
||||
// intentionally leaked at exit, and it MUST NOT hold a frontend SharedPtr - that is
|
||||
// the exit-order rule, stated over every MGPipe process singleton rather than over the
|
||||
// ones a destructor reaches today.
|
||||
static MGPipeProgramEmitter* emitter = new MGPipeProgramEmitter();
|
||||
return *emitter;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
@@ -0,0 +1,615 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/ResourceTracker.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The CLIENT side of P3a's resource family (brief D-A, D-B, D-C, D-D).
|
||||
//
|
||||
// WHERE IT RUNS, and it is the ONE exception to push-at-validate (ARCHITECTURE.md 5.1):
|
||||
// the seven BufferBackendOps hooks already dispatch at the GL call that causes them, so
|
||||
// their pipe calls are emitted from the same BufferObject dispatchers - not from
|
||||
// MGPipeValidateForVerb. Nothing about buffers moves to validate time in P3a.
|
||||
//
|
||||
// WHAT LIVES HERE
|
||||
// * the sticky BindMask, one constexpr BufferTarget -> bit table with a static_assert
|
||||
// that it covers every enumerator, so a new target cannot be silently unmapped;
|
||||
// * the lifetimeId -> {slot, gen} mint (through MGPipeSlots(), the one allocator) and
|
||||
// the slot -> BufferObject* INVERSE the reverse channel resolves a writeback through;
|
||||
// * the nine MGPipeEmitResource* bodies, declared in MG_Pipe/PipeMutation.h so that
|
||||
// MG_State sees a declaration and never this file (the same layering PipeMutation.h
|
||||
// already has for MGP_NOTE_MUTATION: declare in MG_Pipe, define in MG_Impl);
|
||||
// * the MGPSubData range splitter, because one record's box caps the destination at a
|
||||
// 2^31-1 offset and a 2^32-1 size;
|
||||
// * the map-persistent-roundtrips counting site.
|
||||
//
|
||||
// HEADER-ONLY, for the ownership reason Tracker.h states in full: the root CMakeLists.txt
|
||||
// that would name a new .cpp belongs to the contract package and is frozen behind the tag.
|
||||
// MG_Impl/Pipe/PipeFill.cpp is the one translation unit that includes it in the library.
|
||||
//
|
||||
// NO TIMER, and no per-call record copy on a HOT path. The two observables a unit case
|
||||
// needs - the last emitted descriptor and the per-call counts - are written only by
|
||||
// resource_create and resource_respecify, which run once per glBufferData rather than per
|
||||
// upload; resource_subdata, the hot one, is observed through the pure builders below
|
||||
// instead (MGPipeBuildSubDataRecord / MGPipeForEachSubDataRecordRange), which is also what
|
||||
// lets a test drive the splitter at both of its bounds without a 4 GiB buffer.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_Pipe/PipeMutation.h>
|
||||
#include <MG_State/GLState/BufferState/BufferState.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
#include <Config.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-A3: BindMask
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
// MGPResourceDesc::BindMask's twelve bits MOVED TO MG_Pipe/MGPipeTypes.h AT P4a, beside
|
||||
// the field, exactly as the note that stood here said they would when a second producer
|
||||
// appeared: P4a's texture family sets kMGPipeBindSampler / kMGPipeBindShaderImage /
|
||||
// kMGPipeBindRenderTarget / kMGPipeBindDepthStencil, the four bits nothing set before.
|
||||
// No alias is written for them because none is possible or needed - both files are
|
||||
// namespace MobileGL::MG_Pipe and this one includes that header, so every spelling below
|
||||
// and in package B's code is unchanged.
|
||||
//
|
||||
// What stays here is the BUFFER half of the mapping, which is this file's own: the
|
||||
// BufferTarget table, its sentinel and its completeness assert.
|
||||
|
||||
// A sentinel the table below returns for an enumerator it does not name. It is NOT a
|
||||
// legal mask value: every enumerator must be listed, including the ones that map to no
|
||||
// bit at all, so that ADDING a BufferTarget is a build break here rather than a bit
|
||||
// that silently stops being published.
|
||||
inline constexpr Uint32 kMGPipeBindUnmapped = 0x10000u;
|
||||
|
||||
// The one table. No `default:` arm on purpose - that is what makes the static_assert
|
||||
// below able to see an unnamed enumerator.
|
||||
constexpr Uint32 MGPipeBindMaskForBufferTarget(BufferTarget target) {
|
||||
switch (target) {
|
||||
case BufferTarget::Vertex:
|
||||
return kMGPipeBindVertex;
|
||||
// GL_ELEMENT_ARRAY_BUFFER is the VAO's element slot: the same bind is both "this
|
||||
// resource is an index buffer" and "the server may need its bytes on its own side".
|
||||
case BufferTarget::Index:
|
||||
return kMGPipeBindIndex | kMGPipeBindElementArray;
|
||||
case BufferTarget::Uniform:
|
||||
return kMGPipeBindConstant;
|
||||
case BufferTarget::ShaderStorage:
|
||||
return kMGPipeBindShaderBuffer;
|
||||
case BufferTarget::DispatchIndirect:
|
||||
case BufferTarget::DrawIndirect:
|
||||
case BufferTarget::Parameter:
|
||||
return kMGPipeBindIndirect;
|
||||
// A texture buffer's backing store is SAMPLED through the texture that names it.
|
||||
case BufferTarget::Texture:
|
||||
return kMGPipeBindSampler;
|
||||
case BufferTarget::TransformFeedback:
|
||||
return kMGPipeBindStreamOutput;
|
||||
case BufferTarget::AtomicCounter:
|
||||
return kMGPipeBindAtomic;
|
||||
// TRANSFER AND QUERY TARGETS, which the bind mask deliberately does not name: none
|
||||
// of them is a pipeline binding, none of them makes the server keep anything, and
|
||||
// a bit set for them would only widen what a split server mirrors. Listed rather
|
||||
// than defaulted, so the completeness assert still sees them.
|
||||
case BufferTarget::CopyRead:
|
||||
case BufferTarget::CopyWrite:
|
||||
case BufferTarget::PixelPack:
|
||||
case BufferTarget::PixelUnpack:
|
||||
case BufferTarget::Query:
|
||||
return kMGPipeBindNone;
|
||||
case BufferTarget::BufferTargetCount:
|
||||
case BufferTarget::Unknown:
|
||||
return kMGPipeBindNone;
|
||||
}
|
||||
return kMGPipeBindUnmapped;
|
||||
}
|
||||
|
||||
constexpr Bool MGPipeEveryBufferTargetIsMapped() {
|
||||
for (SizeT i = 0; i < static_cast<SizeT>(BufferTarget::BufferTargetCount); ++i) {
|
||||
if (MGPipeBindMaskForBufferTarget(static_cast<BufferTarget>(i)) == kMGPipeBindUnmapped) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
static_assert(MGPipeEveryBufferTargetIsMapped(),
|
||||
"a BufferTarget enumerator has no MGPResourceDesc::BindMask row: add it to "
|
||||
"MGPipeBindMaskForBufferTarget, including a deliberate kMGPipeBindNone, or the "
|
||||
"resource it is bound to stops publishing that binding (D-A3, P8 expectation 1)");
|
||||
static_assert(MGPipeBindMaskForBufferTarget(BufferTarget::Index) & kMGPipeBindElementArray,
|
||||
"the ELEMENT_ARRAY bit is the index host mirror's switch (ARCHITECTURE.md 10.3)");
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// The discriminators MGPResourceDesc / MGPSubData carry for a BUFFER
|
||||
// ---------------------------------------------------------------------------------
|
||||
//
|
||||
// P4a MINTED THE FIRST LIST: MGPipeTypes.h now carries enum MGPipeResourceTarget beside
|
||||
// the field, and kMGPipeResourceTargetBuffer moved there with it - the narrowed
|
||||
// resource_respecify ack predicate lives in that header and has to name the buffer target
|
||||
// explicitly, and it may not reach into MG_Impl to do so. The second discriminator is the
|
||||
// frontend enum, named rather than open-coded, and stays here because only this file
|
||||
// produces it.
|
||||
inline constexpr Uint8 kMGPipeResourceStorageKindBuffer =
|
||||
static_cast<Uint8>(MobileGL::TextureStorageType::Buffer);
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-A2: the payload builders. Pure, so a unit case can assert field by field.
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
// The descriptor for `buffer`. `storageDefined` is false for the create that the
|
||||
// constructor emits - storage is defined lazily by the first respecify and a backend
|
||||
// tolerates a resource that has none - and true for every respecify.
|
||||
inline MGPResourceDesc MGPipeBuildResourceDesc(const MG_State::GLState::BufferObject& buffer,
|
||||
MGPipeHandle handle, Uint16 bindMask,
|
||||
Bool storageDefined) {
|
||||
MGPResourceDesc desc{};
|
||||
desc.Resource = handle;
|
||||
desc.Target = static_cast<Uint8>(kMGPipeResourceTargetBuffer);
|
||||
desc.StorageKind = kMGPipeResourceStorageKindBuffer;
|
||||
desc.BindMask = bindMask;
|
||||
if (storageDefined) {
|
||||
// MGPResourceDesc::Width is a Uint32 and that is the CONTRACT's shape, not this
|
||||
// package's, so a store of 4 GiB or more cannot be declared at all. Truncating it
|
||||
// silently is the one answer that must not happen: the applier's range gate would
|
||||
// then refuse the first legal write past the truncated extent as
|
||||
// Fatal{ProtocolCorruption} and name a corruption that is really a narrowing here.
|
||||
// So it is said out loud, once, in every build - the assertion compiles out at
|
||||
// INFO, which is what all three gate builds are.
|
||||
if (buffer.GetSize() > static_cast<SizeT>(0xFFFFFFFFull)) {
|
||||
MGLOG_E_ONCE("MGPipe: buffer %u declares a store of %llu bytes, which does not fit "
|
||||
"MGPResourceDesc::Width - the descriptor's extent is narrowed and every "
|
||||
"write past 4 GiB will be refused by the applier's range gate",
|
||||
buffer.GetExternalIndex(),
|
||||
static_cast<unsigned long long>(buffer.GetSize()));
|
||||
MOBILEGL_ASSERT(false, "MGPResourceDesc::Width cannot carry this buffer's size");
|
||||
}
|
||||
desc.Width = static_cast<Uint32>(buffer.GetSize());
|
||||
desc.Usage = static_cast<Uint32>(buffer.GetUsage());
|
||||
desc.StorageFlags = static_cast<Uint32>(buffer.GetStorageFlags());
|
||||
desc.Immutable = buffer.IsImmutableStorage() ? 1 : 0;
|
||||
desc.HasDefinedContent = buffer.HasDefinedContent() ? 1 : 0;
|
||||
}
|
||||
// Diagnostics only: a GL name is never an identity, never a memo key and never part
|
||||
// of a content hash (ARCHITECTURE.md 4.2.1).
|
||||
desc.GlNameForDiag = static_cast<Uint32>(buffer.GetExternalIndex());
|
||||
return desc;
|
||||
}
|
||||
|
||||
// The buffer half of MGPSubData: the destination range rides in the box's first
|
||||
// coordinate and first extent, and MGPipeSetSubDataBufferRange is the ONLY spelling of
|
||||
// that convention. Returns false, with the record untouched, when the range does not fit
|
||||
// one record - which is where MGPipeForEachSubDataRecordRange comes in.
|
||||
//
|
||||
// `sourceIsVerbatimLevelShadow` is the record's own question - "are these bytes an
|
||||
// untransformed level shadow?" - and it is a PARAMETER because the answer differs by
|
||||
// caller: resource_subdata hands over the client's own shadow at an offset into it and
|
||||
// says yes; buffer_subdata_resident hands over the application's staging store, or the
|
||||
// locally expanded pattern FillSubData built, and both say no. Nothing reads it on the
|
||||
// buffer path today, which is exactly why it must not be a hard-coded 1 that becomes
|
||||
// wrong the moment something does.
|
||||
//
|
||||
// Blob is FILLED, exactly: Seg is kMGHostSpanSegNone (monolith - the bytes travel beside
|
||||
// the record through the entry point's companion pointer) and Size is the piece's own
|
||||
// byte length, which is what the applier's ONE Blob rule holds a non-zero declaration to
|
||||
// (PipeApply.cpp's SubDataBoxFault: != 0 && != MGPipeSubDataBufferSize is refused).
|
||||
// Leaving it 0 would be legal too; declaring it correctly is the stronger of the two.
|
||||
inline Bool MGPipeBuildSubDataRecord(MGPipeHandle res, Uint64 offset, Uint64 size, MGPSubData& out,
|
||||
Bool sourceIsVerbatimLevelShadow) {
|
||||
out = MGPSubData{};
|
||||
out.Res = res;
|
||||
out.Target = kMGPipeResourceTargetBuffer;
|
||||
out.SourceIsVerbatimLevelShadow = sourceIsVerbatimLevelShadow ? 1 : 0;
|
||||
if (!MGPipeSetSubDataBufferRange(out, offset, size)) return false;
|
||||
out.Blob.Seg = kMGHostSpanSegNone;
|
||||
out.Blob.Size = size;
|
||||
return true;
|
||||
}
|
||||
|
||||
// ONE record's destination box caps the offset at 2^31-1 and the size at 2^32-1
|
||||
// (MGPipeTypes.h), so a range beyond either has to be split. The pieces are CONTIGUOUS
|
||||
// and in ASCENDING order, and both properties are load-bearing rather than tidy:
|
||||
// splitting a content write into overlapping or reordered pieces would change what the
|
||||
// backend's queue-and-drain sees, and the Mali WAR-stall fix depends on that queue being
|
||||
// exactly the writes the application made.
|
||||
inline constexpr Uint64 kMGPipeSubDataMaxRecordOffset = 0x7FFFFFFFull;
|
||||
inline constexpr Uint64 kMGPipeSubDataMaxRecordSize = 0xFFFFFFFFull;
|
||||
|
||||
// WITH THE RECORD'S OWN BOUND THE SPLIT IS NOT REACHABLE, and saying so is better than a
|
||||
// loop that reads as if it were: a second piece starts at least 2^32-1 bytes past the
|
||||
// first, which is already past the OFFSET cap, so a range too big for one record is
|
||||
// REFUSED rather than split. The offset cap cannot be split away at all - every piece of
|
||||
// a range that starts past 2^31-1 starts past it too - and a silent truncation is the one
|
||||
// answer that must not happen, so the walk emits nothing and its caller says so once.
|
||||
//
|
||||
// `maxChunk` exists because the record's bound is not the tight one for long: a transport
|
||||
// segment is far smaller (tens of MiB), and that is where this walk starts producing real
|
||||
// splits. It is a parameter now, and exercised at a reachable value by the unit gate, so
|
||||
// that lowering it is one argument rather than a new code path written under pressure.
|
||||
template <class Fn>
|
||||
inline Bool MGPipeForEachSubDataRecordRange(Uint64 offset, Uint64 size, Fn&& piece,
|
||||
Uint64 maxChunk = kMGPipeSubDataMaxRecordSize) {
|
||||
if (offset > kMGPipeSubDataMaxRecordOffset) return false;
|
||||
if (size == 0) return true;
|
||||
if (maxChunk == 0) return false;
|
||||
// Every piece has to be encodable BEFORE any of them is emitted: a half-emitted range
|
||||
// is a partial content write the backend would land as if it were the whole one.
|
||||
const Uint64 chunkCap = maxChunk < kMGPipeSubDataMaxRecordSize ? maxChunk : kMGPipeSubDataMaxRecordSize;
|
||||
for (Uint64 at = offset; at < offset + size; at += chunkCap) {
|
||||
if (at > kMGPipeSubDataMaxRecordOffset) return false;
|
||||
}
|
||||
for (Uint64 at = offset, left = size; left > 0;) {
|
||||
const Uint64 chunk = left > chunkCap ? chunkCap : left;
|
||||
piece(at, chunk);
|
||||
at += chunk;
|
||||
left -= chunk;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// The tracker: handles, the inverse, the sticky mask, the reverse channel
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
class MGPipeResourceTracker {
|
||||
public:
|
||||
using BufferObject = MG_State::GLState::BufferObject;
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
|
||||
// The handle for `buffer`, minted on first use. Minting is NOT gated on a backend
|
||||
// having registered MGPipeResourceOps: the handle is CLIENT state and
|
||||
// set_vertex_buffers names it whether or not the resource family is switched on, so
|
||||
// gating it would make the vertex-input subsystem emit null handles whenever the
|
||||
// resource subsystem is off. Only the CALLS are gated (D-A1).
|
||||
MGPipeHandle Acquire(BufferObject& buffer) {
|
||||
const MGPipeHandle handle = MGPipeSlots().Acquire(MGPipeKind::Buffer, buffer.GetLifetimeId());
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_bySlot.size()) m_bySlot.resize(slot + 1);
|
||||
m_bySlot[slot].Object = &buffer;
|
||||
m_bySlot[slot].Gen = handle.Gen;
|
||||
return handle;
|
||||
}
|
||||
|
||||
// The handle a buffer already has, or the null handle. Never mints - the emission
|
||||
// path calls Acquire, the query paths call this.
|
||||
MGPipeHandle Find(const BufferObject& buffer) const {
|
||||
return MGPipeSlots().FindByLifetimeId(MGPipeKind::Buffer, buffer.GetLifetimeId());
|
||||
}
|
||||
|
||||
// D-D's inverse, and a RAW pointer is exact here: the entry exists only between the
|
||||
// create the constructor emits and the destroy the destructor emits, and a readback
|
||||
// is only ever issued for a live, bound buffer. A WeakPtr would be wrong - the
|
||||
// object does not own itself through a SharedPtr at those two moments. The Gen
|
||||
// compare is what refuses a stale handle rather than resolving it to whatever now
|
||||
// occupies the slot.
|
||||
BufferObject* Resolve(MGPipeHandle handle) const {
|
||||
const SizeT slot = handle.Slot;
|
||||
if (MGPipeHandleIsNull(handle) || slot >= m_bySlot.size()) return nullptr;
|
||||
const Entry& entry = m_bySlot[slot];
|
||||
if (entry.Object == nullptr || entry.Gen != handle.Gen) return nullptr;
|
||||
if (MGPipeSlots().GenOfSlot(MGPipeKind::Buffer, handle.Slot) != handle.Gen) return nullptr;
|
||||
return entry.Object;
|
||||
}
|
||||
|
||||
// Drops the inverse entry and the sticky mask. The CALLER frees the slot afterwards,
|
||||
// in that order (D-L): MGPipeSlotAllocator::Free erases the lifetimeId -> slot
|
||||
// mapping, so anything that has to resolve the handle must do it first.
|
||||
void Retire(MGPipeHandle handle) {
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_bySlot.size()) return;
|
||||
m_bySlot[slot] = Entry{};
|
||||
}
|
||||
|
||||
// ---- D-L: was resource_create actually EMITTED for this slot? ----
|
||||
//
|
||||
// The create is gated at its call site (BufferObject's constructor) and the destroy
|
||||
// is gated inside MGPipeEmitResourceDestroyAndFree, so the two ask the SAME question
|
||||
// at two different moments. A buffer constructed while a backend's table was
|
||||
// registered and destroyed after UnregisterBufferBackendOps() would take the second
|
||||
// answer, free its slot, and leave the applier's record Live - on a slot the
|
||||
// allocator is about to hand out again, with the backend's twin (a driver buffer id)
|
||||
// still attached to it. So the answer is LATCHED at the create and the destroy uses
|
||||
// the latched one; the two are then a pair by construction rather than by the
|
||||
// registration outliving every buffer.
|
||||
void NotePublished(MGPipeHandle handle) {
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_bySlot.size()) return;
|
||||
m_bySlot[slot].Published = true;
|
||||
}
|
||||
Bool WasPublished(MGPipeHandle handle) const {
|
||||
const SizeT slot = handle.Slot;
|
||||
return slot < m_bySlot.size() && m_bySlot[slot].Published;
|
||||
}
|
||||
|
||||
// The sticky everBoundAs mask. Sticky exactly as MGPResourceDesc::ImageBindableHint's
|
||||
// everImageBound is: ORed, never cleared, so a buffer that was an element array once
|
||||
// keeps saying so.
|
||||
Uint16 BindMask(MGPipeHandle handle) const {
|
||||
const SizeT slot = handle.Slot;
|
||||
return slot < m_bySlot.size() ? m_bySlot[slot].BindMask : Uint16{0};
|
||||
}
|
||||
|
||||
// OR one target's bit into a handle's sticky mask, without looking at the context at
|
||||
// all. This is what closes the sampling window for the two bits anything keys on:
|
||||
// the vertex-input emitters resolve, at EVERY draw, exactly the attribute buffers and
|
||||
// the element-slot buffer, so any buffer ever DRAWN FROM carries its ARRAY_BUFFER /
|
||||
// ELEMENT_ARRAY bit for the rest of its life whether or not it happened to be bound
|
||||
// at a storage op. It grows the table rather than dropping the note: it is called
|
||||
// from the validate point, which is GL-thread by construction, and a slot outside the
|
||||
// table is a buffer whose mint this process has not seen (a unit fixture's
|
||||
// ResetForTest, in practice).
|
||||
void NoteBoundAs(MGPipeHandle handle, BufferTarget target) {
|
||||
if (MGPipeHandleIsNull(handle)) return;
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_bySlot.size()) m_bySlot.resize(slot + 1);
|
||||
m_bySlot[slot].BindMask |= static_cast<Uint16>(MGPipeBindMaskForBufferTarget(target));
|
||||
}
|
||||
|
||||
// Accumulates into the sticky mask every target `buffer` is bound to RIGHT NOW, and
|
||||
// returns the accumulated value.
|
||||
//
|
||||
// [DEVIATION, recorded in client-v2.md] D-A3 asks for the OR at every glBindBuffer /
|
||||
// glBindBufferBase / glBindBufferRange / VAO element-slot bind, and C.1 points at
|
||||
// MG_State/GLState/BufferState/BufferState.{h,cpp} for it - a file this package DOES
|
||||
// own. The brief is wrong about where the entry points are: BufferState only VENDS
|
||||
// BindingSlot<BufferObject>& / BindingSlotRange1D&, and the .Bind() calls are
|
||||
// MG_Impl/GLImpl/Buffer/GL_Buffer.cpp's (BindBuffer_State, BindBufferBase_State,
|
||||
// BindBufferRange_State), which C.5 assigns to no package. So the mask is accumulated
|
||||
// by SAMPLING the frontend's live binding state instead - here, at every create and
|
||||
// respecify, which is where the value is PUBLISHED - and ORed into a per-slot sticky
|
||||
// field that is never cleared.
|
||||
//
|
||||
// WHAT SAMPLING ALONE CANNOT SEE is not "a bind after the last respecify" (which the
|
||||
// specified design misses too) but a TRANSIENT bind: bind an EBO, draw, unbind, then
|
||||
// define it through DSA - the respecify's sample sees no binding at all, and the DSA
|
||||
// idiom makes that the common case rather than a corner (TryAdoptLargeStorage's own
|
||||
// comment names glNamedBufferSubData as what MC 26.3 streams with). That hole is
|
||||
// closed for the two bits anything keys on by NoteBoundAs above, called from
|
||||
// EmitVertexBuffers / EmitIndexBuffer at every draw. What is left unpublished is a
|
||||
// buffer that is bound, never drawn from, and never re-specified afterwards; the
|
||||
// remaining fix is one line in each of GL_Buffer.cpp's three *_State binders, for the
|
||||
// seven bits nothing keys on yet, and it stays handed to whoever owns that file.
|
||||
//
|
||||
// The scan is skipped unless a binding-slot version moved since the last one, which
|
||||
// is one Uint16 load per global target and none per binding point. It is NOT called
|
||||
// from the content emitters, deliberately: it walks the whole context's binding state
|
||||
// and writes the tracker, and one of those emitters (resource_subdata) is on the path
|
||||
// D-A2 preserves as reachable off the render thread. Extra sampling could only widen
|
||||
// a sticky union, but not at the price of a context-wide read from the wrong thread.
|
||||
Uint16 RefreshBindMask(GLContext& ctx, const BufferObject& buffer, MGPipeHandle handle) {
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_bySlot.size()) return 0;
|
||||
Entry& entry = m_bySlot[slot];
|
||||
const Uint64 epoch = BindEpoch(ctx);
|
||||
if (epoch == m_bindEpoch && entry.BindMaskEpoch == epoch) return entry.BindMask;
|
||||
m_bindEpoch = epoch;
|
||||
entry.BindMaskEpoch = epoch;
|
||||
Uint16 mask = entry.BindMask;
|
||||
for (const auto target : MG_State::GLState::GlobalBufferTargets) {
|
||||
if (ctx.GetBufferBindingSlot(target).GetBoundObject().get() == &buffer) {
|
||||
mask |= static_cast<Uint16>(MGPipeBindMaskForBufferTarget(target));
|
||||
}
|
||||
}
|
||||
for (const auto target : MG_State::GLState::BufferBindPointTargets) {
|
||||
const SizeT touched = ctx.GetTouchedBufferBindingPointCount(target);
|
||||
for (SizeT i = 0; i < touched; ++i) {
|
||||
if (ctx.GetBufferBindingPoint(target, static_cast<Uint>(i)).GetBoundObject().get() == &buffer) {
|
||||
mask |= static_cast<Uint16>(MGPipeBindMaskForBufferTarget(target));
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
// The index slot is the BOUND VAO's, not BufferState's, so it is not in
|
||||
// GlobalBufferTargets and GetBufferBindingSlot(Index) asserts without a VAO.
|
||||
if (const auto& vao = ctx.GetBoundVertexArray()) {
|
||||
if (vao->GetIndexBufferBindingSlot().GetBoundObject().get() == &buffer) {
|
||||
mask |= static_cast<Uint16>(MGPipeBindMaskForBufferTarget(BufferTarget::Index));
|
||||
}
|
||||
for (int i = 0; i < MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS; ++i) {
|
||||
if (vao->GetAttribute(static_cast<Uint>(i)).Buffer.get() == &buffer) {
|
||||
mask |= static_cast<Uint16>(MGPipeBindMaskForBufferTarget(BufferTarget::Vertex));
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
entry.BindMask = mask;
|
||||
return mask;
|
||||
}
|
||||
|
||||
// ---- the two observables a unit case reads (see the header comment) ----
|
||||
const MGPResourceDesc& LastDesc() const { return m_lastDesc; }
|
||||
Uint64 CreateCount() const { return m_creates; }
|
||||
Uint64 RespecifyCount() const { return m_respecifies; }
|
||||
Uint64 DestroyCount() const { return m_destroys; }
|
||||
Uint64 MapPersistentCount() const { return m_mapPersistents; }
|
||||
|
||||
void NoteDesc(const MGPResourceDesc& desc, Bool isCreate) {
|
||||
m_lastDesc = desc;
|
||||
if (isCreate) {
|
||||
++m_creates;
|
||||
} else {
|
||||
++m_respecifies;
|
||||
}
|
||||
}
|
||||
void NoteDestroy() { ++m_destroys; }
|
||||
void NoteMapPersistent() { ++m_mapPersistents; }
|
||||
|
||||
// A unit fixture's per-case reset, and the library never calls it. THE RULE, stated
|
||||
// rather than left as an absence, because "nothing resets this" is not a reason:
|
||||
//
|
||||
// A buffer handle and the applier record it names are SHARE-GROUP OBJECT STATE.
|
||||
// A GL object lives in a share group, not in a context, so a make-current changes
|
||||
// neither. The applier's MGPipeApplierReset() is a make-current and deliberately
|
||||
// keeps its Resources / VertexElementsCsos (PipeApply.h says so beside them); the
|
||||
// ONLY things that drop a record are the object's own death signal -
|
||||
// resource_destroy, which ~BufferObject raises through
|
||||
// MGPipeEmitResourceDestroyAndFree, and delete_vertex_elements - and
|
||||
// MGPipeApplierReleaseObjectRecords(), which is the SERVED CONTEXT's teardown and
|
||||
// is deliberately wired to nothing in the monolith (there is one applier behind
|
||||
// every context, so calling it on one context's destruction would drop every other
|
||||
// context's records).
|
||||
//
|
||||
// So this tracker needs no re-publication path on a fresh context and must not have
|
||||
// one: re-emitting resource_create for a record the applier still holds would move
|
||||
// its Serial for nothing. What the client owes instead is the destroy - which
|
||||
// ~BufferObject already emits, in the fixed emit-then-free order (D-L) - and that is
|
||||
// the whole of the client's side of the record lifecycle.
|
||||
//
|
||||
// The vertex-input emitter's latches are the OTHER half and are genuinely per
|
||||
// context: MGPipeVertexInputEmitter::Reset() is called from the FreshlyPrimed arm
|
||||
// because the applier's vertex-input WORKING state (the bound handle, the window, the
|
||||
// fetch shift) IS cleared there. Its vertex-elements RECORDS are not, which is why
|
||||
// the emitter's Reset drops the "already published" latches but no create is lost:
|
||||
// the latch is what says "re-publish", and re-publishing an unchanged configuration
|
||||
// is a bounded over-fire, not a dropped write.
|
||||
void ResetForTest() {
|
||||
m_bySlot.clear();
|
||||
m_bindEpoch = 0;
|
||||
m_lastDesc = MGPResourceDesc{};
|
||||
m_creates = m_respecifies = m_destroys = m_mapPersistents = 0;
|
||||
}
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
BufferObject* Object = nullptr;
|
||||
Uint32 Gen = 0;
|
||||
Uint16 BindMask = 0;
|
||||
Bool Published = false;
|
||||
Uint64 BindMaskEpoch = 0;
|
||||
};
|
||||
|
||||
// "Has any buffer binding moved since the last scan": the sum of the binding-slot
|
||||
// versions, which BindingSlot bumps only on a real change. A collision costs one
|
||||
// skipped rescan of ONE buffer's mask, and the mask is re-scanned at the next
|
||||
// emission whose epoch differs, so it can delay a bit by one storage op and never
|
||||
// drop one - the same over-fire-is-free / under-fire-is-fatal direction every
|
||||
// shutter in Tracker.h takes.
|
||||
//
|
||||
// IT DOES NOT SEE THE 84x4 INDEXED BINDING POINTS, and that is sound only because
|
||||
// BindBufferBase_State / BindBufferRange_State also bind the GENERIC slot for the
|
||||
// same target (GL_Buffer.cpp:1531 says why), so an indexed bind always moves one of
|
||||
// the versions summed here. If that ever stops being true, the CONSTANT /
|
||||
// SHADER_BUFFER / ATOMIC / STREAM_OUTPUT bits start being missed silently and the
|
||||
// repair is to fold GetTouchedBufferBindingPointCount into the epoch.
|
||||
static Uint64 BindEpoch(GLContext& ctx) {
|
||||
Uint64 epoch = 1;
|
||||
for (const auto target : MG_State::GLState::GlobalBufferTargets) {
|
||||
epoch += ctx.GetBufferBindingSlot(target).GetVersion();
|
||||
epoch *= 3;
|
||||
}
|
||||
if (const auto& vao = ctx.GetBoundVertexArray()) {
|
||||
epoch += vao->GetIndexBufferBindingSlot().GetVersion();
|
||||
epoch = MGPipeMixShutterValue(epoch, vao->GetLifetimeId());
|
||||
epoch = MGPipeMixShutterValue(epoch, vao->GetConfigVersion());
|
||||
}
|
||||
return epoch;
|
||||
}
|
||||
|
||||
// The same mix Tracker.h's composite shutters use. Spelled here rather than
|
||||
// included so this header does not depend on the tracker.
|
||||
static constexpr Uint64 MGPipeMixShutterValue(Uint64 accumulator, Uint64 value) {
|
||||
accumulator ^= value + 0x9e3779b97f4a7c15ull + (accumulator << 6) + (accumulator >> 2);
|
||||
return accumulator;
|
||||
}
|
||||
|
||||
Vector<Entry> m_bySlot;
|
||||
Uint64 m_bindEpoch = 0;
|
||||
MGPResourceDesc m_lastDesc{};
|
||||
Uint64 m_creates = 0;
|
||||
Uint64 m_respecifies = 0;
|
||||
Uint64 m_destroys = 0;
|
||||
Uint64 m_mapPersistents = 0;
|
||||
};
|
||||
|
||||
// The monolith's one resource tracker, beside the state tracker, the CSO cache and the
|
||||
// set-hash suppressor.
|
||||
inline MGPipeResourceTracker& MGPipeResourceTrackerInstance() {
|
||||
// NEVER DESTROYED, for MGPipeSlots()' reason (SlotAllocator.cpp): ~BufferObject reads
|
||||
// and writes this tracker, and the objects that own the last reference to a
|
||||
// BufferObject outlive every function-local static.
|
||||
static MGPipeResourceTracker* tracker = new MGPipeResourceTracker();
|
||||
return *tracker;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-D: the client's half of the reverse channel
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
// The backend produced the bytes of a readback and hands them back through the channel.
|
||||
// The client resolves the handle to its own object and writes the shadow; the epoch bump
|
||||
// stays SERVER-side and happens AFTER this returns, never before (ARCHITECTURE.md 7.4:
|
||||
// the reverse channel needs the same ordering guarantee as the forward one).
|
||||
inline void MGPipeClientOnBufferWriteback(MGPipeHandle res, Uint64 offset, MGPBlobRef bytes) {
|
||||
auto* buffer = MGPipeResourceTrackerInstance().Resolve(res);
|
||||
if (buffer == nullptr) {
|
||||
MGLOG_E_ONCE("MGPipe: OnBufferWriteback for a handle {%u,%u} that resolves to no buffer",
|
||||
res.Slot, res.Gen);
|
||||
return;
|
||||
}
|
||||
if (bytes.Seg != kMGHostSpanSegNone) {
|
||||
MGLOG_E_ONCE("MGPipe: OnBufferWriteback carried a transport segment (%u); P3a is monolith only",
|
||||
bytes.Seg);
|
||||
return;
|
||||
}
|
||||
// Monolith: Seg is kMGHostSpanSegNone and Offset IS the address of the backend's
|
||||
// mapped bytes (MGPipeTypes.h says so in as many words). Under a transport the
|
||||
// segment resolves first, and that is the phase's edit, not this one's.
|
||||
buffer->WritebackFromBackend(
|
||||
DataPtr{reinterpret_cast<void*>(static_cast<std::uintptr_t>(bytes.Offset)),
|
||||
static_cast<SizeT>(bytes.Size)},
|
||||
static_cast<SizeT>(offset));
|
||||
}
|
||||
|
||||
// A draw or dispatch wrote these ranges. ARCHITECTURE.md 7.1 calls this a NARROWING
|
||||
// channel - the client builds a conservative pending set at its own emission points and
|
||||
// the callback only ever removes from it - so P3a's implementation marks exactly what
|
||||
// the three Espryt MarkGpuWritten sites mark today and the observable behaviour is
|
||||
// unchanged. The narrowing itself is P8/P9's.
|
||||
inline void MGPipeClientOnGpuWritten(MGPipeHandle res, Uint rangeCount, const MGPRange* ranges) {
|
||||
// THE SHAPE IS A CONTRACT POINT, not a formality: the announcement is ONE range
|
||||
// covering kMGPipeWholeBuffer, deliberately not ZERO ranges, because zero will mean
|
||||
// "a fully narrowed set - nothing is dirty" at P8/P9. Marking the whole buffer
|
||||
// written for a zero-range announcement would be the narrowing channel run backwards,
|
||||
// so the shape is asserted here rather than assumed.
|
||||
MOBILEGL_ASSERT(rangeCount == 1 && ranges != nullptr,
|
||||
"OnGpuWritten {slot=%u, gen=%u}: P3a announces exactly one whole-buffer range, "
|
||||
"not %u",
|
||||
res.Slot, res.Gen, static_cast<Uint>(rangeCount));
|
||||
(void)ranges;
|
||||
if (rangeCount == 0) return;
|
||||
auto* buffer = MGPipeResourceTrackerInstance().Resolve(res);
|
||||
if (buffer == nullptr) {
|
||||
// Loud, like its sibling above: a backend announcing a write against a handle
|
||||
// this client cannot resolve is a dropped MarkGpuWritten, and a dropped
|
||||
// MarkGpuWritten is a stale shadow read back as if it were current.
|
||||
MGLOG_E_ONCE("MGPipe: OnGpuWritten for a handle {%u,%u} that resolves to no buffer", res.Slot,
|
||||
res.Gen);
|
||||
return;
|
||||
}
|
||||
buffer->MarkGpuWritten();
|
||||
}
|
||||
|
||||
// Installed once, and never over an entry a backend already claimed: these two are the
|
||||
// CLIENT's implementations of a backend -> frontend callback, so the backend installs
|
||||
// the rest of the table and these two answer for it.
|
||||
inline void MGPipeInstallClientResourceCallbacks() {
|
||||
if (gMGPipeCallbacks.OnBufferWriteback == nullptr) {
|
||||
gMGPipeCallbacks.OnBufferWriteback = &MGPipeClientOnBufferWriteback;
|
||||
}
|
||||
if (gMGPipeCallbacks.OnGpuWritten == nullptr) {
|
||||
gMGPipeCallbacks.OnGpuWritten = &MGPipeClientOnGpuWritten;
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,111 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/SetHashSuppressor.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// Coalescing rule 4 (ARCHITECTURE.md 5.4, P2 brief D11): every kVarTail set_* hashes the
|
||||
// RESOLVED set on the client and does not emit when the hash has not moved.
|
||||
//
|
||||
// This is the carrier for the ~175 lines of debounce that move off the backends in P3b and
|
||||
// P4b - Espryt's UnitBindingsSnapshot / CaptureUnitBindings / UnitBindingsUnchanged and
|
||||
// Magma's equivalents all answer "is this set the same set as last time", and every one of
|
||||
// them answers it against a shape the backend rediscovered. P2 lands the MECHANISM and ONE
|
||||
// real consumer (SetVertexAttribDefaults) so the shape is pinned by a test rather than by a
|
||||
// plan; the other six slots exist, are unit-tested, and are wired by the phase that moves
|
||||
// the set they name. P3a wires the second, SetVertexBuffers. P4a wires SetSamplerViews,
|
||||
// BindSamplerStates and SetShaderImages, and APPENDS an eighth slot, SetFramebufferState -
|
||||
// which leaves only SetShaderBuffers and SetStreamOutputTargets unwired, both P4b's.
|
||||
//
|
||||
// A WIRED SLOT PUTS A REQUIREMENT ON ITS HASH, and SetVertexBuffers is where that first
|
||||
// bites: the hash has to cover EVERY input the record carries, not only the set. Its
|
||||
// baseInstance is DRAW state and moves without the buffer set moving, so a hash over the
|
||||
// entries alone would suppress a record whose one changed field is the fetch shift and the
|
||||
// server would keep the previous one. MG_Impl/Pipe/VertexInputEmit.h's
|
||||
// MGPipeVertexBufferSetContentHash mixes Start, Count and BaseInstance in for exactly that
|
||||
// reason, and VertexInputEmit's base-instance pair is the test that says so.
|
||||
//
|
||||
// A hash of 0 is reserved for "never emitted", so the first emission always goes out; a
|
||||
// computed 0 is remapped to 1, which costs one collision in 2^64 an extra emission and
|
||||
// never a missed one.
|
||||
//
|
||||
// Header-only for the same ownership reason as Tracker.h and CsoCache.h: the root
|
||||
// CMakeLists.txt that would name a new .cpp belongs to package A and is frozen behind the
|
||||
// p2/contract tag.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// One slot per kVarTail set_* (ARCHITECTURE.md 5.1's call list), PLUS
|
||||
// SetFramebufferState, which is not kVarTail at all: MGPFramebufferState carries a
|
||||
// ContentHash for TWO jobs - the server's render-pass memo key and the client's emission
|
||||
// suppressor - and the second one needs a slot here like any other. The enum is
|
||||
// CLIENT-ONLY and is not a wire opcode, so appending before Count is safe.
|
||||
enum class MGPipeSuppressorSlot : Uint32 {
|
||||
SetVertexBuffers = 0, // P3a - wired, and its hash includes BaseInstance
|
||||
// P4a - WIRED. The three unit sets' suppressors are not optional and were never a
|
||||
// later phase's: MGPipeTypes.h makes the pattern mandatory for every kVarTail set_*,
|
||||
// because GetTextureBindGeneration() bumps on a REDUNDANT rebind - MC 26.2 rebinds the
|
||||
// same sampler at every texture-unit switch - so an unsuppressed set is a
|
||||
// several-hundred-byte variable-length record per batch, which is the exact regression
|
||||
// the design names. What P3b/P4b owns is the ~175-line BACKEND debounce these replace
|
||||
// (UnitBindingsSnapshot / CaptureUnitBindings / UnitBindingsUnchanged and the two
|
||||
// g_*SyncList tables); P4a wires the carrier, P3b/P4b deletes the backend copy.
|
||||
SetSamplerViews, // P4a - wired (backend debounce deletion: P3b/P4b)
|
||||
BindSamplerStates, // P4a - wired (backend debounce deletion: P3b/P4b)
|
||||
SetShaderImages, // P4a - wired (backend debounce deletion: P3b/P4b)
|
||||
SetShaderBuffers, // P4b
|
||||
SetStreamOutputTargets, // P4b
|
||||
SetVertexAttribDefaults, // P2 - the one consumer that is wired
|
||||
SetFramebufferState, // P4a - wired
|
||||
Count,
|
||||
};
|
||||
|
||||
inline constexpr SizeT kMGPipeSuppressorSlotCount = static_cast<SizeT>(MGPipeSuppressorSlot::Count);
|
||||
|
||||
class MGPipeSetHashSuppressor {
|
||||
public:
|
||||
// True when `contentHash` differs from what this slot last emitted, and LATCHES it.
|
||||
// False means the resolved set has not moved and the call must not go out.
|
||||
Bool ShouldEmit(MGPipeSuppressorSlot slot, Uint64 contentHash) {
|
||||
const Uint64 latched = contentHash == 0 ? 1 : contentHash;
|
||||
const SizeT index = static_cast<SizeT>(slot);
|
||||
if (m_lastEmitted[index] == latched) return false;
|
||||
m_lastEmitted[index] = latched;
|
||||
return true;
|
||||
}
|
||||
|
||||
// A context change or a server reset: what the server has is no longer what this
|
||||
// slot last emitted, so the next resolved set must go out whatever it hashes to.
|
||||
void Invalidate(MGPipeSuppressorSlot slot) { m_lastEmitted[static_cast<SizeT>(slot)] = 0; }
|
||||
|
||||
void InvalidateAll() {
|
||||
for (SizeT i = 0; i < kMGPipeSuppressorSlotCount; ++i) m_lastEmitted[i] = 0;
|
||||
}
|
||||
|
||||
// 0 == "never emitted". Exposed for the unit test, which is what pins that the
|
||||
// reserved value really is reserved.
|
||||
Uint64 LastEmitted(MGPipeSuppressorSlot slot) const {
|
||||
return m_lastEmitted[static_cast<SizeT>(slot)];
|
||||
}
|
||||
|
||||
private:
|
||||
Array<Uint64, kMGPipeSuppressorSlotCount> m_lastEmitted{};
|
||||
};
|
||||
|
||||
// The monolith's one suppressor, beside the tracker and the CSO cache.
|
||||
inline MGPipeSetHashSuppressor& MGPipeSetHashSuppressorInstance() {
|
||||
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason (MG_Impl/Pipe/Tracker.h): the
|
||||
// rule covers every MGPipe process singleton, not only the ones on today's death
|
||||
// paths.
|
||||
static MGPipeSetHashSuppressor* suppressor = new MGPipeSetHashSuppressor();
|
||||
return *suppressor;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
@@ -0,0 +1,280 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/SlotAllocator.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
// SlotAllocator.h. Compiled only under MOBILEGL_PIPE_PUSH.
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
namespace {
|
||||
// The ShaderCso band the ordinary allocator must never enter: the top 1/16 of the
|
||||
// ShaderCso slot space is reserved for PROGRAM PIPELINE COMPOSITES, which are minted
|
||||
// client-side out of the stage programs bound to a pipeline object. Reserving a band
|
||||
// rather than a flag keeps the composite resolver's lifetime bookkeeping out of here
|
||||
// (MGPipeHandles.h, ARCHITECTURE.md 5.6.3).
|
||||
Bool SlotIsAllocatable(MGPipeKind kind, Uint32 slot) {
|
||||
if (slot < kMGPipeFirstAllocatableSlot) return false;
|
||||
if (kind != MGPipeKind::ShaderCso) return true;
|
||||
return slot < kMGPipeShaderCsoCompositeSlotBase;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
MGPipeSlotAllocator::KindState& MGPipeSlotAllocator::StateOf(MGPipeKind kind) {
|
||||
const SizeT index = static_cast<SizeT>(kind);
|
||||
MOBILEGL_ASSERT(index < kKindCount, "MGPipeKind %zu out of range", index);
|
||||
return m_kinds[index < kKindCount ? index : 0];
|
||||
}
|
||||
|
||||
const MGPipeSlotAllocator::KindState& MGPipeSlotAllocator::StateOf(MGPipeKind kind) const {
|
||||
const SizeT index = static_cast<SizeT>(kind);
|
||||
MOBILEGL_ASSERT(index < kKindCount, "MGPipeKind %zu out of range", index);
|
||||
return m_kinds[index < kKindCount ? index : 0];
|
||||
}
|
||||
|
||||
MGPipeSlotAllocator::SlotState* MGPipeSlotAllocator::EntryOf(KindState& state, MGPipeKind kind,
|
||||
Uint32 slot) {
|
||||
if (kind == MGPipeKind::ShaderCso && MGPipeIsCompositeShaderSlot(slot)) {
|
||||
const SizeT index = slot - kMGPipeShaderCsoCompositeSlotBase;
|
||||
if (index >= state.BandSlots.size()) return nullptr;
|
||||
return &state.BandSlots[index];
|
||||
}
|
||||
if (slot >= state.Slots.size()) return nullptr;
|
||||
return &state.Slots[slot];
|
||||
}
|
||||
|
||||
const MGPipeSlotAllocator::SlotState*
|
||||
MGPipeSlotAllocator::EntryOf(const KindState& state, MGPipeKind kind, Uint32 slot) {
|
||||
return EntryOf(const_cast<KindState&>(state), kind, slot);
|
||||
}
|
||||
|
||||
MGPipeHandle MGPipeSlotAllocator::Allocate(MGPipeKind kind) {
|
||||
KindState& state = StateOf(kind);
|
||||
if (state.Slots.empty()) {
|
||||
// Slot 0 exists so the vector is slot-indexed, and is never handed out.
|
||||
state.Slots.resize(kMGPipeFirstAllocatableSlot);
|
||||
}
|
||||
|
||||
Uint32 slot = 0;
|
||||
Bool reused = false;
|
||||
while (!state.FreeList.empty()) {
|
||||
const Uint32 candidate = state.FreeList.back();
|
||||
state.FreeList.pop_back();
|
||||
if (!SlotIsAllocatable(kind, candidate)) continue;
|
||||
slot = candidate;
|
||||
reused = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (!reused) {
|
||||
slot = static_cast<Uint32>(state.Slots.size());
|
||||
MOBILEGL_ASSERT(SlotIsAllocatable(kind, slot),
|
||||
"MGPipe slot space of kind %u is exhausted at slot %u",
|
||||
static_cast<Uint32>(kind), slot);
|
||||
if (!SlotIsAllocatable(kind, slot)) return kMGPipeNullHandle;
|
||||
state.Slots.emplace_back();
|
||||
}
|
||||
|
||||
SlotState& entry = state.Slots[slot];
|
||||
if (entry.EverHandedOut) {
|
||||
// The one place Gen may move. 2^32 recycles of ONE slot is ~50 days of continuous
|
||||
// churn at one recycle per frame at 1000 fps, which is why the bound is asserted
|
||||
// in a debug allocator rather than defended in release.
|
||||
MOBILEGL_ASSERT(entry.Gen != ~Uint32{0},
|
||||
"MGPipe handle generation wrapped on kind %u slot %u; {slot, gen} is "
|
||||
"no longer unique",
|
||||
static_cast<Uint32>(kind), slot);
|
||||
++entry.Gen;
|
||||
}
|
||||
entry.EverHandedOut = true;
|
||||
entry.Live = true;
|
||||
entry.LifetimeId = 0;
|
||||
++state.LiveCount;
|
||||
return MGPipeHandle{slot, entry.Gen};
|
||||
}
|
||||
|
||||
MGPipeHandle MGPipeSlotAllocator::AllocateFor(MGPipeKind kind, Uint64 lifetimeId) {
|
||||
const MGPipeHandle handle = Allocate(kind);
|
||||
if (MGPipeHandleIsNull(handle)) return handle;
|
||||
KindState& state = StateOf(kind);
|
||||
state.Slots[handle.Slot].LifetimeId = lifetimeId;
|
||||
if (lifetimeId != 0) {
|
||||
MOBILEGL_ASSERT(state.ByLifetimeId.find(lifetimeId) == state.ByLifetimeId.end(),
|
||||
"lifetime id %llu already owns a slot of kind %u",
|
||||
static_cast<unsigned long long>(lifetimeId), static_cast<Uint32>(kind));
|
||||
state.ByLifetimeId[lifetimeId] = handle.Slot;
|
||||
}
|
||||
return handle;
|
||||
}
|
||||
|
||||
MGPipeHandle MGPipeSlotAllocator::AllocateComposite(Uint64 lifetimeId) {
|
||||
// P4a, D-H7. The mirror image of Allocate() above, restricted to the band that one
|
||||
// refuses, and kept in a table of its own so both spaces stay DENSE: the band's base
|
||||
// is 983040, and minting one composite into the slot-indexed vector would allocate
|
||||
// ~23 MB of SlotState for a single program pipeline.
|
||||
KindState& state = StateOf(MGPipeKind::ShaderCso);
|
||||
|
||||
Uint32 slot = 0;
|
||||
Bool reused = false;
|
||||
if (!state.BandFreeList.empty()) {
|
||||
slot = state.BandFreeList.back();
|
||||
state.BandFreeList.pop_back();
|
||||
reused = true;
|
||||
}
|
||||
|
||||
if (!reused) {
|
||||
const SizeT next = kMGPipeShaderCsoCompositeSlotBase + state.BandSlots.size();
|
||||
slot = static_cast<Uint32>(next);
|
||||
// The band's own exhaustion assert, mirroring Allocate()'s: a composite that
|
||||
// cannot be minted is a NAMED failure, not a silent fall-through into the ordinary
|
||||
// program slots, which is exactly what reserving a band rather than setting a flag
|
||||
// buys.
|
||||
MOBILEGL_ASSERT(next < kMGPipeShaderCsoSlotLimit,
|
||||
"the MGPipe ShaderCso COMPOSITE band is exhausted at slot %zu; a "
|
||||
"program-pipeline composite cannot be minted and must not take an "
|
||||
"ordinary program's slot",
|
||||
next);
|
||||
if (next >= kMGPipeShaderCsoSlotLimit) return kMGPipeNullHandle;
|
||||
state.BandSlots.emplace_back();
|
||||
}
|
||||
|
||||
SlotState* entry = EntryOf(state, MGPipeKind::ShaderCso, slot);
|
||||
if (entry == nullptr) return kMGPipeNullHandle;
|
||||
if (entry->EverHandedOut) {
|
||||
MOBILEGL_ASSERT(entry->Gen != ~Uint32{0},
|
||||
"MGPipe handle generation wrapped on the ShaderCso composite band, "
|
||||
"slot %u; {slot, gen} is no longer unique",
|
||||
slot);
|
||||
++entry->Gen;
|
||||
}
|
||||
entry->EverHandedOut = true;
|
||||
entry->Live = true;
|
||||
entry->LifetimeId = lifetimeId;
|
||||
++state.LiveCount;
|
||||
// The band's share of LiveCount, so CompositeLiveCount() can answer without a walk.
|
||||
++state.BandLiveCount;
|
||||
if (lifetimeId != 0) {
|
||||
MOBILEGL_ASSERT(state.ByLifetimeId.find(lifetimeId) == state.ByLifetimeId.end(),
|
||||
"lifetime id %llu already owns a ShaderCso slot",
|
||||
static_cast<unsigned long long>(lifetimeId));
|
||||
state.ByLifetimeId[lifetimeId] = slot;
|
||||
}
|
||||
return MGPipeHandle{slot, entry->Gen};
|
||||
}
|
||||
|
||||
MGPipeHandle MGPipeSlotAllocator::FindByLifetimeId(MGPipeKind kind, Uint64 lifetimeId) const {
|
||||
if (lifetimeId == 0) return kMGPipeNullHandle;
|
||||
const KindState& state = StateOf(kind);
|
||||
const auto it = state.ByLifetimeId.find(lifetimeId);
|
||||
if (it == state.ByLifetimeId.end()) return kMGPipeNullHandle;
|
||||
const SlotState* entry = EntryOf(state, kind, it->second);
|
||||
if (entry == nullptr || !entry->Live) return kMGPipeNullHandle;
|
||||
return MGPipeHandle{it->second, entry->Gen};
|
||||
}
|
||||
|
||||
MGPipeHandle MGPipeSlotAllocator::Acquire(MGPipeKind kind, Uint64 lifetimeId) {
|
||||
const MGPipeHandle existing = FindByLifetimeId(kind, lifetimeId);
|
||||
if (!MGPipeHandleIsNull(existing)) return existing;
|
||||
return AllocateFor(kind, lifetimeId);
|
||||
}
|
||||
|
||||
void MGPipeSlotAllocator::Free(MGPipeKind kind, MGPipeHandle handle) {
|
||||
KindState& state = StateOf(kind);
|
||||
SlotState* entry = EntryOf(state, kind, handle.Slot);
|
||||
if (entry == nullptr) return;
|
||||
// A stale handle must not free the slot its successor now owns - that is the whole
|
||||
// reason the generation is in the key. It is also what makes the SECOND of a
|
||||
// composite's two independent release paths a proven no-op.
|
||||
if (!entry->Live || entry->Gen != handle.Gen) return;
|
||||
if (entry->LifetimeId != 0) {
|
||||
const auto it = state.ByLifetimeId.find(entry->LifetimeId);
|
||||
if (it != state.ByLifetimeId.end() && it->second == handle.Slot) {
|
||||
state.ByLifetimeId.erase(it);
|
||||
}
|
||||
}
|
||||
entry->Live = false;
|
||||
entry->LifetimeId = 0;
|
||||
--state.LiveCount;
|
||||
if (kind == MGPipeKind::ShaderCso && MGPipeIsCompositeShaderSlot(handle.Slot)) {
|
||||
--state.BandLiveCount;
|
||||
state.BandFreeList.push_back(handle.Slot);
|
||||
} else {
|
||||
state.FreeList.push_back(handle.Slot);
|
||||
}
|
||||
}
|
||||
|
||||
Bool MGPipeSlotAllocator::IsLive(MGPipeKind kind, MGPipeHandle handle) const {
|
||||
const SlotState* entry = EntryOf(StateOf(kind), kind, handle.Slot);
|
||||
return entry != nullptr && entry->Live && entry->Gen == handle.Gen;
|
||||
}
|
||||
|
||||
Uint32 MGPipeSlotAllocator::GenOfSlot(MGPipeKind kind, Uint32 slot) const {
|
||||
const SlotState* entry = EntryOf(StateOf(kind), kind, slot);
|
||||
return entry != nullptr ? entry->Gen : 0;
|
||||
}
|
||||
|
||||
Uint64 MGPipeSlotAllocator::LifetimeIdOfSlot(MGPipeKind kind, Uint32 slot) const {
|
||||
const SlotState* entry = EntryOf(StateOf(kind), kind, slot);
|
||||
return entry != nullptr ? entry->LifetimeId : 0;
|
||||
}
|
||||
|
||||
Uint32 MGPipeSlotAllocator::HighWater(MGPipeKind kind) const {
|
||||
// THE ORDINARY SPACE ONLY, and the band is reported by CompositeHighWater() below.
|
||||
// Folding the two would pin this at ~983k from the first composite mint onward and
|
||||
// take the ordinary space's "the high-water mark did not move" assertion away for the
|
||||
// rest of the process - the assertion that catches a dense table that never shrinks,
|
||||
// which is the leak shape this allocator exists to make visible. Two spaces, two
|
||||
// numbers, two real assertions. See SlotAllocator.h.
|
||||
return static_cast<Uint32>(StateOf(kind).Slots.size());
|
||||
}
|
||||
|
||||
Uint32 MGPipeSlotAllocator::CompositeHighWater() const {
|
||||
const KindState& state = StateOf(MGPipeKind::ShaderCso);
|
||||
// One past the highest composite slot ever handed out; exactly the base when none ever
|
||||
// was, so the number is monotone from the first mint and a LEAKED COMPOSITE MOVES IT.
|
||||
return static_cast<Uint32>(kMGPipeShaderCsoCompositeSlotBase + state.BandSlots.size());
|
||||
}
|
||||
|
||||
Uint32 MGPipeSlotAllocator::LiveCount(MGPipeKind kind) const { return StateOf(kind).LiveCount; }
|
||||
|
||||
Uint32 MGPipeSlotAllocator::CompositeLiveCount() const {
|
||||
return StateOf(MGPipeKind::ShaderCso).BandLiveCount;
|
||||
}
|
||||
|
||||
Uint32 MGPipeSlotAllocator::FreeCount(MGPipeKind kind) const {
|
||||
const KindState& state = StateOf(kind);
|
||||
return static_cast<Uint32>(state.FreeList.size() + state.BandFreeList.size());
|
||||
}
|
||||
|
||||
Uint32 MGPipeSlotAllocator::CompositeFreeCount() const {
|
||||
return static_cast<Uint32>(StateOf(MGPipeKind::ShaderCso).BandFreeList.size());
|
||||
}
|
||||
|
||||
void MGPipeSlotAllocator::Reset() {
|
||||
for (KindState& state : m_kinds) {
|
||||
state.Slots.clear();
|
||||
state.FreeList.clear();
|
||||
state.BandSlots.clear();
|
||||
state.BandFreeList.clear();
|
||||
state.ByLifetimeId.clear();
|
||||
state.LiveCount = 0;
|
||||
state.BandLiveCount = 0;
|
||||
}
|
||||
}
|
||||
|
||||
MGPipeSlotAllocator& MGPipeSlots() {
|
||||
// NEVER DESTROYED, deliberately (one allocation for the life of the process). A
|
||||
// frontend object's destructor reaches this allocator - ~BufferObject through
|
||||
// MGPipeEmitResourceDestroyAndFree, ~VertexArrayObject through the death notice - and
|
||||
// MG_Backend/MGPipe/PipeInputs.h's gPipeInputs holds SharedPtrs to those objects at
|
||||
// namespace scope, so they are destroyed by __run_exit_handlers AFTER this
|
||||
// function-local static would have been. A destroyed allocator then answers
|
||||
// FindByLifetimeId out of a freed hash table and Free() writes into freed vectors -
|
||||
// an exit-time heap corruption whose fatality depends only on the allocator's layout.
|
||||
static MGPipeSlotAllocator* allocator = new MGPipeSlotAllocator();
|
||||
return *allocator;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
@@ -0,0 +1,166 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/SlotAllocator.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
#include <MG_Pipe/MGPipeHandles.h>
|
||||
|
||||
// The CLIENT's slot allocator: the thing that mints every MGPipeHandle in the system
|
||||
// (ARCHITECTURE.md 4.2 - no create_* call in the catalogue returns a server-cast handle,
|
||||
// which is what lets the whole catalogue be remoted with zero creation round trips).
|
||||
//
|
||||
// Per kind: a free list plus a high-water mark, so slots stay DENSE and the server's object
|
||||
// table is an array rather than a hash map. It has nothing to do with MG_State's
|
||||
// IndexGenerator - that container's LIFO GL-name reuse is the very problem {slot, gen}
|
||||
// exists to close, and the whole point of the identity is that an ABA on the GL name, on
|
||||
// the heap address or on the lifetime id cannot reproduce a handle.
|
||||
//
|
||||
// Gen increments ONLY when a slot is reused, never on a respecify: a glBufferData on a live
|
||||
// buffer keeps the same {slot, gen}, because the object is the same object. Two generations
|
||||
// exist in the design and they are strictly separate - this is the client's answer to "is
|
||||
// this still the same GL object"; MGGen is the server's epoch for "did I recast my driver
|
||||
// object", and no MGPipe call may require the client to know it.
|
||||
//
|
||||
// The lifetimeId -> slot map is what keeps a GL NAME out of every key (ARCHITECTURE.md 4.2):
|
||||
// the frontend object's lifetime id is the client's own identity for it, so the backend key
|
||||
// is the handle and the frontend key is the lifetime id, and neither is a recyclable name.
|
||||
//
|
||||
// Lives in MG_Impl (the client side, unrestricted) and is compiled only under
|
||||
// MOBILEGL_PIPE_PUSH. It is in the P2 CONTRACT commit rather than in a Track H package
|
||||
// because both Track H slices - Espryt 0b and Magma subsystem 4 - key off it.
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
class MGPipeSlotAllocator {
|
||||
public:
|
||||
static constexpr SizeT kKindCount = static_cast<SizeT>(MGPipeKind::KindCount);
|
||||
|
||||
// A fresh {slot, gen} of this kind, from the free list if one is waiting and from the
|
||||
// high-water mark otherwise. Never returns slot 0 (reserved: null, and the default
|
||||
// framebuffer for kind Framebuffer), and never returns a ShaderCso slot inside the
|
||||
// composite band, which the program-pipeline resolver mints out of separately.
|
||||
MGPipeHandle Allocate(MGPipeKind kind);
|
||||
// Allocate and remember `lifetimeId` as this handle's frontend identity.
|
||||
MGPipeHandle AllocateFor(MGPipeKind kind, Uint64 lifetimeId);
|
||||
|
||||
// P4a, D-H7: THE ONE ENTRY POINT INTO THE ShaderCso COMPOSITE BAND, and the only one
|
||||
// there will ever be. Allocate() above refuses that band on purpose, so a program
|
||||
// pipeline's flattened composite - minted client-side from the stage programs bound to
|
||||
// the pipeline object, and indistinguishable from an ordinary program to the server -
|
||||
// needs a door of its own rather than a flag on the handle. The kind is implied: only
|
||||
// ShaderCso has a band.
|
||||
//
|
||||
// It behaves exactly like AllocateFor in every other respect (free list first, then
|
||||
// the band's own high-water mark; Gen moves only on reuse; the lifetimeId -> slot map
|
||||
// is written) and it carries the band's own exhaustion assert, so exhausting the
|
||||
// composite space is a NAMED Fatal rather than silent slot theft from ordinary
|
||||
// programs. Returns kMGPipeNullHandle when the band is full.
|
||||
//
|
||||
// Freed through the ordinary Free(MGPipeKind::ShaderCso, handle): a composite's slot
|
||||
// has two independent release paths - the pipeline cache's LRU eviction and the
|
||||
// composite ProgramObject's own destructor - and Free refusing a slot that is not live
|
||||
// at that generation is what makes the second one a proven no-op.
|
||||
MGPipeHandle AllocateComposite(Uint64 lifetimeId);
|
||||
// The handle a lifetime id was allocated for, or kMGPipeNullHandle. A recycled heap
|
||||
// address does NOT reproduce a mapping: MG_State hands out a fresh lifetime id per
|
||||
// object, so the map key is unique for the life of the process.
|
||||
MGPipeHandle FindByLifetimeId(MGPipeKind kind, Uint64 lifetimeId) const;
|
||||
// FindByLifetimeId, then AllocateFor when it misses. The ordinary client path.
|
||||
MGPipeHandle Acquire(MGPipeKind kind, Uint64 lifetimeId);
|
||||
|
||||
// Returns the slot to the free list. The Gen bump happens on the NEXT handout of that
|
||||
// slot, not here, so a handle that is freed twice cannot skip a generation and the
|
||||
// "gen moves only on reuse" contract holds for an object that is never reused.
|
||||
void Free(MGPipeKind kind, MGPipeHandle handle);
|
||||
|
||||
Bool IsLive(MGPipeKind kind, MGPipeHandle handle) const;
|
||||
// 0 for a slot that was never handed out; the generation of the LAST handout
|
||||
// otherwise, live or not.
|
||||
Uint32 GenOfSlot(MGPipeKind kind, Uint32 slot) const;
|
||||
Uint64 LifetimeIdOfSlot(MGPipeKind kind, Uint32 slot) const;
|
||||
// One past the highest ORDINARY slot ever handed out of this kind. For every kind but
|
||||
// ShaderCso that is the whole story; for ShaderCso the composite band is a second,
|
||||
// separately dense space and CompositeHighWater() below answers it.
|
||||
//
|
||||
// THE TWO SPACES ARE REPORTED SEPARATELY, and that is the point rather than a detail.
|
||||
// Folding the band into this number pins it at ~983k from the first composite mint
|
||||
// onward, and every later assertion of the "the high-water mark did not move over N
|
||||
// churn rounds" shape - the one that catches a dense table that never shrinks, which
|
||||
// is the ~1.3 KB-per-record leak C-1 produced - becomes vacuously true for ordinary
|
||||
// ShaderCso slots for the rest of the process. A leak case per space is two real
|
||||
// assertions; one merged number is one real assertion and one that cannot go red.
|
||||
//
|
||||
// It is also NOT a table size for kind ShaderCso even now: the band is sparse against
|
||||
// the ordinary space by design, so a consumer indexing by slot must test
|
||||
// MGPipeIsCompositeShaderSlot(slot) first and keep the band in a table of its own,
|
||||
// exactly as this allocator does.
|
||||
Uint32 HighWater(MGPipeKind kind) const;
|
||||
// One past the highest COMPOSITE slot ever handed out, i.e.
|
||||
// kMGPipeShaderCsoCompositeSlotBase + (band slots ever handed out), and exactly the
|
||||
// base when none ever was. Kind ShaderCso is the only kind with a band, so it is
|
||||
// implied - as it is for AllocateComposite. A LEAKED COMPOSITE MOVES THIS and moves
|
||||
// nothing else, which is what the composite's own leak case asserts on.
|
||||
Uint32 CompositeHighWater() const;
|
||||
// Live slots of this kind, ORDINARY AND COMPOSITE TOGETHER for ShaderCso: a live
|
||||
// composite is a live ShaderCso, the applier's two record tables are one object class,
|
||||
// and a caller asking "how many shader CSOs does this client hold" wants both. The
|
||||
// band's own count is CompositeLiveCount(); the ordinary space's is the difference.
|
||||
Uint32 LiveCount(MGPipeKind kind) const;
|
||||
Uint32 CompositeLiveCount() const;
|
||||
// Slots waiting on a free list. Also BOTH SPACES for ShaderCso, for LiveCount's
|
||||
// reason and with the same caveat: a caller that needs to know WHICH space a slot went
|
||||
// back to reads CompositeFreeCount() and subtracts.
|
||||
Uint32 FreeCount(MGPipeKind kind) const;
|
||||
Uint32 CompositeFreeCount() const;
|
||||
|
||||
// Context teardown / server reset / a unit test's fixture.
|
||||
void Reset();
|
||||
|
||||
private:
|
||||
struct SlotState {
|
||||
Uint32 Gen = 0;
|
||||
Bool Live = false;
|
||||
Bool EverHandedOut = false;
|
||||
Uint64 LifetimeId = 0;
|
||||
};
|
||||
|
||||
struct KindState {
|
||||
// Indexed by slot; [0] is the reserved slot and is never live.
|
||||
Vector<SlotState> Slots;
|
||||
Vector<Uint32> FreeList;
|
||||
// P4a: the ShaderCso COMPOSITE band, indexed by (slot - the band's base) and
|
||||
// EMPTY for every other kind. A SECOND VECTOR RATHER THAN MORE OF THE FIRST, and
|
||||
// it is not a micro-optimisation: the band starts at 983040, so minting one
|
||||
// composite into the slot-indexed vector above would allocate ~983k SlotStates -
|
||||
// ~23 MB - for a single program pipeline, and a consumer that sized a table off
|
||||
// HighWater would pay the same shape again with a far bigger record. Both spaces
|
||||
// stay dense against their own high-water mark, which is the property this
|
||||
// allocator exists to give the server.
|
||||
Vector<SlotState> BandSlots;
|
||||
Vector<Uint32> BandFreeList;
|
||||
UnorderedMap<Uint64, Uint32> ByLifetimeId;
|
||||
Uint32 LiveCount = 0;
|
||||
// The band's share of LiveCount above, so the two spaces can be reported apart
|
||||
// without walking either table. Always 0 for every kind but ShaderCso.
|
||||
Uint32 BandLiveCount = 0;
|
||||
};
|
||||
|
||||
KindState& StateOf(MGPipeKind kind);
|
||||
const KindState& StateOf(MGPipeKind kind) const;
|
||||
// The SlotState a (kind, slot) names, in whichever of the two vectors holds it, or
|
||||
// null when the slot has never been handed out. One resolver, so a caller that forgets
|
||||
// the band cannot exist.
|
||||
static SlotState* EntryOf(KindState& state, MGPipeKind kind, Uint32 slot);
|
||||
static const SlotState* EntryOf(const KindState& state, MGPipeKind kind, Uint32 slot);
|
||||
|
||||
Array<KindState, kKindCount> m_kinds{};
|
||||
};
|
||||
|
||||
// The monolith's one client allocator. Under split there is one per client context.
|
||||
MGPipeSlotAllocator& MGPipeSlots();
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,864 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/Tracker.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The frontend state tracker (ARCHITECTURE.md 5.2, P2 brief D4).
|
||||
//
|
||||
// WHERE IT RUNS. Not above MGP_FILL and not in the GL setter: MGPipeValidateForVerb, the
|
||||
// one statement MGP_FILL already expands to before every gBackendFunctionsTable.GL call
|
||||
// (PipeFill.h). Blaze3D brackets every batch with glEnable/glDisable(GL_BLEND), so a
|
||||
// setter that pushed would push twice per batch for a state the batch may not even read;
|
||||
// the validate point coalesces the whole bracket into the two draws that observe it
|
||||
// (ARCHITECTURE.md 5.1).
|
||||
//
|
||||
// WHAT IT DOES. One Uint32 dirty mask per verb, one bit per row of ARCHITECTURE.md 5.2,
|
||||
// computed by comparing a shutter against what the tracker last pushed. P2 emitted for bits
|
||||
// 0..4 (the value-class ones); P3a adds bits 5, 9 and 10 - the vertex-input family - and P4a
|
||||
// adds SEVEN: 6, 7 and 8 (the program family), 11 (the framebuffer) and 12, 13 and 14 (the
|
||||
// three unit sets). Only bits 15, 16 and 17 - the const-buffer, shader-buffer and
|
||||
// stream-output sets - are still computed, latched and counted without an emitter, so the
|
||||
// per-bit fire rate is a measurement rather than a plan and their fields go through the
|
||||
// residual fill until P4b.
|
||||
//
|
||||
// P4a NARROWS NOTHING AND WIDENS THREE THINGS, and every one of them was an UNDER-FIRE that
|
||||
// only became reachable once the bit gained an emitter:
|
||||
// (1) bit 11's shutter gains the READ framebuffer binding slot's version, because
|
||||
// set_framebuffer_state is emitted per bound TARGET and a glBindFramebuffer(
|
||||
// GL_READ_FRAMEBUFFER, ...) moved no shutter at all before;
|
||||
// (2) bit 13's gains the TEXTURE BIND generation, because glBindSampler moves that one and
|
||||
// not the sampling-resolution one, so bind_sampler_states could not see a sampler bind;
|
||||
// (3) bits 6/7/8 - and with them bit 14's program half - read the EFFECTIVE program source
|
||||
// instead of GetCurrentProgram() alone, which is null for the whole life of a bound
|
||||
// separable program pipeline, so a re-composited pipeline reached no program emitter.
|
||||
// Over-firing is free; all three of those were the other direction.
|
||||
//
|
||||
// WHY EVERY SHUTTER OVER-FIRES. A bit that fires too often costs one extra push. A bit
|
||||
// that fires too rarely renders stale, and ARCHITECTURE.md 13.2 names that as the
|
||||
// dangerous direction precisely because the P1 verify comparator cannot see it for
|
||||
// object-class state (it compares those by identity only). So each shutter below is
|
||||
// deliberately coarser than the state it guards - five bits share one buffer aggregate,
|
||||
// the framebuffer bit fires on any attachment write anywhere - and the narrowing is P3's
|
||||
// work, paid for with the fire rates this file publishes.
|
||||
//
|
||||
// NO TIMER LIVES HERE. ROADMAP.md forbids committing hot-path instrumentation; the
|
||||
// absolute ns/draw comes from DriverBench, which times whole frames from outside the
|
||||
// library (P2 brief D17). The only counting is the per-bit fire tally, behind
|
||||
// PipeStats::Enabled() like every other counting site in the tree.
|
||||
//
|
||||
// HEADER-ONLY, and that is an ownership decision rather than a design one: the P2 brief
|
||||
// asks for Tracker.{h,cpp}, but the root CMakeLists.txt that would have to name a new .cpp
|
||||
// belongs to package A and is frozen behind the p2/contract tag. Everything here is
|
||||
// included by exactly one translation unit in the library (MG_Impl/Pipe/PipeFill.cpp) plus
|
||||
// the unit tests, so inline costs nothing. Splitting it back out is one list(APPEND) line.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/MGPipeValueTypes.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// One bit per row of the ARCHITECTURE.md 5.2 table, hand-written rather than generated:
|
||||
// the list is design, not derived data, and the generator has nothing to derive it from.
|
||||
enum class MGPipeDirty : Uint32 {
|
||||
// ---- value class: P2 emits for these five ----
|
||||
NewRenderState = 0, // RenderState::m_version -> set_dynamic_state
|
||||
NewPipelineState, // RenderState::m_pipelineStateVersion -> create/bind_render_state
|
||||
NewPixelPack, // PixelStoreParameters (pack) -> set_pixel_pack_state
|
||||
NewPatchState, // the patch trio, NaN legal -> set_patch_state
|
||||
NewVertexAttribDefaults, // glVertexAttrib* defaults -> set_vertex_attrib_defaults
|
||||
// ---- value class: NEW_VERTEX_ELEMENTS is emitted from P3a and the other three from
|
||||
// P4a - the program family, one subsystem, three bits because the frontend moves them
|
||||
// as three separate events ----
|
||||
NewVertexElements, // the bound VAO's attribute configuration -> create/bind_vertex_elements
|
||||
NewShader, // the current program's link version -> create/bind_shader_state,
|
||||
// set_draw_program, set_dispatch_program (P4a)
|
||||
NewShaderBindings, // image units, block bindings, uniform write set (P4a)
|
||||
NewGlobalConstants, // the default-uniform-block image -> set_global_constants (P4a)
|
||||
// ---- object class. THE FIRST TWO ARE P3a's, not P3b/P4b's: the roadmap puts
|
||||
// set_vertex_buffers and set_index_buffer in the same phase as the vertex-elements
|
||||
// trio, and this comment said otherwise until the commit that wired them. THE NEXT
|
||||
// FOUR ARE P4a's. The last three are still computed and counted only, until P4b. ----
|
||||
NewVertexBuffers, // -> set_vertex_buffers (P3a)
|
||||
NewIndexBuffer, // -> set_index_buffer (P3a)
|
||||
NewFramebuffer, // -> set_framebuffer_state, per bound target (P4a)
|
||||
NewSamplerViews, // -> set_sampler_views (P4a)
|
||||
NewSamplers, // -> bind_sampler_states (P4a)
|
||||
NewShaderImages, // -> set_shader_images (P4a)
|
||||
NewConstBuffers,
|
||||
NewShaderBuffers,
|
||||
NewSoTargets,
|
||||
Count,
|
||||
};
|
||||
|
||||
inline constexpr SizeT kMGPipeDirtyCount = static_cast<SizeT>(MGPipeDirty::Count);
|
||||
static_assert(kMGPipeDirtyCount <= 32, "the dirty mask is a Uint32");
|
||||
|
||||
inline constexpr Uint32 MGPipeDirtyBit(MGPipeDirty bit) {
|
||||
return Uint32{1} << static_cast<Uint32>(bit);
|
||||
}
|
||||
|
||||
// The five P2 emits for. Each phase's constant survives as the next phase's A/B control
|
||||
// and as what a test compares the subsystem map against, so none of them is edited in
|
||||
// place when a later phase takes more bits over.
|
||||
inline constexpr Uint32 kMGPipeDirtyEmittedAtP2 =
|
||||
MGPipeDirtyBit(MGPipeDirty::NewRenderState) | MGPipeDirtyBit(MGPipeDirty::NewPipelineState) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewPixelPack) | MGPipeDirtyBit(MGPipeDirty::NewPatchState) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewVertexAttribDefaults);
|
||||
|
||||
// The three P3a adds: the vertex-input family, all on one subsystem.
|
||||
inline constexpr Uint32 kMGPipeDirtyEmittedAtP3a =
|
||||
kMGPipeDirtyEmittedAtP2 | MGPipeDirtyBit(MGPipeDirty::NewVertexElements) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewVertexBuffers) | MGPipeDirtyBit(MGPipeDirty::NewIndexBuffer);
|
||||
|
||||
// The SEVEN P4a adds, across FOUR subsystems: bits 6/7/8 are the program family, 11 the
|
||||
// framebuffer, and 12/13/14 the sampler-view / sampler-state / image-unit sets. Added
|
||||
// rather than edited into the two above, for the reason those two exist: each phase's
|
||||
// constant survives as the next phase's A/B control and as what a test compares the
|
||||
// subsystem map against.
|
||||
//
|
||||
// EVERY ONE OF THESE SHUTTERS WAS ALREADY COMPUTED, LATCHED AND COUNTED before P4a; what
|
||||
// P4a adds is an emitter for them. That is why this is a one-line constant and not seven
|
||||
// new shutters - and it is also why the two narrowings below are stated as requirements.
|
||||
inline constexpr Uint32 kMGPipeDirtyEmittedAtP4a =
|
||||
kMGPipeDirtyEmittedAtP3a | MGPipeDirtyBit(MGPipeDirty::NewShader) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewShaderBindings) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewGlobalConstants) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewFramebuffer) | MGPipeDirtyBit(MGPipeDirty::NewSamplerViews) |
|
||||
MGPipeDirtyBit(MGPipeDirty::NewSamplers) | MGPipeDirtyBit(MGPipeDirty::NewShaderImages);
|
||||
|
||||
inline constexpr const char* kMGPipeDirtyNames[kMGPipeDirtyCount] = {
|
||||
"NEW_RENDER_STATE",
|
||||
"NEW_PIPELINE_STATE",
|
||||
"NEW_PIXEL_PACK",
|
||||
"NEW_PATCH_STATE",
|
||||
"NEW_VERTEX_ATTRIB_DEFAULTS",
|
||||
"NEW_VERTEX_ELEMENTS",
|
||||
"NEW_SHADER",
|
||||
"NEW_SHADER_BINDINGS",
|
||||
"NEW_GLOBAL_CONSTANTS",
|
||||
"NEW_VERTEX_BUFFERS",
|
||||
"NEW_INDEX_BUFFER",
|
||||
"NEW_FRAMEBUFFER",
|
||||
"NEW_SAMPLER_VIEWS",
|
||||
"NEW_SAMPLERS",
|
||||
"NEW_SHADER_IMAGES",
|
||||
"NEW_CONST_BUFFERS",
|
||||
"NEW_SHADER_BUFFERS",
|
||||
"NEW_SO_TARGETS",
|
||||
};
|
||||
|
||||
// Which runtime MOBILEGL_PIPE_PUSH subsystem bit gates a dirty bit's emission. Zero for
|
||||
// a bit P2 does not emit, which is what makes "the bitmask is a true per-subsystem A/B"
|
||||
// literally true rather than approximately.
|
||||
inline constexpr Uint64 MGPipeSubsystemForDirty(MGPipeDirty bit) {
|
||||
switch (bit) {
|
||||
case MGPipeDirty::NewRenderState:
|
||||
case MGPipeDirty::NewPipelineState:
|
||||
return kMGPipeSubsystemRenderState;
|
||||
case MGPipeDirty::NewPixelPack:
|
||||
return kMGPipeSubsystemPixelPack;
|
||||
case MGPipeDirty::NewPatchState:
|
||||
return kMGPipeSubsystemPatchState;
|
||||
case MGPipeDirty::NewVertexAttribDefaults:
|
||||
return kMGPipeSubsystemVertexAttribDefaults;
|
||||
// P3a's three, all one subsystem: create/bind_vertex_elements, set_vertex_buffers
|
||||
// and set_index_buffer are the vertex-input family and an operator switching it off
|
||||
// has to get the whole family's legacy arm, not two thirds of it.
|
||||
// PipeFill.cpp's SubsystemForEmitter carries the pairing static_asserts.
|
||||
case MGPipeDirty::NewVertexElements:
|
||||
case MGPipeDirty::NewVertexBuffers:
|
||||
case MGPipeDirty::NewIndexBuffer:
|
||||
return kMGPipeSubsystemVertexInput;
|
||||
// P4a's seven, across four subsystems. FOUR AND NOT ONE for P3a's reason one level
|
||||
// out: a framebuffer path that regressed, a texture path that regressed, a sampler
|
||||
// path that regressed and a program path that regressed are four different findings.
|
||||
//
|
||||
// The program family is three bits because the frontend moves them separately - a
|
||||
// relink, a binding change and a uniform write are three events - but one subsystem,
|
||||
// because an operator switching programs off has to get the whole family's legacy arm.
|
||||
// Same for the three unit sets: create_sampler_state, create_sampler_view and the
|
||||
// three kVarTail sets are one family, and half of it is not a control.
|
||||
case MGPipeDirty::NewShader:
|
||||
case MGPipeDirty::NewShaderBindings:
|
||||
case MGPipeDirty::NewGlobalConstants:
|
||||
return kMGPipeSubsystemPrograms;
|
||||
case MGPipeDirty::NewFramebuffer:
|
||||
return kMGPipeSubsystemFramebuffer;
|
||||
case MGPipeDirty::NewSamplerViews:
|
||||
case MGPipeDirty::NewSamplers:
|
||||
case MGPipeDirty::NewShaderImages:
|
||||
return kMGPipeSubsystemSamplers;
|
||||
// NO BIT NAMES kMGPipeSubsystemTextureResources, and that is deliberate rather than an
|
||||
// omission: the texture and renderbuffer resource_* calls and set_texture_params are
|
||||
// dispatched from the GL entry points that cause them - a constructor, a storage
|
||||
// definition, a glTexParameter - not from a dirty walk, exactly as P3a's buffer family
|
||||
// is. Bit 10 gates those dispatch sites; there is no dirty bit to map onto it and
|
||||
// there must not be one, or the emission would be gated twice and disagree with itself.
|
||||
default:
|
||||
// The remaining bits have no call of their own until P4b, so there is no
|
||||
// subsystem to switch and the residual fill keeps supplying their fields.
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
// A COMPOSITE shutter, for the bits whose "did anything move" is more than one counter.
|
||||
// It is a hash, so two different states can in principle collide and cost a MISSED fire.
|
||||
// The five bits P2 emits for are never composed - they are widened counters and byte
|
||||
// compares, neither of which can collide.
|
||||
//
|
||||
// P3a's three ARE composed, so the risk is now real rather than academic, and it is
|
||||
// accepted with its size stated: each mix takes a 64-bit input into a 64-bit
|
||||
// accumulator, so two DIFFERENT vertex configurations collide with probability ~2^-64
|
||||
// per pair, and the inputs are a monotone lifetime id, a monotone configuration version
|
||||
// and a widened slot version - none of which an application can steer. The alternative,
|
||||
// comparing the whole 32-attribute configuration byte for byte on every verb, is the
|
||||
// per-draw cost the shutter exists to avoid. The narrowing that removes the composition
|
||||
// for bit 10 - its own slot version plus the bound object's identity - is what this
|
||||
// phase already did to the one shutter that was composed over an unrelated aggregate.
|
||||
inline constexpr Uint64 MGPipeMixShutter(Uint64 accumulator, Uint64 value) {
|
||||
accumulator ^= value + 0x9e3779b97f4a7c15ull + (accumulator << 6) + (accumulator >> 2);
|
||||
return accumulator;
|
||||
}
|
||||
|
||||
// A Uint16 counter widened at the TRACKER boundary, never in MG_State
|
||||
// (ARCHITECTURE.md 5.2: MG_State is not changed for this). A decrease is a wrap and adds
|
||||
// 65536. A wrap is harmless locally - one extra re-push, never a missed one - which is
|
||||
// exactly what TrackerTest.WrapAroundRePushesButNeverMisses pins.
|
||||
//
|
||||
// THE ONE CASE IT CANNOT SEE, stated because "never a missed push" is otherwise stronger
|
||||
// than what is true: the wrap test is `now < m_last`, so a counter that advances by
|
||||
// EXACTLY 65536 (or a multiple) between two walks reads as unchanged. That needs 65536
|
||||
// render-state mutations inside one verb boundary, and it is pre-existing in class -
|
||||
// both backends already compare raw Uint16 versions the same way - so P2 records it
|
||||
// rather than widening MG_State's counters, which ARCHITECTURE.md 5.2 rules out.
|
||||
class MGPipeWidenedCounter {
|
||||
public:
|
||||
Uint64 Observe(Uint16 now) {
|
||||
if (m_started && now < m_last) m_high += 0x10000ull;
|
||||
m_started = true;
|
||||
m_last = now;
|
||||
return m_high + now;
|
||||
}
|
||||
void Reset() {
|
||||
m_high = 0;
|
||||
m_last = 0;
|
||||
m_started = false;
|
||||
}
|
||||
|
||||
private:
|
||||
Uint64 m_high = 0;
|
||||
Uint16 m_last = 0;
|
||||
Bool m_started = false;
|
||||
};
|
||||
|
||||
class MGPipeTracker {
|
||||
public:
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
|
||||
// The dirty walk. Compares every shutter against what was last pushed, LATCHES the
|
||||
// new values, counts the fires per verb class, and returns the mask. Latching here
|
||||
// rather than after emission is deliberate: a bit whose subsystem is switched off is
|
||||
// not emitted, but its fields are then still pulled by the residual fill, so the
|
||||
// pushed block is correct either way and a bit can never fire twice for one change.
|
||||
Uint32 Update(GLContext& ctx, MGPipeVerbClass verbClass) {
|
||||
// A different context is a different server: nothing the tracker latched about
|
||||
// the old one says anything about this one, and the first walk on a fresh
|
||||
// context must publish a COMPLETE state rather than an increment.
|
||||
if (m_context != &ctx) {
|
||||
Reset();
|
||||
m_context = &ctx;
|
||||
}
|
||||
const Bool wasPrimed = m_primed;
|
||||
|
||||
Uint64 now[kMGPipeDirtyCount];
|
||||
const RenderStateParameters& render = ctx.GetRenderStateParameters();
|
||||
|
||||
// ---- bits 0..1: the two Uint16 render-state counters, widened HERE ----
|
||||
now[Index(MGPipeDirty::NewRenderState)] =
|
||||
m_renderStateVersion.Observe(static_cast<Uint16>(ctx.GetRenderStateParametersVersion()));
|
||||
now[Index(MGPipeDirty::NewPipelineState)] =
|
||||
m_pipelineStateVersion.Observe(static_cast<Uint16>(ctx.GetPipelineStateVersion()));
|
||||
|
||||
// ---- bit 4 and the value-class bits 5..8 ----
|
||||
now[Index(MGPipeDirty::NewVertexAttribDefaults)] = ctx.GetAnyVertexAttribDefaultGeneration();
|
||||
|
||||
const auto& vao = ctx.GetBoundVertexArray();
|
||||
const Uint64 vaoIdentity =
|
||||
vao ? MGPipeMixShutter(vao->GetLifetimeId(), vao->GetConfigVersion()) : 0;
|
||||
now[Index(MGPipeDirty::NewVertexElements)] = vaoIdentity;
|
||||
|
||||
// Deliberately NOT GetProgramForDraw: that joins a pending link, and the tracker
|
||||
// must not force a compile just to answer "did the shader move". These version
|
||||
// counters are plain members and are exactly what the backends already read
|
||||
// without joining (Core.cpp, the glUseProgram half of join site J1).
|
||||
//
|
||||
// BUT GetCurrentProgram() ALONE IS NOT THE PROGRAM SOURCE, AND AT P4a THAT IS AN
|
||||
// UNDER-FIRE. Under GL_ARB_separate_shader_objects an application drives
|
||||
// `glUseProgram(0); glBindProgramPipeline(P)`, and m_currentProgram is then null
|
||||
// for the whole life of that pipeline (Core.cpp, GetProgramForDraw's second half):
|
||||
// all three of these shutters read 0 == 0 forever, so after the first walk on a
|
||||
// fresh context - the one !m_primed fires unconditionally - bits 6, 7 and 8 never
|
||||
// fire again however the pipeline is restaged.
|
||||
//
|
||||
// WHILE NOTHING WAS EMITTED FOR THEM THAT WAS INVISIBLE, which is how it survived
|
||||
// to P4a: GetProgramForDraw is emitted-and-still-pulled, the residual fill copies
|
||||
// it at every verb, and DirtySurface.def rules BindProgramPipelineObject
|
||||
// kPulledEveryVerb for exactly that reason - the backend still receives the right
|
||||
// SharedPtr and nothing renders wrong. The moment P4a emits off these bits it
|
||||
// stops being invisible: glUseProgramStages rebuilds the composite, EmitShaderState
|
||||
// is never called again, so the new composite gets no ShaderCso handle and no
|
||||
// create_shader_state while set_draw_program keeps naming the previous one - a
|
||||
// program the handle protocol never announced, which is exactly the seam-defect
|
||||
// class P3a spent a phase closing. And bit 8 never firing means
|
||||
// set_global_constants is never sent for a pipeline draw at all, where the pull
|
||||
// rescues nothing.
|
||||
//
|
||||
// SO THE SHUTTER READS THE EFFECTIVE SOURCE: the program in use when there is one,
|
||||
// and the bound pipeline when there is not. What it reads OF that pipeline is the
|
||||
// pair ComputeDrawProgramSignature() is built from - each stage program's lifetime
|
||||
// id and LINK version - so bit 6 fires exactly when GetProgramForDraw would hand
|
||||
// back a different composite, which is exactly when a new ShaderCso handle has to
|
||||
// be minted. Those are the same non-artefact fields the plain-program arm above
|
||||
// reads, and the ones Core.cpp calls out as not passing through ProgramObject's
|
||||
// join gate, so the "must not force a compile" rule survives intact: no join, no
|
||||
// flatten, no Link().
|
||||
//
|
||||
// THE PIPELINE NAME IS MIXED IN because two pipelines can carry the same stage set
|
||||
// and each caches its OWN composite object, so the signature alone would let a
|
||||
// glBindProgramPipeline between two such pipelines pass without a fire. What that
|
||||
// does NOT close is a name RECYCLED (glDeleteProgramPipelines +
|
||||
// glGenProgramPipelines) back onto the same stage programs at the same link
|
||||
// versions with no other program-family change in between: a ProgramPipelineObject
|
||||
// has no lifetime id and no wire object at all - DirtySurface.def says so where it
|
||||
// rules MarkProgramPipelineForDeletion kUnpublishedDestroy - so there is nothing
|
||||
// else here to mix it with. Recorded rather than quietly left: closing it needs a
|
||||
// generation counter on the frontend object, which is an MG_State change and not
|
||||
// this file's to make.
|
||||
const auto& program = ctx.GetCurrentProgram();
|
||||
Uint64 shader = 0;
|
||||
Uint64 bindings = 0;
|
||||
Uint64 constants = 0;
|
||||
Uint64 programImages = 0;
|
||||
// THE PROGRAM INPUT OF THE PROGRAM-RESOLVED VIEW SET (P4a fable seam F-1).
|
||||
// set_sampler_views is resolved for the program in use (SamplerEmit.h: the sampler
|
||||
// uniform's TYPE picks which of a unit's targets is the view) and the emitter
|
||||
// memoises that resolution on (lifetime id, link version, backend state version). A
|
||||
// shutter that read only the texture generations therefore missed a glUseProgram:
|
||||
// `glBindTexture x N; glUseProgram(P1); draw; glUseProgram(P2); draw` moved nothing
|
||||
// bit 12 read, so the view set stayed P1's - and E's record epoch, keyed on the two
|
||||
// set serials, then never rebuilt the texture sync list for P2 either. This value is
|
||||
// that memo key, and bit 12 mixes it in below: over-firing costs one re-resolution
|
||||
// the set-hash suppressor absorbs, under-firing left the record describing the
|
||||
// previous program's units.
|
||||
Uint64 opaqueUnits = 0;
|
||||
if (program) {
|
||||
shader = MGPipeMixShutter(program->GetLifetimeId(), program->GetLinkVersion());
|
||||
bindings = MGPipeMixShutter(
|
||||
MGPipeMixShutter(MGPipeMixShutter(program->GetImageUnitVersion(),
|
||||
program->GetBackendStateVersion()),
|
||||
program->GetBlockBindingVersion()),
|
||||
program->GetUniformWriteSetVersion());
|
||||
constants = MGPipeMixShutter(program->GetLifetimeId(), program->GetUBOContentVersion());
|
||||
// THE IDENTITY IS MIXED IN (P4a fable seam F-2), exactly as the pipeline arm
|
||||
// below mixes stageLinks into its half: the counter alone is a per-program
|
||||
// number two programs routinely share - 0 == 0 for any pair that never moved an
|
||||
// image unit through glUniform1i, and 0 == 0 against no program at all - so a
|
||||
// glUseProgram between them fired nothing, set_shader_images' window stayed the
|
||||
// previous program's, and a program whose only image is a BUFFER image (E's
|
||||
// SD-4: nothing else moves between the bind and the dispatch) never reached the
|
||||
// record at all.
|
||||
programImages = MGPipeMixShutter(shader, program->GetImageUnitVersion());
|
||||
opaqueUnits = MGPipeMixShutter(shader, program->GetBackendStateVersion());
|
||||
} else if (const auto& pipeline = ctx.GetBoundProgramPipeline(); pipeline) {
|
||||
using Pipeline = MG_State::GLState::ProgramPipelineObject;
|
||||
// THE FIELDS ARE READ DIRECTLY RATHER THAN THROUGH THE TWO FUNCTIONS THAT
|
||||
// ALREADY PACK THEM, and that is a gate constraint, not a preference. Calling
|
||||
// ComputeDrawProgramSignature() / ComputeUniformMirrorVersions() would say
|
||||
// "the same pairs the composite cache and the uniform-mirror gate compare"
|
||||
// far better than this loop does - but gen_pipe_dirty_surface.py derives a
|
||||
// shutter by following each accessor to the member it returns, and both of
|
||||
// those build a LOCAL array and return that, which it cannot place. A shutter
|
||||
// naming them is UNRESOLVED, and then every DirtySurface.def row that names
|
||||
// bits 6, 7, 8 or 14 loses its verdict - including the derivation that is the
|
||||
// only mechanism able to catch the next under-fire here. So the pairs are
|
||||
// spelled out, and the two static_asserts below are what say they must stay in
|
||||
// step with the functions they mirror.
|
||||
static_assert(sizeof(Pipeline::DrawProgramSignature) ==
|
||||
2 * Pipeline::kGraphicsStageCount * sizeof(Uint64),
|
||||
"bit 6 reads the {lifetimeId, linkVersion} pair per graphics "
|
||||
"stage that ComputeDrawProgramSignature packs");
|
||||
static_assert(sizeof(Pipeline::UniformMirrorVersions) ==
|
||||
2 * Pipeline::kGraphicsStageCount * sizeof(Uint64),
|
||||
"bits 7 and 8 read the four counters per graphics stage that "
|
||||
"ComputeUniformMirrorVersions packs");
|
||||
|
||||
// Bit 6 is the pipeline's identity plus the composite cache key. Bits 7 and 8
|
||||
// add the per-program state, which under a pipeline is written to the STAGE
|
||||
// programs - glUniform* addresses the pipeline's active program,
|
||||
// glProgramUniform* and the two block-binding calls address a named one - and
|
||||
// only reaches the composite through RefreshCompositeUniforms. Bit 14's half
|
||||
// takes the image-unit generation, which is its own counter for the reason
|
||||
// ProgramObject gives (ES forbids glUniform1i on an image uniform, so Espryt
|
||||
// BAKES the unit into the ESSL it generates and only a regeneration honours a
|
||||
// change) and which D-G4 asks this shutter to keep reading as a FRONTEND
|
||||
// counter rather than any server-side epoch.
|
||||
//
|
||||
// STAGELINKS IS MIXED INTO ALL THREE OF THE OTHERS, ON PURPOSE. A composite
|
||||
// REBUILD hands back a brand-new ProgramObject with an empty default uniform
|
||||
// block and no backend state at all - SetCachedDrawProgram clears the mirror
|
||||
// versions with it - so a shutter watching only the per-stage state counters
|
||||
// would let a rebuilt composite inherit the bindings, the constants and the
|
||||
// image units of the one it replaced.
|
||||
Uint64 stageLinks = static_cast<Uint64>(ctx.GetBoundProgramPipelineName());
|
||||
Uint64 stageState = 0;
|
||||
Uint64 stageImages = 0;
|
||||
// The per-stage sampler/image unit assignments alone (glUniform1i on a stage
|
||||
// program's sampler moves its backend state version and reaches the composite
|
||||
// through the uniform mirror), for bit 12's program input below.
|
||||
Uint64 stageOpaque = 0;
|
||||
for (SizeT stage = 0; stage < Pipeline::kGraphicsStageCount; ++stage) {
|
||||
const auto& staged = pipeline->GetStageProgram(static_cast<ShaderStage>(stage));
|
||||
if (!staged) continue;
|
||||
stageLinks = MGPipeMixShutter(
|
||||
MGPipeMixShutter(stageLinks, staged->GetLifetimeId()), staged->GetLinkVersion());
|
||||
stageState = MGPipeMixShutter(
|
||||
MGPipeMixShutter(MGPipeMixShutter(stageState, staged->GetBackendStateVersion()),
|
||||
MGPipeMixShutter(staged->GetUBOContentVersion(),
|
||||
staged->GetBlockBindingVersion())),
|
||||
staged->GetUniformWriteSetVersion());
|
||||
stageImages = MGPipeMixShutter(stageImages, staged->GetImageUnitVersion());
|
||||
stageOpaque = MGPipeMixShutter(stageOpaque, staged->GetBackendStateVersion());
|
||||
}
|
||||
shader = stageLinks;
|
||||
stageState = MGPipeMixShutter(stageLinks, stageState);
|
||||
bindings = MGPipeMixShutter(stageState, stageImages);
|
||||
constants = stageState;
|
||||
programImages = MGPipeMixShutter(stageLinks, stageImages);
|
||||
opaqueUnits = MGPipeMixShutter(stageLinks, stageOpaque);
|
||||
}
|
||||
now[Index(MGPipeDirty::NewShader)] = shader;
|
||||
now[Index(MGPipeDirty::NewShaderBindings)] = bindings;
|
||||
now[Index(MGPipeDirty::NewGlobalConstants)] = constants;
|
||||
|
||||
// ===========================================================================
|
||||
// THE RECORD-FIELD -> SETTER -> SHUTTER TABLE FOR THE SEVEN P4a BITS.
|
||||
//
|
||||
// THE RULE (P4a fable seam audit, section C.1): every field of every emitted
|
||||
// record names the frontend setter that changes it, and that setter moves a
|
||||
// counter the emitting bit's shutter reads - or the emission is unconditional at
|
||||
// the setter (the resource_* family, set_texture_params). A record field whose
|
||||
// setter moves no shutter input is a stale record with nothing to refuse: c0d
|
||||
// (bit 13 without the bind generation), SD-0 (an image re-bind), F-1 (the
|
||||
// program behind the view set), F-2 (the program behind the image window) and
|
||||
// F-3 (an attached object's storage) were all this one class. DirtySurface.def
|
||||
// cannot catch it - it maps MUTATORS to bits and cannot see that a DERIVED field
|
||||
// depends on a mutator whose row is another family's - so the table lives here,
|
||||
// beside the shutters, and a row is added whenever a record gains a field.
|
||||
//
|
||||
// bit 6 create/bind_shader_state, set_draw/dispatch_program (ProgramEmit.h)
|
||||
// fields: Cso, StageMask, GlobalUboSize, the artefact blob refs, the two
|
||||
// bound handles
|
||||
// setters: glUseProgram (m_currentProgram), glLinkProgram (link version),
|
||||
// glBindProgramPipeline / glUseProgramStages (pipeline name +
|
||||
// per-stage {lifetime id, link version})
|
||||
// shutter: lifetime id x link version, or stageLinks under a pipeline
|
||||
// bit 7 the program's bindings (image units, block bindings, uniform write set)
|
||||
// setters: glUniform1i on an opaque uniform (backend state version, image
|
||||
// unit version), glUniformBlockBinding / glShaderStorageBlockBinding
|
||||
// (block binding version), any glUniform* (uniform write set)
|
||||
// shutter: the four per-program counters, x stageLinks under a pipeline
|
||||
// bit 8 set_global_constants: ShaderCso, Version, the default-block image
|
||||
// setters: any glUniform* on the default block (UBO content version),
|
||||
// glUseProgram (lifetime id)
|
||||
// shutter: lifetime id x UBO content version, or stageState
|
||||
// bit 11 set_framebuffer_state: Fbo, Color[8]/Depth/Stencil/ReadSurface
|
||||
// (Res, Kind, InternalFormat, TextureTarget, Layered, Level, Layer,
|
||||
// UploadTarget), DrawBuffers[8], Width/Height/Layers/Samples/
|
||||
// FixedSampleLocations, IsDefault, Complete, Target
|
||||
// setters: glFramebufferTexture*/glFramebufferRenderbuffer, glDrawBuffer(s),
|
||||
// glReadBuffer, glFramebufferParameteri (the attachment
|
||||
// aggregate); glBindFramebuffer (the two binding slot versions);
|
||||
// AND a storage redefinition of an ATTACHED texture or
|
||||
// renderbuffer - glTexImage*/glTexStorage*/glTexBuffer/
|
||||
// glTextureView/glRenderbufferStorage* - because InternalFormat,
|
||||
// TextureTarget, the extent, Samples and Complete are INLINED at
|
||||
// emission (D-C1): those bump the attachment aggregate from the
|
||||
// object's PipePublishDescriptor (F-3)
|
||||
// shutter: attachment aggregate x draw bind version x read bind version
|
||||
// bit 12 set_sampler_views: per unit {View, Texture}
|
||||
// setters: glBindTexture / glActiveTexture (bind generation), a texture's
|
||||
// or a sampler object's parameters (SamplesAsIncompleteTexture -
|
||||
// the params aggregate), an upload that defines a level (content
|
||||
// aggregate), the default texture's image appearing (bind
|
||||
// generation, TextureObject.cpp); AND the program in use -
|
||||
// glUseProgram, a relink, glUniform1i on a sampler uniform (which
|
||||
// unit a uniform's TYPE resolves) - F-1
|
||||
// shutter: content x params x bind generation x opaqueUnits
|
||||
// bit 13 bind_sampler_states: per unit the sampler CSO handle
|
||||
// setters: glBindSampler (bind generation, c0d), glSamplerParameter* /
|
||||
// glTexParameter* (params aggregate + sampling resolution),
|
||||
// glDeleteSamplers (bind generation)
|
||||
// shutter: params x sampling resolution x bind generation
|
||||
// bit 14 set_shader_images: per unit {Res, InternalFormat, Layer, Level,
|
||||
// Layered, Access} over the program's image-unit window
|
||||
// setters: glBindImageTexture (bind generation, SD-0), a texture's
|
||||
// content/params, glUniform1i on an image uniform (image unit
|
||||
// version); AND the program in use - glUseProgram, a relink -
|
||||
// F-2
|
||||
// shutter: content x params x bind generation x programImages
|
||||
// (lifetime id x link version x image unit version)
|
||||
// ===========================================================================
|
||||
|
||||
// ---- the object-class bits 9..17 ----
|
||||
const Uint64 textureContent = ctx.GetAnyTextureContentGeneration();
|
||||
const Uint64 textureParams = ctx.GetAnyTextureParamsGeneration();
|
||||
const Uint64 buffers = ctx.GetAnyBufferChangeGeneration();
|
||||
|
||||
// Bit 9. The VAO attribute aggregate mixed with the bound VAO's identity is
|
||||
// already exact for the SET - it is bumped by all three Bump*Version functions,
|
||||
// which are the only writers of an attribute's format, buffer or enable state -
|
||||
// and a driver-id re-mint that moves no client counter is caught server-side by
|
||||
// the backend's own id generation.
|
||||
//
|
||||
// THE PENDING BASE INSTANCE IS MIXED IN, and this is a deviation from the design
|
||||
// note that said "keep the shutter" (recorded in client-v1.md): the draw's
|
||||
// baseInstance is now an EXPLICIT field of set_vertex_buffers and a
|
||||
// ContentHash input, and it moves neither the attribute aggregate nor the VAO
|
||||
// identity. Without it here, a draw whose only change is its base instance would
|
||||
// never reach the emitter at all and the server would keep the previous fetch
|
||||
// shift - which is the same silently-wrong-geometry the backend's
|
||||
// baseInstanceDirty flag exists to prevent, one level further out. It fires
|
||||
// extra only on the draws that actually carry one.
|
||||
now[Index(MGPipeDirty::NewVertexBuffers)] = MGPipeMixShutter(
|
||||
MGPipeMixShutter(ctx.GetAnyVaoAttributeGeneration(), vaoIdentity), m_pendingBaseInstance);
|
||||
// Bit 10, NARROWED (P3a, D-I). It used to mix the whole buffer-CONTENT aggregate
|
||||
// with the VAO identity and therefore fired on any buffer write anywhere; what
|
||||
// it guards is one binding slot, so it now reads that slot's own version and the
|
||||
// identity of what is bound to it. The version is a WRAPPING Uint16 bumped only
|
||||
// on a real change, so it goes through the widened counter at this boundary; the
|
||||
// bound object's lifetime id joins it because identity is what closes the wrap
|
||||
// hole. The VAO identity stays in the mix because the element slot BELONGS to
|
||||
// the bound VAO - switching VAOs switches slots.
|
||||
Uint64 indexShutter = 0;
|
||||
if (vao) {
|
||||
const auto& indexSlot = vao->GetIndexBufferBindingSlot();
|
||||
const auto& indexObject = indexSlot.GetBoundObject();
|
||||
indexShutter = MGPipeMixShutter(m_indexSlotVersion.Observe(indexSlot.GetVersion()),
|
||||
indexObject ? indexObject->GetLifetimeId() : 0);
|
||||
}
|
||||
now[Index(MGPipeDirty::NewIndexBuffer)] = MGPipeMixShutter(vaoIdentity, indexShutter);
|
||||
// Bit 11, WIDENED AT P4a AND THIS IS A REQUIREMENT RATHER THAN AN OPTION. The
|
||||
// shutter observed the DRAW binding slot only, so glBindFramebuffer(
|
||||
// GL_READ_FRAMEBUFFER, ...) moved nothing at all - which was harmless while
|
||||
// nothing was emitted for the bit and is an UNDER-FIRE the moment P4a emits
|
||||
// set_framebuffer_state per bound target (D-C2): the read record would never be
|
||||
// sent and the server's ReadSurface would stay the previous framebuffer's. Over-
|
||||
// firing costs one extra push; under-firing renders stale, and this file's own
|
||||
// rule is that under-firing is the dangerous direction.
|
||||
//
|
||||
// A STORAGE REDEFINITION OF AN ATTACHED OBJECT MOVES THIS SHUTTER (P4a fable seam
|
||||
// F-3), and the sentence that stood here - "a renderbuffer respecify is still
|
||||
// invisible here, and deliberately so ... closed by emitting resource_respecify
|
||||
// straight from the storage entry point" - was true of the RESOURCE record only.
|
||||
// set_framebuffer_state inlines each attachment's InternalFormat, TextureTarget,
|
||||
// extent, Samples and Complete (D-C1: "so the four cross-object masks fall out at
|
||||
// push time with no lookup"), so `glTexImage2D(tex, RGB8); attach; draw;
|
||||
// glTexImage2D(tex, RGBA8); draw` left the FRAMEBUFFER record saying RGB8 while the
|
||||
// resource record said RGBA8, and the handle arm answered its alpha-widening,
|
||||
// snorm-clamp and integer masks from the stale copy where the legacy arm re-read
|
||||
// the frontend at the same re-sync - a proven arm divergence on a public-GL
|
||||
// sequence. The fix is at the SETTER, not here: TextureObjectBase::PipePublish
|
||||
// Descriptor and RenderbufferObject::PipePublishDescriptor - the one funnel every
|
||||
// storage-defining entry point of either object takes, push-only - bump the
|
||||
// attachment aggregate this shutter already reads. No counter is added to either
|
||||
// object (G1), nothing widens this shutter onto the texture-content aggregate (which
|
||||
// would fire the 304-byte record build on every glTexSubImage2D), and a storage
|
||||
// definition of an UNATTACHED object over-fires it exactly once at load time.
|
||||
//
|
||||
// AND A TRAP THE NEXT NARROWING WOULD WALK INTO, recorded here because it is
|
||||
// invisible from the shutter: FramebufferObject::SetDrawBuffer versions the VALUE
|
||||
// being written rather than the index being written TO - it calls
|
||||
// BumpAttachmentVersion(buffer). The object version and the aggregate still move,
|
||||
// so THIS shutter is safe; a narrower one built on m_attachmentVersions would not
|
||||
// be, and P4a must not build one.
|
||||
now[Index(MGPipeDirty::NewFramebuffer)] = MGPipeMixShutter(
|
||||
MGPipeMixShutter(
|
||||
ctx.GetAnyFramebufferAttachmentGeneration(),
|
||||
m_framebufferBind.Observe(
|
||||
ctx.GetFramebufferBindingSlot(FramebufferTarget::Draw).GetVersion())),
|
||||
m_readFramebufferBind.Observe(
|
||||
ctx.GetFramebufferBindingSlot(FramebufferTarget::Read).GetVersion()));
|
||||
// Bit 12 reads FOUR things (F-1): the two texture aggregates, the bind generation
|
||||
// and the program input computed above. The params aggregate is here because
|
||||
// SamplerEmit.h drops a unit's view to null when SamplesAsIncompleteTexture says so,
|
||||
// and that predicate reads the effective sampler's filters - a glTexParameteri(
|
||||
// MIN_FILTER) that completes a texture fired bit 13 and not this one, so the entry
|
||||
// stayed null. The program input is here because the set is resolved FOR THE
|
||||
// PROGRAM IN USE, and a glUseProgram alone moved nothing this shutter read.
|
||||
now[Index(MGPipeDirty::NewSamplerViews)] = MGPipeMixShutter(
|
||||
MGPipeMixShutter(MGPipeMixShutter(textureContent, textureParams), ctx.GetTextureBindGeneration()),
|
||||
opaqueUnits);
|
||||
// Bit 13, WIDENED AT P4a FOR BIT 11's REASON and found the same way. glBindSampler
|
||||
// moves NEITHER half of what this used to read: GL_Sampler.cpp's BindSampler_State
|
||||
// goes through NoteTextureUnitTouched and TextureUnit::SetSamplerObject, and both
|
||||
// of those bump the TEXTURE BIND generation - bit 12's. The only two writers of
|
||||
// BumpSamplingResolutionGeneration are PARAMETER changes (SamplerObject.cpp,
|
||||
// TextureObject.cpp). So `glBindSampler(3, a); draw; glBindSampler(3, b); draw`
|
||||
// fired bit 12 twice and bit 13 not once, and the server's BoundSamplerStates[3]
|
||||
// went on naming a's CSO: wrong filtering, with nothing able to see it, because
|
||||
// bind_sampler_states has no pulled twin to fall back on the way the view set does.
|
||||
//
|
||||
// MIXING THE GENERATION IN IS THE FIX RATHER THAN A SECOND GATE ON THE EMITTER,
|
||||
// because that generation is what the unit SET is derived from: a sampler bind
|
||||
// changes which sampler state applies at a unit, and a texture bind changes it too
|
||||
// whenever the unit carries no sampler object and the texture's BUILT-IN sampler is
|
||||
// what applies. Keeping it one shutter per bit is also what keeps the per-subsystem
|
||||
// A/B and the per-bit fire tallies meaning what they say - a bit gated on another
|
||||
// bit's shutter measures neither. The extra fires a plain texture bind now costs
|
||||
// are swallowed by the emitter's own set-hash suppressor, which MGPipeTypes.h makes
|
||||
// mandatory for every kVarTail set for this exact traffic.
|
||||
now[Index(MGPipeDirty::NewSamplers)] = MGPipeMixShutter(
|
||||
MGPipeMixShutter(textureParams, ctx.GetSamplingResolutionGeneration()),
|
||||
ctx.GetTextureBindGeneration());
|
||||
now[Index(MGPipeDirty::NewShaderImages)] = MGPipeMixShutter(
|
||||
MGPipeMixShutter(MGPipeMixShutter(textureContent, textureParams), programImages),
|
||||
ctx.GetTextureBindGeneration());
|
||||
now[Index(MGPipeDirty::NewConstBuffers)] = buffers;
|
||||
now[Index(MGPipeDirty::NewShaderBuffers)] = buffers;
|
||||
now[Index(MGPipeDirty::NewSoTargets)] =
|
||||
MGPipeMixShutter(buffers, ctx.GetTransformFeedbackGeneration());
|
||||
|
||||
Uint32 dirty = 0;
|
||||
for (SizeT i = 0; i < kMGPipeDirtyCount; ++i) {
|
||||
// Bits 2 and 3 are handled below: they are BitwiseEqual shutters, not
|
||||
// counters, so they have no entry in `now`.
|
||||
if (i == Index(MGPipeDirty::NewPixelPack) || i == Index(MGPipeDirty::NewPatchState)) {
|
||||
continue;
|
||||
}
|
||||
if (!m_primed || now[i] != m_lastPushed[i]) dirty |= Uint32{1} << static_cast<Uint32>(i);
|
||||
m_lastPushed[i] = now[i];
|
||||
}
|
||||
|
||||
// ---- bit 2: the PACK half of the pixel store, BitwiseEqual ----
|
||||
const PixelStoreParameters pack = ctx.GetPixelStoreParameters(false);
|
||||
if (!m_primed || std::memcmp(&pack, &m_pack, sizeof(pack)) != 0) {
|
||||
dirty |= MGPipeDirtyBit(MGPipeDirty::NewPixelPack);
|
||||
m_pack = pack;
|
||||
}
|
||||
|
||||
// ---- bit 3: the patch trio, BitwiseEqual, and NaN IS LEGAL ----
|
||||
// A NaN outer level is a legal glPatchParameterfv value and must compare equal to
|
||||
// itself (ARCHITECTURE.md 5.2). Float equality says it is not; memcmp says it is,
|
||||
// which is the whole reason this is a byte compare.
|
||||
PatchTrio patch{};
|
||||
patch.PatchVertices = render.PatchVertices;
|
||||
for (SizeT i = 0; i < 4; ++i) patch.Outer[i] = render.PatchDefaultOuterLevel[i];
|
||||
for (SizeT i = 0; i < 2; ++i) patch.Inner[i] = render.PatchDefaultInnerLevel[i];
|
||||
if (!m_primed || std::memcmp(&patch, &m_patch, sizeof(patch)) != 0) {
|
||||
dirty |= MGPipeDirtyBit(MGPipeDirty::NewPatchState);
|
||||
m_patch = patch;
|
||||
}
|
||||
|
||||
m_primed = true;
|
||||
m_freshlyPrimed = !wasPrimed;
|
||||
m_lastDirty = dirty;
|
||||
|
||||
if (MG_Util::PipeStats::Enabled()) {
|
||||
const SizeT cls = static_cast<SizeT>(verbClass);
|
||||
++m_walks[cls];
|
||||
for (SizeT i = 0; i < kMGPipeDirtyCount; ++i) {
|
||||
if (dirty & (Uint32{1} << static_cast<Uint32>(i))) ++m_fires[i][cls];
|
||||
}
|
||||
}
|
||||
return dirty;
|
||||
}
|
||||
|
||||
// Context teardown, server reset, a unit test's fixture. The next Update returns
|
||||
// every bit set, which is what makes the first verb on a fresh context publish a
|
||||
// complete state rather than an increment. Deliberately does NOT clear the fire
|
||||
// tallies: they are a per-run measurement, not per-context state.
|
||||
//
|
||||
// AND IT DELIBERATELY DOES NOT CLEAR m_pendingBaseInstance. Everything else this
|
||||
// function clears is a LATCH describing what the server was last told; the pending
|
||||
// base instance is THIS CALL'S ARGUMENT, written by the draw entry point one
|
||||
// statement before MGP_FILL and not yet read by anybody. Update() calls Reset() from
|
||||
// inside itself whenever the current GLContext pointer moves, so clearing it here
|
||||
// meant that `eglMakeCurrent(ctxB); glDrawArraysInstancedBaseInstance(..., 7)` put a
|
||||
// BaseInstance of 0 on the wire - one silently mis-shifted instanced draw per context
|
||||
// switch, on the emulation path, with nothing to catch it. The value is cleared by the
|
||||
// verb that consumes it (PipeFill.cpp's step 3, and its no-context early return) and
|
||||
// by MGPipeLeaveVerb, which is where a per-call argument belongs.
|
||||
void Reset() {
|
||||
std::memset(m_lastPushed, 0, sizeof(m_lastPushed));
|
||||
m_renderStateVersion.Reset();
|
||||
m_pipelineStateVersion.Reset();
|
||||
m_framebufferBind.Reset();
|
||||
m_readFramebufferBind.Reset();
|
||||
m_indexSlotVersion.Reset();
|
||||
m_pack = PixelStoreParameters{};
|
||||
m_patch = PatchTrio{};
|
||||
m_staged = RenderStateParameters{};
|
||||
m_stagedAttribs = AttribDefaults{};
|
||||
m_context = nullptr;
|
||||
m_lastDirty = 0;
|
||||
m_primed = false;
|
||||
m_freshlyPrimed = false;
|
||||
}
|
||||
|
||||
void ResetCounters() {
|
||||
std::memset(m_fires, 0, sizeof(m_fires));
|
||||
std::memset(m_walks, 0, sizeof(m_walks));
|
||||
}
|
||||
|
||||
Uint64 FireCount(MGPipeDirty bit, MGPipeVerbClass verbClass) const {
|
||||
return m_fires[Index(bit)][static_cast<SizeT>(verbClass)];
|
||||
}
|
||||
Uint64 FireCount(MGPipeDirty bit) const {
|
||||
Uint64 total = 0;
|
||||
for (SizeT i = 0; i < kMGPipeVerbClassCount; ++i) total += m_fires[Index(bit)][i];
|
||||
return total;
|
||||
}
|
||||
Uint64 WalkCount(MGPipeVerbClass verbClass) const {
|
||||
return m_walks[static_cast<SizeT>(verbClass)];
|
||||
}
|
||||
Uint64 WalkCount() const {
|
||||
Uint64 total = 0;
|
||||
for (SizeT i = 0; i < kMGPipeVerbClassCount; ++i) total += m_walks[i];
|
||||
return total;
|
||||
}
|
||||
|
||||
Uint32 LastDirty() const { return m_lastDirty; }
|
||||
Bool Primed() const { return m_primed; }
|
||||
// True when the LAST Update was the first one after a Reset - a fresh context, or a
|
||||
// server reset. The emission step reads it to send a COMPLETE state rather than an
|
||||
// increment against a staging mirror that describes a context that is gone.
|
||||
Bool FreshlyPrimed() const { return m_freshlyPrimed; }
|
||||
|
||||
// "What the server has" (P2 brief D8). set_dynamic_state sends the dynamic chunks
|
||||
// that differ from this, which is the chunk-level suppressor; a chunk that
|
||||
// memcmp-matches is not sent at all.
|
||||
RenderStateParameters& Staged() { return m_staged; }
|
||||
const RenderStateParameters& Staged() const { return m_staged; }
|
||||
|
||||
// The same mirror for the 32 glVertexAttrib* defaults: set_vertex_attrib_defaults
|
||||
// names only the attributes that differ from it, which is the var-tail's own
|
||||
// suppressor underneath D11's set-hash one.
|
||||
using AttribDefaults = Array<MG_State::GLState::CurrentVertexAttributeValue,
|
||||
MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS>;
|
||||
AttribDefaults& StagedAttribDefaults() { return m_stagedAttribs; }
|
||||
const AttribDefaults& StagedAttribDefaults() const { return m_stagedAttribs; }
|
||||
|
||||
// ---- P3a D-H2: the draw's vertex-FETCH base instance ----
|
||||
//
|
||||
// It lives HERE rather than in a file static because bit 9's shutter has to see it:
|
||||
// an ambient process global cannot cross a pushed boundary, and the value is now an
|
||||
// explicit field of set_vertex_buffers and an input to its content hash, so a draw
|
||||
// whose only change is its base instance has to reach the emitter. Set immediately
|
||||
// before the fill at the three *BaseInstance draw entry points; CONSUMED and cleared
|
||||
// by the validate point once it has been emitted, so a plain draw that follows one
|
||||
// sees 0 again.
|
||||
//
|
||||
// THE CLEAR THAT ACTUALLY RUNS IN PRODUCTION IS THE VALIDATE POINT'S. MGPipeLeaveVerb
|
||||
// clears it too, but no GL entry point calls MGPipeLeaveVerb - only MG_Test's
|
||||
// ScopedPipeVerb and TrackerTest do - so the production guarantee is entirely
|
||||
// PipeFill.cpp's, on BOTH of its exits: the end of step 3, and the no-live-context
|
||||
// early return that skips step 3 altogether. Reset() deliberately does not clear it
|
||||
// (see there): it is this call's argument, not a latch.
|
||||
void SetPendingBaseInstance(Uint32 baseInstance) { m_pendingBaseInstance = baseInstance; }
|
||||
Uint32 PendingBaseInstance() const { return m_pendingBaseInstance; }
|
||||
void ClearPendingBaseInstance() { m_pendingBaseInstance = 0; }
|
||||
|
||||
private:
|
||||
static constexpr SizeT Index(MGPipeDirty bit) { return static_cast<SizeT>(bit); }
|
||||
|
||||
struct PatchTrio {
|
||||
Uint PatchVertices;
|
||||
Float Outer[4];
|
||||
Float Inner[2];
|
||||
};
|
||||
|
||||
Uint64 m_lastPushed[kMGPipeDirtyCount]{};
|
||||
MGPipeWidenedCounter m_renderStateVersion;
|
||||
MGPipeWidenedCounter m_pipelineStateVersion;
|
||||
// The draw framebuffer BINDING slot version, widened for the same reason: a Uint16
|
||||
// that wrapped would let a composite shutter repeat and cost a missed fire.
|
||||
MGPipeWidenedCounter m_framebufferBind;
|
||||
// P4a: the READ framebuffer binding slot's version, its own counter for the same
|
||||
// reason the draw one exists. Two counters rather than one over both slots: a single
|
||||
// widened counter fed two independent Uint16s reads a decrease as a wrap on every
|
||||
// alternation and would add 65536 per switch, which costs nothing in correctness
|
||||
// (over-firing) but makes the high word meaningless.
|
||||
MGPipeWidenedCounter m_readFramebufferBind;
|
||||
// The BOUND VAO's element-array slot version, widened for the same reason. One
|
||||
// counter over a slot that changes with the bound VAO: a stale high word can only
|
||||
// ADD a fire, never drop one, and the VAO identity in the same mix is what makes a
|
||||
// switch between two VAOs differ whatever their slot versions read.
|
||||
MGPipeWidenedCounter m_indexSlotVersion;
|
||||
Uint32 m_pendingBaseInstance = 0;
|
||||
// Bits 2 and 3 are BitwiseEqual shutters, not counters.
|
||||
PixelStoreParameters m_pack{};
|
||||
PatchTrio m_patch{};
|
||||
|
||||
RenderStateParameters m_staged{};
|
||||
AttribDefaults m_stagedAttribs{};
|
||||
|
||||
const void* m_context = nullptr;
|
||||
Uint32 m_lastDirty = 0;
|
||||
Bool m_primed = false;
|
||||
Bool m_freshlyPrimed = false;
|
||||
|
||||
Uint64 m_fires[kMGPipeDirtyCount][kMGPipeVerbClassCount]{};
|
||||
Uint64 m_walks[kMGPipeVerbClassCount]{};
|
||||
};
|
||||
|
||||
// ONE attribute default, flattened onto the wire (P2 brief D10). A named function rather
|
||||
// than four lines inside the emitter because this flattening is the whole correctness
|
||||
// question of set_vertex_attrib_defaults: a CurrentVertexAttributeValue is one value in
|
||||
// three views and GLContext converts NUMERICALLY between them, so four words alone are
|
||||
// not the value - glVertexAttrib4f(loc, 1.5f, ...) leaves 1 in intValue and 0x3FC00000 in
|
||||
// floatValue. MGPAttribValue::ValueClass is what makes the four words readable again, and
|
||||
// TrackerAttribPayload pins that here instead of leaving it to the emitter's shape.
|
||||
inline void MGPipeFillAttribValue(Uint32 location,
|
||||
const MG_State::GLState::CurrentVertexAttributeValue& value,
|
||||
Uint32 writtenClass, MGPAttribValue& out) {
|
||||
out = MGPAttribValue{};
|
||||
out.Location = location;
|
||||
out.ValueClass = static_cast<Uint8>(writtenClass);
|
||||
static_assert(sizeof(out.Data) == sizeof(value.floatValue), "MGPAttribValue::Data is four words");
|
||||
switch (writtenClass) {
|
||||
case MG_State::GLState::kVertexAttribValueClassInt:
|
||||
std::memcpy(out.Data, value.intValue.data(), sizeof(out.Data));
|
||||
break;
|
||||
case MG_State::GLState::kVertexAttribValueClassUint:
|
||||
std::memcpy(out.Data, value.uintValue.data(), sizeof(out.Data));
|
||||
break;
|
||||
default:
|
||||
std::memcpy(out.Data, value.floatValue.data(), sizeof(out.Data));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// The monolith's one tracker. Under split there is one per client context; the context
|
||||
// identity check inside Update is what makes the single instance safe today.
|
||||
inline MGPipeTracker& MGPipeTrackerInstance() {
|
||||
// NEVER DESTROYED, for MGPipeSlots()' reason (MG_Impl/Pipe/SlotAllocator.cpp). The
|
||||
// rule is stated over the SET of MGPipe process singletons rather than over the two
|
||||
// that a frontend destructor reaches today: which of them a destructor reaches is a
|
||||
// property of the emitters, and the emitters change (C-1 added a second reaching
|
||||
// path in one commit). One allocation per process, no destructor to lose - this type
|
||||
// has none - and nothing can then answer a late call out of freed storage.
|
||||
static MGPipeTracker* tracker = new MGPipeTracker();
|
||||
return *tracker;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
@@ -0,0 +1,444 @@
|
||||
// MobileGL - MobileGL/MG_Impl/Pipe/VertexInputEmit.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
// The CLIENT side of P3a's vertex-input family (brief D-G, D-H, D-I): the bound VAO's
|
||||
// format as create/bind_vertex_elements, its buffers as set_vertex_buffers with an explicit
|
||||
// baseInstance, and its element binding as set_index_buffer.
|
||||
//
|
||||
// UNLIKE THE RESOURCE FAMILY, these three emit at the VALIDATE POINT, from
|
||||
// MGPipeValidateForVerb's step 3 in the fixed order elements -> buffers -> index. That is
|
||||
// the ordinary rule (ARCHITECTURE.md 5.1); the resource family is the one exception to it.
|
||||
//
|
||||
// THE CSO IS IDENTITY-ADDRESSED, NOT CONTENT-ADDRESSED (D-G1, a recorded deviation from
|
||||
// ARCHITECTURE.md's 1024-entry content-addressed scheme). One handle per frontend
|
||||
// VertexArrayObject, minted off its lifetime id, and create_vertex_elements is RE-ISSUED on
|
||||
// the same handle whenever the configuration moves - legal, because MGPipeHandle::Gen
|
||||
// increments only on slot reuse and never on a respecify. Espryt has no vertex-elements CSO
|
||||
// to share: its twin owns one driver VAO name plus 64 scratch buffer ids, which two frontend
|
||||
// VAOs cannot share, so content addressing would be strictly slower on the only backend this
|
||||
// phase touches. P7 adds the hash-probe-memcmp layer above these same three calls when
|
||||
// Magma's VertexInputStateFactory takes the CSO over.
|
||||
//
|
||||
// WHAT THE UNIT GATE READS. G6 is "the emitted blob + set + index record reproduce exactly
|
||||
// what the backend's VAO twin reads from the frontend today, field by field, for all 32
|
||||
// slots", and G7 is a scripted control that stops the conversion copying ONE field and
|
||||
// expects the suite to go red NAMING it. So the conversion is a pure function per field
|
||||
// (MGPipeBuildVertexAttribWire / MGPipeBuildVertexBindingPointWire) and the staging buffers
|
||||
// the emitter builds into are readable afterwards - the emitter passes m_blob and m_entries
|
||||
// straight to the applier, so "what was emitted" costs no copy at all.
|
||||
//
|
||||
// HEADER-ONLY, for the ownership reason Tracker.h states in full.
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/ResourceTracker.h>
|
||||
#include <MG_Impl/Pipe/SetHashSuppressor.h>
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#include <MG_Impl/Pipe/Tracker.h>
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
|
||||
#include <xxhash.h>
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace MobileGL::MG_Pipe {
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-G2: the wire conversion, one pure function per view
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
// EVERY FIELD OF VertexAttribute THE WIRE FORM CARRIES, and nothing else:
|
||||
//
|
||||
// Divisor is deliberately absent - it is resolved per binding point and travels in
|
||||
// MGPVertexBuffer::Divisor, which is where the backend's glVertexAttribDivisor reads
|
||||
// it. Carrying it twice would let a malformed record disagree with itself.
|
||||
// LegacyStride / LegacyPointer are deliberately absent - they are the
|
||||
// glGetVertexAttrib* query answers and nothing but the query path reads them, so
|
||||
// they stay client-side.
|
||||
// Buffer is deliberately absent - identity travels in set_vertex_buffers, which is
|
||||
// what keeps this record stable while the buffers under it change.
|
||||
// Stride is the RESOLVED distance and a surviving 0 is MEANINGFUL: a pointer call's 0
|
||||
// was already resolved to the element size by the frontend, so a 0 here can only
|
||||
// have come from the binding model, where it means every vertex reads the SAME
|
||||
// element. Collapsing it back into the element size is what made
|
||||
// KHR-GL43.vertex_attrib_binding.basic-input-case7/8 read past the buffer.
|
||||
// IsLong travels SEPARATELY from Type == Float64: VertexAttribFormat(GL_DOUBLE) reads
|
||||
// doubles and asks for them converted to float, VertexAttribLFormat keeps all 64
|
||||
// bits, and the backend's fp64 narrowing and its Adreno disabled-attribute
|
||||
// workaround both key on telling the two apart.
|
||||
inline MGPVertexAttribWire MGPipeBuildVertexAttribWire(const MG_State::GLState::VertexAttribute& attrib,
|
||||
Uint32 bindingIndex) {
|
||||
// ASSERT RATHER THAN ASSUME, in both directions, because the three narrowing casts
|
||||
// below cross a package boundary: VertexArrayObject is another package's file and its
|
||||
// 32-slot bound is its invariant, not this one's, so a BindingIndex of 256 would wrap
|
||||
// to 0 and silently point every attribute at binding 0, and a negative Stride (the
|
||||
// frontend field is a signed int) would arrive as a ~4 GiB unsigned distance.
|
||||
MOBILEGL_ASSERT(bindingIndex < 256u,
|
||||
"MGPVertexAttribWire::BindingIndex is a Uint8 and cannot carry %u",
|
||||
static_cast<Uint>(bindingIndex));
|
||||
MOBILEGL_ASSERT(attrib.Size >= 0 && attrib.Size <= 255,
|
||||
"MGPVertexAttribWire::Size is a Uint8 and cannot carry %d", attrib.Size);
|
||||
MGPVertexAttribWire wire{};
|
||||
wire.Offset = static_cast<Uint64>(attrib.Offset);
|
||||
wire.Stride = static_cast<Int32>(attrib.Stride);
|
||||
wire.Type = static_cast<Uint32>(attrib.Type);
|
||||
wire.Size = static_cast<Uint8>(attrib.Size);
|
||||
wire.Enabled = attrib.Enabled ? 1 : 0;
|
||||
wire.Normalized = attrib.Normalized ? 1 : 0;
|
||||
wire.IsInteger = attrib.IsInteger ? 1 : 0;
|
||||
wire.IsLong = attrib.IsLong ? 1 : 0;
|
||||
wire.IsBgra = attrib.IsBgra ? 1 : 0;
|
||||
wire.BindingIndex = static_cast<Uint8>(bindingIndex);
|
||||
return wire;
|
||||
}
|
||||
|
||||
// The ARB_vertex_attrib_binding view. Its initial Stride is 16, not 0 (GL 4.6 core table
|
||||
// 23.4), which is why the wire form keeps it signed and copies it verbatim.
|
||||
inline MGPVertexBindingPointWire
|
||||
MGPipeBuildVertexBindingPointWire(const MG_State::GLState::VertexBufferBindingPoint& point) {
|
||||
MGPVertexBindingPointWire wire{};
|
||||
wire.Offset = static_cast<Uint64>(point.Offset);
|
||||
wire.Stride = static_cast<Int32>(point.Stride);
|
||||
wire.Divisor = static_cast<Uint32>(point.Divisor);
|
||||
return wire;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// D-H2.3: the content hash, WITH BaseInstance in it
|
||||
// ---------------------------------------------------------------------------------
|
||||
//
|
||||
// A HARD REQUIREMENT, not a nicety. set_vertex_buffers is suppressed on an unchanged
|
||||
// hash (SetHashSuppressor.h's SetVertexBuffers slot), so a baseInstance that moved while
|
||||
// the buffer set did not would be suppressed and the server would keep the previous
|
||||
// fetch shift - exactly the bug the backend's baseInstanceDirty flag exists to prevent.
|
||||
inline Uint64 MGPipeVertexBufferSetContentHash(const MGPVertexBuffer* entries, Uint32 start, Uint32 count,
|
||||
Uint32 baseInstance) {
|
||||
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPVertexBuffer), 0);
|
||||
hash = MGPipeMixShutter(hash, start);
|
||||
hash = MGPipeMixShutter(hash, count);
|
||||
hash = MGPipeMixShutter(hash, baseInstance);
|
||||
return hash;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------
|
||||
// The emitter
|
||||
// ---------------------------------------------------------------------------------
|
||||
|
||||
class MGPipeVertexInputEmitter {
|
||||
public:
|
||||
using GLContext = MG_State::GLState::GLContext;
|
||||
using VertexArrayObject = MG_State::GLState::VertexArrayObject;
|
||||
static constexpr SizeT kAttribs = static_cast<SizeT>(VertexArrayObject::MAX_VERTEX_ATTRIBS);
|
||||
static constexpr SizeT kBindings = static_cast<SizeT>(VertexArrayObject::MAX_VERTEX_ATTRIB_BINDINGS);
|
||||
static_assert(kAttribs <= kMGPipeMaxVertexAttribs && kBindings <= kMGPipeMaxVertexAttribs,
|
||||
"both declared counts are bounded by kMGPipeMaxVertexAttribs");
|
||||
|
||||
// create/bind_vertex_elements. D-G3's three arms, verbatim:
|
||||
//
|
||||
// no VAO bound -> bind the null handle (legal, and it means
|
||||
// exactly "no VAO bound")
|
||||
// the bound VAO CHANGED -> (re)create if its configuration moved since
|
||||
// this handle last published one, then bind
|
||||
// the same VAO, configuration MOVED-> create on the SAME handle, and do NOT rebind
|
||||
//
|
||||
// The latch is PER HANDLE, in a slot-indexed table, so ping-ponging between two VAOs
|
||||
// re-binds but never re-creates either. A Uint32 configuration version does not wrap
|
||||
// in any realistic run and is compared directly; the tracker's widened counter is
|
||||
// for the Uint16s and is not needed here.
|
||||
Uint64 EmitVertexElements(GLContext& ctx) {
|
||||
const auto& vao = ctx.GetBoundVertexArray();
|
||||
if (!vao) {
|
||||
if (!MGPipeHandleIsNull(m_boundHandle)) {
|
||||
MGPipeApplyBindVertexElements(HandleOnly(kMGPipeNullHandle));
|
||||
++m_binds;
|
||||
m_boundHandle = kMGPipeNullHandle;
|
||||
m_boundLifetimeId = 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
const Uint64 lifetimeId = vao->GetLifetimeId();
|
||||
const Uint32 configVersion = vao->GetConfigVersion();
|
||||
const MGPipeHandle handle = MGPipeSlots().Acquire(MGPipeKind::VertexElementsCso, lifetimeId);
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_latch.size()) m_latch.resize(slot + 1);
|
||||
Latch& latch = m_latch[slot];
|
||||
|
||||
Uint64 bytes = 0;
|
||||
const Bool configMoved = !latch.Published || latch.ConfigVersion != configVersion ||
|
||||
latch.Gen != handle.Gen;
|
||||
if (configMoved) bytes += EmitCreate(*vao, handle, latch, configVersion);
|
||||
if (lifetimeId != m_boundLifetimeId || m_boundHandle != handle) {
|
||||
MGPipeApplyBindVertexElements(HandleOnly(handle));
|
||||
++m_binds;
|
||||
bytes += sizeof(MGPHandleOnly);
|
||||
m_boundHandle = handle;
|
||||
m_boundLifetimeId = lifetimeId;
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
|
||||
// set_vertex_buffers. Espryt consumes RESOLVED attributes, so the set is one entry
|
||||
// per attribute slot with BindingIndex == the attribute index; Start is 0 and Count
|
||||
// is the highest ENABLED attribute plus one, which is the 32-slot prefix walk the
|
||||
// dirty bit is specified over.
|
||||
//
|
||||
// A client-memory array is Res == kMGPipeNullHandle, and that is not a hole: it is
|
||||
// exactly how the server learns "this attribute is client-sourced, upload it
|
||||
// yourself". Its store genuinely does not exist at this moment - the client-array
|
||||
// uploader runs after PrepareForDraw, at the draw entry point - and moving that
|
||||
// resolution to the client is P8's.
|
||||
Uint64 EmitVertexBuffers(GLContext& ctx, Uint32 baseInstance) {
|
||||
const auto& vao = ctx.GetBoundVertexArray();
|
||||
Uint32 count = 0;
|
||||
if (vao) {
|
||||
for (SizeT i = 0; i < kAttribs; ++i) {
|
||||
if (vao->GetAttribute(static_cast<Uint>(i)).Enabled) count = static_cast<Uint32>(i) + 1;
|
||||
}
|
||||
for (SizeT i = 0; i < count; ++i) {
|
||||
const auto& attrib = vao->GetAttribute(static_cast<Uint>(i));
|
||||
MGPVertexBuffer& entry = m_entries[i];
|
||||
entry = MGPVertexBuffer{};
|
||||
entry.Res = attrib.Buffer ? MGPipeSlots().Acquire(MGPipeKind::Buffer,
|
||||
attrib.Buffer->GetLifetimeId())
|
||||
: kMGPipeNullHandle;
|
||||
// D-A3's sticky mask, ORed HERE rather than only sampled at a storage op.
|
||||
// This is the bit that survives the DSA idiom: a buffer defined through
|
||||
// glNamedBuffer* may never be bound at any resource emission, but a draw
|
||||
// that fetches from it resolves it right here, on the GL thread, at every
|
||||
// draw. Sticky, so one draw is enough for the rest of its life.
|
||||
MGPipeResourceTrackerInstance().NoteBoundAs(entry.Res, BufferTarget::Vertex);
|
||||
// The attribute's own byte offset lives in MGPVertexAttribWire::Offset,
|
||||
// so the entry's is the BINDING's, which the frontend already folded in.
|
||||
entry.Offset = 0;
|
||||
// Signed on the frontend, unsigned on the wire, and a negative one would
|
||||
// arrive as a ~4 GiB fetch distance rather than as an error.
|
||||
MOBILEGL_ASSERT(attrib.Stride >= 0, "a resolved vertex stride is never negative (%d)",
|
||||
attrib.Stride);
|
||||
entry.Stride = static_cast<Uint32>(attrib.Stride);
|
||||
entry.Divisor = static_cast<Uint32>(attrib.Divisor);
|
||||
entry.BindingIndex = static_cast<Uint32>(i);
|
||||
}
|
||||
}
|
||||
|
||||
const Uint64 hash = MGPipeVertexBufferSetContentHash(m_entries.data(), 0, count, baseInstance);
|
||||
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetVertexBuffers, hash)) {
|
||||
return 0;
|
||||
}
|
||||
m_lastBuffers = MGPVertexBuffers{};
|
||||
m_lastBuffers.Start = 0;
|
||||
m_lastBuffers.Count = count;
|
||||
// THE DRAW'S RAW value. The client never pre-shifts an offset and never learns
|
||||
// whether the server emulated the shift or let GL_EXT_base_instance do it -
|
||||
// emulation is server-owned.
|
||||
m_lastBuffers.BaseInstance = baseInstance;
|
||||
m_lastBuffers.ContentHash = hash;
|
||||
MGPipeApplySetVertexBuffers(m_lastBuffers, m_entries.data());
|
||||
++m_bufferSets;
|
||||
return sizeof(MGPVertexBuffers) + static_cast<Uint64>(count) * sizeof(MGPVertexBuffer);
|
||||
}
|
||||
|
||||
// set_index_buffer. An INDEPENDENT call, not a subset of the vertex-elements
|
||||
// configuration version (D5) - the index slot is explicitly outside the VAO's
|
||||
// m_configVersion, and the shutter for it is bit 10's, narrowed in Tracker.h.
|
||||
//
|
||||
// Offset and IndexSize are 0 here and the draw verb overrides them: at the validate
|
||||
// point there is no draw to read them from, and the applier stores what it is given.
|
||||
Uint64 EmitIndexBuffer(GLContext& ctx) {
|
||||
const auto& vao = ctx.GetBoundVertexArray();
|
||||
m_lastIndex = MGPIndexBuffer{};
|
||||
if (vao) {
|
||||
if (const auto& bound = vao->GetIndexBufferBindingSlot().GetBoundObject()) {
|
||||
m_lastIndex.Res = MGPipeSlots().Acquire(MGPipeKind::Buffer, bound->GetLifetimeId());
|
||||
// The ELEMENT_ARRAY bit, and it is the one the split path keys on
|
||||
// (kCapNeedsHostIndexBytes -> restart rewriting, multi-draw flattening).
|
||||
// Noted at every draw for RefreshBindMask's reason: an EBO defined through
|
||||
// DSA and unbound before its last respecify would otherwise never publish
|
||||
// it, and getting that bit wrong is invisible in monolith.
|
||||
MGPipeResourceTrackerInstance().NoteBoundAs(m_lastIndex.Res, BufferTarget::Index);
|
||||
}
|
||||
}
|
||||
MGPipeApplySetIndexBuffer(m_lastIndex);
|
||||
++m_indexSets;
|
||||
return sizeof(MGPIndexBuffer);
|
||||
}
|
||||
|
||||
// ---- what a unit case reads. None of it costs a copy: the emitter builds INTO
|
||||
// these and hands the applier the same pointers. ----
|
||||
const Array<MGPVertexAttribWire, kMGPipeMaxVertexAttribs>& LastAttributes() const { return m_attributes; }
|
||||
const Array<MGPVertexBindingPointWire, kMGPipeMaxVertexAttribs>& LastBindingPoints() const {
|
||||
return m_bindingPoints;
|
||||
}
|
||||
const MGPVertexElements& LastElements() const { return m_lastElements; }
|
||||
const MGPVertexBuffers& LastVertexBuffers() const { return m_lastBuffers; }
|
||||
const Array<MGPVertexBuffer, kMGPipeMaxVertexAttribs>& LastEntries() const { return m_entries; }
|
||||
const MGPIndexBuffer& LastIndexBuffer() const { return m_lastIndex; }
|
||||
MGPipeHandle BoundHandle() const { return m_boundHandle; }
|
||||
Uint64 CreateCount() const { return m_creates; }
|
||||
Uint64 BindCount() const { return m_binds; }
|
||||
Uint64 VertexBufferSetCount() const { return m_bufferSets; }
|
||||
Uint64 IndexBufferSetCount() const { return m_indexSets; }
|
||||
|
||||
// ---- C-1: "does the applier hold a record for exactly this handle?" ----
|
||||
//
|
||||
// The CSO's death path (MGPipeEmitVertexElementsDestroyAndFree) needs that answer and
|
||||
// MUST NOT GUESS IT FROM THE SLOT. A VertexElementsCso slot can exist with no record
|
||||
// behind it, because a backend that keys its twins on the handle mints the slot itself
|
||||
// (DirectGLES' BackendSlotTable::GetOrCreate -> MGPipeSlots().Acquire) whether or not
|
||||
// bit 8 ever asked this client to emit anything - which is exactly what a
|
||||
// MOBILEGL_PIPE_PUSH=0x7f lane runs. delete_vertex_elements on such a handle is a
|
||||
// REFUSED call, and the applier's resolver asserts on a refusal
|
||||
// (PipeApply.cpp's ResolveVertexElements), i.e. a stop in a verify build.
|
||||
//
|
||||
// Kept OUT of Reset(), unlike the create/bind latch beside it, and for the mirror
|
||||
// image of Reset()'s own reason: "a fresh context is a fresh server" is true of the
|
||||
// per-context half of this table, and object RECORDS are precisely what
|
||||
// MGPipeApplierReset does not clear (PipeApply.h's two halves). This half tracks those
|
||||
// records, so it lives exactly as long as they do.
|
||||
Bool RecordIsPublished(MGPipeHandle handle) const {
|
||||
if (MGPipeHandleIsNull(handle)) return false;
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot >= m_latch.size()) return false;
|
||||
const Latch& latch = m_latch[slot];
|
||||
return latch.RecordLive && latch.RecordGen == handle.Gen;
|
||||
}
|
||||
|
||||
// The record named by `handle` is gone from the applier. Also drops the bound-handle
|
||||
// memo when it named it, so the client's idea of BoundVertexElements and the applier's
|
||||
// (which MGPipeApplyDeleteVertexElements just cleared for the same handle) stay in
|
||||
// step rather than diverging until the next bind happens to correct it.
|
||||
void NoteRecordDestroyed(MGPipeHandle handle) {
|
||||
if (MGPipeHandleIsNull(handle)) return;
|
||||
const SizeT slot = handle.Slot;
|
||||
if (slot < m_latch.size() && m_latch[slot].RecordGen == handle.Gen) {
|
||||
m_latch[slot] = Latch{};
|
||||
}
|
||||
if (m_boundHandle == handle) {
|
||||
m_boundHandle = kMGPipeNullHandle;
|
||||
m_boundLifetimeId = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// A fresh context is a fresh server: the applier's records are gone, so every latch
|
||||
// this emitter holds describes objects the server no longer has. Called from the
|
||||
// validate point's FreshlyPrimed arm beside MGPipeApplierReset and the suppressor's
|
||||
// InvalidateAll, for the same reason they are.
|
||||
//
|
||||
// The PER-CONTEXT half only - see RecordIsPublished above for why RecordLive/RecordGen
|
||||
// survive. Re-creating a configuration the applier already holds is a bounded
|
||||
// over-fire (MGPipeApplyCreateVertexElements starts the record over); forgetting that
|
||||
// it holds one at all would leak the record and its slot at the object's death.
|
||||
void Reset() {
|
||||
for (Latch& latch : m_latch) {
|
||||
latch.Published = false;
|
||||
latch.Gen = 0;
|
||||
latch.ConfigVersion = 0;
|
||||
}
|
||||
m_boundHandle = kMGPipeNullHandle;
|
||||
m_boundLifetimeId = 0;
|
||||
}
|
||||
|
||||
void ResetCounters() { m_creates = m_binds = m_bufferSets = m_indexSets = 0; }
|
||||
|
||||
private:
|
||||
struct Latch {
|
||||
// The PER-CONTEXT half: "has this emitter told THIS server about this handle's
|
||||
// configuration". Cleared by Reset() at every make-current.
|
||||
Bool Published = false;
|
||||
Uint32 Gen = 0;
|
||||
Uint32 ConfigVersion = 0;
|
||||
// The RECORD half: "does the applier hold a create_vertex_elements record at this
|
||||
// slot, for this generation". Lives as long as the record does - see
|
||||
// RecordIsPublished.
|
||||
Bool RecordLive = false;
|
||||
Uint32 RecordGen = 0;
|
||||
};
|
||||
|
||||
static MGPHandleOnly HandleOnly(MGPipeHandle handle) {
|
||||
MGPHandleOnly only{};
|
||||
only.Handle = handle;
|
||||
only.Kind = static_cast<Uint32>(MGPipeKind::VertexElementsCso);
|
||||
return only;
|
||||
}
|
||||
|
||||
Uint64 EmitCreate(const VertexArrayObject& vao, MGPipeHandle handle, Latch& latch, Uint32 configVersion) {
|
||||
// ALL 32 OF EACH, deliberately. The record DECLARES both counts and the applier
|
||||
// refuses one whose counts do not describe its own blob, so a self-describing
|
||||
// record is the cheap shape - and G6 is stated over all 32 slots, which a
|
||||
// truncated set could not answer. It rides create_vertex_elements only, i.e.
|
||||
// once per configuration change, never per draw.
|
||||
for (SizeT i = 0; i < kAttribs; ++i) {
|
||||
m_attributes[i] = MGPipeBuildVertexAttribWire(vao.GetAttribute(static_cast<Uint>(i)),
|
||||
vao.GetAttributeBindingIndex(static_cast<Uint>(i)));
|
||||
}
|
||||
for (SizeT i = 0; i < kBindings; ++i) {
|
||||
m_bindingPoints[i] = MGPipeBuildVertexBindingPointWire(vao.GetBindingPoint(static_cast<Uint>(i)));
|
||||
}
|
||||
// Attributes first, then binding points, both ascending and contiguous.
|
||||
constexpr SizeT kAttribBytes = kAttribs * sizeof(MGPVertexAttribWire);
|
||||
constexpr SizeT kBindingBytes = kBindings * sizeof(MGPVertexBindingPointWire);
|
||||
std::memcpy(m_blob.data(), m_attributes.data(), kAttribBytes);
|
||||
std::memcpy(m_blob.data() + kAttribBytes, m_bindingPoints.data(), kBindingBytes);
|
||||
|
||||
m_lastElements = MGPVertexElements{};
|
||||
m_lastElements.Cso = handle;
|
||||
m_lastElements.AttributeCount = static_cast<Uint32>(kAttribs);
|
||||
m_lastElements.BindingPointCount = static_cast<Uint32>(kBindings);
|
||||
m_lastElements.Blob.Seg = kMGHostSpanSegNone;
|
||||
m_lastElements.Blob.Offset = 0;
|
||||
m_lastElements.Blob.Size = kAttribBytes + kBindingBytes;
|
||||
MGPipeApplyCreateVertexElements(m_lastElements, m_blob.data());
|
||||
++m_creates;
|
||||
latch.Published = true;
|
||||
latch.Gen = handle.Gen;
|
||||
latch.ConfigVersion = configVersion;
|
||||
// THE ONE PRODUCER of the record half: a create that reached the applier is the
|
||||
// only thing that makes delete_vertex_elements a legal call for this handle.
|
||||
latch.RecordLive = true;
|
||||
latch.RecordGen = handle.Gen;
|
||||
return sizeof(MGPVertexElements) + kAttribBytes + kBindingBytes;
|
||||
}
|
||||
|
||||
Array<MGPVertexAttribWire, kMGPipeMaxVertexAttribs> m_attributes{};
|
||||
Array<MGPVertexBindingPointWire, kMGPipeMaxVertexAttribs> m_bindingPoints{};
|
||||
Array<Uint8, kMGPipeMaxVertexAttribs *(sizeof(MGPVertexAttribWire) + sizeof(MGPVertexBindingPointWire))>
|
||||
m_blob{};
|
||||
Array<MGPVertexBuffer, kMGPipeMaxVertexAttribs> m_entries{};
|
||||
|
||||
MGPVertexElements m_lastElements{};
|
||||
MGPVertexBuffers m_lastBuffers{};
|
||||
MGPIndexBuffer m_lastIndex{};
|
||||
|
||||
Vector<Latch> m_latch;
|
||||
MGPipeHandle m_boundHandle = kMGPipeNullHandle;
|
||||
Uint64 m_boundLifetimeId = 0;
|
||||
|
||||
Uint64 m_creates = 0;
|
||||
Uint64 m_binds = 0;
|
||||
Uint64 m_bufferSets = 0;
|
||||
Uint64 m_indexSets = 0;
|
||||
};
|
||||
|
||||
// The monolith's one vertex-input emitter, beside the tracker, the CSO cache, the
|
||||
// set-hash suppressor and the resource tracker.
|
||||
inline MGPipeVertexInputEmitter& MGPipeVertexInputEmitterInstance() {
|
||||
// NEVER DESTROYED, for MGPipeSlots()' reason (MG_Impl/Pipe/SlotAllocator.cpp), and
|
||||
// this one is not hypothetical: C-1 put this emitter DIRECTLY on ~VertexArrayObject's
|
||||
// path - MGPipeEmitVertexElementsDestroyAndFree asks RecordIsPublished(handle) and
|
||||
// then NoteRecordDestroyed(handle), which read and WRITE m_latch. A destroyed
|
||||
// emitter answers out of a freed Vector and the write grows it, i.e. an operator
|
||||
// new + memcpy + operator delete on an already-freed block.
|
||||
static MGPipeVertexInputEmitter* emitter = new MGPipeVertexInputEmitter();
|
||||
return *emitter;
|
||||
}
|
||||
} // namespace MobileGL::MG_Pipe
|
||||
#endif // MOBILEGL_PIPE_PUSH
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,42 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/BackendCapsPeek.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "BackendCapsPeek.h"
|
||||
|
||||
#if !defined(__ANDROID__)
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
// Declared in MG_Backend/BackendObjects.h, which also pulls in both backends' headers
|
||||
// and, through them, their loaders; the reference alone is all that is needed here.
|
||||
extern UniquePtr<BackendObject>& pActiveBackendObject;
|
||||
} // namespace MobileGL::MG_Backend
|
||||
#endif
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
bool PeekComputeWorkGroupCaps(int outCount[3], int outSize[3]) {
|
||||
#if defined(__ANDROID__)
|
||||
(void)outCount;
|
||||
(void)outSize;
|
||||
return false;
|
||||
#else
|
||||
const auto& backend = MobileGL::MG_Backend::pActiveBackendObject;
|
||||
if (!backend) {
|
||||
return false;
|
||||
}
|
||||
const MobileGL::MG_Backend::DynamicBackendParameters& caps = backend->GetDynamicParameters();
|
||||
for (int axis = 0; axis < 3; ++axis) {
|
||||
outCount[axis] = caps.MaxComputeWorkGroupCount[axis];
|
||||
outSize[axis] = caps.MaxComputeWorkGroupSize[axis];
|
||||
}
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,29 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/BackendCapsPeek.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The one place this module looks past the GL API into the active backend's caps block.
|
||||
//
|
||||
// It exists for exactly one assertion: that the six per-axis compute limits the MGPipe
|
||||
// caps block carries (DynamicBackendParameters::MaxComputeWorkGroupCount/Size, plan B
|
||||
// section 4.4.1) are the same numbers glGetIntegeri_v answers today, since P0.5 retires
|
||||
// the getter in favour of the caps. A separate translation unit, because the scenario
|
||||
// sources include the GL headers with prototypes and MobileGL's umbrella header is not
|
||||
// meant to meet them in one file.
|
||||
|
||||
#pragma once
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// Copies the active backend's MaxComputeWorkGroupCount / MaxComputeWorkGroupSize into the
|
||||
// two arrays and returns true. Returns false, touching nothing, where the caps block is
|
||||
// out of reach: on Android this module links the SHIPPING libMobileGL.so, built
|
||||
// -fvisibility=hidden, so no internal symbol resolves; on desktop it links MobileGL_s and
|
||||
// the read is direct.
|
||||
bool PeekComputeWorkGroupCaps(int outCount[3], int outSize[3]);
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -552,7 +552,15 @@ namespace MGITest {
|
||||
// before the pre-flight forks - the child must measure the same platform
|
||||
// the parent will use.
|
||||
EnsureHeadlessPlatform();
|
||||
m_backendName = EnvOr("MOBILEGL_BACKEND_TYPE", "<unset>");
|
||||
// The backend that is actually about to come up, which is what every
|
||||
// `BackendName() == "DirectGLES"` gate in the scenarios means by the question.
|
||||
// MG_ConfigLoader::InitBackendType defaults an unset MOBILEGL_BACKEND_TYPE to
|
||||
// DirectGLES, so the same default belongs here; this used to report the literal
|
||||
// "<unset>" instead. Under ctest the variable is always set by the ENVIRONMENT
|
||||
// property, which is why that never showed - but run straight from a device
|
||||
// shell, where nothing sets it, DirectGLES came up and every case gated on the
|
||||
// NAME DirectGLES skipped as though it had not.
|
||||
m_backendName = EnvOr("MOBILEGL_BACKEND_TYPE", "DirectGLES");
|
||||
m_usable = BringUp();
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/P4aFinalFixPeek.cpp
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "P4aFinalFixPeek.h"
|
||||
|
||||
#if !defined(__ANDROID__)
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Pipe/MGPipeTypes.h>
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_Util/Metrics/PipeStats.h>
|
||||
#define MGITEST_P4A_FINALFIX_PEEK_LIVE 1
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
#if defined(MGITEST_P4A_FINALFIX_PEEK_LIVE)
|
||||
namespace {
|
||||
namespace MGP = MobileGL::MG_Pipe;
|
||||
} // namespace
|
||||
|
||||
bool PeekPipeTextureResourceRecord(unsigned glTextureName, PipeTextureResourceRecordPeek* out) {
|
||||
if (out == nullptr) return false;
|
||||
const MGP::MGPipeApplierState& applier = MGP::MGPipeApplier();
|
||||
// Slot 0 is the reserved null slot; the walk is the same shape PipeApplyPeek.cpp's
|
||||
// params reading takes. A GL name is never an identity on the wire, which is exactly
|
||||
// why it is the right key for a harness that starts from the application's view.
|
||||
for (MobileGL::SizeT slot = 1; slot < applier.TextureResources.size(); ++slot) {
|
||||
const MGP::MGPipeResourceRecord& record = applier.TextureResources[slot];
|
||||
if (!record.Live) continue;
|
||||
if (record.Desc.GlNameForDiag != static_cast<MobileGL::Uint32>(glTextureName)) continue;
|
||||
out->Slot = static_cast<unsigned>(slot);
|
||||
out->Gen = static_cast<unsigned>(record.Gen);
|
||||
out->Serial = static_cast<unsigned long long>(record.Serial);
|
||||
out->BindMask = static_cast<unsigned>(record.Desc.BindMask);
|
||||
out->ImageBindableHint = static_cast<unsigned>(record.Desc.ImageBindableHint);
|
||||
out->Levels = static_cast<unsigned>(record.Desc.Levels);
|
||||
out->PendingUploads = static_cast<unsigned>(record.PendingUploads.size());
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool PeekPipeStatsTextureRemintPulls(unsigned long long* out) {
|
||||
if (out == nullptr) return false;
|
||||
namespace Stats = MobileGL::MG_Util::PipeStats;
|
||||
if (!Stats::Enabled()) Stats::SetEnabledForTesting(true);
|
||||
*out = static_cast<unsigned long long>(Stats::TotalCalls(Stats::CallClass::TextureRemintPulls));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeStatsTextureUploadEmissions(unsigned long long* out) {
|
||||
if (out == nullptr) return false;
|
||||
namespace Stats = MobileGL::MG_Util::PipeStats;
|
||||
if (!Stats::Enabled()) Stats::SetEnabledForTesting(true);
|
||||
*out = static_cast<unsigned long long>(Stats::TotalCalls(Stats::CallClass::TextureUploadEmissions));
|
||||
return true;
|
||||
}
|
||||
#else
|
||||
bool PeekPipeTextureResourceRecord(unsigned, PipeTextureResourceRecordPeek*) { return false; }
|
||||
bool PeekPipeStatsTextureRemintPulls(unsigned long long*) { return false; }
|
||||
bool PeekPipeStatsTextureUploadEmissions(unsigned long long*) { return false; }
|
||||
#endif
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,41 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/P4aFinalFixPeek.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The white-box readings P4aFinalFixScenario.cpp takes, in a translation unit of their own for
|
||||
// P4aSeamPeek.h's reason: a scenario TU includes the GL prototype headers and cannot include
|
||||
// MG_Pipe/PipeApply.h or the Espryt managers beside them, and PipeApplyPeek.cpp is the gates
|
||||
// package's file. Every entry point answers false where the reading cannot be taken (a pull
|
||||
// build, Android, or an applier that holds no record for the name), and a false teaches the
|
||||
// caller nothing - the case declines that half by name and keeps its public-GL verdict.
|
||||
#pragma once
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// The applier's resource record for a texture, found by its GL name (GlNameForDiag - a
|
||||
// diagnostics-only field, which is exactly what a test harness is).
|
||||
struct PipeTextureResourceRecordPeek {
|
||||
unsigned Slot;
|
||||
unsigned Gen;
|
||||
unsigned long long Serial;
|
||||
unsigned BindMask;
|
||||
unsigned ImageBindableHint;
|
||||
unsigned Levels;
|
||||
unsigned PendingUploads;
|
||||
};
|
||||
bool PeekPipeTextureResourceRecord(unsigned glTextureName, PipeTextureResourceRecordPeek* out);
|
||||
|
||||
// The process-wide texture-remint pull count (PipeStats "tex-remint-pulls", `trp=` on the
|
||||
// summary line; ROADMAP open question 2). Arms the PipeStats counters for this process on
|
||||
// the first call, which is what lets a case read the number without a stats-enabled lane.
|
||||
bool PeekPipeStatsTextureRemintPulls(unsigned long long* out);
|
||||
// Espryt's count of texture uploads it actually issued (PipeStats "tex-upload-emissions"):
|
||||
// what tells a CONSUMED pending upload apart from a DROPPED one, since the record's set is
|
||||
// empty either way. Arms the counters the same way.
|
||||
bool PeekPipeStatsTextureUploadEmissions(unsigned long long* out);
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,107 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/P4aSeamPeek.cpp
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "P4aSeamPeek.h"
|
||||
|
||||
#if !defined(__ANDROID__)
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Backend/DirectGLES/Managers.h>
|
||||
#include <MG_Backend/DirectGLES/DirectGLES.h>
|
||||
#define MGITEST_P4A_SEAM_PEEK_LIVE 1
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
#if defined(MGITEST_P4A_SEAM_PEEK_LIVE)
|
||||
namespace {
|
||||
namespace MGP = MobileGL::MG_Pipe;
|
||||
namespace MGB = MobileGL::MG_Backend::DirectGLES;
|
||||
|
||||
// "Is Espryt the backend running" - the same test PipeApplyPeek.cpp makes through a twin:
|
||||
// on Magma no ES entry point was ever resolved and every member of g_GLESFuncs is null.
|
||||
// It is asked BEFORE SamplerSubsystemEnabled(), which is Espryt's own latch and must not
|
||||
// be resolved on a process whose backend is not Espryt.
|
||||
bool EsprytIsRunning() { return MGB::g_GLESFuncs.glBindSampler != nullptr; }
|
||||
} // namespace
|
||||
|
||||
bool PeekEsprytSamplerHandleArmIsLive(bool* outLive) {
|
||||
if (outLive == nullptr) return false;
|
||||
if (!EsprytIsRunning()) return false;
|
||||
*outLive = MGB::SamplerSubsystemEnabled();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekEsprytFramebufferHandleArmIsLive(bool* outLive) {
|
||||
if (outLive == nullptr) return false;
|
||||
if (!EsprytIsRunning()) return false;
|
||||
*outLive = MGB::FramebufferSubsystemEnabled();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeShaderImageWindow(PipeShaderImageWindowPeek* out) {
|
||||
if (out == nullptr) return false;
|
||||
const MGP::MGPipeApplierState& applier = MGP::MGPipeApplier();
|
||||
out->Start = static_cast<unsigned>(applier.ShaderImageStart);
|
||||
out->Count = static_cast<unsigned>(applier.ShaderImageCount);
|
||||
out->Serial = static_cast<unsigned long long>(applier.ShaderImagesSerial);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekEsprytUnitSampler(unsigned unit, unsigned glSamplerName, EsprytUnitSamplerPeek* out) {
|
||||
if (out == nullptr) return false;
|
||||
if (!EsprytIsRunning()) return false;
|
||||
if (!MobileGL::MG_State::pGLContext) return false;
|
||||
const MGP::MGPipeApplierState& applier = MGP::MGPipeApplier();
|
||||
if (unit >= applier.BoundSamplerStates.size() || unit >= MGB::SamplerImpl::g_boundSamplersCache.size()) {
|
||||
return false;
|
||||
}
|
||||
*out = EsprytUnitSamplerPeek{};
|
||||
|
||||
// Espryt's own binding shadow: every glBindSampler this backend issues routes through it
|
||||
// (BackendSamplerObject::Bind / UnbindSampler), so it IS what the driver holds.
|
||||
if (MGB::SamplerImpl::BackendSamplerObject* const bound = MGB::SamplerImpl::g_boundSamplersCache[unit]) {
|
||||
out->BoundSamplerId = static_cast<unsigned>(bound->GetBackendSamplerId());
|
||||
}
|
||||
|
||||
const MGP::MGPipeHandle cso = applier.BoundSamplerStates[unit];
|
||||
out->CsoHandleSlot = static_cast<unsigned>(cso.Slot);
|
||||
out->CsoHandleGen = static_cast<unsigned>(cso.Gen);
|
||||
out->UnitInsideWindow = unit >= applier.SamplerStateStart &&
|
||||
unit - applier.SamplerStateStart < applier.SamplerStateCount;
|
||||
// The twin AT THE CSO HANDLE, asked of the same table Espryt asks (FindByHandle): a null
|
||||
// here with a live handle is the F-4 shape - a content-addressed handle looked up in a
|
||||
// table that only ever held identity-minted slots.
|
||||
if (!MGP::MGPipeHandleIsNull(cso)) {
|
||||
if (auto* const slot = MGB::SamplerImpl::g_backendSamplerObjects.FindByHandle(cso); slot && *slot) {
|
||||
out->CsoTwinSamplerId = static_cast<unsigned>((*slot)->GetBackendSamplerId());
|
||||
}
|
||||
}
|
||||
|
||||
// And the twin keyed on the frontend OBJECT, which is what the pre-handle program pass
|
||||
// used to mint and bind, so a scenario can say which of the two the driver holds.
|
||||
const auto& object = MobileGL::MG_State::pGLContext->GetSamplerObject(
|
||||
static_cast<MobileGL::Uint>(glSamplerName));
|
||||
if (object) {
|
||||
if (auto* const slot = MGB::SamplerImpl::g_backendSamplerObjects.Find(object.get()); slot && *slot) {
|
||||
out->IdentityTwinSamplerId = static_cast<unsigned>((*slot)->GetBackendSamplerId());
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
#else
|
||||
bool PeekEsprytSamplerHandleArmIsLive(bool*) { return false; }
|
||||
bool PeekEsprytFramebufferHandleArmIsLive(bool*) { return false; }
|
||||
bool PeekPipeShaderImageWindow(PipeShaderImageWindowPeek*) { return false; }
|
||||
bool PeekEsprytUnitSampler(unsigned, unsigned, EsprytUnitSamplerPeek*) { return false; }
|
||||
#endif
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,78 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/P4aSeamPeek.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The three readings P4aSeamAuditScenario.cpp takes from the inside, for the two seams the fable
|
||||
// seam audit proved that PUBLIC GL CANNOT SEE: F-4 (the record arm's sampler bind is a permanent
|
||||
// no-op, hidden by the pre-handle program pass binding the same values) and F-2 / SD-4 (the
|
||||
// shader-image window does not follow a program switch, hidden by the server's window/high-water
|
||||
// union taking the pre-handle bind for the units outside it). Both are correct pictures over a
|
||||
// permanent silent fallback, which is precisely the class ROADMAP.md:20 says a gate has to be
|
||||
// able to make red - and the only place the difference exists is inside.
|
||||
//
|
||||
// A SEPARATE TRANSLATION UNIT for PipeApplyPeek.h's reason, verbatim: this file includes
|
||||
// Espryt's own Managers.h, which may not meet a scenario's GL headers in one file. It is NOT
|
||||
// PipeApplyPeek.cpp because that file is package F's (gates v3) and this round may not edit it.
|
||||
//
|
||||
// EVERY ENTRY POINT RETURNS false, TOUCHING NOTHING, WHERE IT CANNOT LOOK - a pull build, Android,
|
||||
// a backend that is not Espryt - and a caller that gets false has learned NOTHING: "could not
|
||||
// look" is not "was bound". The scenario declines the reading BY NAME and keeps its public-GL
|
||||
// half, which is the shape TextureParamsWithoutASamplerViewScenario.cpp argues for.
|
||||
|
||||
#pragma once
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// ---- is Espryt's sampler family on its HANDLE arm in this process? -------------------
|
||||
//
|
||||
// The gate for every other reading here. True only on DirectGLES, in a push build, with
|
||||
// Espryt's own resolver answering "handle" for kMGPipeSubsystemSamplers (bit 11 set and its
|
||||
// dependency satisfied) - i.e. exactly when bind_sampler_states / set_shader_images are
|
||||
// consumed, so a white-box assertion about them can be red for its own reason and for no
|
||||
// other. Written only on true.
|
||||
bool PeekEsprytSamplerHandleArmIsLive(bool* outLive);
|
||||
|
||||
// The same question for the FRAMEBUFFER family (bit 9): true when Espryt consumes
|
||||
// set_framebuffer_state in this process. The renderbuffer half of the F-3 case asserts only
|
||||
// there - on the pre-handle arm a renderbuffer re-storaged while attached moves nothing the
|
||||
// FBO memo reads (D-D2's documented hole, pre-P4a code), and the record is what closes it.
|
||||
bool PeekEsprytFramebufferHandleArmIsLive(bool* outLive);
|
||||
|
||||
// ---- the applier's shader-image window, as last received ------------------------------
|
||||
//
|
||||
// MGPipeApplierState::ShaderImageStart / ShaderImageCount / ShaderImagesSerial. Count is
|
||||
// "how many units set_shader_images last described" - 0 means the set has NEVER arrived
|
||||
// (MGPipeApplierReset advances the serial whether or not anything was emitted, so the serial
|
||||
// is not that test). Push build only.
|
||||
struct PipeShaderImageWindowPeek {
|
||||
unsigned Start;
|
||||
unsigned Count;
|
||||
unsigned long long Serial;
|
||||
};
|
||||
|
||||
bool PeekPipeShaderImageWindow(PipeShaderImageWindowPeek* out);
|
||||
|
||||
// ---- which driver sampler a texture unit is bound to, and whose twin it is -------------
|
||||
//
|
||||
// For F-4. `BoundSamplerId` is the ES sampler name Espryt's own binding shadow says unit
|
||||
// `unit` carries (0 = none). `CsoHandleSlot/Gen` is bind_sampler_states' handle for the unit,
|
||||
// `CsoTwinSamplerId` the ES name of the twin Espryt holds AT THAT HANDLE (0 = no twin at the
|
||||
// content-addressed slot - the F-4 shape), and `IdentityTwinSamplerId` the ES name of a twin
|
||||
// keyed on the frontend SamplerObject named `glSamplerName` (0 = none). On a correct handle
|
||||
// arm the unit's driver sampler IS the CSO twin. Push build, DirectGLES only.
|
||||
struct EsprytUnitSamplerPeek {
|
||||
unsigned BoundSamplerId;
|
||||
unsigned CsoHandleSlot;
|
||||
unsigned CsoHandleGen;
|
||||
bool UnitInsideWindow;
|
||||
unsigned CsoTwinSamplerId;
|
||||
unsigned IdentityTwinSamplerId;
|
||||
};
|
||||
|
||||
bool PeekEsprytUnitSampler(unsigned unit, unsigned glSamplerName, EsprytUnitSamplerPeek* out);
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,188 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/PipeApplyPeek.cpp
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "PipeApplyPeek.h"
|
||||
|
||||
#if !defined(__ANDROID__)
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Pipe/PipeApply.h>
|
||||
#include <MG_Pipe/MGPipeTypes.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
|
||||
#include <MG_Backend/DirectGLES/Managers.h>
|
||||
#include <MG_Backend/DirectGLES/DirectGLES.h>
|
||||
#define MGITEST_PIPE_APPLY_PEEK_LIVE 1
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
#if defined(MGITEST_PIPE_APPLY_PEEK_LIVE)
|
||||
namespace {
|
||||
namespace MGP = MobileGL::MG_Pipe;
|
||||
namespace MGB = MobileGL::MG_Backend::DirectGLES;
|
||||
|
||||
// The frontend texture object a GL name denotes in the CURRENT context, or null. This is
|
||||
// a LOOKUP KEY and nothing else: every value this file reports comes from the applier or
|
||||
// from Espryt, never from the object found here. (Reading the frontend's own parameter
|
||||
// state would answer the question the scenario is asking with the input to it.)
|
||||
MobileGL::MG_State::GLState::ITextureObject* FrontendTexture(unsigned glTextureName) {
|
||||
if (!MobileGL::MG_State::pGLContext) return nullptr;
|
||||
const auto& object = MobileGL::MG_State::pGLContext->GetTextureObject(
|
||||
static_cast<MobileGL::Uint>(glTextureName));
|
||||
return object ? object.get() : nullptr;
|
||||
}
|
||||
|
||||
// Espryt's twin for that texture, or null - which is also this file's "is Espryt even the
|
||||
// backend running" answer. On Magma no Espryt twin was ever built, so every entry point
|
||||
// below stops here rather than reaching for g_GLESFuncs, whose members are null there.
|
||||
MGB::TextureImpl::BackendTextureObject* EsprytTwin(unsigned glTextureName) {
|
||||
MobileGL::MG_State::GLState::ITextureObject* const object = FrontendTexture(glTextureName);
|
||||
if (object == nullptr) return nullptr;
|
||||
auto* const found = MGB::TextureImpl::g_backendTextureObjects.Find(object);
|
||||
if (found == nullptr || !*found) return nullptr;
|
||||
return found->get();
|
||||
}
|
||||
|
||||
int SwizzleToGLEnum(MobileGL::Uint8 encoded) {
|
||||
return static_cast<int>(MobileGL::MG_Util::ConvertTextureSwizzleParamToGLEnum(
|
||||
static_cast<MobileGL::TextureSwizzleParam>(encoded)));
|
||||
}
|
||||
|
||||
// MGPipeTypes.h owns the two numbers and says why depth is 0 (a zeroed record must decode
|
||||
// to what an untouched texture already has). This is that decode, and nothing else in
|
||||
// this module may open-code it.
|
||||
int DepthStencilModeToGLEnum(MobileGL::Uint8 encoded) {
|
||||
return encoded == MGP::kMGPipeDepthStencilModeStencil ? GL_STENCIL_INDEX
|
||||
: GL_DEPTH_COMPONENT;
|
||||
}
|
||||
|
||||
// The GL_TEXTURE_BINDING_* query for a target, or 0 where this file has no answer. A
|
||||
// guess would be worse than a refusal: the binding is what gets RESTORED, so a wrong
|
||||
// pname would leave the driver bound to this test's texture.
|
||||
int BindingQueryFor(unsigned glTarget) {
|
||||
switch (glTarget) {
|
||||
case GL_TEXTURE_2D: return GL_TEXTURE_BINDING_2D;
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool PeekPipeTextureParamsRecord(unsigned glTextureName, PipeTextureParamsRecordPeek* out) {
|
||||
if (out == nullptr) return false;
|
||||
const MGP::MGPipeApplierState& applier = MGP::MGPipeApplier();
|
||||
// Slot 0 is the reserved null handle and is never live (MGPipeHandles.h), so the scan
|
||||
// starts at 1 and a match at 0 is impossible rather than merely unlikely.
|
||||
for (MobileGL::SizeT slot = 1; slot < applier.TextureResources.size(); ++slot) {
|
||||
const MGP::MGPipeResourceRecord& record = applier.TextureResources[slot];
|
||||
if (!record.Live) continue;
|
||||
if (record.Desc.GlNameForDiag != static_cast<MobileGL::Uint32>(glTextureName)) continue;
|
||||
out->Slot = static_cast<unsigned>(slot);
|
||||
out->Gen = static_cast<unsigned>(record.Gen);
|
||||
out->ParamsSerial = static_cast<unsigned long long>(record.ParamsSerial);
|
||||
for (int channel = 0; channel < 4; ++channel) {
|
||||
out->Swizzle[channel] = SwizzleToGLEnum(record.Params.Swizzle[channel]);
|
||||
}
|
||||
out->DepthStencilMode = DepthStencilModeToGLEnum(record.Params.DepthStencilMode);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool PeekEsprytAppliedTextureParams(unsigned glTextureName, unsigned glTarget,
|
||||
EsprytAppliedTextureParamsPeek* out) {
|
||||
if (out == nullptr) return false;
|
||||
const int bindingQuery = BindingQueryFor(glTarget);
|
||||
if (bindingQuery == 0) return false;
|
||||
MGB::TextureImpl::BackendTextureObject* const twin = EsprytTwin(glTextureName);
|
||||
if (twin == nullptr) return false;
|
||||
const MobileGL::Uint backendId = twin->GetBackendTextureId();
|
||||
if (backendId == 0) return false;
|
||||
if (MGB::g_GLESFuncs.glGetTexParameteriv == nullptr ||
|
||||
MGB::g_GLESFuncs.glBindTexture == nullptr || MGB::g_GLESFuncs.glGetIntegerv == nullptr ||
|
||||
MGB::g_GLESFuncs.glGetError == nullptr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// SAVE / QUERY / RESTORE ON THE UNIT THAT IS ALREADY ACTIVE. No glActiveTexture, so the
|
||||
// only driver state this touches is one unit's binding, and it is put back byte for byte
|
||||
// - which is what keeps Espryt's own g_boundTexturesCache true rather than merely
|
||||
// consistent. (Binding through the twin's own Bind() would update that shadow and would
|
||||
// therefore CHANGE what the scenario measures next; this does not.)
|
||||
GLint previousBinding = 0;
|
||||
MGB::g_GLESFuncs.glGetIntegerv(static_cast<GLenum>(bindingQuery), &previousBinding);
|
||||
MGB::g_GLESFuncs.glBindTexture(static_cast<GLenum>(glTarget), backendId);
|
||||
|
||||
out->BackendTextureId = static_cast<unsigned>(backendId);
|
||||
static const GLenum kSwizzlePnames[4] = {GL_TEXTURE_SWIZZLE_R, GL_TEXTURE_SWIZZLE_G,
|
||||
GL_TEXTURE_SWIZZLE_B, GL_TEXTURE_SWIZZLE_A};
|
||||
for (int channel = 0; channel < 4; ++channel) {
|
||||
GLint value = 0;
|
||||
MGB::g_GLESFuncs.glGetTexParameteriv(static_cast<GLenum>(glTarget),
|
||||
kSwizzlePnames[channel], &value);
|
||||
out->Swizzle[channel] = static_cast<int>(value);
|
||||
}
|
||||
|
||||
// The depth/stencil aspect mode is ES 3.1 and is INVALID_ENUM on a driver without it, so
|
||||
// it is asked for last and its own error decides whether the answer is usable. The queue
|
||||
// is drained first because a stale error from anywhere else would be indistinguishable
|
||||
// from this call's - Espryt drains it the same way at every one of its own sync sites
|
||||
// (DebugImpl::ErrorLopper), and this module's own GL errors are read from the FRONTEND
|
||||
// state (ScenarioTest::FirstGLError), which none of this touches.
|
||||
while (MGB::g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
}
|
||||
GLint mode = 0;
|
||||
MGB::g_GLESFuncs.glGetTexParameteriv(static_cast<GLenum>(glTarget),
|
||||
GL_DEPTH_STENCIL_TEXTURE_MODE, &mode);
|
||||
out->DepthStencilModeIsReadable = MGB::g_GLESFuncs.glGetError() == GL_NO_ERROR;
|
||||
out->DepthStencilMode = static_cast<int>(mode);
|
||||
|
||||
MGB::g_GLESFuncs.glBindTexture(static_cast<GLenum>(glTarget),
|
||||
static_cast<GLuint>(previousBinding));
|
||||
while (MGB::g_GLESFuncs.glGetError() != GL_NO_ERROR) {
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekEsprytHasSamplerViewForTexture(unsigned glTextureName, bool* outExists) {
|
||||
if (outExists == nullptr) return false;
|
||||
MobileGL::MG_State::GLState::ITextureObject* const object = FrontendTexture(glTextureName);
|
||||
if (object == nullptr) return false;
|
||||
// Espryt must be the backend running, or "no view" would be true of every texture on
|
||||
// every other backend and the assertion would be vacuous where it is loudest.
|
||||
if (EsprytTwin(glTextureName) == nullptr) return false;
|
||||
// HandleOfSamplerViewForTexture is the monolith glue that derives the view's handle from
|
||||
// the TEXTURE's lifetime id (D-F2: one view per ITextureObject), so this asks Espryt's
|
||||
// own table the same way Espryt asks it - it does not consult the applier record's
|
||||
// ViewCso, which is the client's statement about the same fact and would make one side
|
||||
// of the seam vouch for the other.
|
||||
const MGP::MGPipeHandle view = MGB::SamplerViewImpl::HandleOfSamplerViewForTexture(object);
|
||||
if (MGP::MGPipeHandleIsNull(view)) {
|
||||
*outExists = false;
|
||||
return true;
|
||||
}
|
||||
*outExists = MGB::SamplerViewImpl::FindSamplerViewForHandle(view) != nullptr;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeApplierRefusedNoConsumer(unsigned long long* outCount) {
|
||||
if (outCount == nullptr) return false;
|
||||
*outCount = static_cast<unsigned long long>(MGP::MGPipeApplier().RefusedNoConsumer);
|
||||
return true;
|
||||
}
|
||||
#else
|
||||
bool PeekPipeTextureParamsRecord(unsigned, PipeTextureParamsRecordPeek*) { return false; }
|
||||
bool PeekEsprytAppliedTextureParams(unsigned, unsigned, EsprytAppliedTextureParamsPeek*) {
|
||||
return false;
|
||||
}
|
||||
bool PeekEsprytHasSamplerViewForTexture(unsigned, bool*) { return false; }
|
||||
bool PeekPipeApplierRefusedNoConsumer(unsigned long long*) { return false; }
|
||||
#endif
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,110 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/PipeApplyPeek.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The APPLIER's texture-parameter record, ESPRYT's applied value for the same texture, and
|
||||
// whether that texture has a sampler view yet. Three readings taken from a scenario, for gate
|
||||
// G9's WHITE-BOX half.
|
||||
//
|
||||
// WHY A WHITE-BOX HALF EXISTS AT ALL (ID-19, brief section F, gates review R1). G9's public-GL
|
||||
// cases in TextureParamsWithoutASamplerViewScenario.cpp catch "the parameter never reached the
|
||||
// driver". They CANNOT catch "the parameter reached the driver LATE", because a texture
|
||||
// parameter's only public-GL observable is a SAMPLE and the sample is itself what repairs an
|
||||
// unsynced parameter: it puts the texture on the unit list, and that walk pushes the parameters
|
||||
// for anything whose params serial moved. A backend that deferred every attachment-only
|
||||
// texture's parameters to the first sampler view would be green on all four of those cases,
|
||||
// forever, on every tree. The distinction only exists on the inside, so the reading has to be
|
||||
// taken there - while the texture is still attachment-only, before any sample.
|
||||
//
|
||||
// A SEPARATE TRANSLATION UNIT for PipeSlotPeek.h's reason, verbatim: the scenario sources
|
||||
// include the GL headers with prototypes and MobileGL's umbrella header is not meant to meet
|
||||
// them in one file. This one goes further than PipeSlotPeek and includes Espryt's own
|
||||
// Managers.h, which is exactly why it may not be anywhere near a scenario's GL headers.
|
||||
//
|
||||
// EVERY ENTRY POINT RETURNS false, TOUCHING NOTHING, WHERE IT CANNOT LOOK, and a caller that
|
||||
// gets false has learned NOTHING - "could not look" is not "was applied". Out of reach means:
|
||||
// a PULL build (there is no applier: it is `#if MOBILEGL_PIPE_PUSH`); Android, where this module
|
||||
// links the shipping libMobileGL.so built -fvisibility=hidden and no internal symbol resolves;
|
||||
// a backend that is not DirectGLES (Espryt is the subject; Magma answers the same GL question
|
||||
// through P7's own paths); and, for the record peek, a mask whose texture-resource bit is off,
|
||||
// where no record exists to find because nothing was ever emitted.
|
||||
|
||||
#pragma once
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// ---- the applier's set_texture_params record for one GL texture name ------------------
|
||||
//
|
||||
// ADDRESSED BY GL NAME, and the search key is MGPResourceDesc::GlNameForDiag. That field is
|
||||
// diagnostics-only by contract - never an identity, never a memo key (MGPipeTypes.h) - and
|
||||
// this is a diagnostic: a test harness looking for the record a named GL object produced.
|
||||
// The alternative would be to ask the CLIENT emitter for the texture's handle, and the
|
||||
// review is explicit that this probe must arm on package D's applier/backend state and not
|
||||
// on the emitter markers B and C set: they are different questions, and a shared marker
|
||||
// would re-create the shape review F-M5 was raised about.
|
||||
struct PipeTextureParamsRecordPeek {
|
||||
// The handle the record sits at, so a caller can print it.
|
||||
unsigned Slot;
|
||||
unsigned Gen;
|
||||
// set_texture_params' own serial. 0 means the record exists (the resource was created)
|
||||
// but NO set_texture_params has ever been applied to it - which is a different finding
|
||||
// from "no record", and the two must not be merged.
|
||||
unsigned long long ParamsSerial;
|
||||
// MGPTextureParams::Swizzle[4], translated to the GL enums the application passed to
|
||||
// glTextureParameteri (GL_ZERO / GL_ONE / GL_RED / GL_GREEN / GL_BLUE / GL_ALPHA), so
|
||||
// the scenario compares what it set against what the record carries in ONE vocabulary
|
||||
// and neither side has to know the other's encoding.
|
||||
int Swizzle[4];
|
||||
// MGPTextureParams::DepthStencilMode, translated the same way: GL_DEPTH_COMPONENT or
|
||||
// GL_STENCIL_INDEX.
|
||||
int DepthStencilMode;
|
||||
};
|
||||
|
||||
bool PeekPipeTextureParamsRecord(unsigned glTextureName, PipeTextureParamsRecordPeek* out);
|
||||
|
||||
// ---- Espryt's APPLIED value for the same texture --------------------------------------
|
||||
//
|
||||
// Read from the DRIVER, through the twin's own ES name, because "applied" means the driver
|
||||
// was told - the same thing package D's white-box unit probe asserts against its mocked
|
||||
// driver (esprytobj-v2 (9)). The current binding on the ACTIVE unit is saved and restored
|
||||
// around the query and no unit is switched, so Espryt's binding shadow still describes
|
||||
// reality afterwards: nothing is perturbed for it to be stale about.
|
||||
//
|
||||
// `glTarget` is the texture's GL target (only GL_TEXTURE_2D is supported today; any other
|
||||
// target returns false rather than guessing a binding query).
|
||||
struct EsprytAppliedTextureParamsPeek {
|
||||
// The driver name Espryt minted for this texture, for the caller's message.
|
||||
unsigned BackendTextureId;
|
||||
int Swizzle[4];
|
||||
int DepthStencilMode;
|
||||
// False when the driver rejected the depth/stencil query - a non-depth texture, or an ES
|
||||
// level without GL_DEPTH_STENCIL_TEXTURE_MODE. The swizzle half is still valid.
|
||||
bool DepthStencilModeIsReadable;
|
||||
};
|
||||
|
||||
bool PeekEsprytAppliedTextureParams(unsigned glTextureName, unsigned glTarget,
|
||||
EsprytAppliedTextureParamsPeek* out);
|
||||
|
||||
// ---- and the claim that makes the two above mean anything ------------------------------
|
||||
//
|
||||
// Whether Espryt holds a SAMPLER VIEW twin for this texture. This is the assertion the
|
||||
// public-GL cases cannot make, because making it there would create the view. `*outExists`
|
||||
// is written only on true.
|
||||
bool PeekEsprytHasSamplerViewForTexture(unsigned glTextureName, bool* outExists);
|
||||
|
||||
// ---- c0f's belt, for the ObjectSubsystemControl arms -----------------------------------
|
||||
//
|
||||
// MGPipeApplierState::RefusedNoConsumer: the number of P4a-family entry points that were
|
||||
// refused because no backend had registered MGPipeResourceOps. On a backend WITH a consumer
|
||||
// it must never move; on one without (Magma, ID-39/ID-40) the client's own gate is supposed
|
||||
// to stop the emission before the belt is reached, so it must never move there either. A
|
||||
// non-zero delta says the gate and the belt disagreed, which is the whole point of having
|
||||
// both. Reset by MGPipeApplierReset, so a caller reads it as a DELTA and treats a value that
|
||||
// went DOWN as "the applier was reset, count everything since as `after`".
|
||||
bool PeekPipeApplierRefusedNoConsumer(unsigned long long* outCount);
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,82 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/PipeSlotPeek.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "PipeSlotPeek.h"
|
||||
|
||||
#if !defined(__ANDROID__)
|
||||
#include <MG_Pipe/MGPipe.h>
|
||||
#if MOBILEGL_PIPE_PUSH
|
||||
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||
#define MGITEST_PIPE_SLOT_PEEK_LIVE 1
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
#if defined(MGITEST_PIPE_SLOT_PEEK_LIVE)
|
||||
namespace {
|
||||
// One arm per member, and NO `default:` on purpose: adding a PipeSlotKind without
|
||||
// deciding which MGPipeKind it names is a compiler warning here (-Wswitch) rather than
|
||||
// a row that silently counts VertexElementsCso and reports "did not leak" about a kind
|
||||
// it never looked at. The trailing return is the unreachable one the compiler needs.
|
||||
MobileGL::MG_Pipe::MGPipeKind Translate(PipeSlotKind kind) {
|
||||
switch (kind) {
|
||||
case PipeSlotKind::Buffer: return MobileGL::MG_Pipe::MGPipeKind::Buffer;
|
||||
case PipeSlotKind::VertexElementsCso:
|
||||
return MobileGL::MG_Pipe::MGPipeKind::VertexElementsCso;
|
||||
case PipeSlotKind::Texture: return MobileGL::MG_Pipe::MGPipeKind::Texture;
|
||||
case PipeSlotKind::Renderbuffer: return MobileGL::MG_Pipe::MGPipeKind::Renderbuffer;
|
||||
case PipeSlotKind::Framebuffer: return MobileGL::MG_Pipe::MGPipeKind::Framebuffer;
|
||||
case PipeSlotKind::SamplerCso: return MobileGL::MG_Pipe::MGPipeKind::SamplerCso;
|
||||
case PipeSlotKind::SamplerViewCso:
|
||||
return MobileGL::MG_Pipe::MGPipeKind::SamplerViewCso;
|
||||
case PipeSlotKind::ShaderCso: return MobileGL::MG_Pipe::MGPipeKind::ShaderCso;
|
||||
}
|
||||
return MobileGL::MG_Pipe::MGPipeKind::None;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool PeekPipeSlotLiveCount(PipeSlotKind kind, unsigned* outLive) {
|
||||
if (outLive == nullptr) return false;
|
||||
*outLive = static_cast<unsigned>(MobileGL::MG_Pipe::MGPipeSlots().LiveCount(Translate(kind)));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeSlotHighWater(PipeSlotKind kind, unsigned* outHighWater) {
|
||||
if (outHighWater == nullptr) return false;
|
||||
// The ORDINARY space only, for every kind including ShaderCso (contract-v2.md 4.3).
|
||||
*outHighWater = static_cast<unsigned>(MobileGL::MG_Pipe::MGPipeSlots().HighWater(Translate(kind)));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeCompositeSlotLiveCount(unsigned* outLive) {
|
||||
if (outLive == nullptr) return false;
|
||||
*outLive = static_cast<unsigned>(MobileGL::MG_Pipe::MGPipeSlots().CompositeLiveCount());
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeCompositeSlotHighWater(unsigned* outHighWater) {
|
||||
if (outHighWater == nullptr) return false;
|
||||
*outHighWater = static_cast<unsigned>(MobileGL::MG_Pipe::MGPipeSlots().CompositeHighWater());
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PeekPipeCompositeSlotBandBase(unsigned* outBandBase) {
|
||||
if (outBandBase == nullptr) return false;
|
||||
*outBandBase = static_cast<unsigned>(MobileGL::MG_Pipe::kMGPipeShaderCsoCompositeSlotBase);
|
||||
return true;
|
||||
}
|
||||
#else
|
||||
bool PeekPipeSlotLiveCount(PipeSlotKind, unsigned*) { return false; }
|
||||
bool PeekPipeSlotHighWater(PipeSlotKind, unsigned*) { return false; }
|
||||
bool PeekPipeCompositeSlotLiveCount(unsigned*) { return false; }
|
||||
bool PeekPipeCompositeSlotHighWater(unsigned*) { return false; }
|
||||
bool PeekPipeCompositeSlotBandBase(unsigned*) { return false; }
|
||||
#endif
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,101 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/PipeSlotPeek.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The CLIENT slot allocator's occupancy, read from a scenario.
|
||||
//
|
||||
// It exists for one assertion, P3a's C-1: a frontend object that dies must return its
|
||||
// MGPipeHandle slot WHATEVER BACKEND IS RUNNING. That question has no answer in the GL API -
|
||||
// the leak it rules out is entirely inside the library, and it is invisible in pixels, in GL
|
||||
// names and in `glGetError` - so the only honest observable is the allocator's own live count
|
||||
// and high-water mark. Reading them is what makes the case fail on the backend it actually
|
||||
// failed on (DirectVulkan, which installs no StateObjectDeathOps) rather than only on the one
|
||||
// where a backend-owned free happened to exist.
|
||||
//
|
||||
// A separate translation unit for BackendCapsPeek.h's reason, verbatim: the scenario sources
|
||||
// include the GL headers with prototypes and MobileGL's umbrella header is not meant to meet
|
||||
// them in one file.
|
||||
|
||||
#pragma once
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// Which client-side object kind to ask about. Mirrors MG_Pipe::MGPipeKind for exactly the
|
||||
// kinds a scenario has a reason to count, so that the enum does not travel through this
|
||||
// header and the GL headers together.
|
||||
enum class PipeSlotKind {
|
||||
Buffer,
|
||||
VertexElementsCso,
|
||||
// P4a's six (G8b). Every one of them is a kind the CLIENT mints and the client alone
|
||||
// frees (BRIEF-P4A.md D-I1: one death helper per kind, called from the frontend
|
||||
// object's own destructor, whatever backend is running), so every one of them can leak
|
||||
// the P3a C-1 way - and the leak is invisible in pixels, in GL names and in
|
||||
// glGetError, exactly as the VertexElementsCso one was.
|
||||
Texture,
|
||||
Renderbuffer,
|
||||
// Framebuffer has a HANDLE but no wire lifetime (D-I2): no create_*, no destroy row in
|
||||
// the catalogue, and its death helper does the notice and the free and emits nothing.
|
||||
// That makes the allocator the ONLY observable of its lifetime, so this row matters
|
||||
// more here than the others rather than less.
|
||||
Framebuffer,
|
||||
SamplerCso,
|
||||
SamplerViewCso,
|
||||
// ShaderCso covers BOTH the ordinary program slots and the program-pipeline COMPOSITES
|
||||
// minted out of the reserved high band (MGPipeHandles.h:86-107, D-H7). One kind, because
|
||||
// that is what the allocator has: the band is a second dense table inside the same kind
|
||||
// and LiveCount counts both.
|
||||
//
|
||||
// THE TWO SPACES' HIGH-WATER MARKS ARE NOT ONE NUMBER, and the correction matters here
|
||||
// more than anywhere else. c0b split them (contract-v2.md 4.3): HighWater(ShaderCso) is
|
||||
// now the ORDINARY space only and the band's own mark is CompositeHighWater(), because
|
||||
// a merged mark is pinned at ~983k from the first composite mint onward and every "the
|
||||
// high-water mark did not move over N churn rounds" assertion about ordinary programs
|
||||
// would be vacuously true for the rest of the process. The composite's leak case is a
|
||||
// separate CASE and reads the BAND'S OWN counters below (PeekPipeCompositeSlot*) - a
|
||||
// composite's slot has TWO independent release paths (the pipeline cache's LRU eviction
|
||||
// and the composite ProgramObject's destructor), and a slot that never comes back to
|
||||
// the band moves neither of the ordinary numbers.
|
||||
ShaderCso,
|
||||
};
|
||||
|
||||
// Live slots of this kind right now, and one past the highest slot ever handed out.
|
||||
// Both return false, touching nothing, where the allocator is out of reach: in a PULL
|
||||
// build there is no allocator at all (it is `#if MOBILEGL_PIPE_PUSH`), and on Android this
|
||||
// module links the shipping libMobileGL.so built -fvisibility=hidden, so no internal symbol
|
||||
// resolves. A caller that gets false must SKIP rather than pass - "could not look" is not
|
||||
// "did not leak".
|
||||
bool PeekPipeSlotLiveCount(PipeSlotKind kind, unsigned* outLive);
|
||||
bool PeekPipeSlotHighWater(PipeSlotKind kind, unsigned* outHighWater);
|
||||
|
||||
// The ShaderCso COMPOSITE BAND's own three numbers, the seventh..ninth members
|
||||
// contract-v2.md 4.3 asks this header for. There is no `kind` argument because the band is
|
||||
// ShaderCso's alone - AllocateComposite is the one door into it and no other kind has one.
|
||||
// All three return false on the same terms as the two above, and a caller that gets false
|
||||
// must SKIP.
|
||||
//
|
||||
// PeekPipeCompositeSlotLiveCount = MGPipeSlotAllocator::CompositeLiveCount(), the band's
|
||||
// share of LiveCount(ShaderCso).
|
||||
// PeekPipeCompositeSlotHighWater = CompositeHighWater() VERBATIM, i.e. one past the
|
||||
// highest band slot ever handed out. It is an ABSOLUTE
|
||||
// slot number and therefore starts at the band's base,
|
||||
// not at zero - "no composite was ever minted" reads as
|
||||
// `high water == band base`, which is what the third
|
||||
// member is for. It is not returned base-relative
|
||||
// because a peek whose name says HighWater and whose
|
||||
// value is a delta is exactly the kind of quietly
|
||||
// redefined counter this member exists to correct.
|
||||
// PeekPipeCompositeSlotBandBase = kMGPipeShaderCsoCompositeSlotBase, the floor the
|
||||
// other two are read against. A constant, but it
|
||||
// reaches a scenario only through this header: the
|
||||
// MG_Pipe headers and the GL headers are not meant to
|
||||
// meet in one translation unit, which is why this
|
||||
// harness exists at all.
|
||||
bool PeekPipeCompositeSlotLiveCount(unsigned* outLive);
|
||||
bool PeekPipeCompositeSlotHighWater(unsigned* outHighWater);
|
||||
bool PeekPipeCompositeSlotBandBase(unsigned* outBandBase);
|
||||
|
||||
} // namespace MGITest
|
||||
@@ -0,0 +1,94 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Harness/PipeStatsWindow.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// Reading ONE PipeStats summary window out of the library's own log, for the scenarios whose
|
||||
// claim is about a counter rather than about pixels.
|
||||
//
|
||||
// WHY THROUGH A LOG FILE AT ALL. MG_Util::PipeStats is internal to the library and this module
|
||||
// cannot link against it (ScenarioFixture.h has the long version: on Android this binary links
|
||||
// the SHIPPING libMobileGL.so, built -fvisibility=hidden). The library's `MGPipe stats:` line is
|
||||
// the only channel, so a lane that wants to read a counter sets MOBILEGL_PIPE_STATS=1,
|
||||
// MOBILEGL_PIPE_STATS_PERIOD=1 - one line per eglSwapBuffers - and a MOBILEGL_LOG_FILE_PATH of
|
||||
// its OWN.
|
||||
//
|
||||
// THE LOG PATH HAS TO BE PRIVATE TO ONE CTEST ENTRY, and that is not a style rule: the library
|
||||
// opens it fopen(path, "w"), so every process launched in a lane TRUNCATES it. Two entries of one
|
||||
// lane reading the same path race under `ctest -j`, and the shape of the failure is an empty read
|
||||
// that looks exactly like "the counter was never emitted". So a case that reads a window gets a
|
||||
// ctest entry whose TEST_FILTER selects that case alone, with a log path nothing else writes -
|
||||
// the rule PipeVerifyArmingScenario and CsoContentAddressingScenario already follow.
|
||||
//
|
||||
// THE WINDOW IS "SINCE THE PREVIOUS LINE" (PipeStats::FormatWindowLine), so the caller closes the
|
||||
// setup window with a swap, runs the workload, swaps again, and reads the LAST line - which then
|
||||
// covers the workload and nothing else.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iterator>
|
||||
#include <string>
|
||||
|
||||
namespace MGITest::PipeStatsWindow {
|
||||
|
||||
// The lane's private log path, or empty when the lane configured none.
|
||||
inline std::string LibraryLogPath() {
|
||||
const char* path = std::getenv("MOBILEGL_LOG_FILE_PATH");
|
||||
return (path != nullptr && *path != '\0') ? std::string(path) : std::string();
|
||||
}
|
||||
|
||||
inline std::string ReadWholeFile(const std::string& path) {
|
||||
if (path.empty()) return {};
|
||||
std::ifstream file(path, std::ios::binary);
|
||||
if (!file.good()) return {};
|
||||
return std::string((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());
|
||||
}
|
||||
|
||||
// The last summary line in the log, verbatim. `found` is false when the library never emitted
|
||||
// one, which is a different failure from "the counter read zero" and has to be reported as
|
||||
// one: it means the stats channel never reached the process, not that the workload did
|
||||
// nothing.
|
||||
struct Window {
|
||||
bool found = false;
|
||||
std::string line;
|
||||
};
|
||||
|
||||
inline Window Last(const std::string& log) {
|
||||
Window window;
|
||||
const std::string marker = "MGPipe stats:";
|
||||
const std::size_t at = log.rfind(marker);
|
||||
if (at == std::string::npos) return window;
|
||||
const std::size_t end = log.find('\n', at);
|
||||
window.line = log.substr(at, end == std::string::npos ? std::string::npos : end - at);
|
||||
window.found = true;
|
||||
return window;
|
||||
}
|
||||
|
||||
inline Window LastFromLaneLog() { return Last(ReadWholeFile(LibraryLogPath())); }
|
||||
|
||||
// One counter out of that line, by its short name ("mpr", "draws", "csom"), or -1 when the
|
||||
// line does not carry it. The search includes the SEPARATOR before the name and the `=` after
|
||||
// it, so "draws" cannot match "draws/f=" and "mpr" cannot match a longer name ending in it -
|
||||
// a substring match here would read a neighbouring counter's value and report it as this
|
||||
// one's, which is the one way a counter assertion can be wrong without ever failing.
|
||||
inline long long CounterOrAbsent(const Window& window, const char* shortName) {
|
||||
if (!window.found) return -1;
|
||||
// A counter is preceded either by a space (` mpr=`, ` draws=`) or by its bracket's
|
||||
// opening (`cso[csom=`, `bytes/f[stage-buffer=`); nothing in the line is preceded by
|
||||
// anything else.
|
||||
for (const char* prefix : {" ", "["}) {
|
||||
const std::string key = std::string(prefix) + shortName + "=";
|
||||
const std::size_t at = window.line.find(key);
|
||||
if (at == std::string::npos) continue;
|
||||
return std::strtoll(window.line.c_str() + at + key.size(), nullptr, 10);
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
} // namespace MGITest::PipeStatsWindow
|
||||
@@ -21,12 +21,49 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cctype>
|
||||
#include <cstdlib>
|
||||
#include <string>
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include "HeadlessGL.h"
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// How a MOBILEGL_* quirk variable reads in THIS process's environment.
|
||||
//
|
||||
// A scenario that needs a non-default configuration takes it from here and skips
|
||||
// when the process it was launched into is not in that configuration, rather than
|
||||
// writing MG_Config::Features itself. Two reasons, and the second one decides it:
|
||||
//
|
||||
// - the feature table is an internal symbol. On Android this module links against
|
||||
// the SHIPPING libMobileGL.so - deliberately, so the on-device run validates the
|
||||
// real artifact - and that library is built -fvisibility=hidden, so nothing
|
||||
// internal is reachable from here at all.
|
||||
// - a quirk poked in-process is already too late for everything latched at
|
||||
// initialization: the compile pool and its threads, and the backend's advertised
|
||||
// extension list, which is built once from the configuration in force at first
|
||||
// use. The process-wide variable is the only spelling that covers the whole
|
||||
// configuration instead of the half of it that is still mutable afterwards.
|
||||
//
|
||||
// The reading rule is MG_ConfigLoader's, character for character (ConfigLoader.cpp,
|
||||
// QueryEnvQuirkOverride / IsTruthyValue): unset is Auto - device auto-detection or a
|
||||
// built-in default, i.e. a value only the implementation knows - a truthy value is
|
||||
// On, and anything else that IS set ("0", "false", "") is Off.
|
||||
enum class AmbientQuirk { Auto, On, Off };
|
||||
|
||||
inline AmbientQuirk AmbientQuirkFromEnvironment(const char* name) {
|
||||
const char* value = std::getenv(name);
|
||||
if (value == nullptr) return AmbientQuirk::Auto;
|
||||
std::string lowered(value);
|
||||
for (char& c : lowered) {
|
||||
c = static_cast<char>(std::tolower(static_cast<unsigned char>(c)));
|
||||
}
|
||||
if (lowered.empty() || lowered == "0" || lowered == "false") return AmbientQuirk::Off;
|
||||
return AmbientQuirk::On;
|
||||
}
|
||||
|
||||
class ScenarioTest : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
|
||||
@@ -26,9 +26,11 @@
|
||||
// quantities, so an entry that only fails on DirectVulkan is a translation bug and one that
|
||||
// fails on both is a table bug.
|
||||
|
||||
#include <algorithm>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "../Harness/BackendCapsPeek.h"
|
||||
#include "../Harness/HeadlessGL.h"
|
||||
#include "../Harness/ScenarioFixture.h"
|
||||
|
||||
@@ -56,7 +58,13 @@ namespace MGITest {
|
||||
|
||||
const std::vector<LimitBound>& BufferLimitTable() {
|
||||
static const std::vector<LimitBound> table = {
|
||||
{GL_MAX_UNIFORM_BUFFER_BINDINGS, "GL_MAX_UNIFORM_BUFFER_BINDINGS", 36, 256},
|
||||
// 84 is the GL 4.5 core table 23.64 minimum, and also the width of the state
|
||||
// layer's indexed-binding array - the two were made to coincide when the array
|
||||
// was widened from 36, which had made the clamp in GL_Getter degenerate.
|
||||
{GL_MAX_UNIFORM_BUFFER_BINDINGS, "GL_MAX_UNIFORM_BUFFER_BINDINGS", 84, 256},
|
||||
// 14 uniform blocks on each of the FIVE graphics stages. The sum used to count
|
||||
// three, and the two tessellation stages were simply missing from it.
|
||||
{GL_MAX_COMBINED_UNIFORM_BLOCKS, "GL_MAX_COMBINED_UNIFORM_BLOCKS", 70, 256},
|
||||
{GL_MAX_COMPUTE_UNIFORM_BLOCKS, "GL_MAX_COMPUTE_UNIFORM_BLOCKS", 12, 256},
|
||||
{GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS, "GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS", 8, 256},
|
||||
{GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS, "GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS", 8, 256},
|
||||
@@ -133,6 +141,49 @@ namespace MGITest {
|
||||
<< relation.blocksName << " = " << blocks << " exceeds " << relation.bindingsName << " = "
|
||||
<< bindings << "; a shader may declare more blocks than there are binding points to bind them to";
|
||||
}
|
||||
|
||||
// THE MIDDLE TERM, which the relation quoted above always had and this case never
|
||||
// checked. It is the one that actually broke: widening the binding-point array to 84
|
||||
// raised what every PER-STAGE count clamps to, while the combined value was a
|
||||
// five-stage sum of 70 - so a device reporting descriptor-indexing-scale uniform
|
||||
// buffers (Adreno: maxPerStageDescriptorUniformBuffers = 16777216) advertised 84
|
||||
// compute uniform blocks inside a combined limit of 70. Per-stage <= combined is
|
||||
// exactly the assertion that says so, and it costs one glGetIntegerv per row.
|
||||
struct StageAgainstCombined {
|
||||
GLenum stage;
|
||||
const char* stageName;
|
||||
GLenum combined;
|
||||
const char* combinedName;
|
||||
};
|
||||
const StageAgainstCombined stageRelations[] = {
|
||||
{GL_MAX_COMPUTE_UNIFORM_BLOCKS, "GL_MAX_COMPUTE_UNIFORM_BLOCKS", GL_MAX_COMBINED_UNIFORM_BLOCKS,
|
||||
"GL_MAX_COMBINED_UNIFORM_BLOCKS"},
|
||||
{GL_MAX_VERTEX_UNIFORM_BLOCKS, "GL_MAX_VERTEX_UNIFORM_BLOCKS", GL_MAX_COMBINED_UNIFORM_BLOCKS,
|
||||
"GL_MAX_COMBINED_UNIFORM_BLOCKS"},
|
||||
{GL_MAX_TESS_CONTROL_UNIFORM_BLOCKS, "GL_MAX_TESS_CONTROL_UNIFORM_BLOCKS",
|
||||
GL_MAX_COMBINED_UNIFORM_BLOCKS, "GL_MAX_COMBINED_UNIFORM_BLOCKS"},
|
||||
{GL_MAX_TESS_EVALUATION_UNIFORM_BLOCKS, "GL_MAX_TESS_EVALUATION_UNIFORM_BLOCKS",
|
||||
GL_MAX_COMBINED_UNIFORM_BLOCKS, "GL_MAX_COMBINED_UNIFORM_BLOCKS"},
|
||||
{GL_MAX_GEOMETRY_UNIFORM_BLOCKS, "GL_MAX_GEOMETRY_UNIFORM_BLOCKS", GL_MAX_COMBINED_UNIFORM_BLOCKS,
|
||||
"GL_MAX_COMBINED_UNIFORM_BLOCKS"},
|
||||
{GL_MAX_FRAGMENT_UNIFORM_BLOCKS, "GL_MAX_FRAGMENT_UNIFORM_BLOCKS", GL_MAX_COMBINED_UNIFORM_BLOCKS,
|
||||
"GL_MAX_COMBINED_UNIFORM_BLOCKS"},
|
||||
{GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS, "GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS",
|
||||
GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS, "GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS"},
|
||||
{GL_MAX_FRAGMENT_SHADER_STORAGE_BLOCKS, "GL_MAX_FRAGMENT_SHADER_STORAGE_BLOCKS",
|
||||
GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS, "GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS"},
|
||||
};
|
||||
for (const StageAgainstCombined& relation : stageRelations) {
|
||||
GLint stage = -1;
|
||||
GLint combined = -1;
|
||||
glGetIntegerv(relation.stage, &stage);
|
||||
glGetIntegerv(relation.combined, &combined);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << relation.stageName;
|
||||
EXPECT_LE(stage, combined)
|
||||
<< relation.stageName << " = " << stage << " exceeds " << relation.combinedName << " = "
|
||||
<< combined << "; GL 4.6 table 23.64 orders MAX_*_BUFFER_BINDINGS >= MAX_COMBINED_*_BLOCKS >= "
|
||||
"every per-stage count, and a single-stage program may use its whole per-stage allowance";
|
||||
}
|
||||
}
|
||||
|
||||
// KHR-GL44.multi_bind.functional_bind_buffers_range sizes each of an indexed target's
|
||||
@@ -199,6 +250,80 @@ namespace MGITest {
|
||||
"derived component limits are computed in";
|
||||
}
|
||||
|
||||
// The GL 4.5 core minimums that had no case in the getter at all, or that were still
|
||||
// carrying an ES/GL3.3-tier number. Every one of these answered GL_INVALID_ENUM or a
|
||||
// too-small value against a context advertising 4.6, and each is the FIRST call its
|
||||
// conformance case makes - so the case died before it could measure anything.
|
||||
//
|
||||
// The cull pair is deliberately absent: zero is a legal answer there (a backend with no
|
||||
// cull-distance route MUST report it), so it is checked for answerability only, below.
|
||||
TEST_F(AdvertisedLimitsScenario, EveryGL45CoreMinimumIsMet) {
|
||||
const std::vector<LimitBound> table = {
|
||||
{GL_MAX_VARYING_VECTORS, "GL_MAX_VARYING_VECTORS", 15, 256},
|
||||
{GL_MAX_VERTEX_UNIFORM_VECTORS, "GL_MAX_VERTEX_UNIFORM_VECTORS", 256, 1 << 20},
|
||||
{GL_MAX_VARYING_COMPONENTS, "GL_MAX_VARYING_COMPONENTS", 60, 1 << 20},
|
||||
// GL_MAX_VERTEX_STREAMS is deliberately absent. GL 4.5 requires 4 and MobileGL
|
||||
// answers 1, which is a KNOWN non-conformance rather than an oversight: raising
|
||||
// the number un-gates two transform-feedback CTS cases per package across
|
||||
// KHR-GL40..GL46 that then fail, because no part of the shader pipeline supports
|
||||
// layout(stream = N). See the GL_MAX_VERTEX_STREAMS case in GL_Getter.cpp. Adding
|
||||
// a row here would pin a number the implementation cannot back.
|
||||
{GL_MAX_GEOMETRY_SHADER_INVOCATIONS, "GL_MAX_GEOMETRY_SHADER_INVOCATIONS", 32, 256},
|
||||
{GL_MAX_SUBROUTINES, "GL_MAX_SUBROUTINES", 256, 1 << 20},
|
||||
{GL_MAX_SUBROUTINE_UNIFORM_LOCATIONS, "GL_MAX_SUBROUTINE_UNIFORM_LOCATIONS", 1024, 1 << 20},
|
||||
{GL_MAX_TESS_CONTROL_INPUT_COMPONENTS, "GL_MAX_TESS_CONTROL_INPUT_COMPONENTS", 128, 1 << 16},
|
||||
{GL_MAX_TESS_CONTROL_OUTPUT_COMPONENTS, "GL_MAX_TESS_CONTROL_OUTPUT_COMPONENTS", 128, 1 << 16},
|
||||
{GL_MAX_TESS_CONTROL_TOTAL_OUTPUT_COMPONENTS, "GL_MAX_TESS_CONTROL_TOTAL_OUTPUT_COMPONENTS", 4096,
|
||||
1 << 20},
|
||||
{GL_MAX_TESS_CONTROL_TEXTURE_IMAGE_UNITS, "GL_MAX_TESS_CONTROL_TEXTURE_IMAGE_UNITS", 16, 256},
|
||||
{GL_MAX_TESS_CONTROL_UNIFORM_COMPONENTS, "GL_MAX_TESS_CONTROL_UNIFORM_COMPONENTS", 1024, 1 << 20},
|
||||
{GL_MAX_TESS_CONTROL_UNIFORM_BLOCKS, "GL_MAX_TESS_CONTROL_UNIFORM_BLOCKS", 14, 256},
|
||||
{GL_MAX_TESS_EVALUATION_INPUT_COMPONENTS, "GL_MAX_TESS_EVALUATION_INPUT_COMPONENTS", 128, 1 << 16},
|
||||
{GL_MAX_TESS_EVALUATION_OUTPUT_COMPONENTS, "GL_MAX_TESS_EVALUATION_OUTPUT_COMPONENTS", 128, 1 << 16},
|
||||
{GL_MAX_TESS_EVALUATION_TEXTURE_IMAGE_UNITS, "GL_MAX_TESS_EVALUATION_TEXTURE_IMAGE_UNITS", 16, 256},
|
||||
{GL_MAX_TESS_EVALUATION_UNIFORM_COMPONENTS, "GL_MAX_TESS_EVALUATION_UNIFORM_COMPONENTS", 1024,
|
||||
1 << 20},
|
||||
{GL_MAX_TESS_EVALUATION_UNIFORM_BLOCKS, "GL_MAX_TESS_EVALUATION_UNIFORM_BLOCKS", 14, 256},
|
||||
{GL_MAX_TESS_PATCH_COMPONENTS, "GL_MAX_TESS_PATCH_COMPONENTS", 120, 1 << 16},
|
||||
{GL_MAX_COMBINED_TESS_CONTROL_UNIFORM_COMPONENTS, "GL_MAX_COMBINED_TESS_CONTROL_UNIFORM_COMPONENTS",
|
||||
58368, 1 << 30},
|
||||
{GL_MAX_COMBINED_TESS_EVALUATION_UNIFORM_COMPONENTS,
|
||||
"GL_MAX_COMBINED_TESS_EVALUATION_UNIFORM_COMPONENTS", 58368, 1 << 30},
|
||||
};
|
||||
for (const LimitBound& bound : table) {
|
||||
GLint value = -424242;
|
||||
glGetIntegerv(bound.pname, &value);
|
||||
const unsigned int error = FirstGLError();
|
||||
EXPECT_EQ(error, GLenum(GL_NO_ERROR)) << bound.name << " is not answerable: " << GLErrorName(error);
|
||||
if (error != GL_NO_ERROR) continue;
|
||||
EXPECT_GE(value, bound.minimum) << bound.name << " = " << value << " is below the GL 4.5 minimum "
|
||||
<< bound.minimum;
|
||||
EXPECT_LE(value, bound.ceiling) << bound.name << " = " << value << " exceeds the ceiling "
|
||||
<< bound.ceiling;
|
||||
}
|
||||
|
||||
// ARB_cull_distance's pair. Zero is honest on a backend with no cull-distance route,
|
||||
// so only answerability and the combined-limit ordering are checked here.
|
||||
GLint cull = -1;
|
||||
GLint clip = -1;
|
||||
GLint combined = -1;
|
||||
glGetIntegerv(GL_MAX_CULL_DISTANCES, &cull);
|
||||
glGetIntegerv(GL_MAX_CLIP_DISTANCES, &clip);
|
||||
glGetIntegerv(GL_MAX_COMBINED_CLIP_AND_CULL_DISTANCES, &combined);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << "the ARB_cull_distance queries must not error";
|
||||
EXPECT_GE(cull, 0);
|
||||
EXPECT_GE(combined, cull) << "GL 4.6 core 11.1.3.10: the combined limit is at least the cull one";
|
||||
EXPECT_GE(combined, clip) << "GL 4.6 core 11.1.3.10: the combined limit is at least the clip one";
|
||||
|
||||
// GL_MAX_ELEMENT_INDEX is 64-bit state: the required 2^32-1 does not fit a GLint, so
|
||||
// the wide query must answer it and the narrow one must saturate rather than wrap.
|
||||
GLint64 elementIndex = -1;
|
||||
glGetInteger64v(GL_MAX_ELEMENT_INDEX, &elementIndex);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_GE(elementIndex, static_cast<GLint64>(4294967295LL))
|
||||
<< "GL 4.5 core table 23.55 sets the GL_MAX_ELEMENT_INDEX minimum at 2^32-1";
|
||||
}
|
||||
|
||||
// ARB_viewport_array's own limits. They are advertised from three different places -
|
||||
// GL_MAX_VIEWPORTS from the frontend's indexed state width, the bounds range and the
|
||||
// subpixel bits from the backend caps table - and each backend fills that table from a
|
||||
@@ -255,5 +380,250 @@ namespace MGITest {
|
||||
EXPECT_GE(viewportDims[1], maxRenderbufferSize);
|
||||
}
|
||||
|
||||
|
||||
// THE INDEXED AND PER-PROGRAM QUERIES THAT NAME FRONTEND STATE, pinned on both lanes.
|
||||
//
|
||||
// Both backends used to carry their own arms for GL_SHADER_STORAGE_BUFFER_* and
|
||||
// GL_IMAGE_BINDING_* inside GLFunctionsTable::GetIntegeri_v, and their own
|
||||
// GetInteger64i_v / GetProgramiv table entries. None of it was reachable: GL_Getter and
|
||||
// GL_Program answer every one of these pnames from the frontend's own state and return
|
||||
// before the table is consulted. The duplicates did not even agree - the backend arms
|
||||
// clamped a bound range to the buffer's current storage, which GL 4.6 core tables
|
||||
// 23.4/23.5 do not permit - so the code was one refactor away from becoming the answer.
|
||||
// These cases pin what the frontend actually reports, so a future move of any of it back
|
||||
// behind the interface has to keep saying the same thing.
|
||||
TEST_F(AdvertisedLimitsScenario, IndexedBufferBindingsAreReportedVerbatimOnBothWidths) {
|
||||
GLuint buffer = 0;
|
||||
glGenBuffers(1, &buffer);
|
||||
glBindBuffer(GL_SHADER_STORAGE_BUFFER, buffer);
|
||||
glBufferData(GL_SHADER_STORAGE_BUFFER, 1024, nullptr, GL_DYNAMIC_DRAW);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
|
||||
// A range that is NOT the whole buffer, so a clamp to the store would be visible.
|
||||
glBindBufferRange(GL_SHADER_STORAGE_BUFFER, 1, buffer, 256, 512);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
|
||||
GLint binding32 = -1;
|
||||
GLint start32 = -1;
|
||||
GLint size32 = -1;
|
||||
glGetIntegeri_v(GL_SHADER_STORAGE_BUFFER_BINDING, 1, &binding32);
|
||||
glGetIntegeri_v(GL_SHADER_STORAGE_BUFFER_START, 1, &start32);
|
||||
glGetIntegeri_v(GL_SHADER_STORAGE_BUFFER_SIZE, 1, &size32);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_EQ(binding32, static_cast<GLint>(buffer));
|
||||
EXPECT_EQ(start32, 256);
|
||||
EXPECT_EQ(size32, 512);
|
||||
|
||||
// The 64-bit width has to agree pname for pname. It has no backend entry of its own
|
||||
// and derives everything from the 32-bit answer above plus its own buffer arm.
|
||||
GLint64 binding64 = -1;
|
||||
GLint64 start64 = -1;
|
||||
GLint64 size64 = -1;
|
||||
glGetInteger64i_v(GL_SHADER_STORAGE_BUFFER_BINDING, 1, &binding64);
|
||||
glGetInteger64i_v(GL_SHADER_STORAGE_BUFFER_START, 1, &start64);
|
||||
glGetInteger64i_v(GL_SHADER_STORAGE_BUFFER_SIZE, 1, &size64);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_EQ(binding64, static_cast<GLint64>(buffer));
|
||||
EXPECT_EQ(start64, static_cast<GLint64>(256));
|
||||
EXPECT_EQ(size64, static_cast<GLint64>(512));
|
||||
|
||||
// An unbound index answers zero rather than erroring or leaking the driver's answer.
|
||||
GLint unbound = -1;
|
||||
glGetIntegeri_v(GL_SHADER_STORAGE_BUFFER_BINDING, 0, &unbound);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_EQ(unbound, 0);
|
||||
|
||||
// THE ARM THAT SEPARATES VERBATIM FROM CLAMPED. GL 4.6 core tables 23.4/23.5 report
|
||||
// the size glBindBufferRange was ASKED for; it does not follow the buffer, so
|
||||
// shrinking the store underneath the binding must not move it. A clamp to the
|
||||
// current storage - which is exactly what both backends' deleted arms did - answers
|
||||
// 128 here, and answers 0 for the bind-then-allocate shape
|
||||
// KHR-GL43.shader_storage_buffer_object.basic-binding uses.
|
||||
glBufferData(GL_SHADER_STORAGE_BUFFER, 128, nullptr, GL_DYNAMIC_DRAW);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
GLint startAfterShrink = -1;
|
||||
GLint sizeAfterShrink = -1;
|
||||
GLint64 sizeAfterShrink64 = -1;
|
||||
glGetIntegeri_v(GL_SHADER_STORAGE_BUFFER_START, 1, &startAfterShrink);
|
||||
glGetIntegeri_v(GL_SHADER_STORAGE_BUFFER_SIZE, 1, &sizeAfterShrink);
|
||||
glGetInteger64i_v(GL_SHADER_STORAGE_BUFFER_SIZE, 1, &sizeAfterShrink64);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_EQ(startAfterShrink, 256)
|
||||
<< "the bound range's start followed the buffer through a re-specification";
|
||||
EXPECT_EQ(sizeAfterShrink, 512)
|
||||
<< "the bound range's size was clamped to the buffer's current 128-byte storage; the range is "
|
||||
"state of the BINDING POINT and is reported verbatim";
|
||||
EXPECT_EQ(sizeAfterShrink64, static_cast<GLint64>(512))
|
||||
<< "the 64-bit width disagreed with the 32-bit one about the same pname";
|
||||
|
||||
glBindBufferBase(GL_SHADER_STORAGE_BUFFER, 1, 0);
|
||||
glDeleteBuffers(1, &buffer);
|
||||
(void)FirstGLError();
|
||||
}
|
||||
|
||||
TEST_F(AdvertisedLimitsScenario, ImageUnitBindingsAreReportedFromTheFrontendState) {
|
||||
GLint maxImageUnits = 0;
|
||||
glGetIntegerv(GL_MAX_IMAGE_UNITS, &maxImageUnits);
|
||||
(void)FirstGLError();
|
||||
if (maxImageUnits < 2) GTEST_SKIP() << "no image units to bind on this lane";
|
||||
|
||||
GLuint texture = 0;
|
||||
glGenTextures(1, &texture);
|
||||
glBindTexture(GL_TEXTURE_2D, texture);
|
||||
glTexStorage2D(GL_TEXTURE_2D, 2, GL_RGBA8, 8, 8);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
|
||||
glBindImageTexture(1, texture, 1, GL_FALSE, 0, GL_READ_ONLY, GL_RGBA8);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
|
||||
struct Expectation {
|
||||
GLenum pname;
|
||||
const char* name;
|
||||
GLint expected;
|
||||
};
|
||||
const Expectation expectations[] = {
|
||||
{GL_IMAGE_BINDING_NAME, "GL_IMAGE_BINDING_NAME", static_cast<GLint>(texture)},
|
||||
{GL_IMAGE_BINDING_LEVEL, "GL_IMAGE_BINDING_LEVEL", 1},
|
||||
{GL_IMAGE_BINDING_LAYERED, "GL_IMAGE_BINDING_LAYERED", GL_FALSE},
|
||||
{GL_IMAGE_BINDING_LAYER, "GL_IMAGE_BINDING_LAYER", 0},
|
||||
{GL_IMAGE_BINDING_ACCESS, "GL_IMAGE_BINDING_ACCESS", GL_READ_ONLY},
|
||||
{GL_IMAGE_BINDING_FORMAT, "GL_IMAGE_BINDING_FORMAT", GL_RGBA8},
|
||||
};
|
||||
for (const Expectation& expectation : expectations) {
|
||||
GLint value = -424242;
|
||||
glGetIntegeri_v(expectation.pname, 1, &value);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << expectation.name;
|
||||
EXPECT_EQ(value, expectation.expected) << expectation.name;
|
||||
|
||||
// Same pname through the wide width - it must not fall through to a driver that
|
||||
// knows nothing about MobileGL's image-unit state.
|
||||
GLint64 wide = -424242;
|
||||
glGetInteger64i_v(expectation.pname, 1, &wide);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << expectation.name << " (64-bit)";
|
||||
EXPECT_EQ(wide, static_cast<GLint64>(expectation.expected)) << expectation.name << " (64-bit)";
|
||||
}
|
||||
|
||||
glBindImageTexture(1, 0, 0, GL_FALSE, 0, GL_READ_ONLY, GL_RGBA8);
|
||||
glDeleteTextures(1, &texture);
|
||||
(void)FirstGLError();
|
||||
}
|
||||
|
||||
// glGetProgramiv(GL_COMPUTE_WORK_GROUP_SIZE) is a LINK ARTIFACT of the program the
|
||||
// application wrote. DirectVulkan used to answer it from its own spirv-reflect cache and
|
||||
// DirectGLES by forwarding to the driver's ESSL program - neither of which the
|
||||
// application ever named - while GL_Program.cpp has always answered it from
|
||||
// ProgramObject::GetComputeLocalSize. This pins the declared local size on both lanes.
|
||||
TEST_F(AdvertisedLimitsScenario, ComputeLocalSizeComesFromTheLinkedProgram) {
|
||||
static const char* kSource = R"(#version 430 core
|
||||
layout(local_size_x = 4, local_size_y = 3, local_size_z = 2) in;
|
||||
layout(std430, binding = 0) buffer Output { uint g_data[]; };
|
||||
void main() { g_data[gl_LocalInvocationIndex] = 1u; }
|
||||
)";
|
||||
const GLuint shader = glCreateShader(GL_COMPUTE_SHADER);
|
||||
glShaderSource(shader, 1, &kSource, nullptr);
|
||||
glCompileShader(shader);
|
||||
GLint compiled = 0;
|
||||
glGetShaderiv(shader, GL_COMPILE_STATUS, &compiled);
|
||||
if (compiled == GL_FALSE) {
|
||||
char log[2048] = {};
|
||||
glGetShaderInfoLog(shader, sizeof(log) - 1, nullptr, log);
|
||||
glDeleteShader(shader);
|
||||
(void)FirstGLError();
|
||||
GTEST_SKIP() << "no compute shader support on this lane: " << log;
|
||||
}
|
||||
const GLuint program = glCreateProgram();
|
||||
glAttachShader(program, shader);
|
||||
glLinkProgram(program);
|
||||
glDeleteShader(shader);
|
||||
GLint linked = 0;
|
||||
glGetProgramiv(program, GL_LINK_STATUS, &linked);
|
||||
if (linked == GL_FALSE) {
|
||||
char log[2048] = {};
|
||||
glGetProgramInfoLog(program, sizeof(log) - 1, nullptr, log);
|
||||
glDeleteProgram(program);
|
||||
(void)FirstGLError();
|
||||
GTEST_SKIP() << "the compute program did not link on this lane: " << log;
|
||||
}
|
||||
|
||||
GLint localSize[3] = {-1, -1, -1};
|
||||
glGetProgramiv(program, GL_COMPUTE_WORK_GROUP_SIZE, localSize);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_EQ(localSize[0], 4);
|
||||
EXPECT_EQ(localSize[1], 3);
|
||||
EXPECT_EQ(localSize[2], 2);
|
||||
|
||||
// A program with no compute stage must answer INVALID_OPERATION, not a stale or
|
||||
// defaulted (1, 1, 1) - the frontend's rule, and the one a backend that answers from
|
||||
// its own reflection cache cannot express.
|
||||
const GLuint empty = glCreateProgram();
|
||||
GLint ignored[3] = {0, 0, 0};
|
||||
glGetProgramiv(empty, GL_COMPUTE_WORK_GROUP_SIZE, ignored);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_INVALID_OPERATION))
|
||||
<< "GL 4.6 core 7.13: the query is only defined for a linked program with a compute shader";
|
||||
|
||||
glDeleteProgram(empty);
|
||||
glDeleteProgram(program);
|
||||
(void)FirstGLError();
|
||||
}
|
||||
|
||||
// THE SIX COMPUTE LIMITS THAT OUTLIVE THE GETTER. GL_MAX_COMPUTE_WORK_GROUP_COUNT and
|
||||
// GL_MAX_COMPUTE_WORK_GROUP_SIZE, three axes each, are the only indexed pnames the
|
||||
// DEVICE answers rather than the frontend (glGetIntegeri_v on Espryt, VkPhysicalDevice-
|
||||
// Limits on Magma), and therefore the only ones that have to cross the MGPipe boundary
|
||||
// once GetIntegeri_v is retired (plan B section 4.4.6 / P0.5). They ride in MGPCaps by
|
||||
// inclusion, as DynamicBackendParameters::MaxComputeWorkGroupCount/Size, filled by both
|
||||
// backends at capability init. This case pins that the caps copy and the live getter
|
||||
// answer are one number - the getter floors the backend's raw answer at the GL 4.3
|
||||
// minimum, so the comparison is against the floored caps value - and pins the
|
||||
// GL-visible half on every lane: answerability, the floors, vector/indexed agreement
|
||||
// and the index bound. On a lane where the caps block is out of reach (Android links
|
||||
// the shipping .so) only the GL-visible half runs.
|
||||
TEST_F(AdvertisedLimitsScenario, ComputeWorkGroupLimitsAreTheCapsBlocksAnswer) {
|
||||
struct Axis {
|
||||
GLenum pname;
|
||||
const char* name;
|
||||
GLint minimum[3]; // GL 4.3 core table 23.60
|
||||
};
|
||||
const Axis axes[] = {
|
||||
{GL_MAX_COMPUTE_WORK_GROUP_COUNT, "GL_MAX_COMPUTE_WORK_GROUP_COUNT", {65535, 65535, 65535}},
|
||||
{GL_MAX_COMPUTE_WORK_GROUP_SIZE, "GL_MAX_COMPUTE_WORK_GROUP_SIZE", {1024, 1024, 64}},
|
||||
};
|
||||
int capsCount[3] = {0, 0, 0};
|
||||
int capsSize[3] = {0, 0, 0};
|
||||
const bool capsVisible = PeekComputeWorkGroupCaps(capsCount, capsSize);
|
||||
|
||||
for (const Axis& axis : axes) {
|
||||
GLint indexed[3] = {-1, -1, -1};
|
||||
for (GLuint i = 0; i < 3; ++i) {
|
||||
glGetIntegeri_v(axis.pname, i, &indexed[i]);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << axis.name << "[" << i << "]";
|
||||
EXPECT_GE(indexed[i], axis.minimum[i])
|
||||
<< axis.name << "[" << i << "] = " << indexed[i]
|
||||
<< " is below the GL 4.3 core table 23.60 minimum " << axis.minimum[i];
|
||||
}
|
||||
GLint vector[3] = {-1, -1, -1};
|
||||
glGetIntegerv(axis.pname, vector);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << axis.name;
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
EXPECT_EQ(vector[i], indexed[i])
|
||||
<< axis.name << "[" << i << "]: the vector query and the indexed query disagree";
|
||||
}
|
||||
GLint outOfRange = -424242;
|
||||
glGetIntegeri_v(axis.pname, 3, &outOfRange);
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_INVALID_VALUE))
|
||||
<< axis.name << "[3]: an index past the three axes is INVALID_VALUE (GL 4.6 core 22.1)";
|
||||
|
||||
if (!capsVisible) continue;
|
||||
const int* capsAxis = axis.pname == GL_MAX_COMPUTE_WORK_GROUP_COUNT ? capsCount : capsSize;
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
EXPECT_EQ(std::max(capsAxis[i], axis.minimum[i]), indexed[i])
|
||||
<< axis.name << "[" << i << "]: MGPCaps carries " << capsAxis[i]
|
||||
<< " but glGetIntegeri_v answers " << indexed[i]
|
||||
<< " - the caps block and the getter path must be one number, because P0.5 retires "
|
||||
"the getter in favour of the caps";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace MGITest
|
||||
|
||||
@@ -25,10 +25,12 @@
|
||||
// be able to turn this into a red.
|
||||
// (b) Forcing the join afterwards produces the right answer for every one of them:
|
||||
// GL_COMPILE_STATUS true, an empty info log, and a program that links.
|
||||
// (c) The extension string matches the configuration. This is the half a recorded
|
||||
// trace can never cover - Iris and Sodium change their submission schedule the
|
||||
// moment they see the string - so it is asserted against a real backend's real
|
||||
// GL_EXTENSIONS, through both glGetString and glGetStringi.
|
||||
// (c) The extension string matches the configuration - where "the configuration" is
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE as this process inherited it, and NOT anything the
|
||||
// implementation says about itself. This is the half a recorded trace can never
|
||||
// cover - Iris and Sodium change their submission schedule the moment they see the
|
||||
// string - so it is asserted against a real backend's real GL_EXTENSIONS, through
|
||||
// both glGetString and glGetStringi.
|
||||
// (d) glMaxShaderCompilerThreadsKHR(0) leaves nothing in flight: every subsequent
|
||||
// GL_COMPLETION_STATUS_KHR reads GL_TRUE immediately, and compilation after it
|
||||
// is synchronous. That is what the extension requires of a zero count.
|
||||
@@ -40,6 +42,27 @@
|
||||
//
|
||||
// Backend selection is the module's usual one process, one backend (MOBILEGL_BACKEND_TYPE),
|
||||
// so this file runs twice per ctest invocation.
|
||||
//
|
||||
// COMPILATION MODE IS PER PROCESS TOO. Every case here needs a particular configuration of
|
||||
// MobileGL's shader compiler, and takes it from the ENVIRONMENT
|
||||
// (MOBILEGL_ASYNC_SHADER_COMPILE, MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS) rather than by
|
||||
// writing MG_Config::Features on the way past. Half of what those variables decide is
|
||||
// latched before the first GL call - the compile pool and its threads, and the advertised
|
||||
// extension list a backend builds once from the configuration in force at its first use -
|
||||
// so an in-process poke could only ever have moved the other half; and on Android it could
|
||||
// move nothing at all, because this module links against the shipping libMobileGL.so, which
|
||||
// exports no such symbol. A case whose process is not in the configuration it needs SKIPS
|
||||
// with that as its reason. CMakeLists.txt registers the extra ctest entries that put a
|
||||
// process into each configuration (AsyncOn., AsyncOff., OptimisticShaderStatus.), so one
|
||||
// ctest run still covers both sides of every switch. Run straight from a shell with nothing
|
||||
// set - the on-device shape - the ambient configuration runs and the rest skip cleanly.
|
||||
//
|
||||
// WITHIN one process, "compiled on a worker" versus "compiled on this thread" is switched
|
||||
// through glMaxShaderCompilerThreadsKHR, the extension's own entry point: a zero count joins
|
||||
// everything outstanding and compiles inline from then on, any nonzero count lifts that
|
||||
// again, and 0xFFFFFFFF asks for the implementation maximum (GL_Program.cpp,
|
||||
// MaxShaderCompilerThreadsKHR_State). Doing it through the public call rather than the
|
||||
// feature table means the switching is itself part of what these cases exercise.
|
||||
|
||||
#include <string>
|
||||
#include <vector>
|
||||
@@ -47,9 +70,6 @@
|
||||
#include "../Harness/HeadlessGL.h"
|
||||
#include "../Harness/ScenarioFixture.h"
|
||||
|
||||
#include "Config.h"
|
||||
#include "MG_Util/Async/ShaderCompilePool.h"
|
||||
|
||||
#ifdef GLAPI
|
||||
#undef GLAPI
|
||||
#endif
|
||||
@@ -76,8 +96,6 @@ extern "C" void glMaxShaderCompilerThreadsKHR(GLuint count);
|
||||
namespace MGITest {
|
||||
namespace {
|
||||
|
||||
using MobileGL::MG_Config::QuirkOverride;
|
||||
|
||||
// Same shape as the other scenarios: a two-attribute pass-through, so the only
|
||||
// thing that can differ between the two compilation modes is the compilation.
|
||||
constexpr const char* kVertexSource = R"(#version 330 core
|
||||
@@ -139,50 +157,40 @@ void main() {
|
||||
return source;
|
||||
}
|
||||
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE decides the ambient mode; a scenario that wants
|
||||
// the other one says so here and gets the ambient one back on scope exit. Forcing
|
||||
// it in-process is what lets ONE ctest run compare the two modes against each
|
||||
// other - the whole point of (e).
|
||||
class AsyncModeScope {
|
||||
public:
|
||||
explicit AsyncModeScope(bool async) : m_saved(MobileGL::MG_Config::Features.AsyncShaderCompile) {
|
||||
MobileGL::MG_Config::Features.AsyncShaderCompile =
|
||||
async ? QuirkOverride::ForceOn : QuirkOverride::ForceOff;
|
||||
// Whether this context advertises GL_KHR_parallel_shader_compile, which is exactly
|
||||
// "MobileGL is configured to compile asynchronously" as an application can see it:
|
||||
// the backends gate the string on AsyncShaderCompileEnabled() and on nothing else
|
||||
// (BackendObject_DirectGLES.cpp / BackendObject_DirectVulkan.cpp), and the string
|
||||
// is the only way MobileGL ever tells anyone. A case that needs asynchronous
|
||||
// compilation checks for it the way an application would, and skips without it.
|
||||
//
|
||||
// The INDEXED form, because that is the one a core-profile application reads.
|
||||
bool HasParallelShaderCompile() {
|
||||
GLint count = 0;
|
||||
glGetIntegerv(GL_NUM_EXTENSIONS, &count);
|
||||
for (GLint i = 0; i < count; ++i) {
|
||||
const char* name = reinterpret_cast<const char*>(glGetStringi(GL_EXTENSIONS, GLuint(i)));
|
||||
if (name != nullptr && std::string(name) == "GL_KHR_parallel_shader_compile") return true;
|
||||
}
|
||||
~AsyncModeScope() { MobileGL::MG_Config::Features.AsyncShaderCompile = m_saved; }
|
||||
AsyncModeScope(const AsyncModeScope&) = delete;
|
||||
AsyncModeScope& operator=(const AsyncModeScope&) = delete;
|
||||
return false;
|
||||
}
|
||||
|
||||
private:
|
||||
const QuirkOverride m_saved;
|
||||
};
|
||||
|
||||
// MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS, forced in-process for the same reason
|
||||
// as AsyncModeScope: one ctest run asserts the quirk against the ambient default.
|
||||
class OptimisticStatusScope {
|
||||
public:
|
||||
explicit OptimisticStatusScope(const QuirkOverride mode)
|
||||
: m_saved(MobileGL::MG_Config::Features.AsyncOptimisticShaderStatus) {
|
||||
MobileGL::MG_Config::Features.AsyncOptimisticShaderStatus = mode;
|
||||
}
|
||||
~OptimisticStatusScope() { MobileGL::MG_Config::Features.AsyncOptimisticShaderStatus = m_saved; }
|
||||
OptimisticStatusScope(const OptimisticStatusScope&) = delete;
|
||||
OptimisticStatusScope& operator=(const OptimisticStatusScope&) = delete;
|
||||
|
||||
private:
|
||||
const QuirkOverride m_saved;
|
||||
};
|
||||
|
||||
// glMaxShaderCompilerThreadsKHR writes process-wide state; a scenario that calls
|
||||
// it has to put the pool back or it changes how every scenario after it compiles.
|
||||
// glMaxShaderCompilerThreadsKHR writes process-wide state; a scenario that calls it
|
||||
// has to put the pool back or it changes how every scenario after it compiles.
|
||||
//
|
||||
// The restore is the extension's own "implementation maximum" spelling rather than a
|
||||
// hand-rolled poke at the pool. glMaxShaderCompilerThreadsKHR(0xFFFFFFFF) is defined
|
||||
// (GL_Program.cpp, MaxShaderCompilerThreadsKHR_State) as precisely the two steps this
|
||||
// used to perform through internal entry points - concurrency := the pool's full
|
||||
// thread count, then lift any suspension a zero count had armed - in the safer order,
|
||||
// since it raises the budget before re-admitting work rather than after. Going through
|
||||
// the public call also puts the restore path itself under test, and it is the only
|
||||
// spelling available on Android, where this module links the shipping shared library
|
||||
// and can reach nothing but the GL entry points.
|
||||
class CompilerThreadScope {
|
||||
public:
|
||||
CompilerThreadScope() = default;
|
||||
~CompilerThreadScope() {
|
||||
MobileGL::MG_Util::Async::SetAsyncShaderCompileSuspended(false);
|
||||
auto& pool = MobileGL::MG_Util::Async::ShaderCompilePool::Get();
|
||||
pool.SetMaxConcurrency(pool.GetThreadCount());
|
||||
}
|
||||
~CompilerThreadScope() { glMaxShaderCompilerThreadsKHR(0xFFFFFFFFu); }
|
||||
CompilerThreadScope(const CompilerThreadScope&) = delete;
|
||||
CompilerThreadScope& operator=(const CompilerThreadScope&) = delete;
|
||||
};
|
||||
@@ -293,7 +301,12 @@ void main() {
|
||||
// interesting for shaders that (a) proved were genuinely still outstanding.
|
||||
TEST_F(AsyncCompileScenario, CompletionStatusPollingThenForcedJoin) {
|
||||
if (!Ready()) return;
|
||||
const AsyncModeScope async(true);
|
||||
if (!HasParallelShaderCompile()) {
|
||||
GTEST_SKIP() << "this process is configured to compile inline "
|
||||
"(GL_KHR_parallel_shader_compile is not advertised), so no compile can be "
|
||||
"outstanding; the AsyncOn. ctest entries run this case with "
|
||||
"MOBILEGL_ASYNC_SHADER_COMPILE=1";
|
||||
}
|
||||
const CompilerThreadScope threads;
|
||||
// One worker, so the queue behind it is what the poll observes.
|
||||
glMaxShaderCompilerThreadsKHR(1);
|
||||
@@ -341,14 +354,33 @@ void main() {
|
||||
}
|
||||
|
||||
// ---- (c) ------------------------------------------------------------------
|
||||
// The extension string, read from a real backend that really brought a driver
|
||||
// up. No mode forcing here: a backend builds its advertised list once, from the
|
||||
// configuration in force at its first use, so the meaningful assertion is
|
||||
// against the AMBIENT configuration - which is exactly what makes this case
|
||||
// worth running in both of the suite's flag states.
|
||||
// The extension string, read from a real backend that really brought a driver up.
|
||||
//
|
||||
// The expectation comes from the ENVIRONMENT, never from the implementation. This
|
||||
// case used to derive it by calling AsyncShaderCompileEnabled() - which is the same
|
||||
// function the backends gate the string on, so the two halves could only ever agree
|
||||
// and the case would have passed however wrong both of them were. Asserting an
|
||||
// implementation against itself pins nothing.
|
||||
//
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE is the whole input: the process inherited it before
|
||||
// any GL call, a backend builds its advertised list once from the configuration in
|
||||
// force at first use, and nothing in this process can move it afterwards. So reading
|
||||
// the variable IS reading the configuration, independently. With the variable unset
|
||||
// the configuration in force is MobileGL's built-in default, which only the
|
||||
// implementation knows - there is nothing independent left to compare against, and
|
||||
// this case says so rather than inventing an expectation. The AsyncOn. and AsyncOff.
|
||||
// ctest entries pin the variable to each of its two values, so one ctest run still
|
||||
// asserts both the advertised and the withdrawn side.
|
||||
TEST_F(AsyncCompileScenario, ExtensionStringMatchesTheConfiguration) {
|
||||
if (!Ready()) return;
|
||||
const bool expected = MobileGL::MG_Util::Async::AsyncShaderCompileEnabled();
|
||||
const AmbientQuirk configured = AmbientQuirkFromEnvironment("MOBILEGL_ASYNC_SHADER_COMPILE");
|
||||
if (configured == AmbientQuirk::Auto) {
|
||||
GTEST_SKIP() << "MOBILEGL_ASYNC_SHADER_COMPILE is unset, so the configuration in force is "
|
||||
"MobileGL's built-in default and the only way to learn it would be to ask "
|
||||
"the implementation this case exists to check; the AsyncOn. and AsyncOff. "
|
||||
"ctest entries run it with the variable pinned to each of its two values";
|
||||
}
|
||||
const bool expected = configured == AmbientQuirk::On;
|
||||
|
||||
const char* extensions = reinterpret_cast<const char*>(glGetString(GL_EXTENSIONS));
|
||||
ASSERT_NE(extensions, nullptr);
|
||||
@@ -385,7 +417,12 @@ void main() {
|
||||
// A zero count must leave nothing in flight and keep it that way.
|
||||
TEST_F(AsyncCompileScenario, ZeroCompilerThreadsSettlesEverythingImmediately) {
|
||||
if (!Ready()) return;
|
||||
const AsyncModeScope async(true);
|
||||
if (!HasParallelShaderCompile()) {
|
||||
GTEST_SKIP() << "this process is configured to compile inline "
|
||||
"(GL_KHR_parallel_shader_compile is not advertised), so a zero count has "
|
||||
"nothing to settle; the AsyncOn. ctest entries run this case with "
|
||||
"MOBILEGL_ASYNC_SHADER_COMPILE=1";
|
||||
}
|
||||
const CompilerThreadScope threads;
|
||||
glMaxShaderCompilerThreadsKHR(1);
|
||||
|
||||
@@ -417,12 +454,28 @@ void main() {
|
||||
// Compared through the DEFAULT framebuffer deliberately: that is where the
|
||||
// backend's orientation and present path live, so the comparison covers the
|
||||
// whole pipeline rather than the reflection tables alone.
|
||||
//
|
||||
// The two modes are selected through glMaxShaderCompilerThreadsKHR, the extension's
|
||||
// own entry point, rather than through the feature table: a zero count joins
|
||||
// everything outstanding and makes every later glCompileShader/glLinkProgram run its
|
||||
// body on the calling thread, and 0xFFFFFFFF lifts that again with the pool at its
|
||||
// full thread count (GL_Program.cpp, MaxShaderCompilerThreadsKHR_State; the compile
|
||||
// and link paths both gate on AsyncShaderCompileActive(), which is what the zero
|
||||
// count switches). So this is still one process comparing worker-built artifacts
|
||||
// against inline-built ones - just asked for the way an application asks.
|
||||
TEST_F(AsyncCompileScenario, AsyncAndSyncProgramsRenderIdenticalFrames) {
|
||||
if (!Ready()) return;
|
||||
if (!HasParallelShaderCompile()) {
|
||||
GTEST_SKIP() << "this process is configured to compile inline "
|
||||
"(GL_KHR_parallel_shader_compile is not advertised), so both halves would "
|
||||
"be the same inline build and the comparison would be vacuous; the "
|
||||
"AsyncOn. ctest entries run this case with MOBILEGL_ASYNC_SHADER_COMPILE=1";
|
||||
}
|
||||
const CompilerThreadScope threads;
|
||||
|
||||
Image asyncImage;
|
||||
{
|
||||
const AsyncModeScope async(true);
|
||||
glMaxShaderCompilerThreadsKHR(0xFFFFFFFFu);
|
||||
const GLuint program = BuildProgram();
|
||||
ASSERT_NE(program, 0u);
|
||||
asyncImage = DrawFrameWith(program);
|
||||
@@ -431,7 +484,7 @@ void main() {
|
||||
|
||||
Image syncImage;
|
||||
{
|
||||
const AsyncModeScope async(false);
|
||||
glMaxShaderCompilerThreadsKHR(0);
|
||||
const GLuint program = BuildProgram();
|
||||
ASSERT_NE(program, 0u);
|
||||
syncImage = DrawFrameWith(program);
|
||||
@@ -456,11 +509,16 @@ void main() {
|
||||
// candidate) shows up here and not in the single-program case above.
|
||||
TEST_F(AsyncCompileScenario, ABatchOfAsyncProgramsAllRenderCorrectly) {
|
||||
if (!Ready()) return;
|
||||
if (!HasParallelShaderCompile()) {
|
||||
GTEST_SKIP() << "this process is configured to compile inline "
|
||||
"(GL_KHR_parallel_shader_compile is not advertised), so nothing would be "
|
||||
"built on a worker and there is no per-worker state to leak; the AsyncOn. "
|
||||
"ctest entries run this case with MOBILEGL_ASYNC_SHADER_COMPILE=1";
|
||||
}
|
||||
constexpr int kPrograms = 12;
|
||||
|
||||
std::vector<GLuint> programs;
|
||||
{
|
||||
const AsyncModeScope async(true);
|
||||
const CompilerThreadScope threads;
|
||||
glMaxShaderCompilerThreadsKHR(1);
|
||||
// Everything enqueued before anything is read: the only shape in which
|
||||
@@ -489,6 +547,21 @@ void main() {
|
||||
// then mis-renders - shows up here as a wrong quadrant signature.
|
||||
TEST_F(AsyncCompileScenario, IrisShapedTwoPhaseBatchRendersCorrectly) {
|
||||
if (!Ready()) return;
|
||||
// The quirk is off by default and never advertised, so unlike the cases above
|
||||
// there is no GL observable that says whether it is in force - only the variable
|
||||
// that put it there. It also has to be set BEFORE this process started for the
|
||||
// shape to be the real one: the optimistic answer is latched per compile, and a
|
||||
// quirk switched on mid-process would only cover the compiles after it.
|
||||
if (AmbientQuirkFromEnvironment("MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS") != AmbientQuirk::On) {
|
||||
GTEST_SKIP() << "this case is the optimistic-status quirk's end-to-end shape and needs it on "
|
||||
"for the whole process; the OptimisticShaderStatus. ctest entries run it with "
|
||||
"MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS=1";
|
||||
}
|
||||
if (!HasParallelShaderCompile()) {
|
||||
GTEST_SKIP() << "the optimistic status only ever applies to a compile that is still in flight "
|
||||
"(OptimisticShaderStatusActive() requires AsyncShaderCompileActive()), and "
|
||||
"this process is configured to compile inline";
|
||||
}
|
||||
constexpr int kPrograms = 12;
|
||||
|
||||
// Distinct per program (so neither the source memo nor the adoption map turns
|
||||
@@ -508,8 +581,6 @@ void main() {
|
||||
|
||||
std::vector<GLuint> programs;
|
||||
{
|
||||
const AsyncModeScope async(true);
|
||||
const OptimisticStatusScope quirk(QuirkOverride::ForceOn);
|
||||
const CompilerThreadScope threads;
|
||||
glMaxShaderCompilerThreadsKHR(1);
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user