mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-07 19:58:32 +09:00
Compare commits
423
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
12df061e0b | ||
|
|
d83a48da5c | ||
|
|
b3100b0de5 | ||
|
|
1cde801a01 | ||
|
|
52c050131e | ||
|
|
ebb8a4cebf | ||
|
|
5398fb4289 | ||
|
|
7f2ca68615 | ||
|
|
b12ef4d717 | ||
|
|
724755d9df | ||
|
|
59f7059bf4 | ||
|
|
4446c861be | ||
|
|
602da1d131 | ||
|
|
06bbaf32b1 | ||
|
|
9bcf0a15a0 | ||
|
|
26e5a946ac | ||
|
|
dc543fa905 | ||
|
|
f559d68728 | ||
|
|
48968a663f | ||
|
|
0fdcb5d6c4 | ||
|
|
6b25e7a7e3 | ||
|
|
50d260c840 | ||
|
|
6a8bf4c03c | ||
|
|
01098e9dd7 | ||
|
|
82244e9048 | ||
|
|
7bd2f08313 | ||
|
|
121b99f8c3 | ||
|
|
e8d79344d6 | ||
|
|
88be35c9ba | ||
|
|
60808b6cf2 | ||
|
|
eb2f14e55a | ||
|
|
0eb5d54bb8 | ||
|
|
6a2e9dc791 | ||
|
|
d5aceebd7b | ||
|
|
473d9951b7 | ||
|
|
13783e3aec | ||
|
|
6bb844b1c1 | ||
|
|
6162603072 | ||
|
|
666f150202 | ||
|
|
2f62970dd5 | ||
|
|
dcb568d445 | ||
|
|
a5f36c8f8d | ||
|
|
1ebe9d11c5 | ||
|
|
a8bb63950d | ||
|
|
7d68a17774 | ||
|
|
415645ccdd | ||
|
|
e6d03eb2a1 | ||
|
|
5e29e7d266 | ||
|
|
7eac33d17b | ||
|
|
8f19ce6fa7 | ||
|
|
d0f7fb99db | ||
|
|
d4247db6c3 | ||
|
|
e4f41e0fd3 | ||
|
|
9cf340cbef | ||
|
|
348a30a816 | ||
|
|
b5e0ada97e | ||
|
|
cb27ac7761 | ||
|
|
38c56a3d38 | ||
|
|
908172ba0f | ||
|
|
e18bac8cb2 | ||
|
|
7b0f443d3a | ||
|
|
f1b4a5e07f | ||
|
|
f3cd4091bf | ||
|
|
529d26f38f | ||
|
|
9bd125aeec | ||
|
|
ece9491d4b | ||
|
|
51b4abd801 | ||
|
|
cbb616093b | ||
|
|
e5846569ca | ||
|
|
194c2f189b | ||
|
|
7de7cfc6eb | ||
|
|
03e69fc9ef | ||
|
|
ee74c8ea3a | ||
|
|
54bbe805e5 | ||
|
|
6e2a3b3496 | ||
|
|
f5a0779385 | ||
|
|
86fdc68efa | ||
|
|
1185265e22 | ||
|
|
77c05b151a | ||
|
|
fdbe0b3117 | ||
|
|
9c9739e1c3 | ||
|
|
87548ae78a | ||
|
|
2a902ff58c | ||
|
|
a79eadd724 | ||
|
|
518e9c7796 | ||
|
|
a4c11f2603 | ||
|
|
37a656dedc | ||
|
|
3cc6b88767 | ||
|
|
b32c35a113 | ||
|
|
8587b83be3 | ||
|
|
0cef345d61 | ||
|
|
31367de628 | ||
|
|
bc4b62026a | ||
|
|
50da7de737 | ||
|
|
d9def5c1bb | ||
|
|
21a4c8aa95 | ||
|
|
6317066add | ||
|
|
02b59bef80 | ||
|
|
668f3e90c9 | ||
|
|
07d6277f87 | ||
|
|
b164692387 | ||
|
|
80ea44573e | ||
|
|
2a7d6f2e16 | ||
|
|
3a12f6d4f3 | ||
|
|
36b9d26b9d | ||
|
|
4b41f01b68 | ||
|
|
e81e938bb8 | ||
|
|
5fa849674e | ||
|
|
aed10f65a6 | ||
|
|
a687fc4577 | ||
|
|
ef66aea73b | ||
|
|
c129cdec2d | ||
|
|
7a7340ebe2 | ||
|
|
325ba07776 | ||
|
|
7aa91e8024 | ||
|
|
1b5a39473e | ||
|
|
0e5f591cfa | ||
|
|
79336c5ccc | ||
|
|
be45dbcf54 | ||
|
|
f7dfa01c18 | ||
|
|
50efa4410a | ||
|
|
bea3086b41 | ||
|
|
8ae93c837d | ||
|
|
4154f2e941 | ||
|
|
cd07d42a47 | ||
|
|
ae0373eb48 | ||
|
|
7480bf4490 | ||
|
|
54b206d90c | ||
|
|
5daf7bf093 | ||
|
|
a8228ca287 | ||
|
|
8b827bd2ce | ||
|
|
6aa161fee7 | ||
|
|
48a70fea81 | ||
|
|
6ea4f32635 | ||
|
|
dc1fffb041 | ||
|
|
d24d5b5ccd | ||
|
|
51883cf1a3 | ||
|
|
6dfadeb7d2 | ||
|
|
4fc3531d0d | ||
|
|
9bde0e500f | ||
|
|
3477d87b50 | ||
|
|
c2a081fa75 | ||
|
|
685fd750c9 | ||
|
|
02c9b8a32d | ||
|
|
26f02567d7 | ||
|
|
a3dbe234d7 | ||
|
|
de8e7a4606 | ||
|
|
31a5da6190 | ||
|
|
8899f065f4 | ||
|
|
a991f63899 | ||
|
|
3ff9cfe5c2 | ||
|
|
6359fba455 | ||
|
|
872876961d | ||
|
|
e2923a239f | ||
|
|
1740a8a41a | ||
|
|
6b1d89f279 | ||
|
|
01fbe0b4b0 | ||
|
|
cb155c5b94 | ||
|
|
04a06438c5 | ||
|
|
db00774224 | ||
|
|
f378c1a064 | ||
|
|
421ccd08c6 | ||
|
|
039af520bf | ||
|
|
cdba7bed2e | ||
|
|
1eeeb44d94 | ||
|
|
fa2e15c27e | ||
|
|
14744f117c | ||
|
|
8329ab4264 | ||
|
|
ee98c453ed | ||
|
|
f88322ce84 | ||
|
|
93f1106ba4 | ||
|
|
31b5b563d6 | ||
|
|
a9fb7ef0af | ||
|
|
6159166d38 | ||
|
|
c6d1b29407 | ||
|
|
5fecfa42f6 | ||
|
|
d48e5d0053 | ||
|
|
0995dfea35 | ||
|
|
7a0182b58f | ||
|
|
0f523db14d | ||
|
|
442cec1a15 | ||
|
|
246a438138 | ||
|
|
042c61fb75 | ||
|
|
a8bebe1a3c | ||
|
|
85cd6913b3 | ||
|
|
afebf38e90 | ||
|
|
898c39f1de | ||
|
|
b1774e80be | ||
|
|
a9b4c47fea | ||
|
|
1c0be3e715 | ||
|
|
27ec3d3438 | ||
|
|
52718ecf84 | ||
|
|
baeb2fa1bc | ||
|
|
54a88ef1e2 | ||
|
|
ee124018a2 | ||
|
|
d7ce0c48ef | ||
|
|
0b3101bf6b | ||
|
|
a4fda520ed | ||
|
|
f0fd6407ae | ||
|
|
bde14cae29 | ||
|
|
f2f6430e34 | ||
|
|
15e36ad1e9 | ||
|
|
7940a09491 | ||
|
|
1e8d4661e6 | ||
|
|
916702629e | ||
|
|
9b37c77ae2 | ||
|
|
085eb5835b | ||
|
|
56377d2025 | ||
|
|
7d2c16a90e | ||
|
|
261cfd1591 | ||
|
|
bbc7b9ca84 | ||
|
|
c1b3b16cab | ||
|
|
f17cb23ea3 | ||
|
|
f297af7d2b | ||
|
|
d9abf1c2c1 | ||
|
|
392736fb6b | ||
|
|
0944925679 | ||
|
|
eadf7bc474 | ||
|
|
00a326ef78 | ||
|
|
c09045fe59 | ||
|
|
5bd8ef01e5 | ||
|
|
3181ed2c5a | ||
|
|
6aed3b08f3 | ||
|
|
281467a345 | ||
|
|
c7e36986e7 | ||
|
|
d8576a2ed3 | ||
|
|
2b6c2b561c | ||
|
|
12c94111b5 | ||
|
|
7769156cfc | ||
|
|
0ecfdff4e7 | ||
|
|
6df5a6137f | ||
|
|
b3794f4e6a | ||
|
|
14d3901d30 | ||
|
|
d4766513e4 | ||
|
|
72dc7aa6aa | ||
|
|
9d1b280375 | ||
|
|
8acd885594 | ||
|
|
10ff5e2b18 | ||
|
|
a6e52476f3 | ||
|
|
0deff52a1b | ||
|
|
50fefca959 | ||
|
|
42ad62b54c | ||
|
|
f41403e227 | ||
|
|
5fbb17f6b9 | ||
|
|
92d8f7269b | ||
|
|
822e405c77 | ||
|
|
b8233f9c4e | ||
|
|
91475a7b6f | ||
|
|
cee17025a0 | ||
|
|
9642ae4d20 | ||
|
|
595d140036 | ||
|
|
b62d1f2078 | ||
|
|
5e82ff968a | ||
|
|
6a80a82dd3 | ||
|
|
f9182a5ca3 | ||
|
|
f37b511fca | ||
|
|
38027d21f8 | ||
|
|
dd2a62228f | ||
|
|
373aa44dd7 | ||
|
|
96646df12e | ||
|
|
f20b20e643 | ||
|
|
2587814970 | ||
|
|
43398e33e8 | ||
|
|
b7557d6615 | ||
|
|
bce34d7fac | ||
|
|
f91857266f | ||
|
|
49cb1be0fd | ||
|
|
a51c68bb2c | ||
|
|
1c723a6cfc | ||
|
|
44805bfa07 | ||
|
|
257fcbfd0b | ||
|
|
2b46a3db96 | ||
|
|
0b36621069 | ||
|
|
9ee2e0a1db | ||
|
|
6b6623ae72 | ||
|
|
a6e029734b | ||
|
|
e7d6bfddac | ||
|
|
f2c879528f | ||
|
|
eaba4ac1dc | ||
|
|
ed6578954e | ||
|
|
e005c8b6cb | ||
|
|
535b5e3095 | ||
|
|
1b05a84928 | ||
|
|
7ccb762936 | ||
|
|
442e7eec1c | ||
|
|
2787d15706 | ||
|
|
74ce58a6c7 | ||
|
|
bb122ebd4f | ||
|
|
9b0ed5b3af | ||
|
|
0e31c1481b | ||
|
|
811f32760e | ||
|
|
01381a0404 | ||
|
|
0f02b0fdb1 | ||
|
|
3be02abf47 | ||
|
|
f91d6b676c | ||
|
|
1ebf191f94 | ||
|
|
ccad803023 | ||
|
|
21159caf31 | ||
|
|
ea4819a21d | ||
|
|
6b882b3ccf | ||
|
|
96bd36c50b | ||
|
|
46fbd837b3 | ||
|
|
796a57a115 | ||
|
|
62a2dae5ba | ||
|
|
532836c058 | ||
|
|
2fced2241b | ||
|
|
21b5fc2d92 | ||
|
|
1f753ab5fa | ||
|
|
3ed9501be5 | ||
|
|
7311251f30 | ||
|
|
7625cf450d | ||
|
|
450eb209b6 | ||
|
|
8c5c39b3c3 | ||
|
|
64a0ea397c | ||
|
|
205d837942 | ||
|
|
b5e9339c66 | ||
|
|
e310e3e9ff | ||
|
|
a0bf4a83bc | ||
|
|
97facf777b | ||
|
|
5cfbb716c0 | ||
|
|
64c3411d70 | ||
|
|
f3a0d9e0a3 | ||
|
|
068786e812 | ||
|
|
8e7cc62c24 | ||
|
|
d2a36d65a3 | ||
|
|
a020de76e3 | ||
|
|
574634adfa | ||
|
|
e71d715e1a | ||
|
|
7b946fd527 | ||
|
|
8f3ce5f5b7 | ||
|
|
21ec744ef2 | ||
|
|
faa7b17da3 | ||
|
|
f6849fc0b3 | ||
|
|
4d1d4f6225 | ||
|
|
cef81df73f | ||
|
|
c4e6ea1f23 | ||
|
|
b4e07ce651 | ||
|
|
ff324057ad | ||
|
|
ef4c6dbe0a | ||
|
|
19fc7346c5 | ||
|
|
bcb0e894ef | ||
|
|
794c10e56c | ||
|
|
28390667d7 | ||
|
|
90dd9bec77 | ||
|
|
7994ca31d3 | ||
|
|
5705e05156 | ||
|
|
ab62f81545 | ||
|
|
f6cf04d6d7 | ||
|
|
38eb9589f9 | ||
|
|
99ebf67a3d | ||
|
|
2292e99476 | ||
|
|
4c5afecc71 | ||
|
|
1c5744f2be | ||
|
|
22859b0958 | ||
|
|
7aa958fbc9 | ||
|
|
a4bd4e04a1 | ||
|
|
09459edb6b | ||
|
|
8af6ebc174 | ||
|
|
577cd8c670 | ||
|
|
5267243404 | ||
|
|
33c2715912 | ||
|
|
3068cdadf8 | ||
|
|
6dd0201bf2 | ||
|
|
05bef7118b | ||
|
|
94e75fef79 | ||
|
|
43bcd03dca | ||
|
|
2ce0595fab | ||
|
|
ba9af18033 | ||
|
|
f7d63f88fa | ||
|
|
6cf5a7744e | ||
|
|
ca3d24f5ea | ||
|
|
cc34d34706 | ||
|
|
757b31592d | ||
|
|
dbae4eda10 | ||
|
|
fa5ff5d168 | ||
|
|
994ae372f8 | ||
|
|
7ba012adf9 | ||
|
|
b1fdffd767 | ||
|
|
18c17ae5ca | ||
|
|
16c010985f | ||
|
|
6f64ec0f51 | ||
|
|
5d47698349 | ||
|
|
7b593e39ef | ||
|
|
d83b4dbbb5 | ||
|
|
964a7fcc92 | ||
|
|
5a7bd9942d | ||
|
|
21a43bf6a4 | ||
|
|
5b6dec2d81 | ||
|
|
543c29bf86 | ||
|
|
ef562ee9b5 | ||
|
|
fa0f6693d0 | ||
|
|
a6c362c6ce | ||
|
|
921504eccf | ||
|
|
7c5fc03b26 | ||
|
|
ed29e63543 | ||
|
|
5c8a9c41d6 | ||
|
|
efa0345c36 | ||
|
|
ce0f18969c | ||
|
|
7ce0966e7d | ||
|
|
1c6ca2753f | ||
|
|
61b0532865 | ||
|
|
f5b8a505ed | ||
|
|
8371365db5 | ||
|
|
b219992ee3 | ||
|
|
94233ef928 | ||
|
|
0827d7a539 | ||
|
|
5248b8b746 | ||
|
|
d868e1c476 | ||
|
|
c3412ca394 | ||
|
|
5ccaff37af | ||
|
|
71e29f9d58 | ||
|
|
8ad07c222c | ||
|
|
5722094d6f | ||
|
|
4831387cf0 | ||
|
|
d03b72267a | ||
|
|
847ec74f48 | ||
|
|
e02e5caa17 | ||
|
|
1958934594 | ||
|
|
dec0c5eaff | ||
|
|
b6a44cd1e2 | ||
|
|
85f45d0e44 | ||
|
|
404236d337 | ||
|
|
6ea948779e |
@@ -1,6 +1,10 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=trace-fixture-lib.sh
|
||||
. "${script_dir}/trace-fixture-lib.sh"
|
||||
|
||||
if [ "$#" -lt 1 ] || [ "$#" -gt 2 ]; then
|
||||
echo "usage: $0 <trace-case> [fixture-dir]" >&2
|
||||
exit 2
|
||||
@@ -62,57 +66,6 @@ if [ "${case_name}" = "OpenRA" ]; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
get_lfs_metadata() {
|
||||
local file="$1"
|
||||
local pointer
|
||||
local expected_oid
|
||||
local expected_size
|
||||
|
||||
if ! pointer="$(git show "HEAD:${file}" 2>/dev/null)"; then
|
||||
echo "failed to read tracked fixture metadata: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
if ! grep -q '^version https://git-lfs.github.com/spec/v1$' <<< "${pointer}"; then
|
||||
echo "tracked fixture is not a Git LFS pointer: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
expected_oid="$(awk '$1 == "oid" && $2 ~ /^sha256:/ { sub(/^sha256:/, "", $2); print $2 }' <<< "${pointer}")"
|
||||
expected_size="$(awk '$1 == "size" { print $2 }' <<< "${pointer}")"
|
||||
if ! [[ "${expected_oid}" =~ ^[0-9a-f]{64}$ ]] || ! [[ "${expected_size}" =~ ^[0-9]+$ ]]; then
|
||||
echo "invalid Git LFS pointer metadata: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
printf '%s %s\n' "${expected_oid}" "${expected_size}"
|
||||
}
|
||||
|
||||
verify_fixture_file() {
|
||||
local downloaded_file="$1"
|
||||
local display_name="$2"
|
||||
local expected_oid="$3"
|
||||
local expected_size="$4"
|
||||
local actual_oid
|
||||
local actual_size
|
||||
|
||||
if [ ! -f "${downloaded_file}" ]; then
|
||||
echo "fixture file is missing: ${display_name}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
actual_size="$(wc -c < "${downloaded_file}" | tr -d '[:space:]')"
|
||||
if [ "${actual_size}" != "${expected_size}" ]; then
|
||||
echo "fixture size mismatch for ${display_name}: expected ${expected_size}, got ${actual_size}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
actual_oid="$(sha256sum "${downloaded_file}" | awk '{ print $1 }')"
|
||||
if [ "${actual_oid}" != "${expected_oid}" ]; then
|
||||
echo "fixture SHA-256 mismatch for ${display_name}: expected ${expected_oid}, got ${actual_oid}" >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
fetch_file_from_mirror() {
|
||||
local file="$1"
|
||||
local url="$2"
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
#!/usr/bin/env bash
|
||||
# Cache-side helper for trace fixtures.
|
||||
#
|
||||
# key <case> [fixture-dir] derive the actions/cache key and path list
|
||||
# verify <case> [fixture-dir] check restored fixtures against their pointers
|
||||
# reset <case> [fixture-dir] drop restored fixtures, leaving the pointers
|
||||
#
|
||||
# The cache key is content-addressed on the Git LFS pointer oids tracked at
|
||||
# HEAD, which are readable from a plain checkout without smudging. Fixture
|
||||
# content therefore maps 1:1 onto a key: unchanged content hits, changed
|
||||
# content is a new key and thus a miss, and the download path handles it. The
|
||||
# key deliberately carries no restore-keys prefix in the workflow - a fixture
|
||||
# that does not match the pointer exactly must never be restored.
|
||||
set -euo pipefail
|
||||
|
||||
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=trace-fixture-lib.sh
|
||||
. "${script_dir}/trace-fixture-lib.sh"
|
||||
|
||||
# Bump when the key derivation changes in a way that must invalidate old
|
||||
# entries; the content digest alone would not notice a format change.
|
||||
key_schema="v1"
|
||||
|
||||
if [ "$#" -lt 2 ] || [ "$#" -gt 3 ]; then
|
||||
echo "usage: $0 <key|verify|reset> <trace-case> [fixture-dir]" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
command_name="$1"
|
||||
case_name="$2"
|
||||
fixture_dir="${3:-tools/trace_replay/fixtures}"
|
||||
python_bin="${PYTHON:-python3}"
|
||||
|
||||
if ! command -v "${python_bin}" >/dev/null 2>&1 && command -v python >/dev/null 2>&1; then
|
||||
python_bin=python
|
||||
fi
|
||||
|
||||
mapfile -t files < <(trace_fixture_files "${case_name}" "${fixture_dir}" "${python_bin}")
|
||||
if [ "${#files[@]}" -eq 0 ]; then
|
||||
echo "no fixture files declared for trace case: ${case_name}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Writes "name=value" to $GITHUB_OUTPUT when running under Actions, and to
|
||||
# stdout otherwise so the script stays runnable (and testable) off-CI.
|
||||
emit_output() {
|
||||
local name="$1"
|
||||
local value="$2"
|
||||
if [ -n "${GITHUB_OUTPUT:-}" ]; then
|
||||
if [[ "${value}" == *$'\n'* ]]; then
|
||||
local delimiter="ghadelim_$(date +%s%N)_$$"
|
||||
{
|
||||
printf '%s<<%s\n' "${name}" "${delimiter}"
|
||||
printf '%s\n' "${value}"
|
||||
printf '%s\n' "${delimiter}"
|
||||
} >> "${GITHUB_OUTPUT}"
|
||||
else
|
||||
printf '%s=%s\n' "${name}" "${value}" >> "${GITHUB_OUTPUT}"
|
||||
fi
|
||||
fi
|
||||
printf '%s=%s\n' "${name}" "${value}"
|
||||
}
|
||||
|
||||
sanitize_case() {
|
||||
printf '%s' "$1" | sed 's/[^A-Za-z0-9._-]/_/g'
|
||||
}
|
||||
|
||||
case "${command_name}" in
|
||||
key)
|
||||
manifest=""
|
||||
for file in "${files[@]}"; do
|
||||
# A case whose fixtures are committed directly rather than through Git LFS
|
||||
# (OpenRA) has no pointer oid to key on, and nothing to download either.
|
||||
# Report it as uncacheable so the workflow skips the cache entirely.
|
||||
if ! metadata="$(get_lfs_metadata "${file}" 2>/dev/null)"; then
|
||||
echo "trace case ${case_name} is not stored in Git LFS; skipping fixture cache" >&2
|
||||
emit_output "cacheable" "false"
|
||||
emit_output "key" ""
|
||||
exit 0
|
||||
fi
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
manifest+="$(basename "${file}") ${expected_oid} ${expected_size}"$'\n'
|
||||
done
|
||||
|
||||
digest="$(printf '%s' "${manifest}" | sha256sum | awk '{ print substr($1, 1, 16) }')"
|
||||
safe_case="$(sanitize_case "${case_name}")"
|
||||
|
||||
emit_output "cacheable" "true"
|
||||
emit_output "key" "trace-fixture-${key_schema}-${safe_case}-${digest}"
|
||||
emit_output "paths" "$(printf '%s\n' "${files[@]}")"
|
||||
;;
|
||||
|
||||
verify)
|
||||
for file in "${files[@]}"; do
|
||||
metadata="$(get_lfs_metadata "${file}")"
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
verify_fixture_file "${file}" "${file}" "${expected_oid}" "${expected_size}"
|
||||
done
|
||||
echo "Verified ${#files[@]} fixture file(s) for ${case_name} against the tracked Git LFS pointers."
|
||||
;;
|
||||
|
||||
reset)
|
||||
# Put the working tree back to the pointer files a fresh checkout would
|
||||
# have, so that a rejected cache entry falls through to exactly the same
|
||||
# download path a cache miss takes.
|
||||
for file in "${files[@]}"; do
|
||||
rm -f "${file}" "${file}.tmp"
|
||||
done
|
||||
git checkout -- "${files[@]}"
|
||||
echo "Reset ${#files[@]} fixture file(s) for ${case_name} to their tracked Git LFS pointers."
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "unknown command: ${command_name}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,73 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared helpers for trace-fixture handling: reading the in-tree Git LFS pointer
|
||||
# metadata and verifying a fixture file against it. Sourced by
|
||||
# fetch-trace-fixture-lfs.sh (verify after download) and by
|
||||
# trace-fixture-cache.sh (cache key derivation and verify after cache restore),
|
||||
# so both paths agree on what a valid fixture is.
|
||||
|
||||
# Reads the Git LFS pointer tracked at HEAD for a fixture path and prints
|
||||
# "<oid> <size>". Fails if the tracked blob is not a well-formed LFS pointer.
|
||||
get_lfs_metadata() {
|
||||
local file="$1"
|
||||
local pointer
|
||||
local expected_oid
|
||||
local expected_size
|
||||
|
||||
if ! pointer="$(git show "HEAD:${file}" 2>/dev/null)"; then
|
||||
echo "failed to read tracked fixture metadata: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
if ! grep -q '^version https://git-lfs.github.com/spec/v1$' <<< "${pointer}"; then
|
||||
echo "tracked fixture is not a Git LFS pointer: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
expected_oid="$(awk '$1 == "oid" && $2 ~ /^sha256:/ { sub(/^sha256:/, "", $2); print $2 }' <<< "${pointer}")"
|
||||
expected_size="$(awk '$1 == "size" { print $2 }' <<< "${pointer}")"
|
||||
if ! [[ "${expected_oid}" =~ ^[0-9a-f]{64}$ ]] || ! [[ "${expected_size}" =~ ^[0-9]+$ ]]; then
|
||||
echo "invalid Git LFS pointer metadata: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
printf '%s %s\n' "${expected_oid}" "${expected_size}"
|
||||
}
|
||||
|
||||
# Checks an on-disk fixture against the size and SHA-256 from its LFS pointer.
|
||||
verify_fixture_file() {
|
||||
local downloaded_file="$1"
|
||||
local display_name="$2"
|
||||
local expected_oid="$3"
|
||||
local expected_size="$4"
|
||||
local actual_oid
|
||||
local actual_size
|
||||
|
||||
if [ ! -f "${downloaded_file}" ]; then
|
||||
echo "fixture file is missing: ${display_name}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
actual_size="$(wc -c < "${downloaded_file}" | tr -d '[:space:]')"
|
||||
if [ "${actual_size}" != "${expected_size}" ]; then
|
||||
echo "fixture size mismatch for ${display_name}: expected ${expected_size}, got ${actual_size}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
actual_oid="$(sha256sum "${downloaded_file}" | awk '{ print $1 }')"
|
||||
if [ "${actual_oid}" != "${expected_oid}" ]; then
|
||||
echo "fixture SHA-256 mismatch for ${display_name}: expected ${expected_oid}, got ${actual_oid}" >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Prints the fixture file paths of a trace case, one per line. Strips CR so the
|
||||
# result is usable when python emits CRLF (Git Bash on Windows).
|
||||
trace_fixture_files() {
|
||||
local case_name="$1"
|
||||
local fixture_dir="$2"
|
||||
local python_bin="${3:-python3}"
|
||||
|
||||
"${python_bin}" tools/trace_replay/trace_cases.py \
|
||||
--format fixture-files \
|
||||
--case "${case_name}" \
|
||||
--fixture-root "${fixture_dir}" | tr -d '\r'
|
||||
}
|
||||
+116
-16
@@ -11,6 +11,9 @@ on:
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
env:
|
||||
CCACHE_BASEDIR: ${{ github.workspace }}
|
||||
CCACHE_COMPRESS: "true"
|
||||
@@ -41,12 +44,11 @@ jobs:
|
||||
gradle-version: 8.10.2
|
||||
|
||||
- name: Restore ccache
|
||||
uses: actions/cache@v5
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-apk-${{ github.job }}-ccache-${{ github.ref_name }}-${{ github.run_id }}
|
||||
key: ${{ runner.os }}-apk-${{ github.job }}-ccache-v1
|
||||
restore-keys: |
|
||||
${{ runner.os }}-apk-${{ github.job }}-ccache-${{ github.ref_name }}-
|
||||
${{ runner.os }}-apk-${{ github.job }}-ccache-
|
||||
|
||||
- name: Install ccache
|
||||
@@ -125,6 +127,28 @@ jobs:
|
||||
if: always()
|
||||
run: ccache --show-stats
|
||||
|
||||
# Rewrite one rolling entry per job on the default branch. The upload stays
|
||||
# cumulative - it carries every object restored at the top of this run plus
|
||||
# the few TUs that actually changed - but Actions cache keys are immutable,
|
||||
# so the superseded blob has to be released before the same key can be
|
||||
# re-uploaded. Running after the build means a failed build leaves the
|
||||
# existing entry untouched. The other trigger branches restore this entry
|
||||
# rather than each writing a ~4 GB one of their own.
|
||||
- name: Release superseded ccache entry
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CACHE_KEY: ${{ runner.os }}-apk-${{ github.job }}-ccache-v1
|
||||
run: gh cache delete "${CACHE_KEY}" || true
|
||||
|
||||
- name: Save ccache
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
continue-on-error: true
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-apk-${{ github.job }}-ccache-v1
|
||||
|
||||
- name: Verify APK metadata and packaging
|
||||
run: |
|
||||
AAPT2="$(find "$ANDROID_HOME/build-tools" -name aapt2 -type f | sort -V | tail -n 1)"
|
||||
@@ -185,7 +209,7 @@ jobs:
|
||||
- name: Load trace cases
|
||||
id: trace-cases
|
||||
run: |
|
||||
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk)" >> "$GITHUB_OUTPUT"
|
||||
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk-matrix)" >> "$GITHUB_OUTPUT"
|
||||
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
trace-fixtures:
|
||||
@@ -201,9 +225,41 @@ jobs:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Derive trace fixture cache key
|
||||
id: fixture-key
|
||||
run: bash .github/scripts/trace-fixture-cache.sh key '${{ matrix.case }}'
|
||||
|
||||
- name: Restore trace fixture cache
|
||||
id: fixture-cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true'
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Verify restored trace fixture
|
||||
id: fixture-verify
|
||||
if: steps.fixture-cache.outputs.cache-hit == 'true'
|
||||
run: |
|
||||
if bash .github/scripts/trace-fixture-cache.sh verify '${{ matrix.case }}'; then
|
||||
echo "ok=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "ok=false" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::Cached fixture for ${{ matrix.case }} failed verification; falling back to the download path"
|
||||
bash .github/scripts/trace-fixture-cache.sh reset '${{ matrix.case }}'
|
||||
fi
|
||||
|
||||
- name: Fetch trace fixture
|
||||
if: steps.fixture-verify.outputs.ok != 'true'
|
||||
run: bash .github/scripts/fetch-trace-fixture-lfs.sh '${{ matrix.case }}'
|
||||
|
||||
- name: Save trace fixture cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true' && steps.fixture-cache.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Stage trace fixture
|
||||
run: |
|
||||
safe_case="$(printf '%s' '${{ matrix.case }}' | sed 's/[^A-Za-z0-9._-]/_/g')"
|
||||
@@ -281,13 +337,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
backend:
|
||||
- name: DirectGLES
|
||||
gpu: software
|
||||
- name: DirectVulkan
|
||||
gpu: lavapipe
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.android) }}
|
||||
matrix: ${{ fromJSON(needs.trace-cases.outputs.android) }}
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@v1.0
|
||||
@@ -370,6 +420,9 @@ jobs:
|
||||
MOBILEGL_USE_ANGLE: ${{ matrix.backend.name == 'DirectGLES' && '1' || '0' }}
|
||||
MOBILEGL_TRACE_ANGLE_VARIANT: ${{ matrix.case.name == 'minecraft-1.21.4-fabric-iris-bliss-in-world' && '90a62123d794' || 'ec889e6ea831' }}
|
||||
MOBILEGL_MAGMA_R11G11B10F_FALLBACK: ${{ matrix.backend.name == 'DirectVulkan' && '1' || '0' }}
|
||||
MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_DERIVE_NUM_SUBGROUPS: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
MOBILEGL_ITERATIONRP_FIX_BARRIER: ${{ matrix.backend.name == 'DirectVulkan' && matrix.case.name == 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' && '1' || '0' }}
|
||||
run: |
|
||||
apk_file="android-retrace-apks/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk"
|
||||
test -f "${apk_file}"
|
||||
@@ -379,6 +432,9 @@ jobs:
|
||||
if [ "${{ matrix.backend.name }}" = "DirectGLES" ] && [ "${{ matrix.case.name }}" = "minecraft-1.21.4-fabric-iris-bliss-in-world" ]; then
|
||||
extra_retrace_args+=(--avoid-angle-llvmpipe-sampler-mipmap-min-filter)
|
||||
fi
|
||||
if [ "${{ matrix.backend.name }}" = "DirectGLES" ] && [ "${{ matrix.case.avoid_angle_llvmpipe_explicit_lod_bias || false }}" = "true" ]; then
|
||||
extra_retrace_args+=(--avoid-angle-llvmpipe-explicit-lod-bias)
|
||||
fi
|
||||
if [ "${{ matrix.case.coherent_as_flush || false }}" = "true" ]; then
|
||||
extra_retrace_args+=(--coherent-as-flush)
|
||||
fi
|
||||
@@ -411,6 +467,24 @@ jobs:
|
||||
run_retrace || retrace_status=$?
|
||||
if [ "${retrace_status}" -eq 75 ]; then
|
||||
echo "::warning::Android emulator infrastructure failed; restarting it and retrying this retrace once."
|
||||
# Surface-lost is retried rather than failed, so it would otherwise
|
||||
# be invisible. Report it per job - a healthy run prints nothing and
|
||||
# a rate spike shows up as a row per affected case.
|
||||
reason_file="android-retrace-result/infrastructure-failure-reason.txt"
|
||||
surface_lost_retries=0
|
||||
if [ -f "${reason_file}" ]; then
|
||||
surface_lost_retries="$(grep -c 'angle-surface-lost' "${reason_file}" || true)"
|
||||
fi
|
||||
if [ "${surface_lost_retries}" -gt 0 ]; then
|
||||
echo "surface-lost retries: ${surface_lost_retries} (${{ matrix.backend.name }}, ${{ matrix.case.name }})" \
|
||||
>> "${GITHUB_STEP_SUMMARY}"
|
||||
fi
|
||||
# The restart truncates EMULATOR_LOG, and the attempt that lost the
|
||||
# emulator is the one worth reading - the retry usually only shows
|
||||
# the wreckage. Keep the first attempt's log before it is clobbered.
|
||||
if [ -f "${EMULATOR_LOG}" ]; then
|
||||
cp "${EMULATOR_LOG}" "${EMULATOR_LOG}.first-attempt" || true
|
||||
fi
|
||||
sh android-plugin/run-avd-ci.sh stop \
|
||||
--avd-name "${AVD_NAME}" \
|
||||
--emulator-log "${EMULATOR_LOG}" \
|
||||
@@ -450,6 +524,13 @@ jobs:
|
||||
if [ -f "${EMULATOR_LOG}" ]; then
|
||||
cp "${EMULATOR_LOG}" android-retrace-result/diagnostics/emulator.log
|
||||
fi
|
||||
if [ -f "${EMULATOR_LOG}.first-attempt" ]; then
|
||||
cp "${EMULATOR_LOG}.first-attempt" android-retrace-result/diagnostics/emulator-first-attempt.log
|
||||
fi
|
||||
# A vanished emulator looks identical whether the host OOM killer took
|
||||
# qemu or the renderer faulted. These two say which.
|
||||
free -h > android-retrace-result/diagnostics/host-memory.txt 2>&1 || true
|
||||
sudo dmesg -T 2>/dev/null | tail -300 > android-retrace-result/diagnostics/host-dmesg.txt || true
|
||||
|
||||
- name: Stop Emulator
|
||||
if: always()
|
||||
@@ -531,22 +612,41 @@ jobs:
|
||||
)
|
||||
|
||||
if ((${#failed_cases[@]})); then
|
||||
echo "Retaining fixtures for failed retrace case(s):"
|
||||
echo "Retaining fixtures and results for failed retrace case(s):"
|
||||
printf ' %s\n' "${!failed_cases[@]}"
|
||||
else
|
||||
echo "All retrace jobs succeeded; no fixtures need to be retained."
|
||||
echo "All retrace jobs succeeded; nothing needs to be retained."
|
||||
fi
|
||||
|
||||
deleted=0
|
||||
retained=0
|
||||
while IFS=$'\t' read -r artifact_id artifact_name; do
|
||||
keep=0
|
||||
if [[ "${artifact_name}" == MobileGL-trace-fixture-* ]]; then
|
||||
case_name="${artifact_name#MobileGL-trace-fixture-}"
|
||||
if [[ -v "failed_cases[${case_name}]" ]]; then
|
||||
echo "Retaining ${artifact_name} (${artifact_id}) for failed retrace."
|
||||
((retained += 1))
|
||||
continue
|
||||
keep=1
|
||||
fi
|
||||
elif [[ "${artifact_name}" == MobileGL-android-retrace-result-* ]]; then
|
||||
# The result artifact carries mobilegl.log, retrace.log, logcat,
|
||||
# the emulator log and the actual/diff images - the only record of
|
||||
# why a retrace failed. Its name ends in -<backend>-<case>, so a
|
||||
# suffix match on the case name keeps both backends' results for a
|
||||
# case that failed on either of them, which is what a comparison
|
||||
# needs. The match is anchored at the end, so a case name that is a
|
||||
# prefix of a longer one does not retain the longer one's results.
|
||||
for case_name in "${!failed_cases[@]}"; do
|
||||
if [[ "${artifact_name}" == *-"${case_name}" ]]; then
|
||||
keep=1
|
||||
break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
if ((keep)); then
|
||||
echo "Retaining ${artifact_name} (${artifact_id}) for failed retrace."
|
||||
((retained += 1))
|
||||
continue
|
||||
fi
|
||||
|
||||
echo "Deleting ${artifact_name} (${artifact_id})"
|
||||
|
||||
+211
-14
@@ -1,4 +1,4 @@
|
||||
name: Test
|
||||
name: Test
|
||||
|
||||
on:
|
||||
push:
|
||||
@@ -11,6 +11,9 @@ on:
|
||||
jobs:
|
||||
build-linux:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
env:
|
||||
BUILD_DIR: build-linux
|
||||
CCACHE_BASEDIR: ${{ github.workspace }}
|
||||
@@ -34,12 +37,11 @@ jobs:
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Restore ccache
|
||||
uses: actions/cache@v5
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-${{ github.ref_name }}-${{ github.run_id }}
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
restore-keys: |
|
||||
${{ runner.os }}-test-${{ github.job }}-ccache-${{ github.ref_name }}-
|
||||
${{ runner.os }}-test-${{ github.job }}-ccache-
|
||||
|
||||
- name: Prepare Vulkan SDK
|
||||
@@ -83,6 +85,8 @@ jobs:
|
||||
-DMOBILEGL_LOG_ACTIVE_LEVEL=MOBILEGL_LOG_LEVEL_INFO \
|
||||
-DMOBILEGL_BUILD_TEST=ON \
|
||||
-DMOBILEGL_BUILD_BENCHMARK=ON \
|
||||
-DMOBILEGL_BUILD_INTEGRATION_TEST=ON \
|
||||
-DMOBILEGL_ITEST_VK_ICD=/usr/share/vulkan/icd.d/lvp_icd.json \
|
||||
-DMOBILEGL_BUILD_TRACE_REPLAY=OFF \
|
||||
-DBENCHMARK_DOWNLOAD_DEPENDENCIES=ON \
|
||||
-DBENCHMARK_ENABLE_TESTING=OFF \
|
||||
@@ -95,6 +99,28 @@ jobs:
|
||||
if: always()
|
||||
run: ccache --show-stats
|
||||
|
||||
# Rewrite one rolling entry per job on the default branch. The upload stays
|
||||
# cumulative - it carries every object restored at the top of this run plus
|
||||
# the few TUs that actually changed - but Actions cache keys are immutable,
|
||||
# so the superseded blob has to be released before the same key can be
|
||||
# re-uploaded. Running after the build means a failed build leaves the
|
||||
# existing entry untouched. The other trigger branches restore this entry
|
||||
# rather than each writing one of their own.
|
||||
- name: Release superseded ccache entry
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CACHE_KEY: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
run: gh cache delete "${CACHE_KEY}" || true
|
||||
|
||||
- name: Save ccache
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
continue-on-error: true
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
|
||||
- name: Package Linux runtime
|
||||
run: |
|
||||
mkdir -p ci-artifacts
|
||||
@@ -110,6 +136,7 @@ jobs:
|
||||
"${BUILD_DIR}/CTestTestfile.cmake" \
|
||||
"${BUILD_DIR}/MobileGL/MG_Test" \
|
||||
"${BUILD_DIR}/MobileGL/MG_Benchmark" \
|
||||
"${BUILD_DIR}/MobileGL/MG_IntegrationTest" \
|
||||
"${SHARED_LIBS[@]}"
|
||||
|
||||
- name: Upload Linux runtime
|
||||
@@ -159,12 +186,105 @@ jobs:
|
||||
- name: Test
|
||||
working-directory: build-linux
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" = "true" ]; then
|
||||
ctest -V -L unit --no-tests=error
|
||||
else
|
||||
ctest --output-on-failure -L unit --no-tests=error
|
||||
fi
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: unit-core-dumps
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
integration:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build-linux
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Install runtime dependencies
|
||||
# Same set as the benchmark job, for the same reason: the scenarios bring
|
||||
# up real headless EGL (llvmpipe) and Vulkan (lavapipe) contexts, and
|
||||
# libegl-mesa0 - the EGL vendor library behind glvnd's libegl1 dispatch -
|
||||
# only arrives as a Recommends.
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libvulkan1 libegl1 libegl-mesa0 libgles2 libgl1-mesa-dri mesa-vulkan-drivers
|
||||
|
||||
- name: Download Linux runtime
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: .
|
||||
|
||||
- name: Unpack Linux runtime
|
||||
run: tar -xzf mobilegl-linux-runtime.tgz
|
||||
|
||||
- name: Normalize CTest command paths
|
||||
run: |
|
||||
python - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
for path in Path('build-linux').rglob('CTestTestfile.cmake'):
|
||||
text = path.read_text()
|
||||
text = re.sub(r'"[^"]*/cmake-[^"]*/bin/cmake"', '"cmake"', text)
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
- name: Integration scenarios
|
||||
working-directory: build-linux
|
||||
# REQUIRE_GPU makes a driverless runner FAIL instead of skipping every
|
||||
# scenario - an all-skip run is otherwise indistinguishable from a pass,
|
||||
# which is how a five-month-old draw-dropping bug survived unseen until
|
||||
# this lane existed.
|
||||
#
|
||||
# The lavapipe ICD pin lives in the build-linux configure
|
||||
# (-DMOBILEGL_ITEST_VK_ICD), NOT here: the configure bakes it into each
|
||||
# test's ctest ENVIRONMENT property, and a property entry OVERRIDES the
|
||||
# job environment - a VK_ICD_FILENAMES exported here would be silently
|
||||
# ignored while looking like it works. This lane runs on lavapipe
|
||||
# deterministically, not on whichever of the eight Mesa ICDs a GPU-less
|
||||
# runner enumerates first.
|
||||
#
|
||||
# Cores are armed so that any crash - the harness pre-flight child's
|
||||
# included - leaves /tmp/core.*, which the failure-only step below ships
|
||||
# as an artifact. Analyzing a downloaded core against the runtime
|
||||
# artifact's binary in an ubuntu-24.04 userspace reproduces the exact
|
||||
# crash stack without burning a CI round on an in-workflow debugger.
|
||||
env:
|
||||
MOBILEGL_ITEST_REQUIRE_GPU: "1"
|
||||
MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: "1"
|
||||
MOBILEGL_DERIVE_NUM_SUBGROUPS: "1"
|
||||
MOBILEGL_ITERATIONRP_FIX_BARRIER: "1"
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" = "true" ]; then
|
||||
ctest -V -L integration-gpu --no-tests=error
|
||||
else
|
||||
ctest --output-on-failure -L integration-gpu --no-tests=error
|
||||
fi
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: integration-core-dumps
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
benchmark:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build-linux
|
||||
@@ -208,7 +328,18 @@ jobs:
|
||||
|
||||
- name: Benchmark
|
||||
working-directory: build-linux
|
||||
run: ctest -V -C Release -L benchmark --no-tests=error
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
ctest -V -C Release -L benchmark --no-tests=error
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: benchmark-core-dumps
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
build-retrace:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -216,6 +347,10 @@ jobs:
|
||||
- build-linux
|
||||
- test
|
||||
- benchmark
|
||||
- integration
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
env:
|
||||
BUILD_DIR: build-retrace
|
||||
CCACHE_BASEDIR: ${{ github.workspace }}
|
||||
@@ -240,12 +375,11 @@ jobs:
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Restore ccache
|
||||
uses: actions/cache@v5
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-${{ github.ref_name }}-${{ github.run_id }}
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
restore-keys: |
|
||||
${{ runner.os }}-test-${{ github.job }}-ccache-${{ github.ref_name }}-
|
||||
${{ runner.os }}-test-${{ github.job }}-ccache-
|
||||
|
||||
- name: Prepare Vulkan SDK
|
||||
@@ -311,6 +445,21 @@ jobs:
|
||||
if: always()
|
||||
run: ccache --show-stats
|
||||
|
||||
- name: Release superseded ccache entry
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CACHE_KEY: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
run: gh cache delete "${CACHE_KEY}" || true
|
||||
|
||||
- name: Save ccache
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
continue-on-error: true
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
|
||||
- name: Normalize CTest command paths
|
||||
run: |
|
||||
python - <<'PY'
|
||||
@@ -343,7 +492,9 @@ jobs:
|
||||
needs:
|
||||
- test
|
||||
- benchmark
|
||||
- integration
|
||||
outputs:
|
||||
matrix: ${{ steps.trace-cases.outputs.matrix }}
|
||||
names: ${{ steps.trace-cases.outputs.names }}
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
@@ -351,7 +502,9 @@ jobs:
|
||||
|
||||
- name: Load trace cases
|
||||
id: trace-cases
|
||||
run: echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
run: |
|
||||
echo "matrix=$(python3 tools/trace_replay/trace_cases.py --ci --format github-test-matrix)" >> "$GITHUB_OUTPUT"
|
||||
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
trace-fixtures:
|
||||
name: trace fixture (${{ matrix.case }})
|
||||
@@ -366,9 +519,41 @@ jobs:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Derive trace fixture cache key
|
||||
id: fixture-key
|
||||
run: bash .github/scripts/trace-fixture-cache.sh key '${{ matrix.case }}'
|
||||
|
||||
- name: Restore trace fixture cache
|
||||
id: fixture-cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true'
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Verify restored trace fixture
|
||||
id: fixture-verify
|
||||
if: steps.fixture-cache.outputs.cache-hit == 'true'
|
||||
run: |
|
||||
if bash .github/scripts/trace-fixture-cache.sh verify '${{ matrix.case }}'; then
|
||||
echo "ok=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "ok=false" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::Cached fixture for ${{ matrix.case }} failed verification; falling back to the download path"
|
||||
bash .github/scripts/trace-fixture-cache.sh reset '${{ matrix.case }}'
|
||||
fi
|
||||
|
||||
- name: Fetch trace fixture
|
||||
if: steps.fixture-verify.outputs.ok != 'true'
|
||||
run: bash .github/scripts/fetch-trace-fixture-lfs.sh '${{ matrix.case }}'
|
||||
|
||||
- name: Save trace fixture cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true' && steps.fixture-cache.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Stage trace fixture
|
||||
run: |
|
||||
safe_case="$(printf '%s' '${{ matrix.case }}' | sed 's/[^A-Za-z0-9._-]/_/g')"
|
||||
@@ -398,11 +583,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
backend:
|
||||
- DirectGLES
|
||||
- DirectVulkan
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.names) }}
|
||||
matrix: ${{ fromJSON(needs.trace-cases.outputs.matrix) }}
|
||||
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
@@ -456,9 +637,17 @@ jobs:
|
||||
- name: Retrace and validate
|
||||
working-directory: build-retrace/tools/trace_replay
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
if [ '${{ matrix.backend }}' = 'DirectVulkan' ]; then
|
||||
export MOBILEGL_MAGMA_R11G11B10F_FALLBACK=1
|
||||
fi
|
||||
if [ '${{ matrix.backend }}' = 'DirectVulkan' ] \
|
||||
&& [ '${{ matrix.case }}' = 'minecraft-1.21.4-fabric-iris-iterationrp-in-world' ]; then
|
||||
export MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH=1
|
||||
export MOBILEGL_DERIVE_NUM_SUBGROUPS=1
|
||||
export MOBILEGL_ITERATIONRP_FIX_BARRIER=1
|
||||
fi
|
||||
# The blended depth-write quirk auto-enables only on Qualcomm, which no CI
|
||||
# runner has, so force it on for the OIT case it exists to fix. ForceOn
|
||||
# bypasses only the vendor gate, so this exercises the real strip on
|
||||
@@ -470,6 +659,14 @@ jobs:
|
||||
fi
|
||||
ctest -V --no-tests=error -R '^MobileGLTraceReplay\.${{ matrix.case }}\.${{ matrix.backend }}$'
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: retrace-core-dumps-${{ matrix.backend }}-${{ matrix.case }}
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Upload actual image
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
|
||||
@@ -27,3 +27,4 @@ MobileGL/MG*/cmake-build*
|
||||
tools/trace_replay/work/
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
/.gradle
|
||||
|
||||
+3
-3
@@ -7,9 +7,6 @@
|
||||
[submodule "3rdparty/SPIRV-Cross"]
|
||||
path = 3rdparty/SPIRV-Cross
|
||||
url = https://github.com/KhronosGroup/SPIRV-Cross.git
|
||||
[submodule "include/FastSTL"]
|
||||
path = include/FastSTL
|
||||
url = https://github.com/MobileGL-Dev/FastSTL.git
|
||||
[submodule "3rdparty/tracy"]
|
||||
path = 3rdparty/tracy
|
||||
url = https://github.com/wolfpld/tracy.git
|
||||
@@ -34,3 +31,6 @@
|
||||
[submodule "3rdparty/asio"]
|
||||
path = 3rdparty/asio
|
||||
url = https://github.com/chriskohlhoff/asio.git
|
||||
[submodule "include/ska"]
|
||||
path = include/ska
|
||||
url = https://github.com/MobileGL-Dev/flat_hash_map.git
|
||||
|
||||
Vendored
+1
-1
Submodule 3rdparty/apitrace updated: 10935bb5e4...c8036190fc
Vendored
+1
-1
Submodule 3rdparty/glslang updated: 900b29d449...fa562bb911
+112
-1
@@ -20,6 +20,81 @@ set(MOBILEGL_VULKAN_LIBRARY "" CACHE FILEPATH "Vulkan loader/MoltenVK library to
|
||||
if (ANDROID)
|
||||
set(MOBILEGL_BUILD_TEST OFF CACHE BOOL "Build MobileGL tests" FORCE)
|
||||
set(MOBILEGL_BUILD_BENCHMARK OFF CACHE BOOL "Build MobileGL benchmarks" FORCE)
|
||||
|
||||
# ------- Android API level policy: minimum 26, decided here and only here -------
|
||||
# MobileGL ships against API 26: the codebase must not use any API introduced
|
||||
# after 26. That usage constraint is enforced where it is real - the shipping
|
||||
# gradle build compiles at minSdk 26, where a newer API is simply undeclared
|
||||
# and fails to compile. Configuring at a HIGHER level is therefore allowed
|
||||
# (nothing in the tree may rely on it), but a LOWER level would change the
|
||||
# libc contract underneath the shipped library and is refused.
|
||||
#
|
||||
# This has to live at configure time because the level cannot be corrected
|
||||
# from a source header. A `#define __ANDROID_API__ 26` in a common header
|
||||
# only rewrites the macro for the bionic headers that happen to be included
|
||||
# after it; any libc++ header pulled in earlier has already latched its
|
||||
# feature macros at the real configure-time level. libc++ and bionic then
|
||||
# disagree about which symbols exist - libc++ calls e.g.
|
||||
# pthread_cond_clockwait while bionic, re-read at the lowered level, has
|
||||
# hidden its declaration. MobileGL/Defines.h carried exactly that pin from
|
||||
# the first commit until it was removed; this guard is what replaces it.
|
||||
#
|
||||
# Read the level back from the compiler target triple first. Its trailing
|
||||
# number (aarch64-none-linux-android26) is precisely what clang turns into
|
||||
# __ANDROID_API__, so it cannot disagree with the compile itself, and it is
|
||||
# already past every NDK normalisation step - codename aliases, "latest",
|
||||
# and per-ABI minimum pull-ups. ANDROID_PLATFORM_LEVEL is the fallback for
|
||||
# generators/languages where the triple variable is not populated.
|
||||
#
|
||||
# Note CMAKE_SYSTEM_VERSION is deliberately NOT consulted: it holds the API
|
||||
# level only under the NDK's newer toolchain path, and is a meaningless 1
|
||||
# when ANDROID_USE_LEGACY_TOOLCHAIN_FILE is on (which is what AGP has been
|
||||
# defaulting to). Reading it would fail every legacy-mode build.
|
||||
set(MOBILEGL_ANDROID_API_LEVEL 26)
|
||||
|
||||
set(_mobilegl_android_api "")
|
||||
foreach (_mobilegl_api_triple "${CMAKE_CXX_COMPILER_TARGET}"
|
||||
"${CMAKE_C_COMPILER_TARGET}")
|
||||
if (NOT _mobilegl_android_api AND
|
||||
_mobilegl_api_triple MATCHES "-android([0-9]+)$")
|
||||
set(_mobilegl_android_api "${CMAKE_MATCH_1}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
foreach (_mobilegl_api_var ANDROID_PLATFORM_LEVEL ANDROID_NATIVE_API_LEVEL
|
||||
ANDROID_PLATFORM)
|
||||
if (NOT _mobilegl_android_api AND ${_mobilegl_api_var})
|
||||
string(REGEX REPLACE "^android-" ""
|
||||
_mobilegl_android_api "${${_mobilegl_api_var}}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (NOT _mobilegl_android_api MATCHES "^[0-9]+$")
|
||||
message(FATAL_ERROR
|
||||
"MobileGL: could not determine the Android API level (got "
|
||||
"\"${_mobilegl_android_api}\"). Configure with the NDK toolchain "
|
||||
"file and -DANDROID_PLATFORM=android-${MOBILEGL_ANDROID_API_LEVEL}.")
|
||||
elseif (_mobilegl_android_api LESS MOBILEGL_ANDROID_API_LEVEL)
|
||||
message(FATAL_ERROR
|
||||
"MobileGL requires at least Android API ${MOBILEGL_ANDROID_API_LEVEL}, "
|
||||
"but this build resolved to API ${_mobilegl_android_api}.\n"
|
||||
"Configure with -DANDROID_PLATFORM=android-${MOBILEGL_ANDROID_API_LEVEL} "
|
||||
"(gradle builds get this from minSdk ${MOBILEGL_ANDROID_API_LEVEL}, so "
|
||||
"check that minSdk instead of adding an override).")
|
||||
elseif (_mobilegl_android_api GREATER MOBILEGL_ANDROID_API_LEVEL)
|
||||
message(STATUS
|
||||
"MobileGL: configuring at Android API ${_mobilegl_android_api} "
|
||||
"(> shipping minimum ${MOBILEGL_ANDROID_API_LEVEL}). Allowed, but the "
|
||||
"tree must not use post-${MOBILEGL_ANDROID_API_LEVEL} APIs - the "
|
||||
"minSdk-${MOBILEGL_ANDROID_API_LEVEL} gradle build is the enforcing "
|
||||
"compile.")
|
||||
endif()
|
||||
|
||||
message(STATUS "MobileGL: Android API level ${_mobilegl_android_api}")
|
||||
|
||||
unset(_mobilegl_android_api)
|
||||
unset(_mobilegl_api_var)
|
||||
unset(_mobilegl_api_triple)
|
||||
endif()
|
||||
|
||||
option(MOBILEGL_ENABLE_LTO "Build with ThinLTO/IPO" OFF)
|
||||
@@ -107,6 +182,7 @@ set(ENABLE_SPVREMAPPER OFF CACHE BOOL "Enable SPVRemapper" FORCE)
|
||||
set(ENABLE_OPT ON CACHE BOOL "Enable SPIRV-Tools opt usage in glslang" FORCE)
|
||||
set(BUILD_EXTERNAL ON CACHE BOOL "Build external deps in External/" FORCE)
|
||||
set(ENABLE_GLSLANG_INSTALL OFF CACHE BOOL "Install glslang targets" FORCE)
|
||||
set(SPIRV_SKIP_EXECUTABLES ON CACHE BOOL "Skip building SPIRV-Tools executables" FORCE)
|
||||
|
||||
set(SPIRV_CROSS_C_API ON CACHE BOOL "Enable C API" FORCE)
|
||||
set(SPIRV_CROSS_ENABLE_GLSL ON CACHE BOOL "Enable GLSL backend" FORCE)
|
||||
@@ -194,6 +270,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpvcSession.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/TranslationCache.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/glslang/TMglGlslIoResolver.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenInterfaceStructPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EliminateFloatEqualsZeroPass.cpp
|
||||
@@ -201,18 +278,41 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RenameBuiltinShadowingFunctionsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecomposeWorkgroupVec3Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecoratePositionInvariantPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DemoteFloat64Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenFloat64StorageBlockPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerDrawParametersPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerViewportIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PackDoubleVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenXfbInterfaceBlocksPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/UniquifyIoBlockNamesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/SplitArrayVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DeriveNumSubgroupsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPBarrierPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FixIterationRPSubgroupScratchPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateSubgroupsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DArrayImagesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DSampledImagesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/BakeImageFormatsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/WidenImageFormatsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ClampMultisampleFetchPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PrivateToEntryLocalPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUniformLocationsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUboMemberRelaxedPrecisionPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripNoPerspectivePass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateNoPerspectivePass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeFragmentOutputIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeResourceArrayIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenAtomicCounterBlockPass.cpp
|
||||
|
||||
MobileGL/MG_Util/BackendLoaders/OpenGL/Loader.cpp
|
||||
MobileGL/MG_Util/BackendLoaders/Vulkan/Loader.cpp
|
||||
|
||||
MobileGL/MG_Util/SelfTest/DriverBugProbes.cpp
|
||||
MobileGL/MG_Util/SelfTest/DriverPost.cpp
|
||||
MobileGL/MG_Util/SelfTest/DriverPostIterationRPWitness.cpp
|
||||
|
||||
MobileGL/MG_Util/Texture/PixelStoreProcessor.cpp
|
||||
MobileGL/MG_Util/Texture/TextureFormatProcessor.cpp
|
||||
@@ -235,6 +335,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Impl/GLImpl/Program/ProgramInterface.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Program/GL_ProgramPipeline.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Texture/GL_Texture.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Debug/GL_Debug.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Texture/Validators.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Texture/ProxyTexture.cpp
|
||||
MobileGL/MG_Impl/GLImpl/VertexArray/GL_VertexArray.cpp
|
||||
@@ -292,10 +393,13 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_State/GLState/TextureState/TextureObject2DCube.cpp
|
||||
MobileGL/MG_State/GLState/TextureState/TextureObject3D.cpp
|
||||
MobileGL/MG_State/GLState/TextureState/TextureObjectBuffer.cpp
|
||||
MobileGL/MG_State/GLState/TextureState/TextureObjectView.cpp
|
||||
MobileGL/MG_State/GLState/TextureState/TextureUnit.cpp
|
||||
MobileGL/MG_State/GLState/TextureState/TextureState.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramObject.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramTranslationCache.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramSpirvTask.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderCompileTask.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderObject.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderPreprocessCache.cpp
|
||||
@@ -372,7 +476,7 @@ set(MOBILEGL_INCLUDE_DIR
|
||||
# Header-only submodule: no add_subdirectory, no link target. Only
|
||||
# MG_Util/Async/ShaderCompilePool.cpp includes it, and it stays behind that file's
|
||||
# pimpl so no consumer target needs this path.
|
||||
${CMAKE_SOURCE_DIR}/3rdparty/asio/asio/include
|
||||
${CMAKE_SOURCE_DIR}/3rdparty/asio/include
|
||||
)
|
||||
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
@@ -584,3 +688,10 @@ if (NOT ANDROID)
|
||||
add_subdirectory(tools/trace_replay)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# The integration binary is also useful as a standalone adb-shell executable.
|
||||
# Android cannot use the desktop-only MobileGL_s target, so its CMake module
|
||||
# links libMobileGL.so and creates an AImageReader-backed window instead.
|
||||
if (ANDROID AND MOBILEGL_BUILD_INTEGRATION_TEST)
|
||||
add_subdirectory(MobileGL/MG_IntegrationTest)
|
||||
endif()
|
||||
|
||||
+101
-5
@@ -69,14 +69,65 @@ namespace MobileGL::MG_Config {
|
||||
struct FeaturesTable {
|
||||
// MOBILEGL_DISABLE_TIMERQUERY: do not advertise or use GPU timer queries.
|
||||
Bool DisableTimerQuery = false;
|
||||
// MOBILEGL_ENABLE_GLES_TEXTURE_VIEW: advertise GL_ARB_texture_view on DirectGLES when
|
||||
// the host ES driver has EXT/OES_texture_view. Off by default: the host extension is
|
||||
// present on Adreno 830 and the functional half of KHR-GL4{2,3}.texture_view still fails
|
||||
// there, because the view's ES internalformat is normalized independently of the storage
|
||||
// it aliases (see BackendObject_DirectGLES::BuildAdvertisedExtensions). The flag exists
|
||||
// so that work can be done without editing the gate.
|
||||
Bool EnableGlesTextureView = false;
|
||||
// MOBILEGL_ENABLE_SPIRV_VALIDATION: validate generated and transformed SPIR-V.
|
||||
// Disabled by default because validation is a diagnostics-only cost.
|
||||
Bool EnableSpirvValidation = false;
|
||||
// MOBILEGL_USE_ANGLE: load ANGLE EGL/GLES libraries.
|
||||
Bool UseAngle = false;
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
// MOBILEGL_TRACE_ANGLE_VARIANT: signed trace-APK ANGLE build short hash.
|
||||
String TraceAngleVariant;
|
||||
#endif
|
||||
// MOBILEGL_DISABLE_SUBGROUP: force-disable Vulkan shader subgroup support.
|
||||
// MOBILEGL_DISABLE_SUBGROUP: force-disable Vulkan shader subgroup support,
|
||||
// including the opt-in emulated compute path below.
|
||||
Bool DisableSubgroup = false;
|
||||
// MOBILEGL_MAGMA_EMULATE_SUBGROUP: implement GL_KHR_shader_subgroup's compute
|
||||
// stage on a 32-lane VIRTUAL subgroup lowered to workgroup-shared memory
|
||||
// (ShaderTranspiler::EmulateSubgroupsPass). Strictly a last resort: it only ever
|
||||
// engages when this flag is set AND the device has no native subgroup support at
|
||||
// all - a device with real subgroup operations always uses them natively,
|
||||
// whatever their width (the known iterationRP defect is patched by
|
||||
// FixIterationRPSubgroupScratch below instead). Off by default.
|
||||
Bool MagmaEmulateSubgroup = false;
|
||||
// MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH: patch iterationRP's own bug - the
|
||||
// pack declares `shared vec2 prefixSumCache[32]` for a 512-invocation exposure
|
||||
// reduction and indexes it by gl_SubgroupID, so any device with sub-16-lane
|
||||
// subgroups (8-lane lavapipe -> 64 subgroups) writes shared memory out of
|
||||
// bounds. The pass grows that one array to what the device's topology needs and
|
||||
// touches nothing else; it only rewrites modules positively matching the pack's
|
||||
// reduction fingerprint (ShaderTranspiler::FixIterationRPSubgroupScratchPass),
|
||||
// so every other shader passes through byte-identical - as does iterationRP
|
||||
// itself on >= 16-lane devices. Auto is ON; ForceOff replays the pack's bug
|
||||
// verbatim.
|
||||
QuirkOverride FixIterationRPSubgroupScratch = QuirkOverride::Auto;
|
||||
// MOBILEGL_ITERATIONRP_FIX_BARRIER: repair Program 203's missing workgroup
|
||||
// rendezvous between its two reductions over prefixSumCache. Off by default and
|
||||
// fingerprint-gated by FixIterationRPBarrierPass when enabled.
|
||||
Bool IterationRPFixBarrier = false;
|
||||
// MOBILEGL_DERIVE_NUM_SUBGROUPS: replace compute gl_NumSubgroups loads with
|
||||
// ceil(workgroup invocations / gl_SubgroupSize) on the NATIVE subgroup path
|
||||
// (ShaderTranspiler::DeriveNumSubgroupsPass). Auto is ON: GL requires
|
||||
// gl_SubgroupID < gl_NumSubgroups, Adreno's builtin reports 1 while the same
|
||||
// dispatch emits IDs 0..7, and the derived value is the one Vulkan guarantees
|
||||
// whenever the pipeline can request REQUIRE_FULL_SUBGROUPS (which the renderer
|
||||
// does whenever local_size_x is a multiple of the native width). ForceOff returns
|
||||
// to the raw driver builtin.
|
||||
QuirkOverride DeriveNumSubgroups = QuirkOverride::Auto;
|
||||
// MOBILEGL_ADVERTISE_FP64: add GL_ARB_gpu_shader_fp64 to the advertised extension
|
||||
// string. `double` in a shader always WORKS - it is narrowed to 32 bits before any
|
||||
// module reaches a backend (ShaderTranspiler::DemoteFloat64Pass) - but the extension
|
||||
// promises 64-bit precision, and that is the one thing the narrowing cannot deliver.
|
||||
// Off by default so an application that checks the string before using doubles keeps
|
||||
// its float path; on for measuring what the conformance suite makes of the demoted
|
||||
// precision. See the DemoteFloat64Pass header and the "fp64" POST row.
|
||||
Bool AdvertiseFp64 = false;
|
||||
// MOBILEGL_MAGMA_R11G11B10F_FALLBACK: use fallback format for R11G11B10F on Vulkan.
|
||||
Bool MagmaR11G11B10FFallback = false;
|
||||
// MOBILEGL_MAGMA_FRAMESINFLIGHT: requested Magma frames in flight, defaulting to 3.
|
||||
@@ -84,6 +135,13 @@ namespace MobileGL::MG_Config {
|
||||
// MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER: avoid mipmap min filters in samplers,
|
||||
// resolves certain rendering bugs on ANGLE + llvmpipe.
|
||||
Bool AvoidSamplerMipmapMinFilter = false;
|
||||
// MOBILEGL_AVOID_EXPLICIT_LOD_BIAS: leave an already-explicit LOD argument alone when
|
||||
// emulating GL_TEXTURE_LOD_BIAS, instead of adding the bias uniform to it. Injecting
|
||||
// the uniform turns a compile-time-constant LOD into a runtime expression, which
|
||||
// sends ANGLE + llvmpipe down a mip-selection path that dereferences a NULL
|
||||
// descriptor and kills the process. Deviates from spec (Vulkan adds the bias to
|
||||
// OpImageSampleExplicitLod), so it is an avoidance for that stack only.
|
||||
Bool AvoidExplicitLodBias = false;
|
||||
// MOBILEGL_COHERENT_AS_FLUSH: app-compat for engines (e.g. Flywheel) that write
|
||||
// GPU-read data through persistent GL_MAP_FLUSH_EXPLICIT_BIT maps they never
|
||||
// flush. Persistent FLUSH_EXPLICIT map requests are rewritten to coherent
|
||||
@@ -97,16 +155,19 @@ namespace MobileGL::MG_Config {
|
||||
// per-draw glBufferSubData path instead of the persistent-mapped ring allocator
|
||||
// (negative control / driver-bug escape hatch).
|
||||
Bool DisableUboRing = false;
|
||||
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
|
||||
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
|
||||
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
|
||||
// Adreno does not), which means the emulation is dead code on exactly the stack the
|
||||
// headless suite runs on. This forces it live so the scenarios and the CTS can
|
||||
// exercise the path, and gives the device an A/B lever over the same choice.
|
||||
Bool EsprytForceDepthStencilReadbackEmulation = false;
|
||||
// MOBILEGL_RELAXED_SEMANTICS: relax strict core-profile rules (e.g. VAO-0 draws,
|
||||
// texture-name reuse after delete) even on contexts that explicitly requested a core
|
||||
// profile. Without it, relaxed semantics still apply to every context that did not
|
||||
// explicitly request a core profile via EGL_CONTEXT_OPENGL_PROFILE_MASK / a >=3.1
|
||||
// version request.
|
||||
Bool RelaxedSemantics = false;
|
||||
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that
|
||||
// rewrites the recognized workgroup prefix-scan template on Qualcomm devices with
|
||||
// subgroups wider than 32 lanes (see ShaderSourceProcessor's quirk registry).
|
||||
QuirkOverride SubgroupPrefixScanQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE: overrides the DirectVulkan quirk that
|
||||
// strips depth writes from accumulation-blended pipelines (MIN/MAX or additive
|
||||
// ONE+ONE - the multi-pass depth-equality signature) on drivers without
|
||||
@@ -136,6 +197,41 @@ namespace MobileGL::MG_Config {
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE_THREADS: shader-compile worker count. 0 (unset) means
|
||||
// auto, which is min(4, big cores); an explicit value is honoured as given.
|
||||
Uint32 AsyncShaderCompileThreads = 0;
|
||||
// MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS: while a compile job is still in flight,
|
||||
// glGetShaderiv(GL_COMPILE_STATUS) answers GL_TRUE and the shader info log reads
|
||||
// empty, WITHOUT joining the job (latched per compile - see
|
||||
// ShaderObject::TakeOptimisticCompileAnswer). A deliberate, bounded spec violation:
|
||||
// a real failure still fails the program link with the compile log quoted. It
|
||||
// exists for applications that compile hundreds of shaders serially and read the
|
||||
// status right after each glCompileShader - Iris's shader-pack load - where those
|
||||
// per-shader joins are what serializes the batch on its main path (Iris's gbuffer
|
||||
// phase issues no program-level query between programs; program-level LINK_STATUS
|
||||
// and the program info log still join truthfully, so paths that check each link
|
||||
// immediately stay serial by their own construction). Off by default; never
|
||||
// advertise it.
|
||||
QuirkOverride AsyncOptimisticShaderStatus = QuirkOverride::Auto;
|
||||
// MOBILEGL_SHADER_CACHE: the three-level, in-memory shader translation memo
|
||||
// (MG_Util/ShaderTranspiler/TranslationCache.h). The levels follow the GL
|
||||
// entry points - L1c memoizes one glCompileShader's PARSE VERDICT, L1 a
|
||||
// linked program's whole front end, L2 DirectGLES's emitted ESSL. Auto is
|
||||
// ON; ForceOff turns ALL THREE off and makes every translation run from
|
||||
// scratch. The escape hatch exists because a wrong cache hit is a silently
|
||||
// miscompiled shader: if a device ever renders differently with the cache
|
||||
// on, one run with this falsy says so.
|
||||
QuirkOverride ShaderTranslationCache = QuirkOverride::Auto;
|
||||
// MOBILEGL_FORCE_VIEWPORT_ARRAY_EMULATION: DirectGLES' gl_ViewportIndex routing
|
||||
// emulation - the builtin becomes a flat varying, the fragment stage gets a
|
||||
// per-pass gate, and a routed draw is REPLAYED once per distinct viewport state
|
||||
// with the real glViewport/glScissor/glDepthRangef set for it. Auto is ON, and
|
||||
// it is ON even where the driver advertises GL_OES_viewport_array, because that
|
||||
// extension only ever gave the SHADER a compilable name: MobileGL has never
|
||||
// programmed a driver's INDEXED viewport state (SyncRenderState pushes index 0
|
||||
// and nothing else), so on an extension-capable driver every index rasterized as
|
||||
// index 0 exactly as it did without one. ForceOff returns to that behaviour -
|
||||
// the pre-emulation path, extension passthrough where it exists and
|
||||
// LowerViewportIndexPass' demote-to-a-plain-global where it does not - and is
|
||||
// the negative control the emulation is measured against.
|
||||
QuirkOverride ViewportArrayEmulation = QuirkOverride::Auto;
|
||||
};
|
||||
extern FeaturesTable Features;
|
||||
} // namespace MobileGL::MG_Config
|
||||
|
||||
@@ -162,20 +162,30 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
inline void InitFeatures() {
|
||||
auto& features = MG_Config::Features;
|
||||
features.DisableTimerQuery = QueryEnvFlag("MOBILEGL_DISABLE_TIMERQUERY");
|
||||
features.EnableGlesTextureView = QueryEnvFlag("MOBILEGL_ENABLE_GLES_TEXTURE_VIEW");
|
||||
features.EnableSpirvValidation = QueryEnvFlag("MOBILEGL_ENABLE_SPIRV_VALIDATION");
|
||||
features.UseAngle = QueryEnvFlag("MOBILEGL_USE_ANGLE");
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
QueryEnvVariable("MOBILEGL_TRACE_ANGLE_VARIANT", features.TraceAngleVariant, "");
|
||||
#endif
|
||||
features.DisableSubgroup = QueryEnvFlag("MOBILEGL_DISABLE_SUBGROUP");
|
||||
features.MagmaEmulateSubgroup = QueryEnvFlag("MOBILEGL_MAGMA_EMULATE_SUBGROUP");
|
||||
features.FixIterationRPSubgroupScratch =
|
||||
QueryEnvQuirkOverride("MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH");
|
||||
features.IterationRPFixBarrier = QueryEnvFlag("MOBILEGL_ITERATIONRP_FIX_BARRIER");
|
||||
features.DeriveNumSubgroups = QueryEnvQuirkOverride("MOBILEGL_DERIVE_NUM_SUBGROUPS");
|
||||
features.AdvertiseFp64 = QueryEnvFlag("MOBILEGL_ADVERTISE_FP64");
|
||||
features.MagmaR11G11B10FFallback = QueryEnvFlag("MOBILEGL_MAGMA_R11G11B10F_FALLBACK");
|
||||
features.MagmaFramesInFlight = QueryEnvUint32("MOBILEGL_MAGMA_FRAMESINFLIGHT", 3, 1, 64);
|
||||
features.AvoidSamplerMipmapMinFilter =
|
||||
QueryEnvFlag("MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER");
|
||||
features.AvoidExplicitLodBias = QueryEnvFlag("MOBILEGL_AVOID_EXPLICIT_LOD_BIAS");
|
||||
features.CoherentAsFlush = QueryEnvFlag("MOBILEGL_COHERENT_AS_FLUSH");
|
||||
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
||||
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
||||
features.EsprytForceDepthStencilReadbackEmulation =
|
||||
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||
features.SubgroupPrefixScanQuirk = QueryEnvQuirkOverride("MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN");
|
||||
features.MagmaDisableBlendedDepthWriteQuirk =
|
||||
QueryEnvQuirkOverride("MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE");
|
||||
features.DisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS");
|
||||
@@ -183,6 +193,11 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
features.EsprytMultiDrawMode = QueryEnvGLESMultiDrawMode("MOBILEGL_ESPRYT_MULTIDRAW_MODE");
|
||||
features.AsyncShaderCompile = QueryEnvQuirkOverride("MOBILEGL_ASYNC_SHADER_COMPILE");
|
||||
features.AsyncShaderCompileThreads = QueryEnvUint32("MOBILEGL_ASYNC_SHADER_COMPILE_THREADS", 0, 0, 64);
|
||||
features.AsyncOptimisticShaderStatus =
|
||||
QueryEnvQuirkOverride("MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS");
|
||||
features.ShaderTranslationCache = QueryEnvQuirkOverride("MOBILEGL_SHADER_CACHE");
|
||||
features.ViewportArrayEmulation =
|
||||
QueryEnvQuirkOverride("MOBILEGL_FORCE_VIEWPORT_ARRAY_EMULATION");
|
||||
}
|
||||
|
||||
inline void InitBackendType() {
|
||||
|
||||
+37
-4
@@ -9,10 +9,20 @@
|
||||
#pragma once
|
||||
|
||||
// ============== Platform-specific definitions and macros ============== //
|
||||
#ifdef __ANDROID__
|
||||
#undef __ANDROID_API__
|
||||
#define __ANDROID_API__ 26 // force Android API level to 26 for compatibility
|
||||
#endif
|
||||
// No __ANDROID_API__ pin here on purpose. The effective API level is owned by
|
||||
// the build system (gradle minSdk 26 -> -DANDROID_PLATFORM=android-26, enforced
|
||||
// by the configure-time guard in CMakeLists.txt), not by a macro.
|
||||
//
|
||||
// History: this used to `#define __ANDROID_API__ 26` to *raise* the level back
|
||||
// when the build configured something lower, so that pthread_getname_np (which
|
||||
// bionic guards with __INTRODUCED_IN(26)) would be declared. Once a later
|
||||
// change added an `#undef` in front of it, the same line started *lowering* the
|
||||
// level whenever the build configured higher than 26 - and that is an
|
||||
// include-order split-brain, not a compatibility knob: a TU that includes any
|
||||
// libc++ header before Includes.h latches libc++'s feature macros at the
|
||||
// configure-time level, and only the bionic headers pulled in afterwards see
|
||||
// the lowered value. The two halves then disagree (e.g. libc++ believes
|
||||
// pthread_cond_clockwait exists while bionic has since hidden its declaration).
|
||||
|
||||
#ifdef _WIN32
|
||||
#ifndef NOMINMAX
|
||||
@@ -37,6 +47,23 @@
|
||||
#define MOBILEGL_WGL_API MOBILEGL_API
|
||||
|
||||
// ====================== MobileGL configurations ======================= //
|
||||
// The numeric log levels live here, not only in Log.h: MOBILEGL_ASSERT below compares
|
||||
// MOBILEGL_LOG_ACTIVE_LEVEL against MOBILEGL_LOG_LEVEL_DEBUG, and in a translation unit
|
||||
// that includes Defines.h without Log.h both tokens would silently evaluate to 0 in the
|
||||
// preprocessor conditional - enabling the assert in exactly the INFO-level builds it is
|
||||
// documented to be compiled out of. Log.h redefines them identically, which is legal.
|
||||
//
|
||||
// Severity order, ascending: DEBUG < INFO < WARN < ERROR < FATAL. MOBILEGL_LOG_ACTIVE_LEVEL
|
||||
// names the lowest severity compiled in, so the production default INFO keeps I/W/E/F and
|
||||
// drops only D. Any edit here must be mirrored in Log.h.
|
||||
#ifndef MOBILEGL_LOG_LEVEL_DEBUG
|
||||
#define MOBILEGL_LOG_LEVEL_DEBUG 0
|
||||
#define MOBILEGL_LOG_LEVEL_INFO 1
|
||||
#define MOBILEGL_LOG_LEVEL_WARN 2
|
||||
#define MOBILEGL_LOG_LEVEL_ERROR 3
|
||||
#define MOBILEGL_LOG_LEVEL_FATAL 4
|
||||
#endif
|
||||
|
||||
#ifndef MOBILEGL_LOG_ACTIVE_LEVEL
|
||||
#define MOBILEGL_LOG_ACTIVE_LEVEL MOBILEGL_LOG_LEVEL_INFO
|
||||
#endif
|
||||
@@ -68,6 +95,12 @@
|
||||
#endif
|
||||
|
||||
// =============================== Utils ================================ //
|
||||
// Asserts are live in exactly the builds where MGLOG_D is live, i.e. DEBUG builds only;
|
||||
// an INFO build (the production default) compiles them out. DEBUG is the lowest severity
|
||||
// in the ordering above, so "ACTIVE <= DEBUG" is true only for ACTIVE == DEBUG - the same
|
||||
// gate MGLOG_D uses in Log.h. That equivalence is what makes this gate survive the
|
||||
// 2026-08-13 renumbering unchanged; the contract is and stays
|
||||
// "INFO builds: asserts OFF; DEBUG builds: asserts ON".
|
||||
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG
|
||||
#define MOBILEGL_ASSERT(condition, ...) \
|
||||
do { \
|
||||
|
||||
+2
-2
@@ -49,8 +49,8 @@
|
||||
#include <stacktrace>
|
||||
#endif
|
||||
|
||||
// Include FastSTL
|
||||
#include <FastSTL/UnorderedMap.h>
|
||||
// Include ska::flat_hash_map
|
||||
#include <ska/flat_hash_map.hpp>
|
||||
|
||||
// Include xxHash
|
||||
#include <xxhash.h>
|
||||
|
||||
@@ -15,8 +15,11 @@
|
||||
#include <MG_Impl/GLImpl/Texture/ProxyTexture.h>
|
||||
#include <MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h>
|
||||
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
||||
#include <MG_Impl/GLImpl/Query/GL_Query.h>
|
||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
#include <MG_State/GLState/ProgramState/ProgramTranslationCache.h>
|
||||
#include <MG_Util/ShaderTranspiler/TranslationCache.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
@@ -51,6 +54,11 @@ namespace MobileGL {
|
||||
// before a re-initialized library could pair them with the wrong
|
||||
// backend's DeleteSync).
|
||||
MG_Impl::GLImpl::DestroyAllSyncObjects();
|
||||
// Queries die with their contexts for the same reason, and their registry
|
||||
// is the same shape of process-global map: drain it here too, while the
|
||||
// function table can still pair each backend handle with the backend that
|
||||
// minted it.
|
||||
MG_Impl::GLImpl::DestroyAllQueryObjects();
|
||||
MG_Backend::pActiveBackendObject.reset();
|
||||
MG_State::pGLContext.reset();
|
||||
MG_State::pEGLContext.reset();
|
||||
@@ -66,6 +74,14 @@ namespace MobileGL {
|
||||
// built-in symbol tables the prewarm latch stands for, so leaving it set would
|
||||
// make the next Initialize() skip a prewarm it genuinely needs.
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::ResetPrewarmLatch();
|
||||
// The two-level translation memo. Nothing in it references a glslang object -
|
||||
// both levels hold plain bytes - so this is RSS hygiene rather than a lifetime
|
||||
// requirement, and it is safe either side of FinalizeProcess. Stats first: an
|
||||
// fordebug build gets one line per level saying how the run went.
|
||||
MG_Util::ShaderTranspiler::LogShaderTranslationCacheStats();
|
||||
MG_Util::ShaderTranspiler::ClearShaderTranslationCaches();
|
||||
MG_State::GLState::LogProgramTranslationCacheStats();
|
||||
MG_State::GLState::ClearProgramTranslationCache();
|
||||
MG_Backend::gBackendFunctionsTable = {};
|
||||
g_isInitialized = false;
|
||||
if (logLifecycle) {
|
||||
|
||||
@@ -14,6 +14,7 @@ namespace MobileGL {
|
||||
namespace MG_State::GLState {
|
||||
class FramebufferObject;
|
||||
class ITextureObject;
|
||||
class RenderbufferObject;
|
||||
}
|
||||
|
||||
enum class BackendType {
|
||||
@@ -24,6 +25,19 @@ namespace MobileGL {
|
||||
};
|
||||
|
||||
namespace MG_Backend {
|
||||
// One endpoint of a glCopyImageSubData. GL 4.6 core 18.3.2 accepts GL_RENDERBUFFER
|
||||
// alongside the ten whole-image texture targets, and a renderbuffer name lives in a
|
||||
// namespace of its own - so an endpoint is a sum type, not an ITextureObject. At most
|
||||
// one of the two pointers is set; neither is set when the name named nothing, which is
|
||||
// the INVALID_VALUE the frontend validator reports.
|
||||
struct CopyImageEndpoint {
|
||||
SharedPtr<MG_State::GLState::ITextureObject> Texture;
|
||||
SharedPtr<MG_State::GLState::RenderbufferObject> Renderbuffer;
|
||||
|
||||
Bool IsRenderbuffer() const { return Renderbuffer != nullptr; }
|
||||
Bool Exists() const { return Texture != nullptr || Renderbuffer != nullptr; }
|
||||
};
|
||||
|
||||
enum class FormatCapability : Uint64 {
|
||||
Creatable = 1ull << 0,
|
||||
|
||||
@@ -160,9 +174,9 @@ namespace MobileGL {
|
||||
GLsizei height, GLint border);
|
||||
void (*CopyTexSubImage2D)(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height);
|
||||
void (*CopyImageSubData)(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
void (*CopyImageSubData)(const CopyImageEndpoint& src,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
const CopyImageEndpoint& dst,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void (*GenerateMipmap)(GLenum target);
|
||||
@@ -236,6 +250,14 @@ namespace MobileGL {
|
||||
// (optional; null = frontend falls back to CPU accounting).
|
||||
BackendQueryHandle (*BeginXfbPrimitivesQuery)(Bool generated);
|
||||
void (*EndXfbPrimitivesQuery)(BackendQueryHandle query);
|
||||
// Whether GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN should be answered from the
|
||||
// frontend's own accounting wherever that accounting is exact - a capture with no
|
||||
// geometry stage - instead of from the query above. Set by DirectGLES, whose result
|
||||
// is whatever the ES driver's PRIMITIVES_WRITTEN counter says: Adreno reports twice
|
||||
// the written count for a vertex-only capture that follows a large render pass,
|
||||
// where the desktop-exact answer is the one the frontend already computed. Defaults
|
||||
// to false, so a backend that never sets it keeps using its GPU result.
|
||||
Bool PrefersCpuXfbPrimitiveAccounting = false;
|
||||
// Transform feedback capture spans, for backends whose own GL/ES driver
|
||||
// performs the capture (DirectGLES). Both optional; null means the backend
|
||||
// drives capture from its draw recording instead (DirectVulkan). End is
|
||||
@@ -279,6 +301,12 @@ namespace MobileGL {
|
||||
|
||||
struct DynamicBackendParameters {
|
||||
SizeT UniformBufferOffsetAlignment = 256;
|
||||
// GL_SHADER_STORAGE_BUFFER_OFFSET_ALIGNMENT, which is a SEPARATE limit from the
|
||||
// uniform one and is routinely larger: Adreno 830 reports 32 for uniform buffers and
|
||||
// 64 for storage buffers. Answering the storage query with the uniform value let an
|
||||
// application bind a storage range at an offset the driver cannot address, which it
|
||||
// accepted without error and then wrote somewhere else entirely.
|
||||
SizeT ShaderStorageBufferOffsetAlignment = 256;
|
||||
// GL_MAX_TEXTURE_MAX_ANISOTROPY_EXT. 1.0 means the backend cannot filter anisotropically,
|
||||
// which is also why the extension is not advertised in that case.
|
||||
Float MaxTextureMaxAnisotropy = 1.0f;
|
||||
@@ -318,6 +346,22 @@ namespace MobileGL {
|
||||
Int MaxVertexAttribs = 16;
|
||||
Int MaxComputeShaderStorageBlocks = 8;
|
||||
Int MaxCombinedShaderStorageBlocks = 32;
|
||||
// Per-stage GL_MAX_*_SHADER_STORAGE_BLOCKS. Zero is a legal answer for the four
|
||||
// non-compute, non-fragment stages and these defaults are the spec minimums, not
|
||||
// placeholders: GL 4.6 table 23.64 and ES 3.2 table 21.44 both set the minimum for
|
||||
// vertex, tessellation control, tessellation evaluation and geometry at 0, and only
|
||||
// fragment (8 in GL, 4 in ES) and compute are guaranteed to have any. Every real ARM
|
||||
// GLES driver takes that allowance - a Mali-G925 reports 0 for all four - so a
|
||||
// backend that cannot honour a graphics-stage storage block MUST report 0 here
|
||||
// rather than a hopeful number. Advertising a non-zero count the driver will refuse
|
||||
// does not make the block work; it only moves the failure from an honest
|
||||
// "unsupported" at query time to a backend link error the frontend never surfaces,
|
||||
// after which every draw with that program silently renders nothing.
|
||||
Int MaxVertexShaderStorageBlocks = 0;
|
||||
Int MaxTessControlShaderStorageBlocks = 0;
|
||||
Int MaxTessEvaluationShaderStorageBlocks = 0;
|
||||
Int MaxGeometryShaderStorageBlocks = 0;
|
||||
Int MaxFragmentShaderStorageBlocks = 8;
|
||||
Int MaxComputeUniformBlocks = 12;
|
||||
Int MaxComputeWorkGroupInvocations = 128;
|
||||
Int MaxShaderStorageBufferBindings = 8;
|
||||
@@ -334,8 +378,32 @@ namespace MobileGL {
|
||||
Int MaxComputeImageUniforms = 8;
|
||||
Int MaxDrawBuffers = 8;
|
||||
Int MaxColorAttachments = 8;
|
||||
// GL_MAX_CLIP_DISTANCES. Zero is a legal answer here, not a placeholder, and a
|
||||
// backend that cannot host a clip distance MUST report it: advertising eight the
|
||||
// backend will refuse does not make gl_ClipDistance work, it only moves the failure
|
||||
// from an honest "unsupported" at query time to a backend shader-compile error the
|
||||
// frontend never surfaces, after which every draw with that program silently renders
|
||||
// nothing. DirectGLES fills it from GL_EXT_clip_cull_distance, DirectVulkan from the
|
||||
// shaderClipDistance device feature. The DEFAULT stays at the GL 4.3 core minimum
|
||||
// because it describes the no-backend case (standalone shader compiles, unit tests),
|
||||
// where there is no device to be honest about and BuildTBuiltInResource still has to
|
||||
// hand glslang a workable gl_MaxClipDistances.
|
||||
Int MaxClipDistances = 8;
|
||||
Int MaxViewports = 16;
|
||||
// GL_LAYER_PROVOKING_VERTEX / GL_VIEWPORT_INDEX_PROVOKING_VERTEX: which vertex of a
|
||||
// primitive supplies gl_Layer and gl_ViewportIndex. GL 4.6 table 23.65 makes
|
||||
// GL_UNDEFINED_VERTEX a legal answer for both, and it is the honest default - naming
|
||||
// a convention is a statement about behaviour, so a backend that does not pin one
|
||||
// must not claim it does. DirectGLES fills the layer one from the ES 3.2 query and
|
||||
// the viewport one from GL_OES_viewport_array, and leaves UNDEFINED where the
|
||||
// capability is absent: without the viewport array extension only viewport 0 is ever
|
||||
// rasterized, so no convention selects anything. DirectVulkan keeps UNDEFINED for
|
||||
// both - which vertex provokes is decided per pipeline by
|
||||
// VulkanRenderer::SelectProvokingVertexMode out of VK_EXT_provoking_vertex,
|
||||
// provokingVertexModePerPipeline and the topology, so no single convention is true
|
||||
// of the backend.
|
||||
GLenum LayerProvokingVertex = GL_UNDEFINED_VERTEX;
|
||||
GLenum ViewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
|
||||
Int MaxViewportWidth = 16384;
|
||||
Int MaxViewportHeight = 16384;
|
||||
Float ViewportBoundsRangeMin = 0.0f;
|
||||
@@ -383,12 +451,36 @@ namespace MobileGL {
|
||||
const Uint32 bit = PerLayerFramebufferAttachmentBit(target);
|
||||
return bit != 0 && (PerLayerFramebufferAttachmentTargets & bit) != 0;
|
||||
}
|
||||
// Whether this backend can CONSUME a shader module that still declares 64-bit floats,
|
||||
// i.e. whether `double` survives the transpile instead of being narrowed to `float`
|
||||
// (ShaderTranspiler::DemoteFloat64Pass). Detected, never assumed:
|
||||
// * DirectVulkan sets it from VkPhysicalDeviceFeatures::shaderFloat64, the feature
|
||||
// VUID-VkShaderModuleCreateInfo-pCode-08740 requires before a module declaring
|
||||
// OpCapability Float64 may be created at all. lavapipe has it; Adreno and Mali
|
||||
// both report VK_FALSE, so no real mobile device does.
|
||||
// * DirectGLES can NEVER have it. GLSL ES has no 64-bit float type in any version
|
||||
// or extension, so SPIRV-Cross cannot emit one ("FP64 not supported in ES
|
||||
// profile") and the demotion there is mathematically mandatory, always.
|
||||
// Defaults to false so a backend that never sets it - and the no-backend case, which
|
||||
// is what standalone shader compiles and the unit tests run under - keeps the
|
||||
// demotion, which is the behaviour that works everywhere.
|
||||
Bool SupportsShaderFloat64 = false;
|
||||
// Whether glVertexAttribLFormat / glVertexArrayAttribLFormat can be honoured, i.e.
|
||||
// whether a 64-bit vertex attribute can actually reach a shader unconverted. Detected,
|
||||
// never assumed: DirectVulkan needs VkPhysicalDeviceFeatures::shaderFloat64 (the
|
||||
// attribute travels as its 32-bit word pair, so no VK_FORMAT_R64* is required, but the
|
||||
// bitcast result is Float64); DirectGLES can never have it, ESSL having no fp64 type at
|
||||
// all. Defaults to false so a backend that never sets it gets the conservative answer.
|
||||
//
|
||||
// INDEPENDENT of SupportsShaderFloat64, and it has to be: this flag decides a VkFormat
|
||||
// from the VAO ATTRIBUTE alone, which does not know what type the shader declared, and
|
||||
// glVertexAttribFormat(GL_DOUBLE) feeding a plain `in vec4` is both legal and common
|
||||
// (KHR-GL43.vertex_attrib_binding.basic-input-case4/5, advanced-bindingUpdate). A
|
||||
// backend with native fp64 that still cannot FETCH 64 bits keeps this false and relies
|
||||
// on the per-MODULE rule in ShaderCompiler::SanitizeAndOptimizeBinary instead: a vertex
|
||||
// module that declares a 64-bit float INPUT is demoted whole, so the two shader-side
|
||||
// halves (PackDoubleVertexInputsPass and VertexInputStateFactory::ToVkVertexFormat)
|
||||
// still see one consistent world.
|
||||
Bool SupportsFloat64VertexAttributes = false;
|
||||
SizeT MaxShaderStorageBlockSize = 128 * 1024 * 1024;
|
||||
Uint32 SubgroupSize = 0;
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "BackendObject_DirectGLES.h"
|
||||
#include "MG_Backend/BackendObject.h"
|
||||
#include "MG_Backend/BackendObjects.h"
|
||||
#include <MG_Backend/DirectGLES/DirectGLES.h>
|
||||
#include <MG_Backend/DirectGLES/Managers.h>
|
||||
#include <MG_Backend/DirectGLES/Utils.h>
|
||||
@@ -212,7 +213,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (options & PixelFormatNormalizeOptionBit::NoThreeChannelRenderTarget) {
|
||||
reasons.push_back("no colour-renderable three-channel format on OpenGL ES");
|
||||
}
|
||||
if (options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) {
|
||||
// A format is either 8- or 16-bit signed normalized, so at most one of the two ever
|
||||
// survives GetApplicablePixelFormatNormalizeOptions and the reason is not duplicated.
|
||||
if ((options & PixelFormatNormalizeOptionBit::NoSnorm16RenderTarget) ||
|
||||
(options & PixelFormatNormalizeOptionBit::NoSnorm8RenderTarget)) {
|
||||
reasons.push_back("EXT_render_snorm not supported");
|
||||
}
|
||||
|
||||
@@ -406,9 +410,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return complete;
|
||||
}
|
||||
|
||||
// `samples` only reaches the multisample targets; every other target ignores it. The
|
||||
// descending sample walk (ProbeTextureSampleCounts) reuses this whole routine rather than
|
||||
// repeating the gen/bind/completeness/delete dance.
|
||||
Bool ProbeTexture(const MG_External::GLESFunctionsTable& gl, TextureTarget target, GLenum internalFormat,
|
||||
GLenum imageFormat, GLenum imageType, TextureInternalFormat logicalFormat,
|
||||
Bool* outRenderable) {
|
||||
Bool* outRenderable, Int samples = 1) {
|
||||
if (!IsGLESProbeTextureTarget(target) || !gl.glGenTextures || !gl.glBindTexture || !gl.glDeleteTextures) {
|
||||
return false;
|
||||
}
|
||||
@@ -428,10 +435,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
const Bool isMultisample = IsGLESProbeMultisampleTarget(target);
|
||||
if (isMultisample) {
|
||||
const auto probeSamples = static_cast<GLsizei>(std::max(samples, 1));
|
||||
if (target == TextureTarget::Texture2DMultisample && gl.glTexStorage2DMultisample) {
|
||||
gl.glTexStorage2DMultisample(glTarget, 1, internalFormat, 1, 1, GL_TRUE);
|
||||
gl.glTexStorage2DMultisample(glTarget, probeSamples, internalFormat, 1, 1, GL_TRUE);
|
||||
} else if (target == TextureTarget::Texture2DMultisampleArray && gl.glTexStorage3DMultisample) {
|
||||
gl.glTexStorage3DMultisample(glTarget, 1, internalFormat, 1, 1, 1, GL_TRUE);
|
||||
gl.glTexStorage3DMultisample(glTarget, probeSamples, internalFormat, 1, 1, 1, GL_TRUE);
|
||||
} else {
|
||||
gl.glBindTexture(glTarget, static_cast<GLuint>(previousBinding));
|
||||
gl.glDeleteTextures(1, &texture);
|
||||
@@ -527,6 +535,29 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return sampleCounts;
|
||||
}
|
||||
|
||||
// The multisample TEXTURE twin of ProbeRenderbufferSampleCounts. It used to be a
|
||||
// hardcoded {1}, which made glGetInternalformativ(GL_SAMPLES) claim a one-sample maximum
|
||||
// for every format on the multisample targets even where glTexImage2DMultisample happily
|
||||
// accepts four - GL 4.6 core 8.8 makes that query the definition of the maximum, so the
|
||||
// two answers cannot both be right. Completeness is required at every count, exactly as
|
||||
// the renderbuffer walk requires it; the caller only reaches here once the one-sample
|
||||
// probe has already succeeded, so 1 terminates the list without being re-probed.
|
||||
Vector<Int> ProbeTextureSampleCounts(const MG_External::GLESFunctionsTable& gl, TextureTarget target,
|
||||
GLenum internalFormat, GLenum imageFormat, GLenum imageType,
|
||||
TextureInternalFormat logicalFormat, Int maxSamples) {
|
||||
Vector<Int> sampleCounts;
|
||||
for (Int samples = std::max(maxSamples, 1); samples > 1; samples >>= 1) {
|
||||
Bool renderable = false;
|
||||
const Bool created = ProbeTexture(gl, target, internalFormat, imageFormat, imageType, logicalFormat,
|
||||
&renderable, samples);
|
||||
if (created && renderable) {
|
||||
sampleCounts.push_back(samples);
|
||||
}
|
||||
}
|
||||
sampleCounts.push_back(1);
|
||||
return sampleCounts;
|
||||
}
|
||||
|
||||
void PopulateFormatCapabilitiesImpl(const MG_External::GLESFunctionsTable& gl,
|
||||
const MG_External::GLESCapabilities& capabilities,
|
||||
FormatCapabilityCache& cache) {
|
||||
@@ -627,7 +658,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
AddFullFormatCaps(cache, targetIndex, formatIndex,
|
||||
BuildTextureCapsFromProbe(logicalFormat, target, nativeRenderable));
|
||||
if (IsGLESProbeMultisampleTarget(target)) {
|
||||
cache.SampleCounts[targetIndex][formatIndex] = {1};
|
||||
const Int maxSamples =
|
||||
GetGLESFormatMaxSamples(capabilities, logicalFormat, nativeInfo.ImageFormat);
|
||||
cache.SampleCounts[targetIndex][formatIndex] = ProbeTextureSampleCounts(
|
||||
gl, probeTarget, nativeInfo.InternalFormat, nativeInfo.ImageFormat,
|
||||
nativeInfo.ImageType, logicalFormat, maxSamples);
|
||||
}
|
||||
}
|
||||
shouldProbeFallback = !nativeCreated || !nativeRenderable;
|
||||
@@ -645,7 +680,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
LogGLESFormatCaveat(logicalFormat, targetIndex, fallbackInfo);
|
||||
}
|
||||
if (IsGLESProbeMultisampleTarget(target)) {
|
||||
cache.SampleCounts[targetIndex][formatIndex] = {1};
|
||||
const Int maxSamples =
|
||||
GetGLESFormatMaxSamples(capabilities, logicalFormat, fallbackInfo.ImageFormat);
|
||||
cache.SampleCounts[targetIndex][formatIndex] = ProbeTextureSampleCounts(
|
||||
gl, probeTarget, fallbackInfo.InternalFormat, fallbackInfo.ImageFormat,
|
||||
fallbackInfo.ImageType, logicalFormat, maxSamples);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -710,11 +749,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
.ExtraVendor = Nullopt, // Extra vendor
|
||||
.RendererGLInfo =
|
||||
{
|
||||
.TargetGLVersion = {4, 0, 0}, // GL target version
|
||||
.TargetGLVersion = {4, 3, 0}, // GL target version
|
||||
.TargetGLSLVersion = {4, 6, 0}, // Target Shading Language Version
|
||||
// Baseline advertisement (no timer queries / anisotropy yet); reconciled
|
||||
// once the ES capabilities exist, see UpdateAdvertisedCapabilityExtensions.
|
||||
.Extensions = BuildAdvertisedExtensions(false, false),
|
||||
// Baseline advertisement (no runtime capabilities yet); reconciled once
|
||||
// the ES capabilities exist, see UpdateAdvertisedCapabilityExtensions.
|
||||
.Extensions = BuildAdvertisedExtensions(false, false, false, false, false, false),
|
||||
.IsCompatibilityProfile = false // Is Compatibility Profile
|
||||
},
|
||||
.StaticBackendCapability = {.AllowVSOnlyPrograms = false} // Backend Capability
|
||||
@@ -734,9 +773,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// thread can only observe the extension string after the
|
||||
// advertisement for its context has settled; rebuilding the whole
|
||||
// list keeps the re-run after a context recreation idempotent.
|
||||
void UpdateAdvertisedCapabilityExtensions(Bool anisotropicFilteringSupported) {
|
||||
MutableRendererInfo().RendererGLInfo.Extensions =
|
||||
BuildAdvertisedExtensions(AreTimerQueriesSupported(), anisotropicFilteringSupported);
|
||||
void UpdateAdvertisedCapabilityExtensions(const MG_External::GLESCapabilities& capabilities) {
|
||||
MutableRendererInfo().RendererGLInfo.Extensions = BuildAdvertisedExtensions(
|
||||
AreTimerQueriesSupported(), capabilities.SupportsTextureFilterAnisotropy,
|
||||
capabilities.SupportsDrawIndirect,
|
||||
capabilities.SupportsDrawIndirect && capabilities.SupportsBaseInstance,
|
||||
capabilities.SupportsTextureView, capabilities.SupportsTextureCubeMapArray);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
@@ -745,6 +787,29 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
PopulateFormatCapabilitiesImpl(gl, capabilities, cache);
|
||||
}
|
||||
|
||||
Int ClampSamplesToBackendSupport(SizeT targetIndex, TextureInternalFormat logicalFormat, GLenum imageFormat,
|
||||
Int samples) {
|
||||
if (samples <= 1) {
|
||||
return samples;
|
||||
}
|
||||
|
||||
Int maxSamples = 0;
|
||||
const SizeT formatIndex = static_cast<SizeT>(logicalFormat);
|
||||
if (pActiveBackendObject && targetIndex < kFormatCapabilityTargetCount &&
|
||||
formatIndex < kFormatCapabilityFormatCount) {
|
||||
// Descending, so the head is the largest count this device actually allocated.
|
||||
const Vector<Int>& probedCounts =
|
||||
pActiveBackendObject->GetFormatCapabilities().SampleCounts[targetIndex][formatIndex];
|
||||
if (!probedCounts.empty()) {
|
||||
maxSamples = probedCounts.front();
|
||||
}
|
||||
}
|
||||
if (maxSamples <= 0) {
|
||||
maxSamples = GetGLESFormatMaxSamples(g_GLESCapabilities, logicalFormat, imageFormat);
|
||||
}
|
||||
return std::min(samples, std::max(maxSamples, 1));
|
||||
}
|
||||
|
||||
BackendObject_DirectGLES::~BackendObject_DirectGLES() {
|
||||
DestroyEGLContext();
|
||||
}
|
||||
@@ -779,11 +844,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return false;
|
||||
}
|
||||
DirectGLES::SetGLESCapabilities(m_GLESCapabilities);
|
||||
// Now that g_GLESCapabilities knows about GL_EXT_disjoint_timer_query and
|
||||
// GL_EXT_texture_filter_anisotropic, reconcile the advertisement (see the comment on
|
||||
// UpdateAdvertisedCapabilityExtensions for why it cannot happen when the extension
|
||||
// list is first built).
|
||||
UpdateAdvertisedCapabilityExtensions(m_GLESCapabilities.SupportsTextureFilterAnisotropy);
|
||||
// Now that g_GLESCapabilities knows the host extensions, entry points, and ES version,
|
||||
// reconcile every runtime-gated advertisement (see the comment on
|
||||
// UpdateAdvertisedCapabilityExtensions for why this cannot happen when the list is first
|
||||
// built).
|
||||
UpdateAdvertisedCapabilityExtensions(m_GLESCapabilities);
|
||||
UpdateDynamicBackendParameters();
|
||||
PopulateFormatCapabilities(m_GLESFunctions, m_GLESCapabilities, MutableFormatCapabilities());
|
||||
PrintFormatCapabilities(GetFormatCapabilities());
|
||||
@@ -924,11 +989,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return MutableRendererInfo();
|
||||
}
|
||||
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported) {
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported,
|
||||
Bool drawIndirectSupported,
|
||||
Bool nonZeroIndirectBaseInstanceSupported,
|
||||
Bool textureViewSupported, Bool cubeMapArraySupported) {
|
||||
Vector<GLExtension> extensions = {
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
|
||||
// The version tokens have to reach the version the backend actually claims:
|
||||
// TargetGLVersion is {4,3,0}, and a list that stopped at OpenGL40 told an
|
||||
// application feature-detecting off these tokens the opposite of what
|
||||
// GL_MAJOR_VERSION / GL_MINOR_VERSION told it.
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, V_OpenGL41, V_OpenGL42, V_OpenGL43,
|
||||
E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
|
||||
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_EXT_framebuffer_object,
|
||||
E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage, E_GL_ARB_texture_storage,
|
||||
E_GL_ARB_texture_storage_multisample, E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access,
|
||||
E_GL_ARB_multi_draw_indirect, E_GL_ARB_indirect_parameters, E_GL_ARB_shader_draw_parameters,
|
||||
@@ -940,10 +1013,105 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// picks a whole different shader for draw_buffers without
|
||||
// explicit_attrib_location. DirectVulkan advertises both.
|
||||
E_GL_ARB_explicit_attrib_location, E_GL_ARB_texture_multisample, E_GL_ARB_shader_image_size,
|
||||
// Core since GL 3.1 and implemented for every version advertised here. The string
|
||||
// matters because applications gate the ENTRY POINTS on it rather than on the
|
||||
// version: a caller that finds the extension missing never resolves
|
||||
// glGetUniformBlockIndex / glUniformBlockBinding, and one that then uses uniform
|
||||
// blocks anyway calls through a null pointer.
|
||||
E_GL_ARB_uniform_buffer_object,
|
||||
// Sampling the stencil aspect through DEPTH_STENCIL_TEXTURE_MODE. Core from 4.3,
|
||||
// so on a 4.0 context the string is the only way to reach it. The host ES driver
|
||||
// has had the same texture parameter since ES 3.1, which every device MobileGL
|
||||
// runs on provides.
|
||||
E_GL_ARB_stencil_texturing,
|
||||
// Core since 3.2 and implemented here on both backends - glDrawElementsBaseVertex,
|
||||
// glDrawRangeElementsBaseVertex, glDrawElementsInstancedBaseVertex and
|
||||
// glMultiDrawElementsBaseVertex all reach real per-draw vertex rebasing. The string
|
||||
// was simply never emitted, which left KHR-GL4*.draw_elements_base_vertex_tests
|
||||
// NotSupported on a feature that works.
|
||||
E_GL_ARB_draw_elements_base_vertex,
|
||||
// The whole sync-object family is real and core since 3.2: glFenceSync, glIsSync,
|
||||
// glDeleteSync, glClientWaitSync, glWaitSync and glGetSynciv all live in GLImpl over a
|
||||
// backend fence (a host GLsync here, a VkFence on DirectVulkan), and glGetInteger64v
|
||||
// answers GL_MAX_SERVER_WAIT_TIMEOUT. The string matters for the same reason
|
||||
// ARB_uniform_buffer_object's does: LWJGL builds GLCapabilities from the extension
|
||||
// list, and a caller that finds GL_ARB_sync missing never resolves the entry points -
|
||||
// then calls through null if it uses fences anyway. Nothing in the CTS gates on this
|
||||
// string, so it is advertised on the strength of the implementation, not a test unlock.
|
||||
E_GL_ARB_sync,
|
||||
// Atomic counters, core since 4.2. glGetActiveAtomicCounterBufferiv and the whole
|
||||
// GL_ATOMIC_COUNTER_BUFFER_* query family are real in GLImpl, and SyncAtomicCounterBuffers
|
||||
// re-issues the counter buffer as an SSBO binding in the range reserved at the top of
|
||||
// the ES driver's shader-storage points, so a counter dispatch reads and writes the
|
||||
// buffer the application bound. DirectVulkan reaches the same place through its own
|
||||
// descriptor resolution, so the string is symmetric.
|
||||
E_GL_ARB_shader_atomic_counters,
|
||||
// glVertexAttribDivisor, core since 3.3 and real on both backends. Applications
|
||||
// (Better Clouds' GLCompat among them) accept the extension string as an
|
||||
// ALTERNATIVE to a 3.3 context when deciding whether instanced rendering is
|
||||
// available, so withholding it makes MobileGL look less capable than it is.
|
||||
E_GL_ARB_instanced_arrays,
|
||||
// The whole of KHR_debug lives in GLImpl - the message log, the group stack and the
|
||||
// object-label table are MobileGL's own state, not the host driver's - so it is as
|
||||
// available here as it is on DirectVulkan, which has advertised it all along.
|
||||
E_GL_KHR_debug,
|
||||
// Core GL 3.0-4.3 plumbing that has been real here for as long as the backend has
|
||||
// existed, and that was simply never named. None of these unlocks a single CTS case -
|
||||
// the conformance suite reaches all of them through the version - so they are
|
||||
// advertised for the OTHER consumer of this list: LWJGL builds GLCapabilities from the
|
||||
// string set, and an application that gates its ENTRY POINTS on the string rather than
|
||||
// on the version never resolves them and then calls through null. Each is backed by
|
||||
// the entry points named beside it.
|
||||
//
|
||||
// glBindVertexArray / glGenVertexArrays / glDeleteVertexArrays / glIsVertexArray.
|
||||
E_GL_ARB_vertex_array_object,
|
||||
// The 14 glSamplerParameter* / glGetSamplerParameter* entry points, including the
|
||||
// integer-valued Iiv/Iuiv forms.
|
||||
E_GL_ARB_sampler_objects,
|
||||
// glMapBufferRange + glFlushMappedBufferRange, which ARB_buffer_storage's persistent
|
||||
// maps are already built on top of.
|
||||
E_GL_ARB_map_buffer_range,
|
||||
// glCopyBufferSubData plus the GL_COPY_READ_BUFFER / GL_COPY_WRITE_BUFFER targets.
|
||||
E_GL_ARB_copy_buffer,
|
||||
// glCopyImageSubData, wired to a real backend hook on both backends.
|
||||
E_GL_ARB_copy_image,
|
||||
// GL_TEXTURE_SWIZZLE_{R,G,B,A,RGBA}, which this backend syncs through to the ES
|
||||
// driver's identical parameters.
|
||||
E_GL_ARB_texture_swizzle,
|
||||
// GL_INT_2_10_10_10_REV / GL_UNSIGNED_INT_2_10_10_10_REV on glVertexAttribPointer plus
|
||||
// the eight glVertexAttribP* entry points.
|
||||
E_GL_ARB_vertex_type_2_10_10_10_rev,
|
||||
// The R/RG internal formats. Named separately from the float ones because an
|
||||
// application may check either.
|
||||
E_GL_ARB_texture_rg,
|
||||
// GL_DEPTH_COMPONENT32F and GL_DEPTH32F_STENCIL8.
|
||||
E_GL_ARB_depth_buffer_float,
|
||||
// The floating-point colour formats. Unlike the rest of this block this string DOES
|
||||
// gate CTS cases - KHR-GL4*.internalformat.texture2d.*{16f,32f} is keyed on it with no
|
||||
// core-version fallback, so eight cases per version list were NotSupported on formats
|
||||
// the backend has always had.
|
||||
E_GL_ARB_texture_float,
|
||||
// glViewportArrayv / glViewportIndexedf{,v} / glScissorArrayv / glScissorIndexed{,v} /
|
||||
// glDepthRangeArrayv / glDepthRangeIndexed / glGetFloati_v / glGetDoublei_v, over the
|
||||
// 16 viewports GL_MAX_VIEWPORTS reports and the per-viewport routing emulation.
|
||||
E_GL_ARB_viewport_array,
|
||||
// Advertised with GL_NUM_PROGRAM_BINARY_FORMATS = 0, which the
|
||||
// extension explicitly permits. It is also the only thing that
|
||||
// exposes glProgramParameteri before GL 4.1.
|
||||
E_GL_ARB_get_program_binary};
|
||||
// Minecraft 26.3 checks this prerequisite before it even considers
|
||||
// GL_ARB_multi_draw_indirect. ES 3.1 supplies both single-draw entry points; the loader
|
||||
// folds the version and pointer checks into SupportsDrawIndirect.
|
||||
if (drawIndirectSupported) {
|
||||
extensions.push_back(E_GL_ARB_draw_indirect);
|
||||
}
|
||||
// ARB_base_instance also defines the last word of an indirect command. Direct calls are
|
||||
// emulated on every Espryt device, but without host GL_EXT_base_instance a native indirect
|
||||
// draw cannot shift divisor attributes by a GPU-authored non-zero value, so do not promise
|
||||
// that incomplete case.
|
||||
if (drawIndirectSupported && nonZeroIndirectBaseInstanceSupported) {
|
||||
extensions.push_back(E_GL_ARB_base_instance);
|
||||
}
|
||||
// GL_KHR_parallel_shader_compile is MobileGL's own capability, not the host ES
|
||||
// driver's: the compiler threads are MobileGL's, and glCompileShader/glLinkProgram
|
||||
// are serviced entirely inside the frontend. Whether the device driver advertises
|
||||
@@ -960,12 +1128,62 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
if (MG_Util::Async::AsyncShaderCompileEnabled()) {
|
||||
extensions.push_back(E_GL_KHR_parallel_shader_compile);
|
||||
}
|
||||
// GL_ARB_gpu_shader_fp64 is opt-in (MOBILEGL_ADVERTISE_FP64). Every `double` in a
|
||||
// shader compiles and runs already - it is narrowed to 32 bits before the module
|
||||
// reaches this backend - so an application that simply uses doubles needs nothing
|
||||
// advertised. What the extension additionally promises is 64-bit PRECISION, which no
|
||||
// mobile GPU has and the narrowing cannot fake, so advertising it by default would
|
||||
// make an application that checks the string take a path MobileGL cannot honour.
|
||||
if (MG_Config::Features.AdvertiseFp64) {
|
||||
extensions.push_back(E_GL_ARB_gpu_shader_fp64);
|
||||
}
|
||||
// Only advertised when the device driver actually has usable timer queries
|
||||
// (GL_EXT_disjoint_timer_query plus its entry points) and the
|
||||
// MOBILEGL_DISABLE_TIMERQUERY escape hatch is off.
|
||||
if (timerQueriesSupported && !MG_Config::Features.DisableTimerQuery) {
|
||||
extensions.push_back(E_GL_ARB_timer_query);
|
||||
}
|
||||
// Cube map arrays are core from GL 4.0 and from ES 3.2, but on a pre-ES-3.2 driver without
|
||||
// EXT/OES_texture_cube_map_array there is nothing underneath: the texture gets no storage
|
||||
// and a samplerCubeArray shader does not even compile, which is exactly what the POST
|
||||
// reports. So the string follows the host capability rather than the version.
|
||||
//
|
||||
// Named for the application's benefit rather than the suite's: measured on Adreno 830,
|
||||
// KHR-GL43.texture_gather.plain-gather-*-cube-array already passed without the string, so
|
||||
// this unlocks no conformance case. It is advertised because the feature is real and
|
||||
// because an application that feature-detects cube map arrays off the string (rather than
|
||||
// off the 4.0 version) would otherwise decline a path this backend serves.
|
||||
if (cubeMapArraySupported) {
|
||||
extensions.push_back(E_GL_ARB_texture_cube_map_array);
|
||||
}
|
||||
// Only advertised when the host ES driver has EXT/OES_texture_view. ES has no core
|
||||
// texture views at any version and no honest emulation exists: a view is a SECOND NAME
|
||||
// over the SAME storage, so that writes through either are visible through the other and
|
||||
// the two carry independent per-texture parameters at the same time - which is exactly
|
||||
// what applications use it for (Better Clouds samples one D24S8 through its own name with
|
||||
// DEPTH_STENCIL_TEXTURE_MODE = STENCIL_INDEX and through a view with DEPTH_COMPONENT, in
|
||||
// a single shading pass). A copy-based fallback satisfies neither half, and fails
|
||||
// silently; withholding the string and answering glTextureView with INVALID_OPERATION is
|
||||
// the only behaviour that cannot be mistaken for success.
|
||||
//
|
||||
// The host extension is necessary and NOT sufficient, which is why this second gate
|
||||
// exists. Adreno 830 has EXT_texture_view, and on it the whole functional half of
|
||||
// KHR-GL4{2,3}.texture_view fails: base_and_max_levels, reference_counting and
|
||||
// view_sampling Fail and view_classes crashes, while only the two pure-API cases
|
||||
// (errors, gettexparameter - neither of which touches the host view) pass. The cause is
|
||||
// known and is MobileGL's, not the driver's: SyncTextureViewToBackend normalizes the
|
||||
// VIEW's ES internalformat independently of the storage it aliases, so whenever the two
|
||||
// land on different renderability carriers the host rejects the pair, the error is
|
||||
// swallowed, and the view is left as a storage-less name that samples as zeros.
|
||||
// DirectVulkan builds the view as a second VkImageView over one VkImage and has no such
|
||||
// seam - it passes 5 of the 7 cases on the same device - so the string stays there.
|
||||
//
|
||||
// Until that reconciliation exists, advertising here would be the same lie the comment
|
||||
// above refuses to tell, just with an extra prerequisite met. Set
|
||||
// MOBILEGL_ENABLE_GLES_TEXTURE_VIEW=1 to re-enable it for that work.
|
||||
if (textureViewSupported && MG_Config::Features.EnableGlesTextureView) {
|
||||
extensions.push_back(E_GL_ARB_texture_view);
|
||||
}
|
||||
// Only advertised when the host ES driver actually filters anisotropically: the sampler
|
||||
// state is accepted regardless, but forwarding it would be a no-op without the extension,
|
||||
// and an app that trusts the string (LWJGL builds GLCapabilities from it) would silently
|
||||
@@ -1002,6 +1220,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
funcsTable.GL.MultiDrawElementsIndirect = MultiDrawElementsIndirect;
|
||||
funcsTable.GL.MultiDrawElementsIndirectCount = MultiDrawElementsIndirectCount;
|
||||
funcsTable.GL.MultiDrawArraysIndirect = MultiDrawArraysIndirect;
|
||||
funcsTable.GL.MultiDrawArraysIndirectCount = MultiDrawArraysIndirectCount;
|
||||
funcsTable.GL.DrawRangeElementsBaseVertex = DrawRangeElementsBaseVertex;
|
||||
funcsTable.GL.DrawRangeElements = DrawRangeElements;
|
||||
funcsTable.GL.DrawElementsInstancedBaseVertexBaseInstance = DrawElementsInstancedBaseVertexBaseInstance;
|
||||
@@ -1069,6 +1288,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// geometry shader's amplification.
|
||||
funcsTable.GL.BeginXfbPrimitivesQuery = BeginXfbPrimitivesQuery;
|
||||
funcsTable.GL.EndXfbPrimitivesQuery = EndXfbPrimitivesQuery;
|
||||
// ...but where it CAN see the whole capture - no geometry stage - the frontend's
|
||||
// own count is the desktop-exact one and the ES driver's is only as good as the
|
||||
// vendor made it (Adreno doubles PRIMITIVES_WRITTEN for a vertex-only capture that
|
||||
// follows a large render pass). The query above stays installed: it is still what
|
||||
// answers an amplifying span, and PRIMITIVES_GENERATED always.
|
||||
funcsTable.GL.PrefersCpuXfbPrimitiveAccounting = true;
|
||||
funcsTable.GL.IsQueryResultAvailable = IsQueryResultAvailable;
|
||||
funcsTable.GL.GetQueryResult64 = GetQueryResult64;
|
||||
funcsTable.GL.DeleteBackendQuery = DeleteBackendQuery;
|
||||
@@ -1098,6 +1323,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
void BackendObject_DirectGLES::UpdateDynamicBackendParameters() {
|
||||
m_dynamicParameters.UniformBufferOffsetAlignment = m_GLESCapabilities.UniformBufferOffsetAlignment;
|
||||
m_dynamicParameters.ShaderStorageBufferOffsetAlignment =
|
||||
m_GLESCapabilities.ShaderStorageBufferOffsetAlignment;
|
||||
m_dynamicParameters.MaxTextureMaxAnisotropy = m_GLESCapabilities.MaxTextureMaxAnisotropy;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMin = m_GLESCapabilities.AliasedLineWidthRangeMin;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMax = m_GLESCapabilities.AliasedLineWidthRangeMax;
|
||||
@@ -1148,9 +1375,41 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
static_cast<Int>(MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS));
|
||||
m_dynamicParameters.MaxComputeShaderStorageBlocks = m_GLESCapabilities.MaxComputeShaderStorageBlocks;
|
||||
m_dynamicParameters.MaxCombinedShaderStorageBlocks = m_GLESCapabilities.MaxCombinedShaderStorageBlocks;
|
||||
// Per-stage storage-block counts, forwarded from the host driver rather than invented.
|
||||
// A stage the driver cannot serve reports 0, which is a legal answer everywhere these
|
||||
// limits appear (GL 4.6 table 23.64, ES 3.2 table 21.44 - the minimum is 0 for every
|
||||
// graphics stage except fragment) and is the only answer that lets an application take
|
||||
// its own fallback instead of building a program the driver will refuse to link. The
|
||||
// stage limit cannot exceed the combined limit or the number of binding points there
|
||||
// are to bind buffers to, so clamp to both.
|
||||
const auto clampStageStorageBlocks = [this](Int stageLimit) {
|
||||
return std::min({std::max(stageLimit, 0), std::max(m_dynamicParameters.MaxCombinedShaderStorageBlocks, 0),
|
||||
std::max(m_dynamicParameters.MaxShaderStorageBufferBindings, 0)});
|
||||
};
|
||||
m_dynamicParameters.MaxShaderStorageBufferBindings = m_GLESCapabilities.MaxShaderStorageBufferBindings;
|
||||
m_dynamicParameters.MaxVertexShaderStorageBlocks =
|
||||
clampStageStorageBlocks(m_GLESCapabilities.MaxVertexShaderStorageBlocks);
|
||||
m_dynamicParameters.MaxTessControlShaderStorageBlocks =
|
||||
clampStageStorageBlocks(m_GLESCapabilities.MaxTessControlShaderStorageBlocks);
|
||||
m_dynamicParameters.MaxTessEvaluationShaderStorageBlocks =
|
||||
clampStageStorageBlocks(m_GLESCapabilities.MaxTessEvaluationShaderStorageBlocks);
|
||||
m_dynamicParameters.MaxGeometryShaderStorageBlocks =
|
||||
clampStageStorageBlocks(m_GLESCapabilities.MaxGeometryShaderStorageBlocks);
|
||||
m_dynamicParameters.MaxFragmentShaderStorageBlocks =
|
||||
clampStageStorageBlocks(m_GLESCapabilities.MaxFragmentShaderStorageBlocks);
|
||||
m_dynamicParameters.MaxComputeUniformBlocks = m_GLESCapabilities.MaxComputeUniformBlocks;
|
||||
m_dynamicParameters.MaxComputeWorkGroupInvocations = m_GLESCapabilities.MaxComputeWorkGroupInvocations;
|
||||
m_dynamicParameters.MaxShaderStorageBufferBindings = m_GLESCapabilities.MaxShaderStorageBufferBindings;
|
||||
// (MaxShaderStorageBufferBindings is assigned above, before the per-stage clamp reads it.)
|
||||
// This is the number glGetIntegerv(GL_MAX_TEXTURE_BUFFER_SIZE) hands the application, and
|
||||
// on a host without buffer textures it is knowingly a floor MobileGL cannot honour rather
|
||||
// than a driver answer (m_GLESCapabilities.MaxTextureBufferSizeIsDriverReported says
|
||||
// which). Reporting 0 instead was considered and rejected: MobileGL advertises an OpenGL
|
||||
// 4.x context, where buffer textures are core and the limit has a spec minimum of 65536,
|
||||
// so 0 is not a legal answer and applications are not written to survive it. GL offers no
|
||||
// way to say "this core feature is missing", so the honesty is carried outside the limit:
|
||||
// FillInGLESCapabilities logs the tier, glTexBuffer and the program build each name the
|
||||
// missing capability at MGLOG_I, and the driver POST carries a "Buffer textures" row that
|
||||
// FAILs on this tier.
|
||||
m_dynamicParameters.MaxTextureBufferSize = m_GLESCapabilities.MaxTextureBufferSize;
|
||||
m_dynamicParameters.TextureBufferOffsetAlignment = m_GLESCapabilities.TextureBufferOffsetAlignment;
|
||||
m_dynamicParameters.MaxUniformBufferBindings = m_GLESCapabilities.MaxUniformBufferBindings;
|
||||
@@ -1193,14 +1452,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
DynParams::PerLayerFramebufferAttachmentBit(TextureTarget::TextureCubeMapArray);
|
||||
}
|
||||
}
|
||||
// Not a driver question and never will be: OpenGL ES has no double-precision vertex format
|
||||
// and ESSL has no fp64 type to consume one with, so a 64-bit vertex attribute has nowhere to
|
||||
// land on this backend regardless of what the driver underneath happens to support.
|
||||
// Not a driver question and never will be: GLSL ES has no 64-bit float type in ANY version
|
||||
// or extension, so SPIRV-Cross cannot emit one ("FP64 not supported in ES profile") and a
|
||||
// module that still declared Float64 would never reach the driver at all. The demotion is
|
||||
// mathematically mandatory here, on every device, forever - which is why this stays false
|
||||
// regardless of what the driver underneath happens to support.
|
||||
m_dynamicParameters.SupportsShaderFloat64 = false;
|
||||
// Follows the line above, and must: OpenGL ES has no double-precision vertex format and no
|
||||
// fp64 type to consume one with, so a 64-bit vertex attribute has nowhere to land here.
|
||||
m_dynamicParameters.SupportsFloat64VertexAttributes = false;
|
||||
m_dynamicParameters.MaxDrawBuffers = m_GLESCapabilities.MaxDrawBuffers;
|
||||
m_dynamicParameters.MaxColorAttachments = m_GLESCapabilities.MaxColorAttachments;
|
||||
m_dynamicParameters.MaxClipDistances = m_GLESCapabilities.MaxClipDistances;
|
||||
m_dynamicParameters.MaxViewports = m_GLESCapabilities.MaxViewports;
|
||||
// Whatever the driver said about which vertex supplies gl_Layer, and GL_UNDEFINED_VERTEX
|
||||
// for gl_ViewportIndex on every driver without GL_OES_viewport_array - which is both test
|
||||
// devices. That is not a shortfall being hidden: without the extension only viewport 0 is
|
||||
// ever rasterized, so no vertex "selects" a viewport index and naming a convention would
|
||||
// describe behaviour this backend does not implement.
|
||||
m_dynamicParameters.LayerProvokingVertex = m_GLESCapabilities.LayerProvokingVertex;
|
||||
m_dynamicParameters.ViewportIndexProvokingVertex = m_GLESCapabilities.ViewportIndexProvokingVertex;
|
||||
m_dynamicParameters.MaxViewportWidth = m_GLESCapabilities.MaxViewportWidth;
|
||||
m_dynamicParameters.MaxViewportHeight = m_GLESCapabilities.MaxViewportHeight;
|
||||
m_dynamicParameters.ViewportBoundsRangeMin = m_GLESCapabilities.ViewportBoundsRangeMin;
|
||||
|
||||
@@ -18,6 +18,16 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const MG_External::GLESCapabilities& capabilities,
|
||||
FormatCapabilityCache& cache);
|
||||
|
||||
// Clamps a requested sample count down to what the ES driver can really deliver for this
|
||||
// format on this format-capability target: the probed per-format list when there is one, the
|
||||
// driver's per-class GL_MAX_*_SAMPLES otherwise. The frontend deliberately validates against
|
||||
// the count MobileGL advertises instead (GL_Getter's GetAdvertisedMaxSamples), which on a
|
||||
// driver reporting GL_MAX_INTEGER_SAMPLES 1 is higher than the driver accepts, so every ES
|
||||
// allocation call has to come through here. The shadow state keeps the requested count, so
|
||||
// GL_TEXTURE_SAMPLES and framebuffer completeness still answer what the application asked for.
|
||||
Int ClampSamplesToBackendSupport(SizeT targetIndex, TextureInternalFormat logicalFormat, GLenum imageFormat,
|
||||
Int samples);
|
||||
|
||||
class BackendObject_DirectGLES : public BackendObject {
|
||||
public:
|
||||
~BackendObject_DirectGLES() override;
|
||||
@@ -67,9 +77,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
const RendererInfo& GetRendererIdentity();
|
||||
|
||||
// The full OpenGL extension list Espryt advertises (glGetString(GL_EXTENSIONS))
|
||||
// for a device whose timer queries / anisotropic filtering are (or are not) usable.
|
||||
// for a device whose timer queries / anisotropic filtering / native indirect draws /
|
||||
// non-zero indirect baseInstance semantics / EXT-OES texture views are (or are not) usable.
|
||||
// The MOBILEGL_DISABLE_TIMERQUERY escape hatch is applied inside.
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported);
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported,
|
||||
Bool drawIndirectSupported,
|
||||
Bool nonZeroIndirectBaseInstanceSupported,
|
||||
Bool textureViewSupported, Bool cubeMapArraySupported);
|
||||
|
||||
// Format: <OpenGL ES Renderer>, OpenGL ES <Major>.<Minor> — the exact string an
|
||||
// initialized backend returns from GetBackendAPIVersionString (and that ends up
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -40,6 +40,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount, GLsizei maxdrawcount,
|
||||
GLsizei stride);
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex);
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices);
|
||||
@@ -74,9 +76,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
GLsizei height, GLint border);
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height);
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
void CopyImageSubData(const CopyImageEndpoint& src,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
const CopyImageEndpoint& dst,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void GenerateMipmap(GLenum target);
|
||||
@@ -117,6 +119,24 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// capability read needs no current ES context, and it stays false until
|
||||
// the ES capabilities have been filled in.
|
||||
Bool AreTimerQueriesSupported();
|
||||
// True when the host ES driver can back a GL_TEXTURE_BUFFER at all - ES 3.2 core, or
|
||||
// EXT/OES_texture_buffer, with glTexBuffer resolved. Desktop GL has had buffer textures as
|
||||
// core since 3.1, so the frontend advertises them unconditionally and an app may call
|
||||
// glTexBuffer whenever it likes; this is the only thing standing between that call and a
|
||||
// null entry point. False also means every shader declaring a samplerBuffer is
|
||||
// uncompilable on this driver, which the program build reports by name.
|
||||
Bool AreBufferTexturesSupported();
|
||||
// Human-readable name of the buffer-texture tier for diagnostics and the driver POST:
|
||||
// "core (ES 3.2)", "GL_EXT_texture_buffer", "GL_OES_texture_buffer" or "unsupported".
|
||||
const char* GetBufferTextureTierName();
|
||||
// glTexBuffer / glTexBufferRange through whichever spelling this driver's buffer-texture
|
||||
// support actually ships: the unsuffixed names are ES 3.2 core, while an EXT/OES driver
|
||||
// exports glTexBuffer{,Range}EXT / OES. Callers must have checked
|
||||
// AreBufferTexturesSupported() first. CallTexBufferRange reports whether it could honour
|
||||
// the range - no tier is required to expose the range form, and the whole-buffer form is
|
||||
// the documented fallback.
|
||||
void CallTexBuffer(GLenum target, GLenum internalFormat, GLuint buffer);
|
||||
Bool CallTexBufferRange(GLenum target, GLenum internalFormat, GLuint buffer, GLintptr offset, GLsizeiptr size);
|
||||
// GL timer-query objects, backed by GL_EXT_disjoint_timer_query. The
|
||||
// creators return null (the frontend then falls back to an immediately
|
||||
// available zero result) when the calling thread does not own the ES
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -21,6 +21,29 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
String EmulateBaseInstanceInVertexShader(String source, GLenum shaderType);
|
||||
String PromoteDrawParameterGlobalsToUniforms(String source, GLenum shaderType);
|
||||
|
||||
// The ESSL half of the gl_ViewportIndex routing emulation, in the order a program's stages
|
||||
// meet it. Both are pure String -> String rewrites over what SPIRV-Cross emitted once
|
||||
// LowerViewportIndexPass has demoted the builtin to the plain global `mg_ViewportIndex`.
|
||||
//
|
||||
// The producing stage's global becomes an ordinary flat varying; true when there was one to
|
||||
// promote, which is also the answer to "does this program route viewports at all".
|
||||
Bool PromoteViewportIndexGlobalToVarying(String& source);
|
||||
// The fragment stage grows a matching flat input, the mg_ViewportPassMask uniform the draw
|
||||
// path writes, and a wrapper entry point that discards every fragment whose primitive routed
|
||||
// to an index the current replay pass is not drawing. False when the stage has no entry point
|
||||
// to wrap, which leaves the program renderable but unrouted.
|
||||
Bool InjectViewportIndexPassGate(String& source);
|
||||
|
||||
// Whether a vertex shader may declare a storage block at all, given what the host driver
|
||||
// reports for GL_MAX_VERTEX_SHADER_STORAGE_BLOCKS. Pure, and separated from the capability
|
||||
// global purely so the decision can be tested without one.
|
||||
//
|
||||
// The indirect half of the gl_BaseInstance lowering in PromoteDrawParameterGlobalsToUniforms
|
||||
// is the only thing that needs this, and it needs exactly one block. A driver reporting 0 is
|
||||
// conformant - the minimum is 0 in GL 4.6 table 23.64 and ES 3.2 table 21.44 - and ARM's
|
||||
// GLES driver does report 0, so this is a live path, not a defensive one.
|
||||
Bool VertexStageStorageBlockUsable(Int maxVertexShaderStorageBlocks);
|
||||
|
||||
// True once the process has entered exit(): past that point the EGL library and
|
||||
// the driver may already be unloaded, so a backend twin's destructor must not
|
||||
// call into g_GLESFuncs (the observed crash is a jump through an unmapped driver
|
||||
@@ -36,6 +59,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Bool InProcessTeardown();
|
||||
void EnsureProcessTeardownSentinel();
|
||||
|
||||
// Generation of the backend ES context that owns the driver ids currently handed
|
||||
// out. Bumped exactly once per DestroyEGLContext. Every backend twin that owns a
|
||||
// driver name (texture, framebuffer, renderbuffer, sampler) stamps this at
|
||||
// construction and compares it in its destructor: a twin outliving its context
|
||||
// must NOT glDelete* its id, because a successor context may already have recycled
|
||||
// that name and the delete would take out a live object of the new context.
|
||||
extern Uint g_backendContextGeneration;
|
||||
|
||||
// Which optional pieces of state a draw needs synchronized before it is issued.
|
||||
// Index/indirect buffer syncs and the instancing-related work are skipped for
|
||||
// draws that provably cannot read them.
|
||||
@@ -74,14 +105,78 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// GLES core supports only GL_PRIMITIVE_RESTART_FIXED_INDEX. Throws when the app enabled
|
||||
// the arbitrary GL_PRIMITIVE_RESTART with a non-fixed index for this index type.
|
||||
void CheckPrimitiveRestartSupported(GLenum indexType);
|
||||
// Feed the current program's gl_BaseInstance / gl_DrawID emulation uniforms. Both are
|
||||
// no-ops when the program does not read the corresponding builtin.
|
||||
// Feed the current program's gl_BaseInstance / gl_DrawID / gl_BaseVertex emulation
|
||||
// uniforms. All are no-ops when the program does not read the corresponding builtin.
|
||||
void SetCurrentBaseInstance(Uint32 baseInstance);
|
||||
void SetCurrentDrawID(Uint32 drawId);
|
||||
// GL's gl_BaseVertex is the base-vertex parameter of an indexed draw and zero for every
|
||||
// command that has none - including all the DrawArrays forms - so every draw path that
|
||||
// does not carry one must leave this at zero rather than inherit the last draw's value.
|
||||
void SetCurrentBaseVertex(Int32 baseVertex);
|
||||
// True when the current program actually reads gl_DrawID, i.e. when a batched
|
||||
// (single driver call) multi-draw tier would have to feed it one value for the whole
|
||||
// batch and would therefore be wrong.
|
||||
Bool CurrentProgramReadsDrawID();
|
||||
// Same question for gl_BaseVertex: a batched multi-draw tier cannot give each sub-draw
|
||||
// its own base vertex through a uniform either.
|
||||
Bool CurrentProgramReadsBaseVertex();
|
||||
// Both of the above, conservatively, for a caller that must decide BEFORE PrepareForDraw
|
||||
// has synced the program - where "does not read it" is indistinguishable from "cannot be
|
||||
// asked yet". Answers true whenever the backend twin is missing or predates the current
|
||||
// link.
|
||||
Bool CurrentProgramMayNeedPerSubDrawBuiltins(Bool batchCarriesBaseVertices);
|
||||
|
||||
// ---- gl_ViewportIndex routing emulation, draw half ---------------------------------------
|
||||
//
|
||||
// GLES has ONE viewport, ONE scissor rectangle and ONE depth range; GL 4.1 has sixteen of
|
||||
// each, selected per primitive by gl_ViewportIndex. There is no ES entry point to program the
|
||||
// other fifteen with (GL_OES_viewport_array exists but Adreno 830 does not have it, verified
|
||||
// three ways), so the only way to rasterize a primitive against index i's rectangle is to
|
||||
// make index i's rectangle THE viewport for the duration of a draw - which means issuing the
|
||||
// draw once per distinct viewport state and letting the fragment stage throw away the
|
||||
// primitives that belong to the other indices (the gate Managers.cpp injects).
|
||||
//
|
||||
// Indices whose whole state tuple (viewport rectangle, scissor rectangle, scissor-test enable,
|
||||
// depth range) is identical share ONE pass, so the overwhelmingly common case - every index
|
||||
// still holding what glViewport/glScissor/glDepthRange broadcast to all sixteen - collapses
|
||||
// to a single pass with an all-ones gate mask, i.e. one draw and no behaviour change at all.
|
||||
//
|
||||
// Whether emulation runs. Off only under MOBILEGL_FORCE_VIEWPORT_ARRAY_EMULATION falsy, which
|
||||
// restores the pre-emulation path as a negative control.
|
||||
Bool ViewportArrayEmulationEnabled();
|
||||
// Whether ANY program built in this process has come out with a viewport gate. Sticky once
|
||||
// true; it exists so that BeginViewportRoutingPasses - which runs on every draw of every
|
||||
// workload - can answer with one static load in the case that matters, which is every
|
||||
// application that has never heard of gl_ViewportIndex.
|
||||
extern Bool g_anyProgramRoutesViewportIndex;
|
||||
// Number of times the current draw has to be issued. Always >= 1, and exactly 1 - with no
|
||||
// state touched - whenever the current program does not route viewports, whenever every
|
||||
// configured index shares one state, and whenever replaying would multiply a side effect the
|
||||
// fragment gate cannot undo (transform feedback, rasterizer discard). Also seeds the pass
|
||||
// mask uniform for that single-pass case, so a gated fragment shader never runs against the
|
||||
// zero every GLSL uniform starts at - which would discard the whole draw.
|
||||
Uint BeginViewportRoutingPasses();
|
||||
// Push pass `pass`'s viewport / scissor / scissor-test / depth range onto the ES context and
|
||||
// set the gate mask to the indices it serves. Only called when the count above exceeds 1.
|
||||
void ApplyViewportRoutingPass(Uint pass);
|
||||
// Restore the gate mask and mark the render-state shadow dirty, so the next ordinary draw
|
||||
// re-pushes index 0's state. Takes the count so it can do nothing at all in the common case.
|
||||
void EndViewportRoutingPasses(Uint passCount);
|
||||
|
||||
// Issue one draw, replayed once per viewport-routing pass. Every application-visible draw
|
||||
// entry point wraps its native glDraw* call in this; the internal blit and clear helpers
|
||||
// deliberately do not, because they bind their own programs, which never route.
|
||||
template <typename IssueDraw>
|
||||
inline void ForEachViewportRoutingPass(IssueDraw&& issue) {
|
||||
const Uint passCount = BeginViewportRoutingPasses();
|
||||
for (Uint pass = 0; pass < passCount; ++pass) {
|
||||
if (passCount > 1) {
|
||||
ApplyViewportRoutingPass(pass);
|
||||
}
|
||||
issue();
|
||||
}
|
||||
EndViewportRoutingPasses(passCount);
|
||||
}
|
||||
|
||||
template <typename StateObject, typename BackendObject>
|
||||
class StateBackendObjectRegistry {
|
||||
@@ -109,7 +204,28 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Twin creation is the moment a driver-owned id starts needing a guarded
|
||||
// destructor; cold path, so the once-guard costs nothing per draw.
|
||||
EnsureProcessTeardownSentinel();
|
||||
// Sweep BEFORE the entry reference below exists: the map is open-addressed and an
|
||||
// erase relocates the rest of the probe cluster, so collecting once that reference
|
||||
// is taken would invalidate it. The sweep is therefore owed from an earlier call
|
||||
// rather than triggered by this one.
|
||||
if (m_creationTick >= kCreationGCInterval) {
|
||||
m_creationTick = 0;
|
||||
CollectGarbage();
|
||||
}
|
||||
const SizeT entryCountBeforeInsert = m_entries.size();
|
||||
auto& entry = m_entries[stateObj.get()];
|
||||
if (m_entries.size() != entryCountBeforeInsert) {
|
||||
// A key the registry has never held. Nothing tells the backend that a texture or
|
||||
// renderbuffer was DELETED - the twin, and the driver storage it owns, lives
|
||||
// until a collection - and CollectGarbageIfNeeded is ticked only from the
|
||||
// per-draw sync paths, which a CTS-shaped workload runs about ten times per
|
||||
// case. 1024 of those ticks then span ~100 cases, so ~100 cases' worth of dead
|
||||
// (and, for this suite, gigabyte-sized) objects stay allocated at once. Object
|
||||
// CHURN rather than draw count is what makes the sweep urgent, so a twin the
|
||||
// registry has never seen ticks it too - and it does so on the path that is
|
||||
// about to allocate, which is exactly when the memory is needed.
|
||||
++m_creationTick;
|
||||
}
|
||||
if (entry.stateRef.expired()) {
|
||||
// The previous owner of this address is gone and the allocator handed it
|
||||
// to a new object: its twin describes ids the new state object never made.
|
||||
@@ -121,6 +237,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
// Null when no live state object owns this key. The result points into the map, so
|
||||
// it stays valid only until the next GetOrCreate/Find/CollectGarbage on this registry.
|
||||
// Take that literally, including for Find: the map is open-addressed and erases by
|
||||
// shifting the rest of the probe cluster into the hole, so an erase relocates entries
|
||||
// OTHER than the erased one - and Find erases, whenever it lands on a key whose state
|
||||
// object has expired. Callers that need the twin across another registry call must copy
|
||||
// the BackendPtr out (or keep only the pointee, which is heap-allocated and never moves).
|
||||
BackendPtr* Find(StateObject* stateObj) {
|
||||
const auto entryIt = m_entries.find(stateObj);
|
||||
if (entryIt == m_entries.end()) {
|
||||
@@ -178,8 +299,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
private:
|
||||
static constexpr Uint32 kGCInterval = 1024;
|
||||
// Creations are far rarer than draws, so this counts in a much smaller unit than
|
||||
// kGCInterval does.
|
||||
static constexpr Uint32 kCreationGCInterval = 64;
|
||||
BackendMap m_entries;
|
||||
Uint32 m_gcTick = 0;
|
||||
Uint32 m_creationTick = 0;
|
||||
Bool m_isCollecting = false;
|
||||
};
|
||||
|
||||
@@ -261,6 +386,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// context loss.
|
||||
Bool persistentMapped = false;
|
||||
void* persistentPtr = nullptr;
|
||||
// The GL store behind `id` was created with glBufferStorageEXT and is
|
||||
// therefore IMMUTABLE - glBufferData cannot respecify it and it must never be
|
||||
// recycled through the size-keyed buffer pool. Tracked separately from
|
||||
// persistentMapped because the two come apart: a glMapBufferRange that fails
|
||||
// after its glBufferStorageEXT succeeded leaves immutable storage behind with
|
||||
// no map, and a respecification then has to retire the id rather than hand it
|
||||
// to glBufferData, which the driver would silently refuse.
|
||||
Bool immutableStorage = false;
|
||||
};
|
||||
|
||||
// Registered as the frontend's BufferBackendOps at backend init and on
|
||||
@@ -313,6 +446,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// client-attribute staging buffers): scrub every buffer-binding shadow that
|
||||
// could false-skip when the name is recycled.
|
||||
void NoteBufferIdDeleted(Uint id);
|
||||
// Bumped whenever a live GLESBufferResource's driver id is retired and re-minted
|
||||
// while its frontend buffer stays alive (persistent-map adoption, immutable-store
|
||||
// retire). The VAO twins' baked glVertexAttribPointer / element-array bindings
|
||||
// key on FRONTEND versions, which a backend-side re-mint does not move - without
|
||||
// this generation the driver VAO would keep fetching through the deleted id (or
|
||||
// its retained store) forever. Compared and stamped by
|
||||
// BackendVertexArrayObject::SyncToBackend.
|
||||
extern Uint64 g_bufferBackendIdGeneration;
|
||||
// Redundant-bind cache for INDEXED buffer bindings (glBindBufferBase/Range on
|
||||
// GL_UNIFORM_BUFFER / GL_SHADER_STORAGE_BUFFER): skips the GL call when the
|
||||
// (id, range) already at that index matches, like the array-buffer/texture/
|
||||
@@ -320,6 +461,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void BindBufferBaseCached(GLenum glTarget, Uint index, Uint id);
|
||||
void BindBufferRangeCached(GLenum glTarget, Uint index, Uint id, GLintptr offset, GLsizeiptr size);
|
||||
void InvalidateIndexedBufferBindingCache();
|
||||
// Re-issues the GL_ATOMIC_COUNTER_BUFFER binding points a program's shaders declare as
|
||||
// GL_SHADER_STORAGE_BUFFER bindings at the reserved slots the transpiled ESSL was built
|
||||
// against (BackendProgramObjectImpl::GetAtomicCounterBindings /
|
||||
// GetAtomicCounterEsslBindingTop). ES has no counter-buffer target at all, so without
|
||||
// this the shader reads a storage block nobody ever bound a buffer to and the buffer the
|
||||
// application bound never reaches the driver.
|
||||
void SyncAtomicCounterBuffers(const Vector<Int>& glBindings, Int esslBindingTop);
|
||||
// Buffer-storage pool maintenance. TrimBufferPool evicts over-budget entries
|
||||
// (called once per frame from Present); ClearBufferPool drops all pooled ids
|
||||
// without glDeleteBuffers (called when the ES context is going away).
|
||||
@@ -426,12 +574,51 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
PendingAttribValueMask& GetPendingAttribValueMaskMemo() { return m_pendingAttribValueMask; }
|
||||
|
||||
private:
|
||||
// Narrows one enabled GL_DOUBLE array into a tightly packed float32 stream held in
|
||||
// this VAO's own scratch buffer and declares the attribute against it. ES has no
|
||||
// 64-bit vertex format, but the source bytes are ordinary IEEE-754 doubles and every
|
||||
// fp64 value in every shader is already narrowed to 32 bits (DemoteFloat64Pass), so
|
||||
// narrowing the ARRAY is the coherent completion of that decision rather than
|
||||
// dropping it. Returns false when the stream cannot be built, in which case the
|
||||
// caller must DISABLE the array - leaving a 64-bit array enabled with no pointer is
|
||||
// what the Adreno driver turns into a SIGSEGV at the next draw.
|
||||
Bool SyncFloat64AttributeAsFloat32(Uint attribIndex, const MG_State::GLState::VertexAttribute& attrib,
|
||||
Uint32 fetchBaseInstance);
|
||||
|
||||
// What the converted float32 stream in m_convertedAttributeBufferIds[i] was built
|
||||
// from. A hit skips the CPU conversion and the re-upload; the buffer's change serial
|
||||
// is part of the key, so a glBufferSubData into the source invalidates it.
|
||||
struct ConvertedFloat64Stream {
|
||||
Bool valid = false;
|
||||
Uint64 sourceLifetimeId = 0;
|
||||
Uint64 sourceChangeSerial = 0;
|
||||
SizeT sourceOffset = 0;
|
||||
SizeT sourceStride = 0;
|
||||
SizeT componentCount = 0;
|
||||
SizeT elementCount = 0;
|
||||
};
|
||||
|
||||
ResolvedDrawBuffers m_resolvedDrawBuffers;
|
||||
PendingAttribValueMask m_pendingAttribValueMask;
|
||||
Uint m_backendVAOId = 0;
|
||||
Array<Uint, MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS> m_clientAttributeBufferIds;
|
||||
// Scratch stores for the buffer-backed GL_DOUBLE narrowing. Deliberately separate
|
||||
// from m_clientAttributeBufferIds: that one holds the per-draw upload of a
|
||||
// CLIENT-MEMORY array, and an attribute index can carry both shapes over its life.
|
||||
Array<Uint, MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS> m_convertedAttributeBufferIds;
|
||||
Array<ConvertedFloat64Stream, MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS>
|
||||
m_convertedAttributeStreams;
|
||||
// True while at least one attribute of this VAO is fed by a converted stream. Such a
|
||||
// stream is derived from buffer CONTENT, which no VAO version covers, so the config
|
||||
// version early-out in SyncToBackend must not be trusted while it is set.
|
||||
Bool m_hasConvertedFloat64Attribute = false;
|
||||
Bool m_isInitialized = false;
|
||||
Uint16 m_syncedIndexBufferVersion = 0;
|
||||
// Identity of the buffer the version above was stamped against. Raw and never
|
||||
// dereferenced: the slot version is a wrapping Uint16 (see the ResolvedDrawBuffers
|
||||
// IBO memo and the packed_pixels postmortem at BindCurrentFBO), so the version
|
||||
// alone would read a wrapped-back count with a different buffer bound as clean.
|
||||
const MG_State::GLState::BufferObject* m_syncedIndexBufferObject = nullptr;
|
||||
// Aggregate gate over the per-attribute walk below: the frontend bumps its config
|
||||
// version on every per-attribute version bump (the three Bump*Version functions are
|
||||
// its only writers), so an unchanged config version proves every per-attribute
|
||||
@@ -441,6 +628,17 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Uint32 m_syncedConfigVersion = 0;
|
||||
Array<MG_State::GLState::VertexAttributeVersion, MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS>
|
||||
m_syncedAttributeVersions;
|
||||
// Byte shift currently baked into the instanced arrays' offsets by the baseInstance
|
||||
// emulation (see SetPendingFetchBaseInstance). It is draw state, not VAO state, so it
|
||||
// is deliberately NOT covered by the config version: the frontend never bumps for it.
|
||||
// Kept here because it describes what was last EMITTED, which is what the next sync
|
||||
// has to correct.
|
||||
Uint32 m_syncedFetchBaseInstance = 0;
|
||||
// BufferImpl::g_bufferBackendIdGeneration as of this twin's last emit. A
|
||||
// mismatch means some live buffer's driver id was re-minted since; the ids
|
||||
// baked into the driver VAO's attribute/element bindings may be dead even
|
||||
// though every frontend version matches, so the next sync re-emits them all.
|
||||
Uint64 m_syncedBufferIdGeneration = 0;
|
||||
};
|
||||
|
||||
extern StateBackendObjectRegistry<MG_State::GLState::VertexArrayObject, BackendVertexArrayObject>
|
||||
@@ -454,6 +652,23 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void InvalidateVAOBindingCache();
|
||||
// ES resets the binding to 0 when the currently bound VAO is deleted.
|
||||
void NoteVAOIdDeleted(Uint id);
|
||||
|
||||
// baseInstance emulation for drivers without GL_EXT_base_instance. GL fetches an
|
||||
// instanced array at element "floor(instance / divisor) + baseInstance", and ES has no
|
||||
// way to say the "+ baseInstance" part - so it is folded into the attribute's own byte
|
||||
// offset (baseInstance * stride) for every divisor'd array, which is exactly equivalent.
|
||||
// Must be set BEFORE PrepareForDraw so the VAO sync sees it, and cleared after the draw
|
||||
// so the next one refetches from element 0; ScopedFetchBaseInstance does both.
|
||||
void SetPendingFetchBaseInstance(Uint32 baseInstance);
|
||||
Uint32 GetPendingFetchBaseInstance();
|
||||
|
||||
class ScopedFetchBaseInstance {
|
||||
public:
|
||||
explicit ScopedFetchBaseInstance(Uint32 baseInstance) { SetPendingFetchBaseInstance(baseInstance); }
|
||||
~ScopedFetchBaseInstance() { SetPendingFetchBaseInstance(0); }
|
||||
ScopedFetchBaseInstance(const ScopedFetchBaseInstance&) = delete;
|
||||
ScopedFetchBaseInstance& operator=(const ScopedFetchBaseInstance&) = delete;
|
||||
};
|
||||
} // namespace VertexArrayImpl
|
||||
|
||||
namespace TextureImpl {
|
||||
@@ -530,9 +745,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Returns `data` untouched when no widening applies. Pure CPU and context-free so a unit
|
||||
// test can exercise the exact packing the driver is handed; `widenedData` is the caller's
|
||||
// scratch buffer and has to outlive the returned pointer.
|
||||
// `alphaOneCodeOverride`, when non-zero, replaces the value written into the synthetic
|
||||
// alpha channel: an image carrier that holds a NORMALIZED format's channel CODES has to
|
||||
// pad alpha with that channel's saturated CODE (65535, 32767, 3), which neither of the
|
||||
// transfer type's own "ones" is.
|
||||
const void* PrepareChannelWidenedUpload(Uint componentCount, const IntVec3& texelSize, const void* data,
|
||||
SizeT byteSize, GLenum uploadType, Vector<Uint8>& widenedData,
|
||||
Bool integerData = false);
|
||||
Bool integerData = false, Uint32 alphaOneCodeOverride = 0u);
|
||||
|
||||
// Splits a GL_UNSIGNED_INT_2_10_10_10_REV shadow (rgb10_a2, rgb10_a2ui) into the four
|
||||
// GL_UNSIGNED_SHORT channel CODES its GL_RGBA16UI image carrier is uploaded as: red in
|
||||
// bits 0-9, green 10-19, blue 20-29, alpha 30-31. Pure CPU and context-free so a unit test
|
||||
// can pin the exact fields; `widenedData` is the caller's scratch and has to outlive the
|
||||
// returned pointer.
|
||||
const void* PreparePackedIntWidenedUpload(const IntVec3& texelSize, const void* data, SizeT byteSize,
|
||||
Vector<Uint8>& widenedData);
|
||||
|
||||
struct StateTextureBasicInfo { // Used for tracking texture state changes
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::Unknown;
|
||||
@@ -565,12 +792,38 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
BackendTextureObject(const BackendTextureObject&) = delete;
|
||||
BackendTextureObject& operator=(const BackendTextureObject&) = delete;
|
||||
void SyncMipmapsToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
// The storage half of the sync for a texture created by glTextureView. Instead of
|
||||
// allocating storage and replaying uploads, it makes this object's ES name BE a view
|
||||
// of the storage texture's ES name (EXT/OES_texture_view), which is what gives the
|
||||
// two names one image and independent per-texture parameters at the same time. The
|
||||
// parameter and sampler halves are unchanged and run on this name as on any other.
|
||||
void SyncTextureViewToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void StampViewSyncKeys(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
// The storage half of the sync for a texture created by glTextureView. Instead of
|
||||
// allocating storage and replaying uploads, it makes this object's ES name BE a view
|
||||
// of the storage texture's ES name (EXT/OES_texture_view), which is what gives the
|
||||
// two names one image and independent per-texture parameters at the same time. The
|
||||
// parameter and sampler halves are unchanged and run on this name as on any other.
|
||||
void SyncBuiltinSamplerToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void SyncTextureParamsToBackend(const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
void RequireImageBindableStorage();
|
||||
// Marks the texture as one whose ES storage has to be image-bindable, which for a
|
||||
// non-core image format means re-minting it in the widening's carrier. Takes the state
|
||||
// object because the levels already uploaded have to be marked dirty again: the
|
||||
// re-mint allocates fresh storage and only replays what the shadow still calls dirty.
|
||||
void RequireImageBindableStorage(
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& stateTextureObject);
|
||||
// Whether this texture's ES storage was minted in an image carrier rather than in the
|
||||
// frontend format's own layout - the readback has to ask, because for a NORMALIZED
|
||||
// carrier the storage is an integer texture holding codes and glGetTexImage still owes
|
||||
// the application floats.
|
||||
Bool RequiresImageBindableStorage() const { return m_imageBindableStorageRequired; }
|
||||
void Bind(GLenum target, Uint unit = TempTextureUnit);
|
||||
Uint GetBackendTextureId() const;
|
||||
|
||||
// The id to hand glBindImageTexture for a SPLIT buffer image, or 0 when this texture
|
||||
// takes no split. See m_bufferImageSplitViewId.
|
||||
Uint GetBufferImageSplitViewId() const { return m_bufferImageSplitViewId; }
|
||||
|
||||
// Aggregate first-level clean gate for the per-draw trio
|
||||
// SyncTextureParamsToBackend + SyncBuiltinSamplerToBackend +
|
||||
// SyncMipmapsToBackend: EXACTLY the conjunction of their own early-outs
|
||||
@@ -583,6 +836,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// `contextId`/`samplingGeneration` are the frontend context's current
|
||||
// values, hoisted by the caller so a per-draw list walk reads them once
|
||||
// instead of per texture. `t` must be the live frontend texture.
|
||||
// True while a driver-side re-mint has left the parameter caches describing a texture
|
||||
// that no longer exists; SyncTextureObjectToBackend re-pushes them in the same sync.
|
||||
Bool NeedsParameterResync() const { return m_forceTextureParamsResync || m_forceSamplerResync; }
|
||||
|
||||
Bool IsDrawSyncClean(const MG_State::GLState::ITextureObject* t, Uint64 contextId,
|
||||
Uint64 samplingGeneration) const {
|
||||
if (!m_isInitialized || m_syncedShapeContextId == 0 || m_syncedShapeContextId != contextId ||
|
||||
@@ -607,12 +864,47 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void RecreateBackendTexture();
|
||||
|
||||
Uint m_backendTextureId = 0;
|
||||
// A SECOND buffer-texture name over the SAME buffer object, viewed in the split's
|
||||
// single-channel base format, used only as the glBindImageTexture target.
|
||||
//
|
||||
// The split needs the view to say r32f where the application said rg32f, but a buffer
|
||||
// texture that is image-bound may ALSO be read through a samplerBuffer - and the
|
||||
// sampler side is not subscript-rewritten, so re-describing the application's own
|
||||
// texture broke it: texelFetch(s, i) returned component 2i of the base view instead of
|
||||
// texel i's pair. That is exactly and only
|
||||
// KHR-GL42/43.shader_image_load_store.advanced-sync-imageAccess, which image-stores
|
||||
// into a GL_RG32F buffer texture and then reads the same texture through both an
|
||||
// imageBuffer and a samplerBuffer in one shader, comparing the two.
|
||||
//
|
||||
// Two names over one buffer cost nothing and alias exactly: a buffer texture owns no
|
||||
// storage, so both views are the application's bytes, and the split's whole premise is
|
||||
// that the two describe the same memory. The application's own name therefore keeps
|
||||
// the format it asked for - rg32f IS a legal SAMPLED buffer-texture format in ES 3.2,
|
||||
// it is only the IMAGE binding ES cannot spell - and the private name below carries
|
||||
// the split the shader was rewritten against. 0 when this texture takes no split.
|
||||
Uint m_bufferImageSplitViewId = 0;
|
||||
// For a texture created by glTextureView: the ES name of the storage texture this
|
||||
// one was last made a view OF. EXT_texture_view may be called only once per name, so
|
||||
// a storage texture that got re-minted underneath (RecreateBackendTexture) has to be
|
||||
// detected here and answered with a fresh name for the view as well - otherwise the
|
||||
// view would keep aliasing storage that no longer exists.
|
||||
Uint m_viewSourceBackendTextureId = 0;
|
||||
// For a texture created by glTextureView: the ES name of the storage texture this
|
||||
// one was last made a view OF. EXT_texture_view may be called only once per name, so
|
||||
// a storage texture that got re-minted underneath (RecreateBackendTexture) has to be
|
||||
// detected here and answered with a fresh name for the view as well - otherwise the
|
||||
// view would keep aliasing storage that no longer exists.
|
||||
// ES context generation the id was created under; a dtor running after
|
||||
// that context died must not delete a foreign (recycled) name.
|
||||
Uint m_contextGeneration = 0;
|
||||
Bool m_isInitialized = false;
|
||||
Bool m_imageBindableStorageRequired = false;
|
||||
Bool m_backendStorageImmutable = false;
|
||||
// Latches the "this driver has no buffer textures" report to once per texture. The
|
||||
// report is emitted from the respecify path, which bails before recording the state
|
||||
// it was asked to apply - so without the latch the texture stays permanently dirty
|
||||
// and every draw of every frame logs the same line.
|
||||
Bool m_bufferTextureUnsupportedReported = false;
|
||||
StateTextureBasicInfo m_prevTextureInfo;
|
||||
// Frontend content version at the last completed mipmap sync. The per-draw
|
||||
// clean probe compares this before rebuilding shape info and scanning
|
||||
@@ -638,8 +930,27 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
FloatVec4 m_cacheBorderColor = {0.0f, 0.0f, 0.0f, 0.0f};
|
||||
Vec4<TextureSwizzleParam> m_cacheSwizzleParams = {TextureSwizzleParam::Red, TextureSwizzleParam::Green,
|
||||
TextureSwizzleParam::Blue, TextureSwizzleParam::Alpha};
|
||||
// GL_DEPTH_STENCIL_TEXTURE_MODE. GL_DEPTH_COMPONENT is the GL and ES default, so a
|
||||
// texture that never asks for the stencil aspect never emits the call. The
|
||||
// depth/stencil readback and replicate-blit emulations also write this parameter
|
||||
// raw, but only ever on their own scratch textures (never on an application
|
||||
// texture), so they cannot desynchronise this cache.
|
||||
GLenum m_cacheDepthStencilTextureMode = GL_DEPTH_COMPONENT;
|
||||
Uint16 m_syncedSamplerVersion = 0;
|
||||
Uint16 m_syncedTextureParamsVersion = 0;
|
||||
// Set when the driver texture underneath was regenerated and has therefore lost every
|
||||
// parameter already pushed onto it: the params-version early-out has to be overridden
|
||||
// once, or an unchanged version would skip the re-push forever.
|
||||
Bool m_forceTextureParamsResync = false;
|
||||
// The same problem for the FILTER state, which lives in m_cacheSamplerParameters and
|
||||
// is gated on the frontend sampler's version rather than on the params version. A
|
||||
// re-mint leaves that cache describing values the new driver texture never received,
|
||||
// and an unchanged sampler version would then skip re-pushing them forever. This
|
||||
// matters more than mis-filtering: ES makes a texture INCOMPLETE when its filters do
|
||||
// not suit its level set (any integer texture with a non-NEAREST filter, or a
|
||||
// single-level texture with a mipmapping filter), and an incomplete texture samples
|
||||
// (0, 0, 0, 1) rather than its contents.
|
||||
Bool m_forceSamplerResync = false;
|
||||
};
|
||||
|
||||
void ActivateTextureUnit(Uint unit);
|
||||
@@ -657,15 +968,20 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS>
|
||||
g_boundTexturesCache;
|
||||
extern Uint g_activeTextureUnit;
|
||||
// Bumped when the backend ES context is destroyed; texture ids stamped with
|
||||
// an older generation belong to a dead context and must not be deleted.
|
||||
extern Uint g_textureContextGeneration;
|
||||
} // namespace TextureImpl
|
||||
|
||||
namespace FramebufferImpl {
|
||||
class BackendFramebufferObject {
|
||||
public:
|
||||
BackendFramebufferObject();
|
||||
// Deletes the driver framebuffer and scrubs the binding shadow. Without it every
|
||||
// frontend glDeleteFramebuffers leaked one ES framebuffer for the process lifetime;
|
||||
// an app that creates a framebuffer per readback (GL CTS packed_pixels does ~3300
|
||||
// per case) walked the driver into hundreds of megabytes of dead framebuffers and
|
||||
// out of the resources a later attachment needs.
|
||||
~BackendFramebufferObject();
|
||||
BackendFramebufferObject(const BackendFramebufferObject&) = delete;
|
||||
BackendFramebufferObject& operator=(const BackendFramebufferObject&) = delete;
|
||||
void SyncToBackend(const SharedPtr<MG_State::GLState::FramebufferObject>& stateFBOObject,
|
||||
FramebufferTarget asTarget);
|
||||
// Apply only this FBO's read buffer (glReadBuffer) to the backend. Split out so it can
|
||||
@@ -680,6 +996,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
private:
|
||||
Uint m_backendFBOId = 0;
|
||||
Uint m_contextGeneration = 0;
|
||||
|
||||
/* this will save buffers in its original form,
|
||||
reversion, absence or not consecutive are all allowed, as long as GL spec allows it
|
||||
@@ -722,6 +1039,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
using FramebufferObject = MG_State::GLState::FramebufferObject;
|
||||
FramebufferObject::FramebufferAttachmentVersionArray m_syncedFrontendAttachmentVersions = {0};
|
||||
// g_attachmentBackendIdGeneration as of this twin's last attachment walk. A
|
||||
// mismatch means some backend texture id was re-minted since, and any of this
|
||||
// twin's attachment points may still hold the dead id even though the frontend
|
||||
// attachment versions match - so the walk re-attaches everything first.
|
||||
Uint64 m_syncedBackendIdGeneration = 0;
|
||||
};
|
||||
|
||||
extern StateBackendObjectRegistry<MG_State::GLState::FramebufferObject, BackendFramebufferObject>
|
||||
@@ -811,6 +1133,19 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
extern Array<MG_State::GLState::FramebufferObject*, SizeT(FramebufferTarget::FramebufferTargetCount)>
|
||||
g_fboSyncedObjects;
|
||||
|
||||
// Bumped whenever a live backend texture's driver id is re-minted while its
|
||||
// frontend texture may still be attached to application FBOs
|
||||
// (BackendTextureObject::RecreateBackendTexture - e.g. a respecify of a texture
|
||||
// whose backend storage went immutable). The FBO twins' attachment memos key on
|
||||
// FRONTEND attachment versions, which a backend-side re-mint does not move, so
|
||||
// the driver FBO would keep the deleted texture name attached forever. The
|
||||
// SyncCurrentFBO gate compares this generation (below) to re-enter the sync,
|
||||
// and each twin re-arms its per-attachment memo on a mismatch (SyncToBackend).
|
||||
extern Uint64 g_attachmentBackendIdGeneration;
|
||||
// What g_attachmentBackendIdGeneration was when SyncCurrentFBO last stamped each
|
||||
// target; part of the synced tuple above.
|
||||
extern Array<Uint64, SizeT(FramebufferTarget::FramebufferTargetCount)> g_fboSyncedBackendIdGenerations;
|
||||
|
||||
// Driver-level READ/DRAW framebuffer-binding shadow. Every backend
|
||||
// glBindFramebuffer routes through BindFramebufferId so scoped helpers can
|
||||
// save/restore the current binding without a glGetIntegerv round-trip (that
|
||||
@@ -821,6 +1156,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void BindFramebufferId(GLenum fbTarget, Uint id);
|
||||
Uint CurrentFramebufferBinding(FramebufferTarget target);
|
||||
void InvalidateFramebufferBindingCache();
|
||||
// A driver framebuffer id is about to be deleted: ES reverts every target that
|
||||
// currently binds it to 0, so the binding shadow has to follow or the next
|
||||
// BindFramebufferId(0) would be deduped away and leave the deleted name bound.
|
||||
void NoteFramebufferIdDeleted(Uint id);
|
||||
} // namespace FramebufferImpl
|
||||
|
||||
// Shared scratch framebuffers for the readback/copy/blit emulation paths, with a
|
||||
@@ -910,23 +1249,51 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Image uniforms take their unit from the layout(binding=N) qualifier baked into
|
||||
// the transpiled ESSL; unlike samplers they must not (and in ES cannot) be
|
||||
// assigned through glUniform1i.
|
||||
//
|
||||
// ALL THIRTY-THREE of them, in the one contiguous block ARB_shader_image_load_store allocated
|
||||
// (GL_IMAGE_1D 0x904C through GL_UNSIGNED_INT_IMAGE_2D_MULTISAMPLE_ARRAY 0x906C). The list
|
||||
// used to hold only the fifteen whose TARGET exists in ES, which read as a reasonable
|
||||
// shortcut and was two bugs: an image uniform this says "no" to is one
|
||||
// CollectImageFormatBakeInputs never walks, so its non-core format is neither baked nor
|
||||
// widened and SPIRV-Cross throws for the whole stage ("Attempting to use image format not
|
||||
// supported in ES profile"), and it is also one SyncToBackend then treats as a SAMPLER and
|
||||
// assigns with glUniform1i, which ES makes an INVALID_OPERATION. A GL_TEXTURE_CUBE_MAP_ARRAY
|
||||
// image - which ES 3.2 has in core, so it is not even an emulated target - hit both.
|
||||
inline Bool IsImageUniformType(GLenum type) {
|
||||
switch (type) {
|
||||
case 0x904C: /*GL_IMAGE_1D*/
|
||||
case 0x904D: /*GL_IMAGE_2D*/
|
||||
case 0x904E: /*GL_IMAGE_3D*/
|
||||
case 0x904F: /*GL_IMAGE_2D_RECT*/
|
||||
case 0x9050: /*GL_IMAGE_CUBE*/
|
||||
case 0x9051: /*GL_IMAGE_BUFFER*/
|
||||
case 0x9052: /*GL_IMAGE_1D_ARRAY*/
|
||||
case 0x9053: /*GL_IMAGE_2D_ARRAY*/
|
||||
case 0x9054: /*GL_IMAGE_CUBE_MAP_ARRAY*/
|
||||
case 0x9055: /*GL_IMAGE_2D_MULTISAMPLE*/
|
||||
case 0x9056: /*GL_IMAGE_2D_MULTISAMPLE_ARRAY*/
|
||||
case 0x9057: /*GL_INT_IMAGE_1D*/
|
||||
case 0x9058: /*GL_INT_IMAGE_2D*/
|
||||
case 0x9059: /*GL_INT_IMAGE_3D*/
|
||||
case 0x905A: /*GL_INT_IMAGE_2D_RECT*/
|
||||
case 0x905B: /*GL_INT_IMAGE_CUBE*/
|
||||
case 0x905C: /*GL_INT_IMAGE_BUFFER*/
|
||||
case 0x905D: /*GL_INT_IMAGE_1D_ARRAY*/
|
||||
case 0x905E: /*GL_INT_IMAGE_2D_ARRAY*/
|
||||
case 0x905F: /*GL_INT_IMAGE_CUBE_MAP_ARRAY*/
|
||||
case 0x9060: /*GL_INT_IMAGE_2D_MULTISAMPLE*/
|
||||
case 0x9061: /*GL_INT_IMAGE_2D_MULTISAMPLE_ARRAY*/
|
||||
case 0x9062: /*GL_UNSIGNED_INT_IMAGE_1D*/
|
||||
case 0x9063: /*GL_UNSIGNED_INT_IMAGE_2D*/
|
||||
case 0x9064: /*GL_UNSIGNED_INT_IMAGE_3D*/
|
||||
case 0x9065: /*GL_UNSIGNED_INT_IMAGE_2D_RECT*/
|
||||
case 0x9066: /*GL_UNSIGNED_INT_IMAGE_CUBE*/
|
||||
case 0x9067: /*GL_UNSIGNED_INT_IMAGE_BUFFER*/
|
||||
case 0x9068: /*GL_UNSIGNED_INT_IMAGE_1D_ARRAY*/
|
||||
case 0x9069: /*GL_UNSIGNED_INT_IMAGE_2D_ARRAY*/
|
||||
case 0x906A: /*GL_UNSIGNED_INT_IMAGE_CUBE_MAP_ARRAY*/
|
||||
case 0x906B: /*GL_UNSIGNED_INT_IMAGE_2D_MULTISAMPLE*/
|
||||
case 0x906C: /*GL_UNSIGNED_INT_IMAGE_2D_MULTISAMPLE_ARRAY*/
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
@@ -934,6 +1301,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
|
||||
namespace PrgramImpl {
|
||||
// Defined further down, next to CollectImageFormatBakeInputs; only referenced here.
|
||||
struct ImageFormatBakeInputs;
|
||||
|
||||
class BackendProgramObjectImpl {
|
||||
public:
|
||||
// Per-link cache of a sampler-style uniform's backend location: built once in
|
||||
@@ -993,13 +1363,25 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
BackendProgramObjectImpl();
|
||||
~BackendProgramObjectImpl();
|
||||
void SyncToBackend(const SharedPtr<MG_State::GLState::ProgramObject>& stateProgramObject);
|
||||
void Use() const;
|
||||
void Use();
|
||||
void SetBaseInstance(Uint32 baseInstance) const;
|
||||
void SetBaseInstanceWordIndex(Int32 wordIndex) const;
|
||||
void SetDrawID(Uint32 drawId) const;
|
||||
void SetBaseVertex(Int32 baseVertex) const;
|
||||
// True when the transpiled program kept a gl_DrawID uniform, i.e. SetDrawID
|
||||
// actually reaches a shader read rather than being discarded.
|
||||
Bool ReadsDrawID() const { return m_drawIdUniformLocation >= 0; }
|
||||
// Same for gl_BaseVertex: only a program that reads it pays for the per-draw
|
||||
// uniform write, and only such a program needs the reset after one.
|
||||
Bool ReadsBaseVertex() const { return m_baseVertexUniformLocation >= 0; }
|
||||
// Which viewport indices the next draw's fragments may keep, one bit each. Written
|
||||
// once per replay pass; see ForEachViewportRoutingPass.
|
||||
void SetViewportPassMask(Uint32 indexMask) const;
|
||||
// True when this build injected the fragment-stage viewport gate, i.e. when a
|
||||
// pre-rasterization stage routes by gl_ViewportIndex AND the fragment stage can act
|
||||
// on it. The uniform is the honest test for both halves: it exists only where the
|
||||
// gate was injected, and the gate is injected only where a stage routes.
|
||||
Bool RoutesViewportIndex() const { return m_viewportPassMaskUniformLocation >= 0; }
|
||||
Int GetIndirectParamsBinding() const { return m_indirectParamsBinding; }
|
||||
Uint GetBackendProgramId() const { return m_backendProgramId; }
|
||||
// False when the last SyncToBackend could not produce a usable program (a
|
||||
@@ -1010,6 +1392,27 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Uint32 GetSnormFallbackClampOutputMask() const { return m_snormFallbackClampOutputMask; }
|
||||
Uint32 GetUnormFallbackClampOutputMask() const { return m_unormFallbackClampOutputMask; }
|
||||
Uint GetFragColorBroadcastCount() const { return m_fragColorBroadcastCount; }
|
||||
// Signature of the glShaderStorageBlockBinding override set the generated ESSL was
|
||||
// transpiled against (ES can only express a storage-block binding as the declared
|
||||
// qualifier, so the overrides are baked into the source). A mismatch means the
|
||||
// program is stale exactly like the clamp masks above.
|
||||
Uint64 GetShaderStorageBlockBindingSignature() const { return m_shaderStorageBlockBindingSignature; }
|
||||
// GL atomic-counter binding points the transpiled stages declare (sorted, unique),
|
||||
// and the top of the reserved shader-storage range their counter blocks were
|
||||
// transpiled against - the slot for GL binding N is `top - N`. Empty for every
|
||||
// program that uses no atomic counter, which is what keeps the per-draw cost of the
|
||||
// counter sync at one empty-vector test.
|
||||
const Vector<Int>& GetAtomicCounterBindings() const { return m_atomicCounterGlBindings; }
|
||||
Int GetAtomicCounterEsslBindingTop() const { return m_atomicCounterEsslBindingTop; }
|
||||
// GL_PATCH_VERTICES the synthesized pass-through tessellation control stage was built
|
||||
// for, or -1 when this program needed no such stage. Another of the same shape as the
|
||||
// signatures above: the value is compiled INTO the synthesized stage as
|
||||
// `layout(vertices = N) out`, so a program built for one patch size is stale for
|
||||
// another and the draw path has to say so. -1 compares equal to itself for every
|
||||
// program that has a control stage of its own, i.e. for all but a handful.
|
||||
Int GetPassthroughTessControlPatchVertices() const {
|
||||
return m_passthroughTessControlPatchVertices;
|
||||
}
|
||||
|
||||
Bool HasGlobalUboBlock() const { return m_globalUboBackendBlockIndex >= 0; }
|
||||
const Vector<Int>& GetUniformBlockBackendIndices() const { return m_uniformBlockBackendIndices; }
|
||||
@@ -1025,23 +1428,98 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Frontend link version this backend program (and its resource caches) was
|
||||
// built from; a mismatch means every link-derived cache here is stale.
|
||||
Uint32 GetSyncedLinkVersion() const { return m_syncedLinkVersion; }
|
||||
// Image-uniform unit generation this backend program was GENERATED against.
|
||||
// Separate from the link version because it is not link state: ES forbids
|
||||
// glUniform1i on an image uniform, so RebindImageUniformsToFrontendUnits bakes the
|
||||
// unit into the ESSL, and a program built before glUniform1i moved that unit is as
|
||||
// stale as one built before a relink - while the sampler half, which really is
|
||||
// re-issued per draw, needs nothing of the sort.
|
||||
Uint32 GetSyncedImageUnitVersion() const { return m_syncedImageUnitVersion; }
|
||||
// Whether the (unit, bound format) pairs this program's FORMAT-LESS image uniforms
|
||||
// resolve to are still the ones its ESSL was generated against.
|
||||
//
|
||||
// A fourth condition of the same family as the three above, and the only one that
|
||||
// reads live state rather than a program-side counter, because that is where the
|
||||
// dependency actually is. GLSL ES requires a format layout qualifier on every image
|
||||
// where desktop GLSL lets a writeonly declaration omit one, and the only correct
|
||||
// qualifier is whatever glBindImageTexture named - so a declaration with no format
|
||||
// is compiled against the BINDING, and a rebind to a different format makes the
|
||||
// built program wrong. Keyed on the units the program's own images address (cached
|
||||
// at sync, since a unit can only move by glUniform1i, which bumps the image-unit
|
||||
// version above and forces a re-sync anyway), so the cost on a program with no
|
||||
// format-less image - which is all but a handful - is one empty-vector test.
|
||||
//
|
||||
// Deliberately NOT reached from glBindImageTexture: that entry point must never
|
||||
// trigger a build (same constraint as glShaderStorageBlockBinding). It moves the
|
||||
// state and this comparison notices at the next Prepare, which is also what makes
|
||||
// an image first bound AFTER link work.
|
||||
Bool ImageUnitFormatsStillMatch() const;
|
||||
// The value ImageUnitFormatsStillMatch() compares against, recomputed from live
|
||||
// image-unit state. 0 when the program has no format-less image uniform.
|
||||
Uint64 ComputeImageUnitFormatSignature() const;
|
||||
|
||||
private:
|
||||
void CacheResourceLocations(const SharedPtr<MG_State::GLState::ProgramObject>& stateProgramObject);
|
||||
|
||||
// Builds, compiles and attaches the pass-through tessellation control stage GL 4.6
|
||||
// core 11.2.2 describes, for a program that has an evaluation stage and none of its
|
||||
// own - which ES 3.2 rejects outright. Called from SyncToBackend after every real
|
||||
// stage has been attached and before the link; see the definition for why it cannot
|
||||
// regress a program that works today.
|
||||
void AttachPassthroughTessControlStage(
|
||||
const MG_State::GLState::ProgramObject& stateProgramObject, Int tessEvalShaderIndex,
|
||||
const Vector<Vector<unsigned int>>& shaderSpirvs, const String& vertexStageEssl,
|
||||
const String& tessEvalStageEssl);
|
||||
|
||||
// One stage's SPIR-V through the DirectGLES pass chain and SPIRV-Cross, producing
|
||||
// the raw emitted ESSL and the interface blocks this stage's XFB flattening
|
||||
// rewrote. This is the segment the L2 shader-translation memo keys on, so every
|
||||
// input it reads must appear in EsslTranslationKeyInputs - see the definition's
|
||||
// header comment in Managers.cpp and MG_Util/ShaderTranspiler/TranslationCache.h.
|
||||
// False means SPIRV-Cross refused the module; `outError` then carries its message.
|
||||
Bool TranspileSpirvToEssl(const Vector<unsigned int>& spirvCode, GLenum glShaderType,
|
||||
const std::set<String>& xfbCaptureBlockNames,
|
||||
const ImageFormatBakeInputs& imageFormatBake,
|
||||
const UnorderedMap<String, Int>& storageBlockBindingOverrides,
|
||||
const std::map<String, String>& inputBlockRenames,
|
||||
const std::map<String, String>& outputBlockRenames,
|
||||
Int atomicCounterEsslBindingTop, Bool enableSpirvValidation,
|
||||
String& outSource,
|
||||
std::set<String>& outFlattenedXfbBlockNames,
|
||||
Vector<Int>& outAtomicCounterGlBindings, String& outError) const;
|
||||
|
||||
Uint m_backendProgramId = 0;
|
||||
// GL name of the frontend program this was last synced from; diagnostics only, so
|
||||
// an unusable backend program can be traced back to the glCreateProgram id the app
|
||||
// knows it by.
|
||||
Uint m_frontendProgramId = 0;
|
||||
Uint m_backendGlobalUBOId = 0;
|
||||
Int m_baseInstanceUniformLocation = -1;
|
||||
Int m_drawIdUniformLocation = -1;
|
||||
Int m_baseVertexUniformLocation = -1;
|
||||
Int m_baseInstanceWordIndexUniformLocation = -1;
|
||||
Int m_viewportPassMaskUniformLocation = -1;
|
||||
Int m_indirectParamsBinding = -1;
|
||||
Uint32 m_snormFallbackClampOutputMask = 0;
|
||||
Uint32 m_unormFallbackClampOutputMask = 0;
|
||||
// Draw buffers a legacy gl_FragColor write has to reach (see
|
||||
// PrgramImpl::BroadcastLegacyFragColor); 1 keeps the plain single-output shader.
|
||||
Uint m_fragColorBroadcastCount = 1;
|
||||
// 0 is the signature of an empty override set, i.e. what almost every program has.
|
||||
Uint64 m_shaderStorageBlockBindingSignature = 0;
|
||||
Vector<Int> m_atomicCounterGlBindings;
|
||||
Int m_atomicCounterEsslBindingTop = -1;
|
||||
// -1 for every program that has a tessellation control stage of its own (or none at
|
||||
// all); otherwise the GL_PATCH_VERTICES the synthesized pass-through stage was built
|
||||
// with. See GetPassthroughTessControlPatchVertices.
|
||||
Int m_passthroughTessControlPatchVertices = -1;
|
||||
Bool m_isInitialized = false;
|
||||
Bool m_backendProgramUsable = false;
|
||||
// Set by SyncToBackend every time it relinks the driver program, cleared by the
|
||||
// next Use(). Use() dedupes on a GL program NAME, and a relink replaces the
|
||||
// executable behind that name without changing it - see the note at the
|
||||
// glLinkProgram in SyncToBackend for what the driver runs otherwise.
|
||||
Bool m_rebindAfterRelink = false;
|
||||
|
||||
Int m_globalUboBackendBlockIndex = -1;
|
||||
Int m_globalUboBackendBlockSize = 0;
|
||||
@@ -1050,6 +1528,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Uint32 m_lastUploadedGlobalUboVersion = ~0u;
|
||||
BufferImpl::UboRingAllocation m_globalUboRingAllocation;
|
||||
Uint32 m_syncedLinkVersion = ~0u;
|
||||
Uint32 m_syncedImageUnitVersion = ~0u;
|
||||
// Image units addressed by the program's FORMAT-LESS image uniforms, and the digest
|
||||
// of the (unit, format) pairs the generated ESSL baked. Empty/0 for every program
|
||||
// that declares a format on all of its images, which is the overwhelming majority -
|
||||
// and what keeps the per-draw comparison free for them.
|
||||
Vector<Int> m_formatlessImageUnits;
|
||||
Uint64 m_imageUnitFormatSignature = 0;
|
||||
SamplerPassMemo m_samplerPassMemo;
|
||||
};
|
||||
|
||||
@@ -1073,26 +1558,87 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// on the backend program (eliminated as unused, or the driver lacks the entry
|
||||
// points), which is not an error - GL_BUFFER_BINDING is served from the frontend
|
||||
// record either way.
|
||||
//
|
||||
// NOT how a rebinding reaches the shader. glShaderStorageBlockBinding has no ES
|
||||
// equivalent and is absent from every real ES driver, so this is a no-op there;
|
||||
// SyncToBackend bakes the effective binding into the ESSL it generates instead
|
||||
// (SpvcSession::SetShaderStorageBlockBinding). This is kept as the cheaper path on
|
||||
// a driver that does happen to expose the entry point.
|
||||
Bool ApplyShaderStorageBlockBinding(Uint backendProgramId, const String& blockName, Uint binding);
|
||||
// Replays every glShaderStorageBlockBinding recorded on the program onto a backend
|
||||
// program that was just built. The frontend record is authoritative (only the
|
||||
// shader's DECLARED binding survives in the SPIR-V), so without this replay any
|
||||
// rebuild would silently revert rebound blocks. Mirrors DirectVulkan's
|
||||
// reseed-on-rebuild in BuildProgramResourceCache.
|
||||
// program that was just built - best effort, on the same "only where the driver has
|
||||
// the entry point" terms as ApplyShaderStorageBlockBinding above. Mirrors
|
||||
// DirectVulkan's reseed-on-rebuild in BuildProgramResourceCache.
|
||||
void ReseedShaderStorageBlockBindings(Uint backendProgramId,
|
||||
const MG_State::GLState::ProgramObject& stateProgramObject);
|
||||
// Order-independent digest of the program's glShaderStorageBlockBinding overrides.
|
||||
// The generated ESSL carries them (ES has no way to move a storage block's binding
|
||||
// after link), so a program built against a different set is stale and the draw path
|
||||
// has to rebuild it. Computed from the values, so re-setting a block to the binding it
|
||||
// already has costs nothing. 0 when nothing was ever rebound.
|
||||
Uint64 ComputeShaderStorageBlockBindingSignature(
|
||||
const MG_State::GLState::ProgramObject& stateProgramObject);
|
||||
|
||||
// Everything the image-format bake needs from one walk of a program's uniform
|
||||
// reflection. GLSL ES requires a format layout qualifier on every image uniform;
|
||||
// desktop GLSL lets a writeonly (or readonly) declaration omit one, and the only
|
||||
// format that is CORRECT to substitute is whatever glBindImageTexture named for the
|
||||
// unit that uniform addresses - so the transpile bakes it in and the build is keyed
|
||||
// on it.
|
||||
struct ImageFormatBakeInputs {
|
||||
// Uniform name (SPIR-V spelling, i.e. an array named once, unsubscripted) to the GL
|
||||
// internal format to bake. Holds only uniforms that DECLARED no format; a declared
|
||||
// one is authoritative and is never overridden.
|
||||
UnorderedMap<String, Uint> glFormatByUniformName;
|
||||
// The same uniforms whose format SPIRV-Cross REFUSES to print for ESSL (it throws on
|
||||
// its desktop-only set, which loses the stage), paired with the ESSL spelling to
|
||||
// write into the emitted declaration instead. Disjoint from the map above by
|
||||
// construction: a format is baked into the module or completed in the text, never
|
||||
// both. r8ui - the stencil half of the packed_depth_stencil case - lands here.
|
||||
UnorderedMap<String, String> esslFormatQualifierByUniformName;
|
||||
// Units those uniforms address, kept so the draw path can re-read their formats
|
||||
// without walking the reflection again.
|
||||
Vector<Int> units;
|
||||
// Digest of the (unit, format) pairs above. 0 when the program has no format-less
|
||||
// image uniform, which is all but a handful.
|
||||
Uint64 signature = 0;
|
||||
// Array uniforms whose elements resolved to units holding DIFFERENT formats: one
|
||||
// declaration carries one qualifier, so there is nothing correct to bake and they
|
||||
// are dropped from the map above. Kept for diagnostics.
|
||||
Vector<String> conflictedNames;
|
||||
// Some format in play - declared or baked - is outside the GLSL ES core image
|
||||
// format set, so the emitted ESSL needs the GL_NV_image_formats directive.
|
||||
Bool needsExtendedImageFormats = false;
|
||||
// Some DECLARED format in play is one WidenImageFormatsForEssl will re-declare in a
|
||||
// core carrier. Answered from the uniform reflection rather than from a module parse
|
||||
// on purpose: the widening is armed on every driver, so a per-stage BuildModule to
|
||||
// find out would land on every stage of every program - which is the cost
|
||||
// SpirvGateFeatures exists to avoid. Program-wide, so it can over-arm a stage that
|
||||
// declares no image; the pass then finds nothing, reports no change, and the caller
|
||||
// keeps the module it already had.
|
||||
Bool declaresWidenableImageFormat = false;
|
||||
};
|
||||
ImageFormatBakeInputs CollectImageFormatBakeInputs(
|
||||
const MG_State::GLState::ProgramObject& stateProgramObject);
|
||||
} // namespace PrgramImpl
|
||||
|
||||
namespace SamplerImpl {
|
||||
class BackendSamplerObject {
|
||||
public:
|
||||
BackendSamplerObject();
|
||||
// Deletes the driver sampler and clears the units whose binding shadow still names
|
||||
// this twin (a recycled heap address would otherwise false-skip a later Bind).
|
||||
// Frontend glDeleteSamplers used to leak the backend id for the process lifetime.
|
||||
~BackendSamplerObject();
|
||||
BackendSamplerObject(const BackendSamplerObject&) = delete;
|
||||
BackendSamplerObject& operator=(const BackendSamplerObject&) = delete;
|
||||
void SyncToBackend(const SharedPtr<MG_State::GLState::SamplerObject>& stateSamplerObject);
|
||||
void Bind(Uint unit);
|
||||
Uint GetBackendSamplerId() const;
|
||||
|
||||
private:
|
||||
Uint m_backendSamplerId = 0;
|
||||
Uint m_contextGeneration = 0;
|
||||
Bool m_isInitialized = false;
|
||||
SamplerParameters m_cacheSamplerParameters;
|
||||
Uint16 m_syncedSamplerVersion = 0;
|
||||
@@ -1110,12 +1656,18 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
class BackendRenderbufferObject {
|
||||
public:
|
||||
BackendRenderbufferObject();
|
||||
// Deletes the driver renderbuffer; frontend glDeleteRenderbuffers used to leak it
|
||||
// (with its whole image allocation) for the process lifetime.
|
||||
~BackendRenderbufferObject();
|
||||
BackendRenderbufferObject(const BackendRenderbufferObject&) = delete;
|
||||
BackendRenderbufferObject& operator=(const BackendRenderbufferObject&) = delete;
|
||||
void SyncToBackend(const SharedPtr<MG_State::GLState::RenderbufferObject>& stateRBOObject);
|
||||
Uint GetBackendRenderbufferId() const { return m_backendRBOId; }
|
||||
void Bind() const;
|
||||
|
||||
private:
|
||||
Uint m_backendRBOId = 0;
|
||||
Uint m_contextGeneration = 0;
|
||||
Bool m_isInitialized = false;
|
||||
TextureInternalFormat m_cacheInternalFormat = TextureInternalFormat::Unknown;
|
||||
Int m_cacheWidth = 0;
|
||||
|
||||
@@ -252,7 +252,7 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
g_resolvedTier =
|
||||
ResolveTier(g_GLESCapabilities, g_GLESFuncs, MG_Config::Features.EsprytMultiDrawMode,
|
||||
&g_tierResolution);
|
||||
MGLOG_I("DirectGLES multi-draw: %s", g_tierResolution.c_str());
|
||||
MGLOG_D("DirectGLES multi-draw: %s", g_tierResolution.c_str());
|
||||
}
|
||||
|
||||
// Which tiers have already announced themselves, one bit per GLESMultiDrawMode.
|
||||
@@ -267,24 +267,29 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
const Uint32 bit = 1u << static_cast<Uint32>(tier);
|
||||
if (g_announcedTiers & bit) return;
|
||||
g_announcedTiers |= bit;
|
||||
MGLOG_I("DirectGLES multi-draw: first batch executed via tier \"%s\"", TierName(tier));
|
||||
MGLOG_D("DirectGLES multi-draw: first batch executed via tier \"%s\"", TierName(tier));
|
||||
}
|
||||
|
||||
// The tier this particular batch can actually take. A tier is demoted here when
|
||||
// the batch's own shape - not the driver - rules it out; the compute tier keeps
|
||||
// its remaining feasibility checks inside its implementation, where the data it
|
||||
// has to walk is already in hand.
|
||||
GLESMultiDrawMode ResolveTierForBatch(Bool programReadsDrawID, Bool hasIndexBuffer) {
|
||||
GLESMultiDrawMode ResolveTierForBatch(Bool programReadsDrawID, Bool perSubDrawBaseVertex,
|
||||
Bool hasIndexBuffer) {
|
||||
ResolveTierOnce();
|
||||
GLESMultiDrawMode tier = g_resolvedTier;
|
||||
|
||||
// Batched tiers issue one driver entry for the whole batch, so the emulated
|
||||
// gl_DrawID uniform can only hold one value across every sub-draw. A program
|
||||
// that reads gl_DrawID gets an unrolled tier, which feeds each sub-draw its
|
||||
// own index (the spec's value); nothing else observes the difference.
|
||||
// own index (the spec's value); nothing else observes the difference. The
|
||||
// emulated gl_BaseVertex is one uniform for the same reason, so a batch whose
|
||||
// sub-draws carry their own base vertices unrolls too - even the Ext tier,
|
||||
// which hands the driver the whole basevertex array, can only leave ONE value
|
||||
// in the uniform the shader reads.
|
||||
const Bool batched = tier == GLESMultiDrawMode::Ext || tier == GLESMultiDrawMode::MultiIndirect ||
|
||||
tier == GLESMultiDrawMode::Compute;
|
||||
if (batched && programReadsDrawID) {
|
||||
if (batched && (programReadsDrawID || perSubDrawBaseVertex)) {
|
||||
tier = SupportsTier(GLESMultiDrawMode::BaseVertex) ? GLESMultiDrawMode::BaseVertex
|
||||
: GLESMultiDrawMode::DrawElements;
|
||||
}
|
||||
@@ -371,7 +376,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunIndirect(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool batched, Bool feedDrawID) {
|
||||
GLsizei drawcount, const GLint* basevertex, Bool batched, Bool feedDrawID,
|
||||
Bool feedBaseVertex) {
|
||||
if (!SupportsTier(batched ? GLESMultiDrawMode::MultiIndirect : GLESMultiDrawMode::Indirect)) return false;
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return false;
|
||||
@@ -408,15 +414,21 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
const Uint previousIndirectBinding = BoundDrawIndirectBufferId();
|
||||
BufferImpl::BindBufferId(GL_DRAW_INDIRECT_BUFFER, g_indirectCommands.id);
|
||||
if (batched) {
|
||||
g_GLESFuncs.glMultiDrawElementsIndirectEXT(mode, type, reinterpret_cast<const void*>(commandBase),
|
||||
drawcount, 0);
|
||||
ForEachViewportRoutingPass([&] {
|
||||
g_GLESFuncs.glMultiDrawElementsIndirectEXT(mode, type, reinterpret_cast<const void*>(commandBase),
|
||||
drawcount, 0);
|
||||
});
|
||||
} else {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
const SizeT commandOffset = commandBase + static_cast<SizeT>(i) * sizeof(DrawElementsIndirectCommand);
|
||||
g_GLESFuncs.glDrawElementsIndirect(mode, type, reinterpret_cast<const void*>(commandOffset));
|
||||
ForEachViewportRoutingPass([&] {
|
||||
g_GLESFuncs.glDrawElementsIndirect(mode, type, reinterpret_cast<const void*>(commandOffset));
|
||||
});
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
}
|
||||
BufferImpl::BindBufferId(GL_DRAW_INDIRECT_BUFFER, previousIndirectBinding);
|
||||
NoteTierExecuted(batched ? GLESMultiDrawMode::MultiIndirect : GLESMultiDrawMode::Indirect);
|
||||
@@ -428,15 +440,19 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunBaseVertexLoop(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID) {
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID, Bool feedBaseVertex) {
|
||||
if (!SupportsTier(GLESMultiDrawMode::BaseVertex)) return false;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
g_GLESFuncs.glDrawElementsBaseVertex(mode, count[i], type, indices[i],
|
||||
basevertex ? basevertex[i] : 0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
ForEachViewportRoutingPass([&] {
|
||||
g_GLESFuncs.glDrawElementsBaseVertex(mode, count[i], type, indices[i],
|
||||
basevertex ? basevertex[i] : 0);
|
||||
});
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
NoteTierExecuted(GLESMultiDrawMode::BaseVertex);
|
||||
return true;
|
||||
}
|
||||
@@ -446,7 +462,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunRebasedDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID) {
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID,
|
||||
Bool feedBaseVertex) {
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return false;
|
||||
|
||||
@@ -479,7 +496,7 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
const Uint8* source = ResolveSubDrawIndices(indexBuffer, indexBufferBytes, indexBufferSize, indices[i],
|
||||
subDrawCount, indexSize);
|
||||
if (!source) {
|
||||
MGLOG_E("DirectGLES multi-draw (drawelements tier): sub-draw %d reads outside the bound index "
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (drawelements tier): sub-draw %d reads outside the bound index "
|
||||
"buffer; skipping the batch",
|
||||
i);
|
||||
return false;
|
||||
@@ -500,11 +517,18 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
g_GLESFuncs.glDrawElements(mode, count[i], GL_UNSIGNED_INT,
|
||||
reinterpret_cast<const void*>(indexBase + cursor * sizeof(Uint32)));
|
||||
// The base vertex is folded into the rewritten index stream here, so the
|
||||
// driver sees none - but gl_BaseVertex still has to report the value the
|
||||
// application passed for this sub-draw.
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
ForEachViewportRoutingPass([&] {
|
||||
g_GLESFuncs.glDrawElements(mode, count[i], GL_UNSIGNED_INT,
|
||||
reinterpret_cast<const void*>(indexBase + cursor * sizeof(Uint32)));
|
||||
});
|
||||
cursor += static_cast<SizeT>(count[i]);
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, previousIndexBinding);
|
||||
NoteTierExecuted(GLESMultiDrawMode::DrawElements);
|
||||
return true;
|
||||
@@ -580,7 +604,7 @@ void main() {
|
||||
|
||||
const GLuint shader = g_GLESFuncs.glCreateShader(GL_COMPUTE_SHADER);
|
||||
if (shader == 0) {
|
||||
MGLOG_E("DirectGLES multi-draw (compute tier): glCreateShader(GL_COMPUTE_SHADER) failed");
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): glCreateShader(GL_COMPUTE_SHADER) failed");
|
||||
return false;
|
||||
}
|
||||
const char* source = kFlattenComputeSource;
|
||||
@@ -591,14 +615,14 @@ void main() {
|
||||
if (status != GL_TRUE) {
|
||||
char log[1024] = {};
|
||||
g_GLESFuncs.glGetShaderInfoLog(shader, sizeof(log) - 1, nullptr, log);
|
||||
MGLOG_E("DirectGLES multi-draw (compute tier): index-flattening shader failed to compile: %s", log);
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): index-flattening shader failed to compile: %s", log);
|
||||
g_GLESFuncs.glDeleteShader(shader);
|
||||
return false;
|
||||
}
|
||||
|
||||
const GLuint program = g_GLESFuncs.glCreateProgram();
|
||||
if (program == 0) {
|
||||
MGLOG_E("DirectGLES multi-draw (compute tier): glCreateProgram failed");
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): glCreateProgram failed");
|
||||
g_GLESFuncs.glDeleteShader(shader);
|
||||
return false;
|
||||
}
|
||||
@@ -609,7 +633,7 @@ void main() {
|
||||
if (status != GL_TRUE) {
|
||||
char log[1024] = {};
|
||||
g_GLESFuncs.glGetProgramInfoLog(program, sizeof(log) - 1, nullptr, log);
|
||||
MGLOG_E("DirectGLES multi-draw (compute tier): index-flattening program failed to link: %s", log);
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): index-flattening program failed to link: %s", log);
|
||||
g_GLESFuncs.glDeleteProgram(program);
|
||||
return false;
|
||||
}
|
||||
@@ -619,7 +643,7 @@ void main() {
|
||||
g_uDrawCount = g_GLESFuncs.glGetUniformLocation(program, "uDrawCount");
|
||||
g_uTotalIndices = g_GLESFuncs.glGetUniformLocation(program, "uTotalIndices");
|
||||
g_computeProgramFailed = false;
|
||||
MGLOG_I("DirectGLES multi-draw: index-flattening compute program ready (id %u)", program);
|
||||
MGLOG_D("DirectGLES multi-draw: index-flattening compute program ready (id %u)", program);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -837,8 +861,15 @@ void main() {
|
||||
// afterwards would mean unpicking the program, SSBO and index bindings
|
||||
// PrepareForDraw just made, and a dispatch inside an open transform feedback
|
||||
// span is not legal at all. On success it hands back a flattened index stream.
|
||||
// A batch whose sub-draws carry their own base vertices cannot be flattened either
|
||||
// when the program reads gl_BaseVertex: one draw call leaves one uniform value.
|
||||
// Asked conservatively because this decision precedes PrepareForDraw - see
|
||||
// CurrentProgramMayNeedPerSubDrawBuiltins. Flattening is the irreversible half:
|
||||
// once the batch is one draw the values are gone, whereas declining to flatten only
|
||||
// costs the unrolled tier.
|
||||
FlattenedStream flattened;
|
||||
if (ResolvedTier() == GLESMultiDrawMode::Compute && !CurrentProgramReadsDrawID()) {
|
||||
if (ResolvedTier() == GLESMultiDrawMode::Compute &&
|
||||
!CurrentProgramMayNeedPerSubDrawBuiltins(basevertex != nullptr)) {
|
||||
FlattenWithCompute(mode, count, type, indices, drawcount, basevertex, flattened);
|
||||
}
|
||||
|
||||
@@ -847,13 +878,18 @@ void main() {
|
||||
if (flattened.indexCount != 0) {
|
||||
const Uint previousIndexBinding = BoundIndexBufferId();
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, flattened.bufferId);
|
||||
g_GLESFuncs.glDrawElements(mode, static_cast<GLsizei>(flattened.indexCount), GL_UNSIGNED_INT, nullptr);
|
||||
ForEachViewportRoutingPass([&] {
|
||||
g_GLESFuncs.glDrawElements(mode, static_cast<GLsizei>(flattened.indexCount), GL_UNSIGNED_INT, nullptr);
|
||||
});
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, previousIndexBinding);
|
||||
return;
|
||||
}
|
||||
|
||||
// Now that PrepareForDraw has synced the program, both questions have real answers;
|
||||
// the tier choice and the per-sub-draw feeds use those, not the guess above.
|
||||
const Bool feedDrawID = CurrentProgramReadsDrawID();
|
||||
const GLESMultiDrawMode tier = ResolveTierForBatch(feedDrawID, hasIndexBuffer);
|
||||
const Bool feedBaseVertex = basevertex != nullptr && CurrentProgramReadsBaseVertex();
|
||||
const GLESMultiDrawMode tier = ResolveTierForBatch(feedDrawID, feedBaseVertex, hasIndexBuffer);
|
||||
|
||||
Bool drawn = false;
|
||||
switch (tier) {
|
||||
@@ -861,16 +897,19 @@ void main() {
|
||||
drawn = RunExt(mode, count, type, indices, drawcount, basevertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::MultiIndirect:
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/true, feedDrawID);
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/true, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::Indirect:
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/false, feedDrawID);
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/false, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::BaseVertex:
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::DrawElements:
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::Compute:
|
||||
// Its pre-pass ran above; reaching here means it declined this batch's shape.
|
||||
@@ -883,10 +922,15 @@ void main() {
|
||||
// below are the floor: a base-vertex replay where the driver has one, and the
|
||||
// rewritten index stream where it does not. Both are safe for any batch these
|
||||
// entry points can receive.
|
||||
if (!drawn) drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
if (!drawn) drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
if (!drawn) {
|
||||
MGLOG_E("DirectGLES multi-draw: no usable tier for a %d sub-draw batch (mode 0x%x, type 0x%x); "
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID,
|
||||
feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw: no usable tier for a %d sub-draw batch (mode 0x%x, type 0x%x); "
|
||||
"the batch was dropped",
|
||||
drawcount, mode, type);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -60,6 +60,115 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Bool BackendTextureFormatAddsAlpha(TextureInternalFormat internalFormat, TextureTarget target);
|
||||
Bool BackendRenderbufferFormatAddsAlpha(TextureInternalFormat internalFormat);
|
||||
Bool ShouldUseCaveatRenderbufferFormat(TextureInternalFormat internalFormat);
|
||||
|
||||
// The CHANNEL WIDENING an image-bindable texture's ES storage takes, so that a format
|
||||
// GLSL ES cannot spell as an image is carried by one it can.
|
||||
//
|
||||
// GL has forty image formats, GLSL ES core has thirteen, and no test device advertises
|
||||
// GL_NV_image_formats - so a shader declaring one of the other twenty-six has no legal
|
||||
// ESSL at all and glBindImageTexture rejects the narrow format outright for most of them
|
||||
// (GL_INVALID_VALUE for nineteen of twenty-six on Adreno, twenty-five on both Malis).
|
||||
// Seventeen have a core format of the SAME per-channel width and component type,
|
||||
// differing only in channel count, and in one of those the emulation is EXACT: GL already
|
||||
// defines an imageLoad from a narrower format as (r, 0, 0, 1) and an imageStore as
|
||||
// dropping the components the format does not have, so the carrier's surplus channels
|
||||
// hold values GL has already named. WidenImageFormatsPass pins them in the shader; this
|
||||
// is the storage half, and DirectGLES::TextureImpl::SyncImageTextureBinding the bind
|
||||
// half. All three ask WidenedCoreEsslImageFormat, so they cannot pick different carriers.
|
||||
//
|
||||
// Reports nothing (InternalFormat == GL_UNKNOWN_MGL) for a format that is core already,
|
||||
// for the nine with no exact carrier (r11f_g11f_b10f, rgb10_a2, rgb10_a2ui, rgba16, rg16,
|
||||
// r16, rgba16_snorm, rg16_snorm, r16_snorm - those keep the honest "no GLSL ES spelling"
|
||||
// diagnostic rather than a silent approximation), and on a driver that HAS
|
||||
// GL_NV_image_formats, where the shader keeps the declared format and no widening may
|
||||
// happen behind it.
|
||||
//
|
||||
// The widened triple REPLACES what GenerateTextureFormatInfo chose, including any
|
||||
// renderability substitution: an image that cannot be image-bound is useless whatever its
|
||||
// attachment behaviour, so the image constraint wins. In practice that only bites
|
||||
// RG8_SNORM/R8_SNORM on a driver without EXT_render_snorm, where the storage stays
|
||||
// signed-normalized instead of becoming the half float that fallback would have picked -
|
||||
// so an image-bound texture in one of those two formats is no longer attachable, and
|
||||
// glGetTexImage on it falls through to the CPU shadow, which a shader-side imageStore
|
||||
// does not update. Accepted deliberately: before the widening, an image binding in either
|
||||
// format was refused outright by every driver tested and the stage that declared it never
|
||||
// compiled at all, so nothing that works today is being given up.
|
||||
//
|
||||
// KNOWN GAP, for the same "all three layers move together" reason: a widened texture that
|
||||
// is ALSO an FBO colour attachment gains one to three writable channels, and a draw into
|
||||
// it can leave values in channels GL says are 0 and 1. Sampling and imageLoad are covered
|
||||
// (the swizzle composition in SyncTextureParamsToBackend and the shader-side mask), but a
|
||||
// glReadPixels/glGetTexImage that asks for more channels than the frontend format has
|
||||
// would see them. Closing it needs the per-draw-buffer colour mask the three-channel
|
||||
// widening already carries (FramebufferImpl::g_alphaWidenedDrawBufferMask) generalized
|
||||
// from "alpha" to a channel count, which is its own change.
|
||||
// How the FRONTEND's CPU shadow for a widened format is laid out relative to the carrier's
|
||||
// transfer, i.e. what the upload has to do to it. Almost every entry is `Components`: the
|
||||
// shadow already holds SourceChannels components of exactly the carrier's own type, so
|
||||
// padding it out to four is the whole conversion. The packed entries do not - their shadow
|
||||
// is ONE 32-bit word per texel - and reading such a word as components of the carrier's
|
||||
// type takes twelve or sixteen bytes out of four and shears the level.
|
||||
enum class ImageWidenSourceEncoding : Uint8 {
|
||||
Components = 0,
|
||||
// r11f_g11f_b10f: GL_UNSIGNED_INT_10F_11F_11F_REV -> four GL_FLOATs of an rgba16f.
|
||||
PackedFloat11f11f10f,
|
||||
// rgb10_a2 and rgb10_a2ui: GL_UNSIGNED_INT_2_10_10_10_REV -> four GL_UNSIGNED_SHORT
|
||||
// channel CODES of an rgba16ui. The same split serves both: the two formats differ
|
||||
// only in what the codes MEAN, which is the shader's business and not the transfer's.
|
||||
PackedInt2101010Rev,
|
||||
};
|
||||
|
||||
struct ImageBindableStorageWidening {
|
||||
GLenum InternalFormat = GL_UNKNOWN_MGL;
|
||||
GLenum Format = GL_UNKNOWN_MGL;
|
||||
GLenum Type = GL_UNKNOWN_MGL;
|
||||
// Channels the FRONTEND format has, i.e. how many of the carrier's four the client
|
||||
// data fills. The rest are uploaded as 0, and the fourth as the format's implied 1.
|
||||
Uint SourceChannels = 0;
|
||||
// Whether that implied 1 is the integer one or a saturated normalized field - the
|
||||
// transfer type cannot tell the two apart (GL_UNSIGNED_BYTE serves both RG8 and
|
||||
// RG8UI), so the carrier decides.
|
||||
Bool IntegerData = false;
|
||||
// What the upload has to do to the frontend shadow before it describes the level to
|
||||
// the driver (PrepareImageWidenedUpload).
|
||||
ImageWidenSourceEncoding SourceEncoding = ImageWidenSourceEncoding::Components;
|
||||
// Non-zero when the carrier holds this format's channels as the INTEGER CODES of a
|
||||
// NORMALIZED value - the seven 16-bit and 10-bit normalized formats, which core ESSL
|
||||
// has no image format of any width for and which a float carrier would requantise.
|
||||
// Each entry is the largest code that channel can hold, i.e. the denominator of GL 4.6
|
||||
// 2.3.5; SignedNormalized picks which of the two conversions it is the denominator of.
|
||||
//
|
||||
// Two things depend on it, both because the ES storage no longer shares the frontend
|
||||
// format's component class: the upload pads a missing alpha with ChannelMax[3] instead
|
||||
// of the transfer type's own "one" (through a uint carrier the saturated field IS the
|
||||
// one), and glGetTexImage divides the codes back out into the floats the application
|
||||
// is still owed.
|
||||
Uint ChannelMax[4] = {0u, 0u, 0u, 0u};
|
||||
Bool SignedNormalized = false;
|
||||
|
||||
Bool CarriesNormalizedCodes() const { return ChannelMax[0] != 0u; }
|
||||
explicit operator Bool() const { return InternalFormat != GL_UNKNOWN_MGL; }
|
||||
};
|
||||
ImageBindableStorageWidening GetImageBindableStorageWidening(TextureInternalFormat internalFormat);
|
||||
|
||||
// The single-channel core format an image-bindable BUFFER texture's view is SPLIT into, or
|
||||
// GL_UNKNOWN_MGL for a format that needs no split (or has no core base).
|
||||
//
|
||||
// A buffer texture cannot be widened: its texels are the application's buffer object, at
|
||||
// the size and layout the application gave it, and it is usually also a vertex, index or
|
||||
// storage buffer whose bytes are not ours to restride. But an rg32f view of N texels and
|
||||
// an r32f view of 2N texels describe exactly the SAME bytes, so the split changes only
|
||||
// how the shader subscripts them - component j of texel i is texel 2i + j of the base
|
||||
// view - which WidenImageFormatsPass rewrites every access to do. The same rule as the
|
||||
// widening decides WHETHER: a driver that can spell rg32f for an imageBuffer needs
|
||||
// nothing.
|
||||
//
|
||||
// KNOWN GAP, and the reason this is not applied to a texture that is merely sampled: a
|
||||
// buffer texture that is BOTH image-bound and read through a samplerBuffer would have its
|
||||
// sampled view split too, and the sampler side is not rewritten. Accepted for the same
|
||||
// reason the storage widening's gaps are - on a driver where the split applies at all
|
||||
// there is no legal ESSL for the image declaration, so such a program did not compile.
|
||||
GLenum GetImageBindableBufferSplitFormat(TextureInternalFormat internalFormat);
|
||||
} // namespace TextureImpl
|
||||
|
||||
namespace FramebufferImpl {} // namespace FramebufferImpl
|
||||
@@ -115,6 +224,16 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Bool StoreWideRowsToClient(const Uint8* wide, GLenum wideType, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, const ReadbackChannelMapping& mapping, GLenum type,
|
||||
void* pixels, Bool applyPackImageParams);
|
||||
|
||||
// Stores packed 32-bit source words verbatim, with the same destination addressing, PACK
|
||||
// parameters and pixel-pack-buffer handling as StoreWideRowsToClient. For the sources whose
|
||||
// storage word already IS the client word (MG_Util::IsRawPackedPixelTransfer): routing those
|
||||
// through the wide float intermediate re-encodes them, and the RGB9_E5 encoder canonicalizes
|
||||
// the shared exponent, so glGetTexImage would answer with different bits than were stored.
|
||||
// `srcWords` holds sliceHeight * sliceCount tightly stacked rows of `width` 32-bit words.
|
||||
// False when `type` is not a 4-byte packed type.
|
||||
Bool StorePackedWordsToClient(const Uint8* srcWords, GLsizei width, GLsizei sliceHeight, GLsizei sliceCount,
|
||||
GLenum type, void* pixels, Bool applyPackImageParams);
|
||||
} // namespace ReadbackImpl
|
||||
|
||||
namespace PrgramImpl {
|
||||
@@ -130,7 +249,247 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// drawBufferCount <= 1, i.e. for everything but a framebuffer that actually
|
||||
// enables several draw buffers, so the ordinary single-target shader is untouched.
|
||||
String BroadcastLegacyFragColor(String glslCode, GLenum shaderType, Uint drawBufferCount);
|
||||
// SPIRV-Cross emits `#extension GL_EXT_texture_buffer : require` for every buffer-texture
|
||||
// sampler when it targets ESSL below 320, and offers no way to ask for the OES spelling.
|
||||
// On a driver that advertises only GL_OES_texture_buffer that directive is a compile
|
||||
// error, so the name is retargeted in the emitted source. A no-op on every other tier:
|
||||
// ES 3.2 needs no directive at all and an EXT driver already has the right one.
|
||||
String RetargetTextureBufferExtension(String glslCode,
|
||||
MG_External::GLESCapabilities::TextureBufferTier tier);
|
||||
// Adds `#extension GL_NV_image_formats : require` when the shader carries an image
|
||||
// format qualifier GLSL ES has no core spelling for. SPIRV-Cross prints the format and
|
||||
// asks for nothing, so the request has to be made here. `needed` is the caller's answer,
|
||||
// because only it knows which formats are in play AND whether the driver advertises the
|
||||
// extension - requesting an unadvertised extension is itself a compile error, so this is
|
||||
// never emitted speculatively. A no-op when not needed or already present.
|
||||
String RequestExtendedImageFormats(String glslCode, Bool needed);
|
||||
// Adds `#extension GL_OES_viewport_array : require` when the emitted ESSL names
|
||||
// gl_ViewportIndex. SPIRV-Cross prints that identifier and asks for nothing (unlike
|
||||
// gl_Layer, which it backs with GL_NV_viewport_array2 on ES) and ESSL has no core
|
||||
// spelling for it at any version, so the request has to be made here or the stage does
|
||||
// not compile - which loses the whole program, not just the multi-viewport routing.
|
||||
// `needed` is the caller's answer for the same reason as above: only it knows whether the
|
||||
// driver advertises the extension, and requesting an unadvertised one is itself a compile
|
||||
// error, so this is never emitted speculatively. A no-op when not needed or already
|
||||
// present.
|
||||
String RequestViewportArrayExtension(String glslCode, Bool needed);
|
||||
// Writes a format layout qualifier into the image declarations named in
|
||||
// `esslFormatByUniformName` that still have none. The completion half of the image-format
|
||||
// bake, and ONLY that: the SPIR-V pass (BakeImageFormatsPass) is what normally puts the
|
||||
// format in, but SPIRV-Cross throws rather than printing the formats it calls
|
||||
// desktop-only when it targets ESSL - r8ui among them, which is what the stencil half of
|
||||
// KHR-GL4x.packed_depth_stencil.stencil_texturing binds - and a throw loses the whole
|
||||
// stage. So those formats stay out of the module and are spelled here instead, on the
|
||||
// emitted text, where nothing can refuse them.
|
||||
//
|
||||
// Declarations that already carry a format are left exactly as they are, whoever wrote
|
||||
// it. Must run before RemoveLayoutBinding, which is where an image's layout qualifier
|
||||
// stops being safe to edit by hand.
|
||||
String BakeImageFormatQualifiers(String glslCode, const UnorderedMap<String, String>& esslFormatByUniformName);
|
||||
String RemoveLayoutBinding(const String& glslCode);
|
||||
// Prefix of the per-element scalar declarations RemapImageArrayElementUnits splits an
|
||||
// image array into; the suffix is the array's own name and the element's index.
|
||||
constexpr const char* IMAGE_ARRAY_ELEMENT_PREFIX = "mg_imageElem_";
|
||||
// One image ARRAY whose elements the application pointed at units that are not
|
||||
// consecutive-from-element-zero.
|
||||
struct ImageArrayUnitPlan {
|
||||
String name; // the array's name, exactly as the emitted ESSL declares it
|
||||
Vector<Int> units; // the frontend image unit element k has to reach
|
||||
};
|
||||
// Desktop GL lets an application give each element of an image array an ARBITRARY unit
|
||||
// (glUniform1i per element). ES has no such call at all - "ES image units come
|
||||
// exclusively from the layout(binding=N) qualifier" - and one declaration carries one
|
||||
// binding, so ESSL nails an array's elements to the CONSECUTIVE units N, N+1, N+2, ...
|
||||
// MobileGL used to stamp element [0]'s unit as the binding and let the rest fall where
|
||||
// they fell: KHR-GL4x.shader_image_load_store.advanced-sso-simple assigns 0,2,4,6 and
|
||||
// 1,3,5,7, so its two programs actually addressed 0,1,2,3 and 1,2,3,4 - one layer got the
|
||||
// wrong value and three were never written, with no GL error and no link log. The same
|
||||
// defect for SAMPLER arrays was fixed API-side (SubscriptUniformNameForElement); an image
|
||||
// array has no API side to fix, because ES makes glUniform1i on an image uniform an
|
||||
// INVALID_OPERATION.
|
||||
//
|
||||
// Repaired by SPLITTING the array into one SCALAR image uniform per element, each with
|
||||
// its own layout(binding = N), and rewriting `name[k]` to the scalar declared for
|
||||
// element k. One declaration carries one binding, so one declaration per unit is the
|
||||
// only spelling that reaches an arbitrary set of them.
|
||||
//
|
||||
// That rewrite needs every k in the emitted text to be a LITERAL, and it is:
|
||||
// LegalizeResourceArrayIndexingForEssl has already folded or lowered every dynamic
|
||||
// image-array subscript in the module, because ESSL forbids one outright ("image arrays
|
||||
// indexed with non-constant expressions are forbidden in GLSL ES", Mesa 26.1.4 at
|
||||
// ES 3.2, on a raw GLES probe with no MobileGL in the loop). The earlier shape here -
|
||||
// widening the array to cover the whole span of units and routing each subscript through
|
||||
// a `const highp int` offset table - was written before that pass covered images, and
|
||||
// the table lookup was itself one of the non-constant expressions the same probe refuses.
|
||||
// The split also costs exactly the image uniforms the application declared, where the
|
||||
// widening cost the whole SPAN (seven for the four elements of
|
||||
// KHR-GL42.shader_image_load_store.advanced-sso-simple), so there is no budget for it to
|
||||
// fail to fit in.
|
||||
//
|
||||
// Declines - leaving the array exactly as it was, and naming it in `outDeclined` for the
|
||||
// caller to report - when the emitted extent disagrees with the reflection, when the
|
||||
// array is reached by anything other than a subscript, or when a subscript is not a
|
||||
// literal element index. Silence was the whole defect here, so a decline must be audible.
|
||||
//
|
||||
// Must run AFTER RebindImageUniformsToFrontendUnits and BakeImageFormatQualifiers (both
|
||||
// key on the GL uniform name and on a binding already being stamped) and BEFORE
|
||||
// SplitReadWriteImageUniforms (so each element that is both read and written is split
|
||||
// with its own binding already on it) and RemoveLayoutBinding (which is what preserves
|
||||
// image bindings). Like them, it is downstream of the L2 shader-translation memo, so the
|
||||
// per-program units it reads need no entry in BuildEsslTranslationKey.
|
||||
String RemapImageArrayElementUnits(const String& glslCode, const Vector<ImageArrayUnitPlan>& plans,
|
||||
Vector<String>* outDeclined = nullptr);
|
||||
// The member list of a `gl_PerVertex { ... }` redeclaration in already-emitted ESSL -
|
||||
// the text between the braces, verbatim - or nullopt when the shader does not redeclare
|
||||
// the block in that direction. `input` selects the `in gl_PerVertex` form over the
|
||||
// `out` one.
|
||||
//
|
||||
// Exists so BuildPassthroughTessControlEssl can MIRROR the stages it has to sit between
|
||||
// rather than guess at them. Whether SPIRV-Cross redeclares the built-in block, and with
|
||||
// which members, depends on what the application's shader touched; a synthesized stage
|
||||
// that redeclares a different shape than its neighbours is an ES link error against a
|
||||
// program that has no other problem.
|
||||
std::optional<String> ExtractPerVertexBlockMembers(const String& essl, Bool input);
|
||||
// The pass-through tessellation control stage GL 4.6 core 11.2.2 describes: "the input
|
||||
// patch is passed through unmodified", the output patch has PATCH_VERTICES vertices, and
|
||||
// the levels come from the PATCH_DEFAULT_OUTER_LEVEL / PATCH_DEFAULT_INNER_LEVEL state.
|
||||
//
|
||||
// Desktop GL makes the control stage OPTIONAL. OpenGL ES 3.2 does not: it has no
|
||||
// PATCH_DEFAULT_*_LEVEL state at all (only glPatchParameteri, for PATCH_VERTICES) and
|
||||
// rejects a program that has an evaluation stage without a control stage - with an EMPTY
|
||||
// info log, verified on an Adreno 830 with no MobileGL in the process. MobileGL's own
|
||||
// frontend link succeeds, so the program reports GL_LINK_STATUS = TRUE, program 0 is
|
||||
// bound in its place, and every draw silently renders nothing.
|
||||
//
|
||||
// `inPerVertexMembers` / `outPerVertexMembers` are the member lists to redeclare gl_in
|
||||
// and gl_out with - normally taken from the neighbouring stages' own emitted ESSL via
|
||||
// ExtractPerVertexBlockMembers, and empty to leave the driver's built-in declaration
|
||||
// alone, which is what matching a neighbour that did not redeclare requires.
|
||||
//
|
||||
// All four outer levels and both inner levels are written unconditionally: writing a
|
||||
// level the evaluation stage's domain does not use is legal and ignored, and it saves
|
||||
// this from having to know the domain. They are literal 1.0 because that is the GL
|
||||
// default and glPatchParameterfv - their only setter - is a stub in this frontend
|
||||
// (MG_Impl/GLImpl/Exporting/Definitions.cpp). Implementing that entry point means making
|
||||
// the levels a parameter here AND part of what makes a built program stale, exactly as
|
||||
// PATCH_VERTICES already is; the two must move together, so they are named together.
|
||||
//
|
||||
// The same stage, for the same reason, that DirectVulkan synthesizes in
|
||||
// ProgramFactory::BuildPassthroughTessControlSource - Vulkan likewise requires both
|
||||
// tessellation stages. Kept as two generators rather than one because the two targets
|
||||
// disagree on everything but the algorithm: desktop GLSL 450 against ESSL, a fixed
|
||||
// gl_PerVertex shape that Vulkan matches structurally against a mirrored one, and a
|
||||
// VkShaderModule against a driver shader object.
|
||||
String BuildPassthroughTessControlEssl(Uint esslVersion, Uint patchVertices,
|
||||
const String& inPerVertexMembers,
|
||||
const String& outPerVertexMembers);
|
||||
// Prefix of the writeonly half a read+write image uniform is split into (see
|
||||
// SplitReadWriteImageUniforms); the suffix is the image's own (already access-tagged) name.
|
||||
constexpr const char* IMAGE_WRITE_ALIAS_PREFIX = "mg_imageWrite_";
|
||||
// The three names SplitReadWriteImageUniforms renames a rewritten image declaration
|
||||
// under, one per REPAIR it can apply. Which one a stage picks is decided by that stage's
|
||||
// own accesses, so two stages that use an image the same way arrive at the SAME name and
|
||||
// two that use it differently arrive at different ones - which is exactly the property
|
||||
// the rename exists for, at no cost to the stages that agree. Exposed for the tests.
|
||||
constexpr const char* IMAGE_READONLY_ALIAS_PREFIX = "mg_imageRo_";
|
||||
constexpr const char* IMAGE_WRITEONLY_ALIAS_PREFIX = "mg_imageWo_";
|
||||
constexpr const char* IMAGE_SPLIT_READ_ALIAS_PREFIX = "mg_imageRw_";
|
||||
// ESSL refuses an image variable that carries a format qualifier other than r32f /
|
||||
// r32i / r32ui unless it also carries `readonly` or `writeonly` (GLSL ES 3.10 4.9 /
|
||||
// 3.20 4.10; glslang enforces it verbatim in ParseHelper.cpp's layoutObjectCheck).
|
||||
// SPIRV-Cross emits NEITHER for an image the shader both reads and writes: it
|
||||
// speculatively decorates every storage image NonWritable+NonReadable
|
||||
// (fixup_image_load_store_access), then OpImageRead clears NonReadable and
|
||||
// OpImageWrite clears NonWritable, and to_qualifiers_glsl only prints `readonly`
|
||||
// from NonWritable and `writeonly` from NonReadable. Desktop GLSL is happy with the
|
||||
// bare declaration, so the frontend raises no error and the illegal ESSL only shows
|
||||
// up as a device compile failure - and then as a silently no-op draw.
|
||||
//
|
||||
// Restores a legal declaration, and RENAMES it after the repair it applied while doing so:
|
||||
// * loaded only -> add `readonly`, rename under IMAGE_READONLY_ALIAS_PREFIX
|
||||
// * stored only -> add `writeonly`, rename under IMAGE_WRITEONLY_ALIAS_PREFIX
|
||||
// * both -> emit TWO declarations on the same binding and of the
|
||||
// same type, `coherent readonly
|
||||
// <IMAGE_SPLIT_READ_ALIAS_PREFIX><name>` and `coherent
|
||||
// writeonly <IMAGE_WRITE_ALIAS_PREFIX><that name>`, point
|
||||
// every imageStore at the second one, and follow each of
|
||||
// those stores with `memoryBarrierImage();`. Several image
|
||||
// variables may share an image unit as long as they have
|
||||
// the same type and format, which is exactly what the pair
|
||||
// is.
|
||||
//
|
||||
// The rename is the other half of the repair and applies to all three cases. The qualifier
|
||||
// chosen above is a decision about ONE STAGE's accesses, and GLSL requires a uniform
|
||||
// declared in two stages to be declared identically - so a shader that stores an image from
|
||||
// the vertex stage and loads it from the fragment stage came out of here `writeonly` in one
|
||||
// and `readonly` in the other. Adreno merges the two same-named declarations and silently
|
||||
// drops the vertex-stage STORES: no GL error, no link log, LINK_STATUS = 1, and the image
|
||||
// still reads back its initial contents
|
||||
// (KHR-GL4x.shader_image_load_store.advanced-memory-dependentInvocation; a raw-ES probe
|
||||
// isolated the trigger to the same-name/mismatched-qualifier pair, and only when both
|
||||
// carry `coherent`). Renaming leaves no cross-stage variable to merge.
|
||||
//
|
||||
// The name is keyed on the REPAIR, not on the stage, and that distinction is the whole
|
||||
// point: two stages that use an image the same way emit byte-identical declarations, so
|
||||
// letting them keep one shared name costs nothing and merging them is correct, while two
|
||||
// stages that use it differently land on different prefixes and cannot be merged at all.
|
||||
// A per-STAGE tag also satisfied the first requirement but violated the second: it made
|
||||
// the SAME image a distinct uniform in every stage that named it, and Adreno allocates
|
||||
// image LOCATIONS per distinct uniform. KHR-GL43.shading_language_420pack.
|
||||
// binding_images_texture_type_* declares three read+write images in each of its five
|
||||
// stages; merged that is 6 image uniforms, per-stage-tagged it is 30, and the Adreno 830
|
||||
// linker answered "Error: Image Image location or component exceeds max allowed. Error:
|
||||
// Linking failed." - which, the frontend having already published LINK_STATUS = TRUE from
|
||||
// glslang's link, surfaced only as every draw silently doing nothing and the images
|
||||
// reading back zero. Mali and Mesa link the same text, so nothing but a device gate
|
||||
// catches this.
|
||||
//
|
||||
// A declaration SPIRV-Cross already tagged `readonly` or `writeonly` needs no qualifier
|
||||
// repair, but it is NOT stage-independent: that tag is derived from the accesses of the
|
||||
// stage being emitted, so an image stored in the vertex stage and loaded in the fragment
|
||||
// stage arrives here as `coherent writeonly g_image` and `coherent readonly g_image` -
|
||||
// one name, two spellings, which is exactly the pair Adreno merges. Those declarations
|
||||
// are therefore renamed too, keyed on the qualifier they already carry (readonly ->
|
||||
// IMAGE_READONLY_ALIAS_PREFIX, writeonly -> IMAGE_WRITEONLY_ALIAS_PREFIX) and with
|
||||
// nothing but the identifier changed. Stages that agree still reach the same alias and
|
||||
// stay merged, so this costs no shader an extra image uniform.
|
||||
//
|
||||
// The declarations this pass still leaves untouched keep their names: one carrying BOTH
|
||||
// readonly and writeonly (a spelling no access analysis produces, so it came from the
|
||||
// application and is identical everywhere), and one carrying NEITHER, which is legal only
|
||||
// for the r32f/r32i/r32ui formats and is likewise spelled the same in every stage.
|
||||
//
|
||||
// The `coherent` on both halves of the pair is load-bearing, not decoration: GLSL only
|
||||
// guarantees a write through one image variable is visible to a read through a DIFFERENT
|
||||
// one when both are coherent, and the split is what makes a same-variable
|
||||
// read-after-write cross-variable. The single-declaration repairs above do not get it -
|
||||
// nothing aliases them.
|
||||
//
|
||||
// The barrier is the other half of the same problem, and coherent alone did not cover it:
|
||||
// visibility is not ORDER. Within one invocation the ES compiler sees a write to one
|
||||
// variable and a read of another it has no reason to believe alias, and is free to serve
|
||||
// the read from before the write - which is what advanced-memory-order's store/load/
|
||||
// compare loop measured on Adreno. memoryBarrierImage() orders exactly those two, is core
|
||||
// GLSL ES 3.10 in every stage, and is not an execution barrier, so it is legal in
|
||||
// non-uniform control flow. It costs something in a shader that stores to a read+write
|
||||
// image in a loop, which is why it is confined to the split pair.
|
||||
//
|
||||
// Budget note: the split DOUBLES the image-uniform count of the stage it fires in, so
|
||||
// a driver advertising a tight GL_MAX_{FRAGMENT,VERTEX,...}_IMAGE_UNIFORMS can turn a
|
||||
// shader that used to compile into a link failure. ES only guarantees 4 fragment image
|
||||
// uniforms, so a shader with more than half the limit in read+write images is the case
|
||||
// to watch.
|
||||
//
|
||||
// Runs on the transpiled ESSL, so it must see the bindings the frontend units were
|
||||
// already rewritten to and must run before those bindings are stripped - see the call
|
||||
// site in Managers.cpp. Its output is a function of the emitted text alone - it needs no
|
||||
// stage and no per-program state - so it adds nothing to BuildEsslTranslationKey either.
|
||||
//
|
||||
// `outSplitCount`, when given, receives the number of declarations that were actually
|
||||
// doubled - i.e. exactly how many image uniforms this stage gained over what the
|
||||
// application declared. Zero for every shader but a handful, and the only number the
|
||||
// budget note above can be reported with.
|
||||
String SplitReadWriteImageUniforms(const String& glslCode, Uint* outSplitCount = nullptr);
|
||||
// Prefix of the per-sampler float uniform that carries GL_TEXTURE_LOD_BIAS into
|
||||
// the shader (see EmulateTextureLodBias); the suffix is the sampler's own name.
|
||||
constexpr const char* LOD_BIAS_UNIFORM_PREFIX = "mg_lodBias_";
|
||||
@@ -142,7 +501,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// the bound texture's (or sampler object's) value into it; a shader whose samplers
|
||||
// all have a zero bias is therefore unaffected. Returns the source unchanged when
|
||||
// there is nothing to rewrite.
|
||||
String EmulateTextureLodBias(const String& glslCode);
|
||||
//
|
||||
// avoidExplicitLodBias leaves lookups that already carry an explicit LOD untouched,
|
||||
// so their constant level stays constant; only the implicit-LOD forms take the bias.
|
||||
// Off by default and only ever set on ANGLE + llvmpipe, where injecting the uniform
|
||||
// into a constant LOD crashes the driver (MOBILEGL_AVOID_EXPLICIT_LOD_BIAS).
|
||||
String EmulateTextureLodBias(const String& glslCode, Bool avoidExplicitLodBias = false);
|
||||
} // namespace PrgramImpl
|
||||
|
||||
namespace Utils {
|
||||
|
||||
@@ -9,7 +9,9 @@
|
||||
#include "BackendObject_DirectVulkan.h"
|
||||
#include "MG_Backend/BackendObject.h"
|
||||
#include "DirectVulkan.h"
|
||||
#include "SubgroupSupportPolicy.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include "MG_State/GLState/TextureState/TextureState.h"
|
||||
#include "MG_Util/Classifiers/TextureEnumClassifier.h"
|
||||
#include "MG_Util/Converters/MGToGL/TextureEnumConverter.h"
|
||||
@@ -383,6 +385,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
if (MG_State::pGLContext) {
|
||||
MG_State::pGLContext->InvalidateCompileEnv();
|
||||
}
|
||||
PopulateFormatCapabilities(physicalDevice.handle, vkGetPhysicalDeviceFormatProperties, m_vulkanCaps,
|
||||
MutableFormatCapabilities());
|
||||
PrintFormatCapabilities(GetFormatCapabilities());
|
||||
@@ -495,32 +500,129 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
.RendererName = "Magma",
|
||||
.BackendName = "Direct (Vulkan)",
|
||||
.ExtraVendor = Nullopt,
|
||||
.RendererGLInfo = {.TargetGLVersion = {4, 0, 0},
|
||||
.RendererGLInfo = {.TargetGLVersion = {4, 3, 0},
|
||||
.TargetGLSLVersion = {4, 6, 0},
|
||||
// Baseline advertisement (no shader subgroup, no timer queries); a
|
||||
// live backend reconciles its copy in UpdateAdvertisedExtensions.
|
||||
.Extensions = BuildAdvertisedExtensions(false, false, false),
|
||||
// Baseline advertisement (no runtime-gated capabilities); a live
|
||||
// backend reconciles its copy in UpdateAdvertisedExtensions.
|
||||
.Extensions = BuildAdvertisedExtensions(false, false, false, false, false),
|
||||
.IsCompatibilityProfile = false},
|
||||
.StaticBackendCapability = {.AllowVSOnlyPrograms = false}};
|
||||
return rendererInfo;
|
||||
}
|
||||
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
|
||||
Bool anisotropicFilteringSupported) {
|
||||
Bool anisotropicFilteringSupported,
|
||||
Bool nonZeroIndirectBaseInstanceSupported,
|
||||
Bool cubeMapArraySupported) {
|
||||
Vector<GLExtension> extensions = {
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
|
||||
// The version tokens have to reach the version the backend actually claims:
|
||||
// TargetGLVersion is {4,3,0}, and a list that stopped at OpenGL40 told an
|
||||
// application feature-detecting off these tokens the opposite of what
|
||||
// GL_MAJOR_VERSION / GL_MINOR_VERSION told it.
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, V_OpenGL41, V_OpenGL42, V_OpenGL43,
|
||||
E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_multi_draw_indirect,
|
||||
E_GL_ARB_clear_buffer_object, E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_draw_indirect,
|
||||
E_GL_ARB_multi_draw_indirect,
|
||||
E_GL_ARB_indirect_parameters, E_GL_EXT_framebuffer_object, E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage,
|
||||
E_GL_ARB_texture_storage, E_GL_ARB_texture_storage_multisample, E_GL_ARB_texture_multisample,
|
||||
E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access, E_GL_ARB_shader_draw_parameters,
|
||||
E_GL_ARB_gpu_shader_int64, E_GL_KHR_debug, E_GL_ARB_gpu_shader5, E_GL_ARB_multi_bind,
|
||||
E_GL_ARB_shading_language_420pack, E_GL_ARB_vertex_attrib_binding, E_GL_ARB_shader_image_size,
|
||||
E_GL_ARB_explicit_attrib_location,
|
||||
// Core since GL 3.1 and implemented for every version advertised here. The string
|
||||
// matters because applications gate the ENTRY POINTS on it rather than on the
|
||||
// version: a caller that finds the extension missing never resolves
|
||||
// glGetUniformBlockIndex / glUniformBlockBinding, and one that then uses uniform
|
||||
// blocks anyway calls through a null pointer.
|
||||
E_GL_ARB_uniform_buffer_object,
|
||||
// Sampling the stencil aspect through DEPTH_STENCIL_TEXTURE_MODE. Core from 4.3,
|
||||
// so on a 4.0 context the string is the only way to reach it.
|
||||
E_GL_ARB_stencil_texturing,
|
||||
// Unconditional, unlike DirectGLES: a GL texture view is a second set of VkImageViews
|
||||
// over the same VkImage with a sub-range and possibly a reinterpreted VkFormat, which
|
||||
// is core Vulkan on every device MobileGL runs on. Format-reinterpreting views need
|
||||
// VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT on the image, which SyncTextureResource sets for
|
||||
// every immutable-storage texture (see the comment there).
|
||||
E_GL_ARB_texture_view,
|
||||
// Core since 3.2 and implemented here on both backends - glDrawElementsBaseVertex,
|
||||
// glDrawRangeElementsBaseVertex, glDrawElementsInstancedBaseVertex and
|
||||
// glMultiDrawElementsBaseVertex all reach real per-draw vertex rebasing. The string
|
||||
// was simply never emitted, which left KHR-GL4*.draw_elements_base_vertex_tests
|
||||
// NotSupported on a feature that works.
|
||||
E_GL_ARB_draw_elements_base_vertex,
|
||||
// The whole sync-object family is real and core since 3.2: glFenceSync, glIsSync,
|
||||
// glDeleteSync, glClientWaitSync, glWaitSync and glGetSynciv all live in GLImpl over a
|
||||
// backend fence (a VkFence here, an EGLSync/GLsync on DirectGLES), and glGetInteger64v
|
||||
// answers GL_MAX_SERVER_WAIT_TIMEOUT. The string matters for the same reason
|
||||
// ARB_uniform_buffer_object's does: LWJGL builds GLCapabilities from the extension
|
||||
// list, and a caller that finds GL_ARB_sync missing never resolves the entry points -
|
||||
// then calls through null if it uses fences anyway. Nothing in the CTS gates on this
|
||||
// string, so it is advertised on the strength of the implementation, not a test unlock.
|
||||
E_GL_ARB_sync,
|
||||
// Atomic counters, core since 4.2. glGetActiveAtomicCounterBufferiv and the whole
|
||||
// GL_ATOMIC_COUNTER_BUFFER_* query family are real in GLImpl, and the counter buffer
|
||||
// now reaches the shader on BOTH backends - Magma resolves the lowered
|
||||
// gl_AtomicCounterBlock_<N> from the atomic-counter binding points rather than the
|
||||
// shader-storage ones (see ResolveStorageBufferDescriptor). Withheld here until that
|
||||
// landed, because the counter silently read whatever was bound as SSBO N instead.
|
||||
E_GL_ARB_shader_atomic_counters,
|
||||
// glVertexAttribDivisor, core since 3.3 and real on both backends. Applications
|
||||
// (Better Clouds' GLCompat among them) accept the extension string as an
|
||||
// ALTERNATIVE to a 3.3 context when deciding whether instanced rendering is
|
||||
// available, so withholding it makes MobileGL look less capable than it is.
|
||||
E_GL_ARB_instanced_arrays,
|
||||
// Core GL 3.0-4.3 plumbing that has been real here for as long as the backend has
|
||||
// existed, and that was simply never named. None of these unlocks a single CTS case -
|
||||
// the conformance suite reaches all of them through the version - so they are
|
||||
// advertised for the OTHER consumer of this list: LWJGL builds GLCapabilities from the
|
||||
// string set, and an application that gates its ENTRY POINTS on the string rather than
|
||||
// on the version never resolves them and then calls through null. Each is backed by
|
||||
// the entry points named beside it. Kept identical to the DirectGLES block so the two
|
||||
// backends do not disagree about what MobileGL is.
|
||||
//
|
||||
// glBindVertexArray / glGenVertexArrays / glDeleteVertexArrays / glIsVertexArray.
|
||||
E_GL_ARB_vertex_array_object,
|
||||
// The 14 glSamplerParameter* / glGetSamplerParameter* entry points, including the
|
||||
// integer-valued Iiv/Iuiv forms.
|
||||
E_GL_ARB_sampler_objects,
|
||||
// glMapBufferRange + glFlushMappedBufferRange, which ARB_buffer_storage's persistent
|
||||
// maps are already built on top of.
|
||||
E_GL_ARB_map_buffer_range,
|
||||
// glCopyBufferSubData plus the GL_COPY_READ_BUFFER / GL_COPY_WRITE_BUFFER targets.
|
||||
E_GL_ARB_copy_buffer,
|
||||
// glCopyImageSubData, wired to a real backend hook on both backends.
|
||||
E_GL_ARB_copy_image,
|
||||
// GL_TEXTURE_SWIZZLE_{R,G,B,A,RGBA}, which map onto a VkImageView's component swizzle.
|
||||
E_GL_ARB_texture_swizzle,
|
||||
// GL_INT_2_10_10_10_REV / GL_UNSIGNED_INT_2_10_10_10_REV on glVertexAttribPointer plus
|
||||
// the eight glVertexAttribP* entry points.
|
||||
E_GL_ARB_vertex_type_2_10_10_10_rev,
|
||||
// The R/RG internal formats. Named separately from the float ones because an
|
||||
// application may check either.
|
||||
E_GL_ARB_texture_rg,
|
||||
// GL_DEPTH_COMPONENT32F and GL_DEPTH32F_STENCIL8.
|
||||
E_GL_ARB_depth_buffer_float,
|
||||
// The floating-point colour formats. Unlike the rest of this block this string DOES
|
||||
// gate CTS cases - KHR-GL4*.internalformat.texture2d.*{16f,32f} is keyed on it with no
|
||||
// core-version fallback, so eight cases per version list were NotSupported on formats
|
||||
// the backend has always had.
|
||||
E_GL_ARB_texture_float,
|
||||
// glViewportArrayv / glViewportIndexedf{,v} / glScissorArrayv / glScissorIndexed{,v} /
|
||||
// glDepthRangeArrayv / glDepthRangeIndexed / glGetFloati_v / glGetDoublei_v, over the
|
||||
// 16 viewports GL_MAX_VIEWPORTS reports.
|
||||
E_GL_ARB_viewport_array,
|
||||
// Advertised with GL_NUM_PROGRAM_BINARY_FORMATS = 0, which the
|
||||
// extension explicitly permits. It is also the only thing that
|
||||
// exposes glProgramParameteri before GL 4.1.
|
||||
E_GL_ARB_get_program_binary};
|
||||
// Vulkan's drawIndirectFirstInstance feature is optional. Direct base-instance calls work
|
||||
// without it, but ARB_base_instance also promises non-zero firstInstance in GPU indirect
|
||||
// commands; the renderer supplies true only when that word is legal and gl_InstanceID can
|
||||
// be rebased to OpenGL's zero-based semantics.
|
||||
if (nonZeroIndirectBaseInstanceSupported) {
|
||||
extensions.push_back(E_GL_ARB_base_instance);
|
||||
}
|
||||
if (shaderSubgroupSupported && !MG_Config::Features.DisableSubgroup) {
|
||||
extensions.push_back(E_GL_KHR_shader_subgroup);
|
||||
}
|
||||
@@ -539,6 +641,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (MG_Util::Async::AsyncShaderCompileEnabled()) {
|
||||
extensions.push_back(E_GL_KHR_parallel_shader_compile);
|
||||
}
|
||||
// GL_ARB_gpu_shader_fp64 is opt-in (MOBILEGL_ADVERTISE_FP64), and stays opt-in even on a
|
||||
// device that HAS shaderFloat64. Every `double` in a shader compiles and runs either way
|
||||
// - narrowed to 32 bits where the device has no 64-bit floats, kept whole where it does -
|
||||
// so an application that simply uses doubles needs nothing advertised. What the extension
|
||||
// additionally promises is the whole GL_ARB_gpu_shader_fp64 SURFACE (glUniform*d
|
||||
// conformance, the fp64 built-ins, the state queries), and turning the string on is a
|
||||
// decision about all of it rather than about the shader path alone.
|
||||
if (MG_Config::Features.AdvertiseFp64) {
|
||||
extensions.push_back(E_GL_ARB_gpu_shader_fp64);
|
||||
}
|
||||
// GL_ARB_timer_query gates MC's F3 GPU% (LWJGL checks the extension string);
|
||||
// only advertised when the device actually supports timestamp queries and the
|
||||
// MOBILEGL_DISABLE_TIMERQUERY escape hatch is off.
|
||||
@@ -552,6 +664,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
extensions.push_back(E_GL_EXT_texture_filter_anisotropic);
|
||||
extensions.push_back(E_GL_ARB_texture_filter_anisotropic);
|
||||
}
|
||||
// A cube map array is a 6n-layer VkImage viewed as VK_IMAGE_VIEW_TYPE_CUBE_ARRAY, and that
|
||||
// view type cannot be created without the imageCubeArray device feature - so the string
|
||||
// follows the feature, not the version, exactly as the per-layer attachment bit does.
|
||||
//
|
||||
// Named for the application's benefit rather than the suite's: measured on Adreno 830,
|
||||
// KHR-GL43.texture_gather.plain-gather-*-cube-array already passed without the string, so
|
||||
// this unlocks no conformance case. It is advertised because the feature is real and
|
||||
// because an application that feature-detects cube map arrays off the string (rather than
|
||||
// off the 4.0 version) would otherwise decline a path this backend serves.
|
||||
if (cubeMapArraySupported) {
|
||||
extensions.push_back(E_GL_ARB_texture_cube_map_array);
|
||||
}
|
||||
return extensions;
|
||||
}
|
||||
|
||||
@@ -660,6 +784,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_vulkanCaps = capabilities;
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
if (MG_State::pGLContext) {
|
||||
MG_State::pGLContext->InvalidateCompileEnv();
|
||||
}
|
||||
MutableFormatCapabilities().Clear();
|
||||
}
|
||||
|
||||
@@ -670,9 +797,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// real device timestamp support. ApplyVulkanCapabilitiesForTesting may
|
||||
// run without a renderer; no timer query is advertised then. Rebuilding
|
||||
// the whole list keeps re-runs idempotent.
|
||||
// The opt-in emulated compute path (SubgroupSupportPolicy.h) carries the
|
||||
// extension by itself on devices with no native subgroup support at all; a
|
||||
// device with native subgroups always advertises - and uses - those.
|
||||
const Bool subgroupSupportAdvertised =
|
||||
m_vulkanCaps.SupportsShaderSubgroup ||
|
||||
ShouldEmulateSubgroups(m_vulkanCaps.SupportsShaderSubgroup);
|
||||
m_rendererInfo.RendererGLInfo.Extensions = BuildAdvertisedExtensions(
|
||||
m_vulkanCaps.SupportsShaderSubgroup, pVulkanRenderer && pVulkanRenderer->IsTimerQuerySupported(),
|
||||
pVulkanRenderer && pVulkanRenderer->IsSamplerAnisotropySupported());
|
||||
subgroupSupportAdvertised, pVulkanRenderer && pVulkanRenderer->IsTimerQuerySupported(),
|
||||
pVulkanRenderer && pVulkanRenderer->IsSamplerAnisotropySupported(),
|
||||
pVulkanRenderer && pVulkanRenderer->IsNonZeroIndirectBaseInstanceSupported(),
|
||||
m_vulkanCaps.SupportsImageCubeArray);
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::UpdateDynamicBackendParameters() {
|
||||
@@ -720,6 +855,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
static constexpr SizeT kMaxAdvertisedShaderStorageBlockSize = 512ull * 1024ull * 1024ull;
|
||||
m_dynamicParameters.UniformBufferOffsetAlignment = m_vulkanCaps.UniformBufferOffsetAlignment;
|
||||
m_dynamicParameters.ShaderStorageBufferOffsetAlignment = m_vulkanCaps.ShaderStorageBufferOffsetAlignment;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMin = m_vulkanCaps.AliasedLineWidthRangeMin;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMax = m_vulkanCaps.AliasedLineWidthRangeMax;
|
||||
// Without the samplerAnisotropy feature the limit is unusable, so report 1.0 (no anisotropy)
|
||||
@@ -768,14 +904,80 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// the Uint32 attribute masks the draw path passes around are both bounded by MAX_VERTEX_ATTRIBS.
|
||||
m_dynamicParameters.MaxVertexAttribs = std::min(
|
||||
m_vulkanCaps.MaxVertexAttribs, static_cast<Int>(MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS));
|
||||
m_dynamicParameters.MaxComputeShaderStorageBlocks = m_vulkanCaps.MaxComputeShaderStorageBlocks;
|
||||
m_dynamicParameters.MaxCombinedShaderStorageBlocks = m_vulkanCaps.MaxCombinedShaderStorageBlocks;
|
||||
m_dynamicParameters.MaxComputeUniformBlocks = m_vulkanCaps.MaxComputeUniformBlocks;
|
||||
// Vulkan descriptor limits are not GL limits, and a GL application reads an advertised
|
||||
// limit as an amount it may actually USE. Adreno answers the per-stage/per-set descriptor
|
||||
// queries at descriptor-indexing scale - the same driver whose
|
||||
// GL_MAX_SHADER_STORAGE_BLOCK_SIZE is clamped from 2147483647 further down - so
|
||||
// KHR-GL44.multi_bind.dispatch_bind_buffers_base read GL_MAX_COMPUTE_UNIFORM_BLOCKS,
|
||||
// created that many buffers and spliced that many UBO declarations into a single compute
|
||||
// shader: ~14 s of allocation, then death on std::bad_alloc. Its sibling
|
||||
// dispatch_bind_buffers_range hard-codes 4 buffers and passes, which is the clean
|
||||
// discriminator. Every ceiling below is far above what any desktop driver advertises for
|
||||
// these (84-96 for the binding families) and far below a descriptor-indexing count, so it
|
||||
// can only lower a limit that was never usable in the first place. The zero floor is not
|
||||
// decoration: a driver reporting UINT32_MAX used to arrive here as -1.
|
||||
const auto clampLimit = [](const char* name, Int reported, Int ceiling) {
|
||||
const Int clamped = std::min(std::max(reported, 0), ceiling);
|
||||
if (clamped != reported) {
|
||||
MGLOG_I("DirectVulkan: clamped %s from %d to %d", name, reported, clamped);
|
||||
}
|
||||
return clamped;
|
||||
};
|
||||
// GL 4.6 required minimums, for the record: MAX_COMPUTE_UNIFORM_BLOCKS 12,
|
||||
// MAX_COMPUTE/COMBINED_SHADER_STORAGE_BLOCKS 8, MAX_SHADER_STORAGE_BUFFER_BINDINGS 8,
|
||||
// MAX_UNIFORM_BUFFER_BINDINGS 84, MAX_TEXTURE_BUFFER_SIZE 65536.
|
||||
constexpr Int kMaxAdvertisedBufferBlocks = 256;
|
||||
constexpr Int kMaxAdvertisedTextureBufferSize = 1 << 27; // texels; what desktop GL reports
|
||||
m_dynamicParameters.MaxComputeShaderStorageBlocks =
|
||||
clampLimit("GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS", m_vulkanCaps.MaxComputeShaderStorageBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxCombinedShaderStorageBlocks =
|
||||
clampLimit("GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS", m_vulkanCaps.MaxCombinedShaderStorageBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxComputeUniformBlocks =
|
||||
clampLimit("GL_MAX_COMPUTE_UNIFORM_BLOCKS", m_vulkanCaps.MaxComputeUniformBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxComputeWorkGroupInvocations = m_vulkanCaps.MaxComputeWorkGroupInvocations;
|
||||
m_dynamicParameters.MaxShaderStorageBufferBindings = m_vulkanCaps.MaxShaderStorageBufferBindings;
|
||||
m_dynamicParameters.MaxTextureBufferSize = m_vulkanCaps.MaxTextureBufferSize;
|
||||
m_dynamicParameters.MaxShaderStorageBufferBindings =
|
||||
clampLimit("GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS", m_vulkanCaps.MaxShaderStorageBufferBindings,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
// Per-stage GL_MAX_*_SHADER_STORAGE_BLOCKS. Vulkan has one descriptor limit for every
|
||||
// stage (maxPerStageDescriptorStorageBuffers, which is what MaxComputeShaderStorageBlocks
|
||||
// carries), so the stage limits differ only by whether the stage can have blocks at all.
|
||||
//
|
||||
// Deliberately NOT gated on vertexPipelineStoresAndAtomics, unlike the per-stage image
|
||||
// uniforms below. That gate reads as the obvious one and is wrong here in practice: a
|
||||
// Mali-G925-Immortalis reports vertexPipelineStoresAndAtomics=false (supported AND
|
||||
// enabled) and yet runs all 433 KHR-GL43.constant_expressions.*_tess_* cases correctly
|
||||
// through this backend - those write their result through a storage block declared in a
|
||||
// tessellation stage. Gating would report 0 and turn 433 passing cases into
|
||||
// "unsupported", removing function that demonstrably works.
|
||||
//
|
||||
// The asymmetry with DirectGLES is real and is the point. There, 0 prevents a program
|
||||
// the driver refuses outright at link time; the honest limit converts a silent
|
||||
// wrong-render into a capability an application can route around. Here there is no such
|
||||
// failure to prevent, so the limit stays at what the device can address. If a Vulkan
|
||||
// device is ever found that genuinely rejects such a pipeline, the gate belongs at
|
||||
// pipeline creation where the rejection is observable, not on a feature bit this driver
|
||||
// reports inaccurately.
|
||||
{
|
||||
const Int maxPerStageStorageBlocks =
|
||||
std::min(std::max(m_dynamicParameters.MaxComputeShaderStorageBlocks, 0),
|
||||
std::min(std::max(m_dynamicParameters.MaxCombinedShaderStorageBlocks, 0),
|
||||
std::max(m_dynamicParameters.MaxShaderStorageBufferBindings, 0)));
|
||||
m_dynamicParameters.MaxVertexShaderStorageBlocks = maxPerStageStorageBlocks;
|
||||
m_dynamicParameters.MaxTessControlShaderStorageBlocks = maxPerStageStorageBlocks;
|
||||
m_dynamicParameters.MaxTessEvaluationShaderStorageBlocks = maxPerStageStorageBlocks;
|
||||
// The one hard capability in the set: no geometry stage means no blocks in it.
|
||||
m_dynamicParameters.MaxGeometryShaderStorageBlocks =
|
||||
m_vulkanCaps.SupportsGeometryShader ? maxPerStageStorageBlocks : 0;
|
||||
m_dynamicParameters.MaxFragmentShaderStorageBlocks = maxPerStageStorageBlocks;
|
||||
}
|
||||
m_dynamicParameters.MaxTextureBufferSize = clampLimit(
|
||||
"GL_MAX_TEXTURE_BUFFER_SIZE", m_vulkanCaps.MaxTextureBufferSize, kMaxAdvertisedTextureBufferSize);
|
||||
m_dynamicParameters.TextureBufferOffsetAlignment = m_vulkanCaps.TextureBufferOffsetAlignment;
|
||||
m_dynamicParameters.MaxUniformBufferBindings = m_vulkanCaps.MaxUniformBufferBindings;
|
||||
m_dynamicParameters.MaxUniformBufferBindings = clampLimit(
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS", m_vulkanCaps.MaxUniformBufferBindings, kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxUniformBlockSize = m_vulkanCaps.MaxUniformBlockSize;
|
||||
m_dynamicParameters.MaxImageUnits = std::max(std::min(m_vulkanCaps.MaxImageUnits, maxSupportedTextureUnits), 0);
|
||||
m_dynamicParameters.MaxCombinedImageUniforms = std::max(m_vulkanCaps.MaxCombinedImageUniforms, 0);
|
||||
@@ -797,8 +999,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const Int maxSupportedDrawBuffers = static_cast<Int>(MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS);
|
||||
m_dynamicParameters.MaxDrawBuffers = std::min(m_vulkanCaps.MaxDrawBuffers, maxSupportedDrawBuffers);
|
||||
m_dynamicParameters.MaxColorAttachments = std::min(m_vulkanCaps.MaxColorAttachments, maxSupportedDrawBuffers);
|
||||
m_dynamicParameters.MaxClipDistances = m_vulkanCaps.MaxClipDistances;
|
||||
// Same shape as the image-uniform limits three lines above: maxClipDistances is reported
|
||||
// by every device, but declaring ClipDistance in a module needs the shaderClipDistance
|
||||
// FEATURE, which VulkanRenderer enables exactly where the physical device has it. Without
|
||||
// it the limit describes a capacity no shader may use, so report none.
|
||||
m_dynamicParameters.MaxClipDistances =
|
||||
m_vulkanCaps.SupportsShaderClipDistance ? std::max(m_vulkanCaps.MaxClipDistances, 0) : 0;
|
||||
m_dynamicParameters.MaxViewports = m_vulkanCaps.MaxViewports;
|
||||
// Assigned explicitly rather than left to the struct's defaults, like every other
|
||||
// parameter here, so a second fill cannot inherit a stale value. GL_UNDEFINED_VERTEX is
|
||||
// the truthful answer for DirectVulkan and a legal one (GL 4.6 table 23.65): which vertex
|
||||
// provokes is chosen per pipeline by VulkanRenderer::SelectProvokingVertexMode out of
|
||||
// VK_EXT_provoking_vertex, provokingVertexModePerPipeline and the topology, so there is no
|
||||
// one convention to name. Vulkan's own default is FIRST, which is the opposite of the
|
||||
// GL_LAST_VERTEX_CONVENTION this used to claim unconditionally.
|
||||
m_dynamicParameters.LayerProvokingVertex = GL_UNDEFINED_VERTEX;
|
||||
m_dynamicParameters.ViewportIndexProvokingVertex = GL_UNDEFINED_VERTEX;
|
||||
m_dynamicParameters.MaxViewportWidth = m_vulkanCaps.MaxViewportWidth;
|
||||
m_dynamicParameters.MaxViewportHeight = m_vulkanCaps.MaxViewportHeight;
|
||||
m_dynamicParameters.ViewportBoundsRangeMin = m_vulkanCaps.ViewportBoundsRangeMin;
|
||||
@@ -843,7 +1059,35 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
DynParams::PerLayerFramebufferAttachmentBit(TextureTarget::TextureCubeMapArray);
|
||||
}
|
||||
}
|
||||
m_dynamicParameters.SupportsFloat64VertexAttributes = m_vulkanCaps.SupportsShaderFloat64;
|
||||
// The device feature the whole fp64 story hangs off. With it, a module keeps its
|
||||
// OpCapability Float64 and real doubles reach the driver; without it the transpile
|
||||
// narrows every 64-bit float to 32 (ShaderTranspiler::DemoteFloat64Pass), because
|
||||
// VUID-VkShaderModuleCreateInfo-pCode-08740 forbids the capability outright and no
|
||||
// pipeline could be built from such a module. lavapipe reports it; Adreno and Mali both
|
||||
// report VK_FALSE, so on every real mobile device this is false and the demotion runs
|
||||
// exactly as it always has.
|
||||
m_dynamicParameters.SupportsShaderFloat64 = m_vulkanCaps.SupportsShaderFloat64;
|
||||
// Never, on any device, and DELIBERATELY NOT COUPLED to the line above even though it
|
||||
// once tracked the same feature. It used to, because a `dvec` input needed Float64 to
|
||||
// exist in the module at all; a 64-bit vertex FETCH was already impossible
|
||||
// (VK_FORMAT_R64*_SFLOAT is optional and lavapipe reports zero bufferFeatures for all
|
||||
// four), so the attribute arrived as its 32-bit word pair and PackDoubleVertexInputsPass
|
||||
// bitcast it back.
|
||||
//
|
||||
// Re-coupling it does not work, and the reason is worth recording because it is not
|
||||
// obvious: this flag decides the VkFormat from the VAO ATTRIBUTE alone, and the attribute
|
||||
// does not know what the shader declared. glVertexAttribFormat(GL_DOUBLE) against a plain
|
||||
// `in vec4` is not only legal but the common case
|
||||
// (KHR-GL43.vertex_attrib_binding.basic-input-case4 does exactly that, and case5 adds
|
||||
// normalized=GL_TRUE), and advanced-bindingUpdate feeds a dvec3 the same way - GL defines
|
||||
// all of them as "doubles in memory, converted to float". Turning the flag on turns the
|
||||
// narrowing OFF for every one of them and the attributes come back unfetched.
|
||||
//
|
||||
// What keeps the two halves honest instead is a per-MODULE decision: a vertex module that
|
||||
// declares a 64-bit float INPUT is demoted whole, even where the backend has native fp64,
|
||||
// so `dvec` inputs are `vec` inputs on this backend exactly as they always were. See
|
||||
// ShaderCompiler::SanitizeAndOptimizeBinary.
|
||||
m_dynamicParameters.SupportsFloat64VertexAttributes = false;
|
||||
m_dynamicParameters.MaxShaderStorageBlockSize =
|
||||
std::min(m_vulkanCaps.MaxShaderStorageBlockSize, kMaxAdvertisedShaderStorageBlockSize);
|
||||
if (m_vulkanCaps.SupportsShaderSubgroup) {
|
||||
@@ -852,6 +1096,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_dynamicParameters.SubgroupSupportedFeatures =
|
||||
mapSubgroupFeatures(m_vulkanCaps.SubgroupSupportedOperations);
|
||||
m_dynamicParameters.SubgroupQuadOperationsInAllStages = m_vulkanCaps.SubgroupQuadOperationsInAllStages;
|
||||
} else if (ShouldEmulateSubgroups(m_vulkanCaps.SupportsShaderSubgroup)) {
|
||||
// MOBILEGL_MAGMA_EMULATE_SUBGROUP on a device with no native subgroups: the
|
||||
// advertised values describe the 32-lane virtual subgroup the compute
|
||||
// lowering implements (SubgroupSupportPolicy.h / EmulateSubgroupsPass).
|
||||
// GL requires the advertisement and the execution to agree, and on this
|
||||
// path the emulation is what executes; only the compute stage is offered.
|
||||
m_dynamicParameters.SubgroupSize = kEmulatedSubgroupSize;
|
||||
m_dynamicParameters.SubgroupSupportedStages = kEmulatedSubgroupStages;
|
||||
m_dynamicParameters.SubgroupSupportedFeatures = kEmulatedSubgroupFeatures;
|
||||
m_dynamicParameters.SubgroupQuadOperationsInAllStages = false;
|
||||
MGLOG_I("DirectVulkan: emulating 32-lane compute subgroups "
|
||||
"(MOBILEGL_MAGMA_EMULATE_SUBGROUP, no native subgroup support)");
|
||||
} else {
|
||||
m_dynamicParameters.SubgroupSize = 0;
|
||||
m_dynamicParameters.SubgroupSupportedStages = 0;
|
||||
|
||||
@@ -62,8 +62,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// POST screen shows.
|
||||
|
||||
// Static identity of the Magma renderer (renderer/backend names, target GL/GLSL
|
||||
// versions, ExtraVendor) with the baseline extension advertisement (no shader
|
||||
// subgroup, no timer queries). A live backend copies this in its constructor and
|
||||
// versions, ExtraVendor) with the baseline extension advertisement (no runtime-gated
|
||||
// capabilities). A live backend copies this in its constructor and
|
||||
// reconciles the Extensions in UpdateAdvertisedExtensions once real capabilities
|
||||
// exist; callers that need the advertised list for a known capability set must
|
||||
// use BuildAdvertisedExtensions instead.
|
||||
@@ -74,7 +74,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// MOBILEGL_DISABLE_TIMERQUERY escape hatches are applied inside, so callers pass
|
||||
// the detected device support (passing an already-gated value is harmless).
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
|
||||
Bool anisotropicFilteringSupported);
|
||||
Bool anisotropicFilteringSupported,
|
||||
Bool nonZeroIndirectBaseInstanceSupported,
|
||||
Bool cubeMapArraySupported);
|
||||
|
||||
// Format: <GPU Name>, Vulkan <Vulkan Version>, Driver <Driver Version> — the exact
|
||||
// string an initialized backend returns from GetBackendAPIVersionString (and that
|
||||
|
||||
@@ -69,6 +69,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// slot's ownership unambiguous.
|
||||
Uint64 programLifetimeId = 0;
|
||||
Uint32 backendStateVersion = 0;
|
||||
// glShaderStorageBlockBinding deliberately does NOT bump the backend state
|
||||
// version, and the pipeline composite is unnamed so the in-place patch in
|
||||
// DirectVulkan::ShaderStorageBlockBinding can never reach its slot - the
|
||||
// mirror replay bumps only the program's block-binding version. Without this
|
||||
// key the composite's slot kept serving the pre-rebind block.binding.
|
||||
Uint32 blockBindingVersion = 0;
|
||||
Vector<StorageBlockResource> storageBlocks;
|
||||
Vector<BufferVariableResource> bufferVariables;
|
||||
GLint computeWorkGroupSize[3] = {1, 1, 1};
|
||||
@@ -156,18 +162,33 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto& cache = g_programResourceCaches[program.GetExternalIndex()];
|
||||
const Uint64 programLifetimeId = program.GetLifetimeId();
|
||||
const Uint32 backendStateVersion = program.GetBackendStateVersion();
|
||||
const Uint32 blockBindingVersion = program.GetBlockBindingVersion();
|
||||
// The lifetime id must match too: a new program that reuses a deleted
|
||||
// program's name and happens to land on the same backendStateVersion (both
|
||||
// count from zero) would otherwise be served the dead program's reflection.
|
||||
if (cache.programLifetimeId == programLifetimeId &&
|
||||
cache.backendStateVersion == backendStateVersion &&
|
||||
(!cache.storageBlocks.empty() || !cache.bufferVariables.empty())) {
|
||||
if (cache.blockBindingVersion != blockBindingVersion) {
|
||||
// Only the block bindings moved (glShaderStorageBlockBinding, or the
|
||||
// pipeline composite's mirror replay - neither touches the backend
|
||||
// state version): the reflection itself is unchanged, so re-apply the
|
||||
// overrides by name instead of re-running spirv-reflect. Overrides
|
||||
// only ever accumulate, so a block without one still holds its
|
||||
// declared binding.
|
||||
for (auto& block : cache.storageBlocks) {
|
||||
const Int rebound = program.GetShaderStorageBlockBindingOverride(block.name);
|
||||
if (rebound >= 0) block.binding = static_cast<Uint32>(rebound);
|
||||
}
|
||||
cache.blockBindingVersion = blockBindingVersion;
|
||||
}
|
||||
return cache;
|
||||
}
|
||||
|
||||
cache = {};
|
||||
cache.programLifetimeId = programLifetimeId;
|
||||
cache.backendStateVersion = backendStateVersion;
|
||||
cache.blockBindingVersion = blockBindingVersion;
|
||||
|
||||
Vector<SpvReflectShaderModule> modules;
|
||||
Vector<Bool> validModules;
|
||||
@@ -269,14 +290,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
drawBuffer->SyncPersistentMappedRange();
|
||||
const SizeT commandOffset = reinterpret_cast<SizeT>(indirect);
|
||||
if (drawBuffer->MappedData() == nullptr || commandOffset + requiredBytes > drawBuffer->GetSize()) {
|
||||
MGLOG_E("%s skipped: invalid GL_DRAW_INDIRECT_BUFFER binding or range", label);
|
||||
MGLOG_E_ONCE("%s skipped: invalid GL_DRAW_INDIRECT_BUFFER binding or range", label);
|
||||
return nullptr;
|
||||
}
|
||||
return drawBuffer->MappedData() + commandOffset;
|
||||
}
|
||||
|
||||
if (!indirect) {
|
||||
MGLOG_E("%s skipped: indirect pointer is null", label);
|
||||
MGLOG_E_ONCE("%s skipped: indirect pointer is null", label);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -398,7 +419,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
stride = sizeof(DrawArraysIndirectCommand);
|
||||
}
|
||||
if (stride < static_cast<GLsizei>(sizeof(DrawArraysIndirectCommand))) {
|
||||
MGLOG_E("MultiDrawArraysIndirect skipped: stride %d is smaller than command size %zu",
|
||||
MGLOG_E_ONCE("MultiDrawArraysIndirect skipped: stride %d is smaller than command size %zu",
|
||||
stride, sizeof(DrawArraysIndirectCommand));
|
||||
return;
|
||||
}
|
||||
@@ -446,20 +467,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
stride = sizeof(DrawArraysIndirectCommand);
|
||||
}
|
||||
if (stride < static_cast<GLsizei>(sizeof(DrawArraysIndirectCommand))) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: stride %d is smaller than command size %zu",
|
||||
MGLOG_E_ONCE("MultiDrawArraysIndirectCount skipped: stride %d is smaller than command size %zu",
|
||||
stride, sizeof(DrawArraysIndirectCommand));
|
||||
return;
|
||||
}
|
||||
|
||||
auto parameterBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
if (!parameterBuffer || drawcount < 0 || static_cast<SizeT>(drawcount) + sizeof(Uint32) > parameterBuffer->GetSize()) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: invalid GL_PARAMETER_BUFFER binding or range");
|
||||
MGLOG_E_ONCE("MultiDrawArraysIndirectCount skipped: invalid GL_PARAMETER_BUFFER binding or range");
|
||||
return;
|
||||
}
|
||||
|
||||
parameterBuffer->SyncPersistentMappedRange();
|
||||
if (parameterBuffer->MappedData() == nullptr) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: CPU fallback cannot read parameter buffer");
|
||||
MGLOG_E_ONCE("MultiDrawArraysIndirectCount skipped: CPU fallback cannot read parameter buffer");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -513,7 +534,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const SizeT indexSize = MG_Util::GetGLTypeSize(type);
|
||||
if (indexSize == 0) {
|
||||
MGLOG_E("DrawElementsIndirect skipped: unsupported index type 0x%x", type);
|
||||
MGLOG_E_ONCE("DrawElementsIndirect skipped: unsupported index type 0x%x", type);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -611,15 +632,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::CopyTexSubImage2D called with null GL context");
|
||||
pVulkanRenderer->CopyTexSubImage2D(target, level, xoffset, yoffset, x, y, width, height);
|
||||
}
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
void CopyImageSubData(const CopyImageEndpoint& src,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
const CopyImageEndpoint& dst,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth) {
|
||||
MOBILEGL_ASSERT(pVulkanRenderer, "DirectVulkan::CopyImageSubData called with null VulkanRenderer");
|
||||
MOBILEGL_ASSERT(MG_State::pGLContext, "DirectVulkan::CopyImageSubData called with null GL context");
|
||||
pVulkanRenderer->CopyImageSubData(srcTexture, srcTarget, srcLevel, srcX, srcY, srcZ,
|
||||
dstTexture, dstTarget, dstLevel, dstX, dstY, dstZ,
|
||||
pVulkanRenderer->CopyImageSubData(src, srcTarget, srcLevel, srcX, srcY, srcZ,
|
||||
dst, dstTarget, dstLevel, dstX, dstY, dstZ,
|
||||
srcWidth, srcHeight, srcDepth);
|
||||
}
|
||||
void GenerateMipmap(GLenum target) {
|
||||
@@ -801,9 +822,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// already recorded the new binding on the program - which is what reseeds this cache
|
||||
// whenever it is rebuilt. Writing the entry here as well keeps an ALREADY-BUILT cache
|
||||
// (the common case: the very next draw reads it) from having to be thrown away.
|
||||
auto& cache = GetProgramResourceCache(*programObject);
|
||||
//
|
||||
// Resolve the index BEFORE taking the reference, and bounds-check the way the
|
||||
// sibling getter does. GetShaderStorageBlockIndex re-enters GetProgramResourceCache,
|
||||
// which indexes g_programResourceCaches and can therefore insert - and that map is
|
||||
// open-addressed, so a rehash MOVES its entries and a reference taken before the
|
||||
// call is left dangling. Binding a program's storage block
|
||||
// while another program's entry was still absent from the cache was a reproducible
|
||||
// segfault (ProgramPipelineScenario's two storage-block cases, in one process).
|
||||
const GLuint blockIndex = GetShaderStorageBlockIndex(*programObject, storageBlockName);
|
||||
if (blockIndex == GL_INVALID_INDEX) return;
|
||||
auto& cache = GetProgramResourceCache(*programObject);
|
||||
if (blockIndex >= cache.storageBlocks.size()) return;
|
||||
cache.storageBlocks[blockIndex].binding = storageBlockBinding;
|
||||
}
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels) {
|
||||
@@ -966,6 +996,29 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (drawcount <= 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// With no element-array buffer bound, every indices[i] is a client pointer into a
|
||||
// separate CPU allocation, not an offset into one shared buffer. The batched payload
|
||||
// below cannot express that: it carries ONE index-buffer view for the whole batch and
|
||||
// turns each pointer into a firstIndex relative to it. Replay the sub-draws through
|
||||
// the single-draw entry point instead - it snapshots each client range into its own
|
||||
// transient slice, which is exactly what the unrolled draws this must match do.
|
||||
// (The batch used to be built this way; the shared-view rewrite that added
|
||||
// MultiDrawIndexedCmd left the client-memory shape addressing a view whose byte
|
||||
// offset is a hardcoded 0, so UploadAndBindIndexBuffer saw a null client pointer,
|
||||
// declined the whole batch and painted nothing.)
|
||||
const auto& vao = *MG_State::pGLContext->GetBoundVertexArray();
|
||||
if (vao.GetIndexBufferBindingSlot().GetBoundObject() == nullptr) {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) {
|
||||
continue;
|
||||
}
|
||||
DrawElementsBaseVertex(mode, count[i], type, indices[i],
|
||||
basevertex != nullptr ? basevertex[i] : 0);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
MultiDrawIndexedCmd payload{};
|
||||
payload.mode = mode;
|
||||
payload.indexBufferView.indexType = type;
|
||||
@@ -977,7 +1030,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// shift - the hardware divide was the hottest instruction of this loop.
|
||||
const SizeT indexSize = MG_Util::GetGLTypeSize(type);
|
||||
if (indexSize == 0) {
|
||||
MGLOG_E("MultiDrawElements skipped: unsupported index type 0x%x", type);
|
||||
MGLOG_E_ONCE("MultiDrawElements skipped: unsupported index type 0x%x", type);
|
||||
return;
|
||||
}
|
||||
const Uint32 indexSizeShift = static_cast<Uint32>(std::countr_zero(indexSize));
|
||||
|
||||
@@ -82,9 +82,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLsizei height, GLint border);
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height);
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
void CopyImageSubData(const CopyImageEndpoint& src,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
const CopyImageEndpoint& dst,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void GenerateMipmap(GLenum target);
|
||||
|
||||
@@ -111,6 +111,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
if (buffer.IsValid()) {
|
||||
// Outgrown, not dead: every BufferSlice handed out from this frame's arena so far
|
||||
// still names it, and those slices stay in service until the frame slot is rewound
|
||||
// (VkBufferResource::transientSlice, the converted-vertex-stream cache, the draw
|
||||
// memos). The release therefore has to survive every mid-frame reclaim and land on
|
||||
// the next ResetFrame of this slot - see VkBufferManager::CollectAllDeferredReleases.
|
||||
m_deferredReleases[frameIndex].push_back(std::move(buffer));
|
||||
}
|
||||
|
||||
|
||||
@@ -205,7 +205,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// commands away. The device is gone on that path anyway - stay silent-safe
|
||||
// rather than trade a lost device for a barrier into a closed buffer.
|
||||
if (frame.hasCommandBufferRecorded) {
|
||||
MGLOG_E("TransitionToPresent: command buffer already closed; skipping the present barrier");
|
||||
MGLOG_E_ONCE("TransitionToPresent: command buffer already closed; skipping the present barrier");
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -206,6 +206,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.primitiveRestartEnable, sizeof(payload.primitiveRestartEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.patchControlPoints, sizeof(payload.patchControlPoints)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.viewportCount, sizeof(payload.viewportCount)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.polygonMode, sizeof(payload.polygonMode)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.cullMode, sizeof(payload.cullMode)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.frontFace, sizeof(payload.frontFace)));
|
||||
@@ -252,6 +253,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
VkPipeline pipeline = CreatePipeline(payload);
|
||||
// A failed creation must never be memoized. Caching VK_NULL_HANDLE served the null back for
|
||||
// the rest of the process, so one transient driver rejection turned every later draw with
|
||||
// the same state into a vkCmdBindPipeline(VK_NULL_HANDLE) - the SIGSEGV behind 9 of the 15
|
||||
// CTS process deaths. Retrying costs one failed vkCreateGraphicsPipelines per draw, which
|
||||
// is the correct price for a broken pipeline and is bounded by the draw itself being
|
||||
// skipped.
|
||||
if (pipeline == VK_NULL_HANDLE) {
|
||||
// Unlatched, like the CreatePipeline report it accompanies: a pipeline MobileGL
|
||||
// assembled and the driver refused is a broken invariant, not an expected failure,
|
||||
// so it stays loud for as long as it is reachable. Raised from MGLOG_I once the
|
||||
// Log.h ordering fix made MGLOG_E live in INFO builds.
|
||||
MGLOG_E("PipelineFactory::GetOrCreatePipeline: creation failed for hash=0x%llx "
|
||||
"programHash=0x%llx; not caching the failure",
|
||||
static_cast<unsigned long long>(hash),
|
||||
static_cast<unsigned long long>(payload.programHash));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
m_cache.emplace(hash, PipelineCacheEntry{pipeline, payload.programHash, payload.renderPass,
|
||||
m_frameCounter});
|
||||
return pipeline;
|
||||
@@ -389,8 +407,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
tessellation.patchControlPoints = payload.patchControlPoints;
|
||||
|
||||
VkPipelineViewportStateCreateInfo vpci{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO};
|
||||
vpci.viewportCount = 1;
|
||||
vpci.scissorCount = 1;
|
||||
// Both counts move together: GL has one scissor rectangle per viewport, and Vulkan
|
||||
// requires viewportCount == scissorCount whenever both are dynamic
|
||||
// (VUID-VkPipelineViewportStateCreateInfo-scissorCount-04136). The caller has already
|
||||
// clamped this to the device's multiViewport capability.
|
||||
vpci.viewportCount = std::max<Uint32>(payload.viewportCount, 1u);
|
||||
vpci.scissorCount = vpci.viewportCount;
|
||||
|
||||
VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO};
|
||||
raster.polygonMode = payload.polygonMode;
|
||||
@@ -458,9 +480,56 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
blend.attachmentCount = payload.colorAttachmentCount;
|
||||
blend.pAttachments = colorAttachments.empty() ? nullptr : colorAttachments.data();
|
||||
|
||||
// A GL program may have a tessellation EVALUATION stage and no CONTROL stage: GL 4.6 core
|
||||
// 11.2.2 gives it a fixed-function pass-through instead. Vulkan has no such stage, and
|
||||
// VUID-VkGraphicsPipelineCreateInfo-pStages-00730 requires both tessellation stages or
|
||||
// neither - so the renderer synthesizes the pass-through GL describes and hands it in
|
||||
// here (see ProgramFactory::GetOrCreatePassthroughTessControlStage).
|
||||
//
|
||||
// The refusal below is what keeps the half-tessellated shape away from the driver when
|
||||
// there is no synthesized stage to add - because Mali does not reject it, it dereferences
|
||||
// null INSIDE vkCreateGraphicsPipelines and takes the process down (SIGSEGV, fault addr
|
||||
// 0x34, on Mali-G715/r54p2 and Mali-G925/r49p1 alike; Adreno and lavapipe merely render
|
||||
// wrong). Returning VK_NULL_HANDLE routes this through the same path a driver rejection
|
||||
// takes: the draw is skipped, nothing is memoised, and the process survives.
|
||||
const Vector<VkPipelineShaderStageCreateInfo>* effectiveStages = payload.stages;
|
||||
Vector<VkPipelineShaderStageCreateInfo> stagesWithPassthrough;
|
||||
if (payload.passthroughTessControlStage.module != VK_NULL_HANDLE) {
|
||||
stagesWithPassthrough = *payload.stages;
|
||||
stagesWithPassthrough.push_back(payload.passthroughTessControlStage);
|
||||
effectiveStages = &stagesWithPassthrough;
|
||||
}
|
||||
{
|
||||
VkShaderStageFlags stagesPresent = 0;
|
||||
for (const auto& stageInfo : *effectiveStages) {
|
||||
stagesPresent |= stageInfo.stage;
|
||||
}
|
||||
const Bool hasTessControl = (stagesPresent & VK_SHADER_STAGE_TESSELLATION_CONTROL_BIT) != 0;
|
||||
const Bool hasTessEval = (stagesPresent & VK_SHADER_STAGE_TESSELLATION_EVALUATION_BIT) != 0;
|
||||
if (hasTessControl != hasTessEval) {
|
||||
// Latched, and the latch is the point: a failed creation is deliberately never
|
||||
// memoised (see GetOrCreatePipeline), so a program in this state re-enters here
|
||||
// once per draw, every frame - and a refusal diagnostic that repeats per draw is
|
||||
// noise, not a diagnostic. One line names the program; the draws it explains are
|
||||
// all the same draw.
|
||||
static Bool s_warnedHalfTessellatedPipeline = false;
|
||||
if (!s_warnedHalfTessellatedPipeline) {
|
||||
s_warnedHalfTessellatedPipeline = true;
|
||||
MGLOG_E_ONCE("PipelineFactory::CreatePipeline: refusing a pipeline with %s tessellation stage and "
|
||||
"no %s stage (VUID-VkGraphicsPipelineCreateInfo-pStages-00730). programHash=0x%llx "
|
||||
"patchControlPoints=%u. Its draws are skipped; logged once.",
|
||||
hasTessEval ? "an evaluation" : "a control",
|
||||
hasTessEval ? "control" : "evaluation",
|
||||
static_cast<unsigned long long>(payload.programHash),
|
||||
payload.patchControlPoints);
|
||||
}
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
|
||||
VkGraphicsPipelineCreateInfo gpi{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO};
|
||||
gpi.stageCount = static_cast<Uint32>(payload.stages->size());
|
||||
gpi.pStages = payload.stages->data();
|
||||
gpi.stageCount = static_cast<Uint32>(effectiveStages->size());
|
||||
gpi.pStages = effectiveStages->data();
|
||||
gpi.pVertexInputState = payload.vertexInputState;
|
||||
gpi.pInputAssemblyState = &ia;
|
||||
gpi.pTessellationState =
|
||||
@@ -477,6 +546,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
const VkResult result = vkCreateGraphicsPipelines(m_device, m_pipelineCache, 1, &gpi, nullptr, &pipeline);
|
||||
// Loud, at MGLOG_F, and deliberately NOT latched. vkCreateGraphicsPipelines refusing a
|
||||
// pipeline MobileGL assembled is a should-never-happen state, and the driver's own
|
||||
// answer is VK_ERROR_UNKNOWN - no information at all - so this dump is the entire
|
||||
// diagnosis. It is not an expected failure mode, so the one-shot rule that quiets W/E
|
||||
// does not apply: while this is reachable it should keep saying so on every draw.
|
||||
// GetOrCreatePipeline deliberately does not cache the failure, which is what makes that
|
||||
// repetition happen; if the repetition ever needs to stop, fix the pipeline, not the log.
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_F("PipelineFactory::CreatePipeline failed: result=%s (%d) programHash=0x%llx vertexInputHash=0x%llx stageCount=%u topology=%s(%d) colorAttachmentCount=%u samples=%s(%d) subpass=%u",
|
||||
VkResultToString(result),
|
||||
@@ -507,6 +583,36 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MGLOG_F("PipelineFactory::CreatePipeline vertex input: bindingCount=%u attributeCount=%u",
|
||||
payload.vertexInputState->vertexBindingDescriptionCount,
|
||||
payload.vertexInputState->vertexAttributeDescriptionCount);
|
||||
// The driver's own answer is VK_ERROR_UNKNOWN, i.e. no information at all, so the only
|
||||
// way to work out WHICH shader it choked on (the open sampler-array-in-struct
|
||||
// investigation) is to name the modules. MGLOG_I, not _D: this is part of a
|
||||
// should-never-happen report and must survive in the INFO-level builds that CTS
|
||||
// actually runs against, alongside the MGLOG_F lines above.
|
||||
if (payload.stageSpirvDigests) {
|
||||
for (SizeT i = 0; i < payload.stageSpirvDigests->size(); ++i) {
|
||||
const auto& digest = (*payload.stageSpirvDigests)[i];
|
||||
MGLOG_I("PipelineFactory::CreatePipeline spirv[%zu]: stage=0x%x words=%u bytes=%zu "
|
||||
"hash=0x%llx",
|
||||
i, digest.stage, digest.wordCount,
|
||||
static_cast<SizeT>(digest.wordCount) * sizeof(Uint32),
|
||||
static_cast<unsigned long long>(digest.hash));
|
||||
}
|
||||
} else {
|
||||
MGLOG_I("PipelineFactory::CreatePipeline: no SPIR-V digests attached to the payload");
|
||||
}
|
||||
if (payload.stages) {
|
||||
for (SizeT i = 0; i < payload.stages->size(); ++i) {
|
||||
const auto& stage = (*payload.stages)[i];
|
||||
// VkShaderModule is a non-dispatchable handle: a pointer on 64-bit but a
|
||||
// plain uint64_t on 32-bit ABIs, where a cast to const void* is ill-formed
|
||||
// (broke the armeabi-v7a build). Print it as the 64-bit value it is.
|
||||
MGLOG_I("PipelineFactory::CreatePipeline stage[%zu]: stage=0x%x module=0x%llx entry=%s "
|
||||
"specialization=%d",
|
||||
i, static_cast<Uint32>(stage.stage),
|
||||
static_cast<unsigned long long>(reinterpret_cast<Uint64>(stage.module)),
|
||||
stage.pName ? stage.pName : "(null)", stage.pSpecializationInfo ? 1 : 0);
|
||||
}
|
||||
}
|
||||
for (Uint32 i = 0; i < payload.colorAttachmentCount; ++i) {
|
||||
const auto& attachment = payload.colorBlendAttachments[i];
|
||||
MGLOG_F("PipelineFactory::CreatePipeline colorAttachment[%u]: blend=%d colorWriteMask=0x%x srcColor=%d dstColor=%d colorOp=%d srcAlpha=%d dstAlpha=%d alphaOp=%d",
|
||||
|
||||
@@ -14,6 +14,16 @@
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Enough of a fingerprint to identify the exact module the driver rejected without keeping the
|
||||
// SPIR-V alive for every program in the cache: a driver that answers VK_ERROR_UNKNOWN tells us
|
||||
// nothing, so the log has to carry the shader's identity itself. Diagnostic only - never part
|
||||
// of any pipeline or program hash.
|
||||
struct ShaderStageSpirvDigest {
|
||||
Uint32 stage = 0; // VkShaderStageFlagBits
|
||||
Uint32 wordCount = 0;
|
||||
Uint64 hash = 0;
|
||||
};
|
||||
|
||||
class PipelineFactory {
|
||||
public:
|
||||
using HashType = Uint64;
|
||||
@@ -32,6 +42,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool primitiveRestartEnable = false;
|
||||
// GL_PATCH_VERTICES; only read for a PATCH_LIST topology.
|
||||
Uint32 patchControlPoints = 3;
|
||||
// How many of ARB_viewport_array's viewports this pipeline rasterizes into. 1 for
|
||||
// every program that never assigns gl_ViewportIndex, which is all of them outside the
|
||||
// conformance suite - the wide shape costs a longer vkCmdSetViewport/Scissor per state
|
||||
// change and can cost hardware fast paths, so it is opt-in per program. Baked into the
|
||||
// pipeline (viewportCount is not dynamic without VK_EXT_extended_dynamic_state) and
|
||||
// therefore hashed; the DYNAMIC viewport/scissor arrays the draw pushes must have
|
||||
// exactly this many elements (VUID-vkCmdDraw-viewportCount-03417/-03418).
|
||||
Uint32 viewportCount = 1;
|
||||
VkPolygonMode polygonMode = VK_POLYGON_MODE_FILL;
|
||||
VkCullModeFlags cullMode = VK_CULL_MODE_BACK_BIT;
|
||||
VkFrontFace frontFace = VK_FRONT_FACE_CLOCKWISE;
|
||||
@@ -61,7 +79,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool fragmentReplacesDepth = false;
|
||||
Array<VkPipelineColorBlendAttachmentState, kMaxColorAttachments> colorBlendAttachments{};
|
||||
const Vector<VkPipelineShaderStageCreateInfo>* stages = nullptr;
|
||||
// The tessellation control stage this renderer synthesized for a program that has
|
||||
// an evaluation stage and none of its own (GL 4.6 core 11.2.2 gives such a program a
|
||||
// fixed-function pass-through; Vulkan has no such thing and
|
||||
// VUID-VkGraphicsPipelineCreateInfo-pStages-00730 forbids the half-tessellated
|
||||
// pipeline outright). Appended to `stages` at creation. A null module means the
|
||||
// renderer could not build one, and CreatePipeline refuses the pipeline - the same
|
||||
// refusal it applies when `stages` itself is half-tessellated.
|
||||
//
|
||||
// NOT hashed: it is a pure function of the program and of patchControlPoints, both
|
||||
// of which ComputeHash already mixes in.
|
||||
VkPipelineShaderStageCreateInfo passthroughTessControlStage{};
|
||||
const VkPipelineVertexInputStateCreateInfo* vertexInputState = nullptr;
|
||||
// Diagnostic only; may be null. Read solely from the pipeline-creation failure path.
|
||||
const Vector<ShaderStageSpirvDigest>* stageSpirvDigests = nullptr;
|
||||
};
|
||||
|
||||
explicit PipelineFactory(VkDevice device, const VulkanRendererConfig& config);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -9,6 +9,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include "PipelineFactory.h"
|
||||
#include "MG_State/GLState/ProgramState/ProgramObject.h"
|
||||
#include "MG_State/GLState/ProgramState/ShaderObject.h"
|
||||
#include "MG_State/GLState/TextureState/TextureEnum.h"
|
||||
@@ -32,7 +33,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
CombinedImageSampler,
|
||||
UniformTexelBuffer,
|
||||
StorageBuffer,
|
||||
StorageImage
|
||||
StorageImage,
|
||||
// GLSL `imageBuffer` - a buffer texture reached through an IMAGE unit rather than a
|
||||
// texture unit. Vulkan spells it VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER, which is a
|
||||
// VkBufferView like UniformTexelBuffer and not a VkImageView like StorageImage: it is
|
||||
// the one image uniform whose descriptor is a buffer. Appended, never inserted -
|
||||
// DescriptorKeyHash mixes the enumerator's value.
|
||||
StorageTexelBuffer
|
||||
};
|
||||
|
||||
enum class CompileOptionBit : Uint {
|
||||
@@ -52,19 +59,56 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// recorded while GL transform feedback is active, so plain draws keep the
|
||||
// undecorated variant.
|
||||
XfbCapture = 1 << 6,
|
||||
// Rewrites the fragment stage's gl_FragCoord reads to GL's bottom-left window
|
||||
// origin. Vulkan's gl_FragCoord.y IS the framebuffer row being written, and the
|
||||
// default framebuffer's image is stored in display (top-left) order, so a shader
|
||||
// that reads gl_FragCoord there sees `height - y_GL`. Set together with
|
||||
// PositionYFlip (the two are the same fact about the same draws) except under a
|
||||
// quarter turn, which this renderer does not convert rectangles for either.
|
||||
FragCoordYFlip = 1 << 7,
|
||||
// Replaces the vertex stage's gl_BaseVertex reads with zero. GL defines the builtin
|
||||
// as zero for every drawing command that has no baseVertex parameter - all the
|
||||
// DrawArrays forms - while Vulkan's BaseVertex reports firstVertex there. Set only
|
||||
// for a non-indexed draw whose program actually reads the builtin, so nothing else
|
||||
// acquires a second program/pipeline variant. See ZeroBaseVertexPass.
|
||||
ZeroBaseVertex = 1 << 8,
|
||||
};
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
|
||||
struct UpdateAfterBindLimits {
|
||||
Bool enabled = false;
|
||||
Uint32 maxPerStageSamplers = 0;
|
||||
Uint32 maxPerStageUniformBuffers = 0;
|
||||
Uint32 maxPerStageStorageBuffers = 0;
|
||||
Uint32 maxPerStageSampledImages = 0;
|
||||
Uint32 maxPerStageStorageImages = 0;
|
||||
Uint32 maxPerStageResources = 0;
|
||||
Uint32 maxSetSamplers = 0;
|
||||
Uint32 maxSetUniformBuffers = 0;
|
||||
Uint32 maxSetUniformBuffersDynamic = 0;
|
||||
Uint32 maxSetStorageBuffers = 0;
|
||||
Uint32 maxSetStorageBuffersDynamic = 0;
|
||||
Uint32 maxSetSampledImages = 0;
|
||||
Uint32 maxSetStorageImages = 0;
|
||||
};
|
||||
|
||||
struct VkProgramObject {
|
||||
static constexpr Uint32 kMaxVertexInputLocations = 32;
|
||||
|
||||
HashType hash = 0;
|
||||
Vector<VkPipelineShaderStageCreateInfo> stages;
|
||||
Vector<VkShaderModule> modules;
|
||||
// Parallel to stages; identifies the exact module bytes handed to the driver when a
|
||||
// pipeline creation fails. Sixteen bytes per stage instead of keeping the SPIR-V.
|
||||
Vector<ShaderStageSpirvDigest> stageSpirvDigests;
|
||||
|
||||
// Layout data (previously in separate VkProgramLayout)
|
||||
VkDescriptorSetLayout descriptorSetLayout = VK_NULL_HANDLE;
|
||||
// True only when this layout passed every descriptor-indexing feature and
|
||||
// update-after-bind limit gate at reflection time. It controls both the
|
||||
// layout/binding flags and the pool class used by UniformManager.
|
||||
Bool usesUpdateAfterBind = false;
|
||||
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
|
||||
Vector<DescriptorBindingKind> bindingKinds;
|
||||
// The bindings this program actually declares, ascending. bindingKinds is sized to the
|
||||
@@ -75,8 +119,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<Uint32> activeBindings;
|
||||
Vector<Uint32> dynamicBindings;
|
||||
Vector<Int> uniformBlockIndexByBinding;
|
||||
// Descriptor count per binding (1 except for UBO instance arrays, which occupy one
|
||||
// binding with descriptorCount = N).
|
||||
// Descriptor count per binding (1 except for a descriptor ARRAY - a UBO or storage
|
||||
// block instance array, an image uniform array or a sampler uniform array - each of
|
||||
// which occupies one binding with descriptorCount = N).
|
||||
Vector<Uint16> bindingDescriptorCounts;
|
||||
// Per-element GL uniform block indices for arrayed UBO bindings (count > 1);
|
||||
// element 0 of a non-arrayed binding stays in uniformBlockIndexByBinding.
|
||||
@@ -85,6 +130,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<Int> samplerUniformLocationByBinding;
|
||||
Vector<TextureTarget> samplerTextureTargetByBinding;
|
||||
Vector<SamplerNumericDomain> samplerNumericDomainByBinding;
|
||||
// Shared by StorageImage and StorageTexelBuffer bindings: a binding is one kind or
|
||||
// the other, never both, and both need exactly the same thing - the format the
|
||||
// shader declared, so the per-draw resolve can tell a typed declaration from a
|
||||
// formatless one. Kept as one pair rather than two so the move operations below
|
||||
// cannot drift out of sync with a field that only one kind populates.
|
||||
Vector<VkFormat> storageImageFormatByBinding;
|
||||
Vector<Bool> storageImageUsesBindingFormatByBinding;
|
||||
Vector<String> storageBlockNameByBinding;
|
||||
@@ -92,6 +142,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Set once during ReflectLayout so the per-draw path can skip the whole
|
||||
// storage-image preparation for the overwhelming majority of programs.
|
||||
Bool hasStorageImages = false;
|
||||
// Something about this program's descriptors could not be resolved - an opaque
|
||||
// uniform array whose elements have no addressable uniform locations (the
|
||||
// multi-dimensional case), or a binding remap that failed outright. The binding
|
||||
// STAYS DECLARED in the descriptor set layout; declining is done here, by refusing
|
||||
// every draw, and BindProgramUniformBuffers returns false so the draw setup skips
|
||||
// the draw exactly as it does for any other bind failure.
|
||||
//
|
||||
// Keeping the layout intact is the load-bearing half. Shrinking it instead - which
|
||||
// is what the first cut of this did - leaves the shader reading a descriptor the
|
||||
// layout never declared, and lavapipe segfaults on that inside PIPELINE CREATION,
|
||||
// in a JIT worker thread, before any draw runs where a refusal could help. The
|
||||
// reason was logged once at MGLOG_I when the descriptor was declined.
|
||||
Bool declinedDescriptors = false;
|
||||
Int globalUboBinding = -1;
|
||||
Uint32 activeVertexInputLocationMask = 0;
|
||||
Array<GLenum, kMaxVertexInputLocations> vertexInputTypes{};
|
||||
@@ -104,6 +167,33 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// gl_FragDepth); shader-computed depth is immune to the cross-pipeline
|
||||
// position-invariance quirk (see PipelineFactory::ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
// The vertex module declares the BaseVertex builtin. Selects the ZeroBaseVertex
|
||||
// program variant for non-indexed draws, and is deliberately a property of the
|
||||
// PROGRAM rather than of the variant: the zeroed variant leaves the variable
|
||||
// declared, so both variants answer the same and the draw path can ask either.
|
||||
Bool readsBaseVertexBuiltin = false;
|
||||
// Some pre-rasterization stage assigns gl_ViewportIndex. Its pipeline declares
|
||||
// viewportCount = the renderer's rasterizable viewport count instead of 1, and its
|
||||
// draws push the whole viewport/scissor array; every other program keeps the
|
||||
// single-viewport fast path untouched. Part of the program's identity (folded into
|
||||
// the pipeline hash through programHash), so no memo can serve the wrong shape.
|
||||
Bool writesViewportIndexBuiltin = false;
|
||||
// This program has a tessellation EVALUATION stage and no tessellation CONTROL
|
||||
// stage. GL allows that (4.6 core 11.2.2: with no control shader the input patch
|
||||
// is passed through unmodified, the output patch size is PATCH_VERTICES, and the
|
||||
// levels come from the PATCH_DEFAULT_*_LEVEL state); Vulkan does not - either both
|
||||
// tessellation stages are present or neither
|
||||
// (VUID-VkGraphicsPipelineCreateInfo-pStages-00730). So the draw path has to supply
|
||||
// the pass-through stage GL describes; see GetOrCreatePassthroughTessControlStage.
|
||||
Bool needsPassthroughTessControl = false;
|
||||
// ...and the pass-through this renderer can synthesize carries gl_Position and
|
||||
// nothing else, so it is only correct when the evaluation stage's inputs are
|
||||
// built-ins. A user-defined varying would arrive at the evaluation stage
|
||||
// UNWRITTEN once a control stage sits between it and the vertex stage, which is
|
||||
// silently wrong pixels rather than a crash - so those programs are declined
|
||||
// instead (PipelineFactory::CreatePipeline refuses the pipeline and the draw is
|
||||
// skipped). See ReflectPassthroughTessControlNeed.
|
||||
Bool passthroughTessControlEmulatable = false;
|
||||
// Frame-boundary counter value of the last GetOrCreateProgram hit; drives
|
||||
// cache eviction (see OnFrameBoundary). Mutable: the draw snapshot's memoised
|
||||
// entry pointer re-stamps use through a const reference (StampProgramUse).
|
||||
@@ -118,7 +208,16 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
hash = other.hash;
|
||||
stages = std::move(other.stages);
|
||||
modules = std::move(other.modules);
|
||||
// Must travel with `modules`: these digests name the SPIR-V those exact
|
||||
// shader modules were built from, and the pipeline-failure diagnostics
|
||||
// print the two together. Leaving it behind used to merely lose the
|
||||
// digests on a rehash; now that the cache is a robin-hood table, insertion
|
||||
// SWAPS two entries, and a field that no move touches stays behind in the
|
||||
// slot - pairing one program's modules with another program's digests, so
|
||||
// a pipeline failure would be reported against the wrong SPIR-V.
|
||||
stageSpirvDigests = std::move(other.stageSpirvDigests);
|
||||
descriptorSetLayout = other.descriptorSetLayout;
|
||||
usesUpdateAfterBind = other.usesUpdateAfterBind;
|
||||
pipelineLayout = other.pipelineLayout;
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
activeBindings = std::move(other.activeBindings);
|
||||
@@ -136,6 +235,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
storageBlockNameByBinding = std::move(other.storageBlockNameByBinding);
|
||||
storageBlockIndexByBinding = std::move(other.storageBlockIndexByBinding);
|
||||
hasStorageImages = other.hasStorageImages;
|
||||
declinedDescriptors = other.declinedDescriptors;
|
||||
globalUboBinding = other.globalUboBinding;
|
||||
activeVertexInputLocationMask = other.activeVertexInputLocationMask;
|
||||
vertexInputTypes = other.vertexInputTypes;
|
||||
@@ -145,11 +245,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
|
||||
writesViewportIndexBuiltin = other.writesViewportIndexBuiltin;
|
||||
needsPassthroughTessControl = other.needsPassthroughTessControl;
|
||||
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.usesUpdateAfterBind = false;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.declinedDescriptors = false;
|
||||
other.globalUboBinding = -1;
|
||||
other.activeVertexInputLocationMask = 0;
|
||||
other.activeFragmentOutputLocationMask = 0;
|
||||
@@ -157,6 +263,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.readsBaseVertexBuiltin = false;
|
||||
other.writesViewportIndexBuiltin = false;
|
||||
other.needsPassthroughTessControl = false;
|
||||
other.passthroughTessControlEmulatable = false;
|
||||
other.lastUsedFrame = 0;
|
||||
}
|
||||
VkProgramObject& operator=(VkProgramObject&& other) noexcept {
|
||||
@@ -167,7 +277,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
hash = other.hash;
|
||||
stages = std::move(other.stages);
|
||||
modules = std::move(other.modules);
|
||||
stageSpirvDigests = std::move(other.stageSpirvDigests); // travels with `modules` - see the move ctor
|
||||
descriptorSetLayout = other.descriptorSetLayout;
|
||||
usesUpdateAfterBind = other.usesUpdateAfterBind;
|
||||
pipelineLayout = other.pipelineLayout;
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
activeBindings = std::move(other.activeBindings);
|
||||
@@ -185,6 +297,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
storageBlockNameByBinding = std::move(other.storageBlockNameByBinding);
|
||||
storageBlockIndexByBinding = std::move(other.storageBlockIndexByBinding);
|
||||
hasStorageImages = other.hasStorageImages;
|
||||
declinedDescriptors = other.declinedDescriptors;
|
||||
globalUboBinding = other.globalUboBinding;
|
||||
activeVertexInputLocationMask = other.activeVertexInputLocationMask;
|
||||
vertexInputTypes = other.vertexInputTypes;
|
||||
@@ -194,11 +307,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
|
||||
writesViewportIndexBuiltin = other.writesViewportIndexBuiltin;
|
||||
needsPassthroughTessControl = other.needsPassthroughTessControl;
|
||||
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.usesUpdateAfterBind = false;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.declinedDescriptors = false;
|
||||
other.globalUboBinding = -1;
|
||||
other.activeVertexInputLocationMask = 0;
|
||||
other.activeFragmentOutputLocationMask = 0;
|
||||
@@ -206,6 +325,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.readsBaseVertexBuiltin = false;
|
||||
other.writesViewportIndexBuiltin = false;
|
||||
other.needsPassthroughTessControl = false;
|
||||
other.passthroughTessControlEmulatable = false;
|
||||
other.lastUsedFrame = 0;
|
||||
return *this;
|
||||
}
|
||||
@@ -233,6 +356,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
modules.clear();
|
||||
stages.clear();
|
||||
stageSpirvDigests.clear(); // the modules they describe are gone
|
||||
}
|
||||
};
|
||||
|
||||
@@ -248,21 +372,62 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
virtual void OnProgramEvicted(HashType programHash, VkDescriptorSetLayout descriptorSetLayout) = 0;
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings = 16,
|
||||
Bool shaderDrawParametersEnabled = false,
|
||||
Bool unformattedFloatStorageImagesEnabled = false)
|
||||
// How this factory's compute modules implement GL_KHR_shader_subgroup. Computed
|
||||
// once at renderer initialization (SubgroupSupportPolicy.h + the device's
|
||||
// subgroup properties) so lowering can never disagree with the advertised
|
||||
// capabilities. Native subgroup operations always execute natively; the two
|
||||
// repair passes patch modules AROUND them, and the emulation only replaces them
|
||||
// on opted-in devices with no subgroup support at all.
|
||||
struct SubgroupLoweringPolicy {
|
||||
Bool emulateSubgroups = false; // MOBILEGL_MAGMA_EMULATE_SUBGROUP, no-native-support devices
|
||||
Bool fixIterationRPSubgroupScratch = false; // patch iterationRP's under-declared scratch
|
||||
Bool fixIterationRPBarrier = false; // repair Program 203's shared-scratch race
|
||||
Bool deriveNumSubgroups = false; // repair the NumSubgroups builtin
|
||||
Bool requireFullSubgroups = false; // computeFullSubgroups enabled on the device
|
||||
Uint32 nativeSubgroupSize = 0;
|
||||
// Full-subgroup launches are bounded by this device limit; a dispatch whose
|
||||
// workgroup needs more subgroups than this cannot request the flag.
|
||||
Uint32 maxComputeWorkgroupSubgroups = 0;
|
||||
// VkPhysicalDeviceLimits::maxComputeSharedMemorySize; bounds the scratch the
|
||||
// emulation pass may add (0 falls back to the Vulkan minimum, 16384).
|
||||
Uint32 maxComputeSharedMemoryBytes = 0;
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings,
|
||||
Bool shaderDrawParametersEnabled,
|
||||
Bool unformattedFloatStorageImagesEnabled,
|
||||
Bool enableSpirvValidation,
|
||||
UpdateAfterBindLimits updateAfterBindLimits,
|
||||
SubgroupLoweringPolicy subgroupPolicy)
|
||||
: m_device(device), m_maxBindings(maxBindings), m_config(config),
|
||||
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled),
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled) {
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled),
|
||||
m_enableSpirvValidation(enableSpirvValidation),
|
||||
m_updateAfterBindLimits(updateAfterBindLimits),
|
||||
m_subgroupPolicy(subgroupPolicy) {
|
||||
VkProgramObject::s_device = device;
|
||||
}
|
||||
~ProgramFactory() = default;
|
||||
// Destroys the pass-through tessellation control modules. Runs while the device is
|
||||
// still alive for the same reason ~VkProgramObject's does: this factory outlives
|
||||
// nothing that owns the device.
|
||||
~ProgramFactory();
|
||||
ProgramFactory(const ProgramFactory&) = delete;
|
||||
|
||||
HashType ComputeHash(const MG_State::GLState::ProgramObject& program, CompileOptionFlags flags) const;
|
||||
const VkProgramObject& GetOrCreateProgram(
|
||||
const MG_State::GLState::ProgramObject& program, CompileOptionFlags flags);
|
||||
|
||||
// The default framebuffer's current image height, baked as a literal into every
|
||||
// FragCoordYFlip variant (there is no push-constant or specialization channel here, and
|
||||
// adding one for a value that changes only on swapchain recreation would cost the draw
|
||||
// path more than a recompile costs a resize). It is therefore part of those variants'
|
||||
// identity: ComputeHash mixes it in when the bit is set, so a height change re-keys them
|
||||
// and leaves every other program's hash untouched. Setting a NEW height also bumps the
|
||||
// cache-structure epoch, because a caller holding a memoised VkProgramObject* would
|
||||
// otherwise keep using a module compiled against the old height.
|
||||
void SetDefaultFramebufferHeight(Uint32 height);
|
||||
Uint32 GetDefaultFramebufferHeight() const { return m_defaultFramebufferHeight; }
|
||||
|
||||
// Bumped whenever m_cache's STRUCTURE changes (any insert or erase): the cache is
|
||||
// an open-addressing map holding entries by value, so both moves existing entries.
|
||||
// A caller that memoised a VkProgramObject* may keep dereferencing it only while
|
||||
@@ -283,6 +448,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static VkShaderStageFlagBits ToVkStage(ShaderStage stage);
|
||||
static VkFormat ConvertSpirvImageFormatToVkFormat(SpvImageFormat format);
|
||||
static SamplerNumericDomain UniformTypeToSamplerNumericDomain(GLenum glType);
|
||||
// The same question for an IMAGE uniform (`image2D`, `uimageBuffer`, ...), which the
|
||||
// sampler form above deliberately does not answer. Kept separate rather than folded in
|
||||
// because the two are asked in different places for different reasons: a sampler's domain
|
||||
// decides a sampled VIEW format, an image's decides what a placeholder descriptor for an
|
||||
// UNBOUND image unit must be (see UniformManager::AcquireUnboundTexelBufferView and
|
||||
// GetUnboundStorageImageTexture) - a formatless `writeonly` declaration reflects no
|
||||
// format at all, and the numeric domain is then the only thing that constrains it.
|
||||
static SamplerNumericDomain UniformTypeToImageNumericDomain(GLenum glType);
|
||||
// True when any entry point declares the DepthReplacing execution mode, i.e. the
|
||||
// shader assigns gl_FragDepth. Exposed so the blended depth-write quirk's exemption
|
||||
// can be pinned by tests. A false negative loses the exemption, so such a shader is
|
||||
@@ -291,6 +464,39 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// True when an entry point reads the InstanceIndex builtin. Only gates a diagnostic:
|
||||
// without shaderDrawParameters such a shader cannot have gl_InstanceID rebased.
|
||||
static Bool ReflectedReadsInstanceIndexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
// True when an entry point declares the BaseVertex builtin, i.e. when a non-indexed
|
||||
// draw with this program has to take the ZeroBaseVertex variant.
|
||||
static Bool ReflectedReadsBaseVertexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
// Shared by the two above: does any entry point list an input variable decorated with
|
||||
// this builtin?
|
||||
static Bool ReflectedDeclaresInputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
|
||||
// True when an entry point writes the ViewportIndex builtin (gl_ViewportIndex), i.e. when
|
||||
// the program can route primitives to a viewport other than 0 and its pipeline therefore
|
||||
// has to declare more than one. Asks about OUTPUT variables because that is the direction
|
||||
// a pre-rasterization stage declares it in.
|
||||
static Bool ReflectedWritesViewportIndexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
static Bool ReflectedDeclaresOutputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
|
||||
|
||||
// The pass-through tessellation control stage GL 4.6 core 11.2.2 describes for a
|
||||
// program that has an evaluation stage and no control stage, for an input patch of
|
||||
// `patchVertices` control points. Returned BY VALUE (a stage description is a POD, and
|
||||
// the cache below is a rehashing map, so a pointer into it would not survive the next
|
||||
// distinct patch size). `.module == VK_NULL_HANDLE` means the stage could not be built:
|
||||
// the caller then has no control stage to inject, and CreatePipeline refuses the
|
||||
// pipeline rather than handing the driver a half-tessellated one.
|
||||
//
|
||||
// Keyed on the patch size because GL takes the output patch size from PATCH_VERTICES,
|
||||
// which is draw state, not link state - the CTS case that motivated this links at the
|
||||
// default 3 and draws at 4. The pipeline cache already re-keys on patchControlPoints,
|
||||
// so the module a pipeline was built with is part of that pipeline's identity.
|
||||
// Compiling is bounded by the number of distinct patch sizes a program draws with
|
||||
// (MAX_PATCH_VERTICES = 32 in the worst case, one or two in practice) and only ever
|
||||
// happens for the rare program that has no control stage at all.
|
||||
VkPipelineShaderStageCreateInfo GetOrCreatePassthroughTessControlStage(Uint32 patchVertices);
|
||||
|
||||
// Source of the module above. Exposed for tests: the generated GLSL is the whole
|
||||
// contract with the evaluation stage, so it is worth pinning independently of a device.
|
||||
static String BuildPassthroughTessControlSource(Uint32 patchVertices);
|
||||
|
||||
private:
|
||||
struct ProgramLookupCache {
|
||||
@@ -301,14 +507,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
|
||||
static TextureTarget UniformTypeToTextureTarget(GLenum glType);
|
||||
void ReflectVertexInputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
|
||||
// `stages` is ALWAYS ProgramObject::GetLinkedShaderStages() - one entry per module of
|
||||
// `spirv`, at the same index. Taking the stages rather than the shader objects is what
|
||||
// keeps the program's live attach list, which is a longer and differently-indexed list
|
||||
// the moment a glAttachShader lands after the link, from being passed here by mistake.
|
||||
void ReflectVertexInputs(const Vector<ShaderStage>& stages,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
void ReflectFragmentOutputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
|
||||
void ReflectViewportIndexUsage(const Vector<ShaderStage>& stages,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
void ReflectFragmentOutputs(const Vector<ShaderStage>& stages,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
void ReflectLayout(const MG_State::GLState::ProgramObject& program, const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
// Fills needsPassthroughTessControl / passthroughTessControlEmulatable off the linked
|
||||
// modules. Const and reflection-only: it decides nothing about the pipeline, it only
|
||||
// records what the evaluation stage's input interface is made of.
|
||||
void ReflectPassthroughTessControlNeed(const Vector<ShaderStage>& stages,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
Uint32 m_maxBindings = 0;
|
||||
@@ -320,12 +539,28 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// True only when the logical device enabled both
|
||||
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// Startup snapshot used only by internally synthesized shader modules, which do not
|
||||
// originate from a ProgramLinkTask.
|
||||
Bool m_enableSpirvValidation = false;
|
||||
// Device feature and limit gate resolved before vkCreateDevice. Keeping it in
|
||||
// the factory lets each reflected layout choose ordinary descriptors when its
|
||||
// own counts would exceed the update-after-bind budget.
|
||||
UpdateAfterBindLimits m_updateAfterBindLimits{};
|
||||
SubgroupLoweringPolicy m_subgroupPolicy{};
|
||||
// See SetDefaultFramebufferHeight. 0 means "not known yet"; the FragCoordYFlip bit is
|
||||
// never set before the swapchain exists, so no variant can be compiled against it.
|
||||
Uint32 m_defaultFramebufferHeight = 0;
|
||||
mutable ProgramLookupCache m_lastLookup;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
// See GetCacheStructureEpoch(). Starts at 1 so a zero-initialized memo can never match.
|
||||
Uint64 m_cacheStructureEpoch = 1;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
// Pass-through tessellation control stages by input patch size. Never evicted: at most
|
||||
// MAX_PATCH_VERTICES entries exist for the lifetime of the device, and every pipeline
|
||||
// ever built from one keeps referencing its module. A failed build is cached as
|
||||
// VK_NULL_HANDLE so a broken generator costs one compile, not one per draw.
|
||||
UnorderedMap<Uint32, VkPipelineShaderStageCreateInfo> m_passthroughTessControlStages;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -157,7 +157,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
MGLOG_I("Got %d surface formats:", swapchainCapabilities.surfaceFormats.size());
|
||||
for (const auto& sf : swapchainCapabilities.surfaceFormats) {
|
||||
MGLOG_I(" [%s, %s]", string_VkFormat(sf.format), string_VkColorSpaceKHR(sf.colorSpace));
|
||||
MGLOG_D(" [%s, %s]", string_VkFormat(sf.format), string_VkColorSpaceKHR(sf.colorSpace));
|
||||
}
|
||||
|
||||
const auto pickedSurfaceFormat = ChooseSwapchainSurfaceFormat(swapchainCapabilities.surfaceFormats);
|
||||
@@ -166,7 +166,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
MGLOG_I("Got %d present modes:", swapchainCapabilities.presentModes.size());
|
||||
for (const auto& pm : swapchainCapabilities.presentModes) {
|
||||
MGLOG_I(" %s", string_VkPresentModeKHR(pm));
|
||||
MGLOG_D(" %s", string_VkPresentModeKHR(pm));
|
||||
}
|
||||
|
||||
const auto presentMode = ChooseSwapchainPresentMode(swapchainCapabilities.presentModes);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -26,12 +26,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
public:
|
||||
struct SamplerBindingOverride {
|
||||
Uint32 binding = 0;
|
||||
Uint32 element = 0;
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
const MG_State::GLState::SamplerObject* sampler = nullptr;
|
||||
VkImageView imageView = VK_NULL_HANDLE;
|
||||
VkImageLayout imageLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
Bool forceNearestFiltering = false;
|
||||
};
|
||||
|
||||
Bool Initialize(VkDevice device, VkBufferManager* bufferManager,
|
||||
struct SamplerImageFeedbackBinding {
|
||||
Uint32 samplerBinding = 0;
|
||||
Uint32 samplerElement = 0;
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
const MG_State::GLState::SamplerObject* sampler = nullptr;
|
||||
SamplerNumericDomain numericDomain = SamplerNumericDomain::Unknown;
|
||||
};
|
||||
|
||||
// `physicalDevice` is only ever asked for format properties: a placeholder descriptor for
|
||||
// an unbound texel-buffer binding has to be built from a format the DEVICE accepts as a
|
||||
// texel buffer, and there is no other route to that answer from here.
|
||||
Bool Initialize(VkDevice device, VkPhysicalDevice physicalDevice, VkBufferManager* bufferManager,
|
||||
ProgramFactory* programFactory,
|
||||
VkDeviceSize minUniformBufferOffsetAlignment, Uint32 frameCount,
|
||||
Uint32 maxBindings = 16, Uint32 setsPerFrame = 64,
|
||||
@@ -53,10 +67,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// caches - a live layout's entry must never be purged (its sets would be
|
||||
// unreachable pool slots), so there is deliberately no age-based sweep here.
|
||||
void OnDescriptorSetLayoutDestroyed(VkDescriptorSetLayout descriptorSetLayout);
|
||||
// One record per visited CombinedImageSampler binding (post fallback substitution,
|
||||
// in binding order): the resolved texture and effective sampler, as never-reused
|
||||
// lifetime ids so a freed-and-reallocated object at the same heap address can only
|
||||
// MISS a comparison, never false-hit it (same ABA rule as SamplerResolveMemo).
|
||||
// One record per visited CombinedImageSampler DESCRIPTOR (post fallback substitution,
|
||||
// in binding order, and within a binding in array-element order): the resolved texture
|
||||
// and effective sampler, as never-reused lifetime ids so a freed-and-reallocated object
|
||||
// at the same heap address can only MISS a comparison, never false-hit it (same ABA
|
||||
// rule as SamplerResolveMemo). An arrayed binding contributes one record per element -
|
||||
// element granularity is required, or swapping the textures of two elements of the same
|
||||
// array would leave the record list identical and the fast path would keep a stale set.
|
||||
struct SampledBindingRecord {
|
||||
Uint64 textureLifetimeId = 0;
|
||||
Uint64 samplerLifetimeId = 0;
|
||||
@@ -76,6 +93,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool CollectStorageImageTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures) const;
|
||||
Bool CollectSamplerImageFeedback(
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<SamplerImageFeedbackBinding>& outBindings) const;
|
||||
static Bool SamplerOverlapsWritableImageSubresource(Int samplerBaseLevel, Int samplerMaxLevel,
|
||||
GLint imageLevel, GLenum imageAccess);
|
||||
// samplerDescriptorsUnchangedHint: the caller (SetupDraw fast path) proved that
|
||||
// every input of every combined-image-sampler resolution is unchanged since the
|
||||
// previous draw's resolve - same (texture, sampler) per binding, texture params
|
||||
@@ -88,7 +111,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 frameIndex,
|
||||
VkPipelineBindPoint bindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS,
|
||||
const SamplerBindingOverride* samplerBindingOverride = nullptr,
|
||||
Bool samplerDescriptorsUnchangedHint = false);
|
||||
Bool samplerDescriptorsUnchangedHint = false,
|
||||
const Vector<SamplerBindingOverride>* samplerBindingOverrides = nullptr);
|
||||
|
||||
// Pure format-policy helper kept public for host regression tests. Formatted storage
|
||||
// images use their shader qualifier; transformed float images use glBindImageTexture's
|
||||
@@ -111,6 +135,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDescriptorPool handle = VK_NULL_HANDLE;
|
||||
Uint32 maxSets = 0;
|
||||
Uint32 allocatedSets = 0;
|
||||
Bool updateAfterBind = false;
|
||||
};
|
||||
|
||||
// A cached descriptor set together with the pool it was allocated from, so a
|
||||
@@ -143,8 +168,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// texture after the fallback substitution (may still be null when no fallback
|
||||
// exists), effective sampler = unit override else the texture's own sampler.
|
||||
// False = the binding is skipped (unbound with a non-2D fallback target).
|
||||
// `element` indexes a sampler array inside the binding; see ResolveSamplerDescriptor.
|
||||
Bool ResolveSampledBinding(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding, Uint32 element,
|
||||
MG_State::GLState::ITextureObject*& outTexture,
|
||||
const MG_State::GLState::SamplerObject*& outSampler) const;
|
||||
// Raw-pointer variant for the per-draw sampled-texture walk (CollectSampledTextures):
|
||||
@@ -152,27 +178,71 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// only need the pointer skip the SharedPtr copy's atomic refcount churn.
|
||||
static MG_State::GLState::ITextureObject* ResolveSamplerTextureRaw(
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding);
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding, Uint32 element);
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetFallbackTexture(TextureTarget target) const;
|
||||
// ---- placeholders for UNBOUND image-backed descriptors -------------------------
|
||||
// GL lets a program declare `samplerBuffer`, `imageBuffer` or `image2D` and bind nothing
|
||||
// to the unit it names: the fetch is then undefined (GL 4.6 core 8.9 for an incomplete
|
||||
// buffer texture, 8.26 for an image unit with no texture) - undefined VALUES, not a
|
||||
// dropped draw. Vulkan has no unwritten descriptor, so something valid has to sit in the
|
||||
// set or the whole draw or dispatch is lost, which is what these two build. Same shape as
|
||||
// VkBufferManager::AcquireUnboundStorageDescriptor, one level up: per FORMAT rather than
|
||||
// one shared object, because a descriptor whose format disagrees with the shader's
|
||||
// declaration is invalid Vulkan even when nothing ever reads it.
|
||||
//
|
||||
// `declaredFormat` is the format the SHADER declared (VK_FORMAT_UNDEFINED for a sampled
|
||||
// texel buffer, which never carries one, or for a formatless `writeonly` image);
|
||||
// `numericDomain` decides the format when there is no declaration and is the fallback
|
||||
// class when the device cannot use the declared one as a texel buffer.
|
||||
VkBufferView AcquireUnboundTexelBufferView(VkFormat declaredFormat, SamplerNumericDomain numericDomain,
|
||||
Bool storage);
|
||||
// A 1x1 (x1 layer, or 6 faces for a cube) texture of `format`, shaped for `target` so the
|
||||
// view the descriptor gets has the view type the shader's image declaration demands.
|
||||
// Null for a target with no single-sampled placeholder shape - multisample images, whose
|
||||
// descriptor needs a multisample view that this cannot stand in for.
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetUnboundStorageImageTexture(TextureTarget target,
|
||||
VkFormat format) const;
|
||||
// The (target, format) pair a storage-image binding's placeholder is keyed by, resolved
|
||||
// from reflection alone. False when the binding has no placeholder shape.
|
||||
Bool ResolveUnboundStorageImagePlaceholder(const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
TextureTarget& outTarget, VkFormat& outFormat) const;
|
||||
// `element` indexes a sampler ARRAY inside one binding; each element carries its own
|
||||
// independently assigned GL texture unit, so it selects the texture, the sampler
|
||||
// override and the fallback separately from its neighbours.
|
||||
//
|
||||
// trustUnchangedHint: reuse this binding's cached VkDescriptorImageInfo outright
|
||||
// (see BindProgramUniformBuffers' samplerDescriptorsUnchangedHint for the proof
|
||||
// obligations the caller carries).
|
||||
// obligations the caller carries). The cache is keyed by binding alone, so it is
|
||||
// used ONLY for single-descriptor bindings - see m_samplerResolveMemo.
|
||||
Bool ResolveSamplerDescriptor(VkCommandBuffer commandBuffer, const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
VkDescriptorImageInfo& outImageInfo,
|
||||
Uint32 element, VkDescriptorImageInfo& outImageInfo,
|
||||
Bool trustUnchangedHint = false) const;
|
||||
Bool ResolveSamplerDescriptorOverride(const SamplerBindingOverride& samplerBindingOverride,
|
||||
VkDescriptorImageInfo& outImageInfo) const;
|
||||
Bool ResolveTexelBufferDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 frameIndex, VkBufferView& outBufferView);
|
||||
// GLSL `imageBuffer`: the same VkBufferView descriptor as the sampled texel buffer above,
|
||||
// but resolved from an IMAGE unit (glBindImageTexture) rather than a texture unit, and
|
||||
// made GPU-resident-writable because the shader may store to it. No `element` parameter:
|
||||
// an imageBuffer ARRAY is refused at program creation, so a binding is always one
|
||||
// descriptor (see the array gate in RemapDescriptorBindingsForVulkan).
|
||||
Bool ResolveStorageTexelBufferDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 frameIndex, VkBufferView& outBufferView);
|
||||
// `element` indexes a block INSTANCE array's descriptors; it is 0 for every ordinary
|
||||
// block. Each element resolves through its own GL storage block, and so its own GL
|
||||
// binding point, buffer and glBindBufferRange window.
|
||||
Bool ResolveStorageBufferDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
VkDescriptorBufferInfo& outBufferInfo) const;
|
||||
Uint32 element, VkDescriptorBufferInfo& outBufferInfo) const;
|
||||
// `element` indexes an image ARRAY inside one binding; each element carries its own
|
||||
// independently assigned GL image unit.
|
||||
Bool ResolveStorageImageDescriptor(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
VkDescriptorImageInfo& outImageInfo) const;
|
||||
Uint32 element, VkDescriptorImageInfo& outImageInfo) const;
|
||||
// Result of resolving a UBO binding: either a zero-copy direct bind to the app's resident
|
||||
// VkBuffer (the GLES backend's approach - no per-draw copy) or the CPU payload to upload.
|
||||
struct UboBindResult {
|
||||
@@ -201,8 +271,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void BindDescriptorSetDeduped(VkCommandBuffer commandBuffer, VkPipelineBindPoint bindPoint,
|
||||
VkPipelineLayout pipelineLayout, VkDescriptorSet descriptorSet,
|
||||
const Vector<Uint32>& dynamicOffsets);
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, Bool updateAfterBind, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex, Bool updateAfterBind);
|
||||
VkResult AllocateDescriptorSetsFromActivePool(
|
||||
Uint32 frameIndex, const ProgramFactory::VkProgramObject& programObj, VkDescriptorSet& outDescriptorSet);
|
||||
VkResult AcquireDescriptorSet(Uint32 frameIndex,
|
||||
@@ -210,6 +280,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDescriptorSet& outDescriptorSet);
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VkBufferManager* m_bufferManager = nullptr;
|
||||
ProgramFactory* m_programFactory = nullptr;
|
||||
Vector<FrameResources> m_frames;
|
||||
@@ -222,6 +293,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkTextureManager* m_textureManager = nullptr;
|
||||
VkSamplerManager* m_samplerManager = nullptr;
|
||||
mutable SharedPtr<MG_State::GLState::ITextureObject> m_fallbackTexture2D;
|
||||
// See AcquireUnboundTexelBufferView / GetUnboundStorageImageTexture. Both are lazily
|
||||
// populated, never evicted (a program's declared formats are a fixed, tiny set) and torn
|
||||
// down with the manager. The texel views are keyed by format AND by storage-vs-sampled
|
||||
// because the two descriptor kinds demand different format FEATURES of the device, so one
|
||||
// format can be usable for one and not the other. Deliberately NOT the per-frame
|
||||
// texelBufferViews list: those are destroyed at every frame boundary, and these must
|
||||
// outlive it or the placeholder would be rebuilt for every unbound binding every frame.
|
||||
UnorderedMap<Uint64, VkBufferView> m_unboundTexelBufferViews;
|
||||
mutable UnorderedMap<Uint64, SharedPtr<MG_State::GLState::ITextureObject>> m_unboundStorageImageTextures;
|
||||
|
||||
// Per-draw scratch buffers for BindProgramUniformBuffers: reused (clear keeps
|
||||
// capacity) so the descriptor-write path stops allocating on every draw.
|
||||
@@ -319,8 +399,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// lifetime id, so a freed-and-reallocated sampler or texture at the same heap address
|
||||
// always gets a fresh id and misses (a raw pointer would false-hit that ABA) - so a
|
||||
// stale guess can only miss and fall through to the hash, never resolve wrong. Still
|
||||
// reset each frame alongside the descriptor-set cache. Indexed by binding.
|
||||
// reset each frame alongside the descriptor-set cache. Indexed by binding, but the
|
||||
// whole-descriptor entry is additionally keyed by program lifetime: Vulkan binding
|
||||
// numbers are layout-local and unrelated programs routinely reuse binding 0/1.
|
||||
struct SamplerResolveMemo {
|
||||
Uint64 infoProgramLifetimeId = 0;
|
||||
Uint64 samplerLifetimeId = 0;
|
||||
Uint64 textureLifetimeId = 0;
|
||||
VkSampler sampler = VK_NULL_HANDLE;
|
||||
@@ -341,6 +424,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// proves every resolve input unchanged; cleared with the per-frame reset
|
||||
// (the cached VkSampler outlives a frame only via a fresh resolve, which
|
||||
// also re-stamps it against VkSamplerManager's frame-boundary sweep).
|
||||
//
|
||||
// This one field is keyed by binding but describes ONE descriptor, so it is
|
||||
// written and read only for single-descriptor bindings. A sampler ARRAY's
|
||||
// elements share the binding and would overwrite each other here - the last
|
||||
// element resolved would then be handed to element 0 on the next hinted draw.
|
||||
// Every other field above is self-validating (each compares its full key
|
||||
// before reuse, and the view-format entry is a pure function of format and
|
||||
// numeric domain), so an arrayed binding may keep using those.
|
||||
VkDescriptorImageInfo info{};
|
||||
Bool infoValid = false;
|
||||
};
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "VertexInputStateFactory.h"
|
||||
#include "MG_Util/Converters/MGToStr/DataTypeConverter.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <utility>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
@@ -107,10 +108,36 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
}
|
||||
|
||||
const VkFormat sourceVkFormat =
|
||||
VkFormat sourceVkFormat =
|
||||
ToVkVertexFormat(attr.Type, attr.Size, attr.Normalized, attr.IsInteger, attr.IsBgra, attr.IsLong);
|
||||
VertexStreamConversion conversion = VertexStreamConversion::None;
|
||||
// Gated on the SAME flag ToVkVertexFormat gates its 64-bit path on, and that is
|
||||
// load-bearing rather than belt-and-braces: the narrowing is only correct because the
|
||||
// shader's `dvec` input is a `vec` by the time the pipeline is built, and what
|
||||
// guarantees that is the flag being clear. It is clear on every backend today, and a
|
||||
// program with a 64-bit float vertex input is demoted WHOLE for the same reason even
|
||||
// where the device has native fp64 (ProgramSpirvTask::GenerateSpirv). With the flag
|
||||
// set, a dvec3/dvec4 would be declined by ToVkVertexFormat AND left 64-bit in the
|
||||
// module, so a float32 stream would be fed to a Float64 input.
|
||||
const Bool narrowFloat64Arrays =
|
||||
MG_Backend::pActiveBackendObject == nullptr ||
|
||||
!MG_Backend::pActiveBackendObject->GetDynamicParameters().SupportsFloat64VertexAttributes;
|
||||
if (sourceVkFormat == VK_FORMAT_UNDEFINED && attr.Type == DataType::Float64 && narrowFloat64Arrays) {
|
||||
// No native 64-bit fetch here (see ToVkVertexFormat's Float64 case), but the
|
||||
// source bytes are ordinary IEEE-754 doubles and DemoteFloat64Pass has already
|
||||
// narrowed every dvec input to a vec, so the array is narrowed to match rather
|
||||
// than dropped. Mirrors what DirectGLES does for the same state.
|
||||
const VkFormat narrowedFormat = ToFloat32VertexFormat(attr.Size);
|
||||
if (narrowedFormat != VK_FORMAT_UNDEFINED && SupportsVertexBufferFormat(narrowedFormat)) {
|
||||
sourceVkFormat = narrowedFormat;
|
||||
conversion = VertexStreamConversion::Float64ToFloat32;
|
||||
MGLOG_W_ONCE("Vertex attribute location=%u is a 64-bit (GL_DOUBLE) array; fetching it at "
|
||||
"float32 precision through format=%d (size=%d long=%s)",
|
||||
location, static_cast<Int>(narrowedFormat), attr.Size, attr.IsLong ? "true" : "false");
|
||||
}
|
||||
}
|
||||
if (sourceVkFormat == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_E("Unsupported vertex attribute layout (location=%u, type=%s, size=%d): the array is "
|
||||
MGLOG_E_ONCE("Unsupported vertex attribute layout (location=%u, type=%s, size=%d): the array is "
|
||||
"enabled but cannot be mapped to a VkFormat",
|
||||
location, MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size);
|
||||
unsupportedAttribMask |= (1u << location);
|
||||
@@ -118,14 +145,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
VkFormat vkFormat = sourceVkFormat;
|
||||
VertexStreamConversion conversion = VertexStreamConversion::None;
|
||||
if (!SupportsVertexBufferFormat(vkFormat)) {
|
||||
if (conversion == VertexStreamConversion::None && !SupportsVertexBufferFormat(vkFormat)) {
|
||||
if (IsScaledIntegerVertexFormat(vkFormat)) {
|
||||
const VkFormat fallbackFormat = ToFloat32VertexFormat(attr.Size);
|
||||
if (fallbackFormat != VK_FORMAT_UNDEFINED && SupportsVertexBufferFormat(fallbackFormat)) {
|
||||
vkFormat = fallbackFormat;
|
||||
conversion = VertexStreamConversion::ScaledIntegerToFloat32;
|
||||
MGLOG_W("Vertex attribute location=%u format=%d lacks "
|
||||
MGLOG_W_ONCE("Vertex attribute location=%u format=%d lacks "
|
||||
"VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT; using float32 stream format=%d "
|
||||
"(type=%s size=%d normalized=%s integer=%s)",
|
||||
location, static_cast<Int>(sourceVkFormat), static_cast<Int>(vkFormat),
|
||||
@@ -135,7 +161,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
if (conversion == VertexStreamConversion::None) {
|
||||
MGLOG_E("Unsupported Vulkan vertex format (location=%u, format=%d, type=%s, size=%d): "
|
||||
MGLOG_E_ONCE("Unsupported Vulkan vertex format (location=%u, format=%d, type=%s, size=%d): "
|
||||
"VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT is unavailable and no semantic fallback exists",
|
||||
location, static_cast<Int>(sourceVkFormat),
|
||||
MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size);
|
||||
@@ -146,15 +172,21 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const SizeT attribByteSize = GetAttributeByteSize(attr.Type, attr.Size, attr.IsBgra);
|
||||
if (attribByteSize == 0) {
|
||||
MGLOG_E("Vertex attribute with unknown component size (location=%u, type=%s): the array is "
|
||||
MGLOG_E_ONCE("Vertex attribute with unknown component size (location=%u, type=%s): the array is "
|
||||
"enabled but cannot be sized",
|
||||
location, MG_Util::ConvertDataTypeToString(attr.Type).c_str());
|
||||
unsupportedAttribMask |= (1u << location);
|
||||
continue;
|
||||
}
|
||||
|
||||
const Uint32 sourceStride =
|
||||
attr.Stride > 0 ? static_cast<Uint32>(attr.Stride) : static_cast<Uint32>(attribByteSize);
|
||||
// Verbatim, zero included. The frontend already resolved a pointer call's
|
||||
// "tightly packed" stride 0 into the element size (see VertexAttribute::Stride),
|
||||
// so a zero here is the binding model's stride 0 - every vertex reads the same
|
||||
// element - which is exactly what a zero VkVertexInputBindingDescription::stride
|
||||
// means. Substituting the element size fetched a fresh element per vertex and ran
|
||||
// off the end of the buffer (KHR-GL43.vertex_attrib_binding.basic-input-case7/8).
|
||||
// Client-memory arrays cannot reach zero: they only exist on the pointer path.
|
||||
const Uint32 sourceStride = static_cast<Uint32>(attr.Stride);
|
||||
const Bool packedAttribute = attr.Type == DataType::Int2101010Rev ||
|
||||
attr.Type == DataType::Uint2101010Rev;
|
||||
const SizeT requiredAlignment = packedAttribute ? attribByteSize : GetComponentSize(attr.Type);
|
||||
@@ -169,16 +201,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// unless VK_EXT_legacy_vertex_attributes is available, so deinterleave this one
|
||||
// attribute into a tightly packed transient stream without changing its format.
|
||||
conversion = VertexStreamConversion::Repack;
|
||||
MGLOG_W("Vertex attribute location=%u uses Vulkan-incompatible alignment "
|
||||
MGLOG_W_ONCE("Vertex attribute location=%u uses Vulkan-incompatible alignment "
|
||||
"(offset=%zu stride=%u required=%zu); using a tightly packed stream",
|
||||
location, attr.Offset, sourceStride, requiredAlignment);
|
||||
}
|
||||
|
||||
Uint32 stride = sourceStride;
|
||||
if (conversion == VertexStreamConversion::Repack) {
|
||||
stride = static_cast<Uint32>(attribByteSize);
|
||||
} else if (conversion == VertexStreamConversion::ScaledIntegerToFloat32) {
|
||||
stride = static_cast<Uint32>(attr.Size * static_cast<Int>(sizeof(Float)));
|
||||
// A converted stream is tightly packed, so its stride is the converted element
|
||||
// size - unless the source stride is zero, which does not describe a packing at
|
||||
// all but "never advance". That survives the conversion unchanged: the draw path
|
||||
// converts exactly one element and every vertex reads it.
|
||||
if (sourceStride != 0) {
|
||||
if (conversion == VertexStreamConversion::Repack) {
|
||||
stride = static_cast<Uint32>(attribByteSize);
|
||||
} else if (conversion == VertexStreamConversion::ScaledIntegerToFloat32 ||
|
||||
conversion == VertexStreamConversion::Float64ToFloat32) {
|
||||
stride = static_cast<Uint32>(attr.Size * static_cast<Int>(sizeof(Float)));
|
||||
}
|
||||
}
|
||||
const VkVertexInputRate inputRate =
|
||||
(attr.Divisor == 0) ? VK_VERTEX_INPUT_RATE_VERTEX : VK_VERTEX_INPUT_RATE_INSTANCE;
|
||||
@@ -275,8 +314,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (m_frameBoundaryCounter - it->second->lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
it = m_cache.erase(it);
|
||||
// Invalidate every VAO's state-pointer memo: the erased node's
|
||||
// address may be reused by a future insert.
|
||||
++m_evictionEpoch;
|
||||
// address may be reused by a future insert. Advance through the
|
||||
// process-wide source so the value stays unique across factory
|
||||
// instances (see the member comment).
|
||||
m_evictionEpoch = ++s_evictionEpochSource;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
@@ -316,6 +357,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// for every R64 float format, so a native 64-bit vertex fetch is simply unavailable there
|
||||
// while shaderFloat64 is not. Both halves key off nothing but the attribute being long,
|
||||
// so they always agree without extra plumbing.
|
||||
//
|
||||
// ... as long as the shader half still runs. It does not when the backend has declared
|
||||
// no 64-bit vertex attribute support: DemoteFloat64Pass has already narrowed every
|
||||
// `dvec` input to a `vec` by then, so PackDoubleVertexInputsPass finds nothing to pack
|
||||
// and a UINT-formatted attribute would be fed to a float input - garbage with no
|
||||
// diagnostic anywhere. Declining here hands the attribute to the caller's
|
||||
// Float64ToFloat32 fallback instead, which narrows the source doubles to match the
|
||||
// demoted `vec` input - the same thing DirectGLES does for the same state. The
|
||||
// frontend RECORDS the format either way, so this gate is the only thing standing
|
||||
// between a legal glVertexAttribLFormat and a mismatched pipeline.
|
||||
if (MG_Backend::pActiveBackendObject == nullptr ||
|
||||
!MG_Backend::pActiveBackendObject->GetDynamicParameters().SupportsFloat64VertexAttributes) {
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
if (!isLong || isInteger || normalized) return VK_FORMAT_UNDEFINED;
|
||||
switch (size) {
|
||||
case 1: return VK_FORMAT_R32G32_UINT;
|
||||
|
||||
@@ -23,6 +23,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
None = 0,
|
||||
Repack,
|
||||
ScaledIntegerToFloat32,
|
||||
// GL_DOUBLE source data narrowed to a tightly packed float32 stream: the fetch half
|
||||
// of the fp64 demotion the shader side already does unconditionally.
|
||||
Float64ToFloat32,
|
||||
};
|
||||
|
||||
struct BackendVertexInputState {
|
||||
@@ -111,10 +114,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
// Values are heap-allocated: FastSTL::unordered_map is open-addressing,
|
||||
// so INSERT invalidates references to stored values. The draw path (and
|
||||
// the VAOs' state-pointer memos) hold entry pointers across inserts;
|
||||
// only the unique_ptr cell moves, never the pointee.
|
||||
// Values are heap-allocated: UnorderedMap is open-addressing, so INSERT
|
||||
// invalidates references to stored values - and so does ERASE, which shifts
|
||||
// the rest of the probe cluster into the hole and therefore moves entries
|
||||
// other than the erased one. The draw path (and the VAOs' state-pointer
|
||||
// memos) hold entry pointers across both; only the unique_ptr cell moves,
|
||||
// never the pointee.
|
||||
UnorderedMap<HashType, UniquePtr<BackendVertexInputState>> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
@@ -123,7 +128,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// construction); a memo is honored only while its recorded epoch
|
||||
// matches, so an evicted entry can never be dereferenced through a
|
||||
// stale memo.
|
||||
Uint64 m_evictionEpoch = 1;
|
||||
//
|
||||
// Drawn from a process-wide source, never a per-instance counter: the VAO
|
||||
// memos outlive this factory (they live on pGLContext's VAOs, the renderer
|
||||
// is destroyed and recreated on EGL surface release/re-create), so a fresh
|
||||
// factory restarting at a dead factory's epoch value would honor its
|
||||
// dangling entry pointers. The constructor takes a value strictly greater
|
||||
// than anything a predecessor ever stamped, so a dead factory's memo can
|
||||
// never compare equal here - the same never-reused idiom as the lifetime ids.
|
||||
// Single-threaded like the rest of the factory (renderer-thread only).
|
||||
static inline Uint64 s_evictionEpochSource = 0;
|
||||
Uint64 m_evictionEpoch = ++s_evictionEpochSource;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -16,6 +16,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT;
|
||||
constexpr SizeT kLiveResourcePruneThreshold = 256;
|
||||
|
||||
// See VkBufferManager::AcquireUnboundStorageDescriptor. 256 bytes: comfortably past
|
||||
// every minStorageBufferOffsetAlignment in the wild, and free.
|
||||
constexpr VkDeviceSize kUnboundStorageDescriptorBytes = 256;
|
||||
// See VkBufferManager::AcquireUnboundTexelBufferDescriptor. The same 256 bytes, for the
|
||||
// same reason plus one: a texel buffer view's range must be a whole number of texels of
|
||||
// whatever format the placeholder is asked for, and 256 divides by every texel size in
|
||||
// the GL image-format table (1, 2, 4, 8 and 16 bytes).
|
||||
constexpr VkDeviceSize kUnboundTexelBufferDescriptorBytes = 256;
|
||||
|
||||
// A zero-copy persistent buffer is created once and never recreated (the app holds
|
||||
// its mapped pointer), and may be bound to any role, so it carries every usage.
|
||||
// TRANSFER_DST is added by CreateResidentStorage.
|
||||
@@ -23,7 +32,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT | VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
|
||||
// "Every usage" has to mean every usage: a buffer texture reached through an IMAGE
|
||||
// unit takes a VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER descriptor, and the write is
|
||||
// invalid unless the buffer was created with this bit. Nothing asked for it until
|
||||
// imageBuffer support existed, so the omission was invisible.
|
||||
VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
|
||||
// Appended to kPersistentBackedUsage when VK_EXT_transform_feedback is enabled
|
||||
// (see VkBufferManagerInitInfo::transformFeedbackUsageEnabled).
|
||||
constexpr VkBufferUsageFlags kTransformFeedbackUsage =
|
||||
@@ -126,6 +139,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
m_transientUploadArena.Shutdown();
|
||||
m_unboundStorageBuffer.Destroy();
|
||||
m_unboundTexelBuffer.Destroy();
|
||||
DestroyAllDeferredReleases();
|
||||
ReleaseAllLiveResources();
|
||||
m_copyProvider = nullptr;
|
||||
@@ -161,12 +176,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
void VkBufferManager::CollectAllDeferredReleases() {
|
||||
// Per-resource releases only. Every one of them was deferred behind a BumpSliceEpoch,
|
||||
// so no memo can still name the handle, and the caller has proved the GPU is idle.
|
||||
//
|
||||
// The transient arena's releases are deliberately NOT collected here. A buffer lands
|
||||
// there when the arena outgrows it mid-frame (BufferArena::EnsureCapacity), and at
|
||||
// that moment every slice already handed out from this frame's arena still names it -
|
||||
// VkBufferResource::transientSlice above all, which AcquireStreamedSlice keeps
|
||||
// serving for the whole frame serial on the strength of transientFrameSerial alone.
|
||||
// Nothing bumps the slice epoch for those other resources, so freeing the buffer
|
||||
// here left the streamed memo handing a destroyed VkBuffer to vkCmdBindIndexBuffer
|
||||
// (llvmpipe then faulted inside the draw; the Create/Flywheel indirect retrace died
|
||||
// exactly this way). Mid-frame drains do not advance m_frameSerial, so they must not
|
||||
// free arena storage either: the arena's own ResetFrame/BeginFrame is the point where
|
||||
// the slot's slices stop being reachable, and that is where these releases land.
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_deferredBufferReleases.size(); ++frameIndex) {
|
||||
CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_transientUploadArena.GetFrameCount(); ++frameIndex) {
|
||||
m_transientUploadArena.CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::NotifyDeviceIdle() {
|
||||
@@ -287,7 +313,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
.requiredFlags = requiredFlags,
|
||||
});
|
||||
if (!created || resource.buffer.Map() == nullptr) {
|
||||
MGLOG_E("VkBufferManager::CreateResidentStorage failed (size=%llu)",
|
||||
MGLOG_E_ONCE("VkBufferManager::CreateResidentStorage failed (size=%llu)",
|
||||
static_cast<unsigned long long>(size));
|
||||
resource.buffer.Destroy();
|
||||
resource.storageSize = 0;
|
||||
@@ -309,7 +335,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
if (!resource.buffer.Upload(bufferObject.MappedData(), size, 0)) {
|
||||
MGLOG_E("VkBufferManager::SwapStorageAndUploadAll: upload failed");
|
||||
MGLOG_E_ONCE("VkBufferManager::SwapStorageAndUploadAll: upload failed");
|
||||
resource.pendingFullUpload = true;
|
||||
return false;
|
||||
}
|
||||
@@ -368,6 +394,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
BumpSliceEpoch(*resource);
|
||||
// Any cached streaming slice refers to the previous contents.
|
||||
resource->transientFrameSerial = 0;
|
||||
// Redefining the store hands any adopted mapping back to the CPU shadow
|
||||
// (BufferObject::RedefineStorage), so a buffer that reaches here persistent-mapped
|
||||
// is an ordinary resident one again: it needs the busy-tracking and conditional
|
||||
// orphan below, and the next AcquirePersistentMap has to mint storage for the new
|
||||
// store rather than hand back a mapping of the old one.
|
||||
resource->persistentMapped = false;
|
||||
if (!resource->buffer.IsValid()) {
|
||||
return; // streaming-only resource: shadow + serial are enough
|
||||
}
|
||||
@@ -388,7 +420,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData(), size, 0)) {
|
||||
MGLOG_E("VkBufferManager::OnRespecify: in-place upload failed");
|
||||
MGLOG_E_ONCE("VkBufferManager::OnRespecify: in-place upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
}
|
||||
}
|
||||
@@ -413,7 +445,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (!IsResourceBusy(*resource)) {
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData() + offset,
|
||||
static_cast<VkDeviceSize>(size), static_cast<VkDeviceSize>(offset))) {
|
||||
MGLOG_E("VkBufferManager::OnSubData: host upload failed");
|
||||
MGLOG_E_ONCE("VkBufferManager::OnSubData: host upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
}
|
||||
return;
|
||||
@@ -450,7 +482,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if ((appAccess & BufferMappingAccessBit::Unsynchronized) || !IsResourceBusy(*resource)) {
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData() + offset,
|
||||
static_cast<VkDeviceSize>(size), static_cast<VkDeviceSize>(offset))) {
|
||||
MGLOG_E("VkBufferManager::OnFlushMappedRange: host upload failed");
|
||||
MGLOG_E_ONCE("VkBufferManager::OnFlushMappedRange: host upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
}
|
||||
return;
|
||||
@@ -542,7 +574,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject->GetSize());
|
||||
if (size == 0) {
|
||||
MGLOG_E("VkBufferManager::AcquireResidentSlice failed: buffer size is zero");
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireResidentSlice failed: buffer size is zero");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -564,7 +596,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
if (!resource->buffer.Upload(bufferObject->MappedData(), size, 0)) {
|
||||
MGLOG_E("VkBufferManager::AcquireResidentSlice failed: initial upload failed");
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireResidentSlice failed: initial upload failed");
|
||||
resource->buffer.Destroy();
|
||||
resource->storageSize = 0;
|
||||
resource->usageFlags = 0;
|
||||
@@ -599,7 +631,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject->GetSize());
|
||||
if (size == 0) {
|
||||
MGLOG_E("VkBufferManager::AcquireStreamedSlice failed: buffer size is zero");
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireStreamedSlice failed: buffer size is zero");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -685,6 +717,70 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_deferredResourceReleases[frameIndex].clear();
|
||||
}
|
||||
|
||||
BufferSlice VkBufferManager::AcquireUnboundStorageDescriptor() {
|
||||
if (!m_unboundStorageBuffer.IsValid()) {
|
||||
if (m_initInfo.allocator == nullptr) {
|
||||
return {};
|
||||
}
|
||||
// Host-visible so the zero fill needs no command buffer: this can be reached from
|
||||
// descriptor resolution, which runs inside an already-open recording and must not
|
||||
// start a copy of its own. The size is a whole minStorageBufferOffsetAlignment-safe
|
||||
// block rather than 4 bytes so that a shader which does read the block gets a
|
||||
// plausible unsized-array length instead of one that rounds to zero.
|
||||
const Bool created = m_unboundStorageBuffer.Create({
|
||||
.allocator = m_initInfo.allocator,
|
||||
.size = kUnboundStorageDescriptorBytes,
|
||||
.usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT,
|
||||
.memoryUsage = VMA_MEMORY_USAGE_AUTO,
|
||||
.allocationFlags = VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT |
|
||||
VMA_ALLOCATION_CREATE_MAPPED_BIT,
|
||||
.requiredFlags = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT,
|
||||
});
|
||||
if (!created) {
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireUnboundStorageDescriptor: placeholder creation failed");
|
||||
m_unboundStorageBuffer.Destroy();
|
||||
return {};
|
||||
}
|
||||
if (void* mapped = m_unboundStorageBuffer.GetMappedData()) {
|
||||
Memset(mapped, 0, static_cast<SizeT>(kUnboundStorageDescriptorBytes));
|
||||
}
|
||||
}
|
||||
return m_unboundStorageBuffer.GetSlice();
|
||||
}
|
||||
|
||||
BufferSlice VkBufferManager::AcquireUnboundTexelBufferDescriptor() {
|
||||
if (!m_unboundTexelBuffer.IsValid()) {
|
||||
if (m_initInfo.allocator == nullptr) {
|
||||
return {};
|
||||
}
|
||||
// A SECOND placeholder rather than more usage bits on the storage-block one. The two
|
||||
// are independent failure domains: a device that refuses this allocation must not
|
||||
// take the storage-block placeholder - and with it the fix this one is a sibling of -
|
||||
// down with it. Host-visible and zero-filled for the same reason as that one: this is
|
||||
// reached from descriptor resolution, inside an already-open recording, which must
|
||||
// not start a copy of its own.
|
||||
const Bool created = m_unboundTexelBuffer.Create({
|
||||
.allocator = m_initInfo.allocator,
|
||||
.size = kUnboundTexelBufferDescriptorBytes,
|
||||
.usage = VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT,
|
||||
.memoryUsage = VMA_MEMORY_USAGE_AUTO,
|
||||
.allocationFlags = VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT |
|
||||
VMA_ALLOCATION_CREATE_MAPPED_BIT,
|
||||
.requiredFlags = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT,
|
||||
});
|
||||
if (!created) {
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireUnboundTexelBufferDescriptor: placeholder creation failed");
|
||||
m_unboundTexelBuffer.Destroy();
|
||||
return {};
|
||||
}
|
||||
if (void* mapped = m_unboundTexelBuffer.GetMappedData()) {
|
||||
Memset(mapped, 0, static_cast<SizeT>(kUnboundTexelBufferDescriptorBytes));
|
||||
}
|
||||
}
|
||||
return m_unboundTexelBuffer.GetSlice();
|
||||
}
|
||||
|
||||
VkBufferUsageFlags VkBufferManager::GetVkBufferUsage(BufferKind kind) {
|
||||
switch (kind) {
|
||||
case BufferKind::Vertex:
|
||||
@@ -697,7 +793,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case BufferKind::Uniform:
|
||||
return VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT;
|
||||
case BufferKind::TextureBuffer:
|
||||
return VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT;
|
||||
// Both texel roles, for the same reason vertex/index carry both bits: one GL buffer
|
||||
// texture can be read as a samplerBuffer and written as an imageBuffer, and which of
|
||||
// the two it is only becomes known when a shader that uses it is bound - long after
|
||||
// the resident buffer was created. A VkBufferView for a storage-texel descriptor is
|
||||
// invalid unless the buffer was created with the storage bit, so a buffer that
|
||||
// acquired only the uniform bit could never be given one.
|
||||
return VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT;
|
||||
case BufferKind::ShaderStorage:
|
||||
return VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT;
|
||||
case BufferKind::Indirect:
|
||||
|
||||
@@ -102,10 +102,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Recreate all per-frame transient arenas
|
||||
Bool RecreateTransientArenas(Uint32 frameCount);
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Drains every frame slot's deferred buffer/resource releases (and the
|
||||
// transient arena's parked superseded blocks). Only valid when the
|
||||
// caller has proven every queue submission complete; used by the
|
||||
// present-less frame-boundary drain.
|
||||
// Drains every frame slot's deferred buffer/resource releases. Only valid when
|
||||
// the caller has proven every queue submission complete; used by the present-less
|
||||
// frame-boundary drain. Deliberately does NOT touch the transient arena's parked
|
||||
// superseded blocks: those are still named by this frame's slices (see the
|
||||
// definition), and only a frame rewind retires them.
|
||||
void CollectAllDeferredReleases();
|
||||
// All previously submitted GPU work has completed (vkDeviceWaitIdle).
|
||||
void NotifyDeviceIdle();
|
||||
@@ -119,6 +120,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool UploadTransient(BufferKind kind, Uint32 frameIndex, const void* data, VkDeviceSize size,
|
||||
VkDeviceSize alignment, BufferSlice& outSlice);
|
||||
|
||||
// The descriptor a shader storage block gets when the program declares it and the
|
||||
// application bound no buffer at its GL binding point. GL 4.6 core 7.8 makes that a
|
||||
// legal state - the block simply has no store, so reads are undefined and writes go
|
||||
// nowhere - whereas Vulkan has no such thing as an unwritten descriptor, so something
|
||||
// real has to sit in the set or the whole draw/dispatch is lost. One zero-filled
|
||||
// buffer, created once and shared by every unbound binding: bindings that are only
|
||||
// declared (the case this exists for) never touch it, and one that is actually read
|
||||
// sees zeros, which is inside GL's "undefined". robustBufferAccess bounds anything
|
||||
// that indexes past it.
|
||||
BufferSlice AcquireUnboundStorageDescriptor();
|
||||
|
||||
// The store a texel-buffer descriptor - `samplerBuffer` or `imageBuffer` - gets when the
|
||||
// unit the program's uniform names has no buffer texture on it, or the buffer texture on
|
||||
// it has no GL buffer attached. Both are legal GL states that make a fetch return
|
||||
// undefined values (GL 4.6 core 8.9: a buffer texture with no attached buffer object is
|
||||
// incomplete, and sampling an incomplete texture is undefined - not a lost draw), and both
|
||||
// used to take the whole draw or dispatch with them. The VIEW over this - one per format,
|
||||
// and the descriptor is a VkBufferView, not a buffer - is built by
|
||||
// UniformManager::AcquireUnboundTexelBufferView.
|
||||
BufferSlice AcquireUnboundTexelBufferDescriptor();
|
||||
|
||||
// Draw-time acquire for resident (device-storage) buffers: ensures the
|
||||
// resource exists and is fully uploaded, marks it used this frame.
|
||||
Bool AcquireResidentSlice(BufferKind kind, const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
@@ -179,6 +201,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkBufferManagerInitInfo m_initInfo{};
|
||||
BufferArena m_transientUploadArena;
|
||||
// See AcquireUnboundStorageDescriptor. Lazily created, never re-created, torn down
|
||||
// with the manager.
|
||||
VkBufferObject m_unboundStorageBuffer;
|
||||
// See AcquireUnboundTexelBufferDescriptor. Same lifetime rules.
|
||||
VkBufferObject m_unboundTexelBuffer;
|
||||
IBufferCopyCommandProvider* m_copyProvider = nullptr;
|
||||
Vector<Vector<VkBufferObject>> m_deferredBufferReleases;
|
||||
Vector<Vector<SharedPtr<VkBufferResource>>> m_deferredResourceReleases;
|
||||
|
||||
@@ -76,7 +76,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VkResult result =
|
||||
vmaCreateBuffer(m_allocator, &bufferInfo, &allocationInfo, &m_buffer, &m_allocation, nullptr);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("VkBufferObject::Create failed: vmaCreateBuffer returned %d", result);
|
||||
MGLOG_E_ONCE("VkBufferObject::Create failed: vmaCreateBuffer returned %d", result);
|
||||
m_allocator = nullptr;
|
||||
m_buffer = VK_NULL_HANDLE;
|
||||
m_allocation = nullptr;
|
||||
@@ -108,7 +108,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VkResult mapResult = vmaMapMemory(m_allocator, m_allocation, &m_mappedData);
|
||||
if (mapResult != VK_SUCCESS || m_mappedData == nullptr) {
|
||||
MGLOG_E("VkBufferObject::Map failed: vmaMapMemory returned %d", mapResult);
|
||||
MGLOG_E_ONCE("VkBufferObject::Map failed: vmaMapMemory returned %d", mapResult);
|
||||
m_mappedData = nullptr;
|
||||
return nullptr;
|
||||
}
|
||||
@@ -138,14 +138,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const Bool wasMapped = IsMapped();
|
||||
void* mapped = wasMapped ? m_mappedData : Map();
|
||||
if (mapped == nullptr) {
|
||||
MGLOG_E("VkBufferObject::Upload failed: unable to map buffer");
|
||||
MGLOG_E_ONCE("VkBufferObject::Upload failed: unable to map buffer");
|
||||
return false;
|
||||
}
|
||||
|
||||
Memcpy(static_cast<Uint8*>(mapped) + offset, data, static_cast<SizeT>(size));
|
||||
const VkResult flushResult = vmaFlushAllocation(m_allocator, m_allocation, offset, size);
|
||||
if (flushResult != VK_SUCCESS) {
|
||||
MGLOG_E("VkBufferObject::Upload failed: vmaFlushAllocation returned %d", flushResult);
|
||||
MGLOG_E_ONCE("VkBufferObject::Upload failed: vmaFlushAllocation returned %d", flushResult);
|
||||
if (!wasMapped) {
|
||||
Unmap();
|
||||
}
|
||||
@@ -170,7 +170,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
const VkResult result = vmaInvalidateAllocation(m_allocator, m_allocation, offset, resolvedSize);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("VkBufferObject::Invalidate failed: vmaInvalidateAllocation returned %d", result);
|
||||
MGLOG_E_ONCE("VkBufferObject::Invalidate failed: vmaInvalidateAllocation returned %d", result);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
|
||||
@@ -122,8 +122,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return &attachment;
|
||||
}
|
||||
|
||||
PendingClearKey VkClearManager::MakePendingClearKey(MG_State::GLState::ITextureObject* texture, Uint32 mipLevel,
|
||||
// The texture a pending clear is actually ABOUT. A clear issued through a GL texture view
|
||||
// (ARB_texture_view) targets the storage it views, so it must queue against - and be found
|
||||
// by - the storage texture; keying it on the view instead left the clear invisible to every
|
||||
// materialisation done through the parent's name (and vice versa), so the image stayed in
|
||||
// VK_IMAGE_LAYOUT_UNDEFINED and the readback was dropped as unreadable.
|
||||
static MG_State::GLState::ITextureObject* ClearStorageTextureOf(MG_State::GLState::ITextureObject* texture) {
|
||||
if (texture == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
const auto& storageOwner = texture->GetViewStorageOwner();
|
||||
return storageOwner ? storageOwner.get() : texture;
|
||||
}
|
||||
|
||||
PendingClearKey VkClearManager::MakePendingClearKey(MG_State::GLState::ITextureObject* rawTexture, Uint32 mipLevel,
|
||||
Uint32 baseArrayLayer, Uint32 layerCount) {
|
||||
MG_State::GLState::ITextureObject* texture = ClearStorageTextureOf(rawTexture);
|
||||
if (rawTexture != nullptr && texture != rawTexture) {
|
||||
// The caller named a level and a layer of the VIEW; the key describes the STORAGE, so
|
||||
// both have to be shifted into its numbering (GL 4.6 core 8.18). Without this a clear
|
||||
// of a view's level 0 would collide with a clear of the storage's level 0 even when
|
||||
// the view opened onto level 1.
|
||||
mipLevel += static_cast<Uint32>(rawTexture->GetViewMinLevel());
|
||||
baseArrayLayer += static_cast<Uint32>(rawTexture->GetViewMinLayer());
|
||||
}
|
||||
return PendingClearKey {
|
||||
.texture = texture,
|
||||
.textureLifetimeId = texture ? texture->GetLifetimeId() : 0,
|
||||
@@ -157,6 +179,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
TextureIdentity VkClearManager::MakeTextureIdentity(MG_State::GLState::ITextureObject* texture) {
|
||||
// Same rule as VkTextureManager::MakeTextureIdentity: a GL texture view is identified by
|
||||
// the storage it views. A clear posted against a view and one posted against its parent
|
||||
// target the same image, so they have to coalesce rather than queue independently.
|
||||
if (texture != nullptr) {
|
||||
const auto& storageOwner = texture->GetViewStorageOwner();
|
||||
if (storageOwner) {
|
||||
texture = storageOwner.get();
|
||||
}
|
||||
}
|
||||
return TextureIdentity {
|
||||
.texture = texture,
|
||||
.lifetimeId = texture ? texture->GetLifetimeId() : 0,
|
||||
@@ -166,7 +197,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void VkClearManager::MergeClearPayload(ClearAttachmentPayload& dst, const ClearAttachmentPayload& src) {
|
||||
dst.mask |= src.mask;
|
||||
if ((src.mask & GL_COLOR_BUFFER_BIT) != 0) {
|
||||
// The whole colour story travels together (same rule as
|
||||
// VkRenderPassManager::QueueRenderbufferClear): a glClearBufferiv/uiv
|
||||
// payload carries its value in colorInt/colorUint and its branch selector
|
||||
// in colorEncoding - dropping them here would leave the pending clear
|
||||
// reading as an all-zero float one.
|
||||
dst.color = src.color;
|
||||
dst.colorEncoding = src.colorEncoding;
|
||||
dst.colorInt = src.colorInt;
|
||||
dst.colorUint = src.colorUint;
|
||||
}
|
||||
if ((src.mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
||||
dst.depth = src.depth;
|
||||
@@ -278,9 +317,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return;
|
||||
}
|
||||
|
||||
const PendingClearKey key = MakePendingClearKey(texture.get());
|
||||
const auto& storageOwner = texture->GetViewStorageOwner();
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& storageTexture = storageOwner ? storageOwner : texture;
|
||||
const PendingClearKey key = MakePendingClearKey(storageTexture.get());
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_aliveObjects[MakeTextureIdentity(texture.get())] = texture;
|
||||
m_aliveObjects[MakeTextureIdentity(storageTexture.get())] = storageTexture;
|
||||
auto& pending = m_pendingClears[key];
|
||||
MergeClearPayload(pending, clearPayload);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
@@ -297,8 +338,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
const PendingClearKey key = MakePendingClearKey(attachment);
|
||||
// The alive entry must hold the STORAGE object, because the key names it:
|
||||
// LockTextureIdentityLocked cross-checks the two, and registering a view here under its
|
||||
// storage's identity made every lookup of this clear fail that check and silently report
|
||||
// "nothing pending" - which is how a clear issued through a view's framebuffer vanished.
|
||||
const auto& storageOwner = texture->GetViewStorageOwner();
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& storageTexture = storageOwner ? storageOwner : texture;
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_aliveObjects[MakeTextureIdentity(texture.get())] = texture;
|
||||
m_aliveObjects[MakeTextureIdentity(storageTexture.get())] = storageTexture;
|
||||
auto& pending = m_pendingClears[key];
|
||||
MergeClearPayload(pending, clearPayload);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
@@ -313,6 +360,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
texture = ClearStorageTextureOf(texture);
|
||||
const Uint64 lifetimeId = texture->GetLifetimeId();
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
for (auto it = m_pendingClears.begin(); it != m_pendingClears.end(); ++it) {
|
||||
@@ -403,6 +451,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
texture = ClearStorageTextureOf(texture);
|
||||
const Uint64 lifetimeId = texture->GetLifetimeId();
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||
|
||||
@@ -114,7 +114,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkClearManager {
|
||||
public:
|
||||
static PendingClearKey MakePendingClearKey(const MG_State::GLState::FramebufferAttachmentObject& attachment);
|
||||
static PendingClearKey MakePendingClearKey(MG_State::GLState::ITextureObject* texture, Uint32 mipLevel = 0,
|
||||
// Resolves a GL texture view to the storage it views before keying; see the definition.
|
||||
static PendingClearKey MakePendingClearKey(MG_State::GLState::ITextureObject* rawTexture, Uint32 mipLevel = 0,
|
||||
Uint32 baseArrayLayer = 0, Uint32 layerCount = 1);
|
||||
|
||||
Bool Initialize();
|
||||
|
||||
@@ -67,19 +67,35 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
static Uint32 ResolveAttachmentBaseArrayLayer(const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
// Every branch has to go through ToStorageArrayLayer, including the two that name layer 0
|
||||
// implicitly: a layered attachment of a texture VIEW starts at the view's first layer, not
|
||||
// at the image's, and a cube FACE index is a layer index like any other. Leaving either
|
||||
// unshifted made the render pass write layers [0, n) while the clear key, the blit, the
|
||||
// copy and the readback for the same attachment all addressed [minLayer, minLayer + n) -
|
||||
// they resolve the layer through their own copies of this helper, which do shift.
|
||||
const auto* texture = attachment.GetTexture().get();
|
||||
if (attachment.IsLayered()) {
|
||||
return 0;
|
||||
return ToStorageArrayLayer(texture, 0);
|
||||
}
|
||||
const TextureUploadTarget uploadTarget = attachment.GetTextureUploadTarget();
|
||||
if (!IsCubeMapFaceUploadTarget(uploadTarget)) {
|
||||
return static_cast<Uint32>(std::max(attachment.GetTextureLayer(), 0));
|
||||
return ToStorageArrayLayer(texture, attachment.GetTextureLayer());
|
||||
}
|
||||
return static_cast<Uint32>(uploadTarget) - static_cast<Uint32>(TextureUploadTarget::CubeMapPositiveX);
|
||||
const Int face =
|
||||
static_cast<Int>(uploadTarget) - static_cast<Int>(TextureUploadTarget::CubeMapPositiveX);
|
||||
return ToStorageArrayLayer(texture, face);
|
||||
}
|
||||
|
||||
// The attachment's size is GL geometry, and GL_TEXTURE_1D_ARRAY keeps its layer count in the
|
||||
// state-side HEIGHT rather than in z (see ToVulkanLevelExtent, which exists for exactly this
|
||||
// remap). Reading z directly gave every layered 1D-array attachment layerCount = 1, so a
|
||||
// geometry shader writing gl_Layer = 1..n had its output silently dropped and the parent's
|
||||
// upper layers were never written at all.
|
||||
static Uint32 ResolveAttachmentLayerCount(const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (attachment.IsLayered()) {
|
||||
return static_cast<Uint32>(std::max(attachment.GetSize().z(), 1));
|
||||
const auto& texture = attachment.GetTexture();
|
||||
const TextureTarget target = texture != nullptr ? texture->GetTarget() : TextureTarget::Unknown;
|
||||
return static_cast<Uint32>(std::max(ToVulkanLevelExtent(target, attachment.GetSize()).z(), 1));
|
||||
}
|
||||
return 1u;
|
||||
}
|
||||
@@ -123,7 +139,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
if (!attachment.IsComplete()) {
|
||||
MGLOG_W("GetOrCreateRenderPass: draw buffer slot %u (%s) on FBO %u has an incomplete texture attachment; using VK_ATTACHMENT_UNUSED",
|
||||
MGLOG_W_ONCE("GetOrCreateRenderPass: draw buffer slot %u (%s) on FBO %u has an incomplete texture attachment; using VK_ATTACHMENT_UNUSED",
|
||||
drawBufferIndex,
|
||||
MG_Util::ConvertFramebufferAttachmentTypeToString(attachmentType).c_str(),
|
||||
fbo.GetExternalIndex());
|
||||
@@ -132,7 +148,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
auto* texture = attachment.GetTexture().get();
|
||||
if (texture == nullptr) {
|
||||
MGLOG_W("GetOrCreateRenderPass: draw buffer slot %u (%s) on FBO %u resolved to a null texture; using VK_ATTACHMENT_UNUSED",
|
||||
MGLOG_W_ONCE("GetOrCreateRenderPass: draw buffer slot %u (%s) on FBO %u resolved to a null texture; using VK_ATTACHMENT_UNUSED",
|
||||
drawBufferIndex,
|
||||
MG_Util::ConvertFramebufferAttachmentTypeToString(attachmentType).c_str(),
|
||||
fbo.GetExternalIndex());
|
||||
@@ -311,7 +327,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
if (!TryResolveSampleCountFlagBits(renderbuffer->GetSamples(), sampleCount)) {
|
||||
MGLOG_E("GetOrCreateRenderbufferResource: unsupported renderbuffer sample count %d for renderbuffer %u",
|
||||
MGLOG_E_ONCE("GetOrCreateRenderbufferResource: unsupported renderbuffer sample count %d for renderbuffer %u",
|
||||
renderbuffer->GetSamples(),
|
||||
renderbuffer->GetExternalIndex());
|
||||
return nullptr;
|
||||
@@ -457,7 +473,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_physicalDevice, format, imageInfo.imageType, imageInfo.tiling, imageInfo.usage, imageInfo.flags,
|
||||
&imageFormatProperties);
|
||||
if (imageFormatResult != VK_SUCCESS || (imageFormatProperties.sampleCounts & sampleCount) == 0) {
|
||||
MGLOG_E("GetOrCreateRenderbufferResource: unsupported renderbuffer format=%d samples=%d for renderbuffer %u",
|
||||
MGLOG_E_ONCE("GetOrCreateRenderbufferResource: unsupported renderbuffer format=%d samples=%d for renderbuffer %u",
|
||||
static_cast<Int>(format),
|
||||
static_cast<Int>(sampleCount),
|
||||
renderbuffer->GetExternalIndex());
|
||||
@@ -633,11 +649,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (att.IsTexture()) {
|
||||
const Uint64 textureLifetimeId = att.GetTexture()->GetLifetimeId();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &textureLifetimeId, sizeof(textureLifetimeId)));
|
||||
const Int textureLevel = att.GetTextureLevel();
|
||||
const Int textureLevel = static_cast<Int>(ToStorageMipLevel(att.GetTexture().get(),
|
||||
att.GetTextureLevel()));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &textureLevel, sizeof(textureLevel)));
|
||||
const TextureUploadTarget textureUploadTarget = att.GetTextureUploadTarget();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &textureUploadTarget, sizeof(textureUploadTarget)));
|
||||
const Int textureLayer = att.GetTextureLayer();
|
||||
const Int textureLayer = static_cast<Int>(ToStorageArrayLayer(att.GetTexture().get(),
|
||||
att.GetTextureLayer()));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &textureLayer, sizeof(textureLayer)));
|
||||
const Bool textureLayered = att.IsLayered();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &textureLayered, sizeof(textureLayered)));
|
||||
@@ -831,6 +849,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// recreated since (texture + renderbuffer image epochs), and no pending clear (which alters
|
||||
// load ops). Any of these differing forces the full recompute below. Portable to VK 1.1.
|
||||
if (activeRenderPass != nullptr && m_rpFastValid && m_rpFastFbo == &fbo &&
|
||||
m_rpFastFboLifetimeId == fbo.GetLifetimeId() &&
|
||||
m_rpFastFboVersion == fbo.GetObjectVersion() && m_rpFastSwapchainIndex == swapchainImageIndex &&
|
||||
m_rpFastTexEpoch == m_textureManager.GetTextureImageEpoch() &&
|
||||
m_rpFastRbEpoch == m_renderbufferImageEpoch &&
|
||||
@@ -855,6 +874,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// epochs AFTER ComputeHash: its attachment SyncTexture can create an image (bump the epoch).
|
||||
m_rpFastValid = true;
|
||||
m_rpFastFbo = &fbo;
|
||||
m_rpFastFboLifetimeId = fbo.GetLifetimeId();
|
||||
m_rpFastFboVersion = fbo.GetObjectVersion();
|
||||
m_rpFastSwapchainIndex = swapchainImageIndex;
|
||||
m_rpFastTexEpoch = m_textureManager.GetTextureImageEpoch();
|
||||
@@ -929,7 +949,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const auto& renderbuffer = rbAtt.GetRenderbuffer();
|
||||
auto* rbResource = GetOrCreateRenderbufferResource(renderbuffer);
|
||||
if (rbResource == nullptr || (rbResource->aspect & VK_IMAGE_ASPECT_COLOR_BIT) == 0) {
|
||||
MGLOG_E("GetOrCreateRenderPass: draw buffer slot %u on FBO %u has an unsupported color "
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: draw buffer slot %u on FBO %u has an unsupported color "
|
||||
"renderbuffer %u; using VK_ATTACHMENT_UNUSED",
|
||||
i, fbo.GetExternalIndex(), renderbuffer->GetExternalIndex());
|
||||
continue;
|
||||
@@ -1004,7 +1024,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
|
||||
auto& att = fbo.GetAttachment(drawbuf);
|
||||
const Uint32 attachmentMipLevel = static_cast<Uint32>(std::max(att.GetTextureLevel(), 0));
|
||||
const Uint32 attachmentMipLevel = ToStorageMipLevel(att.GetTexture().get(), att.GetTextureLevel());
|
||||
const auto textureTarget = texture->GetTarget();
|
||||
const Uint32 attachmentIndex = static_cast<Uint32>(attachmentDescriptions.size());
|
||||
attachmentDescriptions.emplace_back();
|
||||
@@ -1047,8 +1067,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
.key = VkClearManager::MakePendingClearKey(att)
|
||||
});
|
||||
}
|
||||
const IntVec2 attachmentExtent =
|
||||
ResolveRenderPassFramebufferExtent(isDefaultFbo, att.GetSize(), swapchainExtent);
|
||||
// Same remap as ResolveAttachmentLayerCount, for the same reason: a
|
||||
// 1D-array attachment's GL height is its layer count, and using it as the
|
||||
// framebuffer height asks for a framebuffer taller than the VK_IMAGE_TYPE_1D
|
||||
// image it is built over.
|
||||
const IntVec2 attachmentExtent = ResolveRenderPassFramebufferExtent(
|
||||
isDefaultFbo, ToVulkanLevelExtent(texture->GetTarget(), att.GetSize()), swapchainExtent);
|
||||
if (width == 0)
|
||||
width = attachmentExtent.x();
|
||||
if (height == 0)
|
||||
@@ -1105,7 +1129,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
adoptRenderPassSampleCount(attachmentSampleCount, "color", texture->GetExternalIndex());
|
||||
|
||||
if (!hasClear && trackedColorLayout == VK_IMAGE_LAYOUT_UNDEFINED) {
|
||||
MGLOG_W("GetOrCreateRenderPass: color attachment textureId=%d starts with undefined layout and no clear; "
|
||||
MGLOG_W_ONCE("GetOrCreateRenderPass: color attachment textureId=%d starts with undefined layout and no clear; "
|
||||
"using LOAD_OP_DONT_CARE",
|
||||
texture->GetExternalIndex());
|
||||
desc.loadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
|
||||
@@ -1142,7 +1166,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (a.IsTexture() && b.IsTexture()) {
|
||||
return a.GetTexture().get() == b.GetTexture().get() &&
|
||||
a.GetTextureUploadTarget() == b.GetTextureUploadTarget() &&
|
||||
a.GetTextureLevel() == b.GetTextureLevel();
|
||||
ToStorageMipLevel(a.GetTexture().get(), a.GetTextureLevel()) ==
|
||||
ToStorageMipLevel(b.GetTexture().get(), b.GetTextureLevel());
|
||||
}
|
||||
if (a.IsRenderbuffer() && b.IsRenderbuffer()) {
|
||||
return a.GetRenderbuffer().get() == b.GetRenderbuffer().get();
|
||||
@@ -1161,7 +1186,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
isUsableDepthStencilAttachment(depthAtt) && isUsableDepthStencilAttachment(stencilAtt) &&
|
||||
!sameDepthStencilAttachmentObject(depthAtt, stencilAtt);
|
||||
if (hasDistinctDepthAndStencilAttachments) {
|
||||
MGLOG_E("GetOrCreateRenderPass: separate depth/stencil attachments are not supported yet; using the depth attachment and ignoring the standalone stencil attachment for framebuffer %u",
|
||||
MGLOG_E_ONCE("GetOrCreateRenderPass: separate depth/stencil attachments are not supported yet; using the depth attachment and ignoring the standalone stencil attachment for framebuffer %u",
|
||||
fbo.GetExternalIndex());
|
||||
}
|
||||
if (selectedDepthStencilAttachment != nullptr) {
|
||||
@@ -1197,9 +1222,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
depthAttachmentDescription.format = depthTextureResource->format;
|
||||
depthAttachmentSampleCount = depthTextureResource->sampleCount;
|
||||
depthAttachmentId = static_cast<Int>(texture.GetExternalIndex());
|
||||
attachmentExtent =
|
||||
ResolveRenderPassFramebufferExtent(isDefaultFbo, selectedDepthStencilAttachment->GetSize(),
|
||||
swapchainExtent);
|
||||
attachmentExtent = ResolveRenderPassFramebufferExtent(
|
||||
isDefaultFbo,
|
||||
ToVulkanLevelExtent(texture.GetTarget(), selectedDepthStencilAttachment->GetSize()),
|
||||
swapchainExtent);
|
||||
} else {
|
||||
const auto& renderbuffer = selectedDepthStencilAttachment->GetRenderbuffer();
|
||||
depthRenderbufferResource = GetOrCreateRenderbufferResource(renderbuffer);
|
||||
@@ -1223,7 +1249,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
depthAttachmentDescription.finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
||||
depthAttachmentDescription.initialLayout = loadInfo.initialLayout;
|
||||
if (trackedDepthLayout == VK_IMAGE_LAYOUT_UNDEFINED && (!clearDepth || !clearStencil)) {
|
||||
MGLOG_W("GetOrCreateRenderPass: depth/stencil attachment id=%d starts with undefined layout "
|
||||
MGLOG_W_ONCE("GetOrCreateRenderPass: depth/stencil attachment id=%d starts with undefined layout "
|
||||
"and partial/no clear; using DONT_CARE for uncleared aspects",
|
||||
depthAttachmentId);
|
||||
}
|
||||
@@ -1252,7 +1278,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
} else if (selectedDepthStencilAttachment->IsTexture()) {
|
||||
auto& texture = *selectedDepthStencilAttachment->GetTexture();
|
||||
const Uint32 attachmentMipLevel =
|
||||
static_cast<Uint32>(std::max(selectedDepthStencilAttachment->GetTextureLevel(), 0));
|
||||
ToStorageMipLevel(selectedDepthStencilAttachment->GetTexture().get(),
|
||||
selectedDepthStencilAttachment->GetTextureLevel());
|
||||
MOBILEGL_ASSERT(depthTextureResource->layout != VK_IMAGE_LAYOUT_UNDEFINED ||
|
||||
depthAttachmentDescription.loadOp != VK_ATTACHMENT_LOAD_OP_LOAD,
|
||||
"GetOrCreateRenderPass: depth attachment textureId=%d has undefined tracked layout with LOAD_OP_LOAD",
|
||||
@@ -1507,7 +1534,23 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
ClearAttachmentPayload clearPayload{};
|
||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||
if (pending.hasInlinePayload) {
|
||||
clearPayload = pending.inlinePayload;
|
||||
// The inline payload was snapshotted when the entry was CREATED, but the
|
||||
// clear VALUE is not part of the entry's hash - a cache hit with a newer
|
||||
// glClear would replay the creation-time value and drop the new one (the
|
||||
// texture path below is immune because it re-reads the live payload).
|
||||
// Same defense as ClearAttachmentsOnActiveRenderPass: prefer the live
|
||||
// pending clear, fall back to the snapshot only when none is queued.
|
||||
if (s_renderPassManager != nullptr &&
|
||||
s_renderPassManager->GetPendingRenderbufferClear(pending.renderbuffer, clearPayload)) {
|
||||
if ((clearPayload.mask & GL_COLOR_BUFFER_BIT) != 0 && pending.renderbuffer != nullptr &&
|
||||
MG_Util::GetBaseInternalFormatComponentCount(pending.renderbuffer->GetInternalFormat()) ==
|
||||
3) {
|
||||
// RGB renderbuffers are backed by an RGBA image; the missing alpha reads as 1.
|
||||
ForceOpaqueClearAlpha(clearPayload);
|
||||
}
|
||||
} else {
|
||||
clearPayload = pending.inlinePayload;
|
||||
}
|
||||
} else {
|
||||
if (pending.key.texture == nullptr ||
|
||||
!s_clearManager->GetPendingClear(pending.key, clearPayload, liveTexture)) {
|
||||
|
||||
@@ -101,6 +101,42 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
std::swap(layers, that.layers);
|
||||
std::swap(lastUsedFrame, that.lastUsedFrame);
|
||||
}
|
||||
// Move ASSIGNMENT, not just construction. The move constructor above and the
|
||||
// destructor below each independently suppress the implicit one, which left the
|
||||
// type move-constructible but not move-assignable - and therefore not swappable,
|
||||
// which std::swap(pair&, pair&) requires. That was invisible while UnorderedMap
|
||||
// only ever move-CONSTRUCTED an element into a fresh slot. ska::flat_hash_map
|
||||
// probes robin-hood: inserting swaps the entry being placed against the one
|
||||
// already sitting in the slot whenever it has travelled further from its desired
|
||||
// position, so the mapped type has to be swappable or the table fails to
|
||||
// instantiate at all.
|
||||
//
|
||||
// SWAP SEMANTICS, exactly like the move constructor: this does not release the
|
||||
// destination's handles, it parks them in `that`, which destroys them when it
|
||||
// dies. That is correct for the only caller - std::swap, whose temporary expires
|
||||
// immediately - and it is what keeps the three-move sequence from destroying a
|
||||
// live render pass. It is NOT correct for a hand-written `a = std::move(b)` where
|
||||
// `a` held live handles and `b` outlives the statement: those handles would then
|
||||
// survive until `b` dies. There is no such caller; add a destroy-then-steal
|
||||
// assignment before writing one.
|
||||
RenderPassEntry& operator=(RenderPassEntry&& that) noexcept {
|
||||
if (this != &that) {
|
||||
std::swap(hash, that.hash);
|
||||
std::swap(renderPass, that.renderPass);
|
||||
std::swap(framebuffer, that.framebuffer);
|
||||
std::swap(compatibilityHash, that.compatibilityHash);
|
||||
std::swap(pendingClearAttachments, that.pendingClearAttachments);
|
||||
std::swap(trackedAttachmentLayouts, that.trackedAttachmentLayouts);
|
||||
std::swap(attachmentCount, that.attachmentCount);
|
||||
std::swap(colorAttachmentCount, that.colorAttachmentCount);
|
||||
std::swap(hasDepthStencilAttachment, that.hasDepthStencilAttachment);
|
||||
std::swap(sampleCount, that.sampleCount);
|
||||
std::swap(extent, that.extent);
|
||||
std::swap(layers, that.layers);
|
||||
std::swap(lastUsedFrame, that.lastUsedFrame);
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
RenderPassEntry(
|
||||
Uint64 hash,
|
||||
VkRenderPass renderpass,
|
||||
@@ -253,6 +289,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// or a pending clear. Portable to Vulkan 1.1 (no dynamic_rendering / imageless FB needed).
|
||||
Bool m_rpFastValid = false;
|
||||
const MG_State::GLState::FramebufferObject* m_rpFastFbo = nullptr;
|
||||
// The FBO's never-reused lifetime id joins the raw pointer + Uint16 version:
|
||||
// a deleted FBO reallocated at the same address whose fresh setup performed
|
||||
// the same number of version bumps would otherwise compare equal (both count
|
||||
// from 0), serving the dead framebuffer's pass to the new object.
|
||||
Uint64 m_rpFastFboLifetimeId = 0;
|
||||
Uint16 m_rpFastFboVersion = 0;
|
||||
Uint32 m_rpFastSwapchainIndex = 0;
|
||||
Uint64 m_rpFastTexEpoch = 0;
|
||||
@@ -315,26 +356,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 deferredAtFrame = 0;
|
||||
};
|
||||
|
||||
// Node-based std::unordered_map, deliberately not FastSTL's open-addressing UnorderedMap:
|
||||
// Node-based std::unordered_map, deliberately NOT the open-addressing UnorderedMap:
|
||||
// callers cache a RenderbufferResource* - or a bare &resource->layout - and then make further
|
||||
// calls that touch this map. BlitFramebuffer is the one that bit: it resolves the source and
|
||||
// destination colour bindings (ResolveColorBlitBinding caches &rbResource->layout), then
|
||||
// materializes the source's pending clear, which looks that same resource up again. FastSTL's
|
||||
// operator[] runs its load-factor check before find_key and reallocates the whole bucket array
|
||||
// when occupancy crosses it, so even a plain lookup relocates every element; erase only
|
||||
// tombstones and never decrements the occupancy, so the doubling keeps firing. After a
|
||||
// relocation the cached pointer names freed storage still holding the pre-clear
|
||||
// VK_IMAGE_LAYOUT_UNDEFINED, and BlitFramebuffer bails out at "source image layout is
|
||||
// undefined", silently dropping the blit - renderbuffers_storage_multisample read back zero
|
||||
// instead of the clear colour on exactly the iterations that grew the table.
|
||||
// materializes the source's pending clear, which looks that same resource up again. Growing
|
||||
// an open-addressed table relocates every element, so the cached pointer went on to name
|
||||
// freed storage still holding the pre-clear VK_IMAGE_LAYOUT_UNDEFINED; BlitFramebuffer bailed
|
||||
// out at "source image layout is undefined", silently dropping the blit -
|
||||
// renderbuffers_storage_multisample read back zero instead of the clear colour on exactly the
|
||||
// iterations that grew the table.
|
||||
//
|
||||
// Reordering the materialize ahead of the resolves - the fix ReadPixels got - does not cover
|
||||
// this: the destination resolve still runs after the source pointer is taken. The depth blit,
|
||||
// GetOrCreateRenderPass's depthRenderbufferResource and ReadDepthStencilPixels cache the same
|
||||
// kind of pointer, so the invariant belongs in the container rather than in a per-call-site
|
||||
// ordering rule. m_textureResources is node-based for the same reason. This buys stability
|
||||
// across rehash and insert only - erase still invalidates the erased element, which is safe
|
||||
// here because a renderbuffer that is an FBO attachment is held alive by that attachment.
|
||||
// ordering rule. m_textureResources is node-based for the same reason.
|
||||
//
|
||||
// The case for keeping this node-based got STRONGER with ska::flat_hash_map, so do not read
|
||||
// the paragraph above as merely historical: ska erases by shifting the rest of the probe
|
||||
// cluster backwards into the hole, so erasing one renderbuffer relocates OTHER renderbuffers'
|
||||
// entries - a cached pointer can now be invalidated by a key it has nothing to do with, which
|
||||
// no call-site ordering rule can defend against. (What did change: ska's operator[] returns on
|
||||
// a hit before it runs its grow check, so a plain lookup of a PRESENT key no longer relocates.
|
||||
// That narrows the insert hazard; it does not touch the erase one.)
|
||||
std::unordered_map<MG_State::GLState::RenderbufferObject*, RenderbufferResource> m_renderbufferResources;
|
||||
UnorderedMap<MG_State::GLState::RenderbufferObject*, PendingRenderbufferClear> m_pendingRenderbufferClears;
|
||||
Vector<DeferredRenderbufferRelease> m_deferredRenderbufferReleases;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -22,6 +22,46 @@ class ITextureObject;
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class SamplerNumericDomain : Uint8;
|
||||
|
||||
// A GL 1D-ARRAY level keeps its LAYER COUNT in the state-side HEIGHT: that is what
|
||||
// glTexImage2D(GL_TEXTURE_1D_ARRAY, width, layers) means, and the frontend records the level
|
||||
// as {width, layers, 1} (see GL_Texture.cpp's AllocateStorage and the completeness walk in
|
||||
// TextureObject.cpp, which shrinks only x down the chain). Vulkan packs it the other way: a
|
||||
// 1D array is a VK_IMAGE_TYPE_1D image whose extent.height MUST be 1 and whose layers live in
|
||||
// arrayLayers - i.e. in the slot this backend reads out of z. So every place that turns a GL
|
||||
// level size into Vulkan image geometry has to move the count across first, and every GL-space
|
||||
// sub-box that rides along with it has to move its y the same way. DirectGLES performs the
|
||||
// identical remap onto the ES 2D array it maps 1D arrays to (GetBackendUploadSize).
|
||||
//
|
||||
// Applied to nothing else: a 2D array, a cube array and a 3D texture all already carry their
|
||||
// depth/layer count in z, which is where the Vulkan side expects it.
|
||||
inline IntVec3 ToVulkanLevelExtent(TextureTarget stateTarget, const IntVec3& glTexelSize) {
|
||||
if (stateTarget == TextureTarget::Texture1DArray) {
|
||||
return {glTexelSize.x(), 1, glTexelSize.y()};
|
||||
}
|
||||
return glTexelSize;
|
||||
}
|
||||
|
||||
// A GL framebuffer attachment's level/layer, and a GL image unit's, are relative to the texture
|
||||
// the application NAMED. When that texture was created by glTextureView (ARB_texture_view) they
|
||||
// are relative to the VIEW, and have to be shifted into the storage image's numbering before they
|
||||
// can index a Vulkan subresource - DirectVulkan gives a view no image of its own, it shares the
|
||||
// storage texture's (VkTextureManager::StorageTextureOf).
|
||||
//
|
||||
// Apply EXACTLY ONCE, at the boundary where a GL level/layer becomes a subresource index. Every
|
||||
// GetOrCreate*View entry point below expects values that have already been through here, and so
|
||||
// does everything that reads or copies an attachment directly. Both are identity on a plain
|
||||
// texture (TEXTURE_VIEW_MIN_LEVEL / MIN_LAYER are 0 there), so the conversion is unconditional
|
||||
// and there is no second, view-only code path to keep in step.
|
||||
inline Uint32 ToStorageMipLevel(const MG_State::GLState::ITextureObject* texture, Int glLevel) {
|
||||
const Uint32 level = static_cast<Uint32>(glLevel > 0 ? glLevel : 0);
|
||||
return texture != nullptr ? level + static_cast<Uint32>(texture->GetViewMinLevel()) : level;
|
||||
}
|
||||
|
||||
inline Uint32 ToStorageArrayLayer(const MG_State::GLState::ITextureObject* texture, Int glLayer) {
|
||||
const Uint32 layer = static_cast<Uint32>(glLayer > 0 ? glLayer : 0);
|
||||
return texture != nullptr ? layer + static_cast<Uint32>(texture->GetViewMinLayer()) : layer;
|
||||
}
|
||||
|
||||
class VkTextureManager {
|
||||
public:
|
||||
// Monotonic epoch bumped whenever a texture VkImage is (re)created. The render-pass
|
||||
@@ -120,17 +160,35 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
// Layer range and aspect join the key because a GL texture view (ARB_texture_view) can
|
||||
// differ from its storage on either: the Better Clouds shape samples ONE D24S8 image
|
||||
// through two GL names in one draw, the parent with the stencil aspect and the view with
|
||||
// the depth aspect, and a layer-sliced view of an array texture names a sub-range of the
|
||||
// same image. Without these two fields those views would alias each other in the cache.
|
||||
struct SampledImageViewKey {
|
||||
Uint32 baseMipLevel = 0;
|
||||
Uint32 levelCount = 1;
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = 1;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
VkImageAspectFlags aspect = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
// GL_TEXTURE_SWIZZLE_* is per-texture state, so two views over one storage with the
|
||||
// same window but different swizzles are different views. Baked into the key because
|
||||
// a GL texture view's ONLY sampled view lives in this cache: unlike the storage
|
||||
// texture's own sampledView, which SyncTextureViews rebuilds whenever the params
|
||||
// version moves, nothing else would ever notice a swizzle change on a view.
|
||||
Uint32 componentSwizzle = 0;
|
||||
|
||||
Bool operator==(const SampledImageViewKey& other) const {
|
||||
return baseMipLevel == other.baseMipLevel &&
|
||||
levelCount == other.levelCount &&
|
||||
baseArrayLayer == other.baseArrayLayer &&
|
||||
layerCount == other.layerCount &&
|
||||
viewType == other.viewType &&
|
||||
format == other.format;
|
||||
format == other.format &&
|
||||
aspect == other.aspect &&
|
||||
componentSwizzle == other.componentSwizzle;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -138,10 +196,15 @@ public:
|
||||
SizeT operator()(const SampledImageViewKey& key) const {
|
||||
SizeT hash = std::hash<Uint32>{}(key.baseMipLevel);
|
||||
hash ^= std::hash<Uint32>{}(key.levelCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(key.baseArrayLayer) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(key.layerCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewType)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.format)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.aspect)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(key.componentSwizzle) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
@@ -206,6 +269,12 @@ public:
|
||||
// as defense-in-depth: any path that grows the level set (which resizes the sampled view)
|
||||
// busts the skip even if it failed to bump the content version.
|
||||
Uint32 syncedMipLevelCount = 0;
|
||||
// Snapshot of ITextureObject::GetShapeVersion() at the last successful sync. The content
|
||||
// version alone does NOT cover a re-specification: glTexImage2D(..., nullptr) on an
|
||||
// already-defined level changes its size or format and dirties no texel, so it moves the
|
||||
// shape version and nothing else. Without this in the early-out key the image, its views
|
||||
// and therefore imageSize() all keep answering with the texture's PREVIOUS shape.
|
||||
Uint64 syncedShapeVersion = 0;
|
||||
|
||||
TextureResource() = default;
|
||||
TextureResource(const TextureResource&) = delete;
|
||||
@@ -237,6 +306,7 @@ public:
|
||||
std::swap(this->lastRecordingGeneration, that.lastRecordingGeneration);
|
||||
std::swap(this->syncedContentVersion, that.syncedContentVersion);
|
||||
std::swap(this->syncedMipLevelCount, that.syncedMipLevelCount);
|
||||
std::swap(this->syncedShapeVersion, that.syncedShapeVersion);
|
||||
}
|
||||
|
||||
void Reset() {
|
||||
@@ -300,6 +370,7 @@ public:
|
||||
syncedTextureParamsVersion = 0;
|
||||
syncedContentVersion = 0;
|
||||
syncedMipLevelCount = 0;
|
||||
syncedShapeVersion = 0;
|
||||
}
|
||||
|
||||
~TextureResource() {
|
||||
@@ -310,6 +381,11 @@ public:
|
||||
static inline VmaAllocator s_allocator = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
struct SampledTextureSnapshot {
|
||||
VkImageView imageView = VK_NULL_HANDLE;
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
};
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
@@ -326,6 +402,58 @@ public:
|
||||
// present-less frame-boundary drain.
|
||||
void CollectAllDeferredReleases();
|
||||
|
||||
// ---- GL texture views (ARB_texture_view / GL 4.6 core 8.18) ----
|
||||
// The GL texture whose STORAGE backs the given one: itself, or - for a texture created by
|
||||
// glTextureView - the texture it views. Every image-scoped question (which VkImage, its
|
||||
// LAYOUT, its uploads, its extent, its usage) must be asked of this object, because a view
|
||||
// has none of its own; only the VkImageViews differ per GL texture object. Sharing one
|
||||
// TextureResource is not an optimisation, it is the only correct arrangement: layout is a
|
||||
// property of the image, and VulkanRenderer caches raw pointers straight to the resource's
|
||||
// layout field, so a second resource aliasing the same image would desynchronise the moment
|
||||
// either of them transitioned it.
|
||||
static MG_State::GLState::ITextureObject& StorageTextureOf(MG_State::GLState::ITextureObject& texture);
|
||||
|
||||
// The window a GL texture object opens onto its storage image. For a plain texture this is
|
||||
// the resource's own full extent; for a view it is the sub-range, format and aspect
|
||||
// glTextureView gave it. Views built from a non-default window must live in the KEYED caches
|
||||
// (attachmentViews / alternateSampledViews), never in the per-mip vectors, which belong to
|
||||
// the storage texture's own defaults.
|
||||
struct TextureViewWindow {
|
||||
Uint32 baseMipLevel = 0;
|
||||
Uint32 levelCount = 1;
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = 1;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkImageAspectFlags sampledAspect = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
VkComponentMapping components{VK_COMPONENT_SWIZZLE_R, VK_COMPONENT_SWIZZLE_G, VK_COMPONENT_SWIZZLE_B,
|
||||
VK_COMPONENT_SWIZZLE_A};
|
||||
Bool isTextureView = false;
|
||||
};
|
||||
|
||||
// The four component swizzles packed into one value, for the sampled-view cache key.
|
||||
static Uint32 PackComponentSwizzle(const VkComponentMapping& components) {
|
||||
return (static_cast<Uint32>(components.r) & 0xFFu) | ((static_cast<Uint32>(components.g) & 0xFFu) << 8) |
|
||||
((static_cast<Uint32>(components.b) & 0xFFu) << 16) |
|
||||
((static_cast<Uint32>(components.a) & 0xFFu) << 24);
|
||||
}
|
||||
TextureViewWindow ResolveTextureViewWindow(MG_State::GLState::ITextureObject& texture,
|
||||
const TextureResource& resource) const;
|
||||
// Records what a GL texture view needs of the image it views, so the next sync of the
|
||||
// STORAGE texture creates (or recreates and copies forward) an image the view can be built
|
||||
// over. See m_viewRequestedImageFlags for why this is lazy rather than unconditional.
|
||||
void NoteTextureViewImageRequirements(MG_State::GLState::ITextureObject& viewTexture,
|
||||
MG_State::GLState::ITextureObject& storageTexture);
|
||||
VkImageCreateFlags GetViewRequestedImageFlags(const MG_State::GLState::ITextureObject& storageTexture) const;
|
||||
// Appends every format a GL texture view reinterprets this storage as, for the narrowed
|
||||
// VkImageFormatListCreateInfo the image is created with.
|
||||
void AppendViewRequestedFormats(const MG_State::GLState::ITextureObject& storageTexture,
|
||||
Vector<VkFormat>& outFormats) const;
|
||||
// Builds (and caches, keyed by the whole window) one sampled VkImageView over a storage
|
||||
// image. Shared back end of every GL-texture-view sampled path.
|
||||
VkImageView GetOrCreateWindowedSampledView(MG_State::GLState::ITextureObject& texture,
|
||||
TextureResource& resource, const TextureViewWindow& window);
|
||||
|
||||
TextureResource* SyncTextureAndGetDescriptor(
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
VkImageView GetOrCreateViewAtMipLevel(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel);
|
||||
@@ -343,6 +471,13 @@ public:
|
||||
VkImageLayout newLayout);
|
||||
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
// Copies the complete sampler-visible mip range into a transient sampled image. The source is
|
||||
// restored to its prior layout, so image-store descriptors continue to name the original image.
|
||||
// The transient ownership is tied to the current frame slot and is safe through its submission.
|
||||
Bool SnapshotTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture,
|
||||
SamplerNumericDomain numericDomain,
|
||||
VkPipelineStageFlags consumerShaderStageMask,
|
||||
SampledTextureSnapshot& outSnapshot);
|
||||
|
||||
// Recording-generation bookkeeping for the pre-pass command stream. The
|
||||
// generation advances every time the frame command buffer (re)begins
|
||||
@@ -379,17 +514,33 @@ public:
|
||||
// true - a false positive merely ends the render pass, a false negative would skip a barrier.
|
||||
Bool NeedsStorageImagePreparation(MG_State::GLState::ITextureObject& texture) const;
|
||||
|
||||
static VkImageAspectFlags ResolveSampledImageViewAspectMask(VkImageAspectFlags imageAspect);
|
||||
// `depthStencilTextureMode` is the texture's GL_DEPTH_STENCIL_TEXTURE_MODE; it only decides
|
||||
// anything for an image that carries both aspects. Defaulted so the call sites that have no
|
||||
// texture in hand keep the depth-aspect answer they have always given.
|
||||
static VkImageAspectFlags ResolveSampledImageViewAspectMask(VkImageAspectFlags imageAspect,
|
||||
GLenum depthStencilTextureMode = GL_DEPTH_COMPONENT);
|
||||
static VkFormat ResolveSampledImageViewFormat(VkFormat imageFormat, SamplerNumericDomain numericDomain);
|
||||
static Bool AreSampledImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
|
||||
static Bool AreStorageImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
|
||||
|
||||
// Moves `image` to `newLayout` and writes the new layout back through `trackedLayout`.
|
||||
//
|
||||
// The barrier covers EVERY array layer of the image, and there is deliberately no layer
|
||||
// parameter to say otherwise: layout here is tracked per IMAGE (one `TextureResource::layout`,
|
||||
// or one caller-owned variable), so a barrier narrower than the image would leave the layers it
|
||||
// skipped in the old layout while the tracker claims they moved. Every transfer against a
|
||||
// framebuffer attachment above layer 0 - glReadPixels, glBlitFramebuffer, glCopyTexSubImage,
|
||||
// glCopyImageSubData - then ran its copy on a layer no barrier had transitioned.
|
||||
//
|
||||
// The mip range IS a parameter, because mip levels really are transitioned piecewise (see
|
||||
// UpdateTrackedImageLayoutAfterAttachmentWrite and the mipmap generation loops): those callers
|
||||
// move the complement of the level they wrote so the whole image converges on one layout again.
|
||||
// Nothing does, or can, do that per layer.
|
||||
static Bool TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image, VkImageLayout& trackedLayout,
|
||||
VkImageLayout newLayout, VkPipelineStageFlags srcStageMask,
|
||||
VkPipelineStageFlags dstStageMask, VkAccessFlags srcAccessMask,
|
||||
VkAccessFlags dstAccessMask, VkImageAspectFlags aspectMask,
|
||||
Uint32 baseMipLevel = 0, Uint32 levelCount = 1,
|
||||
Uint32 layerCount = 1);
|
||||
Uint32 baseMipLevel = 0, Uint32 levelCount = 1);
|
||||
|
||||
SizeT CollectGarbage();
|
||||
|
||||
@@ -517,6 +668,19 @@ private:
|
||||
std::unordered_map<TextureIdentity, TextureResource, TextureIdentityHash> m_textureResources;
|
||||
// Textures that have been bound to a GL image unit (see MarkStorageImageTexture).
|
||||
std::unordered_set<TextureIdentity, TextureIdentityHash> m_storageImageTextures;
|
||||
// Extra VkImageCreateFlags a GL texture view needs on the storage image it views, keyed by
|
||||
// the STORAGE texture's identity. Requested lazily, exactly like STORAGE usage above and for
|
||||
// the same reason: VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT costs bandwidth compression on tilers
|
||||
// (it is what VK_KHR_image_format_list exists to claw back), so setting it on every
|
||||
// immutable-storage texture would tax every glTexStorage2D render target in a game for a
|
||||
// feature almost none of them use. A SAME-format view - which is the common case, and the
|
||||
// Better Clouds case - needs no flag at all and therefore costs nothing.
|
||||
std::unordered_map<TextureIdentity, VkImageCreateFlags, TextureIdentityHash> m_viewRequestedImageFlags;
|
||||
// Every VkFormat a GL texture view has asked to reinterpret this storage as. The narrowed
|
||||
// VkImageFormatListCreateInfo the image is created with must name them: the list is a promise
|
||||
// that NO other format will ever be viewed, and building a view outside it is
|
||||
// VUID-VkImageViewCreateInfo-pNext-01585. Keyed, like the flags above, by the STORAGE texture.
|
||||
std::unordered_map<TextureIdentity, std::unordered_set<VkFormat>, TextureIdentityHash> m_viewRequestedFormats;
|
||||
// Supported multisample counts per format, so repeat texture syncs do not
|
||||
// re-query vkGetPhysicalDeviceImageFormatProperties.
|
||||
std::unordered_map<VkFormat, VkSampleCountFlags> m_multisampleCountsByFormat;
|
||||
|
||||
@@ -15,7 +15,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(initInfo.device != VK_NULL_HANDLE, "VkTimerQueryManager::Initialize requires valid VkDevice");
|
||||
MOBILEGL_ASSERT(initInfo.frameCount > 0, "VkTimerQueryManager::Initialize requires non-zero frame count");
|
||||
if (initInfo.timestampValidBits == 0 || initInfo.timestampPeriodNs <= 0.0f || initInfo.slotsPerPool == 0) {
|
||||
MGLOG_W("VkTimerQueryManager: timestamps unsupported (validBits=%u, period=%f, slots=%u)",
|
||||
MGLOG_W_ONCE("VkTimerQueryManager: timestamps unsupported (validBits=%u, period=%f, slots=%u)",
|
||||
initInfo.timestampValidBits, initInfo.timestampPeriodNs, initInfo.slotsPerPool);
|
||||
return false;
|
||||
}
|
||||
@@ -35,7 +35,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (auto& poolState : m_pools) {
|
||||
const VkResult result = vkCreateQueryPool(m_device, &poolInfo, nullptr, &poolState.pool);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("VkTimerQueryManager: vkCreateQueryPool failed with %s", VkResultToString(result));
|
||||
MGLOG_E_ONCE("VkTimerQueryManager: vkCreateQueryPool failed with %s", VkResultToString(result));
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
@@ -90,7 +90,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
auto& poolState = m_pools[frameIndex];
|
||||
if (poolState.cursor >= m_slotsPerPool) {
|
||||
if (!poolState.exhaustionWarned) {
|
||||
MGLOG_W("VkTimerQueryManager: frame %u timestamp pool exhausted (%u slots); further timer queries "
|
||||
MGLOG_W_ONCE("VkTimerQueryManager: frame %u timestamp pool exhausted (%u slots); further timer queries "
|
||||
"this frame fall back to the frontend path",
|
||||
frameIndex, m_slotsPerPool);
|
||||
poolState.exhaustionWarned = true;
|
||||
@@ -120,7 +120,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_device, m_pools[record.poolIndex].pool, record.slot, 1, sizeof(resultWithAvailability),
|
||||
resultWithAvailability, sizeof(Uint64), VK_QUERY_RESULT_64_BIT | VK_QUERY_RESULT_WITH_AVAILABILITY_BIT);
|
||||
if (result != VK_SUCCESS && result != VK_NOT_READY) {
|
||||
MGLOG_E("VkTimerQueryManager: vkGetQueryPoolResults failed with %s", VkResultToString(result));
|
||||
MGLOG_E_ONCE("VkTimerQueryManager: vkGetQueryPoolResults failed with %s", VkResultToString(result));
|
||||
return false;
|
||||
}
|
||||
if (resultWithAvailability[1] == 0) {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -23,6 +23,7 @@
|
||||
#include "VkTimerQueryManager.h"
|
||||
#include "MG_Util/Math/VectorTypes.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
@@ -197,9 +198,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLbitfield mask, GLenum filter);
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset,
|
||||
GLint x, GLint y, GLsizei width, GLsizei height);
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
void CopyImageSubData(const CopyImageEndpoint& srcEndpoint,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
const CopyImageEndpoint& dstEndpoint,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void GenerateMipmap(GLenum target);
|
||||
@@ -211,10 +212,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLsizei height, GLenum format, GLenum type, void* pixels);
|
||||
// Copy-and-repack core shared by depth-stencil ReadPixels and GetTexImage;
|
||||
// expects command recording to be active and any render pass already ended.
|
||||
//
|
||||
// `defaultFramebufferOrientation` is set only when the source is the swapchain's
|
||||
// depth/stencil image, which this renderer stores display-side-up: the copy rect then
|
||||
// has to be mapped out of GL's bottom-origin space and the copied rows re-oriented on
|
||||
// the way back, exactly as the colour ReadPixels path does.
|
||||
// `sourceLayerCount` above 1 says the `height` rows the client is owed are stored as that
|
||||
// many ARRAY LAYERS of a one-row image rather than as rows of one layer - the shape a GL
|
||||
// 1D array has in Vulkan. The two produce byte-identical tightly-packed readbacks, so
|
||||
// only the copy region differs; everything after it is written against `height`.
|
||||
void ReadDepthStencilImageToClient(VkImage image, VkFormat vkFormat, VkImageLayout* trackedLayout,
|
||||
VkImageAspectFlags imageAspect, Uint32 mipLevel, Uint32 baseArrayLayer,
|
||||
GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type,
|
||||
void* pixels);
|
||||
void* pixels, Bool defaultFramebufferOrientation = false,
|
||||
Uint32 sourceLayerCount = 1);
|
||||
// Same-extent depth blit between images of different depth formats: host
|
||||
// round-trip with a per-texel re-encode (see BlitNamedFramebuffer).
|
||||
Bool BlitDepthAcrossFormats(FrameContext::FrameData& frame, VkImage srcImage, VkFormat srcFormat,
|
||||
@@ -224,6 +235,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLint dstY, GLint width, GLint height, VkImageLayout srcRestoreLayout,
|
||||
VkImageLayout dstRestoreLayout, Bool stencilAspect);
|
||||
static SizeT GetReadbackTexelSize(VkFormat sourceFormat);
|
||||
// Map a GL bottom-left-origin rectangle into the display-oriented swapchain image.
|
||||
// Quarter-turn surface transforms swap the copy extent's axes.
|
||||
static Bool MapDefaultFramebufferReadbackRect(GLint x, GLint y, GLsizei width, GLsizei height,
|
||||
VkExtent2D imageExtent,
|
||||
VkSurfaceTransformFlagBitsKHR preTransform,
|
||||
VkOffset2D* imageOffset, VkExtent2D* imageCopyExtent);
|
||||
// Reorder a tightly packed block copied with MapDefaultFramebufferReadbackRect back into
|
||||
// GL row order. The input block has swapped dimensions for 90/270 degree transforms.
|
||||
static Bool RemapDefaultFramebufferReadback(const Uint8* rawPixels, Uint32 logicalWidth,
|
||||
Uint32 logicalHeight,
|
||||
VkSurfaceTransformFlagBitsKHR preTransform,
|
||||
SizeT texelSize, Uint8* outPixels);
|
||||
static Bool ConvertReadbackPixels(const Uint8* sourcePixels, VkFormat sourceFormat,
|
||||
GLsizei width, GLsizei height, GLenum destinationFormat,
|
||||
GLenum destinationType, SizeT destinationRowStride,
|
||||
@@ -293,6 +316,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// The samplerAnisotropy device feature was granted, so GL_TEXTURE_MAX_ANISOTROPY_EXT is
|
||||
// honored rather than accepted-and-ignored.
|
||||
Bool IsSamplerAnisotropySupported() const { return m_samplerAnisotropyFeatureEnabled; }
|
||||
// ARB_base_instance extends indirect command records with a non-zero firstInstance and
|
||||
// requires gl_InstanceID to remain zero-based. Vulkan needs both features to honor that
|
||||
// complete contract: one legalizes the command word, the other enables the shader rebase.
|
||||
Bool IsNonZeroIndirectBaseInstanceSupported() const {
|
||||
return m_drawIndirectFirstInstanceFeatureEnabled && m_shaderDrawParametersFeatureEnabled;
|
||||
}
|
||||
// Ensures the frame command buffer is recording (same lazy pattern as
|
||||
// SetupDraw) and writes a bottom-of-pipe timestamp into the current
|
||||
// frame's pool. Null when unsupported or the pool is exhausted.
|
||||
@@ -363,6 +392,31 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 samplerBinding = 0;
|
||||
};
|
||||
|
||||
// A single-sample staging image for multisample-resolve blits that also have to change
|
||||
// orientation. vkCmdResolveImage cannot flip (it takes one offset per side, not the
|
||||
// invertible pair vkCmdBlitImage takes), so a resolve into or out of the default
|
||||
// framebuffer used to land the mirrored band. Resolving here first and then blitting from
|
||||
// here separates the two operations, and each one then does only what it can express.
|
||||
//
|
||||
// Pooled rather than created per blit: the CTS runs hundreds of these back to back, and
|
||||
// create-destroy per call would both cost allocations and, worse, need per-call deferred
|
||||
// destruction to outlive the recording. It grows to the largest extent asked for and is
|
||||
// reused; format changes recreate it.
|
||||
struct MultisampleResolveScratchImage {
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = VK_NULL_HANDLE;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
VkExtent2D extent = {0, 0};
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
};
|
||||
MultisampleResolveScratchImage m_msResolveScratch;
|
||||
// Returns a scratch image at least `extent` in size with exactly `format`, transitioned to
|
||||
// TRANSFER_DST and ready to be resolved into. Null image on failure (the caller then falls
|
||||
// back to the direct resolve).
|
||||
Bool AcquireMultisampleResolveScratchImage(VkCommandBuffer commandBuffer, VkFormat format,
|
||||
VkExtent2D extent);
|
||||
void DestroyMultisampleResolveScratchImage();
|
||||
|
||||
struct DeferredDepthMipmapCleanup {
|
||||
Vector<VkImageView> imageViews;
|
||||
Vector<VkFramebuffer> framebuffers;
|
||||
@@ -445,15 +499,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void* m_platformDisplay = nullptr;
|
||||
void* m_platformLibrary = nullptr;
|
||||
void* m_platformCloseDisplay = nullptr;
|
||||
// Some real ICDs (e.g. NVIDIA's proprietary Linux driver) don't implement
|
||||
// VK_EXT_headless_surface at all. Detected once in CreateInstance() from the
|
||||
// enumerated instance extensions; when false, CreateSurface() falls back to a
|
||||
// hidden Xlib window instead of vkCreateHeadlessSurfaceEXT.
|
||||
// Whether the loader exposes VK_EXT_headless_surface, detected once in
|
||||
// CreateInstance() from the enumerated instance extensions. On desktop an
|
||||
// offscreen surface REQUIRES it: false is a clean, loud bring-up failure, never
|
||||
// a substituted window. (Android is the one exception and has its own path -
|
||||
// no Mali/Adreno driver seen so far exposes the extension, so a windowless
|
||||
// context is given an AImageReader ANativeWindow that is never displayed.)
|
||||
Bool m_headlessSurfaceSupported = true;
|
||||
// Set when CreateSurface() had to create its own Xlib window for the fallback
|
||||
// above (rather than being handed one by the caller), so Shutdown() knows it
|
||||
// owns that window and must destroy it.
|
||||
Bool m_ownsFallbackXlibWindow = false;
|
||||
// Android has the same shortfall: no Mali/Adreno driver seen so far exposes
|
||||
// VK_EXT_headless_surface, so a windowless (EGL pbuffer) context gets an
|
||||
// AImageReader's ANativeWindow to hand the WSI instead. Nothing is ever
|
||||
@@ -508,7 +560,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool m_samplerAnisotropyFeatureEnabled = false;
|
||||
Bool m_shaderDrawParametersExtensionEnabled = false;
|
||||
Bool m_shaderDrawParametersFeatureEnabled = false;
|
||||
// Native subgroup topology, queried at device creation for the compute-module
|
||||
// subgroup repairs (SubgroupSupportPolicy.h) and the REQUIRE_FULL_SUBGROUPS
|
||||
// stage flag; 0 / false when the device has no usable compute subgroups or
|
||||
// MOBILEGL_DISABLE_SUBGROUP forced them off.
|
||||
Uint32 m_nativeSubgroupSize = 0;
|
||||
Bool m_nativeSubgroupSupported = false;
|
||||
Bool m_computeFullSubgroupsFeatureEnabled = false;
|
||||
// VkPhysicalDeviceSubgroupSizeControlProperties::maxComputeWorkgroupSubgroups;
|
||||
// 0 when the extension (and therefore the full-subgroups flag) is unavailable.
|
||||
Uint32 m_maxComputeWorkgroupSubgroups = 0;
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// Set only after descriptor-indexing feature AND property queries prove that
|
||||
// update-after-bind is legal for every descriptor category this renderer emits.
|
||||
ProgramFactory::UpdateAfterBindLimits m_updateAfterBindLimits{};
|
||||
// fillModeNonSolid gates VK_POLYGON_MODE_LINE/_POINT (glPolygonMode); independentBlend gates
|
||||
// per-draw-buffer color write masks (glColorMaski). Both are cached at device creation and
|
||||
// drive a runtime fallback when the device lacks them.
|
||||
@@ -519,6 +584,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// needs no feature). Both cached at device creation and drive a hard-fail-at-draw when absent.
|
||||
Bool m_dualSrcBlendFeatureEnabled = false;
|
||||
Bool m_primitiveTopologyListRestartFeatureEnabled = false;
|
||||
// multiViewport gates rasterizing into more than one of ARB_viewport_array's 16 viewports
|
||||
// (gl_ViewportIndex). m_maxRasterizableViewports is min(MAX_VIEWPORTS, device limit), or 1
|
||||
// when the feature is off, and is the viewportCount a gl_ViewportIndex-writing pipeline
|
||||
// declares - it is NOT what GL_MAX_VIEWPORTS reports, which is the frontend state width.
|
||||
Bool m_multiViewportFeatureEnabled = false;
|
||||
Uint32 m_maxRasterizableViewports = 1;
|
||||
// Union of shader stages sampled-read barriers may name; built at device creation
|
||||
// because geometry/tessellation stage bits are invalid in a barrier when their
|
||||
// feature is off (VUID-vkCmdPipelineBarrier-srcStageMask-04090/-04091), and
|
||||
@@ -742,9 +813,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 m_lastLodProgramVersion = 0;
|
||||
Uint64 m_lastLodBindGeneration = 0;
|
||||
Uint64 m_lastLodParamsSum = 0;
|
||||
// Sampling-resolution generation at probe time. The probe reads the effective
|
||||
// sampler's filters/aniso/LOD range, whose setters bump only this counter -
|
||||
// the params-version sum above never moves for them.
|
||||
Uint64 m_lastLodSamplingGeneration = 0;
|
||||
ProgramFactory::CompileOptionFlags m_lastLodBaseFlags = {};
|
||||
ProgramFactory::CompileOptionFlags m_lastLodResultFlags = {};
|
||||
|
||||
// Does the current program's vertex stage declare the BaseVertex builtin? A property
|
||||
// of the program's SPIR-V, so (lifetime id, backend-state version) is the whole key.
|
||||
//
|
||||
// Memoized rather than re-asked because asking means resolving the UN-zeroed program
|
||||
// variant, and a program that only ever draws non-indexed would then compile a variant
|
||||
// no draw uses AND re-stamp its use every draw, so the idle sweep could never retire
|
||||
// it. With the memo the answer is known before the first lookup and only the variant
|
||||
// the draw actually needs is resolved.
|
||||
Bool m_lastBaseVertexQueryValid = false;
|
||||
Uint64 m_lastBaseVertexProgramLifetimeId = 0;
|
||||
Uint32 m_lastBaseVertexProgramVersion = 0;
|
||||
Bool m_lastBaseVertexReads = false;
|
||||
|
||||
// Snapshot behind TrySetupDrawFastPath. Values only: the program and
|
||||
// render-pass caches are open-addressing maps whose entries move on
|
||||
// insert, so no pointers into them are cached; the pipeline handle is
|
||||
@@ -766,6 +854,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 vaoLifetimeId = 0;
|
||||
Uint32 vaoConfigVersion = 0;
|
||||
const void* drawFbo = nullptr;
|
||||
// Never-reused lifetime id beside the raw pointer + Uint16 version: a
|
||||
// deleted FBO recycled at the same address with the same fresh version
|
||||
// count would otherwise compare equal (same ABA as the render-pass
|
||||
// manager's fast-path memo).
|
||||
Uint64 drawFboLifetimeId = 0;
|
||||
Uint16 fboVersion = 0;
|
||||
Bool drawFboIsDefault = false;
|
||||
Uint renderStateVersion = 0;
|
||||
@@ -789,6 +882,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// re-resolve just the pipeline against the active pass; a change that
|
||||
// flips it must fall back to the full path's pass selection.
|
||||
Bool drawUsesDepthStencil = false;
|
||||
// The snapshotting draw's pipeline viewportCount. A pure function of the PROGRAM
|
||||
// (writesViewportIndexBuiltin) and of a device feature fixed at renderer init, both
|
||||
// of which the programLifetimeId/programVersion guards above already pin - carried
|
||||
// here so the fast path does not re-fetch the program object to re-derive it.
|
||||
Uint32 viewportCount = 1;
|
||||
IntVec2 renderPassExtent = {0, 0};
|
||||
// colorAttachmentCount of the snapshotting draw's render pass: the
|
||||
// pipeline-state hash input, so the fast path can refresh that hash and
|
||||
@@ -856,6 +954,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// already sampleable.
|
||||
Vector<VkTextureManager::TextureResource*> m_sampledResourcesScratch;
|
||||
Vector<MG_State::GLState::ITextureObject*> m_storageImageTexturesScratch;
|
||||
Vector<UniformManager::SamplerImageFeedbackBinding> m_samplerImageFeedbackScratch;
|
||||
Vector<UniformManager::SamplerBindingOverride> m_samplerImageBindingOverridesScratch;
|
||||
Vector<VkBuffer> m_vertexBuffersScratch;
|
||||
Vector<VkDeviceSize> m_vertexOffsetsScratch;
|
||||
Vector<VkVertexInputAttributeDescription> m_patchedAttributesScratch;
|
||||
@@ -982,6 +1082,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkBuffer indexVkBuffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize indexSliceOffset = 0;
|
||||
Uint64 indexFrameSerial = 0;
|
||||
// The EBO carried a host map when the slice was recorded - the mirror of
|
||||
// anyBufferMapped on the vertex half. A shadow-backed (non-adopted)
|
||||
// persistent map mutates its shadow with no API call and no epoch bump, so
|
||||
// the one-compare rescue must decline and re-run the acquire, whose
|
||||
// SyncPersistentMappedRange is the push-down. A map taken AFTER the record
|
||||
// is already covered: AcquirePersistentMap bumps the slice epoch for the
|
||||
// request itself, adopted or declined.
|
||||
Bool indexBufferMapped = false;
|
||||
|
||||
// Bound per draw (first bindingCount elements).
|
||||
VkBuffer vkBuffers[kMaxBindings] = {};
|
||||
@@ -1075,11 +1183,34 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
FrameContext::FrameData& frame,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
// Vulkan forbids a sampled descriptor and writable storage descriptor from naming the
|
||||
// same image subresource in one shader operation. Snapshot only the sampler side; the
|
||||
// storage descriptor continues to name the application texture.
|
||||
Bool PrepareSamplerImageFeedbackSnapshots(
|
||||
FrameContext::FrameData& frame,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
VkPipelineStageFlags consumerShaderStageMask);
|
||||
|
||||
// The per-draw dynamic-state tail (viewport, scissor, blend constants, depth
|
||||
// bias, line width, stencil), gated behind one render-state-parameters-version
|
||||
// compare per command buffer - see the gate fields in DynamicStateShadow.
|
||||
void ApplyDynamicDrawStateTail(FrameContext::FrameData& frame, const IntVec2& extent, Bool isDefaultFbo);
|
||||
// viewportCount is the bound pipeline's declared viewport count: 1 for every program that
|
||||
// does not write gl_ViewportIndex (the memoized fast path), otherwise the renderer's
|
||||
// rasterizable viewport count, which takes the unmemoized array path.
|
||||
void ApplyDynamicDrawStateTail(FrameContext::FrameData& frame, const IntVec2& extent, Bool isDefaultFbo,
|
||||
Uint32 viewportCount = 1);
|
||||
void ApplyMultiViewportDynamicState(VkCommandBuffer commandBuffer, Uint32 viewportCount, const IntVec2& extent,
|
||||
VkSurfaceTransformFlagBitsKHR preTransform, Bool isDefaultFbo);
|
||||
VkRect2D ComputeGLScissorRect(Uint32 index, const IntVec2& extent,
|
||||
VkSurfaceTransformFlagBitsKHR preTransform, Bool isDefaultFbo) const;
|
||||
// How many viewports a draw with this program rasterizes into: 1 unless the program
|
||||
// assigns gl_ViewportIndex AND the device enabled multiViewport. Both the pipeline's
|
||||
// baked viewportCount and the dynamic arrays come from this one answer, so they cannot
|
||||
// disagree.
|
||||
Uint32 ResolveDrawViewportCount(Bool programWritesViewportIndex) const {
|
||||
return programWritesViewportIndex && m_multiViewportFeatureEnabled ? m_maxRasterizableViewports : 1u;
|
||||
}
|
||||
|
||||
Bool UploadAndBindVertexBuffers(VkCommandBuffer commandBuffer, const MG_State::GLState::VertexArrayObject& vao,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
@@ -1120,6 +1251,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool MaterializePendingClearForRenderbuffer(
|
||||
VkCommandBuffer commandBuffer,
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
// The default framebuffer's twin of the two above. It cannot go through
|
||||
// MaterializePendingClearForTexture: the default FBO's colour attachment is a
|
||||
// placeholder texture object, and syncing THAT would clear a texture image nobody
|
||||
// presents instead of the acquired swapchain image.
|
||||
Bool MaterializePendingClearForDefaultFramebuffer(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::FramebufferObject& fbo,
|
||||
FramebufferAttachmentType attachmentType);
|
||||
// Its depth/stencil half: a different image (the swapchain's depth/stencil twin), a
|
||||
// different clear command and per-aspect masking.
|
||||
Bool MaterializePendingDepthStencilClearForDefaultFramebuffer(
|
||||
VkCommandBuffer commandBuffer, const MG_State::GLState::FramebufferAttachmentObject& attachment,
|
||||
const ClearAttachmentPayload& payload);
|
||||
VkPipeline GetOrCreateBlitPipeline(const RenderPassEntry& renderPassEntry);
|
||||
Bool GenerateDepthMipmapWithShader(FrameContext::FrameData& frame,
|
||||
MG_State::GLState::ITextureObject& texture,
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/SubgroupSupportPolicy.h
|
||||
// Copyright (c) 2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <Config.h>
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// The single decision point for how DirectVulkan implements GL_KHR_shader_subgroup,
|
||||
// shared by capability advertisement (BackendObject) and module lowering
|
||||
// (VulkanRenderer / ProgramFactory) so the two can never disagree.
|
||||
//
|
||||
// Native subgroups are the implementation whenever the device has them, whatever
|
||||
// their width - subgroup operations execute on the hardware paths they were made
|
||||
// for. Module-level repairs keep the GL contract intact around them:
|
||||
// - FixIterationRPSubgroupScratchPass patches the one known pack bug: iterationRP's
|
||||
// prefixSumCache[32], under-declared for sub-16-lane devices (8-lane lavapipe);
|
||||
// - FixIterationRPBarrierPass repairs Program 203's race between two reductions
|
||||
// reusing that scratch, when explicitly enabled;
|
||||
// - DeriveNumSubgroupsPass replaces the one builtin drivers get wrong
|
||||
// (gl_NumSubgroups) with the value the rest of the topology implies.
|
||||
// The 32-lane shared-memory emulation (EmulateSubgroupsPass) is a LAST RESORT for
|
||||
// devices with no subgroup support at all, and only when the user opts in with
|
||||
// MOBILEGL_MAGMA_EMULATE_SUBGROUP=1; it never replaces available native operations.
|
||||
|
||||
inline constexpr Uint32 kEmulatedSubgroupSize = 32u;
|
||||
inline constexpr Uint32 kEmulatedSubgroupStages = GL_COMPUTE_SHADER_BIT;
|
||||
inline constexpr Uint32 kEmulatedSubgroupFeatures =
|
||||
GL_SUBGROUP_FEATURE_BASIC_BIT_KHR | GL_SUBGROUP_FEATURE_VOTE_BIT_KHR |
|
||||
GL_SUBGROUP_FEATURE_ARITHMETIC_BIT_KHR | GL_SUBGROUP_FEATURE_BALLOT_BIT_KHR |
|
||||
GL_SUBGROUP_FEATURE_SHUFFLE_BIT_KHR | GL_SUBGROUP_FEATURE_SHUFFLE_RELATIVE_BIT_KHR |
|
||||
GL_SUBGROUP_FEATURE_CLUSTERED_BIT_KHR | GL_SUBGROUP_FEATURE_QUAD_BIT_KHR;
|
||||
|
||||
inline Bool ShouldEmulateSubgroups(const Bool nativeSubgroupSupported) {
|
||||
return MG_Config::Features.MagmaEmulateSubgroup && !nativeSubgroupSupported &&
|
||||
!MG_Config::Features.DisableSubgroup;
|
||||
}
|
||||
|
||||
inline Bool ShouldFixIterationRPSubgroupScratch() {
|
||||
// Auto is ON: the patch is fingerprint-gated to iterationRP's reduction and
|
||||
// grows one under-declared array; every other module passes through untouched.
|
||||
return MG_Config::Features.FixIterationRPSubgroupScratch !=
|
||||
MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
|
||||
inline Bool ShouldFixIterationRPBarrier() {
|
||||
return MG_Config::Features.IterationRPFixBarrier;
|
||||
}
|
||||
|
||||
inline Bool ShouldDeriveNumSubgroups() {
|
||||
// Auto is ON: gl_NumSubgroups must agree with the gl_SubgroupID range for the GL
|
||||
// contract to hold, and the derived ceil() value is the one the renderer can pin
|
||||
// with REQUIRE_FULL_SUBGROUPS - the driver builtin is the value with no
|
||||
// cross-driver guarantee (Adreno returns 1 for an 8-subgroup dispatch).
|
||||
return MG_Config::Features.DeriveNumSubgroups != MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -74,6 +74,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// The context line (__VA_ARGS__ = its own format string + args) must be a SEPARATE log
|
||||
// call: appending its format to the base format while its arguments precede the base
|
||||
// arguments makes every conversion read the wrong slot (a %s pulling an int crashes).
|
||||
//
|
||||
// MGLOG_F and deliberately NOT latched. VK_VERIFY is the invariant-check macro: a Vulkan call
|
||||
// MobileGL believes it has already made legal came back non-success, which is a
|
||||
// should-never-happen state, not an expected failure mode a user hits. Those fast-fail loudly
|
||||
// and keep saying so - the log-quietness rules that latch W/E cover expected failures (driver
|
||||
// capability gaps, app misuse), not broken internal invariants. MOBILEGL_ASSERT below traps in
|
||||
// a DEBUG build; MGLOG_F is what makes the same condition visible in an INFO test run, where
|
||||
// the assert is compiled out by contract.
|
||||
//
|
||||
// A soft, recoverable failure must therefore NOT be routed through VK_VERIFY. Check the
|
||||
// VkResult directly and report it with MGLOG_E_ONCE - see VkTextureManager::SyncTextureResource,
|
||||
// where a driver legitimately refuses an image the format pre-check accepted.
|
||||
#define VK_VERIFY(expr, ...) \
|
||||
do { \
|
||||
VkResult _vk_verify_result = (expr); \
|
||||
|
||||
@@ -42,4 +42,7 @@ set_tests_properties(SanityBench PROPERTIES LABELS benchmark)
|
||||
|
||||
add_subdirectory(Program)
|
||||
add_subdirectory(Buffer)
|
||||
add_subdirectory(Driver)
|
||||
add_subdirectory(Driver)
|
||||
add_subdirectory(Container)
|
||||
add_subdirectory(ShaderCache)
|
||||
add_subdirectory(Transpile)
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
cmake_minimum_required(VERSION 3.24)
|
||||
|
||||
add_executable(
|
||||
UnorderedMapBench
|
||||
UnorderedMapBench.cpp
|
||||
)
|
||||
|
||||
target_include_directories(UnorderedMapBench PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
UnorderedMapBench PRIVATE
|
||||
benchmark::benchmark
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
add_test(NAME UnorderedMapBench COMMAND UnorderedMapBench --benchmark_counters_tabular=true)
|
||||
set_tests_properties(UnorderedMapBench PROPERTIES LABELS benchmark)
|
||||
@@ -0,0 +1,248 @@
|
||||
// MobileGL - MobileGL/MG_Benchmark/Container/UnorderedMapBench.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The standing performance observatory for MobileGL::UnorderedMap.
|
||||
//
|
||||
// This benchmarks the ALIAS, never a concrete table, so whatever UnorderedMap
|
||||
// names today is what gets measured - swap the container in MG_Util/Types.h and
|
||||
// re-run this same binary to get a directly comparable set of numbers. That is
|
||||
// the point of it: the container sits on per-draw paths, so a change to it needs
|
||||
// evidence, and the evidence should be produced the same way every time.
|
||||
//
|
||||
// The workloads are the shapes the tree actually exercises, not generic hash-map
|
||||
// microbenchmarks. Four key shapes, because they stress a hash function very
|
||||
// differently:
|
||||
// * SEQUENTIAL dense small integers - GL object names from the index generator
|
||||
// (buffer/texture/framebuffer/sampler registries).
|
||||
// * POINTER real heap addresses - StateBackendObjectRegistry keys on
|
||||
// StateObject*. These are aligned, so their low bits are the
|
||||
// least random part of the key; a table that indexes on raw low
|
||||
// bits clusters badly here and one that mixes first does not.
|
||||
// Taken from the real allocator rather than a synthetic stride,
|
||||
// which would flatter whichever table mixes its bits.
|
||||
// * DIGEST already well-mixed 64-bit values - the XXH64 pipeline,
|
||||
// vertex-input-state and program memos.
|
||||
// * NAME short strings - uniform/attribute name to location maps.
|
||||
//
|
||||
// Sizes sweep from 8 upward because the per-draw memos are usually SMALL; a table
|
||||
// that only wins at 4096 entries has not won anything that matters here.
|
||||
//
|
||||
// Run: build-linux/MobileGL/MG_Benchmark/Container/UnorderedMapBench
|
||||
// or: ctest -R UnorderedMapBench (label: benchmark)
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include "MG_Util/Types.h"
|
||||
|
||||
using namespace MobileGL;
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr Int64 kMinSize = 8;
|
||||
constexpr Int64 kMaxSize = 4096;
|
||||
|
||||
// Keep the real allocations alive for the whole process: the POINTER shape is
|
||||
// only honest if the keys are addresses the allocator actually handed out, and
|
||||
// they have to stay unique (a freed address can be handed out twice).
|
||||
std::vector<std::unique_ptr<char[]>>& PointerKeyStorage() {
|
||||
static std::vector<std::unique_ptr<char[]>> storage;
|
||||
return storage;
|
||||
}
|
||||
|
||||
Vector<Uint64> SequentialKeys(SizeT n) {
|
||||
Vector<Uint64> keys;
|
||||
keys.reserve(n);
|
||||
for (SizeT i = 0; i < n; ++i) keys.push_back(static_cast<Uint64>(i) + 1);
|
||||
return keys;
|
||||
}
|
||||
|
||||
Vector<Uint64> PointerKeys(SizeT n) {
|
||||
auto& storage = PointerKeyStorage();
|
||||
Vector<Uint64> keys;
|
||||
keys.reserve(n);
|
||||
std::mt19937_64 rng(0xBEEF);
|
||||
std::vector<std::unique_ptr<char[]>> churn;
|
||||
for (SizeT i = 0; i < n; ++i) {
|
||||
// State objects are not all one size, and the allocator sees other
|
||||
// traffic between them - a single uniform stride is not what this
|
||||
// registry ever sees.
|
||||
const SizeT sz = 96 + (rng() % 192);
|
||||
auto p = std::make_unique<char[]>(sz);
|
||||
keys.push_back(reinterpret_cast<Uint64>(p.get()));
|
||||
storage.push_back(std::move(p));
|
||||
if ((rng() & 3) == 0) churn.push_back(std::make_unique<char[]>(32 + (rng() % 128)));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
Vector<Uint64> DigestKeys(SizeT n) {
|
||||
Vector<Uint64> keys;
|
||||
keys.reserve(n);
|
||||
std::mt19937_64 rng(0xC0FFEE);
|
||||
for (SizeT i = 0; i < n; ++i) keys.push_back(rng());
|
||||
return keys;
|
||||
}
|
||||
|
||||
Vector<String> NameKeys(SizeT n) {
|
||||
static const char* kPrefixes[] = {"u_", "a_", "mc_", "iris_", "gl_", "v_"};
|
||||
Vector<String> keys;
|
||||
keys.reserve(n);
|
||||
for (SizeT i = 0; i < n; ++i) {
|
||||
keys.push_back(String(kPrefixes[i % 6]) + "Uniform" + std::to_string(i) + "_xyz");
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
// Key sets are built once per size and shared: generating them inside the timed
|
||||
// loop would measure the generator (and, for POINTER, the allocator) instead of
|
||||
// the table.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
const KeyVec& CachedKeys(SizeT n) {
|
||||
static UnorderedMap<SizeT, KeyVec> cache;
|
||||
auto it = cache.find(n);
|
||||
if (it != cache.end()) return it->second;
|
||||
return cache.emplace(n, Make(n)).first->second;
|
||||
}
|
||||
|
||||
template <typename Key>
|
||||
UnorderedMap<Key, Uint64> Populated(const Vector<Key>& keys) {
|
||||
UnorderedMap<Key, Uint64> map;
|
||||
for (SizeT i = 0; i < keys.size(); ++i) map[keys[i]] = i;
|
||||
return map;
|
||||
}
|
||||
|
||||
// ---- the workloads ----------------------------------------------------
|
||||
|
||||
// The dominant per-draw operation by a wide margin: a populated cache that is
|
||||
// read far more often than it is written.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void LookupHit(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
auto map = Populated(keys);
|
||||
for (auto _ : state) {
|
||||
for (const auto& k : keys) {
|
||||
auto it = map.find(k);
|
||||
benchmark::DoNotOptimize(it->second);
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// "Is this resource cached yet?" answered NO - the probe length on a miss is a
|
||||
// different cost from a hit, and resource caches ask this constantly.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void LookupMiss(benchmark::State& state) {
|
||||
const SizeT n = static_cast<SizeT>(state.range(0));
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(n);
|
||||
auto map = Populated(keys);
|
||||
const KeyVec absent = Make(n); // same shape, never inserted
|
||||
for (auto _ : state) {
|
||||
for (const auto& k : absent) {
|
||||
benchmark::DoNotOptimize(map.find(k) != map.end());
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(absent.size()));
|
||||
}
|
||||
|
||||
// Building a cache from empty, rehashes included.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void InsertGrow(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
UnorderedMap<typename KeyVec::value_type, Uint64> map;
|
||||
for (SizeT i = 0; i < keys.size(); ++i) map[keys[i]] = i;
|
||||
benchmark::DoNotOptimize(map.size());
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// Cache eviction and refill: erase half by key, put them back. This is the
|
||||
// aged-out-entry sweep the pipeline and vertex-input caches do.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void EraseChurn(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
state.PauseTiming();
|
||||
auto map = Populated(keys);
|
||||
state.ResumeTiming();
|
||||
for (SizeT i = 0; i < keys.size(); i += 2) benchmark::DoNotOptimize(map.erase(keys[i]));
|
||||
for (SizeT i = 0; i < keys.size(); i += 2) map[keys[i]] = i;
|
||||
benchmark::DoNotOptimize(map.size());
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// Mass eviction: erase-while-iterating across the whole table. This is the loop
|
||||
// shape that a container's erase()-return contract can get wrong, and the one
|
||||
// that fed garbage handles to vkDestroyPipeline when it was wrong before.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void EraseSweep(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
state.PauseTiming();
|
||||
auto map = Populated(keys);
|
||||
state.ResumeTiming();
|
||||
for (auto it = map.begin(); it != map.end();) it = map.erase(it);
|
||||
benchmark::DoNotOptimize(map.size());
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// Whole-table walks: the per-frame sweeps that age entries out, and the
|
||||
// teardown loops that destroy every Vulkan object a cache owns.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void Iterate(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
auto map = Populated(keys);
|
||||
for (auto _ : state) {
|
||||
Uint64 acc = 0;
|
||||
for (const auto& entry : map) acc += entry.second;
|
||||
benchmark::DoNotOptimize(acc);
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
#define MGL_MAP_BENCH(WORKLOAD, SHAPE, VEC, MAKER) \
|
||||
BENCHMARK_TEMPLATE(WORKLOAD, VEC, MAKER) \
|
||||
->Name(#WORKLOAD "/" #SHAPE) \
|
||||
->RangeMultiplier(8) \
|
||||
->Range(kMinSize, kMaxSize)
|
||||
|
||||
MGL_MAP_BENCH(LookupHit, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(LookupHit, pointer, Vector<Uint64>, PointerKeys);
|
||||
MGL_MAP_BENCH(LookupHit, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(LookupHit, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(LookupMiss, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(LookupMiss, pointer, Vector<Uint64>, PointerKeys);
|
||||
MGL_MAP_BENCH(LookupMiss, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(LookupMiss, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(InsertGrow, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(InsertGrow, pointer, Vector<Uint64>, PointerKeys);
|
||||
MGL_MAP_BENCH(InsertGrow, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(InsertGrow, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(EraseChurn, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(EraseChurn, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(EraseChurn, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(EraseSweep, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(EraseSweep, digest, Vector<Uint64>, DigestKeys);
|
||||
|
||||
MGL_MAP_BENCH(Iterate, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(Iterate, digest, Vector<Uint64>, DigestKeys);
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,21 @@
|
||||
cmake_minimum_required(VERSION 3.24)
|
||||
|
||||
add_executable(
|
||||
TranslationCacheBench
|
||||
TranslationCacheBench.cpp
|
||||
)
|
||||
|
||||
target_include_directories(TranslationCacheBench PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
${MGL_ROOT}/3rdparty/SPIRV-Reflect
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
TranslationCacheBench PRIVATE
|
||||
benchmark::benchmark
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
add_test(NAME TranslationCacheBench COMMAND TranslationCacheBench --benchmark_counters_tabular=true)
|
||||
set_tests_properties(TranslationCacheBench PROPERTIES LABELS benchmark)
|
||||
@@ -0,0 +1,457 @@
|
||||
// MobileGL - MobileGL/MG_Benchmark/ShaderCache/TranslationCacheBench.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
// What the two-level shader translation memo is worth, measured on the workload that
|
||||
// motivated it: the KHR-GL33.texture_swizzle.smoke_* shape, where one case builds 2592
|
||||
// programs out of a handful of distinct sources.
|
||||
//
|
||||
// Four pairs of cases, each Off/On:
|
||||
//
|
||||
// ProgramLink - the whole glCompileShader + glLinkProgram path for one program, with
|
||||
// FRESH SHADER OBJECTS every iteration. This is the CTS shape exactly,
|
||||
// and it is the headline case now. It used to be the PESSIMISTIC one:
|
||||
// a hit still paid for both glslang parses, because the parse happens
|
||||
// at glCompileShader - a different entry point from the one L1
|
||||
// memoizes - and fresh shader objects meant ShaderCompileAdoptionMap
|
||||
// could not hand the earlier parse over either. L1c is what closed
|
||||
// that: the compile half of the memo recognises each stage's source
|
||||
// and publishes its verdict without parsing, so on a hit this case now
|
||||
// constructs no glslang object at all.
|
||||
//
|
||||
// SharedShaderLink - the same program population with the shader objects KEPT ALIVE, so
|
||||
// the parses happen once outside the measured loop whatever the cache
|
||||
// does. That makes it the CONTROL for L1c rather than a target: its
|
||||
// numbers should not move, and if they do, L1c has added cost to a
|
||||
// path it was supposed to leave alone.
|
||||
//
|
||||
// DeferredParseLink - the shape where L1c could LOSE: a constant vertex source (which
|
||||
// hits L1c and therefore skips its parse) against a fresh fragment
|
||||
// source every iteration (which makes the PROGRAM key miss, so the
|
||||
// skipped parse has to happen inside the link after all). Same parse
|
||||
// count either way, so the pair should land within noise; see its own
|
||||
// header below.
|
||||
//
|
||||
// EsslTranspile - the DirectGLES backend segment: the SPIR-V pass chain plus
|
||||
// SPIRV-Cross. Runs the driver-INDEPENDENT half of the real chain (the
|
||||
// passes SyncToBackend runs unconditionally, plus the two stage-gated
|
||||
// ones a fragment module reaches) so the miss path costs what
|
||||
// production costs; the capability-gated passes need a live ES driver
|
||||
// and are not reachable from a benchmark process.
|
||||
//
|
||||
// Every On case runs with a warm cache: the first iteration misses and every one after it
|
||||
// hits, which is exactly the steady state of a 2592-program smoke case.
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include <string>
|
||||
|
||||
#include "Config.h"
|
||||
#include "Includes.h"
|
||||
#include "Init.h"
|
||||
#include "MG_Impl/GLImpl/Program/GL_Program.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include "MG_State/GLState/ProgramState/ProgramTranslationCache.h"
|
||||
#include "MG_Util/ShaderTranspiler/ShaderCompiler.h"
|
||||
#include "MG_Util/ShaderTranspiler/SpvcSession.h"
|
||||
#include "MG_Util/ShaderTranspiler/TranslationCache.h"
|
||||
#include "MG_Util/ShaderTranspiler/Types.h"
|
||||
|
||||
using namespace MobileGL;
|
||||
using namespace MobileGL::MG_Util::ShaderTranspiler;
|
||||
|
||||
namespace {
|
||||
const char* kVertexSource = R"(#version 460
|
||||
layout(location = 0) in vec3 aPos;
|
||||
out vec3 vPos;
|
||||
out vec2 vUv;
|
||||
void main() {
|
||||
vPos = aPos;
|
||||
vUv = aPos.xy * 0.5 + 0.5;
|
||||
gl_Position = vec4(aPos, 1.0);
|
||||
}
|
||||
)";
|
||||
|
||||
// Shaped after gl3cTextureSwizzleTests.cpp's template: a sampler of one type, one
|
||||
// TEXTURE_ACCESS, one CHANNEL, and an output whose BASIC_TYPE is the only thing that
|
||||
// varies within a case. Padded with enough real arithmetic that the translation chain
|
||||
// is doing work rather than measuring fixed overheads.
|
||||
// `padLines` = 0 is the honest CTS size: gl3cTextureSwizzleTests' smoke template is a
|
||||
// handful of lines, and that is the workload the memo exists for. The padded variant is
|
||||
// kept alongside it because a shaderpack stage is orders of magnitude bigger, and the
|
||||
// two bracket the ratio the cache is worth in practice.
|
||||
String SwizzleLikeFragment(const String& prefix, const int padLines) {
|
||||
String source = "#version 460\n";
|
||||
source += "in vec3 vPos;\n";
|
||||
source += "in vec2 vUv;\n";
|
||||
source += "layout(location = 0) out " + prefix + "vec4 fragColor;\n";
|
||||
source += "uniform sampler2D uTex;\n";
|
||||
source += "uniform vec4 uTint;\n";
|
||||
source += "uniform mat4 uModel;\n";
|
||||
source += "uniform float uArr[8];\n";
|
||||
source += "void main() {\n";
|
||||
source += " vec4 s = texture(uTex, vUv);\n";
|
||||
source += " float acc = s.r;\n";
|
||||
for (int i = 0; i < padLines; ++i) {
|
||||
source += " acc = acc * 1.0001 + sin(acc + " + std::to_string(i) + ".0) * cos(acc);\n";
|
||||
}
|
||||
source += " for (int i = 0; i < 8; ++i) acc += uArr[i];\n";
|
||||
source += " vec4 p = uModel * vec4(vPos, 1.0);\n";
|
||||
source += " fragColor = " + prefix + "vec4((s + uTint) * acc + p);\n";
|
||||
source += "}\n";
|
||||
return source;
|
||||
}
|
||||
|
||||
class CacheModeScope {
|
||||
public:
|
||||
explicit CacheModeScope(const Bool enabled)
|
||||
: m_saved(MG_Config::Features.ShaderTranslationCache) {
|
||||
MG_Config::Features.ShaderTranslationCache =
|
||||
enabled ? MG_Config::QuirkOverride::ForceOn : MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
~CacheModeScope() { MG_Config::Features.ShaderTranslationCache = m_saved; }
|
||||
|
||||
private:
|
||||
const MG_Config::QuirkOverride m_saved;
|
||||
};
|
||||
|
||||
class SyncCompileScope {
|
||||
public:
|
||||
SyncCompileScope() : m_saved(MG_Config::Features.AsyncShaderCompile) {
|
||||
MG_Config::Features.AsyncShaderCompile = MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
~SyncCompileScope() { MG_Config::Features.AsyncShaderCompile = m_saved; }
|
||||
|
||||
private:
|
||||
const MG_Config::QuirkOverride m_saved;
|
||||
};
|
||||
|
||||
// One program, built the way the CTS builds one: fresh shader objects every time.
|
||||
void LinkOneProgram(const String& vertexSource, const String& fragmentSource) {
|
||||
using namespace MG_Impl::GLImpl;
|
||||
const GLuint vs = CreateShader(GL_VERTEX_SHADER);
|
||||
const char* vsText = vertexSource.c_str();
|
||||
ShaderSource(vs, 1, &vsText, nullptr);
|
||||
CompileShader(vs);
|
||||
|
||||
const GLuint fs = CreateShader(GL_FRAGMENT_SHADER);
|
||||
const char* fsText = fragmentSource.c_str();
|
||||
ShaderSource(fs, 1, &fsText, nullptr);
|
||||
CompileShader(fs);
|
||||
|
||||
const GLuint program = CreateProgram();
|
||||
AttachShader(program, vs);
|
||||
AttachShader(program, fs);
|
||||
LinkProgram(program);
|
||||
benchmark::DoNotOptimize(program);
|
||||
|
||||
DeleteProgram(program);
|
||||
DeleteShader(vs);
|
||||
DeleteShader(fs);
|
||||
}
|
||||
|
||||
Vector<Uint32> BuildSanitizedFragmentSpirv(const String& fragmentSource) {
|
||||
ShaderAttrib attrib{.shaderType = GL_FRAGMENT_SHADER, .sourceStr = fragmentSource};
|
||||
auto shader = ShaderCompiler::CompileShader(attrib);
|
||||
if (!shader) return {};
|
||||
ProgramAttrib programAttrib{.shaders = {shader.value()}};
|
||||
auto program = ShaderCompiler::LinkProgram(programAttrib);
|
||||
if (!program) return {};
|
||||
ProgramBinaryAttrib binaryAttrib{.shaderTypes = {GL_FRAGMENT_SHADER}, .program = *program.value()};
|
||||
auto binary = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
|
||||
if (!binary || binary->empty()) return {};
|
||||
Vector<Uint32> sanitized;
|
||||
if (!ShaderCompiler::SanitizeAndOptimizeBinary(binary->front(), sanitized)) return {};
|
||||
return sanitized;
|
||||
}
|
||||
|
||||
// The driver-independent part of BackendProgramObjectImpl::TranspileSpirvToEssl, in the
|
||||
// same order. What is missing is only the capability-gated passes (viewport lowering,
|
||||
// multisample clamping, noperspective emulation, the image-format bake), which cannot
|
||||
// fire without a live ES driver to arm them.
|
||||
Bool TranspileLikeDirectGles(const Vector<Uint32>& spirv, const Uint esslVersion, String& outEssl) {
|
||||
Vector<Uint32> a;
|
||||
const Vector<Uint32>* effective = &spirv;
|
||||
if (ShaderCompiler::StripUboMemberRelaxedPrecisionForEssl(*effective, a, false) && !a.empty()) {
|
||||
effective = &a;
|
||||
}
|
||||
Vector<Uint32> b;
|
||||
if (ShaderCompiler::LowerRectImages(*effective, b, false) && !b.empty()) effective = &b;
|
||||
Vector<Uint32> c;
|
||||
if (ShaderCompiler::Lower1DArrayImagesForEssl(*effective, c, false) && !c.empty()) effective = &c;
|
||||
Vector<Uint32> d;
|
||||
if (ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(*effective, d, false) && !d.empty()) {
|
||||
effective = &d;
|
||||
}
|
||||
|
||||
SpvcSession session(*effective, SessionUsageBit::Transpile);
|
||||
spvc_compiler_options options;
|
||||
if (session.CreateOptions(&options) != SPVC_SUCCESS) return false;
|
||||
spvc_compiler_options_set_uint(options, SPVC_COMPILER_OPTION_GLSL_VERSION, esslVersion);
|
||||
spvc_compiler_options_set_bool(options, SPVC_COMPILER_OPTION_GLSL_ES, SPVC_TRUE);
|
||||
spvc_compiler_options_set_bool(options, SPVC_COMPILER_OPTION_GLSL_VULKAN_SEMANTICS, SPVC_FALSE);
|
||||
session.SetOptions(options);
|
||||
const char* result = nullptr;
|
||||
session.Compile(&result);
|
||||
if (!result) return false;
|
||||
outEssl = result;
|
||||
return true;
|
||||
}
|
||||
|
||||
EsslTranslationKeyInputs EsslInputsFor(const Vector<Uint32>& spirv) {
|
||||
EsslTranslationKeyInputs inputs;
|
||||
inputs.spirv = &spirv;
|
||||
inputs.shaderType = GL_FRAGMENT_SHADER;
|
||||
inputs.maxColorTextureSamples = 4;
|
||||
inputs.maxIntegerSamples = 1;
|
||||
inputs.maxDepthTextureSamples = 4;
|
||||
inputs.advertisedMaxSamples = 4;
|
||||
inputs.esslVersion = 320;
|
||||
return inputs;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// L1, in situ: the full glCompileShader + glLinkProgram path for a repeated program.
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// Arg(0) = the CTS smoke size; Arg(120) = a heavy stage, bracketing the ratio.
|
||||
static void BM_ProgramLink_CacheOff(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const SyncCompileScope sync;
|
||||
const CacheModeScope cache(false);
|
||||
const String vs = kVertexSource;
|
||||
const String fs = SwizzleLikeFragment("", static_cast<int>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
LinkOneProgram(vs, fs);
|
||||
}
|
||||
state.SetLabel("MOBILEGL_SHADER_CACHE=0");
|
||||
}
|
||||
BENCHMARK(BM_ProgramLink_CacheOff)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
static void BM_ProgramLink_CacheOn(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const SyncCompileScope sync;
|
||||
const CacheModeScope cache(true);
|
||||
const String vs = kVertexSource;
|
||||
const String fs = SwizzleLikeFragment("", static_cast<int>(state.range(0)));
|
||||
LinkOneProgram(vs, fs); // prime, so the measured loop is the steady state
|
||||
const TranslationCacheStats before = MG_State::GLState::GetProgramTranslationCache().Stats();
|
||||
const TranslationCacheStats parseBefore = GetShaderParseVerdictCache().Stats();
|
||||
for (auto _ : state) {
|
||||
LinkOneProgram(vs, fs);
|
||||
}
|
||||
const TranslationCacheStats stats = MG_State::GLState::GetProgramTranslationCache().Stats();
|
||||
const TranslationCacheStats parseStats = GetShaderParseVerdictCache().Stats();
|
||||
state.counters["L1_hits"] = static_cast<double>(stats.hits - before.hits);
|
||||
state.counters["L1_misses"] = static_cast<double>(stats.misses - before.misses);
|
||||
// Two stages per iteration, so a clean run shows L1c_hits == 2 * iterations and zero
|
||||
// misses: every glCompileShader in the loop skipped its parse.
|
||||
state.counters["L1c_hits"] = static_cast<double>(parseStats.hits - parseBefore.hits);
|
||||
state.counters["L1c_misses"] = static_cast<double>(parseStats.misses - parseBefore.misses);
|
||||
}
|
||||
BENCHMARK(BM_ProgramLink_CacheOn)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// L1, the shape the memo actually exists for: MANY PROGRAMS OUT OF THE SAME SHADERS.
|
||||
//
|
||||
// The pair above deletes its shader objects every iteration, which forces a fresh glslang
|
||||
// parse per iteration no matter what the link does - glCompileShader parses, and that is a
|
||||
// DIFFERENT entry point from the one L1 memoizes. It is a real workload (what an application
|
||||
// that never reuses a shader object pays) but it is the pessimistic one, and the residual it
|
||||
// leaves is the parse, not the link.
|
||||
//
|
||||
// This pair keeps the shader objects alive, so the parses happen once before the measured
|
||||
// loop and the L1 hit then skips the link, mapIO, the SPIR-V, the reflection and the routing
|
||||
// outright.
|
||||
//
|
||||
// SINCE L1c THIS IS THE CONTROL, NOT THE TARGET. Nothing inside the measured loop calls
|
||||
// glCompileShader, so L1c cannot fire here at all - which is exactly what makes the pair
|
||||
// useful: it is the shape that says whether the compile-side memo has slowed the LINK path
|
||||
// down. Its numbers should be indistinguishable from the pre-L1c ones.
|
||||
// ---------------------------------------------------------------------------------------
|
||||
namespace {
|
||||
struct SharedShaders {
|
||||
GLuint vs = 0;
|
||||
GLuint fs = 0;
|
||||
};
|
||||
|
||||
SharedShaders MakeSharedShaders(const String& vertexSource, const String& fragmentSource) {
|
||||
using namespace MG_Impl::GLImpl;
|
||||
SharedShaders shaders;
|
||||
shaders.vs = CreateShader(GL_VERTEX_SHADER);
|
||||
const char* vsText = vertexSource.c_str();
|
||||
ShaderSource(shaders.vs, 1, &vsText, nullptr);
|
||||
CompileShader(shaders.vs);
|
||||
shaders.fs = CreateShader(GL_FRAGMENT_SHADER);
|
||||
const char* fsText = fragmentSource.c_str();
|
||||
ShaderSource(shaders.fs, 1, &fsText, nullptr);
|
||||
CompileShader(shaders.fs);
|
||||
return shaders;
|
||||
}
|
||||
|
||||
void LinkFromSharedShaders(const SharedShaders& shaders) {
|
||||
using namespace MG_Impl::GLImpl;
|
||||
const GLuint program = CreateProgram();
|
||||
AttachShader(program, shaders.vs);
|
||||
AttachShader(program, shaders.fs);
|
||||
LinkProgram(program);
|
||||
benchmark::DoNotOptimize(program);
|
||||
DeleteProgram(program);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
static void BM_SharedShaderLink_CacheOff(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const SyncCompileScope sync;
|
||||
const CacheModeScope cache(false);
|
||||
const SharedShaders shaders =
|
||||
MakeSharedShaders(kVertexSource, SwizzleLikeFragment("", static_cast<int>(state.range(0))));
|
||||
for (auto _ : state) {
|
||||
LinkFromSharedShaders(shaders);
|
||||
}
|
||||
state.SetLabel("MOBILEGL_SHADER_CACHE=0");
|
||||
}
|
||||
BENCHMARK(BM_SharedShaderLink_CacheOff)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
static void BM_SharedShaderLink_CacheOn(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const SyncCompileScope sync;
|
||||
const CacheModeScope cache(true);
|
||||
const SharedShaders shaders =
|
||||
MakeSharedShaders(kVertexSource, SwizzleLikeFragment("", static_cast<int>(state.range(0))));
|
||||
LinkFromSharedShaders(shaders); // prime, so the measured loop is the steady state
|
||||
const TranslationCacheStats before = MG_State::GLState::GetProgramTranslationCache().Stats();
|
||||
for (auto _ : state) {
|
||||
LinkFromSharedShaders(shaders);
|
||||
}
|
||||
const TranslationCacheStats stats = MG_State::GLState::GetProgramTranslationCache().Stats();
|
||||
state.counters["L1_hits"] = static_cast<double>(stats.hits - before.hits);
|
||||
state.counters["L1_misses"] = static_cast<double>(stats.misses - before.misses);
|
||||
}
|
||||
BENCHMARK(BM_SharedShaderLink_CacheOn)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// L2, component: the DirectGLES SPIR-V pass chain plus SPIRV-Cross for one stage.
|
||||
// ---------------------------------------------------------------------------------------
|
||||
static void BM_EsslTranspile_CacheOff(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const Vector<Uint32> spirv =
|
||||
BuildSanitizedFragmentSpirv(SwizzleLikeFragment("", static_cast<int>(state.range(0))));
|
||||
if (spirv.empty()) {
|
||||
state.SkipWithError("could not build the fragment module");
|
||||
return;
|
||||
}
|
||||
String essl;
|
||||
for (auto _ : state) {
|
||||
if (!TranspileLikeDirectGles(spirv, 320, essl)) {
|
||||
state.SkipWithError("transpile failed");
|
||||
break;
|
||||
}
|
||||
benchmark::DoNotOptimize(essl.data());
|
||||
}
|
||||
state.SetLabel("MOBILEGL_SHADER_CACHE=0");
|
||||
}
|
||||
BENCHMARK(BM_EsslTranspile_CacheOff)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
static void BM_EsslTranspile_CacheOn(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const Vector<Uint32> spirv =
|
||||
BuildSanitizedFragmentSpirv(SwizzleLikeFragment("", static_cast<int>(state.range(0))));
|
||||
if (spirv.empty()) {
|
||||
state.SkipWithError("could not build the fragment module");
|
||||
return;
|
||||
}
|
||||
BoundedTranslationCache<EsslTranslationResult> cache("bench L2", 64, 8u << 20);
|
||||
const EsslTranslationKeyInputs inputs = EsslInputsFor(spirv);
|
||||
for (auto _ : state) {
|
||||
const TranslationCacheKey key = BuildEsslTranslationKey(inputs);
|
||||
EsslTranslationResultPtr hit = cache.Find(key);
|
||||
if (!hit) {
|
||||
auto payload = MakeShared<EsslTranslationResult>();
|
||||
if (!TranspileLikeDirectGles(spirv, inputs.esslVersion, payload->essl)) {
|
||||
state.SkipWithError("transpile failed");
|
||||
break;
|
||||
}
|
||||
cache.Insert(key, EsslTranslationResultPtr(payload), EsslTranslationResultBytes(*payload));
|
||||
hit = payload;
|
||||
}
|
||||
benchmark::DoNotOptimize(hit->essl.data());
|
||||
}
|
||||
const TranslationCacheStats stats = cache.Stats();
|
||||
state.counters["L2_hits"] = static_cast<double>(stats.hits);
|
||||
state.counters["L2_misses"] = static_cast<double>(stats.misses);
|
||||
}
|
||||
BENCHMARK(BM_EsslTranspile_CacheOn)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// L1c, the shape where it could LOSE rather than win: the DEFERRED PARSE.
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// A stage whose compile hits L1c holds no AST, so if the program-level key then MISSES, the
|
||||
// parse it skipped has to happen anyway - inside the link, via ClaimParsedShader. The parse
|
||||
// is moved, not removed, and this pair is what says whether moving it costs anything.
|
||||
//
|
||||
// The shape forces exactly that, every iteration: one CONSTANT vertex source (hits L1c after
|
||||
// the first iteration) linked against a FRESH fragment source each time (misses L1c, and
|
||||
// makes the program key miss too). So:
|
||||
//
|
||||
// cache off - two parses at glCompileShader, then the link.
|
||||
// cache on - one parse at glCompileShader (the fragment), one deferred parse inside the
|
||||
// link (the vertex), then the link.
|
||||
//
|
||||
// The parse count is identical, so these two should land within noise of each other. If the
|
||||
// On arm is materially SLOWER, L1c is charging for something - the per-compile key build and
|
||||
// hash over the full preprocessed source, or the loss of the claim-CAS reuse - and that cost
|
||||
// shows up here and nowhere else.
|
||||
//
|
||||
// The distinct fragment sources also churn both front-end levels through their FIFO caps,
|
||||
// which is the eviction behaviour a real shaderpack load produces; over a long run the
|
||||
// constant vertex entry is occasionally evicted by that churn and re-inserted, so the L1c
|
||||
// hit rate reported below is high but not exactly 1.0 per iteration.
|
||||
namespace {
|
||||
String UniqueFragmentSource(const Uint64 serial, const int padLines) {
|
||||
return SwizzleLikeFragment("", padLines) +
|
||||
"\n// unique-" + std::to_string(serial) + "\n";
|
||||
}
|
||||
} // namespace
|
||||
|
||||
static void BM_DeferredParseLink_CacheOff(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const SyncCompileScope sync;
|
||||
const CacheModeScope cache(false);
|
||||
const String vs = kVertexSource;
|
||||
Uint64 serial = 0;
|
||||
for (auto _ : state) {
|
||||
LinkOneProgram(vs, UniqueFragmentSource(serial++, static_cast<int>(state.range(0))));
|
||||
}
|
||||
state.SetLabel("MOBILEGL_SHADER_CACHE=0");
|
||||
}
|
||||
BENCHMARK(BM_DeferredParseLink_CacheOff)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
static void BM_DeferredParseLink_CacheOn(benchmark::State& state) {
|
||||
MobileGL::Initialize();
|
||||
const SyncCompileScope sync;
|
||||
const CacheModeScope cache(true);
|
||||
const String vs = kVertexSource;
|
||||
Uint64 serial = 0;
|
||||
LinkOneProgram(vs, UniqueFragmentSource(~0ull, static_cast<int>(state.range(0)))); // prime the vertex entry
|
||||
const TranslationCacheStats before = MG_State::GLState::GetProgramTranslationCache().Stats();
|
||||
const TranslationCacheStats parseBefore = GetShaderParseVerdictCache().Stats();
|
||||
for (auto _ : state) {
|
||||
LinkOneProgram(vs, UniqueFragmentSource(serial++, static_cast<int>(state.range(0))));
|
||||
}
|
||||
const TranslationCacheStats stats = MG_State::GLState::GetProgramTranslationCache().Stats();
|
||||
const TranslationCacheStats parseStats = GetShaderParseVerdictCache().Stats();
|
||||
// Expected shape: L1 all misses (every program is new), L1c one hit (vertex) and one miss
|
||||
// (fragment) per iteration.
|
||||
state.counters["L1_hits"] = static_cast<double>(stats.hits - before.hits);
|
||||
state.counters["L1_misses"] = static_cast<double>(stats.misses - before.misses);
|
||||
state.counters["L1c_hits"] = static_cast<double>(parseStats.hits - parseBefore.hits);
|
||||
state.counters["L1c_misses"] = static_cast<double>(parseStats.misses - parseBefore.misses);
|
||||
}
|
||||
BENCHMARK(BM_DeferredParseLink_CacheOn)->Arg(0)->Arg(120)->Unit(benchmark::kMicrosecond);
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,20 @@
|
||||
cmake_minimum_required(VERSION 3.24)
|
||||
|
||||
# Deliberately NOT a google-benchmark target: the interesting quantity is a per-stage
|
||||
# breakdown of one program build, which needs its own clock around sub-steps that share
|
||||
# set-up, and a plain main() keeps the output a table this can be read straight out of.
|
||||
add_executable(
|
||||
TranspileProfile
|
||||
TranspileProfile.cpp
|
||||
)
|
||||
|
||||
target_include_directories(TranspileProfile PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
${MGL_ROOT}/3rdparty/SPIRV-Reflect
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
TranspileProfile PRIVATE
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -21,7 +21,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
EGLStateContext* GetState() {
|
||||
if (!MG_State::pEGLContext) {
|
||||
MGLOG_E("pEGLContext is null. MG_State may not be initialized.");
|
||||
MGLOG_E_ONCE("pEGLContext is null. MG_State may not be initialized.");
|
||||
}
|
||||
return MG_State::pEGLContext.get();
|
||||
}
|
||||
@@ -146,7 +146,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
state->DestroySurface(dpy, surface);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
@@ -172,11 +172,11 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!backendObject->SwapEGLBuffers(dpy, draw)) {
|
||||
MGLOG_E("eglSwapBuffers failed on thread=%s dpy=%p draw=%p", CurrentThreadIdString().c_str(), dpy, draw);
|
||||
MGLOG_E_ONCE("eglSwapBuffers failed on thread=%s dpy=%p draw=%p", CurrentThreadIdString().c_str(), dpy, draw);
|
||||
state->SetError(EGL_BAD_SURFACE);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -211,7 +211,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!backendObject->InitializeEGLDisplay(dpy, major, minor)) {
|
||||
@@ -265,7 +265,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (releaseCurrentRequest) {
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
if (!backendObject->MakeEGLCurrent(dpy, draw, read, ctx)) {
|
||||
MGLOG_E("eglMakeCurrent release failed in backend thread=%s", threadId.c_str());
|
||||
MGLOG_E_ONCE("eglMakeCurrent release failed in backend thread=%s", threadId.c_str());
|
||||
state->MakeCurrent(oldDisplay, oldDraw, oldRead, oldContext);
|
||||
state->SetError(EGL_BAD_ACCESS);
|
||||
return EGL_FALSE;
|
||||
@@ -277,12 +277,12 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
state->MakeCurrent(oldDisplay, oldDraw, oldRead, oldContext);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!backendObject->MakeEGLCurrent(dpy, draw, read, ctx)) {
|
||||
MGLOG_E("eglMakeCurrent backend attach failed thread=%s dpy=%p draw=%p read=%p ctx=%p", threadId.c_str(),
|
||||
MGLOG_E_ONCE("eglMakeCurrent backend attach failed thread=%s dpy=%p draw=%p read=%p ctx=%p", threadId.c_str(),
|
||||
dpy, draw, read, ctx);
|
||||
state->SetError(EGL_BAD_ACCESS);
|
||||
state->MakeCurrent(oldDisplay, oldDraw, oldRead, oldContext);
|
||||
@@ -703,7 +703,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
state->DestroySurface(dpy, surface);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
@@ -726,7 +726,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
return EGL_FALSE;
|
||||
}
|
||||
width = std::max<EGLint>(width, 1);
|
||||
@@ -764,7 +764,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
MGLOG_D("eglGetProcAddress(%s)", name);
|
||||
void* proc = MG_Impl::GetProcAddress(name);
|
||||
if (!proc) {
|
||||
MGLOG_W("Failed to get function: %s", name);
|
||||
MGLOG_D("Failed to get function: %s", name);
|
||||
return nullptr;
|
||||
}
|
||||
return (__eglMustCastToProperFunctionPointerType)proc;
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/BufferEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/BufferEnumConverter.h>
|
||||
#include <MG_Util/Texture/PixelStoreProcessor.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
@@ -31,6 +32,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
NamedBufferData,
|
||||
NamedBufferSubData,
|
||||
CopyNamedBufferSubData,
|
||||
ClearBufferData,
|
||||
ClearBufferSubData,
|
||||
ClearNamedBufferData,
|
||||
ClearNamedBufferSubData,
|
||||
MapBufferRange,
|
||||
@@ -65,6 +68,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return "NamedBufferSubData";
|
||||
case BufferOp::CopyNamedBufferSubData:
|
||||
return "CopyNamedBufferSubData";
|
||||
case BufferOp::ClearBufferData:
|
||||
return "ClearBufferData";
|
||||
case BufferOp::ClearBufferSubData:
|
||||
return "ClearBufferSubData";
|
||||
case BufferOp::ClearNamedBufferData:
|
||||
return "ClearNamedBufferData";
|
||||
case BufferOp::ClearNamedBufferSubData:
|
||||
@@ -143,16 +150,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// The pattern is replicated verbatim, which is only the whole story while the client
|
||||
// layout already matches the internal format - the case every entry point in practice
|
||||
// uses, and the only one the conversion machinery here can express. Say so rather than
|
||||
// quietly writing a differently-sized pattern.
|
||||
const SizeT sourceSize = MG_Util::GetInputBytesPerPixel(inputFormat, pixelType);
|
||||
if (sourceSize != elementSize) {
|
||||
MGLOG_W("%s: clear pattern is %zu bytes but internalformat 0x%X stores %zu; "
|
||||
"converting between them is not implemented",
|
||||
GetBufferOpName(op), sourceSize, internalformat, elementSize);
|
||||
}
|
||||
return elementSize;
|
||||
}
|
||||
|
||||
@@ -194,27 +191,59 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearNamedBufferRange_State(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data, BufferOp op) {
|
||||
Bool BuildClearPattern(GLenum internalformat, GLenum format, GLenum type, const void* data,
|
||||
SizeT patternSize, BufferOp op, Vector<Uint8>& pattern) {
|
||||
const TextureInternalFormat internal = MG_Util::ConvertGLEnumToTextureInternalFormat(internalformat);
|
||||
const TextureInputFormat inputFormat = MG_Util::ConvertGLEnumToTextureInputFormat(format);
|
||||
const TexturePixelDataType inputType = MG_Util::ConvertGLEnumToTexturePixelDataType(type);
|
||||
|
||||
Vector<Uint8> zeroInput;
|
||||
const void* inputPixel = data;
|
||||
if (inputPixel == nullptr) {
|
||||
const SizeT inputSize = MG_Util::GetInputBytesPerPixel(inputFormat, inputType);
|
||||
if (inputSize == 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", GetBufferOpName(op),
|
||||
"format and type do not describe a source pixel."));
|
||||
return false;
|
||||
}
|
||||
zeroInput.resize(inputSize);
|
||||
inputPixel = zeroInput.data();
|
||||
}
|
||||
|
||||
if (!MG_Util::PixelStoreProcessor::ConvertOnePixelToInternal(
|
||||
internal, inputFormat, inputType, inputPixel, pattern)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", GetBufferOpName(op),
|
||||
std::format("Cannot convert one ({}, {}) pixel into internalformat 0x{:X}.",
|
||||
MG_Util::ConvertGLEnumToString(format), MG_Util::ConvertGLEnumToString(type),
|
||||
internalformat)));
|
||||
return false;
|
||||
}
|
||||
|
||||
if (data == nullptr) {
|
||||
// GL defines a null clear value as all zero bits in the destination store, while
|
||||
// retaining the format/type validation above.
|
||||
pattern.assign(patternSize, 0);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearBufferRange_State(const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data, BufferOp op) {
|
||||
const SizeT patternSize = GetClearPatternSize(internalformat, format, type, op);
|
||||
if (patternSize == 0) return;
|
||||
|
||||
auto bufferObject = GetNamedBufferObject(buffer, op);
|
||||
if (!bufferObject) return;
|
||||
if (!ValidateBufferClearRange(bufferObject, offset, size, patternSize, op)) return;
|
||||
if (size == 0) return;
|
||||
|
||||
Vector<Uint8> clearData(static_cast<SizeT>(size));
|
||||
if (data) {
|
||||
const auto* pattern = static_cast<const Uint8*>(data);
|
||||
for (SizeT at = 0; at < clearData.size(); at += patternSize) {
|
||||
Memcpy(clearData.data() + at, pattern, patternSize);
|
||||
}
|
||||
} else {
|
||||
Memset(clearData.data(), 0, clearData.size());
|
||||
}
|
||||
|
||||
bufferObject->UploadSubData({clearData.data(), clearData.size()}, static_cast<SizeT>(offset));
|
||||
Vector<Uint8> pattern;
|
||||
if (!BuildClearPattern(internalformat, format, type, data, patternSize, op, pattern)) return;
|
||||
bufferObject->FillSubData({pattern.data(), pattern.size()}, static_cast<SizeT>(offset),
|
||||
static_cast<SizeT>(size));
|
||||
}
|
||||
|
||||
auto& GetBufferBindingSlot(BufferTarget target) {
|
||||
@@ -1197,17 +1226,34 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
static_cast<SizeT>(writeOffset), static_cast<SizeT>(size));
|
||||
}
|
||||
|
||||
void ClearBufferData_State(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
auto bufferObject = GetBoundBufferObject(target, BufferOp::ClearBufferData);
|
||||
if (!bufferObject) return;
|
||||
ClearBufferRange_State(bufferObject, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
|
||||
type, data, BufferOp::ClearBufferData);
|
||||
}
|
||||
|
||||
void ClearBufferSubData_State(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data) {
|
||||
auto bufferObject = GetBoundBufferObject(target, BufferOp::ClearBufferSubData);
|
||||
if (!bufferObject) return;
|
||||
ClearBufferRange_State(bufferObject, internalformat, offset, size, format, type, data,
|
||||
BufferOp::ClearBufferSubData);
|
||||
}
|
||||
|
||||
void ClearNamedBufferData_State(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
auto bufferObject = GetNamedBufferObject(buffer, BufferOp::ClearNamedBufferData);
|
||||
if (!bufferObject) return;
|
||||
ClearNamedBufferRange_State(buffer, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
|
||||
type, data, BufferOp::ClearNamedBufferData);
|
||||
ClearBufferRange_State(bufferObject, internalformat, 0, static_cast<GLsizeiptr>(bufferObject->GetSize()), format,
|
||||
type, data, BufferOp::ClearNamedBufferData);
|
||||
}
|
||||
|
||||
void ClearNamedBufferSubData_State(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size,
|
||||
GLenum format, GLenum type, const void* data) {
|
||||
ClearNamedBufferRange_State(buffer, internalformat, offset, size, format, type, data,
|
||||
BufferOp::ClearNamedBufferSubData);
|
||||
auto bufferObject = GetNamedBufferObject(buffer, BufferOp::ClearNamedBufferSubData);
|
||||
if (!bufferObject) return;
|
||||
ClearBufferRange_State(bufferObject, internalformat, offset, size, format, type, data,
|
||||
BufferOp::ClearNamedBufferSubData);
|
||||
}
|
||||
|
||||
void* MapNamedBuffer_State(GLuint buffer, GLenum access) {
|
||||
@@ -1491,8 +1537,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// offset and size, which is also how glBindBuffersRange spells "reset this element"
|
||||
// (a NULL buffers array, or a zero entry inside one).
|
||||
static Bool ValidateBufferRangeOffsetAndSize(GLenum target, GLintptr offset, GLsizeiptr size,
|
||||
const char* funcName) {
|
||||
if (size <= 0) {
|
||||
const char* funcName, Bool hasBuffer = true) {
|
||||
if (hasBuffer && size <= 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", funcName,
|
||||
@@ -1527,16 +1573,27 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
// A transform feedback capture binding is addressed in 32-bit components, so BOTH the
|
||||
// offset and the size must be multiples of 4.
|
||||
if (target == GL_TRANSFORM_FEEDBACK_BUFFER && ((offset % 4) != 0 || (size % 4) != 0)) {
|
||||
// GL 4.6 core 6.1.1 constrains the OFFSET to a multiple of four for both
|
||||
// TRANSFORM_FEEDBACK_BUFFER and ATOMIC_COUNTER_BUFFER (the atomic-counter one has no
|
||||
// queryable alignment pname, which is why it was missing here), and the SIZE only for
|
||||
// transform feedback, whose capture is written in whole 32-bit components. Extending the
|
||||
// size rule to atomic counters as well breaks a legal bind: the conformance suite splits
|
||||
// MAX_ATOMIC_COUNTER_BUFFER_SIZE evenly across the binding points and that quotient is
|
||||
// not required to land on four.
|
||||
if ((target == GL_TRANSFORM_FEEDBACK_BUFFER || target == GL_ATOMIC_COUNTER_BUFFER) && (offset % 4) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", funcName,
|
||||
std::format("offset ({}) must be a multiple of 4 for {}.", offset,
|
||||
MG_Util::ConvertGLEnumToString(target))));
|
||||
return false;
|
||||
}
|
||||
if (target == GL_TRANSFORM_FEEDBACK_BUFFER && hasBuffer && (size % 4) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", funcName,
|
||||
std::format("offset ({}) and size ({}) must both be multiples of 4 for "
|
||||
"GL_TRANSFORM_FEEDBACK_BUFFER.",
|
||||
offset, size)));
|
||||
std::format("size ({}) must be a multiple of 4 for GL_TRANSFORM_FEEDBACK_BUFFER.", size)));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
@@ -1548,7 +1605,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
BufferTarget bufferTarget = MG_Util::ConvertGLEnumToBufferTarget(target);
|
||||
if (!BufferImpl::ValidateBufferBindingPointTarget(bufferTarget)) return;
|
||||
if (!BufferImpl::ValidateBufferBindingPointIndex(bufferTarget, index)) return;
|
||||
if (buffer != 0 && !ValidateBufferRangeOffsetAndSize(target, offset, size, __func__)) return;
|
||||
// The target's alignment rules are a property of the BINDING POINT, not of the buffer,
|
||||
// so they apply even when buffer is zero - which is exactly how
|
||||
// KHR-GL43.shader_storage_buffer_object.negative-api-bind probes the SSBO alignment
|
||||
// (glBindBufferRange(SHADER_STORAGE_BUFFER, 0, 0, alignment - 1, 0)). Only the size
|
||||
// rules need a buffer, since buffer 0 detaches the binding point and ignores size.
|
||||
if (!ValidateBufferRangeOffsetAndSize(target, offset, size, __func__, /*hasBuffer: */ buffer != 0)) return;
|
||||
if (bufferTarget == BufferTarget::TransformFeedback && MG_State::pGLContext->IsTransformFeedbackActive()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
@@ -1646,6 +1708,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
CopyNamedBufferSubData_State(readBuffer, writeBuffer, readOffset, writeOffset, size);
|
||||
}
|
||||
|
||||
void ClearBufferData(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
ClearBufferData_State(target, internalformat, format, type, data);
|
||||
}
|
||||
|
||||
void ClearBufferSubData(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
|
||||
GLenum type, const void* data) {
|
||||
ClearBufferSubData_State(target, internalformat, offset, size, format, type, data);
|
||||
}
|
||||
|
||||
void ClearNamedBufferData(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data) {
|
||||
ClearNamedBufferData_State(buffer, internalformat, format, type, data);
|
||||
}
|
||||
@@ -1732,10 +1803,30 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return BufferImpl::ValidateBufferBindingPointRange(bufferTarget, first, count, funcName);
|
||||
}
|
||||
|
||||
// ARB_multi_bind states the equivalence to a loop of single binds "except that ... buffers
|
||||
// will not be created if they do not exist": glBindBuffer instantiates a name glGenBuffers
|
||||
// merely reserved, glBindBuffers* must refuse it and raise INVALID_OPERATION instead
|
||||
// (KHR-GL44.multi_bind.errors_bind_buffers).
|
||||
//
|
||||
// Deliberately PER ELEMENT, not all-or-nothing: the equivalence the extension defines is a
|
||||
// loop, so a bad entry costs its own binding point and nothing else. Rejecting the whole
|
||||
// call instead cost multi_bind.functional_bind_buffers_base its bindings.
|
||||
static Bool IsExistingBufferForMultiBind(GLuint buffer, GLsizei index, const char* funcName) {
|
||||
if (buffer == 0 || MG_State::pGLContext->ValidateBufferObject(buffer)) return true;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", funcName,
|
||||
std::format("buffers[{}] ({}) is not the name of an existing buffer object.", index, buffer)));
|
||||
return false;
|
||||
}
|
||||
|
||||
void BindBuffersBase(GLenum target, GLuint first, GLsizei count, const GLuint* buffers) {
|
||||
if (!ValidateMultiBindBufferRange(target, first, count, __func__)) return;
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
BindBufferBase_State(target, first + i, buffers ? buffers[i] : 0);
|
||||
const GLuint buffer = buffers ? buffers[i] : 0;
|
||||
if (!IsExistingBufferForMultiBind(buffer, i, __func__)) continue;
|
||||
BindBufferBase_State(target, first + i, buffer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1749,6 +1840,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
const GLsizeiptr* sizes) {
|
||||
if (!ValidateMultiBindBufferRange(target, first, count, __func__)) return;
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
if (buffers && !IsExistingBufferForMultiBind(buffers[i], i, __func__)) continue;
|
||||
if (!buffers || buffers[i] == 0) {
|
||||
BindBufferBase_State(target, first + i, 0);
|
||||
} else {
|
||||
|
||||
@@ -27,6 +27,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void NamedBufferSubData(GLuint buffer, GLintptr offset, GLsizeiptr size, const void* data);
|
||||
void CopyNamedBufferSubData(GLuint readBuffer, GLuint writeBuffer, GLintptr readOffset, GLintptr writeOffset,
|
||||
GLsizeiptr size);
|
||||
void ClearBufferData(GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data);
|
||||
void ClearBufferSubData(GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
|
||||
GLenum type, const void* data);
|
||||
void ClearNamedBufferData(GLuint buffer, GLenum internalformat, GLenum format, GLenum type, const void* data);
|
||||
void ClearNamedBufferSubData(GLuint buffer, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format,
|
||||
GLenum type, const void* data);
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/BufferEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToStr/BufferEnumConverter.h>
|
||||
#include <MG_Util/ShaderTranspiler/Types.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl::BufferImpl {
|
||||
Bool ValidateBufferTarget(BufferTarget target) {
|
||||
@@ -67,6 +68,13 @@ namespace MobileGL::MG_Impl::GLImpl::BufferImpl {
|
||||
// binding points in GL 3.3 (no ARB_transform_feedback3).
|
||||
pointCount = std::min<SizeT>(pointCount, 4);
|
||||
}
|
||||
if (target == BufferTarget::AtomicCounter) {
|
||||
// GL_MAX_ATOMIC_COUNTER_BUFFER_BINDINGS, which is NOT the state layer's array
|
||||
// size: a counter buffer reaches a shader only as a lowered storage block, so the
|
||||
// reserved range is the ceiling, and glGetIntegerv advertises the same number.
|
||||
pointCount = std::min<SizeT>(
|
||||
pointCount, static_cast<SizeT>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTER_BUFFER_BINDINGS));
|
||||
}
|
||||
return pointCount;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
@@ -0,0 +1,271 @@
|
||||
// MobileGL - MobileGL/MG_Impl/GLImpl/Debug/GL_Debug.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "GL_Debug.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/GLState/ErrorState/Error.h>
|
||||
#include <MG_Impl/GLImpl/Query/GL_Query.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
// Must agree with what GL_Getter answers for GL_MAX_DEBUG_GROUP_STACK_DEPTH and
|
||||
// GL_MAX_DEBUG_MESSAGE_LENGTH / GL_MAX_LABEL_LENGTH; an application that sizes a buffer
|
||||
// off the query and then trips a different limit here would have no way to explain it.
|
||||
constexpr SizeT kMaxDebugGroupStackDepth = 64;
|
||||
constexpr GLsizei kMaxDebugMessageLength = 1024;
|
||||
constexpr GLsizei kMaxLabelLength = 256;
|
||||
|
||||
// The debug state KHR_debug makes per-context. Held here rather than on GLContext because
|
||||
// nothing else in MobileGL reads it, and it is keyed on the context id so a
|
||||
// destroyed-and-recreated context starts with an empty stack and no labels - which the
|
||||
// unit tests, which recreate the context between cases, depend on.
|
||||
struct DebugState {
|
||||
Uint64 contextId = 0;
|
||||
// The messages pushed with glPushDebugGroup, innermost last. The base group GL creates
|
||||
// the context with is implicit and is what makes the reported depth start at 1.
|
||||
Vector<String> groupStack;
|
||||
// Keyed by (identifier, name); see MakeObjectLabelKey.
|
||||
UnorderedMap<Uint64, String> objectLabels;
|
||||
};
|
||||
|
||||
DebugState& State() {
|
||||
static DebugState state;
|
||||
const Uint64 contextId = MG_State::pGLContext ? MG_State::pGLContext->GetTextureContextId() : 0;
|
||||
if (state.contextId != contextId) {
|
||||
state.contextId = contextId;
|
||||
state.groupStack.clear();
|
||||
state.objectLabels.clear();
|
||||
}
|
||||
return state;
|
||||
}
|
||||
|
||||
Uint64 MakeObjectLabelKey(GLenum identifier, GLuint name) {
|
||||
return (static_cast<Uint64>(identifier) << 32) | static_cast<Uint64>(name);
|
||||
}
|
||||
|
||||
void RecordDebugError(ErrorCode code, const char* caller, const String& message) {
|
||||
MG_State::pGLContext->RecordError(code, MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller, message));
|
||||
}
|
||||
|
||||
// GL 4.6 core 20.2: only an APPLICATION or THIRD_PARTY source may be injected; the rest
|
||||
// are reserved for the implementation itself.
|
||||
Bool ValidateInjectedSource(GLenum source, const char* caller) {
|
||||
if (source == GL_DEBUG_SOURCE_APPLICATION || source == GL_DEBUG_SOURCE_THIRD_PARTY) {
|
||||
return true;
|
||||
}
|
||||
RecordDebugError(ErrorCode::InvalidEnum, caller,
|
||||
std::format("source {} is not GL_DEBUG_SOURCE_APPLICATION or "
|
||||
"GL_DEBUG_SOURCE_THIRD_PARTY.",
|
||||
MG_Util::ConvertGLEnumToString(source)));
|
||||
return false;
|
||||
}
|
||||
|
||||
// A negative length means the string is NUL-terminated (GL 4.6 core 20.2), which is how
|
||||
// every one of these entry points spells "just use the whole thing".
|
||||
Bool ValidateDebugStringLength(GLsizei length, const GLchar* text, GLsizei limit, const char* caller,
|
||||
const char* what) {
|
||||
const GLsizei effective =
|
||||
length < 0 ? static_cast<GLsizei>(text != nullptr ? std::strlen(text) : 0) : length;
|
||||
if (effective < limit) {
|
||||
return true;
|
||||
}
|
||||
RecordDebugError(ErrorCode::InvalidValue, caller,
|
||||
std::format("{} length {} is not less than the {} limit of {}.", what, effective, what,
|
||||
limit));
|
||||
return false;
|
||||
}
|
||||
|
||||
String MakeDebugString(GLsizei length, const GLchar* text) {
|
||||
if (text == nullptr) return {};
|
||||
return length < 0 ? String(text) : String(text, static_cast<SizeT>(length));
|
||||
}
|
||||
|
||||
// Whether `name` currently names an object of `identifier`'s type. GL 4.6 core 20.5 makes
|
||||
// labelling something that does not exist INVALID_VALUE, and every type KHR_debug lists
|
||||
// has a frontend name check - so this is answered exactly rather than waved through.
|
||||
// GL_DISPLAY_LIST is deliberately absent: it exists only in the compatibility profile,
|
||||
// which MobileGL does not expose, so it falls to the INVALID_ENUM path below.
|
||||
Bool ValidateLabelledObject(GLenum identifier, GLuint name, Bool& outIdentifierKnown) {
|
||||
outIdentifierKnown = true;
|
||||
auto* context = MG_State::pGLContext.get();
|
||||
switch (identifier) {
|
||||
case GL_BUFFER:
|
||||
return context->ValidateBufferName(name);
|
||||
case GL_SHADER:
|
||||
return context->ValidateShaderName(name);
|
||||
case GL_PROGRAM:
|
||||
return context->ValidateProgramName(name);
|
||||
case GL_VERTEX_ARRAY:
|
||||
return context->ValidateVertexArrayName(name);
|
||||
case GL_QUERY:
|
||||
return IsQuery(name) == GL_TRUE;
|
||||
case GL_PROGRAM_PIPELINE:
|
||||
return context->ValidateProgramPipelineName(name);
|
||||
case GL_TRANSFORM_FEEDBACK:
|
||||
return context->ValidateTransformFeedbackName(name);
|
||||
case GL_SAMPLER:
|
||||
return context->ValidateSamplerName(name);
|
||||
case GL_TEXTURE:
|
||||
return context->ValidateTextureName(name);
|
||||
case GL_RENDERBUFFER:
|
||||
return context->ValidateRenderbufferName(name);
|
||||
case GL_FRAMEBUFFER:
|
||||
// Name 0 is the default framebuffer, which is a real, labellable object.
|
||||
return name == 0 || context->ValidateFramebufferName(name);
|
||||
default:
|
||||
outIdentifierKnown = false;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
GLint GetDebugGroupStackDepth() {
|
||||
// GL 4.6 core 20.6: the context is created with one group already on the stack, so the
|
||||
// reported depth is one more than the number of pushes the application has made.
|
||||
return static_cast<GLint>(State().groupStack.size()) + 1;
|
||||
}
|
||||
|
||||
void PushDebugGroup(GLenum source, GLuint id, GLsizei length, const GLchar* message) {
|
||||
static_cast<void>(id);
|
||||
if (!ValidateInjectedSource(source, __func__)) return;
|
||||
if (!ValidateDebugStringLength(length, message, kMaxDebugMessageLength, __func__, "message")) return;
|
||||
|
||||
auto& state = State();
|
||||
if (state.groupStack.size() + 1 >= kMaxDebugGroupStackDepth) {
|
||||
// Not INVALID_*: KHR_debug gives the group stack its own error code.
|
||||
RecordDebugError(ErrorCode::StackOverflow, __func__,
|
||||
std::format("the debug group stack is already {} deep, which is its maximum.",
|
||||
kMaxDebugGroupStackDepth));
|
||||
return;
|
||||
}
|
||||
state.groupStack.push_back(MakeDebugString(length, message));
|
||||
MGLOG_D("glPushDebugGroup(%s) -> depth %d", state.groupStack.back().c_str(), GetDebugGroupStackDepth());
|
||||
}
|
||||
|
||||
void PopDebugGroup() {
|
||||
auto& state = State();
|
||||
if (state.groupStack.empty()) {
|
||||
// The base group the context was created with may not be popped (GL 4.6 core 20.6).
|
||||
RecordDebugError(ErrorCode::StackUnderflow, __func__,
|
||||
"the debug group stack holds only the group the context was created with.");
|
||||
return;
|
||||
}
|
||||
MGLOG_D("glPopDebugGroup(%s)", state.groupStack.back().c_str());
|
||||
state.groupStack.pop_back();
|
||||
}
|
||||
|
||||
void DebugMessageInsert(GLenum source, GLenum type, GLuint id, GLenum severity, GLsizei length,
|
||||
const GLchar* buf) {
|
||||
static_cast<void>(id);
|
||||
if (!ValidateInjectedSource(source, __func__)) return;
|
||||
switch (type) {
|
||||
case GL_DEBUG_TYPE_ERROR:
|
||||
case GL_DEBUG_TYPE_DEPRECATED_BEHAVIOR:
|
||||
case GL_DEBUG_TYPE_UNDEFINED_BEHAVIOR:
|
||||
case GL_DEBUG_TYPE_PORTABILITY:
|
||||
case GL_DEBUG_TYPE_PERFORMANCE:
|
||||
case GL_DEBUG_TYPE_MARKER:
|
||||
case GL_DEBUG_TYPE_PUSH_GROUP:
|
||||
case GL_DEBUG_TYPE_POP_GROUP:
|
||||
case GL_DEBUG_TYPE_OTHER:
|
||||
break;
|
||||
default:
|
||||
RecordDebugError(ErrorCode::InvalidEnum, __func__,
|
||||
std::format("type {} is not a debug message type.",
|
||||
MG_Util::ConvertGLEnumToString(type)));
|
||||
return;
|
||||
}
|
||||
switch (severity) {
|
||||
case GL_DEBUG_SEVERITY_HIGH:
|
||||
case GL_DEBUG_SEVERITY_MEDIUM:
|
||||
case GL_DEBUG_SEVERITY_LOW:
|
||||
case GL_DEBUG_SEVERITY_NOTIFICATION:
|
||||
break;
|
||||
default:
|
||||
RecordDebugError(ErrorCode::InvalidEnum, __func__,
|
||||
std::format("severity {} is not a debug message severity.",
|
||||
MG_Util::ConvertGLEnumToString(severity)));
|
||||
return;
|
||||
}
|
||||
if (!ValidateDebugStringLength(length, buf, kMaxDebugMessageLength, __func__, "message")) return;
|
||||
|
||||
// No callback is ever invoked and the message log is empty by construction
|
||||
// (GL_MAX_DEBUG_LOGGED_MESSAGES is 1 and glGetDebugMessageLog returns nothing), so the
|
||||
// application-visible effect is exactly the error checking above. The text still reaches
|
||||
// MobileGL's own log, where it is worth having next to the calls it annotates - at debug
|
||||
// level, so an application that inserts a message per draw costs nothing in a release build.
|
||||
MGLOG_D("glDebugMessageInsert: %s", MakeDebugString(length, buf).c_str());
|
||||
}
|
||||
|
||||
void ObjectLabel(GLenum identifier, GLuint name, GLsizei length, const GLchar* label) {
|
||||
Bool identifierKnown = false;
|
||||
const Bool objectExists = ValidateLabelledObject(identifier, name, identifierKnown);
|
||||
if (!identifierKnown) {
|
||||
RecordDebugError(ErrorCode::InvalidEnum, __func__,
|
||||
std::format("identifier {} is not a labellable object type.",
|
||||
MG_Util::ConvertGLEnumToString(identifier)));
|
||||
return;
|
||||
}
|
||||
if (!objectExists) {
|
||||
RecordDebugError(ErrorCode::InvalidValue, __func__,
|
||||
std::format("{} {} is not the name of an existing object.",
|
||||
MG_Util::ConvertGLEnumToString(identifier), name));
|
||||
return;
|
||||
}
|
||||
if (!ValidateDebugStringLength(length, label, kMaxLabelLength, __func__, "label")) return;
|
||||
|
||||
auto& labels = State().objectLabels;
|
||||
const Uint64 key = MakeObjectLabelKey(identifier, name);
|
||||
if (label == nullptr) {
|
||||
// GL 4.6 core 20.5: a NULL label removes any label the object had.
|
||||
labels.erase(key);
|
||||
return;
|
||||
}
|
||||
labels[key] = MakeDebugString(length, label);
|
||||
}
|
||||
|
||||
void GetObjectLabel(GLenum identifier, GLuint name, GLsizei bufSize, GLsizei* length, GLchar* label) {
|
||||
if (bufSize < 0) {
|
||||
RecordDebugError(ErrorCode::InvalidValue, __func__, "bufSize must not be negative.");
|
||||
return;
|
||||
}
|
||||
Bool identifierKnown = false;
|
||||
const Bool objectExists = ValidateLabelledObject(identifier, name, identifierKnown);
|
||||
if (!identifierKnown) {
|
||||
RecordDebugError(ErrorCode::InvalidEnum, __func__,
|
||||
std::format("identifier {} is not a labellable object type.",
|
||||
MG_Util::ConvertGLEnumToString(identifier)));
|
||||
return;
|
||||
}
|
||||
if (!objectExists) {
|
||||
RecordDebugError(ErrorCode::InvalidValue, __func__,
|
||||
std::format("{} {} is not the name of an existing object.",
|
||||
MG_Util::ConvertGLEnumToString(identifier), name));
|
||||
return;
|
||||
}
|
||||
|
||||
const auto& labels = State().objectLabels;
|
||||
const auto it = labels.find(MakeObjectLabelKey(identifier, name));
|
||||
const String& text = it != labels.end() ? it->second : String{};
|
||||
// GL 4.6 core 20.5: the returned length excludes the NUL, and an unlabelled object hands
|
||||
// back an empty string with length 0 rather than an error.
|
||||
SizeT copied = 0;
|
||||
if (label != nullptr && bufSize > 0) {
|
||||
copied = std::min(text.size(), static_cast<SizeT>(bufSize) - 1);
|
||||
std::memcpy(label, text.data(), copied);
|
||||
label[copied] = '\0';
|
||||
}
|
||||
if (length != nullptr) {
|
||||
*length = static_cast<GLsizei>(copied);
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
@@ -0,0 +1,42 @@
|
||||
// MobileGL - MobileGL/MG_Impl/GLImpl/Debug/GL_Debug.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
// KHR_debug, core since GL 4.3 (GL 4.6 core 20). Applications use these to annotate a capture
|
||||
// and to name their objects; Better Clouds calls all four for exactly that.
|
||||
//
|
||||
// MobileGL implements the STATE and the ERRORS, and deliberately does not forward the calls to
|
||||
// the host driver. Two independent reasons:
|
||||
//
|
||||
// * glObjectLabel names a FRONTEND object. MobileGL's texture 5 is not the ES driver's
|
||||
// texture 5 (and under DirectVulkan it is not a driver object at all), so forwarding the
|
||||
// pair verbatim would label an unrelated object or a nonexistent one - worse than not
|
||||
// labelling.
|
||||
// * A debug GROUP is only meaningful if it brackets the commands the application issued
|
||||
// inside it. Neither backend emits its work at the moment the GL call arrives: DirectGLES
|
||||
// defers and reorders state sync and uploads around draws, and DirectVulkan is usually not
|
||||
// even recording a command buffer here. A forwarded push/pop would therefore enclose the
|
||||
// wrong commands, which is a misleading capture rather than a helpful one.
|
||||
//
|
||||
// What the application can rely on is the observable contract: the group stack depth is real
|
||||
// (GL_DEBUG_GROUP_STACK_DEPTH tracks it, and over/underflow raise the errors KHR_debug
|
||||
// specifies), and a label written with glObjectLabel comes back from glGetObjectLabel.
|
||||
void PushDebugGroup(GLenum source, GLuint id, GLsizei length, const GLchar* message);
|
||||
void PopDebugGroup();
|
||||
void DebugMessageInsert(GLenum source, GLenum type, GLuint id, GLenum severity, GLsizei length,
|
||||
const GLchar* buf);
|
||||
void ObjectLabel(GLenum identifier, GLuint name, GLsizei length, const GLchar* label);
|
||||
void GetObjectLabel(GLenum identifier, GLuint name, GLsizei bufSize, GLsizei* length, GLchar* label);
|
||||
|
||||
// Current depth of the debug group stack, for GL_DEBUG_GROUP_STACK_DEPTH. The base group the
|
||||
// context is created with counts, so this is never below 1 (GL 4.6 core 20.6).
|
||||
GLint GetDebugGroupStackDepth();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
@@ -14,8 +14,8 @@
|
||||
#include "../Getter/GL_Getter.h"
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
static Bool ValidateCurrentProgramForExecution(const char* functionName) {
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDraw();
|
||||
static Bool ValidateProgramForExecution(const SharedPtr<MG_State::GLState::ProgramObject>& currentProgram,
|
||||
const char* functionName) {
|
||||
if (!currentProgram) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
@@ -34,11 +34,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
static Bool ValidateCurrentProgramForCompute(const char* functionName) {
|
||||
if (!ValidateCurrentProgramForExecution(functionName)) return false;
|
||||
static Bool ValidateCurrentProgramForExecution(const char* functionName) {
|
||||
return ValidateProgramForExecution(MG_State::pGLContext->GetProgramForDraw(), functionName);
|
||||
}
|
||||
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDraw();
|
||||
if (currentProgram->GetShaderIndexByStage(ShaderStage::Compute) < 0) {
|
||||
// A dispatch resolves its program through the DISPATCH accessor: with a pipeline bound
|
||||
// that is the pipeline's compute stage program, not the graphics composite a draw would
|
||||
// build - which no longer contains a compute stage to find at all.
|
||||
static Bool ValidateCurrentProgramForCompute(const char* functionName) {
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDispatch();
|
||||
if (!ValidateProgramForExecution(currentProgram, functionName)) return false;
|
||||
|
||||
// Of the EXECUTABLE, not the live attach list: attaching a compute shader to an
|
||||
// already-linked graphics program does not give that program a compute stage to
|
||||
// dispatch (GL 4.6 core 7.3), and letting the dispatch through on the strength of the
|
||||
// attach hands the backend a program whose SPIR-V has no compute module in it.
|
||||
if (!currentProgram->HasLinkedShaderStage(ShaderStage::Compute)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
@@ -101,6 +112,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
const auto& program = MG_State::pGLContext->GetTransformFeedbackProgram();
|
||||
if (program != nullptr) {
|
||||
// A geometry stage writes what it emits, not what the draw assembled, and the
|
||||
// amplification factor lives in the shader. Record that this span contained such
|
||||
// a draw so the transform feedback queries keep their backend result for it.
|
||||
if (program->HasLinkedShaderStage(ShaderStage::Geometry)) {
|
||||
MG_State::pGLContext->AddTransformFeedbackGeometryCaptureDraw();
|
||||
}
|
||||
// Capacity in captured vertices = the tightest bound buffer.
|
||||
Uint64 capacityVertices = ~0ull;
|
||||
for (SizeT i = 0; i < program->GetTransformFeedbackBufferCount(); ++i) {
|
||||
@@ -120,6 +137,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
MG_State::pGLContext->AddTransformFeedbackPrimitives(primitives);
|
||||
MG_State::pGLContext->AddTransformFeedbackCapturedVertices(primitives * verticesPerPrimitive);
|
||||
// Only draws that get this far are in the written counter at all. The instanced and
|
||||
// indirect entry points never call this function, so a span that contains one is NOT
|
||||
// fully accounted, and the queries must be able to tell: they compare this counter's
|
||||
// delta against zero before standing in for the backend's own result.
|
||||
MG_State::pGLContext->AddTransformFeedbackAccountedCaptureDraw();
|
||||
}
|
||||
|
||||
// Every primitive mode a draw command accepts (GL 4.6 core table 10.1, plus
|
||||
@@ -144,11 +166,23 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// The `mode` INVALID_ENUM in isolation, so a draw entry point can raise it BEFORE any of the
|
||||
// state-dependent INVALID_OPERATIONs below. GL 4.6 core 10.4 makes a bad mode INVALID_ENUM
|
||||
// unconditionally, while "no current program" is not even a spec-listed draw error - it is
|
||||
// MobileGL's own null-dereference guard - so it must never shadow the enum check
|
||||
// (KHR-GL31.api.coverage calls glDrawArraysInstanced/glDrawElementsInstanced with mode
|
||||
// GL_POINTS-1 against a bare context and pins GL_INVALID_ENUM).
|
||||
static Bool ValidatePrimitiveModeEnum(const char* functionName, GLenum mode) {
|
||||
if (IsAcceptedPrimitiveMode(mode)) return true;
|
||||
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "mode is not an accepted primitive type."));
|
||||
return false;
|
||||
}
|
||||
|
||||
static Bool ValidatePrimitiveModeForBackend(const char* functionName, GLenum mode) {
|
||||
if (!IsAcceptedPrimitiveMode(mode)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "mode is not an accepted primitive type."));
|
||||
if (!ValidatePrimitiveModeEnum(functionName, mode)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -169,13 +203,58 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDraw();
|
||||
|
||||
// GL 4.6 core 10.1: the tessellation pipeline's only input primitive is GL_PATCHES, and
|
||||
// GL_PATCHES has no meaning without it. Both directions are INVALID_OPERATION, and
|
||||
// neither was implemented - which is two of the four sites
|
||||
// KHR-GL43.transform_feedback.api_errors_test checks with one shared message string.
|
||||
// The EVALUATION stage is what decides: a control stage cannot run without one, and a
|
||||
// program carrying only an evaluation stage still tessellates, through GL's
|
||||
// fixed-function pass-through control stage (11.2.2).
|
||||
// Asked of the LAST LINK, not the live attach list (GL 4.6 core 7.3): attaching a
|
||||
// tessellation evaluation shader to an already-linked program does not put it in the
|
||||
// executable, so reading the live list here would reject every non-GL_PATCHES draw
|
||||
// against a program that does not tessellate - and keep rejecting them, since a detach
|
||||
// is likewise deferred to the next link.
|
||||
const Bool tessellationActive = currentProgram && currentProgram->HasLinkedShaderStage(ShaderStage::TessEval);
|
||||
if (tessellationActive && mode != GL_PATCHES) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", functionName,
|
||||
"A program with a tessellation evaluation shader can only be drawn with GL_PATCHES."));
|
||||
return false;
|
||||
}
|
||||
if (!tessellationActive && mode == GL_PATCHES) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"GL_PATCHES requires an active tessellation evaluation shader."));
|
||||
return false;
|
||||
}
|
||||
|
||||
// A geometry stage only accepts the primitive types that decompose into its declared
|
||||
// input primitive (GL 4.6 core 11.3.1); anything else is INVALID_OPERATION. GL_PATCHES
|
||||
// is the tessellation pipeline's input and reaches the geometry stage already
|
||||
// converted, so it is not constrained here.
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDraw();
|
||||
const GLenum gsInput = currentProgram ? currentProgram->GetGeometryInputType() : GL_NONE;
|
||||
if (gsInput != GL_NONE && mode != GL_PATCHES) {
|
||||
//
|
||||
// "Is there a geometry stage at all" has to be asked of the STAGE, never of the input
|
||||
// primitive: GL_NONE and GL_POINTS are both 0, so a `layout(points) in` geometry shader
|
||||
// is indistinguishable from no geometry shader by its reflected input type alone. The
|
||||
// sentinel test this replaces therefore skipped the whole rule for exactly the geometry
|
||||
// shaders whose input is the most restrictive one - every mode but GL_POINTS was
|
||||
// accepted (KHR-GL43.transform_feedback.api_errors_test draws a points-in geometry
|
||||
// program with GL_LINES and requires INVALID_OPERATION).
|
||||
//
|
||||
// And it has to be asked of the LAST LINK: gsInputPrimitive is a link artifact, so
|
||||
// pairing it with the live attach list would re-point the very same 0-aliasing rather
|
||||
// than remove it. In the window after glAttachShader(GS) on a linked program the live
|
||||
// list says "geometry present" while the artifact still reads GL_NONE == GL_POINTS, and
|
||||
// the switch below would silently reject every mode but GL_POINTS.
|
||||
const Bool geometryActive = currentProgram && currentProgram->HasLinkedShaderStage(ShaderStage::Geometry);
|
||||
const GLenum gsInput = geometryActive ? currentProgram->GetGeometryInputType() : GL_NONE;
|
||||
if (geometryActive && mode != GL_PATCHES) {
|
||||
Bool compatible = false;
|
||||
switch (gsInput) {
|
||||
case GL_POINTS:
|
||||
@@ -209,13 +288,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// While transform feedback is active the draw's primitive type must match
|
||||
// the feedback primitive mode (GL 3.3 core 13.2.2). With a geometry shader
|
||||
// the constraint moves to the shader's output primitive type instead, so
|
||||
// the draw mode itself is unconstrained here. A paused span is exempt: it
|
||||
// captures nothing, so there is nothing for the mode to be incompatible with
|
||||
// (GL 4.6 core 13.2.3).
|
||||
// the draw mode itself is unconstrained here - and a TESSELLATION EVALUATION
|
||||
// stage relocates it exactly the same way (GL 4.6 core 13.2.2 names both):
|
||||
// what is captured is the tessellator's output primitive, and the draw mode
|
||||
// can only ever be GL_PATCHES. A paused span is exempt: it captures nothing,
|
||||
// so there is nothing for the mode to be incompatible with (GL 4.6 core 13.2.3).
|
||||
const auto& feedbackProgram = MG_State::pGLContext->GetTransformFeedbackProgram();
|
||||
// Both stage tests are asked of the last link, for the same reason as the two guards
|
||||
// above: what relocates the constraint is a stage the program actually RUNS, and an
|
||||
// attach that has not been linked in yet gives it none.
|
||||
const Bool feedbackModeIsProgramDriven =
|
||||
feedbackProgram && (feedbackProgram->HasLinkedShaderStage(ShaderStage::Geometry) ||
|
||||
feedbackProgram->HasLinkedShaderStage(ShaderStage::TessEval));
|
||||
if (MG_State::pGLContext->IsTransformFeedbackActive() &&
|
||||
!MG_State::pGLContext->IsTransformFeedbackPaused() &&
|
||||
!(MG_State::pGLContext->GetTransformFeedbackProgram() &&
|
||||
MG_State::pGLContext->GetTransformFeedbackProgram()->GetShaderIndexByStage(ShaderStage::Geometry) >= 0)) {
|
||||
!MG_State::pGLContext->IsTransformFeedbackPaused() && !feedbackModeIsProgramDriven) {
|
||||
const GLenum feedbackMode = MG_State::pGLContext->GetTransformFeedbackPrimitiveMode();
|
||||
Bool compatible = false;
|
||||
switch (feedbackMode) {
|
||||
@@ -296,10 +382,48 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// GL 4.6 core 10.3.9: every DrawElements-family count is a sizei and "if count is negative, an
|
||||
// INVALID_VALUE error is generated". The same sentence covers instancecount and the
|
||||
// MultiDraw* drawcount, so one helper serves all of them; the parameter is named for the
|
||||
// caller so the message says which argument the application actually got wrong.
|
||||
static Bool ValidateNonNegativeDrawArgument(const char* functionName, const char* argumentName, GLsizei value) {
|
||||
if (value >= 0) return true;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
String(argumentName) + " must be non-negative."));
|
||||
return false;
|
||||
}
|
||||
|
||||
// GL 4.6 core 10.3.9 for DrawRangeElements*: "if end < start, an INVALID_VALUE error is
|
||||
// generated". Both are uints, so a caller that passes -1 for start arrives here as
|
||||
// 0xFFFFFFFF and is caught by the same comparison - which is exactly what
|
||||
// KHR-GL4x.draw_elements_base_vertex_tests.invalid_count_argument checks.
|
||||
static Bool ValidateDrawElementsRange(const char* functionName, GLuint start, GLuint end) {
|
||||
if (end >= start) return true;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "end must not be less than start."));
|
||||
return false;
|
||||
}
|
||||
|
||||
// GL 4.6 core 10.9: inside a conditional block whose predicate did not pass, the drawing
|
||||
// commands, Clear, ClearBuffer* and the compute dispatches are DISCARDED. The gate sits on the
|
||||
// wrappers that ISSUE the backend call rather than at the top of each entry point, so that
|
||||
// everything a real driver would still do inside the block - argument validation and the
|
||||
// errors it raises - happens exactly as it does outside one, and only the command itself is
|
||||
// dropped. It is deliberately not on the frontend's transform-feedback accounting either:
|
||||
// that mirrors what the capture stage would have written, and a conditional block around a
|
||||
// capturing draw has no test coverage in either direction.
|
||||
static Bool ConditionalRenderDiscardsCommand() {
|
||||
return MG_State::pGLContext->ConditionalRenderDiscardsCommands();
|
||||
}
|
||||
|
||||
void Clear_Backend(GLbitfield mask) {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.Clear(mask);
|
||||
}
|
||||
|
||||
@@ -307,6 +431,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElements(mode, count, type, indices);
|
||||
}
|
||||
|
||||
@@ -315,6 +440,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElements(mode, count, type, indices, drawcount);
|
||||
}
|
||||
|
||||
@@ -323,6 +449,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsBaseVertex(mode, count, type, indices, drawcount,
|
||||
basevertex);
|
||||
}
|
||||
@@ -331,6 +458,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArrays(mode, first, count);
|
||||
}
|
||||
|
||||
@@ -338,6 +466,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawArrays(mode, first, count, drawcount);
|
||||
}
|
||||
|
||||
@@ -346,6 +475,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsBaseVertex(mode, count, type, indices, basevertex);
|
||||
}
|
||||
|
||||
@@ -354,6 +484,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsIndirect(mode, type, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
@@ -361,6 +492,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawArraysIndirect(mode, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
@@ -369,6 +501,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsIndirectCount(mode, type, indirect, drawcount,
|
||||
maxdrawcount, stride);
|
||||
}
|
||||
@@ -378,6 +511,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.MultiDrawArraysIndirectCount(mode, indirect, drawcount, maxdrawcount,
|
||||
stride);
|
||||
}
|
||||
@@ -387,6 +521,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawRangeElementsBaseVertex(mode, start, end, count, type, indices,
|
||||
basevertex);
|
||||
}
|
||||
@@ -396,6 +531,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawRangeElements(mode, start, end, count, type, indices);
|
||||
}
|
||||
|
||||
@@ -405,6 +541,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstancedBaseVertexBaseInstance(
|
||||
mode, count, type, indices, instancecount, basevertex, baseinstance);
|
||||
}
|
||||
@@ -414,6 +551,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstancedBaseVertex(mode, count, type, indices, instancecount,
|
||||
basevertex);
|
||||
}
|
||||
@@ -423,6 +561,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstancedBaseInstance(mode, count, type, indices,
|
||||
instancecount, baseinstance);
|
||||
}
|
||||
@@ -432,6 +571,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsInstanced(mode, count, type, indices, instancecount);
|
||||
}
|
||||
|
||||
@@ -439,6 +579,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawElementsIndirect(mode, type, indirect);
|
||||
}
|
||||
void DrawArraysInstancedBaseInstance_Backend(GLenum mode, GLint first, GLsizei count, GLsizei instancecount,
|
||||
@@ -446,6 +587,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArraysInstancedBaseInstance(mode, first, count, instancecount,
|
||||
baseinstance);
|
||||
}
|
||||
@@ -454,6 +596,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArraysInstanced(mode, first, count, instancecount);
|
||||
}
|
||||
|
||||
@@ -461,6 +604,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
#ifdef TRACY_ENABLE
|
||||
ZoneScopedC(TRACY_ZONECOLOR_BACKEND);
|
||||
#endif
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.DrawArraysIndirect(mode, indirect);
|
||||
}
|
||||
|
||||
@@ -489,19 +633,19 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
}
|
||||
// GL 4.3 added both dispatches to the conditional-render set (GL 4.6 core 10.9), which is
|
||||
// exactly what KHR-GL43.compute_shader.conditional-dispatching checks.
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
dispatchCompute(numGroupsX, numGroupsY, numGroupsZ);
|
||||
}
|
||||
|
||||
void DispatchComputeIndirect(GLintptr indirect) {
|
||||
auto dispatchComputeIndirect = MG_Backend::gBackendFunctionsTable.GL.DispatchComputeIndirect;
|
||||
if (!dispatchComputeIndirect) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"Backend does not support indirect compute dispatch."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateCurrentProgramForCompute(__func__)) return;
|
||||
// Argument and binding validation runs FIRST. Both are properties of the call and of GL
|
||||
// state, so a context whose backend cannot dispatch at all must still report the
|
||||
// argument error the spec names rather than masking every one of them with
|
||||
// "unsupported" - which is what put GL_INVALID_OPERATION where
|
||||
// KHR-GL43.compute_shader.api-indirect expects GL_INVALID_VALUE.
|
||||
//
|
||||
// GL 4.6 core 19: `indirect` is a byte offset into GL_DISPATCH_INDIRECT_BUFFER -
|
||||
// negative or misaligned is INVALID_VALUE, nothing bound is INVALID_OPERATION.
|
||||
if (indirect < 0 || (indirect % 4) != 0) {
|
||||
@@ -520,6 +664,30 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"No buffer is bound to GL_DISPATCH_INDIRECT_BUFFER."));
|
||||
return;
|
||||
}
|
||||
// ...and the same INVALID_OPERATION covers "the command would source data beyond the end
|
||||
// of the bound buffer object" (GL 4.6 core 19): the dispatch reads three uints starting
|
||||
// at `indirect`.
|
||||
constexpr SizeT kDispatchIndirectCommandSize = 3 * sizeof(Uint32);
|
||||
if (static_cast<SizeT>(indirect) + kDispatchIndirectCommandSize > indirectBuffer->GetSize()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", __func__,
|
||||
std::format("indirect ({}) + 12 bytes runs past the end of the {}-byte buffer bound to "
|
||||
"GL_DISPATCH_INDIRECT_BUFFER.",
|
||||
indirect, indirectBuffer->GetSize())));
|
||||
return;
|
||||
}
|
||||
auto dispatchComputeIndirect = MG_Backend::gBackendFunctionsTable.GL.DispatchComputeIndirect;
|
||||
if (!dispatchComputeIndirect) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"Backend does not support indirect compute dispatch."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateCurrentProgramForCompute(__func__)) return;
|
||||
if (ConditionalRenderDiscardsCommand()) return;
|
||||
dispatchComputeIndirect(indirect);
|
||||
}
|
||||
|
||||
@@ -545,7 +713,34 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
namespace {
|
||||
// GL 4.6 core 7.11.2 (and ARB_shader_image_load_store, which introduced the call): the
|
||||
// barrier bitfield is INVALID_VALUE unless every bit is one of the defined ones, with
|
||||
// GL_ALL_BARRIER_BITS - which is 0xFFFFFFFF, not the union of the list - accepted whole.
|
||||
// Forwarding an undefined bit to the host driver let a caller that had computed its mask
|
||||
// wrongly (or reused an ES-only bit) get silence instead of the error the spec promises.
|
||||
constexpr GLbitfield kAllDefinedBarrierBits =
|
||||
GL_VERTEX_ATTRIB_ARRAY_BARRIER_BIT | GL_ELEMENT_ARRAY_BARRIER_BIT | GL_UNIFORM_BARRIER_BIT |
|
||||
GL_TEXTURE_FETCH_BARRIER_BIT | GL_SHADER_IMAGE_ACCESS_BARRIER_BIT | GL_COMMAND_BARRIER_BIT |
|
||||
GL_PIXEL_BUFFER_BARRIER_BIT | GL_TEXTURE_UPDATE_BARRIER_BIT | GL_BUFFER_UPDATE_BARRIER_BIT |
|
||||
GL_FRAMEBUFFER_BARRIER_BIT | GL_TRANSFORM_FEEDBACK_BARRIER_BIT | GL_ATOMIC_COUNTER_BARRIER_BIT |
|
||||
GL_SHADER_STORAGE_BARRIER_BIT | GL_CLIENT_MAPPED_BUFFER_BARRIER_BIT | GL_QUERY_BUFFER_BARRIER_BIT;
|
||||
|
||||
Bool ValidateMemoryBarrierBits(const char* function, GLbitfield barriers) {
|
||||
if (barriers == GL_ALL_BARRIER_BITS) return true;
|
||||
if ((barriers & ~kAllDefinedBarrierBits) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", function,
|
||||
"barriers contains bits that are not defined barrier bits."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void MemoryBarrier(GLbitfield barriers) {
|
||||
if (!ValidateMemoryBarrierBits(__func__, barriers)) return;
|
||||
auto memoryBarrier = MG_Backend::gBackendFunctionsTable.GL.MemoryBarrier;
|
||||
if (!memoryBarrier) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -557,6 +752,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void MemoryBarrierByRegion(GLbitfield barriers) {
|
||||
if (!ValidateMemoryBarrierBits(__func__, barriers)) return;
|
||||
auto memoryBarrierByRegion = MG_Backend::gBackendFunctionsTable.GL.MemoryBarrierByRegion;
|
||||
if (!memoryBarrierByRegion) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -569,19 +765,93 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
MultiDrawElementsIndirect_Backend(mode, type, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
MultiDrawArraysIndirect_Backend(mode, indirect, drawcount, stride);
|
||||
}
|
||||
|
||||
// ARB_indirect_parameters / GL 4.6 core 10.4: `drawcount` is a byte offset into the buffer
|
||||
// bound to PARAMETER_BUFFER and holds one uint draw count. Three errors have to be raised
|
||||
// before the call reaches a backend, and none of them was
|
||||
// (KHR-GL46.indirect_parameters_tests.MultiDraw{Arrays,Elements}IndirectCount):
|
||||
// * drawcount not a multiple of four INVALID_VALUE
|
||||
// * nothing bound to PARAMETER_BUFFER, or the uint at `drawcount`
|
||||
// lies past its end INVALID_OPERATION
|
||||
// * maxdrawcount commands from `indirect` run past the end of the
|
||||
// buffer bound to DRAW_INDIRECT_BUFFER INVALID_OPERATION
|
||||
static Bool ValidateIndirectCountDraw(GLintptr indirect, GLintptr drawcount, GLsizei maxdrawcount,
|
||||
GLsizei stride, SizeT commandSize, const char* funcName) {
|
||||
if (drawcount < 0 || (drawcount % 4) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", funcName,
|
||||
"drawcount must be non-negative and a multiple of four."));
|
||||
return false;
|
||||
}
|
||||
const auto& parameterBuffer =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
if (!parameterBuffer ||
|
||||
static_cast<SizeT>(drawcount) + sizeof(Uint32) > parameterBuffer->GetSize()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", funcName,
|
||||
"No buffer is bound to GL_PARAMETER_BUFFER, or drawcount runs past "
|
||||
"the end of the one that is."));
|
||||
return false;
|
||||
}
|
||||
if (maxdrawcount < 0 || stride < 0 || indirect < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", funcName,
|
||||
"indirect, maxdrawcount and stride must all be non-negative."));
|
||||
return false;
|
||||
}
|
||||
const SizeT effectiveStride = stride != 0 ? static_cast<SizeT>(stride) : commandSize;
|
||||
const auto& indirectBuffer =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
// A zero maxdrawcount sources nothing, so it cannot run past anything.
|
||||
const SizeT requiredBytes =
|
||||
maxdrawcount == 0 ? 0
|
||||
: static_cast<SizeT>(indirect) +
|
||||
static_cast<SizeT>(maxdrawcount - 1) * effectiveStride + commandSize;
|
||||
if (!indirectBuffer || requiredBytes > indirectBuffer->GetSize()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", funcName,
|
||||
"maxdrawcount commands would be sourced from beyond the end of the "
|
||||
"buffer bound to GL_DRAW_INDIRECT_BUFFER."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride) {
|
||||
// Argument validation before the backend-availability check: see DispatchComputeIndirect.
|
||||
// DrawElementsIndirectCommand: count, instanceCount, firstIndex, baseVertex, baseInstance.
|
||||
if (!ValidateIndirectCountDraw(reinterpret_cast<GLintptr>(indirect), drawcount, maxdrawcount, stride,
|
||||
5 * sizeof(Uint32), __func__)) {
|
||||
return;
|
||||
}
|
||||
// The only two draw entry points that were missing this. Every backend draw path
|
||||
// dereferences GetProgramForDraw() unconditionally, so "no current program" has to be
|
||||
// stopped here or it is a null dereference rather than the INVALID_OPERATION the spec
|
||||
// asks for - reachable through a bound pipeline that supplies no graphics stage.
|
||||
//
|
||||
// AFTER the argument checks, unlike the sibling draw entry points, and deliberately:
|
||||
// the argument rules here are properties of the call rather than of GL state, and
|
||||
// NegativeApiErrorsTest.IndirectParameterDrawsCheckBothBuffers pins the INVALID_VALUE
|
||||
// they produce for a call made with no program bound. Same precedence decision, and
|
||||
// the same reason, as DispatchComputeIndirect above.
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
auto multiDrawElementsIndirectCount = MG_Backend::gBackendFunctionsTable.GL.MultiDrawElementsIndirectCount;
|
||||
if (!multiDrawElementsIndirectCount) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -595,6 +865,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride) {
|
||||
// Argument validation before the backend-availability check: see DispatchComputeIndirect.
|
||||
// DrawArraysIndirectCommand: count, instanceCount, first, baseInstance.
|
||||
if (!ValidateIndirectCountDraw(reinterpret_cast<GLintptr>(indirect), drawcount, maxdrawcount, stride,
|
||||
4 * sizeof(Uint32), __func__)) {
|
||||
return;
|
||||
}
|
||||
// See MultiDrawElementsIndirectCount, including why this one goes last.
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
auto multiDrawArraysIndirectCount = MG_Backend::gBackendFunctionsTable.GL.MultiDrawArraysIndirectCount;
|
||||
if (!multiDrawArraysIndirectCount) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -608,12 +886,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "count", count)) return;
|
||||
if (!ValidateDrawElementsRange(__func__, start, end)) return;
|
||||
DrawRangeElementsBaseVertex_Backend(mode, start, end, count, type, indices, basevertex);
|
||||
}
|
||||
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawRangeElements_Backend(mode, start, end, count, type, indices);
|
||||
@@ -621,6 +904,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawElementsInstancedBaseVertexBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLint basevertex, GLuint baseinstance) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawElementsInstancedBaseVertexBaseInstance_Backend(mode, count, type, indices, instancecount, basevertex,
|
||||
@@ -629,25 +913,32 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawElementsInstancedBaseVertex(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLint basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "count", count)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "instancecount", instancecount)) return;
|
||||
DrawElementsInstancedBaseVertex_Backend(mode, count, type, indices, instancecount, basevertex);
|
||||
}
|
||||
|
||||
void DrawElementsInstancedBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLsizei instancecount, GLuint baseinstance) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawElementsInstancedBaseInstance_Backend(mode, count, type, indices, instancecount, baseinstance);
|
||||
}
|
||||
|
||||
void DrawElementsInstanced(GLenum mode, GLsizei count, GLenum type, const void* indices, GLsizei instancecount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawElementsInstanced_Backend(mode, count, type, indices, instancecount);
|
||||
}
|
||||
|
||||
void DrawElementsIndirect(GLenum mode, GLenum type, const void* indirect) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
@@ -657,18 +948,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void DrawArraysInstancedBaseInstance(GLenum mode, GLint first, GLsizei count, GLsizei instancecount,
|
||||
GLuint baseinstance) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawArraysInstancedBaseInstance_Backend(mode, first, count, instancecount, baseinstance);
|
||||
}
|
||||
|
||||
void DrawArraysInstanced(GLenum mode, GLint first, GLsizei count, GLsizei instancecount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
DrawArraysInstanced_Backend(mode, first, count, instancecount);
|
||||
}
|
||||
|
||||
void DrawArraysIndirect(GLenum mode, const void* indirect) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateIndirectDrawSource(__func__, indirect, kDrawArraysIndirectCommandBytes)) return;
|
||||
@@ -676,13 +970,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void DrawElementsBaseVertex(GLenum mode, GLsizei count, GLenum type, const void* indices, GLint basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "count", count)) return;
|
||||
AccountTransformFeedbackPrimitives(mode, count);
|
||||
DrawElementsBaseVertex_Backend(mode, count, type, indices, basevertex);
|
||||
}
|
||||
|
||||
void DrawArrays(GLenum mode, GLint first, GLsizei count) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
AccountTransformFeedbackPrimitives(mode, count);
|
||||
@@ -690,6 +988,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (drawcount < 0) {
|
||||
@@ -703,6 +1002,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void MultiDrawElements(GLenum mode, const GLsizei* count, GLenum type, const void* const* indices,
|
||||
GLsizei drawcount) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
MultiDrawElements_Backend(mode, count, type, indices, drawcount);
|
||||
@@ -710,8 +1010,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void MultiDrawElementsBaseVertex(GLenum mode, const GLsizei* count, GLenum type, const void* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
if (!ValidateDrawElementsIndexType(__func__, type)) return;
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "drawcount", drawcount)) return;
|
||||
// GL 4.6 core 10.5 defines MultiDrawElementsBaseVertex as drawcount separate
|
||||
// DrawElementsBaseVertex calls, so each element of the count array carries the same
|
||||
// non-negative requirement the single-draw entry point applies to its own count. The
|
||||
// whole call is rejected before any sub-draw is issued, which is what makes the error
|
||||
// observable at all - a driver that drew the valid prefix first would leave the
|
||||
// framebuffer half-written.
|
||||
if (count != nullptr) {
|
||||
for (GLsizei draw = 0; draw < drawcount; ++draw) {
|
||||
if (!ValidateNonNegativeDrawArgument(__func__, "every element of count", count[draw])) return;
|
||||
}
|
||||
}
|
||||
MultiDrawElementsBaseVertex_Backend(mode, count, type, indices, drawcount, basevertex);
|
||||
}
|
||||
|
||||
@@ -720,6 +1034,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void DrawElements(GLenum mode, GLsizei count, GLenum type, const void* indices) {
|
||||
if (!ValidatePrimitiveModeEnum(__func__, mode)) return;
|
||||
if (!ValidateCurrentProgramForExecution(__func__)) return;
|
||||
if (!ValidatePrimitiveModeForBackend(__func__, mode)) return;
|
||||
AccountTransformFeedbackPrimitives(mode, count);
|
||||
@@ -1152,7 +1467,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "instancecount must be non-negative."));
|
||||
return;
|
||||
}
|
||||
if (!MG_State::pGLContext->ValidateTransformFeedbackName(id)) {
|
||||
// "id is not the name of a transform feedback object" has to mean the same thing here
|
||||
// as it does to glIsTransformFeedback, and the two predicates are not interchangeable:
|
||||
// a name glGenTransformFeedbacks handed out is only reserved until it is first bound,
|
||||
// and only the bind turns it into an object (GL 4.6 core 13.2.1). ValidateTransformFeedbackName
|
||||
// answers the reservation question - the right one for glBindTransformFeedback, which is
|
||||
// what turns a reserved name into an object - so using it here let a generated-but-unbound
|
||||
// name through to the completed-span check below and raised INVALID_OPERATION where the
|
||||
// spec asks for INVALID_VALUE. Name 0 is the default object and always drawable.
|
||||
if (id != 0 && !MG_State::pGLContext->IsTransformFeedbackObject(id)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
|
||||
@@ -20,17 +20,18 @@
|
||||
#include "../Framebuffer/GL_Framebuffer.h"
|
||||
#include "../VertexArray/GL_VertexArray.h"
|
||||
#include "../Sync/GL_Sync.h"
|
||||
#include "../Debug/GL_Debug.h"
|
||||
#include <MG_State/GLState/Core.h>
|
||||
|
||||
#define DECLARE_GL_FUNCTION_STUB_HEAD(type, name, ...) MOBILEGL_GL_API type gl##name(__VA_ARGS__) {
|
||||
|
||||
#define DECLARE_GL_FUNCTION_STUB_END(type, name, ...) \
|
||||
MGLOG_W("Stub function: %s(...)", __FUNCTION__); \
|
||||
MGLOG_W_ONCE("Stub function: %s(...)", __FUNCTION__); \
|
||||
return (type)1; \
|
||||
}
|
||||
|
||||
#define DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(type, name, ...) \
|
||||
MGLOG_W("Stub function: %s(...)", __FUNCTION__); \
|
||||
MGLOG_W_ONCE("Stub function: %s(...)", __FUNCTION__); \
|
||||
}
|
||||
|
||||
#define DECLARE_GL_FUNCTION_HEAD(type, name, ...) MOBILEGL_GL_API type gl##name(__VA_ARGS__) {
|
||||
@@ -378,27 +379,13 @@ DECLARE_GL_FUNCTION_HEAD(void, VertexBindingDivisor, GLuint bindingindex, GLuint
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BlendBarrier) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BlendBarrier)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CopyImageSubData, GLuint srcName, GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ, GLuint dstName, GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ, GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CopyImageSubData, srcName, srcTarget, srcLevel, srcX, srcY, srcZ, dstName, dstTarget, dstLevel, dstX, dstY, dstZ, srcWidth, srcHeight, srcDepth)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DebugMessageControl, GLenum source, GLenum type, GLenum severity, GLsizei count, const GLuint* ids, GLboolean enabled) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DebugMessageControl, source, type, severity, count, ids, enabled)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DebugMessageInsert, GLenum source, GLenum type, GLuint id, GLenum severity, GLsizei length, const GLchar* buf) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DebugMessageInsert, source, type, id, severity, length, buf)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DebugMessageInsert, GLenum source, GLenum type, GLuint id, GLenum severity, GLsizei length, const GLchar* buf) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DebugMessageInsert, source, type, id, severity, length, buf)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DebugMessageCallback, GLDEBUGPROC callback, const void* userParam) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DebugMessageCallback, callback, userParam)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(GLuint, GetDebugMessageLog, GLuint count, GLsizei bufSize, GLenum* sources, GLenum* types, GLuint* ids, GLenum* severities, GLsizei* lengths, GLchar* messageLog) DECLARE_GL_FUNCTION_STUB_END(GLuint, GetDebugMessageLog, count, bufSize, sources, types, ids, severities, lengths, messageLog)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PushDebugGroup, GLenum source, GLuint id, GLsizei length, const GLchar* message) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PushDebugGroup, source, id, length, message)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PopDebugGroup) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PopDebugGroup)
|
||||
MOBILEGL_GL_API void glObjectLabel(GLenum identifier, GLuint name, GLsizei length, const GLchar* label) {
|
||||
(void)identifier;
|
||||
(void)name;
|
||||
(void)length;
|
||||
(void)label;
|
||||
}
|
||||
MOBILEGL_GL_API void glGetObjectLabel(GLenum identifier, GLuint name, GLsizei bufSize, GLsizei* length, GLchar* label) {
|
||||
(void)identifier;
|
||||
(void)name;
|
||||
if (length) {
|
||||
*length = 0;
|
||||
}
|
||||
if (label && bufSize > 0) {
|
||||
label[0] = '\0';
|
||||
}
|
||||
}
|
||||
DECLARE_GL_FUNCTION_HEAD(void, PushDebugGroup, GLenum source, GLuint id, GLsizei length, const GLchar* message) DECLARE_GL_FUNCTION_END_NO_RETURN(void, PushDebugGroup, source, id, length, message)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, PopDebugGroup) DECLARE_GL_FUNCTION_END_NO_RETURN(void, PopDebugGroup)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ObjectLabel, GLenum identifier, GLuint name, GLsizei length, const GLchar* label) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ObjectLabel, identifier, name, length, label)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, GetObjectLabel, GLenum identifier, GLuint name, GLsizei bufSize, GLsizei* length, GLchar* label) DECLARE_GL_FUNCTION_END_NO_RETURN(void, GetObjectLabel, identifier, name, bufSize, length, label)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ObjectPtrLabel, const void* ptr, GLsizei length, const GLchar* label) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ObjectPtrLabel, ptr, length, label)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetObjectPtrLabel, const void* ptr, GLsizei bufSize, GLsizei* length, GLchar* label) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetObjectPtrLabel, ptr, bufSize, length, label)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetPointerv, GLenum pname, void** params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetPointerv, pname, params)
|
||||
@@ -725,8 +712,8 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, LoadName, GLuint name) DECLARE_GL_FUNCTION_S
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PushName, GLuint name) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PushName, name)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PopName) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PopName)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClampColor, GLenum target, GLenum clamp) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClampColor, target, clamp)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BeginConditionalRender, GLuint id, GLenum mode) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BeginConditionalRender, id, mode)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, EndConditionalRender, void) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, EndConditionalRender)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BeginConditionalRender, GLuint id, GLenum mode) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BeginConditionalRender, id, mode)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, EndConditionalRender) DECLARE_GL_FUNCTION_END_NO_RETURN(void, EndConditionalRender)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, VertexAttribI1i, GLuint index, GLint x) DECLARE_GL_FUNCTION_END_NO_RETURN(void, VertexAttribI1i, index, x)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, VertexAttribI2i, GLuint index, GLint x, GLint y) DECLARE_GL_FUNCTION_END_NO_RETURN(void, VertexAttribI2i, index, x, y)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, VertexAttribI3i, GLuint index, GLint x, GLint y, GLint z) DECLARE_GL_FUNCTION_END_NO_RETURN(void, VertexAttribI3i, index, x, y, z)
|
||||
@@ -969,24 +956,24 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribL3dv, GLuint index, const GLdoub
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribL4dv, GLuint index, const GLdouble* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, VertexAttribL4dv, index, v)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, VertexAttribLPointer, GLuint index, GLint size, GLenum type, GLsizei stride, const void* pointer) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, VertexAttribLPointer, index, size, type, stride, pointer)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetVertexAttribLdv, GLuint index, GLenum pname, GLdouble* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetVertexAttribLdv, index, pname, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ViewportArrayv, GLuint first, GLsizei count, const GLfloat* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ViewportArrayv, first, count, v)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ViewportIndexedf, GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ViewportIndexedf, index, x, y, w, h)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ViewportIndexedfv, GLuint index, const GLfloat* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ViewportIndexedfv, index, v)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ScissorArrayv, GLuint first, GLsizei count, const GLint* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ScissorArrayv, first, count, v)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ScissorIndexed, GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ScissorIndexed, index, left, bottom, width, height)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ScissorIndexedv, GLuint index, const GLint* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ScissorIndexedv, index, v)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DepthRangeArrayv, GLuint first, GLsizei count, const GLdouble* v) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DepthRangeArrayv, first, count, v)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DepthRangeIndexed, GLuint index, GLdouble n, GLdouble f) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DepthRangeIndexed, index, n, f)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ViewportArrayv, GLuint first, GLsizei count, const GLfloat* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ViewportArrayv, first, count, v)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ViewportIndexedf, GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ViewportIndexedf, index, x, y, w, h)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ViewportIndexedfv, GLuint index, const GLfloat* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ViewportIndexedfv, index, v)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ScissorArrayv, GLuint first, GLsizei count, const GLint* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ScissorArrayv, first, count, v)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ScissorIndexed, GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ScissorIndexed, index, left, bottom, width, height)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ScissorIndexedv, GLuint index, const GLint* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ScissorIndexedv, index, v)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DepthRangeArrayv, GLuint first, GLsizei count, const GLdouble* v) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DepthRangeArrayv, first, count, v)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DepthRangeIndexed, GLuint index, GLdouble n, GLdouble f) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DepthRangeIndexed, index, n, f)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, GetFloati_v, GLenum target, GLuint index, GLfloat* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, GetFloati_v, target, index, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, GetDoublei_v, GLenum target, GLuint index, GLdouble* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, GetDoublei_v, target, index, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawArraysInstancedBaseInstance, GLenum mode, GLint first, GLsizei count, GLsizei instancecount, GLuint baseinstance) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawArraysInstancedBaseInstance, mode, first, count, instancecount, baseinstance)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawElementsInstancedBaseInstance, GLenum mode, GLsizei count, GLenum type, const void* indices, GLsizei instancecount, GLuint baseinstance) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawElementsInstancedBaseInstance, mode, count, type, indices, instancecount, baseinstance)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawElementsInstancedBaseVertexBaseInstance, GLenum mode, GLsizei count, GLenum type, const void* indices, GLsizei instancecount, GLint basevertex, GLuint baseinstance) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawElementsInstancedBaseVertexBaseInstance, mode, count, type, indices, instancecount, basevertex, baseinstance)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetActiveAtomicCounterBufferiv, GLuint program, GLuint bufferIndex, GLenum pname, GLint* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetActiveAtomicCounterBufferiv, program, bufferIndex, pname, params)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, GetActiveAtomicCounterBufferiv, GLuint program, GLuint bufferIndex, GLenum pname, GLint* params) DECLARE_GL_FUNCTION_END_NO_RETURN(void, GetActiveAtomicCounterBufferiv, program, bufferIndex, pname, params)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackInstanced, GLenum mode, GLuint id, GLsizei instancecount) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackInstanced, mode, id, instancecount)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, DrawTransformFeedbackStreamInstanced, GLenum mode, GLuint id, GLuint stream, GLsizei instancecount) DECLARE_GL_FUNCTION_END_NO_RETURN(void, DrawTransformFeedbackStreamInstanced, mode, id, stream, instancecount)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearBufferData, GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearBufferData, target, internalformat, format, type, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClearBufferSubData, GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClearBufferSubData, target, internalformat, offset, size, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearBufferData, GLenum target, GLenum internalformat, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearBufferData, target, internalformat, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearBufferSubData, GLenum target, GLenum internalformat, GLintptr offset, GLsizeiptr size, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearBufferSubData, target, internalformat, offset, size, format, type, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetInternalformati64v, GLenum target, GLenum internalformat, GLenum pname, GLsizei count, GLint64* params) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetInternalformati64v, target, internalformat, pname, count, params)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, InvalidateTexSubImage, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, InvalidateTexSubImage, texture, level, xoffset, yoffset, zoffset, width, height, depth)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, InvalidateTexImage, GLuint texture, GLint level) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, InvalidateTexImage, texture, level)
|
||||
@@ -996,16 +983,16 @@ DECLARE_GL_FUNCTION_HEAD(void, MultiDrawArraysIndirect, GLenum mode, const void*
|
||||
DECLARE_GL_FUNCTION_HEAD(void, MultiDrawElementsIndirect, GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride) DECLARE_GL_FUNCTION_END_NO_RETURN(void, MultiDrawElementsIndirect, mode, type, indirect, drawcount, stride)
|
||||
DECLARE_GL_FUNCTION_HEAD(GLint, GetProgramResourceLocationIndex, GLuint program, GLenum programInterface, const GLchar* name) DECLARE_GL_FUNCTION_END(GLint, GetProgramResourceLocationIndex, program, programInterface, name)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ShaderStorageBlockBinding, GLuint program, GLuint storageBlockIndex, GLuint storageBlockBinding) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ShaderStorageBlockBinding, program, storageBlockIndex, storageBlockBinding)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, TextureView, GLuint texture, GLenum target, GLuint origtexture, GLenum internalformat, GLuint minlevel, GLuint numlevels, GLuint minlayer, GLuint numlayers) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, TextureView, texture, target, origtexture, internalformat, minlevel, numlevels, minlayer, numlayers)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureView, GLuint texture, GLenum target, GLuint origtexture, GLenum internalformat, GLuint minlevel, GLuint numlevels, GLuint minlayer, GLuint numlayers) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureView, texture, target, origtexture, internalformat, minlevel, numlevels, minlayer, numlayers)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, VertexAttribLFormat, GLuint attribindex, GLint size, GLenum type, GLuint relativeoffset) DECLARE_GL_FUNCTION_END_NO_RETURN(void, VertexAttribLFormat, attribindex, size, type, relativeoffset)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BufferStorage, GLenum target, GLsizeiptr size, const void* data, GLbitfield flags) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BufferStorage, target, size, data, flags)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearTexImage, GLuint texture, GLint level, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearTexImage, texture, level, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, ClearTexSubImage, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLenum type, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, ClearTexSubImage, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, type, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindBuffersBase, GLenum target, GLuint first, GLsizei count, const GLuint* buffers) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindBuffersBase, target, first, count, buffers)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindBuffersRange, GLenum target, GLuint first, GLsizei count, const GLuint* buffers, const GLintptr* offsets, const GLsizeiptr* sizes) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindBuffersRange, target, first, count, buffers, offsets, sizes)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BindTextures, GLuint first, GLsizei count, const GLuint* textures) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BindTextures, first, count, textures)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindTextures, GLuint first, GLsizei count, const GLuint* textures) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindTextures, first, count, textures)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindSamplers, GLuint first, GLsizei count, const GLuint* samplers) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindSamplers, first, count, samplers)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, BindImageTextures, GLuint first, GLsizei count, const GLuint* textures) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, BindImageTextures, first, count, textures)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindImageTextures, GLuint first, GLsizei count, const GLuint* textures) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindImageTextures, first, count, textures)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, BindVertexBuffers, GLuint first, GLsizei count, const GLuint* buffers, const GLintptr* offsets, const GLsizei* strides) DECLARE_GL_FUNCTION_END_NO_RETURN(void, BindVertexBuffers, first, count, buffers, offsets, strides)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, ClipControl, GLenum origin, GLenum depth) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, ClipControl, origin, depth)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CreateTransformFeedbacks, GLsizei n, GLuint* ids) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CreateTransformFeedbacks, n, ids)
|
||||
@@ -1060,9 +1047,9 @@ DECLARE_GL_FUNCTION_HEAD(void, TextureStorage3DMultisample, GLuint texture, GLsi
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureSubImage1D, GLuint texture, GLint level, GLint xoffset, GLsizei width, GLenum format, GLenum type, const void* pixels) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureSubImage1D, texture, level, xoffset, width, format, type, pixels)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLsizei width, GLsizei height, GLenum format, GLenum type, const void* pixels) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureSubImage2D, texture, level, xoffset, yoffset, width, height, format, type, pixels)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, TextureSubImage3D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLenum type, const void* pixels) DECLARE_GL_FUNCTION_END_NO_RETURN(void, TextureSubImage3D, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, type, pixels)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage1D, GLuint texture, GLint level, GLint xoffset, GLsizei width, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage1D, texture, level, xoffset, width, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLsizei width, GLsizei height, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage2D, texture, level, xoffset, yoffset, width, height, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage3D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage3D, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CompressedTextureSubImage1D, GLuint texture, GLint level, GLint xoffset, GLsizei width, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CompressedTextureSubImage1D, texture, level, xoffset, width, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CompressedTextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLsizei width, GLsizei height, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CompressedTextureSubImage2D, texture, level, xoffset, yoffset, width, height, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CompressedTextureSubImage3D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize, const void* data) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CompressedTextureSubImage3D, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, imageSize, data)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CopyTextureSubImage1D, GLuint texture, GLint level, GLint xoffset, GLint x, GLint y, GLsizei width) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CopyTextureSubImage1D, texture, level, xoffset, x, y, width)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CopyTextureSubImage2D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CopyTextureSubImage2D, texture, level, xoffset, yoffset, x, y, width, height)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CopyTextureSubImage3D, GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLint x, GLint y, GLsizei width, GLsizei height) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CopyTextureSubImage3D, texture, level, xoffset, yoffset, zoffset, x, y, width, height)
|
||||
@@ -1848,9 +1835,9 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, GetBooleanIndexedvEXT, GLenum target, GLuint
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureImage3DEXT, GLuint texture, GLenum target, GLint level, GLenum internalformat, GLsizei width, GLsizei height, GLsizei depth, GLint border, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureImage3DEXT, texture, target, level, internalformat, width, height, depth, border, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureImage2DEXT, GLuint texture, GLenum target, GLint level, GLenum internalformat, GLsizei width, GLsizei height, GLint border, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureImage2DEXT, texture, target, level, internalformat, width, height, border, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureImage1DEXT, GLuint texture, GLenum target, GLint level, GLenum internalformat, GLsizei width, GLint border, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureImage1DEXT, texture, target, level, internalformat, width, border, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage3DEXT, GLuint texture, GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage3DEXT, texture, target, level, xoffset, yoffset, zoffset, width, height, depth, format, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage2DEXT, GLuint texture, GLenum target, GLint level, GLint xoffset, GLint yoffset, GLsizei width, GLsizei height, GLenum format, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage2DEXT, texture, target, level, xoffset, yoffset, width, height, format, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedTextureSubImage1DEXT, GLuint texture, GLenum target, GLint level, GLint xoffset, GLsizei width, GLenum format, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedTextureSubImage1DEXT, texture, target, level, xoffset, width, format, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CompressedTextureSubImage3DEXT, GLuint texture, GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CompressedTextureSubImage3D, texture, level, xoffset, yoffset, zoffset, width, height, depth, format, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CompressedTextureSubImage2DEXT, GLuint texture, GLenum target, GLint level, GLint xoffset, GLint yoffset, GLsizei width, GLsizei height, GLenum format, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CompressedTextureSubImage2D, texture, level, xoffset, yoffset, width, height, format, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_HEAD(void, CompressedTextureSubImage1DEXT, GLuint texture, GLenum target, GLint level, GLint xoffset, GLsizei width, GLenum format, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_END_NO_RETURN(void, CompressedTextureSubImage1D, texture, level, xoffset, width, format, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GetCompressedTextureImageEXT, GLuint texture, GLenum target, GLint lod, void* img) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GetCompressedTextureImageEXT, texture, target, lod, img)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedMultiTexImage3DEXT, GLenum texunit, GLenum target, GLint level, GLenum internalformat, GLsizei width, GLsizei height, GLsizei depth, GLint border, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedMultiTexImage3DEXT, texunit, target, level, internalformat, width, height, depth, border, imageSize, bits)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, CompressedMultiTexImage2DEXT, GLenum texunit, GLenum target, GLint level, GLenum internalformat, GLsizei width, GLsizei height, GLint border, GLsizei imageSize, const void* bits) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, CompressedMultiTexImage2DEXT, texunit, target, level, internalformat, width, height, border, imageSize, bits)
|
||||
@@ -2585,7 +2572,7 @@ DECLARE_GL_FUNCTION_STUB_HEAD(void, BindTransformFeedbackNV, GLenum target, GLui
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, DeleteTransformFeedbacksNV, GLsizei n, const GLuint* ids) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, DeleteTransformFeedbacksNV, n, ids)
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, GenTransformFeedbacksNV, GLsizei n, GLuint* ids) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, GenTransformFeedbacksNV, n, ids)
|
||||
MOBILEGL_GL_API GLboolean glIsTransformFeedbackNV(GLuint id) {
|
||||
MGLOG_W("Stub function: %s(...)", __FUNCTION__);
|
||||
MGLOG_W_ONCE("Stub function: %s(...)", __FUNCTION__);
|
||||
return GL_FALSE;
|
||||
}
|
||||
DECLARE_GL_FUNCTION_STUB_HEAD(void, PauseTransformFeedbackNV, void) DECLARE_GL_FUNCTION_STUB_END_NO_RETURN(void, PauseTransformFeedbackNV, )
|
||||
@@ -3181,5 +3168,5 @@ MOBILEGL_GL_API void glVertexAttribDivisorARB(GLuint index, GLuint divisor) {
|
||||
}
|
||||
|
||||
MOBILEGL_GL_API void glWindowRectanglesEXT(GLenum mode, GLsizei count, const GLint* box) {
|
||||
MGLOG_W("Stub function: %s(...)", __FUNCTION__);
|
||||
MGLOG_W_ONCE("Stub function: %s(...)", __FUNCTION__);
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Util/Metrics/TextureMetrics.h>
|
||||
#include <MG_Impl/GLImpl/Texture/Validators.h>
|
||||
#include <MG_Impl/GLImpl/Getter/GL_Getter.h>
|
||||
#include <MG_State/GLState/ErrorState/Error.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/TextureEnumConverter.h>
|
||||
@@ -547,7 +548,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLint dstX1, GLint dstY1, GLbitfield mask, GLenum filter) {
|
||||
auto blitNamedFramebuffer = MG_Backend::gBackendFunctionsTable.GL.BlitNamedFramebuffer;
|
||||
if (!blitNamedFramebuffer) {
|
||||
MGLOG_E("glBlitNamedFramebuffer skipped: backend does not implement explicit framebuffer blit.");
|
||||
MGLOG_E_ONCE("glBlitNamedFramebuffer skipped: backend does not implement explicit framebuffer blit.");
|
||||
return;
|
||||
}
|
||||
blitNamedFramebuffer(readFramebuffer, drawFramebuffer, srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1,
|
||||
@@ -558,7 +559,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLenum buffer, GLint drawbuffer, const GLfloat* value) {
|
||||
auto clearNamedFramebufferfv = MG_Backend::gBackendFunctionsTable.GL.ClearNamedFramebufferfv;
|
||||
if (!clearNamedFramebufferfv) {
|
||||
MGLOG_E("glClearNamedFramebufferfv skipped: backend does not implement explicit framebuffer clear.");
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferfv skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
clearNamedFramebufferfv(framebuffer, buffer, drawbuffer, value);
|
||||
@@ -568,7 +569,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil) {
|
||||
auto clearNamedFramebufferfi = MG_Backend::gBackendFunctionsTable.GL.ClearNamedFramebufferfi;
|
||||
if (!clearNamedFramebufferfi) {
|
||||
MGLOG_E("glClearNamedFramebufferfi skipped: backend does not implement explicit framebuffer clear.");
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferfi skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
clearNamedFramebufferfi(framebuffer, buffer, drawbuffer, depth, stencil);
|
||||
@@ -578,7 +579,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLenum buffer, GLint drawbuffer, const GLint* value) {
|
||||
auto clearNamedFramebufferiv = MG_Backend::gBackendFunctionsTable.GL.ClearNamedFramebufferiv;
|
||||
if (!clearNamedFramebufferiv) {
|
||||
MGLOG_E("glClearNamedFramebufferiv skipped: backend does not implement explicit framebuffer clear.");
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferiv skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
clearNamedFramebufferiv(framebuffer, buffer, drawbuffer, value);
|
||||
@@ -588,7 +589,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLenum buffer, GLint drawbuffer, const GLuint* value) {
|
||||
auto clearNamedFramebufferuiv = MG_Backend::gBackendFunctionsTable.GL.ClearNamedFramebufferuiv;
|
||||
if (!clearNamedFramebufferuiv) {
|
||||
MGLOG_E("glClearNamedFramebufferuiv skipped: backend does not implement explicit framebuffer clear.");
|
||||
MGLOG_E_ONCE("glClearNamedFramebufferuiv skipped: backend does not implement explicit framebuffer clear.");
|
||||
return;
|
||||
}
|
||||
clearNamedFramebufferuiv(framebuffer, buffer, drawbuffer, value);
|
||||
@@ -617,7 +618,39 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return std::numeric_limits<Int>::max();
|
||||
}
|
||||
return std::max(MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxSamples, 1);
|
||||
return GetAdvertisedMaxSamples();
|
||||
}
|
||||
|
||||
// GL_MAX_SAMPLES is the ceiling over all formats; an integer format has its own
|
||||
// (GL_MAX_INTEGER_SAMPLES) and GL 4.6 core 9.2.4 makes exceeding it INVALID_OPERATION.
|
||||
// The multisample TEXTURE path resolves the limit per format the same way
|
||||
// (GL_Texture.cpp, GetMaxSupportedTextureSamples). Both are floored to the value MobileGL
|
||||
// advertises: on a driver where the two differ - Adreno reports GL_MAX_SAMPLES 4 and
|
||||
// GL_MAX_INTEGER_SAMPLES 1 - rejecting the advertised count here only moves the failure
|
||||
// from the driver into MobileGL, so the frontend accepts it and the backend clamps the
|
||||
// count it actually hands the driver.
|
||||
Int GetMaxRenderbufferSamplesForFormat_State(TextureInternalFormat format) {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return std::numeric_limits<Int>::max();
|
||||
}
|
||||
const auto& dynamicParameters = MG_Backend::pActiveBackendObject->GetDynamicParameters();
|
||||
|
||||
GLenum normalizedInternalFormat = MG_Util::ConvertTextureInternalFormatToGLEnum(format);
|
||||
GLenum normalizedFormat = GL_RGBA;
|
||||
GLenum normalizedType = GL_UNSIGNED_BYTE;
|
||||
MG_Util::TextureFormatProcessor::NormalizePixelFormat(normalizedInternalFormat,
|
||||
PixelFormatNormalizeOptionBit::None,
|
||||
&normalizedInternalFormat, &normalizedFormat,
|
||||
&normalizedType);
|
||||
const Bool isIntegerFormat = normalizedFormat == GL_RED_INTEGER || normalizedFormat == GL_RG_INTEGER ||
|
||||
normalizedFormat == GL_RGB_INTEGER || normalizedFormat == GL_RGBA_INTEGER;
|
||||
if (!isIntegerFormat) {
|
||||
return GetMaxRenderbufferSamples_State();
|
||||
}
|
||||
// Per-format still, but never below the ceiling glGetIntegerv(GL_MAX_SAMPLES) promised:
|
||||
// the driver's raw GL_MAX_INTEGER_SAMPLES stays the *backend* limit and the backend
|
||||
// clamps to it, while the frontend honours what it advertised.
|
||||
return std::max(dynamicParameters.MaxIntegerSamples, GetAdvertisedMaxSamples());
|
||||
}
|
||||
|
||||
Bool ValidateRenderbufferStorageSize_State(GLsizei width, GLsizei height, const char* caller) {
|
||||
@@ -641,7 +674,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool ValidateRenderbufferStorageSamples_State(GLsizei samples, const char* caller) {
|
||||
Bool ValidateRenderbufferStorageSamples_State(GLsizei samples, TextureInternalFormat format, const char* caller) {
|
||||
if (samples < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
@@ -649,9 +682,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Int maxSamples = GetMaxRenderbufferSamples_State();
|
||||
// TODO: Resolve the remaining per-internalformat renderbuffer sample limits once
|
||||
// glGetInternalformativ is backed; integer formats are handled below.
|
||||
const Int maxSamples = GetMaxRenderbufferSamplesForFormat_State(format);
|
||||
if (samples > maxSamples) {
|
||||
// TODO: Use per-internalformat renderbuffer sample limits once glGetInternalformativ is backed.
|
||||
// GL 4.6 core 9.2.4 makes asking for more samples than the format supports
|
||||
// INVALID_OPERATION, not INVALID_VALUE - the count is well formed, this format just
|
||||
// cannot deliver it. Only a negative count is INVALID_VALUE.
|
||||
@@ -659,7 +693,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", caller,
|
||||
std::format("Sample count {} exceeds GL_MAX_SAMPLES ({}).", samples, maxSamples)));
|
||||
std::format("Sample count {} exceeds this format's sample limit ({}).", samples, maxSamples)));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
@@ -684,7 +718,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
TextureInternalFormat format = MG_Util::ConvertGLEnumToTextureInternalFormat(internalformat);
|
||||
if (!TextureImpl::ValidateTextureInternalFormat(format)) return;
|
||||
|
||||
if (!ValidateRenderbufferStorageSamples_State(samples, kCaller)) return;
|
||||
if (!ValidateRenderbufferStorageSamples_State(samples, format, kCaller)) return;
|
||||
if (!ValidateRenderbufferStorageSize_State(width, height, kCaller)) return;
|
||||
|
||||
renderbufferObject->AllocateStorage({width, height});
|
||||
@@ -931,7 +965,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
TextureInternalFormat format = MG_Util::ConvertGLEnumToTextureInternalFormat(internalformat);
|
||||
if (!TextureImpl::ValidateTextureInternalFormat(format)) return;
|
||||
if (!ValidateRenderbufferStorageSamples_State(samples, "NamedRenderbufferStorageMultisample_State")) return;
|
||||
if (!ValidateRenderbufferStorageSamples_State(samples, format, "NamedRenderbufferStorageMultisample_State"))
|
||||
return;
|
||||
if (!ValidateRenderbufferStorageSize_State(width, height, "NamedRenderbufferStorageMultisample_State")) return;
|
||||
|
||||
renderbufferObject->AllocateStorage({width, height});
|
||||
@@ -2578,18 +2613,26 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void ClearBufferfi_Backend(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferfi(buffer, drawbuffer, depth, stencil);
|
||||
}
|
||||
|
||||
void ClearBufferfv_Backend(GLenum buffer, GLint drawbuffer, const GLfloat* value) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferfv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearBufferuiv_Backend(GLenum buffer, GLint drawbuffer, const GLuint* value) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferuiv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
void ClearBufferiv_Backend(GLenum buffer, GLint drawbuffer, const GLint* value) {
|
||||
// GL 4.6 core 10.9 makes ClearBuffer* conditional alongside the drawing commands.
|
||||
if (MG_State::pGLContext->ConditionalRenderDiscardsCommands()) return;
|
||||
MG_Backend::gBackendFunctionsTable.GL.ClearBufferiv(buffer, drawbuffer, value);
|
||||
}
|
||||
|
||||
@@ -3118,15 +3161,55 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GetNamedFramebufferAttachmentParameteriv_State(framebuffer, attachment, pname, params);
|
||||
}
|
||||
|
||||
// The three argument errors GL 4.6 core 18.3.1 asks a blit for. They have to be raised here,
|
||||
// in the backend-independent frontend: DirectGLES drains the driver's error queue around the
|
||||
// blit on purpose (that is how the resolve fallback probes the driver), so an ES-side
|
||||
// rejection never reaches the application and glGetError() answered GL_NO_ERROR for a call
|
||||
// the spec requires to fail (KHR-GL30.api.coverage's glBlitFramebuffer sub-check). DirectVulkan
|
||||
// already dropped the bad-filter and LINEAR-with-depth/stencil calls on the floor with a log
|
||||
// line (VulkanRenderer::BlitFramebuffer), so the only thing that changes for it is that the
|
||||
// error is now visible where the spec says it should be.
|
||||
static Bool ValidateBlitMaskAndFilter(const char* functionName, GLbitfield mask, GLenum filter) {
|
||||
constexpr GLbitfield kBlitMaskBits = GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT | GL_STENCIL_BUFFER_BIT;
|
||||
if ((mask & ~kBlitMaskBits) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"mask contains bits other than GL_COLOR_BUFFER_BIT, "
|
||||
"GL_DEPTH_BUFFER_BIT and GL_STENCIL_BUFFER_BIT."));
|
||||
return false;
|
||||
}
|
||||
if (filter != GL_NEAREST && filter != GL_LINEAR) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"filter must be GL_NEAREST or GL_LINEAR."));
|
||||
return false;
|
||||
}
|
||||
// Depth and stencil have no meaningful interpolation, so GL_LINEAR is rejected outright
|
||||
// rather than downgraded - even when the mask also carries the colour bit.
|
||||
if (filter == GL_LINEAR && (mask & (GL_DEPTH_BUFFER_BIT | GL_STENCIL_BUFFER_BIT)) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"GL_LINEAR filtering is not allowed when mask includes "
|
||||
"GL_DEPTH_BUFFER_BIT or GL_STENCIL_BUFFER_BIT."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void BlitNamedFramebuffer(GLuint readFramebuffer, GLuint drawFramebuffer, GLint srcX0, GLint srcY0, GLint srcX1,
|
||||
GLint srcY1, GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1, GLbitfield mask,
|
||||
GLenum filter) {
|
||||
if (!ValidateBlitMaskAndFilter(__func__, mask, filter)) return;
|
||||
BlitNamedFramebuffer_State(readFramebuffer, drawFramebuffer, srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1,
|
||||
dstY1, mask, filter);
|
||||
}
|
||||
|
||||
void BlitFramebuffer(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1, GLint dstX0, GLint dstY0, GLint dstX1,
|
||||
GLint dstY1, GLbitfield mask, GLenum filter) {
|
||||
if (!ValidateBlitMaskAndFilter(__func__, mask, filter)) return;
|
||||
BlitFramebuffer_Backend(srcX0, srcY0, srcX1, srcY1, dstX0, dstY0, dstX1, dstY1, mask, filter);
|
||||
}
|
||||
|
||||
|
||||
@@ -10,12 +10,14 @@
|
||||
#include <cmath>
|
||||
#include <Config.h>
|
||||
#include <MGGitHash.h>
|
||||
#include <MG_Impl/GLImpl/Debug/GL_Debug.h>
|
||||
#include <MG_Impl/GLImpl/VertexArray/Validators.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/GLState/ErrorState/ErrorInfo.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/BufferEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/RenderStateEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/FramebufferEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/ErrorCodeConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
|
||||
@@ -24,9 +26,15 @@
|
||||
#include <MG_State/GLState/FramebufferState/FramebufferObject.h>
|
||||
#include <MG_Util/Texture/TextureFormatProcessor.h>
|
||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||
#include <MG_Util/ShaderTranspiler/Types.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
// Declared rather than #included from GL_RenderState.h on purpose: that header also declares
|
||||
// a free function named BlendEquation, which would hide the ::MobileGL::BlendEquation enum
|
||||
// this file's blend-state queries name unqualified.
|
||||
GLboolean IsEnabledi(GLenum target, GLuint index);
|
||||
|
||||
namespace {
|
||||
enum class IndexedBufferQueryKind {
|
||||
Binding,
|
||||
@@ -40,21 +48,47 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
constexpr GLint kFrontendMaxComputeUniformComponents = 1024;
|
||||
constexpr GLint kFrontendMaxComputeAtomicCounters = 8;
|
||||
constexpr GLint kFrontendMaxComputeAtomicCounterBuffers = 8;
|
||||
// Shared with the glslang resource table for the same reason as the atomic-counter
|
||||
// limits below: gl_MaxComputeUniformComponents expands from BuildTBuiltInResource.
|
||||
constexpr GLint kFrontendMaxComputeUniformComponents =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_COMPUTE_UNIFORM_COMPONENTS);
|
||||
// Every atomic-counter limit is shared with the glslang resource table
|
||||
// (BuildTBuiltInResource) through MG_Util/ShaderTranspiler/Types.h: GL 4.6 requires
|
||||
// glGetIntegerv and the gl_MaxAtomicCounter* built-in constants to agree, and the two
|
||||
// used to be independent tables that disagreed on both the binding count and the buffer
|
||||
// size. Never move one of these without the other.
|
||||
constexpr GLint kFrontendMaxComputeAtomicCounters =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTERS_PER_STAGE);
|
||||
constexpr GLint kFrontendMaxComputeAtomicCounterBuffers =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTER_BUFFERS_PER_STAGE);
|
||||
constexpr GLint kFrontendMaxComputeSharedMemorySize = 32768;
|
||||
constexpr GLint kFrontendMaxComputeWorkGroupInvocations = 1024;
|
||||
constexpr GLint kFrontendMaxCombinedAtomicCounters = 8;
|
||||
constexpr GLint kFrontendMaxFragmentAtomicCounters = 8;
|
||||
constexpr GLint kFrontendMaxCombinedAtomicCounters =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTERS_PER_STAGE);
|
||||
constexpr GLint kFrontendMaxCombinedAtomicCounterBuffers =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTER_BUFFERS_PER_STAGE);
|
||||
constexpr GLint kFrontendMaxFragmentAtomicCounters =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTERS_PER_STAGE);
|
||||
constexpr GLint kFrontendMaxFragmentAtomicCounterBuffers =
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTER_BUFFERS_PER_STAGE);
|
||||
constexpr GLint kFrontendMaxGeometryAtomicCounters = 0;
|
||||
constexpr GLint kFrontendMaxTessControlAtomicCounters = 0;
|
||||
constexpr GLint kFrontendMaxTessEvaluationAtomicCounters = 0;
|
||||
constexpr GLint kFrontendMaxVertexAtomicCounters = 0;
|
||||
// One atomic counter is a uint, and a buffer never has to hold more counters than the
|
||||
// combined limit the frontend advertises. GL 4.6 table 23.63 floors this at 32 bytes.
|
||||
// Zero counters means zero buffers to hold them. These have to be ANSWERED rather than
|
||||
// left to the default INVALID_ENUM: a well-behaved application queries the limit exactly
|
||||
// to find out that the stage cannot do this, and an error instead both leaves its output
|
||||
// untouched (so it reads uninitialised memory and may conclude the opposite) and leaves a
|
||||
// GL error pending that surfaces at whatever unrelated call checks next.
|
||||
constexpr GLint kFrontendMaxGeometryAtomicCounterBuffers = 0;
|
||||
constexpr GLint kFrontendMaxTessControlAtomicCounterBuffers = 0;
|
||||
constexpr GLint kFrontendMaxTessEvaluationAtomicCounterBuffers = 0;
|
||||
constexpr GLint kFrontendMaxVertexAtomicCounterBuffers = 0;
|
||||
// GL_MAX_ATOMIC_COUNTER_BUFFER_SIZE: the byte offset ceiling a counter may be declared
|
||||
// at. The matching binding count is applied in GetIndexedBufferQueryPointCount, so that
|
||||
// the getter, the indexed queries and glBindBufferBase all share one ceiling.
|
||||
constexpr GLint kFrontendMaxAtomicCounterBufferSize =
|
||||
kFrontendMaxCombinedAtomicCounters * static_cast<GLint>(sizeof(GLuint));
|
||||
static_cast<GLint>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTER_BUFFER_SIZE);
|
||||
// KHR_debug minima (GL 4.6 table 23.66); the debug entry points are stubs, but the
|
||||
// limits they advertise still have to be legal.
|
||||
constexpr GLint kFrontendMaxDebugGroupStackDepth = 64;
|
||||
@@ -88,12 +122,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
constexpr GLint kFrontendSubpixelBits = 4;
|
||||
constexpr GLint kFrontendMaxSamples = 4;
|
||||
|
||||
// The floors under GL_MAX_COMPUTE_WORK_GROUP_COUNT / _SIZE. Shared with the compile
|
||||
// pipeline (CaptureCompileEnv floors the same driver answers at them, and
|
||||
// BuildTBuiltInResource expands gl_MaxComputeWorkGroup* from the result), because a
|
||||
// shader is allowed to compare the built-in constant against this query.
|
||||
constexpr GLint GetMinComputeWorkGroupCount(GLuint index) {
|
||||
return index < 3 ? 65535 : 0;
|
||||
return index < 3 ? static_cast<GLint>(MG_Util::ShaderTranspiler::MIN_COMPUTE_WORK_GROUP_COUNT[index]) : 0;
|
||||
}
|
||||
|
||||
constexpr GLint GetMinComputeWorkGroupSize(GLuint index) {
|
||||
return index < 2 ? 1024 : (index == 2 ? 64 : 0);
|
||||
return index < 3 ? static_cast<GLint>(MG_Util::ShaderTranspiler::MIN_COMPUTE_WORK_GROUP_SIZE[index]) : 0;
|
||||
}
|
||||
|
||||
GLint GetMaxCombinedUniformComponents(GLint maxDefaultUniformComponents, GLint maxUniformBlocks,
|
||||
@@ -171,9 +209,60 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxShaderStorageBufferBindings;
|
||||
return std::min(frontendCount, static_cast<SizeT>(std::max(backendCount, 0)));
|
||||
}
|
||||
if (bufferTarget == BufferTarget::AtomicCounter) {
|
||||
// The counter family's binding count is NOT the state layer's array size: a
|
||||
// counter buffer only reaches a shader as a lowered storage block, so what an
|
||||
// implementation can serve is the reserved range, and that number is also what
|
||||
// glslang compiles a layout(binding = N) atomic_uint against. Clamped here so
|
||||
// GL_MAX_ATOMIC_COUNTER_BUFFER_BINDINGS, the indexed getters' index check and
|
||||
// glBindBufferBase's all report the same ceiling.
|
||||
return std::min(frontendCount,
|
||||
static_cast<SizeT>(MG_Util::ShaderTranspiler::MAX_ATOMIC_COUNTER_BUFFER_BINDINGS));
|
||||
}
|
||||
return frontendCount;
|
||||
}
|
||||
|
||||
// A per-stage or combined BLOCK count is an amount of indexed binding points an
|
||||
// application will occupy, and GL 4.6 table 23.64 orders the two accordingly:
|
||||
// MAX_UNIFORM_BUFFER_BINDINGS >= MAX_COMBINED_UNIFORM_BLOCKS >= every per-stage count,
|
||||
// and the same for the shader-storage family. The two families are answered from
|
||||
// unrelated places here - frontend constants, backend dynamic parameters, and a few
|
||||
// hard-coded TODOs - so nothing kept them ordered, and a backend that reports Vulkan
|
||||
// descriptor-indexing counts advertised 256 compute uniform blocks over 36 binding
|
||||
// points. KHR-GL44.multi_bind.dispatch_bind_buffers_base reads the block count and binds
|
||||
// that many buffers in ONE glBindBuffersBase, which is then INVALID_OPERATION before it
|
||||
// binds anything. Clamping is the only direction available: the binding count is the
|
||||
// capacity of the state layer's indexed-binding array, not a number we may inflate.
|
||||
GLint ClampBlockCountToBindingPoints(GLint blockCount, BufferTarget bufferTarget) {
|
||||
const GLint bindingPoints = static_cast<GLint>(GetIndexedBufferQueryPointCount(bufferTarget));
|
||||
return std::min(std::max(blockCount, 0), bindingPoints);
|
||||
}
|
||||
|
||||
GLint ClampUniformBlockCount(GLint blockCount) {
|
||||
return ClampBlockCountToBindingPoints(blockCount, BufferTarget::Uniform);
|
||||
}
|
||||
|
||||
GLint ClampStorageBlockCount(GLint blockCount) {
|
||||
return ClampBlockCountToBindingPoints(blockCount, BufferTarget::ShaderStorage);
|
||||
}
|
||||
|
||||
// The per-stage GL_MAX_*_SHADER_STORAGE_BLOCKS answers. Backend-derived, and NOT a
|
||||
// constant to be "restored" - these used to return a flat 16 for vertex, geometry and
|
||||
// both tessellation stages, which is wrong on any host that does not serve storage
|
||||
// blocks in those stages. Zero is a legal answer: GL 4.6 table 23.64 and ES 3.2 table
|
||||
// 21.44 both set the minimum at 0 for every graphics stage except fragment, which is
|
||||
// why the conformance suite gates each such test on the query instead of assuming it.
|
||||
// ARM's GLES driver reports 0 for all four (a Mali-G925 does), and advertising 16 there
|
||||
// bought nothing: the program still failed to link inside the backend, the frontend
|
||||
// still reported LINK_STATUS as true, and every draw with it silently rendered nothing.
|
||||
GLint StageStorageBlockCount(Int MG_Backend::DynamicBackendParameters::*stageLimit) {
|
||||
static const MG_Backend::DynamicBackendParameters kBackendlessDefaults{};
|
||||
const MG_Backend::DynamicBackendParameters& parameters =
|
||||
MG_Backend::pActiveBackendObject ? MG_Backend::pActiveBackendObject->GetDynamicParameters()
|
||||
: kBackendlessDefaults;
|
||||
return ClampStorageBlockCount(static_cast<GLint>(parameters.*stageLimit));
|
||||
}
|
||||
|
||||
bool TryDecodeDrawBufferQuery(GLenum pname, SizeT& drawBufferIndex) {
|
||||
if (pname == GL_DRAW_BUFFER) {
|
||||
drawBufferIndex = 0;
|
||||
@@ -306,26 +395,70 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return sampler ? static_cast<GLint>(sampler->GetExternalIndex()) : 0;
|
||||
}
|
||||
|
||||
// The ARB_viewport_array indexed rectangles. MobileGL keeps exactly one viewport, one
|
||||
// scissor box and one depth range, so every in-range index answers with that single
|
||||
// value - but it has to come from the frontend state the non-indexed getters read.
|
||||
// The generic path at the bottom of GetIntegeri_v is a raw backend passthrough that
|
||||
// has no case for these, so routing them through it returned zeros.
|
||||
// The ARB_viewport_array indexed rectangles. Each of these is genuinely per-viewport
|
||||
// frontend state (RenderStateParameters::Viewports / ScissorBoxes / DepthRanges), so the
|
||||
// indexed getters must read the indexed storage - the generic path at the bottom of
|
||||
// GetIntegeri_v is a raw backend passthrough that has no case for them and returned
|
||||
// zeros, and routing them to the NON-indexed getter (what this used to do) answered every
|
||||
// index with viewport 0's value, which is what
|
||||
// KHR-GL43.viewport_array.{viewport,scissor,depth_range}_api caught.
|
||||
Bool IsIndexedViewportQuery(GLenum target) {
|
||||
return target == GL_VIEWPORT || target == GL_SCISSOR_BOX || target == GL_DEPTH_RANGE;
|
||||
}
|
||||
|
||||
// ARB_viewport_array: `index` selects a viewport and MAX_VIEWPORTS bounds it.
|
||||
// Component count of an indexed viewport-array query, so every width of getter writes the
|
||||
// caller's whole buffer instead of just element 0 (GL 4.6 core 22.1).
|
||||
GLsizei IndexedViewportQueryComponents(GLenum target) {
|
||||
return target == GL_DEPTH_RANGE ? 2 : 4;
|
||||
}
|
||||
|
||||
// ARB_viewport_array: `index` selects a viewport and MAX_VIEWPORTS bounds it. The bound is
|
||||
// the frontend's own state width, which is also exactly what GL_MAX_VIEWPORTS reports -
|
||||
// taking it from the backend caps instead would let a device limit of 1 (a Vulkan device
|
||||
// without the multiViewport feature) make index 1 illegal even though the state exists.
|
||||
Bool ValidateViewportQueryIndex(GLuint index, const char* caller) {
|
||||
GLint maxViewports = 0;
|
||||
GetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
|
||||
if (index < static_cast<GLuint>(std::max(maxViewports, 1))) return true;
|
||||
if (index < RenderStateParameters::MAX_VIEWPORTS) return true;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller, "Viewport index is out of range."));
|
||||
return false;
|
||||
}
|
||||
|
||||
// The indexed viewport/scissor/depth-range state as floats, which is the widest lossless
|
||||
// shape MobileGL stores (the viewport really is float state; the scissor box is integral
|
||||
// and well inside float's exact range, and every depth range is in [0, 1]). Every indexed
|
||||
// getter width funnels through this so they can never disagree with each other.
|
||||
void ReadIndexedViewportStateFloat(GLenum target, GLuint index, GLfloat* out) {
|
||||
switch (target) {
|
||||
case GL_VIEWPORT: {
|
||||
const FloatVec4& viewport = MG_State::pGLContext->GetViewportIndexed(index);
|
||||
out[0] = viewport.x();
|
||||
out[1] = viewport.y();
|
||||
out[2] = viewport.z();
|
||||
out[3] = viewport.w();
|
||||
return;
|
||||
}
|
||||
case GL_SCISSOR_BOX: {
|
||||
const IntVec4& box = MG_State::pGLContext->GetScissorBoxIndexed(index);
|
||||
out[0] = static_cast<GLfloat>(box.x());
|
||||
out[1] = static_cast<GLfloat>(box.y());
|
||||
out[2] = static_cast<GLfloat>(box.z());
|
||||
out[3] = static_cast<GLfloat>(box.w());
|
||||
return;
|
||||
}
|
||||
case GL_DEPTH_RANGE: {
|
||||
const FloatVec2& range = MG_State::pGLContext->GetDepthRangeIndexed(index);
|
||||
out[0] = range.x();
|
||||
out[1] = range.y();
|
||||
return;
|
||||
}
|
||||
default:
|
||||
MOBILEGL_ASSERT(false, "ReadIndexedViewportStateFloat: unexpected target 0x%x",
|
||||
static_cast<Uint32>(target));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
void CopyIntsToBooleans(const GLint* src, SizeT count, GLboolean* dst) {
|
||||
for (SizeT i = 0; i < count; ++i) {
|
||||
dst[i] = src[i] ? GL_TRUE : GL_FALSE;
|
||||
@@ -339,6 +472,18 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// GL 4.6 core table 23.53 requires GL_MAX_SAMPLES >= 4, so the driver's value is floored
|
||||
// before it is advertised. Every other multisample ceiling MobileGL advertises has to be
|
||||
// floored the same way: promising 4 samples globally while answering GL_MAX_INTEGER_SAMPLES
|
||||
// 1 - which is exactly what Adreno reports - makes the frontend reject the very count it
|
||||
// just told the application to use. The backends clamp the realised count instead.
|
||||
GLint GetAdvertisedMaxSamples() {
|
||||
if (MG_Backend::pActiveBackendObject == nullptr) {
|
||||
return kFrontendMaxSamples;
|
||||
}
|
||||
return std::max(MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxSamples, kFrontendMaxSamples);
|
||||
}
|
||||
|
||||
/* @INSERTION_POINT:FUNCTION_IMPLEMENTATION@ */
|
||||
const GLubyte* GetString(GLenum name) {
|
||||
static String vendorString;
|
||||
@@ -350,7 +495,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
MGLOG_D("glGetString, name: %s", MG_Util::ConvertGLEnumToString(name).c_str());
|
||||
if (!activeBackendObject) {
|
||||
MGLOG_E("activeBackendObject is not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject is not initialized!");
|
||||
return (GLubyte*)"Unknown";
|
||||
}
|
||||
|
||||
@@ -409,7 +554,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackendObject) {
|
||||
MGLOG_E("activeBackendObject is not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject is not initialized!");
|
||||
return (GLubyte*)"Unknown";
|
||||
}
|
||||
const auto& rendererInfo = activeBackendObject->GetRendererInfo();
|
||||
@@ -596,6 +741,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
params[1] = dynamicParameters.ViewportBoundsRangeMax;
|
||||
return;
|
||||
}
|
||||
// Viewport 0's rectangle, verbatim. Falling through to the integer width below would
|
||||
// round the fractional rectangle a glViewportIndexedf(0, ...) is allowed to set, and
|
||||
// glGetFloatv(GL_VIEWPORT) is a lossless query of float state.
|
||||
case GL_VIEWPORT: {
|
||||
const FloatVec4& viewport = MG_State::pGLContext->GetViewportIndexed(0);
|
||||
params[0] = viewport.x();
|
||||
params[1] = viewport.y();
|
||||
params[2] = viewport.z();
|
||||
params[3] = viewport.w();
|
||||
return;
|
||||
}
|
||||
case GL_MIN_FRAGMENT_INTERPOLATION_OFFSET:
|
||||
case GL_MAX_FRAGMENT_INTERPOLATION_OFFSET:
|
||||
case GL_FRAGMENT_INTERPOLATION_OFFSET_BITS: {
|
||||
@@ -722,10 +878,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
// GL 4.6 core table 23.4/23.5: *_BUFFER_SIZE reports the size glBindBufferRange
|
||||
// was ASKED for, verbatim. It is not clamped to the buffer's storage, and it does
|
||||
// not follow the buffer when a later glBufferData resizes it - a range may legally
|
||||
// name bytes the buffer does not have yet. Clamping it here answered 0 for the
|
||||
// common conformance shape of binding a range on a buffer that has no storage
|
||||
// yet (KHR-GL43.shader_storage_buffer_object.basic-binding).
|
||||
const Range1D range = bindingPoint.GetRange();
|
||||
const auto start = std::min(range.start, bufferObject->GetSize());
|
||||
const auto end = std::min(range.end, bufferObject->GetSize());
|
||||
*data = static_cast<GLint>(end - start);
|
||||
*data = static_cast<GLint>(range.end - range.start);
|
||||
return;
|
||||
}
|
||||
default:
|
||||
@@ -755,15 +915,32 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
// GL 4.6 core 22.1: an indexed query answers EVERY indexed state, and GL_SCISSOR_TEST is
|
||||
// indexed by viewport just like GL_BLEND is by draw buffer. Without this the integer
|
||||
// width fell through to the backend passthrough and answered GL_INVALID_ENUM, which is
|
||||
// the sticky error KHR-GL43.viewport_array.queries trips over at its next error check.
|
||||
if (MG_Util::ConvertGLEnumToCapabilityInput(target) != CapabilityInput::Unknown) {
|
||||
*data = IsEnabledi(target, index);
|
||||
return;
|
||||
}
|
||||
|
||||
switch (target) {
|
||||
// ARB_viewport_array queries the indexed rectangles through glGetIntegeri_v as well
|
||||
// (gl4cMultiBindTests and the viewport_array group both do). The frontend keeps one
|
||||
// viewport and one scissor box, so every in-range index reports that one.
|
||||
// (gl4cMultiBindTests and the viewport_array group both do).
|
||||
case GL_VIEWPORT:
|
||||
case GL_SCISSOR_BOX:
|
||||
case GL_DEPTH_RANGE: {
|
||||
if (!ValidateViewportQueryIndex(index, __func__)) return;
|
||||
GetIntegerv(target, data);
|
||||
GLfloat values[4] = {};
|
||||
ReadIndexedViewportStateFloat(target, index, values);
|
||||
const GLsizei components = IndexedViewportQueryComponents(target);
|
||||
for (GLsizei i = 0; i < components; ++i) {
|
||||
// Round, not truncate: glGetIntegerv on floating-point state rounds to nearest
|
||||
// (GL 4.6 core 22.2), so a 255.875-wide viewport reads back as 256 and not 255.
|
||||
data[i] = static_cast<GLint>(std::lround(values[i]));
|
||||
}
|
||||
return;
|
||||
}
|
||||
// The vertex buffer binding points of the vertex array object that is bound. Indexed by
|
||||
// binding point, not by attribute (GL 4.6 core 10.3.1).
|
||||
case GL_VERTEX_BINDING_BUFFER:
|
||||
@@ -890,7 +1067,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
if (IsIndexedViewportQuery(target)) {
|
||||
if (!ValidateViewportQueryIndex(index, __func__)) return;
|
||||
GetFloatv(target, data);
|
||||
// Verbatim, NOT via the integer width: the viewport is float state and
|
||||
// KHR-GL43.viewport_array.viewport_api compares the read-back with ==, so a
|
||||
// glViewportIndexedf(i, 0.125f, ...) has to come back as 0.125f exactly.
|
||||
ReadIndexedViewportStateFloat(target, index, data);
|
||||
return;
|
||||
}
|
||||
GLint ints[4] = {};
|
||||
@@ -907,7 +1087,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
if (IsIndexedViewportQuery(target)) {
|
||||
if (!ValidateViewportQueryIndex(index, __func__)) return;
|
||||
GetDoublev(target, data);
|
||||
GLfloat values[4] = {};
|
||||
ReadIndexedViewportStateFloat(target, index, values);
|
||||
const GLsizei components = IndexedViewportQueryComponents(target);
|
||||
for (GLsizei i = 0; i < components; ++i) {
|
||||
data[i] = static_cast<GLdouble>(values[i]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
GLint ints[4] = {};
|
||||
@@ -951,9 +1136,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*data = 0;
|
||||
return;
|
||||
}
|
||||
const auto start = std::min(range.start, bufferObject->GetSize());
|
||||
const auto end = std::min(range.end, bufferObject->GetSize());
|
||||
*data = static_cast<GLint64>(end - start);
|
||||
// Verbatim, unclamped - see the GetIntegeri_v arm.
|
||||
*data = static_cast<GLint64>(range.end - range.start);
|
||||
return;
|
||||
}
|
||||
default:
|
||||
@@ -961,15 +1145,35 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
auto getInteger64i = MG_Backend::gBackendFunctionsTable.GL.GetInteger64i_v;
|
||||
if (!getInteger64i) {
|
||||
*data = 0;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "Backend does not support indexed integer queries."));
|
||||
// The one indexed pname whose value genuinely needs 64 bits: a vertex buffer binding
|
||||
// offset is an intptr, so taking the 32-bit route below would truncate it.
|
||||
if (target == GL_VERTEX_BINDING_OFFSET) {
|
||||
if (index >= VertexArrayImpl::GetMaxVertexAttribBindings()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"Vertex buffer binding index is out of range."));
|
||||
return;
|
||||
}
|
||||
const auto& vao = MG_State::pGLContext->GetBoundVertexArray();
|
||||
*data = vao ? static_cast<GLint64>(vao->GetBindingPoint(index).Offset) : 0;
|
||||
return;
|
||||
}
|
||||
getInteger64i(target, index, data);
|
||||
|
||||
// Everything else is 32-bit indexed state that the glGetIntegeri_v pname table already
|
||||
// owns, and GL 4.6 core 22.1 says every indexed query answers every indexed pname.
|
||||
// Handing the leftovers straight to the backend instead made glGetInteger64i_v disagree
|
||||
// with glGetIntegeri_v on the very same pname - GL_MAX_COMPUTE_WORK_GROUP_COUNT read
|
||||
// back 0 while the 32-bit view said 65535 (KHR-GL43.compute_shader.max), because a
|
||||
// frontend-only value simply is not in the driver's table.
|
||||
GLint values[4] = {};
|
||||
GetIntegeri_v(target, index, values);
|
||||
// The viewport-array rectangles are the only multi-component indexed state here; every
|
||||
// other pname is scalar, so widening element 0 alone would silently truncate them.
|
||||
const GLsizei components = IsIndexedViewportQuery(target) ? IndexedViewportQueryComponents(target) : 1;
|
||||
for (GLsizei i = 0; i < components; ++i) {
|
||||
data[i] = static_cast<GLint64>(values[i]);
|
||||
}
|
||||
}
|
||||
|
||||
void GetInteger64v(GLenum pname, GLint64* params) {
|
||||
@@ -1224,19 +1428,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
: 0;
|
||||
return;
|
||||
case GL_MAX_DEBUG_GROUP_STACK_DEPTH:
|
||||
// KHR_debug floors this at 64 even when the group entry points are stubs: the
|
||||
// limit describes how deep glPushDebugGroup may nest, and 0 is not a legal answer.
|
||||
// KHR_debug floors this at 64. It must agree with what GL_Debug.cpp actually enforces,
|
||||
// or an application that nests to the reported limit would take a STACK_OVERFLOW.
|
||||
*params = kFrontendMaxDebugGroupStackDepth;
|
||||
return;
|
||||
case GL_MAX_DEBUG_MESSAGE_LENGTH:
|
||||
*params = 1024; // debug-message entrypoints are stubbed, but KHR_debug requires a valid limit
|
||||
*params = 1024; // agrees with GL_Debug.cpp's kMaxDebugMessageLength
|
||||
return;
|
||||
case GL_MAX_DEBUG_LOGGED_MESSAGES:
|
||||
// Size of the message log ring; KHR_debug requires at least 1.
|
||||
*params = kFrontendMaxDebugLoggedMessages;
|
||||
return;
|
||||
case GL_DEBUG_GROUP_STACK_DEPTH:
|
||||
*params = 0; // debug-group entrypoints are stubbed
|
||||
// The live depth, which is never 0: GL 4.6 core 20.6 creates the context with one
|
||||
// group already on the stack, and that is the one glPopDebugGroup may not pop.
|
||||
*params = GetDebugGroupStackDepth();
|
||||
return;
|
||||
case GL_CONTEXT_FLAGS: {
|
||||
*params = MG_State::pEGLContext ? MG_State::pEGLContext->GetCurrentContextFlags() : 0;
|
||||
@@ -1369,17 +1575,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_LINE_WIDTH:
|
||||
*params = static_cast<GLint>(MG_State::pGLContext->GetLineWidth());
|
||||
return;
|
||||
case GL_LAYER_PROVOKING_VERTEX:
|
||||
*params = GL_LAST_VERTEX_CONVENTION;
|
||||
return;
|
||||
case GL_LOGIC_OP_MODE:
|
||||
*params = static_cast<GLint>(MG_Util::ConvertLogicOperationToGLEnum(MG_State::pGLContext->GetLogicOp()));
|
||||
return;
|
||||
case GL_MAX_COMBINED_ATOMIC_COUNTERS:
|
||||
*params = kFrontendMaxCombinedAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_COMBINED_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = kFrontendMaxCombinedAtomicCounterBuffers;
|
||||
return;
|
||||
case GL_MAX_COMBINED_UNIFORM_BLOCKS:
|
||||
*params = kFrontendMaxCombinedUniformBlocks;
|
||||
*params = ClampUniformBlockCount(kFrontendMaxCombinedUniformBlocks);
|
||||
return;
|
||||
case GL_MAX_DUAL_SOURCE_DRAW_BUFFERS:
|
||||
*params = 1; // TODO
|
||||
@@ -1393,8 +1599,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_FRAGMENT_ATOMIC_COUNTERS:
|
||||
*params = kFrontendMaxFragmentAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_FRAGMENT_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = kFrontendMaxFragmentAtomicCounterBuffers;
|
||||
return;
|
||||
case GL_MAX_FRAGMENT_SHADER_STORAGE_BLOCKS:
|
||||
*params = 16; // TODO
|
||||
*params = StageStorageBlockCount(&MG_Backend::DynamicBackendParameters::MaxFragmentShaderStorageBlocks);
|
||||
return;
|
||||
case GL_MAX_FRAGMENT_INPUT_COMPONENTS:
|
||||
*params = kFrontendMaxFragmentInputComponents;
|
||||
@@ -1411,13 +1620,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxFragmentUniformVectors;
|
||||
return;
|
||||
case GL_MAX_FRAGMENT_UNIFORM_BLOCKS:
|
||||
*params = kFrontendMaxFragmentUniformBlocks;
|
||||
*params = ClampUniformBlockCount(kFrontendMaxFragmentUniformBlocks);
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_ATOMIC_COUNTERS:
|
||||
*params = kFrontendMaxGeometryAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = kFrontendMaxGeometryAtomicCounterBuffers;
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_SHADER_STORAGE_BLOCKS:
|
||||
*params = 16; // TODO
|
||||
*params = StageStorageBlockCount(&MG_Backend::DynamicBackendParameters::MaxGeometryShaderStorageBlocks);
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_INPUT_COMPONENTS:
|
||||
*params = kFrontendMaxGeometryInputComponents;
|
||||
@@ -1440,7 +1652,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxGeometryTotalOutputComponents;
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_UNIFORM_BLOCKS:
|
||||
*params = kFrontendMaxGeometryUniformBlocks;
|
||||
*params = ClampUniformBlockCount(kFrontendMaxGeometryUniformBlocks);
|
||||
return;
|
||||
case GL_MAX_GEOMETRY_UNIFORM_COMPONENTS:
|
||||
*params = kFrontendMaxGeometryUniformComponents;
|
||||
@@ -1452,7 +1664,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::Multisample) ? GL_TRUE : GL_FALSE;
|
||||
return;
|
||||
case GL_MIN_MAP_BUFFER_ALIGNMENT:
|
||||
*params = 64; // TODO
|
||||
// The same constant the map paths align to (MG_State/GLState/BufferState/
|
||||
// PipeResource.h), never a literal: this number is a PROMISE about the pointers
|
||||
// glMapBuffer and glMapBufferRange return, and the two used to be unrelated - the
|
||||
// query said 64 while the pointers came out of a std::vector aligned to 16.
|
||||
*params = static_cast<GLint>(MG_State::GLState::MIN_MAP_BUFFER_ALIGNMENT);
|
||||
return;
|
||||
case GL_MAX_LABEL_LENGTH:
|
||||
*params = 256; // TODO
|
||||
@@ -1472,9 +1688,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_TESS_CONTROL_ATOMIC_COUNTERS:
|
||||
*params = kFrontendMaxTessControlAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = kFrontendMaxTessControlAtomicCounterBuffers;
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_ATOMIC_COUNTERS:
|
||||
*params = kFrontendMaxTessEvaluationAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = kFrontendMaxTessEvaluationAtomicCounterBuffers;
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_IMAGE_UNIFORMS:
|
||||
*params = 0;
|
||||
return;
|
||||
@@ -1482,16 +1704,18 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = 0;
|
||||
return;
|
||||
case GL_MAX_TESS_CONTROL_SHADER_STORAGE_BLOCKS:
|
||||
*params = 16; // TODO
|
||||
*params = StageStorageBlockCount(&MG_Backend::DynamicBackendParameters::MaxTessControlShaderStorageBlocks);
|
||||
return;
|
||||
case GL_MAX_TESS_EVALUATION_SHADER_STORAGE_BLOCKS:
|
||||
*params = 16; // TODO
|
||||
*params =
|
||||
StageStorageBlockCount(&MG_Backend::DynamicBackendParameters::MaxTessEvaluationShaderStorageBlocks);
|
||||
return;
|
||||
case GL_MAX_TEXTURE_LOD_BIAS:
|
||||
*params = 15; // TODO
|
||||
return;
|
||||
case GL_MAX_UNIFORM_LOCATIONS:
|
||||
*params = 1024 * 4; // TODO
|
||||
// The same constant the link's location allocator enforces - see ProgramObject.
|
||||
*params = MG_State::GLState::ProgramObject::MAX_UNIFORM_LOCATIONS;
|
||||
return;
|
||||
case GL_MAX_VARYING_COMPONENTS:
|
||||
*params = kFrontendMaxVaryingComponents;
|
||||
@@ -1502,13 +1726,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_VERTEX_ATOMIC_COUNTERS:
|
||||
*params = kFrontendMaxVertexAtomicCounters;
|
||||
return;
|
||||
case GL_MAX_VERTEX_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = kFrontendMaxVertexAtomicCounterBuffers;
|
||||
return;
|
||||
case GL_MAX_VERTEX_IMAGE_UNIFORMS:
|
||||
*params = MG_Backend::pActiveBackendObject
|
||||
? MG_Backend::pActiveBackendObject->GetDynamicParameters().MaxVertexImageUniforms
|
||||
: MG_Backend::DynamicBackendParameters{}.MaxVertexImageUniforms;
|
||||
return;
|
||||
case GL_MAX_VERTEX_SHADER_STORAGE_BLOCKS:
|
||||
*params = 16; // TODO
|
||||
*params = StageStorageBlockCount(&MG_Backend::DynamicBackendParameters::MaxVertexShaderStorageBlocks);
|
||||
return;
|
||||
case GL_MAX_VERTEX_UNIFORM_COMPONENTS:
|
||||
*params = kFrontendMaxVertexUniformComponents;
|
||||
@@ -1520,7 +1747,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = kFrontendMaxVertexOutputComponents;
|
||||
return;
|
||||
case GL_MAX_VERTEX_UNIFORM_BLOCKS:
|
||||
*params = kFrontendMaxVertexUniformBlocks;
|
||||
*params = ClampUniformBlockCount(kFrontendMaxVertexUniformBlocks);
|
||||
return;
|
||||
case GL_NUM_COMPRESSED_TEXTURE_FORMATS:
|
||||
*params = 0; // compressed texture upload entrypoints are still unimplemented
|
||||
@@ -1818,6 +2045,24 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_UNIFORM_BUFFER_START:
|
||||
RecordIndexedOnlyGetterError(__func__, pname);
|
||||
return;
|
||||
// glBindBufferBase/Range set the GENERIC binding point too (GL 4.6 core 6.1.1), and this
|
||||
// is the one indexed-buffer family whose non-indexed query was never answered - so it
|
||||
// fell through to INVALID_ENUM and left the caller's variable holding whatever was in its
|
||||
// stack slot. _START/_SIZE stay indexed-only, exactly like their uniform-buffer siblings.
|
||||
case GL_ATOMIC_COUNTER_BUFFER_BINDING:
|
||||
if (const auto& obj =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::AtomicCounter).GetBoundObject()) {
|
||||
*params = static_cast<GLint>(obj->GetExternalIndex());
|
||||
} else {
|
||||
*params = 0;
|
||||
}
|
||||
return;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_START:
|
||||
RecordIndexedOnlyGetterError(__func__, pname);
|
||||
return;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_SIZE:
|
||||
RecordIndexedOnlyGetterError(__func__, pname);
|
||||
return;
|
||||
case GL_UNPACK_ALIGNMENT:
|
||||
*params = MG_State::pGLContext->GetPixelStoreParam(PixelStoreParam::UnpackAlignment);
|
||||
return;
|
||||
@@ -1872,9 +2117,6 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
params[3] = vp.w();
|
||||
return;
|
||||
}
|
||||
case GL_VIEWPORT_INDEX_PROVOKING_VERTEX:
|
||||
*params = GL_LAST_VERTEX_CONVENTION;
|
||||
return;
|
||||
case GL_MAX_ELEMENT_INDEX:
|
||||
*params = 1024 * 1024; // TODO
|
||||
return;
|
||||
@@ -1891,7 +2133,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackendObject) {
|
||||
MGLOG_E("activeBackendObject is not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject is not initialized!");
|
||||
return;
|
||||
}
|
||||
const auto& rendererInfo = activeBackendObject->GetRendererInfo();
|
||||
@@ -1920,13 +2162,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = dynamicParameters.SubgroupQuadOperationsInAllStages ? GL_TRUE : GL_FALSE;
|
||||
break;
|
||||
case GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS:
|
||||
*params = dynamicParameters.MaxComputeShaderStorageBlocks;
|
||||
*params = ClampStorageBlockCount(dynamicParameters.MaxComputeShaderStorageBlocks);
|
||||
break;
|
||||
case GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS:
|
||||
*params = dynamicParameters.MaxCombinedShaderStorageBlocks;
|
||||
*params = ClampStorageBlockCount(dynamicParameters.MaxCombinedShaderStorageBlocks);
|
||||
break;
|
||||
case GL_MAX_COMPUTE_UNIFORM_BLOCKS:
|
||||
*params = dynamicParameters.MaxComputeUniformBlocks;
|
||||
*params = ClampUniformBlockCount(dynamicParameters.MaxComputeUniformBlocks);
|
||||
break;
|
||||
case GL_MAX_COMPUTE_TEXTURE_IMAGE_UNITS:
|
||||
*params = dynamicParameters.MaxComputeTextureImageUnits;
|
||||
@@ -1962,8 +2204,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_MAX_CLIP_DISTANCES:
|
||||
*params = dynamicParameters.MaxClipDistances;
|
||||
break;
|
||||
// Both were a hard-coded GL_LAST_VERTEX_CONVENTION, derived from nothing. GL 4.6 table
|
||||
// 23.65 permits GL_UNDEFINED_VERTEX for either, and that is what the backends report
|
||||
// wherever they do not actually pin a convention - claiming one is a statement about
|
||||
// which vertex of a primitive supplies gl_Layer / gl_ViewportIndex, and DirectGLES
|
||||
// rasterizes only viewport 0 on a driver without GL_OES_viewport_array while
|
||||
// DirectVulkan picks its provoking mode per pipeline. KHR-GLxx.viewport_array.query
|
||||
// accepts all four values, and .provoking_vertex - which failed on both devices, in
|
||||
// OPPOSITE directions - stops verifying as soon as either answer is undefined.
|
||||
case GL_LAYER_PROVOKING_VERTEX:
|
||||
*params = static_cast<GLint>(dynamicParameters.LayerProvokingVertex);
|
||||
break;
|
||||
case GL_VIEWPORT_INDEX_PROVOKING_VERTEX:
|
||||
*params = static_cast<GLint>(dynamicParameters.ViewportIndexProvokingVertex);
|
||||
break;
|
||||
case GL_MAX_COLOR_TEXTURE_SAMPLES:
|
||||
*params = dynamicParameters.MaxColorTextureSamples;
|
||||
*params = std::max(dynamicParameters.MaxColorTextureSamples, GetAdvertisedMaxSamples());
|
||||
break;
|
||||
case GL_MAX_COMBINED_FRAGMENT_UNIFORM_COMPONENTS:
|
||||
*params = GetMaxCombinedUniformComponents(kFrontendMaxFragmentUniformComponents,
|
||||
@@ -1993,7 +2249,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = dynamicParameters.MaxCubeMapTextureSize;
|
||||
break;
|
||||
case GL_MAX_DEPTH_TEXTURE_SAMPLES:
|
||||
*params = dynamicParameters.MaxDepthTextureSamples;
|
||||
*params = std::max(dynamicParameters.MaxDepthTextureSamples, GetAdvertisedMaxSamples());
|
||||
break;
|
||||
case GL_MAX_FRAMEBUFFER_WIDTH:
|
||||
*params = dynamicParameters.MaxFramebufferWidth;
|
||||
@@ -2020,7 +2276,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = dynamicParameters.MaxComputeImageUniforms;
|
||||
break;
|
||||
case GL_MAX_INTEGER_SAMPLES:
|
||||
*params = dynamicParameters.MaxIntegerSamples;
|
||||
*params = std::max(dynamicParameters.MaxIntegerSamples, GetAdvertisedMaxSamples());
|
||||
break;
|
||||
case GL_MAX_RENDERBUFFER_SIZE:
|
||||
*params = dynamicParameters.MaxRenderbufferSize;
|
||||
@@ -2053,9 +2309,18 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
static_cast<Uint64>(INT32_MAX)));
|
||||
break;
|
||||
case GL_MAX_ATOMIC_COUNTER_BUFFER_BINDINGS:
|
||||
// NOT the frontend's binding-point array size: GetIndexedBufferQueryPointCount
|
||||
// clamps this family to the range a lowered counter block can actually be served
|
||||
// from, which is the same number glslang compiles a layout(binding = N) atomic_uint
|
||||
// against and the same one glBindBufferBase validates an index against.
|
||||
*params = static_cast<GLint>(GetIndexedBufferQueryPointCount(BufferTarget::AtomicCounter));
|
||||
break;
|
||||
case GL_MAX_ATOMIC_COUNTER_BUFFER_SIZE:
|
||||
// The conformance suite splits this evenly across every advertised binding point and
|
||||
// binds all of them in one glBindBuffersRange
|
||||
// (KHR-GL44.multi_bind.functional_bind_buffers_range), so the pair has to divide -
|
||||
// a zero-sized range is INVALID_VALUE before BindBufferRange binds anything. The
|
||||
// shared constant is 16384 over 8 binding points, which divides.
|
||||
*params = kFrontendMaxAtomicCounterBufferSize;
|
||||
break;
|
||||
case GL_MAX_TEXTURE_BUFFER_SIZE:
|
||||
@@ -2121,7 +2386,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
params[1] = dynamicParameters.MaxViewportHeight;
|
||||
break;
|
||||
case GL_MAX_VIEWPORTS:
|
||||
*params = dynamicParameters.MaxViewports;
|
||||
// The frontend's own state width, not the backend's device limit. GL 4.3 core
|
||||
// requires MAX_VIEWPORTS >= 16 and every indexed viewport entry point validates
|
||||
// against RenderStateParameters::MAX_VIEWPORTS, so reporting anything else would
|
||||
// either advertise viewports the state cannot hold or reject indices it can. A
|
||||
// Vulkan device without the multiViewport feature reports maxViewports == 1, which
|
||||
// limits what can be RASTERIZED to more than one rectangle (see the multiViewport
|
||||
// gate in VulkanRenderer), not what the GL state can hold; caps.MaxViewports keeps
|
||||
// carrying that device number for exactly that decision.
|
||||
*params = static_cast<GLint>(RenderStateParameters::MAX_VIEWPORTS);
|
||||
break;
|
||||
case GL_MINOR_VERSION:
|
||||
*params = rendererInfo.RendererGLInfo.TargetGLVersion.Minor;
|
||||
@@ -2133,7 +2406,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = static_cast<GLint>(dynamicParameters.PointSizeGranularity);
|
||||
break;
|
||||
case GL_SHADER_STORAGE_BUFFER_OFFSET_ALIGNMENT:
|
||||
*params = static_cast<GLint>(dynamicParameters.UniformBufferOffsetAlignment);
|
||||
// The STORAGE alignment, which is its own limit - this used to answer with the
|
||||
// uniform one. They differ on real hardware (Adreno 830: 32 uniform, 64 storage), and
|
||||
// under-reporting it is silent: ValidateBindBufferRange accepts the offset, the ES
|
||||
// driver accepts it too without raising an error, and the shader's writes then land
|
||||
// at an address the application never bound.
|
||||
*params = static_cast<GLint>(dynamicParameters.ShaderStorageBufferOffsetAlignment);
|
||||
break;
|
||||
case GL_SMOOTH_LINE_WIDTH_RANGE:
|
||||
params[0] = static_cast<GLint>(dynamicParameters.SmoothLineWidthRangeMin);
|
||||
@@ -2170,14 +2448,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
: dynamicParameters.MaxDrawBuffers;
|
||||
break;
|
||||
case GL_MAX_SAMPLES:
|
||||
*params = std::max(dynamicParameters.MaxSamples, kFrontendMaxSamples);
|
||||
*params = GetAdvertisedMaxSamples();
|
||||
break;
|
||||
case GL_MAX_TEXTURE_MAX_ANISOTROPY_EXT:
|
||||
// Float state (see GetFloatv); rounded to nearest for the integer query per GL 3.3 6.1.2.
|
||||
*params = static_cast<GLint>(std::lround(dynamicParameters.MaxTextureMaxAnisotropy));
|
||||
break;
|
||||
default:
|
||||
MGLOG_E("glGetIntegerv: Invalid enum %s (0x%X)", MG_Util::ConvertGLEnumToString(pname).c_str(), pname);
|
||||
MGLOG_D("glGetIntegerv: Invalid enum %s (0x%X)", MG_Util::ConvertGLEnumToString(pname).c_str(), pname);
|
||||
MG_State::pGLContext->RecordError(ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "GetIntegerv",
|
||||
std::format("Invalid enum: 0x{:X}", pname)));
|
||||
|
||||
@@ -24,4 +24,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
GLenum GetError();
|
||||
GLenum GetGraphicsResetStatus();
|
||||
// The GL_MAX_SAMPLES value MobileGL advertises, i.e. the driver's value floored to the GL
|
||||
// core minimum. Frontend multisample validators have to honour this ceiling for every
|
||||
// format, otherwise MobileGL rejects a sample count it advertised itself.
|
||||
GLint GetAdvertisedMaxSamples();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -21,6 +21,9 @@
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
// The flattened uniform type these helpers used to take as a raw glslang::TType*
|
||||
// pointing into the TProgram's pool allocator. See ProgramObject::TypeFacts.
|
||||
using TypeFactsRef = const MG_State::GLState::ProgramObject::TypeFacts&;
|
||||
static GLint BoolToGLInt(bool value) {
|
||||
return value ? GL_TRUE : GL_FALSE;
|
||||
}
|
||||
@@ -223,14 +226,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return false;
|
||||
}
|
||||
|
||||
GLint GetOpaqueUniformUnitLimit(const glslang::TType* type) {
|
||||
GLint GetOpaqueUniformUnitLimit(const TypeFactsRef type) {
|
||||
const auto& dynamicParameters = MG_Backend::pActiveBackendObject->GetDynamicParameters();
|
||||
if (type && type->isImage()) return dynamicParameters.MaxImageUnits;
|
||||
if (type && type->isTexture()) return dynamicParameters.MaxCombinedTextureImageUnits;
|
||||
if (type.isImage) return dynamicParameters.MaxImageUnits;
|
||||
if (type.isTexture) return dynamicParameters.MaxCombinedTextureImageUnits;
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool ValidateOpaqueUniformUnit(const char* functionName, const glslang::TType* type, GLint unit) {
|
||||
bool ValidateOpaqueUniformUnit(const char* functionName, const TypeFactsRef type, GLint unit) {
|
||||
const GLint limit = GetOpaqueUniformUnitLimit(type);
|
||||
if (unit < 0 || unit >= limit) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -525,6 +528,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_UNIFORM_ARRAY_STRIDE:
|
||||
case GL_UNIFORM_MATRIX_STRIDE:
|
||||
case GL_UNIFORM_IS_ROW_MAJOR:
|
||||
// GL 4.2 / ARB_shader_atomic_counters adds this one to the accepted set. Leaving it
|
||||
// out did not merely lose the answer: the leftover GL_INVALID_ENUM is what made
|
||||
// KHR-GL43.shader_atomic_counters.basic-program-query force a FAIL.
|
||||
case GL_UNIFORM_ATOMIC_COUNTER_BUFFER_INDEX:
|
||||
break;
|
||||
default:
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -580,6 +587,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_UNIFORM_IS_ROW_MAJOR:
|
||||
params[i] = programObject->GetActiveUniformIsRowMajor(idx);
|
||||
break;
|
||||
case GL_UNIFORM_ATOMIC_COUNTER_BUFFER_INDEX:
|
||||
// Index into the GL_ACTIVE_ATOMIC_COUNTER_BUFFERS list, -1 for every uniform
|
||||
// that is not an atomic counter (GL 4.6 core table 7.6).
|
||||
params[i] = programObject->GetActiveUniformAtomicCounterBufferIndex(idx);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -642,7 +654,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
break;
|
||||
}
|
||||
case GL_ACTIVE_ATOMIC_COUNTER_BUFFERS:
|
||||
*params = programObject->GetActiveAtomicCounterCount();
|
||||
// Counter BUFFERS, not counters, and glslang's own getNumAtomicCounters() answers
|
||||
// neither: the relaxed parse has already turned every atomic_uint into a plain uint
|
||||
// member of a synthesized storage block by the time it builds its reflection, so it
|
||||
// reports zero. The interface-query model recovers the buffers from those blocks and
|
||||
// is what glGetProgramInterfaceiv(GL_ATOMIC_COUNTER_BUFFER, GL_ACTIVE_RESOURCES)
|
||||
// already answers - the two queries are required to agree.
|
||||
*params = ProgramInterface::GetActiveResourceCount(*programObject, GL_ATOMIC_COUNTER_BUFFER);
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
case GL_ACTIVE_ATTRIBUTES:
|
||||
@@ -662,7 +680,9 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
case GL_ACTIVE_UNIFORM_BLOCKS: // GL >= 3.1
|
||||
*params = programObject->GetActiveUniformBlocksCount();
|
||||
// Uniform blocks only. GetActiveUniformBlocksCount() is the internal block space,
|
||||
// which also carries the storage blocks and the synthesized atomic counter blocks.
|
||||
*params = programObject->GetGlUniformBlockCount();
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
case GL_ACTIVE_UNIFORM_BLOCK_MAX_NAME_LENGTH: // ditto.
|
||||
@@ -682,7 +702,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
MGLOG_D("%s: %s = %d", __func__, MG_Util::ConvertGLEnumToString(pname).c_str(), *params);
|
||||
break;
|
||||
case GL_COMPUTE_WORK_GROUP_SIZE: { // GL >= 4.3
|
||||
if (!programObject->GetLinkStatus() || programObject->GetShaderIndexByStage(ShaderStage::Compute) < 0) {
|
||||
// "a linked program object with a compute shader" is one whose EXECUTABLE has the
|
||||
// stage: the local size below is a link artifact, so an attached-but-not-yet-linked
|
||||
// compute shader would answer this query with the previous link's (absent) value
|
||||
// instead of the INVALID_OPERATION GL 4.6 core 7.13 asks for.
|
||||
if (!programObject->GetLinkStatus() || !programObject->HasLinkedShaderStage(ShaderStage::Compute)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
@@ -744,6 +768,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
CopyStr(bufSize, length, infoLog, log.c_str(), (GLsizei)log.length());
|
||||
}
|
||||
|
||||
// MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS: while the compile job is still in flight -
|
||||
// and, via the latch below, for the rest of that node's life once any query was
|
||||
// answered this way - GL_COMPILE_STATUS reads GL_TRUE and the info log reads empty,
|
||||
// WITHOUT joining. The latch (TakeOptimisticCompileAnswer) is what makes the three
|
||||
// sites tell ONE story: without it, a job settling between an application's info-log
|
||||
// read and its status read would produce the torn pair "GL_FALSE with an empty log",
|
||||
// and an application that aborts on that never reaches the link join that carries the
|
||||
// real diagnostic. A failure hidden here still fails the program link, with the
|
||||
// compile log quoted in the program info log (ProgramLinkTask::ConsumeShaders), which
|
||||
// is where the serial compile-then-check applications this exists for do their error
|
||||
// handling.
|
||||
static Bool AnswerCompileOptimistically(const SharedPtr<MG_State::GLState::ShaderObject>& shaderObject) {
|
||||
return MG_Util::Async::OptimisticShaderStatusActive() && shaderObject->TakeOptimisticCompileAnswer();
|
||||
}
|
||||
|
||||
void GetShaderiv_State(GLuint shader, GLenum pname, GLint* params) {
|
||||
auto& shaderObject = TryToGetShaderObject(shader);
|
||||
if (!shaderObject) return;
|
||||
@@ -756,9 +795,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*params = shaderObject->GetDeleteStatus();
|
||||
break;
|
||||
case GL_COMPILE_STATUS:
|
||||
if (AnswerCompileOptimistically(shaderObject)) {
|
||||
*params = GL_TRUE;
|
||||
break;
|
||||
}
|
||||
*params = shaderObject->GetCompileStatus();
|
||||
break;
|
||||
case GL_INFO_LOG_LENGTH:
|
||||
// Not cosmetic: LWJGL's one-argument glGetShaderInfoLog convenience overload
|
||||
// sizes its buffer from this query, so a joining answer here would defeat the
|
||||
// non-joining GetShaderInfoLog below.
|
||||
if (AnswerCompileOptimistically(shaderObject)) {
|
||||
*params = 0;
|
||||
break;
|
||||
}
|
||||
*params = shaderObject->GetInfoLog().empty() ? 0 : (GLint)shaderObject->GetInfoLog().length() + 1;
|
||||
break;
|
||||
case GL_SHADER_SOURCE_LENGTH:
|
||||
@@ -784,6 +834,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto& shaderObject = TryToGetShaderObject(shader);
|
||||
if (!shaderObject) return;
|
||||
|
||||
// See AnswerCompileOptimistically: an in-flight compile reads as an empty log. The
|
||||
// cost is a lost compile WARNING (a successful compile whose log the application
|
||||
// reads exactly once, now, and never after the join) - accepted as part of the
|
||||
// opt-in.
|
||||
if (AnswerCompileOptimistically(shaderObject)) {
|
||||
CopyStr(bufSize, length, infoLog, "", 0);
|
||||
return;
|
||||
}
|
||||
|
||||
const auto& log = shaderObject->GetInfoLog();
|
||||
CopyStr(bufSize, length, infoLog, log.c_str(), (GLsizei)log.length());
|
||||
}
|
||||
@@ -815,11 +874,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// vector per column - while the value glGetUniform* must return is tightly packed
|
||||
// columns * rows floats. Only mat4 is the same either way; every other shape needs the
|
||||
// padding undone, and the readback has to undo exactly what UniformMatrixfv_Object put
|
||||
// there. Returns false when `ttype` is not a float matrix (nothing to unpack).
|
||||
Bool TryGatherFloatMatrixColumns(const glslang::TType* ttype, const char* pBase, void* params) {
|
||||
if (ttype == nullptr || !ttype->isMatrix() || ttype->getBasicType() == glslang::EbtDouble) return false;
|
||||
const Int columns = ttype->getMatrixCols();
|
||||
const Int rows = ttype->getMatrixRows();
|
||||
// there. Returns false when there is nothing here to unpack.
|
||||
//
|
||||
// A DOUBLE matrix is declined not because it is laid out differently - it is not, the
|
||||
// demotion makes a dmat4 a mat4 in the shader and a mat4-shaped slot here - but because it
|
||||
// is ROUTED differently: the caller's component-by-component EbtDouble branch has to widen
|
||||
// each float back to the queried type, and it undoes the same padding itself.
|
||||
// Float matrices only, in both senses: a DOUBLE matrix never comes through here, whether its
|
||||
// program was demoted (components are floats, the query is not) or kept its doubles (the
|
||||
// column stride is a dvec4's, and the caller's converting branch already walks it component
|
||||
// by component with the right one).
|
||||
Bool TryGatherFloatMatrixColumns(const TypeFactsRef ttype, const char* pBase, void* params) {
|
||||
if (!ttype.isMatrix || ttype.isDouble) return false;
|
||||
const Int columns = ttype.matrixCols;
|
||||
const Int rows = ttype.matrixRows;
|
||||
for (Int column = 0; column < columns; ++column) {
|
||||
Memcpy(static_cast<char*>(params) + static_cast<SizeT>(column) * rows * sizeof(GLfloat),
|
||||
pBase + static_cast<SizeT>(column) * 4 * sizeof(GLfloat), rows * sizeof(GLfloat));
|
||||
@@ -828,12 +896,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
// Bytes a uniform actually occupies in the global UBO. It is the tight GL type size for
|
||||
// everything except a float matrix, whose padded columns make it wider.
|
||||
SizeT UniformStorageSpanInBytes(const glslang::TType* ttype, SizeT tightSize) {
|
||||
if (ttype != nullptr && ttype->isMatrix() && ttype->getBasicType() != glslang::EbtDouble) {
|
||||
return static_cast<SizeT>(ttype->getMatrixCols()) * 4 * sizeof(GLfloat);
|
||||
}
|
||||
return tightSize;
|
||||
// everything except a matrix, whose padded columns make it wider, and a `double` on a
|
||||
// program whose modules were demoted, where it is half. The rule itself lives on
|
||||
// ProgramObject, because the pipeline composite's uniform refresh needs the same one and
|
||||
// two copies of a layout rule is one too many.
|
||||
SizeT UniformStorageSpanInBytes(const TypeFactsRef ttype, SizeT tightSize, const Bool nativeFloat64) {
|
||||
return MG_State::GLState::ProgramObject::UniformStorageSpanInBytes(ttype, tightSize, nativeFloat64);
|
||||
}
|
||||
|
||||
void GetUniform_State(GLuint program, GLint location, void* params) {
|
||||
@@ -865,17 +933,24 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto offset = programObject->GetUniformOffset(location);
|
||||
auto size = programObject->GetUniformSizesInBytes(location);
|
||||
char* pUBO = (char*)programObject->MapUBO();
|
||||
auto* ttype = programObject->GetUniformTType(location);
|
||||
const SizeT span = UniformStorageSpanInBytes(ttype, size);
|
||||
const auto& ttype = programObject->GetUniformTypeFacts(location);
|
||||
const Bool nativeFloat64 = programObject->UsesNativeFloat64();
|
||||
const SizeT span = UniformStorageSpanInBytes(ttype, size, nativeFloat64);
|
||||
if (pUBO == nullptr || offset == MG_State::GLState::ProgramObject::kInvalidUniformOffset ||
|
||||
offset + span > programObject->GetUBOSize()) {
|
||||
MGLOG_E("%s: uniform at program %u location %d has no backing storage; returning nothing", __func__,
|
||||
MGLOG_E_ONCE("%s: uniform at program %u location %d has no backing storage; returning nothing", __func__,
|
||||
program, location);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!TryGatherFloatMatrixColumns(ttype, pUBO + offset, params)) {
|
||||
Memcpy(params, pUBO + offset, size);
|
||||
// Never more than the uniform actually occupies. `size` is the GL type size,
|
||||
// which on a DEMOTED program is twice a `double` uniform's storage - its 64-bit
|
||||
// floats were narrowed before the module reached a backend, so the slot holds
|
||||
// floats. The typed entry points (glGetUniformdv and friends) go through
|
||||
// GetUniformScalar_State, which converts component by component; this raw
|
||||
// copy has no type to convert with, so it is bounded rather than converted.
|
||||
Memcpy(params, pUBO + offset, std::min<SizeT>(size, span));
|
||||
}
|
||||
}
|
||||
// TODO: handle 1i variant as texture unit
|
||||
@@ -913,11 +988,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
auto offset = programObject->GetUniformOffset(location);
|
||||
auto size = programObject->GetUniformSizesInBytes(location);
|
||||
char* pUBO = static_cast<char*>(programObject->MapUBO());
|
||||
auto* ttype = programObject->GetUniformTType(location);
|
||||
const SizeT span = UniformStorageSpanInBytes(ttype, size);
|
||||
const auto& ttype = programObject->GetUniformTypeFacts(location);
|
||||
const Bool nativeFloat64 = programObject->UsesNativeFloat64();
|
||||
const SizeT span = UniformStorageSpanInBytes(ttype, size, nativeFloat64);
|
||||
if (pUBO == nullptr || offset == MG_State::GLState::ProgramObject::kInvalidUniformOffset ||
|
||||
offset + span > programObject->GetUBOSize()) {
|
||||
MGLOG_E("%s: uniform at program %u location %d has no backing storage; returning nothing", __func__,
|
||||
MGLOG_E_ONCE("%s: uniform at program %u location %d has no backing storage; returning nothing", __func__,
|
||||
program, location);
|
||||
return;
|
||||
}
|
||||
@@ -926,23 +1002,38 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (TryGatherFloatMatrixColumns(ttype, pUBO + offset, params)) return;
|
||||
}
|
||||
|
||||
// A double-precision uniform is the one case where the stored component type can
|
||||
// differ from the queried one for a non-opaque uniform, and the difference is not
|
||||
// just a reinterpretation: it is twice as wide, so a raw copy would overrun the
|
||||
// caller's buffer as well as return nonsense. Read component by component and let
|
||||
// GL's conversion rules (7.6: round to nearest for the integer queries) apply.
|
||||
if (ttype->getBasicType() == glslang::EbtDouble) {
|
||||
const Int columns = ttype->isMatrix() ? ttype->getMatrixCols() : 1;
|
||||
const Int rows = ttype->isMatrix() ? ttype->getMatrixRows()
|
||||
: (ttype->isVector() ? ttype->getVectorSize() : 1);
|
||||
// The slot the linker handed out is exactly `columns` columns wide, so it also
|
||||
// states the column stride - which for a double matrix is not a float's 16 bytes.
|
||||
const SizeT columnStride = columns > 0 ? size / static_cast<SizeT>(columns) : size;
|
||||
// A double-precision uniform is the one case where the stored component type can differ
|
||||
// from the DECLARED one for a non-opaque uniform: on a DEMOTED program the shader's
|
||||
// 64-bit floats were narrowed to 32 before the module reached the backend
|
||||
// (ShaderTranspiler::DemoteFloat64Pass), so what is in the global UBO is a float per
|
||||
// component, laid out exactly like the float-typed twin of this uniform - std140
|
||||
// 16-byte column stride for a matrix included. Reading it as a GLdouble would return
|
||||
// two components reinterpreted as one. A program that KEPT its doubles stores real ones
|
||||
// at the dvec4 column stride instead, so the width and the stride both move; everything
|
||||
// else about this walk is the same. Read component by component either way and let GL's
|
||||
// conversion rules (7.6: round to nearest for the integer queries) apply; the value
|
||||
// widens back to the queried type, having lost precision - where it lost any - at the
|
||||
// glUniform*d that stored it and not here.
|
||||
if (ttype.isDouble) {
|
||||
const Int columns = ttype.isMatrix ? ttype.matrixCols : 1;
|
||||
const Int rows = ttype.isMatrix ? ttype.matrixRows
|
||||
: (ttype.isVector ? ttype.vectorSize : 1);
|
||||
// A non-matrix is one tightly packed run and never reaches the stride at all.
|
||||
const SizeT columnStride =
|
||||
MG_State::GLState::ProgramObject::UniformMatrixColumnStride(ttype, nativeFloat64);
|
||||
const SizeT componentSize = nativeFloat64 ? sizeof(GLdouble) : sizeof(GLfloat);
|
||||
for (Int column = 0; column < columns; ++column) {
|
||||
for (Int row = 0; row < rows; ++row) {
|
||||
GLdouble component = 0.0;
|
||||
Memcpy(&component, pUBO + offset + column * columnStride + row * sizeof(GLdouble),
|
||||
sizeof(component));
|
||||
if (nativeFloat64) {
|
||||
Memcpy(&component, pUBO + offset + column * columnStride + row * componentSize,
|
||||
sizeof(GLdouble));
|
||||
} else {
|
||||
GLfloat narrow = 0.0f;
|
||||
Memcpy(&narrow, pUBO + offset + column * columnStride + row * componentSize,
|
||||
sizeof(narrow));
|
||||
component = static_cast<GLdouble>(narrow);
|
||||
}
|
||||
if constexpr (std::is_integral_v<T>) {
|
||||
// Rounded to the nearest integer and clamped into the queried type's
|
||||
// range, so a negative double read through glGetUniformuiv is 0
|
||||
@@ -1007,21 +1098,20 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
static Bool allowVSOnlyPrograms;
|
||||
static Bool initialized = false;
|
||||
if (!initialized) {
|
||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
||||
if (!activeBackendObject) {
|
||||
MGLOG_E("activeBackendObject is not initialized!");
|
||||
return;
|
||||
}
|
||||
const auto& rendererInfo = activeBackendObject->GetRendererInfo();
|
||||
allowVSOnlyPrograms = (Int)rendererInfo.StaticBackendCapability.AllowVSOnlyPrograms;
|
||||
}
|
||||
// Read fresh every link, never latched in a static: the capability is
|
||||
// per-backend, and a latch would freeze it across a backend teardown +
|
||||
// re-initialization (the previous function-static memo here never even set
|
||||
// its own initialized flag, so it re-read every call anyway - this makes
|
||||
// the always-fresh behavior the stated one). A struct-field read per
|
||||
// glLinkProgram costs nothing.
|
||||
const auto& activeBackendObject = MG_Backend::pActiveBackendObject;
|
||||
if (activeBackendObject) {
|
||||
programObject->SetMaxFragmentOutputColorNumber(activeBackendObject->GetDynamicParameters().MaxDrawBuffers);
|
||||
if (!activeBackendObject) {
|
||||
MGLOG_E_ONCE("activeBackendObject is not initialized!");
|
||||
return;
|
||||
}
|
||||
const Bool allowVSOnlyPrograms =
|
||||
activeBackendObject->GetRendererInfo().StaticBackendCapability.AllowVSOnlyPrograms;
|
||||
programObject->SetMaxFragmentOutputColorNumber(activeBackendObject->GetDynamicParameters().MaxDrawBuffers);
|
||||
programObject->Link(!allowVSOnlyPrograms);
|
||||
}
|
||||
|
||||
@@ -1085,23 +1175,45 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!programObject.IsUniformOpaqueAtLocation(location)) {
|
||||
MGLOG_D("%s: program = %d, location = %d, maxLocation = %d", __func__, programObject.GetExternalIndex(),
|
||||
location, programObject.GetMaxUniformLocation());
|
||||
// Record the write for the pipeline composite's uniform mirror, which copies only
|
||||
// the locations a stage program has actually been written to (see
|
||||
// ProgramObject::MarkUniformWrittenAtLocation). Here rather than further down
|
||||
// because every exit below is still a write as far as GL is concerned: the
|
||||
// buffered-write detour returns early, the bytes-equal dedupe returns early, and
|
||||
// even the no-backing-storage bail is a uniform the application addressed. This is
|
||||
// the funnel EVERY glUniform* and glProgramUniform* entry point reaches, once per
|
||||
// LOCATION - so an array element write marks that element and nothing else. On a
|
||||
// program that can never be a pipeline stage - the monolithic glUseProgram path,
|
||||
// which is where the thousands of calls per frame are - this is one bool branch.
|
||||
programObject.MarkUniformWrittenAtLocation(location);
|
||||
// Everything up to and including the clamp is phase-A data (the uniform's GL type
|
||||
// decides its size), so it is answered without joining anything.
|
||||
const SizeT size = programObject.GetUniformSizesInBytes(location);
|
||||
const Uint offset = programObject.GetUniformOffset(location);
|
||||
char* pUBO = static_cast<char*>(programObject.MapUBO());
|
||||
const SizeT uboSize = programObject.GetUBOSize();
|
||||
SizeT writeSize = ItemCount * sizeof(T);
|
||||
if (size < writeSize) {
|
||||
// Metadata bug: degrade to a clamped copy instead of killing the process.
|
||||
MGLOG_E("%s: uniform size mismatch at program %u location %u: expected at least %zu bytes, got %zu "
|
||||
MGLOG_E_ONCE("%s: uniform size mismatch at program %u location %u: expected at least %zu bytes, got %zu "
|
||||
"bytes; clamping",
|
||||
__func__, programObject.GetExternalIndex(), location, ItemCount * sizeof(T), size);
|
||||
writeSize = size;
|
||||
}
|
||||
// The uniform shadow's LAYOUT is phase-B data, so a write that lands while the
|
||||
// SPIR-V job is still running is recorded and replayed at its publish instead of
|
||||
// joining it. This is the hot path for a shaderpack that sets its uniforms
|
||||
// immediately after glLinkProgram. BufferUniformWrite declines (and we fall
|
||||
// through, joining) only past its size budget.
|
||||
if (programObject.IsSpirvPending() &&
|
||||
programObject.BufferUniformWrite(location, byteOffsetInsideUniform, value, writeSize)) {
|
||||
return;
|
||||
}
|
||||
const Uint offset = programObject.GetUniformOffset(location);
|
||||
char* pUBO = static_cast<char*>(programObject.MapUBO());
|
||||
const SizeT uboSize = programObject.GetUBOSize();
|
||||
if (pUBO == nullptr || offset == MG_State::GLState::ProgramObject::kInvalidUniformOffset ||
|
||||
offset + byteOffsetInsideUniform + writeSize > uboSize) {
|
||||
// Should not happen: linking gives every settable uniform backing
|
||||
// storage. Log and drop the write instead of faulting.
|
||||
MGLOG_E("%s: uniform at program %u location %u has no backing storage (ubo=%p offset=%u size=%zu "
|
||||
MGLOG_E_ONCE("%s: uniform at program %u location %u has no backing storage (ubo=%p offset=%u size=%zu "
|
||||
"uboSize=%zu); dropping write",
|
||||
__func__, programObject.GetExternalIndex(), location, static_cast<void*>(pUBO), offset,
|
||||
writeSize, uboSize);
|
||||
@@ -1120,8 +1232,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
Memcpy(pUBO + offset + byteOffsetInsideUniform, value, writeSize);
|
||||
programObject.MarkUBOContentDirty();
|
||||
} else {
|
||||
auto* ttype = programObject.GetUniformTType(location);
|
||||
if (!ttype->isTexture() && !ttype->isImage()) return;
|
||||
const auto& ttype = programObject.GetUniformTypeFacts(location);
|
||||
if (!ttype.isTexture && !ttype.isImage) return;
|
||||
if constexpr (!std::is_same_v<std::remove_cv_t<T>, GLint> || ItemCount != 1) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
@@ -1192,36 +1304,75 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// glUniform*d / glUniformMatrix*dv. The vector forms need nothing beyond the shared
|
||||
// upload template - it is already typed on the component - but a matrix does: the
|
||||
// column stride the linker used for a double matrix is not the 16 bytes a float one
|
||||
// gets. It is not guessed here; the slot the uniform was given is exactly `columns`
|
||||
// columns wide, so dividing states the stride the rest of the pipeline agreed on.
|
||||
template <typename Program>
|
||||
void UniformMatrixdv_Object(Program& programObject, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value, Int columns, Int rows) {
|
||||
const SizeT slotSize = programObject.GetUniformSizesInBytes(location);
|
||||
const SizeT columnStride = columns > 0 ? slotSize / static_cast<SizeT>(columns) : slotSize;
|
||||
const SizeT componentCount = static_cast<SizeT>(columns) * static_cast<SizeT>(rows);
|
||||
Vector<GLdouble> column(static_cast<SizeT>(rows));
|
||||
for (GLint matrix = 0; matrix < count; ++matrix) {
|
||||
if (matrix > 0 && !programObject.UniformLocationsAliasSameUniform(location, location + matrix)) break;
|
||||
if (!programObject.IsValidUniformLocation(location + matrix)) {
|
||||
RecordInvalidUniformLocationError(__func__, location + matrix, "the current program object");
|
||||
return;
|
||||
}
|
||||
const GLdouble* source = value + matrix * componentCount;
|
||||
for (Int c = 0; c < columns; ++c) {
|
||||
for (Int r = 0; r < rows; ++r) {
|
||||
column[r] = transpose == GL_TRUE ? source[r * columns + c] : source[c * rows + r];
|
||||
}
|
||||
Uniform_State<1>(programObject, location + matrix, column.data(), c * columnStride);
|
||||
for (Int r = 1; r < rows; ++r) {
|
||||
Uniform_State<1>(programObject, location + matrix, column.data() + r,
|
||||
c * columnStride + r * sizeof(GLdouble));
|
||||
}
|
||||
}
|
||||
// Whether the program a uniform write is about to land in stores 64-bit floats at their
|
||||
// declared width. Answered off the PROGRAM, never off the live backend: it describes the
|
||||
// modules that were actually built for it, and a backend with native fp64 still demotes a
|
||||
// program whose vertex stage declares a Float64 input (see ProgramSpirvTask::GenerateSpirv).
|
||||
// Nullptr - no current program, or a name that is not a program - answers false and lets the
|
||||
// callee record the same error it always did.
|
||||
Bool CurrentProgramUsesNativeFloat64() {
|
||||
if (MG_State::pGLContext == nullptr) return false;
|
||||
const auto& programObject = MG_State::pGLContext->GetProgramForUniform();
|
||||
return programObject != nullptr && programObject->UsesNativeFloat64();
|
||||
}
|
||||
|
||||
Bool NamedProgramUsesNativeFloat64(GLuint program) {
|
||||
const auto& programObject = TryToGetProgramObject(program);
|
||||
return programObject != nullptr && programObject->GetLinkStatus() && programObject->UsesNativeFloat64();
|
||||
}
|
||||
|
||||
// glUniform*d / glUniformMatrix*dv. On a DEMOTED program neither needs a layout of its own:
|
||||
// the transpile chain narrowed every 64-bit float in the shader to 32
|
||||
// (ShaderTranspiler::DemoteFloat64Pass) and the global UBO is laid out by reflecting that
|
||||
// demoted module, so a double uniform's storage IS a float uniform's - same offset, same
|
||||
// 4-byte components, same std140 column padding for matrices. Narrowing here, at the one
|
||||
// place the 64-bit value enters, and then handing the bytes to the ordinary float upload
|
||||
// path is what keeps the two in step; a separate double-shaped layout there would write
|
||||
// 8-byte components into 4-byte slots and silently address the wrong ones.
|
||||
//
|
||||
// The narrowing is the same static_cast the demoted shader's own arithmetic performs, so the
|
||||
// value the shader reads is the value glUniform*d was given, at float precision.
|
||||
//
|
||||
// On a program that KEPT its doubles the reverse is true and for the same reason: its global
|
||||
// UBO really does hold 8-byte components, so narrowing would leave a float bit pattern in the
|
||||
// low half of a double slot - which is not a precision loss but a garbage value. The 64-bit
|
||||
// values go through unchanged then, and the upload path is width-agnostic (it is templated on
|
||||
// the component type and bounded by the uniform's own slot span).
|
||||
//
|
||||
// Note TryToGetProgramObject / GetProgramForUniform run TWICE on this path, once for the
|
||||
// width question and once inside the call below. That is a lookup and a join on an entry
|
||||
// point no shader pack uses; the alternative is duplicating both functions' whole validation
|
||||
// sequence here, which is the thing that must not drift.
|
||||
template <GLsizei ItemCount>
|
||||
void UniformvNarrowed_State(GLint location, GLsizei count, const GLdouble* value) {
|
||||
if (value == nullptr || count <= 0) {
|
||||
// Same shape as the float entry points: the location validation still runs, and a
|
||||
// null pointer is left to fault exactly where glUniform*fv would.
|
||||
Uniformv_State<ItemCount>(location, count, reinterpret_cast<const GLfloat*>(value));
|
||||
return;
|
||||
}
|
||||
if (location != -1 && CurrentProgramUsesNativeFloat64()) {
|
||||
Uniformv_State<ItemCount>(location, count, value);
|
||||
return;
|
||||
}
|
||||
Vector<GLfloat> narrowed(static_cast<SizeT>(count) * ItemCount);
|
||||
for (SizeT i = 0; i < narrowed.size(); ++i) narrowed[i] = static_cast<GLfloat>(value[i]);
|
||||
Uniformv_State<ItemCount>(location, count, narrowed.data());
|
||||
}
|
||||
|
||||
template <GLsizei ItemCount>
|
||||
void ProgramUniformvNarrowed_State(GLuint program, GLint location, GLsizei count, const GLdouble* value) {
|
||||
if (value == nullptr || count <= 0) {
|
||||
ProgramUniformv_State<ItemCount>(program, location, count, reinterpret_cast<const GLfloat*>(value));
|
||||
return;
|
||||
}
|
||||
if (location != -1 && NamedProgramUsesNativeFloat64(program)) {
|
||||
ProgramUniformv_State<ItemCount>(program, location, count, value);
|
||||
return;
|
||||
}
|
||||
Vector<GLfloat> narrowed(static_cast<SizeT>(count) * ItemCount);
|
||||
for (SizeT i = 0; i < narrowed.size(); ++i) narrowed[i] = static_cast<GLfloat>(value[i]);
|
||||
ProgramUniformv_State<ItemCount>(program, location, count, narrowed.data());
|
||||
}
|
||||
|
||||
// glUniformMatrix*fv / glProgramUniformMatrix*fv, every shape (square and non-square).
|
||||
@@ -1270,6 +1421,70 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// glUniformMatrix*dv / glProgramUniformMatrix*dv on a program that KEPT its doubles. Same
|
||||
// walk as UniformMatrixfv_Object down to the last branch, and deliberately a copy of it
|
||||
// rather than a template over the component type: the two differ in exactly one number that
|
||||
// is not derivable from the component type alone - std140 pads a double matrix's column out
|
||||
// to a dvec4 (32 bytes) unless the column is a dvec2, which is already 16 - and folding that
|
||||
// into the float version would put a per-call branch on the hot glUniformMatrix4fv path
|
||||
// Minecraft calls thousands of times a frame for a case no shader pack ever takes.
|
||||
template <typename Program>
|
||||
void UniformMatrixdvNative_Object(Program& programObject, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value, Int columns, Int rows,
|
||||
const String& ownerDescription) {
|
||||
const SizeT columnStride = rows <= 2 ? 2 * sizeof(GLdouble) : 4 * sizeof(GLdouble);
|
||||
const SizeT componentCount = static_cast<SizeT>(columns) * static_cast<SizeT>(rows);
|
||||
GLdouble column[4] = {};
|
||||
for (GLint matrix = 0; matrix < count; ++matrix) {
|
||||
if (matrix > 0 && !programObject.UniformLocationsAliasSameUniform(location, location + matrix)) break;
|
||||
if (!programObject.IsValidUniformLocation(location + matrix)) {
|
||||
RecordInvalidUniformLocationError("glUniformMatrixdv", location + matrix, ownerDescription);
|
||||
return;
|
||||
}
|
||||
if (programObject.IsUniformOpaqueAtLocation(location + matrix)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "glUniformMatrixdv",
|
||||
"Opaque uniforms cannot be set with matrix Uniform calls."));
|
||||
return;
|
||||
}
|
||||
const GLdouble* source = value + static_cast<SizeT>(matrix) * componentCount;
|
||||
for (Int c = 0; c < columns; ++c) {
|
||||
for (Int r = 0; r < rows; ++r) {
|
||||
column[r] = transpose == GL_TRUE ? source[r * columns + c] : source[c * rows + r];
|
||||
}
|
||||
const SizeT byteOffset = static_cast<SizeT>(c) * columnStride;
|
||||
switch (rows) {
|
||||
case 2: Uniform_State<2>(programObject, location + matrix, column, byteOffset); break;
|
||||
case 3: Uniform_State<3>(programObject, location + matrix, column, byteOffset); break;
|
||||
default: Uniform_State<4>(programObject, location + matrix, column, byteOffset); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// glUniformMatrix*dv / glProgramUniformMatrix*dv. On a DEMOTED program this narrows to the
|
||||
// float form and hands it straight over: after DemoteFloat64Pass a `dmat4` uniform is a
|
||||
// `mat4` in the shader and a mat4-shaped slot in the global UBO, columns padded to a vec4
|
||||
// and all. Everything else about the call - transpose handling, the array-element walk, the
|
||||
// opaque-uniform refusal - is then the one implementation both spellings share. A program
|
||||
// that kept its doubles gets the same walk at double width and the wider column stride.
|
||||
template <typename Program>
|
||||
void UniformMatrixdv_Object(Program& programObject, GLint location, GLsizei count, GLboolean transpose,
|
||||
const GLdouble* value, Int columns, Int rows) {
|
||||
if (value == nullptr || count <= 0) return;
|
||||
if (programObject.UsesNativeFloat64()) {
|
||||
UniformMatrixdvNative_Object(programObject, location, count, transpose, value, columns, rows,
|
||||
"the current program object");
|
||||
return;
|
||||
}
|
||||
const SizeT componentCount = static_cast<SizeT>(columns) * static_cast<SizeT>(rows);
|
||||
Vector<GLfloat> narrowed(static_cast<SizeT>(count) * componentCount);
|
||||
for (SizeT i = 0; i < narrowed.size(); ++i) narrowed[i] = static_cast<GLfloat>(value[i]);
|
||||
UniformMatrixfv_Object(programObject, "glUniformMatrixdv", location, count, transpose, narrowed.data(),
|
||||
columns, rows, "the current program object");
|
||||
}
|
||||
|
||||
// Helper function to transpose a 2x2 matrix
|
||||
void TransposeMatrix2x2(const GLfloat* input, GLfloat* output) {
|
||||
// Input matrix is in column-major order (OpenGL default)
|
||||
@@ -1604,7 +1819,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return GL_INVALID_INDEX;
|
||||
}
|
||||
|
||||
const auto& index = programObject->GetUniformBlockIndex(uniformBlockName);
|
||||
// GetGlUniformBlockIndex, not GetUniformBlockIndex: the latter answers in the internal
|
||||
// block space, which also resolves storage blocks and the synthesized atomic counter
|
||||
// blocks. Neither is a uniform block (GL 4.6 core 7.6), so both are GL_INVALID_INDEX here.
|
||||
const auto index = programObject->GetGlUniformBlockIndex(uniformBlockName);
|
||||
MGLOG_D("GBI prog=%u name='%s' -> %d", program, uniformBlockName ? uniformBlockName : "(null)", (Int)index);
|
||||
return index;
|
||||
}
|
||||
@@ -1618,7 +1836,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Program object" + std::to_string(program) + " that has been linked."));
|
||||
return;
|
||||
}
|
||||
if (!programObject->IsActiveUniformBlock(uniformBlockIndex)) {
|
||||
if (!programObject->IsActiveGlUniformBlock(uniformBlockIndex)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
@@ -1629,8 +1847,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::to_string(program) + "."));
|
||||
return;
|
||||
}
|
||||
// The GL_UNIFORM_BLOCK index space skips the storage and atomic counter blocks the
|
||||
// block-keyed tables still carry; translate before touching them.
|
||||
const Uint blockIndex = static_cast<Uint>(programObject->BlockIndexFromGlUniformBlock(uniformBlockIndex));
|
||||
MGLOG_D("UBB prog=%u idx=%u binding=%u", program, uniformBlockIndex, uniformBlockBinding);
|
||||
programObject->SetUniformBlockBinding(uniformBlockIndex, uniformBlockBinding);
|
||||
programObject->SetUniformBlockBinding(blockIndex, uniformBlockBinding);
|
||||
}
|
||||
|
||||
void GetActiveUniformBlockiv_State(GLuint program, GLuint uniformBlockIndex, GLenum pname, GLint* params) {
|
||||
@@ -1642,7 +1863,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"Program object" + std::to_string(program) + " that has been linked."));
|
||||
return;
|
||||
}
|
||||
if (!programObject->IsActiveUniformBlock(uniformBlockIndex)) {
|
||||
if (!programObject->IsActiveGlUniformBlock(uniformBlockIndex)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
@@ -1653,61 +1874,68 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
std::to_string(program) + "."));
|
||||
return;
|
||||
}
|
||||
// The GL_UNIFORM_BLOCK index space skips the storage and atomic counter blocks the
|
||||
// block-keyed tables still carry; every accessor below is indexed by the block space.
|
||||
const Uint blockIndex = static_cast<Uint>(programObject->BlockIndexFromGlUniformBlock(uniformBlockIndex));
|
||||
switch (pname) {
|
||||
case GL_UNIFORM_BLOCK_DATA_SIZE: {
|
||||
*params = (GLint)programObject->GetUBOSizeAt(uniformBlockIndex);
|
||||
*params = (GLint)programObject->GetUBOSizeAt(blockIndex);
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_DATA_SIZE = %d", __func__, *params);
|
||||
break;
|
||||
}
|
||||
case GL_UNIFORM_BLOCK_NAME_LENGTH: {
|
||||
*params = (GLint)programObject->GetUniformBlockName(uniformBlockIndex).length() + 1;
|
||||
*params = (GLint)programObject->GetUniformBlockName(blockIndex).length() + 1;
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_NAME_LENGTH = %d", __func__, *params);
|
||||
break;
|
||||
}
|
||||
case GL_UNIFORM_BLOCK_ACTIVE_UNIFORMS: {
|
||||
*params = programObject->GetUniformBlockActiveUniformCount(uniformBlockIndex);
|
||||
*params = programObject->GetUniformBlockActiveUniformCount(blockIndex);
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_ACTIVE_UNIFORMS = %d", __func__, *params);
|
||||
break;
|
||||
}
|
||||
case GL_UNIFORM_BLOCK_BINDING: {
|
||||
*params = static_cast<GLint>(programObject->GetUniformBlockBinding(uniformBlockIndex));
|
||||
*params = static_cast<GLint>(programObject->GetUniformBlockBinding(blockIndex));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_BINDING = %d", __func__, *params);
|
||||
break;
|
||||
}
|
||||
case GL_UNIFORM_BLOCK_REFERENCED_BY_VERTEX_SHADER:
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(uniformBlockIndex, EShLangVertex));
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(blockIndex, EShLangVertex));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_REFERENCED_BY_VERTEX_SHADER = %d", __func__, *params);
|
||||
break;
|
||||
case GL_UNIFORM_BLOCK_REFERENCED_BY_TESS_CONTROL_SHADER:
|
||||
*params =
|
||||
BoolToGLInt(programObject->IsUniformBlockReferencedByStage(uniformBlockIndex, EShLangTessControl));
|
||||
BoolToGLInt(programObject->IsUniformBlockReferencedByStage(blockIndex, EShLangTessControl));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_REFERENCED_BY_TESS_CONTROL_SHADER = %d", __func__, *params);
|
||||
break;
|
||||
case GL_UNIFORM_BLOCK_REFERENCED_BY_TESS_EVALUATION_SHADER:
|
||||
*params =
|
||||
BoolToGLInt(programObject->IsUniformBlockReferencedByStage(uniformBlockIndex, EShLangTessEvaluation));
|
||||
BoolToGLInt(programObject->IsUniformBlockReferencedByStage(blockIndex, EShLangTessEvaluation));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_REFERENCED_BY_TESS_EVALUATION_SHADER = %d", __func__, *params);
|
||||
break;
|
||||
case GL_UNIFORM_BLOCK_REFERENCED_BY_GEOMETRY_SHADER:
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(uniformBlockIndex, EShLangGeometry));
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(blockIndex, EShLangGeometry));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_REFERENCED_BY_GEOMETRY_SHADER = %d", __func__, *params);
|
||||
break;
|
||||
case GL_UNIFORM_BLOCK_REFERENCED_BY_FRAGMENT_SHADER:
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(uniformBlockIndex, EShLangFragment));
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(blockIndex, EShLangFragment));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_REFERENCED_BY_FRAGMENT_SHADER = %d", __func__, *params);
|
||||
break;
|
||||
case GL_UNIFORM_BLOCK_REFERENCED_BY_COMPUTE_SHADER:
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(uniformBlockIndex, EShLangCompute));
|
||||
*params = BoolToGLInt(programObject->IsUniformBlockReferencedByStage(blockIndex, EShLangCompute));
|
||||
MGLOG_D("%s: GL_UNIFORM_BLOCK_REFERENCED_BY_COMPUTE_SHADER = %d", __func__, *params);
|
||||
break;
|
||||
case GL_UNIFORM_BLOCK_ACTIVE_UNIFORM_INDICES: {
|
||||
// Member entries of an arrayed block are recorded against the first instance;
|
||||
// every instance of the array reports that shared member set (matches
|
||||
// GL_UNIFORM_BLOCK_ACTIVE_UNIFORMS, which scans with the same owner index).
|
||||
const Int ownerIndex = static_cast<Int>(programObject->GetUniformBlockMemberOwnerIndex(uniformBlockIndex));
|
||||
//
|
||||
// Both sides of the comparison are BLOCK indices: GetUniformBlockMemberOwnerIndex
|
||||
// answers in that space, so the scan uses GetActiveUniformOwnerBlockIndex rather
|
||||
// than the GL_UNIFORM_BLOCK-space GetActiveUniformBlockIndex.
|
||||
const Int ownerIndex = static_cast<Int>(programObject->GetUniformBlockMemberOwnerIndex(blockIndex));
|
||||
GLint uniformIndexCount = 0;
|
||||
for (Uint uniformIndex = 0; uniformIndex < programObject->GetUniformCount(); ++uniformIndex) {
|
||||
if (programObject->GetActiveUniformBlockIndex(uniformIndex) != ownerIndex) {
|
||||
if (programObject->GetActiveUniformOwnerBlockIndex(uniformIndex) != ownerIndex) {
|
||||
continue;
|
||||
}
|
||||
params[uniformIndexCount++] = static_cast<GLint>(uniformIndex);
|
||||
@@ -1716,7 +1944,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
break;
|
||||
}
|
||||
default:
|
||||
MGLOG_E("%s: unknown pname = %p %s", __func__, pname, MG_Util::ConvertGLEnumToString(pname).c_str());
|
||||
MGLOG_D("%s: unknown pname = %p %s", __func__, pname, MG_Util::ConvertGLEnumToString(pname).c_str());
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
@@ -1737,7 +1965,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
" is not a program object that has been linked."));
|
||||
return;
|
||||
}
|
||||
if (!programObject->IsActiveUniformBlock(uniformBlockIndex)) {
|
||||
if (!programObject->IsActiveGlUniformBlock(uniformBlockIndex)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
@@ -1747,7 +1975,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
"not the index of an active uniform block in program."));
|
||||
return;
|
||||
}
|
||||
const auto& name = programObject->GetUniformBlockName(uniformBlockIndex);
|
||||
const auto& name = programObject->GetUniformBlockName(
|
||||
static_cast<Uint>(programObject->BlockIndexFromGlUniformBlock(uniformBlockIndex)));
|
||||
CopyStr(bufSize, length, uniformBlockName, name.c_str(), (GLsizei)name.length());
|
||||
MGLOG_D("%s: \"%s\" at uniformBlockIndex %02d, length = %d", __func__, uniformBlockName, uniformBlockIndex,
|
||||
length ? *length : 0);
|
||||
@@ -2033,71 +2262,71 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
void Uniform1d(GLint location, GLdouble v0) {
|
||||
const GLdouble v[] = {v0};
|
||||
Uniformv_State<1>(location, 1, v);
|
||||
UniformvNarrowed_State<1>(location, 1, v);
|
||||
}
|
||||
|
||||
void Uniform1dv(GLint location, GLsizei count, const GLdouble* value) {
|
||||
Uniformv_State<1>(location, count, value);
|
||||
UniformvNarrowed_State<1>(location, count, value);
|
||||
}
|
||||
|
||||
void ProgramUniform1d(GLuint program, GLint location, GLdouble v0) {
|
||||
const GLdouble v[] = {v0};
|
||||
ProgramUniformv_State<1>(program, location, 1, v);
|
||||
ProgramUniformvNarrowed_State<1>(program, location, 1, v);
|
||||
}
|
||||
|
||||
void ProgramUniform1dv(GLuint program, GLint location, GLsizei count, const GLdouble* value) {
|
||||
ProgramUniformv_State<1>(program, location, count, value);
|
||||
ProgramUniformvNarrowed_State<1>(program, location, count, value);
|
||||
}
|
||||
void Uniform2d(GLint location, GLdouble v0, GLdouble v1) {
|
||||
const GLdouble v[] = {v0, v1};
|
||||
Uniformv_State<2>(location, 1, v);
|
||||
UniformvNarrowed_State<2>(location, 1, v);
|
||||
}
|
||||
|
||||
void Uniform2dv(GLint location, GLsizei count, const GLdouble* value) {
|
||||
Uniformv_State<2>(location, count, value);
|
||||
UniformvNarrowed_State<2>(location, count, value);
|
||||
}
|
||||
|
||||
void ProgramUniform2d(GLuint program, GLint location, GLdouble v0, GLdouble v1) {
|
||||
const GLdouble v[] = {v0, v1};
|
||||
ProgramUniformv_State<2>(program, location, 1, v);
|
||||
ProgramUniformvNarrowed_State<2>(program, location, 1, v);
|
||||
}
|
||||
|
||||
void ProgramUniform2dv(GLuint program, GLint location, GLsizei count, const GLdouble* value) {
|
||||
ProgramUniformv_State<2>(program, location, count, value);
|
||||
ProgramUniformvNarrowed_State<2>(program, location, count, value);
|
||||
}
|
||||
void Uniform3d(GLint location, GLdouble v0, GLdouble v1, GLdouble v2) {
|
||||
const GLdouble v[] = {v0, v1, v2};
|
||||
Uniformv_State<3>(location, 1, v);
|
||||
UniformvNarrowed_State<3>(location, 1, v);
|
||||
}
|
||||
|
||||
void Uniform3dv(GLint location, GLsizei count, const GLdouble* value) {
|
||||
Uniformv_State<3>(location, count, value);
|
||||
UniformvNarrowed_State<3>(location, count, value);
|
||||
}
|
||||
|
||||
void ProgramUniform3d(GLuint program, GLint location, GLdouble v0, GLdouble v1, GLdouble v2) {
|
||||
const GLdouble v[] = {v0, v1, v2};
|
||||
ProgramUniformv_State<3>(program, location, 1, v);
|
||||
ProgramUniformvNarrowed_State<3>(program, location, 1, v);
|
||||
}
|
||||
|
||||
void ProgramUniform3dv(GLuint program, GLint location, GLsizei count, const GLdouble* value) {
|
||||
ProgramUniformv_State<3>(program, location, count, value);
|
||||
ProgramUniformvNarrowed_State<3>(program, location, count, value);
|
||||
}
|
||||
void Uniform4d(GLint location, GLdouble v0, GLdouble v1, GLdouble v2, GLdouble v3) {
|
||||
const GLdouble v[] = {v0, v1, v2, v3};
|
||||
Uniformv_State<4>(location, 1, v);
|
||||
UniformvNarrowed_State<4>(location, 1, v);
|
||||
}
|
||||
|
||||
void Uniform4dv(GLint location, GLsizei count, const GLdouble* value) {
|
||||
Uniformv_State<4>(location, count, value);
|
||||
UniformvNarrowed_State<4>(location, count, value);
|
||||
}
|
||||
|
||||
void ProgramUniform4d(GLuint program, GLint location, GLdouble v0, GLdouble v1, GLdouble v2, GLdouble v3) {
|
||||
const GLdouble v[] = {v0, v1, v2, v3};
|
||||
ProgramUniformv_State<4>(program, location, 1, v);
|
||||
ProgramUniformvNarrowed_State<4>(program, location, 1, v);
|
||||
}
|
||||
|
||||
void ProgramUniform4dv(GLuint program, GLint location, GLsizei count, const GLdouble* value) {
|
||||
ProgramUniformv_State<4>(program, location, count, value);
|
||||
ProgramUniformvNarrowed_State<4>(program, location, count, value);
|
||||
}
|
||||
void UniformMatrix2dv(GLint location, GLsizei count, GLboolean transpose, const GLdouble* value) {
|
||||
if (location == -1) return;
|
||||
@@ -2658,6 +2887,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
void GetProgramResourceiv(GLuint program, GLenum programInterface, GLuint index, GLsizei propCount,
|
||||
const GLenum* props, GLsizei bufSize, GLsizei* length, GLint* params) {
|
||||
// Every early-out below reports "nothing was written", and it has to say so before it can
|
||||
// take one: callers legitimately leave *length uninitialised and then loop to it. The CTS
|
||||
// does exactly that (gl4cProgramInterfaceQueryTests.cpp:2172 declares `GLsizei length;` and
|
||||
// walks `for (i = 0; i < length; ++i)` over a 1000-entry stack array), so an untouched
|
||||
// *length turned every error path here into a stack overrun inside the caller -
|
||||
// KHR-GL43.program_interface_query.subroutines-vertex read 0x20202020 entries and died on
|
||||
// both backends. The success path overwrites this with the real count.
|
||||
if (length) *length = 0;
|
||||
|
||||
auto& programObject = TryToGetProgramForInterfaceQuery(program, __func__);
|
||||
if (!programObject) return;
|
||||
if (!ProgramInterface::IsInterfaceEnum(programInterface)) {
|
||||
@@ -2736,6 +2974,73 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return ProgramInterface::GetResourceLocationIndex(*programObject, programInterface, name);
|
||||
}
|
||||
|
||||
// GL 4.6 §7.7. Every property this reports is one the GL_ATOMIC_COUNTER_BUFFER interface
|
||||
// already carries, so this is a rename of glGetProgramResourceiv's props onto the older
|
||||
// entry point's - and the two are required to agree, which is only true while both read the
|
||||
// same model. It was a silent stub: it wrote nothing, raised nothing, and left every probe
|
||||
// reading its own uninitialised output.
|
||||
static Bool TryMapActiveAtomicCounterBufferProp(GLenum pname, GLenum& outProp) {
|
||||
switch (pname) {
|
||||
case GL_ATOMIC_COUNTER_BUFFER_BINDING:
|
||||
outProp = GL_BUFFER_BINDING;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_DATA_SIZE:
|
||||
outProp = GL_BUFFER_DATA_SIZE;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_ACTIVE_ATOMIC_COUNTERS:
|
||||
outProp = GL_NUM_ACTIVE_VARIABLES;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_ACTIVE_ATOMIC_COUNTER_INDICES:
|
||||
outProp = GL_ACTIVE_VARIABLES;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_REFERENCED_BY_VERTEX_SHADER:
|
||||
outProp = GL_REFERENCED_BY_VERTEX_SHADER;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_REFERENCED_BY_TESS_CONTROL_SHADER:
|
||||
outProp = GL_REFERENCED_BY_TESS_CONTROL_SHADER;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_REFERENCED_BY_TESS_EVALUATION_SHADER:
|
||||
outProp = GL_REFERENCED_BY_TESS_EVALUATION_SHADER;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_REFERENCED_BY_GEOMETRY_SHADER:
|
||||
outProp = GL_REFERENCED_BY_GEOMETRY_SHADER;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_REFERENCED_BY_FRAGMENT_SHADER:
|
||||
outProp = GL_REFERENCED_BY_FRAGMENT_SHADER;
|
||||
return true;
|
||||
case GL_ATOMIC_COUNTER_BUFFER_REFERENCED_BY_COMPUTE_SHADER:
|
||||
outProp = GL_REFERENCED_BY_COMPUTE_SHADER;
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
void GetActiveAtomicCounterBufferiv(GLuint program, GLuint bufferIndex, GLenum pname, GLint* params) {
|
||||
auto& programObject = TryToGetProgramForInterfaceQuery(program, __func__);
|
||||
if (!programObject) return;
|
||||
GLenum prop = GL_NONE;
|
||||
if (!TryMapActiveAtomicCounterBufferProp(pname, prop)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"pname is not an active atomic counter buffer property."));
|
||||
return;
|
||||
}
|
||||
Vector<GLint> values;
|
||||
if (!ProgramInterface::GetResourceProp(*programObject, GL_ATOMIC_COUNTER_BUFFER, bufferIndex, prop, values)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"bufferIndex is not an active atomic counter buffer index."));
|
||||
return;
|
||||
}
|
||||
if (params == nullptr) return;
|
||||
// GL_ATOMIC_COUNTER_BUFFER_ACTIVE_ATOMIC_COUNTER_INDICES is the only multi-value property
|
||||
// here, and the caller sized its array from _ACTIVE_ATOMIC_COUNTERS.
|
||||
for (SizeT i = 0; i < values.size(); ++i) params[i] = values[i];
|
||||
}
|
||||
|
||||
// GL 4.6 §7.6.2: <storageBlockIndex> is an active shader storage block index of <program>
|
||||
// - that is, exactly what glGetProgramResourceIndex(GL_SHADER_STORAGE_BLOCK) returned.
|
||||
// Since wave 2 that index is the interface-query layer's, so this is where the one index
|
||||
|
||||
@@ -140,6 +140,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
const GLenum* props, GLsizei bufSize, GLsizei* length, GLint* params);
|
||||
GLint GetProgramResourceLocation(GLuint program, GLenum programInterface, const GLchar* name);
|
||||
GLint GetProgramResourceLocationIndex(GLuint program, GLenum programInterface, const GLchar* name);
|
||||
void GetActiveAtomicCounterBufferiv(GLuint program, GLuint bufferIndex, GLenum pname, GLint* params);
|
||||
void ShaderStorageBlockBinding(GLuint program, GLuint storageBlockIndex, GLuint storageBlockBinding);
|
||||
void Uniform1d(GLint location, GLdouble v0);
|
||||
void Uniform1dv(GLint location, GLsizei count, const GLdouble* value);
|
||||
|
||||
@@ -19,16 +19,26 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
code, MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", function, Move(message)));
|
||||
}
|
||||
|
||||
// A pipeline name only names an object once it has been bound or created; querying a
|
||||
// reserved-but-unmaterialised name is INVALID_OPERATION (GL 4.6 core 7.4).
|
||||
// GL 4.6 core 7.4 asks only that the name came from GenProgramPipelines and has not been
|
||||
// deleted - so a name that was reserved and never bound is legal here, and the command
|
||||
// MATERIALIZES it rather than rejecting it.
|
||||
//
|
||||
// Requiring a bound object instead is what broke every separable-program conformance case
|
||||
// across three families: the CTS reserves a name, calls glUseProgramStages three times and
|
||||
// only then binds, which is the order the spec's own example uses. Each of those calls
|
||||
// failed with INVALID_OPERATION, so the stage programs were never recorded - the pipeline
|
||||
// stayed empty, GetProgramForDraw flattened nothing and the draw painted nothing, and the
|
||||
// rejected calls' error was left in the queue for the harness to find. One cause, both
|
||||
// symptoms.
|
||||
const SharedPtr<MG_State::GLState::ProgramPipelineObject>* TryGetPipeline(GLuint pipeline,
|
||||
const char* function) {
|
||||
if (!MG_State::pGLContext->IsProgramPipelineObject(pipeline)) {
|
||||
const auto& object = MG_State::pGLContext->MaterializeProgramPipelineObject(pipeline);
|
||||
if (!object) {
|
||||
RecordPipelineError(ErrorCode::InvalidOperation, function,
|
||||
std::format("Program pipeline {} does not exist.", pipeline));
|
||||
return nullptr;
|
||||
}
|
||||
return &MG_State::pGLContext->GetProgramPipelineObject(pipeline);
|
||||
return &object;
|
||||
}
|
||||
|
||||
Bool ValidatePipelineCount(GLsizei n, const char* function) {
|
||||
|
||||
@@ -19,7 +19,7 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
// "<getAtomicCounterBlockName()>_<binding>" (ParseContextBase.cpp), one per GL
|
||||
// atomic-counter binding point. That block IS the GL_ATOMIC_COUNTER_BUFFER resource
|
||||
// and its trailing number IS GL_BUFFER_BINDING; its members stay GL_UNIFORMs.
|
||||
constexpr const char* kAtomicCounterBlockPrefix = "gl_AtomicCounterBlock";
|
||||
constexpr const char* kAtomicCounterBlockPrefix = MG_Util::ShaderTranspiler::ATOMIC_COUNTER_BLOCK_PREFIX;
|
||||
|
||||
enum class BlockKind {
|
||||
Uniform, // a real GL uniform block
|
||||
@@ -81,19 +81,18 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
// The enumerated spelling of an array resource is "name[0]". glslang already applies
|
||||
// that to uniforms and buffer variables (EShReflectionBasicArraySuffix), but never to
|
||||
// stage inputs/outputs, so those get it here.
|
||||
String WithArraySuffix(const String& name, const glslang::TType* type) {
|
||||
if (type == nullptr || !type->isArray() || EndsWithZeroSubscript(name)) return name;
|
||||
String WithArraySuffix(const String& name, const ProgramObject::TypeFacts& type) {
|
||||
if (!type.isArray || EndsWithZeroSubscript(name)) return name;
|
||||
return name + "[0]";
|
||||
}
|
||||
|
||||
// GL_ARRAY_SIZE: element count for a sized array, 0 for a runtime-sized one
|
||||
// (a shader storage block's unsized trailing member), 1 for a non-array.
|
||||
GLint ArraySizeOf(const glslang::TType* type, GLint reflectedSize) {
|
||||
if (type != nullptr && type->isArray()) {
|
||||
if (!type->isSizedArray()) return 0;
|
||||
return type->getOuterArraySize();
|
||||
}
|
||||
return reflectedSize < 1 ? 1 : reflectedSize;
|
||||
// `record.arraySize` is already the sized-array/reflected-size resolution; the only
|
||||
// extra rule here is GL's 0 for a runtime-sized array.
|
||||
GLint ArraySizeOf(const ProgramObject::ResourceReflection& record) {
|
||||
if (record.type.isArray && !record.type.isSizedArray) return 0;
|
||||
return record.arraySize;
|
||||
}
|
||||
|
||||
// Two spellings name the same resource when they are equal, or differ only by the
|
||||
@@ -174,22 +173,21 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
return static_cast<GLint>(element);
|
||||
}
|
||||
|
||||
BlockKind ClassifyBlock(const glslang::TObjectReflection& block) {
|
||||
BlockKind ClassifyBlock(const ProgramObject::BlockReflection& block) {
|
||||
if (std::strstr(block.name.c_str(), MG_Util::ShaderTranspiler::GLOBAL_UBO_NAME) != nullptr) {
|
||||
return BlockKind::GlobalUbo;
|
||||
}
|
||||
if (IsAtomicCounterBlockName(block.name)) return BlockKind::AtomicCounter;
|
||||
const glslang::TType* type = block.getType();
|
||||
if (type != nullptr && type->getQualifier().storage == glslang::EvqBuffer) return BlockKind::Storage;
|
||||
if (block.type.isBuffer) return BlockKind::Storage;
|
||||
return BlockKind::Uniform;
|
||||
}
|
||||
|
||||
// std140/std430 column stride, the same vec4-rounded rule ProgramObject applies to
|
||||
// uniform matrices. 0 for a non-matrix.
|
||||
GLint MatrixStrideOf(const glslang::TType* type) {
|
||||
if (type == nullptr || !type->isMatrix()) return 0;
|
||||
const bool rowMajor = type->getQualifier().layoutMatrix == glslang::ElmRowMajor;
|
||||
const int strideVectorComponents = rowMajor ? type->getMatrixCols() : type->getMatrixRows();
|
||||
GLint MatrixStrideOf(const ProgramObject::TypeFacts& type) {
|
||||
if (!type.isMatrix) return 0;
|
||||
const bool rowMajor = type.layoutMatrix == static_cast<Int>(glslang::ElmRowMajor);
|
||||
const int strideVectorComponents = rowMajor ? type.matrixCols : type.matrixRows;
|
||||
constexpr int scalarSize = 4;
|
||||
const int vectorAlignment = (strideVectorComponents <= 1) ? scalarSize
|
||||
: (strideVectorComponents == 2) ? 2 * scalarSize
|
||||
@@ -197,9 +195,9 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
return (vectorAlignment + 15) & ~15;
|
||||
}
|
||||
|
||||
GLint IsRowMajorOf(const glslang::TType* type) {
|
||||
if (type == nullptr || !type->isMatrix()) return 0;
|
||||
return type->getQualifier().layoutMatrix == glslang::ElmRowMajor ? 1 : 0;
|
||||
GLint IsRowMajorOf(const ProgramObject::TypeFacts& type) {
|
||||
if (!type.isMatrix) return 0;
|
||||
return type.layoutMatrix == static_cast<Int>(glslang::ElmRowMajor) ? 1 : 0;
|
||||
}
|
||||
|
||||
GLint MappedLocation(Int rawLocation) {
|
||||
@@ -210,14 +208,69 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
|
||||
// ---- model construction --------------------------------------------------------
|
||||
|
||||
void BuildBlocks(ProgramObject& program, const glslang::TProgram& reflection, Model& model,
|
||||
// GL_REFERENCED_BY_*_SHADER for an ARRAYED block instance, refined per element.
|
||||
//
|
||||
// glslang records a block reference by walking up to the base symbol and calling
|
||||
// addBlockName with the whole ARRAY type, which ORs the referencing stage into every
|
||||
// element at once - it has not resolved the subscript yet at that point. So reading
|
||||
// "e[0].b" marks both TrickyBlock[0] and TrickyBlock[1] as referenced by the fragment
|
||||
// stage (KHR-GL43.program_interface_query.uniform-block-types).
|
||||
//
|
||||
// The MEMBER masks are exact: EShReflectionAllBlockVariables enumerates every member of
|
||||
// every element with the stage mask suppressed, and only the dereference chain actually
|
||||
// walked turns a bit on - and that chain carries the subscript. So the union of a block
|
||||
// instance's members is the reference set of that instance.
|
||||
//
|
||||
// Applied ONLY to arrayed instances, because for a scalar block glslang is already exact.
|
||||
// Note the union is used even when it is empty: an array element nobody dereferenced has
|
||||
// no member bits and is genuinely referenced by nobody, which is the whole point - falling
|
||||
// back to the block's own mask there would restore the over-approximation.
|
||||
Vector<Uint32> BuildBlockStagesFromMembers(const ProgramObject::LinkArtifacts& reflection,
|
||||
Int blockCount) {
|
||||
Vector<Uint32> stagesByBlock(static_cast<SizeT>(blockCount < 0 ? 0 : blockCount), 0u);
|
||||
const Int uniformCount = static_cast<Int>(reflection.uniformReflection.size());
|
||||
for (Int index = 0; index < uniformCount; ++index) {
|
||||
const auto& uniform = reflection.uniformReflection[index];
|
||||
const Int owner = uniform.index;
|
||||
if (owner < 0 || owner >= blockCount) continue;
|
||||
stagesByBlock[static_cast<SizeT>(owner)] |= static_cast<Uint32>(uniform.stages);
|
||||
}
|
||||
return stagesByBlock;
|
||||
}
|
||||
|
||||
// UNIFORM blocks only, and that scope is load-bearing rather than cautious. The member
|
||||
// names glslang produces for a uniform block array carry the subscript
|
||||
// ("TrickyBlock[0].b", via EShReflectionStrictArraySuffix), so each element's members are
|
||||
// distinct entries and the bits land on the right one. A SHADER STORAGE block array does
|
||||
// NOT get that treatment - its buffer variables reflect under one subscript-free spelling
|
||||
// shared by every element - so a union over them credits element 0 and starves the rest.
|
||||
// KHR-GL43.program_interface_query.ssb-types is the case that says so: it reads ss[0] and
|
||||
// ss[1] and requires both to report the fragment stage, which only glslang's own
|
||||
// (deliberately over-approximating) block mask gets right. Storage and atomic-counter
|
||||
// blocks therefore keep that mask untouched.
|
||||
Uint32 UniformBlockStages(const ProgramObject::BlockReflection& block, const Vector<Uint32>& stagesFromMembers,
|
||||
Int tIndex) {
|
||||
String arrayBase;
|
||||
Uint element = 0;
|
||||
Bool malformed = false;
|
||||
if (!SplitTrailingSubscript(block.name, arrayBase, element, malformed) || malformed) {
|
||||
return static_cast<Uint32>(block.stages);
|
||||
}
|
||||
if (tIndex < 0 || tIndex >= static_cast<Int>(stagesFromMembers.size())) {
|
||||
return static_cast<Uint32>(block.stages);
|
||||
}
|
||||
return stagesFromMembers[static_cast<SizeT>(tIndex)];
|
||||
}
|
||||
|
||||
void BuildBlocks(ProgramObject& program, const ProgramObject::LinkArtifacts& reflection, Model& model,
|
||||
Vector<BlockKind>& blockKind, Vector<Int>& blockInterfaceIndex) {
|
||||
const Int blockCount = const_cast<glslang::TProgram&>(reflection).getNumUniformBlocks();
|
||||
const Int blockCount = static_cast<Int>(reflection.blockReflection.size());
|
||||
blockKind.assign(blockCount, BlockKind::Uniform);
|
||||
blockInterfaceIndex.assign(blockCount, -1);
|
||||
const Vector<Uint32> stagesFromMembers = BuildBlockStagesFromMembers(reflection, blockCount);
|
||||
|
||||
for (Int tIndex = 0; tIndex < blockCount; ++tIndex) {
|
||||
const auto& block = const_cast<glslang::TProgram&>(reflection).getUniformBlock(tIndex);
|
||||
const auto& block = reflection.blockReflection[tIndex];
|
||||
const BlockKind kind = ClassifyBlock(block);
|
||||
blockKind[tIndex] = kind;
|
||||
if (kind == BlockKind::AtomicCounter) {
|
||||
@@ -238,7 +291,7 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
// glShaderStorageBlockBinding wins over the declaration (GL 4.6 §7.6.2 -
|
||||
// exactly the same rule GL_UNIFORM_BLOCK follows through
|
||||
// GetUniformBlockBinding below).
|
||||
const GLint declared = block.getBinding();
|
||||
const GLint declared = block.binding;
|
||||
resource.bufferBinding = declared < 0 ? 0 : declared + BlockArrayElement(block.name);
|
||||
const Int rebound = program.GetShaderStorageBlockBindingOverride(block.name);
|
||||
if (rebound >= 0) resource.bufferBinding = static_cast<GLint>(rebound);
|
||||
@@ -252,38 +305,53 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
// GL_UNIFORM_BLOCK keeps the index space glUniformBlockBinding and
|
||||
// glGetActiveUniformBlockiv already use, so an index handed out here is usable
|
||||
// with them (which is exactly what the CTS does).
|
||||
const Int glBlockCount = program.GetActiveUniformBlocksCount();
|
||||
const Int glBlockCount = program.GetGlUniformBlockCount();
|
||||
for (Int glIndex = 0; glIndex < glBlockCount; ++glIndex) {
|
||||
// The block-space index the block-keyed accessors want; the two spaces differ
|
||||
// whenever the program also has a storage or atomic counter block, which
|
||||
// glslang files under the same reflection list (no EShReflectionSeparateBuffers).
|
||||
const Int blockIndex = program.BlockIndexFromGlUniformBlock(static_cast<Uint>(glIndex));
|
||||
Resource resource;
|
||||
resource.name = program.GetUniformBlockName(glIndex);
|
||||
resource.bufferBinding = static_cast<GLint>(program.GetUniformBlockBinding(glIndex));
|
||||
resource.bufferDataSize = static_cast<GLint>(program.GetUBOSizeAt(glIndex));
|
||||
const Int tIndex = program.TProgramBlockIndex(static_cast<Uint>(glIndex));
|
||||
resource.name = program.GetUniformBlockName(static_cast<Uint>(blockIndex));
|
||||
resource.bufferBinding = static_cast<GLint>(program.GetUniformBlockBinding(static_cast<Uint>(blockIndex)));
|
||||
resource.bufferDataSize = static_cast<GLint>(program.GetUBOSizeAt(static_cast<Uint>(blockIndex)));
|
||||
const Int tIndex = program.TProgramBlockIndex(static_cast<Uint>(blockIndex));
|
||||
if (tIndex >= 0 && tIndex < blockCount) {
|
||||
resource.stages =
|
||||
static_cast<Uint32>(const_cast<glslang::TProgram&>(reflection).getUniformBlock(tIndex).stages);
|
||||
resource.stages = UniformBlockStages(reflection.blockReflection[tIndex],
|
||||
stagesFromMembers, tIndex);
|
||||
}
|
||||
model.uniformBlocks.push_back(Move(resource));
|
||||
}
|
||||
}
|
||||
|
||||
void BuildUniformsAndBufferVariables(ProgramObject& program, const glslang::TProgram& reflection, Model& model,
|
||||
void BuildUniformsAndBufferVariables(ProgramObject& program,
|
||||
const ProgramObject::LinkArtifacts& reflection, Model& model,
|
||||
const Vector<BlockKind>& blockKind,
|
||||
const Vector<Int>& blockInterfaceIndex) {
|
||||
const Uint uniformCount = program.GetUniformCount();
|
||||
for (Uint glIndex = 0; glIndex < uniformCount; ++glIndex) {
|
||||
const Int tIndex = program.TProgramUniformIndex(glIndex);
|
||||
const auto& refl = const_cast<glslang::TProgram&>(reflection).getUniform(tIndex);
|
||||
const glslang::TType* type = refl.getType();
|
||||
// Walks the TPROGRAM uniform space, not the GL one. A buffer variable is not a GL
|
||||
// uniform (GL 4.6 core 7.3.1) and DoReflection therefore keeps it out of the GL
|
||||
// active-uniform index space - but GL_BUFFER_VARIABLE still has to enumerate it, and
|
||||
// this is the only place that does. GL uniforms keep their GL index as their
|
||||
// GL_UNIFORM resource index: the GL space is a subsequence of this one, so pushing
|
||||
// the GL-visible entries in this order preserves the correspondence.
|
||||
const Int tUniformCount = static_cast<Int>(reflection.uniformReflection.size());
|
||||
for (Int tIndex = 0; tIndex < tUniformCount; ++tIndex) {
|
||||
const auto& refl = ProgramObject::UniformAtIn(reflection, tIndex);
|
||||
const auto& type = refl.type;
|
||||
const Int owner = refl.index;
|
||||
const BlockKind kind = (owner >= 0 && owner < static_cast<Int>(blockKind.size()))
|
||||
? blockKind[owner]
|
||||
: BlockKind::GlobalUbo;
|
||||
const Int glIndex = program.GlUniformIndexFromTProgram(tIndex);
|
||||
// Everything except a buffer variable is enumerated through the GL space, so a
|
||||
// uniform the relaxed parse swept out of it (a declared-but-dead default-block
|
||||
// one) stays out of GL_UNIFORM too.
|
||||
if (kind != BlockKind::Storage && glIndex < 0) continue;
|
||||
|
||||
Resource resource;
|
||||
resource.name = refl.name;
|
||||
resource.type = static_cast<GLenum>(refl.glDefineType);
|
||||
resource.arraySize = ArraySizeOf(type, refl.size);
|
||||
resource.arraySize = ArraySizeOf(refl);
|
||||
resource.stages = static_cast<Uint32>(refl.stages);
|
||||
|
||||
if (kind == BlockKind::Storage) {
|
||||
@@ -311,11 +379,12 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
resource.atomicCounterBufferIndex = blockInterfaceIndex[owner];
|
||||
resource.location = -1;
|
||||
} else {
|
||||
resource.blockIndex = program.GetActiveUniformBlockIndex(glIndex);
|
||||
resource.offset = program.GetActiveUniformOffset(glIndex);
|
||||
resource.arrayStride = program.GetActiveUniformArrayStride(glIndex);
|
||||
resource.matrixStride = program.GetActiveUniformMatrixStride(glIndex);
|
||||
resource.isRowMajor = program.GetActiveUniformIsRowMajor(glIndex);
|
||||
const Uint glUniformIndex = static_cast<Uint>(glIndex);
|
||||
resource.blockIndex = program.GetActiveUniformBlockIndex(glUniformIndex);
|
||||
resource.offset = program.GetActiveUniformOffset(glUniformIndex);
|
||||
resource.arrayStride = program.GetActiveUniformArrayStride(glUniformIndex);
|
||||
resource.matrixStride = program.GetActiveUniformMatrixStride(glUniformIndex);
|
||||
resource.isRowMajor = program.GetActiveUniformIsRowMajor(glUniformIndex);
|
||||
// A member of a named uniform block has no location, whatever the
|
||||
// frontend's own location table says (it hands one out to every uniform
|
||||
// so glUniform* can address block members through the global UBO).
|
||||
@@ -334,12 +403,16 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
static_cast<GLuint>(i));
|
||||
}
|
||||
}
|
||||
for (SizeT blockIndex = 0; blockIndex < model.uniformBlocks.size(); ++blockIndex) {
|
||||
for (SizeT glBlockIndex = 0; glBlockIndex < model.uniformBlocks.size(); ++glBlockIndex) {
|
||||
// Members of an arrayed block are reflected once, against instance [0].
|
||||
const Int owner = static_cast<Int>(program.GetUniformBlockMemberOwnerIndex(static_cast<Uint>(blockIndex)));
|
||||
// GetUniformBlockMemberOwnerIndex takes and answers BLOCK indices, while
|
||||
// Resource::blockIndex is a GL_UNIFORM_BLOCK index, so translate both ways.
|
||||
const Int blockIndex = program.BlockIndexFromGlUniformBlock(static_cast<Uint>(glBlockIndex));
|
||||
const Int owner = program.GlUniformBlockIndexFromBlock(
|
||||
static_cast<Int>(program.GetUniformBlockMemberOwnerIndex(static_cast<Uint>(blockIndex))));
|
||||
for (SizeT i = 0; i < model.uniforms.size(); ++i) {
|
||||
if (model.uniforms[i].blockIndex == owner) {
|
||||
model.uniformBlocks[blockIndex].activeVariables.push_back(static_cast<GLuint>(i));
|
||||
model.uniformBlocks[glBlockIndex].activeVariables.push_back(static_cast<GLuint>(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -352,49 +425,67 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
}
|
||||
}
|
||||
|
||||
void BuildStageIO(ProgramObject& program, const glslang::TProgram& reflection, Model& model) {
|
||||
auto& mutableReflection = const_cast<glslang::TProgram&>(reflection);
|
||||
// A built-in interface block that a shader redeclares with fewer members keeps the
|
||||
// omitted ones in its type when the redeclaration is ANONYMOUS - glslang hides them
|
||||
// (basic type void) instead of erasing them, because the original shared declaration
|
||||
// has to stay usable. Only the instance-named form erases. So a separable vertex
|
||||
// program that redeclares `out gl_PerVertex { vec4 gl_Position; }` still carries
|
||||
// gl_PointSize and gl_ClipDistance through the block-unwrapping reflection, and they
|
||||
// are not part of its output interface.
|
||||
Bool IsHiddenBlockMember(const ProgramObject::TypeFacts& type) { return type.isVoid; }
|
||||
|
||||
const Int inputCount = mutableReflection.getNumPipeInputs();
|
||||
void BuildStageIO(ProgramObject& program, const ProgramObject::LinkArtifacts& reflection, Model& model) {
|
||||
const Int inputCount = static_cast<Int>(reflection.pipeInputReflection.size());
|
||||
for (Int index = 0; index < inputCount; ++index) {
|
||||
const auto& refl = mutableReflection.getPipeInput(index);
|
||||
const glslang::TType* type = refl.getType();
|
||||
const auto& refl = reflection.pipeInputReflection[index];
|
||||
const auto& type = refl.type;
|
||||
if (IsHiddenBlockMember(type)) continue;
|
||||
Resource resource;
|
||||
// The Vulkan-semantics parse reflects the vertex builtins under their SPIR-V
|
||||
// names; GL enumerates the GL spellings.
|
||||
const String& glName = ProgramObject::NormalizeBuiltinPipeInputName(refl.name);
|
||||
resource.name = WithArraySuffix(glName, type);
|
||||
resource.type = static_cast<GLenum>(refl.glDefineType);
|
||||
resource.arraySize = ArraySizeOf(type, refl.size);
|
||||
resource.arraySize = ArraySizeOf(refl);
|
||||
resource.location = program.GetAttributeLocation(refl.name);
|
||||
if (resource.location < 0) resource.location = MappedLocation(static_cast<Int>(refl.layoutLocation()));
|
||||
resource.isPerPatch = (type != nullptr && type->getQualifier().patch) ? 1 : 0;
|
||||
if (resource.location < 0) resource.location = MappedLocation(refl.location);
|
||||
resource.isPerPatch = type.isPatch ? 1 : 0;
|
||||
resource.stages = static_cast<Uint32>(refl.stages);
|
||||
model.programInputs.push_back(Move(resource));
|
||||
}
|
||||
|
||||
const Int outputCount = mutableReflection.getNumPipeOutputs();
|
||||
// A color number, and therefore a color INDEX, exists only for a fragment stage's
|
||||
// outputs. The output interface belongs to the program's last stage, so for a
|
||||
// separable tessellation/geometry/vertex program these are varyings: asking the
|
||||
// frag-data maps about them can still answer a location (a tess-control output
|
||||
// carries its own layout(location=N)), and a location then manufactures a color
|
||||
// index of 0 where GL requires -1
|
||||
// (KHR-GL43.program_interface_query.separate-programs-tess-control).
|
||||
const Bool lastStageIsFragment = reflection.lastStageIsFragment;
|
||||
const Int outputCount = static_cast<Int>(reflection.pipeOutputReflection.size());
|
||||
for (Int index = 0; index < outputCount; ++index) {
|
||||
const auto& refl = mutableReflection.getPipeOutput(index);
|
||||
const glslang::TType* type = refl.getType();
|
||||
const auto& refl = reflection.pipeOutputReflection[index];
|
||||
const auto& type = refl.type;
|
||||
if (IsHiddenBlockMember(type)) continue;
|
||||
Resource resource;
|
||||
resource.name = WithArraySuffix(refl.name, type);
|
||||
resource.type = static_cast<GLenum>(refl.glDefineType);
|
||||
resource.arraySize = ArraySizeOf(type, refl.size);
|
||||
resource.arraySize = ArraySizeOf(refl);
|
||||
resource.location = MappedLocation(program.GetFragmentDataLocation(refl.name.c_str()));
|
||||
if (resource.location < 0) {
|
||||
// A built-in output (gl_FragDepth, gl_SampleMask) and a non-fragment stage
|
||||
// output both have no location, and therefore no color index either.
|
||||
if (resource.location < 0 || !lastStageIsFragment) {
|
||||
// A built-in output (gl_FragDepth, gl_SampleMask) has no location, and a
|
||||
// non-fragment stage's outputs have no color number at all - either way there
|
||||
// is no color index.
|
||||
resource.locationIndex = -1;
|
||||
} else {
|
||||
resource.locationIndex = program.GetFragmentDataIndex(refl.name.c_str());
|
||||
// glBindFragDataLocationIndexed wins; otherwise the shader's
|
||||
// layout(index = N), which the frag-data maps never saw.
|
||||
if (resource.locationIndex == 0 && type != nullptr && type->getQualifier().hasIndex()) {
|
||||
resource.locationIndex = static_cast<GLint>(type->getQualifier().layoutIndex);
|
||||
if (resource.locationIndex == 0 && type.hasIndex) {
|
||||
resource.locationIndex = static_cast<GLint>(type.layoutIndex);
|
||||
}
|
||||
}
|
||||
resource.isPerPatch = (type != nullptr && type->getQualifier().patch) ? 1 : 0;
|
||||
resource.isPerPatch = type.isPatch ? 1 : 0;
|
||||
resource.stages = static_cast<Uint32>(refl.stages);
|
||||
model.programOutputs.push_back(Move(resource));
|
||||
}
|
||||
@@ -434,15 +525,14 @@ namespace MobileGL::MG_Impl::GLImpl::ProgramInterface {
|
||||
Model BuildModel(ProgramObject& program) {
|
||||
Model model;
|
||||
if (!program.GetLinkStatus()) return model;
|
||||
const glslang::TProgram* reflection = program.GetReflection();
|
||||
if (reflection == nullptr) return model;
|
||||
const ProgramObject::LinkArtifacts& reflection = program.GetLinkReflection();
|
||||
model.valid = true;
|
||||
|
||||
Vector<BlockKind> blockKind;
|
||||
Vector<Int> blockInterfaceIndex;
|
||||
BuildBlocks(program, *reflection, model, blockKind, blockInterfaceIndex);
|
||||
BuildUniformsAndBufferVariables(program, *reflection, model, blockKind, blockInterfaceIndex);
|
||||
BuildStageIO(program, *reflection, model);
|
||||
BuildBlocks(program, reflection, model, blockKind, blockInterfaceIndex);
|
||||
BuildUniformsAndBufferVariables(program, reflection, model, blockKind, blockInterfaceIndex);
|
||||
BuildStageIO(program, reflection, model);
|
||||
BuildXfb(program, model);
|
||||
return model;
|
||||
}
|
||||
|
||||
@@ -31,8 +31,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
Bool ended = false;
|
||||
Bool resultCached = false;
|
||||
Uint64 cachedResult = 0;
|
||||
// Transform feedback primitive counter at BeginQuery time.
|
||||
// The transform feedback primitive counter matching this query's target, at
|
||||
// BeginQuery time.
|
||||
Uint64 counterSnapshot = 0;
|
||||
// Capture-draw counters at BeginQuery time: how many capture draws the CPU
|
||||
// accounting had reproduced exactly, and how many of those it could not (a
|
||||
// geometry stage amplifies). Their deltas decide whether the CPU result may
|
||||
// stand in for the backend's.
|
||||
Uint64 accountedCaptureDrawSnapshot = 0;
|
||||
Uint64 geometryCaptureDrawSnapshot = 0;
|
||||
};
|
||||
|
||||
// Query calls may arrive from any thread (launchers migrate the context
|
||||
@@ -122,6 +129,46 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
g_activeTimeElapsedQueryId = 0;
|
||||
}
|
||||
|
||||
// The CPU accounting counter a transform feedback query target reads: what the capture
|
||||
// buffers took for GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN, and everything the capture
|
||||
// stage assembled - a paused span included - for GL_PRIMITIVES_GENERATED. One counter
|
||||
// for both targets would report the clamped written count as the generated one.
|
||||
Uint64 TransformFeedbackCounterForTarget(GLenum target) {
|
||||
return target == GL_PRIMITIVES_GENERATED
|
||||
? MG_State::pGLContext->GetTransformFeedbackGeneratedCounter()
|
||||
: MG_State::pGLContext->GetTransformFeedbackPrimitiveCounter();
|
||||
}
|
||||
|
||||
// The span's CPU accounting delta. Saturating: a snapshot left above its counter (a
|
||||
// context switch between Begin and End, a counter that never moved) would otherwise
|
||||
// wrap to 2^64-1, which GetQueryObjectuiv hands the app as 4294967295.
|
||||
Uint64 TransformFeedbackCpuResult(const QueryObject* queryObject) {
|
||||
const Uint64 counter = TransformFeedbackCounterForTarget(queryObject->target);
|
||||
return counter > queryObject->counterSnapshot ? counter - queryObject->counterSnapshot : 0;
|
||||
}
|
||||
|
||||
// Whether this ended span's result should come from the CPU accounting rather than from
|
||||
// the backend query it also ran. Three conditions, all necessary:
|
||||
// * the backend asked for it (DirectGLES, whose ES driver counter is the unreliable
|
||||
// one; DirectVulkan never sets the bit and so is untouched by any of this);
|
||||
// * the target is PRIMITIVES_WRITTEN. GL_PRIMITIVES_GENERATED counts primitives
|
||||
// whether or not a capture is active, and the accounting only ever sees capture
|
||||
// draws, so the backend's counter is the more complete answer there;
|
||||
// * the span was fully accounted: at least one capture draw reached the accounting
|
||||
// (the instanced, indirect and multi-draw entry points do not call it at all, so a
|
||||
// span made of those is invisible to it) and none of them amplified through a
|
||||
// geometry stage, which the CPU cannot model.
|
||||
Bool PrefersCpuTransformFeedbackResult(const QueryObject* queryObject) {
|
||||
if (!MG_Backend::gBackendFunctionsTable.GL.PrefersCpuXfbPrimitiveAccounting) return false;
|
||||
if (queryObject->target != GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN) return false;
|
||||
if (MG_State::pGLContext->GetTransformFeedbackGeometryCaptureDraws() !=
|
||||
queryObject->geometryCaptureDrawSnapshot) {
|
||||
return false;
|
||||
}
|
||||
return MG_State::pGLContext->GetTransformFeedbackAccountedCaptureDraws() !=
|
||||
queryObject->accountedCaptureDrawSnapshot;
|
||||
}
|
||||
|
||||
// Shared GetQueryObject* implementation. Returns false when an error
|
||||
// was recorded and no value should be written back. `outValueProduced`, when given,
|
||||
// additionally distinguishes "succeeded with a value" from "succeeded but the result is not
|
||||
@@ -407,7 +454,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
const auto beginXfbPrimitivesQuery = MG_Backend::gBackendFunctionsTable.GL.BeginXfbPrimitivesQuery;
|
||||
queryObject->backendHandle =
|
||||
beginXfbPrimitivesQuery ? beginXfbPrimitivesQuery(target == GL_PRIMITIVES_GENERATED) : nullptr;
|
||||
queryObject->counterSnapshot = MG_State::pGLContext->GetTransformFeedbackPrimitiveCounter();
|
||||
queryObject->counterSnapshot = TransformFeedbackCounterForTarget(target);
|
||||
queryObject->accountedCaptureDrawSnapshot =
|
||||
MG_State::pGLContext->GetTransformFeedbackAccountedCaptureDraws();
|
||||
queryObject->geometryCaptureDrawSnapshot =
|
||||
MG_State::pGLContext->GetTransformFeedbackGeometryCaptureDraws();
|
||||
} else if (isOcclusionQuery) {
|
||||
queryObject->backendHandle = MG_Backend::gBackendFunctionsTable.GL.BeginOcclusionQuery();
|
||||
} else {
|
||||
@@ -448,12 +499,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (const auto endXfbPrimitivesQuery = MG_Backend::gBackendFunctionsTable.GL.EndXfbPrimitivesQuery) {
|
||||
endXfbPrimitivesQuery(queryObject->backendHandle);
|
||||
}
|
||||
// Result comes from the GPU query at read time.
|
||||
} else {
|
||||
queryObject->cachedResult =
|
||||
MG_State::pGLContext->GetTransformFeedbackPrimitiveCounter() - queryObject->counterSnapshot;
|
||||
}
|
||||
// A backend query that is not going to be read is released here, not left to be
|
||||
// collected later: the span is over, the driver object has nothing left to say.
|
||||
// Ending it first is what makes that legal.
|
||||
if (!queryObject->backendHandle || PrefersCpuTransformFeedbackResult(queryObject)) {
|
||||
if (queryObject->backendHandle) {
|
||||
if (const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery) {
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
queryObject->backendHandle = nullptr;
|
||||
}
|
||||
queryObject->cachedResult = TransformFeedbackCpuResult(queryObject);
|
||||
queryObject->resultCached = true;
|
||||
}
|
||||
// Otherwise the result comes from the GPU query at read time.
|
||||
queryObject->active = false;
|
||||
queryObject->ended = true;
|
||||
activeQueryId = 0;
|
||||
@@ -505,6 +565,75 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
queryObject->ended = true;
|
||||
}
|
||||
|
||||
void BeginConditionalRender(GLuint id, GLenum mode) {
|
||||
// GL 4.6 core 10.9's eight modes. The _INVERTED half flips the sense of the predicate;
|
||||
// the BY_REGION half only narrows WHERE an implementation is permitted to discard, so
|
||||
// treating it as its whole-framebuffer sibling is what an implementation without region
|
||||
// granularity does. The _NO_WAIT half is a permission to render rather than stall, not an
|
||||
// obligation - see the resolve below.
|
||||
Bool inverted = false;
|
||||
switch (mode) {
|
||||
case GL_QUERY_WAIT:
|
||||
case GL_QUERY_NO_WAIT:
|
||||
case GL_QUERY_BY_REGION_WAIT:
|
||||
case GL_QUERY_BY_REGION_NO_WAIT:
|
||||
inverted = false;
|
||||
break;
|
||||
case GL_QUERY_WAIT_INVERTED:
|
||||
case GL_QUERY_NO_WAIT_INVERTED:
|
||||
case GL_QUERY_BY_REGION_WAIT_INVERTED:
|
||||
case GL_QUERY_BY_REGION_NO_WAIT_INVERTED:
|
||||
inverted = true;
|
||||
break;
|
||||
default:
|
||||
RecordQueryError(ErrorCode::InvalidEnum, __FUNCTION__, "mode is not a conditional render mode.");
|
||||
return;
|
||||
}
|
||||
|
||||
if (MG_State::pGLContext->IsConditionalRenderActive()) {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__, "Conditional rendering is already active.");
|
||||
return;
|
||||
}
|
||||
|
||||
{
|
||||
const std::lock_guard<std::mutex> lock(g_queryObjectsMutex);
|
||||
const auto* queryObject = FindQueryObjectLocked(id);
|
||||
// A generated NAME is not yet a query object; it becomes one at its first use with a
|
||||
// target (the same rule glIsQuery answers by).
|
||||
if (!queryObject || (!queryObject->created && queryObject->target == 0)) {
|
||||
RecordQueryError(ErrorCode::InvalidValue, __FUNCTION__, "id is not the name of a query object.");
|
||||
return;
|
||||
}
|
||||
if (queryObject->active) {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__, "The query object is still active.");
|
||||
return;
|
||||
}
|
||||
if (queryObject->target != GL_SAMPLES_PASSED && queryObject->target != GL_ANY_SAMPLES_PASSED &&
|
||||
queryObject->target != GL_ANY_SAMPLES_PASSED_CONSERVATIVE) {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__,
|
||||
"Conditional rendering requires an occlusion query object.");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// Resolved ONCE, here, and by WAITING even for the _NO_WAIT modes: the spec lets those
|
||||
// render instead of stalling, so always waiting is conforming and is the only choice that
|
||||
// gives the whole block one deterministic verdict. Reading it per command instead would
|
||||
// let a result that lands mid-block change the answer half way through.
|
||||
Uint64 samplesPassed = 0;
|
||||
if (!GetQueryObjectValue(id, GL_QUERY_RESULT, __FUNCTION__, samplesPassed)) return;
|
||||
const Bool passed = samplesPassed != 0;
|
||||
MG_State::pGLContext->BeginConditionalRender(id, mode, inverted ? passed : !passed);
|
||||
}
|
||||
|
||||
void EndConditionalRender() {
|
||||
if (!MG_State::pGLContext->IsConditionalRenderActive()) {
|
||||
RecordQueryError(ErrorCode::InvalidOperation, __FUNCTION__, "Conditional rendering is not active.");
|
||||
return;
|
||||
}
|
||||
MG_State::pGLContext->EndConditionalRender();
|
||||
}
|
||||
|
||||
void GetQueryiv(GLenum target, GLenum pname, GLint* params) {
|
||||
if (!params) {
|
||||
return;
|
||||
@@ -648,4 +777,39 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
if (!ValidateQueryStreamIndex(__FUNCTION__, target, index)) return;
|
||||
GetQueryiv(target, pname, params);
|
||||
}
|
||||
|
||||
void DestroyAllQueryObjects() {
|
||||
// Detach the registry under the lock, release outside it - same discipline
|
||||
// (and the same accepted teardown race) as DestroyAllSyncObjects. Without
|
||||
// this drain, every query the app left undeleted survived full library
|
||||
// teardown in the process-global registry: the objects and their backend
|
||||
// wrappers leaked across Destroy/Initialize cycles, stale ids kept
|
||||
// answering IsQuery == GL_TRUE in the re-initialized library, and a later
|
||||
// glDeleteQueries could hand the OLD backend's handle to a DIFFERENT
|
||||
// backend's DeleteBackendQuery, which casts it to the wrong wrapper type.
|
||||
UnorderedMap<GLuint, QueryObject*> orphans;
|
||||
{
|
||||
const std::lock_guard<std::mutex> lock(g_queryObjectsMutex);
|
||||
orphans.swap(g_liveQueryObjects);
|
||||
g_activeTimeElapsedQueryId = 0;
|
||||
g_activePrimitivesWrittenQueryId = 0;
|
||||
g_activePrimitivesGeneratedQueryId = 0;
|
||||
g_activeSamplesPassedQueryId = 0;
|
||||
}
|
||||
if (orphans.empty()) {
|
||||
return;
|
||||
}
|
||||
// Backend handles must be released by the backend that created them, so
|
||||
// this runs while the function table is still populated. Both backends'
|
||||
// DeleteBackendQuery are generation-guarded, so a handle whose renderer
|
||||
// or ES context is already gone frees only the wrapper.
|
||||
const auto deleteBackendQuery = MG_Backend::gBackendFunctionsTable.GL.DeleteBackendQuery;
|
||||
for (const auto& [_, queryObject] : orphans) {
|
||||
if (deleteBackendQuery && queryObject->backendHandle) {
|
||||
deleteBackendQuery(queryObject->backendHandle);
|
||||
}
|
||||
delete queryObject;
|
||||
}
|
||||
MGLOG_D("DestroyAllQueryObjects: reclaimed %zu query object(s) the app left undeleted", orphans.size());
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -29,4 +29,18 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void GetQueryBufferObjecti64v(GLuint id, GLuint buffer, GLenum pname, GLintptr offset);
|
||||
void GetQueryBufferObjectui64v(GLuint id, GLuint buffer, GLenum pname, GLintptr offset);
|
||||
void QueryCounter(GLuint id, GLenum target);
|
||||
// Conditional rendering (GL 4.6 core 10.9). Implemented here rather than beside the drawing
|
||||
// entry points because the predicate is a QUERY OBJECT's result, and the object registry -
|
||||
// with the lock that guards it - lives in this file.
|
||||
void BeginConditionalRender(GLuint id, GLenum mode);
|
||||
void EndConditionalRender();
|
||||
// Destroys every still-registered query object exactly as DeleteQueries would.
|
||||
// GL requires queries to die with their context; called only from full library
|
||||
// teardown (DestroyImpl), where no context survives on any thread, so the
|
||||
// process-global registry can be drained wholesale. Must run while the backend
|
||||
// function table is still populated: each backend handle has to be released by
|
||||
// the backend that created it, never by a later re-initialized one (whose
|
||||
// DeleteBackendQuery would cast the wrapper to the wrong backend's type).
|
||||
// Same contract as DestroyAllSyncObjects.
|
||||
void DestroyAllQueryObjects();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "GL_RenderState.h"
|
||||
#include <cmath>
|
||||
#include <MG_Impl/GLImpl/Getter/GL_Getter.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
|
||||
#include <MG_Util/Converters/GLToMG/RenderStateEnumConverter.h>
|
||||
@@ -19,28 +20,118 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return std::clamp(static_cast<Float>(value), 0.0f, 1.0f);
|
||||
}
|
||||
|
||||
static Bool ValidateIndexedBlendCapability(GLenum target, GLuint index, const char* functionName) {
|
||||
if (target != GL_BLEND) {
|
||||
// GL 4.6 core 17.3.2 and 22.1 give exactly two indexed capabilities: GL_BLEND, indexed by
|
||||
// draw buffer, and GL_SCISSOR_TEST, indexed by viewport. They have DIFFERENT bounds
|
||||
// (MAX_DRAW_BUFFERS vs MAX_VIEWPORTS), so the limit is picked per target rather than shared.
|
||||
static Bool ValidateIndexedCapability(GLenum target, GLuint index, const char* functionName) {
|
||||
GLuint limit = 0;
|
||||
const char* indexName = nullptr;
|
||||
switch (target) {
|
||||
case GL_BLEND:
|
||||
limit = MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS;
|
||||
indexName = "Buffer";
|
||||
break;
|
||||
case GL_SCISSOR_TEST:
|
||||
limit = RenderStateParameters::MAX_VIEWPORTS;
|
||||
indexName = "Viewport";
|
||||
break;
|
||||
default:
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"Only GL_BLEND is supported for indexed capability state."));
|
||||
"Only GL_BLEND and GL_SCISSOR_TEST are supported for indexed "
|
||||
"capability state."));
|
||||
return false;
|
||||
}
|
||||
|
||||
if (index >= MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS) {
|
||||
if (index >= limit) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", functionName,
|
||||
"Buffer index " + std::to_string(index) + " is out of range. Max supported is " +
|
||||
std::to_string(MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS - 1) + "."));
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
String(indexName) + " index " + std::to_string(index) +
|
||||
" is out of range. Max supported is " + std::to_string(limit - 1) +
|
||||
"."));
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// ------------------ ARB_viewport_array parameter validation ------------------
|
||||
// All three families share the same two shapes, so they share the two checkers. GL 4.6 core
|
||||
// 13.6.1/17.3.2: an out-of-range index is GL_INVALID_VALUE, and so is a negative width or
|
||||
// height. `first + count == MAX_VIEWPORTS` is LEGAL - only strictly greater is an error,
|
||||
// which KHR-GL43.viewport_array.api_errors checks explicitly in both directions.
|
||||
static Bool ValidateViewportIndex(GLuint index, const char* functionName) {
|
||||
if (index < RenderStateParameters::MAX_VIEWPORTS) return true;
|
||||
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"Viewport index " + std::to_string(index) +
|
||||
" is out of range. Max supported is " +
|
||||
std::to_string(RenderStateParameters::MAX_VIEWPORTS - 1) + "."));
|
||||
return false;
|
||||
}
|
||||
|
||||
static Bool ValidateViewportRange(GLuint first, GLsizei count, const char* functionName) {
|
||||
if (count < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "count must not be negative."));
|
||||
return false;
|
||||
}
|
||||
// Widened before adding: first is a GLuint and count a GLsizei, so `first + count` in
|
||||
// 32 bits can wrap past MAX_VIEWPORTS and let an out-of-range range through.
|
||||
const Uint64 last = static_cast<Uint64>(first) + static_cast<Uint64>(count);
|
||||
if (last > RenderStateParameters::MAX_VIEWPORTS) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"first (" + std::to_string(first) + ") + count (" +
|
||||
std::to_string(count) + ") exceeds GL_MAX_VIEWPORTS (" +
|
||||
std::to_string(RenderStateParameters::MAX_VIEWPORTS) + ")."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static Bool ValidateNonNegativeExtent(T width, T height, const char* functionName) {
|
||||
if (width >= T(0) && height >= T(0)) return true;
|
||||
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "Width and height must be non-negative."));
|
||||
return false;
|
||||
}
|
||||
|
||||
// The array forms are all-or-nothing: one bad element rejects the whole call with a SINGLE
|
||||
// GL_INVALID_VALUE and leaves every rectangle untouched. api_errors relies on both halves -
|
||||
// it passes a full 16-element array with exactly one negative extent and then asserts the
|
||||
// error queue holds exactly one entry.
|
||||
template <typename T>
|
||||
static Bool ValidateArrayExtents(GLsizei count, const T* v, const char* functionName) {
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
if (v[i * 4 + 2] >= T(0) && v[i * 4 + 3] >= T(0)) continue;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName,
|
||||
"Width and height must be non-negative (element " + std::to_string(i) +
|
||||
")."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static Bool ValidateNonNullArray(const void* v, const char* functionName) {
|
||||
if (v != nullptr) return true;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", functionName, "value pointer cannot be null."));
|
||||
return false;
|
||||
}
|
||||
|
||||
static Bool TryConvertBlendEquation(GLenum mode, const char* functionName,
|
||||
::MobileGL::BlendEquation& outEquation) {
|
||||
outEquation = MG_Util::ConvertGLEnumToBlendEquation(mode);
|
||||
@@ -92,16 +183,70 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void Viewport_State(GLint x, GLint y, GLsizei width, GLsizei height) {
|
||||
if (width < 0 || height < 0) {
|
||||
MG_State::pGLContext->RecordError(ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "Viewport_State",
|
||||
"Width abd height must be non-negative."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateNonNegativeExtent(width, height, "Viewport_State")) return;
|
||||
|
||||
MG_State::pGLContext->SetViewport(IntVec4(x, y, width, height));
|
||||
}
|
||||
|
||||
// ------------------ ARB_viewport_array setters ------------------
|
||||
void ViewportArrayv_State(GLuint first, GLsizei count, const GLfloat* v) {
|
||||
if (!ValidateViewportRange(first, count, "ViewportArrayv_State")) return;
|
||||
if (count == 0) return;
|
||||
if (!ValidateNonNullArray(v, "ViewportArrayv_State")) return;
|
||||
if (!ValidateArrayExtents(count, v, "ViewportArrayv_State")) return;
|
||||
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
MG_State::pGLContext->SetViewportIndexed(first + static_cast<GLuint>(i),
|
||||
FloatVec4(v[i * 4 + 0], v[i * 4 + 1], v[i * 4 + 2], v[i * 4 + 3]));
|
||||
}
|
||||
}
|
||||
|
||||
void ViewportIndexedf_State(GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) {
|
||||
if (!ValidateViewportIndex(index, "ViewportIndexedf_State")) return;
|
||||
if (!ValidateNonNegativeExtent(w, h, "ViewportIndexedf_State")) return;
|
||||
|
||||
MG_State::pGLContext->SetViewportIndexed(index, FloatVec4(x, y, w, h));
|
||||
}
|
||||
|
||||
void ScissorArrayv_State(GLuint first, GLsizei count, const GLint* v) {
|
||||
if (!ValidateViewportRange(first, count, "ScissorArrayv_State")) return;
|
||||
if (count == 0) return;
|
||||
if (!ValidateNonNullArray(v, "ScissorArrayv_State")) return;
|
||||
if (!ValidateArrayExtents(count, v, "ScissorArrayv_State")) return;
|
||||
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
MG_State::pGLContext->SetScissorBoxIndexed(first + static_cast<GLuint>(i),
|
||||
IntVec4(v[i * 4 + 0], v[i * 4 + 1], v[i * 4 + 2], v[i * 4 + 3]));
|
||||
}
|
||||
}
|
||||
|
||||
void ScissorIndexed_State(GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) {
|
||||
if (!ValidateViewportIndex(index, "ScissorIndexed_State")) return;
|
||||
if (!ValidateNonNegativeExtent(width, height, "ScissorIndexed_State")) return;
|
||||
|
||||
MG_State::pGLContext->SetScissorBoxIndexed(index, IntVec4(left, bottom, width, height));
|
||||
}
|
||||
|
||||
void DepthRangeArrayv_State(GLuint first, GLsizei count, const GLdouble* v) {
|
||||
if (!ValidateViewportRange(first, count, "DepthRangeArrayv_State")) return;
|
||||
if (count == 0) return;
|
||||
if (!ValidateNonNullArray(v, "DepthRangeArrayv_State")) return;
|
||||
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
MG_State::pGLContext->SetDepthRangeIndexed(
|
||||
first + static_cast<GLuint>(i),
|
||||
FloatVec2(ClampUnitFloat(static_cast<GLfloat>(v[i * 2 + 0])),
|
||||
ClampUnitFloat(static_cast<GLfloat>(v[i * 2 + 1]))));
|
||||
}
|
||||
}
|
||||
|
||||
void DepthRangeIndexed_State(GLuint index, GLdouble n, GLdouble f) {
|
||||
if (!ValidateViewportIndex(index, "DepthRangeIndexed_State")) return;
|
||||
|
||||
MG_State::pGLContext->SetDepthRangeIndexed(
|
||||
index, FloatVec2(ClampUnitFloat(static_cast<GLfloat>(n)), ClampUnitFloat(static_cast<GLfloat>(f))));
|
||||
}
|
||||
|
||||
void StencilOpSeparate_State(GLenum face, GLenum sfail, GLenum dpfail, GLenum dppass) {
|
||||
Bool applyFront = false;
|
||||
Bool applyBack = false;
|
||||
@@ -174,12 +319,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void Scissor_State(GLint x, GLint y, GLsizei width, GLsizei height) {
|
||||
if (width < 0 || height < 0) {
|
||||
MG_State::pGLContext->RecordError(ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "Scissor_State",
|
||||
"Width abd height must be non-negative."));
|
||||
return;
|
||||
}
|
||||
if (!ValidateNonNegativeExtent(width, height, "Scissor_State")) return;
|
||||
|
||||
MG_State::pGLContext->SetScissorBox(IntVec4(x, y, width, height));
|
||||
}
|
||||
@@ -335,7 +475,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
GLboolean IsEnabledi_State(GLenum target, GLuint index) {
|
||||
if (!ValidateIndexedBlendCapability(target, index, "IsEnabledi_State")) {
|
||||
if (!ValidateIndexedCapability(target, index, "IsEnabledi_State")) {
|
||||
return GL_FALSE;
|
||||
}
|
||||
|
||||
@@ -380,7 +520,25 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
*data = IsEnabledi_State(target, index);
|
||||
// GL 4.6 core 22.1: glGetBooleani_v answers EVERY indexed state, not just the indexed
|
||||
// capabilities - a non-boolean value simply reads back as "is it non-zero". Routing the
|
||||
// non-capability enums to the pname table glGetIntegeri_v already owns is what makes
|
||||
// that true; without it a query like glGetBooleani_v(GL_MAX_COMPUTE_WORK_GROUP_COUNT, 0)
|
||||
// came back GL_INVALID_ENUM (KHR-GL43.compute_shader.max).
|
||||
if (MG_Util::ConvertGLEnumToCapabilityInput(target) != CapabilityInput::Unknown) {
|
||||
*data = IsEnabledi_State(target, index);
|
||||
return;
|
||||
}
|
||||
GLint values[4] = {};
|
||||
GetIntegeri_v(target, index, values);
|
||||
// The ARB_viewport_array rectangles are the only multi-component indexed state that
|
||||
// reaches here; writing element 0 alone would leave the caller's other three untouched.
|
||||
const GLsizei components = target == GL_VIEWPORT || target == GL_SCISSOR_BOX
|
||||
? 4
|
||||
: (target == GL_DEPTH_RANGE ? 2 : 1);
|
||||
for (GLsizei i = 0; i < components; ++i) {
|
||||
data[i] = values[i] != 0 ? GL_TRUE : GL_FALSE;
|
||||
}
|
||||
}
|
||||
|
||||
GLboolean IsEnabled_State(GLenum cap) {
|
||||
@@ -713,7 +871,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void Disablei_State(GLenum target, GLuint index) {
|
||||
if (!ValidateIndexedBlendCapability(target, index, "Disablei_State")) {
|
||||
if (!ValidateIndexedCapability(target, index, "Disablei_State")) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -731,7 +889,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void Enablei_State(GLenum target, GLuint index) {
|
||||
if (!ValidateIndexedBlendCapability(target, index, "Enablei_State")) {
|
||||
if (!ValidateIndexedCapability(target, index, "Enablei_State")) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -785,6 +943,44 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
Viewport_State(x, y, width, height);
|
||||
}
|
||||
|
||||
void ViewportArrayv(GLuint first, GLsizei count, const GLfloat* v) {
|
||||
ViewportArrayv_State(first, count, v);
|
||||
}
|
||||
|
||||
void ViewportIndexedf(GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h) {
|
||||
ViewportIndexedf_State(index, x, y, w, h);
|
||||
}
|
||||
|
||||
void ViewportIndexedfv(GLuint index, const GLfloat* v) {
|
||||
// The index is validated before the pointer is touched: glViewportIndexedfv(MAX, nullptr)
|
||||
// must be one GL_INVALID_VALUE, not a null dereference.
|
||||
if (!ValidateViewportIndex(index, "ViewportIndexedfv")) return;
|
||||
if (!ValidateNonNullArray(v, "ViewportIndexedfv")) return;
|
||||
ViewportIndexedf_State(index, v[0], v[1], v[2], v[3]);
|
||||
}
|
||||
|
||||
void ScissorArrayv(GLuint first, GLsizei count, const GLint* v) {
|
||||
ScissorArrayv_State(first, count, v);
|
||||
}
|
||||
|
||||
void ScissorIndexed(GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height) {
|
||||
ScissorIndexed_State(index, left, bottom, width, height);
|
||||
}
|
||||
|
||||
void ScissorIndexedv(GLuint index, const GLint* v) {
|
||||
if (!ValidateViewportIndex(index, "ScissorIndexedv")) return;
|
||||
if (!ValidateNonNullArray(v, "ScissorIndexedv")) return;
|
||||
ScissorIndexed_State(index, v[0], v[1], v[2], v[3]);
|
||||
}
|
||||
|
||||
void DepthRangeArrayv(GLuint first, GLsizei count, const GLdouble* v) {
|
||||
DepthRangeArrayv_State(first, count, v);
|
||||
}
|
||||
|
||||
void DepthRangeIndexed(GLuint index, GLdouble n, GLdouble f) {
|
||||
DepthRangeIndexed_State(index, n, f);
|
||||
}
|
||||
|
||||
void StencilOpSeparate(GLenum face, GLenum sfail, GLenum dpfail, GLenum dppass) {
|
||||
StencilOpSeparate_State(face, sfail, dpfail, dppass);
|
||||
}
|
||||
|
||||
@@ -20,6 +20,16 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void Enablei(GLenum target, GLuint index);
|
||||
void BlendFunc(GLenum sfactor, GLenum dfactor);
|
||||
void Viewport(GLint x, GLint y, GLsizei width, GLsizei height);
|
||||
// ARB_viewport_array (core since GL 4.1). Every one of these addresses the same 16-element
|
||||
// indexed state the classic glViewport/glScissor/glDepthRange trio broadcasts to.
|
||||
void ViewportArrayv(GLuint first, GLsizei count, const GLfloat* v);
|
||||
void ViewportIndexedf(GLuint index, GLfloat x, GLfloat y, GLfloat w, GLfloat h);
|
||||
void ViewportIndexedfv(GLuint index, const GLfloat* v);
|
||||
void ScissorArrayv(GLuint first, GLsizei count, const GLint* v);
|
||||
void ScissorIndexed(GLuint index, GLint left, GLint bottom, GLsizei width, GLsizei height);
|
||||
void ScissorIndexedv(GLuint index, const GLint* v);
|
||||
void DepthRangeArrayv(GLuint first, GLsizei count, const GLdouble* v);
|
||||
void DepthRangeIndexed(GLuint index, GLdouble n, GLdouble f);
|
||||
void StencilOpSeparate(GLenum face, GLenum sfail, GLenum dpfail, GLenum dppass);
|
||||
void StencilOp(GLenum fail, GLenum zfail, GLenum zpass);
|
||||
void StencilMaskSeparate(GLenum face, GLuint mask);
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "GL_Sampler.h"
|
||||
#include "Validators.h"
|
||||
#include "../Getter/GL_Getter.h"
|
||||
#include "../Texture/GL_Texture.h"
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/Converters/GLToMG/TextureEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
|
||||
@@ -269,15 +270,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
}
|
||||
|
||||
// The number of texture units a sampler may be bound to. GL 3.3 core 3.8.2 names
|
||||
// GL_MAX_COMBINED_TEXTURE_IMAGE_UNITS, which is what the backend advertises; the frontend's
|
||||
// MAX_TEXTURE_IMAGE_UNITS is only the capacity of the unit array, so it is a clamp on the
|
||||
// answer and never the answer itself - gating on it alone accepts every unit up to 192 no
|
||||
// matter what the driver reports.
|
||||
// The number of texture units a sampler may be bound to is the same count a TEXTURE may be
|
||||
// bound to - GL 3.3 core 3.8.2 names GL_MAX_COMBINED_TEXTURE_IMAGE_UNITS for both - so it is
|
||||
// computed once, in GetCombinedTextureImageUnitCount, and named here for the sampler-side
|
||||
// readers below. Two copies of that arithmetic is how glBindSamplers and glBindTextures would
|
||||
// come to disagree about which units exist.
|
||||
static GLint GetSamplerBindableTextureUnitCount() {
|
||||
GLint maxTextureUnits = 0;
|
||||
GetIntegerv(GL_MAX_COMBINED_TEXTURE_IMAGE_UNITS, &maxTextureUnits);
|
||||
return std::min<GLint>(std::max(maxTextureUnits, 0), MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS);
|
||||
return GetCombinedTextureImageUnitCount();
|
||||
}
|
||||
|
||||
void BindSampler_State(GLuint unit, GLuint sampler) {
|
||||
@@ -336,8 +335,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return;
|
||||
}
|
||||
|
||||
// ARB_multi_bind adds one rule the single-bind path does not have: "samplers will not be
|
||||
// created if they do not exist", so a name that is not an existing sampler OBJECT is
|
||||
// INVALID_OPERATION here (KHR-GL44.multi_bind.errors_bind_samplers). Per element, not
|
||||
// all-or-nothing - the extension defines glBindSamplers as a loop, so a bad entry costs
|
||||
// its own texture unit and leaves the rest of the range bound.
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
BindSampler_State(first + i, samplers ? samplers[i] : 0);
|
||||
const GLuint sampler = samplers ? samplers[i] : 0;
|
||||
if (sampler != 0 && !MG_State::pGLContext->ValidateSamplerObject(sampler)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", "BindSamplers",
|
||||
std::format("samplers[{}] ({}) is not the name of an existing sampler object.", i, sampler)));
|
||||
continue;
|
||||
}
|
||||
BindSampler_State(first + i, sampler);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "GL_Sync.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl {
|
||||
namespace {
|
||||
@@ -35,6 +36,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
} // namespace
|
||||
|
||||
GLsync FenceSync(GLenum condition, GLbitfield flags) {
|
||||
// GL 4.6 core 4.1.2: GL_SYNC_GPU_COMMANDS_COMPLETE is the only condition and the only
|
||||
// legal flags value is zero; both violations return 0 rather than a handle. A caller that
|
||||
// then hands the 0 back to glDeleteSync hits the glDeleteSync(0) no-op below.
|
||||
if (condition != GL_SYNC_GPU_COMMANDS_COMPLETE) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"condition must be GL_SYNC_GPU_COMMANDS_COMPLETE."));
|
||||
return nullptr;
|
||||
}
|
||||
if (flags != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "flags must be zero."));
|
||||
return nullptr;
|
||||
}
|
||||
auto* syncObject = new SyncObject;
|
||||
syncObject->condition = condition;
|
||||
syncObject->flags = flags;
|
||||
@@ -52,8 +69,25 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
GLenum ClientWaitSync(GLsync sync, GLbitfield flags, GLuint64 timeout) {
|
||||
// GL 4.6 core 4.1.1: GL_SYNC_FLUSH_COMMANDS_BIT is the only bit this call accepts, and
|
||||
// any other bit is INVALID_VALUE. Silently ignoring the stray bits used to make a caller
|
||||
// that passed, say, GL_SYNC_GPU_COMMANDS_COMPLETE by mistake think it had asked for a
|
||||
// flush it never got.
|
||||
if ((flags & ~static_cast<GLbitfield>(GL_SYNC_FLUSH_COMMANDS_BIT)) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"flags must be zero or GL_SYNC_FLUSH_COMMANDS_BIT."));
|
||||
return GL_WAIT_FAILED;
|
||||
}
|
||||
const auto* syncObject = FindSyncObject(sync);
|
||||
if (!syncObject) {
|
||||
// The spec pairs the GL_WAIT_FAILED return with a recorded INVALID_VALUE; returning
|
||||
// the enum alone left glGetError() clean and the failure indistinguishable from a
|
||||
// genuine wait failure on a live sync.
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "sync is not the name of a sync object."));
|
||||
return GL_WAIT_FAILED;
|
||||
}
|
||||
const auto backendClientWaitSync = MG_Backend::gBackendFunctionsTable.GL.ClientWaitSync;
|
||||
@@ -64,8 +98,23 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void WaitSync(GLsync sync, GLbitfield flags, GLuint64 timeout) {
|
||||
// GL 4.6 core 4.1.2: the server-side wait takes no flags and no finite timeout - both
|
||||
// arguments exist only to be forward-compatible, and anything else is INVALID_VALUE.
|
||||
// Neither backend ever honored a nonzero timeout (DirectGLES hard-codes
|
||||
// 0/GL_TIMEOUT_IGNORED, DirectVulkan's queue ordering makes the wait implicit), so
|
||||
// rejecting the call loses no wait that used to happen.
|
||||
if (flags != 0 || timeout != GL_TIMEOUT_IGNORED) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"flags must be zero and timeout must be GL_TIMEOUT_IGNORED."));
|
||||
return;
|
||||
}
|
||||
const auto* syncObject = FindSyncObject(sync);
|
||||
if (!syncObject) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "sync is not the name of a sync object."));
|
||||
return;
|
||||
}
|
||||
const auto backendWaitSync = MG_Backend::gBackendFunctionsTable.GL.WaitSync;
|
||||
@@ -96,8 +145,22 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void GetSynciv(GLsync sync, GLenum pname, GLsizei bufSize, GLsizei* length, GLint* values) {
|
||||
// GL 4.6 core 4.1: a negative bufSize is INVALID_VALUE, an unnamed sync is INVALID_VALUE
|
||||
// and an unrecognised pname is INVALID_ENUM. All three used to leave glGetError() clean
|
||||
// and write a plausible-looking zero, which is the one failure mode a caller cannot tell
|
||||
// apart from a real answer - GL_SYNC_STATUS legitimately answers GL_UNSIGNALED (0x9118),
|
||||
// but a mistyped pname answered a bare 0 that no query ever returns.
|
||||
if (bufSize < 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "bufSize must not be negative."));
|
||||
return;
|
||||
}
|
||||
const auto* syncObject = FindSyncObject(sync);
|
||||
if (!syncObject) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__, "sync is not the name of a sync object."));
|
||||
if (length) {
|
||||
*length = 0;
|
||||
}
|
||||
@@ -123,7 +186,15 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
value = static_cast<GLint>(syncObject->flags);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidEnum,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", __func__,
|
||||
"pname must be GL_OBJECT_TYPE, GL_SYNC_STATUS, GL_SYNC_CONDITION or "
|
||||
"GL_SYNC_FLAGS."));
|
||||
if (length) {
|
||||
*length = 0;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (length) {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -37,6 +37,13 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
GLenum format, GLenum type, const void* pixels);
|
||||
void TextureSubImage3D(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset, GLsizei width,
|
||||
GLsizei height, GLsizei depth, GLenum format, GLenum type, const void* pixels);
|
||||
void CompressedTextureSubImage1D(GLuint texture, GLint level, GLint xoffset, GLsizei width, GLenum format,
|
||||
GLsizei imageSize, const void* data);
|
||||
void CompressedTextureSubImage2D(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLsizei width,
|
||||
GLsizei height, GLenum format, GLsizei imageSize, const void* data);
|
||||
void CompressedTextureSubImage3D(GLuint texture, GLint level, GLint xoffset, GLint yoffset, GLint zoffset,
|
||||
GLsizei width, GLsizei height, GLsizei depth, GLenum format, GLsizei imageSize,
|
||||
const void* data);
|
||||
void TextureParameterf(GLuint texture, GLenum pname, GLfloat param);
|
||||
void TextureParameterfv(GLuint texture, GLenum pname, const GLfloat* params);
|
||||
void TextureParameteri(GLuint texture, GLenum pname, GLint param);
|
||||
@@ -58,6 +65,8 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void GetTextureParameteriv(GLuint texture, GLenum pname, GLint* params);
|
||||
void GetTextureLevelParameterfv(GLuint texture, GLint level, GLenum pname, GLfloat* params);
|
||||
void GetTextureLevelParameteriv(GLuint texture, GLint level, GLenum pname, GLint* params);
|
||||
void TextureView(GLuint texture, GLenum target, GLuint origtexture, GLenum internalformat, GLuint minlevel,
|
||||
GLuint numlevels, GLuint minlayer, GLuint numlayers);
|
||||
void TexStorage1D(GLenum target, GLsizei levels, GLenum internalformat, GLsizei width);
|
||||
void TexStorage2D(GLenum target, GLsizei levels, GLenum internalformat, GLsizei width, GLsizei height);
|
||||
void TexStorage3D(GLenum target, GLsizei levels, GLenum internalformat, GLsizei width, GLsizei height,
|
||||
@@ -132,5 +141,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
void CompressedTexImage1D(GLenum target, GLint level, GLenum internalformat, GLsizei width, GLint border,
|
||||
GLsizei imageSize, const void* data);
|
||||
void BindTexture(GLenum target, GLuint texture);
|
||||
void BindTextures(GLuint first, GLsizei count, const GLuint* textures);
|
||||
void BindImageTextures(GLuint first, GLsizei count, const GLuint* textures);
|
||||
void ActiveTexture(GLenum texture);
|
||||
// The number of texture image units a texture or a sampler may be bound to: what the backend
|
||||
// advertises as GL_MAX_COMBINED_TEXTURE_IMAGE_UNITS, clamped by the frontend's fixed unit-array
|
||||
// capacity. Shared so the texture and sampler multi-bind range checks cannot drift apart.
|
||||
GLint GetCombinedTextureImageUnitCount();
|
||||
} // namespace MobileGL::MG_Impl::GLImpl
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
#include <MG_Util/Converters/MGToGL/TextureEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToMG/TextureEnumConverter.h>
|
||||
#include <MG_Util/Converters/MGToStr/TextureEnumConverter.h>
|
||||
#include <MG_Util/Metrics/TextureMetrics.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
Bool ValidateTextureTarget(TextureTarget target) {
|
||||
@@ -312,9 +313,13 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
return false;
|
||||
}
|
||||
|
||||
// TexImage in core 3.3 has no stencil-only upload path (that arrived with GL 4.4).
|
||||
if (format == TextureInputFormat::StencilIndex) {
|
||||
return recordInvalidOperation("STENCIL_INDEX is not a valid texture upload format");
|
||||
// The stencil-only transfer path arrived with GL 4.4 / ARB_texture_stencil8, and only ever
|
||||
// pairs with stencil-only storage: against a depth, depth-stencil or colour internal format
|
||||
// STENCIL_INDEX keeps the pre-4.4 answer (GL CTS packed_pixels feeds exactly that pairing
|
||||
// and expects INVALID_OPERATION).
|
||||
if (format == TextureInputFormat::StencilIndex &&
|
||||
internalFormat != TextureInternalFormat::StencilIndex8) {
|
||||
return recordInvalidOperation("STENCIL_INDEX requires a stencil-only internal format");
|
||||
}
|
||||
|
||||
if (IsDepthLikeInputFormat(format) != IsDepthLikeInternalFormat(internalFormat)) {
|
||||
@@ -353,6 +358,63 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool ValidateTextureLevelExists(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject, Int level,
|
||||
const char* caller) {
|
||||
// A null object is somebody else's error to report - ValidateTextureObject runs
|
||||
// first at every call site and has already recorded it.
|
||||
if (!textureObject) return false;
|
||||
|
||||
const auto* mipmapTexture = MG_State::GLState::AsMipmapTexture(textureObject.get());
|
||||
if (mipmapTexture == nullptr) {
|
||||
// The only non-mipmap storage class is a buffer texture, and GL_TEXTURE_BUFFER is
|
||||
// not a target glCopyImageSubData accepts at all (it is in the CTS's invalid-target
|
||||
// set). Declining here is not the error code the spec asks for - that would be
|
||||
// INVALID_ENUM from a target check this validator is not - but it does keep a
|
||||
// texture with no image levels whatsoever from reaching a backend that would
|
||||
// dereference a backend texture it never created.
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller,
|
||||
"Texture has no mipmap levels to address."));
|
||||
return false;
|
||||
}
|
||||
|
||||
// What this number is, exactly, because two other things are almost it and neither is
|
||||
// safe to assume: it is the number of level SLOTS the shadow has allocated - holes
|
||||
// included, since MipmapStorage::AllocateLevel grows to level+1 and never fills the gap.
|
||||
// For a cube map MipmapUploadTargetArray reports face +X's chain rather than the union.
|
||||
//
|
||||
// The guarantee that matters is one-sided: this count is always >= the level count the
|
||||
// backends derive (VkTextureManager::GetUploadMipLevelCount stops at the first level
|
||||
// with a non-positive extent, so it can only be shorter). That is the safe direction -
|
||||
// no copy to a level the texture genuinely has is ever rejected here. It is NOT an
|
||||
// exact match, so the backends keep their own range guard for the band in between: a
|
||||
// chain with a hole (level 0 and 2 defined, 1 not) is accepted by this predicate and
|
||||
// declined by the backend, which is a silent no-op rather than a copy. That band is a
|
||||
// backend storage limitation, not a validation one - rejecting it here with
|
||||
// INVALID_VALUE would be refusing a copy the spec permits.
|
||||
const Uint levelCount = mipmapTexture->GetMipmapLevelCount();
|
||||
|
||||
if (levelCount == 0) {
|
||||
// No image has ever been defined on this texture, so the fault is the texture,
|
||||
// not the number: GL 4.6 core 18.3.2 asks for INVALID_OPERATION when an object a
|
||||
// copy names is an incomplete texture.
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller,
|
||||
"Texture has no image defined at any level."));
|
||||
return false;
|
||||
}
|
||||
if (level < 0 || static_cast<Uint>(level) >= levelCount) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", caller,
|
||||
"Texture level does not exist in this texture."));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool ValidateTextureObject(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject) {
|
||||
if (!textureObject) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
@@ -424,19 +486,281 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool ValidateBaseInternalFormatMatch(TextureInternalFormat format1, TextureInternalFormat format2) {
|
||||
auto unsizedFormat1 = MG_Util::ConvertInternalFormatToUnsized(format1);
|
||||
auto unsizedFormat2 = MG_Util::ConvertInternalFormatToUnsized(format2);
|
||||
if (unsizedFormat1 != unsizedFormat2) {
|
||||
namespace {
|
||||
// Component set of an UNSIZED base internal format, as the bitmask GL 4.6 SS 8.6
|
||||
// reasons about. Colour components are independent bits so "subset" is a plain
|
||||
// mask test; depth and stencil are their own components and never satisfy a
|
||||
// colour request (or each other).
|
||||
enum : Uint32 {
|
||||
kComponentR = 1u << 0,
|
||||
kComponentG = 1u << 1,
|
||||
kComponentB = 1u << 2,
|
||||
kComponentA = 1u << 3,
|
||||
kComponentDepth = 1u << 4,
|
||||
kComponentStencil = 1u << 5,
|
||||
};
|
||||
|
||||
Uint32 BaseFormatComponents(TextureInternalFormat unsizedFormat) {
|
||||
switch (unsizedFormat) {
|
||||
case TextureInternalFormat::Red:
|
||||
return kComponentR;
|
||||
case TextureInternalFormat::RG:
|
||||
return kComponentR | kComponentG;
|
||||
case TextureInternalFormat::RGB:
|
||||
return kComponentR | kComponentG | kComponentB;
|
||||
case TextureInternalFormat::RGBA:
|
||||
return kComponentR | kComponentG | kComponentB | kComponentA;
|
||||
case TextureInternalFormat::DepthComponent:
|
||||
return kComponentDepth;
|
||||
case TextureInternalFormat::DepthStencil:
|
||||
return kComponentDepth | kComponentStencil;
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
CopyImageTexelBlock ResolveCopyImageTexelBlock(TextureInternalFormat format, GLenum compressedFormat) {
|
||||
CopyImageTexelBlock block{};
|
||||
if (compressedFormat != GL_NONE) {
|
||||
const auto info = MG_Util::GetCompressedFormatInfo(compressedFormat);
|
||||
if (info.blockByteSize != 0) {
|
||||
block.byteSize = info.blockByteSize;
|
||||
block.blockWidth = info.blockWidth;
|
||||
block.blockHeight = info.blockHeight;
|
||||
block.compressed = true;
|
||||
return block;
|
||||
}
|
||||
}
|
||||
// The size MobileGL actually stores a texel of this format in, which for every format GL
|
||||
// gives a required size is that required size. The handful of legacy formats GL leaves
|
||||
// implementation-defined (R3_G3_B2, RGB4/5/10/12, RGBA2/12) have no view class in table
|
||||
// 8.22 to be compared against anyway, and this is the size that decides whether a raw
|
||||
// copy between them would in fact preserve the bytes.
|
||||
block.byteSize = MG_Util::GetSizedInternalFormatSizeInBytes(format);
|
||||
return block;
|
||||
}
|
||||
|
||||
Bool ValidateCopyImageFormatCompatibility(const CopyImageTexelBlock& srcBlock,
|
||||
const CopyImageTexelBlock& dstBlock) {
|
||||
if (srcBlock.byteSize == 0 || dstBlock.byteSize == 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "ValidateCopyImageFormatCompatibility",
|
||||
"A copied image has no storage whose texel size is known."));
|
||||
return false;
|
||||
}
|
||||
if (srcBlock.byteSize != dstBlock.byteSize) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
std::format("MG_Impl/GLImpl", "ValidateBaseInternalFormatMatch",
|
||||
"The base internal format of the two formats do not match ({} vs. {})",
|
||||
MG_Util::ConvertTextureInternalFormatToString(unsizedFormat1).c_str(),
|
||||
MG_Util::ConvertTextureInternalFormatToString(unsizedFormat2).c_str())));
|
||||
"MG_Impl/GLImpl", "ValidateCopyImageFormatCompatibility",
|
||||
std::format("The two images' texel blocks are different sizes ({} vs. {} bytes), so the "
|
||||
"formats are not copy-compatible.",
|
||||
srcBlock.byteSize, dstBlock.byteSize)));
|
||||
return false;
|
||||
}
|
||||
// Two compressed images additionally have to agree on the SHAPE of the block, not only
|
||||
// its size: an 8-byte 4x4 block and a hypothetical 8-byte 8x8 one hold different texel
|
||||
// counts, and GL 4.6 core 18.3.2 requires both dimensions to match.
|
||||
if (srcBlock.compressed && dstBlock.compressed &&
|
||||
(srcBlock.blockWidth != dstBlock.blockWidth || srcBlock.blockHeight != dstBlock.blockHeight)) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", "ValidateCopyImageFormatCompatibility",
|
||||
std::format("The two compressed images have different block dimensions ({}x{} vs. {}x{}).",
|
||||
srcBlock.blockWidth, srcBlock.blockHeight, dstBlock.blockWidth,
|
||||
dstBlock.blockHeight)));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
} // namespace TextureImpl
|
||||
}
|
||||
|
||||
Bool ValidateCopyImageBlockAlignment(const CopyImageTexelBlock& block, Int x, Int y, Int width, Int height,
|
||||
Int imageWidth, Int imageHeight, const char* endpointName) {
|
||||
if (!block.compressed) return true;
|
||||
const Int blockWidth = static_cast<Int>(block.blockWidth);
|
||||
const Int blockHeight = static_cast<Int>(block.blockHeight);
|
||||
if (blockWidth <= 1 && blockHeight <= 1) return true;
|
||||
// The origin is unconditional; the extent gets the "or it reaches the edge of the image"
|
||||
// exemption GL 4.6 core 18.3.2 grants, which is what lets a 16x16 BPTC image be copied
|
||||
// whole even when the last block is partial.
|
||||
const Bool originAligned = (x % blockWidth == 0) && (y % blockHeight == 0);
|
||||
const Bool widthOk = (width % blockWidth == 0) || (x + width == imageWidth);
|
||||
const Bool heightOk = (height % blockHeight == 0) || (y + height == imageHeight);
|
||||
if (originAligned && widthOk && heightOk) return true;
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidValue,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", "ValidateCopyImageBlockAlignment",
|
||||
std::format("The {} region [{}, {}] + [{} x {}] is not aligned to the {}x{} compressed block "
|
||||
"grid of a {} x {} image.",
|
||||
endpointName, x, y, width, height, blockWidth, blockHeight, imageWidth, imageHeight)));
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool ValidateCopyTexImageBaseFormatSubset(TextureInternalFormat destFormat, TextureInternalFormat srcFormat) {
|
||||
const auto unsizedDest = MG_Util::ConvertInternalFormatToUnsized(destFormat);
|
||||
const auto unsizedSrc = MG_Util::ConvertInternalFormatToUnsized(srcFormat);
|
||||
// GL 4.6 SS 8.6: glCopyTexImage* may request a SUBSET of the read buffer's components,
|
||||
// not an exact match - GL_RGB from an RGBA8 framebuffer is textbook legal and is what
|
||||
// Minecraft and its mods do. glCopyTexImage2D used to run the exact-match predicate
|
||||
// above and turn its rejection into an uncaught exception through the C GL ABI, so the
|
||||
// app died rather than seeing a GL error.
|
||||
const Uint32 destComponents = BaseFormatComponents(unsizedDest);
|
||||
const Uint32 srcComponents = BaseFormatComponents(unsizedSrc);
|
||||
if (destComponents == 0 || srcComponents == 0 || (destComponents & ~srcComponents) != 0) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", "ValidateCopyTexImageBaseFormatSubset",
|
||||
std::format("the read buffer's base internal format {} does not provide every component of "
|
||||
"the requested internal format {}",
|
||||
MG_Util::ConvertTextureInternalFormatToString(unsizedSrc),
|
||||
MG_Util::ConvertTextureInternalFormatToString(unsizedDest))));
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// GL 4.6 core table 8.21 ("Compatible internal formats for TextureView"), transcribed whole.
|
||||
// Written against the raw GLenum rather than TextureInternalFormat on purpose: MobileGL's own
|
||||
// enum collapses every compressed format onto uncompressed storage and drops formats it
|
||||
// cannot carry, so classifying the converted value would silently widen the compatibility
|
||||
// rule - GL_COMPRESSED_RG_RGTC2 and GL_RGBA8 would end up in the same class.
|
||||
TextureViewClass GetTextureViewClass(GLenum internalformat) {
|
||||
switch (internalformat) {
|
||||
case GL_RGBA32F:
|
||||
case GL_RGBA32UI:
|
||||
case GL_RGBA32I:
|
||||
return TextureViewClass::Bits128;
|
||||
case GL_RGB32F:
|
||||
case GL_RGB32UI:
|
||||
case GL_RGB32I:
|
||||
return TextureViewClass::Bits96;
|
||||
case GL_RGBA16F:
|
||||
case GL_RG32F:
|
||||
case GL_RGBA16UI:
|
||||
case GL_RG32UI:
|
||||
case GL_RGBA16I:
|
||||
case GL_RG32I:
|
||||
case GL_RGBA16:
|
||||
case GL_RGBA16_SNORM:
|
||||
return TextureViewClass::Bits64;
|
||||
case GL_RGB16:
|
||||
case GL_RGB16_SNORM:
|
||||
case GL_RGB16F:
|
||||
case GL_RGB16UI:
|
||||
case GL_RGB16I:
|
||||
return TextureViewClass::Bits48;
|
||||
case GL_RG16F:
|
||||
case GL_R11F_G11F_B10F:
|
||||
case GL_R32F:
|
||||
case GL_RGB10_A2UI:
|
||||
case GL_RGBA8UI:
|
||||
case GL_RG16UI:
|
||||
case GL_R32UI:
|
||||
case GL_RGBA8I:
|
||||
case GL_RG16I:
|
||||
case GL_R32I:
|
||||
case GL_RGB10_A2:
|
||||
case GL_RGBA8:
|
||||
case GL_RG16:
|
||||
case GL_RGBA8_SNORM:
|
||||
case GL_RG16_SNORM:
|
||||
case GL_SRGB8_ALPHA8:
|
||||
case GL_RGB9_E5:
|
||||
return TextureViewClass::Bits32;
|
||||
case GL_RGB8:
|
||||
case GL_RGB8_SNORM:
|
||||
case GL_SRGB8:
|
||||
case GL_RGB8UI:
|
||||
case GL_RGB8I:
|
||||
return TextureViewClass::Bits24;
|
||||
case GL_R16F:
|
||||
case GL_RG8UI:
|
||||
case GL_R16UI:
|
||||
case GL_RG8I:
|
||||
case GL_R16I:
|
||||
case GL_RG8:
|
||||
case GL_R16:
|
||||
case GL_RG8_SNORM:
|
||||
case GL_R16_SNORM:
|
||||
return TextureViewClass::Bits16;
|
||||
case GL_R8UI:
|
||||
case GL_R8I:
|
||||
case GL_R8:
|
||||
case GL_R8_SNORM:
|
||||
return TextureViewClass::Bits8;
|
||||
case GL_COMPRESSED_RED_RGTC1:
|
||||
case GL_COMPRESSED_SIGNED_RED_RGTC1:
|
||||
return TextureViewClass::Rgtc1Red;
|
||||
case GL_COMPRESSED_RG_RGTC2:
|
||||
case GL_COMPRESSED_SIGNED_RG_RGTC2:
|
||||
return TextureViewClass::Rgtc2Rg;
|
||||
case GL_COMPRESSED_RGBA_BPTC_UNORM:
|
||||
case GL_COMPRESSED_SRGB_ALPHA_BPTC_UNORM:
|
||||
return TextureViewClass::BptcUnorm;
|
||||
case GL_COMPRESSED_RGB_BPTC_SIGNED_FLOAT:
|
||||
case GL_COMPRESSED_RGB_BPTC_UNSIGNED_FLOAT:
|
||||
return TextureViewClass::BptcFloat;
|
||||
default:
|
||||
// Every depth/stencil format, every S3TC/ETC/ASTC format and every unsized format
|
||||
// reaches here. The caller must then demand an EXACT format match.
|
||||
return TextureViewClass::None;
|
||||
}
|
||||
}
|
||||
|
||||
// GL 4.6 core table 8.20 ("Legal texture targets for TextureView").
|
||||
Bool IsLegalTextureViewTargetPair(TextureTarget origTarget, TextureTarget viewTarget) {
|
||||
switch (origTarget) {
|
||||
case TextureTarget::Texture1D:
|
||||
return viewTarget == TextureTarget::Texture1D || viewTarget == TextureTarget::Texture1DArray;
|
||||
case TextureTarget::Texture2D:
|
||||
return viewTarget == TextureTarget::Texture2D || viewTarget == TextureTarget::Texture2DArray;
|
||||
case TextureTarget::Texture3D:
|
||||
return viewTarget == TextureTarget::Texture3D;
|
||||
case TextureTarget::TextureCubeMap:
|
||||
return viewTarget == TextureTarget::TextureCubeMap || viewTarget == TextureTarget::Texture2D ||
|
||||
viewTarget == TextureTarget::Texture2DArray || viewTarget == TextureTarget::TextureCubeMapArray;
|
||||
case TextureTarget::TextureRectangle:
|
||||
return viewTarget == TextureTarget::TextureRectangle;
|
||||
case TextureTarget::Texture1DArray:
|
||||
return viewTarget == TextureTarget::Texture1DArray || viewTarget == TextureTarget::Texture1D;
|
||||
case TextureTarget::Texture2DArray:
|
||||
return viewTarget == TextureTarget::Texture2DArray || viewTarget == TextureTarget::Texture2D ||
|
||||
viewTarget == TextureTarget::TextureCubeMap || viewTarget == TextureTarget::TextureCubeMapArray;
|
||||
case TextureTarget::TextureCubeMapArray:
|
||||
return viewTarget == TextureTarget::TextureCubeMapArray || viewTarget == TextureTarget::Texture2DArray ||
|
||||
viewTarget == TextureTarget::Texture2D || viewTarget == TextureTarget::TextureCubeMap;
|
||||
case TextureTarget::Texture2DMultisample:
|
||||
case TextureTarget::Texture2DMultisampleArray:
|
||||
return viewTarget == TextureTarget::Texture2DMultisample ||
|
||||
viewTarget == TextureTarget::Texture2DMultisampleArray;
|
||||
case TextureTarget::TextureBuffer:
|
||||
// The table lists no legal target for a buffer texture: its storage is a buffer
|
||||
// object, and there is nothing to make a view of.
|
||||
return false;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
Uint RequiredTextureViewLayerCount(TextureTarget viewTarget) {
|
||||
switch (viewTarget) {
|
||||
case TextureTarget::TextureCubeMap:
|
||||
return 6;
|
||||
case TextureTarget::Texture1D:
|
||||
case TextureTarget::Texture2D:
|
||||
case TextureTarget::Texture3D:
|
||||
case TextureTarget::TextureRectangle:
|
||||
case TextureTarget::Texture2DMultisample:
|
||||
return 1;
|
||||
default:
|
||||
// 1D/2D array, cube-map array, 2D multisample array: any count (the cube-map array's
|
||||
// "multiple of 6" is checked by the caller).
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::GLImpl::TextureImpl
|
||||
|
||||
@@ -30,6 +30,16 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
TextureInternalFormat internalFormat,
|
||||
TexturePixelDataType type);
|
||||
Bool ValidateTextureLevelWithUploadTarget(TextureUploadTarget target, Int level);
|
||||
// "Is <level> a level this texture actually has?", which ValidateTextureLevelNumber above
|
||||
// does NOT answer - that one only bounds the index by GL_MAX_TEXTURE_SIZE and knows nothing
|
||||
// about the object. Entry points that resolve a level straight into a backend image
|
||||
// subresource need this one: a level the texture never had is GL_INVALID_VALUE (GL 4.6 core
|
||||
// 18.3.2), and passing it through instead reaches the driver as an out-of-range subresource.
|
||||
// Note the error split is per-entry-point, so this is not universally reusable:
|
||||
// glClearTexImage owes INVALID_OPERATION for the same out-of-range level and spells its own
|
||||
// copy of this predicate in GL_Texture.cpp (GetClearTextureObject).
|
||||
Bool ValidateTextureLevelExists(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject, Int level,
|
||||
const char* caller);
|
||||
Bool ValidateTextureObject(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject);
|
||||
// Rejects the per-target default texture objects (name 0) with GL_INVALID_OPERATION for entry
|
||||
// points that require a GenTextures-created texture, e.g. TexStorage* ("An INVALID_OPERATION
|
||||
@@ -40,5 +50,62 @@ namespace MobileGL::MG_Impl::GLImpl::TextureImpl {
|
||||
TextureTarget target);
|
||||
Bool ValidateTextureSubImageOffsets(const SharedPtr<MG_State::GLState::ITextureObject>& textureObject, Int xoffset,
|
||||
Int width, Int yoffset = 0, Int height = 0, Int zoffset = 0, Int depth = 0);
|
||||
Bool ValidateBaseInternalFormatMatch(TextureInternalFormat format1, TextureInternalFormat format2);
|
||||
// The texel block of one glCopyImageSubData endpoint, resolved to the two things the
|
||||
// compatibility rule actually asks about. `compressed` is not redundant with a block bigger
|
||||
// than 1x1: it is what distinguishes "compressed, and so the region is measured in texels of
|
||||
// a blocked image" from "uncompressed, and so it is measured in texels".
|
||||
struct CopyImageTexelBlock {
|
||||
SizeT byteSize = 0;
|
||||
Uint blockWidth = 1;
|
||||
Uint blockHeight = 1;
|
||||
Bool compressed = false;
|
||||
};
|
||||
// `compressedFormat` is the GLenum a glCompressedTexImage* upload recorded for the level, or
|
||||
// GL_NONE. It has to be asked for separately because MobileGL stores every compressed format
|
||||
// in uncompressed storage (ConvertGLEnumToTextureInternalFormat), so the TextureInternalFormat
|
||||
// alone can no longer tell a BPTC image from the RGBA8 backing it.
|
||||
CopyImageTexelBlock ResolveCopyImageTexelBlock(TextureInternalFormat format, GLenum compressedFormat);
|
||||
// GL 4.6 core 18.3.2: the two images must be COMPATIBLE, and compatible means their texel
|
||||
// blocks are the same SIZE - not that they share a base internal format. RGBA32UI into
|
||||
// RGBA32F is legal (both 128-bit) while RGBA8 into RGBA32F is not, and a compressed image
|
||||
// pairs with an uncompressed one whose texel is as big as the compressed block.
|
||||
Bool ValidateCopyImageFormatCompatibility(const CopyImageTexelBlock& srcBlock,
|
||||
const CopyImageTexelBlock& dstBlock);
|
||||
// GL 4.6 core 18.3.2: for a compressed image the region's origin must sit on a block
|
||||
// boundary and its size must be a whole number of blocks - unless the edge it runs to is
|
||||
// the edge of the image.
|
||||
Bool ValidateCopyImageBlockAlignment(const CopyImageTexelBlock& block, Int x, Int y, Int width, Int height,
|
||||
Int imageWidth, Int imageHeight, const char* endpointName);
|
||||
// GL 4.6 SS 8.6 subset rule for glCopyTexImage*: the read buffer must supply every component
|
||||
// the requested internalformat asks for, but may supply more.
|
||||
Bool ValidateCopyTexImageBaseFormatSubset(TextureInternalFormat destFormat, TextureInternalFormat srcFormat);
|
||||
|
||||
// ---- glTextureView (ARB_texture_view / GL 4.6 core 8.18) ----
|
||||
// Table 8.21's view classes. `None` is not a class - it means the format has NO entry in the
|
||||
// table, which the spec turns into a much stricter rule than "same class": such a format can
|
||||
// only ever be viewed as ITSELF. Every depth, stencil and depth/stencil format lands here,
|
||||
// which is why the Better Clouds D24S8 view must name GL_DEPTH24_STENCIL8 exactly.
|
||||
enum class TextureViewClass {
|
||||
None = 0,
|
||||
Bits128,
|
||||
Bits96,
|
||||
Bits64,
|
||||
Bits48,
|
||||
Bits32,
|
||||
Bits24,
|
||||
Bits16,
|
||||
Bits8,
|
||||
Rgtc1Red,
|
||||
Rgtc2Rg,
|
||||
BptcUnorm,
|
||||
BptcFloat,
|
||||
};
|
||||
TextureViewClass GetTextureViewClass(GLenum internalformat);
|
||||
// Table 8.20: which <target> values glTextureView accepts for a given origtexture target.
|
||||
Bool IsLegalTextureViewTargetPair(TextureTarget origTarget, TextureTarget viewTarget);
|
||||
// Table 8.20 again, read the other way: how many layers <target> requires. Returns 0 for the
|
||||
// targets whose layer count is unconstrained (the array targets), 6 for GL_TEXTURE_CUBE_MAP,
|
||||
// and 1 for every single-layer target. GL_TEXTURE_CUBE_MAP_ARRAY is special-cased by the
|
||||
// caller because its constraint is "a multiple of 6", not an exact count.
|
||||
Uint RequiredTextureViewLayerCount(TextureTarget viewTarget);
|
||||
} // namespace MobileGL::MG_Impl::GLImpl::TextureImpl
|
||||
|
||||
@@ -179,6 +179,28 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
return vao;
|
||||
}
|
||||
|
||||
// The ARB_vertex_attrib_binding entry points that take no vertex array name modify the
|
||||
// *bound* vertex array, and in a core profile the default vertex array (name 0) is not
|
||||
// one: every one of them is INVALID_OPERATION there (GL 4.6 core 10.3.1, and the tail of
|
||||
// each KHR-GL4x.vertex_attrib_binding.negative-* case checks exactly this). MobileGL
|
||||
// keeps a real object at name 0 for the compatibility paths, so GetBoundVertexArray
|
||||
// never returns null and the rule has to be spelled out - behind the same gate the VAO-0
|
||||
// draw rule already uses (MOBILEGL_RELAXED_SEMANTICS, plus "the context never asked for
|
||||
// a core profile"), so applications that legitimately run relaxed keep working.
|
||||
static SharedPtr<MG_State::GLState::VertexArrayObject> GetBoundVertexArrayForBindingApi(const char* funcName) {
|
||||
auto vao = GetBoundVertexArrayOrError(funcName);
|
||||
if (!vao) return nullptr;
|
||||
if (vao->GetExternalIndex() == 0 && !MG_State::IsRelaxedSemanticsActive()) {
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>(
|
||||
"MG_Impl/GLImpl", funcName,
|
||||
"The default vertex array object cannot be modified in a core profile."));
|
||||
return nullptr;
|
||||
}
|
||||
return vao;
|
||||
}
|
||||
|
||||
static bool ValidateVertexAttribPname(GLenum pname) {
|
||||
switch (pname) {
|
||||
case GL_VERTEX_ATTRIB_ARRAY_ENABLED:
|
||||
@@ -293,9 +315,10 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
auto offset = reinterpret_cast<SizeT>(pointer);
|
||||
|
||||
vao->SetAttributeFormat(index, size, dataType, false, stride, offset, true, false);
|
||||
const int effectiveStride = EffectiveVertexStride(stride, size, type);
|
||||
vao->SetAttributeFormat(index, size, dataType, false, stride, offset, true, false, effectiveStride);
|
||||
vao->BindAttributeBuffer(index, vbo);
|
||||
vao->MirrorPointerIntoBinding(index, vbo, offset, EffectiveVertexStride(stride, size, type));
|
||||
vao->MirrorPointerIntoBinding(index, vbo, offset, effectiveStride);
|
||||
}
|
||||
|
||||
void VertexAttribPointer_State(GLuint index, GLint size, GLenum type, GLboolean normalized, GLsizei stride,
|
||||
@@ -323,9 +346,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// backend can pick the reversed VkFormat / pass GL_BGRA through to a GLES driver.
|
||||
const bool isBgra = (size == static_cast<GLint>(GL_BGRA));
|
||||
const int effectiveSize = isBgra ? 4 : size;
|
||||
vao->SetAttributeFormat(index, effectiveSize, dataType, normalized, stride, offset, false, isBgra);
|
||||
const int effectiveStride = EffectiveVertexStride(stride, effectiveSize, type);
|
||||
vao->SetAttributeFormat(index, effectiveSize, dataType, normalized, stride, offset, false, isBgra,
|
||||
effectiveStride);
|
||||
vao->BindAttributeBuffer(index, vbo);
|
||||
vao->MirrorPointerIntoBinding(index, vbo, offset, EffectiveVertexStride(stride, effectiveSize, type));
|
||||
vao->MirrorPointerIntoBinding(index, vbo, offset, effectiveStride);
|
||||
}
|
||||
|
||||
void BindVertexArray_State(GLuint array) {
|
||||
@@ -489,10 +514,17 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
// recorded DataType is always Float64 - what IsLong adds is that this is the *unconverted* form,
|
||||
// as opposed to VertexAttribFormat(GL_DOUBLE), which asks for a float conversion.
|
||||
//
|
||||
// Whether the backend can feed it is detected, not assumed: DirectVulkan needs shaderFloat64,
|
||||
// and DirectGLES can never have it at all. A backend without it declines here, loudly - GL error
|
||||
// plus a log line naming the reason - rather than accepting state no draw could honour and
|
||||
// rendering garbage. The matching startup POST row is in MG_Util/SelfTest/DriverPost.cpp.
|
||||
// Whether the backend can FEED it at full precision is detected, not assumed: DirectVulkan
|
||||
// needs shaderFloat64, and DirectGLES can never have it at all. What that costs is PRECISION,
|
||||
// not the call and no longer the array: GL 4.6 core 10.3.2 defines no error for a well-formed
|
||||
// glVertexAttribLFormat, and a GL 4.3 context has 64-bit attributes in core, so declining the
|
||||
// call would be non-conformant and would make the four pure state queries
|
||||
// (VERTEX_ATTRIB_ARRAY_SIZE / _TYPE / _LONG / _RELATIVE_OFFSET) unanswerable
|
||||
// (KHR-GL43.vertex_attrib_binding.basic-state1/3). The format is therefore RECORDED here and
|
||||
// the array is NARROWED to float32 at draw, matching the fp64 demotion every shader already
|
||||
// gets (DemoteFloat64Pass) - loudly, once, naming the cost. The matching startup POST row is in
|
||||
// MG_Util/SelfTest/DriverPost.cpp; the draw-side narrowing is DirectGLES/Managers.cpp and, on
|
||||
// DirectVulkan, VertexInputStateFactory's Float64 case.
|
||||
static void VertexAttribLFormatSeparate_State(const SharedPtr<MG_State::GLState::VertexArrayObject>& vao,
|
||||
GLuint attribindex, GLint size, GLenum type,
|
||||
GLuint relativeoffset) {
|
||||
@@ -502,15 +534,12 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
|
||||
if (!MG_Backend::pActiveBackendObject ||
|
||||
!MG_Backend::pActiveBackendObject->GetDynamicParameters().SupportsFloat64VertexAttributes) {
|
||||
MGLOG_I("VertexAttribLFormat: attribute %u asked for a 64-bit (GL_DOUBLE) format, but this "
|
||||
"backend has no double-precision vertex attribute support - see the "
|
||||
"\"64-bit vertex attributes\" / \"shaderFloat64\" POST row for what that costs",
|
||||
MGLOG_W_ONCE("VertexAttribLFormat: attribute %u asked for a 64-bit (GL_DOUBLE) format, but this "
|
||||
"backend has no double-precision vertex attribute support - the format is recorded "
|
||||
"and queryable, and the array is FETCHED AT FLOAT32 PRECISION at draw (the same "
|
||||
"narrowing the shader's dvec inputs already get); see the \"64-bit vertex "
|
||||
"attributes\" / \"shaderFloat64\" POST row for what that costs",
|
||||
attribindex);
|
||||
MG_State::pGLContext->RecordError(
|
||||
ErrorCode::InvalidOperation,
|
||||
MakeUnique<GenericErrorInfo>("MG_Impl/GLImpl", "VertexAttribLFormat",
|
||||
"64-bit vertex attributes are not supported by this backend."));
|
||||
return;
|
||||
}
|
||||
|
||||
vao->SetAttributeFormatSeparate(attribindex, size, MG_Util::ConvertGLEnumToDataType(type),
|
||||
@@ -944,7 +973,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
params[0] = static_cast<GLfloat>(attr->Size);
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_STRIDE:
|
||||
params[0] = static_cast<GLfloat>(attr->Stride);
|
||||
params[0] = static_cast<GLfloat>(attr->LegacyStride);
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_TYPE:
|
||||
params[0] = static_cast<GLfloat>(MG_Util::ConvertDataTypeToGLEnum(attr->Type));
|
||||
@@ -1014,7 +1043,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
params[0] = static_cast<GLdouble>(attr->Size);
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_STRIDE:
|
||||
params[0] = static_cast<GLdouble>(attr->Stride);
|
||||
params[0] = static_cast<GLdouble>(attr->LegacyStride);
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_TYPE:
|
||||
params[0] = static_cast<GLdouble>(MG_Util::ConvertDataTypeToGLEnum(attr->Type));
|
||||
@@ -1079,8 +1108,11 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
case GL_VERTEX_ATTRIB_ARRAY_SIZE:
|
||||
params[0] = attr->Size;
|
||||
return;
|
||||
// The legacy shadow, not the resolved draw stride: GL 4.6 core table 23.3 defines this
|
||||
// as the last glVertexAttrib*Pointer argument, which glBindVertexBuffer must not
|
||||
// overwrite even though it does overwrite what the backend actually reads.
|
||||
case GL_VERTEX_ATTRIB_ARRAY_STRIDE:
|
||||
params[0] = attr->Stride;
|
||||
params[0] = attr->LegacyStride;
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_TYPE:
|
||||
params[0] = static_cast<GLint>(MG_Util::ConvertDataTypeToGLEnum(attr->Type));
|
||||
@@ -1138,7 +1170,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
const auto& attr = vao->GetAttribute(index);
|
||||
*pointer = reinterpret_cast<void*>(attr.Offset);
|
||||
*pointer = reinterpret_cast<void*>(attr.LegacyPointer);
|
||||
}
|
||||
|
||||
void GetVertexAttribIiv(GLuint index, GLenum pname, GLint* params) {
|
||||
@@ -1222,7 +1254,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
*param = static_cast<GLint>(attr.Size);
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_STRIDE:
|
||||
*param = static_cast<GLint>(attr.Stride);
|
||||
*param = static_cast<GLint>(attr.LegacyStride);
|
||||
return;
|
||||
case GL_VERTEX_ATTRIB_ARRAY_TYPE:
|
||||
*param = static_cast<GLint>(MG_Util::ConvertDataTypeToGLEnum(attr.Type));
|
||||
@@ -1294,14 +1326,14 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void BindVertexBuffer(GLuint bindingindex, GLuint buffer, GLintptr offset, GLsizei stride) {
|
||||
auto vao = GetBoundVertexArrayOrError("BindVertexBuffer");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("BindVertexBuffer");
|
||||
if (!vao) return;
|
||||
VertexBufferBinding_State(vao, bindingindex, buffer, offset, stride, "BindVertexBuffer");
|
||||
}
|
||||
|
||||
void BindVertexBuffers(GLuint first, GLsizei count, const GLuint* buffers, const GLintptr* offsets,
|
||||
const GLsizei* strides) {
|
||||
auto vao = GetBoundVertexArrayOrError("BindVertexBuffers");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("BindVertexBuffers");
|
||||
if (!vao) return;
|
||||
if (!ValidateVertexBindingRange(first, count, "BindVertexBuffers")) return;
|
||||
for (GLsizei i = 0; i < count; ++i) {
|
||||
@@ -1315,21 +1347,21 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void VertexAttribFormat(GLuint attribindex, GLint size, GLenum type, GLboolean normalized, GLuint relativeoffset) {
|
||||
auto vao = GetBoundVertexArrayOrError("VertexAttribFormat");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("VertexAttribFormat");
|
||||
if (!vao) return;
|
||||
VertexAttribFormatSeparate_State(vao, attribindex, size, type, normalized, relativeoffset, false,
|
||||
"VertexAttribFormat");
|
||||
}
|
||||
|
||||
void VertexAttribIFormat(GLuint attribindex, GLint size, GLenum type, GLuint relativeoffset) {
|
||||
auto vao = GetBoundVertexArrayOrError("VertexAttribIFormat");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("VertexAttribIFormat");
|
||||
if (!vao) return;
|
||||
VertexAttribFormatSeparate_State(vao, attribindex, size, type, GL_FALSE, relativeoffset, true,
|
||||
"VertexAttribIFormat");
|
||||
}
|
||||
|
||||
void VertexAttribLFormat(GLuint attribindex, GLint size, GLenum type, GLuint relativeoffset) {
|
||||
auto vao = GetBoundVertexArrayOrError("VertexAttribLFormat");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("VertexAttribLFormat");
|
||||
if (!vao) return;
|
||||
VertexAttribLFormatSeparate_State(vao, attribindex, size, type, relativeoffset);
|
||||
}
|
||||
@@ -1341,7 +1373,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void VertexAttribBinding(GLuint attribindex, GLuint bindingindex) {
|
||||
auto vao = GetBoundVertexArrayOrError("VertexAttribBinding");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("VertexAttribBinding");
|
||||
if (!vao) return;
|
||||
if (!VertexArrayImpl::ValidateVertexAttributeIndex(attribindex)) return;
|
||||
if (!ValidateVertexBindingIndex(bindingindex, "VertexAttribBinding")) return;
|
||||
@@ -1349,7 +1381,7 @@ namespace MobileGL::MG_Impl::GLImpl {
|
||||
}
|
||||
|
||||
void VertexBindingDivisor(GLuint bindingindex, GLuint divisor) {
|
||||
auto vao = GetBoundVertexArrayOrError("VertexBindingDivisor");
|
||||
auto vao = GetBoundVertexArrayForBindingApi("VertexBindingDivisor");
|
||||
if (!vao) return;
|
||||
if (!ValidateVertexBindingIndex(bindingindex, "VertexBindingDivisor")) return;
|
||||
vao->SetBindingDivisor(bindingindex, divisor);
|
||||
|
||||
@@ -166,32 +166,32 @@ MOBILEGL_GLX_API int glXSwapIntervalSGI(int interval) {
|
||||
|
||||
// Legacy entry points some loaders probe for; harmless no-op stubs.
|
||||
MOBILEGL_GLX_API void glXCopyContext(Display*, void*, void*, unsigned long) {
|
||||
MGLOG_W("glx: glXCopyContext is not supported");
|
||||
MGLOG_W_ONCE("glx: glXCopyContext is not supported");
|
||||
}
|
||||
|
||||
MOBILEGL_GLX_API unsigned long glXCreateGLXPixmap(Display*, void*, unsigned long) {
|
||||
MGLOG_W("glx: glXCreateGLXPixmap is not supported");
|
||||
MGLOG_W_ONCE("glx: glXCreateGLXPixmap is not supported");
|
||||
return 0;
|
||||
}
|
||||
|
||||
MOBILEGL_GLX_API void glXDestroyGLXPixmap(Display*, unsigned long) {}
|
||||
|
||||
MOBILEGL_GLX_API unsigned long glXCreatePixmap(Display*, void*, unsigned long, const int*) {
|
||||
MGLOG_W("glx: glXCreatePixmap is not supported");
|
||||
MGLOG_W_ONCE("glx: glXCreatePixmap is not supported");
|
||||
return 0;
|
||||
}
|
||||
|
||||
MOBILEGL_GLX_API void glXDestroyPixmap(Display*, unsigned long) {}
|
||||
|
||||
MOBILEGL_GLX_API unsigned long glXCreatePbuffer(Display*, void*, const int*) {
|
||||
MGLOG_W("glx: glXCreatePbuffer is not supported");
|
||||
MGLOG_W_ONCE("glx: glXCreatePbuffer is not supported");
|
||||
return 0;
|
||||
}
|
||||
|
||||
MOBILEGL_GLX_API void glXDestroyPbuffer(Display*, unsigned long) {}
|
||||
|
||||
MOBILEGL_GLX_API void glXUseXFont(unsigned long, int, int, int) {
|
||||
MGLOG_W("glx: glXUseXFont is not supported");
|
||||
MGLOG_W_ONCE("glx: glXUseXFont is not supported");
|
||||
}
|
||||
|
||||
MOBILEGL_GLX_API void glXSelectEvent(Display*, unsigned long, unsigned long) {}
|
||||
|
||||
@@ -149,7 +149,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
fns->Sync = reinterpret_cast<decltype(fns->Sync)>(dlsym(fns->Library, "XSync"));
|
||||
}
|
||||
if (!fns->Valid()) {
|
||||
MGLOG_E("glx: failed to load libX11 entry points");
|
||||
MGLOG_E_ONCE("glx: failed to load libX11 entry points");
|
||||
}
|
||||
return fns;
|
||||
}();
|
||||
@@ -314,7 +314,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
Uint32 width = 0;
|
||||
Uint32 height = 0;
|
||||
if (!QueryDrawableSize(dpy, drawable, width, height)) {
|
||||
MGLOG_E("glx: XGetGeometry failed for drawable 0x%lx", drawable);
|
||||
MGLOG_E_ONCE("glx: XGetGeometry failed for drawable 0x%lx", drawable);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -326,7 +326,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
EGLSurface surface = EGLImpl::CreatePlatformWindowSurface(
|
||||
context.Display, context.Config, reinterpret_cast<void*>(drawable), attribs);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E("glx: failed to create window surface for drawable 0x%lx (%ux%u)", drawable,
|
||||
MGLOG_E_ONCE("glx: failed to create window surface for drawable 0x%lx (%ux%u)", drawable,
|
||||
width, height);
|
||||
return nullptr;
|
||||
}
|
||||
@@ -347,7 +347,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
EGLDisplay display = EnsureDisplay();
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
MGLOG_E("glx: no EGL display");
|
||||
MGLOG_E_ONCE("glx: no EGL display");
|
||||
return nullptr;
|
||||
}
|
||||
EGLImpl::BindAPI(EGL_OPENGL_API);
|
||||
@@ -376,13 +376,13 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
EGLint configCount = 0;
|
||||
if (!EGLImpl::ChooseConfig(display, configAttribs, &config, 1, &configCount) ||
|
||||
configCount <= 0) {
|
||||
MGLOG_E("glx: eglChooseConfig failed");
|
||||
MGLOG_E_ONCE("glx: eglChooseConfig failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
EGLContext eglContext = EGLImpl::CreateContext(display, config, shareContext, contextAttribs);
|
||||
if (eglContext == EGL_NO_CONTEXT) {
|
||||
MGLOG_E("glx: eglCreateContext failed");
|
||||
MGLOG_E_ONCE("glx: eglCreateContext failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -931,7 +931,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
|
||||
if (!EGLImpl::MakeCurrent(object->Display, surface->Surface, surface->Surface,
|
||||
object->Context)) {
|
||||
MGLOG_E("glx: eglMakeCurrent failed (drawable=0x%lx, ctx=%p)", drawable, context);
|
||||
MGLOG_E_ONCE("glx: eglMakeCurrent failed (drawable=0x%lx, ctx=%p)", drawable, context);
|
||||
return 0;
|
||||
}
|
||||
t_current = {dpy, drawable, drawable, context};
|
||||
@@ -943,7 +943,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
if (context && draw != read) {
|
||||
// MobileGL's backends reject split draw/read surfaces; bind the draw
|
||||
// drawable for both, which is what every real caller here needs.
|
||||
MGLOG_W("glx: glXMakeContextCurrent draw 0x%lx != read 0x%lx, using draw for both", draw,
|
||||
MGLOG_W_ONCE("glx: glXMakeContextCurrent draw 0x%lx != read 0x%lx, using draw for both", draw,
|
||||
read);
|
||||
}
|
||||
const int result = MakeCurrent(dpy, draw, context);
|
||||
@@ -958,7 +958,7 @@ namespace MobileGL::MG_Impl::GLXImpl {
|
||||
auto& surfaces = DrawableSurfaces();
|
||||
auto it = surfaces.find(drawable);
|
||||
if (it == surfaces.end()) {
|
||||
MGLOG_W("glx: glXSwapBuffers with no surface for drawable 0x%lx", drawable);
|
||||
MGLOG_W_ONCE("glx: glXSwapBuffers with no surface for drawable 0x%lx", drawable);
|
||||
return;
|
||||
}
|
||||
SyncSurfaceSize(dpy, drawable, it->second);
|
||||
|
||||
@@ -31,7 +31,7 @@ namespace MG_Impl::GLXImpl {
|
||||
#endif
|
||||
void* proc = MobileGL::MG_Impl::GetProcAddress(name);
|
||||
if (!proc) {
|
||||
MGLOG_W("Failed to get function: %s", (const char*)name);
|
||||
MGLOG_D("Failed to get function: %s", (const char*)name);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
|
||||
@@ -1403,7 +1403,7 @@ namespace MobileGL::MG_Impl {
|
||||
GETPROC(glFramebufferTextureMultiviewOVR, name);
|
||||
// GETPROC(glNamedFramebufferTextureMultiviewOVR, name);
|
||||
|
||||
MGLOG_W("GetProcAddress(%s) = nullptr!", name);
|
||||
MGLOG_D("GetProcAddress(%s) = nullptr!", name);
|
||||
return nullptr;
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl
|
||||
|
||||
@@ -269,7 +269,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
}
|
||||
id metalLayerClass = reinterpret_cast<id>(objc_getClass("CAMetalLayer"));
|
||||
if (!metalLayerClass) {
|
||||
MGLOG_E("NSOpenGLImpl: CAMetalLayer class not found");
|
||||
MGLOG_E_ONCE("NSOpenGLImpl: CAMetalLayer class not found");
|
||||
return nil;
|
||||
}
|
||||
|
||||
@@ -310,7 +310,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
static_cast<GLint>(geometry.DrawableSize.width),
|
||||
static_cast<GLint>(geometry.DrawableSize.height));
|
||||
if (error != kCGLNoError) {
|
||||
MGLOG_E("NSOpenGLImpl: failed to attach drawable: %s", CGLImpl::ErrorString(error));
|
||||
MGLOG_E_ONCE("NSOpenGLImpl: failed to attach drawable: %s", CGLImpl::ErrorString(error));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -325,7 +325,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
}
|
||||
const auto error = CGLImpl::SetCurrentContext(context);
|
||||
if (error != kCGLNoError) {
|
||||
MGLOG_E("NSOpenGLImpl: makeCurrentContext failed: %s", CGLImpl::ErrorString(error));
|
||||
MGLOG_E_ONCE("NSOpenGLImpl: makeCurrentContext failed: %s", CGLImpl::ErrorString(error));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -345,7 +345,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
}
|
||||
const auto error = CGLImpl::FlushDrawable(context);
|
||||
if (error != kCGLNoError) {
|
||||
MGLOG_E("NSOpenGLImpl: flushBuffer failed: %s", CGLImpl::ErrorString(error));
|
||||
MGLOG_E_ONCE("NSOpenGLImpl: flushBuffer failed: %s", CGLImpl::ErrorString(error));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -377,7 +377,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
static_cast<GLint>(geometry.DrawableSize.width),
|
||||
static_cast<GLint>(geometry.DrawableSize.height));
|
||||
if (error != kCGLNoError) {
|
||||
MGLOG_E("NSOpenGLImpl: update failed to attach drawable: %s", CGLImpl::ErrorString(error));
|
||||
MGLOG_E_ONCE("NSOpenGLImpl: update failed to attach drawable: %s", CGLImpl::ErrorString(error));
|
||||
return;
|
||||
}
|
||||
CGLImpl::UpdateContext(context);
|
||||
@@ -421,7 +421,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
SEL selector = sel_registerName(selectorName);
|
||||
Method method = class_getInstanceMethod(cls, selector);
|
||||
if (!method) {
|
||||
MGLOG_W("NSOpenGLImpl: missing instance method %s", selectorName);
|
||||
MGLOG_W_ONCE("NSOpenGLImpl: missing instance method %s", selectorName);
|
||||
return;
|
||||
}
|
||||
if (original) {
|
||||
@@ -434,7 +434,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
SEL selector = sel_registerName(selectorName);
|
||||
Method method = class_getClassMethod(cls, selector);
|
||||
if (!method) {
|
||||
MGLOG_W("NSOpenGLImpl: missing class method %s", selectorName);
|
||||
MGLOG_W_ONCE("NSOpenGLImpl: missing class method %s", selectorName);
|
||||
return;
|
||||
}
|
||||
method_setImplementation(method, replacement);
|
||||
@@ -444,7 +444,7 @@ namespace MobileGL::MG_Impl::NSOpenGLImpl {
|
||||
Class pixelFormatClass = objc_getClass("NSOpenGLPixelFormat");
|
||||
Class contextClass = objc_getClass("NSOpenGLContext");
|
||||
if (!pixelFormatClass || !contextClass) {
|
||||
MGLOG_W("NSOpenGLImpl: NSOpenGL classes are not loaded; hooks not installed");
|
||||
MGLOG_W_ONCE("NSOpenGLImpl: NSOpenGL classes are not loaded; hooks not installed");
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -56,7 +56,7 @@ extern "C" HGLRC WINAPI wglCreateLayerContext(HDC hdc, int iLayerPlane) {
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglCopyContext(HGLRC, HGLRC, UINT) {
|
||||
MGLOG_W("wglCopyContext is not supported");
|
||||
MGLOG_W_ONCE("wglCopyContext is not supported");
|
||||
SetLastError(ERROR_NOT_SUPPORTED);
|
||||
return FALSE;
|
||||
}
|
||||
@@ -132,24 +132,24 @@ extern "C" DWORD WINAPI wglSwapMultipleBuffers(UINT n, CONST WGLSWAP* ps) {
|
||||
// ---- Font rendering (legacy immediate-mode feature; not supported) ----
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontBitmapsA(HDC, DWORD, DWORD, DWORD) {
|
||||
MGLOG_W("wglUseFontBitmapsA is not supported");
|
||||
MGLOG_W_ONCE("wglUseFontBitmapsA is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontBitmapsW(HDC, DWORD, DWORD, DWORD) {
|
||||
MGLOG_W("wglUseFontBitmapsW is not supported");
|
||||
MGLOG_W_ONCE("wglUseFontBitmapsW is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontOutlinesA(HDC, DWORD, DWORD, DWORD, FLOAT, FLOAT, int,
|
||||
LPGLYPHMETRICSFLOAT) {
|
||||
MGLOG_W("wglUseFontOutlinesA is not supported");
|
||||
MGLOG_W_ONCE("wglUseFontOutlinesA is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
extern "C" BOOL WINAPI wglUseFontOutlinesW(HDC, DWORD, DWORD, DWORD, FLOAT, FLOAT, int,
|
||||
LPGLYPHMETRICSFLOAT) {
|
||||
MGLOG_W("wglUseFontOutlinesW is not supported");
|
||||
MGLOG_W_ONCE("wglUseFontOutlinesW is not supported");
|
||||
return FALSE;
|
||||
}
|
||||
|
||||
|
||||
@@ -215,7 +215,7 @@ namespace MobileGL::MG_Impl::WGLImpl {
|
||||
Uint32 width = 0;
|
||||
Uint32 height = 0;
|
||||
if (!QueryClientSize(hwnd, width, height)) {
|
||||
MGLOG_E("wgl: GetClientRect failed for HWND %p", hwnd);
|
||||
MGLOG_E_ONCE("wgl: GetClientRect failed for HWND %p", hwnd);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -227,7 +227,7 @@ namespace MobileGL::MG_Impl::WGLImpl {
|
||||
EGLSurface surface =
|
||||
EGLImpl::CreatePlatformWindowSurface(context.Display, context.Config, hwnd, attribs);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E("wgl: failed to create window surface for HWND %p (%ux%u)", hwnd, width, height);
|
||||
MGLOG_E_ONCE("wgl: failed to create window surface for HWND %p (%ux%u)", hwnd, width, height);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -244,7 +244,7 @@ namespace MobileGL::MG_Impl::WGLImpl {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
EGLDisplay display = EnsureDisplay();
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
MGLOG_E("wgl: no EGL display");
|
||||
MGLOG_E_ONCE("wgl: no EGL display");
|
||||
return nullptr;
|
||||
}
|
||||
EGLImpl::BindAPI(EGL_OPENGL_API);
|
||||
@@ -275,13 +275,13 @@ namespace MobileGL::MG_Impl::WGLImpl {
|
||||
EGLConfig config = nullptr;
|
||||
EGLint configCount = 0;
|
||||
if (!EGLImpl::ChooseConfig(display, configAttribs, &config, 1, &configCount) || configCount <= 0) {
|
||||
MGLOG_E("wgl: eglChooseConfig failed");
|
||||
MGLOG_E_ONCE("wgl: eglChooseConfig failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
EGLContext eglContext = EGLImpl::CreateContext(display, config, shareContext, contextAttribs);
|
||||
if (eglContext == EGL_NO_CONTEXT) {
|
||||
MGLOG_E("wgl: eglCreateContext failed");
|
||||
MGLOG_E_ONCE("wgl: eglCreateContext failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -612,7 +612,7 @@ namespace MobileGL::MG_Impl::WGLImpl {
|
||||
auto& surfaces = WindowSurfaces();
|
||||
auto it = surfaces.find(hwnd);
|
||||
if (it == surfaces.end()) {
|
||||
MGLOG_W("wglSwapBuffers: no surface for HWND %p", hwnd);
|
||||
MGLOG_W_ONCE("wglSwapBuffers: no surface for HWND %p", hwnd);
|
||||
return FALSE;
|
||||
}
|
||||
SyncSurfaceSize(hwnd, it->second);
|
||||
@@ -685,7 +685,7 @@ namespace MobileGL::MG_Impl::WGLImpl {
|
||||
}
|
||||
|
||||
if (!EGLImpl::MakeCurrent(object->Display, surface->Surface, surface->Surface, object->Context)) {
|
||||
MGLOG_E("wglMakeCurrent: eglMakeCurrent failed (hdc=%p, hglrc=%p)", hdc, hglrc);
|
||||
MGLOG_E_ONCE("wglMakeCurrent: eglMakeCurrent failed (hdc=%p, hglrc=%p)", hdc, hglrc);
|
||||
return FALSE;
|
||||
}
|
||||
t_current = {hdc, hglrc};
|
||||
|
||||
@@ -24,9 +24,14 @@ set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
|
||||
set(MGL_ITEST_ROOT ${CMAKE_CURRENT_LIST_DIR}/../..)
|
||||
|
||||
# Only meaningful where MobileGL_s exists (i.e. not Android).
|
||||
if (NOT TARGET MobileGL_s)
|
||||
message(STATUS "MobileGL_s is not available; skipping the integration test module")
|
||||
# Desktop links the static implementation directly. Android runs the same
|
||||
# executable from adb shell and links the shipping shared library instead.
|
||||
if (ANDROID)
|
||||
set(MGL_ITEST_MOBILEGL_TARGET MobileGL)
|
||||
elseif (TARGET MobileGL_s)
|
||||
set(MGL_ITEST_MOBILEGL_TARGET MobileGL_s)
|
||||
else()
|
||||
message(STATUS "No MobileGL library target is available; skipping the integration test module")
|
||||
return()
|
||||
endif()
|
||||
|
||||
@@ -50,9 +55,59 @@ add_executable(MobileGLIntegrationTest
|
||||
Scenarios/CrossFrameBufferScenario.cpp
|
||||
Scenarios/ResidentIndexScenario.cpp
|
||||
Scenarios/MultiDrawScenario.cpp
|
||||
Scenarios/DrawParametersScenario.cpp
|
||||
Scenarios/AsyncCompileScenario.cpp
|
||||
Scenarios/XfbAfterClipDistanceScenario.cpp
|
||||
Scenarios/ThreeChannelAttachmentScenario.cpp
|
||||
Scenarios/SnormAttachmentScenario.cpp
|
||||
Scenarios/PipelineFailureScenario.cpp
|
||||
Scenarios/AdvertisedLimitsScenario.cpp
|
||||
Scenarios/PixelStoreSweepScenario.cpp
|
||||
Scenarios/FragCoordOriginScenario.cpp
|
||||
Scenarios/ClearThenReadPixelsScenario.cpp
|
||||
Scenarios/DepthStencilReadbackScenario.cpp
|
||||
Scenarios/DepthStencilReadbackMatrixScenario.cpp
|
||||
Scenarios/DepthStencilReadbackAttachmentShapeScenario.cpp
|
||||
Scenarios/ClipDistanceScenario.cpp
|
||||
Scenarios/ViewportArrayScenario.cpp
|
||||
Scenarios/SsboArrayLengthScenario.cpp
|
||||
Scenarios/DoublePrecisionScenario.cpp
|
||||
Scenarios/UniformInitializerScenario.cpp
|
||||
Scenarios/SwizzleAccessRoutineScenario.cpp
|
||||
Scenarios/IterationRPFirstReductionScenario.cpp
|
||||
Scenarios/IterationRPProgram203Scenario.cpp
|
||||
Scenarios/IterationRPScratchFixScenario.cpp
|
||||
Scenarios/ProgramPipelineScenario.cpp
|
||||
Scenarios/ImageLoadStoreSsoScenario.cpp
|
||||
Scenarios/ImageTargetKindScenario.cpp
|
||||
Scenarios/ImageFormatQualifierScenario.cpp
|
||||
Scenarios/NonCoreImageFormatScenario.cpp
|
||||
Scenarios/ImageSizeAfterRespecScenario.cpp
|
||||
Scenarios/SsboDeclarationFormScenario.cpp
|
||||
Scenarios/Glsl420DeclarationScenario.cpp
|
||||
Scenarios/IoBlockNameCollisionScenario.cpp
|
||||
Scenarios/TessellationDrawModeScenario.cpp
|
||||
Scenarios/GeometryDrawModeScenario.cpp
|
||||
Scenarios/PostLinkAttachScenario.cpp
|
||||
Scenarios/FormatlessImageBakeScenario.cpp
|
||||
Scenarios/FragmentOutputArrayIndexScenario.cpp
|
||||
Scenarios/BufferTextureScenario.cpp
|
||||
Scenarios/VertexAttribBindingScenario.cpp
|
||||
Scenarios/XfbCaptureBufferReuseScenario.cpp
|
||||
Scenarios/XfbPrimitiveQueryScenario.cpp
|
||||
Scenarios/VertexArrayEnableDisableScenario.cpp
|
||||
Scenarios/CopyImageLevelRangeScenario.cpp
|
||||
Scenarios/CopyImageLayeredScenario.cpp
|
||||
Scenarios/TextureViewScenario.cpp
|
||||
Scenarios/PackedWordReadbackScenario.cpp
|
||||
Scenarios/LayeredAttachmentBarrierScenario.cpp
|
||||
Scenarios/LayeredTextureReadbackScenario.cpp
|
||||
Scenarios/AtomicCounterScenario.cpp
|
||||
Scenarios/SsboArrayDynamicIndexScenario.cpp
|
||||
Scenarios/StorageBufferRegrowScenario.cpp
|
||||
Scenarios/RelinkStageSetScenario.cpp
|
||||
Scenarios/GuiBatchScenario.cpp
|
||||
Scenarios/UnboundImageDescriptorScenario.cpp
|
||||
)
|
||||
|
||||
target_include_directories(MobileGLIntegrationTest PRIVATE
|
||||
@@ -63,9 +118,20 @@ target_include_directories(MobileGLIntegrationTest PRIVATE
|
||||
# gtest, not gtest_main: Main.cpp installs the harness banner itself.
|
||||
target_link_libraries(MobileGLIntegrationTest PRIVATE
|
||||
GTest::gtest
|
||||
MobileGL_s
|
||||
${MGL_ITEST_MOBILEGL_TARGET}
|
||||
)
|
||||
|
||||
if (ANDROID)
|
||||
find_library(MGL_ITEST_ANDROID_LIBRARY android REQUIRED)
|
||||
find_library(MGL_ITEST_LOG_LIBRARY log REQUIRED)
|
||||
find_library(MGL_ITEST_MEDIANDK_LIBRARY mediandk REQUIRED)
|
||||
target_link_libraries(MobileGLIntegrationTest PRIVATE
|
||||
${MGL_ITEST_ANDROID_LIBRARY}
|
||||
${MGL_ITEST_LOG_LIBRARY}
|
||||
${MGL_ITEST_MEDIANDK_LIBRARY}
|
||||
)
|
||||
endif()
|
||||
|
||||
if (MSVC)
|
||||
# Same reason as MG_Test/Backend/DirectVulkan: the GLES headers declare gl*
|
||||
# as dllimport on Windows, so the in-library GL entry-point definitions only
|
||||
@@ -74,6 +140,10 @@ if (MSVC)
|
||||
endif()
|
||||
target_compile_definitions(MobileGLIntegrationTest PRIVATE -DNOMINMAX)
|
||||
|
||||
if (ANDROID)
|
||||
return()
|
||||
endif()
|
||||
|
||||
# --- ctest wiring --------------------------------------------------------
|
||||
# A bare libEGL on a glvnd box resolves to whatever vendor comes first, which is
|
||||
# usually Mesa/llvmpipe - a software rasteriser silently replacing the GPU under
|
||||
@@ -169,25 +239,24 @@ endif()
|
||||
option(MOBILEGL_ITEST_REQUIRE_GPU
|
||||
"Fail (rather than skip) the integration scenarios when the headless harness is unusable" OFF)
|
||||
|
||||
# DirectGLES asks the system EGL for a pbuffer config, and on Mesa the default
|
||||
# platform is not X11 unless it is said out loud (run_driver_bench.sh sets the
|
||||
# same variable). Wrong platform here is not a soft failure: eglCreatePbuffer
|
||||
# fails and every scenario skips.
|
||||
if (UNIX AND NOT APPLE AND NOT ANDROID)
|
||||
set(MOBILEGL_ITEST_EGL_PLATFORM "x11" CACHE STRING
|
||||
"EGL_PLATFORM for the integration tests (empty: leave the loader alone)")
|
||||
else()
|
||||
set(MOBILEGL_ITEST_EGL_PLATFORM "" CACHE STRING
|
||||
"EGL_PLATFORM for the integration tests (empty: leave the loader alone)")
|
||||
endif()
|
||||
# No EGL_PLATFORM knob here on purpose. The harness pins EGL_PLATFORM=surfaceless
|
||||
# itself before its first EGL call (HeadlessGL.cpp, EnsureHeadlessPlatform) so a
|
||||
# developer's machine and a CI runner take the SAME path whether or not a window
|
||||
# system happens to be running. This used to inject "x11", which is how the lane
|
||||
# came up green on a workstation with WSLg and died on a runner with no X server.
|
||||
#
|
||||
# A build-system knob would not just be redundant, it would be a trap: `set(...
|
||||
# CACHE ...)` does not rewrite an existing cache, so every build directory
|
||||
# configured before this change would keep injecting EGL_PLATFORM=x11 and go on
|
||||
# binding to a window system - silently, and only on the machines that have one.
|
||||
# Someone reproducing a platform-specific bug sets EGL_PLATFORM in their own
|
||||
# environment, which the harness still honours.
|
||||
|
||||
set(MGL_ITEST_COMMON_ENV "")
|
||||
if (MOBILEGL_ITEST_EGL_VENDOR)
|
||||
list(APPEND MGL_ITEST_COMMON_ENV "__EGL_VENDOR_LIBRARY_FILENAMES=${MOBILEGL_ITEST_EGL_VENDOR}")
|
||||
endif()
|
||||
if (MOBILEGL_ITEST_EGL_PLATFORM)
|
||||
list(APPEND MGL_ITEST_COMMON_ENV "EGL_PLATFORM=${MOBILEGL_ITEST_EGL_PLATFORM}")
|
||||
endif()
|
||||
unset(MOBILEGL_ITEST_EGL_PLATFORM CACHE) # see above: an old cache must not resurrect x11
|
||||
if (MOBILEGL_ITEST_REQUIRE_GPU)
|
||||
list(APPEND MGL_ITEST_COMMON_ENV "MOBILEGL_ITEST_REQUIRE_GPU=1")
|
||||
endif()
|
||||
@@ -195,6 +264,19 @@ endif()
|
||||
set(MGL_ITEST_VULKAN_ENV ${MGL_ITEST_COMMON_ENV})
|
||||
if (MOBILEGL_ITEST_VK_ICD)
|
||||
list(APPEND MGL_ITEST_VULKAN_ENV "VK_ICD_FILENAMES=${MOBILEGL_ITEST_VK_ICD}")
|
||||
# The three iterationRP repairs are tri-state quirks that default to device
|
||||
# auto-detection, and lavapipe is not on any auto list - so on lavapipe the
|
||||
# iterationRP scenarios run unrepaired and Program 203 misses its golden
|
||||
# output. CI's integration-gpu job exports these three by hand; pinning them
|
||||
# to the ICD instead means a local `ctest -L integration-gpu` measures the
|
||||
# same thing the gate does, with no environment to remember.
|
||||
if (MOBILEGL_ITEST_VK_ICD MATCHES "lvp_icd|lavapipe")
|
||||
message(STATUS "Integration tests: lavapipe ICD - forcing the iterationRP repairs on")
|
||||
list(APPEND MGL_ITEST_VULKAN_ENV
|
||||
"MOBILEGL_FIX_ITERATIONRP_SUBGROUP_SCRATCH=1"
|
||||
"MOBILEGL_DERIVE_NUM_SUBGROUPS=1"
|
||||
"MOBILEGL_ITERATIONRP_FIX_BARRIER=1")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# The ENVIRONMENT test property is itself a `;`-list, and gtest_discover_tests
|
||||
@@ -221,6 +303,42 @@ mgl_itest_join_environment(MGL_ITEST_VULKAN_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectVulkan" ${MGL_ITEST_VULKAN_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_VULKAN_ASYNC_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectVulkan" "MOBILEGL_ASYNC_SHADER_COMPILE=1" ${MGL_ITEST_VULKAN_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_GLES_FORCED_DS_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectGLES" "MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION=1" ${MGL_ITEST_COMMON_ENV})
|
||||
|
||||
# The shader-compiler configurations AsyncCompileScenario needs, and the one
|
||||
# ViewportArrayScenario's negative control needs.
|
||||
#
|
||||
# These used to be poked into MG_Config::Features from inside the test bodies. They
|
||||
# cannot be any more - on Android this module links the SHIPPING libMobileGL.so, which
|
||||
# exports nothing internal - and they should not have been anyway: half of what each of
|
||||
# them decides is latched before the first GL call (the compile pool and its threads;
|
||||
# the advertised extension list, which a backend builds once from the configuration in
|
||||
# force at its first use), so an in-process write could only ever have moved the other
|
||||
# half. Every one of them is a whole-process property, and a whole-process property is
|
||||
# spelled with an environment variable and a ctest entry of its own.
|
||||
#
|
||||
# Note the shape of every list here: it APPENDS to MGL_ITEST_COMMON_ENV /
|
||||
# MGL_ITEST_VULKAN_ENV rather than standing alone. A ctest ENVIRONMENT property REPLACES
|
||||
# the job environment rather than adding to it, so an entry that lists only its mode
|
||||
# variable would silently lose the EGL vendor and Vulkan ICD pinning and run against
|
||||
# whatever the loader found first.
|
||||
mgl_itest_join_environment(MGL_ITEST_GLES_ASYNC_ON_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectGLES" "MOBILEGL_ASYNC_SHADER_COMPILE=1" ${MGL_ITEST_COMMON_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_GLES_ASYNC_OFF_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectGLES" "MOBILEGL_ASYNC_SHADER_COMPILE=0" ${MGL_ITEST_COMMON_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_VULKAN_ASYNC_ON_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectVulkan" "MOBILEGL_ASYNC_SHADER_COMPILE=1" ${MGL_ITEST_VULKAN_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_VULKAN_ASYNC_OFF_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectVulkan" "MOBILEGL_ASYNC_SHADER_COMPILE=0" ${MGL_ITEST_VULKAN_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_GLES_OPTIMISTIC_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectGLES" "MOBILEGL_ASYNC_SHADER_COMPILE=1"
|
||||
"MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS=1" ${MGL_ITEST_COMMON_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_VULKAN_OPTIMISTIC_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectVulkan" "MOBILEGL_ASYNC_SHADER_COMPILE=1"
|
||||
"MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS=1" ${MGL_ITEST_VULKAN_ENV})
|
||||
mgl_itest_join_environment(MGL_ITEST_GLES_NO_VIEWPORT_EMULATION_ENVIRONMENT
|
||||
"MOBILEGL_BACKEND_TYPE=DirectGLES" "MOBILEGL_FORCE_VIEWPORT_ARRAY_EMULATION=0" ${MGL_ITEST_COMMON_ENV})
|
||||
|
||||
# TIMEOUT on every entry: a GPU test that wedges must fail the run, not hang it.
|
||||
set(MGL_ITEST_TIMEOUT 120)
|
||||
@@ -268,3 +386,118 @@ gtest_discover_tests(MobileGLIntegrationTest
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_VULKAN_ASYNC_ENVIRONMENT}"
|
||||
)
|
||||
|
||||
# A fourth registration, of the depth/stencil readback scenarios, with the ES
|
||||
# shader-sampling emulation forced on. Not paranoia - without it these scenarios are
|
||||
# UNFALSIFIABLE on the machines this suite runs on: OpenGL ES has no depth or stencil
|
||||
# readback in core, but Mesa accepts the reads anyway, so on llvmpipe every one of them
|
||||
# goes green through a native path that the Adreno device does not have. Deleting the
|
||||
# entire emulation left all of them passing. With the flag the native spellings are off
|
||||
# the table and only the path the device actually takes remains. DirectGLES only - the
|
||||
# emulation is DirectGLES's.
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectGLES.ForcedDepthStencilEmulation."
|
||||
TEST_FILTER "DepthStencilReadback*Scenario.*"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_GLES_FORCED_DS_ENVIRONMENT}"
|
||||
)
|
||||
|
||||
# AsyncCompileScenario, with asynchronous compilation PINNED ON per backend.
|
||||
#
|
||||
# Not a duplicate of what the two ambient registrations already run: they run whatever
|
||||
# MobileGL's built-in default happens to be, and the day that default flips they would
|
||||
# stop covering the asynchronous path without anything going red. These entries are the
|
||||
# ones that keep the asynchronous half tested no matter what ships. They are also the
|
||||
# only place ExtensionStringMatchesTheConfiguration can assert that the extension IS
|
||||
# advertised - the case derives its expectation from this variable and nothing else, and
|
||||
# skips where it is unset, precisely so that it is not asserting the implementation
|
||||
# against itself.
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectGLES.AsyncOn."
|
||||
TEST_FILTER "AsyncCompileScenario.*"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_GLES_ASYNC_ON_ENVIRONMENT}"
|
||||
)
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectVulkan.AsyncOn."
|
||||
TEST_FILTER "AsyncCompileScenario.*"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_VULKAN_ASYNC_ON_ENVIRONMENT}"
|
||||
)
|
||||
|
||||
# The other side of the same switch: asynchronous compilation OFF, so
|
||||
# GL_KHR_parallel_shader_compile must be WITHDRAWN from both spellings of the extension
|
||||
# list and GL_MAX_SHADER_COMPILER_THREADS_KHR must read 0. Only that one case is
|
||||
# registered here because it is the only one that has anything to say in this
|
||||
# configuration - the other four exist to observe worker-built artifacts, and there are
|
||||
# none - so registering the whole scenario would buy four guaranteed skips per backend.
|
||||
# Together with the AsyncOn. entries above, one ctest run still covers both flag states,
|
||||
# which is what the in-process forcing used to be for.
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectGLES.AsyncOff."
|
||||
TEST_FILTER "AsyncCompileScenario.ExtensionStringMatchesTheConfiguration"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_GLES_ASYNC_OFF_ENVIRONMENT}"
|
||||
)
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectVulkan.AsyncOff."
|
||||
TEST_FILTER "AsyncCompileScenario.ExtensionStringMatchesTheConfiguration"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_VULKAN_ASYNC_OFF_ENVIRONMENT}"
|
||||
)
|
||||
|
||||
# The optimistic-status quirk's end-to-end shape. Its own entries and not part of the
|
||||
# AsyncOn. ones because the quirk is not neutral for the rest of the scenario: with it in
|
||||
# force glGetShaderiv(GL_COMPILE_STATUS) deliberately answers without joining, which is
|
||||
# exactly what CompletionStatusPollingThenForcedJoin asserts must NOT happen. Off by
|
||||
# default and never advertised, so - unlike asynchronous compilation, which announces
|
||||
# itself through the extension string - the variable is the only thing that can tell the
|
||||
# case it is in force.
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectGLES.OptimisticShaderStatus."
|
||||
TEST_FILTER "AsyncCompileScenario.IrisShapedTwoPhaseBatchRendersCorrectly"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_GLES_OPTIMISTIC_ENVIRONMENT}"
|
||||
)
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectVulkan.OptimisticShaderStatus."
|
||||
TEST_FILTER "AsyncCompileScenario.IrisShapedTwoPhaseBatchRendersCorrectly"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_VULKAN_OPTIMISTIC_ENVIRONMENT}"
|
||||
)
|
||||
|
||||
# The negative control for the DirectGLES gl_ViewportIndex emulation, in a process that
|
||||
# has it switched off. One case, because it is the only one the switch may touch: with
|
||||
# the emulation off the three positive cases in the same fixture describe behaviour the
|
||||
# backend does not have, so a whole-scenario registration would be three guaranteed reds.
|
||||
# DirectGLES only - the flag steers nothing on DirectVulkan, which routes natively.
|
||||
gtest_discover_tests(MobileGLIntegrationTest
|
||||
TEST_PREFIX "DirectGLES.NoViewportArrayEmulation."
|
||||
TEST_FILTER "ViewportArrayScenario.WithoutTheEmulationEveryIndexCollapsesOntoViewportZero"
|
||||
DISCOVERY_TIMEOUT 30
|
||||
PROPERTIES
|
||||
LABELS integration-gpu
|
||||
TIMEOUT ${MGL_ITEST_TIMEOUT}
|
||||
ENVIRONMENT "${MGL_ITEST_GLES_NO_VIEWPORT_EMULATION_ENVIRONMENT}"
|
||||
)
|
||||
|
||||
@@ -15,6 +15,16 @@
|
||||
#include <ostream>
|
||||
#include <sstream>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#include <windows.h>
|
||||
#elif defined(__ANDROID__)
|
||||
#include <android/hardware_buffer.h>
|
||||
#include <android/native_window.h>
|
||||
#include <media/NdkImage.h>
|
||||
#include <media/NdkImageReader.h>
|
||||
#endif
|
||||
|
||||
// MobileGL's own headers, in the order MobileGL/Includes.h uses them: GL/gl.h
|
||||
// first, then glcorearb.h for the 3.x+ entry points. This binary links
|
||||
// MobileGL_s, so every gl*/egl* below binds to MobileGL's implementation, not
|
||||
@@ -32,7 +42,7 @@
|
||||
// the only construction that is actually predictive here: MobileGL ABORTS
|
||||
// (MOBILEGL_ASSERT -> SIGTRAP) rather than returning an error on an unusable
|
||||
// platform, so nothing the parent can call in-process is allowed to be wrong.
|
||||
#if !defined(_WIN32) && !defined(__APPLE__) && __has_include(<sys/wait.h>)
|
||||
#if !defined(_WIN32) && !defined(__APPLE__) && !defined(__ANDROID__) && __has_include(<sys/wait.h>)
|
||||
#define MGITEST_HAVE_FORK_PREFLIGHT 1
|
||||
#include <csignal>
|
||||
#include <ctime>
|
||||
@@ -53,6 +63,83 @@ namespace MGITest {
|
||||
constexpr int kSurfaceWidth = 128;
|
||||
constexpr int kSurfaceHeight = 96;
|
||||
|
||||
#if defined(_WIN32)
|
||||
HWND g_testWindow = nullptr;
|
||||
|
||||
HWND CreateTestWindow() {
|
||||
static const wchar_t* const kClassName = L"MobileGLIntegrationTestWindow";
|
||||
static bool registered = false;
|
||||
if (!registered) {
|
||||
WNDCLASSW windowClass{};
|
||||
windowClass.lpfnWndProc = DefWindowProcW;
|
||||
windowClass.hInstance = GetModuleHandleW(nullptr);
|
||||
windowClass.lpszClassName = kClassName;
|
||||
if (RegisterClassW(&windowClass) == 0 && GetLastError() != ERROR_CLASS_ALREADY_EXISTS) {
|
||||
return nullptr;
|
||||
}
|
||||
registered = true;
|
||||
}
|
||||
return CreateWindowExW(0, kClassName, L"MobileGL Integration Test", WS_OVERLAPPEDWINDOW,
|
||||
CW_USEDEFAULT, CW_USEDEFAULT, kSurfaceWidth, kSurfaceHeight, nullptr, nullptr,
|
||||
GetModuleHandleW(nullptr), nullptr);
|
||||
}
|
||||
#elif defined(__ANDROID__)
|
||||
AImageReader* g_imageReader = nullptr;
|
||||
ANativeWindow* g_imageReaderWindow = nullptr;
|
||||
|
||||
void DrainImageReader(void*, AImageReader* reader) {
|
||||
AImage* image = nullptr;
|
||||
if (AImageReader_acquireNextImage(reader, &image) == AMEDIA_OK && image != nullptr) {
|
||||
AImage_delete(image);
|
||||
}
|
||||
}
|
||||
|
||||
bool CreateImageReaderWindow() {
|
||||
if (g_imageReaderWindow != nullptr) return true;
|
||||
constexpr int kMaxImages = 4;
|
||||
const media_status_t status = AImageReader_newWithUsage(
|
||||
kSurfaceWidth, kSurfaceHeight, AIMAGE_FORMAT_RGBA_8888,
|
||||
AHARDWAREBUFFER_USAGE_GPU_SAMPLED_IMAGE | AHARDWAREBUFFER_USAGE_GPU_COLOR_OUTPUT,
|
||||
kMaxImages, &g_imageReader);
|
||||
if (status != AMEDIA_OK || g_imageReader == nullptr) return false;
|
||||
|
||||
AImageReader_ImageListener listener = {nullptr, DrainImageReader};
|
||||
AImageReader_setImageListener(g_imageReader, &listener);
|
||||
if (AImageReader_getWindow(g_imageReader, &g_imageReaderWindow) != AMEDIA_OK ||
|
||||
g_imageReaderWindow == nullptr) {
|
||||
AImageReader_setImageListener(g_imageReader, nullptr);
|
||||
AImageReader_delete(g_imageReader);
|
||||
g_imageReader = nullptr;
|
||||
return false;
|
||||
}
|
||||
ANativeWindow_acquire(g_imageReaderWindow);
|
||||
return true;
|
||||
}
|
||||
|
||||
void DestroyImageReaderWindow() {
|
||||
if (g_imageReaderWindow != nullptr) {
|
||||
ANativeWindow_release(g_imageReaderWindow);
|
||||
g_imageReaderWindow = nullptr;
|
||||
}
|
||||
if (g_imageReader != nullptr) {
|
||||
AImageReader_setImageListener(g_imageReader, nullptr);
|
||||
AImageReader_delete(g_imageReader);
|
||||
g_imageReader = nullptr;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
bool UseWindowSurface() {
|
||||
#if defined(_WIN32)
|
||||
const char* value = std::getenv("MOBILEGL_ITEST_WINDOW_SURFACE");
|
||||
return value != nullptr && value[0] != '\0' && std::strcmp(value, "0") != 0;
|
||||
#elif defined(__ANDROID__)
|
||||
return true;
|
||||
#else
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
std::string EnvOr(const char* name, const char* fallback) {
|
||||
const char* value = std::getenv(name);
|
||||
return (value != nullptr && value[0] != '\0') ? std::string(value) : std::string(fallback);
|
||||
@@ -74,6 +161,40 @@ namespace MGITest {
|
||||
std::string renderer;
|
||||
};
|
||||
|
||||
// The harness is headless BY CONSTRUCTION, on every machine: it must never
|
||||
// reach a window system, not even where one happens to be running. This is
|
||||
// not a CI accommodation - it is what keeps a developer's run and a CI run
|
||||
// the same run. The lane was wired up green on a workstation and immediately
|
||||
// died on the runner precisely because the workstation had a DISPLAY (WSLg)
|
||||
// and took Mesa's x11 platform, while the runner has none; that divergence
|
||||
// is the bug, and pinning the platform here is the fix for it.
|
||||
//
|
||||
// Mesa selects its EGL platform from EGL_PLATFORM at loader time, so this
|
||||
// has to run before the first EGL call in the process (see EnsureHeadless
|
||||
// callers). surfaceless is the platform with no window-system dependency at
|
||||
// all; the surface this file then creates is still a pbuffer, which every
|
||||
// platform supports and which the amendment to this rule requires as the
|
||||
// fallback shape on desktop. Android instead supplies an AImageReader
|
||||
// ANativeWindow. DISPLAY/WAYLAND_DISPLAY are cleared as well so that a
|
||||
// driver that consults them directly cannot reintroduce the dependency
|
||||
// behind EGL's back.
|
||||
void EnsureHeadlessPlatform() {
|
||||
#if defined(__linux__) && !defined(__ANDROID__)
|
||||
static bool done = false;
|
||||
if (done) {
|
||||
return;
|
||||
}
|
||||
done = true;
|
||||
// An explicit EGL_PLATFORM from the operator still wins: pinning a
|
||||
// platform is exactly how someone reproduces a platform-specific bug.
|
||||
if (std::getenv("EGL_PLATFORM") == nullptr) {
|
||||
setenv("EGL_PLATFORM", "surfaceless", 1);
|
||||
}
|
||||
unsetenv("DISPLAY");
|
||||
unsetenv("WAYLAND_DISPLAY");
|
||||
#endif
|
||||
}
|
||||
|
||||
// THE bring-up, in one function so the pre-flight child and the parent run
|
||||
// literally the same sequence - a pre-flight that tests something narrower
|
||||
// than what the parent will do is exactly the kind of "predictive" check
|
||||
@@ -82,6 +203,9 @@ namespace MGITest {
|
||||
// Returns 0 on success, or the 1-based index of the step that failed, and
|
||||
// fills outReason either way.
|
||||
int RunEglBringUp(EglBringUp& out, std::string& outReason) {
|
||||
// Belt and braces: the pre-flight child and the parent both enter here,
|
||||
// and neither may be the first to touch EGL without this having run.
|
||||
EnsureHeadlessPlatform();
|
||||
EGLDisplay display = eglGetDisplay(EGL_DEFAULT_DISPLAY);
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
outReason = WithEglError("eglGetDisplay(EGL_DEFAULT_DISPLAY) returned EGL_NO_DISPLAY");
|
||||
@@ -97,8 +221,9 @@ namespace MGITest {
|
||||
return 3;
|
||||
}
|
||||
|
||||
const bool useWindowSurface = UseWindowSurface();
|
||||
const EGLint configAttribs[] = {EGL_SURFACE_TYPE,
|
||||
EGL_PBUFFER_BIT,
|
||||
useWindowSurface ? EGL_WINDOW_BIT : EGL_PBUFFER_BIT,
|
||||
EGL_RED_SIZE,
|
||||
8,
|
||||
EGL_GREEN_SIZE,
|
||||
@@ -115,7 +240,9 @@ namespace MGITest {
|
||||
EGLConfig config = nullptr;
|
||||
EGLint configCount = 0;
|
||||
if (eglChooseConfig(display, configAttribs, &config, 1, &configCount) != EGL_TRUE || configCount < 1) {
|
||||
outReason = WithEglError("eglChooseConfig found no pbuffer-capable RGBA8/D24 config");
|
||||
outReason = WithEglError(useWindowSurface
|
||||
? "eglChooseConfig found no window-capable RGBA8/D24 config"
|
||||
: "eglChooseConfig found no pbuffer-capable RGBA8/D24 config");
|
||||
return 4;
|
||||
}
|
||||
|
||||
@@ -129,10 +256,32 @@ namespace MGITest {
|
||||
return 5;
|
||||
}
|
||||
|
||||
const EGLint pbufferAttribs[] = {EGL_WIDTH, kSurfaceWidth, EGL_HEIGHT, kSurfaceHeight, EGL_NONE};
|
||||
EGLSurface surface = eglCreatePbufferSurface(display, config, pbufferAttribs);
|
||||
EGLSurface surface = EGL_NO_SURFACE;
|
||||
if (useWindowSurface) {
|
||||
#if defined(_WIN32)
|
||||
if (g_testWindow == nullptr) g_testWindow = CreateTestWindow();
|
||||
if (g_testWindow == nullptr) {
|
||||
outReason = "failed to create the Windows integration-test window";
|
||||
return 6;
|
||||
}
|
||||
surface = eglCreateWindowSurface(display, config, g_testWindow, nullptr);
|
||||
#elif defined(__ANDROID__)
|
||||
if (!CreateImageReaderWindow()) {
|
||||
outReason = "failed to create the Android AImageReader integration-test window";
|
||||
return 6;
|
||||
}
|
||||
surface = eglCreateWindowSurface(display, config, g_imageReaderWindow, nullptr);
|
||||
#endif
|
||||
} else {
|
||||
const EGLint pbufferAttribs[] = {EGL_WIDTH, kSurfaceWidth, EGL_HEIGHT, kSurfaceHeight, EGL_NONE};
|
||||
surface = eglCreatePbufferSurface(display, config, pbufferAttribs);
|
||||
}
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
outReason = WithEglError("eglCreatePbufferSurface failed");
|
||||
#if defined(__ANDROID__)
|
||||
DestroyImageReaderWindow();
|
||||
#endif
|
||||
outReason = WithEglError(useWindowSurface ? "eglCreateWindowSurface failed"
|
||||
: "eglCreatePbufferSurface failed");
|
||||
return 6;
|
||||
}
|
||||
// The step that brings the whole backend up (DirectVulkan creates its
|
||||
@@ -199,11 +348,10 @@ namespace MGITest {
|
||||
}
|
||||
if (child == 0) {
|
||||
close(channel[0]);
|
||||
// The child is EXPECTED to die on a signal on an unusable
|
||||
// platform; that is the measurement. Do not let each such
|
||||
// measurement drop a core file next to the test binary.
|
||||
const rlimit noCore{0, 0};
|
||||
setrlimit(RLIMIT_CORE, &noCore);
|
||||
// No core suppression here, deliberately: when the child dies on a
|
||||
// signal, the core IS the diagnosis (an rlimit that used to sit here
|
||||
// made a CI-only crash undebuggable). Machines that do not want
|
||||
// cores control that with the usual ulimit/core_pattern knobs.
|
||||
std::fprintf(stderr, "[itest] pre-flight child: attempting a full EGL bring-up\n");
|
||||
EglBringUp local;
|
||||
std::string reason;
|
||||
@@ -284,9 +432,19 @@ namespace MGITest {
|
||||
}
|
||||
} // namespace
|
||||
|
||||
namespace {
|
||||
bool EnvFlag(const char* name) {
|
||||
const char* value = std::getenv(name);
|
||||
return value != nullptr && value[0] != '\0' && std::strcmp(value, "0") != 0;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool RequireGpu() {
|
||||
const char* value = std::getenv("MOBILEGL_ITEST_REQUIRE_GPU");
|
||||
return value != nullptr && value[0] != '\0' && std::strcmp(value, "0") != 0;
|
||||
return EnvFlag("MOBILEGL_ITEST_REQUIRE_GPU");
|
||||
}
|
||||
|
||||
bool RequireHardwareGpu() {
|
||||
return EnvFlag("MOBILEGL_ITEST_REQUIRE_HARDWARE_GPU");
|
||||
}
|
||||
|
||||
std::ostream& operator<<(std::ostream& os, const Rgba8& c) {
|
||||
@@ -390,7 +548,19 @@ namespace MGITest {
|
||||
}
|
||||
|
||||
HeadlessGL::HeadlessGL() {
|
||||
m_backendName = EnvOr("MOBILEGL_BACKEND_TYPE", "<unset>");
|
||||
// Before anything else in this process can reach EGL, and in particular
|
||||
// before the pre-flight forks - the child must measure the same platform
|
||||
// the parent will use.
|
||||
EnsureHeadlessPlatform();
|
||||
// The backend that is actually about to come up, which is what every
|
||||
// `BackendName() == "DirectGLES"` gate in the scenarios means by the question.
|
||||
// MG_ConfigLoader::InitBackendType defaults an unset MOBILEGL_BACKEND_TYPE to
|
||||
// DirectGLES, so the same default belongs here; this used to report the literal
|
||||
// "<unset>" instead. Under ctest the variable is always set by the ENVIRONMENT
|
||||
// property, which is why that never showed - but run straight from a device
|
||||
// shell, where nothing sets it, DirectGLES came up and every case gated on the
|
||||
// NAME DirectGLES skipped as though it had not.
|
||||
m_backendName = EnvOr("MOBILEGL_BACKEND_TYPE", "DirectGLES");
|
||||
m_usable = BringUp();
|
||||
}
|
||||
|
||||
@@ -441,6 +611,14 @@ namespace MGITest {
|
||||
if (m_context != nullptr) eglDestroyContext(display, static_cast<EGLContext>(m_context));
|
||||
if (m_surface != nullptr) eglDestroySurface(display, static_cast<EGLSurface>(m_surface));
|
||||
eglTerminate(display);
|
||||
#if defined(_WIN32)
|
||||
if (g_testWindow != nullptr) {
|
||||
DestroyWindow(g_testWindow);
|
||||
g_testWindow = nullptr;
|
||||
}
|
||||
#elif defined(__ANDROID__)
|
||||
DestroyImageReaderWindow();
|
||||
#endif
|
||||
m_context = nullptr;
|
||||
m_surface = nullptr;
|
||||
m_display = nullptr;
|
||||
@@ -551,9 +729,13 @@ namespace MGITest {
|
||||
}
|
||||
|
||||
Image ReadPixels(int width, int height) {
|
||||
return ReadPixelsRect(0, 0, width, height);
|
||||
}
|
||||
|
||||
Image ReadPixelsRect(int x, int y, int width, int height) {
|
||||
Image image(width, height);
|
||||
glPixelStorei(GL_PACK_ALIGNMENT, 1);
|
||||
glReadPixels(0, 0, width, height, GL_RGBA, GL_UNSIGNED_BYTE, image.Data());
|
||||
glReadPixels(x, y, width, height, GL_RGBA, GL_UNSIGNED_BYTE, image.Data());
|
||||
return image;
|
||||
}
|
||||
|
||||
|
||||
@@ -14,11 +14,11 @@
|
||||
// inspects backend state - both bugs this module pins were invisible to
|
||||
// state-level assertions and visible only in pixels.
|
||||
//
|
||||
// Headless by construction, following MG_Benchmark/Driver/DriverBench.c: an EGL
|
||||
// context on a PBUFFER surface. No window, no window manager, no human. Unlike
|
||||
// DriverBench the scenarios do draw to the DEFAULT framebuffer (that is where
|
||||
// the Y-flip lives) and do call eglSwapBuffers (that is the frame boundary the
|
||||
// cross-frame scenarios need to be real).
|
||||
// Headless by construction: desktop uses an EGL pbuffer and Android uses an
|
||||
// AImageReader-backed ANativeWindow that needs no Activity. No window manager,
|
||||
// no human. Unlike DriverBench the scenarios do draw to the DEFAULT framebuffer
|
||||
// (that is where the Y-flip lives) and do call eglSwapBuffers (that is the frame
|
||||
// boundary the cross-frame scenarios need to be real).
|
||||
//
|
||||
// One process is one backend: MOBILEGL_BACKEND_TYPE is latched at
|
||||
// initialization, so the CMake wiring runs this binary once per backend rather
|
||||
@@ -41,6 +41,15 @@ namespace MGITest {
|
||||
// a job that ran everything.
|
||||
bool RequireGpu();
|
||||
|
||||
// True when MOBILEGL_ITEST_REQUIRE_HARDWARE_GPU is set: additionally asserts
|
||||
// that the context did NOT land on a software rasterizer. Deliberately a
|
||||
// SEPARATE switch from RequireGpu - a GPU-less CI runner is a supported and
|
||||
// intended configuration for these scenarios (they pin backend draw logic,
|
||||
// which llvmpipe/lavapipe execute faithfully), so CI wants the falsifiability
|
||||
// of REQUIRE_GPU without the hardware demand. Use this one only where a vendor
|
||||
// pin silently degrading to software would invalidate the measurement.
|
||||
bool RequireHardwareGpu();
|
||||
|
||||
struct Rgba8 {
|
||||
std::uint8_t r = 0, g = 0, b = 0, a = 0;
|
||||
|
||||
@@ -175,11 +184,18 @@ namespace MGITest {
|
||||
|
||||
void ClearTo(float r, float g, float b, float a);
|
||||
|
||||
// Reads back the whole currently bound READ framebuffer. width/height must
|
||||
// be the target's full size - DirectVulkan's default-framebuffer readback
|
||||
// only re-orients a full-extent read.
|
||||
// Reads back the whole currently bound READ framebuffer.
|
||||
Image ReadPixels(int width, int height);
|
||||
|
||||
// A PARTIAL glReadPixels. Row 0 of the returned image is GL row `y` of the
|
||||
// framebuffer, i.e. the bottom row of the requested rect - the same
|
||||
// convention ReadPixels uses, just with an origin. This is the shape the
|
||||
// conformance suite reads in (a random sub-rect of the default
|
||||
// framebuffer), and the shape DirectVulkan's default-FBO readback used to
|
||||
// hand back in Vulkan row order because its re-orientation only ran on an
|
||||
// exact full-extent read.
|
||||
Image ReadPixelsRect(int x, int y, int width, int height);
|
||||
|
||||
// Drains any GL error queue and returns the first error, or 0.
|
||||
unsigned int FirstGLError();
|
||||
const char* GLErrorName(unsigned int error);
|
||||
|
||||
@@ -21,12 +21,49 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cctype>
|
||||
#include <cstdlib>
|
||||
#include <string>
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include "HeadlessGL.h"
|
||||
|
||||
namespace MGITest {
|
||||
|
||||
// How a MOBILEGL_* quirk variable reads in THIS process's environment.
|
||||
//
|
||||
// A scenario that needs a non-default configuration takes it from here and skips
|
||||
// when the process it was launched into is not in that configuration, rather than
|
||||
// writing MG_Config::Features itself. Two reasons, and the second one decides it:
|
||||
//
|
||||
// - the feature table is an internal symbol. On Android this module links against
|
||||
// the SHIPPING libMobileGL.so - deliberately, so the on-device run validates the
|
||||
// real artifact - and that library is built -fvisibility=hidden, so nothing
|
||||
// internal is reachable from here at all.
|
||||
// - a quirk poked in-process is already too late for everything latched at
|
||||
// initialization: the compile pool and its threads, and the backend's advertised
|
||||
// extension list, which is built once from the configuration in force at first
|
||||
// use. The process-wide variable is the only spelling that covers the whole
|
||||
// configuration instead of the half of it that is still mutable afterwards.
|
||||
//
|
||||
// The reading rule is MG_ConfigLoader's, character for character (ConfigLoader.cpp,
|
||||
// QueryEnvQuirkOverride / IsTruthyValue): unset is Auto - device auto-detection or a
|
||||
// built-in default, i.e. a value only the implementation knows - a truthy value is
|
||||
// On, and anything else that IS set ("0", "false", "") is Off.
|
||||
enum class AmbientQuirk { Auto, On, Off };
|
||||
|
||||
inline AmbientQuirk AmbientQuirkFromEnvironment(const char* name) {
|
||||
const char* value = std::getenv(name);
|
||||
if (value == nullptr) return AmbientQuirk::Auto;
|
||||
std::string lowered(value);
|
||||
for (char& c : lowered) {
|
||||
c = static_cast<char>(std::tolower(static_cast<unsigned char>(c)));
|
||||
}
|
||||
if (lowered.empty() || lowered == "0" || lowered == "false") return AmbientQuirk::Off;
|
||||
return AmbientQuirk::On;
|
||||
}
|
||||
|
||||
class ScenarioTest : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
@@ -45,12 +82,18 @@ namespace MGITest {
|
||||
}
|
||||
GTEST_SKIP() << "no usable GPU/display/ICD for backend " << gl.BackendName() << ": " << gl.SkipReason();
|
||||
}
|
||||
if (RequireGpu() && LooksLikeSoftwareRasterizer(gl.RendererString())) {
|
||||
// "Ran on llvmpipe" must not be able to pass as "ran on the GPU":
|
||||
// a misconfigured vendor pin silently lands on the software
|
||||
// rasterizer, and REQUIRE_GPU exists precisely to make that loud.
|
||||
FAIL() << "MOBILEGL_ITEST_REQUIRE_GPU is set but the context landed on a software rasterizer: "
|
||||
<< gl.RendererString();
|
||||
if (RequireHardwareGpu() && LooksLikeSoftwareRasterizer(gl.RendererString())) {
|
||||
// Only when hardware was asked for BY NAME. REQUIRE_GPU means "an
|
||||
// unusable harness is a failure, not a silent skip" - it is the
|
||||
// falsifiability switch, and CI is exactly where it belongs. But CI
|
||||
// runners have no GPU, so folding "must not be llvmpipe" into the
|
||||
// same switch made the CI lane unpassable by construction: the
|
||||
// scenarios pin backend draw logic, which a software rasterizer
|
||||
// executes just as faithfully. Landing on llvmpipe/lavapipe there is
|
||||
// the intended configuration, not a misconfiguration. A vendor pin
|
||||
// that must not silently degrade sets REQUIRE_HARDWARE_GPU.
|
||||
FAIL() << "MOBILEGL_ITEST_REQUIRE_HARDWARE_GPU is set but the context landed on a software "
|
||||
<< "rasterizer: " << gl.RendererString();
|
||||
}
|
||||
// A scenario starts from a clean slate but shares the context (and so
|
||||
// the renderer's memos) with every other scenario in this process -
|
||||
|
||||
@@ -26,8 +26,20 @@ namespace {
|
||||
const MGITest::HeadlessGL& gl = MGITest::HeadlessGL::Get();
|
||||
std::fprintf(stderr, "MobileGL integration scenarios: backend=%s\n", gl.BackendName().c_str());
|
||||
if (gl.Usable()) {
|
||||
std::fprintf(stderr, " renderer: %s\n surface: %dx%d pbuffer (headless)\n",
|
||||
gl.RendererString().c_str(), gl.Width(), gl.Height());
|
||||
// EGL_PLATFORM is echoed because it is the invariant this harness
|
||||
// rests on: the run is headless on every machine, so a run that
|
||||
// silently bound to a workstation's window system is a different
|
||||
// run from CI's and must be visible as one in the log.
|
||||
const char* eglPlatform = std::getenv("EGL_PLATFORM");
|
||||
#if defined(__ANDROID__)
|
||||
constexpr const char* surfaceKind = "AImageReader window";
|
||||
#else
|
||||
constexpr const char* surfaceKind = "pbuffer";
|
||||
#endif
|
||||
std::fprintf(stderr, " renderer: %s\n surface: %dx%d %s (headless, EGL_PLATFORM=%s)\n",
|
||||
gl.RendererString().c_str(), gl.Width(), gl.Height(),
|
||||
surfaceKind,
|
||||
eglPlatform != nullptr ? eglPlatform : "<unset>");
|
||||
} else if (MGITest::RequireGpu()) {
|
||||
std::fprintf(stderr,
|
||||
" FAILING every scenario (MOBILEGL_ITEST_REQUIRE_GPU is set): %s\n",
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/AdvertisedLimitsScenario.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// "The limit we advertise is a promise, and an application will hold us to it."
|
||||
//
|
||||
// DirectVulkan copied Vulkan descriptor limits straight into the GL limit table. Those are not
|
||||
// the same quantity: Adreno answers maxPerStageDescriptorUniformBuffers at descriptor-indexing
|
||||
// scale, and GL_MAX_COMPUTE_UNIFORM_BLOCKS is a count an app will allocate. KHR-GL44.multi_bind
|
||||
// .dispatch_bind_buffers_base does exactly that - createsO(limit) buffers and splices O(limit)
|
||||
// UBO declarations into one compute shader - and spent ~14 s allocating before dying on
|
||||
// std::bad_alloc. Its sibling dispatch_bind_buffers_range hard-codes 4 buffers and passes.
|
||||
//
|
||||
// Two failure modes, one table:
|
||||
// - too LARGE: an unusable promise (the OOM above).
|
||||
// - too SMALL or negative: a uint32 limit that lost its top bit on the way to a signed Int -
|
||||
// UINT32_MAX arrived as -1, which every downstream std::min then accepted as "small enough".
|
||||
// A conformant GL 4.x implementation may never advertise below the spec minimum either.
|
||||
//
|
||||
// Every bound below is checked on BOTH backends, because the loader casts are shared and the
|
||||
// DirectGLES lane is the control: it takes its limits from a driver that already reports GL
|
||||
// quantities, so an entry that only fails on DirectVulkan is a translation bug and one that
|
||||
// fails on both is a table bug.
|
||||
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "../Harness/HeadlessGL.h"
|
||||
#include "../Harness/ScenarioFixture.h"
|
||||
|
||||
#ifdef GLAPI
|
||||
#undef GLAPI
|
||||
#endif
|
||||
#define GL_GLEXT_PROTOTYPES
|
||||
#include <GL/gl.h>
|
||||
#include <GL/glcorearb.h>
|
||||
#undef GL_GLEXT_PROTOTYPES
|
||||
|
||||
namespace MGITest {
|
||||
namespace {
|
||||
|
||||
struct LimitBound {
|
||||
GLenum pname;
|
||||
const char* name;
|
||||
// The GL 4.x required minimum. A value below this is a conformance failure in its own
|
||||
// right, and is what a sign-flipped uint32 looks like.
|
||||
int minimum;
|
||||
// The largest value this implementation is willing to promise. Chosen well above every
|
||||
// desktop driver's answer, so it can only catch a descriptor-scale number.
|
||||
int ceiling;
|
||||
};
|
||||
|
||||
const std::vector<LimitBound>& BufferLimitTable() {
|
||||
static const std::vector<LimitBound> table = {
|
||||
{GL_MAX_UNIFORM_BUFFER_BINDINGS, "GL_MAX_UNIFORM_BUFFER_BINDINGS", 36, 256},
|
||||
{GL_MAX_COMPUTE_UNIFORM_BLOCKS, "GL_MAX_COMPUTE_UNIFORM_BLOCKS", 12, 256},
|
||||
{GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS, "GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS", 8, 256},
|
||||
{GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS, "GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS", 8, 256},
|
||||
{GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS, "GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS", 8, 256},
|
||||
{GL_MAX_TEXTURE_BUFFER_SIZE, "GL_MAX_TEXTURE_BUFFER_SIZE", 65536, 1 << 27},
|
||||
{GL_MAX_UNIFORM_BLOCK_SIZE, "GL_MAX_UNIFORM_BLOCK_SIZE", 16384, 1 << 30},
|
||||
// Already clamped before this campaign; in the table so a regression there is
|
||||
// caught by the same case.
|
||||
{GL_MAX_SHADER_STORAGE_BLOCK_SIZE, "GL_MAX_SHADER_STORAGE_BLOCK_SIZE", 1 << 24, 512 * 1024 * 1024},
|
||||
{GL_MAX_TEXTURE_IMAGE_UNITS, "GL_MAX_TEXTURE_IMAGE_UNITS", 16, 32},
|
||||
{GL_MAX_COMBINED_TEXTURE_IMAGE_UNITS, "GL_MAX_COMBINED_TEXTURE_IMAGE_UNITS", 48, 192},
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
class AdvertisedLimitsScenario : public ScenarioTest {};
|
||||
|
||||
TEST_F(AdvertisedLimitsScenario, EveryBufferLimitIsWithinItsAdvertisedRange) {
|
||||
for (const LimitBound& bound : BufferLimitTable()) {
|
||||
GLint value = -424242;
|
||||
glGetIntegerv(bound.pname, &value);
|
||||
const unsigned int error = FirstGLError();
|
||||
EXPECT_EQ(error, GLenum(GL_NO_ERROR))
|
||||
<< bound.name << " is not answerable: " << GLErrorName(error);
|
||||
if (error != GL_NO_ERROR) continue;
|
||||
|
||||
EXPECT_GE(value, bound.minimum)
|
||||
<< bound.name << " = " << value << " is below the GL required minimum "
|
||||
<< bound.minimum << " (a negative or tiny value here is a uint32 limit that lost "
|
||||
"its top bit on the way to a signed Int)";
|
||||
EXPECT_LE(value, bound.ceiling)
|
||||
<< bound.name << " = " << value << " exceeds the ceiling " << bound.ceiling
|
||||
<< " this implementation is willing to promise - an application that allocates "
|
||||
"what we advertise will run out of memory";
|
||||
}
|
||||
}
|
||||
|
||||
// A per-stage block count is an amount of BINDING POINTS an application will use, so it
|
||||
// can never exceed the number of binding points that exist. GL 4.6 Table 23.64 states the
|
||||
// relation the other way round (MAX_UNIFORM_BUFFER_BINDINGS >= MAX_COMBINED_UNIFORM_BLOCKS
|
||||
// >= every per-stage count), and DirectVulkan broke it by clamping the two families
|
||||
// independently: a device reporting 256 compute uniform blocks and 84 uniform binding
|
||||
// points passes both ceilings and still cannot serve
|
||||
// KHR-GL44.multi_bind.dispatch_bind_buffers_base, which reads the block count and binds
|
||||
// that many buffers in one glBindBuffersBase - INVALID_OPERATION before a single bind.
|
||||
TEST_F(AdvertisedLimitsScenario, PerStageBlockCountsFitInTheirBindingPoints) {
|
||||
struct Relation {
|
||||
GLenum blocks;
|
||||
const char* blocksName;
|
||||
GLenum bindings;
|
||||
const char* bindingsName;
|
||||
};
|
||||
const Relation relations[] = {
|
||||
{GL_MAX_COMPUTE_UNIFORM_BLOCKS, "GL_MAX_COMPUTE_UNIFORM_BLOCKS", GL_MAX_UNIFORM_BUFFER_BINDINGS,
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS"},
|
||||
{GL_MAX_VERTEX_UNIFORM_BLOCKS, "GL_MAX_VERTEX_UNIFORM_BLOCKS", GL_MAX_UNIFORM_BUFFER_BINDINGS,
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS"},
|
||||
{GL_MAX_FRAGMENT_UNIFORM_BLOCKS, "GL_MAX_FRAGMENT_UNIFORM_BLOCKS", GL_MAX_UNIFORM_BUFFER_BINDINGS,
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS"},
|
||||
{GL_MAX_COMBINED_UNIFORM_BLOCKS, "GL_MAX_COMBINED_UNIFORM_BLOCKS", GL_MAX_UNIFORM_BUFFER_BINDINGS,
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS"},
|
||||
{GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS, "GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS",
|
||||
GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS, "GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS"},
|
||||
{GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS, "GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS",
|
||||
GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS, "GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS"},
|
||||
};
|
||||
for (const Relation& relation : relations) {
|
||||
GLint blocks = -1;
|
||||
GLint bindings = -1;
|
||||
glGetIntegerv(relation.blocks, &blocks);
|
||||
glGetIntegerv(relation.bindings, &bindings);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << relation.blocksName;
|
||||
EXPECT_LE(blocks, bindings)
|
||||
<< relation.blocksName << " = " << blocks << " exceeds " << relation.bindingsName << " = "
|
||||
<< bindings << "; a shader may declare more blocks than there are binding points to bind them to";
|
||||
}
|
||||
}
|
||||
|
||||
// KHR-GL44.multi_bind.functional_bind_buffers_range sizes each of an indexed target's
|
||||
// binding points at MAX_<target>_SIZE / MAX_<target>_BINDINGS and binds all of them in
|
||||
// one glBindBuffersRange. That quotient has to be a legal BindBufferRange size, which
|
||||
// makes the two limits of every indexed family a PAIR: advertise a size that does not
|
||||
// survive division by the binding count and the call fails with INVALID_VALUE before any
|
||||
// of it binds.
|
||||
TEST_F(AdvertisedLimitsScenario, IndexedTargetSizeSurvivesDivisionByItsBindingCount) {
|
||||
struct IndexedFamily {
|
||||
GLenum maxSize;
|
||||
const char* maxSizeName;
|
||||
GLenum maxBindings;
|
||||
const char* maxBindingsName;
|
||||
GLint sizeGranularity; // BindBufferRange's size rule for the target
|
||||
};
|
||||
const IndexedFamily families[] = {
|
||||
{GL_MAX_ATOMIC_COUNTER_BUFFER_SIZE, "GL_MAX_ATOMIC_COUNTER_BUFFER_SIZE",
|
||||
GL_MAX_ATOMIC_COUNTER_BUFFER_BINDINGS, "GL_MAX_ATOMIC_COUNTER_BUFFER_BINDINGS", 1},
|
||||
{GL_MAX_TRANSFORM_FEEDBACK_INTERLEAVED_COMPONENTS, "GL_MAX_TRANSFORM_FEEDBACK_INTERLEAVED_COMPONENTS",
|
||||
GL_MAX_TRANSFORM_FEEDBACK_BUFFERS, "GL_MAX_TRANSFORM_FEEDBACK_BUFFERS", 4},
|
||||
{GL_MAX_UNIFORM_BLOCK_SIZE, "GL_MAX_UNIFORM_BLOCK_SIZE", GL_MAX_UNIFORM_BUFFER_BINDINGS,
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS", 1},
|
||||
{GL_MAX_SHADER_STORAGE_BLOCK_SIZE, "GL_MAX_SHADER_STORAGE_BLOCK_SIZE",
|
||||
GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS, "GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS", 1},
|
||||
};
|
||||
for (const IndexedFamily& family : families) {
|
||||
GLint maxSize = -1;
|
||||
GLint maxBindings = -1;
|
||||
glGetIntegerv(family.maxSize, &maxSize);
|
||||
glGetIntegerv(family.maxBindings, &maxBindings);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << family.maxSizeName;
|
||||
ASSERT_GT(maxBindings, 0) << family.maxBindingsName;
|
||||
const GLint perBinding = maxSize / maxBindings;
|
||||
EXPECT_GT(perBinding, 0)
|
||||
<< family.maxSizeName << " (" << maxSize << ") / " << family.maxBindingsName << " ("
|
||||
<< maxBindings << ") is zero, and BindBufferRange rejects a zero size";
|
||||
EXPECT_EQ(perBinding % family.sizeGranularity, 0)
|
||||
<< family.maxSizeName << " (" << maxSize << ") / " << family.maxBindingsName << " ("
|
||||
<< maxBindings << ") = " << perBinding << " is not a multiple of the "
|
||||
<< family.sizeGranularity << "-byte size granularity BindBufferRange requires for it";
|
||||
}
|
||||
}
|
||||
|
||||
// The OOM case in isolation, because it is the one with a known CTS victim and the one a
|
||||
// future refactor is most likely to reintroduce by copying the Vulkan limit back.
|
||||
TEST_F(AdvertisedLimitsScenario, ComputeUniformBlocksIsAnAmountAnApplicationCouldActuallyAllocate) {
|
||||
GLint blocks = -1;
|
||||
glGetIntegerv(GL_MAX_COMPUTE_UNIFORM_BLOCKS, &blocks);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_GE(blocks, 12);
|
||||
EXPECT_LE(blocks, 256) << "KHR-GL44.multi_bind.dispatch_bind_buffers_base creates one GL buffer "
|
||||
"and one UBO declaration per advertised block";
|
||||
|
||||
GLint blockSize = -1;
|
||||
glGetIntegerv(GL_MAX_UNIFORM_BLOCK_SIZE, &blockSize);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_GT(blockSize, 0);
|
||||
// GL_MAX_COMBINED_COMPUTE_UNIFORM_COMPONENTS is derived from the product of these two,
|
||||
// so their product has to stay representable.
|
||||
EXPECT_LE(static_cast<long long>(blocks) * blockSize,
|
||||
static_cast<long long>(2147483647))
|
||||
<< "blocks(" << blocks << ") * blockSize(" << blockSize << ") overflows the GLint the "
|
||||
"derived component limits are computed in";
|
||||
}
|
||||
|
||||
// ARB_viewport_array's own limits. They are advertised from three different places -
|
||||
// GL_MAX_VIEWPORTS from the frontend's indexed state width, the bounds range and the
|
||||
// subpixel bits from the backend caps table - and each backend fills that table from a
|
||||
// different source, so all three are checked on both lanes.
|
||||
//
|
||||
// GL_VIEWPORT_BOUNDS_RANGE is the one that shipped wrong: GLES has no such query, the
|
||||
// DirectGLES loader's glGetFloatv(GL_VIEWPORT_BOUNDS_RANGE) therefore raised
|
||||
// GL_INVALID_ENUM and left the probe's zero-initialized array in place, and MobileGL
|
||||
// advertised [0, 0] - a range that admits no viewport origin at all, and the check that
|
||||
// kept KHR-GL43.viewport_array.queries red on Espryt after the indexed-state work.
|
||||
TEST_F(AdvertisedLimitsScenario, ViewportArrayLimitsMeetTheirGL43Floors) {
|
||||
GLint maxViewports = -1;
|
||||
glGetIntegerv(GL_MAX_VIEWPORTS, &maxViewports);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_GE(maxViewports, 16) << "GL 4.3 core table 23.53 sets the MAX_VIEWPORTS minimum at 16";
|
||||
EXPECT_LE(maxViewports, 256) << "one viewport rectangle of indexed state is allocated per advertised "
|
||||
"viewport, and the CTS sizes its arrays off this number";
|
||||
|
||||
GLfloat boundsRange[2] = {1.0f, -1.0f};
|
||||
glGetFloatv(GL_VIEWPORT_BOUNDS_RANGE, boundsRange);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_LE(boundsRange[0], -32768.0f)
|
||||
<< "GL 4.6 core table 23.60 sets the VIEWPORT_BOUNDS_RANGE minimum at [-32768, 32767]; got ["
|
||||
<< boundsRange[0] << ", " << boundsRange[1] << "]";
|
||||
EXPECT_GE(boundsRange[1], 32767.0f)
|
||||
<< "GL 4.6 core table 23.60 sets the VIEWPORT_BOUNDS_RANGE minimum at [-32768, 32767]; got ["
|
||||
<< boundsRange[0] << ", " << boundsRange[1] << "]";
|
||||
|
||||
// KNOWN INFIDELITY, pinned here rather than hidden. MobileGL reports the driver's own
|
||||
// VIEWPORT_SUBPIXEL_BITS (4 on llvmpipe, i.e. 1/16-pixel viewport precision), but the
|
||||
// float viewport rectangle glViewportIndexedf stores is snapped to integers on its
|
||||
// way to both backends (ComputeGLViewport, DirectGLES SyncRenderState). The STATE
|
||||
// round trip is exact - which is all KHR-GL43.viewport_array.viewport_api checks, and
|
||||
// all this cluster set out to fix - so the gap is in rasterization only: a fractional
|
||||
// viewport origin rasterizes as if it had been rounded. Nothing in the suite or in
|
||||
// Minecraft sets one. Only the spec floor is asserted; tightening this to EQ(0) would
|
||||
// mean advertising no subpixel precision at all, which is a separate decision about a
|
||||
// limit MobileGL currently passes through from the driver.
|
||||
GLint subpixelBits = -1;
|
||||
glGetIntegerv(GL_VIEWPORT_SUBPIXEL_BITS, &subpixelBits);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
EXPECT_GE(subpixelBits, 0) << "GL 4.6 core table 23.60: VIEWPORT_SUBPIXEL_BITS has a minimum of 0, and "
|
||||
"a negative value is what a sign-flipped uint32 looks like";
|
||||
|
||||
GLint viewportDims[2] = {-1, -1};
|
||||
glGetIntegerv(GL_MAX_VIEWPORT_DIMS, viewportDims);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
GLint maxRenderbufferSize = -1;
|
||||
glGetIntegerv(GL_MAX_RENDERBUFFER_SIZE, &maxRenderbufferSize);
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
// GL 4.6 core 13.6.1: MAX_VIEWPORT_DIMS must be at least as large as the largest
|
||||
// renderable surface, or a full-size framebuffer could not be fully viewported.
|
||||
EXPECT_GE(viewportDims[0], maxRenderbufferSize);
|
||||
EXPECT_GE(viewportDims[1], maxRenderbufferSize);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace MGITest
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user