mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-08 04:08:32 +09:00
Compare commits
1126
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
42ad62b54c | ||
|
|
f41403e227 | ||
|
|
5fbb17f6b9 | ||
|
|
92d8f7269b | ||
|
|
822e405c77 | ||
|
|
b8233f9c4e | ||
|
|
91475a7b6f | ||
|
|
cee17025a0 | ||
|
|
9642ae4d20 | ||
|
|
595d140036 | ||
|
|
b62d1f2078 | ||
|
|
5e82ff968a | ||
|
|
6a80a82dd3 | ||
|
|
f9182a5ca3 | ||
|
|
f37b511fca | ||
|
|
38027d21f8 | ||
|
|
dd2a62228f | ||
|
|
373aa44dd7 | ||
|
|
96646df12e | ||
|
|
f20b20e643 | ||
|
|
2587814970 | ||
|
|
43398e33e8 | ||
|
|
b7557d6615 | ||
|
|
bce34d7fac | ||
|
|
f91857266f | ||
|
|
49cb1be0fd | ||
|
|
a51c68bb2c | ||
|
|
1c723a6cfc | ||
|
|
44805bfa07 | ||
|
|
257fcbfd0b | ||
|
|
2b46a3db96 | ||
|
|
0b36621069 | ||
|
|
9ee2e0a1db | ||
|
|
6b6623ae72 | ||
|
|
a6e029734b | ||
|
|
e7d6bfddac | ||
|
|
f2c879528f | ||
|
|
eaba4ac1dc | ||
|
|
ed6578954e | ||
|
|
e005c8b6cb | ||
|
|
535b5e3095 | ||
|
|
1b05a84928 | ||
|
|
7ccb762936 | ||
|
|
442e7eec1c | ||
|
|
2787d15706 | ||
|
|
74ce58a6c7 | ||
|
|
bb122ebd4f | ||
|
|
9b0ed5b3af | ||
|
|
0e31c1481b | ||
|
|
811f32760e | ||
|
|
01381a0404 | ||
|
|
0f02b0fdb1 | ||
|
|
3be02abf47 | ||
|
|
f91d6b676c | ||
|
|
1ebf191f94 | ||
|
|
ccad803023 | ||
|
|
21159caf31 | ||
|
|
ea4819a21d | ||
|
|
6b882b3ccf | ||
|
|
96bd36c50b | ||
|
|
46fbd837b3 | ||
|
|
796a57a115 | ||
|
|
62a2dae5ba | ||
|
|
532836c058 | ||
|
|
2fced2241b | ||
|
|
21b5fc2d92 | ||
|
|
1f753ab5fa | ||
|
|
3ed9501be5 | ||
|
|
7311251f30 | ||
|
|
7625cf450d | ||
|
|
450eb209b6 | ||
|
|
8c5c39b3c3 | ||
|
|
64a0ea397c | ||
|
|
205d837942 | ||
|
|
b5e9339c66 | ||
|
|
e310e3e9ff | ||
|
|
a0bf4a83bc | ||
|
|
97facf777b | ||
|
|
5cfbb716c0 | ||
|
|
64c3411d70 | ||
|
|
f3a0d9e0a3 | ||
|
|
068786e812 | ||
|
|
8e7cc62c24 | ||
|
|
d2a36d65a3 | ||
|
|
a020de76e3 | ||
|
|
574634adfa | ||
|
|
e71d715e1a | ||
|
|
7b946fd527 | ||
|
|
8f3ce5f5b7 | ||
|
|
21ec744ef2 | ||
|
|
faa7b17da3 | ||
|
|
f6849fc0b3 | ||
|
|
4d1d4f6225 | ||
|
|
cef81df73f | ||
|
|
c4e6ea1f23 | ||
|
|
b4e07ce651 | ||
|
|
ff324057ad | ||
|
|
ef4c6dbe0a | ||
|
|
19fc7346c5 | ||
|
|
bcb0e894ef | ||
|
|
794c10e56c | ||
|
|
28390667d7 | ||
|
|
90dd9bec77 | ||
|
|
7994ca31d3 | ||
|
|
5705e05156 | ||
|
|
ab62f81545 | ||
|
|
f6cf04d6d7 | ||
|
|
38eb9589f9 | ||
|
|
99ebf67a3d | ||
|
|
2292e99476 | ||
|
|
4c5afecc71 | ||
|
|
1c5744f2be | ||
|
|
22859b0958 | ||
|
|
7aa958fbc9 | ||
|
|
a4bd4e04a1 | ||
|
|
09459edb6b | ||
|
|
8af6ebc174 | ||
|
|
577cd8c670 | ||
|
|
5267243404 | ||
|
|
33c2715912 | ||
|
|
3068cdadf8 | ||
|
|
6dd0201bf2 | ||
|
|
05bef7118b | ||
|
|
94e75fef79 | ||
|
|
43bcd03dca | ||
|
|
2ce0595fab | ||
|
|
ba9af18033 | ||
|
|
f7d63f88fa | ||
|
|
6cf5a7744e | ||
|
|
ca3d24f5ea | ||
|
|
cc34d34706 | ||
|
|
757b31592d | ||
|
|
dbae4eda10 | ||
|
|
fa5ff5d168 | ||
|
|
994ae372f8 | ||
|
|
7ba012adf9 | ||
|
|
b1fdffd767 | ||
|
|
18c17ae5ca | ||
|
|
16c010985f | ||
|
|
6f64ec0f51 | ||
|
|
5d47698349 | ||
|
|
7b593e39ef | ||
|
|
d83b4dbbb5 | ||
|
|
964a7fcc92 | ||
|
|
5a7bd9942d | ||
|
|
21a43bf6a4 | ||
|
|
5b6dec2d81 | ||
|
|
543c29bf86 | ||
|
|
ef562ee9b5 | ||
|
|
fa0f6693d0 | ||
|
|
a6c362c6ce | ||
|
|
921504eccf | ||
|
|
7c5fc03b26 | ||
|
|
ed29e63543 | ||
|
|
5c8a9c41d6 | ||
|
|
efa0345c36 | ||
|
|
ce0f18969c | ||
|
|
7ce0966e7d | ||
|
|
1c6ca2753f | ||
|
|
61b0532865 | ||
|
|
f5b8a505ed | ||
|
|
8371365db5 | ||
|
|
b219992ee3 | ||
|
|
94233ef928 | ||
|
|
0827d7a539 | ||
|
|
5248b8b746 | ||
|
|
d868e1c476 | ||
|
|
c3412ca394 | ||
|
|
5ccaff37af | ||
|
|
71e29f9d58 | ||
|
|
8ad07c222c | ||
|
|
5722094d6f | ||
|
|
4831387cf0 | ||
|
|
d03b72267a | ||
|
|
847ec74f48 | ||
|
|
e02e5caa17 | ||
|
|
1958934594 | ||
|
|
dec0c5eaff | ||
|
|
b6a44cd1e2 | ||
|
|
85f45d0e44 | ||
|
|
404236d337 | ||
|
|
6ea948779e | ||
|
|
d8d7530011 | ||
|
|
0f394fa46f | ||
|
|
107669b3db | ||
|
|
3e0460e472 | ||
|
|
33ff177bb2 | ||
|
|
2e6fc1ffc0 | ||
|
|
c6299f754f | ||
|
|
dcf918b9ee | ||
|
|
d98f72447d | ||
|
|
f15cb8900f | ||
|
|
bd0def6133 | ||
|
|
6f8b7fbc40 | ||
|
|
e5fb57f7eb | ||
|
|
c93e5fa409 | ||
|
|
8191075133 | ||
|
|
d6caed7822 | ||
|
|
9152e88734 | ||
|
|
2406e2d219 | ||
|
|
b228f813c0 | ||
|
|
0d0527192a | ||
|
|
81bcbd6c14 | ||
|
|
867fe3e0ef | ||
|
|
0ec487c993 | ||
|
|
23b880c8be | ||
|
|
ebc5bff9b1 | ||
|
|
231d5c90e4 | ||
|
|
d5f5e6405b | ||
|
|
ac3a83b207 | ||
|
|
335f2decbd | ||
|
|
fb1ad96c04 | ||
|
|
313b75a7c0 | ||
|
|
d7976326fa | ||
|
|
72ee7c439c | ||
|
|
cdea275227 | ||
|
|
6a02c5fea0 | ||
|
|
7db5b35a3e | ||
|
|
990e518e33 | ||
|
|
25a8f51db5 | ||
|
|
8f2b766b56 | ||
|
|
b9d8ad0421 | ||
|
|
f8069c0624 | ||
|
|
d0aae85da2 | ||
|
|
b904658b10 | ||
|
|
4b3fd11462 | ||
|
|
9be5d95440 | ||
|
|
d49d79a64b | ||
|
|
f5761ea1f3 | ||
|
|
b3f774d2c0 | ||
|
|
d524330032 | ||
|
|
f2d210b12d | ||
|
|
fd40960f70 | ||
|
|
49aab57f03 | ||
|
|
62dea3bea4 | ||
|
|
57aeeec053 | ||
|
|
9c0144d24a | ||
|
|
1e45958e01 | ||
|
|
6e6f5268fb | ||
|
|
08f98ad9ce | ||
|
|
d39a706d57 | ||
|
|
f3d52faad4 | ||
|
|
ba81ee114e | ||
|
|
c8c7b19579 | ||
|
|
0e7692251d | ||
|
|
34f09291da | ||
|
|
3b65e646e1 | ||
|
|
25b9370815 | ||
|
|
4ce808b9f2 | ||
|
|
5545d31c37 | ||
|
|
588ddba722 | ||
|
|
62301b1061 | ||
|
|
f3a846d336 | ||
|
|
9cdc82fbdd | ||
|
|
4a9d20c49f | ||
|
|
394d1ce748 | ||
|
|
300b458132 | ||
|
|
1f1a331a44 | ||
|
|
817091641c | ||
|
|
e64c7c7e65 | ||
|
|
dd60ff39ce | ||
|
|
96ad7ca0cc | ||
|
|
e80a23eae6 | ||
|
|
088f263495 | ||
|
|
3b3b6e5b8b | ||
|
|
9eda2147b1 | ||
|
|
a63699cde6 | ||
|
|
765aaec6dc | ||
|
|
bcd669bd25 | ||
|
|
f3405d1d53 | ||
|
|
81604d5596 | ||
|
|
31ea6aa5a3 | ||
|
|
7d6f6603c1 | ||
|
|
39c17c0b1b | ||
|
|
88138b48ec | ||
|
|
c114ce750b | ||
|
|
1ca2d3c0fe | ||
|
|
58c17f85a5 | ||
|
|
4873da6844 | ||
|
|
efeb24ff9b | ||
|
|
18c1a4d586 | ||
|
|
66ac3486e1 | ||
|
|
764a44f589 | ||
|
|
d96acb7972 | ||
|
|
bd710078fc | ||
|
|
95a7b17d45 | ||
|
|
19932f9e49 | ||
|
|
1011d9fea1 | ||
|
|
b5565ae503 | ||
|
|
b06ad3f877 | ||
|
|
3311e6034a | ||
|
|
027c1bd4ab | ||
|
|
da52cc3906 | ||
|
|
ebe4fe133f | ||
|
|
534ec65dda | ||
|
|
35ad1ae7fc | ||
|
|
6152ee933f | ||
|
|
dac02ca044 | ||
|
|
42fd02d82f | ||
|
|
6359b0002b | ||
|
|
c186f5f255 | ||
|
|
3d97f6fa8f | ||
|
|
bb582203d9 | ||
|
|
f39e6eb82d | ||
|
|
cb2ba71feb | ||
|
|
6ea7ccdf64 | ||
|
|
14605723f0 | ||
|
|
a680611c9f | ||
|
|
93224ca406 | ||
|
|
fbed4485b7 | ||
|
|
90ae0f048c | ||
|
|
cff959b2e8 | ||
|
|
a50b2c422b | ||
|
|
9dcda82d71 | ||
|
|
28d0af6f04 | ||
|
|
44ee6b66b3 | ||
|
|
28c3cfc1d6 | ||
|
|
38e04eefae | ||
|
|
fd29cb914e | ||
|
|
38497174c8 | ||
|
|
b95fcb7bca | ||
|
|
5fce287de5 | ||
|
|
ae6949e459 | ||
|
|
5437947240 | ||
|
|
f38dbf018d | ||
|
|
2dcc15bb0e | ||
|
|
ff76af9df7 | ||
|
|
76f37a18e6 | ||
|
|
8d1a734c22 | ||
|
|
7d215028fb | ||
|
|
00534d8bbc | ||
|
|
d81a6a0998 | ||
|
|
9bf23d7ffd | ||
|
|
41e45f7d48 | ||
|
|
3dc6a1b6db | ||
|
|
598c5497b0 | ||
|
|
86c00bdf18 | ||
|
|
2f2f95498f | ||
|
|
7105c2ebdc | ||
|
|
13bab780f2 | ||
|
|
027310f993 | ||
|
|
c741a938bc | ||
|
|
01d0f01d13 | ||
|
|
e45f7ae5d4 | ||
|
|
a687873d32 | ||
|
|
43a43c1180 | ||
|
|
65dbfa6f26 | ||
|
|
6de38c666c | ||
|
|
9dff24f3e1 | ||
|
|
5a400e0297 | ||
|
|
b6a2bf08d4 | ||
|
|
6b2a2b5e00 | ||
|
|
512c857f18 | ||
|
|
9f3cac6691 | ||
|
|
4c7332d5e6 | ||
|
|
1c5f6c0986 | ||
|
|
8b75628dec | ||
|
|
8269a1786f | ||
|
|
3019c68945 | ||
|
|
800142c104 | ||
|
|
e9382f5329 | ||
|
|
df7d1edeca | ||
|
|
8025745fa3 | ||
|
|
4a533a215a | ||
|
|
0a5d7ceb6c | ||
|
|
ac81185968 | ||
|
|
9e861f3f7a | ||
|
|
da6f75dbd1 | ||
|
|
951da362f7 | ||
|
|
4c929b9b3f | ||
|
|
8d5072543a | ||
|
|
5de2b9e3e9 | ||
|
|
9192d156d1 | ||
|
|
27cdfbc0ca | ||
|
|
4567c3b468 | ||
|
|
d18c6a1bae | ||
|
|
dd745d7547 | ||
|
|
4407be89cd | ||
|
|
0c1a433af6 | ||
|
|
a2f3efe22c | ||
|
|
b9a15aed61 | ||
|
|
f748a06632 | ||
|
|
4532cae175 | ||
|
|
107b56d603 | ||
|
|
22b749dd37 | ||
|
|
f0c0211767 | ||
|
|
9ebbb76df1 | ||
|
|
95547ab9ce | ||
|
|
8eaf2d0069 | ||
|
|
54a8609c64 | ||
|
|
1fb0eb0737 | ||
|
|
282dd69230 | ||
|
|
e6ebe7078d | ||
|
|
30a91023f6 | ||
|
|
47dd8cdc05 | ||
|
|
8b36a15fb3 | ||
|
|
07fa84fb8d | ||
|
|
a03817b4ee | ||
|
|
641bc0cdd9 | ||
|
|
c069890ac7 | ||
|
|
48dd1c5956 | ||
|
|
a389477f78 | ||
|
|
94a8f1e3f3 | ||
|
|
45d506545e | ||
|
|
d9d63c9496 | ||
|
|
92140405c1 | ||
|
|
b9ecfef0b6 | ||
|
|
fba26ea169 | ||
|
|
93a3b55907 | ||
|
|
d1487bedf0 | ||
|
|
1bb736c57e | ||
|
|
eb76686c1e | ||
|
|
1a04fb8c0c | ||
|
|
306790ee7c | ||
|
|
170ccda3e7 | ||
|
|
e86a9bbec5 | ||
|
|
c5569e71b3 | ||
|
|
7e8c32a063 | ||
|
|
77bd03d962 | ||
|
|
37111ae992 | ||
|
|
6b0c2a15ab | ||
|
|
e9ffd99313 | ||
|
|
7c01ddea0c | ||
|
|
2d4d6e9cfb | ||
|
|
76b8957b99 | ||
|
|
ec685b9fa7 | ||
|
|
a12068df52 | ||
|
|
0b344792cc | ||
|
|
9fa32bdad0 | ||
|
|
a4980f2b56 | ||
|
|
8ca20e28ca | ||
|
|
c353a2055f | ||
|
|
421c20984e | ||
|
|
fc4cd980f2 | ||
|
|
992d16267c | ||
|
|
0ea9e6de5f | ||
|
|
241ed377b4 | ||
|
|
bf312a4b67 | ||
|
|
56b31a9587 | ||
|
|
8a0a8a0274 | ||
|
|
d076c29146 | ||
|
|
930a607bdf | ||
|
|
34685b4bb0 | ||
|
|
c540fb88ee | ||
|
|
6ae3245a0d | ||
|
|
7e048fc2bf | ||
|
|
83cdfd6bdd | ||
|
|
1c76f886cf | ||
|
|
a2e109beff | ||
|
|
63f0756644 | ||
|
|
450215d12c | ||
|
|
3a9e520170 | ||
|
|
d2996ba1cf | ||
|
|
c8632dfefe | ||
|
|
b8a8a660e1 | ||
|
|
7ab83861ca | ||
|
|
eaeba556a3 | ||
|
|
72fa1221a5 | ||
|
|
c4254c4bbd | ||
|
|
199164c2e0 | ||
|
|
e87063e90c | ||
|
|
122da27249 | ||
|
|
bc2d698b3e | ||
|
|
3049c4b82b | ||
|
|
2b3850b76b | ||
|
|
2a0ae743a0 | ||
|
|
6839219c10 | ||
|
|
52ddb440ca | ||
|
|
79feeffd25 | ||
|
|
202037b5a3 | ||
|
|
bce9c48c8e | ||
|
|
b6a7807a3a | ||
|
|
48ba622387 | ||
|
|
05260d1262 | ||
|
|
6eb5ff51c5 | ||
|
|
e526f8e8ac | ||
|
|
e724e88eec | ||
|
|
3b175fb88a | ||
|
|
b5a4e7075a | ||
|
|
520c2b6750 | ||
|
|
57cc652b1d | ||
|
|
65ea54da9e | ||
|
|
c81dd04f08 | ||
|
|
c158bfa584 | ||
|
|
f9f455144c | ||
|
|
64e4840de2 | ||
|
|
bf7b5755cc | ||
|
|
293f64b3c2 | ||
|
|
f0cc07c937 | ||
|
|
4658536652 | ||
|
|
04b4627c65 | ||
|
|
68e13705c8 | ||
|
|
e5388c0e7e | ||
|
|
8bc4808b1a | ||
|
|
5d8a5387e2 | ||
|
|
aa2184e47a | ||
|
|
9152a4a4bc | ||
|
|
1963b427db | ||
|
|
f3def150e7 | ||
|
|
fc0688c223 | ||
|
|
626c7f26fd | ||
|
|
1d947d934b | ||
|
|
4203837648 | ||
|
|
f5cba4c2f1 | ||
|
|
c2a1db3fcb | ||
|
|
594916850f | ||
|
|
c718d6bad8 | ||
|
|
abfda60ff1 | ||
|
|
92cced9bcc | ||
|
|
1929a7c546 | ||
|
|
df7f5a369d | ||
|
|
0de9861da4 | ||
|
|
6b223e4d23 | ||
|
|
48568cdb89 | ||
|
|
8d83dedc0f | ||
|
|
a25d8ee0e7 | ||
|
|
c30bd0fabb | ||
|
|
c9fbf79d6a | ||
|
|
f39e8738da | ||
|
|
c6d22e6ece | ||
|
|
981f10e4da | ||
|
|
076cd0d19d | ||
|
|
5cd82f2002 | ||
|
|
d4922cb0fb | ||
|
|
870d882fef | ||
|
|
e2f873c95c | ||
|
|
efd7b47388 | ||
|
|
a08669df72 | ||
|
|
b8db509581 | ||
|
|
5d6b544021 | ||
|
|
346cd417ca | ||
|
|
f61675e9ce | ||
|
|
8026838563 | ||
|
|
5bd8fa8c4e | ||
|
|
37ef2cb600 | ||
|
|
5184901a5b | ||
|
|
4172959e49 | ||
|
|
eb5b4bca3e | ||
|
|
56db115a1a | ||
|
|
b14edd515a | ||
|
|
57030b5fb9 | ||
|
|
72532de780 | ||
|
|
25323bfb8e | ||
|
|
9ed5dbf483 | ||
|
|
3ca57068a8 | ||
|
|
075471cdbd | ||
|
|
62a0ad5639 | ||
|
|
8bf8f6f906 | ||
|
|
249c1ca574 | ||
|
|
6a843b3088 | ||
|
|
2e94314b78 | ||
|
|
c59c15f66a | ||
|
|
4d1613ba55 | ||
|
|
254cf1dc21 | ||
|
|
607e84deed | ||
|
|
df0e9fca71 | ||
|
|
b831dae8d5 | ||
|
|
0005a50517 | ||
|
|
9f302373d6 | ||
|
|
f896c7396f | ||
|
|
6ca48e40fe | ||
|
|
1cefb9780b | ||
|
|
a1a8a18575 | ||
|
|
bae222227a | ||
|
|
37a050b106 | ||
|
|
12a67f596c | ||
|
|
164bfd810b | ||
|
|
274c234aff | ||
|
|
b8dc4a6004 | ||
|
|
20e1b417cc | ||
|
|
3e8c8b756f | ||
|
|
4dbdbd3bcd | ||
|
|
763d4c3207 | ||
|
|
5eeba29579 | ||
|
|
9351d66dbc | ||
|
|
dae8a40af1 | ||
|
|
e9bd520d9b | ||
|
|
8b78379f6f | ||
|
|
1dc217b32c | ||
|
|
535e9f8b15 | ||
|
|
35626da5c4 | ||
|
|
b9844ed7c1 | ||
|
|
8496e7c7eb | ||
|
|
f0ed5c1b8e | ||
|
|
c247ad5807 | ||
|
|
b604188849 | ||
|
|
7514587b5a | ||
|
|
cf8f928db8 | ||
|
|
5b38f61961 | ||
|
|
176d130f09 | ||
|
|
8bdab8005b | ||
|
|
0b94e02de5 | ||
|
|
273c7ebcf0 | ||
|
|
fe7a5ee1b2 | ||
|
|
3a40778c4b | ||
|
|
5331150cb9 | ||
|
|
37a7f35a27 | ||
|
|
dc2f3a477b | ||
|
|
c6ed9429be | ||
|
|
21753b0e4c | ||
|
|
cb6af44984 | ||
|
|
f509b19b1d | ||
|
|
7464179249 | ||
|
|
c7ac5de28e | ||
|
|
315e9cb194 | ||
|
|
2e7073a890 | ||
|
|
e8d9a913d8 | ||
|
|
305701326c | ||
|
|
4453f1910d | ||
|
|
8c89b1618a | ||
|
|
1b0be9a997 | ||
|
|
3e4ce5caa7 | ||
|
|
f80f6f4a62 | ||
|
|
66caf907fd | ||
|
|
d3150399c7 | ||
|
|
f1c6a12c06 | ||
|
|
947442ec78 | ||
|
|
165dd003d7 | ||
|
|
b852ced3c1 | ||
|
|
4a03d62b91 | ||
|
|
056574eebe | ||
|
|
e61685547a | ||
|
|
7ebaf43282 | ||
|
|
15580ff6a6 | ||
|
|
fd6f5bca83 | ||
|
|
b6d311f20b | ||
|
|
9c0d5517bd | ||
|
|
3445ab9304 | ||
|
|
533219ede7 | ||
|
|
b1f55026af | ||
|
|
a26e9aaf25 | ||
|
|
a55a0645e2 | ||
|
|
e529e12d27 | ||
|
|
e78eee972e | ||
|
|
37cd5b42de | ||
|
|
7fe5247626 | ||
|
|
f098983c9f | ||
|
|
516d2a659e | ||
|
|
808c5dcc46 | ||
|
|
ecea8054a6 | ||
|
|
b0076af9bd | ||
|
|
24cf1e3a7f | ||
|
|
e5ee4cde4f | ||
|
|
e8e1521972 | ||
|
|
6f53b9a6bb | ||
|
|
acaa9f6dc7 | ||
|
|
375f2df694 | ||
|
|
542e50be33 | ||
|
|
ad9ee99521 | ||
|
|
25395a9f9a | ||
|
|
d7029952bb | ||
|
|
340449b77e | ||
|
|
527e229ac8 | ||
|
|
d5bc753764 | ||
|
|
436f7f7e86 | ||
|
|
009e37ec6f | ||
|
|
b253df881d | ||
|
|
c1743aa42d | ||
|
|
0f99d93300 | ||
|
|
e9fa99e16b | ||
|
|
e18d369adf | ||
|
|
22ac8a8c10 | ||
|
|
bebe534bad | ||
|
|
9ffcb23877 | ||
|
|
dc3c2cc5c7 | ||
|
|
4dd2b2216c | ||
|
|
0cada09aa7 | ||
|
|
3ff8cafac6 | ||
|
|
041de6cba3 | ||
|
|
aa5c33a42d | ||
|
|
f892f609c2 | ||
|
|
5e8106114f | ||
|
|
95876d9d8c | ||
|
|
e460536119 | ||
|
|
561d8992bc | ||
|
|
d5e19cb7ba | ||
|
|
d40f753983 | ||
|
|
eb090c6170 | ||
|
|
7b00255b11 | ||
|
|
f0dd5d667b | ||
|
|
7eb3994b02 | ||
|
|
7d31a6fcd7 | ||
|
|
096d6f591b | ||
|
|
19348631ab | ||
|
|
4e558ee142 | ||
|
|
effdaabab3 | ||
|
|
a394fe1af3 | ||
|
|
d16b7ccd6a | ||
|
|
28facc1c3f | ||
|
|
85b68a9969 | ||
|
|
a6a5edf573 | ||
|
|
bd208783d7 | ||
|
|
08e808ef20 | ||
|
|
bb3a18c627 | ||
|
|
74ae1e4a29 | ||
|
|
139de76347 | ||
|
|
2395a6ded2 | ||
|
|
41b15955b0 | ||
|
|
07055bb531 | ||
|
|
638999213e | ||
|
|
5ea49f3c50 | ||
|
|
9fbb708e64 | ||
|
|
f355080b6f | ||
|
|
b8ffd25148 | ||
|
|
2ed96e9678 | ||
|
|
fbbf3c4beb | ||
|
|
b88066b73b | ||
|
|
83d475eb02 | ||
|
|
292576d2a1 | ||
|
|
0c8af978db | ||
|
|
23671f1a99 | ||
|
|
94e882762e | ||
|
|
672538f4f1 | ||
|
|
a9763639ed | ||
|
|
616e694bdd | ||
|
|
0bee379b61 | ||
|
|
d35e452368 | ||
|
|
86f322e252 | ||
|
|
b40def47eb | ||
|
|
93cf3559e1 | ||
|
|
8f947253ad | ||
|
|
8cce59302b | ||
|
|
74ad6d76eb | ||
|
|
95f2f5bab9 | ||
|
|
e5ef9b2ace | ||
|
|
0a138276f8 | ||
|
|
1549598e52 | ||
|
|
1902518cd6 | ||
|
|
bcb8a9b57b | ||
|
|
bdb276cd68 | ||
|
|
79aa381722 | ||
|
|
45f1a13cc3 | ||
|
|
afdbf0a194 | ||
|
|
940ab5fd8e | ||
|
|
6418561d3c | ||
|
|
ce2b5a793f | ||
|
|
6ecefaec75 | ||
|
|
bdf29fc4d5 | ||
|
|
51d9fa91ed | ||
|
|
5cdc6c902e | ||
|
|
03696f8a1a | ||
|
|
7c26ff1b81 | ||
|
|
d472d8c32e | ||
|
|
75e5fe1dd3 | ||
|
|
fb1da4bbd0 | ||
|
|
ac43c0224c | ||
|
|
011b2ad6f8 | ||
|
|
e83e6ed76e | ||
|
|
edec6e4e62 | ||
|
|
cd96db7829 | ||
|
|
ca38d8fe86 | ||
|
|
f10f7df389 | ||
|
|
1cd75f7f54 | ||
|
|
1c84422f8e | ||
|
|
ba330cd70a | ||
|
|
4439162fea | ||
|
|
91120d86ba | ||
|
|
71f5ba9601 | ||
|
|
2e14b44349 | ||
|
|
ac66ea7790 | ||
|
|
632a4f0859 | ||
|
|
e2e4b6e579 | ||
|
|
37255523b1 | ||
|
|
c3a830e9e6 | ||
|
|
8266376838 | ||
|
|
d8e3c29744 | ||
|
|
ef3273674b | ||
|
|
633a25b456 | ||
|
|
b9de562491 | ||
|
|
76f5a23b7f | ||
|
|
4a3a226f27 | ||
|
|
ced7f28898 | ||
|
|
bc2db26e07 | ||
|
|
6649241193 | ||
|
|
fea8e615f9 | ||
|
|
57eb9bd415 | ||
|
|
0cd236414e | ||
|
|
e4957e089a | ||
|
|
3c643d943a | ||
|
|
9b06475811 | ||
|
|
fabae2465b | ||
|
|
233277d94b | ||
|
|
59976a7f7b | ||
|
|
4613167abb | ||
|
|
f5f63aa044 | ||
|
|
6495c6dad9 | ||
|
|
2d1b8cdd30 | ||
|
|
d2cbd2f596 | ||
|
|
6c7c5a1bc7 | ||
|
|
5b116696f0 | ||
|
|
2f1949e093 | ||
|
|
cca4df17d9 | ||
|
|
ef06d90b6b | ||
|
|
e7e6888768 | ||
|
|
a762346a3b | ||
|
|
d78892cad0 | ||
|
|
3fc6357f28 | ||
|
|
65ff056b2e | ||
|
|
f121aeb57f | ||
|
|
d61a0b6904 | ||
|
|
0ef9c76224 | ||
|
|
46f3192c67 | ||
|
|
b2aafc3f95 | ||
|
|
531ebb3537 | ||
|
|
6a11f96a5b | ||
|
|
4fa2e0a58d | ||
|
|
c3d08a2125 | ||
|
|
56be5318ab | ||
|
|
db388e64b9 | ||
|
|
5af927224f | ||
|
|
b425b37e19 | ||
|
|
0132781fff | ||
|
|
9db9513c30 | ||
|
|
2f62b90d7b | ||
|
|
f68c7296a6 | ||
|
|
8fd25acbb6 | ||
|
|
59c9b94d76 | ||
|
|
d23e08f564 | ||
|
|
9a48c3f10c | ||
|
|
790b542163 | ||
|
|
5032148cf0 | ||
|
|
1c14c7b3ba | ||
|
|
82fabe90b3 | ||
|
|
acf7341fb8 | ||
|
|
ac33292e1b | ||
|
|
5d6cb7dfed | ||
|
|
195330ccea | ||
|
|
f6114c9e15 | ||
|
|
d2f2a0039f | ||
|
|
943600edb2 | ||
|
|
76acae9889 | ||
|
|
9cca0a8753 | ||
|
|
8fe8d096fb | ||
|
|
496fa50a23 | ||
|
|
9137396eae | ||
|
|
2f52264f01 | ||
|
|
f0f6d1e5fa | ||
|
|
18d19a9a8b | ||
|
|
3f53041ed2 | ||
|
|
c08ac7db72 | ||
|
|
7419f62159 | ||
|
|
252e59334d | ||
|
|
565dc90bf0 | ||
|
|
84eddaef2f | ||
|
|
641bfb1dd9 | ||
|
|
b87b698148 | ||
|
|
dac5f8964f | ||
|
|
85ffcb74d8 | ||
|
|
9e719461e2 | ||
|
|
f270988e03 | ||
|
|
b4f9401395 | ||
|
|
a1e2007b82 | ||
|
|
e92a57011f | ||
|
|
3736e1fc38 | ||
|
|
2660e1669c | ||
|
|
be6818effe | ||
|
|
a49a463acf | ||
|
|
317b3602c3 | ||
|
|
b637962da0 | ||
|
|
44f9dcb3a2 | ||
|
|
1a2b337618 | ||
|
|
d219ac3f6f | ||
|
|
1a2f867ba9 | ||
|
|
0329f40df5 | ||
|
|
193204de52 | ||
|
|
3f945dcfeb | ||
|
|
2f9ccb84d9 | ||
|
|
f10ea4ebcd | ||
|
|
ada6923039 | ||
|
|
f4cf398651 | ||
|
|
84dba77275 | ||
|
|
19e4ba386d | ||
|
|
7d101182cd | ||
|
|
e60b044ff7 | ||
|
|
4366909cf0 | ||
|
|
7b1d8ce9dd | ||
|
|
94fdeb633f | ||
|
|
50eda634e0 | ||
|
|
2bf75537d8 | ||
|
|
f780736f29 | ||
|
|
85e71622f8 | ||
|
|
ea02a2fac9 | ||
|
|
8368901bf9 | ||
|
|
a18170cdb1 | ||
|
|
e70fb39cb9 | ||
|
|
e7ba689ca2 | ||
|
|
78dcf43c72 | ||
|
|
70951f46e5 | ||
|
|
a6b4c4b049 | ||
|
|
db3d569ed0 | ||
|
|
ec0a1a0b70 | ||
|
|
42e3cce8c3 | ||
|
|
c82062a51b | ||
|
|
61349ac0c2 | ||
|
|
102bd2cfd2 | ||
|
|
e7bb46e819 | ||
|
|
02f8c7ab56 | ||
|
|
96ecabc38b | ||
|
|
57600f2bd3 | ||
|
|
a8628feb3c | ||
|
|
e2695235ab | ||
|
|
44c3ea7844 | ||
|
|
7c81906459 | ||
|
|
4e3679c7b7 | ||
|
|
19c0ae4e33 | ||
|
|
565649e3b6 | ||
|
|
527ef9653a | ||
|
|
23bb5245bd | ||
|
|
e3c4b94b11 | ||
|
|
3fa89640cd | ||
|
|
722bddf916 | ||
|
|
294dad1773 | ||
|
|
b33ae1c481 | ||
|
|
0dc3a0916e | ||
|
|
24372c8558 | ||
|
|
49faa5e749 | ||
|
|
69a3985f9d | ||
|
|
d335131f52 | ||
|
|
fe1b6db653 | ||
|
|
4c2c4cb565 | ||
|
|
bc8109b695 | ||
|
|
a88ca75c14 | ||
|
|
e453a75ea9 | ||
|
|
627f737bbe | ||
|
|
ef3e8da1b5 | ||
|
|
c2d6134cfa | ||
|
|
d51723c153 | ||
|
|
94b2a863bf | ||
|
|
442510f793 | ||
|
|
75a0fc2c0c | ||
|
|
72b1f50314 | ||
|
|
66c36cd945 | ||
|
|
d091d9c460 | ||
|
|
d3a2e2e666 | ||
|
|
a4261f8b19 | ||
|
|
6051588458 | ||
|
|
5c1fd733da | ||
|
|
c499480692 | ||
|
|
3da0b1dfd4 | ||
|
|
720919455f | ||
|
|
df90753d85 | ||
|
|
7bee645b9a | ||
|
|
83a6f24f93 | ||
|
|
8d35be14a0 | ||
|
|
85bd0613ca | ||
|
|
727939af5b | ||
|
|
dd52f0381a | ||
|
|
cf165c0db5 | ||
|
|
ab9db43599 | ||
|
|
d553e363a7 | ||
|
|
be3c3eb9bb | ||
|
|
357807666d | ||
|
|
d1a5a4e39c | ||
|
|
a701c896f0 | ||
|
|
b0de886f8e | ||
|
|
ad1ca4ea92 | ||
|
|
fbaf5261e2 | ||
|
|
19ada4b8f9 | ||
|
|
ff41e59282 | ||
|
|
a15a13ca46 | ||
|
|
2937043e77 | ||
|
|
24efb0bb17 | ||
|
|
dd2bd267c1 | ||
|
|
5cfe9c8998 | ||
|
|
3bd8a62aa8 | ||
|
|
dcd37f3c38 | ||
|
|
a579fd2342 | ||
|
|
f93ff00642 | ||
|
|
8a628631fb | ||
|
|
59733a2a26 | ||
|
|
c7d385c03d | ||
|
|
27a8aa057c | ||
|
|
e63bfbabf0 | ||
|
|
c107d9edcf | ||
|
|
49e69c1c7d | ||
|
|
482b6d7bbf | ||
|
|
dcfe91bdfc | ||
|
|
6072b5b703 | ||
|
|
9c50ac8a69 | ||
|
|
8b7acad257 | ||
|
|
cb8cb48a60 | ||
|
|
8c00ae1d38 | ||
|
|
633b3d3b0a | ||
|
|
60199c0323 | ||
|
|
9e0e03d4c0 | ||
|
|
4972ebf914 | ||
|
|
4a36d63c1b | ||
|
|
d7e768096e | ||
|
|
e13b5d0618 | ||
|
|
c5bf0dc07d | ||
|
|
9fbd83d602 | ||
|
|
39c5b28dfb | ||
|
|
be533994e5 | ||
|
|
368c089172 | ||
|
|
30c6f55c2d | ||
|
|
407d4344c4 | ||
|
|
76e3950028 | ||
|
|
7337b00ca8 | ||
|
|
22582dea3e | ||
|
|
f600a07404 | ||
|
|
ff790e1ff1 | ||
|
|
a345369269 | ||
|
|
8e4a4359b4 | ||
|
|
0a000d1628 | ||
|
|
5bd8e369c4 | ||
|
|
777d756d1b | ||
|
|
7716a8e05d | ||
|
|
94e93b171c | ||
|
|
1bf12be278 | ||
|
|
99b753be9a | ||
|
|
bc2f2d896b | ||
|
|
659fbf26da | ||
|
|
f61a7bb9c0 | ||
|
|
212309083a | ||
|
|
c1ebd01a70 | ||
|
|
af58e2c7b4 | ||
|
|
b373535d6e | ||
|
|
4a3a44924a | ||
|
|
1b2a7989af | ||
|
|
83d1ce177e | ||
|
|
4321c4a827 | ||
|
|
a039ce0987 | ||
|
|
accfaab720 | ||
|
|
35658eb998 | ||
|
|
ccbde0196e | ||
|
|
0209c5461f | ||
|
|
810b4886c7 | ||
|
|
7dfc2149d8 | ||
|
|
9d1e8bd7cd | ||
|
|
bfa4049ac1 | ||
|
|
87c56ce3d5 | ||
|
|
f2ab50b84e | ||
|
|
b6ac6c182e | ||
|
|
5e9f0f0c70 | ||
|
|
e4455aed9a | ||
|
|
bc018c9513 | ||
|
|
2ecba4d70c | ||
|
|
1e11a5950e | ||
|
|
b324363db0 | ||
|
|
a654b15190 | ||
|
|
75ea7f8c0c | ||
|
|
09aeee4c03 | ||
|
|
87122973be | ||
|
|
8b9adc5121 | ||
|
|
9e42e40705 | ||
|
|
8a91225eb1 | ||
|
|
e82815802e | ||
|
|
67069d4a12 | ||
|
|
4a38e7224e | ||
|
|
645f2f748f | ||
|
|
5652f9ae5f | ||
|
|
f3b0b242dd | ||
|
|
876cb7bcd3 | ||
|
|
ee9a121c6f | ||
|
|
00439c5c4a | ||
|
|
d7286a5832 | ||
|
|
972804fdc7 | ||
|
|
bca7328421 | ||
|
|
8097c0d7b1 | ||
|
|
641f54e1b7 | ||
|
|
c36bfe0843 | ||
|
|
8ae66521e3 | ||
|
|
6e21fd9a35 | ||
|
|
a2e8faafe5 | ||
|
|
668fa3033e | ||
|
|
139f3f978d | ||
|
|
09f96e5ba5 | ||
|
|
592a69b1c4 | ||
|
|
9211341ae7 | ||
|
|
5f9d354adb | ||
|
|
0ae36b71f5 | ||
|
|
20e2768f27 | ||
|
|
00c8b5ae3a | ||
|
|
7687c19b33 | ||
|
|
fc3a7cd5f1 | ||
|
|
0b7f906533 | ||
|
|
c78a94ed09 | ||
|
|
debcb171ce | ||
|
|
282468c9d5 | ||
|
|
34b06da83b | ||
|
|
3d55f38dab | ||
|
|
16426db244 | ||
|
|
6f8c76f06c | ||
|
|
43f2043478 | ||
|
|
f61f068e28 | ||
|
|
e234d471fe | ||
|
|
6ff97a9bb1 | ||
|
|
8adf1d3b8a | ||
|
|
2935279603 | ||
|
|
918b39bb63 | ||
|
|
963e88943c | ||
|
|
1a75bcba5e | ||
|
|
b64309edfc | ||
|
|
e3f1db0a9a | ||
|
|
2c848ab44b | ||
|
|
9b23594415 | ||
|
|
e2e52965df | ||
|
|
6e7907f183 | ||
|
|
b4be1292b8 | ||
|
|
db58f4c214 | ||
|
|
59bad77176 | ||
|
|
3504fca8da | ||
|
|
e61817ff6a | ||
|
|
b9df28838c | ||
|
|
fff00e65c4 | ||
|
|
8320b4f456 | ||
|
|
a757669df5 | ||
|
|
e16c0645c0 | ||
|
|
8d37763fd3 | ||
|
|
e0d0551e7b | ||
|
|
caa814e8f9 | ||
|
|
2502d32a2e | ||
|
|
e57dd3529e | ||
|
|
4b8d809237 | ||
|
|
d42182befc | ||
|
|
5efdb1016e | ||
|
|
15e24cda78 | ||
|
|
5013108a7f | ||
|
|
359ba1c572 | ||
|
|
733523011b | ||
|
|
14fc13c34d | ||
|
|
e9fee1358d | ||
|
|
dae3dd9996 | ||
|
|
556db8b02e | ||
|
|
96e993eecf | ||
|
|
fb3e2123cc |
@@ -14,7 +14,6 @@ bugprone-forwarding-reference-overload,
|
||||
bugprone-inaccurate-erase,
|
||||
bugprone-incorrect-roundings,
|
||||
bugprone-integer-division,
|
||||
bugprone-lambda-function-name,
|
||||
bugprone-macro-parentheses,
|
||||
bugprone-macro-repeated-side-effects,
|
||||
bugprone-misplaced-operator-in-strlen-in-alloc,
|
||||
@@ -63,7 +62,6 @@ cert-str34-c,
|
||||
cppcoreguidelines-interfaces-global-init,
|
||||
cppcoreguidelines-narrowing-conversions,
|
||||
cppcoreguidelines-pro-type-member-init,
|
||||
cppcoreguidelines-pro-type-static-cast-downcast,
|
||||
cppcoreguidelines-slicing,
|
||||
google-default-arguments,
|
||||
google-runtime-operator,
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
tools/trace_replay/fixtures/*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
tools/trace_replay/fixtures/*.png filter=lfs diff=lfs merge=lfs -text
|
||||
tools/trace_replay/fixtures/openra.tgz -filter -diff -merge -text
|
||||
tools/trace_replay/fixtures/openra.0000031249.png -filter -diff -merge -text
|
||||
@@ -0,0 +1,199 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=trace-fixture-lib.sh
|
||||
. "${script_dir}/trace-fixture-lib.sh"
|
||||
|
||||
if [ "$#" -lt 1 ] || [ "$#" -gt 2 ]; then
|
||||
echo "usage: $0 <trace-case> [fixture-dir]" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
case_name="$1"
|
||||
fixture_dir="${2:-tools/trace_replay/fixtures}"
|
||||
python_bin="${PYTHON:-python3}"
|
||||
# Fixture mirrors, tried in order before falling back to Git LFS. Override the
|
||||
# whole list with MOBILEGL_TRACE_FIXTURE_MIRROR_BASES (whitespace separated);
|
||||
# MOBILEGL_TRACE_FIXTURE_MIRROR_BASE still works and is tried first.
|
||||
default_mirror_bases=(
|
||||
"https://git.hit.moe/swung0x48/MobileGL/media/branch/dev/tools/trace_replay/fixtures"
|
||||
"https://repo.miawa.cn/mgl/tools/trace_replay/fixtures"
|
||||
)
|
||||
if [ -n "${MOBILEGL_TRACE_FIXTURE_MIRROR_BASES:-}" ]; then
|
||||
read -r -a mirror_bases <<< "${MOBILEGL_TRACE_FIXTURE_MIRROR_BASES}"
|
||||
else
|
||||
mirror_bases=("${default_mirror_bases[@]}")
|
||||
fi
|
||||
if [ -n "${MOBILEGL_TRACE_FIXTURE_MIRROR_BASE:-}" ]; then
|
||||
mirror_bases=("${MOBILEGL_TRACE_FIXTURE_MIRROR_BASE}" "${mirror_bases[@]}")
|
||||
fi
|
||||
# Optional bearer token for mirrors that require authentication (private Gitea).
|
||||
mirror_token="${MOBILEGL_TRACE_FIXTURE_MIRROR_TOKEN:-}"
|
||||
download_attempts="${MOBILEGL_TRACE_FIXTURE_DOWNLOAD_ATTEMPTS:-5}"
|
||||
retry_delay="${MOBILEGL_TRACE_FIXTURE_RETRY_DELAY:-2}"
|
||||
|
||||
if ! command -v "${python_bin}" >/dev/null 2>&1 && command -v python >/dev/null 2>&1; then
|
||||
python_bin=python
|
||||
fi
|
||||
|
||||
if ! [[ "${download_attempts}" =~ ^[1-9][0-9]*$ ]]; then
|
||||
echo "MOBILEGL_TRACE_FIXTURE_DOWNLOAD_ATTEMPTS must be a positive integer: ${download_attempts}" >&2
|
||||
exit 2
|
||||
fi
|
||||
if ! [[ "${retry_delay}" =~ ^[0-9]+$ ]]; then
|
||||
echo "MOBILEGL_TRACE_FIXTURE_RETRY_DELAY must be a non-negative integer: ${retry_delay}" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
fixture_list="$("${python_bin}" tools/trace_replay/trace_cases.py \
|
||||
--format fixture-files \
|
||||
--case "${case_name}" \
|
||||
--fixture-root "${fixture_dir}")"
|
||||
# Strip CR so the script also works when python emits CRLF (Git Bash on Windows).
|
||||
mapfile -t files < <(printf '%s\n' "${fixture_list}" | tr -d '\r')
|
||||
|
||||
include="$(IFS=,; echo "${files[*]}")"
|
||||
if [ "${case_name}" = "OpenRA" ]; then
|
||||
echo "Fixture files for ${case_name} are stored in Git: ${include}"
|
||||
for file in "${files[@]}"; do
|
||||
test -s "${file}"
|
||||
if head -n 1 "${file}" | grep -q "version https://git-lfs.github.com/spec/v1"; then
|
||||
echo "fixture should not be stored as an LFS pointer: ${file}" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
exit 0
|
||||
fi
|
||||
|
||||
fetch_file_from_mirror() {
|
||||
local file="$1"
|
||||
local url="$2"
|
||||
local metadata
|
||||
local expected_oid
|
||||
local expected_size
|
||||
local tmp_file="${file}.tmp"
|
||||
local attempt
|
||||
local partial_size
|
||||
local curl_status
|
||||
local curl_auth
|
||||
|
||||
metadata="$(get_lfs_metadata "${file}")" || return 1
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
|
||||
if [ -f "${tmp_file}" ]; then
|
||||
partial_size="$(wc -c < "${tmp_file}" | tr -d '[:space:]')"
|
||||
if [ "${partial_size}" -gt "${expected_size}" ]; then
|
||||
echo "Discarding oversized partial fixture ${tmp_file}: ${partial_size} > ${expected_size}" >&2
|
||||
rm -f "${tmp_file}"
|
||||
elif [ "${partial_size}" = "${expected_size}" ]; then
|
||||
if verify_fixture_file "${tmp_file}" "${file}" "${expected_oid}" "${expected_size}"; then
|
||||
mv "${tmp_file}" "${file}"
|
||||
return 0
|
||||
fi
|
||||
rm -f "${tmp_file}"
|
||||
fi
|
||||
fi
|
||||
|
||||
for ((attempt = 1; attempt <= download_attempts; attempt++)); do
|
||||
partial_size=0
|
||||
if [ -f "${tmp_file}" ]; then
|
||||
partial_size="$(wc -c < "${tmp_file}" | tr -d '[:space:]')"
|
||||
fi
|
||||
|
||||
if [ "${partial_size}" -gt 0 ]; then
|
||||
echo "Resuming mirror download for ${file} at byte ${partial_size} (attempt ${attempt}/${download_attempts})"
|
||||
else
|
||||
echo "Starting mirror download for ${file} (attempt ${attempt}/${download_attempts})"
|
||||
fi
|
||||
|
||||
curl_auth=()
|
||||
if [ -n "${mirror_token}" ]; then
|
||||
curl_auth=(--header "Authorization: token ${mirror_token}")
|
||||
fi
|
||||
if curl -L --fail --show-error --continue-at - "${curl_auth[@]}" --output "${tmp_file}" "${url}"; then
|
||||
if verify_fixture_file "${tmp_file}" "${file}" "${expected_oid}" "${expected_size}"; then
|
||||
mv "${tmp_file}" "${file}"
|
||||
return 0
|
||||
fi
|
||||
echo "Mirror download failed integrity verification; retrying from the beginning: ${file}" >&2
|
||||
rm -f "${tmp_file}"
|
||||
else
|
||||
curl_status=$?
|
||||
partial_size=0
|
||||
if [ -f "${tmp_file}" ]; then
|
||||
partial_size="$(wc -c < "${tmp_file}" | tr -d '[:space:]')"
|
||||
fi
|
||||
|
||||
if [ "${partial_size}" = "${expected_size}" ]; then
|
||||
if verify_fixture_file "${tmp_file}" "${file}" "${expected_oid}" "${expected_size}"; then
|
||||
mv "${tmp_file}" "${file}"
|
||||
return 0
|
||||
fi
|
||||
rm -f "${tmp_file}"
|
||||
partial_size=0
|
||||
elif [ "${partial_size}" -gt "${expected_size}" ]; then
|
||||
echo "Discarding oversized partial fixture ${tmp_file}: ${partial_size} > ${expected_size}" >&2
|
||||
rm -f "${tmp_file}"
|
||||
partial_size=0
|
||||
elif [ "${curl_status}" -eq 33 ]; then
|
||||
echo "Mirror refused the resume request; retrying from the beginning: ${file}" >&2
|
||||
rm -f "${tmp_file}"
|
||||
partial_size=0
|
||||
fi
|
||||
|
||||
echo "Mirror download attempt ${attempt}/${download_attempts} failed with curl exit ${curl_status}; retained ${partial_size} bytes for resume: ${file}" >&2
|
||||
fi
|
||||
|
||||
if [ "${attempt}" -lt "${download_attempts}" ]; then
|
||||
sleep "${retry_delay}"
|
||||
fi
|
||||
done
|
||||
|
||||
rm -f "${tmp_file}"
|
||||
return 1
|
||||
}
|
||||
|
||||
# Files no mirror could serve, even after retrying every mirror. Only these fall
|
||||
# back to Git LFS, so a mirror that served the rest of the case still spares
|
||||
# GitHub the bandwidth for those files.
|
||||
mirror_failures=()
|
||||
|
||||
fetch_from_mirror() {
|
||||
mkdir -p "${fixture_dir}"
|
||||
for file in "${files[@]}"; do
|
||||
local name
|
||||
local url
|
||||
local base
|
||||
local fetched=0
|
||||
name="$(basename "${file}")"
|
||||
for base in "${mirror_bases[@]}"; do
|
||||
url="${base%/}/${name}"
|
||||
echo "Fetching trace fixture from mirror: ${url}"
|
||||
if fetch_file_from_mirror "${file}" "${url}"; then
|
||||
fetched=1
|
||||
break
|
||||
fi
|
||||
echo "Mirror did not serve ${name}; trying the next mirror" >&2
|
||||
done
|
||||
if [ "${fetched}" -ne 1 ]; then
|
||||
mirror_failures+=("${file}")
|
||||
fi
|
||||
done
|
||||
[ "${#mirror_failures[@]}" -eq 0 ]
|
||||
}
|
||||
|
||||
if fetch_from_mirror; then
|
||||
echo "Fetched trace fixture files for ${case_name} from mirror: ${include}"
|
||||
else
|
||||
fallback_include="$(IFS=,; echo "${mirror_failures[*]}")"
|
||||
echo "All mirrors failed for ${#mirror_failures[@]} of ${#files[@]} file(s) of ${case_name}; falling back to Git LFS: ${fallback_include}"
|
||||
git lfs install --local
|
||||
git lfs pull --include="${fallback_include}" --exclude=""
|
||||
fi
|
||||
|
||||
for file in "${files[@]}"; do
|
||||
metadata="$(get_lfs_metadata "${file}")"
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
verify_fixture_file "${file}" "${file}" "${expected_oid}" "${expected_size}"
|
||||
done
|
||||
@@ -0,0 +1,117 @@
|
||||
#!/usr/bin/env bash
|
||||
# Cache-side helper for trace fixtures.
|
||||
#
|
||||
# key <case> [fixture-dir] derive the actions/cache key and path list
|
||||
# verify <case> [fixture-dir] check restored fixtures against their pointers
|
||||
# reset <case> [fixture-dir] drop restored fixtures, leaving the pointers
|
||||
#
|
||||
# The cache key is content-addressed on the Git LFS pointer oids tracked at
|
||||
# HEAD, which are readable from a plain checkout without smudging. Fixture
|
||||
# content therefore maps 1:1 onto a key: unchanged content hits, changed
|
||||
# content is a new key and thus a miss, and the download path handles it. The
|
||||
# key deliberately carries no restore-keys prefix in the workflow - a fixture
|
||||
# that does not match the pointer exactly must never be restored.
|
||||
set -euo pipefail
|
||||
|
||||
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=trace-fixture-lib.sh
|
||||
. "${script_dir}/trace-fixture-lib.sh"
|
||||
|
||||
# Bump when the key derivation changes in a way that must invalidate old
|
||||
# entries; the content digest alone would not notice a format change.
|
||||
key_schema="v1"
|
||||
|
||||
if [ "$#" -lt 2 ] || [ "$#" -gt 3 ]; then
|
||||
echo "usage: $0 <key|verify|reset> <trace-case> [fixture-dir]" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
command_name="$1"
|
||||
case_name="$2"
|
||||
fixture_dir="${3:-tools/trace_replay/fixtures}"
|
||||
python_bin="${PYTHON:-python3}"
|
||||
|
||||
if ! command -v "${python_bin}" >/dev/null 2>&1 && command -v python >/dev/null 2>&1; then
|
||||
python_bin=python
|
||||
fi
|
||||
|
||||
mapfile -t files < <(trace_fixture_files "${case_name}" "${fixture_dir}" "${python_bin}")
|
||||
if [ "${#files[@]}" -eq 0 ]; then
|
||||
echo "no fixture files declared for trace case: ${case_name}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Writes "name=value" to $GITHUB_OUTPUT when running under Actions, and to
|
||||
# stdout otherwise so the script stays runnable (and testable) off-CI.
|
||||
emit_output() {
|
||||
local name="$1"
|
||||
local value="$2"
|
||||
if [ -n "${GITHUB_OUTPUT:-}" ]; then
|
||||
if [[ "${value}" == *$'\n'* ]]; then
|
||||
local delimiter="ghadelim_$(date +%s%N)_$$"
|
||||
{
|
||||
printf '%s<<%s\n' "${name}" "${delimiter}"
|
||||
printf '%s\n' "${value}"
|
||||
printf '%s\n' "${delimiter}"
|
||||
} >> "${GITHUB_OUTPUT}"
|
||||
else
|
||||
printf '%s=%s\n' "${name}" "${value}" >> "${GITHUB_OUTPUT}"
|
||||
fi
|
||||
fi
|
||||
printf '%s=%s\n' "${name}" "${value}"
|
||||
}
|
||||
|
||||
sanitize_case() {
|
||||
printf '%s' "$1" | sed 's/[^A-Za-z0-9._-]/_/g'
|
||||
}
|
||||
|
||||
case "${command_name}" in
|
||||
key)
|
||||
manifest=""
|
||||
for file in "${files[@]}"; do
|
||||
# A case whose fixtures are committed directly rather than through Git LFS
|
||||
# (OpenRA) has no pointer oid to key on, and nothing to download either.
|
||||
# Report it as uncacheable so the workflow skips the cache entirely.
|
||||
if ! metadata="$(get_lfs_metadata "${file}" 2>/dev/null)"; then
|
||||
echo "trace case ${case_name} is not stored in Git LFS; skipping fixture cache" >&2
|
||||
emit_output "cacheable" "false"
|
||||
emit_output "key" ""
|
||||
exit 0
|
||||
fi
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
manifest+="$(basename "${file}") ${expected_oid} ${expected_size}"$'\n'
|
||||
done
|
||||
|
||||
digest="$(printf '%s' "${manifest}" | sha256sum | awk '{ print substr($1, 1, 16) }')"
|
||||
safe_case="$(sanitize_case "${case_name}")"
|
||||
|
||||
emit_output "cacheable" "true"
|
||||
emit_output "key" "trace-fixture-${key_schema}-${safe_case}-${digest}"
|
||||
emit_output "paths" "$(printf '%s\n' "${files[@]}")"
|
||||
;;
|
||||
|
||||
verify)
|
||||
for file in "${files[@]}"; do
|
||||
metadata="$(get_lfs_metadata "${file}")"
|
||||
read -r expected_oid expected_size <<< "${metadata}"
|
||||
verify_fixture_file "${file}" "${file}" "${expected_oid}" "${expected_size}"
|
||||
done
|
||||
echo "Verified ${#files[@]} fixture file(s) for ${case_name} against the tracked Git LFS pointers."
|
||||
;;
|
||||
|
||||
reset)
|
||||
# Put the working tree back to the pointer files a fresh checkout would
|
||||
# have, so that a rejected cache entry falls through to exactly the same
|
||||
# download path a cache miss takes.
|
||||
for file in "${files[@]}"; do
|
||||
rm -f "${file}" "${file}.tmp"
|
||||
done
|
||||
git checkout -- "${files[@]}"
|
||||
echo "Reset ${#files[@]} fixture file(s) for ${case_name} to their tracked Git LFS pointers."
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "unknown command: ${command_name}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,73 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared helpers for trace-fixture handling: reading the in-tree Git LFS pointer
|
||||
# metadata and verifying a fixture file against it. Sourced by
|
||||
# fetch-trace-fixture-lfs.sh (verify after download) and by
|
||||
# trace-fixture-cache.sh (cache key derivation and verify after cache restore),
|
||||
# so both paths agree on what a valid fixture is.
|
||||
|
||||
# Reads the Git LFS pointer tracked at HEAD for a fixture path and prints
|
||||
# "<oid> <size>". Fails if the tracked blob is not a well-formed LFS pointer.
|
||||
get_lfs_metadata() {
|
||||
local file="$1"
|
||||
local pointer
|
||||
local expected_oid
|
||||
local expected_size
|
||||
|
||||
if ! pointer="$(git show "HEAD:${file}" 2>/dev/null)"; then
|
||||
echo "failed to read tracked fixture metadata: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
if ! grep -q '^version https://git-lfs.github.com/spec/v1$' <<< "${pointer}"; then
|
||||
echo "tracked fixture is not a Git LFS pointer: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
expected_oid="$(awk '$1 == "oid" && $2 ~ /^sha256:/ { sub(/^sha256:/, "", $2); print $2 }' <<< "${pointer}")"
|
||||
expected_size="$(awk '$1 == "size" { print $2 }' <<< "${pointer}")"
|
||||
if ! [[ "${expected_oid}" =~ ^[0-9a-f]{64}$ ]] || ! [[ "${expected_size}" =~ ^[0-9]+$ ]]; then
|
||||
echo "invalid Git LFS pointer metadata: ${file}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
printf '%s %s\n' "${expected_oid}" "${expected_size}"
|
||||
}
|
||||
|
||||
# Checks an on-disk fixture against the size and SHA-256 from its LFS pointer.
|
||||
verify_fixture_file() {
|
||||
local downloaded_file="$1"
|
||||
local display_name="$2"
|
||||
local expected_oid="$3"
|
||||
local expected_size="$4"
|
||||
local actual_oid
|
||||
local actual_size
|
||||
|
||||
if [ ! -f "${downloaded_file}" ]; then
|
||||
echo "fixture file is missing: ${display_name}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
actual_size="$(wc -c < "${downloaded_file}" | tr -d '[:space:]')"
|
||||
if [ "${actual_size}" != "${expected_size}" ]; then
|
||||
echo "fixture size mismatch for ${display_name}: expected ${expected_size}, got ${actual_size}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
actual_oid="$(sha256sum "${downloaded_file}" | awk '{ print $1 }')"
|
||||
if [ "${actual_oid}" != "${expected_oid}" ]; then
|
||||
echo "fixture SHA-256 mismatch for ${display_name}: expected ${expected_oid}, got ${actual_oid}" >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Prints the fixture file paths of a trace case, one per line. Strips CR so the
|
||||
# result is usable when python emits CRLF (Git Bash on Windows).
|
||||
trace_fixture_files() {
|
||||
local case_name="$1"
|
||||
local fixture_dir="$2"
|
||||
local python_bin="${3:-python3}"
|
||||
|
||||
"${python_bin}" tools/trace_replay/trace_cases.py \
|
||||
--format fixture-files \
|
||||
--case "${case_name}" \
|
||||
--fixture-root "${fixture_dir}" | tr -d '\r'
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
if [[ $# -ne 3 ]]; then
|
||||
echo "Usage: $0 <aapt2> <plugin-apk> <trace-apk>" >&2
|
||||
exit 64
|
||||
fi
|
||||
|
||||
aapt2=$1
|
||||
plugin_apk=$2
|
||||
trace_apk=$3
|
||||
|
||||
require() {
|
||||
local needle=$1
|
||||
local content=$2
|
||||
local description=$3
|
||||
if ! grep -Fq -- "$needle" <<<"$content"; then
|
||||
echo "::error::Missing ${description}: ${needle}" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
for apk in "$plugin_apk" "$trace_apk"; do
|
||||
[[ -f "$apk" ]] || { echo "::error::APK not found: $apk" >&2; exit 1; }
|
||||
done
|
||||
|
||||
plugin_manifest=$("$aapt2" dump xmltree --file AndroidManifest.xml "$plugin_apk")
|
||||
plugin_resources=$("$aapt2" dump resources "$plugin_apk")
|
||||
plugin_resource_text=$(tr -d '"' <<<"$plugin_resources")
|
||||
trace_manifest=$("$aapt2" dump xmltree --file AndroidManifest.xml "$trace_apk")
|
||||
plugin_contents=$(unzip -Z1 "$plugin_apk")
|
||||
|
||||
require 'top.mobilegl.plugin' "$plugin_manifest" 'plugin package name'
|
||||
require 'MobileGL' "$plugin_manifest" 'plugin label'
|
||||
require 'fclPlugin' "$plugin_manifest" 'legacy plugin marker'
|
||||
require 'fclPlugin_V2' "$plugin_manifest" 'V2 plugin marker'
|
||||
require 'LIBGL_ES=3:POJAV_RENDERER=opengles3:MOBILEGL_BACKEND_TYPE=DirectGLES' "$plugin_manifest" 'V1 DirectGLES fallback'
|
||||
require 'string/config' "$plugin_resources" 'V2 renderer configuration resource'
|
||||
require '{displayName:MobileGL,rendererId:opengles3' "$plugin_resource_text" 'V2 MobileGL entry and renderer ID'
|
||||
require 'rendererGLPath:**|libMobileGL.so' "$plugin_resource_text" 'V2 GL library path'
|
||||
require 'rendererEGLPath:**|libMobileGL.so' "$plugin_resource_text" 'V2 EGL library path'
|
||||
require 'key:LIBGL_ES,value:3' "$plugin_resource_text" 'V2 fixed LIBGL_ES variable'
|
||||
require 'key:MOBILEGL_BACKEND_TYPE' "$plugin_resource_text" 'V2 backend variable'
|
||||
require 'defaultValue:DirectGLES' "$plugin_resource_text" 'V2 DirectGLES default'
|
||||
require 'DirectVulkan' "$plugin_resource_text" 'V2 DirectVulkan option'
|
||||
require 'key:MOBILEGL_DISABLE_TIMERQUERY' "$plugin_resource_text" 'V2 timer-query toggle'
|
||||
require 'key:MOBILEGL_DISABLE_SUBGROUP' "$plugin_resource_text" 'V2 Vulkan subgroup toggle'
|
||||
require 'key:MOBILEGL_MAGMA_R11G11B10F_FALLBACK' "$plugin_resource_text" 'V2 Magma format fallback toggle'
|
||||
require 'key:MOBILEGL_MAGMA_FRAMESINFLIGHT' "$plugin_resource_text" 'V2 Magma frames-in-flight setting'
|
||||
require 'key:MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER' "$plugin_resource_text" 'V2 sampler workaround toggle'
|
||||
require 'key:MOBILEGL_COHERENT_AS_FLUSH' "$plugin_resource_text" 'V2 coherent-as-flush toggle'
|
||||
require 'key:MOBILEGL_USE_ANGLE' "$plugin_resource_text" 'V2 ANGLE toggle'
|
||||
|
||||
if [[ $(grep -Fc 'fclPlugin_V2' <<<"$plugin_manifest") -ne 1 ]]; then
|
||||
echo '::error::Plugin manifest must expose exactly one V2 descriptor' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! grep -Eq '^lib/[^/]+/libMobileGL\.so$' <<<"$plugin_contents"; then
|
||||
echo '::error::Plugin APK does not contain libMobileGL.so' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
require 'top.mobilegl.plugin.trace' "$trace_manifest" 'trace package name'
|
||||
require 'top.mobilegl.plugin.TRACE_REPLAY' "$trace_manifest" 'trace replay action'
|
||||
if grep -Fq 'fclPlugin' <<<"$trace_manifest"; then
|
||||
echo '::error::Trace APK must not advertise renderer-plugin metadata' >&2
|
||||
exit 1
|
||||
fi
|
||||
if grep -Fq 'android.intent.action.MAIN' <<<"$trace_manifest"; then
|
||||
echo '::error::Trace APK must not expose a launcher activity' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo 'Validated unified MobileGL plugin APK and isolated trace APK.'
|
||||
@@ -0,0 +1,663 @@
|
||||
name: MobileGL APK
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- dev
|
||||
- Feat/Backend-Direct-GLES
|
||||
- Feat/Backend-Direct-Vulkan
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
env:
|
||||
CCACHE_BASEDIR: ${{ github.workspace }}
|
||||
CCACHE_COMPRESS: "true"
|
||||
CCACHE_DIR: ${{ github.workspace }}/.ccache
|
||||
CCACHE_MAXSIZE: 4G
|
||||
CCACHE_NOHASHDIR: "true"
|
||||
MOBILEGL_CMAKE_COMPILER_LAUNCHER: ccache
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
submodules: recursive
|
||||
|
||||
- name: Set artifact metadata
|
||||
run: |
|
||||
echo "date_today=$(date +'%Y-%m-%d')" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Set up JDK
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: zulu
|
||||
java-version: '17'
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
gradle-version: 8.10.2
|
||||
|
||||
- name: Restore ccache
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-apk-${{ github.job }}-ccache-v1
|
||||
restore-keys: |
|
||||
${{ runner.os }}-apk-${{ github.job }}-ccache-
|
||||
|
||||
- name: Install ccache
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y ccache
|
||||
ccache --version
|
||||
|
||||
- name: Setup Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
with:
|
||||
accept-android-sdk-licenses: false
|
||||
|
||||
- name: Accept Android SDK licenses
|
||||
run: yes | sdkmanager --licenses >/dev/null
|
||||
|
||||
- name: Install Android NDK
|
||||
run: |
|
||||
sdkmanager "ndk;27.3.13750724"
|
||||
echo "ndk.dir=$ANDROID_HOME/ndk/27.3.13750724" >> android-plugin/local.properties
|
||||
|
||||
- name: Update glslang external sources
|
||||
working-directory: 3rdparty/glslang
|
||||
run: python update_glslang_sources.py
|
||||
|
||||
- name: Build plugin APK
|
||||
run: gradle --no-daemon -p android-plugin :app:assemblePluginRelease -Pmobilegl.apkSuffix="${GITHUB_SHA}" -Pmobilegl.logLevel=MOBILEGL_LOG_LEVEL_INFO --parallel --max-workers "$(nproc)"
|
||||
env:
|
||||
SIGNING_STORE_PASSWORD: ${{ secrets.SIGNING_STORE_PASSWORD }}
|
||||
SIGNING_KEY_ALIAS: ${{ secrets.SIGNING_KEY_ALIAS }}
|
||||
SIGNING_KEY_PASSWORD: ${{ secrets.SIGNING_KEY_PASSWORD }}
|
||||
|
||||
- name: Download ANGLE x86_64 libraries
|
||||
run: |
|
||||
angle_dir="android-plugin/app/src/trace/jniLibs/x86_64"
|
||||
rm -rf "${angle_dir}"
|
||||
mkdir -p "${angle_dir}"
|
||||
|
||||
package_angle_variant() {
|
||||
variant="$1"
|
||||
commit="$2"
|
||||
egl_sha="$3"
|
||||
gles_sha="$4"
|
||||
source_dir="${RUNNER_TEMP}/mobilegl-angle-${variant}"
|
||||
base="https://raw.githubusercontent.com/FCL-Team/FoldCraftLauncher/${commit}/FCLauncher/src/main/jniLibs/x86_64"
|
||||
mkdir -p "${source_dir}"
|
||||
curl -L --fail --retry 3 -o "${source_dir}/libEGL_angle.so" "${base}/libEGL_angle.so"
|
||||
curl -L --fail --retry 3 -o "${source_dir}/libGLESv2_angle.so" "${base}/libGLESv2_angle.so"
|
||||
echo "${egl_sha} ${source_dir}/libEGL_angle.so" | sha256sum -c -
|
||||
echo "${gles_sha} ${source_dir}/libGLESv2_angle.so" | sha256sum -c -
|
||||
for library in libEGL_angle libGLESv2_angle; do
|
||||
filename="${library}_${variant}.so"
|
||||
cp "${source_dir}/${library}.so" "${angle_dir}/${filename}"
|
||||
done
|
||||
}
|
||||
|
||||
package_angle_variant \
|
||||
ec889e6ea831 \
|
||||
f2a3d510dffd8f6540a52e1a7d0c5787d151075b \
|
||||
c41828768d089899fa058ec0bee711a91be88347f29bdb935223da6be1149c40 \
|
||||
e4f820d99f94365c66df868c7740fef142fe5c0cd7c941790a9e30638857ca4d
|
||||
package_angle_variant \
|
||||
90a62123d794 \
|
||||
bdcc96ac11c79001018ae4375eb73cb54a9f682f \
|
||||
d0f4298ccc770cc801fc52e21733521646161e8a4adb3bd0052d9a1b57ee0ca8 \
|
||||
66fdc867e552192d553d59095ea2e3cef4829de65c356f1fd826027b1905972e
|
||||
|
||||
- name: Build retrace APK
|
||||
run: gradle --no-daemon -p android-plugin :app:assembleTraceRelease -Pmobilegl.apkSuffix="${GITHUB_SHA}" -Pmobilegl.abis=all -Pmobilegl.debuggableRelease=true -Pmobilegl.logLevel=MOBILEGL_LOG_LEVEL_INFO --parallel --max-workers "$(nproc)"
|
||||
env:
|
||||
SIGNING_STORE_PASSWORD: ${{ secrets.SIGNING_STORE_PASSWORD }}
|
||||
SIGNING_KEY_ALIAS: ${{ secrets.SIGNING_KEY_ALIAS }}
|
||||
SIGNING_KEY_PASSWORD: ${{ secrets.SIGNING_KEY_PASSWORD }}
|
||||
|
||||
- name: Show ccache stats
|
||||
if: always()
|
||||
run: ccache --show-stats
|
||||
|
||||
# Rewrite one rolling entry per job on the default branch. The upload stays
|
||||
# cumulative - it carries every object restored at the top of this run plus
|
||||
# the few TUs that actually changed - but Actions cache keys are immutable,
|
||||
# so the superseded blob has to be released before the same key can be
|
||||
# re-uploaded. Running after the build means a failed build leaves the
|
||||
# existing entry untouched. The other trigger branches restore this entry
|
||||
# rather than each writing a ~4 GB one of their own.
|
||||
- name: Release superseded ccache entry
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CACHE_KEY: ${{ runner.os }}-apk-${{ github.job }}-ccache-v1
|
||||
run: gh cache delete "${CACHE_KEY}" || true
|
||||
|
||||
- name: Save ccache
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
continue-on-error: true
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-apk-${{ github.job }}-ccache-v1
|
||||
|
||||
- name: Verify APK metadata and packaging
|
||||
run: |
|
||||
AAPT2="$(find "$ANDROID_HOME/build-tools" -name aapt2 -type f | sort -V | tail -n 1)"
|
||||
plugin_apk="android-plugin/app/build/outputs/apk/plugin/release/MobileGL-plugin-release-${GITHUB_SHA}.apk"
|
||||
trace_apk="android-plugin/app/build/outputs/apk/trace/release/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk"
|
||||
test -f "${plugin_apk}"
|
||||
test -f "${trace_apk}"
|
||||
bash .github/scripts/validate-plugin-apks.sh "$AAPT2" "$plugin_apk" "$trace_apk"
|
||||
|
||||
- name: Verify signed APKs
|
||||
run: |
|
||||
APKSIGNER="$(find "$ANDROID_HOME/build-tools" -name apksigner -type f | sort -V | tail -n 1)"
|
||||
mapfile -t APKS < <(printf '%s\n' \
|
||||
"android-plugin/app/build/outputs/apk/plugin/release/MobileGL-plugin-release-${GITHUB_SHA}.apk" \
|
||||
"android-plugin/app/build/outputs/apk/trace/release/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk")
|
||||
for APK in "${APKS[@]}"; do
|
||||
if [[ ! -f "$APK" ]]; then
|
||||
echo "::error::Expected release APK was not produced: $APK"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
for APK in "${APKS[@]}"; do
|
||||
if [[ "$APK" == *-unsigned.apk ]]; then
|
||||
echo "::error::Unsigned release APK produced: $APK"
|
||||
exit 1
|
||||
fi
|
||||
"$APKSIGNER" verify --verbose "$APK"
|
||||
done
|
||||
|
||||
- name: Upload plugin APK
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: MobileGL-plugin-${{ env.date_today }}-${{ github.sha }}
|
||||
path: android-plugin/app/build/outputs/apk/plugin/release/MobileGL-plugin-release-${{ github.sha }}.apk
|
||||
archive: false
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Upload retrace APK
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: MobileGL-retrace-apk-${{ env.date_today }}-${{ github.sha }}
|
||||
path: android-plugin/app/build/outputs/apk/trace/release/MobileGL-plugin-trace-release-${{ github.sha }}.apk
|
||||
archive: false
|
||||
if-no-files-found: error
|
||||
|
||||
trace-cases:
|
||||
name: trace case matrix
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
outputs:
|
||||
android: ${{ steps.trace-cases.outputs.android }}
|
||||
names: ${{ steps.trace-cases.outputs.names }}
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Load trace cases
|
||||
id: trace-cases
|
||||
run: |
|
||||
echo "android=$(python3 tools/trace_replay/trace_cases.py --ci --format github-apk)" >> "$GITHUB_OUTPUT"
|
||||
echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
trace-fixtures:
|
||||
name: trace fixture (${{ matrix.case }})
|
||||
runs-on: ubuntu-latest
|
||||
needs: trace-cases
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.names) }}
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Derive trace fixture cache key
|
||||
id: fixture-key
|
||||
run: bash .github/scripts/trace-fixture-cache.sh key '${{ matrix.case }}'
|
||||
|
||||
- name: Restore trace fixture cache
|
||||
id: fixture-cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true'
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Verify restored trace fixture
|
||||
id: fixture-verify
|
||||
if: steps.fixture-cache.outputs.cache-hit == 'true'
|
||||
run: |
|
||||
if bash .github/scripts/trace-fixture-cache.sh verify '${{ matrix.case }}'; then
|
||||
echo "ok=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "ok=false" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::Cached fixture for ${{ matrix.case }} failed verification; falling back to the download path"
|
||||
bash .github/scripts/trace-fixture-cache.sh reset '${{ matrix.case }}'
|
||||
fi
|
||||
|
||||
- name: Fetch trace fixture
|
||||
if: steps.fixture-verify.outputs.ok != 'true'
|
||||
run: bash .github/scripts/fetch-trace-fixture-lfs.sh '${{ matrix.case }}'
|
||||
|
||||
- name: Save trace fixture cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true' && steps.fixture-cache.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Stage trace fixture
|
||||
run: |
|
||||
safe_case="$(printf '%s' '${{ matrix.case }}' | sed 's/[^A-Za-z0-9._-]/_/g')"
|
||||
stage_dir="trace-fixtures/${safe_case}"
|
||||
mkdir -p "${stage_dir}"
|
||||
python3 tools/trace_replay/trace_cases.py --format fixture-files --case '${{ matrix.case }}' |
|
||||
while IFS= read -r file; do
|
||||
cp "${file}" "${stage_dir}/"
|
||||
done
|
||||
|
||||
- name: Upload trace fixture
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: MobileGL-trace-fixture-${{ matrix.case }}
|
||||
path: trace-fixtures/**
|
||||
if-no-files-found: error
|
||||
|
||||
android-avd:
|
||||
name: android avd image
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
AVD_NAME: mobilegl-ci
|
||||
ANDROID_AVD_HOME: ${{ github.workspace }}/.android/avd
|
||||
ANDROID_HOME: ${{ github.workspace }}/.android/sdk
|
||||
ANDROID_SDK_ROOT: ${{ github.workspace }}/.android/sdk
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Setup Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
with:
|
||||
accept-android-sdk-licenses: false
|
||||
|
||||
- name: Accept Android SDK licenses
|
||||
run: yes | sdkmanager --licenses >/dev/null
|
||||
|
||||
- name: Restore Android AVD cache
|
||||
id: android-avd-cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: |
|
||||
${{ env.ANDROID_AVD_HOME }}
|
||||
${{ env.ANDROID_SDK_ROOT }}/emulator
|
||||
${{ env.ANDROID_SDK_ROOT }}/platform-tools
|
||||
${{ env.ANDROID_SDK_ROOT }}/platforms/android-35
|
||||
${{ env.ANDROID_SDK_ROOT }}/system-images/android-35/google_apis/x86_64
|
||||
key: ${{ runner.os }}-mobilegl-avd-api35-google_apis-x86_64-pixel_6-v2-${{ hashFiles('android-plugin/run-avd-ci.sh') }}
|
||||
|
||||
- name: Create AVD
|
||||
if: steps.android-avd-cache.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
sh android-plugin/run-avd-ci.sh create \
|
||||
--api-level 35 \
|
||||
--target google_apis \
|
||||
--arch x86_64 \
|
||||
--profile pixel_6 \
|
||||
--avd-name "${AVD_NAME}"
|
||||
|
||||
retrace:
|
||||
name: retrace (${{ matrix.backend.name }}, ${{ matrix.case.name }})
|
||||
runs-on: ubuntu-latest
|
||||
needs:
|
||||
- build
|
||||
- android-avd
|
||||
- trace-cases
|
||||
- trace-fixtures
|
||||
if: ${{ always() && needs.build.result == 'success' && needs.android-avd.result == 'success' && needs.trace-cases.result == 'success' }}
|
||||
timeout-minutes: 75
|
||||
env:
|
||||
AVD_NAME: mobilegl-ci
|
||||
ANDROID_AVD_HOME: ${{ github.workspace }}/.android/avd
|
||||
ANDROID_HOME: ${{ github.workspace }}/.android/sdk
|
||||
ANDROID_SDK_ROOT: ${{ github.workspace }}/.android/sdk
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
backend:
|
||||
- name: DirectGLES
|
||||
gpu: software
|
||||
- name: DirectVulkan
|
||||
gpu: lavapipe
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.android) }}
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@v1.0
|
||||
with:
|
||||
swap-size-gb: 8
|
||||
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Download trace fixture
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: MobileGL-trace-fixture-${{ matrix.case.name }}
|
||||
path: trace-fixture-download
|
||||
|
||||
- name: Install trace fixture
|
||||
run: |
|
||||
mkdir -p tools/trace_replay/fixtures
|
||||
find trace-fixture-download -type f -exec cp {} tools/trace_replay/fixtures/ \;
|
||||
|
||||
- name: Set artifact metadata
|
||||
run: |
|
||||
echo "date_today=$(date +'%Y-%m-%d')" >> "$GITHUB_ENV"
|
||||
echo "EMULATOR_LOG=${RUNNER_TEMP}/mobilegl-emulator.log" >> "$GITHUB_ENV"
|
||||
echo "EMULATOR_PID_FILE=${RUNNER_TEMP}/mobilegl-emulator.pid" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Setup Android SDK
|
||||
uses: android-actions/setup-android@v4
|
||||
with:
|
||||
accept-android-sdk-licenses: false
|
||||
|
||||
- name: Accept Android SDK licenses
|
||||
run: yes | sdkmanager --licenses >/dev/null
|
||||
|
||||
- name: Restore Android AVD cache
|
||||
id: android-avd-cache
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: |
|
||||
${{ env.ANDROID_AVD_HOME }}
|
||||
${{ env.ANDROID_SDK_ROOT }}/emulator
|
||||
${{ env.ANDROID_SDK_ROOT }}/platform-tools
|
||||
${{ env.ANDROID_SDK_ROOT }}/platforms/android-35
|
||||
${{ env.ANDROID_SDK_ROOT }}/system-images/android-35/google_apis/x86_64
|
||||
key: ${{ runner.os }}-mobilegl-avd-api35-google_apis-x86_64-pixel_6-v2-${{ hashFiles('android-plugin/run-avd-ci.sh') }}
|
||||
|
||||
- name: Download retrace APK
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: MobileGL-plugin-trace-release-${{ github.sha }}.apk
|
||||
path: android-retrace-apks
|
||||
|
||||
- name: Enable KVM
|
||||
run: |
|
||||
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules
|
||||
sudo udevadm control --reload-rules
|
||||
sudo udevadm trigger --name-match=kvm
|
||||
|
||||
- name: Create AVD
|
||||
if: steps.android-avd-cache.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
sh android-plugin/run-avd-ci.sh create \
|
||||
--api-level 35 \
|
||||
--target google_apis \
|
||||
--arch x86_64 \
|
||||
--profile pixel_6 \
|
||||
--avd-name "${AVD_NAME}"
|
||||
|
||||
- name: Launch Emulator
|
||||
run: |
|
||||
sh android-plugin/run-avd-ci.sh start \
|
||||
--avd-name "${AVD_NAME}" \
|
||||
--gpu "${{ matrix.backend.gpu }}" \
|
||||
--emulator-log "${EMULATOR_LOG}" \
|
||||
--pid-file "${EMULATOR_PID_FILE}" \
|
||||
--boot-timeout 300
|
||||
|
||||
- name: Retrace and validate
|
||||
env:
|
||||
MOBILEGL_USE_ANGLE: ${{ matrix.backend.name == 'DirectGLES' && '1' || '0' }}
|
||||
MOBILEGL_TRACE_ANGLE_VARIANT: ${{ matrix.case.name == 'minecraft-1.21.4-fabric-iris-bliss-in-world' && '90a62123d794' || 'ec889e6ea831' }}
|
||||
MOBILEGL_MAGMA_R11G11B10F_FALLBACK: ${{ matrix.backend.name == 'DirectVulkan' && '1' || '0' }}
|
||||
run: |
|
||||
apk_file="android-retrace-apks/MobileGL-plugin-trace-release-${GITHUB_SHA}.apk"
|
||||
test -f "${apk_file}"
|
||||
extra_retrace_args=()
|
||||
# Bliss needs the newer signed ANGLE variant plus sampler mipmap
|
||||
# min-filter downgrading on ANGLE llvmpipe.
|
||||
if [ "${{ matrix.backend.name }}" = "DirectGLES" ] && [ "${{ matrix.case.name }}" = "minecraft-1.21.4-fabric-iris-bliss-in-world" ]; then
|
||||
extra_retrace_args+=(--avoid-angle-llvmpipe-sampler-mipmap-min-filter)
|
||||
fi
|
||||
if [ "${{ matrix.backend.name }}" = "DirectGLES" ] && [ "${{ matrix.case.avoid_angle_llvmpipe_explicit_lod_bias || false }}" = "true" ]; then
|
||||
extra_retrace_args+=(--avoid-angle-llvmpipe-explicit-lod-bias)
|
||||
fi
|
||||
if [ "${{ matrix.case.coherent_as_flush || false }}" = "true" ]; then
|
||||
extra_retrace_args+=(--coherent-as-flush)
|
||||
fi
|
||||
|
||||
run_retrace() {
|
||||
timeout "$(( ${{ matrix.case.timeout_seconds }} + 300 ))" sh android-plugin/trace-replay-ci.sh \
|
||||
--apk-file "${apk_file}" \
|
||||
--package top.mobilegl.plugin.trace \
|
||||
--backend "${{ matrix.backend.name }}" \
|
||||
--result-root android-retrace-result \
|
||||
--fixture-root android-retrace-fixture \
|
||||
--case "${{ matrix.case.name }}" \
|
||||
--trace-archive "${{ matrix.case.trace_archive }}" \
|
||||
--trace-file "${{ matrix.case.trace_file }}" \
|
||||
--golden "${{ matrix.case.golden }}" \
|
||||
--alternate-golden "${{ matrix.case.alternate_golden || '' }}" \
|
||||
--target-call "${{ matrix.case.target_call }}" \
|
||||
--width "${{ matrix.case.width }}" \
|
||||
--height "${{ matrix.case.height }}" \
|
||||
--ssim-threshold "${{ matrix.case.ssim_threshold || '0.99' }}" \
|
||||
--crop-x "${{ matrix.case.crop_x }}" \
|
||||
--crop-y "${{ matrix.case.crop_y }}" \
|
||||
--crop-width "${{ matrix.case.crop_width }}" \
|
||||
--crop-height "${{ matrix.case.crop_height }}" \
|
||||
--timeout-seconds "${{ matrix.case.timeout_seconds }}" \
|
||||
"${extra_retrace_args[@]}"
|
||||
}
|
||||
|
||||
retrace_status=0
|
||||
run_retrace || retrace_status=$?
|
||||
if [ "${retrace_status}" -eq 75 ]; then
|
||||
echo "::warning::Android emulator infrastructure failed; restarting it and retrying this retrace once."
|
||||
# Surface-lost is retried rather than failed, so it would otherwise
|
||||
# be invisible. Report it per job - a healthy run prints nothing and
|
||||
# a rate spike shows up as a row per affected case.
|
||||
reason_file="android-retrace-result/infrastructure-failure-reason.txt"
|
||||
surface_lost_retries=0
|
||||
if [ -f "${reason_file}" ]; then
|
||||
surface_lost_retries="$(grep -c 'angle-surface-lost' "${reason_file}" || true)"
|
||||
fi
|
||||
if [ "${surface_lost_retries}" -gt 0 ]; then
|
||||
echo "surface-lost retries: ${surface_lost_retries} (${{ matrix.backend.name }}, ${{ matrix.case.name }})" \
|
||||
>> "${GITHUB_STEP_SUMMARY}"
|
||||
fi
|
||||
# The restart truncates EMULATOR_LOG, and the attempt that lost the
|
||||
# emulator is the one worth reading - the retry usually only shows
|
||||
# the wreckage. Keep the first attempt's log before it is clobbered.
|
||||
if [ -f "${EMULATOR_LOG}" ]; then
|
||||
cp "${EMULATOR_LOG}" "${EMULATOR_LOG}.first-attempt" || true
|
||||
fi
|
||||
sh android-plugin/run-avd-ci.sh stop \
|
||||
--avd-name "${AVD_NAME}" \
|
||||
--emulator-log "${EMULATOR_LOG}" \
|
||||
--pid-file "${EMULATOR_PID_FILE}"
|
||||
adb kill-server || true
|
||||
sleep 2
|
||||
sh android-plugin/run-avd-ci.sh start \
|
||||
--avd-name "${AVD_NAME}" \
|
||||
--gpu "${{ matrix.backend.gpu }}" \
|
||||
--emulator-log "${EMULATOR_LOG}" \
|
||||
--pid-file "${EMULATOR_PID_FILE}" \
|
||||
--boot-timeout 300
|
||||
run_retrace
|
||||
elif [ "${retrace_status}" -ne 0 ]; then
|
||||
exit "${retrace_status}"
|
||||
fi
|
||||
|
||||
- name: Collect retrace summary inputs
|
||||
if: always()
|
||||
run: |
|
||||
safe_case="$(printf '%s' '${{ matrix.case.name }}' | sed 's/[^A-Za-z0-9._-]/_/g')"
|
||||
result_dir="android-retrace-result/${safe_case}-${{ matrix.backend.name }}"
|
||||
mkdir -p "${result_dir}"
|
||||
if [ -s "${{ matrix.case.golden }}" ]; then
|
||||
cp "${{ matrix.case.golden }}" "${result_dir}/${safe_case}-${{ matrix.backend.name }}-golden.png"
|
||||
fi
|
||||
if [ -n "${{ matrix.case.alternate_golden || '' }}" ] && [ -s "${{ matrix.case.alternate_golden || '' }}" ]; then
|
||||
cp "${{ matrix.case.alternate_golden || '' }}" "${result_dir}/${safe_case}-${{ matrix.backend.name }}-alternate-golden.png"
|
||||
fi
|
||||
|
||||
- name: Collect emulator diagnostics
|
||||
if: always()
|
||||
run: |
|
||||
mkdir -p android-retrace-result/diagnostics
|
||||
adb devices -l > android-retrace-result/diagnostics/adb-devices.txt || true
|
||||
timeout 30 adb logcat -d -t 1000 > android-retrace-result/diagnostics/logcat.txt || true
|
||||
if [ -f "${EMULATOR_LOG}" ]; then
|
||||
cp "${EMULATOR_LOG}" android-retrace-result/diagnostics/emulator.log
|
||||
fi
|
||||
if [ -f "${EMULATOR_LOG}.first-attempt" ]; then
|
||||
cp "${EMULATOR_LOG}.first-attempt" android-retrace-result/diagnostics/emulator-first-attempt.log
|
||||
fi
|
||||
# A vanished emulator looks identical whether the host OOM killer took
|
||||
# qemu or the renderer faulted. These two say which.
|
||||
free -h > android-retrace-result/diagnostics/host-memory.txt 2>&1 || true
|
||||
sudo dmesg -T 2>/dev/null | tail -300 > android-retrace-result/diagnostics/host-dmesg.txt || true
|
||||
|
||||
- name: Stop Emulator
|
||||
if: always()
|
||||
run: |
|
||||
sh android-plugin/run-avd-ci.sh stop \
|
||||
--avd-name "${AVD_NAME}" \
|
||||
--emulator-log "${EMULATOR_LOG}" \
|
||||
--pid-file "${EMULATOR_PID_FILE}"
|
||||
|
||||
- name: Upload Android retrace result
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: MobileGL-android-retrace-result-${{ env.date_today }}-${{ github.sha }}-${{ matrix.backend.name }}-${{ matrix.case.name }}
|
||||
path: android-retrace-result/**
|
||||
if-no-files-found: warn
|
||||
|
||||
retrace-summary:
|
||||
name: retrace summary
|
||||
runs-on: ubuntu-latest
|
||||
needs: retrace
|
||||
if: always()
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set artifact metadata
|
||||
run: |
|
||||
echo "date_today=$(date +'%Y-%m-%d')" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
|
||||
- name: Download Android retrace results
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
pattern: MobileGL-android-retrace-result-*
|
||||
path: retrace-artifacts
|
||||
|
||||
- name: Render retrace summary
|
||||
run: |
|
||||
node tools/trace_replay/render_retrace_summary.mjs \
|
||||
--input retrace-artifacts \
|
||||
--output-dir android-retrace-summary \
|
||||
--title "MobileGL Android retrace overview" \
|
||||
--group-label "Android Emulator" \
|
||||
--html mobilegl-android-retrace-overview.html
|
||||
|
||||
- name: Upload Android retrace summary
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: android-retrace-summary/mobilegl-android-retrace-overview.html
|
||||
archive: false
|
||||
if-no-files-found: error
|
||||
|
||||
remove-artifact-clutter:
|
||||
name: remove artifact clutter
|
||||
runs-on: ubuntu-latest
|
||||
needs: retrace-summary
|
||||
if: always()
|
||||
permissions:
|
||||
actions: write
|
||||
steps:
|
||||
- name: Delete intermediate Android retrace artifacts
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
declare -A failed_cases=()
|
||||
while IFS= read -r job_name; do
|
||||
case_name="${job_name#retrace (*, }"
|
||||
case_name="${case_name%)}"
|
||||
failed_cases["${case_name}"]=1
|
||||
done < <(
|
||||
gh api --paginate "repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/jobs?per_page=100" \
|
||||
--jq '.jobs[] | select(.name | startswith("retrace (")) | select(.conclusion == "failure" or .conclusion == "cancelled" or .conclusion == "timed_out" or .conclusion == "action_required") | .name'
|
||||
)
|
||||
|
||||
if ((${#failed_cases[@]})); then
|
||||
echo "Retaining fixtures and results for failed retrace case(s):"
|
||||
printf ' %s\n' "${!failed_cases[@]}"
|
||||
else
|
||||
echo "All retrace jobs succeeded; nothing needs to be retained."
|
||||
fi
|
||||
|
||||
deleted=0
|
||||
retained=0
|
||||
while IFS=$'\t' read -r artifact_id artifact_name; do
|
||||
keep=0
|
||||
if [[ "${artifact_name}" == MobileGL-trace-fixture-* ]]; then
|
||||
case_name="${artifact_name#MobileGL-trace-fixture-}"
|
||||
if [[ -v "failed_cases[${case_name}]" ]]; then
|
||||
keep=1
|
||||
fi
|
||||
elif [[ "${artifact_name}" == MobileGL-android-retrace-result-* ]]; then
|
||||
# The result artifact carries mobilegl.log, retrace.log, logcat,
|
||||
# the emulator log and the actual/diff images - the only record of
|
||||
# why a retrace failed. Its name ends in -<backend>-<case>, so a
|
||||
# suffix match on the case name keeps both backends' results for a
|
||||
# case that failed on either of them, which is what a comparison
|
||||
# needs. The match is anchored at the end, so a case name that is a
|
||||
# prefix of a longer one does not retain the longer one's results.
|
||||
for case_name in "${!failed_cases[@]}"; do
|
||||
if [[ "${artifact_name}" == *-"${case_name}" ]]; then
|
||||
keep=1
|
||||
break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
if ((keep)); then
|
||||
echo "Retaining ${artifact_name} (${artifact_id}) for failed retrace."
|
||||
((retained += 1))
|
||||
continue
|
||||
fi
|
||||
|
||||
echo "Deleting ${artifact_name} (${artifact_id})"
|
||||
gh api --method DELETE "repos/${GITHUB_REPOSITORY}/actions/artifacts/${artifact_id}"
|
||||
((deleted += 1))
|
||||
done < <(
|
||||
gh api --paginate "repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/artifacts?per_page=100" \
|
||||
--jq '.artifacts[] | select(.name | startswith("MobileGL-trace-fixture-") or startswith("MobileGL-android-retrace-result-") or startswith("trace-fixture-") or startswith("retrace-result-")) | [.id, .name] | @tsv'
|
||||
)
|
||||
|
||||
echo "Deleted ${deleted} intermediate Android artifact(s); retained ${retained} failed-retrace fixture(s)."
|
||||
@@ -1,64 +0,0 @@
|
||||
name: Benchmark
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- dev
|
||||
- Feat/Backend-Direct-GLES
|
||||
- Feat/Backend-Direct-Vulkan
|
||||
|
||||
jobs:
|
||||
benchmark:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
# BENCH_ROOT: ${{github.workspace}}/MobileGL/MG_Benchmark
|
||||
BENCH_ROOT: ${{github.workspace}}
|
||||
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
with:
|
||||
swap-size-gb: 32
|
||||
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
submodules: true
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@latest
|
||||
|
||||
- name: Prepare Vulkan SDK
|
||||
uses: humbletim/setup-vulkan-sdk@v1.2.1
|
||||
with:
|
||||
vulkan-query-version: 1.4.304.1
|
||||
vulkan-components: Vulkan-Headers, Vulkan-Loader
|
||||
vulkan-use-cache: true
|
||||
|
||||
- name: Update glslang external sources
|
||||
working-directory: ${{env.BENCH_ROOT}}/3rdparty/glslang
|
||||
run: python update_glslang_sources.py
|
||||
|
||||
- name: Install clang-20
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y clang-20 clang++-20 lld-20 libc++-20-dev libc++abi-20-dev libvulkan-dev
|
||||
|
||||
- name: Show installed toolchain
|
||||
run: |
|
||||
clang-20 --version
|
||||
clang++-20 --version
|
||||
ld.lld-20 --version || ld.lld --version || true
|
||||
dpkg -l 'libc++*' || true
|
||||
|
||||
- name: Configure CMake
|
||||
working-directory: ${{env.BENCH_ROOT}}
|
||||
run: cmake -S . -B build-bench -G Ninja -DCMAKE_BUILD_TYPE=Release -DCMAKE_C_COMPILER=clang-20 -DCMAKE_CXX_COMPILER=clang++-20 -DBENCHMARK_DOWNLOAD_DEPENDENCIES=ON -DBENCHMARK_ENABLE_TESTING=OFF -DMOBILEGL_BUILD_TEST=OFF -DMOBILEGL_BUILD_BENCHMARK=ON -DCMAKE_POLICY_VERSION_MINIMUM=3.5
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{env.BENCH_ROOT}}/build-bench
|
||||
run: cmake --build .
|
||||
|
||||
- name: Benchmark
|
||||
working-directory: ${{env.BENCH_ROOT}}/build-bench/MobileGL/MG_Benchmark
|
||||
run: ctest -V -C Release
|
||||
+712
-24
@@ -1,4 +1,4 @@
|
||||
name: Test
|
||||
name: Test
|
||||
|
||||
on:
|
||||
push:
|
||||
@@ -6,27 +6,43 @@ on:
|
||||
- dev
|
||||
- Feat/Backend-Direct-GLES
|
||||
- Feat/Backend-Direct-Vulkan
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
test:
|
||||
build-linux:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
env:
|
||||
# TEST_ROOT: ${{github.workspace}}/MobileGL/MG_Test
|
||||
TEST_ROOT: ${{github.workspace}}
|
||||
BUILD_DIR: build-linux
|
||||
CCACHE_BASEDIR: ${{ github.workspace }}
|
||||
CCACHE_COMPRESS: "true"
|
||||
CCACHE_DIR: ${{ github.workspace }}/.ccache
|
||||
CCACHE_MAXSIZE: 4G
|
||||
CCACHE_NOHASHDIR: "true"
|
||||
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@master
|
||||
uses: pierotofy/set-swap-space@v1.0
|
||||
with:
|
||||
swap-size-gb: 32
|
||||
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
submodules: true
|
||||
submodules: recursive
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@latest
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Restore ccache
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
restore-keys: |
|
||||
${{ runner.os }}-test-${{ github.job }}-ccache-
|
||||
|
||||
- name: Prepare Vulkan SDK
|
||||
uses: humbletim/setup-vulkan-sdk@v1.2.1
|
||||
@@ -36,39 +52,711 @@ jobs:
|
||||
vulkan-use-cache: true
|
||||
|
||||
- name: Update glslang external sources
|
||||
working-directory: ${{env.TEST_ROOT}}/3rdparty/glslang
|
||||
working-directory: 3rdparty/glslang
|
||||
run: python update_glslang_sources.py
|
||||
|
||||
- name: Install clang-20
|
||||
- name: Install build dependencies
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y clang-20 clang++-20 lld-20 libc++-20-dev libc++abi-20-dev libvulkan-dev
|
||||
sudo apt-get install -y ccache clang-20 clang++-20 lld-20 libc++-20-dev libc++abi-20-dev libvulkan-dev libegl1-mesa-dev libgles2-mesa-dev libgl1-mesa-dri mesa-vulkan-drivers ninja-build
|
||||
|
||||
- name: Show installed toolchain
|
||||
run: |
|
||||
ccache --version
|
||||
clang-20 --version
|
||||
clang++-20 --version
|
||||
ld.lld-20 --version || ld.lld --version || true
|
||||
dpkg -l 'libc++*' || true
|
||||
|
||||
dpkg -l 'libc++*' 'libegl*' 'libgles*' 'mesa*' 'vulkan*' || true
|
||||
|
||||
- name: Configure CMake
|
||||
working-directory: ${{env.TEST_ROOT}}
|
||||
run: |
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" == "true" ]; then
|
||||
cmake -S . -B build-test -G Ninja -DCMAKE_C_COMPILER=clang-20 -DCMAKE_CXX_COMPILER=clang++-20 -DCMAKE_BUILD_TYPE=Debug -DMOBILEGL_BUILD_TEST=ON -DMOBILEGL_BUILD_BENCHMARK=OFF -DCMAKE_POLICY_VERSION_MINIMUM=3.5
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" = "true" ]; then
|
||||
BUILD_TYPE=Debug
|
||||
else
|
||||
cmake -S . -B build-test -G Ninja -DCMAKE_C_COMPILER=clang-20 -DCMAKE_CXX_COMPILER=clang++-20 -DCMAKE_BUILD_TYPE=Release -DMOBILEGL_BUILD_TEST=ON -DMOBILEGL_BUILD_BENCHMARK=OFF -DCMAKE_POLICY_VERSION_MINIMUM=3.5
|
||||
BUILD_TYPE=Release
|
||||
fi
|
||||
|
||||
|
||||
cmake -S . -B "${BUILD_DIR}" -G Ninja \
|
||||
-DCMAKE_C_COMPILER=clang-20 \
|
||||
-DCMAKE_CXX_COMPILER=clang++-20 \
|
||||
-DCMAKE_C_COMPILER_LAUNCHER=ccache \
|
||||
-DCMAKE_CXX_COMPILER_LAUNCHER=ccache \
|
||||
-DCMAKE_BUILD_TYPE="${BUILD_TYPE}" \
|
||||
-DMOBILEGL_LOG_ACTIVE_LEVEL=MOBILEGL_LOG_LEVEL_INFO \
|
||||
-DMOBILEGL_BUILD_TEST=ON \
|
||||
-DMOBILEGL_BUILD_BENCHMARK=ON \
|
||||
-DMOBILEGL_BUILD_INTEGRATION_TEST=ON \
|
||||
-DMOBILEGL_ITEST_VK_ICD=/usr/share/vulkan/icd.d/lvp_icd.json \
|
||||
-DMOBILEGL_BUILD_TRACE_REPLAY=OFF \
|
||||
-DBENCHMARK_DOWNLOAD_DEPENDENCIES=ON \
|
||||
-DBENCHMARK_ENABLE_TESTING=OFF \
|
||||
-DCMAKE_POLICY_VERSION_MINIMUM=3.5
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{env.TEST_ROOT}}/build-test
|
||||
run: cmake --build .
|
||||
run: cmake --build "${BUILD_DIR}" --parallel "$(nproc)"
|
||||
|
||||
- name: Show ccache stats
|
||||
if: always()
|
||||
run: ccache --show-stats
|
||||
|
||||
# Rewrite one rolling entry per job on the default branch. The upload stays
|
||||
# cumulative - it carries every object restored at the top of this run plus
|
||||
# the few TUs that actually changed - but Actions cache keys are immutable,
|
||||
# so the superseded blob has to be released before the same key can be
|
||||
# re-uploaded. Running after the build means a failed build leaves the
|
||||
# existing entry untouched. The other trigger branches restore this entry
|
||||
# rather than each writing one of their own.
|
||||
- name: Release superseded ccache entry
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CACHE_KEY: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
run: gh cache delete "${CACHE_KEY}" || true
|
||||
|
||||
- name: Save ccache
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
continue-on-error: true
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
|
||||
- name: Package Linux runtime
|
||||
run: |
|
||||
mkdir -p ci-artifacts
|
||||
mapfile -t SHARED_LIBS < <(find "${BUILD_DIR}" -type f \( -name '*.so' -o -name '*.so.*' \) -print | sort)
|
||||
tar \
|
||||
--exclude='*/CMakeFiles' \
|
||||
--exclude='*.o' \
|
||||
--exclude='*.a' \
|
||||
--exclude='*.ninja*' \
|
||||
--exclude='build.ninja' \
|
||||
--exclude='cmake_install.cmake' \
|
||||
-czf ci-artifacts/mobilegl-linux-runtime.tgz \
|
||||
"${BUILD_DIR}/CTestTestfile.cmake" \
|
||||
"${BUILD_DIR}/MobileGL/MG_Test" \
|
||||
"${BUILD_DIR}/MobileGL/MG_Benchmark" \
|
||||
"${BUILD_DIR}/MobileGL/MG_IntegrationTest" \
|
||||
"${SHARED_LIBS[@]}"
|
||||
|
||||
- name: Upload Linux runtime
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: ci-artifacts/mobilegl-linux-runtime.tgz
|
||||
if-no-files-found: error
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build-linux
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Install runtime dependencies
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libvulkan1 libegl1 libgles2 libgl1-mesa-dri mesa-vulkan-drivers
|
||||
|
||||
- name: Download Linux runtime
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: .
|
||||
|
||||
- name: Unpack Linux runtime
|
||||
run: tar -xzf mobilegl-linux-runtime.tgz
|
||||
|
||||
- name: Normalize CTest command paths
|
||||
run: |
|
||||
python - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
for path in Path('build-linux').rglob('CTestTestfile.cmake'):
|
||||
text = path.read_text()
|
||||
text = re.sub(r'"[^"]*/cmake-[^"]*/bin/cmake"', '"cmake"', text)
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
- name: Test
|
||||
working-directory: ${{env.TEST_ROOT}}/build-test/MobileGL/MG_Test
|
||||
working-directory: build-linux
|
||||
run: |
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" == "true" ]; then
|
||||
ctest -V
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" = "true" ]; then
|
||||
ctest -V -L unit --no-tests=error
|
||||
else
|
||||
ctest
|
||||
ctest --output-on-failure -L unit --no-tests=error
|
||||
fi
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: unit-core-dumps
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
integration:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build-linux
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Install runtime dependencies
|
||||
# Same set as the benchmark job, for the same reason: the scenarios bring
|
||||
# up real headless EGL (llvmpipe) and Vulkan (lavapipe) contexts, and
|
||||
# libegl-mesa0 - the EGL vendor library behind glvnd's libegl1 dispatch -
|
||||
# only arrives as a Recommends.
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libvulkan1 libegl1 libegl-mesa0 libgles2 libgl1-mesa-dri mesa-vulkan-drivers
|
||||
|
||||
- name: Download Linux runtime
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: .
|
||||
|
||||
- name: Unpack Linux runtime
|
||||
run: tar -xzf mobilegl-linux-runtime.tgz
|
||||
|
||||
- name: Normalize CTest command paths
|
||||
run: |
|
||||
python - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
for path in Path('build-linux').rglob('CTestTestfile.cmake'):
|
||||
text = path.read_text()
|
||||
text = re.sub(r'"[^"]*/cmake-[^"]*/bin/cmake"', '"cmake"', text)
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
- name: Integration scenarios
|
||||
working-directory: build-linux
|
||||
# REQUIRE_GPU makes a driverless runner FAIL instead of skipping every
|
||||
# scenario - an all-skip run is otherwise indistinguishable from a pass,
|
||||
# which is how a five-month-old draw-dropping bug survived unseen until
|
||||
# this lane existed.
|
||||
#
|
||||
# The lavapipe ICD pin lives in the build-linux configure
|
||||
# (-DMOBILEGL_ITEST_VK_ICD), NOT here: the configure bakes it into each
|
||||
# test's ctest ENVIRONMENT property, and a property entry OVERRIDES the
|
||||
# job environment - a VK_ICD_FILENAMES exported here would be silently
|
||||
# ignored while looking like it works. This lane runs on lavapipe
|
||||
# deterministically, not on whichever of the eight Mesa ICDs a GPU-less
|
||||
# runner enumerates first.
|
||||
#
|
||||
# Cores are armed so that any crash - the harness pre-flight child's
|
||||
# included - leaves /tmp/core.*, which the failure-only step below ships
|
||||
# as an artifact. Analyzing a downloaded core against the runtime
|
||||
# artifact's binary in an ubuntu-24.04 userspace reproduces the exact
|
||||
# crash stack without burning a CI round on an in-workflow debugger.
|
||||
env:
|
||||
MOBILEGL_ITEST_REQUIRE_GPU: "1"
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" = "true" ]; then
|
||||
ctest -V -L integration-gpu --no-tests=error
|
||||
else
|
||||
ctest --output-on-failure -L integration-gpu --no-tests=error
|
||||
fi
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: integration-core-dumps
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
benchmark:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build-linux
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Install runtime dependencies
|
||||
# libegl-mesa0 is the EGL vendor library itself: DriverBench brings up a
|
||||
# real GL context, and libegl1 is only glvnd's dispatch. It normally
|
||||
# arrives as a Recommends of libegl1, which is too quiet a dependency for
|
||||
# the one job that needs a working driver.
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libvulkan1 libegl1 libegl-mesa0 libgles2 libgl1-mesa-dri mesa-vulkan-drivers
|
||||
|
||||
- name: Download Linux runtime
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: .
|
||||
|
||||
- name: Unpack Linux runtime
|
||||
run: tar -xzf mobilegl-linux-runtime.tgz
|
||||
|
||||
- name: Normalize CTest command paths
|
||||
run: |
|
||||
python - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
for path in Path('build-linux').rglob('CTestTestfile.cmake'):
|
||||
text = path.read_text()
|
||||
text = re.sub(r'"[^"]*/cmake-[^"]*/bin/cmake"', '"cmake"', text)
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
- name: Benchmark
|
||||
working-directory: build-linux
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
ctest -V -C Release -L benchmark --no-tests=error
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: benchmark-core-dumps
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
build-retrace:
|
||||
runs-on: ubuntu-latest
|
||||
needs:
|
||||
- build-linux
|
||||
- test
|
||||
- benchmark
|
||||
- integration
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
env:
|
||||
BUILD_DIR: build-retrace
|
||||
CCACHE_BASEDIR: ${{ github.workspace }}
|
||||
CCACHE_COMPRESS: "true"
|
||||
CCACHE_DIR: ${{ github.workspace }}/.ccache
|
||||
CCACHE_MAXSIZE: 4G
|
||||
CCACHE_NOHASHDIR: "true"
|
||||
MOBILEGL_LIBRARY: ${{ github.workspace }}/build-linux/libMobileGL.so
|
||||
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@v1.0
|
||||
with:
|
||||
swap-size-gb: 32
|
||||
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
submodules: recursive
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Restore ccache
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
restore-keys: |
|
||||
${{ runner.os }}-test-${{ github.job }}-ccache-
|
||||
|
||||
- name: Prepare Vulkan SDK
|
||||
uses: humbletim/setup-vulkan-sdk@v1.2.1
|
||||
with:
|
||||
vulkan-query-version: 1.4.304.1
|
||||
vulkan-components: Vulkan-Headers, Vulkan-Loader
|
||||
vulkan-use-cache: true
|
||||
|
||||
- name: Update glslang external sources
|
||||
working-directory: 3rdparty/glslang
|
||||
run: python update_glslang_sources.py
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y ccache clang-20 clang++-20 lld-20 libc++-20-dev libc++abi-20-dev libvulkan-dev libegl1-mesa-dev libgles2-mesa-dev libgl1-mesa-dri mesa-vulkan-drivers ninja-build
|
||||
|
||||
- name: Show installed toolchain
|
||||
run: |
|
||||
ccache --version
|
||||
clang-20 --version
|
||||
clang++-20 --version
|
||||
ld.lld-20 --version || ld.lld --version || true
|
||||
dpkg -l 'libc++*' 'libegl*' 'libgles*' 'mesa*' 'vulkan*' || true
|
||||
|
||||
- name: Download Linux runtime
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: .
|
||||
|
||||
- name: Unpack Linux runtime
|
||||
run: |
|
||||
tar -xzf mobilegl-linux-runtime.tgz
|
||||
test -f "${MOBILEGL_LIBRARY}"
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
if [ "${{ secrets.ACTIONS_STEP_DEBUG }}" = "true" ]; then
|
||||
BUILD_TYPE=Debug
|
||||
else
|
||||
BUILD_TYPE=Release
|
||||
fi
|
||||
|
||||
cmake -S . -B "${BUILD_DIR}" -G Ninja \
|
||||
-DCMAKE_C_COMPILER=clang-20 \
|
||||
-DCMAKE_CXX_COMPILER=clang++-20 \
|
||||
-DCMAKE_C_COMPILER_LAUNCHER=ccache \
|
||||
-DCMAKE_CXX_COMPILER_LAUNCHER=ccache \
|
||||
-DCMAKE_BUILD_TYPE="${BUILD_TYPE}" \
|
||||
-DMOBILEGL_LOG_ACTIVE_LEVEL=MOBILEGL_LOG_LEVEL_INFO \
|
||||
-DMOBILEGL_BUILD_TEST=OFF \
|
||||
-DMOBILEGL_BUILD_BENCHMARK=OFF \
|
||||
-DMOBILEGL_BUILD_TRACE_REPLAY=ON \
|
||||
-DMOBILEGL_TRACE_REPLAY_MOBILEGL_LIBRARY="${MOBILEGL_LIBRARY}" \
|
||||
-DCMAKE_POLICY_VERSION_MINIMUM=3.5
|
||||
|
||||
- name: Build trace replay
|
||||
run: cmake --build "${BUILD_DIR}" --target mobilegl_trace_replay --parallel "$(nproc)"
|
||||
|
||||
- name: Show ccache stats
|
||||
if: always()
|
||||
run: ccache --show-stats
|
||||
|
||||
- name: Release superseded ccache entry
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CACHE_KEY: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
run: gh cache delete "${CACHE_KEY}" || true
|
||||
|
||||
- name: Save ccache
|
||||
if: github.ref_name == github.event.repository.default_branch
|
||||
continue-on-error: true
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ${{ runner.os }}-test-${{ github.job }}-ccache-v1
|
||||
|
||||
- name: Normalize CTest command paths
|
||||
run: |
|
||||
python - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
for path in Path('build-retrace').rglob('CTestTestfile.cmake'):
|
||||
text = path.read_text()
|
||||
text = re.sub(r'"[^"]*/cmake-[^"]*/bin/cmake"', '"cmake"', text)
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
- name: Package trace replay
|
||||
run: |
|
||||
mkdir -p ci-artifacts
|
||||
tar -czf ci-artifacts/mobilegl-trace-replay.tgz \
|
||||
build-retrace/tools/trace_replay/mobilegl_trace_replay \
|
||||
build-retrace/tools/trace_replay/CTestTestfile.cmake
|
||||
|
||||
- name: Upload trace replay
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: mobilegl-trace-replay
|
||||
path: ci-artifacts/mobilegl-trace-replay.tgz
|
||||
if-no-files-found: error
|
||||
|
||||
trace-cases:
|
||||
name: trace case matrix
|
||||
runs-on: ubuntu-latest
|
||||
needs:
|
||||
- test
|
||||
- benchmark
|
||||
- integration
|
||||
outputs:
|
||||
names: ${{ steps.trace-cases.outputs.names }}
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Load trace cases
|
||||
id: trace-cases
|
||||
run: echo "names=$(python3 tools/trace_replay/trace_cases.py --ci --format names)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
trace-fixtures:
|
||||
name: trace fixture (${{ matrix.case }})
|
||||
runs-on: ubuntu-latest
|
||||
needs: trace-cases
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.names) }}
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Derive trace fixture cache key
|
||||
id: fixture-key
|
||||
run: bash .github/scripts/trace-fixture-cache.sh key '${{ matrix.case }}'
|
||||
|
||||
- name: Restore trace fixture cache
|
||||
id: fixture-cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true'
|
||||
uses: actions/cache/restore@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Verify restored trace fixture
|
||||
id: fixture-verify
|
||||
if: steps.fixture-cache.outputs.cache-hit == 'true'
|
||||
run: |
|
||||
if bash .github/scripts/trace-fixture-cache.sh verify '${{ matrix.case }}'; then
|
||||
echo "ok=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "ok=false" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::Cached fixture for ${{ matrix.case }} failed verification; falling back to the download path"
|
||||
bash .github/scripts/trace-fixture-cache.sh reset '${{ matrix.case }}'
|
||||
fi
|
||||
|
||||
- name: Fetch trace fixture
|
||||
if: steps.fixture-verify.outputs.ok != 'true'
|
||||
run: bash .github/scripts/fetch-trace-fixture-lfs.sh '${{ matrix.case }}'
|
||||
|
||||
- name: Save trace fixture cache
|
||||
if: steps.fixture-key.outputs.cacheable == 'true' && steps.fixture-cache.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@v5
|
||||
with:
|
||||
path: ${{ steps.fixture-key.outputs.paths }}
|
||||
key: ${{ steps.fixture-key.outputs.key }}
|
||||
|
||||
- name: Stage trace fixture
|
||||
run: |
|
||||
safe_case="$(printf '%s' '${{ matrix.case }}' | sed 's/[^A-Za-z0-9._-]/_/g')"
|
||||
stage_dir="trace-fixtures/${safe_case}"
|
||||
mkdir -p "${stage_dir}"
|
||||
python3 tools/trace_replay/trace_cases.py --format fixture-files --case '${{ matrix.case }}' |
|
||||
while IFS= read -r file; do
|
||||
cp "${file}" "${stage_dir}/"
|
||||
done
|
||||
|
||||
- name: Upload trace fixture
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: trace-fixture-${{ matrix.case }}
|
||||
path: trace-fixtures/**
|
||||
if-no-files-found: error
|
||||
|
||||
retrace:
|
||||
name: retrace (${{ matrix.backend }}, ${{ matrix.case }})
|
||||
runs-on: ubuntu-latest
|
||||
needs:
|
||||
- build-linux
|
||||
- build-retrace
|
||||
- trace-cases
|
||||
- trace-fixtures
|
||||
if: ${{ always() && needs.build-linux.result == 'success' && needs.build-retrace.result == 'success' && needs.trace-cases.result == 'success' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
matrix:
|
||||
backend:
|
||||
- DirectGLES
|
||||
- DirectVulkan
|
||||
case: ${{ fromJSON(needs.trace-cases.outputs.names) }}
|
||||
|
||||
steps:
|
||||
- name: Set Swap Space
|
||||
uses: pierotofy/set-swap-space@v1.0
|
||||
with:
|
||||
swap-size-gb: 16
|
||||
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Download trace fixture
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: trace-fixture-${{ matrix.case }}
|
||||
path: trace-fixture-download
|
||||
|
||||
- name: Install trace fixture
|
||||
run: |
|
||||
mkdir -p tools/trace_replay/fixtures
|
||||
find trace-fixture-download -type f -exec cp {} tools/trace_replay/fixtures/ \;
|
||||
|
||||
- name: Get CMake
|
||||
uses: lukka/get-cmake@v4.3.3
|
||||
|
||||
- name: Install runtime dependencies
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libvulkan1 libegl1-mesa-dev libgles2-mesa-dev libgl1-mesa-dri mesa-vulkan-drivers
|
||||
test -e /usr/lib/x86_64-linux-gnu/libEGL.so
|
||||
test -e /usr/lib/x86_64-linux-gnu/libGLESv2.so
|
||||
|
||||
- name: Download Linux runtime
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-linux-runtime
|
||||
path: .
|
||||
|
||||
- name: Download trace replay
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: mobilegl-trace-replay
|
||||
path: .
|
||||
|
||||
- name: Unpack retrace runtime
|
||||
run: |
|
||||
tar -xzf mobilegl-linux-runtime.tgz
|
||||
tar -xzf mobilegl-trace-replay.tgz
|
||||
test -f build-linux/libMobileGL.so
|
||||
test -f build-retrace/tools/trace_replay/mobilegl_trace_replay
|
||||
|
||||
- name: Retrace and validate
|
||||
working-directory: build-retrace/tools/trace_replay
|
||||
run: |
|
||||
ulimit -c unlimited
|
||||
sudo sysctl -w kernel.core_pattern='/tmp/core.%e.%p'
|
||||
if [ '${{ matrix.backend }}' = 'DirectVulkan' ]; then
|
||||
export MOBILEGL_MAGMA_R11G11B10F_FALLBACK=1
|
||||
fi
|
||||
# The blended depth-write quirk auto-enables only on Qualcomm, which no CI
|
||||
# runner has, so force it on for the OIT case it exists to fix. ForceOn
|
||||
# bypasses only the vendor gate, so this exercises the real strip on
|
||||
# lavapipe. The Android AVD lane deliberately leaves it off, keeping the
|
||||
# unstripped path covered for the same trace.
|
||||
if [ '${{ matrix.backend }}' = 'DirectVulkan' ] \
|
||||
&& [ '${{ matrix.case }}' = 'improved-transparency-minecraft-26.3' ]; then
|
||||
export MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE=1
|
||||
fi
|
||||
ctest -V --no-tests=error -R '^MobileGLTraceReplay\.${{ matrix.case }}\.${{ matrix.backend }}$'
|
||||
|
||||
- name: Upload core dumps
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: retrace-core-dumps-${{ matrix.backend }}-${{ matrix.case }}
|
||||
path: /tmp/core.*
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Upload actual image
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: retrace-result-${{ matrix.backend }}-${{ matrix.case }}
|
||||
path: |
|
||||
build-retrace/tools/trace_replay/${{ matrix.case }}/actual-images/**
|
||||
build-retrace/tools/trace_replay/${{ matrix.case }}/${{ matrix.backend }}/output/**
|
||||
if-no-files-found: warn
|
||||
|
||||
retrace-summary:
|
||||
name: retrace summary
|
||||
runs-on: ubuntu-latest
|
||||
needs: retrace
|
||||
if: ${{ always() && needs.retrace.result != 'skipped' }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set artifact metadata
|
||||
run: |
|
||||
echo "date_today=$(date +'%Y-%m-%d')" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: '22'
|
||||
|
||||
- name: Download retrace results
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
pattern: retrace-result-*
|
||||
path: retrace-artifacts
|
||||
|
||||
- name: Render retrace summary
|
||||
run: |
|
||||
node tools/trace_replay/render_retrace_summary.mjs \
|
||||
--input retrace-artifacts \
|
||||
--output-dir retrace-summary \
|
||||
--title "MobileGL Linux retrace overview" \
|
||||
--group-label "Linux" \
|
||||
--html mobilegl-linux-retrace-overview.html
|
||||
|
||||
- name: Upload retrace summary
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: retrace-summary/mobilegl-linux-retrace-overview.html
|
||||
archive: false
|
||||
if-no-files-found: error
|
||||
|
||||
remove-artifact-clutter:
|
||||
name: remove artifact clutter
|
||||
runs-on: ubuntu-latest
|
||||
needs: retrace-summary
|
||||
if: always()
|
||||
permissions:
|
||||
actions: write
|
||||
steps:
|
||||
- name: Delete intermediate Linux retrace artifacts
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
declare -A failed_cases=()
|
||||
while IFS= read -r job_name; do
|
||||
case_name="${job_name#retrace (*, }"
|
||||
case_name="${case_name%)}"
|
||||
failed_cases["${case_name}"]=1
|
||||
done < <(
|
||||
gh api --paginate "repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/jobs?per_page=100" \
|
||||
--jq '.jobs[] | select(.name | startswith("retrace (")) | select(.conclusion == "failure" or .conclusion == "cancelled" or .conclusion == "timed_out" or .conclusion == "action_required") | .name'
|
||||
)
|
||||
|
||||
if ((${#failed_cases[@]})); then
|
||||
echo "Retaining fixtures for failed retrace case(s):"
|
||||
printf ' %s\n' "${!failed_cases[@]}"
|
||||
else
|
||||
echo "All retrace jobs succeeded; no fixtures need to be retained."
|
||||
fi
|
||||
|
||||
deleted=0
|
||||
retained=0
|
||||
while IFS=$'\t' read -r artifact_id artifact_name; do
|
||||
if [[ "${artifact_name}" == trace-fixture-* ]]; then
|
||||
case_name="${artifact_name#trace-fixture-}"
|
||||
if [[ -v "failed_cases[${case_name}]" ]]; then
|
||||
echo "Retaining ${artifact_name} (${artifact_id}) for failed retrace."
|
||||
((retained += 1))
|
||||
continue
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "Deleting ${artifact_name} (${artifact_id})"
|
||||
gh api --method DELETE "repos/${GITHUB_REPOSITORY}/actions/artifacts/${artifact_id}"
|
||||
((deleted += 1))
|
||||
done < <(
|
||||
gh api --paginate "repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/artifacts?per_page=100" \
|
||||
--jq '.artifacts[] | select(.name | startswith("trace-fixture-") or startswith("retrace-result-")) | [.id, .name] | @tsv'
|
||||
)
|
||||
|
||||
echo "Deleted ${deleted} intermediate Linux artifact(s); retained ${retained} failed-retrace fixture(s)."
|
||||
|
||||
+9
-1
@@ -18,4 +18,12 @@ MobileGL/MG_Test/build
|
||||
/cmake-build*
|
||||
.idea
|
||||
MobileGL/MG*/build*
|
||||
MobileGL/MG*/cmake-build*
|
||||
MobileGL/MG*/cmake-build*
|
||||
/android-plugin/.gradle
|
||||
/android-plugin/build
|
||||
/android-plugin/app/build
|
||||
/android-plugin/app/src/trace/jniLibs
|
||||
/android-plugin/local.properties
|
||||
tools/trace_replay/work/
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
|
||||
+18
-3
@@ -7,9 +7,6 @@
|
||||
[submodule "3rdparty/SPIRV-Cross"]
|
||||
path = 3rdparty/SPIRV-Cross
|
||||
url = https://github.com/KhronosGroup/SPIRV-Cross.git
|
||||
[submodule "include/FastSTL"]
|
||||
path = include/FastSTL
|
||||
url = https://github.com/MobileGL-Dev/FastSTL.git
|
||||
[submodule "3rdparty/tracy"]
|
||||
path = 3rdparty/tracy
|
||||
url = https://github.com/wolfpld/tracy.git
|
||||
@@ -19,3 +16,21 @@
|
||||
[submodule "3rdparty/VulkanMemoryAllocator"]
|
||||
path = 3rdparty/VulkanMemoryAllocator
|
||||
url = https://github.com/GPUOpen-LibrariesAndSDKs/VulkanMemoryAllocator.git
|
||||
[submodule "3rdparty/Vulkan-Utility-Libraries"]
|
||||
path = 3rdparty/Vulkan-Utility-Libraries
|
||||
url = https://github.com/KhronosGroup/Vulkan-Utility-Libraries.git
|
||||
[submodule "3rdparty/Vulkan-Headers"]
|
||||
path = 3rdparty/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
[submodule "3rdparty/SPIRV-Reflect"]
|
||||
path = 3rdparty/SPIRV-Reflect
|
||||
url = https://github.com/KhronosGroup/SPIRV-Reflect.git
|
||||
[submodule "3rdparty/apitrace"]
|
||||
path = 3rdparty/apitrace
|
||||
url = https://github.com/MobileGL-Dev/apitrace.git
|
||||
[submodule "3rdparty/asio"]
|
||||
path = 3rdparty/asio
|
||||
url = https://github.com/chriskohlhoff/asio.git
|
||||
[submodule "include/ska"]
|
||||
path = include/ska
|
||||
url = https://github.com/MobileGL-Dev/flat_hash_map.git
|
||||
|
||||
+1
Submodule 3rdparty/SPIRV-Reflect added at 10b4f09a24
+1
Submodule 3rdparty/Vulkan-Headers added at ad9ce1235e
+1
Submodule 3rdparty/Vulkan-Utility-Libraries added at 738ec97a3f
+1
Submodule 3rdparty/apitrace added at 10935bb5e4
+1
Submodule 3rdparty/asio added at 8806a6803c
Vendored
+1
-1
Submodule 3rdparty/glslang updated: 26fe5ceb45...6f12598784
+296
-6
@@ -4,15 +4,102 @@ project("MobileGL")
|
||||
|
||||
option(MOBILEGL_BUILD_TEST "Build MobileGL tests" ON )
|
||||
option(MOBILEGL_BUILD_BENCHMARK "Build MobileGL benchmarks" ON )
|
||||
# Headless end-to-end GPU scenarios (MobileGL/MG_IntegrationTest). They need a
|
||||
# real GPU/ICD to do anything, so they are off by default for CI; every scenario
|
||||
# skips cleanly where there is none. Registered under the `integration-gpu`
|
||||
# ctest label so a run can select or exclude them.
|
||||
option(MOBILEGL_BUILD_INTEGRATION_TEST "Build MobileGL headless GPU integration tests" OFF)
|
||||
option(MOBILEGL_FORCE_RELEASE_OPT "Enable Release optimization flags in Debug build" ON )
|
||||
option(MOBILEGL_ENABLE_TRACY "Enable tracy for profiling" OFF)
|
||||
option(MOBILEGL_BUILD_TRACE_REPLAY "Build desktop apitrace replay runner" OFF)
|
||||
option(MOBILEGL_TRACE_ANGLE_VARIANTS "Enable signed trace-APK ANGLE variant loading" OFF)
|
||||
option(MOBILEGL_IOS "Build MobileGL for iOS instead of macOS when APPLE is set" OFF)
|
||||
set(MOBILEGL_LOG_ACTIVE_LEVEL "MOBILEGL_LOG_LEVEL_INFO" CACHE STRING "MobileGL active log level macro")
|
||||
set(MOBILEGL_VULKAN_LIBRARY "" CACHE FILEPATH "Vulkan loader/MoltenVK library to link for iOS builds")
|
||||
|
||||
if (ANDROID)
|
||||
set(MOBILEGL_BUILD_TEST OFF CACHE BOOL "Build MobileGL tests" FORCE)
|
||||
set(MOBILEGL_BUILD_BENCHMARK OFF CACHE BOOL "Build MobileGL benchmarks" FORCE)
|
||||
|
||||
# ------- Android API level policy: minimum 26, decided here and only here -------
|
||||
# MobileGL ships against API 26: the codebase must not use any API introduced
|
||||
# after 26. That usage constraint is enforced where it is real - the shipping
|
||||
# gradle build compiles at minSdk 26, where a newer API is simply undeclared
|
||||
# and fails to compile. Configuring at a HIGHER level is therefore allowed
|
||||
# (nothing in the tree may rely on it), but a LOWER level would change the
|
||||
# libc contract underneath the shipped library and is refused.
|
||||
#
|
||||
# This has to live at configure time because the level cannot be corrected
|
||||
# from a source header. A `#define __ANDROID_API__ 26` in a common header
|
||||
# only rewrites the macro for the bionic headers that happen to be included
|
||||
# after it; any libc++ header pulled in earlier has already latched its
|
||||
# feature macros at the real configure-time level. libc++ and bionic then
|
||||
# disagree about which symbols exist - libc++ calls e.g.
|
||||
# pthread_cond_clockwait while bionic, re-read at the lowered level, has
|
||||
# hidden its declaration. MobileGL/Defines.h carried exactly that pin from
|
||||
# the first commit until it was removed; this guard is what replaces it.
|
||||
#
|
||||
# Read the level back from the compiler target triple first. Its trailing
|
||||
# number (aarch64-none-linux-android26) is precisely what clang turns into
|
||||
# __ANDROID_API__, so it cannot disagree with the compile itself, and it is
|
||||
# already past every NDK normalisation step - codename aliases, "latest",
|
||||
# and per-ABI minimum pull-ups. ANDROID_PLATFORM_LEVEL is the fallback for
|
||||
# generators/languages where the triple variable is not populated.
|
||||
#
|
||||
# Note CMAKE_SYSTEM_VERSION is deliberately NOT consulted: it holds the API
|
||||
# level only under the NDK's newer toolchain path, and is a meaningless 1
|
||||
# when ANDROID_USE_LEGACY_TOOLCHAIN_FILE is on (which is what AGP has been
|
||||
# defaulting to). Reading it would fail every legacy-mode build.
|
||||
set(MOBILEGL_ANDROID_API_LEVEL 26)
|
||||
|
||||
set(_mobilegl_android_api "")
|
||||
foreach (_mobilegl_api_triple "${CMAKE_CXX_COMPILER_TARGET}"
|
||||
"${CMAKE_C_COMPILER_TARGET}")
|
||||
if (NOT _mobilegl_android_api AND
|
||||
_mobilegl_api_triple MATCHES "-android([0-9]+)$")
|
||||
set(_mobilegl_android_api "${CMAKE_MATCH_1}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
foreach (_mobilegl_api_var ANDROID_PLATFORM_LEVEL ANDROID_NATIVE_API_LEVEL
|
||||
ANDROID_PLATFORM)
|
||||
if (NOT _mobilegl_android_api AND ${_mobilegl_api_var})
|
||||
string(REGEX REPLACE "^android-" ""
|
||||
_mobilegl_android_api "${${_mobilegl_api_var}}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (NOT _mobilegl_android_api MATCHES "^[0-9]+$")
|
||||
message(FATAL_ERROR
|
||||
"MobileGL: could not determine the Android API level (got "
|
||||
"\"${_mobilegl_android_api}\"). Configure with the NDK toolchain "
|
||||
"file and -DANDROID_PLATFORM=android-${MOBILEGL_ANDROID_API_LEVEL}.")
|
||||
elseif (_mobilegl_android_api LESS MOBILEGL_ANDROID_API_LEVEL)
|
||||
message(FATAL_ERROR
|
||||
"MobileGL requires at least Android API ${MOBILEGL_ANDROID_API_LEVEL}, "
|
||||
"but this build resolved to API ${_mobilegl_android_api}.\n"
|
||||
"Configure with -DANDROID_PLATFORM=android-${MOBILEGL_ANDROID_API_LEVEL} "
|
||||
"(gradle builds get this from minSdk ${MOBILEGL_ANDROID_API_LEVEL}, so "
|
||||
"check that minSdk instead of adding an override).")
|
||||
elseif (_mobilegl_android_api GREATER MOBILEGL_ANDROID_API_LEVEL)
|
||||
message(STATUS
|
||||
"MobileGL: configuring at Android API ${_mobilegl_android_api} "
|
||||
"(> shipping minimum ${MOBILEGL_ANDROID_API_LEVEL}). Allowed, but the "
|
||||
"tree must not use post-${MOBILEGL_ANDROID_API_LEVEL} APIs - the "
|
||||
"minSdk-${MOBILEGL_ANDROID_API_LEVEL} gradle build is the enforcing "
|
||||
"compile.")
|
||||
endif()
|
||||
|
||||
message(STATUS "MobileGL: Android API level ${_mobilegl_android_api}")
|
||||
|
||||
unset(_mobilegl_android_api)
|
||||
unset(_mobilegl_api_var)
|
||||
unset(_mobilegl_api_triple)
|
||||
endif()
|
||||
|
||||
if (NOT CMAKE_BUILD_TYPE STREQUAL "Debug" OR MOBILEGL_FORCE_RELEASE_OPT)
|
||||
option(MOBILEGL_ENABLE_LTO "Build with ThinLTO/IPO" OFF)
|
||||
|
||||
if ((NOT CMAKE_BUILD_TYPE STREQUAL "Debug" OR MOBILEGL_FORCE_RELEASE_OPT) AND MOBILEGL_ENABLE_LTO)
|
||||
# Check if ThinLTO or LTO is suppported
|
||||
include(CheckIPOSupported)
|
||||
include(CheckCCompilerFlag)
|
||||
@@ -104,10 +191,20 @@ set(SPIRV_CROSS_ENABLE_CPP OFF CACHE BOOL "Disable C++ API target" FORCE)
|
||||
set(SPIRV_CROSS_CLI OFF CACHE BOOL "Disable CLI binary" FORCE)
|
||||
set(SPIRV_CROSS_STATIC ON CACHE BOOL "Prefer static libs" FORCE)
|
||||
|
||||
set(SPIRV_REFLECT_EXECUTABLE OFF CACHE BOOL "Build spirv-reflect executable" FORCE)
|
||||
set(SPIRV_REFLECT_STATIC_LIB ON CACHE BOOL "Build a SPIRV-Reflect static library" FORCE)
|
||||
set(SPIRV_REFLECT_BUILD_TESTS OFF CACHE BOOL "Build the SPIRV-Reflect test suite" FORCE)
|
||||
set(SPIRV_REFLECT_ENABLE_ASSERTS OFF CACHE BOOL "Enable asserts for debugging" FORCE)
|
||||
set(SPIRV_REFLECT_ENABLE_ASAN OFF CACHE BOOL "Use address sanitization" FORCE)
|
||||
set(SPIRV_REFLECT_INSTALL OFF CACHE BOOL "Whether to install" FORCE)
|
||||
|
||||
# add_subdirectory(3rdparty/DiligentCore)
|
||||
add_subdirectory(3rdparty/glslang)
|
||||
add_subdirectory(3rdparty/SPIRV-Cross)
|
||||
add_subdirectory(3rdparty/VulkanMemoryAllocator)
|
||||
add_subdirectory(3rdparty/Vulkan-Headers)
|
||||
add_subdirectory(3rdparty/Vulkan-Utility-Libraries)
|
||||
add_subdirectory(3rdparty/SPIRV-Reflect)
|
||||
|
||||
set(XXHASH_BUILD_XXHSUM OFF)
|
||||
option(BUILD_SHARED_LIBS OFF)
|
||||
@@ -132,6 +229,9 @@ set(SOURCE_FILES
|
||||
|
||||
MobileGL/MG_Util/Debug/Log.cpp
|
||||
|
||||
MobileGL/MG_Util/Async/JobNode.cpp
|
||||
MobileGL/MG_Util/Async/ShaderCompilePool.cpp
|
||||
|
||||
MobileGL/MG_Util/Math/VectorTypes.cpp
|
||||
MobileGL/MG_Util/Metrics/TextureMetrics.cpp
|
||||
|
||||
@@ -160,22 +260,49 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/Converters/GLToMG/RenderStateEnumConverter.cpp
|
||||
MobileGL/MG_Util/Converters/GLToMG/ProgramEnumConverter.cpp
|
||||
MobileGL/MG_Util/Converters/MGToMG/TextureEnumConverter.cpp
|
||||
MobileGL/MG_Util/Converters/MGToVk/RenderStateEnumConverter.cpp
|
||||
MobileGL/MG_Util/Converters/MGToVk/TextureEnumConverter.cpp
|
||||
|
||||
MobileGL/MG_Util/Classifiers/TextureEnumClassifier.cpp
|
||||
|
||||
MobileGL/MG_Util/ShaderTranspiler/CompileEnv.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/ShaderCompiler.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpvcSession.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/ShaderSourceProcessor.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/glslang/TMglGlslIoResolver.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenInterfaceStructPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EliminateFloatEqualsZeroPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RenameSamplerFunctionParameterPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RenameBuiltinShadowingFunctionsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecomposeWorkgroupVec3Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DecoratePositionInvariantPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/DemoteFloat64Pass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerDrawParametersPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PackDoubleVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/FlattenXfbInterfaceBlocksPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/SplitArrayVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/Lower1DArrayImagesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/BakeImageFormatsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PrivateToEntryLocalPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUniformLocationsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUboMemberRelaxedPrecisionPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripNoPerspectivePass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/EmulateNoPerspectivePass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeFragmentOutputIndexPass.cpp
|
||||
|
||||
MobileGL/MG_Util/BackendLoaders/OpenGL/Loader.cpp
|
||||
MobileGL/MG_Util/BackendLoaders/Vulkan/Loader.cpp
|
||||
|
||||
MobileGL/MG_Util/SelfTest/DriverPost.cpp
|
||||
|
||||
MobileGL/MG_Util/Texture/PixelStoreProcessor.cpp
|
||||
MobileGL/MG_Util/Texture/TextureFormatProcessor.cpp
|
||||
|
||||
MobileGL/MG_Impl/GLXImpl/Exporting/Definitions.cpp
|
||||
MobileGL/MG_Impl/GLXImpl/GLXImpl.cpp
|
||||
MobileGL/MG_Impl/GLXImpl/LookUp/LookUp.cpp
|
||||
|
||||
MobileGL/MG_Impl/EGLImpl/Exporting/Definitions.cpp
|
||||
@@ -189,6 +316,8 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Impl/GLImpl/Framebuffer/Validators.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Program/GL_Program.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Program/ProgramInterface.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Program/GL_ProgramPipeline.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Texture/GL_Texture.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Texture/Validators.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Texture/ProxyTexture.cpp
|
||||
@@ -199,6 +328,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Impl/GLImpl/Exporting/Definitions.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Getter/GL_Getter.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Sync/GL_Sync.cpp
|
||||
MobileGL/MG_Impl/GLImpl/Query/GL_Query.cpp
|
||||
|
||||
MobileGL/MG_Impl/Init.cpp
|
||||
MobileGL/MG_Impl/GetProcAddress.cpp
|
||||
@@ -210,6 +340,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Backend/DirectGLES/BackendObject_DirectGLES.cpp
|
||||
MobileGL/MG_Backend/DirectGLES/Utils.cpp
|
||||
MobileGL/MG_Backend/DirectGLES/Managers.cpp
|
||||
MobileGL/MG_Backend/DirectGLES/MultiDraw.cpp
|
||||
|
||||
MobileGL/MG_Backend/DirectVulkan/DirectVulkan.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/BackendObject_DirectVulkan.cpp
|
||||
@@ -219,13 +350,17 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/FrameContext.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/PipelineFactory.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/ProgramFactory.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/UniformDescriptorBinder.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/UniformManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/BufferArena.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkBufferManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VertexInputStateBuilder.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VertexInputStateFactory.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkBufferObject.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkFramebufferManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkTextureManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkTimerQueryManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkSamplerManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkClearManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkRenderPassManager.cpp
|
||||
MobileGL/MG_Backend/DirectVulkan/Renderer/VkTextureSamplerManager.cpp
|
||||
|
||||
MobileGL/MG_State/GLState/Core.cpp
|
||||
MobileGL/MG_State/EGLState/Core.cpp
|
||||
@@ -244,7 +379,12 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_State/GLState/TextureState/TextureUnit.cpp
|
||||
MobileGL/MG_State/GLState/TextureState/TextureState.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramObject.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramLinkTask.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramSpirvTask.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderCompileTask.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderObject.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderPreprocessCache.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ShaderCompileAdoptionMap.cpp
|
||||
MobileGL/MG_State/GLState/ProgramState/ProgramState.cpp
|
||||
MobileGL/MG_State/GLState/RenderState/RenderState.cpp
|
||||
MobileGL/MG_State/GLState/FramebufferState/FramebufferObject.cpp
|
||||
@@ -255,6 +395,34 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_State/GLState/RenderbufferState/RenderbufferState.cpp
|
||||
)
|
||||
|
||||
if (APPLE AND NOT MOBILEGL_IOS)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Impl/CGLImpl/CGLImpl.cpp
|
||||
MobileGL/MG_Impl/CGLImpl/Exporting/Definitions.cpp
|
||||
MobileGL/MG_Impl/DyldInterpose/DyldInterpose.cpp
|
||||
MobileGL/MG_Impl/NSOpenGLImpl/NSOpenGLImpl.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
if (ANDROID)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Util/SelfTest/DriverPostJni.cpp
|
||||
MobileGL/MG_Util/SelfTest/DriverBenchJni.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
if (WIN32)
|
||||
list(APPEND SOURCE_FILES
|
||||
MobileGL/MG_Impl/WGLImpl/WGLImpl.cpp
|
||||
MobileGL/MG_Impl/WGLImpl/Exporting/Definitions.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
# The shader-compile pool runs standalone Asio on real threads. This host's glibc (>= 2.34)
|
||||
# merged pthread into libc, so it links without asking, but the NDK and musl are not
|
||||
# guaranteed to be as forgiving - ask for it explicitly rather than rely on the accident.
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
set(MOBILEGL_LINK_LIBRARIES
|
||||
glslang::glslang
|
||||
spirv-cross-c
|
||||
@@ -262,12 +430,19 @@ set(MOBILEGL_LINK_LIBRARIES
|
||||
SPIRV-Tools
|
||||
xxHash::xxhash
|
||||
GPUOpen::VulkanMemoryAllocator
|
||||
Vulkan::UtilityHeaders
|
||||
spirv-reflect-static
|
||||
Threads::Threads
|
||||
)
|
||||
|
||||
set(MOBILEGL_COMPILE_DEF
|
||||
-DVMA_STATIC_VULKAN_FUNCTIONS=0
|
||||
-DVMA_DYNAMIC_VULKAN_FUNCTIONS=1
|
||||
-DVMA_VULKAN_VERSION=1001000
|
||||
# Header-only Asio, no Boost, no deprecated interfaces. Set on the definition list
|
||||
# rather than per-target so the shared library and the _s static target agree.
|
||||
-DASIO_STANDALONE
|
||||
-DASIO_NO_DEPRECATED
|
||||
)
|
||||
|
||||
message(STATUS "MOBILEGL_COMPILE_DEF=${MOBILEGL_COMPILE_DEF}")
|
||||
@@ -279,12 +454,24 @@ set(MOBILEGL_INCLUDE_DIR
|
||||
${spirv-tools_SOURCE_DIR}/include
|
||||
${spirv-tools_BINARY_DIR}
|
||||
${SPIRV-Headers_SOURCE_DIR}/include
|
||||
# Header-only submodule: no add_subdirectory, no link target. Only
|
||||
# MG_Util/Async/ShaderCompilePool.cpp includes it, and it stays behind that file's
|
||||
# pimpl so no consumer target needs this path.
|
||||
${CMAKE_SOURCE_DIR}/3rdparty/asio/asio/include
|
||||
)
|
||||
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
add_library(${CMAKE_PROJECT_NAME} SHARED
|
||||
${SOURCE_FILES}
|
||||
)
|
||||
|
||||
if (WIN32)
|
||||
# The wgl* entry points are exported via .def (see the comment in wgl.def);
|
||||
# only the shared library links it.
|
||||
target_sources(${CMAKE_PROJECT_NAME} PRIVATE
|
||||
MobileGL/MG_Impl/WGLImpl/Exporting/wgl.def
|
||||
)
|
||||
endif()
|
||||
|
||||
if (CMAKE_BUILD_TYPE STREQUAL "Debug")
|
||||
set_target_properties(${CMAKE_PROJECT_NAME} PROPERTIES
|
||||
C_VISIBILITY_PRESET default
|
||||
@@ -311,8 +498,34 @@ target_link_libraries(${CMAKE_PROJECT_NAME}
|
||||
target_compile_definitions(${CMAKE_PROJECT_NAME}
|
||||
PUBLIC
|
||||
${MOBILEGL_COMPILE_DEF}
|
||||
MOBILEGL_LOG_ACTIVE_LEVEL=${MOBILEGL_LOG_ACTIVE_LEVEL}
|
||||
$<$<BOOL:${MOBILEGL_TRACE_ANGLE_VARIANTS}>:MOBILEGL_TRACE_ANGLE_VARIANTS=1>
|
||||
)
|
||||
|
||||
if(UNIX AND NOT APPLE AND NOT ANDROID)
|
||||
foreach(MOBILEGL_LOADER_ALIAS
|
||||
libEGL.so libEGL.so.1)
|
||||
add_custom_command(TARGET ${CMAKE_PROJECT_NAME} POST_BUILD
|
||||
COMMAND ${CMAKE_COMMAND} -E create_symlink
|
||||
"$<TARGET_FILE_NAME:${CMAKE_PROJECT_NAME}>"
|
||||
"$<TARGET_FILE_DIR:${CMAKE_PROJECT_NAME}>/${MOBILEGL_LOADER_ALIAS}"
|
||||
COMMENT "Creating ${MOBILEGL_LOADER_ALIAS} alias for Linux GL/EGL loaders"
|
||||
)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if(WIN32)
|
||||
# Drop-in for the classic GL loader path: a copy named opengl32.dll placed
|
||||
# next to a host executable is what LoadLibrary("opengl32.dll") and gdi32's
|
||||
# pixel-format forwarding will resolve.
|
||||
add_custom_command(TARGET ${CMAKE_PROJECT_NAME} POST_BUILD
|
||||
COMMAND ${CMAKE_COMMAND} -E copy_if_different
|
||||
"$<TARGET_FILE:${CMAKE_PROJECT_NAME}>"
|
||||
"$<TARGET_FILE_DIR:${CMAKE_PROJECT_NAME}>/opengl32.dll"
|
||||
COMMENT "Creating opengl32.dll drop-in copy"
|
||||
)
|
||||
endif()
|
||||
|
||||
if(NOT ANDROID)
|
||||
add_library(${CMAKE_PROJECT_NAME}_s STATIC
|
||||
${SOURCE_FILES}
|
||||
@@ -344,6 +557,7 @@ if(NOT ANDROID)
|
||||
target_compile_definitions(${CMAKE_PROJECT_NAME}_s
|
||||
PUBLIC
|
||||
${MOBILEGL_COMPILE_DEF}
|
||||
MOBILEGL_LOG_ACTIVE_LEVEL=${MOBILEGL_LOG_ACTIVE_LEVEL}
|
||||
)
|
||||
endif()
|
||||
|
||||
@@ -362,7 +576,62 @@ if (ANDROID)
|
||||
)
|
||||
endif()
|
||||
|
||||
if (NOT ANDROID)
|
||||
if (APPLE AND NOT MOBILEGL_IOS)
|
||||
# MobileGL statically embeds glslang, SPIRV-Tools, and SPIRV-Cross. When
|
||||
# this dylib is injected with DYLD_INSERT_LIBRARIES, exporting those C++
|
||||
# symbols interposes incompatible copies embedded by host libraries such
|
||||
# as shaderc. Keep only the public GL/EGL/CGL loader surface globally
|
||||
# visible; GetProcAddress can still return pointers to hidden internals.
|
||||
set(MOBILEGL_MACOS_EXPORTED_SYMBOLS
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/MobileGL/MG_Impl/DyldInterpose/ExportedSymbols.txt")
|
||||
target_link_options(${CMAKE_PROJECT_NAME} PRIVATE
|
||||
"LINKER:-exported_symbols_list,${MOBILEGL_MACOS_EXPORTED_SYMBOLS}")
|
||||
set_property(TARGET ${CMAKE_PROJECT_NAME} APPEND PROPERTY
|
||||
LINK_DEPENDS "${MOBILEGL_MACOS_EXPORTED_SYMBOLS}")
|
||||
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME} PUBLIC
|
||||
"-framework Cocoa"
|
||||
"-framework CoreVideo"
|
||||
"-framework QuartzCore"
|
||||
"-framework Foundation"
|
||||
"-framework OpenGL"
|
||||
objc)
|
||||
if(TARGET ${CMAKE_PROJECT_NAME}_s)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME}_s PUBLIC
|
||||
"-framework Cocoa"
|
||||
"-framework CoreVideo"
|
||||
"-framework QuartzCore"
|
||||
"-framework Foundation"
|
||||
"-framework OpenGL"
|
||||
objc)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (APPLE AND MOBILEGL_IOS)
|
||||
target_compile_definitions(${CMAKE_PROJECT_NAME} PUBLIC MOBILEGL_IOS=1 _LIBCPP_DISABLE_AVAILABILITY)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME} PUBLIC
|
||||
"-framework CoreGraphics"
|
||||
"-framework Foundation"
|
||||
"-framework QuartzCore"
|
||||
objc)
|
||||
if (MOBILEGL_VULKAN_LIBRARY)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME} PUBLIC "${MOBILEGL_VULKAN_LIBRARY}")
|
||||
endif()
|
||||
|
||||
if(TARGET ${CMAKE_PROJECT_NAME}_s)
|
||||
target_compile_definitions(${CMAKE_PROJECT_NAME}_s PUBLIC MOBILEGL_IOS=1 _LIBCPP_DISABLE_AVAILABILITY)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME}_s PUBLIC
|
||||
"-framework CoreGraphics"
|
||||
"-framework Foundation"
|
||||
"-framework QuartzCore"
|
||||
objc)
|
||||
if (MOBILEGL_VULKAN_LIBRARY)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME}_s PUBLIC "${MOBILEGL_VULKAN_LIBRARY}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (NOT ANDROID AND NOT MOBILEGL_IOS)
|
||||
find_package(Vulkan)
|
||||
if (Vulkan_FOUND)
|
||||
target_link_libraries(${CMAKE_PROJECT_NAME} PUBLIC Vulkan::Vulkan Vulkan::Headers)
|
||||
@@ -370,12 +639,33 @@ if (NOT ANDROID)
|
||||
target_include_directories(${CMAKE_PROJECT_NAME} PUBLIC ${Vulkan_INCLUDE_DIR})
|
||||
target_include_directories(${CMAKE_PROJECT_NAME}_s PUBLIC ${Vulkan_INCLUDE_DIR})
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (NOT ANDROID)
|
||||
# Enable testing in the top-level scope so a CTestTestfile.cmake is emitted
|
||||
# at the build-tree root. This lets `ctest` be invoked from the top-level
|
||||
# build directory (IDE "run all tests", CI) and discover every test in the
|
||||
# subdirectories below, instead of having to descend into each
|
||||
# MG_Test/MG_Benchmark subdirectory. Tests are tagged with CTest labels
|
||||
# (unit / benchmark / integration), so e.g. `ctest -L unit` selects just
|
||||
# the unit suite.
|
||||
enable_testing()
|
||||
|
||||
if (MOBILEGL_BUILD_TEST)
|
||||
add_subdirectory(MobileGL/MG_Test)
|
||||
endif()
|
||||
|
||||
# After MG_Test so googletest is already available when the unit tests are
|
||||
# built; the module fetches its own copy when they are not.
|
||||
if (MOBILEGL_BUILD_INTEGRATION_TEST)
|
||||
add_subdirectory(MobileGL/MG_IntegrationTest)
|
||||
endif()
|
||||
|
||||
if (MOBILEGL_BUILD_BENCHMARK)
|
||||
add_subdirectory(MobileGL/MG_Benchmark)
|
||||
endif()
|
||||
|
||||
if (MOBILEGL_BUILD_TRACE_REPLAY)
|
||||
add_subdirectory(tools/trace_replay)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
+160
-1
@@ -14,9 +14,168 @@ namespace MobileGL::MG_Config {
|
||||
inline const String ProjectName = "MobileGL";
|
||||
inline const String CoreName = "MobileGL Core";
|
||||
inline const String CoreVendor = "MobileGL-Dev (BZLZHH, Swung0x48, Tungsten)";
|
||||
inline const Version CoreVersion = {26, 2, 0, "-dev", VersionType::Development};
|
||||
inline const Version CoreVersion = {26, 8, 0, "-dev", VersionType::Development};
|
||||
inline const VersionStringFormatAttrib DefaultVersionStringFormatAttrib = {2, 2, 0, true, true};
|
||||
inline const Uint64 CacheVersion = 0;
|
||||
|
||||
extern BackendType ActiveBackendType;
|
||||
|
||||
// Tri-state override for device-specific quirks: Auto lets the detected device decide,
|
||||
// ForceOn/ForceOff bypass the detection in either direction. ForceOn only bypasses the
|
||||
// device gate - each quirk keeps its structural safety checks.
|
||||
enum class QuirkOverride : Uint8 {
|
||||
Auto = 0,
|
||||
ForceOn,
|
||||
ForceOff,
|
||||
};
|
||||
|
||||
// Preferred DirectVulkan dispatch tier for the glMultiDraw* families. A preference,
|
||||
// never a demand: the renderer clamps it to what the device supports at device
|
||||
// creation, falling down the chain ext -> indirect -> unroll with one log line.
|
||||
enum class MultiDrawMode : Uint8 {
|
||||
Auto = 0, // unset: best supported tier
|
||||
Ext, // VK_EXT_multi_draw: one vkCmdDrawMultiEXT / vkCmdDrawMultiIndexedEXT
|
||||
Indirect, // multiDrawIndirect feature: one vkCmdDraw*Indirect over a transient command array
|
||||
Unroll, // one vkCmdDraw* per sub-draw
|
||||
};
|
||||
|
||||
// Preferred DirectGLES emulation tier for glMultiDrawElements(BaseVertex). GLES has no
|
||||
// such entry point in core, so every tier below is an emulation; they differ only in
|
||||
// which driver capability they lean on and how many driver calls a batch costs. Like
|
||||
// the Magma knob this is a preference, clamped at resolution time to what the ES
|
||||
// driver actually supports, with one log line when it falls back.
|
||||
enum class GLESMultiDrawMode : Uint8 {
|
||||
Auto = 0, // unset: best supported tier
|
||||
Ext, // one glMultiDrawElementsBaseVertexEXT
|
||||
MultiIndirect, // one glMultiDrawElementsIndirectEXT over a scratch command buffer
|
||||
Indirect, // one glDrawElementsIndirect per sub-draw over that same buffer
|
||||
BaseVertex, // one glDrawElementsBaseVertex per sub-draw
|
||||
DrawElements, // baseVertex folded into a scratch index buffer on the CPU, then plain
|
||||
// glDrawElements per sub-draw (for drivers with no base-vertex draw at all)
|
||||
Compute, // a compute shader flattens every sub-draw into one rebased index buffer,
|
||||
// drawn by a single glDrawElements
|
||||
};
|
||||
|
||||
// Feature toggles parsed once from environment variables in MG_ConfigLoader::Init()
|
||||
// (ConfigLoader.cpp), before the accepted-env map is destroyed. All Bool fields share
|
||||
// one truthy rule: the variable is set, non-empty, not "0", and not "false"
|
||||
// (case-insensitive).
|
||||
//
|
||||
// Env variables intentionally NOT mirrored here (kept as live std::getenv at their
|
||||
// call sites):
|
||||
// - DISPLAY: X11 session variable, not MobileGL configuration.
|
||||
// - MOBILEGL_LOG_FILE_PATH: log-file init runs before MG_ConfigLoader::Init
|
||||
// (see MG_Util/Debug/Log.cpp).
|
||||
// - MOBILEGL_VALIDATE_SPIRV: test suites like SpirvPassTest exercise
|
||||
// ShaderCompiler without ever running MobileGL::Initialize(), and every
|
||||
// Initialize() re-runs MG_ConfigLoader::Init, which would clobber a
|
||||
// programmatic override stored here (see ShaderCompiler.cpp,
|
||||
// SpirvValidationEnabled).
|
||||
struct FeaturesTable {
|
||||
// MOBILEGL_DISABLE_TIMERQUERY: do not advertise or use GPU timer queries.
|
||||
Bool DisableTimerQuery = false;
|
||||
// MOBILEGL_USE_ANGLE: load ANGLE EGL/GLES libraries.
|
||||
Bool UseAngle = false;
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
// MOBILEGL_TRACE_ANGLE_VARIANT: signed trace-APK ANGLE build short hash.
|
||||
String TraceAngleVariant;
|
||||
#endif
|
||||
// MOBILEGL_DISABLE_SUBGROUP: force-disable Vulkan shader subgroup support.
|
||||
Bool DisableSubgroup = false;
|
||||
// MOBILEGL_ADVERTISE_FP64: add GL_ARB_gpu_shader_fp64 to the advertised extension
|
||||
// string. `double` in a shader always WORKS - it is narrowed to 32 bits before any
|
||||
// module reaches a backend (ShaderTranspiler::DemoteFloat64Pass) - but the extension
|
||||
// promises 64-bit precision, and that is the one thing the narrowing cannot deliver.
|
||||
// Off by default so an application that checks the string before using doubles keeps
|
||||
// its float path; on for measuring what the conformance suite makes of the demoted
|
||||
// precision. See the DemoteFloat64Pass header and the "fp64" POST row.
|
||||
Bool AdvertiseFp64 = false;
|
||||
// MOBILEGL_MAGMA_R11G11B10F_FALLBACK: use fallback format for R11G11B10F on Vulkan.
|
||||
Bool MagmaR11G11B10FFallback = false;
|
||||
// MOBILEGL_MAGMA_FRAMESINFLIGHT: requested Magma frames in flight, defaulting to 3.
|
||||
Uint32 MagmaFramesInFlight = 3;
|
||||
// MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER: avoid mipmap min filters in samplers,
|
||||
// resolves certain rendering bugs on ANGLE + llvmpipe.
|
||||
Bool AvoidSamplerMipmapMinFilter = false;
|
||||
// MOBILEGL_AVOID_EXPLICIT_LOD_BIAS: leave an already-explicit LOD argument alone when
|
||||
// emulating GL_TEXTURE_LOD_BIAS, instead of adding the bias uniform to it. Injecting
|
||||
// the uniform turns a compile-time-constant LOD into a runtime expression, which
|
||||
// sends ANGLE + llvmpipe down a mip-selection path that dereferences a NULL
|
||||
// descriptor and kills the process. Deviates from spec (Vulkan adds the bias to
|
||||
// OpImageSampleExplicitLod), so it is an avoidance for that stack only.
|
||||
Bool AvoidExplicitLodBias = false;
|
||||
// MOBILEGL_COHERENT_AS_FLUSH: app-compat for engines (e.g. Flywheel) that write
|
||||
// GPU-read data through persistent GL_MAP_FLUSH_EXPLICIT_BIT maps they never
|
||||
// flush. Persistent FLUSH_EXPLICIT map requests are rewritten to coherent
|
||||
// semantics: writes reach the backend without glFlushMappedBufferRange, and
|
||||
// flush calls on rewritten maps become error-free no-ops. Non-persistent maps
|
||||
// keep spec FLUSH_EXPLICIT behavior.
|
||||
Bool CoherentAsFlush = false;
|
||||
// MOBILEGL_TRACE_SKIP_AUTODESTROY: skip teardown in the ELF destructor (Init.cpp).
|
||||
Bool TraceSkipAutodestroy = false;
|
||||
// MOBILEGL_DISABLE_UBO_RING: force the DirectGLES global-UBO upload back to the
|
||||
// per-draw glBufferSubData path instead of the persistent-mapped ring allocator
|
||||
// (negative control / driver-bug escape hatch).
|
||||
Bool DisableUboRing = false;
|
||||
// MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION: make DirectGLES skip the native ES
|
||||
// depth/stencil reads and always go through the shader-sampling emulation. Core GL
|
||||
// ES has no depth or stencil readback, but some drivers accept it anyway (Mesa does,
|
||||
// Adreno does not), which means the emulation is dead code on exactly the stack the
|
||||
// headless suite runs on. This forces it live so the scenarios and the CTS can
|
||||
// exercise the path, and gives the device an A/B lever over the same choice.
|
||||
Bool EsprytForceDepthStencilReadbackEmulation = false;
|
||||
// MOBILEGL_RELAXED_SEMANTICS: relax strict core-profile rules (e.g. VAO-0 draws,
|
||||
// texture-name reuse after delete) even on contexts that explicitly requested a core
|
||||
// profile. Without it, relaxed semantics still apply to every context that did not
|
||||
// explicitly request a core profile via EGL_CONTEXT_OPENGL_PROFILE_MASK / a >=3.1
|
||||
// version request.
|
||||
Bool RelaxedSemantics = false;
|
||||
// MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN: overrides the shader-source quirk that
|
||||
// rewrites the recognized workgroup prefix-scan template on Qualcomm devices with
|
||||
// subgroups wider than 32 lanes (see ShaderSourceProcessor's quirk registry).
|
||||
QuirkOverride SubgroupPrefixScanQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE: overrides the DirectVulkan quirk that
|
||||
// strips depth writes from accumulation-blended pipelines (MIN/MAX or additive
|
||||
// ONE+ONE - the multi-pass depth-equality signature) on drivers without
|
||||
// cross-pipeline vertex position invariance. Sorted-transparency "over" blends,
|
||||
// gl_FragDepth writers, and fully color-masked attachments are exempt (see
|
||||
// PipelineFactory::ShouldSuppressDepthWrite). Auto detects Qualcomm.
|
||||
QuirkOverride MagmaDisableBlendedDepthWriteQuirk = QuirkOverride::Auto;
|
||||
// MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS: leave the Vulkan robustBufferAccess device
|
||||
// feature off. It is enabled by default to match GL's defined out-of-range fetch
|
||||
// behavior; this escape hatch exists to measure or dodge its GPU cost on a device.
|
||||
Bool DisableRobustBufferAccess = false;
|
||||
// MOBILEGL_MAGMA_MULTIDRAW_MODE: preferred DirectVulkan multi-draw dispatch tier
|
||||
// ("ext" | "indirect" | "unroll", see MultiDrawMode). Clamped to device support;
|
||||
// unset picks the best supported tier.
|
||||
MultiDrawMode MagmaMultiDrawMode = MultiDrawMode::Auto;
|
||||
// MOBILEGL_ESPRYT_MULTIDRAW_MODE: preferred DirectGLES glMultiDrawElements emulation
|
||||
// tier ("ext" | "multiindirect" | "indirect" | "basevertex" | "drawelements" |
|
||||
// "compute", see GLESMultiDrawMode). Clamped to driver support; unset picks the best
|
||||
// supported tier, which never includes "compute" - see the note on its resolution.
|
||||
GLESMultiDrawMode EsprytMultiDrawMode = GLESMultiDrawMode::Auto;
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE: overrides asynchronous shader compilation. Unset
|
||||
// keeps the built-in default (MG_Util::Async::kAsyncShaderCompileDefault); falsy
|
||||
// forces every glCompileShader/glLinkProgram to run synchronously on the calling
|
||||
// thread AND withdraws GL_KHR_parallel_shader_compile, so the single switch reverts
|
||||
// both the threading and the application-visible behaviour change.
|
||||
QuirkOverride AsyncShaderCompile = QuirkOverride::Auto;
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE_THREADS: shader-compile worker count. 0 (unset) means
|
||||
// auto, which is min(4, big cores); an explicit value is honoured as given.
|
||||
Uint32 AsyncShaderCompileThreads = 0;
|
||||
// MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS: while a compile job is still in flight,
|
||||
// glGetShaderiv(GL_COMPILE_STATUS) answers GL_TRUE and the shader info log reads
|
||||
// empty, WITHOUT joining the job (latched per compile - see
|
||||
// ShaderObject::TakeOptimisticCompileAnswer). A deliberate, bounded spec violation:
|
||||
// a real failure still fails the program link with the compile log quoted. It
|
||||
// exists for applications that compile hundreds of shaders serially and read the
|
||||
// status right after each glCompileShader - Iris's shader-pack load - where those
|
||||
// per-shader joins are what serializes the batch on its main path (Iris's gbuffer
|
||||
// phase issues no program-level query between programs; program-level LINK_STATUS
|
||||
// and the program info log still join truthfully, so paths that check each link
|
||||
// immediately stay serial by their own construction). Off by default; never
|
||||
// advertise it.
|
||||
QuirkOverride AsyncOptimisticShaderStatus = QuirkOverride::Auto;
|
||||
};
|
||||
extern FeaturesTable Features;
|
||||
} // namespace MobileGL::MG_Config
|
||||
|
||||
@@ -8,6 +8,19 @@
|
||||
|
||||
#include "Config.h"
|
||||
|
||||
#include <cerrno>
|
||||
#include <cstdlib>
|
||||
|
||||
#ifndef _WIN32
|
||||
extern char** environ;
|
||||
#endif
|
||||
|
||||
namespace MobileGL::MG_Config {
|
||||
// Zero/default-initialized at static-init time (all fields have constexpr-friendly
|
||||
// defaults), so it is safe to read even if MG_ConfigLoader::Init has not run yet.
|
||||
FeaturesTable Features;
|
||||
} // namespace MobileGL::MG_Config
|
||||
|
||||
namespace MobileGL::MG_ConfigLoader {
|
||||
static UniquePtr<UnorderedMap<String, String>> acceptedEnvVariablesMap;
|
||||
|
||||
@@ -56,6 +69,128 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
}
|
||||
}
|
||||
|
||||
// Unified truthy rule for boolean feature env variables: set, non-empty, not "0",
|
||||
// and not "false" (case-insensitive).
|
||||
static Bool IsTruthyValue(const String& value) {
|
||||
if (value.empty() || value == "0") {
|
||||
return false;
|
||||
}
|
||||
String lowered = value;
|
||||
std::transform(lowered.begin(), lowered.end(), lowered.begin(),
|
||||
[](unsigned char c) { return static_cast<char>(std::tolower(c)); });
|
||||
return lowered != "false";
|
||||
}
|
||||
|
||||
inline Bool QueryEnvFlag(const String& key) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
return it != acceptedEnvVariablesMap->end() && IsTruthyValue(it->second);
|
||||
}
|
||||
|
||||
// Quirk overrides are tri-state: an unset variable keeps device auto-detection, a truthy
|
||||
// value forces the quirk on, anything else set ("0", "false", "") forces it off.
|
||||
inline MG_Config::QuirkOverride QueryEnvQuirkOverride(const String& key) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
return MG_Config::QuirkOverride::Auto;
|
||||
}
|
||||
return IsTruthyValue(it->second) ? MG_Config::QuirkOverride::ForceOn
|
||||
: MG_Config::QuirkOverride::ForceOff;
|
||||
}
|
||||
|
||||
// Multi-draw mode is a named-value preference: unset keeps Auto (best supported tier),
|
||||
// a recognized name selects that tier as the ceiling, anything else warns and keeps Auto.
|
||||
inline MG_Config::MultiDrawMode QueryEnvMultiDrawMode(const String& key) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
return MG_Config::MultiDrawMode::Auto;
|
||||
}
|
||||
String lowered = it->second;
|
||||
std::transform(lowered.begin(), lowered.end(), lowered.begin(),
|
||||
[](unsigned char c) { return static_cast<char>(std::tolower(c)); });
|
||||
if (lowered == "ext") return MG_Config::MultiDrawMode::Ext;
|
||||
if (lowered == "indirect") return MG_Config::MultiDrawMode::Indirect;
|
||||
if (lowered == "unroll") return MG_Config::MultiDrawMode::Unroll;
|
||||
if (lowered.empty() || lowered == "auto") return MG_Config::MultiDrawMode::Auto;
|
||||
MGLOG_W("Config: Ignoring invalid env variable %s='%s'; expected ext|indirect|unroll|auto, using auto",
|
||||
key.c_str(), it->second.c_str());
|
||||
return MG_Config::MultiDrawMode::Auto;
|
||||
}
|
||||
|
||||
// Same contract as QueryEnvMultiDrawMode, over the DirectGLES tier names.
|
||||
inline MG_Config::GLESMultiDrawMode QueryEnvGLESMultiDrawMode(const String& key) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
return MG_Config::GLESMultiDrawMode::Auto;
|
||||
}
|
||||
String lowered = it->second;
|
||||
std::transform(lowered.begin(), lowered.end(), lowered.begin(),
|
||||
[](unsigned char c) { return static_cast<char>(std::tolower(c)); });
|
||||
if (lowered == "ext") return MG_Config::GLESMultiDrawMode::Ext;
|
||||
if (lowered == "multiindirect") return MG_Config::GLESMultiDrawMode::MultiIndirect;
|
||||
if (lowered == "indirect") return MG_Config::GLESMultiDrawMode::Indirect;
|
||||
if (lowered == "basevertex") return MG_Config::GLESMultiDrawMode::BaseVertex;
|
||||
if (lowered == "drawelements") return MG_Config::GLESMultiDrawMode::DrawElements;
|
||||
if (lowered == "compute") return MG_Config::GLESMultiDrawMode::Compute;
|
||||
if (lowered.empty() || lowered == "auto") return MG_Config::GLESMultiDrawMode::Auto;
|
||||
MGLOG_W("Config: Ignoring invalid env variable %s='%s'; expected "
|
||||
"ext|multiindirect|indirect|basevertex|drawelements|compute|auto, using auto",
|
||||
key.c_str(), it->second.c_str());
|
||||
return MG_Config::GLESMultiDrawMode::Auto;
|
||||
}
|
||||
|
||||
inline Uint32 QueryEnvUint32(const String& key, Uint32 defaultValue, Uint32 minValue, Uint32 maxValue) {
|
||||
auto it = acceptedEnvVariablesMap->find(key);
|
||||
if (it == acceptedEnvVariablesMap->end()) {
|
||||
return defaultValue;
|
||||
}
|
||||
|
||||
const String& value = it->second;
|
||||
char* parseEnd = nullptr;
|
||||
errno = 0;
|
||||
const unsigned long parsedValue = std::strtoul(value.c_str(), &parseEnd, 10);
|
||||
if (parseEnd == value.c_str() || *parseEnd != '\0' || errno == ERANGE || parsedValue < minValue ||
|
||||
parsedValue > maxValue) {
|
||||
MGLOG_W("Config: Ignoring invalid env variable %s='%s'; expected an integer in range [%u, %u], "
|
||||
"using default %u",
|
||||
key.c_str(), value.c_str(), minValue, maxValue, defaultValue);
|
||||
return defaultValue;
|
||||
}
|
||||
|
||||
return static_cast<Uint32>(parsedValue);
|
||||
}
|
||||
|
||||
inline void InitFeatures() {
|
||||
auto& features = MG_Config::Features;
|
||||
features.DisableTimerQuery = QueryEnvFlag("MOBILEGL_DISABLE_TIMERQUERY");
|
||||
features.UseAngle = QueryEnvFlag("MOBILEGL_USE_ANGLE");
|
||||
#if defined(MOBILEGL_TRACE_ANGLE_VARIANTS)
|
||||
QueryEnvVariable("MOBILEGL_TRACE_ANGLE_VARIANT", features.TraceAngleVariant, "");
|
||||
#endif
|
||||
features.DisableSubgroup = QueryEnvFlag("MOBILEGL_DISABLE_SUBGROUP");
|
||||
features.AdvertiseFp64 = QueryEnvFlag("MOBILEGL_ADVERTISE_FP64");
|
||||
features.MagmaR11G11B10FFallback = QueryEnvFlag("MOBILEGL_MAGMA_R11G11B10F_FALLBACK");
|
||||
features.MagmaFramesInFlight = QueryEnvUint32("MOBILEGL_MAGMA_FRAMESINFLIGHT", 3, 1, 64);
|
||||
features.AvoidSamplerMipmapMinFilter =
|
||||
QueryEnvFlag("MOBILEGL_AVOID_SAMPLER_MIPMAP_MIN_FILTER");
|
||||
features.AvoidExplicitLodBias = QueryEnvFlag("MOBILEGL_AVOID_EXPLICIT_LOD_BIAS");
|
||||
features.CoherentAsFlush = QueryEnvFlag("MOBILEGL_COHERENT_AS_FLUSH");
|
||||
features.TraceSkipAutodestroy = QueryEnvFlag("MOBILEGL_TRACE_SKIP_AUTODESTROY");
|
||||
features.DisableUboRing = QueryEnvFlag("MOBILEGL_DISABLE_UBO_RING");
|
||||
features.EsprytForceDepthStencilReadbackEmulation =
|
||||
QueryEnvFlag("MOBILEGL_ESPRYT_FORCE_DS_READBACK_EMULATION");
|
||||
features.RelaxedSemantics = QueryEnvFlag("MOBILEGL_RELAXED_SEMANTICS");
|
||||
features.SubgroupPrefixScanQuirk = QueryEnvQuirkOverride("MOBILEGL_QUIRK_SUBGROUP_PREFIX_SCAN");
|
||||
features.MagmaDisableBlendedDepthWriteQuirk =
|
||||
QueryEnvQuirkOverride("MOBILEGL_MAGMA_DISABLE_BLENDED_DEPTH_WRITE");
|
||||
features.DisableRobustBufferAccess = QueryEnvFlag("MOBILEGL_DISABLE_ROBUST_BUFFER_ACCESS");
|
||||
features.MagmaMultiDrawMode = QueryEnvMultiDrawMode("MOBILEGL_MAGMA_MULTIDRAW_MODE");
|
||||
features.EsprytMultiDrawMode = QueryEnvGLESMultiDrawMode("MOBILEGL_ESPRYT_MULTIDRAW_MODE");
|
||||
features.AsyncShaderCompile = QueryEnvQuirkOverride("MOBILEGL_ASYNC_SHADER_COMPILE");
|
||||
features.AsyncShaderCompileThreads = QueryEnvUint32("MOBILEGL_ASYNC_SHADER_COMPILE_THREADS", 0, 0, 64);
|
||||
features.AsyncOptimisticShaderStatus =
|
||||
QueryEnvQuirkOverride("MOBILEGL_ASYNC_OPTIMISTIC_SHADER_STATUS");
|
||||
}
|
||||
|
||||
inline void InitBackendType() {
|
||||
String backendTypeStr;
|
||||
QueryEnvVariable("MOBILEGL_BACKEND_TYPE", backendTypeStr, "DirectGLES");
|
||||
@@ -77,6 +212,7 @@ namespace MobileGL::MG_ConfigLoader {
|
||||
InitializeAcceptedEnvVariables();
|
||||
|
||||
InitBackendType();
|
||||
InitFeatures();
|
||||
|
||||
// Destroy the map since we won't need it anymore
|
||||
acceptedEnvVariablesMap.reset();
|
||||
|
||||
+54
-12
@@ -9,10 +9,20 @@
|
||||
#pragma once
|
||||
|
||||
// ============== Platform-specific definitions and macros ============== //
|
||||
#ifdef __ANDROID__
|
||||
#undef __ANDROID_API__
|
||||
#define __ANDROID_API__ 26 // force Android API level to 26 for compatibility
|
||||
#endif
|
||||
// No __ANDROID_API__ pin here on purpose. The effective API level is owned by
|
||||
// the build system (gradle minSdk 26 -> -DANDROID_PLATFORM=android-26, enforced
|
||||
// by the configure-time guard in CMakeLists.txt), not by a macro.
|
||||
//
|
||||
// History: this used to `#define __ANDROID_API__ 26` to *raise* the level back
|
||||
// when the build configured something lower, so that pthread_getname_np (which
|
||||
// bionic guards with __INTRODUCED_IN(26)) would be declared. Once a later
|
||||
// change added an `#undef` in front of it, the same line started *lowering* the
|
||||
// level whenever the build configured higher than 26 - and that is an
|
||||
// include-order split-brain, not a compatibility knob: a TU that includes any
|
||||
// libc++ header before Includes.h latches libc++'s feature macros at the
|
||||
// configure-time level, and only the bionic headers pulled in afterwards see
|
||||
// the lowered value. The two halves then disagree (e.g. libc++ believes
|
||||
// pthread_cond_clockwait exists while bionic has since hidden its declaration).
|
||||
|
||||
#ifdef _WIN32
|
||||
#ifndef NOMINMAX
|
||||
@@ -32,9 +42,31 @@
|
||||
#define MOBILEGL_GLX_API MOBILEGL_API
|
||||
#define MOBILEGL_GL_API MOBILEGL_API
|
||||
#define MOBILEGL_EGL_API MOBILEGL_API
|
||||
#define MOBILEGL_CGL_API MOBILEGL_API
|
||||
#define MOBILEGL_NSOPENGL_API MOBILEGL_API
|
||||
#define MOBILEGL_WGL_API MOBILEGL_API
|
||||
|
||||
// ====================== MobileGL configurations ======================= //
|
||||
// The numeric log levels live here, not only in Log.h: MOBILEGL_ASSERT below compares
|
||||
// MOBILEGL_LOG_ACTIVE_LEVEL against MOBILEGL_LOG_LEVEL_DEBUG, and in a translation unit
|
||||
// that includes Defines.h without Log.h both tokens would silently evaluate to 0 in the
|
||||
// preprocessor conditional - enabling the assert in exactly the INFO-level builds it is
|
||||
// documented to be compiled out of. Log.h redefines them identically, which is legal.
|
||||
//
|
||||
// Severity order, ascending: DEBUG < INFO < WARN < ERROR < FATAL. MOBILEGL_LOG_ACTIVE_LEVEL
|
||||
// names the lowest severity compiled in, so the production default INFO keeps I/W/E/F and
|
||||
// drops only D. Any edit here must be mirrored in Log.h.
|
||||
#ifndef MOBILEGL_LOG_LEVEL_DEBUG
|
||||
#define MOBILEGL_LOG_LEVEL_DEBUG 0
|
||||
#define MOBILEGL_LOG_LEVEL_INFO 1
|
||||
#define MOBILEGL_LOG_LEVEL_WARN 2
|
||||
#define MOBILEGL_LOG_LEVEL_ERROR 3
|
||||
#define MOBILEGL_LOG_LEVEL_FATAL 4
|
||||
#endif
|
||||
|
||||
#ifndef MOBILEGL_LOG_ACTIVE_LEVEL
|
||||
#define MOBILEGL_LOG_ACTIVE_LEVEL MOBILEGL_LOG_LEVEL_INFO
|
||||
#endif
|
||||
|
||||
#define MOBILEGL_LOG_ENABLE_CONSOLE 0
|
||||
#define MOBILEGL_LOG_ENABLE_FILE 1
|
||||
@@ -63,11 +95,21 @@
|
||||
#endif
|
||||
|
||||
// =============================== Utils ================================ //
|
||||
#define MOBILEGL_ASSERT(condition, ...) \
|
||||
do { \
|
||||
if (!(condition)) { \
|
||||
MGLOG_F("Assertion failed" __VA_OPT__(": ") __VA_ARGS__); \
|
||||
MGLOG_F(" at %s:%d (%s)", __FILE__, __LINE__, __func__); \
|
||||
TRAP; \
|
||||
} \
|
||||
} while (0)
|
||||
// Asserts are live in exactly the builds where MGLOG_D is live, i.e. DEBUG builds only;
|
||||
// an INFO build (the production default) compiles them out. DEBUG is the lowest severity
|
||||
// in the ordering above, so "ACTIVE <= DEBUG" is true only for ACTIVE == DEBUG - the same
|
||||
// gate MGLOG_D uses in Log.h. That equivalence is what makes this gate survive the
|
||||
// 2026-08-13 renumbering unchanged; the contract is and stays
|
||||
// "INFO builds: asserts OFF; DEBUG builds: asserts ON".
|
||||
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG
|
||||
#define MOBILEGL_ASSERT(condition, ...) \
|
||||
do { \
|
||||
if (!(condition)) { \
|
||||
MGLOG_F("Assertion failed" __VA_OPT__(": ") __VA_ARGS__); \
|
||||
MGLOG_F(" at %s:%d (%s)", __FILE__, __LINE__, __func__); \
|
||||
TRAP; \
|
||||
} \
|
||||
} while (0)
|
||||
#else
|
||||
#define MOBILEGL_ASSERT(condition, ...)
|
||||
#endif
|
||||
|
||||
@@ -14,7 +14,13 @@ namespace MobileGL {
|
||||
} // namespace MG_Config
|
||||
|
||||
namespace MG_Backend {
|
||||
UniquePtr<BackendObject> pActiveBackendObject;
|
||||
// Leak-at-exit storage: the UniquePtr itself lives on the heap and is
|
||||
// never destroyed by the runtime, so process exit runs no backend
|
||||
// destructors (static destruction order across TUs is undefined).
|
||||
// Deterministic teardown happens inside the EGL lifecycle instead:
|
||||
// the last eglTerminate calls MobileGL::Destroy(), which .reset()s
|
||||
// these singletons while the process is still healthy.
|
||||
UniquePtr<BackendObject>& pActiveBackendObject = *new UniquePtr<BackendObject>();
|
||||
GlobalBackendFunctionsTable gBackendFunctionsTable;
|
||||
} // namespace MG_Backend
|
||||
} // namespace MobileGL
|
||||
|
||||
+26
-2
@@ -32,6 +32,8 @@
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
#include <cassert>
|
||||
#include <climits>
|
||||
#include <cstdlib>
|
||||
#include <cstdarg>
|
||||
#include <cstring>
|
||||
#include <numeric>
|
||||
@@ -47,8 +49,8 @@
|
||||
#include <stacktrace>
|
||||
#endif
|
||||
|
||||
// Include FastSTL
|
||||
#include <FastSTL/UnorderedMap.h>
|
||||
// Include ska::flat_hash_map
|
||||
#include <ska/flat_hash_map.hpp>
|
||||
|
||||
// Include xxHash
|
||||
#include <xxhash.h>
|
||||
@@ -108,10 +110,32 @@
|
||||
#define VK_USE_PLATFORM_WIN32_KHR
|
||||
#elif defined(__APPLE__)
|
||||
#define VK_USE_PLATFORM_METAL_EXT
|
||||
#elif defined(__linux__)
|
||||
#define VK_USE_PLATFORM_XLIB_KHR
|
||||
typedef struct _XDisplay Display;
|
||||
typedef unsigned long XID;
|
||||
typedef XID Window;
|
||||
typedef unsigned long VisualID;
|
||||
#else
|
||||
#warning "VK_USE_PLATFORM_*_KHR not defined for this platform!"
|
||||
#endif
|
||||
#if defined(VK_USE_PLATFORM_XLIB_KHR)
|
||||
#pragma push_macro("Bool")
|
||||
#pragma push_macro("None")
|
||||
#pragma push_macro("Always")
|
||||
#pragma push_macro("Status")
|
||||
#pragma push_macro("LSBFirst")
|
||||
#pragma push_macro("DestroyAll")
|
||||
#endif
|
||||
#include <vulkan/vulkan.h>
|
||||
#if defined(VK_USE_PLATFORM_XLIB_KHR)
|
||||
#pragma pop_macro("DestroyAll")
|
||||
#pragma pop_macro("LSBFirst")
|
||||
#pragma pop_macro("Status")
|
||||
#pragma pop_macro("Always")
|
||||
#pragma pop_macro("None")
|
||||
#pragma pop_macro("Bool")
|
||||
#endif
|
||||
|
||||
#ifdef TRACY_ENABLE
|
||||
#include <tracy/Tracy.hpp>
|
||||
|
||||
+111
-34
@@ -9,13 +9,79 @@
|
||||
#include "Init.h"
|
||||
#include "Config.h"
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_Backend/DirectVulkan/DirectVulkan.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <MG_Impl/GLImpl/Texture/ProxyTexture.h>
|
||||
#include <MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h>
|
||||
#include <MG_Impl/GLImpl/Sync/GL_Sync.h>
|
||||
#include <MG_Util/Async/ShaderCompilePool.h>
|
||||
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace {
|
||||
std::atomic<Bool> g_isInitialized = false;
|
||||
thread_local Bool tl_initializing = false;
|
||||
|
||||
std::mutex& InitMutex() {
|
||||
static std::mutex mutex;
|
||||
return mutex;
|
||||
}
|
||||
|
||||
void DestroyImpl(Bool logLifecycle) {
|
||||
if (!g_isInitialized) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (logLifecycle) {
|
||||
MGLOG_I("MobileGL closing...");
|
||||
}
|
||||
// First, before anything else is torn down. In-flight compile/link jobs own
|
||||
// their own inputs and are safe against everything below EXCEPT glslang's
|
||||
// process globals and the TShader/TProgram objects hanging off pGLContext,
|
||||
// both of which this function is about to destroy. This is the one
|
||||
// cancellation path in the whole design that waits.
|
||||
MG_Util::Async::ShaderCompilePool::Get().StopAndDrain();
|
||||
// GL syncs die with their contexts, and every context is gone by the
|
||||
// time full teardown runs: drain the live-sync registry while the
|
||||
// backend function table can still release the backend handles (and
|
||||
// before a re-initialized library could pair them with the wrong
|
||||
// backend's DeleteSync).
|
||||
MG_Impl::GLImpl::DestroyAllSyncObjects();
|
||||
MG_Backend::pActiveBackendObject.reset();
|
||||
MG_State::pGLContext.reset();
|
||||
MG_State::pEGLContext.reset();
|
||||
MG_Impl::GLImpl::TextureImpl::pProxyTextureManager.reset();
|
||||
MG_Impl::GLImpl::FramebufferImpl::pDefaultFramebufferInfo.reset();
|
||||
// Must run AFTER pGLContext.reset(). FinalizeProcess -> ShFinalize deletes
|
||||
// glslang's process-wide pool allocator and every cached built-in symbol table,
|
||||
// while the TShader/TProgram objects owned by the shader and program objects
|
||||
// still reference levels adopted from those tables. Finalizing first left live
|
||||
// glslang objects pointing at freed memory for the rest of the teardown.
|
||||
glslang::FinalizeProcess();
|
||||
// Immediately after, and never apart from it: FinalizeProcess just deleted the
|
||||
// built-in symbol tables the prewarm latch stands for, so leaving it set would
|
||||
// make the next Initialize() skip a prewarm it genuinely needs.
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::ResetPrewarmLatch();
|
||||
MG_Backend::gBackendFunctionsTable = {};
|
||||
g_isInitialized = false;
|
||||
if (logLifecycle) {
|
||||
MG_Util::Debug::Close();
|
||||
}
|
||||
|
||||
// TODO: add and use Destroy functions for other subsystems
|
||||
}
|
||||
}
|
||||
|
||||
void Initialize() {
|
||||
if (g_isInitialized) {
|
||||
MGLOG_D("MobileGL already initialized; skipping duplicate Initialize()");
|
||||
return;
|
||||
}
|
||||
|
||||
MG_Util::Debug::InitFile();
|
||||
MGLOG_I("Initializing MobileGL...");
|
||||
MG_ConfigLoader::Init();
|
||||
@@ -27,44 +93,55 @@ namespace MobileGL {
|
||||
MG_Impl::Init();
|
||||
MGLOG_D("MG_Impl initialized");
|
||||
glslang::InitializeProcess();
|
||||
// On the GL thread, before any worker can exist. glslang builds its built-in symbol
|
||||
// tables lazily under a process-wide lock held for the whole build, so without this
|
||||
// the first concurrent compiles of a shaderpack all serialize behind the very first
|
||||
// parse and asynchronous compilation looks like it is doing nothing.
|
||||
//
|
||||
// Gated on the flag, because the problem it solves only exists when there are
|
||||
// workers: with compilation synchronous, nothing ever contends for that lock and the
|
||||
// three throwaway parses buy nothing - they just add to every eglInitialize. Read the
|
||||
// flag here rather than inside PrewarmBuiltins so ShaderCompiler keeps no dependency
|
||||
// on the async subsystem (ProgramUtilTest compiles that file without it).
|
||||
if (MG_Util::Async::AsyncShaderCompileEnabled()) {
|
||||
MG_Util::ShaderTranspiler::ShaderCompiler::PrewarmBuiltins();
|
||||
}
|
||||
MGLOG_D("glslang initialized");
|
||||
g_isInitialized = true;
|
||||
MGLOG_I("MobileGL initialized");
|
||||
}
|
||||
|
||||
void Destroy() {
|
||||
MGLOG_I("MobileGL closing...");
|
||||
glslang::FinalizeProcess();
|
||||
delete MG_State::pGLContext;
|
||||
delete MG_State::pEGLContext;
|
||||
delete MG_Impl::GLImpl::TextureImpl::pProxyTextureManager;
|
||||
delete MG_Impl::GLImpl::FramebufferImpl::pDefaultFramebufferInfo;
|
||||
MG_Util::Debug::Close();
|
||||
|
||||
// TODO: add and use Destroy functions for other subsystems
|
||||
}
|
||||
|
||||
#if defined(__linux__) || defined(__APPLE__)
|
||||
__attribute__((constructor)) static void AutoInit() {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
__attribute__((destructor)) static void AutoDestroy() {
|
||||
Destroy();
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef _WIN32
|
||||
BOOL WINAPI DllMain(HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved) {
|
||||
switch (ul_reason_for_call) {
|
||||
case DLL_PROCESS_ATTACH:
|
||||
Initialize();
|
||||
break;
|
||||
|
||||
case DLL_PROCESS_DETACH:
|
||||
Destroy();
|
||||
break;
|
||||
void EnsureInitialized() {
|
||||
if (g_isInitialized.load(std::memory_order_acquire)) {
|
||||
return;
|
||||
}
|
||||
return TRUE;
|
||||
// Re-entrant call while this thread is already inside Initialize()
|
||||
// (e.g. an init step routing back through a public entry point).
|
||||
if (tl_initializing) {
|
||||
return;
|
||||
}
|
||||
const std::lock_guard<std::mutex> lock(InitMutex());
|
||||
if (g_isInitialized.load(std::memory_order_acquire)) {
|
||||
return;
|
||||
}
|
||||
tl_initializing = true;
|
||||
Initialize();
|
||||
tl_initializing = false;
|
||||
}
|
||||
#endif
|
||||
|
||||
void Destroy() {
|
||||
DestroyImpl(true);
|
||||
}
|
||||
|
||||
// MobileGL's lifecycle is owned entirely by the host-API layers
|
||||
// (EGL/WGL/CGL): initialization happens lazily on the first entry point
|
||||
// via EnsureInitialized(), and full teardown happens deterministically
|
||||
// when the last EGL display is terminated with nothing current (EGLImpl
|
||||
// calls Destroy()). There is intentionally no backend-initializing static
|
||||
// constructor, no static destructor, and no DllMain: the global singletons
|
||||
// use leak-at-exit storage (see GlobalObjects.cpp), so a process that exits
|
||||
// without eglTerminate simply leaks them to the OS instead of running
|
||||
// backend destructors during static teardown. macOS has a lightweight
|
||||
// dyld constructor that installs NSOpenGL dispatch hooks only; full backend
|
||||
// initialization still enters here from the first hooked CGL context.
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -11,6 +11,13 @@
|
||||
|
||||
namespace MobileGL {
|
||||
void Initialize();
|
||||
// Thread-safe, idempotent, and re-entrant wrapper around Initialize().
|
||||
// Host layers (EGL/WGL/CGL entry points) call this lazily on first use so
|
||||
// full backend initialization never depends on ELF/DLL static constructors,
|
||||
// and so a fresh init can follow a full Destroy() (e.g. after the last
|
||||
// eglTerminate). The macOS dyld bootstrap installs only lightweight
|
||||
// NSOpenGL method hooks.
|
||||
void EnsureInitialized();
|
||||
void Destroy();
|
||||
|
||||
namespace MG_Util::Debug {
|
||||
|
||||
@@ -7,18 +7,164 @@
|
||||
// End of Source File Header
|
||||
|
||||
#include "BackendObject.h"
|
||||
#include "MG_Util/Converters/MGToStr/TextureEnumConverter.h"
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <iomanip>
|
||||
#include <sstream>
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
namespace {
|
||||
Bool IsReleaseCurrentRequest(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx) {
|
||||
return dpy == EGL_NO_DISPLAY && draw == EGL_NO_SURFACE && read == EGL_NO_SURFACE && ctx == EGL_NO_CONTEXT;
|
||||
(void)dpy;
|
||||
return draw == EGL_NO_SURFACE && read == EGL_NO_SURFACE && ctx == EGL_NO_CONTEXT;
|
||||
}
|
||||
|
||||
std::thread::id CurrentThreadKey() {
|
||||
return std::this_thread::get_id();
|
||||
}
|
||||
|
||||
const char* GetFormatCapabilitySupportString(const FormatCapabilityCache& cache,
|
||||
SizeT targetIndex,
|
||||
SizeT formatIndex,
|
||||
FormatCapability capability) {
|
||||
if (HasFormatCapability(cache.FullCaps[targetIndex][formatIndex], capability)) return "Full";
|
||||
if (HasFormatCapability(cache.CaveatCaps[targetIndex][formatIndex], capability)) return "Caveat";
|
||||
return "None";
|
||||
}
|
||||
|
||||
SizeT GetPrintedFormatNameWidth() {
|
||||
SizeT width = 0;
|
||||
for (SizeT formatIndex = 0; formatIndex < kFormatCapabilityFormatCount; ++formatIndex) {
|
||||
const auto format = static_cast<TextureInternalFormat>(formatIndex);
|
||||
width = std::max(width, MG_Util::ConvertTextureInternalFormatToString(format).size());
|
||||
}
|
||||
return width;
|
||||
}
|
||||
|
||||
SizeT GetCapabilityColumnWidth(FormatCapability capability) {
|
||||
SizeT width = std::strlen(GetFormatCapabilityName(capability));
|
||||
width = std::max<SizeT>(width, std::strlen("Caveat"));
|
||||
return width;
|
||||
}
|
||||
|
||||
String BuildFormatCapabilityHeader(SizeT formatNameWidth) {
|
||||
std::ostringstream line;
|
||||
line << std::left << std::setw(static_cast<Int>(formatNameWidth)) << "";
|
||||
for (FormatCapability capability : kReportedFormatCapabilities) {
|
||||
line << " | " << std::left << std::setw(static_cast<Int>(GetCapabilityColumnWidth(capability)))
|
||||
<< GetFormatCapabilityName(capability);
|
||||
}
|
||||
return line.str();
|
||||
}
|
||||
|
||||
String BuildFormatCapabilityRow(const FormatCapabilityCache& cache,
|
||||
SizeT targetIndex,
|
||||
SizeT formatIndex,
|
||||
SizeT formatNameWidth) {
|
||||
const auto format = static_cast<TextureInternalFormat>(formatIndex);
|
||||
std::ostringstream line;
|
||||
line << std::left << std::setw(static_cast<Int>(formatNameWidth))
|
||||
<< MG_Util::ConvertTextureInternalFormatToString(format);
|
||||
for (FormatCapability capability : kReportedFormatCapabilities) {
|
||||
line << " | " << std::left << std::setw(static_cast<Int>(GetCapabilityColumnWidth(capability)))
|
||||
<< GetFormatCapabilitySupportString(cache, targetIndex, formatIndex, capability);
|
||||
}
|
||||
return line.str();
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void FormatCapabilityCache::Clear() {
|
||||
for (auto& row : FullCaps) {
|
||||
row.fill(FormatCapabilityFlags{});
|
||||
}
|
||||
for (auto& row : CaveatCaps) {
|
||||
row.fill(FormatCapabilityFlags{});
|
||||
}
|
||||
for (auto& row : SampleCounts) {
|
||||
for (auto& counts : row) {
|
||||
counts.clear();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Bool HasFormatCapability(FormatCapabilityFlags caps, FormatCapability capability) {
|
||||
return static_cast<Bool>(caps & capability);
|
||||
}
|
||||
|
||||
SizeT GetFormatCapabilityTargetIndex(TextureTarget target) {
|
||||
if (target == TextureTarget::Unknown || static_cast<Int>(target) < 0 ||
|
||||
static_cast<SizeT>(target) >= kFormatCapabilityTextureTargetCount) {
|
||||
return kFormatCapabilityTargetCount;
|
||||
}
|
||||
return static_cast<SizeT>(target);
|
||||
}
|
||||
|
||||
SizeT GetRenderbufferFormatCapabilityTargetIndex() {
|
||||
return kFormatCapabilityRenderbufferTargetIndex;
|
||||
}
|
||||
|
||||
const char* GetFormatCapabilityName(FormatCapability capability) {
|
||||
switch (capability) {
|
||||
case FormatCapability::Creatable:
|
||||
return "Creatable";
|
||||
case FormatCapability::Sampled:
|
||||
return "Sampled";
|
||||
case FormatCapability::LinearFilter:
|
||||
return "LinearFilter";
|
||||
case FormatCapability::GenerateMipmap:
|
||||
return "GenerateMipmap";
|
||||
case FormatCapability::TextureGather:
|
||||
return "TextureGather";
|
||||
case FormatCapability::TextureShadow:
|
||||
return "TextureShadow";
|
||||
case FormatCapability::FramebufferRenderable:
|
||||
return "FramebufferRenderable";
|
||||
case FormatCapability::FramebufferLayered:
|
||||
return "FramebufferLayered";
|
||||
case FormatCapability::MultisampleTexture:
|
||||
return "MultisampleTexture";
|
||||
case FormatCapability::MultisampleRenderbuffer:
|
||||
return "MultisampleRenderbuffer";
|
||||
case FormatCapability::ColorAttachment:
|
||||
return "ColorAttachment";
|
||||
case FormatCapability::DepthAttachment:
|
||||
return "DepthAttachment";
|
||||
case FormatCapability::StencilAttachment:
|
||||
return "StencilAttachment";
|
||||
case FormatCapability::TextureBuffer:
|
||||
return "TextureBuffer";
|
||||
}
|
||||
return "Unknown";
|
||||
}
|
||||
|
||||
String GetFormatCapabilityTargetName(SizeT targetIndex) {
|
||||
if (targetIndex == kFormatCapabilityRenderbufferTargetIndex) {
|
||||
return "Renderbuffer";
|
||||
}
|
||||
if (targetIndex >= kFormatCapabilityTextureTargetCount) {
|
||||
return "Unknown";
|
||||
}
|
||||
return MG_Util::ConvertTextureTargetToString(static_cast<TextureTarget>(targetIndex));
|
||||
}
|
||||
|
||||
void PrintFormatCapabilities(const FormatCapabilityCache& cache) {
|
||||
const SizeT formatNameWidth = GetPrintedFormatNameWidth();
|
||||
|
||||
MGLOG_D("Backend format capabilities:");
|
||||
for (SizeT targetIndex = 0; targetIndex < kFormatCapabilityTargetCount; ++targetIndex) {
|
||||
MGLOG_D("");
|
||||
const String targetName = GetFormatCapabilityTargetName(targetIndex);
|
||||
MGLOG_D("- %s", targetName.c_str());
|
||||
const String header = BuildFormatCapabilityHeader(formatNameWidth);
|
||||
MGLOG_D("%s", header.c_str());
|
||||
for (SizeT formatIndex = 0; formatIndex < kFormatCapabilityFormatCount; ++formatIndex) {
|
||||
const String row = BuildFormatCapabilityRow(cache, targetIndex, formatIndex, formatNameWidth);
|
||||
MGLOG_D("%s", row.c_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Bool BackendObject::InitializeEGLDisplay(EGLDisplay dpy, EGLint* major, EGLint* minor) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (dpy == EGL_NO_DISPLAY) {
|
||||
@@ -42,28 +188,116 @@ namespace MobileGL::MG_Backend {
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool BackendObject::CreateEGLWindowSurface(const WindowHandle& handle) {
|
||||
Bool BackendObject::CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
return RegisterEGLWindowSurface(surface, handle) && ActivateEGLSurface(surface);
|
||||
}
|
||||
|
||||
Bool BackendObject::ResizeEGLWindowSurface(EGLSurface surface, Uint32 width, Uint32 height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
auto surfaceIt = m_eglSurfaces.find(surface);
|
||||
if (surfaceIt == m_eglSurfaces.end() || surfaceIt->second.Kind != SurfaceKind::Window) {
|
||||
MGLOG_E("ResizeEGLWindowSurface failed: no window surface is initialized");
|
||||
return false;
|
||||
}
|
||||
surfaceIt->second.Window.Width = width;
|
||||
surfaceIt->second.Window.Height = height;
|
||||
if (m_eglSurface == surface) {
|
||||
m_windowHandle.Width = width;
|
||||
m_windowHandle.Height = height;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool BackendObject::CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
return RegisterEGLPbufferSurface(surface, width, height) && ActivateEGLSurface(surface);
|
||||
}
|
||||
|
||||
Bool BackendObject::RegisterEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_eglDisplayInitialized) {
|
||||
MGLOG_E("CreateEGLWindowSurface failed: EGL display is not initialized");
|
||||
MGLOG_E("RegisterEGLWindowSurface failed: EGL display is not initialized");
|
||||
return false;
|
||||
}
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E("RegisterEGLWindowSurface failed: invalid EGLSurface");
|
||||
return false;
|
||||
}
|
||||
if (handle.Backend == WindowBackend::Unknown || !handle.Handle) {
|
||||
MGLOG_E("CreateEGLWindowSurface failed: invalid native window handle");
|
||||
MGLOG_E("RegisterEGLWindowSurface failed: invalid native window handle");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (m_eglWindowSurfaceInitialized && m_windowHandle.Backend == handle.Backend && m_windowHandle.Handle == handle.Handle) {
|
||||
auto& state = m_eglSurfaces[surface];
|
||||
state = EGLSurfaceState{
|
||||
.Kind = SurfaceKind::Window,
|
||||
.Window = handle,
|
||||
.Width = static_cast<EGLint>(std::max<Uint32>(handle.Width, 1)),
|
||||
.Height = static_cast<EGLint>(std::max<Uint32>(handle.Height, 1)),
|
||||
};
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool BackendObject::RegisterEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_eglDisplayInitialized) {
|
||||
MGLOG_E("RegisterEGLPbufferSurface failed: EGL display is not initialized");
|
||||
return false;
|
||||
}
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
MGLOG_E("RegisterEGLPbufferSurface failed: invalid EGLSurface");
|
||||
return false;
|
||||
}
|
||||
if (width <= 0 || height <= 0) {
|
||||
MGLOG_E("RegisterEGLPbufferSurface failed: invalid size %dx%d", width, height);
|
||||
return false;
|
||||
}
|
||||
|
||||
m_eglSurfaces[surface] = EGLSurfaceState{
|
||||
.Kind = SurfaceKind::Pbuffer,
|
||||
.Width = width,
|
||||
.Height = height,
|
||||
};
|
||||
return true;
|
||||
}
|
||||
|
||||
const BackendObject::EGLSurfaceState* BackendObject::GetRegisteredEGLSurface(EGLSurface surface) const {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
auto surfaceIt = m_eglSurfaces.find(surface);
|
||||
return surfaceIt == m_eglSurfaces.end() ? nullptr : &surfaceIt->second;
|
||||
}
|
||||
|
||||
Bool BackendObject::ActivateEGLSurface(EGLSurface surface) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
const auto* surfaceState = GetRegisteredEGLSurface(surface);
|
||||
if (!surfaceState) {
|
||||
MGLOG_E("ActivateEGLSurface failed: EGL surface is not registered");
|
||||
return false;
|
||||
}
|
||||
if (m_eglSurfaceInitialized && m_eglSurface == surface) {
|
||||
return true;
|
||||
}
|
||||
|
||||
SetWindowHandle(handle);
|
||||
if (!InitWindowSurface()) {
|
||||
MGLOG_E("CreateEGLWindowSurface failed: backend InitWindowSurface failed");
|
||||
if (surfaceState->Kind == SurfaceKind::Window) {
|
||||
SetWindowHandle(surfaceState->Window);
|
||||
if (!InitWindowSurface()) {
|
||||
MGLOG_E("ActivateEGLSurface failed: backend InitWindowSurface failed");
|
||||
return false;
|
||||
}
|
||||
} else if (surfaceState->Kind == SurfaceKind::Pbuffer) {
|
||||
if (!InitPbufferSurface(surfaceState->Width, surfaceState->Height)) {
|
||||
MGLOG_E("ActivateEGLSurface failed: backend InitPbufferSurface failed");
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
MGLOG_E("ActivateEGLSurface failed: unsupported surface kind");
|
||||
return false;
|
||||
}
|
||||
|
||||
m_eglWindowSurfaceInitialized = true;
|
||||
m_eglSurface = surface;
|
||||
m_eglSurfaceInitialized = true;
|
||||
m_eglSurfaceKind = surfaceState->Kind;
|
||||
m_eglCurrentThreads.clear();
|
||||
m_backendCapabilitiesInitialized = false;
|
||||
return true;
|
||||
@@ -73,7 +307,7 @@ namespace MobileGL::MG_Backend {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
const auto threadKey = CurrentThreadKey();
|
||||
if (IsReleaseCurrentRequest(dpy, draw, read, ctx)) {
|
||||
m_eglCurrentThreads.erase(threadKey);
|
||||
ReleaseEGLCurrentThread(threadKey);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -81,8 +315,22 @@ namespace MobileGL::MG_Backend {
|
||||
MGLOG_E("MakeEGLCurrent failed: EGL display mismatch or not initialized");
|
||||
return false;
|
||||
}
|
||||
if (!m_eglWindowSurfaceInitialized) {
|
||||
MGLOG_E("MakeEGLCurrent failed: EGL window surface is not initialized");
|
||||
if (!m_eglSurfaceInitialized) {
|
||||
if (draw != read || !ActivateEGLSurface(draw)) {
|
||||
MGLOG_E("MakeEGLCurrent failed: EGL surface is not initialized");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!GetRegisteredEGLSurface(draw) || !GetRegisteredEGLSurface(read)) {
|
||||
MGLOG_E("MakeEGLCurrent failed: EGL surface is not registered");
|
||||
return false;
|
||||
}
|
||||
if (draw != read) {
|
||||
MGLOG_E("MakeEGLCurrent failed: separate draw/read surfaces are not supported");
|
||||
return false;
|
||||
}
|
||||
if (draw != m_eglSurface && !ActivateEGLSurface(draw)) {
|
||||
MGLOG_E("MakeEGLCurrent failed: EGL surface is not backed by this backend");
|
||||
return false;
|
||||
}
|
||||
if (draw == EGL_NO_SURFACE || read == EGL_NO_SURFACE || ctx == EGL_NO_CONTEXT) {
|
||||
@@ -98,14 +346,23 @@ namespace MobileGL::MG_Backend {
|
||||
m_backendCapabilitiesInitialized = true;
|
||||
}
|
||||
|
||||
m_eglCurrentThreads[threadKey] = true;
|
||||
ReleaseEGLCurrentThread(threadKey);
|
||||
m_eglCurrentThreads[threadKey] = EGLCurrentState{
|
||||
.Display = dpy,
|
||||
.DrawSurface = draw,
|
||||
.ReadSurface = read,
|
||||
.Context = ctx,
|
||||
};
|
||||
return true;
|
||||
}
|
||||
|
||||
void BackendObject::ResetEGLRuntimeState() {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
m_eglWindowSurfaceInitialized = false;
|
||||
m_eglSurfaceInitialized = false;
|
||||
m_backendCapabilitiesInitialized = false;
|
||||
m_eglSurfaceKind = SurfaceKind::None;
|
||||
m_eglSurface = EGL_NO_SURFACE;
|
||||
m_windowHandle = {};
|
||||
m_eglCurrentThreads.clear();
|
||||
}
|
||||
|
||||
@@ -115,11 +372,17 @@ namespace MobileGL::MG_Backend {
|
||||
MGLOG_E("SwapEGLBuffers failed: EGL display mismatch or not initialized");
|
||||
return false;
|
||||
}
|
||||
if (m_eglCurrentThreads.find(CurrentThreadKey()) == m_eglCurrentThreads.end()) {
|
||||
const auto currentIt = m_eglCurrentThreads.find(CurrentThreadKey());
|
||||
if (currentIt == m_eglCurrentThreads.end()) {
|
||||
MGLOG_E("SwapEGLBuffers failed: no current context attached");
|
||||
return false;
|
||||
}
|
||||
if (!m_eglWindowSurfaceInitialized || draw == EGL_NO_SURFACE) {
|
||||
if (currentIt->second.Display != dpy || currentIt->second.DrawSurface != draw ||
|
||||
currentIt->second.Context == EGL_NO_CONTEXT) {
|
||||
MGLOG_E("SwapEGLBuffers failed: draw surface is not current on this thread");
|
||||
return false;
|
||||
}
|
||||
if (!m_eglSurfaceInitialized || draw == EGL_NO_SURFACE || draw != m_eglSurface) {
|
||||
MGLOG_E("SwapEGLBuffers failed: invalid draw surface");
|
||||
return false;
|
||||
}
|
||||
@@ -134,8 +397,99 @@ namespace MobileGL::MG_Backend {
|
||||
return true;
|
||||
}
|
||||
|
||||
void BackendObject::SetEGLSwapInterval(Int interval) {
|
||||
const auto& backendFunctions = GetBackendFunctions();
|
||||
if (backendFunctions.SetSwapInterval) {
|
||||
backendFunctions.SetSwapInterval(interval);
|
||||
}
|
||||
}
|
||||
|
||||
Bool BackendObject::IsEGLSurfaceCurrent(EGLSurface surface) const {
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
return false;
|
||||
}
|
||||
for (const auto& current : m_eglCurrentThreads) {
|
||||
if (current.second.DrawSurface == surface || current.second.ReadSurface == surface) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void BackendObject::DestroyPendingEGLSurfaceIfUnused(EGLSurface surface) {
|
||||
auto surfaceIt = m_eglSurfaces.find(surface);
|
||||
if (surfaceIt == m_eglSurfaces.end() || !surfaceIt->second.DestroyPending ||
|
||||
IsEGLSurfaceCurrent(surface)) {
|
||||
return;
|
||||
}
|
||||
|
||||
m_eglSurfaces.erase(surfaceIt);
|
||||
if (m_eglSurface == surface) {
|
||||
OnEGLSurfaceReleased(surface);
|
||||
ResetEGLRuntimeState();
|
||||
}
|
||||
}
|
||||
|
||||
void BackendObject::ReleaseEGLCurrentThread(const std::thread::id& threadKey) {
|
||||
auto currentIt = m_eglCurrentThreads.find(threadKey);
|
||||
if (currentIt == m_eglCurrentThreads.end()) {
|
||||
return;
|
||||
}
|
||||
|
||||
const EGLSurface drawSurface = currentIt->second.DrawSurface;
|
||||
const EGLSurface readSurface = currentIt->second.ReadSurface;
|
||||
m_eglCurrentThreads.erase(currentIt);
|
||||
DestroyPendingEGLSurfaceIfUnused(drawSurface);
|
||||
DestroyPendingEGLSurfaceIfUnused(readSurface);
|
||||
}
|
||||
|
||||
void BackendObject::ReleaseEGLSurface(EGLSurface surface) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
auto surfaceIt = m_eglSurfaces.find(surface);
|
||||
if (surfaceIt == m_eglSurfaces.end()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsEGLSurfaceCurrent(surface)) {
|
||||
surfaceIt->second.DestroyPending = true;
|
||||
return;
|
||||
}
|
||||
|
||||
m_eglSurfaces.erase(surfaceIt);
|
||||
if (m_eglSurface == surface) {
|
||||
OnEGLSurfaceReleased(surface);
|
||||
ResetEGLRuntimeState();
|
||||
}
|
||||
}
|
||||
|
||||
void BackendObject::ReleaseEGLResources() {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
ResetEGLRuntimeState();
|
||||
m_eglSurfaces.clear();
|
||||
m_eglDisplay = EGL_NO_DISPLAY;
|
||||
m_eglDisplayInitialized = false;
|
||||
}
|
||||
|
||||
void BackendObject::SetWindowHandle(const WindowHandle& handle) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
m_windowHandle = handle;
|
||||
}
|
||||
|
||||
const FormatCapabilityCache& BackendObject::GetFormatCapabilities() const {
|
||||
return m_formatCapabilities;
|
||||
}
|
||||
|
||||
FormatCapabilityCache& BackendObject::MutableFormatCapabilities() {
|
||||
return m_formatCapabilities;
|
||||
}
|
||||
|
||||
Bool BackendObject::InitPbufferSurface(EGLint width, EGLint height) {
|
||||
(void)width;
|
||||
(void)height;
|
||||
return false;
|
||||
}
|
||||
|
||||
void BackendObject::OnEGLSurfaceReleased(EGLSurface surface) {
|
||||
(void)surface;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend
|
||||
|
||||
@@ -8,8 +8,14 @@
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
#include "MG_State/GLState/TextureState/TextureEnum.h"
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_State::GLState {
|
||||
class FramebufferObject;
|
||||
class ITextureObject;
|
||||
}
|
||||
|
||||
enum class BackendType {
|
||||
DirectGLES,
|
||||
DirectVulkan,
|
||||
@@ -18,11 +24,88 @@ namespace MobileGL {
|
||||
};
|
||||
|
||||
namespace MG_Backend {
|
||||
enum class FormatCapability : Uint64 {
|
||||
Creatable = 1ull << 0,
|
||||
|
||||
Sampled = 1ull << 1,
|
||||
LinearFilter = 1ull << 2,
|
||||
GenerateMipmap = 1ull << 3,
|
||||
TextureGather = 1ull << 4,
|
||||
TextureShadow = 1ull << 5,
|
||||
|
||||
FramebufferRenderable = 1ull << 6,
|
||||
FramebufferLayered = 1ull << 7,
|
||||
MultisampleTexture = 1ull << 8,
|
||||
MultisampleRenderbuffer = 1ull << 9,
|
||||
|
||||
ColorAttachment = 1ull << 10,
|
||||
DepthAttachment = 1ull << 11,
|
||||
StencilAttachment = 1ull << 12,
|
||||
|
||||
TextureBuffer = 1ull << 13
|
||||
};
|
||||
|
||||
using FormatCapabilityFlags = Flags<FormatCapability>;
|
||||
|
||||
inline constexpr Array<FormatCapability, 14> kReportedFormatCapabilities = {
|
||||
FormatCapability::Creatable,
|
||||
FormatCapability::Sampled,
|
||||
FormatCapability::LinearFilter,
|
||||
FormatCapability::GenerateMipmap,
|
||||
FormatCapability::TextureGather,
|
||||
FormatCapability::TextureShadow,
|
||||
FormatCapability::FramebufferRenderable,
|
||||
FormatCapability::FramebufferLayered,
|
||||
FormatCapability::MultisampleTexture,
|
||||
FormatCapability::MultisampleRenderbuffer,
|
||||
FormatCapability::ColorAttachment,
|
||||
FormatCapability::DepthAttachment,
|
||||
FormatCapability::StencilAttachment,
|
||||
FormatCapability::TextureBuffer,
|
||||
};
|
||||
|
||||
inline constexpr SizeT kFormatCapabilityTextureTargetCount =
|
||||
static_cast<SizeT>(TextureTarget::TextureTargetCount);
|
||||
inline constexpr SizeT kFormatCapabilityRenderbufferTargetIndex = kFormatCapabilityTextureTargetCount;
|
||||
inline constexpr SizeT kFormatCapabilityTargetCount = kFormatCapabilityTextureTargetCount + 1;
|
||||
inline constexpr SizeT kFormatCapabilityFormatCount =
|
||||
static_cast<SizeT>(TextureInternalFormat::TextureInternalFormatCount);
|
||||
|
||||
using FormatCapabilityTable =
|
||||
Array<Array<FormatCapabilityFlags, kFormatCapabilityFormatCount>, kFormatCapabilityTargetCount>;
|
||||
using FormatSampleCountTable =
|
||||
Array<Array<Vector<Int>, kFormatCapabilityFormatCount>, kFormatCapabilityTargetCount>;
|
||||
|
||||
struct FormatCapabilityCache {
|
||||
FormatCapabilityTable FullCaps{};
|
||||
FormatCapabilityTable CaveatCaps{};
|
||||
FormatSampleCountTable SampleCounts{};
|
||||
|
||||
void Clear();
|
||||
};
|
||||
|
||||
Bool HasFormatCapability(FormatCapabilityFlags caps, FormatCapability capability);
|
||||
SizeT GetFormatCapabilityTargetIndex(TextureTarget target);
|
||||
SizeT GetRenderbufferFormatCapabilityTargetIndex();
|
||||
const char* GetFormatCapabilityName(FormatCapability capability);
|
||||
String GetFormatCapabilityTargetName(SizeT targetIndex);
|
||||
void PrintFormatCapabilities(const FormatCapabilityCache& cache);
|
||||
|
||||
// Opaque backend fence-sync handle, created by GLFunctionsTable::FenceSync
|
||||
// and released by GLFunctionsTable::DeleteSync.
|
||||
using BackendSyncHandle = void*;
|
||||
|
||||
// Opaque backend timer-query handle, created by
|
||||
// GLFunctionsTable::BeginTimeElapsedQuery / QueryCounterTimestamp and
|
||||
// released by GLFunctionsTable::DeleteBackendQuery.
|
||||
using BackendQueryHandle = void*;
|
||||
|
||||
struct GLFunctionsTable {
|
||||
void (*DrawArrays)(GLenum mode, GLint first, GLsizei count);
|
||||
void (*DrawElements)(GLenum mode, GLsizei count, GLenum type, const void* indices);
|
||||
void (*DrawElementsBaseVertex)(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
GLint basevertex);
|
||||
void (*MultiDrawArrays)(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount);
|
||||
void (*MultiDrawElements)(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount);
|
||||
void (*MultiDrawElementsBaseVertex)(GLenum mode, const GLsizei* count, GLenum type,
|
||||
@@ -31,6 +114,10 @@ namespace MobileGL {
|
||||
void (*MultiDrawElementsIndirect)(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount,
|
||||
GLsizei stride);
|
||||
void (*MultiDrawArraysIndirect)(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void (*MultiDrawElementsIndirectCount)(GLenum mode, GLenum type, const void* indirect,
|
||||
GLintptr drawcount, GLsizei maxdrawcount, GLsizei stride);
|
||||
void (*MultiDrawArraysIndirectCount)(GLenum mode, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride);
|
||||
void (*DrawRangeElementsBaseVertex)(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex);
|
||||
void (*DrawRangeElements)(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
@@ -54,29 +141,269 @@ namespace MobileGL {
|
||||
void (*ClearBufferfv)(GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void (*ClearBufferuiv)(GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
void (*ClearBufferiv)(GLenum buffer, GLint drawbuffer, const GLint* value);
|
||||
void (*ClearNamedFramebufferfv)(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void (*ClearNamedFramebufferfi)(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void (*ClearNamedFramebufferiv)(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, const GLint* value);
|
||||
void (*ClearNamedFramebufferuiv)(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
void (*BlitFramebuffer)(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1, GLint dstX0, GLint dstY0,
|
||||
GLint dstX1, GLint dstY1, GLbitfield mask, GLenum filter);
|
||||
void (*BlitNamedFramebuffer)(const SharedPtr<MG_State::GLState::FramebufferObject>& readFramebuffer,
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& drawFramebuffer,
|
||||
GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLbitfield mask, GLenum filter);
|
||||
void (*CopyTexImage2D)(GLenum target, GLint level, GLenum internalformat, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height, GLint border);
|
||||
void (*CopyTexSubImage2D)(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y,
|
||||
GLsizei width, GLsizei height);
|
||||
void (*CopyImageSubData)(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void (*GenerateMipmap)(GLenum target);
|
||||
void (*ReadPixels)(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type,
|
||||
void* pixels);
|
||||
void (*GetTexImage)(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels);
|
||||
void (*GetTextureImage)(const SharedPtr<MG_State::GLState::ITextureObject>& texture,
|
||||
TextureUploadTarget uploadTarget, GLint level, GLenum format, GLenum type,
|
||||
GLsizei bufSize, GLvoid* pixels);
|
||||
void (*DispatchCompute)(GLuint numGroupsX, GLuint numGroupsY, GLuint numGroupsZ);
|
||||
void (*DispatchComputeIndirect)(GLintptr indirect);
|
||||
void (*MemoryBarrier)(GLbitfield barriers);
|
||||
void (*MemoryBarrierByRegion)(GLbitfield barriers);
|
||||
void (*BindImageTexture)(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer,
|
||||
GLenum access, GLenum format);
|
||||
void (*GetIntegeri_v)(GLenum target, GLuint index, GLint* data);
|
||||
void (*GetInteger64i_v)(GLenum target, GLuint index, GLint64* data);
|
||||
void (*GetProgramiv)(GLuint program, GLenum pname, GLint* params);
|
||||
// The GL program interface (glGetProgramInterfaceiv / glGetProgramResource*) is NOT
|
||||
// a backend query: it describes the program the application wrote, in the
|
||||
// application's namespace, which neither backend program is in. It is answered
|
||||
// entirely by MG_Impl/GLImpl/Program/ProgramInterface from the frontend reflection.
|
||||
// Takes the block's GL NAME, not glShaderStorageBlockBinding's index. The index
|
||||
// the application passes is the frontend interface-query enumeration's, and no
|
||||
// backend shares that index space: DirectVulkan enumerates SPIR-V descriptor
|
||||
// bindings and DirectGLES asks a real driver about SPIRV-Cross-generated ESSL.
|
||||
// The name is the one coordinate all three agree on, so the frontend resolves the
|
||||
// index against its own enumeration and each backend maps the name to its own.
|
||||
void (*ShaderStorageBlockBinding)(GLuint program, const GLchar* storageBlockName,
|
||||
GLuint storageBlockBinding);
|
||||
// GL fence sync objects. All entries are optional (may be null); the
|
||||
// frontend then falls back to always-signaled sync semantics.
|
||||
// FenceSync may itself return null when the backend cannot create a
|
||||
// fence right now (e.g. the calling thread does not own the backend
|
||||
// context); the frontend treats such a sync as always signaled.
|
||||
BackendSyncHandle (*FenceSync)();
|
||||
GLenum (*ClientWaitSync)(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout);
|
||||
void (*WaitSync)(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout);
|
||||
void (*DeleteSync)(BackendSyncHandle sync);
|
||||
Bool (*GetSyncStatus)(BackendSyncHandle sync); // true = signaled
|
||||
// GL timer-query objects (GL_ARB_timer_query). All entries are
|
||||
// optional (may be null); the frontend then falls back to zero
|
||||
// results and reports GL_QUERY_COUNTER_BITS == 0.
|
||||
// BeginTimeElapsedQuery / QueryCounterTimestamp may themselves
|
||||
// return null when the backend cannot create a query right now;
|
||||
// the frontend treats such a query as immediately available with
|
||||
// a zero result.
|
||||
// Dynamic support check: true only when the live backend can
|
||||
// actually time at the moment of the call (extension / entry
|
||||
// points / timestamp valid bits are known then, not at table
|
||||
// init). Gates the advertised GL_QUERY_COUNTER_BITS.
|
||||
Bool (*IsTimerQuerySupported)();
|
||||
BackendQueryHandle (*BeginTimeElapsedQuery)(); // starts a TIME_ELAPSED span
|
||||
void (*EndTimeElapsedQuery)(BackendQueryHandle query); // ends the span
|
||||
BackendQueryHandle (*QueryCounterTimestamp)(); // glQueryCounter(GL_TIMESTAMP) one-shot
|
||||
Bool (*IsQueryResultAvailable)(BackendQueryHandle query); // non-blocking
|
||||
// Returns true when a final value was produced (*outNanoseconds
|
||||
// written; the frontend may cache it and release the handle).
|
||||
// Returns false when the result could not be obtained YET - e.g.
|
||||
// a Vulkan wait that refuses to block on a not-yet-submitted
|
||||
// frame serial - in which case the frontend must keep the handle
|
||||
// and leave the query readable later.
|
||||
Bool (*GetQueryResult64)(BackendQueryHandle query, Bool wait, Uint64* outNanoseconds);
|
||||
void (*DeleteBackendQuery)(BackendQueryHandle query);
|
||||
// GL_SAMPLES_PASSED occlusion queries (optional; null = unsupported,
|
||||
// the frontend then rejects the target). Results/deletion flow through
|
||||
// GetQueryResult64 / DeleteBackendQuery like timer queries.
|
||||
BackendQueryHandle (*BeginOcclusionQuery)();
|
||||
void (*EndOcclusionQuery)(BackendQueryHandle query);
|
||||
// Transform feedback primitive queries backed by real GPU query pools
|
||||
// (optional; null = frontend falls back to CPU accounting).
|
||||
BackendQueryHandle (*BeginXfbPrimitivesQuery)(Bool generated);
|
||||
void (*EndXfbPrimitivesQuery)(BackendQueryHandle query);
|
||||
// Transform feedback capture spans, for backends whose own GL/ES driver
|
||||
// performs the capture (DirectGLES). Both optional; null means the backend
|
||||
// drives capture from its draw recording instead (DirectVulkan). End is
|
||||
// called while the frontend capture state is still active, so the backend
|
||||
// can still see the capture program and buffer bindings.
|
||||
// GL_PATCH_VERTICES; ES 3.2 spells it the same way.
|
||||
void (*PatchParameteri)(GLenum pname, GLint value);
|
||||
void (*BeginTransformFeedback)(GLenum primitiveMode);
|
||||
void (*EndTransformFeedback)();
|
||||
// ARB_transform_feedback2. A backend that leaves these null keeps the single
|
||||
// implicit capture span the frontend has always modelled; the frontend state
|
||||
// (paused flag, per-object bindings) is tracked either way.
|
||||
void (*PauseTransformFeedback)();
|
||||
void (*ResumeTransformFeedback)();
|
||||
void (*BindTransformFeedback)(GLuint name);
|
||||
void (*DeleteTransformFeedback)(GLuint name);
|
||||
Int64 (*GetGpuTimestampNs)(); // glGetInteger64v(GL_TIMESTAMP); 0 if unsupported
|
||||
};
|
||||
struct GlobalBackendFunctionsTable {
|
||||
GLFunctionsTable GL;
|
||||
void (*Present)();
|
||||
// Optional: applies the app-requested eglSwapInterval to the native
|
||||
// presentation path (null = backend keeps its own pacing policy).
|
||||
void (*SetSwapInterval)(Int interval);
|
||||
};
|
||||
|
||||
// Coarse GPU vendor identity for gating device-specific quirks. Detected from the
|
||||
// Vulkan physical-device vendorID or the GLES GL_VENDOR/GL_RENDERER strings; stays
|
||||
// Unknown when detection is inconclusive, in which case auto-gated quirks stay off.
|
||||
enum class GpuVendorKind : Uint8 {
|
||||
Unknown = 0,
|
||||
Qualcomm,
|
||||
Arm,
|
||||
Nvidia,
|
||||
Amd,
|
||||
Intel,
|
||||
ImgTec,
|
||||
// Software rasterizers (llvmpipe/lavapipe, SwiftShader).
|
||||
Software,
|
||||
};
|
||||
|
||||
struct DynamicBackendParameters {
|
||||
SizeT UniformBufferOffsetAlignment = 256;
|
||||
// GL_MAX_TEXTURE_MAX_ANISOTROPY_EXT. 1.0 means the backend cannot filter anisotropically,
|
||||
// which is also why the extension is not advertised in that case.
|
||||
Float MaxTextureMaxAnisotropy = 1.0f;
|
||||
Float AliasedLineWidthRangeMin = 1.0f;
|
||||
Float AliasedLineWidthRangeMax = 1.0f;
|
||||
Float SmoothLineWidthRangeMin = 1.0f;
|
||||
Float SmoothLineWidthRangeMax = 1.0f;
|
||||
Float SmoothLineWidthGranularity = 1.0f;
|
||||
Float PointSizeRangeMin = 1.0f;
|
||||
Float PointSizeRangeMax = 1.0f;
|
||||
Float PointSizeGranularity = 1.0f;
|
||||
Int Max3DTextureSize = 16384;
|
||||
Int MaxArrayTextureLayers = 2048;
|
||||
Int MaxCubeMapTextureSize = 16384;
|
||||
Int MaxFramebufferWidth = 16384;
|
||||
Int MaxFramebufferHeight = 16384;
|
||||
Int MaxFramebufferLayers = 2048;
|
||||
Int MaxRenderbufferSize = 16384;
|
||||
Int MaxTextureSize = 16384;
|
||||
Int MaxColorTextureSamples = 1;
|
||||
Int MaxDepthTextureSamples = 1;
|
||||
Int MaxFramebufferSamples = 1;
|
||||
Int MaxIntegerSamples = 1;
|
||||
Int MaxSamples = 1;
|
||||
Int MaxSampleMaskWords = 1;
|
||||
// Tessellation limits; defaults are the GL 4.0 core minimums.
|
||||
Int MaxPatchVertices = 32;
|
||||
Int MaxTessGenLevel = 64;
|
||||
// GL_MIN/MAX_PROGRAM_TEXTURE_GATHER_OFFSET. Defaults are the GL 4.0 core
|
||||
// minimums, which every ES 3.1 driver also guarantees.
|
||||
Int MinProgramTextureGatherOffset = -8;
|
||||
Int MaxProgramTextureGatherOffset = 7;
|
||||
Int MaxTextureImageUnits = 32;
|
||||
Int MaxVertexTextureImageUnits = 32;
|
||||
Int MaxComputeTextureImageUnits = 32;
|
||||
Int MaxCombinedTextureImageUnits = 192;
|
||||
Int MaxVertexAttribs = 16;
|
||||
Int MaxComputeShaderStorageBlocks = 8;
|
||||
Int MaxCombinedShaderStorageBlocks = 32;
|
||||
Int MaxComputeUniformBlocks = 12;
|
||||
Int MaxComputeWorkGroupInvocations = 128;
|
||||
Int MaxShaderStorageBufferBindings = 8;
|
||||
Int MaxTextureBufferSize = 65536;
|
||||
// GL_TEXTURE_BUFFER_OFFSET_ALIGNMENT; 1 means the offset is unconstrained.
|
||||
Int TextureBufferOffsetAlignment = 1;
|
||||
Int MaxUniformBufferBindings = 24;
|
||||
Int MaxUniformBlockSize = 16384;
|
||||
Int MaxImageUnits = 8;
|
||||
Int MaxCombinedImageUniforms = 8;
|
||||
Int MaxVertexImageUniforms = 0;
|
||||
Int MaxGeometryImageUniforms = 0;
|
||||
Int MaxFragmentImageUniforms = 8;
|
||||
Int MaxComputeImageUniforms = 8;
|
||||
Int MaxDrawBuffers = 8;
|
||||
Int MaxColorAttachments = 8;
|
||||
Int MaxClipDistances = 8;
|
||||
Int MaxViewports = 16;
|
||||
Int MaxViewportWidth = 16384;
|
||||
Int MaxViewportHeight = 16384;
|
||||
Float ViewportBoundsRangeMin = 0.0f;
|
||||
Float ViewportBoundsRangeMax = 0.0f;
|
||||
Int ViewportSubpixelBits = 0;
|
||||
// GL 4.x fragment-interpolation offset limits. These defaults are the
|
||||
// core minimums and are replaced by live GLES/Vulkan device limits.
|
||||
Float MinFragmentInterpolationOffset = -0.5f;
|
||||
// For four fractional bits the greatest required legal offset is
|
||||
// 0.5 - 2^-4 = 0.4375 (GL 4.6 table 23.70).
|
||||
Float MaxFragmentInterpolationOffset = 0.4375f;
|
||||
Int FragmentInterpolationOffsetBits = 4;
|
||||
Bool SupportsWideLines = false;
|
||||
// Whether a framebuffer whose depth and stencil attachments are distinct
|
||||
// images can be rendered to. GL only requires support when both refer to the
|
||||
// same image and lets an implementation answer GL_FRAMEBUFFER_UNSUPPORTED
|
||||
// otherwise, which is what DirectVulkan (one combined attachment) and the
|
||||
// real ES drivers behind DirectGLES both do. Defaults to true so a backend
|
||||
// that never sets it keeps the permissive behaviour.
|
||||
Bool SupportsDistinctDepthStencilAttachments = true;
|
||||
// Whether attaching a single layer of a 3D or array texture to a framebuffer actually
|
||||
// renders to that layer. DirectGLES hands the layer straight to
|
||||
// glFramebufferTextureLayer, so it does; DirectVulkan maps a GL layer onto a Vulkan
|
||||
// array layer with no notion of a 3D depth slice, so it does not yet. Defaults to false
|
||||
// so a backend that never sets it gets the conservative answer.
|
||||
// Which layered texture targets this backend can attach ONE layer of to a framebuffer
|
||||
// and then really clear, render and read back that layer. Bit (1u << TextureTarget) is
|
||||
// set for each supported target. Deliberately per target rather than one flag: the three
|
||||
// ways a GL layer maps onto Vulkan are independent capabilities. A 2D or 2D multisample
|
||||
// array layer IS a VkImage array layer and needs nothing extra; a 3D texture's layer is
|
||||
// a z slice, which needs a 2D-array-compatible image and a per-slice clear that
|
||||
// vkCmdClearColorImage cannot express; a cube map array needs an image shape and the
|
||||
// imageCubeArray feature before it can be attached at any layer at all. Defaults to 0 so
|
||||
// a backend that never sets it gets the conservative answer.
|
||||
Uint32 PerLayerFramebufferAttachmentTargets = 0;
|
||||
|
||||
static constexpr Uint32 PerLayerFramebufferAttachmentBit(TextureTarget target) {
|
||||
return (static_cast<Int>(target) >= 0 &&
|
||||
static_cast<Int>(target) < static_cast<Int>(TextureTarget::TextureTargetCount))
|
||||
? (1u << static_cast<Uint32>(target))
|
||||
: 0u;
|
||||
}
|
||||
|
||||
Bool SupportsPerLayerFramebufferAttachment(TextureTarget target) const {
|
||||
const Uint32 bit = PerLayerFramebufferAttachmentBit(target);
|
||||
return bit != 0 && (PerLayerFramebufferAttachmentTargets & bit) != 0;
|
||||
}
|
||||
// Whether glVertexAttribLFormat / glVertexArrayAttribLFormat can be honoured, i.e.
|
||||
// whether a 64-bit vertex attribute can actually reach a shader unconverted. Detected,
|
||||
// never assumed: DirectVulkan needs VkPhysicalDeviceFeatures::shaderFloat64 (the
|
||||
// attribute travels as its 32-bit word pair, so no VK_FORMAT_R64* is required, but the
|
||||
// bitcast result is Float64); DirectGLES can never have it, ESSL having no fp64 type at
|
||||
// all. Defaults to false so a backend that never sets it gets the conservative answer.
|
||||
Bool SupportsFloat64VertexAttributes = false;
|
||||
SizeT MaxShaderStorageBlockSize = 128 * 1024 * 1024;
|
||||
Uint32 SubgroupSize = 0;
|
||||
Uint32 SubgroupSupportedStages = 0;
|
||||
Uint32 SubgroupSupportedFeatures = 0;
|
||||
Bool SubgroupQuadOperationsInAllStages = false;
|
||||
GpuVendorKind GpuVendor = GpuVendorKind::Unknown;
|
||||
};
|
||||
|
||||
enum class WindowBackend {
|
||||
Android,
|
||||
// TODO: X11, Wayland, Windows, macOS, etc.
|
||||
X11,
|
||||
MetalLayer,
|
||||
Win32, // Handle is an HWND
|
||||
// TODO: Wayland, etc.
|
||||
WindowBackendCount,
|
||||
Unknown = -1
|
||||
};
|
||||
@@ -84,6 +411,8 @@ namespace MobileGL {
|
||||
struct WindowHandle {
|
||||
WindowBackend Backend = WindowBackend::Unknown;
|
||||
void* Handle = nullptr;
|
||||
Uint32 Width = 0;
|
||||
Uint32 Height = 0;
|
||||
};
|
||||
|
||||
class BackendObject {
|
||||
@@ -95,9 +424,16 @@ namespace MobileGL {
|
||||
virtual Bool InitWindowSurface() = 0;
|
||||
|
||||
virtual Bool InitializeEGLDisplay(EGLDisplay dpy, EGLint* major, EGLint* minor);
|
||||
virtual Bool CreateEGLWindowSurface(const WindowHandle& handle);
|
||||
virtual Bool CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle);
|
||||
virtual Bool ResizeEGLWindowSurface(EGLSurface surface, Uint32 width, Uint32 height);
|
||||
virtual Bool CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height);
|
||||
virtual Bool MakeEGLCurrent(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx);
|
||||
virtual Bool SwapEGLBuffers(EGLDisplay dpy, EGLSurface draw);
|
||||
// Forwards the app-requested eglSwapInterval to the backend's native
|
||||
// presentation path (no-op for backends without a SetSwapInterval hook).
|
||||
virtual void SetEGLSwapInterval(Int interval);
|
||||
virtual void ReleaseEGLSurface(EGLSurface surface);
|
||||
virtual void ReleaseEGLResources();
|
||||
|
||||
void SetWindowHandle(const WindowHandle& handle);
|
||||
|
||||
@@ -105,18 +441,56 @@ namespace MobileGL {
|
||||
virtual String GetBackendAPIVersionString() const = 0;
|
||||
virtual const GlobalBackendFunctionsTable& GetBackendFunctions() const = 0;
|
||||
virtual const DynamicBackendParameters& GetDynamicParameters() const = 0;
|
||||
const FormatCapabilityCache& GetFormatCapabilities() const;
|
||||
virtual BackendType GetBackendType() const = 0;
|
||||
|
||||
protected:
|
||||
enum class SurfaceKind {
|
||||
None,
|
||||
Window,
|
||||
Pbuffer
|
||||
};
|
||||
|
||||
struct EGLCurrentState {
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLSurface DrawSurface = EGL_NO_SURFACE;
|
||||
EGLSurface ReadSurface = EGL_NO_SURFACE;
|
||||
EGLContext Context = EGL_NO_CONTEXT;
|
||||
};
|
||||
|
||||
struct EGLSurfaceState {
|
||||
SurfaceKind Kind = SurfaceKind::None;
|
||||
Bool DestroyPending = false;
|
||||
WindowHandle Window;
|
||||
EGLint Width = 1;
|
||||
EGLint Height = 1;
|
||||
};
|
||||
|
||||
void ResetEGLRuntimeState();
|
||||
Bool RegisterEGLWindowSurface(EGLSurface surface, const WindowHandle& handle);
|
||||
Bool RegisterEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height);
|
||||
const EGLSurfaceState* GetRegisteredEGLSurface(EGLSurface surface) const;
|
||||
Bool ActivateEGLSurface(EGLSurface surface);
|
||||
virtual Bool InitPbufferSurface(EGLint width, EGLint height);
|
||||
virtual void OnEGLSurfaceReleased(EGLSurface surface);
|
||||
FormatCapabilityCache& MutableFormatCapabilities();
|
||||
|
||||
mutable std::recursive_mutex m_eglStateMutex;
|
||||
FormatCapabilityCache m_formatCapabilities;
|
||||
WindowHandle m_windowHandle;
|
||||
EGLDisplay m_eglDisplay = EGL_NO_DISPLAY;
|
||||
EGLSurface m_eglSurface = EGL_NO_SURFACE;
|
||||
Bool m_eglDisplayInitialized = false;
|
||||
Bool m_eglWindowSurfaceInitialized = false;
|
||||
Bool m_eglSurfaceInitialized = false;
|
||||
Bool m_backendCapabilitiesInitialized = false;
|
||||
UnorderedMap<std::thread::id, Bool> m_eglCurrentThreads;
|
||||
SurfaceKind m_eglSurfaceKind = SurfaceKind::None;
|
||||
UnorderedMap<std::thread::id, EGLCurrentState> m_eglCurrentThreads;
|
||||
UnorderedMap<EGLSurface, EGLSurfaceState> m_eglSurfaces;
|
||||
|
||||
private:
|
||||
Bool IsEGLSurfaceCurrent(EGLSurface surface) const;
|
||||
void DestroyPendingEGLSurfaceIfUnused(EGLSurface surface);
|
||||
void ReleaseEGLCurrentThread(const std::thread::id& threadKey);
|
||||
};
|
||||
} // namespace MG_Backend
|
||||
} // namespace MobileGL
|
||||
|
||||
@@ -13,6 +13,6 @@
|
||||
#include "DirectVulkan/BackendObject_DirectVulkan.h"
|
||||
|
||||
namespace MobileGL::MG_Backend {
|
||||
extern UniquePtr<BackendObject> pActiveBackendObject;
|
||||
extern UniquePtr<BackendObject>& pActiveBackendObject;
|
||||
extern GlobalBackendFunctionsTable gBackendFunctionsTable;
|
||||
} // namespace MobileGL::MG_Backend
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -12,6 +12,12 @@
|
||||
#include <MG_Util/BackendLoaders/OpenGL/Loader.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Populates the same format-capability cache used by backend startup. The caller
|
||||
// must keep the supplied GLES context current for the duration of this call.
|
||||
void PopulateFormatCapabilities(const MG_External::GLESFunctionsTable& gl,
|
||||
const MG_External::GLESCapabilities& capabilities,
|
||||
FormatCapabilityCache& cache);
|
||||
|
||||
class BackendObject_DirectGLES : public BackendObject {
|
||||
public:
|
||||
~BackendObject_DirectGLES() override;
|
||||
@@ -20,9 +26,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Bool InitCapabilities() override;
|
||||
Bool InitWindowSurface() override;
|
||||
Bool InitializeEGLDisplay(EGLDisplay dpy, EGLint* major, EGLint* minor) override;
|
||||
Bool CreateEGLWindowSurface(const WindowHandle& handle) override;
|
||||
Bool CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) override;
|
||||
Bool CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) override;
|
||||
Bool MakeEGLCurrent(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx) override;
|
||||
Bool SwapEGLBuffers(EGLDisplay dpy, EGLSurface draw) override;
|
||||
void ReleaseEGLSurface(EGLSurface surface) override;
|
||||
void ReleaseEGLResources() override;
|
||||
|
||||
const RendererInfo& GetRendererInfo() const override;
|
||||
String GetBackendAPIVersionString() const override;
|
||||
@@ -32,9 +41,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
|
||||
const MG_External::GLESFunctionsTable& GetGLESFunctions() const;
|
||||
const MG_External::EGLFunctionsTable& GetEGLFunctions() const;
|
||||
void ApplyGLESCapabilitiesForTesting(const MG_External::GLESCapabilities& capabilities);
|
||||
|
||||
private:
|
||||
void UpdateDynamicBackendParameters();
|
||||
Bool InitPbufferSurface(EGLint width, EGLint height) override;
|
||||
void OnEGLSurfaceReleased(EGLSurface surface) override;
|
||||
|
||||
Bool m_initialized = false;
|
||||
MG_External::EGLFunctionsTable m_EGLFunctions;
|
||||
@@ -42,4 +54,25 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
MG_External::GLESCapabilities m_GLESCapabilities;
|
||||
DynamicBackendParameters m_dynamicParameters;
|
||||
};
|
||||
|
||||
// Single-source-of-truth helpers shared with the driver POST
|
||||
// (MG_Util/SelfTest/DriverPost.cpp), so the identity strings and extension list
|
||||
// MobileGL reports to applications on this backend cannot drift from what the
|
||||
// POST screen shows.
|
||||
|
||||
// Static identity of the Espryt renderer (renderer/backend names, target GL/GLSL
|
||||
// versions, ExtraVendor). The Extensions vector inside is live backend state that
|
||||
// is reconciled after capability init; callers that need the advertised list for
|
||||
// a known capability set must use BuildAdvertisedExtensions instead.
|
||||
const RendererInfo& GetRendererIdentity();
|
||||
|
||||
// The full OpenGL extension list Espryt advertises (glGetString(GL_EXTENSIONS))
|
||||
// for a device whose timer queries / anisotropic filtering are (or are not) usable.
|
||||
// The MOBILEGL_DISABLE_TIMERQUERY escape hatch is applied inside.
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool timerQueriesSupported, Bool anisotropicFilteringSupported);
|
||||
|
||||
// Format: <OpenGL ES Renderer>, OpenGL ES <Major>.<Minor> — the exact string an
|
||||
// initialized backend returns from GetBackendAPIVersionString (and that ends up
|
||||
// inside the application-visible GL_RENDERER string).
|
||||
String FormatBackendAPIVersionString(const String& glesRendererString, Int glesMajor, Int glesMinor);
|
||||
} // namespace MobileGL::MG_Backend::DirectGLES
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -8,6 +8,8 @@
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
#include <MG_State/GLState/FramebufferState/FramebufferObject.h>
|
||||
#include <MG_State/GLState/TextureState/TextureState.h>
|
||||
#include <MG_State/GLState/SamplerState/SamplerObject.h>
|
||||
#include <MG_Util/BackendLoaders/OpenGL/Loader.h>
|
||||
@@ -17,6 +19,10 @@
|
||||
operation Utils::CheckGLESError();
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// Re-establishes the frontend texture-unit bindings on the native ES context.
|
||||
// Content uploads use scratch bindings, so draws and dispatches call this after
|
||||
// texture synchronization.
|
||||
void BindCurrentTextures();
|
||||
void ClearBufferfi(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void ClearBufferfv(GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void ClearBufferuiv(GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
@@ -25,12 +31,17 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void DrawElements(GLenum mode, GLsizei count, GLenum type, const void* indices);
|
||||
void DrawArrays(GLenum mode, GLint first, GLsizei count);
|
||||
void DrawElementsBaseVertex(GLenum mode, GLsizei count, GLenum type, const GLvoid* indices, GLint basevertex);
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount);
|
||||
void MultiDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount);
|
||||
void MultiDrawElementsBaseVertex(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex);
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount, GLsizei maxdrawcount,
|
||||
GLsizei stride);
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex);
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices);
|
||||
@@ -46,23 +57,177 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
GLuint baseinstance);
|
||||
void DrawArraysInstanced(GLenum mode, GLint first, GLsizei count, GLsizei instancecount);
|
||||
void DrawArraysIndirect(GLenum mode, const void* indirect);
|
||||
void ClearNamedFramebufferfv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void ClearNamedFramebufferfi(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void ClearNamedFramebufferiv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, const GLint* value);
|
||||
void ClearNamedFramebufferuiv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer,
|
||||
GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
void BlitFramebuffer(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1, GLint dstX0, GLint dstY0, GLint dstX1,
|
||||
GLint dstY1, GLbitfield mask, GLenum filter);
|
||||
void BlitNamedFramebuffer(const SharedPtr<MG_State::GLState::FramebufferObject>& readFramebuffer,
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& drawFramebuffer,
|
||||
GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLbitfield mask, GLenum filter);
|
||||
void CopyTexImage2D(GLenum target, GLint level, GLenum internalformat, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height, GLint border);
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height);
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void GenerateMipmap(GLenum target);
|
||||
const GLubyte* GetString(GLenum name);
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels);
|
||||
void GetTexImage(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels);
|
||||
void DispatchCompute(GLuint numGroupsX, GLuint numGroupsY, GLuint numGroupsZ);
|
||||
void DispatchComputeIndirect(GLintptr indirect);
|
||||
void MemoryBarrier(GLbitfield barriers);
|
||||
void MemoryBarrierByRegion(GLbitfield barriers);
|
||||
void BindImageTexture(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer, GLenum access,
|
||||
GLenum format);
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data);
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
void GetProgramiv(GLuint program, GLenum pname, GLint* params);
|
||||
void ShaderStorageBlockBinding(GLuint program, const GLchar* storageBlockName, GLuint storageBlockBinding);
|
||||
Bool InitWindowSurface(NativeWindowType window);
|
||||
Bool InitPbufferSurface(EGLint width, EGLint height);
|
||||
Bool MakeCurrent();
|
||||
Bool ReleaseCurrent();
|
||||
// True when the backend ES context is current on the calling thread, i.e.
|
||||
// immediate buffer ops may issue GL calls right now.
|
||||
Bool IsBackendContextCurrentOnThisThread();
|
||||
// GL fence sync objects, backed by native ES fences. FenceSync returns null
|
||||
// (the frontend then falls back to an always-signaled sync) when the calling
|
||||
// thread does not own the ES context. Waits/queries degrade to "signaled" in
|
||||
// the same situation, and handles created under a since-destroyed ES context
|
||||
// are always treated as signaled.
|
||||
BackendSyncHandle FenceSync();
|
||||
GLenum ClientWaitSync(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout);
|
||||
void WaitSync(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout);
|
||||
void DeleteSync(BackendSyncHandle sync);
|
||||
Bool GetSyncStatus(BackendSyncHandle sync);
|
||||
// True when GL_EXT_disjoint_timer_query and every entry point the timer
|
||||
// hooks below need are present. Also gates the E_GL_ARB_timer_query
|
||||
// advertisement in BackendObject_DirectGLES::InitCapabilities, and is
|
||||
// registered as the GLFunctionsTable::IsTimerQuerySupported hook: a pure
|
||||
// capability read needs no current ES context, and it stays false until
|
||||
// the ES capabilities have been filled in.
|
||||
Bool AreTimerQueriesSupported();
|
||||
// True when the host ES driver can back a GL_TEXTURE_BUFFER at all - ES 3.2 core, or
|
||||
// EXT/OES_texture_buffer, with glTexBuffer resolved. Desktop GL has had buffer textures as
|
||||
// core since 3.1, so the frontend advertises them unconditionally and an app may call
|
||||
// glTexBuffer whenever it likes; this is the only thing standing between that call and a
|
||||
// null entry point. False also means every shader declaring a samplerBuffer is
|
||||
// uncompilable on this driver, which the program build reports by name.
|
||||
Bool AreBufferTexturesSupported();
|
||||
// Human-readable name of the buffer-texture tier for diagnostics and the driver POST:
|
||||
// "core (ES 3.2)", "GL_EXT_texture_buffer", "GL_OES_texture_buffer" or "unsupported".
|
||||
const char* GetBufferTextureTierName();
|
||||
// glTexBuffer / glTexBufferRange through whichever spelling this driver's buffer-texture
|
||||
// support actually ships: the unsuffixed names are ES 3.2 core, while an EXT/OES driver
|
||||
// exports glTexBuffer{,Range}EXT / OES. Callers must have checked
|
||||
// AreBufferTexturesSupported() first. CallTexBufferRange reports whether it could honour
|
||||
// the range - no tier is required to expose the range form, and the whole-buffer form is
|
||||
// the documented fallback.
|
||||
void CallTexBuffer(GLenum target, GLenum internalFormat, GLuint buffer);
|
||||
Bool CallTexBufferRange(GLenum target, GLenum internalFormat, GLuint buffer, GLintptr offset, GLsizeiptr size);
|
||||
// GL timer-query objects, backed by GL_EXT_disjoint_timer_query. The
|
||||
// creators return null (the frontend then falls back to an immediately
|
||||
// available zero result) when the calling thread does not own the ES
|
||||
// context or the extension/entry points are missing, and handles created
|
||||
// under a since-destroyed ES context are always treated as complete with
|
||||
// a zero result (mirrors the fence-sync handles above).
|
||||
BackendQueryHandle BeginTimeElapsedQuery();
|
||||
void EndTimeElapsedQuery(BackendQueryHandle query);
|
||||
BackendQueryHandle QueryCounterTimestamp();
|
||||
// GL_ANY_SAMPLES_PASSED(_CONSERVATIVE) occlusion queries: core ES3, independent of
|
||||
// GL_EXT_disjoint_timer_query and of MOBILEGL_DISABLE_TIMERQUERY. Results/deletion
|
||||
// flow through GetQueryResult64/DeleteBackendQuery like the timer queries above.
|
||||
BackendQueryHandle BeginOcclusionQuery();
|
||||
void EndOcclusionQuery(BackendQueryHandle query);
|
||||
// GL_TRANSFORM_FEEDBACK_PRIMITIVES_WRITTEN / GL_PRIMITIVES_GENERATED, also core ES
|
||||
// (GL_PRIMITIVES_GENERATED from ES 3.2 on). Null when the target is unavailable, in
|
||||
// which case the frontend falls back to counting primitives from the draw calls.
|
||||
BackendQueryHandle BeginXfbPrimitivesQuery(Bool generated);
|
||||
void EndXfbPrimitivesQuery(BackendQueryHandle query);
|
||||
Bool IsQueryResultAvailable(BackendQueryHandle query);
|
||||
// Returns true when a final value landed in *outNanoseconds (a zero for
|
||||
// null or stale-generation handles IS final: the frontend may cache it
|
||||
// and release the handle). Returns false only when the calling thread
|
||||
// does not own the ES context, so the value is genuinely unobtainable
|
||||
// right now; the handle stays alive and readable later.
|
||||
Bool GetQueryResult64(BackendQueryHandle query, Bool wait, Uint64* outNanoseconds);
|
||||
void DeleteBackendQuery(BackendQueryHandle query);
|
||||
Int64 GetGpuTimestampNs();
|
||||
void Present();
|
||||
// Frame-completion watermarks for the buffer-storage pool: CurrentFrameSerial()
|
||||
// is bumped once per Present(); CompletedFrameSerial() is the newest frame whose
|
||||
// GPU work has provably finished (advanced by polling a one-fence-per-frame ring).
|
||||
// A buffer retired during frame N is safe to recycle once CompletedFrameSerial() >= N.
|
||||
Uint64 CurrentFrameSerial();
|
||||
Uint64 CompletedFrameSerial();
|
||||
// Block (up to timeoutNs) until the given frame serial provably retired on the
|
||||
// GPU, using the per-frame fence ring. False when no usable fence covers the
|
||||
// serial (fence-less context, foreign thread, or the slot was recycled);
|
||||
// completion state is untouched in that case.
|
||||
Bool WaitForFrameSerialCompleted(Uint64 serial, Uint64 timeoutNs);
|
||||
// Applies (or defers until the window surface exists) the app-requested
|
||||
// eglSwapInterval on the native EGL surface.
|
||||
void SetSwapInterval(Int interval);
|
||||
void SetEGLFuncsTable(const MG_External::EGLFunctionsTable& eglFuncs);
|
||||
void SetGLESFuncsTable(const MG_External::GLESFunctionsTable& glesFuncs);
|
||||
void SetGLESCapabilities(const MG_External::GLESCapabilities& capabilities);
|
||||
void DestroyEGLContext();
|
||||
|
||||
// Transform feedback capture spans, performed by the real ES driver. The
|
||||
// capture set is declared on the backend program at link time; the driver-side
|
||||
// begin is deferred to the first draw of the span (ES needs the capturing
|
||||
// program current and the capture buffers bound), and the end also mirrors the
|
||||
// captured bytes back into the frontend buffer shadows.
|
||||
void PatchParameteri(GLenum pname, GLint value);
|
||||
|
||||
namespace XfbImpl {
|
||||
Bool AreTransformFeedbacksSupported();
|
||||
// True while a capture span is open on the current transform feedback object
|
||||
// (frontend Begin seen and not paused), whether or not the deferred driver-side
|
||||
// Begin has been issued yet. Draw paths that would restructure the primitive
|
||||
// stream, or that need to dispatch compute mid-draw, decline while it is set.
|
||||
Bool IsCaptureSpanOpen();
|
||||
void BeginTransformFeedback(GLenum primitiveMode);
|
||||
void EndTransformFeedback();
|
||||
void PauseTransformFeedback();
|
||||
void ResumeTransformFeedback();
|
||||
void BindTransformFeedback(GLuint name);
|
||||
void DeleteTransformFeedback(GLuint name);
|
||||
void OnBackendContextDestroyed();
|
||||
} // namespace XfbImpl
|
||||
|
||||
namespace RenderStateImpl {
|
||||
// Pushes the frontend's render-state block to the ES driver, diffed against what was
|
||||
// last pushed.
|
||||
//
|
||||
// `forColorClear` names the CALLER, and the only thing it changes is the colour write
|
||||
// mask handed to the driver. A draw into a colour attachment the backend widened from
|
||||
// three channels to four gets that buffer's alpha channel masked OFF, so nothing can
|
||||
// move the stored alpha away from the 1.0 the application's three-channel format
|
||||
// implies (see FramebufferImpl::g_alphaWidenedDrawBufferMask). A CLEAR is how that 1.0
|
||||
// gets there in the first place, so it must be allowed to write alpha - hence the flag
|
||||
// rather than an unconditional doctoring. It is part of the sync memo, so a clear
|
||||
// followed by a draw re-pushes the mask instead of early-outing on an unchanged
|
||||
// frontend version.
|
||||
//
|
||||
// The application's own colour mask is never modified: glGet(GL_COLOR_WRITEMASK)
|
||||
// answers from the frontend state, which this function only reads.
|
||||
void SyncRenderState(Bool forColorClear = false);
|
||||
void InvalidateSyncedRenderState();
|
||||
} // namespace RenderStateImpl
|
||||
|
||||
extern MG_External::EGLFunctionsTable g_EGLFuncs;
|
||||
extern MG_External::GLESFunctionsTable g_GLESFuncs;
|
||||
extern MG_External::GLESCapabilities g_GLESCapabilities;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,928 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectGLES/MultiDraw.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "MultiDraw.h"
|
||||
#include "Managers.h"
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
using MG_Config::GLESMultiDrawMode;
|
||||
|
||||
namespace {
|
||||
// ---------------------------------------------------------------------------
|
||||
// Batch shape
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
SizeT IndexTypeSize(GLenum type) {
|
||||
switch (type) {
|
||||
case GL_UNSIGNED_BYTE: return 1;
|
||||
case GL_UNSIGNED_SHORT: return 2;
|
||||
case GL_UNSIGNED_INT: return 4;
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
|
||||
// The all-ones value of an index type, which is what GL restarts on once
|
||||
// primitive restart is in play. CheckPrimitiveRestartSupported has already
|
||||
// rejected the arbitrary-index form of GL_PRIMITIVE_RESTART, so an enabled
|
||||
// restart always restarts here and nowhere else.
|
||||
Uint32 RestartSentinelFor(GLenum type) {
|
||||
switch (type) {
|
||||
case GL_UNSIGNED_BYTE: return 0xFFu;
|
||||
case GL_UNSIGNED_SHORT: return 0xFFFFu;
|
||||
default: return 0xFFFFFFFFu;
|
||||
}
|
||||
}
|
||||
|
||||
Bool RestartActive() {
|
||||
return MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::PrimitiveRestart) ||
|
||||
MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::PrimitiveRestartFixedIndex);
|
||||
}
|
||||
|
||||
// Vertices per primitive for the modes whose sub-draws may be concatenated into a
|
||||
// single draw without changing the primitive stream. Zero for strip/loop/fan modes
|
||||
// (concatenation would weld one sub-draw's last primitive to the next sub-draw's
|
||||
// first) and for GL_PATCHES, whose primitive size is dynamic tessellation state.
|
||||
Uint32 ConcatenablePrimitiveSize(GLenum mode) {
|
||||
switch (mode) {
|
||||
case GL_POINTS: return 1;
|
||||
case GL_LINES: return 2;
|
||||
case GL_TRIANGLES: return 3;
|
||||
case GL_LINES_ADJACENCY: return 4;
|
||||
case GL_TRIANGLES_ADJACENCY: return 6;
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Beyond this an emulated batch would ask for a scratch allocation measured in
|
||||
// hundreds of megabytes (and the scratch ring never shrinks again); decline and let
|
||||
// a per-sub-draw tier handle it instead of trying and failing inside the driver.
|
||||
constexpr SizeT kMaxFlattenedIndices = SizeT{1} << 24;
|
||||
|
||||
// The flattening dispatch is one invocation per output index. ES 3.1 only
|
||||
// guarantees 65535 work groups per dimension, and exceeding it makes
|
||||
// glDispatchCompute an INVALID_VALUE no-op - which would leave the draw reading an
|
||||
// uninitialised index buffer rather than failing visibly. Cap the tier there
|
||||
// instead of querying: 4.19M indices is far past any real multi-draw batch, and
|
||||
// beyond it the per-sub-draw tiers are the better answer anyway.
|
||||
constexpr SizeT kComputeWorkGroupSize = 64;
|
||||
constexpr SizeT kMaxComputeWorkGroups = 65535;
|
||||
constexpr SizeT kMaxComputeFlattenedIndices = kMaxComputeWorkGroups * kComputeWorkGroupSize;
|
||||
|
||||
Uint BoundDrawIndirectBufferId() {
|
||||
const auto& indirect =
|
||||
MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
if (!indirect) return 0;
|
||||
const auto* resource = BufferImpl::EnsureBufferResource(indirect);
|
||||
return resource ? resource->id : 0;
|
||||
}
|
||||
|
||||
const SharedPtr<MG_State::GLState::BufferObject>& BoundIndexBuffer() {
|
||||
static const SharedPtr<MG_State::GLState::BufferObject> none;
|
||||
const auto& vao = MG_State::pGLContext->GetBoundVertexArray();
|
||||
if (!vao) return none;
|
||||
return vao->GetIndexBufferBindingSlot().GetBoundObject();
|
||||
}
|
||||
|
||||
// The GL name PrepareForDraw left on GL_ELEMENT_ARRAY_BUFFER, i.e. what a tier
|
||||
// that swaps in a scratch index buffer has to put back. Restoring the exact name
|
||||
// matters beyond tidiness: the VAO twin memoises that it already synced this
|
||||
// index binding and will not re-issue it on the next draw.
|
||||
Uint BoundIndexBufferId() {
|
||||
const auto& ibo = BoundIndexBuffer();
|
||||
if (!ibo) return 0;
|
||||
const auto* resource = BufferImpl::EnsureBufferResource(ibo);
|
||||
return resource ? resource->id : 0;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Scratch GL objects
|
||||
//
|
||||
// All of them belong to the ES context and are abandoned (not deleted) when it
|
||||
// dies, exactly like XfbImpl's scatter buffer: the names are the dead context's
|
||||
// to reclaim, and deleting them would target whatever the successor context
|
||||
// handed out for the same name.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
struct ScratchBuffer {
|
||||
Uint id = 0;
|
||||
SizeT capacity = 0;
|
||||
SizeT cursor = 0; // ring buffers only: next free byte
|
||||
};
|
||||
|
||||
ScratchBuffer g_indirectCommands; // synthesized DrawElementsIndirectCommand array
|
||||
ScratchBuffer g_rebasedIndices; // CPU-rebased index stream
|
||||
ScratchBuffer g_drawInfo; // compute tier: per-sub-draw descriptors
|
||||
ScratchBuffer g_flattenedIndices; // compute tier: flattened index stream
|
||||
|
||||
Uint g_computeProgram = 0;
|
||||
Bool g_computeProgramFailed = false;
|
||||
GLint g_uElementSize = -1;
|
||||
GLint g_uDrawCount = -1;
|
||||
GLint g_uTotalIndices = -1;
|
||||
|
||||
// Reused staging, so a steady stream of batches allocates nothing.
|
||||
Vector<DrawElementsIndirectCommand> g_commandStaging;
|
||||
Vector<Uint32> g_indexStaging;
|
||||
Vector<Uint32> g_drawInfoStaging;
|
||||
Vector<GLint> g_zeroBaseVertices;
|
||||
|
||||
// Everything below stages through GL_ARRAY_BUFFER, the manager-wide staging target
|
||||
// (BufferImpl::TempBufferTarget); binding it disturbs no VAO state.
|
||||
Bool EnsureScratchName(ScratchBuffer& buffer) {
|
||||
if (buffer.id != 0) return true;
|
||||
GLuint id = 0;
|
||||
g_GLESFuncs.glGenBuffers(1, &id);
|
||||
if (id == 0) return false;
|
||||
buffer.id = id;
|
||||
buffer.capacity = 0;
|
||||
buffer.cursor = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Whole-buffer upload, for the two buffers that are read from offset 0 because they
|
||||
// are bound as storage blocks. Respecifies rather than sub-updates: glBufferData
|
||||
// orphans the previous store, so the upload never waits on a dispatch still reading
|
||||
// the old contents out of the same name.
|
||||
Bool UploadScratch(ScratchBuffer& buffer, SizeT bytes, const void* data) {
|
||||
if (bytes == 0) return true;
|
||||
if (!EnsureScratchName(buffer)) return false;
|
||||
BufferImpl::BindBufferId(BufferImpl::TempBufferTarget, buffer.id);
|
||||
// Grow in powers of two so a batch that creeps up in size stops respecifying.
|
||||
SizeT capacity = buffer.capacity == 0 ? bytes : buffer.capacity;
|
||||
while (capacity < bytes) capacity *= 2;
|
||||
g_GLESFuncs.glBufferData(BufferImpl::TempBufferTarget, static_cast<GLsizeiptr>(capacity), nullptr,
|
||||
GL_STREAM_DRAW);
|
||||
buffer.capacity = capacity;
|
||||
buffer.cursor = 0;
|
||||
if (data) {
|
||||
g_GLESFuncs.glBufferSubData(BufferImpl::TempBufferTarget, 0, static_cast<GLsizeiptr>(bytes), data);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// Ring upload, for the buffers whose consumers can address a byte offset (indirect
|
||||
// commands and rewritten index streams). Respecifying per batch is what an
|
||||
// orphan-every-time scheme costs, and on a desktop-class driver that allocation
|
||||
// dominated the tiers that use these buffers - a multi-draw of 32 sub-draws stages
|
||||
// 640 bytes and paid for a fresh store to hold them. Bump-allocating instead means
|
||||
// one respecify per wrap; every byte between two wraps is written exactly once, so
|
||||
// nothing in flight is overwritten, and the wrap itself orphans.
|
||||
constexpr SizeT kRingAlignment = 16; // >= 4, so both command and uint32-index offsets stay legal
|
||||
constexpr SizeT kMinRingBytes = 1u << 16;
|
||||
|
||||
Bool UploadScratchRing(ScratchBuffer& buffer, SizeT bytes, const void* data, SizeT& outOffset) {
|
||||
outOffset = 0;
|
||||
if (bytes == 0) return true;
|
||||
if (!EnsureScratchName(buffer)) return false;
|
||||
BufferImpl::BindBufferId(BufferImpl::TempBufferTarget, buffer.id);
|
||||
|
||||
const SizeT aligned = (bytes + kRingAlignment - 1) & ~(kRingAlignment - 1);
|
||||
if (buffer.capacity < aligned) {
|
||||
SizeT capacity = buffer.capacity == 0 ? kMinRingBytes : buffer.capacity;
|
||||
while (capacity < aligned) capacity *= 2;
|
||||
g_GLESFuncs.glBufferData(BufferImpl::TempBufferTarget, static_cast<GLsizeiptr>(capacity), nullptr,
|
||||
GL_STREAM_DRAW);
|
||||
buffer.capacity = capacity;
|
||||
buffer.cursor = 0;
|
||||
} else if (buffer.cursor + aligned > buffer.capacity) {
|
||||
g_GLESFuncs.glBufferData(BufferImpl::TempBufferTarget, static_cast<GLsizeiptr>(buffer.capacity),
|
||||
nullptr, GL_STREAM_DRAW);
|
||||
buffer.cursor = 0;
|
||||
}
|
||||
|
||||
outOffset = buffer.cursor;
|
||||
if (data) {
|
||||
g_GLESFuncs.glBufferSubData(BufferImpl::TempBufferTarget, static_cast<GLintptr>(outOffset),
|
||||
static_cast<GLsizeiptr>(bytes), data);
|
||||
}
|
||||
buffer.cursor += aligned;
|
||||
return true;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tier resolution
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Best-first, and measured rather than assumed. MobileGlues orders its own Auto
|
||||
// multiindirect -> indirect -> basevertex; on both ES drivers available here that
|
||||
// is backwards, because staging a command buffer per batch costs more than the
|
||||
// driver entries it saves. mc_sodium_multidraw (132 batches x 32 sub-draws),
|
||||
// ns/op, median of three:
|
||||
//
|
||||
// NVIDIA ES 3.2 Mesa llvmpipe ES 3.2
|
||||
// ext n/a 19300
|
||||
// basevertex 2500 25200
|
||||
// multiindirect 5700 27600
|
||||
// drawelements 5600 28700
|
||||
// indirect 5800 31000
|
||||
//
|
||||
// Ring-allocating the command staging (instead of respecifying per batch) was
|
||||
// tried first and moved the indirect tiers by less than noise, so the cost is the
|
||||
// indirect draw path itself, not the upload. Only "ext" - a real multi-draw entry
|
||||
// point rather than an indirect one - actually beats replaying the sub-draws.
|
||||
//
|
||||
// The compute tier is deliberately absent from the ladder: it rewrites the
|
||||
// primitive stream rather than replaying it, and it measured slowest of all here,
|
||||
// so it stays opt-in behind the env knob (the same call MobileGlues makes - its
|
||||
// Auto never selects Compute either).
|
||||
constexpr GLESMultiDrawMode kAutoLadder[] = {
|
||||
GLESMultiDrawMode::Ext, GLESMultiDrawMode::BaseVertex, GLESMultiDrawMode::MultiIndirect,
|
||||
GLESMultiDrawMode::Indirect, GLESMultiDrawMode::DrawElements,
|
||||
};
|
||||
|
||||
Bool SupportsTier(GLESMultiDrawMode tier) {
|
||||
return IsTierSupported(g_GLESCapabilities, g_GLESFuncs, tier);
|
||||
}
|
||||
|
||||
GLESMultiDrawMode g_resolvedTier = GLESMultiDrawMode::Auto;
|
||||
Bool g_tierResolved = false;
|
||||
String g_tierResolution;
|
||||
|
||||
void ResolveTierOnce() {
|
||||
if (g_tierResolved) return;
|
||||
g_tierResolved = true;
|
||||
g_resolvedTier =
|
||||
ResolveTier(g_GLESCapabilities, g_GLESFuncs, MG_Config::Features.EsprytMultiDrawMode,
|
||||
&g_tierResolution);
|
||||
MGLOG_D("DirectGLES multi-draw: %s", g_tierResolution.c_str());
|
||||
}
|
||||
|
||||
// Which tiers have already announced themselves, one bit per GLESMultiDrawMode.
|
||||
// The resolution line above says which tier was CHOSEN; this says which one a
|
||||
// batch actually went through, and the two differ whenever a batch's shape
|
||||
// demotes it. Worth a line each: a multi-draw path that resolves to a tier and
|
||||
// then quietly runs a different one is exactly how "the batch drew nothing"
|
||||
// hides.
|
||||
Uint32 g_announcedTiers = 0;
|
||||
|
||||
void NoteTierExecuted(GLESMultiDrawMode tier) {
|
||||
const Uint32 bit = 1u << static_cast<Uint32>(tier);
|
||||
if (g_announcedTiers & bit) return;
|
||||
g_announcedTiers |= bit;
|
||||
MGLOG_D("DirectGLES multi-draw: first batch executed via tier \"%s\"", TierName(tier));
|
||||
}
|
||||
|
||||
// The tier this particular batch can actually take. A tier is demoted here when
|
||||
// the batch's own shape - not the driver - rules it out; the compute tier keeps
|
||||
// its remaining feasibility checks inside its implementation, where the data it
|
||||
// has to walk is already in hand.
|
||||
GLESMultiDrawMode ResolveTierForBatch(Bool programReadsDrawID, Bool perSubDrawBaseVertex,
|
||||
Bool hasIndexBuffer) {
|
||||
ResolveTierOnce();
|
||||
GLESMultiDrawMode tier = g_resolvedTier;
|
||||
|
||||
// Batched tiers issue one driver entry for the whole batch, so the emulated
|
||||
// gl_DrawID uniform can only hold one value across every sub-draw. A program
|
||||
// that reads gl_DrawID gets an unrolled tier, which feeds each sub-draw its
|
||||
// own index (the spec's value); nothing else observes the difference. The
|
||||
// emulated gl_BaseVertex is one uniform for the same reason, so a batch whose
|
||||
// sub-draws carry their own base vertices unrolls too - even the Ext tier,
|
||||
// which hands the driver the whole basevertex array, can only leave ONE value
|
||||
// in the uniform the shader reads.
|
||||
const Bool batched = tier == GLESMultiDrawMode::Ext || tier == GLESMultiDrawMode::MultiIndirect ||
|
||||
tier == GLESMultiDrawMode::Compute;
|
||||
if (batched && (programReadsDrawID || perSubDrawBaseVertex)) {
|
||||
tier = SupportsTier(GLESMultiDrawMode::BaseVertex) ? GLESMultiDrawMode::BaseVertex
|
||||
: GLESMultiDrawMode::DrawElements;
|
||||
}
|
||||
|
||||
// The indirect tiers describe each sub-draw as an element offset into the
|
||||
// bound element array buffer. A client-memory index array has no such buffer,
|
||||
// and indirect draws are not defined without one.
|
||||
if (!hasIndexBuffer &&
|
||||
(tier == GLESMultiDrawMode::MultiIndirect || tier == GLESMultiDrawMode::Indirect)) {
|
||||
tier = SupportsTier(GLESMultiDrawMode::BaseVertex) ? GLESMultiDrawMode::BaseVertex
|
||||
: GLESMultiDrawMode::DrawElements;
|
||||
}
|
||||
return tier;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Index rewriting, shared by the two tiers that fold base vertices into indices
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Both of those tiers emit GL_UNSIGNED_INT regardless of the source type. Keeping
|
||||
// the source width would be wrong, not merely tight: GL adds baseVertex to the
|
||||
// index at full precision, so a GL_UNSIGNED_SHORT index plus a base vertex past
|
||||
// 65535 addresses a vertex the source type cannot spell. Widening also gives the
|
||||
// rewritten stream a restart sentinel (0xFFFFFFFF) that survives the rebase.
|
||||
void RebaseIndices(const Uint8* source, SizeT sourceIndexCount, SizeT indexSize, Int32 baseVertex,
|
||||
Bool restartActive, Uint32 restartSentinel, Uint32* out) {
|
||||
const Uint32 baseVertexBits = static_cast<Uint32>(baseVertex);
|
||||
for (SizeT i = 0; i < sourceIndexCount; ++i) {
|
||||
Uint32 value = 0;
|
||||
switch (indexSize) {
|
||||
case 1: value = source[i]; break;
|
||||
case 2: {
|
||||
Uint16 narrow = 0;
|
||||
std::memcpy(&narrow, source + i * 2, sizeof(narrow));
|
||||
value = narrow;
|
||||
break;
|
||||
}
|
||||
default: std::memcpy(&value, source + i * 4, sizeof(value)); break;
|
||||
}
|
||||
// Unsigned wraparound is the defined behaviour for a negative base vertex.
|
||||
out[i] = (restartActive && value == restartSentinel) ? 0xFFFFFFFFu : value + baseVertexBits;
|
||||
}
|
||||
}
|
||||
|
||||
// CPU-readable bytes of one sub-draw's indices, from the frontend shadow of the
|
||||
// bound index buffer or straight from the client array. Null when the sub-draw
|
||||
// would read outside the buffer.
|
||||
const Uint8* ResolveSubDrawIndices(const SharedPtr<MG_State::GLState::BufferObject>& indexBuffer,
|
||||
const Uint8* indexBufferBytes, SizeT indexBufferSize, const void* indices,
|
||||
SizeT indexCount, SizeT indexSize) {
|
||||
if (!indexBuffer) {
|
||||
return static_cast<const Uint8*>(indices);
|
||||
}
|
||||
if (!indexBufferBytes) return nullptr;
|
||||
const SizeT byteOffset = reinterpret_cast<SizeT>(indices);
|
||||
const SizeT byteEnd = byteOffset + indexCount * indexSize;
|
||||
if (byteEnd > indexBufferSize || byteEnd < byteOffset) return nullptr;
|
||||
return indexBufferBytes + byteOffset;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tier: Ext - one glMultiDrawElementsBaseVertexEXT
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunExt(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices, GLsizei drawcount,
|
||||
const GLint* basevertex) {
|
||||
if (!SupportsTier(GLESMultiDrawMode::Ext)) return false;
|
||||
const GLint* baseVertices = basevertex;
|
||||
if (!baseVertices) {
|
||||
// glMultiDrawElements: every base vertex is 0, but the entry point still
|
||||
// wants an array. One permanently-zero vector serves every such batch.
|
||||
if (g_zeroBaseVertices.size() < static_cast<SizeT>(drawcount)) {
|
||||
g_zeroBaseVertices.resize(static_cast<SizeT>(drawcount), 0);
|
||||
}
|
||||
baseVertices = g_zeroBaseVertices.data();
|
||||
}
|
||||
g_GLESFuncs.glMultiDrawElementsBaseVertexEXT(mode, count, type, indices, drawcount, baseVertices);
|
||||
NoteTierExecuted(GLESMultiDrawMode::Ext);
|
||||
return true;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tiers: MultiIndirect / Indirect - synthesized indirect commands
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunIndirect(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool batched, Bool feedDrawID,
|
||||
Bool feedBaseVertex) {
|
||||
if (!SupportsTier(batched ? GLESMultiDrawMode::MultiIndirect : GLESMultiDrawMode::Indirect)) return false;
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return false;
|
||||
// Indirect commands address indices as an element offset into the bound element
|
||||
// array buffer, and an indirect draw is not defined without one.
|
||||
const auto& indexBuffer = BoundIndexBuffer();
|
||||
if (!indexBuffer) return false;
|
||||
|
||||
g_commandStaging.resize(static_cast<SizeT>(drawcount));
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
const SizeT byteOffset = reinterpret_cast<SizeT>(indices[i]);
|
||||
// firstIndex counts elements, so an offset that is not a whole number of
|
||||
// them cannot be expressed as a command at all.
|
||||
if (byteOffset % indexSize != 0) return false;
|
||||
auto& command = g_commandStaging[static_cast<SizeT>(i)];
|
||||
command.count = count[i] > 0 ? static_cast<Uint32>(count[i]) : 0u;
|
||||
command.instanceCount = 1;
|
||||
command.firstIndex = static_cast<Uint32>(byteOffset / indexSize);
|
||||
command.baseVertex = basevertex ? basevertex[i] : 0;
|
||||
command.baseInstance = 0;
|
||||
}
|
||||
|
||||
const SizeT commandBytes = g_commandStaging.size() * sizeof(DrawElementsIndirectCommand);
|
||||
SizeT commandBase = 0;
|
||||
if (!UploadScratchRing(g_indirectCommands, commandBytes, g_commandStaging.data(), commandBase)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Every synthesized command carries baseInstance 0. Say so through the direct
|
||||
// path, which also clears the indirect-params word index a preceding real
|
||||
// indirect draw may have left pointing into its own command buffer.
|
||||
SetCurrentBaseInstance(0);
|
||||
|
||||
const Uint previousIndirectBinding = BoundDrawIndirectBufferId();
|
||||
BufferImpl::BindBufferId(GL_DRAW_INDIRECT_BUFFER, g_indirectCommands.id);
|
||||
if (batched) {
|
||||
g_GLESFuncs.glMultiDrawElementsIndirectEXT(mode, type, reinterpret_cast<const void*>(commandBase),
|
||||
drawcount, 0);
|
||||
} else {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
const SizeT commandOffset = commandBase + static_cast<SizeT>(i) * sizeof(DrawElementsIndirectCommand);
|
||||
g_GLESFuncs.glDrawElementsIndirect(mode, type, reinterpret_cast<const void*>(commandOffset));
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
}
|
||||
BufferImpl::BindBufferId(GL_DRAW_INDIRECT_BUFFER, previousIndirectBinding);
|
||||
NoteTierExecuted(batched ? GLESMultiDrawMode::MultiIndirect : GLESMultiDrawMode::Indirect);
|
||||
return true;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tier: BaseVertex - the per-sub-draw replay
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunBaseVertexLoop(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID, Bool feedBaseVertex) {
|
||||
if (!SupportsTier(GLESMultiDrawMode::BaseVertex)) return false;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
g_GLESFuncs.glDrawElementsBaseVertex(mode, count[i], type, indices[i],
|
||||
basevertex ? basevertex[i] : 0);
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
NoteTierExecuted(GLESMultiDrawMode::BaseVertex);
|
||||
return true;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tier: DrawElements - base vertices folded into a scratch index stream
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunRebasedDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID,
|
||||
Bool feedBaseVertex) {
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return false;
|
||||
|
||||
SizeT total = 0;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] > 0) total += static_cast<SizeT>(count[i]);
|
||||
}
|
||||
if (total == 0) return true;
|
||||
if (total > kMaxFlattenedIndices) return false;
|
||||
|
||||
const auto& indexBuffer = BoundIndexBuffer();
|
||||
const Uint8* indexBufferBytes = nullptr;
|
||||
SizeT indexBufferSize = 0;
|
||||
if (indexBuffer) {
|
||||
// The shadow is the source of truth for CPU reads, but a persistent map or
|
||||
// a shader write may have moved past it since the last sync.
|
||||
indexBuffer->SyncPersistentMappedRange();
|
||||
indexBuffer->SyncGpuWrites();
|
||||
indexBufferBytes = indexBuffer->MappedData();
|
||||
indexBufferSize = indexBuffer->GetSize();
|
||||
}
|
||||
|
||||
const Bool restartActive = RestartActive();
|
||||
const Uint32 restartSentinel = RestartSentinelFor(type);
|
||||
g_indexStaging.resize(total);
|
||||
SizeT cursor = 0;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
const SizeT subDrawCount = static_cast<SizeT>(count[i]);
|
||||
const Uint8* source = ResolveSubDrawIndices(indexBuffer, indexBufferBytes, indexBufferSize, indices[i],
|
||||
subDrawCount, indexSize);
|
||||
if (!source) {
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (drawelements tier): sub-draw %d reads outside the bound index "
|
||||
"buffer; skipping the batch",
|
||||
i);
|
||||
return false;
|
||||
}
|
||||
RebaseIndices(source, subDrawCount, indexSize, basevertex ? basevertex[i] : 0, restartActive,
|
||||
restartSentinel, g_indexStaging.data() + cursor);
|
||||
cursor += subDrawCount;
|
||||
}
|
||||
|
||||
SizeT indexBase = 0;
|
||||
if (!UploadScratchRing(g_rebasedIndices, total * sizeof(Uint32), g_indexStaging.data(), indexBase)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Uint previousIndexBinding = BoundIndexBufferId();
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, g_rebasedIndices.id);
|
||||
cursor = 0;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
// The base vertex is folded into the rewritten index stream here, so the
|
||||
// driver sees none - but gl_BaseVertex still has to report the value the
|
||||
// application passed for this sub-draw.
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
g_GLESFuncs.glDrawElements(mode, count[i], GL_UNSIGNED_INT,
|
||||
reinterpret_cast<const void*>(indexBase + cursor * sizeof(Uint32)));
|
||||
cursor += static_cast<SizeT>(count[i]);
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, previousIndexBinding);
|
||||
NoteTierExecuted(GLESMultiDrawMode::DrawElements);
|
||||
return true;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tier: Compute - the whole batch flattened into one rebased index stream
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// One index per invocation. The sub-draw an output slot belongs to is found by
|
||||
// binary search over the inclusive prefix sums of the sub-draw counts, which is
|
||||
// why the descriptors are sorted by construction. Sub-draws with a zero count
|
||||
// repeat the previous prefix sum and are therefore skipped by the search.
|
||||
//
|
||||
// Three storage blocks, not the five the shape suggests: ES 3.1 only guarantees
|
||||
// four per compute stage, so the per-sub-draw descriptors share one buffer.
|
||||
constexpr const char* kFlattenComputeSource = R"(#version 310 es
|
||||
layout(local_size_x = 64) in;
|
||||
|
||||
uniform uint uElementSize;
|
||||
uniform uint uDrawCount;
|
||||
uniform uint uTotalIndices;
|
||||
|
||||
layout(std430, binding = 0) readonly buffer SourceIndices { uint sourceWords[]; };
|
||||
layout(std430, binding = 1) readonly buffer DrawInfo { uint drawInfo[]; };
|
||||
layout(std430, binding = 2) writeonly buffer FlatIndices { uint flatIndices[]; };
|
||||
|
||||
uint ReadSourceIndex(uint element) {
|
||||
if (uElementSize == 4u) {
|
||||
return sourceWords[element];
|
||||
}
|
||||
if (uElementSize == 2u) {
|
||||
uint word = sourceWords[element >> 1u];
|
||||
return (word >> ((element & 1u) * 16u)) & 0xFFFFu;
|
||||
}
|
||||
uint word = sourceWords[element >> 2u];
|
||||
return (word >> ((element & 3u) * 8u)) & 0xFFu;
|
||||
}
|
||||
|
||||
void main() {
|
||||
uint outIndex = gl_GlobalInvocationID.x;
|
||||
if (outIndex >= uTotalIndices) {
|
||||
return;
|
||||
}
|
||||
|
||||
uint low = 0u;
|
||||
uint high = uDrawCount - 1u;
|
||||
while (low < high) {
|
||||
uint mid = low + (high - low) / 2u;
|
||||
if (drawInfo[mid * 3u + 2u] > outIndex) {
|
||||
high = mid;
|
||||
} else {
|
||||
low = mid + 1u;
|
||||
}
|
||||
}
|
||||
|
||||
uint localIndex = outIndex - (low == 0u ? 0u : drawInfo[(low - 1u) * 3u + 2u]);
|
||||
// Unsigned wraparound is the defined behaviour for a negative base vertex. No
|
||||
// restart sentinel handling: the tier declines outright while restart is enabled.
|
||||
flatIndices[outIndex] = ReadSourceIndex(localIndex + drawInfo[low * 3u]) + drawInfo[low * 3u + 1u];
|
||||
}
|
||||
)";
|
||||
|
||||
struct FlattenedStream {
|
||||
Uint bufferId = 0;
|
||||
SizeT indexCount = 0;
|
||||
};
|
||||
|
||||
Bool EnsureComputeProgram() {
|
||||
if (g_computeProgram != 0) return true;
|
||||
if (g_computeProgramFailed) return false;
|
||||
g_computeProgramFailed = true; // cleared again only on a complete success
|
||||
|
||||
const GLuint shader = g_GLESFuncs.glCreateShader(GL_COMPUTE_SHADER);
|
||||
if (shader == 0) {
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): glCreateShader(GL_COMPUTE_SHADER) failed");
|
||||
return false;
|
||||
}
|
||||
const char* source = kFlattenComputeSource;
|
||||
g_GLESFuncs.glShaderSource(shader, 1, &source, nullptr);
|
||||
g_GLESFuncs.glCompileShader(shader);
|
||||
GLint status = GL_FALSE;
|
||||
g_GLESFuncs.glGetShaderiv(shader, GL_COMPILE_STATUS, &status);
|
||||
if (status != GL_TRUE) {
|
||||
char log[1024] = {};
|
||||
g_GLESFuncs.glGetShaderInfoLog(shader, sizeof(log) - 1, nullptr, log);
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): index-flattening shader failed to compile: %s", log);
|
||||
g_GLESFuncs.glDeleteShader(shader);
|
||||
return false;
|
||||
}
|
||||
|
||||
const GLuint program = g_GLESFuncs.glCreateProgram();
|
||||
if (program == 0) {
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): glCreateProgram failed");
|
||||
g_GLESFuncs.glDeleteShader(shader);
|
||||
return false;
|
||||
}
|
||||
g_GLESFuncs.glAttachShader(program, shader);
|
||||
g_GLESFuncs.glLinkProgram(program);
|
||||
g_GLESFuncs.glDeleteShader(shader);
|
||||
g_GLESFuncs.glGetProgramiv(program, GL_LINK_STATUS, &status);
|
||||
if (status != GL_TRUE) {
|
||||
char log[1024] = {};
|
||||
g_GLESFuncs.glGetProgramInfoLog(program, sizeof(log) - 1, nullptr, log);
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw (compute tier): index-flattening program failed to link: %s", log);
|
||||
g_GLESFuncs.glDeleteProgram(program);
|
||||
return false;
|
||||
}
|
||||
|
||||
g_computeProgram = program;
|
||||
g_uElementSize = g_GLESFuncs.glGetUniformLocation(program, "uElementSize");
|
||||
g_uDrawCount = g_GLESFuncs.glGetUniformLocation(program, "uDrawCount");
|
||||
g_uTotalIndices = g_GLESFuncs.glGetUniformLocation(program, "uTotalIndices");
|
||||
g_computeProgramFailed = false;
|
||||
MGLOG_D("DirectGLES multi-draw: index-flattening compute program ready (id %u)", program);
|
||||
return true;
|
||||
}
|
||||
|
||||
// Builds the flattened stream, or leaves `out` empty when this batch's shape rules
|
||||
// the tier out. Runs BEFORE PrepareForDraw - see the call site - so it may leave
|
||||
// the compute program current and the first storage points unbound; the
|
||||
// preparation that follows re-establishes both.
|
||||
void FlattenWithCompute(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, FlattenedStream& out) {
|
||||
if (!SupportsTier(GLESMultiDrawMode::Compute)) return;
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return;
|
||||
|
||||
// Merging sub-draws into a single draw only reproduces the original primitive
|
||||
// stream for list-shaped modes: a strip, loop or fan would gain primitives
|
||||
// spanning the seam between two sub-draws.
|
||||
const Uint32 primitiveSize = ConcatenablePrimitiveSize(mode);
|
||||
if (primitiveSize == 0) return;
|
||||
|
||||
// Primitive restart defeats the whole-multiple-of-a-primitive argument below,
|
||||
// even for a list mode. A restart ends the current primitive, so a sub-draw of
|
||||
// six GL_TRIANGLES indices with a restart after the third emits ONE triangle
|
||||
// and drops the two leftover vertices - and once concatenated those leftovers
|
||||
// find a third vertex in the next sub-draw and become a triangle that GL never
|
||||
// draws. Splicing separator sentinels into the flattened stream could fix it,
|
||||
// at the cost of a per-sub-draw offset the prefix-sum layout does not carry;
|
||||
// declining is the honest trade for a tier that is already opt-in.
|
||||
if (RestartActive()) return;
|
||||
|
||||
// The shader reads the source indices as a storage buffer, so there has to be
|
||||
// a real buffer to read - a client-memory index array has none.
|
||||
const auto& indexBuffer = BoundIndexBuffer();
|
||||
if (!indexBuffer) return;
|
||||
|
||||
// A dispatch inside an open capture span is not legal, and the span would also
|
||||
// observe one merged draw rather than the batch it asked for.
|
||||
if (XfbImpl::IsCaptureSpanOpen()) return;
|
||||
|
||||
auto* sourceResource = BufferImpl::EnsureBufferResource(indexBuffer);
|
||||
if (!sourceResource || sourceResource->id == 0) return;
|
||||
const SizeT sourceSize = indexBuffer->GetSize();
|
||||
// std430 addresses the source as uint[]; a tail shorter than a word is not
|
||||
// reachable, so a narrow index type needs a word-multiple buffer.
|
||||
if (indexSize < 4 && (sourceSize % 4) != 0) return;
|
||||
|
||||
g_drawInfoStaging.resize(3 * static_cast<SizeT>(drawcount));
|
||||
SizeT total = 0;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
const SizeT subDrawCount = count[i] > 0 ? static_cast<SizeT>(count[i]) : 0;
|
||||
// GL drops a trailing partial primitive per sub-draw; concatenation would
|
||||
// instead splice it onto the next sub-draw's first vertices.
|
||||
if (subDrawCount % primitiveSize != 0) return;
|
||||
const SizeT byteOffset = reinterpret_cast<SizeT>(indices[i]);
|
||||
if (byteOffset % indexSize != 0) return;
|
||||
if (subDrawCount != 0) {
|
||||
const SizeT byteEnd = byteOffset + subDrawCount * indexSize;
|
||||
if (byteEnd > sourceSize || byteEnd < byteOffset) return;
|
||||
}
|
||||
total += subDrawCount;
|
||||
if (total > kMaxComputeFlattenedIndices) return;
|
||||
const SizeT slot = 3 * static_cast<SizeT>(i);
|
||||
g_drawInfoStaging[slot] = static_cast<Uint32>(byteOffset / indexSize);
|
||||
g_drawInfoStaging[slot + 1] = static_cast<Uint32>(basevertex ? basevertex[i] : 0);
|
||||
g_drawInfoStaging[slot + 2] = static_cast<Uint32>(total);
|
||||
}
|
||||
if (total == 0) return; // nothing to draw; the ordinary tiers no-op just as well
|
||||
|
||||
if (!EnsureComputeProgram()) return;
|
||||
if (!UploadScratch(g_drawInfo, g_drawInfoStaging.size() * sizeof(Uint32), g_drawInfoStaging.data())) {
|
||||
return;
|
||||
}
|
||||
if (!UploadScratch(g_flattenedIndices, total * sizeof(Uint32), nullptr)) return;
|
||||
|
||||
BufferImpl::BindBufferBaseCached(GL_SHADER_STORAGE_BUFFER, 0, sourceResource->id);
|
||||
BufferImpl::BindBufferBaseCached(GL_SHADER_STORAGE_BUFFER, 1, g_drawInfo.id);
|
||||
BufferImpl::BindBufferBaseCached(GL_SHADER_STORAGE_BUFFER, 2, g_flattenedIndices.id);
|
||||
|
||||
g_GLESFuncs.glUseProgram(g_computeProgram);
|
||||
PrgramImpl::g_lastUsedBackendProgramId = g_computeProgram;
|
||||
if (g_uElementSize >= 0) g_GLESFuncs.glUniform1ui(g_uElementSize, static_cast<GLuint>(indexSize));
|
||||
if (g_uDrawCount >= 0) g_GLESFuncs.glUniform1ui(g_uDrawCount, static_cast<GLuint>(drawcount));
|
||||
if (g_uTotalIndices >= 0) g_GLESFuncs.glUniform1ui(g_uTotalIndices, static_cast<GLuint>(total));
|
||||
|
||||
g_GLESFuncs.glDispatchCompute(
|
||||
static_cast<GLuint>((total + kComputeWorkGroupSize - 1) / kComputeWorkGroupSize), 1, 1);
|
||||
g_GLESFuncs.glMemoryBarrier(GL_SHADER_STORAGE_BARRIER_BIT | GL_ELEMENT_ARRAY_BARRIER_BIT);
|
||||
|
||||
// Hand the storage points back to their GL default. PrepareForDraw re-syncs
|
||||
// only the points the app has actually touched, so leaving a scratch buffer on
|
||||
// an untouched point would keep it visible to the next shader that declares one.
|
||||
for (Uint point = 0; point < 3; ++point) {
|
||||
BufferImpl::BindBufferBaseCached(GL_SHADER_STORAGE_BUFFER, point, 0);
|
||||
}
|
||||
|
||||
NoteTierExecuted(GLESMultiDrawMode::Compute);
|
||||
out.bufferId = g_flattenedIndices.id;
|
||||
out.indexCount = total;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// -------------------------------------------------------------------------------
|
||||
// Public surface
|
||||
// -------------------------------------------------------------------------------
|
||||
|
||||
Bool IsTierSupported(const MG_External::GLESCapabilities& caps, const MG_External::GLESFunctionsTable& funcs,
|
||||
GLESMultiDrawMode tier) {
|
||||
const Bool esAtLeast31 =
|
||||
caps.GLESVersion.Major > 3 || (caps.GLESVersion.Major == 3 && caps.GLESVersion.Minor >= 1);
|
||||
switch (tier) {
|
||||
case GLESMultiDrawMode::Ext:
|
||||
return caps.SupportsMultiDrawElementsBaseVertex;
|
||||
case GLESMultiDrawMode::MultiIndirect:
|
||||
return caps.SupportsMultiDrawIndirect && esAtLeast31 && funcs.glDrawElementsIndirect != nullptr;
|
||||
case GLESMultiDrawMode::Indirect:
|
||||
return esAtLeast31 && funcs.glDrawElementsIndirect != nullptr;
|
||||
case GLESMultiDrawMode::BaseVertex:
|
||||
return caps.SupportsDrawElementsBaseVertex;
|
||||
case GLESMultiDrawMode::DrawElements:
|
||||
// Plain glDrawElements over a rewritten index stream: ES 2 core, so this is
|
||||
// the floor every other tier can fall back to.
|
||||
return true;
|
||||
case GLESMultiDrawMode::Compute:
|
||||
// Three storage blocks, which is inside the four ES 3.1 guarantees per stage.
|
||||
return caps.SupportsComputeShader && caps.MaxComputeShaderStorageBlocks >= 3 &&
|
||||
funcs.glBindBufferBase != nullptr;
|
||||
case GLESMultiDrawMode::Auto:
|
||||
break;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
GLESMultiDrawMode ResolveTier(const MG_External::GLESCapabilities& caps,
|
||||
const MG_External::GLESFunctionsTable& funcs, GLESMultiDrawMode requested,
|
||||
String* explanation) {
|
||||
const auto bestAuto = [&]() {
|
||||
for (const GLESMultiDrawMode tier : kAutoLadder) {
|
||||
if (IsTierSupported(caps, funcs, tier)) return tier;
|
||||
}
|
||||
return GLESMultiDrawMode::DrawElements;
|
||||
};
|
||||
|
||||
GLESMultiDrawMode resolved = GLESMultiDrawMode::DrawElements;
|
||||
String line;
|
||||
if (requested == GLESMultiDrawMode::Auto) {
|
||||
resolved = bestAuto();
|
||||
line = String("auto -> ") + TierName(resolved);
|
||||
} else if (IsTierSupported(caps, funcs, requested)) {
|
||||
resolved = requested;
|
||||
line = String("MOBILEGL_ESPRYT_MULTIDRAW_MODE=") + TierName(requested) + " -> " + TierName(resolved);
|
||||
} else {
|
||||
resolved = bestAuto();
|
||||
line = String("MOBILEGL_ESPRYT_MULTIDRAW_MODE=") + TierName(requested) +
|
||||
" requested but unsupported by this driver -> " + TierName(resolved);
|
||||
}
|
||||
|
||||
if (explanation) {
|
||||
String supported;
|
||||
for (const GLESMultiDrawMode tier : kAutoLadder) {
|
||||
if (!IsTierSupported(caps, funcs, tier)) continue;
|
||||
if (!supported.empty()) supported += ", ";
|
||||
supported += TierName(tier);
|
||||
}
|
||||
if (IsTierSupported(caps, funcs, GLESMultiDrawMode::Compute)) {
|
||||
supported += supported.empty() ? "compute (opt-in)" : ", compute (opt-in)";
|
||||
}
|
||||
*explanation = line + " (driver supports: " + supported + ")";
|
||||
}
|
||||
return resolved;
|
||||
}
|
||||
|
||||
const char* TierName(GLESMultiDrawMode tier) {
|
||||
switch (tier) {
|
||||
case GLESMultiDrawMode::Auto: return "auto";
|
||||
case GLESMultiDrawMode::Ext: return "ext";
|
||||
case GLESMultiDrawMode::MultiIndirect: return "multiindirect";
|
||||
case GLESMultiDrawMode::Indirect: return "indirect";
|
||||
case GLESMultiDrawMode::BaseVertex: return "basevertex";
|
||||
case GLESMultiDrawMode::DrawElements: return "drawelements";
|
||||
case GLESMultiDrawMode::Compute: return "compute";
|
||||
}
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
GLESMultiDrawMode ResolvedTier() {
|
||||
ResolveTierOnce();
|
||||
return g_resolvedTier;
|
||||
}
|
||||
|
||||
String DescribeTierResolution() {
|
||||
ResolveTierOnce();
|
||||
return g_tierResolution;
|
||||
}
|
||||
|
||||
void OnBackendContextDestroyed() {
|
||||
g_indirectCommands = {};
|
||||
g_rebasedIndices = {};
|
||||
g_drawInfo = {};
|
||||
g_flattenedIndices = {};
|
||||
g_computeProgram = 0;
|
||||
g_computeProgramFailed = false;
|
||||
g_uElementSize = -1;
|
||||
g_uDrawCount = -1;
|
||||
g_uTotalIndices = -1;
|
||||
}
|
||||
|
||||
void DrawElementsBatch(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex) {
|
||||
if (drawcount <= 0 || !count || !indices) return;
|
||||
// State-independent and possibly throwing, so it runs before any GL work.
|
||||
CheckPrimitiveRestartSupported(type);
|
||||
|
||||
const Bool hasIndexBuffer = BoundIndexBuffer() != nullptr;
|
||||
|
||||
// The compute tier dispatches BEFORE the draw state is established: doing it
|
||||
// afterwards would mean unpicking the program, SSBO and index bindings
|
||||
// PrepareForDraw just made, and a dispatch inside an open transform feedback
|
||||
// span is not legal at all. On success it hands back a flattened index stream.
|
||||
// A batch whose sub-draws carry their own base vertices cannot be flattened either
|
||||
// when the program reads gl_BaseVertex: one draw call leaves one uniform value.
|
||||
// Asked conservatively because this decision precedes PrepareForDraw - see
|
||||
// CurrentProgramMayNeedPerSubDrawBuiltins. Flattening is the irreversible half:
|
||||
// once the batch is one draw the values are gone, whereas declining to flatten only
|
||||
// costs the unrolled tier.
|
||||
FlattenedStream flattened;
|
||||
if (ResolvedTier() == GLESMultiDrawMode::Compute &&
|
||||
!CurrentProgramMayNeedPerSubDrawBuiltins(basevertex != nullptr)) {
|
||||
FlattenWithCompute(mode, count, type, indices, drawcount, basevertex, flattened);
|
||||
}
|
||||
|
||||
PrepareForDraw(DrawSyncBit::IndexBuffer);
|
||||
|
||||
if (flattened.indexCount != 0) {
|
||||
const Uint previousIndexBinding = BoundIndexBufferId();
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, flattened.bufferId);
|
||||
g_GLESFuncs.glDrawElements(mode, static_cast<GLsizei>(flattened.indexCount), GL_UNSIGNED_INT, nullptr);
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, previousIndexBinding);
|
||||
return;
|
||||
}
|
||||
|
||||
// Now that PrepareForDraw has synced the program, both questions have real answers;
|
||||
// the tier choice and the per-sub-draw feeds use those, not the guess above.
|
||||
const Bool feedDrawID = CurrentProgramReadsDrawID();
|
||||
const Bool feedBaseVertex = basevertex != nullptr && CurrentProgramReadsBaseVertex();
|
||||
const GLESMultiDrawMode tier = ResolveTierForBatch(feedDrawID, feedBaseVertex, hasIndexBuffer);
|
||||
|
||||
Bool drawn = false;
|
||||
switch (tier) {
|
||||
case GLESMultiDrawMode::Ext:
|
||||
drawn = RunExt(mode, count, type, indices, drawcount, basevertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::MultiIndirect:
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/true, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::Indirect:
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/false, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::BaseVertex:
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::DrawElements:
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::Compute:
|
||||
// Its pre-pass ran above; reaching here means it declined this batch's shape.
|
||||
break;
|
||||
case GLESMultiDrawMode::Auto:
|
||||
break; // resolution never yields Auto
|
||||
}
|
||||
|
||||
// Every tier above may decline a batch whose shape it cannot express. The two
|
||||
// below are the floor: a base-vertex replay where the driver has one, and the
|
||||
// rewritten index stream where it does not. Both are safe for any batch these
|
||||
// entry points can receive.
|
||||
if (!drawn) {
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID,
|
||||
feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
MGLOG_E_ONCE("DirectGLES multi-draw: no usable tier for a %d sub-draw batch (mode 0x%x, type 0x%x); "
|
||||
"the batch was dropped",
|
||||
drawcount, mode, type);
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl
|
||||
@@ -0,0 +1,64 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectGLES/MultiDraw.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
#include <Config.h>
|
||||
#include "DirectGLES.h"
|
||||
|
||||
// Emulation of the desktop glMultiDrawElements / glMultiDrawElementsBaseVertex entry
|
||||
// points on OpenGL ES, which has neither in core.
|
||||
//
|
||||
// Every strategy below is an emulation; they differ only in which driver capability
|
||||
// they lean on and in how many driver entries a batch of N sub-draws costs. The design
|
||||
// follows MobileGlues (MobileGL-Dev/MobileGlues, gl/multidraw.cpp) tier for tier, plus
|
||||
// the native GL_EXT_multi_draw_arrays interaction that MobileGL already had:
|
||||
//
|
||||
// Ext one glMultiDrawElementsBaseVertexEXT 1 driver entry
|
||||
// MultiIndirect one glMultiDrawElementsIndirectEXT 1 driver entry + 1 upload
|
||||
// Indirect N x glDrawElementsIndirect N + 1 upload
|
||||
// BaseVertex N x glDrawElementsBaseVertex N
|
||||
// DrawElements N x glDrawElements over CPU-rebased indices N + 1 upload
|
||||
// Compute 1 x glDrawElements over a GPU-flattened, 1 dispatch + 1 entry
|
||||
// rebased index stream
|
||||
//
|
||||
// Which one runs is resolved once per ES context from the driver's capabilities,
|
||||
// capped by MOBILEGL_ESPRYT_MULTIDRAW_MODE, and can additionally be demoted per batch
|
||||
// when the batch's own shape rules a tier out (see ResolveTierForBatch in the .cpp).
|
||||
namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// The tier this ES context resolved to, computed on first use and stable after.
|
||||
MG_Config::GLESMultiDrawMode ResolvedTier();
|
||||
// "multiindirect", "compute", ... - stable identifiers, also used by the POST row.
|
||||
const char* TierName(MG_Config::GLESMultiDrawMode tier);
|
||||
// One line naming the resolved tier, the tiers the driver can support, and the env
|
||||
// clamp if one applied. For DriverPost and the startup log.
|
||||
String DescribeTierResolution();
|
||||
|
||||
// The resolution itself, as a pure function of a capability set: the backend feeds
|
||||
// it the live ES context's capabilities, DriverPost feeds it the ones it probed
|
||||
// standalone, and both therefore report the same tier. `explanation`, when non-null,
|
||||
// receives the "requested -> resolved (driver supports: ...)" line.
|
||||
MG_Config::GLESMultiDrawMode ResolveTier(const MG_External::GLESCapabilities& caps,
|
||||
const MG_External::GLESFunctionsTable& funcs,
|
||||
MG_Config::GLESMultiDrawMode requested, String* explanation);
|
||||
// Whether one tier is runnable on the given capability set, for per-row POST output.
|
||||
Bool IsTierSupported(const MG_External::GLESCapabilities& caps, const MG_External::GLESFunctionsTable& funcs,
|
||||
MG_Config::GLESMultiDrawMode tier);
|
||||
|
||||
// Runs `drawcount` indexed sub-draws as one glMultiDrawElements(BaseVertex) call
|
||||
// would. `basevertex` is null for the plain glMultiDrawElements entry point (every
|
||||
// base vertex is 0). Owns the whole draw, preparation included: callers must not
|
||||
// have run PrepareForDraw, because the compute tier has to dispatch before the
|
||||
// draw state is established.
|
||||
void DrawElementsBatch(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex);
|
||||
|
||||
// The ES context is gone: every scratch buffer and the compute program belonged to
|
||||
// it, so drop the names without deleting them (the dead context reclaims them).
|
||||
void OnBackendContextDestroyed();
|
||||
} // namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl
|
||||
File diff suppressed because it is too large
Load Diff
@@ -9,20 +9,22 @@
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/Core.h>
|
||||
#include <MG_Util/BackendLoaders/OpenGL/Loader.h>
|
||||
#include <MG_Util/Texture/TextureFormatProcessor.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectGLES {
|
||||
namespace DebugImpl {
|
||||
class ErrorLopper {
|
||||
public:
|
||||
void Loop(std::function<void(GLenum)>);
|
||||
void Clear();
|
||||
static void Loop(const std::function<void(GLenum)>&);
|
||||
static void Clear();
|
||||
ErrorLopper();
|
||||
~ErrorLopper();
|
||||
};
|
||||
|
||||
class OpenGLScopeMarker {
|
||||
public:
|
||||
explicit OpenGLScopeMarker(String scopeName);
|
||||
explicit OpenGLScopeMarker(const String& scopeName);
|
||||
~OpenGLScopeMarker();
|
||||
};
|
||||
} // namespace DebugImpl
|
||||
@@ -34,16 +36,189 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
} // namespace VertexArrayImpl
|
||||
|
||||
namespace TextureImpl {
|
||||
// Whether images on this format-capability target can back a colour attachment, and so
|
||||
// need a colour-renderable storage format even when the frontend asked for a
|
||||
// three-channel one ES never renders to. Shared by the capability probe (which passes the
|
||||
// capabilities it has just queried, before the globals are published) and by the
|
||||
// allocation path (which reads the active backend's), so the format the cache was probed
|
||||
// with is always the format the image is created with.
|
||||
Bool TargetRequiresRenderableFormat(SizeT targetIndex);
|
||||
Flags<PixelFormatNormalizeOptionBit> GetRenderTargetNormalizeOptions(
|
||||
const MG_External::GLESCapabilities& capabilities, SizeT targetIndex);
|
||||
|
||||
void GenerateTextureFormatInfo(TextureInternalFormat internalFormat, GLenum* outInternalFormat,
|
||||
GLenum* outFormat, GLenum* outType);
|
||||
GLenum* outFormat, GLenum* outType,
|
||||
TextureTarget target = TextureTarget::Unknown);
|
||||
void GenerateRenderbufferFormatInfo(TextureInternalFormat internalFormat, GLenum* outInternalFormat,
|
||||
GLenum* outFormat, GLenum* outType);
|
||||
Bool ShouldUseCaveatTextureFormat(TextureInternalFormat internalFormat, TextureTarget target);
|
||||
|
||||
// True when the format the image is actually created with has an alpha channel the
|
||||
// frontend format does not (the three-channel colour-renderable widening). GL reads such
|
||||
// a channel back as 1.0, so any swizzle source of ALPHA has to be answered with ONE and
|
||||
// any readback of the image has to overwrite the alpha the draw happened to leave there.
|
||||
Bool BackendTextureFormatAddsAlpha(TextureInternalFormat internalFormat, TextureTarget target);
|
||||
Bool BackendRenderbufferFormatAddsAlpha(TextureInternalFormat internalFormat);
|
||||
Bool ShouldUseCaveatRenderbufferFormat(TextureInternalFormat internalFormat);
|
||||
} // namespace TextureImpl
|
||||
|
||||
namespace FramebufferImpl {} // namespace FramebufferImpl
|
||||
|
||||
// Pure CPU helpers of the client-format readback conversion (ReadPixels/GetTexImage repack a wide
|
||||
// RGBA(_INTEGER) read into the caller's (format, type) layout). Kept context-free so unit tests can
|
||||
// exercise the exact packing the GL CTS packed_pixels oracle compares against.
|
||||
namespace ReadbackImpl {
|
||||
struct ReadbackChannelMapping {
|
||||
Int sourceChannel[4]; // RGBA source channel feeding each destination component
|
||||
Int channelCount; // destination component count
|
||||
Bool isInteger;
|
||||
};
|
||||
Bool GetReadbackChannelMapping(GLenum format, ReadbackChannelMapping& outMapping);
|
||||
|
||||
// Byte size of one destination component of `type`; packed types report the packed word size.
|
||||
// 0 = type not supported by the conversion path.
|
||||
SizeT GetReadbackComponentSize(GLenum type);
|
||||
|
||||
// Bit-field layout of a GL packed pixel type. width/shift are indexed in the client format's
|
||||
// component order (matching ReadbackChannelMapping); shift is the LSB position of the field in
|
||||
// the packed word: non-REV types pack the first component from the MSB, *_REV types from the
|
||||
// LSB (GL 3.3 table 3.6; field positions mirror the GL CTS glcPackedPixelsTests pack_* oracle).
|
||||
struct PackedReadbackLayout {
|
||||
Int fieldCount; // format components stored in the packed word
|
||||
Int width[4]; // bit width of each component's field
|
||||
Int shift[4]; // LSB bit position of each component's field
|
||||
SizeT byteSize; // packed word size in bytes (1, 2 or 4)
|
||||
Bool isFloatPacked; // 10F_11F_11F_REV / 5_9_9_9_REV: fields hold unsigned small floats
|
||||
};
|
||||
Bool GetPackedReadbackLayout(GLenum type, PackedReadbackLayout& out);
|
||||
|
||||
// Unsigned small-float encoders (EXT_packed_float / EXT_texture_shared_exponent semantics).
|
||||
Uint32 EncodeFloatToUnsignedF11(Float value);
|
||||
Uint32 EncodeFloatToUnsignedF10(Float value);
|
||||
Uint32 EncodeSharedExponentRGB9E5(const Float rgb[3]);
|
||||
|
||||
// Destination bytes per pixel for a (format mapping, type) readback pair; 0 when the pair is
|
||||
// not convertible (unknown type, packed field count != format component count, floating-point
|
||||
// or packed-float type with an integer format).
|
||||
SizeT GetReadbackDstPixelSize(const ReadbackChannelMapping& mapping, GLenum type);
|
||||
|
||||
// Repacks one row of wide RGBA(_INTEGER) texels (4 components of wideType each) into the
|
||||
// client's (format, type) layout. src holds width * 4 * GetReadbackComponentSize(wideType)
|
||||
// bytes, dst receives width * GetReadbackDstPixelSize(mapping, type) bytes.
|
||||
void ConvertWideReadbackRow(const Uint8* src, Uint8* dst, SizeT width, GLenum wideType,
|
||||
const ReadbackChannelMapping& mapping, GLenum type);
|
||||
|
||||
// Stores wide RGBA(_INTEGER) rows into the client pointer or the bound PACK pixel buffer,
|
||||
// honoring the client-side PACK pixel-store parameters (row length, alignment, skips,
|
||||
// swap-bytes, and - when applyPackImageParams - image height/skip images). Shared by the
|
||||
// DirectGLES and DirectVulkan readback conversion paths.
|
||||
Bool StoreWideRowsToClient(const Uint8* wide, GLenum wideType, GLsizei width, GLsizei sliceHeight,
|
||||
GLsizei sliceCount, const ReadbackChannelMapping& mapping, GLenum type,
|
||||
void* pixels, Bool applyPackImageParams);
|
||||
|
||||
// Stores packed 32-bit source words verbatim, with the same destination addressing, PACK
|
||||
// parameters and pixel-pack-buffer handling as StoreWideRowsToClient. For the sources whose
|
||||
// storage word already IS the client word (MG_Util::IsRawPackedPixelTransfer): routing those
|
||||
// through the wide float intermediate re-encodes them, and the RGB9_E5 encoder canonicalizes
|
||||
// the shared exponent, so glGetTexImage would answer with different bits than were stored.
|
||||
// `srcWords` holds sliceHeight * sliceCount tightly stacked rows of `width` 32-bit words.
|
||||
// False when `type` is not a 4-byte packed type.
|
||||
Bool StorePackedWordsToClient(const Uint8* srcWords, GLsizei width, GLsizei sliceHeight, GLsizei sliceCount,
|
||||
GLenum type, void* pixels, Bool applyPackImageParams);
|
||||
} // namespace ReadbackImpl
|
||||
|
||||
namespace PrgramImpl {
|
||||
String ProcessOutColorLocations(const String& glslCode);
|
||||
String ForceSupporterOutput(const String& glslCode);
|
||||
String ClampNormFallbackOutputs(String glslCode, GLenum shaderType, Uint32 snormOutputMask,
|
||||
Uint32 unormOutputMask);
|
||||
String ForceFlatIntegerVaryings(const String& glslCode, GLenum shaderType);
|
||||
// Legacy GLSL's gl_FragColor is broadcast to every enabled draw buffer (GL 4.6
|
||||
// 15.2.3), but ShaderSourceProcessor lowers it to the single output mg_FragColor,
|
||||
// which only ever reaches draw buffer 0. Replicates it across `drawBufferCount`
|
||||
// outputs and copies the value into them at the end of main. A no-op for
|
||||
// drawBufferCount <= 1, i.e. for everything but a framebuffer that actually
|
||||
// enables several draw buffers, so the ordinary single-target shader is untouched.
|
||||
String BroadcastLegacyFragColor(String glslCode, GLenum shaderType, Uint drawBufferCount);
|
||||
// SPIRV-Cross emits `#extension GL_EXT_texture_buffer : require` for every buffer-texture
|
||||
// sampler when it targets ESSL below 320, and offers no way to ask for the OES spelling.
|
||||
// On a driver that advertises only GL_OES_texture_buffer that directive is a compile
|
||||
// error, so the name is retargeted in the emitted source. A no-op on every other tier:
|
||||
// ES 3.2 needs no directive at all and an EXT driver already has the right one.
|
||||
String RetargetTextureBufferExtension(String glslCode,
|
||||
MG_External::GLESCapabilities::TextureBufferTier tier);
|
||||
// Adds `#extension GL_NV_image_formats : require` when the shader carries an image
|
||||
// format qualifier GLSL ES has no core spelling for. SPIRV-Cross prints the format and
|
||||
// asks for nothing, so the request has to be made here. `needed` is the caller's answer,
|
||||
// because only it knows which formats are in play AND whether the driver advertises the
|
||||
// extension - requesting an unadvertised extension is itself a compile error, so this is
|
||||
// never emitted speculatively. A no-op when not needed or already present.
|
||||
String RequestExtendedImageFormats(String glslCode, Bool needed);
|
||||
// Writes a format layout qualifier into the image declarations named in
|
||||
// `esslFormatByUniformName` that still have none. The completion half of the image-format
|
||||
// bake, and ONLY that: the SPIR-V pass (BakeImageFormatsPass) is what normally puts the
|
||||
// format in, but SPIRV-Cross throws rather than printing the formats it calls
|
||||
// desktop-only when it targets ESSL - r8ui among them, which is what the stencil half of
|
||||
// KHR-GL4x.packed_depth_stencil.stencil_texturing binds - and a throw loses the whole
|
||||
// stage. So those formats stay out of the module and are spelled here instead, on the
|
||||
// emitted text, where nothing can refuse them.
|
||||
//
|
||||
// Declarations that already carry a format are left exactly as they are, whoever wrote
|
||||
// it. Must run before RemoveLayoutBinding, which is where an image's layout qualifier
|
||||
// stops being safe to edit by hand.
|
||||
String BakeImageFormatQualifiers(String glslCode, const UnorderedMap<String, String>& esslFormatByUniformName);
|
||||
String RemoveLayoutBinding(const String& glslCode);
|
||||
// Prefix of the writeonly half a read+write image uniform is split into (see
|
||||
// SplitReadWriteImageUniforms); the suffix is the image's own name.
|
||||
constexpr const char* IMAGE_WRITE_ALIAS_PREFIX = "mg_imageWrite_";
|
||||
// ESSL refuses an image variable that carries a format qualifier other than r32f /
|
||||
// r32i / r32ui unless it also carries `readonly` or `writeonly` (GLSL ES 3.10 4.9 /
|
||||
// 3.20 4.10; glslang enforces it verbatim in ParseHelper.cpp's layoutObjectCheck).
|
||||
// SPIRV-Cross emits NEITHER for an image the shader both reads and writes: it
|
||||
// speculatively decorates every storage image NonWritable+NonReadable
|
||||
// (fixup_image_load_store_access), then OpImageRead clears NonReadable and
|
||||
// OpImageWrite clears NonWritable, and to_qualifiers_glsl only prints `readonly`
|
||||
// from NonWritable and `writeonly` from NonReadable. Desktop GLSL is happy with the
|
||||
// bare declaration, so the frontend raises no error and the illegal ESSL only shows
|
||||
// up as a device compile failure - and then as a silently no-op draw.
|
||||
//
|
||||
// Restores a legal declaration:
|
||||
// * loaded only -> add `readonly`
|
||||
// * stored only -> add `writeonly`
|
||||
// * both -> emit TWO declarations on the same binding and of the
|
||||
// same type, `readonly <name>` and `writeonly
|
||||
// <IMAGE_WRITE_ALIAS_PREFIX><name>`, and point every
|
||||
// imageStore at the second one. Several image variables
|
||||
// may share an image unit as long as they have the same
|
||||
// type and format, which is exactly what the pair is.
|
||||
//
|
||||
// Budget note: the split DOUBLES the image-uniform count of the stage it fires in, so
|
||||
// a driver advertising a tight GL_MAX_{FRAGMENT,VERTEX,...}_IMAGE_UNIFORMS can turn a
|
||||
// shader that used to compile into a link failure. ES only guarantees 4 fragment image
|
||||
// uniforms, so a shader with more than half the limit in read+write images is the case
|
||||
// to watch.
|
||||
//
|
||||
// Runs on the transpiled ESSL, so it must see the bindings the frontend units were
|
||||
// already rewritten to and must run before those bindings are stripped - see the call
|
||||
// site in Managers.cpp.
|
||||
String SplitReadWriteImageUniforms(const String& glslCode);
|
||||
// Prefix of the per-sampler float uniform that carries GL_TEXTURE_LOD_BIAS into
|
||||
// the shader (see EmulateTextureLodBias); the suffix is the sampler's own name.
|
||||
constexpr const char* LOD_BIAS_UNIFORM_PREFIX = "mg_lodBias_";
|
||||
// ES has no per-texture/sampler LOD bias at all (GL_TEXTURE_LOD_BIAS is desktop
|
||||
// only; Vulkan spells it VkSamplerCreateInfo::mipLodBias), so it has to reach the
|
||||
// shader as a uniform and be folded into every lookup's level of detail. Declares
|
||||
// one `uniform highp float mg_lodBias_<sampler>;` per mip-capable sampler and adds
|
||||
// it to the bias / explicit-LOD argument of every lookup that takes one. Draws push
|
||||
// the bound texture's (or sampler object's) value into it; a shader whose samplers
|
||||
// all have a zero bias is therefore unaffected. Returns the source unchanged when
|
||||
// there is nothing to rewrite.
|
||||
//
|
||||
// avoidExplicitLodBias leaves lookups that already carry an explicit LOD untouched,
|
||||
// so their constant level stays constant; only the implicit-LOD forms take the bias.
|
||||
// Off by default and only ever set on ANGLE + llvmpipe, where injecting the uniform
|
||||
// into a constant LOD crashes the driver (MOBILEGL_AVOID_EXPLICIT_LOD_BIAS).
|
||||
String EmulateTextureLodBias(const String& glslCode, Bool avoidExplicitLodBias = false);
|
||||
} // namespace PrgramImpl
|
||||
|
||||
namespace Utils {
|
||||
|
||||
@@ -9,16 +9,328 @@
|
||||
#include "BackendObject_DirectVulkan.h"
|
||||
#include "MG_Backend/BackendObject.h"
|
||||
#include "DirectVulkan.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
#include "MG_State/GLState/TextureState/TextureState.h"
|
||||
#include "MG_Util/Classifiers/TextureEnumClassifier.h"
|
||||
#include "MG_Util/Converters/MGToGL/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include "MG_Util/Texture/TextureFormatProcessor.h"
|
||||
#include "MG_Util/Async/ShaderCompilePool.h"
|
||||
|
||||
#include <Config.h>
|
||||
#include <cmath>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
Bool IsR11G11B10FFallbackEnabled() {
|
||||
return MG_Config::Features.MagmaR11G11B10FFallback;
|
||||
}
|
||||
|
||||
Bool IsReleaseCurrentRequest(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx) {
|
||||
return dpy == EGL_NO_DISPLAY && draw == EGL_NO_SURFACE && read == EGL_NO_SURFACE && ctx == EGL_NO_CONTEXT;
|
||||
(void)dpy;
|
||||
return draw == EGL_NO_SURFACE && read == EGL_NO_SURFACE && ctx == EGL_NO_CONTEXT;
|
||||
}
|
||||
|
||||
Bool IsFormatIndexValid(TextureInternalFormat format) {
|
||||
return format != TextureInternalFormat::Unknown && static_cast<Int>(format) >= 0 &&
|
||||
static_cast<SizeT>(format) < kFormatCapabilityFormatCount;
|
||||
}
|
||||
|
||||
Bool IsLayeredTarget(TextureTarget target) {
|
||||
return target == TextureTarget::Texture3D || target == TextureTarget::Texture1DArray ||
|
||||
target == TextureTarget::Texture2DArray || target == TextureTarget::TextureCubeMap ||
|
||||
target == TextureTarget::TextureCubeMapArray || target == TextureTarget::Texture2DMultisampleArray;
|
||||
}
|
||||
|
||||
Bool IsMultisampleTarget(TextureTarget target) {
|
||||
return target == TextureTarget::Texture2DMultisample || target == TextureTarget::Texture2DMultisampleArray;
|
||||
}
|
||||
|
||||
Bool IsTextureBufferTarget(TextureTarget target) {
|
||||
return target == TextureTarget::TextureBuffer;
|
||||
}
|
||||
|
||||
Bool IsIntegerInternalFormat(TextureInternalFormat format) {
|
||||
const GLenum glFormat = MG_Util::ConvertTextureInternalFormatToGLEnum(format);
|
||||
GLenum normalizedInternalFormat = glFormat;
|
||||
GLenum imageFormat = GL_RGBA;
|
||||
GLenum imageType = GL_UNSIGNED_BYTE;
|
||||
MG_Util::TextureFormatProcessor::NormalizePixelFormat(glFormat, PixelFormatNormalizeOptionBit::None,
|
||||
&normalizedInternalFormat, &imageFormat, &imageType);
|
||||
return imageFormat == GL_RED_INTEGER || imageFormat == GL_RG_INTEGER || imageFormat == GL_RGB_INTEGER ||
|
||||
imageFormat == GL_RGBA_INTEGER;
|
||||
}
|
||||
|
||||
FormatCapabilityFlags GetAttachmentCaps(TextureInternalFormat format) {
|
||||
FormatCapabilityFlags caps = FormatCapability::FramebufferRenderable;
|
||||
const Bool isDepth = MG_Util::IsDepthFormatInternalFormat(format);
|
||||
const Bool isStencil = MG_Util::IsStencilFormatInternalFormat(format);
|
||||
if (!isDepth && !isStencil) {
|
||||
caps |= FormatCapability::ColorAttachment;
|
||||
}
|
||||
if (isDepth) {
|
||||
caps |= FormatCapability::DepthAttachment;
|
||||
}
|
||||
if (isStencil) {
|
||||
caps |= FormatCapability::StencilAttachment;
|
||||
}
|
||||
return caps;
|
||||
}
|
||||
|
||||
FormatCapabilityFlags BuildVulkanCaps(TextureInternalFormat logicalFormat, TextureTarget target,
|
||||
VkFormatFeatureFlags features) {
|
||||
FormatCapabilityFlags caps;
|
||||
const Bool isDepth = MG_Util::IsDepthFormatInternalFormat(logicalFormat);
|
||||
const Bool isStencil = MG_Util::IsStencilFormatInternalFormat(logicalFormat);
|
||||
const Bool isInteger = IsIntegerInternalFormat(logicalFormat);
|
||||
|
||||
if (IsTextureBufferTarget(target)) {
|
||||
if ((features & VK_FORMAT_FEATURE_UNIFORM_TEXEL_BUFFER_BIT) != 0) {
|
||||
caps |= FormatCapability::Creatable;
|
||||
caps |= FormatCapability::Sampled;
|
||||
caps |= FormatCapability::TextureBuffer;
|
||||
}
|
||||
return caps;
|
||||
}
|
||||
|
||||
const Bool sampled = (features & VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT) != 0;
|
||||
const Bool linearFilter = (features & VK_FORMAT_FEATURE_SAMPLED_IMAGE_FILTER_LINEAR_BIT) != 0;
|
||||
const Bool colorRenderable = (features & VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT) != 0;
|
||||
const Bool depthStencilRenderable = (features & VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT) != 0;
|
||||
const Bool renderable = (isDepth || isStencil) ? depthStencilRenderable : colorRenderable;
|
||||
|
||||
if (sampled || renderable) {
|
||||
caps |= FormatCapability::Creatable;
|
||||
}
|
||||
if (sampled) {
|
||||
caps |= FormatCapability::Sampled;
|
||||
if (linearFilter && !isInteger && !isStencil) {
|
||||
caps |= FormatCapability::LinearFilter;
|
||||
}
|
||||
if (!isStencil && (features & VK_FORMAT_FEATURE_BLIT_SRC_BIT) != 0 &&
|
||||
(features & VK_FORMAT_FEATURE_BLIT_DST_BIT) != 0) {
|
||||
caps |= FormatCapability::GenerateMipmap;
|
||||
}
|
||||
if (!isInteger && !isDepth && !isStencil) {
|
||||
caps |= FormatCapability::TextureGather;
|
||||
}
|
||||
if (isDepth && !isStencil) {
|
||||
caps |= FormatCapability::TextureShadow;
|
||||
}
|
||||
}
|
||||
if (renderable) {
|
||||
caps |= GetAttachmentCaps(logicalFormat);
|
||||
if (IsLayeredTarget(target)) {
|
||||
caps |= FormatCapability::FramebufferLayered;
|
||||
}
|
||||
}
|
||||
if (IsMultisampleTarget(target)) {
|
||||
caps |= FormatCapability::MultisampleTexture;
|
||||
}
|
||||
return caps;
|
||||
}
|
||||
|
||||
Optional<TextureInternalFormat> ResolveVulkanFallbackLogicalFormat(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::RGB:
|
||||
case TextureInternalFormat::RGB8:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
// Legacy low-bit-depth formats with no (or rarely supported) native Vulkan
|
||||
// encoding; a wider normalized fallback keeps at least the required precision.
|
||||
case TextureInternalFormat::R3G3B2:
|
||||
case TextureInternalFormat::RGB4:
|
||||
case TextureInternalFormat::RGB5:
|
||||
case TextureInternalFormat::RGBA2:
|
||||
case TextureInternalFormat::RGBA4:
|
||||
case TextureInternalFormat::RGB5A1:
|
||||
return TextureInternalFormat::RGBA8;
|
||||
case TextureInternalFormat::RGB10:
|
||||
return TextureInternalFormat::RGB10A2;
|
||||
case TextureInternalFormat::RGB12:
|
||||
case TextureInternalFormat::RGBA12:
|
||||
return TextureInternalFormat::RGBA16;
|
||||
case TextureInternalFormat::SRGB8:
|
||||
return TextureInternalFormat::SRGB8Alpha8;
|
||||
case TextureInternalFormat::RGB8Snorm:
|
||||
return TextureInternalFormat::RGBA8Snorm;
|
||||
case TextureInternalFormat::RGB16:
|
||||
return TextureInternalFormat::RGBA16;
|
||||
case TextureInternalFormat::RGB16Snorm:
|
||||
return TextureInternalFormat::RGBA16Snorm;
|
||||
case TextureInternalFormat::RGB16F:
|
||||
return TextureInternalFormat::RGBA16F;
|
||||
case TextureInternalFormat::R11FG11FB10F:
|
||||
if (IsR11G11B10FFallbackEnabled()) {
|
||||
return TextureInternalFormat::RGBA16F;
|
||||
}
|
||||
return Nullopt;
|
||||
case TextureInternalFormat::RGB32F:
|
||||
return TextureInternalFormat::RGBA32F;
|
||||
case TextureInternalFormat::RGB8I:
|
||||
return TextureInternalFormat::RGBA8I;
|
||||
case TextureInternalFormat::RGB8UI:
|
||||
return TextureInternalFormat::RGBA8UI;
|
||||
case TextureInternalFormat::RGB16I:
|
||||
return TextureInternalFormat::RGBA16I;
|
||||
case TextureInternalFormat::RGB16UI:
|
||||
return TextureInternalFormat::RGBA16UI;
|
||||
case TextureInternalFormat::RGB32I:
|
||||
return TextureInternalFormat::RGBA32I;
|
||||
case TextureInternalFormat::RGB32UI:
|
||||
return TextureInternalFormat::RGBA32UI;
|
||||
default:
|
||||
return Nullopt;
|
||||
}
|
||||
}
|
||||
|
||||
Optional<VkFormat> ResolveVulkanFallbackFormat(TextureInternalFormat format) {
|
||||
const Optional<TextureInternalFormat> fallbackLogicalFormat = ResolveVulkanFallbackLogicalFormat(format);
|
||||
if (!fallbackLogicalFormat) {
|
||||
return Nullopt;
|
||||
}
|
||||
return MG_Util::ConvertTextureInternalFormatToVkEnum(*fallbackLogicalFormat);
|
||||
}
|
||||
|
||||
Bool HasNewCaveatFormatCaps(FormatCapabilityFlags nativeCaps, FormatCapabilityFlags fallbackCaps) {
|
||||
for (FormatCapability capability : kReportedFormatCapabilities) {
|
||||
if (HasFormatCapability(fallbackCaps, capability) && !HasFormatCapability(nativeCaps, capability)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void LogVulkanFormatCaveat(TextureInternalFormat logicalFormat, SizeT targetIndex,
|
||||
TextureInternalFormat fallbackFormat) {
|
||||
MGLOG_D(
|
||||
"Caveat: %s %s not fully supported. Reason: native Vulkan format is not fully supported. Fallback: %s",
|
||||
GetFormatCapabilityTargetName(targetIndex).c_str(),
|
||||
MG_Util::ConvertTextureInternalFormatToString(logicalFormat).c_str(),
|
||||
MG_Util::ConvertTextureInternalFormatToString(fallbackFormat).c_str());
|
||||
}
|
||||
|
||||
Vector<Int> BuildSampleCounts(Int maxSamples) {
|
||||
Vector<Int> counts;
|
||||
for (Int samples = std::max(maxSamples, 1); samples > 1; samples >>= 1) {
|
||||
counts.push_back(samples);
|
||||
}
|
||||
counts.push_back(1);
|
||||
return counts;
|
||||
}
|
||||
|
||||
void PopulateFormatCapabilitiesImpl(VkPhysicalDevice physicalDevice,
|
||||
PFN_vkGetPhysicalDeviceFormatProperties getFormatProperties,
|
||||
const MG_External::VulkanCapabilities& capabilities,
|
||||
FormatCapabilityCache& cache) {
|
||||
cache.Clear();
|
||||
if (physicalDevice == VK_NULL_HANDLE || getFormatProperties == nullptr) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (SizeT formatIndex = 0; formatIndex < kFormatCapabilityFormatCount; ++formatIndex) {
|
||||
const auto logicalFormat = static_cast<TextureInternalFormat>(formatIndex);
|
||||
if (!IsFormatIndexValid(logicalFormat)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
VkFormat nativeFormat = MG_Util::ConvertTextureInternalFormatToVkEnum(logicalFormat);
|
||||
const Optional<TextureInternalFormat> fallbackLogicalFormat =
|
||||
ResolveVulkanFallbackLogicalFormat(logicalFormat);
|
||||
VkFormat fallbackFormat = ResolveVulkanFallbackFormat(logicalFormat).value_or(VK_FORMAT_UNDEFINED);
|
||||
|
||||
VkFormatProperties nativeProperties{};
|
||||
if (nativeFormat != VK_FORMAT_UNDEFINED) {
|
||||
getFormatProperties(physicalDevice, nativeFormat, &nativeProperties);
|
||||
}
|
||||
|
||||
VkFormatProperties fallbackProperties{};
|
||||
if (fallbackFormat != VK_FORMAT_UNDEFINED && fallbackFormat != nativeFormat) {
|
||||
getFormatProperties(physicalDevice, fallbackFormat, &fallbackProperties);
|
||||
}
|
||||
|
||||
for (SizeT targetIndex = 0; targetIndex < kFormatCapabilityTextureTargetCount; ++targetIndex) {
|
||||
const auto target = static_cast<TextureTarget>(targetIndex);
|
||||
const VkFormatFeatureFlags nativeFeatures = IsTextureBufferTarget(target)
|
||||
? nativeProperties.bufferFeatures
|
||||
: nativeProperties.optimalTilingFeatures;
|
||||
FormatCapabilityFlags nativeCaps = BuildVulkanCaps(logicalFormat, target, nativeFeatures);
|
||||
cache.FullCaps[targetIndex][formatIndex] |= nativeCaps;
|
||||
|
||||
const VkFormatFeatureFlags fallbackFeatures = IsTextureBufferTarget(target)
|
||||
? fallbackProperties.bufferFeatures
|
||||
: fallbackProperties.optimalTilingFeatures;
|
||||
FormatCapabilityFlags fallbackCaps = BuildVulkanCaps(logicalFormat, target, fallbackFeatures);
|
||||
if (fallbackFormat != VK_FORMAT_UNDEFINED && fallbackFormat != nativeFormat) {
|
||||
cache.CaveatCaps[targetIndex][formatIndex] |= fallbackCaps;
|
||||
if (fallbackLogicalFormat && HasNewCaveatFormatCaps(nativeCaps, fallbackCaps)) {
|
||||
LogVulkanFormatCaveat(logicalFormat, targetIndex, *fallbackLogicalFormat);
|
||||
}
|
||||
}
|
||||
|
||||
if (HasFormatCapability(nativeCaps | fallbackCaps, FormatCapability::MultisampleTexture)) {
|
||||
const Bool isDepth = MG_Util::IsDepthFormatInternalFormat(logicalFormat);
|
||||
const Bool isStencil = MG_Util::IsStencilFormatInternalFormat(logicalFormat);
|
||||
const Bool isInteger = IsIntegerInternalFormat(logicalFormat);
|
||||
Int maxSamples = capabilities.MaxColorTextureSamples;
|
||||
if (isDepth || isStencil) {
|
||||
maxSamples = capabilities.MaxDepthTextureSamples;
|
||||
} else if (isInteger) {
|
||||
maxSamples = capabilities.MaxIntegerSamples;
|
||||
}
|
||||
cache.SampleCounts[targetIndex][formatIndex] = BuildSampleCounts(maxSamples);
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT renderbufferTargetIndex = GetRenderbufferFormatCapabilityTargetIndex();
|
||||
FormatCapabilityFlags renderbufferCaps =
|
||||
BuildVulkanCaps(logicalFormat, TextureTarget::Texture2D, nativeProperties.optimalTilingFeatures);
|
||||
renderbufferCaps &= FormatCapability::Creatable;
|
||||
if ((nativeProperties.optimalTilingFeatures &
|
||||
(VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT)) != 0) {
|
||||
renderbufferCaps |= GetAttachmentCaps(logicalFormat);
|
||||
renderbufferCaps |= FormatCapability::MultisampleRenderbuffer;
|
||||
}
|
||||
cache.FullCaps[renderbufferTargetIndex][formatIndex] |= renderbufferCaps;
|
||||
|
||||
if (fallbackFormat != VK_FORMAT_UNDEFINED && fallbackFormat != nativeFormat) {
|
||||
FormatCapabilityFlags fallbackRenderbufferCaps = BuildVulkanCaps(
|
||||
logicalFormat, TextureTarget::Texture2D, fallbackProperties.optimalTilingFeatures);
|
||||
fallbackRenderbufferCaps &= FormatCapability::Creatable;
|
||||
if ((fallbackProperties.optimalTilingFeatures &
|
||||
(VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT)) !=
|
||||
0) {
|
||||
fallbackRenderbufferCaps |= GetAttachmentCaps(logicalFormat);
|
||||
fallbackRenderbufferCaps |= FormatCapability::MultisampleRenderbuffer;
|
||||
}
|
||||
cache.CaveatCaps[renderbufferTargetIndex][formatIndex] |= fallbackRenderbufferCaps;
|
||||
if (fallbackLogicalFormat && HasNewCaveatFormatCaps(renderbufferCaps, fallbackRenderbufferCaps)) {
|
||||
LogVulkanFormatCaveat(logicalFormat, renderbufferTargetIndex, *fallbackLogicalFormat);
|
||||
}
|
||||
}
|
||||
|
||||
const FormatCapabilityFlags rbCaps = cache.FullCaps[renderbufferTargetIndex][formatIndex] |
|
||||
cache.CaveatCaps[renderbufferTargetIndex][formatIndex];
|
||||
if (HasFormatCapability(rbCaps, FormatCapability::MultisampleRenderbuffer)) {
|
||||
cache.SampleCounts[renderbufferTargetIndex][formatIndex] =
|
||||
BuildSampleCounts(capabilities.MaxFramebufferSamples);
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void PopulateFormatCapabilities(VkPhysicalDevice physicalDevice,
|
||||
PFN_vkGetPhysicalDeviceFormatProperties getFormatProperties,
|
||||
const MG_External::VulkanCapabilities& capabilities, FormatCapabilityCache& cache) {
|
||||
PopulateFormatCapabilitiesImpl(physicalDevice, getFormatProperties, capabilities, cache);
|
||||
}
|
||||
|
||||
BackendObject_DirectVulkan::~BackendObject_DirectVulkan() = default;
|
||||
|
||||
BackendObject_DirectVulkan::BackendObject_DirectVulkan() : m_rendererInfo{GetRendererIdentity()} {}
|
||||
|
||||
Bool BackendObject_DirectVulkan::InitWindowSurface() {
|
||||
if (!m_windowHandle.Handle) {
|
||||
MGLOG_E("Cannot initialize DirectVulkan window surface: native window handle is null");
|
||||
@@ -27,7 +339,24 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
auto nativeWindow = reinterpret_cast<NativeWindowType>(m_windowHandle.Handle);
|
||||
|
||||
// Any renderer instance this assignment replaces is destroyed here;
|
||||
// fence/timer-query handles stamped with the old generation go stale.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer = MakeUnique<MG_Backend::DirectVulkan::VulkanRenderer>(nativeWindow);
|
||||
MOBILEGL_ASSERT(pVulkanRenderer != nullptr, "InitWindowSurface: VulkanRenderer creation failed");
|
||||
pVulkanRenderer->Initialize();
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool BackendObject_DirectVulkan::InitPbufferSurface(EGLint width, EGLint height) {
|
||||
VulkanRendererConfig config;
|
||||
config.SurfaceWidth = static_cast<Uint32>(std::max<EGLint>(width, 1));
|
||||
config.SurfaceHeight = static_cast<Uint32>(std::max<EGLint>(height, 1));
|
||||
// Any renderer instance this assignment replaces is destroyed here;
|
||||
// fence/timer-query handles stamped with the old generation go stale.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer = MakeUnique<MG_Backend::DirectVulkan::VulkanRenderer>(NativeWindowType{}, config);
|
||||
MOBILEGL_ASSERT(pVulkanRenderer != nullptr, "InitPbufferSurface: VulkanRenderer creation failed");
|
||||
pVulkanRenderer->Initialize();
|
||||
return true;
|
||||
}
|
||||
@@ -46,8 +375,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
|
||||
MG_Util::BackendLoader::FillInVulkanCapabilities(m_vulkanCaps, pVulkanRenderer->GetPhysicalDevice().properties);
|
||||
const auto& physicalDevice = pVulkanRenderer->GetPhysicalDevice();
|
||||
if (!MG_Util::BackendLoader::QueryVulkanCapabilities(m_vulkanCaps, pVulkanRenderer->GetInstance(),
|
||||
physicalDevice.handle)) {
|
||||
MGLOG_W("DirectVulkan: failed to query extended Vulkan capabilities, using basic properties");
|
||||
MG_Util::BackendLoader::FillInVulkanCapabilities(m_vulkanCaps, physicalDevice.properties);
|
||||
}
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
PopulateFormatCapabilities(physicalDevice.handle, vkGetPhysicalDeviceFormatProperties, m_vulkanCaps,
|
||||
MutableFormatCapabilities());
|
||||
PrintFormatCapabilities(GetFormatCapabilities());
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -59,40 +397,47 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return BackendObject::InitializeEGLDisplay(dpy, major, minor);
|
||||
}
|
||||
|
||||
Bool BackendObject_DirectVulkan::CreateEGLWindowSurface(const WindowHandle& handle) {
|
||||
Bool BackendObject_DirectVulkan::CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_initialized) {
|
||||
MGLOG_E("DirectVulkan backend not initialized");
|
||||
return false;
|
||||
}
|
||||
if (handle.Backend != WindowBackend::Android || !handle.Handle) {
|
||||
MGLOG_E("DirectVulkan backend only supports Android native windows");
|
||||
if (!handle.Handle || (handle.Backend != WindowBackend::Android && handle.Backend != WindowBackend::X11 &&
|
||||
handle.Backend != WindowBackend::MetalLayer && handle.Backend != WindowBackend::Win32)) {
|
||||
MGLOG_E("DirectVulkan backend only supports Android, X11, CAMetalLayer, and Win32 native windows");
|
||||
return false;
|
||||
}
|
||||
|
||||
const Bool sameHandle =
|
||||
m_eglWindowSurfaceInitialized && m_windowHandle.Backend == handle.Backend && m_windowHandle.Handle == handle.Handle;
|
||||
if (sameHandle) {
|
||||
return true;
|
||||
}
|
||||
return RegisterEGLWindowSurface(surface, handle);
|
||||
}
|
||||
|
||||
if (m_eglWindowSurfaceInitialized || pVulkanRenderer) {
|
||||
pVulkanRenderer.reset();
|
||||
ResetEGLRuntimeState();
|
||||
Bool BackendObject_DirectVulkan::ResizeEGLWindowSurface(EGLSurface surface, Uint32 width, Uint32 height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_initialized) {
|
||||
MGLOG_E("DirectVulkan backend not initialized");
|
||||
return false;
|
||||
}
|
||||
if (!BackendObject::ResizeEGLWindowSurface(surface, width, height)) {
|
||||
return false;
|
||||
}
|
||||
if (pVulkanRenderer && m_eglSurface == surface) {
|
||||
pVulkanRenderer->RequestSwapchainResize(width, height);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
return BackendObject::CreateEGLWindowSurface(handle);
|
||||
Bool BackendObject_DirectVulkan::CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (!m_initialized) {
|
||||
MGLOG_E("DirectVulkan backend not initialized");
|
||||
return false;
|
||||
}
|
||||
return RegisterEGLPbufferSurface(surface, width, height);
|
||||
}
|
||||
|
||||
Bool BackendObject_DirectVulkan::MakeEGLCurrent(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
if (IsReleaseCurrentRequest(dpy, draw, read, ctx)) {
|
||||
return BackendObject::MakeEGLCurrent(dpy, draw, read, ctx);
|
||||
}
|
||||
if (!pVulkanRenderer) {
|
||||
MGLOG_E("DirectVulkan renderer is not initialized");
|
||||
return false;
|
||||
}
|
||||
return BackendObject::MakeEGLCurrent(dpy, draw, read, ctx);
|
||||
}
|
||||
|
||||
@@ -105,33 +450,134 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return BackendObject::SwapEGLBuffers(dpy, draw);
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::ReleaseEGLSurface(EGLSurface surface) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
BackendObject::ReleaseEGLSurface(surface);
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::ReleaseEGLResources() {
|
||||
const std::lock_guard<std::recursive_mutex> lock(m_eglStateMutex);
|
||||
// Outstanding fence/timer-query handles now refer to a dead renderer;
|
||||
// treat them as signaled/available with zero results from here on.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer.reset();
|
||||
// The reflection cache is file-scope, not renderer-owned; without this the
|
||||
// deleted programs' reflection strings survive full context teardown.
|
||||
ClearProgramResourceCaches();
|
||||
BackendObject::ReleaseEGLResources();
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::OnEGLSurfaceReleased(EGLSurface surface) {
|
||||
(void)surface;
|
||||
// Outstanding fence/timer-query handles now refer to a dead renderer;
|
||||
// treat them as signaled/available with zero results from here on.
|
||||
BumpRendererGeneration();
|
||||
pVulkanRenderer.reset();
|
||||
// The reflection cache is file-scope, not renderer-owned; without this the
|
||||
// deleted programs' reflection strings survive full context teardown.
|
||||
ClearProgramResourceCaches();
|
||||
}
|
||||
|
||||
const RendererInfo& BackendObject_DirectVulkan::GetRendererInfo() const {
|
||||
static RendererInfo RendererInfo = {
|
||||
.RendererName = "Magma", // Renderer Name
|
||||
.BackendName = "Direct (Vulkan)", // Backend Name
|
||||
.ExtraVendor = Nullopt, // Extra vendor
|
||||
.RendererGLInfo =
|
||||
{
|
||||
.TargetGLVersion = {3, 3, 0}, // Target OpenGL Version
|
||||
.TargetGLSLVersion = {4, 6, 0}, // Target Shading Language Version
|
||||
.Extensions = {V_OpenGL30, V_OpenGL31, V_OpenGL32, // OpenGL Extensions
|
||||
V_OpenGL33},
|
||||
.IsCompatibilityProfile = false // Is Compatibility Profile
|
||||
},
|
||||
.StaticBackendCapability = {.AllowVSOnlyPrograms = false} // Backend Capability
|
||||
};
|
||||
return RendererInfo;
|
||||
return m_rendererInfo;
|
||||
}
|
||||
|
||||
String BackendObject_DirectVulkan::GetBackendAPIVersionString() const {
|
||||
if (!m_initialized) {
|
||||
return "<uninitialized DirectVulkan backend>";
|
||||
}
|
||||
return FormatBackendAPIVersionString(m_vulkanCaps.DeviceName, m_vulkanCaps.VulkanAPIVersion.toString(),
|
||||
m_vulkanCaps.DriverVersionString);
|
||||
}
|
||||
|
||||
const RendererInfo& GetRendererIdentity() {
|
||||
static const RendererInfo rendererInfo = {
|
||||
.RendererName = "Magma",
|
||||
.BackendName = "Direct (Vulkan)",
|
||||
.ExtraVendor = Nullopt,
|
||||
.RendererGLInfo = {.TargetGLVersion = {4, 0, 0},
|
||||
.TargetGLSLVersion = {4, 6, 0},
|
||||
// Baseline advertisement (no shader subgroup, no timer queries); a
|
||||
// live backend reconciles its copy in UpdateAdvertisedExtensions.
|
||||
.Extensions = BuildAdvertisedExtensions(false, false, false),
|
||||
.IsCompatibilityProfile = false},
|
||||
.StaticBackendCapability = {.AllowVSOnlyPrograms = false}};
|
||||
return rendererInfo;
|
||||
}
|
||||
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
|
||||
Bool anisotropicFilteringSupported) {
|
||||
Vector<GLExtension> extensions = {
|
||||
V_OpenGL30, V_OpenGL31, V_OpenGL32, V_OpenGL33, V_OpenGL40, E_GL_ARB_draw_buffers_blend,
|
||||
E_GL_ARB_compute_shader, E_GL_ARB_shader_storage_buffer_object, E_GL_ARB_shader_image_load_store,
|
||||
E_GL_ARB_program_interface_query, E_GL_ARB_framebuffer_object, E_GL_ARB_multi_draw_indirect,
|
||||
E_GL_ARB_indirect_parameters, E_GL_EXT_framebuffer_object, E_GL_ARB_depth_texture, E_GL_ARB_buffer_storage,
|
||||
E_GL_ARB_texture_storage, E_GL_ARB_texture_storage_multisample, E_GL_ARB_texture_multisample,
|
||||
E_GL_ARB_clear_texture, E_GL_ARB_direct_state_access, E_GL_ARB_shader_draw_parameters,
|
||||
E_GL_ARB_gpu_shader_int64, E_GL_KHR_debug, E_GL_ARB_gpu_shader5, E_GL_ARB_multi_bind,
|
||||
E_GL_ARB_shading_language_420pack, E_GL_ARB_vertex_attrib_binding, E_GL_ARB_shader_image_size,
|
||||
E_GL_ARB_explicit_attrib_location,
|
||||
// Core since GL 3.1 and implemented for every version advertised here. The string
|
||||
// matters because applications gate the ENTRY POINTS on it rather than on the
|
||||
// version: a caller that finds the extension missing never resolves
|
||||
// glGetUniformBlockIndex / glUniformBlockBinding, and one that then uses uniform
|
||||
// blocks anyway calls through a null pointer.
|
||||
E_GL_ARB_uniform_buffer_object,
|
||||
// Sampling the stencil aspect through DEPTH_STENCIL_TEXTURE_MODE. Core from 4.3,
|
||||
// so on a 4.0 context the string is the only way to reach it.
|
||||
E_GL_ARB_stencil_texturing,
|
||||
// Advertised with GL_NUM_PROGRAM_BINARY_FORMATS = 0, which the
|
||||
// extension explicitly permits. It is also the only thing that
|
||||
// exposes glProgramParameteri before GL 4.1.
|
||||
E_GL_ARB_get_program_binary};
|
||||
if (shaderSubgroupSupported && !MG_Config::Features.DisableSubgroup) {
|
||||
extensions.push_back(E_GL_KHR_shader_subgroup);
|
||||
}
|
||||
// GL_KHR_parallel_shader_compile is MobileGL's own capability, not the Vulkan
|
||||
// device's: the compiler threads belong to MobileGL's shader pool and
|
||||
// glCompileShader/glLinkProgram are serviced entirely inside the frontend, so there
|
||||
// is no device feature to condition this on.
|
||||
//
|
||||
// Gated on the async flag deliberately, and this is the whole reason the gate
|
||||
// exists. Advertising the string is the one part of asynchronous compilation that a
|
||||
// recorded trace can never cover: Iris and Sodium change their SUBMISSION SCHEDULE
|
||||
// the moment they see it - they enqueue whole pipeline batches and poll
|
||||
// GL_COMPLETION_STATUS_KHR instead of compiling one program at a time - so
|
||||
// MOBILEGL_ASYNC_SHADER_COMPILE=0 has to withdraw the application-visible behaviour
|
||||
// change as well as the threading, or the kill switch would only be half a switch.
|
||||
if (MG_Util::Async::AsyncShaderCompileEnabled()) {
|
||||
extensions.push_back(E_GL_KHR_parallel_shader_compile);
|
||||
}
|
||||
// GL_ARB_gpu_shader_fp64 is opt-in (MOBILEGL_ADVERTISE_FP64). Every `double` in a
|
||||
// shader compiles and runs already - it is narrowed to 32 bits before the module
|
||||
// reaches this backend - so an application that simply uses doubles needs nothing
|
||||
// advertised. What the extension additionally promises is 64-bit PRECISION, which no
|
||||
// mobile GPU has and the narrowing cannot fake, so advertising it by default would
|
||||
// make an application that checks the string take a path MobileGL cannot honour.
|
||||
if (MG_Config::Features.AdvertiseFp64) {
|
||||
extensions.push_back(E_GL_ARB_gpu_shader_fp64);
|
||||
}
|
||||
// GL_ARB_timer_query gates MC's F3 GPU% (LWJGL checks the extension string);
|
||||
// only advertised when the device actually supports timestamp queries and the
|
||||
// MOBILEGL_DISABLE_TIMERQUERY escape hatch is off.
|
||||
if (timerQueriesSupported && !MG_Config::Features.DisableTimerQuery) {
|
||||
extensions.push_back(E_GL_ARB_timer_query);
|
||||
}
|
||||
// Only advertised when the samplerAnisotropy device feature was granted: without it the
|
||||
// sampler state is accepted but never applied, and an app trusting the string (LWJGL builds
|
||||
// GLCapabilities from it) would think it enabled anisotropic filtering.
|
||||
if (anisotropicFilteringSupported) {
|
||||
extensions.push_back(E_GL_EXT_texture_filter_anisotropic);
|
||||
extensions.push_back(E_GL_ARB_texture_filter_anisotropic);
|
||||
}
|
||||
return extensions;
|
||||
}
|
||||
|
||||
String FormatBackendAPIVersionString(const String& deviceName, const String& vulkanApiVersionString,
|
||||
const String& driverVersionString) {
|
||||
// Format:
|
||||
// <GPU Name>, Vulkan <Vulkan Version>, Driver <Driver Version>
|
||||
String str = m_vulkanCaps.DeviceName + ", Vulkan " + m_vulkanCaps.VulkanAPIVersion.toString() + ", Driver " +
|
||||
m_vulkanCaps.DriverVersionString;
|
||||
return str;
|
||||
return deviceName + ", Vulkan " + vulkanApiVersionString + ", Driver " + driverVersionString;
|
||||
}
|
||||
|
||||
BackendType BackendObject_DirectVulkan::GetBackendType() const {
|
||||
@@ -146,10 +592,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
funcsTable.GL.DrawArrays = DrawArrays;
|
||||
funcsTable.GL.DrawElements = DrawElements;
|
||||
funcsTable.GL.DrawElementsBaseVertex = DrawElementsBaseVertex;
|
||||
funcsTable.GL.MultiDrawArrays = MultiDrawArrays;
|
||||
funcsTable.GL.MultiDrawElements = MultiDrawElements;
|
||||
funcsTable.GL.MultiDrawElementsBaseVertex = MultiDrawElementsBaseVertex;
|
||||
funcsTable.GL.MultiDrawElementsIndirect = MultiDrawElementsIndirect;
|
||||
funcsTable.GL.MultiDrawArraysIndirect = MultiDrawArraysIndirect;
|
||||
funcsTable.GL.MultiDrawElementsIndirectCount = MultiDrawElementsIndirectCount;
|
||||
funcsTable.GL.MultiDrawArraysIndirectCount = MultiDrawArraysIndirectCount;
|
||||
funcsTable.GL.DrawRangeElementsBaseVertex = DrawRangeElementsBaseVertex;
|
||||
funcsTable.GL.DrawRangeElements = DrawRangeElements;
|
||||
funcsTable.GL.DrawElementsInstancedBaseVertexBaseInstance = DrawElementsInstancedBaseVertexBaseInstance;
|
||||
@@ -165,12 +614,56 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
funcsTable.GL.ClearBufferfv = ClearBufferfv;
|
||||
funcsTable.GL.ClearBufferuiv = ClearBufferuiv;
|
||||
funcsTable.GL.ClearBufferiv = ClearBufferiv;
|
||||
funcsTable.GL.ClearNamedFramebufferfv = ClearNamedFramebufferfv;
|
||||
funcsTable.GL.ClearNamedFramebufferfi = ClearNamedFramebufferfi;
|
||||
funcsTable.GL.ClearNamedFramebufferiv = ClearNamedFramebufferiv;
|
||||
funcsTable.GL.ClearNamedFramebufferuiv = ClearNamedFramebufferuiv;
|
||||
funcsTable.GL.BlitFramebuffer = BlitFramebuffer;
|
||||
funcsTable.GL.BlitNamedFramebuffer = BlitNamedFramebuffer;
|
||||
funcsTable.GL.CopyTexImage2D = CopyTexImage2D;
|
||||
funcsTable.GL.CopyTexSubImage2D = CopyTexSubImage2D;
|
||||
funcsTable.GL.CopyImageSubData = CopyImageSubData;
|
||||
funcsTable.GL.GenerateMipmap = GenerateMipmap;
|
||||
funcsTable.GL.ReadPixels = ReadPixels;
|
||||
funcsTable.GL.GetTexImage = GetTexImage;
|
||||
funcsTable.GL.GetTextureImage = GetTextureImage;
|
||||
funcsTable.GL.DispatchCompute = DispatchCompute;
|
||||
funcsTable.GL.DispatchComputeIndirect = DispatchComputeIndirect;
|
||||
funcsTable.GL.MemoryBarrier = MemoryBarrier;
|
||||
funcsTable.GL.MemoryBarrierByRegion = MemoryBarrierByRegion;
|
||||
funcsTable.GL.BindImageTexture = BindImageTexture;
|
||||
funcsTable.GL.GetIntegeri_v = GetIntegeri_v;
|
||||
funcsTable.GL.GetInteger64i_v = GetInteger64i_v;
|
||||
funcsTable.GL.GetProgramiv = GetProgramiv;
|
||||
funcsTable.GL.ShaderStorageBlockBinding = ShaderStorageBlockBinding;
|
||||
funcsTable.GL.FenceSync = FenceSync;
|
||||
funcsTable.GL.ClientWaitSync = ClientWaitSync;
|
||||
funcsTable.GL.WaitSync = WaitSync;
|
||||
funcsTable.GL.DeleteSync = DeleteSync;
|
||||
funcsTable.GL.GetSyncStatus = GetSyncStatus;
|
||||
// Optional timer-query group: left null (the frontend then falls
|
||||
// back) when disabled via MOBILEGL_DISABLE_TIMERQUERY. The hooks
|
||||
// themselves additionally degrade to null handles when the device
|
||||
// lacks timestamp support.
|
||||
if (!MG_Config::Features.DisableTimerQuery) {
|
||||
funcsTable.GL.IsTimerQuerySupported = IsTimerQuerySupported;
|
||||
funcsTable.GL.BeginTimeElapsedQuery = BeginTimeElapsedQuery;
|
||||
funcsTable.GL.EndTimeElapsedQuery = EndTimeElapsedQuery;
|
||||
funcsTable.GL.QueryCounterTimestamp = QueryCounterTimestamp;
|
||||
funcsTable.GL.IsQueryResultAvailable = IsQueryResultAvailable;
|
||||
funcsTable.GL.GetQueryResult64 = GetQueryResult64;
|
||||
funcsTable.GL.DeleteBackendQuery = DeleteBackendQuery;
|
||||
funcsTable.GL.GetGpuTimestampNs = GetGpuTimestampNs;
|
||||
}
|
||||
// Occlusion queries share the handle-based result/delete entries, which must
|
||||
// exist even when timer queries are disabled.
|
||||
funcsTable.GL.BeginOcclusionQuery = BeginOcclusionQuery;
|
||||
funcsTable.GL.EndOcclusionQuery = EndOcclusionQuery;
|
||||
funcsTable.GL.BeginXfbPrimitivesQuery = BeginXfbPrimitivesQuery;
|
||||
funcsTable.GL.EndXfbPrimitivesQuery = EndXfbPrimitivesQuery;
|
||||
funcsTable.GL.IsQueryResultAvailable = IsQueryResultAvailable;
|
||||
funcsTable.GL.GetQueryResult64 = GetQueryResult64;
|
||||
funcsTable.GL.DeleteBackendQuery = DeleteBackendQuery;
|
||||
funcsTableInitialized = true;
|
||||
}
|
||||
return funcsTable;
|
||||
@@ -180,7 +673,293 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return m_dynamicParameters;
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::ApplyVulkanCapabilitiesForTesting(
|
||||
const MG_External::VulkanCapabilities& capabilities) {
|
||||
m_vulkanCaps = capabilities;
|
||||
UpdateDynamicBackendParameters();
|
||||
UpdateAdvertisedExtensions();
|
||||
MutableFormatCapabilities().Clear();
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::UpdateAdvertisedExtensions() {
|
||||
// GL_ARB_timer_query gates MC's F3 GPU% (LWJGL checks the extension
|
||||
// string). InitCapabilities runs after InitWindowSurface has created
|
||||
// and initialized the renderer, so the advertisement can be gated on
|
||||
// real device timestamp support. ApplyVulkanCapabilitiesForTesting may
|
||||
// run without a renderer; no timer query is advertised then. Rebuilding
|
||||
// the whole list keeps re-runs idempotent.
|
||||
m_rendererInfo.RendererGLInfo.Extensions = BuildAdvertisedExtensions(
|
||||
m_vulkanCaps.SupportsShaderSubgroup, pVulkanRenderer && pVulkanRenderer->IsTimerQuerySupported(),
|
||||
pVulkanRenderer && pVulkanRenderer->IsSamplerAnisotropySupported());
|
||||
}
|
||||
|
||||
void BackendObject_DirectVulkan::UpdateDynamicBackendParameters() {
|
||||
const auto mapShaderStages = [](Uint32 vkStages) {
|
||||
Uint32 glStages = 0;
|
||||
if ((vkStages & VK_SHADER_STAGE_VERTEX_BIT) != 0) glStages |= GL_VERTEX_SHADER_BIT;
|
||||
if ((vkStages & VK_SHADER_STAGE_TESSELLATION_CONTROL_BIT) != 0) glStages |= GL_TESS_CONTROL_SHADER_BIT;
|
||||
if ((vkStages & VK_SHADER_STAGE_TESSELLATION_EVALUATION_BIT) != 0) {
|
||||
glStages |= GL_TESS_EVALUATION_SHADER_BIT;
|
||||
}
|
||||
if ((vkStages & VK_SHADER_STAGE_GEOMETRY_BIT) != 0) glStages |= GL_GEOMETRY_SHADER_BIT;
|
||||
if ((vkStages & VK_SHADER_STAGE_FRAGMENT_BIT) != 0) glStages |= GL_FRAGMENT_SHADER_BIT;
|
||||
if ((vkStages & VK_SHADER_STAGE_COMPUTE_BIT) != 0) glStages |= GL_COMPUTE_SHADER_BIT;
|
||||
return glStages;
|
||||
};
|
||||
|
||||
const auto mapSubgroupFeatures = [](Uint32 vkFeatures) {
|
||||
Uint32 glFeatures = 0;
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_BASIC_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_BASIC_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_VOTE_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_VOTE_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_ARITHMETIC_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_ARITHMETIC_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_BALLOT_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_BALLOT_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_SHUFFLE_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_SHUFFLE_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_SHUFFLE_RELATIVE_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_SHUFFLE_RELATIVE_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_CLUSTERED_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_CLUSTERED_BIT_KHR;
|
||||
}
|
||||
if ((vkFeatures & VK_SUBGROUP_FEATURE_QUAD_BIT) != 0) {
|
||||
glFeatures |= GL_SUBGROUP_FEATURE_QUAD_BIT_KHR;
|
||||
}
|
||||
return glFeatures;
|
||||
};
|
||||
|
||||
static constexpr SizeT kMaxAdvertisedShaderStorageBlockSize = 512ull * 1024ull * 1024ull;
|
||||
m_dynamicParameters.UniformBufferOffsetAlignment = m_vulkanCaps.UniformBufferOffsetAlignment;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMin = m_vulkanCaps.AliasedLineWidthRangeMin;
|
||||
m_dynamicParameters.AliasedLineWidthRangeMax = m_vulkanCaps.AliasedLineWidthRangeMax;
|
||||
// Without the samplerAnisotropy feature the limit is unusable, so report 1.0 (no anisotropy)
|
||||
// rather than a maximum the sampler manager will never apply.
|
||||
m_dynamicParameters.MaxTextureMaxAnisotropy =
|
||||
(pVulkanRenderer && pVulkanRenderer->IsSamplerAnisotropySupported()) ? m_vulkanCaps.MaxSamplerAnisotropy
|
||||
: 1.0f;
|
||||
m_dynamicParameters.SmoothLineWidthRangeMin = m_vulkanCaps.SmoothLineWidthRangeMin;
|
||||
m_dynamicParameters.SmoothLineWidthRangeMax = m_vulkanCaps.SmoothLineWidthRangeMax;
|
||||
m_dynamicParameters.SmoothLineWidthGranularity = m_vulkanCaps.SmoothLineWidthGranularity;
|
||||
m_dynamicParameters.PointSizeRangeMin = m_vulkanCaps.PointSizeRangeMin;
|
||||
m_dynamicParameters.PointSizeRangeMax = m_vulkanCaps.PointSizeRangeMax;
|
||||
m_dynamicParameters.PointSizeGranularity = m_vulkanCaps.PointSizeGranularity;
|
||||
m_dynamicParameters.Max3DTextureSize = m_vulkanCaps.Max3DTextureSize;
|
||||
m_dynamicParameters.MaxArrayTextureLayers = m_vulkanCaps.MaxArrayTextureLayers;
|
||||
m_dynamicParameters.MaxCubeMapTextureSize = m_vulkanCaps.MaxCubeMapTextureSize;
|
||||
m_dynamicParameters.MaxFramebufferWidth = m_vulkanCaps.MaxFramebufferWidth;
|
||||
m_dynamicParameters.MaxFramebufferHeight = m_vulkanCaps.MaxFramebufferHeight;
|
||||
m_dynamicParameters.MaxFramebufferLayers = m_vulkanCaps.MaxFramebufferLayers;
|
||||
m_dynamicParameters.MaxRenderbufferSize = m_vulkanCaps.MaxRenderbufferSize;
|
||||
m_dynamicParameters.MaxTextureSize = m_vulkanCaps.MaxTextureSize;
|
||||
m_dynamicParameters.MaxColorTextureSamples = m_vulkanCaps.MaxColorTextureSamples;
|
||||
m_dynamicParameters.MaxDepthTextureSamples = m_vulkanCaps.MaxDepthTextureSamples;
|
||||
m_dynamicParameters.MaxFramebufferSamples = m_vulkanCaps.MaxFramebufferSamples;
|
||||
m_dynamicParameters.MaxIntegerSamples = m_vulkanCaps.MaxIntegerSamples;
|
||||
m_dynamicParameters.MaxSamples = m_vulkanCaps.MaxSamples;
|
||||
m_dynamicParameters.MaxSampleMaskWords = m_vulkanCaps.MaxSampleMaskWords;
|
||||
const Int maxSupportedTextureUnits = static_cast<Int>(MG_State::GLState::TextureState::MAX_TEXTURE_IMAGE_UNITS);
|
||||
// GL_MAX_TEXTURE_IMAGE_UNITS is a *per-stage* sampler limit. Adreno/Qualcomm report a huge
|
||||
// maxPerStageDescriptorSampledImages (descriptor-indexing scale), so clamping it only to our
|
||||
// combined array capacity (192) still advertises 192 per stage. Host code treats this value as
|
||||
// an array bound: Minecraft's Blaze3D GlStateManager.TEXTURES[] holds 128 entries and Iris
|
||||
// iterates [0, GL_MAX_TEXTURE_IMAGE_UNITS) over it (CompositeRenderer.renderAll), so any value
|
||||
// > 128 throws ArrayIndexOutOfBoundsException. Match desktop drivers (32) for the per-stage
|
||||
// limits while keeping the combined limit at our texture-unit array capacity.
|
||||
constexpr Int maxPerStageTextureUnits =
|
||||
static_cast<Int>(MG_State::GLState::TextureState::MAX_PER_STAGE_TEXTURE_IMAGE_UNITS);
|
||||
m_dynamicParameters.MaxTextureImageUnits = std::min(m_vulkanCaps.MaxTextureImageUnits, maxPerStageTextureUnits);
|
||||
m_dynamicParameters.MaxVertexTextureImageUnits =
|
||||
std::min(m_vulkanCaps.MaxVertexTextureImageUnits, maxPerStageTextureUnits);
|
||||
m_dynamicParameters.MaxComputeTextureImageUnits =
|
||||
std::min(m_vulkanCaps.MaxComputeTextureImageUnits, maxPerStageTextureUnits);
|
||||
m_dynamicParameters.MaxCombinedTextureImageUnits =
|
||||
std::min(m_vulkanCaps.MaxCombinedTextureImageUnits, maxSupportedTextureUnits);
|
||||
// Never advertise more attributes than the state layer can store: the current-value array and
|
||||
// the Uint32 attribute masks the draw path passes around are both bounded by MAX_VERTEX_ATTRIBS.
|
||||
m_dynamicParameters.MaxVertexAttribs = std::min(
|
||||
m_vulkanCaps.MaxVertexAttribs, static_cast<Int>(MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS));
|
||||
// Vulkan descriptor limits are not GL limits, and a GL application reads an advertised
|
||||
// limit as an amount it may actually USE. Adreno answers the per-stage/per-set descriptor
|
||||
// queries at descriptor-indexing scale - the same driver whose
|
||||
// GL_MAX_SHADER_STORAGE_BLOCK_SIZE is clamped from 2147483647 further down - so
|
||||
// KHR-GL44.multi_bind.dispatch_bind_buffers_base read GL_MAX_COMPUTE_UNIFORM_BLOCKS,
|
||||
// created that many buffers and spliced that many UBO declarations into a single compute
|
||||
// shader: ~14 s of allocation, then death on std::bad_alloc. Its sibling
|
||||
// dispatch_bind_buffers_range hard-codes 4 buffers and passes, which is the clean
|
||||
// discriminator. Every ceiling below is far above what any desktop driver advertises for
|
||||
// these (84-96 for the binding families) and far below a descriptor-indexing count, so it
|
||||
// can only lower a limit that was never usable in the first place. The zero floor is not
|
||||
// decoration: a driver reporting UINT32_MAX used to arrive here as -1.
|
||||
const auto clampLimit = [](const char* name, Int reported, Int ceiling) {
|
||||
const Int clamped = std::min(std::max(reported, 0), ceiling);
|
||||
if (clamped != reported) {
|
||||
MGLOG_I("DirectVulkan: clamped %s from %d to %d", name, reported, clamped);
|
||||
}
|
||||
return clamped;
|
||||
};
|
||||
// GL 4.6 required minimums, for the record: MAX_COMPUTE_UNIFORM_BLOCKS 12,
|
||||
// MAX_COMPUTE/COMBINED_SHADER_STORAGE_BLOCKS 8, MAX_SHADER_STORAGE_BUFFER_BINDINGS 8,
|
||||
// MAX_UNIFORM_BUFFER_BINDINGS 84, MAX_TEXTURE_BUFFER_SIZE 65536.
|
||||
constexpr Int kMaxAdvertisedBufferBlocks = 256;
|
||||
constexpr Int kMaxAdvertisedTextureBufferSize = 1 << 27; // texels; what desktop GL reports
|
||||
m_dynamicParameters.MaxComputeShaderStorageBlocks =
|
||||
clampLimit("GL_MAX_COMPUTE_SHADER_STORAGE_BLOCKS", m_vulkanCaps.MaxComputeShaderStorageBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxCombinedShaderStorageBlocks =
|
||||
clampLimit("GL_MAX_COMBINED_SHADER_STORAGE_BLOCKS", m_vulkanCaps.MaxCombinedShaderStorageBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxComputeUniformBlocks =
|
||||
clampLimit("GL_MAX_COMPUTE_UNIFORM_BLOCKS", m_vulkanCaps.MaxComputeUniformBlocks,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxComputeWorkGroupInvocations = m_vulkanCaps.MaxComputeWorkGroupInvocations;
|
||||
m_dynamicParameters.MaxShaderStorageBufferBindings =
|
||||
clampLimit("GL_MAX_SHADER_STORAGE_BUFFER_BINDINGS", m_vulkanCaps.MaxShaderStorageBufferBindings,
|
||||
kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxTextureBufferSize = clampLimit(
|
||||
"GL_MAX_TEXTURE_BUFFER_SIZE", m_vulkanCaps.MaxTextureBufferSize, kMaxAdvertisedTextureBufferSize);
|
||||
m_dynamicParameters.TextureBufferOffsetAlignment = m_vulkanCaps.TextureBufferOffsetAlignment;
|
||||
m_dynamicParameters.MaxUniformBufferBindings = clampLimit(
|
||||
"GL_MAX_UNIFORM_BUFFER_BINDINGS", m_vulkanCaps.MaxUniformBufferBindings, kMaxAdvertisedBufferBlocks);
|
||||
m_dynamicParameters.MaxUniformBlockSize = m_vulkanCaps.MaxUniformBlockSize;
|
||||
m_dynamicParameters.MaxImageUnits = std::max(std::min(m_vulkanCaps.MaxImageUnits, maxSupportedTextureUnits), 0);
|
||||
m_dynamicParameters.MaxCombinedImageUniforms = std::max(m_vulkanCaps.MaxCombinedImageUniforms, 0);
|
||||
const Int maxPerStageImageUniforms =
|
||||
std::min(m_dynamicParameters.MaxImageUnits, m_dynamicParameters.MaxCombinedImageUniforms);
|
||||
// Vulkan uses one descriptor limit for every stage, but non-compute stores/atomics are
|
||||
// optional device features. VulkanRenderer enables each feature whenever the physical
|
||||
// device reports it, so these are the exact limits the logical device can compile and run.
|
||||
m_dynamicParameters.MaxVertexImageUniforms =
|
||||
m_vulkanCaps.SupportsVertexPipelineStoresAndAtomics ? maxPerStageImageUniforms : 0;
|
||||
m_dynamicParameters.MaxGeometryImageUniforms =
|
||||
m_vulkanCaps.SupportsVertexPipelineStoresAndAtomics && m_vulkanCaps.SupportsGeometryShader
|
||||
? maxPerStageImageUniforms
|
||||
: 0;
|
||||
m_dynamicParameters.MaxFragmentImageUniforms =
|
||||
m_vulkanCaps.SupportsFragmentStoresAndAtomics ? maxPerStageImageUniforms : 0;
|
||||
m_dynamicParameters.MaxComputeImageUniforms =
|
||||
std::min(std::max(m_vulkanCaps.MaxComputeImageUniforms, 0), maxPerStageImageUniforms);
|
||||
const Int maxSupportedDrawBuffers = static_cast<Int>(MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS);
|
||||
m_dynamicParameters.MaxDrawBuffers = std::min(m_vulkanCaps.MaxDrawBuffers, maxSupportedDrawBuffers);
|
||||
m_dynamicParameters.MaxColorAttachments = std::min(m_vulkanCaps.MaxColorAttachments, maxSupportedDrawBuffers);
|
||||
m_dynamicParameters.MaxClipDistances = m_vulkanCaps.MaxClipDistances;
|
||||
m_dynamicParameters.MaxViewports = m_vulkanCaps.MaxViewports;
|
||||
m_dynamicParameters.MaxViewportWidth = m_vulkanCaps.MaxViewportWidth;
|
||||
m_dynamicParameters.MaxViewportHeight = m_vulkanCaps.MaxViewportHeight;
|
||||
m_dynamicParameters.ViewportBoundsRangeMin = m_vulkanCaps.ViewportBoundsRangeMin;
|
||||
m_dynamicParameters.ViewportBoundsRangeMax = m_vulkanCaps.ViewportBoundsRangeMax;
|
||||
m_dynamicParameters.ViewportSubpixelBits = m_vulkanCaps.ViewportSubpixelBits;
|
||||
m_dynamicParameters.MinFragmentInterpolationOffset =
|
||||
std::isfinite(m_vulkanCaps.MinFragmentInterpolationOffset) &&
|
||||
m_vulkanCaps.MinFragmentInterpolationOffset <= -0.5f
|
||||
? m_vulkanCaps.MinFragmentInterpolationOffset
|
||||
: -0.5f;
|
||||
m_dynamicParameters.MaxFragmentInterpolationOffset = 0.4375f;
|
||||
m_dynamicParameters.FragmentInterpolationOffsetBits = 4;
|
||||
if (m_vulkanCaps.FragmentInterpolationOffsetBits >= 4 &&
|
||||
std::isfinite(m_vulkanCaps.MaxFragmentInterpolationOffset)) {
|
||||
const Float requiredMaxOffset = 0.5f - std::ldexp(1.0f, -m_vulkanCaps.FragmentInterpolationOffsetBits);
|
||||
if (m_vulkanCaps.MaxFragmentInterpolationOffset >= requiredMaxOffset) {
|
||||
m_dynamicParameters.MaxFragmentInterpolationOffset = m_vulkanCaps.MaxFragmentInterpolationOffset;
|
||||
m_dynamicParameters.FragmentInterpolationOffsetBits = m_vulkanCaps.FragmentInterpolationOffsetBits;
|
||||
}
|
||||
}
|
||||
m_dynamicParameters.SupportsWideLines = m_vulkanCaps.SupportsWideLines;
|
||||
// A 2D or 2D multisample array texture is a VK_IMAGE_TYPE_2D image whose GL depth IS its
|
||||
// arrayLayers, so a GL layer is a Vulkan array layer with nothing to translate.
|
||||
// ResolveAttachmentBaseArrayLayer already passes the attachment's layer through. The other
|
||||
// layered targets are declared separately as their own machinery lands.
|
||||
{
|
||||
using DynParams = MG_Backend::DynamicBackendParameters;
|
||||
m_dynamicParameters.PerLayerFramebufferAttachmentTargets |=
|
||||
DynParams::PerLayerFramebufferAttachmentBit(TextureTarget::Texture2DArray) |
|
||||
DynParams::PerLayerFramebufferAttachmentBit(TextureTarget::Texture2DMultisampleArray);
|
||||
// A cube map array is one 2D image with arrayLayers = 6 * cubeCount, so a GL layer is a
|
||||
// Vulkan array layer here too - but the image cannot be created without imageCubeArray.
|
||||
// A 3D texture's GL layer is a z slice, which only a 2D view over a 2D-array-compatible
|
||||
// image can name. Optimistic: a format that refuses the flag is caught at image creation
|
||||
// and declines the slice view there, which the clear path handles as a soft miss.
|
||||
if (m_vulkanCaps.Supports2DArrayCompatible3DImages) {
|
||||
m_dynamicParameters.PerLayerFramebufferAttachmentTargets |=
|
||||
DynParams::PerLayerFramebufferAttachmentBit(TextureTarget::Texture3D);
|
||||
}
|
||||
if (m_vulkanCaps.SupportsImageCubeArray) {
|
||||
m_dynamicParameters.PerLayerFramebufferAttachmentTargets |=
|
||||
DynParams::PerLayerFramebufferAttachmentBit(TextureTarget::TextureCubeMapArray);
|
||||
}
|
||||
}
|
||||
// Never, on any device, and no longer for the reason it used to be. It used to track
|
||||
// shaderFloat64 because a `dvec3` input needed the Float64 capability to exist in the
|
||||
// module at all; a 64-bit vertex FETCH was already impossible (VK_FORMAT_R64*_SFLOAT is
|
||||
// optional and lavapipe reports zero bufferFeatures for all four), so the attribute
|
||||
// arrived as its 32-bit word pair and PackDoubleVertexInputsPass bitcast it back.
|
||||
//
|
||||
// The shader half of that is gone: every 64-bit float is narrowed before any module
|
||||
// reaches a backend (ShaderTranspiler::DemoteFloat64Pass), so there is no `double` input
|
||||
// left to bitcast INTO, and feeding a UINT-formatted attribute to what is now a `float`
|
||||
// input would be silent garbage. Reconstructing the value would mean decoding the
|
||||
// IEEE-754 double bit pattern in the shader - software fp64, which is precisely what the
|
||||
// demotion exists to avoid - and on Espryt it would additionally need the ES driver to
|
||||
// fetch 2N uint components where the application declared N doubles, which a dvec3 or
|
||||
// dvec4 cannot even express within one attribute location.
|
||||
//
|
||||
// So glVertexAttribLFormat / glVertexAttribLPointer are declined here exactly as they
|
||||
// already were on Espryt and on every real mobile device (Adreno and Mali both report
|
||||
// shaderFloat64 == VK_FALSE), and for the same visible reason. A `dvec3` INPUT still
|
||||
// compiles and draws - it is a `vec3` after demotion - as long as the application feeds
|
||||
// it with glVertexAttribPointer(GL_FLOAT) rather than 64-bit data.
|
||||
m_dynamicParameters.SupportsFloat64VertexAttributes = false;
|
||||
m_dynamicParameters.MaxShaderStorageBlockSize =
|
||||
std::min(m_vulkanCaps.MaxShaderStorageBlockSize, kMaxAdvertisedShaderStorageBlockSize);
|
||||
if (m_vulkanCaps.SupportsShaderSubgroup) {
|
||||
m_dynamicParameters.SubgroupSize = m_vulkanCaps.SubgroupSize;
|
||||
m_dynamicParameters.SubgroupSupportedStages = mapShaderStages(m_vulkanCaps.SubgroupSupportedStages);
|
||||
m_dynamicParameters.SubgroupSupportedFeatures =
|
||||
mapSubgroupFeatures(m_vulkanCaps.SubgroupSupportedOperations);
|
||||
m_dynamicParameters.SubgroupQuadOperationsInAllStages = m_vulkanCaps.SubgroupQuadOperationsInAllStages;
|
||||
} else {
|
||||
m_dynamicParameters.SubgroupSize = 0;
|
||||
m_dynamicParameters.SubgroupSupportedStages = 0;
|
||||
m_dynamicParameters.SubgroupSupportedFeatures = 0;
|
||||
m_dynamicParameters.SubgroupQuadOperationsInAllStages = false;
|
||||
}
|
||||
if (m_dynamicParameters.MaxShaderStorageBlockSize != m_vulkanCaps.MaxShaderStorageBlockSize) {
|
||||
MGLOG_I("DirectVulkan: clamped GL_MAX_SHADER_STORAGE_BLOCK_SIZE from %zu to %zu",
|
||||
m_vulkanCaps.MaxShaderStorageBlockSize, m_dynamicParameters.MaxShaderStorageBlockSize);
|
||||
}
|
||||
switch (m_vulkanCaps.VendorId) {
|
||||
case 0x5143u: // VK_VENDOR_ID: Qualcomm
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Qualcomm;
|
||||
break;
|
||||
case 0x13B5u: // ARM
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Arm;
|
||||
break;
|
||||
case 0x10DEu: // NVIDIA
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Nvidia;
|
||||
break;
|
||||
case 0x1002u: // AMD
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Amd;
|
||||
break;
|
||||
case 0x8086u: // Intel
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Intel;
|
||||
break;
|
||||
case 0x1010u: // Imagination
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::ImgTec;
|
||||
break;
|
||||
case 0x10005u: // Mesa software (lavapipe)
|
||||
case 0x1AE0u: // Google (SwiftShader)
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Software;
|
||||
break;
|
||||
default:
|
||||
m_dynamicParameters.GpuVendor = GpuVendorKind::Unknown;
|
||||
break;
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -12,29 +12,73 @@
|
||||
#include <MG_Util/BackendLoaders/Vulkan/Loader.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Populates the same format-capability cache used by backend startup. Passing the
|
||||
// instance-resolved function keeps standalone callers independent of global loader
|
||||
// initialization; the physical device must remain valid for the duration of the call.
|
||||
void PopulateFormatCapabilities(VkPhysicalDevice physicalDevice,
|
||||
PFN_vkGetPhysicalDeviceFormatProperties getFormatProperties,
|
||||
const MG_External::VulkanCapabilities& capabilities,
|
||||
FormatCapabilityCache& cache);
|
||||
|
||||
class BackendObject_DirectVulkan : public BackendObject {
|
||||
public:
|
||||
BackendObject_DirectVulkan();
|
||||
~BackendObject_DirectVulkan() override;
|
||||
|
||||
void Initialize() override;
|
||||
Bool InitWindowSurface() override;
|
||||
Bool InitCapabilities() override;
|
||||
Bool InitializeEGLDisplay(EGLDisplay dpy, EGLint* major, EGLint* minor) override;
|
||||
Bool CreateEGLWindowSurface(const WindowHandle& handle) override;
|
||||
Bool CreateEGLWindowSurface(EGLSurface surface, const WindowHandle& handle) override;
|
||||
Bool ResizeEGLWindowSurface(EGLSurface surface, Uint32 width, Uint32 height) override;
|
||||
Bool CreateEGLPbufferSurface(EGLSurface surface, EGLint width, EGLint height) override;
|
||||
Bool MakeEGLCurrent(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx) override;
|
||||
Bool SwapEGLBuffers(EGLDisplay dpy, EGLSurface draw) override;
|
||||
void ReleaseEGLSurface(EGLSurface surface) override;
|
||||
void ReleaseEGLResources() override;
|
||||
|
||||
const RendererInfo& GetRendererInfo() const override;
|
||||
String GetBackendAPIVersionString() const override;
|
||||
const GlobalBackendFunctionsTable& GetBackendFunctions() const override;
|
||||
const DynamicBackendParameters& GetDynamicParameters() const override;
|
||||
BackendType GetBackendType() const override;
|
||||
void ApplyVulkanCapabilitiesForTesting(const MG_External::VulkanCapabilities& capabilities);
|
||||
|
||||
private:
|
||||
Bool InitPbufferSurface(EGLint width, EGLint height) override;
|
||||
void OnEGLSurfaceReleased(EGLSurface surface) override;
|
||||
void UpdateAdvertisedExtensions();
|
||||
void UpdateDynamicBackendParameters();
|
||||
|
||||
Bool m_initialized = false;
|
||||
DynamicBackendParameters m_dynamicParameters;
|
||||
MG_External::VulkanCapabilities m_vulkanCaps;
|
||||
RendererInfo m_rendererInfo;
|
||||
};
|
||||
|
||||
// Single-source-of-truth helpers shared with the driver POST
|
||||
// (MG_Util/SelfTest/DriverPost.cpp), so the identity strings and extension list
|
||||
// MobileGL reports to applications on this backend cannot drift from what the
|
||||
// POST screen shows.
|
||||
|
||||
// Static identity of the Magma renderer (renderer/backend names, target GL/GLSL
|
||||
// versions, ExtraVendor) with the baseline extension advertisement (no shader
|
||||
// subgroup, no timer queries). A live backend copies this in its constructor and
|
||||
// reconciles the Extensions in UpdateAdvertisedExtensions once real capabilities
|
||||
// exist; callers that need the advertised list for a known capability set must
|
||||
// use BuildAdvertisedExtensions instead.
|
||||
const RendererInfo& GetRendererIdentity();
|
||||
|
||||
// The full OpenGL extension list Magma advertises (glGetString(GL_EXTENSIONS)) for
|
||||
// a device with the given raw capabilities. The MOBILEGL_DISABLE_SUBGROUP and
|
||||
// MOBILEGL_DISABLE_TIMERQUERY escape hatches are applied inside, so callers pass
|
||||
// the detected device support (passing an already-gated value is harmless).
|
||||
Vector<GLExtension> BuildAdvertisedExtensions(Bool shaderSubgroupSupported, Bool timerQueriesSupported,
|
||||
Bool anisotropicFilteringSupported);
|
||||
|
||||
// Format: <GPU Name>, Vulkan <Vulkan Version>, Driver <Driver Version> — the exact
|
||||
// string an initialized backend returns from GetBackendAPIVersionString (and that
|
||||
// ends up inside the application-visible GL_RENDERER string).
|
||||
String FormatBackendAPIVersionString(const String& deviceName, const String& vulkanApiVersionString,
|
||||
const String& driverVersionString);
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -8,25 +8,54 @@
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
#include <MG_Backend/BackendObject.h>
|
||||
#include "Renderer/VulkanRenderer.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
extern UniquePtr<VulkanRenderer> pVulkanRenderer;
|
||||
extern UniquePtr<VulkanRenderer>& pVulkanRenderer;
|
||||
|
||||
// Generation of the live VulkanRenderer instance, mirroring DirectGLES's
|
||||
// g_syncContextGeneration. BackendObject_DirectVulkan bumps it wherever
|
||||
// pVulkanRenderer is reset or recreated; fence and timer-query handles
|
||||
// stamped with an older generation are stale and resolve as signaled /
|
||||
// available with zero results instead of dereferencing the destroyed
|
||||
// renderer's frame serials and query-pool slots.
|
||||
Uint64 GetRendererGeneration();
|
||||
void BumpRendererGeneration();
|
||||
|
||||
// Drops every cached program-resource reflection entry (CPU-side strings/vectors
|
||||
// only, no Vulkan handles). Called at EGL teardown next to the renderer reset;
|
||||
// safe because GL calls are serialized in this codebase, and any still-live
|
||||
// program rebuilds its entry from the retained generated SPIR-V on demand.
|
||||
void ClearProgramResourceCaches();
|
||||
|
||||
void ClearBufferfi(GLenum buffer, GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void ClearBufferfv(GLenum buffer, GLint drawbuffer, const GLfloat* value);
|
||||
void ClearBufferuiv(GLenum buffer, GLint drawbuffer, const GLuint* value);
|
||||
void ClearBufferiv(GLenum buffer, GLint drawbuffer, const GLint* value);
|
||||
void ClearNamedFramebufferfv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, const GLfloat* value);
|
||||
void ClearNamedFramebufferiv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, const GLint* value);
|
||||
void ClearNamedFramebufferuiv(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, const GLuint* value);
|
||||
void ClearNamedFramebufferfi(const SharedPtr<MG_State::GLState::FramebufferObject>& framebuffer, GLenum buffer,
|
||||
GLint drawbuffer, GLfloat depth, GLint stencil);
|
||||
void Clear(GLbitfield mask);
|
||||
void DrawElements(GLenum mode, GLsizei count, GLenum type, const void* indices);
|
||||
void DrawArrays(GLenum mode, GLint first, GLsizei count);
|
||||
void DrawElementsBaseVertex(GLenum mode, GLsizei count, GLenum type, const GLvoid* indices, GLint basevertex);
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount);
|
||||
void MultiDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount);
|
||||
void MultiDrawElementsBaseVertex(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex);
|
||||
void MultiDrawElementsIndirect(GLenum mode, GLenum type, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride);
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex);
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices);
|
||||
@@ -44,12 +73,68 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void DrawArraysIndirect(GLenum mode, const void* indirect);
|
||||
void BlitFramebuffer(GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1, GLint dstX0, GLint dstY0, GLint dstX1,
|
||||
GLint dstY1, GLbitfield mask, GLenum filter);
|
||||
void BlitNamedFramebuffer(const SharedPtr<MG_State::GLState::FramebufferObject>& readFramebuffer,
|
||||
const SharedPtr<MG_State::GLState::FramebufferObject>& drawFramebuffer,
|
||||
GLint srcX0, GLint srcY0, GLint srcX1, GLint srcY1,
|
||||
GLint dstX0, GLint dstY0, GLint dstX1, GLint dstY1,
|
||||
GLbitfield mask, GLenum filter);
|
||||
void CopyTexImage2D(GLenum target, GLint level, GLenum internalformat, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height, GLint border);
|
||||
void CopyTexSubImage2D(GLenum target, GLint level, GLint xoffset, GLint yoffset, GLint x, GLint y, GLsizei width,
|
||||
GLsizei height);
|
||||
void CopyImageSubData(const SharedPtr<MG_State::GLState::ITextureObject>& srcTexture,
|
||||
GLenum srcTarget, GLint srcLevel, GLint srcX, GLint srcY, GLint srcZ,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& dstTexture,
|
||||
GLenum dstTarget, GLint dstLevel, GLint dstX, GLint dstY, GLint dstZ,
|
||||
GLsizei srcWidth, GLsizei srcHeight, GLsizei srcDepth);
|
||||
void GenerateMipmap(GLenum target);
|
||||
void DispatchCompute(GLuint numGroupsX, GLuint numGroupsY, GLuint numGroupsZ);
|
||||
void DispatchComputeIndirect(GLintptr indirect);
|
||||
void MemoryBarrier(GLbitfield barriers);
|
||||
void MemoryBarrierByRegion(GLbitfield barriers);
|
||||
void BindImageTexture(GLuint unit, GLuint texture, GLint level, GLboolean layered, GLint layer, GLenum access,
|
||||
GLenum format);
|
||||
void GetIntegeri_v(GLenum target, GLuint index, GLint* data);
|
||||
void GetInteger64i_v(GLenum target, GLuint index, GLint64* data);
|
||||
void GetProgramiv(GLuint program, GLenum pname, GLint* params);
|
||||
void ShaderStorageBlockBinding(GLuint program, const GLchar* storageBlockName, GLuint storageBlockBinding);
|
||||
void ReadPixels(GLint x, GLint y, GLsizei width, GLsizei height, GLenum format, GLenum type, void* pixels);
|
||||
void GetTexImage(GLenum target, GLint level, GLenum format, GLenum type, GLvoid* pixels);
|
||||
void GetTextureImage(const SharedPtr<MG_State::GLState::ITextureObject>& texture, TextureUploadTarget uploadTarget,
|
||||
GLint level, GLenum format, GLenum type, GLsizei bufSize, GLvoid* pixels);
|
||||
// GL fence sync objects, mapped onto the renderer's frame-serial busy
|
||||
// tracking: a fence captures the frame serial current at creation and is
|
||||
// signaled once every command recorded under that serial has completed on
|
||||
// the GPU.
|
||||
BackendSyncHandle FenceSync();
|
||||
GLenum ClientWaitSync(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout);
|
||||
void WaitSync(BackendSyncHandle sync, GLbitfield flags, GLuint64 timeout);
|
||||
void DeleteSync(BackendSyncHandle sync);
|
||||
Bool GetSyncStatus(BackendSyncHandle sync);
|
||||
// GPU timer queries (GL_TIME_ELAPSED spans and GL_TIMESTAMP one-shots),
|
||||
// backed by per-frame VkQueryPool timestamp slots. All hooks degrade
|
||||
// gracefully: null handles when the renderer is absent, the device lacks
|
||||
// timestamp support, or the frame's pool is exhausted.
|
||||
// Dynamic support check (GLFunctionsTable::IsTimerQuerySupported): true
|
||||
// only while a live renderer exists whose device can actually time.
|
||||
Bool IsTimerQuerySupported();
|
||||
BackendQueryHandle BeginTimeElapsedQuery();
|
||||
BackendQueryHandle BeginXfbPrimitivesQuery(Bool generated);
|
||||
void EndXfbPrimitivesQuery(BackendQueryHandle query);
|
||||
BackendQueryHandle BeginOcclusionQuery();
|
||||
void EndOcclusionQuery(BackendQueryHandle query);
|
||||
void EndTimeElapsedQuery(BackendQueryHandle query);
|
||||
BackendQueryHandle QueryCounterTimestamp();
|
||||
Bool IsQueryResultAvailable(BackendQueryHandle query);
|
||||
// Returns true when a final value was produced (outNanoseconds set; the
|
||||
// frontend may cache it and release the handle), false when the result
|
||||
// cannot be obtained yet (e.g. a wait refused because the records' frame
|
||||
// serial is the current unsubmitted frame) - the handle then stays
|
||||
// readable later.
|
||||
Bool GetQueryResult64(BackendQueryHandle query, Bool wait, Uint64* outNanoseconds);
|
||||
void DeleteBackendQuery(BackendQueryHandle query);
|
||||
// Always 0: Vulkan cannot synchronously sample the GPU clock (timestamps
|
||||
// only exist as vkCmdWriteTimestamp results); the frontend falls back.
|
||||
Int64 GetGpuTimestampNs();
|
||||
void Present();
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/DirectVulkanResourceState.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ProgramObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
GLuint GetShaderStorageBlockIndex(const MG_State::GLState::ProgramObject& program, const String& name);
|
||||
GLuint GetShaderStorageBlockBinding(const MG_State::GLState::ProgramObject& program, GLuint blockIndex);
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/BufferArena.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "BufferArena.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool BufferArena::Initialize(const BufferArenaDesc& desc) {
|
||||
Shutdown();
|
||||
|
||||
MOBILEGL_ASSERT(desc.allocator != nullptr, "BufferArena::Initialize requires valid allocator");
|
||||
MOBILEGL_ASSERT(desc.frameCount > 0, "BufferArena::Initialize requires non-zero frame count");
|
||||
MOBILEGL_ASSERT(desc.usage != 0, "BufferArena::Initialize requires non-zero buffer usage");
|
||||
|
||||
m_desc = desc;
|
||||
m_frames.clear();
|
||||
m_frames.resize(desc.frameCount);
|
||||
m_deferredReleases.resize(desc.frameCount);
|
||||
return true;
|
||||
}
|
||||
|
||||
void BufferArena::Shutdown() {
|
||||
for (auto& frame : m_frames) {
|
||||
frame.buffer.Destroy();
|
||||
frame.writeCursor = 0;
|
||||
}
|
||||
m_frames.clear();
|
||||
m_deferredReleases.clear();
|
||||
m_desc = {};
|
||||
}
|
||||
|
||||
void BufferArena::BeginFrame(Uint32 frameIndex) {
|
||||
CollectDeferredReleases(frameIndex);
|
||||
ResetFrame(frameIndex);
|
||||
}
|
||||
|
||||
void BufferArena::ResetFrame(Uint32 frameIndex) {
|
||||
AssertValidFrameIndex(frameIndex);
|
||||
m_frames[frameIndex].writeCursor = 0;
|
||||
}
|
||||
|
||||
void BufferArena::CollectDeferredReleases(Uint32 frameIndex) {
|
||||
AssertValidFrameIndex(frameIndex);
|
||||
m_deferredReleases[frameIndex].clear();
|
||||
}
|
||||
|
||||
Bool BufferArena::Allocate(Uint32 frameIndex, VkDeviceSize size, VkDeviceSize alignment, BufferSlice& outSlice) {
|
||||
AssertValidFrameIndex(frameIndex);
|
||||
MOBILEGL_ASSERT(size > 0, "BufferArena::Allocate requires non-zero size");
|
||||
|
||||
auto& frame = m_frames[frameIndex];
|
||||
const VkDeviceSize resolvedAlignment = alignment > 0 ? alignment : 1;
|
||||
const VkDeviceSize offset = (frame.writeCursor + resolvedAlignment - 1) & ~(resolvedAlignment - 1);
|
||||
const VkDeviceSize endOffset = offset + size;
|
||||
|
||||
if (!EnsureCapacity(frameIndex, endOffset)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
frame.writeCursor = endOffset;
|
||||
outSlice = frame.buffer.GetSlice(offset, size);
|
||||
return outSlice.IsValid();
|
||||
}
|
||||
|
||||
Bool BufferArena::Upload(Uint32 frameIndex, const void* data, VkDeviceSize size, VkDeviceSize alignment,
|
||||
BufferSlice& outSlice) {
|
||||
MOBILEGL_ASSERT(data != nullptr || size == 0, "BufferArena::Upload data pointer is null");
|
||||
if (!Allocate(frameIndex, size, alignment, outSlice)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (outSlice.mapped != nullptr) {
|
||||
Memcpy(outSlice.mapped, data, static_cast<SizeT>(size));
|
||||
return true;
|
||||
}
|
||||
|
||||
return m_frames[frameIndex].buffer.Upload(data, size, outSlice.offset);
|
||||
}
|
||||
|
||||
VkDeviceSize BufferArena::GetWriteCursor(Uint32 frameIndex) const {
|
||||
AssertValidFrameIndex(frameIndex);
|
||||
return m_frames[frameIndex].writeCursor;
|
||||
}
|
||||
|
||||
Uint32 BufferArena::GetFrameCount() const {
|
||||
return static_cast<Uint32>(m_frames.size());
|
||||
}
|
||||
|
||||
Bool BufferArena::EnsureCapacity(Uint32 frameIndex, VkDeviceSize requiredEndOffset) {
|
||||
AssertValidFrameIndex(frameIndex);
|
||||
auto& frame = m_frames[frameIndex];
|
||||
auto& buffer = frame.buffer;
|
||||
|
||||
if (buffer.IsValid() && buffer.GetSize() >= requiredEndOffset) {
|
||||
return true;
|
||||
}
|
||||
|
||||
VkDeviceSize newCapacity = buffer.IsValid() ? buffer.GetSize() : 0;
|
||||
if (newCapacity < m_desc.minBufferSize) {
|
||||
newCapacity = m_desc.minBufferSize;
|
||||
}
|
||||
if (newCapacity == 0) {
|
||||
newCapacity = requiredEndOffset;
|
||||
}
|
||||
while (newCapacity < requiredEndOffset) {
|
||||
newCapacity *= 2;
|
||||
}
|
||||
|
||||
if (buffer.IsValid()) {
|
||||
// Outgrown, not dead: every BufferSlice handed out from this frame's arena so far
|
||||
// still names it, and those slices stay in service until the frame slot is rewound
|
||||
// (VkBufferResource::transientSlice, the converted-vertex-stream cache, the draw
|
||||
// memos). The release therefore has to survive every mid-frame reclaim and land on
|
||||
// the next ResetFrame of this slot - see VkBufferManager::CollectAllDeferredReleases.
|
||||
m_deferredReleases[frameIndex].push_back(std::move(buffer));
|
||||
}
|
||||
|
||||
VkBufferObjectDesc bufferDesc{};
|
||||
bufferDesc.allocator = m_desc.allocator;
|
||||
bufferDesc.size = newCapacity;
|
||||
bufferDesc.usage = m_desc.usage;
|
||||
bufferDesc.memoryUsage = m_desc.memoryUsage;
|
||||
bufferDesc.allocationFlags = m_desc.allocationFlags;
|
||||
if (!buffer.Create(bufferDesc)) {
|
||||
return false;
|
||||
}
|
||||
if (m_desc.persistentlyMapped && buffer.Map() == nullptr) {
|
||||
buffer.Destroy();
|
||||
return false;
|
||||
}
|
||||
|
||||
frame.writeCursor = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
void BufferArena::AssertValidFrameIndex(Uint32 frameIndex) const {
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(), "BufferArena frame index out of range");
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,56 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/BufferArena.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "BufferSlice.h"
|
||||
#include "VkBufferObject.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
struct BufferArenaDesc {
|
||||
VmaAllocator allocator = nullptr;
|
||||
Uint32 frameCount = 0;
|
||||
VkBufferUsageFlags usage = 0;
|
||||
VmaMemoryUsage memoryUsage = VMA_MEMORY_USAGE_AUTO;
|
||||
VmaAllocationCreateFlags allocationFlags = 0;
|
||||
VkDeviceSize minBufferSize = 0;
|
||||
Bool persistentlyMapped = false;
|
||||
};
|
||||
|
||||
class BufferArena {
|
||||
public:
|
||||
Bool Initialize(const BufferArenaDesc& desc);
|
||||
void Shutdown();
|
||||
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
void ResetFrame(Uint32 frameIndex);
|
||||
void CollectDeferredReleases(Uint32 frameIndex);
|
||||
|
||||
Bool Allocate(Uint32 frameIndex, VkDeviceSize size, VkDeviceSize alignment, BufferSlice& outSlice);
|
||||
Bool Upload(Uint32 frameIndex, const void* data, VkDeviceSize size, VkDeviceSize alignment, BufferSlice& outSlice);
|
||||
|
||||
VkDeviceSize GetWriteCursor(Uint32 frameIndex) const;
|
||||
Uint32 GetFrameCount() const;
|
||||
|
||||
private:
|
||||
struct FrameResources {
|
||||
VkBufferObject buffer;
|
||||
VkDeviceSize writeCursor = 0;
|
||||
};
|
||||
|
||||
Bool EnsureCapacity(Uint32 frameIndex, VkDeviceSize requiredEndOffset);
|
||||
void AssertValidFrameIndex(Uint32 frameIndex) const;
|
||||
|
||||
BufferArenaDesc m_desc{};
|
||||
Vector<FrameResources> m_frames;
|
||||
Vector<Vector<VkBufferObject>> m_deferredReleases;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,23 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/BufferSlice.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
struct BufferSlice {
|
||||
VkBuffer buffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize offset = 0;
|
||||
VkDeviceSize size = 0;
|
||||
void* mapped = nullptr;
|
||||
|
||||
Bool IsValid() const { return buffer != VK_NULL_HANDLE; }
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -13,19 +13,22 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Destroy(device, commandPool);
|
||||
m_frames.assign(frameCount, {});
|
||||
currentFrameIndex = 0;
|
||||
m_device = device;
|
||||
m_commandPool = commandPool;
|
||||
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount, VK_NULL_HANDLE);
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount * 2, VK_NULL_HANDLE);
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
allocInfo.commandPool = commandPool;
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = frameCount;
|
||||
allocInfo.commandBufferCount = frameCount * 2;
|
||||
VkResult result = vkAllocateCommandBuffers(device, &allocInfo, commandBuffers.data());
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
m_frames[i].commandBuffer = commandBuffers[i];
|
||||
m_frames[i].preCommandBuffer = commandBuffers[frameCount + i];
|
||||
}
|
||||
|
||||
VkSemaphoreCreateInfo semaphoreInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
|
||||
@@ -45,9 +48,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void FrameContext::Destroy(VkDevice device, VkCommandPool commandPool) {
|
||||
const Uint32 frameCount = static_cast<Uint32>(m_frames.size());
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount, VK_NULL_HANDLE);
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount * 2, VK_NULL_HANDLE);
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
commandBuffers[i] = m_frames[i].commandBuffer;
|
||||
commandBuffers[frameCount + i] = m_frames[i].preCommandBuffer;
|
||||
}
|
||||
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
@@ -55,10 +59,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
DestroySwapchainSemaphores(device);
|
||||
if (device != VK_NULL_HANDLE && commandPool != VK_NULL_HANDLE && !m_frames.empty()) {
|
||||
vkFreeCommandBuffers(device, commandPool, frameCount, commandBuffers.data());
|
||||
for (auto& frame : m_frames) {
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
}
|
||||
vkFreeCommandBuffers(device, commandPool, frameCount * 2, commandBuffers.data());
|
||||
}
|
||||
m_frames.clear();
|
||||
currentFrameIndex = 0;
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_commandPool = VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
FrameContext::FrameData& FrameContext::GetCurrent() {
|
||||
@@ -80,6 +89,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
currentFrameIndex = (currentFrameIndex + 1) % static_cast<Uint32>(m_frames.size());
|
||||
GetCurrent().isCommandRecording = false;
|
||||
GetCurrent().hasCommandBufferRecorded = false;
|
||||
GetCurrent().isPreCommandRecording = false;
|
||||
GetCurrent().hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
|
||||
VkCommandBuffer& FrameContext::BeginCommandRecording(VkCommandBufferUsageFlags flags,
|
||||
@@ -97,6 +108,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VK_VERIFY(vkBeginCommandBuffer(frame.commandBuffer, &beginInfo), "BeginCommandRecording, vkBeginCommandBuffer");
|
||||
|
||||
frame.isCommandRecording = true;
|
||||
if (m_recordingObserver != nullptr) {
|
||||
m_recordingObserver->OnFrameCommandRecordingBegan(frame.commandBuffer);
|
||||
}
|
||||
return frame.commandBuffer;
|
||||
}
|
||||
|
||||
@@ -108,6 +122,41 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
}
|
||||
|
||||
VkCommandBuffer FrameContext::BeginPreCommandRecording() {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.isPreCommandRecording) {
|
||||
return frame.preCommandBuffer;
|
||||
}
|
||||
MOBILEGL_ASSERT(!frame.hasPreCommandBufferRecorded,
|
||||
"BeginPreCommandRecording: a recorded pre stream is still awaiting submission");
|
||||
VK_VERIFY(vkResetCommandBuffer(frame.preCommandBuffer, 0), "BeginPreCommandRecording, vkResetCommandBuffer");
|
||||
VkCommandBufferBeginInfo beginInfo{};
|
||||
beginInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO;
|
||||
VK_VERIFY(vkBeginCommandBuffer(frame.preCommandBuffer, &beginInfo),
|
||||
"BeginPreCommandRecording, vkBeginCommandBuffer");
|
||||
frame.isPreCommandRecording = true;
|
||||
return frame.preCommandBuffer;
|
||||
}
|
||||
|
||||
void FrameContext::EndPreCommandRecordingIfOpen() {
|
||||
auto& frame = GetCurrent();
|
||||
if (!frame.isPreCommandRecording) {
|
||||
return;
|
||||
}
|
||||
VK_VERIFY(vkEndCommandBuffer(frame.preCommandBuffer), "EndPreCommandRecordingIfOpen, vkEndCommandBuffer");
|
||||
frame.isPreCommandRecording = false;
|
||||
frame.hasPreCommandBufferRecorded = true;
|
||||
}
|
||||
|
||||
void FrameContext::AbandonPreCommandRecording() {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.isPreCommandRecording) {
|
||||
VK_VERIFY(vkEndCommandBuffer(frame.preCommandBuffer), "AbandonPreCommandRecording, vkEndCommandBuffer");
|
||||
}
|
||||
frame.isPreCommandRecording = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
|
||||
VkResult FrameContext::InitializeSwapchainSemaphores(VkDevice device, Uint32 swapchainImageCount) {
|
||||
DestroySwapchainSemaphores(device);
|
||||
if (swapchainImageCount == 0) {
|
||||
@@ -140,12 +189,30 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
Bool FrameContext::TransitionToPresent(VkImage image, VkImageLayout oldLayout, VkImageLayout presentLayout) {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.hasCommandBufferRecorded || frame.isCommandRecording || oldLayout == presentLayout ||
|
||||
oldLayout == VK_IMAGE_LAYOUT_SHARED_PRESENT_KHR) {
|
||||
if (oldLayout == presentLayout || oldLayout == VK_IMAGE_LAYOUT_SHARED_PRESENT_KHR) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& commandBuffer = BeginCommandRecording();
|
||||
// The barrier belongs in the frame's own recording. Bailing out because
|
||||
// something was already recorded (the previous behaviour) dropped the
|
||||
// transition entirely for every frame that never ran a default-framebuffer
|
||||
// render pass - the only other thing that carries the image to
|
||||
// PRESENT_SRC_KHR, via that pass's finalLayout - so the swapchain image was
|
||||
// handed to the WSI still in the layout it was acquired in.
|
||||
// A closed-but-unsubmitted buffer can only come from a submit that already
|
||||
// failed (SubmitPendingCommandBuffer leaves the flag set on error), and
|
||||
// appending to it is illegal while reopening would reset the frame's own
|
||||
// commands away. The device is gone on that path anyway - stay silent-safe
|
||||
// rather than trade a lost device for a barrier into a closed buffer.
|
||||
if (frame.hasCommandBufferRecorded) {
|
||||
MGLOG_E_ONCE("TransitionToPresent: command buffer already closed; skipping the present barrier");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Reopening a recording here would vkResetCommandBuffer this frame's own
|
||||
// commands away, so append to the open one and let the caller close it.
|
||||
const Bool openedRecording = !frame.isCommandRecording;
|
||||
VkCommandBuffer commandBuffer = openedRecording ? BeginCommandRecording() : frame.commandBuffer;
|
||||
|
||||
VkImageMemoryBarrier presentBarrier{};
|
||||
presentBarrier.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
@@ -164,7 +231,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0,
|
||||
nullptr, 0, nullptr, 1, &presentBarrier);
|
||||
|
||||
EndCommandRecording();
|
||||
if (openedRecording) {
|
||||
EndCommandRecording();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -172,34 +241,44 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 swapchainImageIndex) const {
|
||||
const auto& frame = GetCurrent();
|
||||
MOBILEGL_ASSERT(!frame.isCommandRecording, "GetSubmitInfo called while command buffer recording is still active");
|
||||
MOBILEGL_ASSERT(!frame.isPreCommandRecording,
|
||||
"GetSubmitInfo called while the pre-pass stream is still recording");
|
||||
AssertValidSwapchainImageIndex(swapchainImageIndex);
|
||||
SubmitInfoPacket packet{};
|
||||
packet.waitSemaphore = frame.imageAvailableSemaphore;
|
||||
packet.signalSemaphore = m_swapchainImageRenderFinishedSemaphores[swapchainImageIndex];
|
||||
packet.commandBuffer = frame.commandBuffer;
|
||||
|
||||
packet.submitInfo.waitSemaphoreCount = 1;
|
||||
packet.submitInfo.pWaitSemaphores = &packet.waitSemaphore;
|
||||
packet.submitInfo.pWaitDstStageMask = &packet.waitDstStageMask;
|
||||
packet.submitInfo.commandBufferCount = shouldSubmitCommandBuffer ? 1U : 0U;
|
||||
packet.submitInfo.pCommandBuffers = shouldSubmitCommandBuffer ? &packet.commandBuffer : nullptr;
|
||||
Uint32 commandBufferCount = 0;
|
||||
// The pre-pass stream executes strictly before the frame's commands.
|
||||
if (frame.hasPreCommandBufferRecorded) {
|
||||
packet.commandBuffers[commandBufferCount++] = frame.preCommandBuffer;
|
||||
}
|
||||
if (shouldSubmitCommandBuffer) {
|
||||
packet.commandBuffers[commandBufferCount++] = frame.commandBuffer;
|
||||
}
|
||||
|
||||
packet.submitInfo.waitSemaphoreCount = frame.imageAvailableSemaphoreConsumed ? 0U : 1U;
|
||||
packet.submitInfo.pWaitSemaphores = frame.imageAvailableSemaphoreConsumed ? nullptr : &packet.waitSemaphore;
|
||||
packet.submitInfo.pWaitDstStageMask = frame.imageAvailableSemaphoreConsumed ? nullptr : &packet.waitDstStageMask;
|
||||
packet.submitInfo.commandBufferCount = commandBufferCount;
|
||||
packet.submitInfo.pCommandBuffers = commandBufferCount > 0 ? packet.commandBuffers : nullptr;
|
||||
packet.submitInfo.signalSemaphoreCount = 1;
|
||||
packet.submitInfo.pSignalSemaphores = &packet.signalSemaphore;
|
||||
return packet;
|
||||
}
|
||||
|
||||
FrameContext::PresentInfoPacket FrameContext::GetPresentInfo(VkSwapchainKHR swapchain, const Uint32& imageIndex) const {
|
||||
FrameContext::PresentInfoPacket FrameContext::GetPresentInfo(VkSwapchainKHR swapchain, Uint32 imageIndex) const {
|
||||
AssertValidSwapchainImageIndex(imageIndex);
|
||||
PresentInfoPacket packet{};
|
||||
packet.waitSemaphore = m_swapchainImageRenderFinishedSemaphores[imageIndex];
|
||||
packet.swapchain = swapchain;
|
||||
packet.imageIndex = &imageIndex;
|
||||
packet.imageIndex = imageIndex;
|
||||
|
||||
packet.presentInfo.waitSemaphoreCount = 1;
|
||||
packet.presentInfo.pWaitSemaphores = &packet.waitSemaphore;
|
||||
packet.presentInfo.swapchainCount = 1;
|
||||
packet.presentInfo.pSwapchains = &packet.swapchain;
|
||||
packet.presentInfo.pImageIndices = packet.imageIndex;
|
||||
packet.presentInfo.pImageIndices = &packet.imageIndex;
|
||||
packet.presentInfo.pResults = nullptr;
|
||||
return packet;
|
||||
}
|
||||
@@ -211,14 +290,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
// The slot's fence has been waited: every command buffer this slot
|
||||
// submitted (including mid-frame flushes) has finished executing.
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
|
||||
result = vkResetFences(device, 1, &frame.imageInFlightFence);
|
||||
if (result != VK_SUCCESS) {
|
||||
result = vkAcquireNextImageKHR(device, swapchain, timeout, frame.imageAvailableSemaphore, acquireFence,
|
||||
&outImageIndex);
|
||||
// VK_SUBOPTIMAL_KHR is a success code: an image *was* acquired and
|
||||
// imageAvailableSemaphore *will* be signaled. Bailing out on it skipped both
|
||||
// the consumed-flag reset (leaving a stale "already consumed", so the next
|
||||
// submit never waited on the pending signal) and the fence reset (leaving
|
||||
// the slot's fence signaled for the next submit to reuse). Only a genuine
|
||||
// failure - VK_ERROR_OUT_OF_DATE_KHR and friends, where nothing is acquired
|
||||
// and nothing is signaled - skips the bookkeeping.
|
||||
if (result != VK_SUCCESS && result != VK_SUBOPTIMAL_KHR) {
|
||||
return result;
|
||||
}
|
||||
|
||||
return vkAcquireNextImageKHR(device, swapchain, timeout, frame.imageAvailableSemaphore, acquireFence,
|
||||
&outImageIndex);
|
||||
frame.imageAvailableSemaphoreConsumed = false;
|
||||
const VkResult resetResult = vkResetFences(device, 1, &frame.imageInFlightFence);
|
||||
// Hand the acquire's own code back so the caller can schedule a rebuild.
|
||||
return resetResult == VK_SUCCESS ? result : resetResult;
|
||||
}
|
||||
|
||||
Uint32 FrameContext::GetCurrentFrameIndex() const {
|
||||
@@ -229,6 +321,85 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return static_cast<Uint32>(m_frames.size());
|
||||
}
|
||||
|
||||
void FrameContext::SetRecordingObserver(IRecordingObserver* observer) {
|
||||
m_recordingObserver = observer;
|
||||
}
|
||||
|
||||
VkResult FrameContext::RetireCurrentCommandBuffer(Bool retirePreCommandBuffer) {
|
||||
MOBILEGL_ASSERT(m_device != VK_NULL_HANDLE && m_commandPool != VK_NULL_HANDLE,
|
||||
"RetireCurrentCommandBuffer requires an initialized FrameContext");
|
||||
auto& frame = GetCurrent();
|
||||
MOBILEGL_ASSERT(!frame.isCommandRecording,
|
||||
"RetireCurrentCommandBuffer called while the command buffer is still recording");
|
||||
MOBILEGL_ASSERT(!frame.isPreCommandRecording,
|
||||
"RetireCurrentCommandBuffer called while the pre-pass stream is still recording");
|
||||
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
allocInfo.commandPool = m_commandPool;
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = 1;
|
||||
VkCommandBuffer replacement = VK_NULL_HANDLE;
|
||||
VkResult result = vkAllocateCommandBuffers(m_device, &allocInfo, &replacement);
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
if (retirePreCommandBuffer) {
|
||||
VkCommandBuffer preReplacement = VK_NULL_HANDLE;
|
||||
result = vkAllocateCommandBuffers(m_device, &allocInfo, &preReplacement);
|
||||
if (result != VK_SUCCESS) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &replacement);
|
||||
return result;
|
||||
}
|
||||
frame.retiredCommandBuffers.push_back({frame.preCommandBuffer, frame.lastSubmitIndex});
|
||||
frame.preCommandBuffer = preReplacement;
|
||||
}
|
||||
// lastSubmitIndex was just written by the renderer for the submission
|
||||
// that carried this command buffer.
|
||||
frame.retiredCommandBuffers.push_back({frame.commandBuffer, frame.lastSubmitIndex});
|
||||
frame.commandBuffer = replacement;
|
||||
return VK_SUCCESS;
|
||||
}
|
||||
|
||||
void FrameContext::FreeRetiredCommandBuffers(FrameData& frame) {
|
||||
if (frame.retiredCommandBuffers.empty()) {
|
||||
return;
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && m_commandPool != VK_NULL_HANDLE) {
|
||||
for (const auto& retired : frame.retiredCommandBuffers) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &retired.commandBuffer);
|
||||
}
|
||||
}
|
||||
frame.retiredCommandBuffers.clear();
|
||||
}
|
||||
|
||||
void FrameContext::FreeRetiredCommandBuffersCompletedUpTo(Uint64 completedSubmitIndex) {
|
||||
if (m_device == VK_NULL_HANDLE || m_commandPool == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto& frame : m_frames) {
|
||||
// Retired buffers are appended in submit order, so the completed
|
||||
// ones form a prefix.
|
||||
SizeT completedCount = 0;
|
||||
while (completedCount < frame.retiredCommandBuffers.size() &&
|
||||
frame.retiredCommandBuffers[completedCount].submitIndex <= completedSubmitIndex) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1,
|
||||
&frame.retiredCommandBuffers[completedCount].commandBuffer);
|
||||
++completedCount;
|
||||
}
|
||||
if (completedCount > 0) {
|
||||
frame.retiredCommandBuffers.erase(frame.retiredCommandBuffers.begin(),
|
||||
frame.retiredCommandBuffers.begin() + completedCount);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void FrameContext::FreeAllRetiredCommandBuffers() {
|
||||
for (auto& frame : m_frames) {
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
}
|
||||
}
|
||||
|
||||
void FrameContext::AssertValidFrameIndex(Uint32 frameIndex) const {
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(), "FrameContext index out of range");
|
||||
}
|
||||
@@ -260,6 +431,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
frame.hasCommandBufferRecorded = false;
|
||||
frame.isCommandRecording = false;
|
||||
frame.imageAvailableSemaphoreConsumed = false;
|
||||
return VK_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -277,5 +449,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frame.imageAvailableSemaphore = VK_NULL_HANDLE;
|
||||
frame.isCommandRecording = false;
|
||||
frame.hasCommandBufferRecorded = false;
|
||||
frame.imageAvailableSemaphoreConsumed = false;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -14,27 +14,66 @@
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class FrameContext {
|
||||
public:
|
||||
// Notified immediately after a frame command buffer begins recording
|
||||
// (before any render pass has been begun); every BeginCommandRecording
|
||||
// caller funnels through this single seam. Implemented by the renderer
|
||||
// to prepare per-frame timer-query pools (vkCmdResetQueryPool must be
|
||||
// recorded outside a render pass).
|
||||
class IRecordingObserver {
|
||||
public:
|
||||
virtual ~IRecordingObserver() = default;
|
||||
virtual void OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer) = 0;
|
||||
};
|
||||
|
||||
struct SubmitInfoPacket {
|
||||
VkPipelineStageFlags waitDstStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
VkSemaphore waitSemaphore = VK_NULL_HANDLE;
|
||||
VkSemaphore signalSemaphore = VK_NULL_HANDLE;
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
// [0] = pre-pass command buffer (when recorded), then the frame
|
||||
// command buffer; submitInfo.pCommandBuffers points here.
|
||||
VkCommandBuffer commandBuffers[2] = {VK_NULL_HANDLE, VK_NULL_HANDLE};
|
||||
VkSubmitInfo submitInfo{VK_STRUCTURE_TYPE_SUBMIT_INFO};
|
||||
};
|
||||
|
||||
struct PresentInfoPacket {
|
||||
VkSemaphore waitSemaphore = VK_NULL_HANDLE;
|
||||
VkSwapchainKHR swapchain = VK_NULL_HANDLE;
|
||||
const Uint32* imageIndex = nullptr;
|
||||
Uint32 imageIndex = 0;
|
||||
VkPresentInfoKHR presentInfo{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR};
|
||||
};
|
||||
|
||||
// A command buffer submitted mid-frame (FlushPendingCommands), tagged
|
||||
// with the submit-tracker index it was submitted under so it can be
|
||||
// freed as soon as that submission is observed complete - without
|
||||
// waiting for the slot's fence to be waited again (present-less flush
|
||||
// loops never wait it).
|
||||
struct RetiredCommandBuffer {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
Uint64 submitIndex = 0;
|
||||
};
|
||||
|
||||
struct FrameData {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
// Pre-pass work stream: out-of-pass commands (deferred clear
|
||||
// materialization, sampled-layout transitions) for resources the
|
||||
// frame's recording has not touched yet. Submitted immediately
|
||||
// BEFORE commandBuffer in the same vkQueueSubmit, so recording
|
||||
// into it never has to split the frame's active render pass.
|
||||
VkCommandBuffer preCommandBuffer = VK_NULL_HANDLE;
|
||||
VkSemaphore imageAvailableSemaphore = VK_NULL_HANDLE;
|
||||
VkFence imageInFlightFence = VK_NULL_HANDLE;
|
||||
Bool isCommandRecording = false;
|
||||
Bool hasCommandBufferRecorded = false;
|
||||
Bool isPreCommandRecording = false;
|
||||
Bool hasPreCommandBufferRecorded = false;
|
||||
Bool imageAvailableSemaphoreConsumed = false;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands),
|
||||
// appended in submit order; freed once their submission is known
|
||||
// complete (fence wait or completion poll).
|
||||
Vector<RetiredCommandBuffer> retiredCommandBuffers;
|
||||
// Submit-tracker index of this slot's most recent queue submission
|
||||
// (written by the renderer at submit time).
|
||||
Uint64 lastSubmitIndex = 0;
|
||||
};
|
||||
|
||||
VkResult Initialize(VkDevice device, VkCommandPool commandPool, Uint32 frameCount);
|
||||
@@ -48,18 +87,45 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkCommandBuffer& BeginCommandRecording(VkCommandBufferUsageFlags flags = 0,
|
||||
const VkCommandBufferInheritanceInfo* pInheritanceInfo = nullptr);
|
||||
void EndCommandRecording();
|
||||
// Lazily opens the pre-pass work stream (see FrameData::preCommandBuffer).
|
||||
VkCommandBuffer BeginPreCommandRecording();
|
||||
// Closes the pre stream if open, marking it for submission ahead of the
|
||||
// frame command buffer. Safe to call when it never opened.
|
||||
void EndPreCommandRecordingIfOpen();
|
||||
// Drops an in-progress or recorded-but-unsubmitted pre stream (dropped
|
||||
// frame recordings, swapchain recreation).
|
||||
void AbandonPreCommandRecording();
|
||||
VkResult InitializeSwapchainSemaphores(VkDevice device, Uint32 swapchainImageCount);
|
||||
void DestroySwapchainSemaphores(VkDevice device);
|
||||
Bool TransitionToPresent(VkImage image, VkImageLayout oldLayout,
|
||||
VkImageLayout presentLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR);
|
||||
SubmitInfoPacket GetSubmitInfo(Bool shouldSubmitCommandBuffer, Uint32 swapchainImageIndex) const;
|
||||
PresentInfoPacket GetPresentInfo(VkSwapchainKHR swapchain, const Uint32& imageIndex) const;
|
||||
PresentInfoPacket GetPresentInfo(VkSwapchainKHR swapchain, Uint32 imageIndex) const;
|
||||
VkResult WaitAndAcquireNextImage(VkDevice device, VkSwapchainKHR swapchain, Uint32& outImageIndex,
|
||||
Uint64 timeout = UINT64_MAX, VkFence acquireFence = VK_NULL_HANDLE);
|
||||
|
||||
// Parks the current (already ended and submitted) command buffer on the
|
||||
// slot's retired list and installs a freshly allocated one, so recording
|
||||
// can restart while the submitted buffer is still executing. Retired
|
||||
// buffers are freed after the slot's fence is next waited, or as soon
|
||||
// as their submission is observed complete.
|
||||
VkResult RetireCurrentCommandBuffer(Bool retirePreCommandBuffer = false);
|
||||
|
||||
// Frees every retired command buffer whose tagged submission index is
|
||||
// known complete. Driven by the renderer's submit tracker on completion
|
||||
// events (fence waits and non-blocking polls), so present-less flush
|
||||
// loops reclaim their buffers without any extra wait.
|
||||
void FreeRetiredCommandBuffersCompletedUpTo(Uint64 completedSubmitIndex);
|
||||
// Frees every slot's retired command buffers. Only valid when the
|
||||
// caller has proven every queue submission complete.
|
||||
void FreeAllRetiredCommandBuffers();
|
||||
|
||||
Uint32 GetCurrentFrameIndex() const;
|
||||
Uint32 GetFrameCount() const;
|
||||
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetRecordingObserver(IRecordingObserver* observer);
|
||||
|
||||
private:
|
||||
void AssertValidFrameIndex(Uint32 frameIndex) const;
|
||||
void AssertValidSwapchainImageIndex(Uint32 imageIndex) const;
|
||||
@@ -68,9 +134,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const VkSemaphoreCreateInfo& semaphoreInfo,
|
||||
const VkFenceCreateInfo& fenceInfo);
|
||||
void DestroySyncObjectsForFrame(VkDevice device, Uint32 frameIndex);
|
||||
void FreeRetiredCommandBuffers(FrameData& frame);
|
||||
|
||||
Vector<FrameData> m_frames;
|
||||
Vector<VkSemaphore> m_swapchainImageRenderFinishedSemaphores;
|
||||
Uint32 currentFrameIndex = 0;
|
||||
IRecordingObserver* m_recordingObserver = nullptr;
|
||||
// Stored at Initialize for retired-command-buffer management.
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkCommandPool m_commandPool = VK_NULL_HANDLE;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -8,9 +8,189 @@
|
||||
|
||||
#include "PipelineFactory.h"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static const char* PrimitiveTopologyToString(VkPrimitiveTopology topology) {
|
||||
switch (topology) {
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_POINT_LIST)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_LINE_LIST)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_LINE_STRIP)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_TRIANGLE_STRIP)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_TRIANGLE_FAN)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_LINE_LIST_WITH_ADJACENCY)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_LINE_STRIP_WITH_ADJACENCY)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST_WITH_ADJACENCY)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_TRIANGLE_STRIP_WITH_ADJACENCY)
|
||||
ENUM_STR_CASE(VK_PRIMITIVE_TOPOLOGY_PATCH_LIST)
|
||||
default:
|
||||
return "VK_PRIMITIVE_TOPOLOGY_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
static const char* SampleCountToString(VkSampleCountFlagBits sampleCount) {
|
||||
switch (sampleCount) {
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_1_BIT)
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_2_BIT)
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_4_BIT)
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_8_BIT)
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_16_BIT)
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_32_BIT)
|
||||
ENUM_STR_CASE(VK_SAMPLE_COUNT_64_BIT)
|
||||
default:
|
||||
return "VK_SAMPLE_COUNT_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
static const char* CullModeToString(VkCullModeFlags cullMode) {
|
||||
switch (cullMode) {
|
||||
case VK_CULL_MODE_NONE:
|
||||
return "VK_CULL_MODE_NONE";
|
||||
case VK_CULL_MODE_FRONT_BIT:
|
||||
return "VK_CULL_MODE_FRONT_BIT";
|
||||
case VK_CULL_MODE_BACK_BIT:
|
||||
return "VK_CULL_MODE_BACK_BIT";
|
||||
case VK_CULL_MODE_FRONT_AND_BACK:
|
||||
return "VK_CULL_MODE_FRONT_AND_BACK";
|
||||
default:
|
||||
return "VK_CULL_MODE_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
static const char* CompareOpToString(VkCompareOp compareOp) {
|
||||
switch (compareOp) {
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_NEVER)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_LESS)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_EQUAL)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_LESS_OR_EQUAL)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_GREATER)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_NOT_EQUAL)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_GREATER_OR_EQUAL)
|
||||
ENUM_STR_CASE(VK_COMPARE_OP_ALWAYS)
|
||||
default:
|
||||
return "VK_COMPARE_OP_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
static const char* LogicOpToString(VkLogicOp logicOp) {
|
||||
switch (logicOp) {
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_CLEAR)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_AND)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_AND_REVERSE)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_COPY)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_AND_INVERTED)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_NO_OP)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_XOR)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_OR)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_NOR)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_EQUIVALENT)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_INVERT)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_OR_REVERSE)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_COPY_INVERTED)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_OR_INVERTED)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_NAND)
|
||||
ENUM_STR_CASE(VK_LOGIC_OP_SET)
|
||||
default:
|
||||
return "VK_LOGIC_OP_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
PipelineFactory::PipelineFactory(VkDevice device, const VulkanRendererConfig& config):
|
||||
m_device(device), m_config(config) {
|
||||
MOBILEGL_ASSERT(m_device != VK_NULL_HANDLE, "PipelineFactory: device is null");
|
||||
|
||||
if (m_config.DisablePipelineCache) {
|
||||
MGLOG_I("DirectVulkan: pipeline cache disabled");
|
||||
return;
|
||||
}
|
||||
|
||||
VkPipelineCacheCreateInfo pipelineCacheInfo{VK_STRUCTURE_TYPE_PIPELINE_CACHE_CREATE_INFO};
|
||||
VK_VERIFY(vkCreatePipelineCache(m_device, &pipelineCacheInfo, nullptr, &m_pipelineCache),
|
||||
"vkCreatePipelineCache");
|
||||
}
|
||||
|
||||
// Must be called once, before any pipeline is created: the flag is not part of the
|
||||
// pipeline hash, so flipping it mid-life would serve cached pipelines built under the
|
||||
// old value.
|
||||
void PipelineFactory::SetSuppressBlendedDepthWrite(Bool enabled) {
|
||||
s_suppressBlendedDepthWrite = enabled;
|
||||
}
|
||||
|
||||
Bool PipelineFactory::ShouldSuppressBlendedDepthWriteForDevice(MG_Config::QuirkOverride quirkOverride,
|
||||
Uint32 vendorId) {
|
||||
static constexpr Uint32 kVendorIdQualcomm = 0x5143;
|
||||
switch (quirkOverride) {
|
||||
case MG_Config::QuirkOverride::ForceOn:
|
||||
return true;
|
||||
case MG_Config::QuirkOverride::ForceOff:
|
||||
return false;
|
||||
case MG_Config::QuirkOverride::Auto:
|
||||
default:
|
||||
return vendorId == kVendorIdQualcomm;
|
||||
}
|
||||
}
|
||||
|
||||
namespace {
|
||||
// MIN/MAX extremum blending: the signature of a depth-bounds accumulation pass
|
||||
// (MC 26.3 OIT writes vec4(-linD, linD, deviceZ, 0) under GL_MAX while writing
|
||||
// depth for its equality chain). MIN/MAX ignore blend factors per the Vulkan spec.
|
||||
//
|
||||
// Deliberately the ONLY shape stripped. A quirk should touch as little unrelated
|
||||
// content as possible, and a trace sweep of every fixture showed the wider
|
||||
// alternatives all cost more than they fix:
|
||||
// - additive ONE+ONE with a depth write matched zero draws of the 26.3 chain
|
||||
// (its transmittance/accumulate passes disable depth writes themselves) - the
|
||||
// only real content it caught was harmless additive glow effects (Create);
|
||||
// - sorted-transparency "over" blends (SRC_ALPHA-style) are order-dependent,
|
||||
// drawn once per surface, and rely on their depth writes for occlusion;
|
||||
// - separate-alpha accumulation over an over-blending color channel has no
|
||||
// known pairing with a depth-equality chain (color channel only, see tests).
|
||||
// If a future workload pairs another blend shape with an equality chain, widen
|
||||
// this with that evidence in hand rather than pre-emptively.
|
||||
Bool IsAccumulationBlend(const VkPipelineColorBlendAttachmentState& attachment) {
|
||||
return attachment.colorBlendOp == VK_BLEND_OP_MIN ||
|
||||
attachment.colorBlendOp == VK_BLEND_OP_MAX;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool PipelineFactory::ShouldSuppressDepthWrite(const PipelineCreatePayload& payload) {
|
||||
if (!payload.depthWriteEnable) {
|
||||
return false;
|
||||
}
|
||||
// A shader that assigns gl_FragDepth supplies depth itself rather than taking the
|
||||
// pipeline's interpolated Z, so a driver that varies the vertex position math
|
||||
// between pipelines cannot desynchronize it. (A gl_FragDepth = gl_FragCoord.z
|
||||
// passthrough is the exception that stays exposed; no known content pairs one with
|
||||
// an equality chain, and 26.3's composite is a genuine computed-depth writer.)
|
||||
if (payload.fragmentReplacesDepth) {
|
||||
return false;
|
||||
}
|
||||
for (Uint32 i = 0; i < payload.colorAttachmentCount; ++i) {
|
||||
const VkPipelineColorBlendAttachmentState& attachment = payload.colorBlendAttachments[i];
|
||||
if (attachment.blendEnable != VK_TRUE) {
|
||||
continue;
|
||||
}
|
||||
// All color writes masked: blending is moot (depth-prepass pattern that left
|
||||
// GL_BLEND enabled); stripping the depth write would delete the whole prepass.
|
||||
if (attachment.colorWriteMask == 0) {
|
||||
continue;
|
||||
}
|
||||
// Any attachment qualifies, not just attachment 0: the 26.3 transmittance pass
|
||||
// accumulates into a 2-target MRT and must stay stripped.
|
||||
if (IsAccumulationBlend(attachment)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
PipelineFactory::~PipelineFactory() {
|
||||
DestroyAll();
|
||||
if (m_pipelineCache != VK_NULL_HANDLE) {
|
||||
vkDestroyPipelineCache(m_device, m_pipelineCache, nullptr);
|
||||
m_pipelineCache = VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
|
||||
PipelineFactory::HashType PipelineFactory::ComputeHash(const PipelineCreatePayload& payload) const {
|
||||
@@ -19,17 +199,48 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.vertexInputHash, sizeof(payload.vertexInputHash)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.pipelineLayout, sizeof(payload.pipelineLayout)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.renderPass, sizeof(payload.renderPass)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.colorAttachmentCount, sizeof(payload.colorAttachmentCount)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.rasterizationSamples, sizeof(payload.rasterizationSamples)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.subpass, sizeof(payload.subpass)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.topology, sizeof(payload.topology)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.primitiveRestartEnable, sizeof(payload.primitiveRestartEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.patchControlPoints, sizeof(payload.patchControlPoints)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.viewportCount, sizeof(payload.viewportCount)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.polygonMode, sizeof(payload.polygonMode)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.cullMode, sizeof(payload.cullMode)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.frontFace, sizeof(payload.frontFace)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.provokingVertexMode, sizeof(payload.provokingVertexMode)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.depthTestEnable, sizeof(payload.depthTestEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.depthWriteEnable, sizeof(payload.depthWriteEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.depthBiasEnable, sizeof(payload.depthBiasEnable)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.rasterizerDiscardEnable, sizeof(payload.rasterizerDiscardEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.logicOpEnable, sizeof(payload.logicOpEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.stencilTestEnable, sizeof(payload.stencilTestEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.depthCompareOp, sizeof(payload.depthCompareOp)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.blendEnable, sizeof(payload.blendEnable)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.srcColorBlendFactor, sizeof(payload.srcColorBlendFactor)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.dstColorBlendFactor, sizeof(payload.dstColorBlendFactor)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.srcAlphaBlendFactor, sizeof(payload.srcAlphaBlendFactor)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.dstAlphaBlendFactor, sizeof(payload.dstAlphaBlendFactor)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.colorWriteMask, sizeof(payload.colorWriteMask)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.logicOp, sizeof(payload.logicOp)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.frontStencilFailOp, sizeof(payload.frontStencilFailOp)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.frontStencilPassOp, sizeof(payload.frontStencilPassOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.frontStencilDepthFailOp, sizeof(payload.frontStencilDepthFailOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.frontStencilCompareOp, sizeof(payload.frontStencilCompareOp)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.backStencilFailOp, sizeof(payload.backStencilFailOp)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &payload.backStencilPassOp, sizeof(payload.backStencilPassOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.backStencilDepthFailOp, sizeof(payload.backStencilDepthFailOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.backStencilCompareOp, sizeof(payload.backStencilCompareOp)));
|
||||
XXHASH_VERIFY(
|
||||
XXH64_update(m_hashState, &payload.fragmentReplacesDepth, sizeof(payload.fragmentReplacesDepth)));
|
||||
if (payload.colorAttachmentCount > 0) {
|
||||
XXHASH_VERIFY(XXH64_update(
|
||||
m_hashState,
|
||||
payload.colorBlendAttachments.data(),
|
||||
sizeof(payload.colorBlendAttachments[0]) * payload.colorAttachmentCount));
|
||||
}
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
@@ -37,32 +248,148 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
const HashType hash = ComputeHash(payload);
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
return it->second;
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
return it->second.pipeline;
|
||||
}
|
||||
|
||||
VkPipeline pipeline = CreatePipeline(payload);
|
||||
m_cache.emplace(hash, pipeline);
|
||||
// A failed creation must never be memoized. Caching VK_NULL_HANDLE served the null back for
|
||||
// the rest of the process, so one transient driver rejection turned every later draw with
|
||||
// the same state into a vkCmdBindPipeline(VK_NULL_HANDLE) - the SIGSEGV behind 9 of the 15
|
||||
// CTS process deaths. Retrying costs one failed vkCreateGraphicsPipelines per draw, which
|
||||
// is the correct price for a broken pipeline and is bounded by the draw itself being
|
||||
// skipped.
|
||||
if (pipeline == VK_NULL_HANDLE) {
|
||||
// Unlatched, like the CreatePipeline report it accompanies: a pipeline MobileGL
|
||||
// assembled and the driver refused is a broken invariant, not an expected failure,
|
||||
// so it stays loud for as long as it is reachable. Raised from MGLOG_I once the
|
||||
// Log.h ordering fix made MGLOG_E live in INFO builds.
|
||||
MGLOG_E("PipelineFactory::GetOrCreatePipeline: creation failed for hash=0x%llx "
|
||||
"programHash=0x%llx; not caching the failure",
|
||||
static_cast<unsigned long long>(hash),
|
||||
static_cast<unsigned long long>(payload.programHash));
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
m_cache.emplace(hash, PipelineCacheEntry{pipeline, payload.programHash, payload.renderPass,
|
||||
m_frameCounter});
|
||||
return pipeline;
|
||||
}
|
||||
|
||||
void PipelineFactory::DestroyAll() {
|
||||
for (auto& pair : m_cache) {
|
||||
if (pair.second != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, pair.second, nullptr);
|
||||
if (pair.second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, pair.second.pipeline, nullptr);
|
||||
}
|
||||
}
|
||||
m_cache.clear();
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::OnFrameBoundary() {
|
||||
++m_frameCounter;
|
||||
|
||||
// Sweep cadence and retire age mirror VkRenderPassManager::OnPresent: an entry
|
||||
// idle for more than kRetireAgeFrames frame boundaries cannot be referenced by
|
||||
// any in-flight command buffer (frames-in-flight <= MOBILEGL_MAGMA_FRAMESINFLIGHT),
|
||||
// so immediate vkDestroyPipeline is safe. The caller must drop its "last
|
||||
// pipeline" memo when this returns non-zero: the memo can return a cached
|
||||
// handle without touching this cache, so an evicted pipeline may still be
|
||||
// memoized (present-less flush loops never reset the memo per frame).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeFrames = 1024;
|
||||
if ((m_frameCounter % kSweepInterval) != 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameCounter - it->second.lastUsedFrame > kRetireAgeFrames) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::OnFrameBoundary: evicted %u idle pipelines (%zu remain)", evicted,
|
||||
m_cache.size());
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::EvictByRenderPasses(const Vector<VkRenderPass>& renderPasses) {
|
||||
if (renderPasses.empty() || m_cache.empty()) {
|
||||
return 0;
|
||||
}
|
||||
// Sorted-batch membership test keeps a mass eviction (shader-pack switch,
|
||||
// dimension exit) at one O(cache * log batch) scan instead of one full scan
|
||||
// per dying pass.
|
||||
Vector<VkRenderPass> sortedPasses = renderPasses;
|
||||
std::sort(sortedPasses.begin(), sortedPasses.end());
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (std::binary_search(sortedPasses.begin(), sortedPasses.end(), it->second.renderPass)) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::EvictByRenderPasses: evicted %u pipelines for %zu destroyed render passes",
|
||||
evicted, sortedPasses.size());
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
Uint32 PipelineFactory::EvictByProgramHash(HashType programHash) {
|
||||
Uint32 evicted = 0;
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (it->second.programHash == programHash) {
|
||||
if (it->second.pipeline != VK_NULL_HANDLE) {
|
||||
vkDestroyPipeline(m_device, it->second.pipeline, nullptr);
|
||||
}
|
||||
it = m_cache.erase(it);
|
||||
++evicted;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
if (evicted > 0) {
|
||||
MGLOG_D("PipelineFactory::EvictByProgramHash: evicted %u pipelines for program hash 0x%llx",
|
||||
evicted, static_cast<unsigned long long>(programHash));
|
||||
}
|
||||
return evicted;
|
||||
}
|
||||
|
||||
VkPipeline PipelineFactory::CreatePipeline(const PipelineCreatePayload& payload) const {
|
||||
MOBILEGL_ASSERT(payload.stages != nullptr && !payload.stages->empty(), "PipelineFactory: stages are empty");
|
||||
MOBILEGL_ASSERT(payload.vertexInputState != nullptr, "PipelineFactory: vertexInputState is null");
|
||||
MOBILEGL_ASSERT(payload.pipelineLayout != VK_NULL_HANDLE, "PipelineFactory: pipelineLayout is null");
|
||||
MOBILEGL_ASSERT(payload.renderPass != VK_NULL_HANDLE, "PipelineFactory: renderPass is null");
|
||||
MOBILEGL_ASSERT(payload.colorAttachmentCount <= PipelineCreatePayload::kMaxColorAttachments,
|
||||
"PipelineFactory: colorAttachmentCount=%u is unexpectedly large",
|
||||
payload.colorAttachmentCount);
|
||||
MGLOG_D("PipelineFactory::CreatePipeline: programHash=0x%llx vertexInputHash=0x%llx colorAttachmentCount=%u subpass=%u",
|
||||
static_cast<unsigned long long>(payload.programHash),
|
||||
static_cast<unsigned long long>(payload.vertexInputHash),
|
||||
payload.colorAttachmentCount,
|
||||
payload.subpass);
|
||||
|
||||
static constexpr VkDynamicState kDynamicStates[] = {
|
||||
VK_DYNAMIC_STATE_VIEWPORT,
|
||||
VK_DYNAMIC_STATE_SCISSOR
|
||||
VK_DYNAMIC_STATE_SCISSOR,
|
||||
VK_DYNAMIC_STATE_BLEND_CONSTANTS,
|
||||
VK_DYNAMIC_STATE_DEPTH_BIAS,
|
||||
VK_DYNAMIC_STATE_LINE_WIDTH,
|
||||
VK_DYNAMIC_STATE_STENCIL_COMPARE_MASK,
|
||||
VK_DYNAMIC_STATE_STENCIL_WRITE_MASK,
|
||||
VK_DYNAMIC_STATE_STENCIL_REFERENCE
|
||||
};
|
||||
|
||||
VkPipelineDynamicStateCreateInfo dynamicState{};
|
||||
@@ -72,45 +399,141 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
VkPipelineInputAssemblyStateCreateInfo ia{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO};
|
||||
ia.topology = payload.topology;
|
||||
ia.primitiveRestartEnable = payload.primitiveRestartEnable ? VK_TRUE : VK_FALSE;
|
||||
|
||||
// Only a patch topology has a tessellation stage to configure; leaving the pointer null
|
||||
// otherwise is what the spec expects.
|
||||
VkPipelineTessellationStateCreateInfo tessellation{VK_STRUCTURE_TYPE_PIPELINE_TESSELLATION_STATE_CREATE_INFO};
|
||||
tessellation.patchControlPoints = payload.patchControlPoints;
|
||||
|
||||
VkPipelineViewportStateCreateInfo vpci{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO};
|
||||
vpci.viewportCount = 1;
|
||||
vpci.scissorCount = 1;
|
||||
// Both counts move together: GL has one scissor rectangle per viewport, and Vulkan
|
||||
// requires viewportCount == scissorCount whenever both are dynamic
|
||||
// (VUID-VkPipelineViewportStateCreateInfo-scissorCount-04136). The caller has already
|
||||
// clamped this to the device's multiViewport capability.
|
||||
vpci.viewportCount = std::max<Uint32>(payload.viewportCount, 1u);
|
||||
vpci.scissorCount = vpci.viewportCount;
|
||||
|
||||
VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO};
|
||||
raster.polygonMode = VK_POLYGON_MODE_FILL;
|
||||
raster.cullMode = VK_CULL_MODE_NONE;
|
||||
raster.frontFace = VK_FRONT_FACE_CLOCKWISE;
|
||||
raster.polygonMode = payload.polygonMode;
|
||||
raster.cullMode = payload.cullMode;
|
||||
raster.frontFace = payload.frontFace;
|
||||
raster.depthBiasEnable = payload.depthBiasEnable ? VK_TRUE : VK_FALSE;
|
||||
raster.rasterizerDiscardEnable = payload.rasterizerDiscardEnable ? VK_TRUE : VK_FALSE;
|
||||
raster.lineWidth = 1.0f;
|
||||
// Only chain the struct when the mode is not Vulkan's implicit default: a device without
|
||||
// VK_EXT_provoking_vertex enabled must never see this pNext entry, and the renderer's
|
||||
// selector already collapses to FIRST in exactly that case - so a device without the
|
||||
// extension produces a byte-identical VkGraphicsPipelineCreateInfo to before.
|
||||
VkPipelineRasterizationProvokingVertexStateCreateInfoEXT provokingVertexState{
|
||||
VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_PROVOKING_VERTEX_STATE_CREATE_INFO_EXT};
|
||||
if (payload.provokingVertexMode != VK_PROVOKING_VERTEX_MODE_FIRST_VERTEX_EXT) {
|
||||
provokingVertexState.provokingVertexMode = payload.provokingVertexMode;
|
||||
provokingVertexState.pNext = raster.pNext;
|
||||
raster.pNext = &provokingVertexState;
|
||||
}
|
||||
|
||||
VkPipelineMultisampleStateCreateInfo ms{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO};
|
||||
ms.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT;
|
||||
ms.rasterizationSamples = payload.rasterizationSamples;
|
||||
|
||||
VkPipelineDepthStencilStateCreateInfo depthStencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO};
|
||||
depthStencil.depthTestEnable = payload.depthTestEnable ? VK_TRUE : VK_FALSE;
|
||||
depthStencil.depthWriteEnable = payload.depthWriteEnable ? VK_TRUE : VK_FALSE;
|
||||
depthStencil.depthCompareOp = payload.depthCompareOp;
|
||||
depthStencil.depthBoundsTestEnable = VK_FALSE;
|
||||
depthStencil.stencilTestEnable = VK_FALSE;
|
||||
depthStencil.stencilTestEnable = payload.stencilTestEnable ? VK_TRUE : VK_FALSE;
|
||||
if (payload.stencilTestEnable) {
|
||||
depthStencil.front.failOp = payload.frontStencilFailOp;
|
||||
depthStencil.front.passOp = payload.frontStencilPassOp;
|
||||
depthStencil.front.depthFailOp = payload.frontStencilDepthFailOp;
|
||||
depthStencil.front.compareOp = payload.frontStencilCompareOp;
|
||||
depthStencil.front.compareMask = 0xffffffffu;
|
||||
depthStencil.front.writeMask = 0xffffffffu;
|
||||
depthStencil.front.reference = 0;
|
||||
depthStencil.back.failOp = payload.backStencilFailOp;
|
||||
depthStencil.back.passOp = payload.backStencilPassOp;
|
||||
depthStencil.back.depthFailOp = payload.backStencilDepthFailOp;
|
||||
depthStencil.back.compareOp = payload.backStencilCompareOp;
|
||||
depthStencil.back.compareMask = 0xffffffffu;
|
||||
depthStencil.back.writeMask = 0xffffffffu;
|
||||
depthStencil.back.reference = 0;
|
||||
}
|
||||
|
||||
VkPipelineColorBlendAttachmentState colorAttach{};
|
||||
colorAttach.colorWriteMask = payload.colorWriteMask;
|
||||
colorAttach.blendEnable = payload.blendEnable ? VK_TRUE : VK_FALSE;
|
||||
colorAttach.srcColorBlendFactor = payload.srcColorBlendFactor;
|
||||
colorAttach.dstColorBlendFactor = payload.dstColorBlendFactor;
|
||||
colorAttach.colorBlendOp = VK_BLEND_OP_ADD;
|
||||
colorAttach.srcAlphaBlendFactor = payload.srcAlphaBlendFactor;
|
||||
colorAttach.dstAlphaBlendFactor = payload.dstAlphaBlendFactor;
|
||||
colorAttach.alphaBlendOp = VK_BLEND_OP_ADD;
|
||||
Vector<VkPipelineColorBlendAttachmentState> colorAttachments(payload.colorAttachmentCount);
|
||||
for (Uint32 i = 0; i < payload.colorAttachmentCount; ++i) {
|
||||
colorAttachments[i] = payload.colorBlendAttachments[i];
|
||||
}
|
||||
// Suppress depth writes on accumulation-blended pipelines when the active driver
|
||||
// cannot keep vertex positions invariant across the pipelines of a multi-pass
|
||||
// depth-equality chain (see SetSuppressBlendedDepthWrite). The decision is narrowed
|
||||
// in ShouldSuppressDepthWrite: sorted-transparency "over" blends (vanilla MC water),
|
||||
// gl_FragDepth writers, and masked-out attachments keep their depth writes.
|
||||
// This bakes the decision into the pipeline, which only works because depth write is
|
||||
// static state here - adding VK_DYNAMIC_STATE_DEPTH_WRITE_ENABLE to kDynamicStates
|
||||
// would let the record-time value override it and silently disable the quirk.
|
||||
if (s_suppressBlendedDepthWrite && ShouldSuppressDepthWrite(payload)) {
|
||||
depthStencil.depthWriteEnable = VK_FALSE;
|
||||
}
|
||||
VkPipelineColorBlendStateCreateInfo blend{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO};
|
||||
blend.attachmentCount = 1;
|
||||
blend.pAttachments = &colorAttach;
|
||||
blend.logicOpEnable = payload.logicOpEnable ? VK_TRUE : VK_FALSE;
|
||||
blend.logicOp = payload.logicOp;
|
||||
blend.attachmentCount = payload.colorAttachmentCount;
|
||||
blend.pAttachments = colorAttachments.empty() ? nullptr : colorAttachments.data();
|
||||
|
||||
// A GL program may have a tessellation EVALUATION stage and no CONTROL stage: GL 4.6 core
|
||||
// 11.2.2 gives it a fixed-function pass-through instead. Vulkan has no such stage, and
|
||||
// VUID-VkGraphicsPipelineCreateInfo-pStages-00730 requires both tessellation stages or
|
||||
// neither - so the renderer synthesizes the pass-through GL describes and hands it in
|
||||
// here (see ProgramFactory::GetOrCreatePassthroughTessControlStage).
|
||||
//
|
||||
// The refusal below is what keeps the half-tessellated shape away from the driver when
|
||||
// there is no synthesized stage to add - because Mali does not reject it, it dereferences
|
||||
// null INSIDE vkCreateGraphicsPipelines and takes the process down (SIGSEGV, fault addr
|
||||
// 0x34, on Mali-G715/r54p2 and Mali-G925/r49p1 alike; Adreno and lavapipe merely render
|
||||
// wrong). Returning VK_NULL_HANDLE routes this through the same path a driver rejection
|
||||
// takes: the draw is skipped, nothing is memoised, and the process survives.
|
||||
const Vector<VkPipelineShaderStageCreateInfo>* effectiveStages = payload.stages;
|
||||
Vector<VkPipelineShaderStageCreateInfo> stagesWithPassthrough;
|
||||
if (payload.passthroughTessControlStage.module != VK_NULL_HANDLE) {
|
||||
stagesWithPassthrough = *payload.stages;
|
||||
stagesWithPassthrough.push_back(payload.passthroughTessControlStage);
|
||||
effectiveStages = &stagesWithPassthrough;
|
||||
}
|
||||
{
|
||||
VkShaderStageFlags stagesPresent = 0;
|
||||
for (const auto& stageInfo : *effectiveStages) {
|
||||
stagesPresent |= stageInfo.stage;
|
||||
}
|
||||
const Bool hasTessControl = (stagesPresent & VK_SHADER_STAGE_TESSELLATION_CONTROL_BIT) != 0;
|
||||
const Bool hasTessEval = (stagesPresent & VK_SHADER_STAGE_TESSELLATION_EVALUATION_BIT) != 0;
|
||||
if (hasTessControl != hasTessEval) {
|
||||
// Latched, and the latch is the point: a failed creation is deliberately never
|
||||
// memoised (see GetOrCreatePipeline), so a program in this state re-enters here
|
||||
// once per draw, every frame - and a refusal diagnostic that repeats per draw is
|
||||
// noise, not a diagnostic. One line names the program; the draws it explains are
|
||||
// all the same draw.
|
||||
static Bool s_warnedHalfTessellatedPipeline = false;
|
||||
if (!s_warnedHalfTessellatedPipeline) {
|
||||
s_warnedHalfTessellatedPipeline = true;
|
||||
MGLOG_E_ONCE("PipelineFactory::CreatePipeline: refusing a pipeline with %s tessellation stage and "
|
||||
"no %s stage (VUID-VkGraphicsPipelineCreateInfo-pStages-00730). programHash=0x%llx "
|
||||
"patchControlPoints=%u. Its draws are skipped; logged once.",
|
||||
hasTessEval ? "an evaluation" : "a control",
|
||||
hasTessEval ? "control" : "evaluation",
|
||||
static_cast<unsigned long long>(payload.programHash),
|
||||
payload.patchControlPoints);
|
||||
}
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
|
||||
VkGraphicsPipelineCreateInfo gpi{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO};
|
||||
gpi.stageCount = static_cast<Uint32>(payload.stages->size());
|
||||
gpi.pStages = payload.stages->data();
|
||||
gpi.stageCount = static_cast<Uint32>(effectiveStages->size());
|
||||
gpi.pStages = effectiveStages->data();
|
||||
gpi.pVertexInputState = payload.vertexInputState;
|
||||
gpi.pInputAssemblyState = &ia;
|
||||
gpi.pTessellationState =
|
||||
payload.topology == VK_PRIMITIVE_TOPOLOGY_PATCH_LIST ? &tessellation : nullptr;
|
||||
gpi.pViewportState = &vpci;
|
||||
gpi.pRasterizationState = &raster;
|
||||
gpi.pMultisampleState = &ms;
|
||||
@@ -122,8 +545,89 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
gpi.subpass = payload.subpass;
|
||||
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateGraphicsPipelines(m_device, VK_NULL_HANDLE, 1, &gpi, nullptr, &pipeline),
|
||||
"vkCreateGraphicsPipelines");
|
||||
const VkResult result = vkCreateGraphicsPipelines(m_device, m_pipelineCache, 1, &gpi, nullptr, &pipeline);
|
||||
// Loud, at MGLOG_F, and deliberately NOT latched. vkCreateGraphicsPipelines refusing a
|
||||
// pipeline MobileGL assembled is a should-never-happen state, and the driver's own
|
||||
// answer is VK_ERROR_UNKNOWN - no information at all - so this dump is the entire
|
||||
// diagnosis. It is not an expected failure mode, so the one-shot rule that quiets W/E
|
||||
// does not apply: while this is reachable it should keep saying so on every draw.
|
||||
// GetOrCreatePipeline deliberately does not cache the failure, which is what makes that
|
||||
// repetition happen; if the repetition ever needs to stop, fix the pipeline, not the log.
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_F("PipelineFactory::CreatePipeline failed: result=%s (%d) programHash=0x%llx vertexInputHash=0x%llx stageCount=%u topology=%s(%d) colorAttachmentCount=%u samples=%s(%d) subpass=%u",
|
||||
VkResultToString(result),
|
||||
result,
|
||||
static_cast<unsigned long long>(payload.programHash),
|
||||
static_cast<unsigned long long>(payload.vertexInputHash),
|
||||
gpi.stageCount,
|
||||
PrimitiveTopologyToString(payload.topology),
|
||||
payload.topology,
|
||||
payload.colorAttachmentCount,
|
||||
SampleCountToString(payload.rasterizationSamples),
|
||||
payload.rasterizationSamples,
|
||||
payload.subpass);
|
||||
MGLOG_F("PipelineFactory::CreatePipeline state: cullMode=%s(0x%x) frontFace=%d depthTest=%d depthWrite=%d depthCompare=%s(%d) depthBias=%d rasterizerDiscard=%d stencilTest=%d logicOpEnable=%d logicOp=%s(%d)",
|
||||
CullModeToString(payload.cullMode),
|
||||
static_cast<Uint32>(payload.cullMode),
|
||||
payload.frontFace,
|
||||
payload.depthTestEnable ? 1 : 0,
|
||||
payload.depthWriteEnable ? 1 : 0,
|
||||
CompareOpToString(payload.depthCompareOp),
|
||||
payload.depthCompareOp,
|
||||
payload.depthBiasEnable ? 1 : 0,
|
||||
payload.rasterizerDiscardEnable ? 1 : 0,
|
||||
payload.stencilTestEnable ? 1 : 0,
|
||||
payload.logicOpEnable ? 1 : 0,
|
||||
LogicOpToString(payload.logicOp),
|
||||
payload.logicOp);
|
||||
MGLOG_F("PipelineFactory::CreatePipeline vertex input: bindingCount=%u attributeCount=%u",
|
||||
payload.vertexInputState->vertexBindingDescriptionCount,
|
||||
payload.vertexInputState->vertexAttributeDescriptionCount);
|
||||
// The driver's own answer is VK_ERROR_UNKNOWN, i.e. no information at all, so the only
|
||||
// way to work out WHICH shader it choked on (the open sampler-array-in-struct
|
||||
// investigation) is to name the modules. MGLOG_I, not _D: this is part of a
|
||||
// should-never-happen report and must survive in the INFO-level builds that CTS
|
||||
// actually runs against, alongside the MGLOG_F lines above.
|
||||
if (payload.stageSpirvDigests) {
|
||||
for (SizeT i = 0; i < payload.stageSpirvDigests->size(); ++i) {
|
||||
const auto& digest = (*payload.stageSpirvDigests)[i];
|
||||
MGLOG_I("PipelineFactory::CreatePipeline spirv[%zu]: stage=0x%x words=%u bytes=%zu "
|
||||
"hash=0x%llx",
|
||||
i, digest.stage, digest.wordCount,
|
||||
static_cast<SizeT>(digest.wordCount) * sizeof(Uint32),
|
||||
static_cast<unsigned long long>(digest.hash));
|
||||
}
|
||||
} else {
|
||||
MGLOG_I("PipelineFactory::CreatePipeline: no SPIR-V digests attached to the payload");
|
||||
}
|
||||
if (payload.stages) {
|
||||
for (SizeT i = 0; i < payload.stages->size(); ++i) {
|
||||
const auto& stage = (*payload.stages)[i];
|
||||
// VkShaderModule is a non-dispatchable handle: a pointer on 64-bit but a
|
||||
// plain uint64_t on 32-bit ABIs, where a cast to const void* is ill-formed
|
||||
// (broke the armeabi-v7a build). Print it as the 64-bit value it is.
|
||||
MGLOG_I("PipelineFactory::CreatePipeline stage[%zu]: stage=0x%x module=0x%llx entry=%s "
|
||||
"specialization=%d",
|
||||
i, static_cast<Uint32>(stage.stage),
|
||||
static_cast<unsigned long long>(reinterpret_cast<Uint64>(stage.module)),
|
||||
stage.pName ? stage.pName : "(null)", stage.pSpecializationInfo ? 1 : 0);
|
||||
}
|
||||
}
|
||||
for (Uint32 i = 0; i < payload.colorAttachmentCount; ++i) {
|
||||
const auto& attachment = payload.colorBlendAttachments[i];
|
||||
MGLOG_F("PipelineFactory::CreatePipeline colorAttachment[%u]: blend=%d colorWriteMask=0x%x srcColor=%d dstColor=%d colorOp=%d srcAlpha=%d dstAlpha=%d alphaOp=%d",
|
||||
i,
|
||||
attachment.blendEnable == VK_TRUE ? 1 : 0,
|
||||
static_cast<Uint32>(attachment.colorWriteMask),
|
||||
attachment.srcColorBlendFactor,
|
||||
attachment.dstColorBlendFactor,
|
||||
attachment.colorBlendOp,
|
||||
attachment.srcAlphaBlendFactor,
|
||||
attachment.dstAlphaBlendFactor,
|
||||
attachment.alphaBlendOp);
|
||||
}
|
||||
}
|
||||
VK_VERIFY(result, "vkCreateGraphicsPipelines");
|
||||
return pipeline;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -10,37 +10,92 @@
|
||||
|
||||
#include "Config.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// Enough of a fingerprint to identify the exact module the driver rejected without keeping the
|
||||
// SPIR-V alive for every program in the cache: a driver that answers VK_ERROR_UNKNOWN tells us
|
||||
// nothing, so the log has to carry the shader's identity itself. Diagnostic only - never part
|
||||
// of any pipeline or program hash.
|
||||
struct ShaderStageSpirvDigest {
|
||||
Uint32 stage = 0; // VkShaderStageFlagBits
|
||||
Uint32 wordCount = 0;
|
||||
Uint64 hash = 0;
|
||||
};
|
||||
|
||||
class PipelineFactory {
|
||||
public:
|
||||
using HashType = Uint64;
|
||||
|
||||
struct PipelineCreatePayload {
|
||||
static constexpr Uint32 kMaxColorAttachments = MG_State::GLState::FramebufferObject::MAX_DRAW_BUFFERS;
|
||||
|
||||
HashType programHash = 0;
|
||||
HashType vertexInputHash = 0;
|
||||
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
Uint32 colorAttachmentCount = 1;
|
||||
VkSampleCountFlagBits rasterizationSamples = VK_SAMPLE_COUNT_1_BIT;
|
||||
Uint32 subpass = 0;
|
||||
VkPrimitiveTopology topology = VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST;
|
||||
Bool primitiveRestartEnable = false;
|
||||
// GL_PATCH_VERTICES; only read for a PATCH_LIST topology.
|
||||
Uint32 patchControlPoints = 3;
|
||||
// How many of ARB_viewport_array's viewports this pipeline rasterizes into. 1 for
|
||||
// every program that never assigns gl_ViewportIndex, which is all of them outside the
|
||||
// conformance suite - the wide shape costs a longer vkCmdSetViewport/Scissor per state
|
||||
// change and can cost hardware fast paths, so it is opt-in per program. Baked into the
|
||||
// pipeline (viewportCount is not dynamic without VK_EXT_extended_dynamic_state) and
|
||||
// therefore hashed; the DYNAMIC viewport/scissor arrays the draw pushes must have
|
||||
// exactly this many elements (VUID-vkCmdDraw-viewportCount-03417/-03418).
|
||||
Uint32 viewportCount = 1;
|
||||
VkPolygonMode polygonMode = VK_POLYGON_MODE_FILL;
|
||||
VkCullModeFlags cullMode = VK_CULL_MODE_BACK_BIT;
|
||||
VkFrontFace frontFace = VK_FRONT_FACE_CLOCKWISE;
|
||||
// GL's provoking vertex, baked into the pipeline (VK_EXT_provoking_vertex). It selects
|
||||
// which vertex a flat varying takes AND the vertex order transform feedback records for
|
||||
// strips/fans, so it is part of the pipeline's identity, not dynamic state. Defaults to
|
||||
// Vulkan's own convention, which is what a device without the extension gets.
|
||||
VkProvokingVertexModeEXT provokingVertexMode = VK_PROVOKING_VERTEX_MODE_FIRST_VERTEX_EXT;
|
||||
Bool depthTestEnable = false;
|
||||
Bool depthWriteEnable = false;
|
||||
Bool depthBiasEnable = false;
|
||||
Bool rasterizerDiscardEnable = false;
|
||||
Bool logicOpEnable = false;
|
||||
Bool stencilTestEnable = false;
|
||||
VkCompareOp depthCompareOp = VK_COMPARE_OP_ALWAYS;
|
||||
Bool blendEnable = false;
|
||||
VkBlendFactor srcColorBlendFactor = VK_BLEND_FACTOR_ONE;
|
||||
VkBlendFactor dstColorBlendFactor = VK_BLEND_FACTOR_ZERO;
|
||||
VkBlendFactor srcAlphaBlendFactor = VK_BLEND_FACTOR_ONE;
|
||||
VkBlendFactor dstAlphaBlendFactor = VK_BLEND_FACTOR_ZERO;
|
||||
VkColorComponentFlags colorWriteMask =
|
||||
VK_COLOR_COMPONENT_R_BIT | VK_COLOR_COMPONENT_G_BIT |
|
||||
VK_COLOR_COMPONENT_B_BIT | VK_COLOR_COMPONENT_A_BIT;
|
||||
VkLogicOp logicOp = VK_LOGIC_OP_COPY;
|
||||
VkStencilOp frontStencilFailOp = VK_STENCIL_OP_KEEP;
|
||||
VkStencilOp frontStencilPassOp = VK_STENCIL_OP_KEEP;
|
||||
VkStencilOp frontStencilDepthFailOp = VK_STENCIL_OP_KEEP;
|
||||
VkCompareOp frontStencilCompareOp = VK_COMPARE_OP_ALWAYS;
|
||||
VkStencilOp backStencilFailOp = VK_STENCIL_OP_KEEP;
|
||||
VkStencilOp backStencilPassOp = VK_STENCIL_OP_KEEP;
|
||||
VkStencilOp backStencilDepthFailOp = VK_STENCIL_OP_KEEP;
|
||||
VkCompareOp backStencilCompareOp = VK_COMPARE_OP_ALWAYS;
|
||||
// The fragment module writes gl_FragDepth (SPIR-V DepthReplacing); exempts the
|
||||
// pipeline from the blended depth-write quirk (see ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
Array<VkPipelineColorBlendAttachmentState, kMaxColorAttachments> colorBlendAttachments{};
|
||||
const Vector<VkPipelineShaderStageCreateInfo>* stages = nullptr;
|
||||
// The tessellation control stage this renderer synthesized for a program that has
|
||||
// an evaluation stage and none of its own (GL 4.6 core 11.2.2 gives such a program a
|
||||
// fixed-function pass-through; Vulkan has no such thing and
|
||||
// VUID-VkGraphicsPipelineCreateInfo-pStages-00730 forbids the half-tessellated
|
||||
// pipeline outright). Appended to `stages` at creation. A null module means the
|
||||
// renderer could not build one, and CreatePipeline refuses the pipeline - the same
|
||||
// refusal it applies when `stages` itself is half-tessellated.
|
||||
//
|
||||
// NOT hashed: it is a pure function of the program and of patchControlPoints, both
|
||||
// of which ComputeHash already mixes in.
|
||||
VkPipelineShaderStageCreateInfo passthroughTessControlStage{};
|
||||
const VkPipelineVertexInputStateCreateInfo* vertexInputState = nullptr;
|
||||
// Diagnostic only; may be null. Read solely from the pipeline-creation failure path.
|
||||
const Vector<ShaderStageSpirvDigest>* stageSpirvDigests = nullptr;
|
||||
};
|
||||
|
||||
explicit PipelineFactory(VkDevice device, const VulkanRendererConfig& config):
|
||||
m_device(device), m_config(config) {}
|
||||
explicit PipelineFactory(VkDevice device, const VulkanRendererConfig& config);
|
||||
~PipelineFactory();
|
||||
PipelineFactory(const PipelineFactory&) = delete;
|
||||
|
||||
@@ -48,12 +103,69 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPipeline GetOrCreatePipeline(const PipelineCreatePayload& payload);
|
||||
void DestroyAll();
|
||||
|
||||
// Frame boundary hook: ages the pipeline cache and destroys long-unused entries
|
||||
// (their command buffers retired many frames ago), mirroring
|
||||
// VkRenderPassManager::OnPresent's sweep. Returns the number of pipelines
|
||||
// destroyed so the caller can drop any memoized VkPipeline handle.
|
||||
Uint32 OnFrameBoundary();
|
||||
// Destroys every cached pipeline hashed on one of `renderPasses`. Only safe
|
||||
// when the caller guarantees GPU idleness for them - the render-pass manager
|
||||
// calls this (via the renderer) for passes its own >1024-boundary-idle sweep
|
||||
// just evicted, and a pipeline hashed on those handles is only ever bound by
|
||||
// draws that also hit the render-pass entries. Also closes the handle-recycling
|
||||
// hazard: a recycled VkRenderPass value must never serve a stale pipeline.
|
||||
// Batched: one cache scan regardless of how many passes died in the sweep.
|
||||
// Returns the number destroyed (callers invalidate memos when non-zero).
|
||||
Uint32 EvictByRenderPasses(const Vector<VkRenderPass>& renderPasses);
|
||||
// Destroys every cached pipeline built from the program with content hash
|
||||
// `programHash`. Called from the ProgramFactory eviction path, which proves the
|
||||
// same >1024-boundary idleness (the program's pipelines are only bound by draws
|
||||
// that stamp its factory entry). Returns the number destroyed.
|
||||
Uint32 EvictByProgramHash(HashType programHash);
|
||||
|
||||
// Driver quirk: suppress depth writes on accumulation-blended pipelines. Multi-pass
|
||||
// depth-equality rendering (a blended prepass writes depth that later passes re-test
|
||||
// with an equality-inclusive compare on the re-rasterized geometry) requires
|
||||
// cross-pipeline position invariance that some mobile compilers do not provide, even
|
||||
// with the SPIR-V Invariant decoration; whole primitives then drop out of the later
|
||||
// passes. Only MIN/MAX extremum blends are stripped - the signature of such a
|
||||
// chain's depth-bounds pass (MC 26.3 OIT), and per a fixture-wide trace sweep the
|
||||
// only depth-writing shape the chain actually uses - so every other blend
|
||||
// (sorted-transparency "over" like vanilla MC water, additive glows, ...) keeps
|
||||
// its depth writes. Set at renderer initialization based on the active driver.
|
||||
static void SetSuppressBlendedDepthWrite(Bool enabled);
|
||||
static Bool IsSuppressBlendedDepthWriteEnabled() { return s_suppressBlendedDepthWrite; }
|
||||
// Device gate for the quirk: ForceOn/ForceOff bypass detection, Auto enables it on
|
||||
// the known-affected vendor (Qualcomm).
|
||||
static Bool ShouldSuppressBlendedDepthWriteForDevice(MG_Config::QuirkOverride quirkOverride,
|
||||
Uint32 vendorId);
|
||||
// Pure per-pipeline strip decision (exempts gl_FragDepth writers, masked-out and
|
||||
// non-accumulation blends); combined with the device flag in CreatePipeline. Static
|
||||
// and payload-only so tests can pin the contract without a VkDevice.
|
||||
static Bool ShouldSuppressDepthWrite(const PipelineCreatePayload& payload);
|
||||
|
||||
private:
|
||||
struct PipelineCacheEntry {
|
||||
VkPipeline pipeline = VK_NULL_HANDLE;
|
||||
// The hashed inputs the eviction paths key on: programHash ties the entry to
|
||||
// its ProgramFactory entry, renderPass records the exact handle the hash
|
||||
// folded in (the hash is one-way, so targeted eviction needs them verbatim).
|
||||
HashType programHash = 0;
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
// Frame-boundary counter value of the last GetOrCreatePipeline hit; drives
|
||||
// cache eviction (see OnFrameBoundary).
|
||||
Uint64 lastUsedFrame = 0;
|
||||
};
|
||||
|
||||
VkPipeline CreatePipeline(const PipelineCreatePayload& payload) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
const VulkanRendererConfig& m_config;
|
||||
UnorderedMap<HashType, VkPipeline> m_cache;
|
||||
VkPipelineCache m_pipelineCache = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, PipelineCacheEntry> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
static inline Bool s_suppressBlendedDepthWrite = false;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -9,13 +9,39 @@
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include "PipelineFactory.h"
|
||||
#include "MG_State/GLState/ProgramState/ProgramObject.h"
|
||||
#include "MG_State/GLState/ProgramState/ShaderObject.h"
|
||||
#include "MG_State/GLState/TextureState/TextureEnum.h"
|
||||
|
||||
#include <Includes.h>
|
||||
#include <spirv_reflect.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class SamplerNumericDomain : Uint8 {
|
||||
Unknown = 0,
|
||||
Float,
|
||||
SignedInteger,
|
||||
UnsignedInteger,
|
||||
};
|
||||
|
||||
class ProgramFactory {
|
||||
public:
|
||||
enum class DescriptorBindingKind : Uint8 {
|
||||
None = 0,
|
||||
UniformBufferDynamic,
|
||||
CombinedImageSampler,
|
||||
UniformTexelBuffer,
|
||||
StorageBuffer,
|
||||
StorageImage,
|
||||
// GLSL `imageBuffer` - a buffer texture reached through an IMAGE unit rather than a
|
||||
// texture unit. Vulkan spells it VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER, which is a
|
||||
// VkBufferView like UniformTexelBuffer and not a VkImageView like StorageImage: it is
|
||||
// the one image uniform whose descriptor is a buffer. Appended, never inserted -
|
||||
// DescriptorKeyHash mixes the enumerator's value.
|
||||
StorageTexelBuffer
|
||||
};
|
||||
|
||||
enum class CompileOptionBit : Uint {
|
||||
None = 0,
|
||||
PositionYFlip = 1 << 0,
|
||||
@@ -23,72 +49,442 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
SurfaceRotate90 = 1 << 2,
|
||||
SurfaceRotate180 = 1 << 3,
|
||||
SurfaceRotate270 = 1 << 4,
|
||||
// Rewrites the fragment stage's implicit-LOD image samples to explicit LOD 0.
|
||||
// Only ever set for a draw whose every sampler binding is clamped to a single mip
|
||||
// level, which makes the two forms produce identical texels (the implicit lambda is
|
||||
// clamped into [minLod, maxLod] = [0, 0] regardless of derivatives or bias).
|
||||
ExplicitLod0Sampling = 1 << 5,
|
||||
// Decorates the last vertex-processing stage's captured varyings with
|
||||
// XfbBuffer/XfbStride/Offset (VK_EXT_transform_feedback). Set only for draws
|
||||
// recorded while GL transform feedback is active, so plain draws keep the
|
||||
// undecorated variant.
|
||||
XfbCapture = 1 << 6,
|
||||
// Rewrites the fragment stage's gl_FragCoord reads to GL's bottom-left window
|
||||
// origin. Vulkan's gl_FragCoord.y IS the framebuffer row being written, and the
|
||||
// default framebuffer's image is stored in display (top-left) order, so a shader
|
||||
// that reads gl_FragCoord there sees `height - y_GL`. Set together with
|
||||
// PositionYFlip (the two are the same fact about the same draws) except under a
|
||||
// quarter turn, which this renderer does not convert rectangles for either.
|
||||
FragCoordYFlip = 1 << 7,
|
||||
// Replaces the vertex stage's gl_BaseVertex reads with zero. GL defines the builtin
|
||||
// as zero for every drawing command that has no baseVertex parameter - all the
|
||||
// DrawArrays forms - while Vulkan's BaseVertex reports firstVertex there. Set only
|
||||
// for a non-indexed draw whose program actually reads the builtin, so nothing else
|
||||
// acquires a second program/pipeline variant. See ZeroBaseVertexPass.
|
||||
ZeroBaseVertex = 1 << 8,
|
||||
};
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
struct BackendProgramObject {
|
||||
|
||||
struct VkProgramObject {
|
||||
static constexpr Uint32 kMaxVertexInputLocations = 32;
|
||||
|
||||
HashType hash = 0;
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
Vector<VkPipelineShaderStageCreateInfo> stages;
|
||||
Vector<VkShaderModule> modules;
|
||||
// Parallel to stages; identifies the exact module bytes handed to the driver when a
|
||||
// pipeline creation fails. Sixteen bytes per stage instead of keeping the SPIR-V.
|
||||
Vector<ShaderStageSpirvDigest> stageSpirvDigests;
|
||||
|
||||
BackendProgramObject() = default;
|
||||
BackendProgramObject(const BackendProgramObject&) = delete;
|
||||
BackendProgramObject& operator=(const BackendProgramObject&) = delete;
|
||||
BackendProgramObject(BackendProgramObject&& other) noexcept {
|
||||
// Layout data (previously in separate VkProgramLayout)
|
||||
VkDescriptorSetLayout descriptorSetLayout = VK_NULL_HANDLE;
|
||||
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
|
||||
Vector<DescriptorBindingKind> bindingKinds;
|
||||
// The bindings this program actually declares, ascending. bindingKinds is sized to the
|
||||
// 256-binding cap while a real GL program uses 1-8, so the per-draw descriptor walk was
|
||||
// scanning 256 slots to find a handful. MUST stay ascending: Vulkan consumes
|
||||
// pDynamicOffsets in binding order and the writer pushes them in iteration order, so an
|
||||
// unordered list would silently mis-pair dynamic offsets with their uniform blocks.
|
||||
Vector<Uint32> activeBindings;
|
||||
Vector<Uint32> dynamicBindings;
|
||||
Vector<Int> uniformBlockIndexByBinding;
|
||||
// Descriptor count per binding (1 except for a descriptor ARRAY - a UBO or storage
|
||||
// block instance array, an image uniform array or a sampler uniform array - each of
|
||||
// which occupies one binding with descriptorCount = N).
|
||||
Vector<Uint16> bindingDescriptorCounts;
|
||||
// Per-element GL uniform block indices for arrayed UBO bindings (count > 1);
|
||||
// element 0 of a non-arrayed binding stays in uniformBlockIndexByBinding.
|
||||
UnorderedMap<Uint32, Vector<Int>> arrayedUniformBlockIndicesByBinding;
|
||||
Vector<String> samplerNameByBinding;
|
||||
Vector<Int> samplerUniformLocationByBinding;
|
||||
Vector<TextureTarget> samplerTextureTargetByBinding;
|
||||
Vector<SamplerNumericDomain> samplerNumericDomainByBinding;
|
||||
// Shared by StorageImage and StorageTexelBuffer bindings: a binding is one kind or
|
||||
// the other, never both, and both need exactly the same thing - the format the
|
||||
// shader declared, so the per-draw resolve can tell a typed declaration from a
|
||||
// formatless one. Kept as one pair rather than two so the move operations below
|
||||
// cannot drift out of sync with a field that only one kind populates.
|
||||
Vector<VkFormat> storageImageFormatByBinding;
|
||||
Vector<Bool> storageImageUsesBindingFormatByBinding;
|
||||
Vector<String> storageBlockNameByBinding;
|
||||
Vector<Int> storageBlockIndexByBinding;
|
||||
// Set once during ReflectLayout so the per-draw path can skip the whole
|
||||
// storage-image preparation for the overwhelming majority of programs.
|
||||
Bool hasStorageImages = false;
|
||||
// Something about this program's descriptors could not be resolved - an opaque
|
||||
// uniform array whose elements have no addressable uniform locations (the
|
||||
// multi-dimensional case), or a binding remap that failed outright. The binding
|
||||
// STAYS DECLARED in the descriptor set layout; declining is done here, by refusing
|
||||
// every draw, and BindProgramUniformBuffers returns false so the draw setup skips
|
||||
// the draw exactly as it does for any other bind failure.
|
||||
//
|
||||
// Keeping the layout intact is the load-bearing half. Shrinking it instead - which
|
||||
// is what the first cut of this did - leaves the shader reading a descriptor the
|
||||
// layout never declared, and lavapipe segfaults on that inside PIPELINE CREATION,
|
||||
// in a JIT worker thread, before any draw runs where a refusal could help. The
|
||||
// reason was logged once at MGLOG_I when the descriptor was declined.
|
||||
Bool declinedDescriptors = false;
|
||||
Int globalUboBinding = -1;
|
||||
Uint32 activeVertexInputLocationMask = 0;
|
||||
Array<GLenum, kMaxVertexInputLocations> vertexInputTypes{};
|
||||
Uint32 activeFragmentOutputLocationMask = 0;
|
||||
Array<GLenum, kMaxVertexInputLocations> fragmentOutputTypes{};
|
||||
ShaderStage rasterizationProducerStage = ShaderStage::Unknown;
|
||||
Uint32 producerOutputComponentCount = 0;
|
||||
Uint32 fragmentInputComponentCount = 0;
|
||||
// The fragment module declares the DepthReplacing execution mode (writes
|
||||
// gl_FragDepth); shader-computed depth is immune to the cross-pipeline
|
||||
// position-invariance quirk (see PipelineFactory::ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
// The vertex module declares the BaseVertex builtin. Selects the ZeroBaseVertex
|
||||
// program variant for non-indexed draws, and is deliberately a property of the
|
||||
// PROGRAM rather than of the variant: the zeroed variant leaves the variable
|
||||
// declared, so both variants answer the same and the draw path can ask either.
|
||||
Bool readsBaseVertexBuiltin = false;
|
||||
// Some pre-rasterization stage assigns gl_ViewportIndex. Its pipeline declares
|
||||
// viewportCount = the renderer's rasterizable viewport count instead of 1, and its
|
||||
// draws push the whole viewport/scissor array; every other program keeps the
|
||||
// single-viewport fast path untouched. Part of the program's identity (folded into
|
||||
// the pipeline hash through programHash), so no memo can serve the wrong shape.
|
||||
Bool writesViewportIndexBuiltin = false;
|
||||
// This program has a tessellation EVALUATION stage and no tessellation CONTROL
|
||||
// stage. GL allows that (4.6 core 11.2.2: with no control shader the input patch
|
||||
// is passed through unmodified, the output patch size is PATCH_VERTICES, and the
|
||||
// levels come from the PATCH_DEFAULT_*_LEVEL state); Vulkan does not - either both
|
||||
// tessellation stages are present or neither
|
||||
// (VUID-VkGraphicsPipelineCreateInfo-pStages-00730). So the draw path has to supply
|
||||
// the pass-through stage GL describes; see GetOrCreatePassthroughTessControlStage.
|
||||
Bool needsPassthroughTessControl = false;
|
||||
// ...and the pass-through this renderer can synthesize carries gl_Position and
|
||||
// nothing else, so it is only correct when the evaluation stage's inputs are
|
||||
// built-ins. A user-defined varying would arrive at the evaluation stage
|
||||
// UNWRITTEN once a control stage sits between it and the vertex stage, which is
|
||||
// silently wrong pixels rather than a crash - so those programs are declined
|
||||
// instead (PipelineFactory::CreatePipeline refuses the pipeline and the draw is
|
||||
// skipped). See ReflectPassthroughTessControlNeed.
|
||||
Bool passthroughTessControlEmulatable = false;
|
||||
// Frame-boundary counter value of the last GetOrCreateProgram hit; drives
|
||||
// cache eviction (see OnFrameBoundary). Mutable: the draw snapshot's memoised
|
||||
// entry pointer re-stamps use through a const reference (StampProgramUse).
|
||||
mutable Uint64 lastUsedFrame = 0;
|
||||
|
||||
static inline VkDevice s_device = VK_NULL_HANDLE;
|
||||
|
||||
VkProgramObject() = default;
|
||||
VkProgramObject(const VkProgramObject&) = delete;
|
||||
VkProgramObject& operator=(const VkProgramObject&) = delete;
|
||||
VkProgramObject(VkProgramObject&& other) noexcept {
|
||||
hash = other.hash;
|
||||
device = other.device;
|
||||
stages = std::move(other.stages);
|
||||
modules = std::move(other.modules);
|
||||
// Must travel with `modules`: these digests name the SPIR-V those exact
|
||||
// shader modules were built from, and the pipeline-failure diagnostics
|
||||
// print the two together. Leaving it behind used to merely lose the
|
||||
// digests on a rehash; now that the cache is a robin-hood table, insertion
|
||||
// SWAPS two entries, and a field that no move touches stays behind in the
|
||||
// slot - pairing one program's modules with another program's digests, so
|
||||
// a pipeline failure would be reported against the wrong SPIR-V.
|
||||
stageSpirvDigests = std::move(other.stageSpirvDigests);
|
||||
descriptorSetLayout = other.descriptorSetLayout;
|
||||
pipelineLayout = other.pipelineLayout;
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
activeBindings = std::move(other.activeBindings);
|
||||
dynamicBindings = std::move(other.dynamicBindings);
|
||||
uniformBlockIndexByBinding = std::move(other.uniformBlockIndexByBinding);
|
||||
bindingDescriptorCounts = std::move(other.bindingDescriptorCounts);
|
||||
arrayedUniformBlockIndicesByBinding = std::move(other.arrayedUniformBlockIndicesByBinding);
|
||||
samplerNameByBinding = std::move(other.samplerNameByBinding);
|
||||
samplerUniformLocationByBinding = std::move(other.samplerUniformLocationByBinding);
|
||||
samplerTextureTargetByBinding = std::move(other.samplerTextureTargetByBinding);
|
||||
samplerNumericDomainByBinding = std::move(other.samplerNumericDomainByBinding);
|
||||
storageImageFormatByBinding = std::move(other.storageImageFormatByBinding);
|
||||
storageImageUsesBindingFormatByBinding =
|
||||
std::move(other.storageImageUsesBindingFormatByBinding);
|
||||
storageBlockNameByBinding = std::move(other.storageBlockNameByBinding);
|
||||
storageBlockIndexByBinding = std::move(other.storageBlockIndexByBinding);
|
||||
hasStorageImages = other.hasStorageImages;
|
||||
declinedDescriptors = other.declinedDescriptors;
|
||||
globalUboBinding = other.globalUboBinding;
|
||||
activeVertexInputLocationMask = other.activeVertexInputLocationMask;
|
||||
vertexInputTypes = other.vertexInputTypes;
|
||||
activeFragmentOutputLocationMask = other.activeFragmentOutputLocationMask;
|
||||
fragmentOutputTypes = other.fragmentOutputTypes;
|
||||
rasterizationProducerStage = other.rasterizationProducerStage;
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
|
||||
needsPassthroughTessControl = other.needsPassthroughTessControl;
|
||||
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.device = VK_NULL_HANDLE;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.declinedDescriptors = false;
|
||||
other.globalUboBinding = -1;
|
||||
other.activeVertexInputLocationMask = 0;
|
||||
other.activeFragmentOutputLocationMask = 0;
|
||||
other.rasterizationProducerStage = ShaderStage::Unknown;
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.readsBaseVertexBuiltin = false;
|
||||
other.needsPassthroughTessControl = false;
|
||||
other.passthroughTessControlEmulatable = false;
|
||||
other.lastUsedFrame = 0;
|
||||
}
|
||||
BackendProgramObject& operator=(BackendProgramObject&& other) noexcept {
|
||||
VkProgramObject& operator=(VkProgramObject&& other) noexcept {
|
||||
if (this == &other) {
|
||||
return *this;
|
||||
}
|
||||
DestroyModules();
|
||||
stages.clear();
|
||||
Destroy();
|
||||
hash = other.hash;
|
||||
device = other.device;
|
||||
stages = std::move(other.stages);
|
||||
modules = std::move(other.modules);
|
||||
stageSpirvDigests = std::move(other.stageSpirvDigests); // travels with `modules` - see the move ctor
|
||||
descriptorSetLayout = other.descriptorSetLayout;
|
||||
pipelineLayout = other.pipelineLayout;
|
||||
bindingKinds = std::move(other.bindingKinds);
|
||||
activeBindings = std::move(other.activeBindings);
|
||||
dynamicBindings = std::move(other.dynamicBindings);
|
||||
uniformBlockIndexByBinding = std::move(other.uniformBlockIndexByBinding);
|
||||
bindingDescriptorCounts = std::move(other.bindingDescriptorCounts);
|
||||
arrayedUniformBlockIndicesByBinding = std::move(other.arrayedUniformBlockIndicesByBinding);
|
||||
samplerNameByBinding = std::move(other.samplerNameByBinding);
|
||||
samplerUniformLocationByBinding = std::move(other.samplerUniformLocationByBinding);
|
||||
samplerTextureTargetByBinding = std::move(other.samplerTextureTargetByBinding);
|
||||
samplerNumericDomainByBinding = std::move(other.samplerNumericDomainByBinding);
|
||||
storageImageFormatByBinding = std::move(other.storageImageFormatByBinding);
|
||||
storageImageUsesBindingFormatByBinding =
|
||||
std::move(other.storageImageUsesBindingFormatByBinding);
|
||||
storageBlockNameByBinding = std::move(other.storageBlockNameByBinding);
|
||||
storageBlockIndexByBinding = std::move(other.storageBlockIndexByBinding);
|
||||
hasStorageImages = other.hasStorageImages;
|
||||
declinedDescriptors = other.declinedDescriptors;
|
||||
globalUboBinding = other.globalUboBinding;
|
||||
activeVertexInputLocationMask = other.activeVertexInputLocationMask;
|
||||
vertexInputTypes = other.vertexInputTypes;
|
||||
activeFragmentOutputLocationMask = other.activeFragmentOutputLocationMask;
|
||||
fragmentOutputTypes = other.fragmentOutputTypes;
|
||||
rasterizationProducerStage = other.rasterizationProducerStage;
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
|
||||
needsPassthroughTessControl = other.needsPassthroughTessControl;
|
||||
passthroughTessControlEmulatable = other.passthroughTessControlEmulatable;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.device = VK_NULL_HANDLE;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
other.pipelineLayout = VK_NULL_HANDLE;
|
||||
other.hasStorageImages = false;
|
||||
other.declinedDescriptors = false;
|
||||
other.globalUboBinding = -1;
|
||||
other.activeVertexInputLocationMask = 0;
|
||||
other.activeFragmentOutputLocationMask = 0;
|
||||
other.rasterizationProducerStage = ShaderStage::Unknown;
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.readsBaseVertexBuiltin = false;
|
||||
other.needsPassthroughTessControl = false;
|
||||
other.passthroughTessControlEmulatable = false;
|
||||
other.lastUsedFrame = 0;
|
||||
return *this;
|
||||
}
|
||||
|
||||
~BackendProgramObject() {
|
||||
DestroyModules();
|
||||
stages.clear();
|
||||
~VkProgramObject() {
|
||||
Destroy();
|
||||
}
|
||||
|
||||
private:
|
||||
void DestroyModules() {
|
||||
for (auto module : modules) {
|
||||
if (module != VK_NULL_HANDLE && device != VK_NULL_HANDLE) {
|
||||
vkDestroyShaderModule(device, module, nullptr);
|
||||
void Destroy() {
|
||||
if (s_device != VK_NULL_HANDLE) {
|
||||
if (pipelineLayout != VK_NULL_HANDLE) {
|
||||
vkDestroyPipelineLayout(s_device, pipelineLayout, nullptr);
|
||||
pipelineLayout = VK_NULL_HANDLE;
|
||||
}
|
||||
if (descriptorSetLayout != VK_NULL_HANDLE) {
|
||||
vkDestroyDescriptorSetLayout(s_device, descriptorSetLayout, nullptr);
|
||||
descriptorSetLayout = VK_NULL_HANDLE;
|
||||
}
|
||||
for (auto module : modules) {
|
||||
if (module != VK_NULL_HANDLE) {
|
||||
vkDestroyShaderModule(s_device, module, nullptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
modules.clear();
|
||||
stages.clear();
|
||||
stageSpirvDigests.clear(); // the modules they describe are gone
|
||||
}
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config)
|
||||
: m_device(device), m_config(config) {}
|
||||
// Notified when the OnFrameBoundary sweep destroys an aged-out cache entry,
|
||||
// carrying the entry's content hash and the VkDescriptorSetLayout it owned.
|
||||
// Dependent caches (compute pipelines, PipelineFactory entries, UniformManager's
|
||||
// per-layout descriptor sets) must purge in the same step: after vkDestroy the
|
||||
// layout handle value may be recycled for an unrelated layout, and the program
|
||||
// hash may be re-inserted by a later rebuild of the same content.
|
||||
class IEvictionObserver {
|
||||
public:
|
||||
virtual ~IEvictionObserver() = default;
|
||||
virtual void OnProgramEvicted(HashType programHash, VkDescriptorSetLayout descriptorSetLayout) = 0;
|
||||
};
|
||||
|
||||
explicit ProgramFactory(VkDevice device, const VulkanRendererConfig& config, Uint32 maxBindings = 16,
|
||||
Bool shaderDrawParametersEnabled = false,
|
||||
Bool unformattedFloatStorageImagesEnabled = false)
|
||||
: m_device(device), m_maxBindings(maxBindings), m_config(config),
|
||||
m_shaderDrawParametersEnabled(shaderDrawParametersEnabled),
|
||||
m_unformattedFloatStorageImagesEnabled(unformattedFloatStorageImagesEnabled) {
|
||||
VkProgramObject::s_device = device;
|
||||
}
|
||||
// Destroys the pass-through tessellation control modules. Runs while the device is
|
||||
// still alive for the same reason ~VkProgramObject's does: this factory outlives
|
||||
// nothing that owns the device.
|
||||
~ProgramFactory();
|
||||
ProgramFactory(const ProgramFactory&) = delete;
|
||||
|
||||
HashType ComputeHash(const MG_State::GLState::ProgramObject& program, CompileOptionFlags flags) const;
|
||||
Vector<VkPipelineShaderStageCreateInfo>& GetOrCreatePipelineShaderStages(
|
||||
const VkProgramObject& GetOrCreateProgram(
|
||||
const MG_State::GLState::ProgramObject& program, CompileOptionFlags flags);
|
||||
|
||||
// The default framebuffer's current image height, baked as a literal into every
|
||||
// FragCoordYFlip variant (there is no push-constant or specialization channel here, and
|
||||
// adding one for a value that changes only on swapchain recreation would cost the draw
|
||||
// path more than a recompile costs a resize). It is therefore part of those variants'
|
||||
// identity: ComputeHash mixes it in when the bit is set, so a height change re-keys them
|
||||
// and leaves every other program's hash untouched. Setting a NEW height also bumps the
|
||||
// cache-structure epoch, because a caller holding a memoised VkProgramObject* would
|
||||
// otherwise keep using a module compiled against the old height.
|
||||
void SetDefaultFramebufferHeight(Uint32 height);
|
||||
Uint32 GetDefaultFramebufferHeight() const { return m_defaultFramebufferHeight; }
|
||||
|
||||
// Bumped whenever m_cache's STRUCTURE changes (any insert or erase): the cache is
|
||||
// an open-addressing map holding entries by value, so both moves existing entries.
|
||||
// A caller that memoised a VkProgramObject* may keep dereferencing it only while
|
||||
// this is unchanged; on a bump it must re-run GetOrCreateProgram.
|
||||
Uint64 GetCacheStructureEpoch() const { return m_cacheStructureEpoch; }
|
||||
// A memoised entry pointer bypasses GetOrCreateProgram, whose per-lookup stamp is
|
||||
// what keeps an in-use entry out of OnFrameBoundary's idle sweep - so such a
|
||||
// caller must re-stamp the entry itself, at least once per frame boundary.
|
||||
void StampProgramUse(const VkProgramObject& entry) const { entry.lastUsedFrame = m_frameCounter; }
|
||||
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetEvictionObserver(IEvictionObserver* observer) { m_evictionObserver = observer; }
|
||||
// Frame boundary hook: ages the program cache and evicts long-unused entries
|
||||
// (their command buffers retired many frames ago), mirroring
|
||||
// VkRenderPassManager::OnPresent's sweep.
|
||||
void OnFrameBoundary();
|
||||
|
||||
static VkShaderStageFlagBits ToVkStage(ShaderStage stage);
|
||||
static VkFormat ConvertSpirvImageFormatToVkFormat(SpvImageFormat format);
|
||||
static SamplerNumericDomain UniformTypeToSamplerNumericDomain(GLenum glType);
|
||||
// True when any entry point declares the DepthReplacing execution mode, i.e. the
|
||||
// shader assigns gl_FragDepth. Exposed so the blended depth-write quirk's exemption
|
||||
// can be pinned by tests. A false negative loses the exemption, so such a shader is
|
||||
// stripped conservatively and forfeits its depth write.
|
||||
static Bool ReflectedFragmentReplacesDepth(const SpvReflectShaderModule& reflectModule);
|
||||
// True when an entry point reads the InstanceIndex builtin. Only gates a diagnostic:
|
||||
// without shaderDrawParameters such a shader cannot have gl_InstanceID rebased.
|
||||
static Bool ReflectedReadsInstanceIndexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
// True when an entry point declares the BaseVertex builtin, i.e. when a non-indexed
|
||||
// draw with this program has to take the ZeroBaseVertex variant.
|
||||
static Bool ReflectedReadsBaseVertexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
// Shared by the two above: does any entry point list an input variable decorated with
|
||||
// this builtin?
|
||||
static Bool ReflectedDeclaresInputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
|
||||
// True when an entry point writes the ViewportIndex builtin (gl_ViewportIndex), i.e. when
|
||||
// the program can route primitives to a viewport other than 0 and its pipeline therefore
|
||||
// has to declare more than one. Asks about OUTPUT variables because that is the direction
|
||||
// a pre-rasterization stage declares it in.
|
||||
static Bool ReflectedWritesViewportIndexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
static Bool ReflectedDeclaresOutputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
|
||||
|
||||
// The pass-through tessellation control stage GL 4.6 core 11.2.2 describes for a
|
||||
// program that has an evaluation stage and no control stage, for an input patch of
|
||||
// `patchVertices` control points. Returned BY VALUE (a stage description is a POD, and
|
||||
// the cache below is a rehashing map, so a pointer into it would not survive the next
|
||||
// distinct patch size). `.module == VK_NULL_HANDLE` means the stage could not be built:
|
||||
// the caller then has no control stage to inject, and CreatePipeline refuses the
|
||||
// pipeline rather than handing the driver a half-tessellated one.
|
||||
//
|
||||
// Keyed on the patch size because GL takes the output patch size from PATCH_VERTICES,
|
||||
// which is draw state, not link state - the CTS case that motivated this links at the
|
||||
// default 3 and draws at 4. The pipeline cache already re-keys on patchControlPoints,
|
||||
// so the module a pipeline was built with is part of that pipeline's identity.
|
||||
// Compiling is bounded by the number of distinct patch sizes a program draws with
|
||||
// (MAX_PATCH_VERTICES = 32 in the worst case, one or two in practice) and only ever
|
||||
// happens for the rare program that has no control stage at all.
|
||||
VkPipelineShaderStageCreateInfo GetOrCreatePassthroughTessControlStage(Uint32 patchVertices);
|
||||
|
||||
// Source of the module above. Exposed for tests: the generated GLSL is the whole
|
||||
// contract with the evaluation stage, so it is worth pinning independently of a device.
|
||||
static String BuildPassthroughTessControlSource(Uint32 patchVertices);
|
||||
|
||||
private:
|
||||
struct ProgramLookupCache {
|
||||
const MG_State::GLState::ProgramObject* program = nullptr;
|
||||
Uint32 backendStateVersion = 0;
|
||||
CompileOptionFlags flags{};
|
||||
HashType hash = 0;
|
||||
};
|
||||
|
||||
static TextureTarget UniformTypeToTextureTarget(GLenum glType);
|
||||
void ReflectVertexInputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
void ReflectViewportIndexUsage(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
void ReflectFragmentOutputs(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
void ReflectLayout(const MG_State::GLState::ProgramObject& program, const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
// Fills needsPassthroughTessControl / passthroughTessControlEmulatable off the linked
|
||||
// modules. Const and reflection-only: it decides nothing about the pipeline, it only
|
||||
// records what the evaluation stage's input interface is made of.
|
||||
void ReflectPassthroughTessControlNeed(const Vector<SharedPtr<MG_State::GLState::ShaderObject>>& shaders,
|
||||
const Vector<Vector<Uint>>& spirv,
|
||||
VkProgramObject& entry) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
UnorderedMap<HashType, BackendProgramObject> m_cache;
|
||||
Uint32 m_maxBindings = 0;
|
||||
UnorderedMap<HashType, VkProgramObject> m_cache;
|
||||
const VulkanRendererConfig& m_config;
|
||||
// True when the device enabled shaderDrawParameters; gates the InstanceIndex rebase pass
|
||||
// (which needs the DrawParameters capability / gl_BaseInstance builtin).
|
||||
Bool m_shaderDrawParametersEnabled = false;
|
||||
// True only when the logical device enabled both
|
||||
// shaderStorageImageReadWithoutFormat and shaderStorageImageWriteWithoutFormat.
|
||||
Bool m_unformattedFloatStorageImagesEnabled = false;
|
||||
// See SetDefaultFramebufferHeight. 0 means "not known yet"; the FragCoordYFlip bit is
|
||||
// never set before the swapchain exists, so no variant can be compiled against it.
|
||||
Uint32 m_defaultFramebufferHeight = 0;
|
||||
mutable ProgramLookupCache m_lastLookup;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
// See GetCacheStructureEpoch(). Starts at 1 so a zero-initialized memo can never match.
|
||||
Uint64 m_cacheStructureEpoch = 1;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
// Pass-through tessellation control stages by input patch size. Never evicted: at most
|
||||
// MAX_PATCH_VERTICES entries exist for the lifetime of the device, and every pipeline
|
||||
// ever built from one keeps referencing its module. A failed build is cached as
|
||||
// VK_NULL_HANDLE so a broken generator costs one compile, not one per draw.
|
||||
UnorderedMap<Uint32, VkPipelineShaderStageCreateInfo> m_passthroughTessControlStages;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -8,6 +8,9 @@
|
||||
|
||||
#include "SwapchainObject.h"
|
||||
|
||||
#include "MG_Impl/GLImpl/Framebuffer/GL_Framebuffer.h"
|
||||
#include "MG_State/GLState/TextureState/TextureObject2D.h"
|
||||
|
||||
#if defined(__has_include)
|
||||
#if __has_include(<vulkan/vk_enum_string_helper.h>)
|
||||
#include <vulkan/vk_enum_string_helper.h>
|
||||
@@ -28,8 +31,19 @@ static const char* string_VkColorSpaceKHR(VkColorSpaceKHR) {
|
||||
return "VkColorSpaceKHR(unknown)";
|
||||
}
|
||||
|
||||
static const char* string_VkPresentModeKHR(VkPresentModeKHR) {
|
||||
return "VkPresentModeKHR(unknown)";
|
||||
static const char* string_VkPresentModeKHR(VkPresentModeKHR presentMode) {
|
||||
switch (presentMode) {
|
||||
case VK_PRESENT_MODE_IMMEDIATE_KHR:
|
||||
return "VK_PRESENT_MODE_IMMEDIATE_KHR";
|
||||
case VK_PRESENT_MODE_MAILBOX_KHR:
|
||||
return "VK_PRESENT_MODE_MAILBOX_KHR";
|
||||
case VK_PRESENT_MODE_FIFO_KHR:
|
||||
return "VK_PRESENT_MODE_FIFO_KHR";
|
||||
case VK_PRESENT_MODE_FIFO_RELAXED_KHR:
|
||||
return "VK_PRESENT_MODE_FIFO_RELAXED_KHR";
|
||||
default:
|
||||
return "VkPresentModeKHR(unknown)";
|
||||
}
|
||||
}
|
||||
|
||||
static const char* string_VkSurfaceTransformFlagBitsKHR(VkSurfaceTransformFlagBitsKHR) {
|
||||
@@ -38,6 +52,40 @@ static const char* string_VkSurfaceTransformFlagBitsKHR(VkSurfaceTransformFlagBi
|
||||
#endif
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
Bool HasStencilComponent(VkFormat format) {
|
||||
return format == VK_FORMAT_D24_UNORM_S8_UINT || format == VK_FORMAT_D32_SFLOAT_S8_UINT;
|
||||
}
|
||||
|
||||
VkFormat FindSupportedDepthStencilFormat(VkPhysicalDevice physicalDevice) {
|
||||
const VkFormat candidates[] = {VK_FORMAT_D24_UNORM_S8_UINT, VK_FORMAT_D32_SFLOAT_S8_UINT,
|
||||
VK_FORMAT_D32_SFLOAT};
|
||||
for (VkFormat format : candidates) {
|
||||
VkFormatProperties props{};
|
||||
vkGetPhysicalDeviceFormatProperties(physicalDevice, format, &props);
|
||||
if ((props.optimalTilingFeatures & VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT) != 0) {
|
||||
return format;
|
||||
}
|
||||
}
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
|
||||
Uint32 FindMemoryType(VkPhysicalDevice physicalDevice, Uint32 typeFilter, VkMemoryPropertyFlags properties) {
|
||||
VkPhysicalDeviceMemoryProperties memProperties{};
|
||||
vkGetPhysicalDeviceMemoryProperties(physicalDevice, &memProperties);
|
||||
|
||||
for (Uint32 i = 0; i < memProperties.memoryTypeCount; i++) {
|
||||
if ((typeFilter & (1 << i)) &&
|
||||
(memProperties.memoryTypes[i].propertyFlags & properties) == properties) {
|
||||
return i;
|
||||
}
|
||||
}
|
||||
|
||||
MOBILEGL_ASSERT(false, "Failed to find suitable memory type.");
|
||||
return 0;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
SwapchainObject::SwapchainCapabilities SwapchainObject::GetSwapchainCapabilities(VkPhysicalDevice physicalDevice,
|
||||
VkSurfaceKHR surface) {
|
||||
SwapchainCapabilities swapchainCapabilities{};
|
||||
@@ -68,7 +116,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSurfaceFormatKHR SwapchainObject::ChooseSwapchainSurfaceFormat(
|
||||
const Vector<VkSurfaceFormatKHR>& availableFormats) {
|
||||
for (const auto& availableFormat : availableFormats) {
|
||||
if (availableFormat.format == VK_FORMAT_B8G8R8A8_SRGB &&
|
||||
if ((availableFormat.format == VK_FORMAT_B8G8R8A8_UNORM ||
|
||||
availableFormat.format == VK_FORMAT_R8G8B8A8_UNORM) &&
|
||||
availableFormat.colorSpace == VK_COLOR_SPACE_SRGB_NONLINEAR_KHR) {
|
||||
return availableFormat;
|
||||
}
|
||||
}
|
||||
for (const auto& availableFormat : availableFormats) {
|
||||
if ((availableFormat.format == VK_FORMAT_B8G8R8A8_SRGB ||
|
||||
availableFormat.format == VK_FORMAT_R8G8B8A8_SRGB) &&
|
||||
availableFormat.colorSpace == VK_COLOR_SPACE_SRGB_NONLINEAR_KHR) {
|
||||
return availableFormat;
|
||||
}
|
||||
@@ -93,14 +149,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
void SwapchainObject::Create(VkDevice device, VkPhysicalDevice physicalDevice, VkSurfaceKHR surface,
|
||||
Uint32 graphicsQueueFamily, Uint32 presentQueueFamily, Uint32 minImageCountHint) {
|
||||
Uint32 graphicsQueueFamily, Uint32 presentQueueFamily, Uint32 minImageCountHint,
|
||||
VkExtent2D desiredExtent) {
|
||||
const auto swapchainCapabilities = GetSwapchainCapabilities(physicalDevice, surface);
|
||||
MOBILEGL_ASSERT(swapchainCapabilities.IsComplete(),
|
||||
"SwapchainObject::Create failed: incomplete swapchain capabilities");
|
||||
|
||||
MGLOG_I("Got %d surface formats:", swapchainCapabilities.surfaceFormats.size());
|
||||
for (const auto& sf : swapchainCapabilities.surfaceFormats) {
|
||||
MGLOG_I(" [%s, %s]", string_VkFormat(sf.format), string_VkColorSpaceKHR(sf.colorSpace));
|
||||
MGLOG_D(" [%s, %s]", string_VkFormat(sf.format), string_VkColorSpaceKHR(sf.colorSpace));
|
||||
}
|
||||
|
||||
const auto pickedSurfaceFormat = ChooseSwapchainSurfaceFormat(swapchainCapabilities.surfaceFormats);
|
||||
@@ -109,14 +166,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
MGLOG_I("Got %d present modes:", swapchainCapabilities.presentModes.size());
|
||||
for (const auto& pm : swapchainCapabilities.presentModes) {
|
||||
MGLOG_I(" %s", string_VkPresentModeKHR(pm));
|
||||
MGLOG_D(" %s", string_VkPresentModeKHR(pm));
|
||||
}
|
||||
|
||||
const auto presentMode = ChooseSwapchainPresentMode(swapchainCapabilities.presentModes);
|
||||
MGLOG_I("Picked present mode: %s", string_VkPresentModeKHR(presentMode));
|
||||
|
||||
const auto& swapchainCaps = swapchainCapabilities.capabilities;
|
||||
const auto targetImageCount = std::max<Uint32>(minImageCountHint, swapchainCaps.minImageCount);
|
||||
Uint32 targetImageCount = std::max<Uint32>(minImageCountHint, swapchainCaps.minImageCount);
|
||||
if (swapchainCaps.maxImageCount != 0) {
|
||||
targetImageCount = std::min(targetImageCount, swapchainCaps.maxImageCount);
|
||||
}
|
||||
MGLOG_I("Set minImageCount = %u", targetImageCount);
|
||||
MGLOG_I("Swapchain currentTransform = %s",
|
||||
string_VkSurfaceTransformFlagBitsKHR(swapchainCaps.currentTransform));
|
||||
@@ -127,6 +187,20 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
createInfo.imageFormat = pickedSurfaceFormat.format;
|
||||
createInfo.imageColorSpace = pickedSurfaceFormat.colorSpace;
|
||||
createInfo.imageExtent = swapchainCaps.currentExtent;
|
||||
if (createInfo.imageExtent.width == UINT32_MAX || createInfo.imageExtent.height == UINT32_MAX) {
|
||||
createInfo.imageExtent.width = std::clamp(desiredExtent.width,
|
||||
swapchainCaps.minImageExtent.width,
|
||||
swapchainCaps.maxImageExtent.width);
|
||||
createInfo.imageExtent.height = std::clamp(desiredExtent.height,
|
||||
swapchainCaps.minImageExtent.height,
|
||||
swapchainCaps.maxImageExtent.height);
|
||||
}
|
||||
const VkExtent2D defaultFramebufferExtent = createInfo.imageExtent;
|
||||
if (swapchainCaps.currentTransform == VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR ||
|
||||
swapchainCaps.currentTransform == VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR) {
|
||||
std::swap(createInfo.imageExtent.width, createInfo.imageExtent.height);
|
||||
}
|
||||
|
||||
createInfo.imageArrayLayers = 1;
|
||||
const VkImageUsageFlags requiredImageUsage =
|
||||
VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT;
|
||||
@@ -173,6 +247,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_surfaceFormat = {createInfo.imageFormat, createInfo.imageColorSpace};
|
||||
m_extent = createInfo.imageExtent;
|
||||
// The surface-space extent this swapchain was built from, i.e. before the
|
||||
// quarter-turn swap above. Out-of-date checks must compare in THIS space: comparing a
|
||||
// freshly queried currentExtent against the swapped m_extent flips axes every rotation
|
||||
// and makes the comparison alternate forever.
|
||||
m_surfaceExtent = defaultFramebufferExtent;
|
||||
m_preTransform = createInfo.preTransform;
|
||||
|
||||
VK_VERIFY(vkCreateSwapchainKHR(device, &createInfo, nullptr, &m_swapchain));
|
||||
@@ -183,14 +262,168 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_images.resize(imageCount, VK_NULL_HANDLE);
|
||||
VK_VERIFY(vkGetSwapchainImagesKHR(device, m_swapchain, &imageCount, m_images.data()));
|
||||
m_imageLayouts.assign(imageCount, VK_IMAGE_LAYOUT_UNDEFINED);
|
||||
// Fresh swapchain images hold garbage until a render pass stores into them.
|
||||
m_imageContentDefined.assign(imageCount, false);
|
||||
m_depthStencilContentDefined.assign(imageCount, false);
|
||||
|
||||
CreateImageViews(device);
|
||||
CreateDepthStencilResources(device, physicalDevice);
|
||||
|
||||
MGLOG_I("Swapchain created, extent = %dx%d, swapchain imageCount = %d", m_extent.width, m_extent.height,
|
||||
imageCount);
|
||||
|
||||
// Properly initialize Default FBO here
|
||||
auto& defaultFBOInfo = MG_Impl::GLImpl::FramebufferImpl::pDefaultFramebufferInfo;
|
||||
const Int extentWidth = static_cast<Int>(defaultFramebufferExtent.width);
|
||||
const Int extentHeight = static_cast<Int>(defaultFramebufferExtent.height);
|
||||
const SizeT defaultAttachmentByteSize =
|
||||
static_cast<SizeT>(defaultFramebufferExtent.width) *
|
||||
static_cast<SizeT>(defaultFramebufferExtent.height) * 4;
|
||||
|
||||
auto* colorTex = static_cast<MG_State::GLState::TextureObject2D*>(defaultFBOInfo->colorAttachment.get());
|
||||
colorTex->AllocateStorage(
|
||||
TextureUploadTarget::Texture2D, 0, {
|
||||
{extentWidth, extentHeight, 1},
|
||||
defaultAttachmentByteSize}); // TODO: 4 is format size
|
||||
TextureInternalFormat depthFormat = TextureInternalFormat::Depth24Stencil8;
|
||||
switch (m_depthStencilFormat) {
|
||||
case VK_FORMAT_D24_UNORM_S8_UINT:
|
||||
depthFormat = TextureInternalFormat::Depth24Stencil8;
|
||||
break;
|
||||
case VK_FORMAT_D32_SFLOAT_S8_UINT:
|
||||
depthFormat = TextureInternalFormat::Depth32FStencil8;
|
||||
break;
|
||||
case VK_FORMAT_D32_SFLOAT:
|
||||
depthFormat = TextureInternalFormat::DepthComponent32F;
|
||||
break;
|
||||
default:
|
||||
depthFormat = TextureInternalFormat::Depth24Stencil8;
|
||||
break;
|
||||
}
|
||||
auto* depthTex = static_cast<MG_State::GLState::TextureObject2D*>(defaultFBOInfo->depthAttachment.get());
|
||||
depthTex->SetInternalFormat(depthFormat);
|
||||
depthTex->AllocateStorage(TextureUploadTarget::Texture2D, 0, {
|
||||
{extentWidth, extentHeight, 1},
|
||||
defaultAttachmentByteSize}); // TODO: 4 is format size
|
||||
|
||||
// The default FBO's stencil attachment must track the swapchain extent:
|
||||
// FramebufferObject::CheckCompleteness requires every valid attachment
|
||||
// to share the same dimensions, and Init.cpp leaves a 512x512 placeholder.
|
||||
// Without this the retrace-layer glReadPixels snapshot fails with
|
||||
// GL_INVALID_FRAMEBUFFER_OPERATION on DirectVulkan.
|
||||
TextureInternalFormat stencilFormat = TextureInternalFormat::Depth24Stencil8;
|
||||
switch (m_depthStencilFormat) {
|
||||
case VK_FORMAT_D32_SFLOAT_S8_UINT:
|
||||
stencilFormat = TextureInternalFormat::Depth32FStencil8;
|
||||
break;
|
||||
case VK_FORMAT_D24_UNORM_S8_UINT:
|
||||
stencilFormat = TextureInternalFormat::Depth24Stencil8;
|
||||
break;
|
||||
default:
|
||||
// No stencil plane; mirror the depth format for consistency.
|
||||
stencilFormat = depthFormat;
|
||||
break;
|
||||
}
|
||||
auto* stencilTex = static_cast<MG_State::GLState::TextureObject2D*>(defaultFBOInfo->stencilAttachment.get());
|
||||
stencilTex->SetInternalFormat(stencilFormat);
|
||||
stencilTex->AllocateStorage(TextureUploadTarget::Texture2D, 0, {
|
||||
{extentWidth, extentHeight, 1},
|
||||
defaultAttachmentByteSize}); // TODO: 4 is format size
|
||||
|
||||
}
|
||||
|
||||
void SwapchainObject::CreateDepthStencilResources(VkDevice device, VkPhysicalDevice physicalDevice) {
|
||||
DestroyDepthStencilResources(device);
|
||||
|
||||
const auto imageCount = static_cast<Uint32>(m_images.size());
|
||||
if (imageCount == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
m_depthStencilFormat = FindSupportedDepthStencilFormat(physicalDevice);
|
||||
MOBILEGL_ASSERT(m_depthStencilFormat != VK_FORMAT_UNDEFINED, "No supported depth/stencil format found.");
|
||||
|
||||
m_depthStencilImages.assign(imageCount, VK_NULL_HANDLE);
|
||||
m_depthStencilImageMemories.assign(imageCount, VK_NULL_HANDLE);
|
||||
m_depthStencilImageViews.assign(imageCount, VK_NULL_HANDLE);
|
||||
m_depthStencilImageLayouts.assign(imageCount, VK_IMAGE_LAYOUT_UNDEFINED);
|
||||
|
||||
for (Uint32 i = 0; i < imageCount; ++i) {
|
||||
VkImageCreateInfo imageInfo{};
|
||||
imageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
|
||||
imageInfo.imageType = VK_IMAGE_TYPE_2D;
|
||||
imageInfo.extent.width = m_extent.width;
|
||||
imageInfo.extent.height = m_extent.height;
|
||||
imageInfo.extent.depth = 1;
|
||||
imageInfo.mipLevels = 1;
|
||||
imageInfo.arrayLayers = 1;
|
||||
imageInfo.format = m_depthStencilFormat;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
|
||||
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateImage(device, &imageInfo, nullptr, &m_depthStencilImages[i]), "vkCreateImage(depth)");
|
||||
|
||||
VkMemoryRequirements memRequirements{};
|
||||
vkGetImageMemoryRequirements(device, m_depthStencilImages[i], &memRequirements);
|
||||
|
||||
VkMemoryAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
allocInfo.allocationSize = memRequirements.size;
|
||||
allocInfo.memoryTypeIndex =
|
||||
FindMemoryType(physicalDevice, memRequirements.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(device, &allocInfo, nullptr, &m_depthStencilImageMemories[i]),
|
||||
"vkAllocateMemory(depth)");
|
||||
VK_VERIFY(vkBindImageMemory(device, m_depthStencilImages[i], m_depthStencilImageMemories[i], 0),
|
||||
"vkBindImageMemory(depth)");
|
||||
|
||||
VkImageViewCreateInfo viewInfo{};
|
||||
viewInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
|
||||
viewInfo.image = m_depthStencilImages[i];
|
||||
viewInfo.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
viewInfo.format = m_depthStencilFormat;
|
||||
viewInfo.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
||||
if (HasStencilComponent(m_depthStencilFormat)) {
|
||||
viewInfo.subresourceRange.aspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
|
||||
}
|
||||
viewInfo.subresourceRange.baseMipLevel = 0;
|
||||
viewInfo.subresourceRange.levelCount = 1;
|
||||
viewInfo.subresourceRange.baseArrayLayer = 0;
|
||||
viewInfo.subresourceRange.layerCount = 1;
|
||||
VK_VERIFY(vkCreateImageView(device, &viewInfo, nullptr, &m_depthStencilImageViews[i]),
|
||||
"vkCreateImageView(depth)");
|
||||
}
|
||||
}
|
||||
|
||||
void SwapchainObject::DestroyDepthStencilResources(VkDevice device) {
|
||||
for (auto view : m_depthStencilImageViews) {
|
||||
if (view != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(device, view, nullptr);
|
||||
}
|
||||
}
|
||||
m_depthStencilImageViews.clear();
|
||||
|
||||
for (auto image : m_depthStencilImages) {
|
||||
if (image != VK_NULL_HANDLE) {
|
||||
vkDestroyImage(device, image, nullptr);
|
||||
}
|
||||
}
|
||||
m_depthStencilImages.clear();
|
||||
|
||||
for (auto memory : m_depthStencilImageMemories) {
|
||||
if (memory != VK_NULL_HANDLE) {
|
||||
vkFreeMemory(device, memory, nullptr);
|
||||
}
|
||||
}
|
||||
m_depthStencilImageMemories.clear();
|
||||
m_depthStencilImageLayouts.clear();
|
||||
m_depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
|
||||
void SwapchainObject::Shutdown(VkDevice device) {
|
||||
DestroyDepthStencilResources(device);
|
||||
|
||||
for (auto imageView : m_imageViews) {
|
||||
vkDestroyImageView(device, imageView, nullptr);
|
||||
}
|
||||
@@ -203,9 +436,39 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_images.clear();
|
||||
m_imageLayouts.clear();
|
||||
m_imageContentDefined.clear();
|
||||
m_depthStencilContentDefined.clear();
|
||||
m_preTransform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
|
||||
}
|
||||
|
||||
Bool SwapchainObject::IsImageContentDefined(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_imageContentDefined.size(), "Swapchain image content index out of range");
|
||||
return m_imageContentDefined[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetImageContentDefined(Uint32 index, Bool defined) {
|
||||
MOBILEGL_ASSERT(index < m_imageContentDefined.size(), "Swapchain image content index out of range");
|
||||
m_imageContentDefined[index] = defined;
|
||||
}
|
||||
|
||||
Bool SwapchainObject::IsDepthStencilContentDefined(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilContentDefined.size(),
|
||||
"Swapchain depth/stencil content index out of range");
|
||||
return m_depthStencilContentDefined[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetDepthStencilContentDefined(Uint32 index, Bool defined) {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilContentDefined.size(),
|
||||
"Swapchain depth/stencil content index out of range");
|
||||
m_depthStencilContentDefined[index] = defined;
|
||||
}
|
||||
|
||||
void SwapchainObject::SetAllDepthStencilContentUndefined() {
|
||||
for (SizeT i = 0; i < m_depthStencilContentDefined.size(); ++i) {
|
||||
m_depthStencilContentDefined[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
VkImage SwapchainObject::GetImage(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_images.size(), "Swapchain image index out of range");
|
||||
return m_images[index];
|
||||
@@ -221,6 +484,29 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_imageLayouts[index] = layout;
|
||||
}
|
||||
|
||||
VkImage SwapchainObject::GetDepthStencilImage(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilImages.size(), "Swapchain depth/stencil image index out of range");
|
||||
return m_depthStencilImages[index];
|
||||
}
|
||||
|
||||
VkImageView SwapchainObject::GetDepthStencilImageView(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilImageViews.size(),
|
||||
"Swapchain depth/stencil image view index out of range");
|
||||
return m_depthStencilImageViews[index];
|
||||
}
|
||||
|
||||
VkImageLayout SwapchainObject::GetDepthStencilImageLayout(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilImageLayouts.size(),
|
||||
"Swapchain depth/stencil image layout index out of range");
|
||||
return m_depthStencilImageLayouts[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetDepthStencilImageLayout(Uint32 index, VkImageLayout layout) {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilImageLayouts.size(),
|
||||
"Swapchain depth/stencil image layout index out of range");
|
||||
m_depthStencilImageLayouts[index] = layout;
|
||||
}
|
||||
|
||||
void SwapchainObject::CreateImageViews(VkDevice device) {
|
||||
m_imageViews.resize(m_images.size(), VK_NULL_HANDLE);
|
||||
for (SizeT i = 0; i < m_imageViews.size(); i++) {
|
||||
|
||||
@@ -29,22 +29,48 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static VkPresentModeKHR ChooseSwapchainPresentMode(const Vector<VkPresentModeKHR>& availablePresentModes);
|
||||
|
||||
void Create(VkDevice device, VkPhysicalDevice physicalDevice, VkSurfaceKHR surface, Uint32 graphicsQueueFamily,
|
||||
Uint32 presentQueueFamily, Uint32 minImageCountHint);
|
||||
Uint32 presentQueueFamily, Uint32 minImageCountHint, VkExtent2D desiredExtent);
|
||||
void Shutdown(VkDevice device);
|
||||
|
||||
VkSwapchainKHR GetHandle() const { return m_swapchain; }
|
||||
const VkSurfaceFormatKHR& GetSurfaceFormat() const { return m_surfaceFormat; }
|
||||
VkExtent2D GetExtent() const { return m_extent; }
|
||||
// Surface-space extent (before the pre-rotation quarter-turn swap) this swapchain was
|
||||
// created from - the value to compare a freshly queried currentExtent against.
|
||||
VkExtent2D GetSurfaceExtent() const { return m_surfaceExtent; }
|
||||
VkSurfaceTransformFlagBitsKHR GetPreTransform() const { return m_preTransform; }
|
||||
const Vector<VkImage>& GetImages() const { return m_images; }
|
||||
const Vector<VkImageView>& GetImageViews() const { return m_imageViews; }
|
||||
VkFormat GetDepthStencilFormat() const { return m_depthStencilFormat; }
|
||||
const Vector<VkImageView>& GetDepthStencilImageViews() const { return m_depthStencilImageViews; }
|
||||
VkImage GetDepthStencilImage(Uint32 index) const;
|
||||
VkImageView GetDepthStencilImageView(Uint32 index) const;
|
||||
VkImageLayout GetDepthStencilImageLayout(Uint32 index) const;
|
||||
void SetDepthStencilImageLayout(Uint32 index, VkImageLayout layout);
|
||||
VkImage GetImage(Uint32 index) const;
|
||||
VkImageLayout GetImageLayout(Uint32 index) const;
|
||||
void SetImageLayout(Uint32 index, VkImageLayout layout);
|
||||
SizeT GetImageCount() const { return m_images.size(); }
|
||||
|
||||
// EGL content-validity tracking for the default framebuffer. A color
|
||||
// buffer's content is undefined once its image has been presented
|
||||
// (EGL_BUFFER_DESTROYED swap behaviour, the implementation default),
|
||||
// and every ancillary (depth/stencil) buffer's content is undefined
|
||||
// after ANY swap regardless of swap behaviour (EGL 1.5 §3.10.1). The
|
||||
// render-pass manager turns an undefined attachment's tile load into
|
||||
// LOAD_OP_DONT_CARE. Flags start false (a fresh swapchain image holds
|
||||
// garbage) and a render pass storing into an attachment sets it back
|
||||
// to defined.
|
||||
Bool IsImageContentDefined(Uint32 index) const;
|
||||
void SetImageContentDefined(Uint32 index, Bool defined);
|
||||
Bool IsDepthStencilContentDefined(Uint32 index) const;
|
||||
void SetDepthStencilContentDefined(Uint32 index, Bool defined);
|
||||
void SetAllDepthStencilContentUndefined();
|
||||
|
||||
private:
|
||||
void CreateImageViews(VkDevice device);
|
||||
void CreateDepthStencilResources(VkDevice device, VkPhysicalDevice physicalDevice);
|
||||
void DestroyDepthStencilResources(VkDevice device);
|
||||
static constexpr VkPresentModeKHR s_desiredPresentModes[] {
|
||||
VK_PRESENT_MODE_MAILBOX_KHR,
|
||||
VK_PRESENT_MODE_IMMEDIATE_KHR,
|
||||
@@ -55,9 +81,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkSwapchainKHR m_swapchain = VK_NULL_HANDLE;
|
||||
VkSurfaceFormatKHR m_surfaceFormat{};
|
||||
VkExtent2D m_extent{};
|
||||
VkExtent2D m_surfaceExtent{};
|
||||
VkSurfaceTransformFlagBitsKHR m_preTransform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
|
||||
Vector<VkImage> m_images;
|
||||
Vector<VkImageView> m_imageViews;
|
||||
Vector<VkImageLayout> m_imageLayouts;
|
||||
|
||||
VkFormat m_depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
Vector<VkImage> m_depthStencilImages;
|
||||
Vector<VkDeviceMemory> m_depthStencilImageMemories;
|
||||
Vector<VkImageView> m_depthStencilImageViews;
|
||||
Vector<VkImageLayout> m_depthStencilImageLayouts;
|
||||
Vector<Bool> m_imageContentDefined;
|
||||
Vector<Bool> m_depthStencilContentDefined;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -1,875 +0,0 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/UniformDescriptorBinder.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "UniformDescriptorBinder.h"
|
||||
|
||||
#include "VkFramebufferManager.h"
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include "MG_State/GLState/ProgramState/ProgramObject.h"
|
||||
#include "MG_Util/ShaderTranspiler/Types.h"
|
||||
#include <limits>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkDeviceSize UniformDescriptorBinder::AlignUp(VkDeviceSize value, VkDeviceSize alignment) {
|
||||
if (alignment == 0) {
|
||||
return value;
|
||||
}
|
||||
return (value + alignment - 1) / alignment * alignment;
|
||||
}
|
||||
|
||||
Uint64 UniformDescriptorBinder::ComputeProgramHash(const MG_State::GLState::ProgramObject& program) {
|
||||
XXH64_state_t* state = XXH64_createState();
|
||||
XXHASH_VERIFY(XXH64_reset(state, 0xC0D3A11ULL));
|
||||
const auto& spirv = program.GetGeneratedSpirv();
|
||||
for (const auto& module : spirv) {
|
||||
XXHASH_VERIFY(XXH64_update(state, module.data(), module.size() * sizeof(Uint)));
|
||||
}
|
||||
const Uint32 blockCount = static_cast<Uint32>(program.GetActiveUniformBlocksCount());
|
||||
XXHASH_VERIFY(XXH64_update(state, &blockCount, sizeof(blockCount)));
|
||||
for (Uint32 i = 0; i < blockCount; ++i) {
|
||||
const Uint32 binding = program.GetUniformBlockBinding(i);
|
||||
XXHASH_VERIFY(XXH64_update(state, &binding, sizeof(binding)));
|
||||
}
|
||||
const Uint64 hash = XXH64_digest(state);
|
||||
XXH64_freeState(state);
|
||||
return hash;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::IsSamplerUniformType(GLenum glType) {
|
||||
switch (glType) {
|
||||
case GL_SAMPLER_1D:
|
||||
case GL_SAMPLER_2D:
|
||||
case GL_SAMPLER_3D:
|
||||
case GL_SAMPLER_CUBE:
|
||||
case GL_SAMPLER_1D_SHADOW:
|
||||
case GL_SAMPLER_2D_SHADOW:
|
||||
case GL_SAMPLER_1D_ARRAY:
|
||||
case GL_SAMPLER_2D_ARRAY:
|
||||
case GL_SAMPLER_1D_ARRAY_SHADOW:
|
||||
case GL_SAMPLER_2D_ARRAY_SHADOW:
|
||||
case GL_SAMPLER_2D_MULTISAMPLE:
|
||||
case GL_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||
case GL_SAMPLER_CUBE_SHADOW:
|
||||
case GL_SAMPLER_BUFFER:
|
||||
case GL_SAMPLER_2D_RECT:
|
||||
case GL_SAMPLER_2D_RECT_SHADOW:
|
||||
case GL_INT_SAMPLER_1D:
|
||||
case GL_INT_SAMPLER_2D:
|
||||
case GL_INT_SAMPLER_3D:
|
||||
case GL_INT_SAMPLER_CUBE:
|
||||
case GL_INT_SAMPLER_1D_ARRAY:
|
||||
case GL_INT_SAMPLER_2D_ARRAY:
|
||||
case GL_INT_SAMPLER_2D_MULTISAMPLE:
|
||||
case GL_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||
case GL_INT_SAMPLER_BUFFER:
|
||||
case GL_INT_SAMPLER_2D_RECT:
|
||||
case GL_UNSIGNED_INT_SAMPLER_1D:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D:
|
||||
case GL_UNSIGNED_INT_SAMPLER_3D:
|
||||
case GL_UNSIGNED_INT_SAMPLER_CUBE:
|
||||
case GL_UNSIGNED_INT_SAMPLER_1D_ARRAY:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_ARRAY:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||
case GL_UNSIGNED_INT_SAMPLER_BUFFER:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_RECT:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
TextureTarget UniformDescriptorBinder::UniformTypeToTextureTarget(GLenum glType) {
|
||||
switch (glType) {
|
||||
case GL_SAMPLER_1D:
|
||||
case GL_INT_SAMPLER_1D:
|
||||
case GL_UNSIGNED_INT_SAMPLER_1D:
|
||||
return TextureTarget::Texture1D;
|
||||
case GL_SAMPLER_3D:
|
||||
case GL_INT_SAMPLER_3D:
|
||||
case GL_UNSIGNED_INT_SAMPLER_3D:
|
||||
return TextureTarget::Texture3D;
|
||||
case GL_SAMPLER_CUBE:
|
||||
case GL_SAMPLER_CUBE_SHADOW:
|
||||
case GL_INT_SAMPLER_CUBE:
|
||||
case GL_UNSIGNED_INT_SAMPLER_CUBE:
|
||||
return TextureTarget::TextureCubeMap;
|
||||
case GL_SAMPLER_2D_MULTISAMPLE:
|
||||
case GL_INT_SAMPLER_2D_MULTISAMPLE:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE:
|
||||
return TextureTarget::Texture2DMultisample;
|
||||
case GL_SAMPLER_BUFFER:
|
||||
case GL_INT_SAMPLER_BUFFER:
|
||||
case GL_UNSIGNED_INT_SAMPLER_BUFFER:
|
||||
return TextureTarget::TextureBuffer;
|
||||
case GL_SAMPLER_1D_ARRAY:
|
||||
case GL_SAMPLER_1D_ARRAY_SHADOW:
|
||||
case GL_INT_SAMPLER_1D_ARRAY:
|
||||
case GL_UNSIGNED_INT_SAMPLER_1D_ARRAY:
|
||||
return TextureTarget::Texture1DArray;
|
||||
case GL_SAMPLER_2D_ARRAY:
|
||||
case GL_SAMPLER_2D_ARRAY_SHADOW:
|
||||
case GL_INT_SAMPLER_2D_ARRAY:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_ARRAY:
|
||||
return TextureTarget::Texture2DArray;
|
||||
case GL_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||
case GL_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||
return TextureTarget::Texture2DMultisampleArray;
|
||||
case GL_SAMPLER_2D_RECT:
|
||||
case GL_SAMPLER_2D_RECT_SHADOW:
|
||||
case GL_INT_SAMPLER_2D_RECT:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D_RECT:
|
||||
return TextureTarget::TextureRectangle;
|
||||
case GL_SAMPLER_2D:
|
||||
case GL_SAMPLER_2D_SHADOW:
|
||||
case GL_INT_SAMPLER_2D:
|
||||
case GL_UNSIGNED_INT_SAMPLER_2D:
|
||||
default:
|
||||
return TextureTarget::Texture2D;
|
||||
}
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::Initialize(VkDevice device, VmaAllocator allocator,
|
||||
VkDeviceSize minUniformBufferOffsetAlignment, Uint32 frameCount,
|
||||
Uint32 maxBindings, Uint32 setsPerFrame, VkDeviceSize perFrameUploadBytes,
|
||||
VkTextureSamplerManager* textureSamplerManager,
|
||||
VkFramebufferManager* framebufferManager) {
|
||||
Shutdown();
|
||||
|
||||
MOBILEGL_ASSERT(device != VK_NULL_HANDLE, "UniformDescriptorBinder::Initialize requires valid VkDevice");
|
||||
MOBILEGL_ASSERT(allocator != nullptr, "UniformDescriptorBinder::Initialize requires valid VMA allocator");
|
||||
MOBILEGL_ASSERT(frameCount > 0, "UniformDescriptorBinder::Initialize requires frameCount > 0");
|
||||
MOBILEGL_ASSERT(maxBindings > 0, "UniformDescriptorBinder::Initialize requires maxBindings > 0");
|
||||
MOBILEGL_ASSERT(setsPerFrame > 0, "UniformDescriptorBinder::Initialize requires setsPerFrame > 0");
|
||||
|
||||
m_device = device;
|
||||
m_allocator = allocator;
|
||||
m_minDynamicOffsetAlignment = std::max<VkDeviceSize>(1, minUniformBufferOffsetAlignment);
|
||||
m_perFrameUploadBytes = perFrameUploadBytes;
|
||||
m_frameCount = frameCount;
|
||||
m_maxBindings = maxBindings;
|
||||
m_setsPerFrame = setsPerFrame;
|
||||
m_peakDescriptorSetsObserved = 0;
|
||||
m_textureSamplerManager = textureSamplerManager;
|
||||
m_framebufferManager = framebufferManager;
|
||||
|
||||
m_frames.resize(m_frameCount);
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_frameCount; ++frameIndex) {
|
||||
auto& frame = m_frames[frameIndex];
|
||||
frame.writeCursor = 0;
|
||||
frame.activeDescriptorPoolIndex = 0;
|
||||
frame.allocatedSetsThisFrame = 0;
|
||||
frame.peakAllocatedSetsThisFrame = 0;
|
||||
frame.descriptorPools.clear();
|
||||
|
||||
VkDescriptorPool initialPool = VK_NULL_HANDLE;
|
||||
if (!CreateDescriptorPool(m_setsPerFrame, initialPool)) {
|
||||
MGLOG_E("UniformDescriptorBinder::Initialize failed: cannot create frame descriptor pool %u",
|
||||
frameIndex);
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
frame.descriptorPools.push_back({initialPool, m_setsPerFrame, 0});
|
||||
MGLOG_D("UniformDescriptorBinder: frame %u descriptor pool created (maxSets=%u)", frameIndex, m_setsPerFrame);
|
||||
|
||||
const Bool created = frame.uploadBuffer.Create(
|
||||
m_allocator, m_perFrameUploadBytes, VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT, VMA_MEMORY_USAGE_AUTO,
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT);
|
||||
if (!created) {
|
||||
MGLOG_E("UniformDescriptorBinder::Initialize failed: cannot create frame upload buffer %u", frameIndex);
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void UniformDescriptorBinder::Shutdown() {
|
||||
for (auto& frame : m_frames) {
|
||||
frame.uploadBuffer.Destroy();
|
||||
if (m_device != VK_NULL_HANDLE) {
|
||||
for (auto& bucket : frame.descriptorPools) {
|
||||
if (bucket.handle != VK_NULL_HANDLE) {
|
||||
vkDestroyDescriptorPool(m_device, bucket.handle, nullptr);
|
||||
bucket.handle = VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
}
|
||||
frame.descriptorPools.clear();
|
||||
frame.activeDescriptorPoolIndex = 0;
|
||||
frame.allocatedSetsThisFrame = 0;
|
||||
frame.peakAllocatedSetsThisFrame = 0;
|
||||
frame.writeCursor = 0;
|
||||
}
|
||||
m_frames.clear();
|
||||
DestroyProgramLayouts();
|
||||
|
||||
m_allocator = nullptr;
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_minDynamicOffsetAlignment = 1;
|
||||
m_perFrameUploadBytes = 0;
|
||||
m_frameCount = 0;
|
||||
m_maxBindings = 0;
|
||||
m_setsPerFrame = 0;
|
||||
m_peakDescriptorSetsObserved = 0;
|
||||
m_textureSamplerManager = nullptr;
|
||||
m_framebufferManager = nullptr;
|
||||
}
|
||||
|
||||
void UniformDescriptorBinder::BeginFrame(Uint32 frameIndex) {
|
||||
MOBILEGL_ASSERT(frameIndex < m_frames.size(), "UniformDescriptorBinder::BeginFrame invalid frame index");
|
||||
auto& frame = m_frames[frameIndex];
|
||||
if (frame.peakAllocatedSetsThisFrame > m_peakDescriptorSetsObserved) {
|
||||
m_peakDescriptorSetsObserved = frame.peakAllocatedSetsThisFrame;
|
||||
MGLOG_D(
|
||||
"UniformDescriptorBinder: new descriptor set peak observed=%u (base setsPerFrame=%u, frame=%u, pools=%zu)",
|
||||
m_peakDescriptorSetsObserved, m_setsPerFrame, frameIndex, frame.descriptorPools.size());
|
||||
}
|
||||
frame.writeCursor = 0;
|
||||
frame.activeDescriptorPoolIndex = 0;
|
||||
frame.allocatedSetsThisFrame = 0;
|
||||
frame.peakAllocatedSetsThisFrame = 0;
|
||||
for (auto& bucket : frame.descriptorPools) {
|
||||
bucket.allocatedSets = 0;
|
||||
if (bucket.handle == VK_NULL_HANDLE) {
|
||||
continue;
|
||||
}
|
||||
VK_VERIFY(vkResetDescriptorPool(m_device, bucket.handle, 0),
|
||||
"UniformDescriptorBinder::BeginFrame, vkResetDescriptorPool");
|
||||
}
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::ReflectBindingKinds(const MG_State::GLState::ProgramObject& program,
|
||||
Vector<BindingKind>& outKinds) const {
|
||||
outKinds.assign(m_maxBindings, BindingKind::None);
|
||||
|
||||
const auto& spirv = program.GetGeneratedSpirv();
|
||||
for (const auto& module : spirv) {
|
||||
if (module.empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
spvc_context context = nullptr;
|
||||
spvc_parsed_ir ir = nullptr;
|
||||
spvc_compiler compiler = nullptr;
|
||||
spvc_resources resources = nullptr;
|
||||
|
||||
if (spvc_context_create(&context) != SPVC_SUCCESS) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const spvc_result parseResult = spvc_context_parse_spirv(context, module.data(), module.size(), &ir);
|
||||
if (parseResult != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
|
||||
const spvc_result compilerResult =
|
||||
spvc_context_create_compiler(context, SPVC_BACKEND_GLSL, ir, SPVC_CAPTURE_MODE_TAKE_OWNERSHIP, &compiler);
|
||||
if (compilerResult != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (spvc_compiler_create_shader_resources(compiler, &resources) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto applyBindings = [&](spvc_resource_type resourceType, BindingKind kind) {
|
||||
const spvc_reflected_resource* list = nullptr;
|
||||
size_t count = 0;
|
||||
if (spvc_resources_get_resource_list_for_type(resources, resourceType, &list, &count) != SPVC_SUCCESS) {
|
||||
return;
|
||||
}
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
const Uint32 binding =
|
||||
spvc_compiler_get_decoration(compiler, list[i].id, SpvDecorationBinding);
|
||||
if (binding >= m_maxBindings) {
|
||||
continue;
|
||||
}
|
||||
if (kind == BindingKind::CombinedImageSampler) {
|
||||
outKinds[binding] = BindingKind::CombinedImageSampler;
|
||||
} else if (outKinds[binding] == BindingKind::None) {
|
||||
outKinds[binding] = kind;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
applyBindings(SPVC_RESOURCE_TYPE_UNIFORM_BUFFER, BindingKind::UniformBufferDynamic);
|
||||
applyBindings(SPVC_RESOURCE_TYPE_SAMPLED_IMAGE, BindingKind::CombinedImageSampler);
|
||||
|
||||
spvc_context_destroy(context);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::ReflectSamplerBindings(const MG_State::GLState::ProgramObject& program,
|
||||
ProgramLayout& layout) const {
|
||||
layout.samplerUniformLocationByBinding.assign(m_maxBindings, -1);
|
||||
layout.samplerTextureTargetByBinding.assign(m_maxBindings, TextureTarget::Texture2D);
|
||||
|
||||
const auto& spirv = program.GetGeneratedSpirv();
|
||||
for (const auto& module : spirv) {
|
||||
if (module.empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
spvc_context context = nullptr;
|
||||
spvc_parsed_ir ir = nullptr;
|
||||
spvc_compiler compiler = nullptr;
|
||||
spvc_resources resources = nullptr;
|
||||
|
||||
if (spvc_context_create(&context) != SPVC_SUCCESS) {
|
||||
return false;
|
||||
}
|
||||
if (spvc_context_parse_spirv(context, module.data(), module.size(), &ir) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
if (spvc_context_create_compiler(context, SPVC_BACKEND_GLSL, ir, SPVC_CAPTURE_MODE_TAKE_OWNERSHIP,
|
||||
&compiler) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
if (spvc_compiler_create_shader_resources(compiler, &resources) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
|
||||
const spvc_reflected_resource* list = nullptr;
|
||||
size_t count = 0;
|
||||
if (spvc_resources_get_resource_list_for_type(resources, SPVC_RESOURCE_TYPE_SAMPLED_IMAGE, &list, &count) ==
|
||||
SPVC_SUCCESS) {
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
const Uint32 binding =
|
||||
spvc_compiler_get_decoration(compiler, list[i].id, SpvDecorationBinding);
|
||||
if (binding >= m_maxBindings) {
|
||||
continue;
|
||||
}
|
||||
|
||||
String uniformName = list[i].name ? list[i].name : "";
|
||||
Int location = program.GetUniformLocation(uniformName);
|
||||
if (location < 0) {
|
||||
const auto arraySuffix = uniformName.find("[0]");
|
||||
if (arraySuffix != String::npos) {
|
||||
uniformName = uniformName.substr(0, arraySuffix);
|
||||
location = program.GetUniformLocation(uniformName);
|
||||
}
|
||||
}
|
||||
if (location < 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
layout.samplerUniformLocationByBinding[binding] = location;
|
||||
layout.samplerTextureTargetByBinding[binding] =
|
||||
UniformTypeToTextureTarget(program.GetUniformType(static_cast<Uint>(location)));
|
||||
}
|
||||
}
|
||||
|
||||
spvc_context_destroy(context);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::ReflectGlobalUboBinding(const MG_State::GLState::ProgramObject& program,
|
||||
ProgramLayout& layout) const {
|
||||
layout.globalUboBinding = -1;
|
||||
|
||||
const auto& spirv = program.GetGeneratedSpirv();
|
||||
for (const auto& module : spirv) {
|
||||
if (module.empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
spvc_context context = nullptr;
|
||||
spvc_parsed_ir ir = nullptr;
|
||||
spvc_compiler compiler = nullptr;
|
||||
spvc_resources resources = nullptr;
|
||||
|
||||
if (spvc_context_create(&context) != SPVC_SUCCESS) {
|
||||
return false;
|
||||
}
|
||||
if (spvc_context_parse_spirv(context, module.data(), module.size(), &ir) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
if (spvc_context_create_compiler(context, SPVC_BACKEND_GLSL, ir, SPVC_CAPTURE_MODE_TAKE_OWNERSHIP,
|
||||
&compiler) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
if (spvc_compiler_create_shader_resources(compiler, &resources) != SPVC_SUCCESS) {
|
||||
spvc_context_destroy(context);
|
||||
continue;
|
||||
}
|
||||
|
||||
const spvc_reflected_resource* list = nullptr;
|
||||
size_t count = 0;
|
||||
if (spvc_resources_get_resource_list_for_type(resources, SPVC_RESOURCE_TYPE_UNIFORM_BUFFER, &list, &count) ==
|
||||
SPVC_SUCCESS) {
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
const char* name = list[i].name ? list[i].name : "";
|
||||
if (std::strstr(name, MG_Util::ShaderTranspiler::GLOBAL_UBO_NAME) == nullptr) {
|
||||
continue;
|
||||
}
|
||||
const Uint32 binding =
|
||||
spvc_compiler_get_decoration(compiler, list[i].id, SpvDecorationBinding);
|
||||
if (binding < m_maxBindings) {
|
||||
layout.globalUboBinding = static_cast<Int>(binding);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
spvc_context_destroy(context);
|
||||
if (layout.globalUboBinding >= 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::ResolveSamplerDescriptor(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramLayout& layout, Uint32 binding,
|
||||
VkDescriptorImageInfo& outImageInfo) const {
|
||||
if (!m_textureSamplerManager || !MG_State::pGLContext || binding >= layout.samplerUniformLocationByBinding.size()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Int location = layout.samplerUniformLocationByBinding[binding];
|
||||
if (location < 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Int unit = program.GetUniformSamplerOrImageUnitIndex(static_cast<Uint>(location));
|
||||
if (unit < 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& textureUnit = MG_State::pGLContext->GetTextureUnitObject(unit);
|
||||
const auto samplerOverride = textureUnit.GetSamplerObject();
|
||||
|
||||
const TextureTarget preferredTarget = layout.samplerTextureTargetByBinding[binding];
|
||||
auto texture = textureUnit.GetBindingSlot(preferredTarget).GetBoundObject();
|
||||
if (!texture) {
|
||||
auto& slots = textureUnit.GetAllBindingSlots();
|
||||
for (auto& slot : slots) {
|
||||
texture = slot.GetBoundObject();
|
||||
if (texture) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!texture) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (m_framebufferManager &&
|
||||
m_framebufferManager->TransitionOffscreenColorTextureToShaderRead(commandBuffer, texture->GetExternalIndex())) {
|
||||
VkImageView offscreenView = VK_NULL_HANDLE;
|
||||
if (m_framebufferManager->GetOffscreenColorViewByTexture(texture->GetExternalIndex(), offscreenView) &&
|
||||
offscreenView != VK_NULL_HANDLE) {
|
||||
VkDescriptorImageInfo sampledInfo{};
|
||||
if (!m_textureSamplerManager->SyncTextureAndGetDescriptor(*texture, samplerOverride.get(), sampledInfo)) {
|
||||
return false;
|
||||
}
|
||||
sampledInfo.imageView = offscreenView;
|
||||
sampledInfo.imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
||||
outImageInfo = sampledInfo;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return m_textureSamplerManager->SyncTextureAndGetDescriptor(*texture, samplerOverride.get(), outImageInfo);
|
||||
}
|
||||
|
||||
UniformDescriptorBinder::ProgramLayout* UniformDescriptorBinder::GetOrCreateProgramLayout(
|
||||
const MG_State::GLState::ProgramObject& program) {
|
||||
const Uint64 hash = ComputeProgramHash(program);
|
||||
auto it = m_programLayouts.find(hash);
|
||||
if (it != m_programLayouts.end()) {
|
||||
return &it->second;
|
||||
}
|
||||
|
||||
ProgramLayout layout{};
|
||||
layout.hash = hash;
|
||||
if (!ReflectBindingKinds(program, layout.bindingKinds)) {
|
||||
MGLOG_E("UniformDescriptorBinder::GetOrCreateProgramLayout failed: reflection failed");
|
||||
return nullptr;
|
||||
}
|
||||
if (!ReflectSamplerBindings(program, layout)) {
|
||||
MGLOG_E("UniformDescriptorBinder::GetOrCreateProgramLayout failed: sampler reflection failed");
|
||||
return nullptr;
|
||||
}
|
||||
if (!ReflectGlobalUboBinding(program, layout)) {
|
||||
MGLOG_E("UniformDescriptorBinder::GetOrCreateProgramLayout failed: global UBO reflection failed");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
Vector<VkDescriptorSetLayoutBinding> bindings;
|
||||
bindings.reserve(m_maxBindings);
|
||||
for (Uint32 binding = 0; binding < m_maxBindings; ++binding) {
|
||||
const auto kind = layout.bindingKinds[binding];
|
||||
if (kind == BindingKind::None) {
|
||||
continue;
|
||||
}
|
||||
|
||||
VkDescriptorSetLayoutBinding layoutBinding{};
|
||||
layoutBinding.binding = binding;
|
||||
layoutBinding.descriptorCount = 1;
|
||||
layoutBinding.stageFlags = VK_SHADER_STAGE_ALL_GRAPHICS;
|
||||
layoutBinding.pImmutableSamplers = nullptr;
|
||||
if (kind == BindingKind::UniformBufferDynamic) {
|
||||
layoutBinding.descriptorType = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC;
|
||||
layout.dynamicBindings.push_back(binding);
|
||||
} else {
|
||||
layoutBinding.descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
||||
}
|
||||
bindings.push_back(layoutBinding);
|
||||
}
|
||||
|
||||
VkDescriptorSetLayoutCreateInfo setLayoutInfo{};
|
||||
setLayoutInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO;
|
||||
setLayoutInfo.bindingCount = static_cast<Uint32>(bindings.size());
|
||||
setLayoutInfo.pBindings = bindings.data();
|
||||
VK_VERIFY(vkCreateDescriptorSetLayout(m_device, &setLayoutInfo, nullptr, &layout.descriptorSetLayout),
|
||||
"UniformDescriptorBinder::GetOrCreateProgramLayout, vkCreateDescriptorSetLayout");
|
||||
|
||||
VkPipelineLayoutCreateInfo pipelineLayoutInfo{};
|
||||
pipelineLayoutInfo.sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
|
||||
pipelineLayoutInfo.setLayoutCount = 1;
|
||||
pipelineLayoutInfo.pSetLayouts = &layout.descriptorSetLayout;
|
||||
VK_VERIFY(vkCreatePipelineLayout(m_device, &pipelineLayoutInfo, nullptr, &layout.pipelineLayout),
|
||||
"UniformDescriptorBinder::GetOrCreateProgramLayout, vkCreatePipelineLayout");
|
||||
|
||||
auto [insertIt, _] = m_programLayouts.emplace(hash, std::move(layout));
|
||||
return &insertIt->second;
|
||||
}
|
||||
|
||||
VkPipelineLayout UniformDescriptorBinder::GetOrCreatePipelineLayout(const MG_State::GLState::ProgramObject& program) {
|
||||
auto* layout = GetOrCreateProgramLayout(program);
|
||||
return layout ? layout->pipelineLayout : VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::AllocateUploadRegion(FrameResources& frame, VkDeviceSize size, VkDeviceSize& outOffset) {
|
||||
const VkDeviceSize alignedOffset = AlignUp(frame.writeCursor, m_minDynamicOffsetAlignment);
|
||||
if (alignedOffset + size > m_perFrameUploadBytes) {
|
||||
return false;
|
||||
}
|
||||
outOffset = alignedOffset;
|
||||
frame.writeCursor = alignedOffset + size;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::GatherBindingPayloads(const MG_State::GLState::ProgramObject& program,
|
||||
Vector<const void*>& outData,
|
||||
Vector<VkDeviceSize>& outSizes) const {
|
||||
outData.assign(m_maxBindings, nullptr);
|
||||
outSizes.assign(m_maxBindings, 0);
|
||||
|
||||
if (MG_State::pGLContext == nullptr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Uint32 activeUniformBlockCount = static_cast<Uint32>(program.GetActiveUniformBlocksCount());
|
||||
const Uint32 uniformBindingPointCount =
|
||||
static_cast<Uint32>(MG_State::pGLContext->GetBufferBindingPointCount(BufferTarget::Uniform));
|
||||
|
||||
for (Uint32 blockIndex = 0; blockIndex < activeUniformBlockCount; ++blockIndex) {
|
||||
const Uint32 binding = program.GetUniformBlockBinding(blockIndex);
|
||||
if (binding >= m_maxBindings) {
|
||||
continue;
|
||||
}
|
||||
|
||||
VkDeviceSize blockSize = static_cast<VkDeviceSize>(program.GetUBOSizeAt(blockIndex));
|
||||
if (blockSize == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (binding >= uniformBindingPointCount) {
|
||||
continue;
|
||||
}
|
||||
auto& bindingPoint = MG_State::pGLContext->GetBufferBindingPoint(BufferTarget::Uniform, binding);
|
||||
const auto bufferObject = bindingPoint.GetBoundObject();
|
||||
if (!bufferObject) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto bufferData = bufferObject->GetDataReadOnly();
|
||||
if (!bufferData || bufferData->empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto range = bindingPoint.GetRange();
|
||||
const VkDeviceSize bufferSize = static_cast<VkDeviceSize>(bufferObject->GetSize());
|
||||
VkDeviceSize rangeStart = static_cast<VkDeviceSize>(range.start);
|
||||
VkDeviceSize rangeEnd = static_cast<VkDeviceSize>(range.end);
|
||||
|
||||
if (rangeStart >= bufferSize) {
|
||||
continue;
|
||||
}
|
||||
if (rangeEnd <= rangeStart || rangeEnd > bufferSize) {
|
||||
rangeEnd = bufferSize;
|
||||
}
|
||||
|
||||
VkDeviceSize available = rangeEnd - rangeStart;
|
||||
if (available == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
outData[binding] = bufferData->data() + static_cast<SizeT>(rangeStart);
|
||||
outSizes[binding] = std::min(blockSize, available);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const {
|
||||
outPool = VK_NULL_HANDLE;
|
||||
if (m_device == VK_NULL_HANDLE || maxSets == 0 || m_maxBindings == 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Uint64 descriptorCount64 = static_cast<Uint64>(maxSets) * static_cast<Uint64>(m_maxBindings);
|
||||
if (descriptorCount64 > static_cast<Uint64>(std::numeric_limits<Uint32>::max())) {
|
||||
MGLOG_E("UniformDescriptorBinder::CreateDescriptorPool failed: descriptorCount overflow");
|
||||
return false;
|
||||
}
|
||||
|
||||
const Uint32 descriptorCount = static_cast<Uint32>(descriptorCount64);
|
||||
VkDescriptorPoolSize poolSizes[2]{};
|
||||
poolSizes[0].type = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC;
|
||||
poolSizes[0].descriptorCount = descriptorCount;
|
||||
poolSizes[1].type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
||||
poolSizes[1].descriptorCount = descriptorCount;
|
||||
|
||||
VkDescriptorPoolCreateInfo poolInfo{};
|
||||
poolInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO;
|
||||
poolInfo.maxSets = maxSets;
|
||||
poolInfo.poolSizeCount = static_cast<Uint32>(std::size(poolSizes));
|
||||
poolInfo.pPoolSizes = poolSizes;
|
||||
|
||||
const VkResult result = vkCreateDescriptorPool(m_device, &poolInfo, nullptr, &outPool);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("UniformDescriptorBinder::CreateDescriptorPool failed: vkCreateDescriptorPool returned %d", result);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex) {
|
||||
if (frame.descriptorPools.empty()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto& currentBucket = frame.descriptorPools[frame.activeDescriptorPoolIndex];
|
||||
const Uint32 currentMaxSets = std::max<Uint32>(1, currentBucket.maxSets);
|
||||
const Uint32 grownMaxSets = currentMaxSets <= (std::numeric_limits<Uint32>::max() / 2) ? (currentMaxSets * 2)
|
||||
: currentMaxSets;
|
||||
|
||||
VkDescriptorPool grownPool = VK_NULL_HANDLE;
|
||||
if (!CreateDescriptorPool(grownMaxSets, grownPool)) {
|
||||
MGLOG_E("UniformDescriptorBinder::GrowFrameDescriptorPool failed: cannot create grown pool (%u -> %u sets)",
|
||||
currentMaxSets, grownMaxSets);
|
||||
return false;
|
||||
}
|
||||
|
||||
frame.descriptorPools.push_back({grownPool, grownMaxSets, 0});
|
||||
frame.activeDescriptorPoolIndex = static_cast<Uint32>(frame.descriptorPools.size() - 1);
|
||||
MGLOG_D(
|
||||
"UniformDescriptorBinder: frame %u descriptor pool exhausted, grew pool (%u -> %u sets), poolCount=%zu",
|
||||
frameIndex, currentMaxSets, grownMaxSets, frame.descriptorPools.size());
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformDescriptorBinder::BindProgramUniformBuffers(VkCommandBuffer commandBuffer, VkPipelineLayout pipelineLayout,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
Uint32 frameIndex) {
|
||||
if (m_frames.empty()) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: binder is not initialized");
|
||||
return false;
|
||||
}
|
||||
if (frameIndex >= m_frames.size()) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: invalid frame index %u", frameIndex);
|
||||
return false;
|
||||
}
|
||||
|
||||
ProgramLayout* layout = GetOrCreateProgramLayout(program);
|
||||
if (!layout || layout->pipelineLayout == VK_NULL_HANDLE || layout->descriptorSetLayout == VK_NULL_HANDLE) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: cannot get program layout");
|
||||
return false;
|
||||
}
|
||||
if (layout->pipelineLayout != pipelineLayout) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: pipelineLayout mismatch");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& frame = m_frames[frameIndex];
|
||||
if (frame.descriptorPools.empty()) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: frame descriptor pools are invalid");
|
||||
return false;
|
||||
}
|
||||
if (frame.activeDescriptorPoolIndex >= frame.descriptorPools.size()) {
|
||||
frame.activeDescriptorPoolIndex = 0;
|
||||
}
|
||||
|
||||
VkDescriptorSetAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO;
|
||||
allocInfo.descriptorSetCount = 1;
|
||||
allocInfo.pSetLayouts = &layout->descriptorSetLayout;
|
||||
VkDescriptorSet descriptorSet = VK_NULL_HANDLE;
|
||||
|
||||
auto allocateFromActivePool = [&](VkResult& outResult) {
|
||||
auto& bucket = frame.descriptorPools[frame.activeDescriptorPoolIndex];
|
||||
allocInfo.descriptorPool = bucket.handle;
|
||||
outResult = vkAllocateDescriptorSets(m_device, &allocInfo, &descriptorSet);
|
||||
if (outResult == VK_SUCCESS) {
|
||||
++bucket.allocatedSets;
|
||||
++frame.allocatedSetsThisFrame;
|
||||
frame.peakAllocatedSetsThisFrame = std::max(frame.peakAllocatedSetsThisFrame, frame.allocatedSetsThisFrame);
|
||||
}
|
||||
};
|
||||
|
||||
VkResult allocResult = VK_SUCCESS;
|
||||
allocateFromActivePool(allocResult);
|
||||
if (allocResult == VK_ERROR_OUT_OF_POOL_MEMORY || allocResult == VK_ERROR_FRAGMENTED_POOL) {
|
||||
if (!GrowFrameDescriptorPool(frame, frameIndex)) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: descriptor pool growth failed");
|
||||
return false;
|
||||
}
|
||||
allocateFromActivePool(allocResult);
|
||||
}
|
||||
if (allocResult != VK_SUCCESS || descriptorSet == VK_NULL_HANDLE) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: vkAllocateDescriptorSets returned %d",
|
||||
allocResult);
|
||||
return false;
|
||||
}
|
||||
|
||||
Vector<const void*> bindingData;
|
||||
Vector<VkDeviceSize> bindingSizes;
|
||||
if (!GatherBindingPayloads(program, bindingData, bindingSizes)) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: cannot gather UBO payloads");
|
||||
return false;
|
||||
}
|
||||
|
||||
static const Uint8 kFallbackData[16] = {};
|
||||
VkDescriptorImageInfo fallbackImageInfo{};
|
||||
const Bool hasFallbackImage = m_textureSamplerManager && m_textureSamplerManager->GetFallbackDescriptor(fallbackImageInfo);
|
||||
|
||||
Vector<VkWriteDescriptorSet> writes;
|
||||
writes.reserve(m_maxBindings);
|
||||
Vector<VkDescriptorBufferInfo> bufferInfos;
|
||||
Vector<VkDescriptorImageInfo> imageInfos;
|
||||
Vector<Uint32> dynamicOffsets;
|
||||
bufferInfos.reserve(m_maxBindings);
|
||||
imageInfos.reserve(m_maxBindings);
|
||||
dynamicOffsets.reserve(layout->dynamicBindings.size());
|
||||
|
||||
for (Uint32 binding = 0; binding < m_maxBindings; ++binding) {
|
||||
const auto kind = layout->bindingKinds[binding];
|
||||
if (kind == BindingKind::None) {
|
||||
continue;
|
||||
}
|
||||
|
||||
VkWriteDescriptorSet write{};
|
||||
write.sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
|
||||
write.dstSet = descriptorSet;
|
||||
write.dstBinding = binding;
|
||||
write.dstArrayElement = 0;
|
||||
write.descriptorCount = 1;
|
||||
|
||||
if (kind == BindingKind::UniformBufferDynamic) {
|
||||
const void* payload = bindingData[binding];
|
||||
VkDeviceSize payloadSize = bindingSizes[binding];
|
||||
if (payload == nullptr || payloadSize == 0) {
|
||||
if (layout->globalUboBinding == static_cast<Int>(binding)) {
|
||||
const void* globalUboData = program.GetUBOData();
|
||||
const VkDeviceSize globalUboSize = static_cast<VkDeviceSize>(program.GetUBOSize());
|
||||
if (globalUboData != nullptr && globalUboSize > 0) {
|
||||
payload = globalUboData;
|
||||
payloadSize = globalUboSize;
|
||||
}
|
||||
}
|
||||
if (payload == nullptr || payloadSize == 0) {
|
||||
payload = kFallbackData;
|
||||
payloadSize = sizeof(kFallbackData);
|
||||
}
|
||||
}
|
||||
|
||||
VkDeviceSize payloadOffset = 0;
|
||||
if (!AllocateUploadRegion(frame, payloadSize, payloadOffset)) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: frame upload buffer exhausted");
|
||||
return false;
|
||||
}
|
||||
if (!frame.uploadBuffer.Upload(payload, payloadSize, payloadOffset)) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: UBO upload failed on binding %u",
|
||||
binding);
|
||||
return false;
|
||||
}
|
||||
|
||||
VkDescriptorBufferInfo bufferInfo{};
|
||||
bufferInfo.buffer = frame.uploadBuffer.GetHandle();
|
||||
bufferInfo.offset = 0;
|
||||
bufferInfo.range = payloadSize;
|
||||
bufferInfos.push_back(bufferInfo);
|
||||
|
||||
write.descriptorType = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC;
|
||||
write.pBufferInfo = &bufferInfos.back();
|
||||
writes.push_back(write);
|
||||
dynamicOffsets.push_back(static_cast<Uint32>(payloadOffset));
|
||||
} else {
|
||||
VkDescriptorImageInfo imageInfo{};
|
||||
Bool hasImage = ResolveSamplerDescriptor(commandBuffer, program, *layout, binding, imageInfo);
|
||||
if (!hasImage) {
|
||||
if (!hasFallbackImage) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: fallback sampler/texture is unavailable");
|
||||
return false;
|
||||
}
|
||||
imageInfo = fallbackImageInfo;
|
||||
}
|
||||
if (imageInfo.sampler == VK_NULL_HANDLE || imageInfo.imageView == VK_NULL_HANDLE) {
|
||||
MGLOG_E("UniformDescriptorBinder::BindProgramUniformBuffers failed: fallback sampler/texture is unavailable");
|
||||
return false;
|
||||
}
|
||||
imageInfos.push_back(imageInfo);
|
||||
write.descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
||||
write.pImageInfo = &imageInfos.back();
|
||||
writes.push_back(write);
|
||||
}
|
||||
}
|
||||
|
||||
if (!writes.empty()) {
|
||||
vkUpdateDescriptorSets(m_device, static_cast<Uint32>(writes.size()), writes.data(), 0, nullptr);
|
||||
}
|
||||
|
||||
vkCmdBindDescriptorSets(commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, pipelineLayout, 0, 1, &descriptorSet,
|
||||
static_cast<Uint32>(dynamicOffsets.size()), dynamicOffsets.data());
|
||||
return true;
|
||||
}
|
||||
|
||||
void UniformDescriptorBinder::DestroyProgramLayouts() {
|
||||
for (auto& [_, layout] : m_programLayouts) {
|
||||
if (layout.pipelineLayout != VK_NULL_HANDLE) {
|
||||
vkDestroyPipelineLayout(m_device, layout.pipelineLayout, nullptr);
|
||||
layout.pipelineLayout = VK_NULL_HANDLE;
|
||||
}
|
||||
if (layout.descriptorSetLayout != VK_NULL_HANDLE) {
|
||||
vkDestroyDescriptorSetLayout(m_device, layout.descriptorSetLayout, nullptr);
|
||||
layout.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
m_programLayouts.clear();
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -1,104 +0,0 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/UniformDescriptorBinder.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "VkBufferObject.h"
|
||||
#include "VkTextureSamplerManager.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ProgramObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkFramebufferManager;
|
||||
|
||||
class UniformDescriptorBinder {
|
||||
public:
|
||||
enum class BindingKind : Uint8 {
|
||||
None = 0,
|
||||
UniformBufferDynamic,
|
||||
CombinedImageSampler
|
||||
};
|
||||
|
||||
Bool Initialize(VkDevice device, VmaAllocator allocator, VkDeviceSize minUniformBufferOffsetAlignment,
|
||||
Uint32 frameCount, Uint32 maxBindings = 16, Uint32 setsPerFrame = 64,
|
||||
VkDeviceSize perFrameUploadBytes = 4 * 1024 * 1024,
|
||||
VkTextureSamplerManager* textureSamplerManager = nullptr,
|
||||
VkFramebufferManager* framebufferManager = nullptr);
|
||||
void Shutdown();
|
||||
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
VkPipelineLayout GetOrCreatePipelineLayout(const MG_State::GLState::ProgramObject& program);
|
||||
Bool BindProgramUniformBuffers(VkCommandBuffer commandBuffer, VkPipelineLayout pipelineLayout,
|
||||
const MG_State::GLState::ProgramObject& program, Uint32 frameIndex);
|
||||
|
||||
private:
|
||||
struct DescriptorPoolBucket {
|
||||
VkDescriptorPool handle = VK_NULL_HANDLE;
|
||||
Uint32 maxSets = 0;
|
||||
Uint32 allocatedSets = 0;
|
||||
};
|
||||
|
||||
struct FrameResources {
|
||||
VkBufferObject uploadBuffer;
|
||||
Vector<DescriptorPoolBucket> descriptorPools;
|
||||
Uint32 activeDescriptorPoolIndex = 0;
|
||||
Uint32 allocatedSetsThisFrame = 0;
|
||||
Uint32 peakAllocatedSetsThisFrame = 0;
|
||||
VkDeviceSize writeCursor = 0;
|
||||
};
|
||||
|
||||
struct ProgramLayout {
|
||||
Uint64 hash = 0;
|
||||
VkDescriptorSetLayout descriptorSetLayout = VK_NULL_HANDLE;
|
||||
VkPipelineLayout pipelineLayout = VK_NULL_HANDLE;
|
||||
Vector<BindingKind> bindingKinds;
|
||||
Vector<Uint32> dynamicBindings;
|
||||
Vector<Int> samplerUniformLocationByBinding;
|
||||
Vector<TextureTarget> samplerTextureTargetByBinding;
|
||||
Int globalUboBinding = -1;
|
||||
};
|
||||
|
||||
static VkDeviceSize AlignUp(VkDeviceSize value, VkDeviceSize alignment);
|
||||
static Uint64 ComputeProgramHash(const MG_State::GLState::ProgramObject& program);
|
||||
static Bool IsSamplerUniformType(GLenum glType);
|
||||
static TextureTarget UniformTypeToTextureTarget(GLenum glType);
|
||||
Bool ReflectSamplerBindings(const MG_State::GLState::ProgramObject& program, ProgramLayout& layout) const;
|
||||
Bool ReflectGlobalUboBinding(const MG_State::GLState::ProgramObject& program, ProgramLayout& layout) const;
|
||||
Bool ResolveSamplerDescriptor(VkCommandBuffer commandBuffer, const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramLayout& layout, Uint32 binding,
|
||||
VkDescriptorImageInfo& outImageInfo) const;
|
||||
Bool ReflectBindingKinds(const MG_State::GLState::ProgramObject& program, Vector<BindingKind>& outKinds) const;
|
||||
ProgramLayout* GetOrCreateProgramLayout(const MG_State::GLState::ProgramObject& program);
|
||||
Bool AllocateUploadRegion(FrameResources& frame, VkDeviceSize size, VkDeviceSize& outOffset);
|
||||
Bool GatherBindingPayloads(const MG_State::GLState::ProgramObject& program, Vector<const void*>& outData,
|
||||
Vector<VkDeviceSize>& outSizes) const;
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
|
||||
void DestroyProgramLayouts();
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
Vector<FrameResources> m_frames;
|
||||
UnorderedMap<Uint64, ProgramLayout> m_programLayouts;
|
||||
|
||||
VkDeviceSize m_minDynamicOffsetAlignment = 1;
|
||||
VkDeviceSize m_perFrameUploadBytes = 0;
|
||||
Uint32 m_frameCount = 0;
|
||||
Uint32 m_maxBindings = 0;
|
||||
Uint32 m_setsPerFrame = 0;
|
||||
Uint32 m_peakDescriptorSetsObserved = 0;
|
||||
VkTextureSamplerManager* m_textureSamplerManager = nullptr;
|
||||
VkFramebufferManager* m_framebufferManager = nullptr;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,393 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/UniformDescriptorBinder.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "ProgramFactory.h"
|
||||
#include "VkBufferManager.h"
|
||||
#include "VkSamplerManager.h"
|
||||
#include "VkTextureManager.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject;
|
||||
class ProgramObject;
|
||||
class SamplerObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class UniformManager {
|
||||
public:
|
||||
struct SamplerBindingOverride {
|
||||
Uint32 binding = 0;
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
const MG_State::GLState::SamplerObject* sampler = nullptr;
|
||||
VkImageView imageView = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
Bool Initialize(VkDevice device, VkBufferManager* bufferManager,
|
||||
ProgramFactory* programFactory,
|
||||
VkDeviceSize minUniformBufferOffsetAlignment, Uint32 frameCount,
|
||||
Uint32 maxBindings = 16, Uint32 setsPerFrame = 64,
|
||||
VkTextureManager* textureManager = nullptr, VkSamplerManager* samplerManager = nullptr);
|
||||
void Shutdown();
|
||||
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// A command buffer (re)began recording: descriptor bindings recorded into
|
||||
// the previous buffer do not carry over, so drop the bind-dedup shadow.
|
||||
void OnCommandBufferBoundary() { m_lastBindValid = false; }
|
||||
// A ProgramFactory eviction just destroyed this layout: purge every frame
|
||||
// slot's cached descriptor sets for it, so a recycled handle value can never
|
||||
// stale-hit sets written for the dead layout's bindings. The sets are
|
||||
// vkFreeDescriptorSets'd back to their pools (created with
|
||||
// FREE_DESCRIPTOR_SET_BIT) and the pool accounting is credited, so program
|
||||
// churn recycles pool capacity instead of abandoning it. GPU-safe: the layout
|
||||
// only dies after >1024 idle frame boundaries, so no in-flight command buffer
|
||||
// references its sets. This is the only eviction path for the per-layout
|
||||
// caches - a live layout's entry must never be purged (its sets would be
|
||||
// unreachable pool slots), so there is deliberately no age-based sweep here.
|
||||
void OnDescriptorSetLayoutDestroyed(VkDescriptorSetLayout descriptorSetLayout);
|
||||
// One record per visited CombinedImageSampler DESCRIPTOR (post fallback substitution,
|
||||
// in binding order, and within a binding in array-element order): the resolved texture
|
||||
// and effective sampler, as never-reused lifetime ids so a freed-and-reallocated object
|
||||
// at the same heap address can only MISS a comparison, never false-hit it (same ABA
|
||||
// rule as SamplerResolveMemo). An arrayed binding contributes one record per element -
|
||||
// element granularity is required, or swapping the textures of two elements of the same
|
||||
// array would leave the record list identical and the fast path would keep a stale set.
|
||||
struct SampledBindingRecord {
|
||||
Uint64 textureLifetimeId = 0;
|
||||
Uint64 samplerLifetimeId = 0;
|
||||
};
|
||||
Bool CollectSampledTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures,
|
||||
Vector<SampledBindingRecord>* outBindingRecords = nullptr);
|
||||
// Shadow-compare for the SetupDraw fast path: re-runs the CollectSampledTextures
|
||||
// walk and reports whether every visited binding still resolves to the recorded
|
||||
// (texture, effective sampler) pair. A texture bind generation bump alone (e.g. a
|
||||
// redundant glBindSampler, which always bumps it) does not prove the sampled set
|
||||
// moved; this walk does, without rebuilding the set or falling off the fast path.
|
||||
Bool SampledBindingsUnchanged(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
const Vector<SampledBindingRecord>& previousRecords) const;
|
||||
Bool CollectStorageImageTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Vector<MG_State::GLState::ITextureObject*>& outTextures) const;
|
||||
// samplerDescriptorsUnchangedHint: the caller (SetupDraw fast path) proved that
|
||||
// every input of every combined-image-sampler resolution is unchanged since the
|
||||
// previous draw's resolve - same (texture, sampler) per binding, texture params
|
||||
// sum, sampling-resolution generation (sampler params + texture shape), image
|
||||
// epochs AND per-resource layout values - so the per-binding cached
|
||||
// VkDescriptorImageInfo may be reused without re-running the resolve chain.
|
||||
Bool BindProgramUniformBuffers(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
Uint32 frameIndex,
|
||||
VkPipelineBindPoint bindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS,
|
||||
const SamplerBindingOverride* samplerBindingOverride = nullptr,
|
||||
Bool samplerDescriptorsUnchangedHint = false);
|
||||
|
||||
// Pure format-policy helper kept public for host regression tests. Formatted storage
|
||||
// images use their shader qualifier; transformed float images use glBindImageTexture's
|
||||
// format and never silently fall back to the backing image format.
|
||||
static VkFormat ResolveStorageImageViewFormat(VkFormat reflectedFormat, GLenum bindingFormat,
|
||||
VkFormat resourceFormat, Bool useBindingFormat);
|
||||
|
||||
// True when the program reads at least one sampler and every one of them is bound to a
|
||||
// texture whose GL level range is a single level. Such a sampler resolves to
|
||||
// minLod = maxLod = 0 (see VkSamplerManager::GetOrCreateSampler), so an implicit-LOD sample
|
||||
// and an explicit LOD 0 sample must read the same texel - which is what makes the
|
||||
// ExplicitLod0Sampling SPIR-V rewrite safe to request. Deliberately conservative: it reads
|
||||
// only GL state, so a texture that ends up single-level for another reason (one uploaded
|
||||
// level under a wide level range) merely misses the rewrite.
|
||||
static Bool ProgramSamplesOnlySingleLevelTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
|
||||
private:
|
||||
struct DescriptorPoolBucket {
|
||||
VkDescriptorPool handle = VK_NULL_HANDLE;
|
||||
Uint32 maxSets = 0;
|
||||
Uint32 allocatedSets = 0;
|
||||
};
|
||||
|
||||
// A cached descriptor set together with the pool it was allocated from, so a
|
||||
// layout-destroyed purge can vkFreeDescriptorSets it back and credit the
|
||||
// owning bucket's accounting.
|
||||
struct CachedDescriptorSet {
|
||||
VkDescriptorSet set = VK_NULL_HANDLE;
|
||||
VkDescriptorPool pool = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
struct DescriptorSetCacheEntry {
|
||||
Vector<CachedDescriptorSet> sets;
|
||||
Uint32 cursor = 0;
|
||||
};
|
||||
|
||||
struct FrameResources {
|
||||
Vector<DescriptorPoolBucket> descriptorPools;
|
||||
UnorderedMap<VkDescriptorSetLayout, DescriptorSetCacheEntry> descriptorSetCacheByLayout;
|
||||
Vector<VkBufferView> texelBufferViews;
|
||||
Uint32 activeDescriptorPoolIndex = 0;
|
||||
Uint32 allocatedSetsThisFrame = 0;
|
||||
Uint32 peakAllocatedSetsThisFrame = 0;
|
||||
};
|
||||
|
||||
static Bool ResolveSamplerTexture(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture);
|
||||
// Shared per-binding resolution for CollectSampledTextures and
|
||||
// SampledBindingsUnchanged, so membership and comparison can never diverge:
|
||||
// texture after the fallback substitution (may still be null when no fallback
|
||||
// exists), effective sampler = unit override else the texture's own sampler.
|
||||
// False = the binding is skipped (unbound with a non-2D fallback target).
|
||||
// `element` indexes a sampler array inside the binding; see ResolveSamplerDescriptor.
|
||||
Bool ResolveSampledBinding(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding, Uint32 element,
|
||||
MG_State::GLState::ITextureObject*& outTexture,
|
||||
const MG_State::GLState::SamplerObject*& outSampler) const;
|
||||
// Raw-pointer variant for the per-draw sampled-texture walk (CollectSampledTextures):
|
||||
// the bound texture stays alive through the draw via GL binding state, so callers that
|
||||
// only need the pointer skip the SharedPtr copy's atomic refcount churn.
|
||||
static MG_State::GLState::ITextureObject* ResolveSamplerTextureRaw(
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding, Uint32 element);
|
||||
SharedPtr<MG_State::GLState::ITextureObject> GetFallbackTexture(TextureTarget target) const;
|
||||
// `element` indexes a sampler ARRAY inside one binding; each element carries its own
|
||||
// independently assigned GL texture unit, so it selects the texture, the sampler
|
||||
// override and the fallback separately from its neighbours.
|
||||
//
|
||||
// trustUnchangedHint: reuse this binding's cached VkDescriptorImageInfo outright
|
||||
// (see BindProgramUniformBuffers' samplerDescriptorsUnchangedHint for the proof
|
||||
// obligations the caller carries). The cache is keyed by binding alone, so it is
|
||||
// used ONLY for single-descriptor bindings - see m_samplerResolveMemo.
|
||||
Bool ResolveSamplerDescriptor(VkCommandBuffer commandBuffer, const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 element, VkDescriptorImageInfo& outImageInfo,
|
||||
Bool trustUnchangedHint = false) const;
|
||||
Bool ResolveSamplerDescriptorOverride(const SamplerBindingOverride& samplerBindingOverride,
|
||||
VkDescriptorImageInfo& outImageInfo) const;
|
||||
Bool ResolveTexelBufferDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 frameIndex, VkBufferView& outBufferView);
|
||||
// GLSL `imageBuffer`: the same VkBufferView descriptor as the sampled texel buffer above,
|
||||
// but resolved from an IMAGE unit (glBindImageTexture) rather than a texture unit, and
|
||||
// made GPU-resident-writable because the shader may store to it. No `element` parameter:
|
||||
// an imageBuffer ARRAY is refused at program creation, so a binding is always one
|
||||
// descriptor (see the array gate in RemapDescriptorBindingsForVulkan).
|
||||
Bool ResolveStorageTexelBufferDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 frameIndex, VkBufferView& outBufferView);
|
||||
// `element` indexes a block INSTANCE array's descriptors; it is 0 for every ordinary
|
||||
// block. Each element resolves through its own GL storage block, and so its own GL
|
||||
// binding point, buffer and glBindBufferRange window.
|
||||
Bool ResolveStorageBufferDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 element, VkDescriptorBufferInfo& outBufferInfo) const;
|
||||
// `element` indexes an image ARRAY inside one binding; each element carries its own
|
||||
// independently assigned GL image unit.
|
||||
Bool ResolveStorageImageDescriptor(VkCommandBuffer commandBuffer,
|
||||
const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 element, VkDescriptorImageInfo& outImageInfo) const;
|
||||
// Result of resolving a UBO binding: either a zero-copy direct bind to the app's resident
|
||||
// VkBuffer (the GLES backend's approach - no per-draw copy) or the CPU payload to upload.
|
||||
struct UboBindResult {
|
||||
Bool directBindable = false;
|
||||
VkBuffer buffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize range = 0; // reflected block size; constant across draws (hashed)
|
||||
VkDeviceSize dynamicOffset = 0; // block range start; moves per draw (NOT hashed)
|
||||
const void* payload = nullptr; // fallback UploadTransient path
|
||||
VkDeviceSize payloadSize = 0;
|
||||
};
|
||||
Bool ResolveUniformBufferPayload(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 arrayElement, UboBindResult& out) const;
|
||||
// Shared resolution of one dynamic-UBO binding element into the
|
||||
// (buffer, range, dynamicOffset) triple the descriptor consumes: direct
|
||||
// bind, global-slice reuse, or transient upload. Used by the full walk
|
||||
// and by the dynamic-offset-only rebind (see FastRebindMemo).
|
||||
Bool ResolveDynamicUboDescriptor(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj, Uint32 binding,
|
||||
Uint32 arrayElement, Uint32 frameIndex, VkBuffer& outBuffer,
|
||||
VkDeviceSize& outRange, Uint32& outDynamicOffset);
|
||||
// The vkCmdBindDescriptorSets tail shared by the full walk and the
|
||||
// dynamic-offset-only rebind: skips the driver call when this exact
|
||||
// binding is already live on the command buffer (see the bind-dedup
|
||||
// shadow below), otherwise binds and refreshes the shadow.
|
||||
void BindDescriptorSetDeduped(VkCommandBuffer commandBuffer, VkPipelineBindPoint bindPoint,
|
||||
VkPipelineLayout pipelineLayout, VkDescriptorSet descriptorSet,
|
||||
const Vector<Uint32>& dynamicOffsets);
|
||||
Bool CreateDescriptorPool(Uint32 maxSets, VkDescriptorPool& outPool) const;
|
||||
Bool GrowFrameDescriptorPool(FrameResources& frame, Uint32 frameIndex);
|
||||
VkResult AllocateDescriptorSetsFromActivePool(
|
||||
Uint32 frameIndex, const ProgramFactory::VkProgramObject& programObj, VkDescriptorSet& outDescriptorSet);
|
||||
VkResult AcquireDescriptorSet(Uint32 frameIndex,
|
||||
const ProgramFactory::VkProgramObject& programObj,
|
||||
VkDescriptorSet& outDescriptorSet);
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkBufferManager* m_bufferManager = nullptr;
|
||||
ProgramFactory* m_programFactory = nullptr;
|
||||
Vector<FrameResources> m_frames;
|
||||
|
||||
VkDeviceSize m_minDynamicOffsetAlignment = 1;
|
||||
Uint32 m_frameCount = 0;
|
||||
Uint32 m_maxBindings = 0;
|
||||
Uint32 m_setsPerFrame = 0;
|
||||
Uint32 m_peakDescriptorSetsObserved = 0;
|
||||
VkTextureManager* m_textureManager = nullptr;
|
||||
VkSamplerManager* m_samplerManager = nullptr;
|
||||
mutable SharedPtr<MG_State::GLState::ITextureObject> m_fallbackTexture2D;
|
||||
|
||||
// Per-draw scratch buffers for BindProgramUniformBuffers: reused (clear keeps
|
||||
// capacity) so the descriptor-write path stops allocating on every draw.
|
||||
Vector<VkWriteDescriptorSet> m_writesScratch;
|
||||
Vector<VkDescriptorBufferInfo> m_bufferInfosScratch;
|
||||
Vector<VkDescriptorImageInfo> m_imageInfosScratch;
|
||||
Vector<VkBufferView> m_texelBufferViewsScratch;
|
||||
Vector<Uint32> m_dynamicOffsetsScratch;
|
||||
|
||||
// Descriptor-set reuse across recent draws (see BindProgramUniformBuffers).
|
||||
// When a draw's resolved descriptor content is byte-identical to one memoized
|
||||
// earlier, reuse that VkDescriptorSet and skip AcquireDescriptorSet +
|
||||
// vkUpdateDescriptorSets - only the bind-time dynamic offsets differ. Four
|
||||
// entries with round-robin replacement rather than one: draws alternating
|
||||
// between two programs (MC's chunk<->entity ping-pong) would thrash a single
|
||||
// slot into a full re-allocate+write every draw. Reset each frame in BeginFrame
|
||||
// because the frame's descriptor sets are recycled there.
|
||||
struct DescriptorReuseEntry {
|
||||
Uint64 signature = 0;
|
||||
VkDescriptorSet set = VK_NULL_HANDLE;
|
||||
Bool valid = false;
|
||||
};
|
||||
static constexpr Uint32 kDescriptorReuseMemoSize = 4;
|
||||
DescriptorReuseEntry m_descriptorReuseMemo[kDescriptorReuseMemoSize];
|
||||
Uint32 m_descriptorReuseMemoNext = 0;
|
||||
|
||||
// Dynamic-offset-only rebind (see BindProgramUniformBuffers): records the
|
||||
// descriptor set selected by the last cacheable full walk of a program
|
||||
// whose active bindings are exactly one dynamic UBO (single descriptor)
|
||||
// plus combined-image samplers. When the next call proves every sampler
|
||||
// descriptor input unchanged (samplerDescriptorsUnchangedHint) and the
|
||||
// UBO re-resolves to the SAME VkBuffer+range - only the dynamic offset
|
||||
// moved, the per-draw glUniform case - the walk collapses to: resolve one
|
||||
// offset, rebind the recorded set with new pDynamicOffsets (Vulkan allows
|
||||
// rebinding the same set with different dynamic offsets).
|
||||
// Invalidation inventory: BeginFrame clears it (the frame's sets are
|
||||
// recycled) and the frameIndex field guards cross-frame confusion on top;
|
||||
// OnDescriptorSetLayoutDestroyed clears it (the set may be freed); a
|
||||
// sampler-override walk clears it (mirrors m_descriptorReuseMemo); a
|
||||
// program relink bumps the backend state version and thus programObj.hash
|
||||
// so the key misses; the program lifetime id is never reused, so a
|
||||
// deleted-and-recreated program misses; a texture/sampler/binding change
|
||||
// drops the hint upstream; an arena wrap or growth resolves a different
|
||||
// VkBuffer and misses. AcquireDescriptorSet's per-frame cursor only
|
||||
// advances, so the recorded set is never re-written within its frame.
|
||||
struct FastRebindMemo {
|
||||
Bool valid = false;
|
||||
Uint32 frameIndex = 0;
|
||||
Uint64 programLifetimeId = 0;
|
||||
ProgramFactory::HashType programHash = 0;
|
||||
Uint32 uboBinding = 0;
|
||||
VkBuffer uboBuffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize uboRange = 0;
|
||||
VkDescriptorSet set = VK_NULL_HANDLE;
|
||||
};
|
||||
FastRebindMemo m_fastRebindMemo;
|
||||
|
||||
// vkCmdBindDescriptorSets dedup: consecutive draws with a static uniform
|
||||
// block resolve to the same set AND the same dynamic offsets, so the
|
||||
// driver call can be skipped outright. Command-buffer-scope state; reset
|
||||
// via OnCommandBufferBoundary whenever a recording (re)begins. Keyed on
|
||||
// layout+bind point, so a pipeline-layout switch always rebinds.
|
||||
static constexpr Uint32 kMaxShadowedDynamicOffsets = 8;
|
||||
Bool m_lastBindValid = false;
|
||||
VkDescriptorSet m_lastBindSet = VK_NULL_HANDLE;
|
||||
VkPipelineLayout m_lastBindLayout = VK_NULL_HANDLE;
|
||||
VkPipelineBindPoint m_lastBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
|
||||
Uint32 m_lastBindOffsetCount = 0;
|
||||
Uint32 m_lastBindOffsets[kMaxShadowedDynamicOffsets] = {};
|
||||
|
||||
// Global-UBO transient-slice reuse: MC leaves the default uniform block
|
||||
// untouched across long GUI/terrain runs, so the per-draw re-upload of
|
||||
// the same bytes can reuse the slice uploaded earlier THIS frame (frame
|
||||
// serial guards arena recycling; the content version guards writes).
|
||||
struct GlobalUboSliceMemo {
|
||||
Uint64 programLifetimeId = 0;
|
||||
Uint64 frameSerial = 0;
|
||||
Uint32 uboContentVersion = 0;
|
||||
VkBuffer buffer = VK_NULL_HANDLE;
|
||||
VkDeviceSize offset = 0;
|
||||
VkDeviceSize range = 0;
|
||||
};
|
||||
static constexpr Uint32 kGlobalUboMemoSize = 4;
|
||||
GlobalUboSliceMemo m_globalUboMemo[kGlobalUboMemoSize];
|
||||
Uint32 m_globalUboMemoNext = 0;
|
||||
|
||||
// Per-binding fast path over VkSamplerManager's content-hashed sampler cache, which
|
||||
// stays the source of truth: its key hashes all sampler+texture state, so two distinct
|
||||
// sampler objects with identical state still resolve to one VkSampler. This memo only
|
||||
// skips recomputing that hash. Across a draw batch the bound sampler set is stable, so a
|
||||
// binding whose sampler (lifetime id + version, bumped on every setter) and texture
|
||||
// (lifetime id + params version, bumped on the format/border-color setters that feed the
|
||||
// key) are unchanged recycles the VkSampler it resolved last draw; a param change bumps
|
||||
// a version and forces a re-resolve. Both objects are keyed by a never-reused monotonic
|
||||
// lifetime id, so a freed-and-reallocated sampler or texture at the same heap address
|
||||
// always gets a fresh id and misses (a raw pointer would false-hit that ABA) - so a
|
||||
// stale guess can only miss and fall through to the hash, never resolve wrong. Still
|
||||
// reset each frame alongside the descriptor-set cache. Indexed by binding.
|
||||
struct SamplerResolveMemo {
|
||||
Uint64 samplerLifetimeId = 0;
|
||||
Uint64 textureLifetimeId = 0;
|
||||
VkSampler sampler = VK_NULL_HANDLE;
|
||||
Uint32 viewLevelCount = 0;
|
||||
Uint16 samplerVersion = 0;
|
||||
Uint16 textureParamsVersion = 0;
|
||||
Bool forceNearestFiltering = false;
|
||||
Bool valid = false;
|
||||
// ResolveSampledImageViewFormat is pure in (image format, numeric domain), but a
|
||||
// domain mismatch walks a ~184-entry format table. Memo the resolution per binding
|
||||
// so a reinterpreted sampler pays that scan once, not once per draw.
|
||||
VkFormat viewFormatSource = VK_FORMAT_UNDEFINED;
|
||||
SamplerNumericDomain viewFormatDomain = SamplerNumericDomain::Unknown;
|
||||
VkFormat viewFormat = VK_FORMAT_UNDEFINED;
|
||||
Bool viewFormatValid = false;
|
||||
// Whole resolved descriptor from this binding's last full resolve. Reused
|
||||
// ONLY under ResolveSamplerDescriptor's trustUnchangedHint, whose caller
|
||||
// proves every resolve input unchanged; cleared with the per-frame reset
|
||||
// (the cached VkSampler outlives a frame only via a fresh resolve, which
|
||||
// also re-stamps it against VkSamplerManager's frame-boundary sweep).
|
||||
//
|
||||
// This one field is keyed by binding but describes ONE descriptor, so it is
|
||||
// written and read only for single-descriptor bindings. A sampler ARRAY's
|
||||
// elements share the binding and would overwrite each other here - the last
|
||||
// element resolved would then be handed to element 0 on the next hinted draw.
|
||||
// Every other field above is self-validating (each compares its full key
|
||||
// before reuse, and the view-format entry is a pure function of format and
|
||||
// numeric domain), so an arrayed binding may keep using those.
|
||||
VkDescriptorImageInfo info{};
|
||||
Bool infoValid = false;
|
||||
};
|
||||
mutable Vector<SamplerResolveMemo> m_samplerResolveMemo;
|
||||
// Exclusive upper bound on the entries of m_samplerResolveMemo that any resolve
|
||||
// has ever written. The vector is sized to the DEVICE binding cap (256 on desktop
|
||||
// NVIDIA), but a program declares 1-8 bindings, so the per-frame reset below was
|
||||
// memsetting ~22 KB of never-touched entries every frame - a measurable slice of
|
||||
// the per-frame fixed cost on draw-light frames. Every site that can turn any of
|
||||
// an entry's *Valid flags on raises this mark first, so entries at or above it are
|
||||
// provably still in their constructed (all-invalid) state and clearing them is a
|
||||
// no-op. Never lowered except by Initialize/Shutdown, which rebuild the vector.
|
||||
mutable Uint32 m_samplerResolveMemoHighWater = 0;
|
||||
void NoteSamplerResolveMemoTouched(Uint32 binding) const {
|
||||
if (binding >= m_samplerResolveMemoHighWater) {
|
||||
m_samplerResolveMemoHighWater = binding + 1;
|
||||
}
|
||||
}
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -29,96 +29,314 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.Stride, sizeof(attr.Stride)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.Offset, sizeof(attr.Offset)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.IsInteger, sizeof(attr.IsInteger)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.IsLong, sizeof(attr.IsLong)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.IsBgra, sizeof(attr.IsBgra)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attr.Divisor, sizeof(attr.Divisor)));
|
||||
|
||||
const SizeT bufferKey = reinterpret_cast<SizeT>(attr.Buffer.get());
|
||||
// The bound buffer's IDENTITY is a component of the key, and it has to be the
|
||||
// buffer's never-reused lifetime id - NOT its heap address, which this used to
|
||||
// hash. An address is recycled by the allocator, so a deleted-and-recreated
|
||||
// buffer reproduces it; combined with a byte-identical attribute layout that
|
||||
// reproduces the WHOLE content hash, and the hash is what
|
||||
// TryBindResolvedVertexBindings accepts as proof that a memoised binding still
|
||||
// reads the buffer it was resolved from. It did not: a destroyed buffer's GPU
|
||||
// slice was bound for its successor's draw, which is how a transform-feedback
|
||||
// capture came back holding a dead VAO's vertex data (0,0,0,1 - the previous
|
||||
// test's positions) instead of its own.
|
||||
// Zero for client memory (no buffer), which is a distinct identity of its own.
|
||||
const Uint64 bufferKey = attr.Buffer ? attr.Buffer->GetLifetimeId() : 0;
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &bufferKey, sizeof(bufferKey)));
|
||||
}
|
||||
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
VertexInputStateFactory::HashType VertexInputStateFactory::GetOrComputeHash(
|
||||
const MG_State::GLState::VertexArrayObject& vao) const {
|
||||
HashType hash = 0;
|
||||
if (!vao.GetBackendHashMemo(hash)) {
|
||||
hash = ComputeHash(vao);
|
||||
vao.SetBackendHashMemo(hash);
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
const VertexInputStateFactory::BackendVertexInputState& VertexInputStateFactory::GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao) {
|
||||
const HashType hash = ComputeHash(vao);
|
||||
// Per-draw fast path: the VAO carries a pointer to its resolved entry,
|
||||
// valid while its config version and the cache's eviction epoch both
|
||||
// match - no re-hash, no map lookup.
|
||||
const void* memoState = nullptr;
|
||||
Uint64 memoEpoch = 0;
|
||||
if (vao.GetBackendStateMemo(memoState, memoEpoch) && memoEpoch == m_evictionEpoch) {
|
||||
const auto* entry = static_cast<const BackendVertexInputState*>(memoState);
|
||||
entry->lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return *entry;
|
||||
}
|
||||
const BackendVertexInputState& entry = GetOrCreateVertexInputState(vao, GetOrComputeHash(vao));
|
||||
vao.SetBackendStateMemo(&entry, m_evictionEpoch);
|
||||
// Also mirror the layout identity and the two per-draw masks into the VAO's aux
|
||||
// memo (pure VALUES derived from the VAO configuration, so config-version
|
||||
// guarding alone is sound). The draw fast path reads them from the VAO object it
|
||||
// already touched instead of chasing into this entry - see PackVertexInputAuxMemo.
|
||||
vao.SetBackendAuxMemo(entry.layoutHash,
|
||||
PackVertexInputAuxMasks(entry.unsupportedAttribMask, entry.attributeLocationMask));
|
||||
return entry;
|
||||
}
|
||||
|
||||
const VertexInputStateFactory::BackendVertexInputState& VertexInputStateFactory::GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash) {
|
||||
auto it = m_cache.find(hash);
|
||||
if (it != m_cache.end()) {
|
||||
return it->second;
|
||||
it->second->lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return *it->second;
|
||||
}
|
||||
|
||||
VertexInputStateBuilder builder;
|
||||
UnorderedMap<SizeT, Uint32> bindingByBufferKey;
|
||||
UnorderedMap<SizeT, Uint32> strideByBufferKey;
|
||||
UnorderedMap<SizeT, VkVertexInputRate> inputRateByBufferKey;
|
||||
Vector<SizeT> bindingBufferKeys;
|
||||
Vector<SizeT> bindingBaseOffsets;
|
||||
Vector<Uint32> bindingAttributeLocations;
|
||||
Vector<Bool> bindingUsesClientMemory;
|
||||
Vector<VertexStreamConversion> bindingConversions;
|
||||
Vector<VkVertexInputBindingDivisorDescriptionEXT> bindingDivisors;
|
||||
Uint32 unsupportedAttribMask = 0;
|
||||
|
||||
for (Uint32 location = 0; location < MG_State::GLState::VertexArrayObject::MAX_VERTEX_ATTRIBS; ++location) {
|
||||
const auto& attr = vao.GetAttribute(location);
|
||||
if (!attr.Enabled || !attr.Buffer) {
|
||||
if (!attr.Enabled) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto vkFormat = ToVkVertexFormat(attr.Type, attr.Size, attr.Normalized, attr.IsInteger);
|
||||
if (vkFormat == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_D("Skipping unsupported vertex attribute layout (location=%u, type=%s, size=%d)",
|
||||
const VkFormat sourceVkFormat =
|
||||
ToVkVertexFormat(attr.Type, attr.Size, attr.Normalized, attr.IsInteger, attr.IsBgra, attr.IsLong);
|
||||
if (sourceVkFormat == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_E_ONCE("Unsupported vertex attribute layout (location=%u, type=%s, size=%d): the array is "
|
||||
"enabled but cannot be mapped to a VkFormat",
|
||||
location, MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size);
|
||||
unsupportedAttribMask |= (1u << location);
|
||||
continue;
|
||||
}
|
||||
|
||||
const SizeT componentSize = GetComponentSize(attr.Type);
|
||||
if (componentSize == 0) {
|
||||
MGLOG_D("Skipping vertex attribute with unknown component size (location=%u, type=%s)",
|
||||
VkFormat vkFormat = sourceVkFormat;
|
||||
VertexStreamConversion conversion = VertexStreamConversion::None;
|
||||
if (!SupportsVertexBufferFormat(vkFormat)) {
|
||||
if (IsScaledIntegerVertexFormat(vkFormat)) {
|
||||
const VkFormat fallbackFormat = ToFloat32VertexFormat(attr.Size);
|
||||
if (fallbackFormat != VK_FORMAT_UNDEFINED && SupportsVertexBufferFormat(fallbackFormat)) {
|
||||
vkFormat = fallbackFormat;
|
||||
conversion = VertexStreamConversion::ScaledIntegerToFloat32;
|
||||
MGLOG_W_ONCE("Vertex attribute location=%u format=%d lacks "
|
||||
"VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT; using float32 stream format=%d "
|
||||
"(type=%s size=%d normalized=%s integer=%s)",
|
||||
location, static_cast<Int>(sourceVkFormat), static_cast<Int>(vkFormat),
|
||||
MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size,
|
||||
attr.Normalized ? "true" : "false", attr.IsInteger ? "true" : "false");
|
||||
}
|
||||
}
|
||||
|
||||
if (conversion == VertexStreamConversion::None) {
|
||||
MGLOG_E_ONCE("Unsupported Vulkan vertex format (location=%u, format=%d, type=%s, size=%d): "
|
||||
"VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT is unavailable and no semantic fallback exists",
|
||||
location, static_cast<Int>(sourceVkFormat),
|
||||
MG_Util::ConvertDataTypeToString(attr.Type).c_str(), attr.Size);
|
||||
unsupportedAttribMask |= (1u << location);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
const SizeT attribByteSize = GetAttributeByteSize(attr.Type, attr.Size, attr.IsBgra);
|
||||
if (attribByteSize == 0) {
|
||||
MGLOG_E_ONCE("Vertex attribute with unknown component size (location=%u, type=%s): the array is "
|
||||
"enabled but cannot be sized",
|
||||
location, MG_Util::ConvertDataTypeToString(attr.Type).c_str());
|
||||
unsupportedAttribMask |= (1u << location);
|
||||
continue;
|
||||
}
|
||||
|
||||
const Uint32 stride = attr.Stride > 0
|
||||
? static_cast<Uint32>(attr.Stride)
|
||||
: static_cast<Uint32>(componentSize * static_cast<SizeT>(attr.Size));
|
||||
// Verbatim, zero included. The frontend already resolved a pointer call's
|
||||
// "tightly packed" stride 0 into the element size (see VertexAttribute::Stride),
|
||||
// so a zero here is the binding model's stride 0 - every vertex reads the same
|
||||
// element - which is exactly what a zero VkVertexInputBindingDescription::stride
|
||||
// means. Substituting the element size fetched a fresh element per vertex and ran
|
||||
// off the end of the buffer (KHR-GL43.vertex_attrib_binding.basic-input-case7/8).
|
||||
// Client-memory arrays cannot reach zero: they only exist on the pointer path.
|
||||
const Uint32 sourceStride = static_cast<Uint32>(attr.Stride);
|
||||
const Bool packedAttribute = attr.Type == DataType::Int2101010Rev ||
|
||||
attr.Type == DataType::Uint2101010Rev;
|
||||
const SizeT requiredAlignment = packedAttribute ? attribByteSize : GetComponentSize(attr.Type);
|
||||
// For a client-memory array attr.Offset holds the raw client pointer, and the
|
||||
// draw path re-uploads the data to a 16-aligned transient slice with attribute
|
||||
// offset 0, so only the stride can violate Vulkan's fetch alignment there.
|
||||
const Bool clientMemoryAttribute = attr.Buffer == nullptr;
|
||||
if (conversion == VertexStreamConversion::None && requiredAlignment > 1 &&
|
||||
((sourceStride % requiredAlignment) != 0 ||
|
||||
(!clientMemoryAttribute && (attr.Offset % requiredAlignment) != 0))) {
|
||||
// GL accepts arbitrary byte strides and offsets. Core Vulkan vertex fetches do not
|
||||
// unless VK_EXT_legacy_vertex_attributes is available, so deinterleave this one
|
||||
// attribute into a tightly packed transient stream without changing its format.
|
||||
conversion = VertexStreamConversion::Repack;
|
||||
MGLOG_W_ONCE("Vertex attribute location=%u uses Vulkan-incompatible alignment "
|
||||
"(offset=%zu stride=%u required=%zu); using a tightly packed stream",
|
||||
location, attr.Offset, sourceStride, requiredAlignment);
|
||||
}
|
||||
|
||||
Uint32 stride = sourceStride;
|
||||
// A converted stream is tightly packed, so its stride is the converted element
|
||||
// size - unless the source stride is zero, which does not describe a packing at
|
||||
// all but "never advance". That survives the conversion unchanged: the draw path
|
||||
// converts exactly one element and every vertex reads it.
|
||||
if (sourceStride != 0) {
|
||||
if (conversion == VertexStreamConversion::Repack) {
|
||||
stride = static_cast<Uint32>(attribByteSize);
|
||||
} else if (conversion == VertexStreamConversion::ScaledIntegerToFloat32) {
|
||||
stride = static_cast<Uint32>(attr.Size * static_cast<Int>(sizeof(Float)));
|
||||
}
|
||||
}
|
||||
const VkVertexInputRate inputRate =
|
||||
(attr.Divisor == 0) ? VK_VERTEX_INPUT_RATE_VERTEX : VK_VERTEX_INPUT_RATE_INSTANCE;
|
||||
|
||||
const SizeT bufferKey = reinterpret_cast<SizeT>(attr.Buffer.get());
|
||||
Uint32 binding = 0;
|
||||
auto itBinding = bindingByBufferKey.find(bufferKey);
|
||||
if (itBinding == bindingByBufferKey.end()) {
|
||||
binding = static_cast<Uint32>(bindingByBufferKey.size());
|
||||
bindingByBufferKey.emplace(bufferKey, binding);
|
||||
strideByBufferKey.emplace(bufferKey, stride);
|
||||
inputRateByBufferKey.emplace(bufferKey, inputRate);
|
||||
bindingBufferKeys.push_back(bufferKey);
|
||||
builder.AddBinding(binding, stride, inputRate);
|
||||
} else {
|
||||
binding = itBinding->second;
|
||||
if (strideByBufferKey[bufferKey] != stride) {
|
||||
MGLOG_D("Skipping vertex attribute at location %u: stride mismatch (%u vs %u) on same buffer",
|
||||
location, stride, strideByBufferKey[bufferKey]);
|
||||
continue;
|
||||
}
|
||||
if (inputRateByBufferKey[bufferKey] != inputRate) {
|
||||
MGLOG_D("Skipping vertex attribute at location %u: input-rate mismatch on same buffer", location);
|
||||
continue;
|
||||
}
|
||||
const Uint32 binding = static_cast<Uint32>(bindingBufferKeys.size());
|
||||
bindingBufferKeys.push_back(bufferKey);
|
||||
bindingBaseOffsets.push_back(attr.Buffer ? attr.Offset : 0);
|
||||
bindingAttributeLocations.push_back(location);
|
||||
bindingUsesClientMemory.push_back(attr.Buffer == nullptr);
|
||||
bindingConversions.push_back(conversion);
|
||||
builder.AddBinding(binding, stride, inputRate);
|
||||
builder.AddAttribute(location, binding, vkFormat, 0);
|
||||
// Divisor 1 is what VK_VERTEX_INPUT_RATE_INSTANCE already means; only anything
|
||||
// else needs the extension to say it.
|
||||
if (inputRate == VK_VERTEX_INPUT_RATE_INSTANCE && attr.Divisor != 1) {
|
||||
bindingDivisors.push_back({binding, static_cast<Uint32>(attr.Divisor)});
|
||||
}
|
||||
|
||||
builder.AddAttribute(location, binding, vkFormat, static_cast<Uint32>(attr.Offset));
|
||||
}
|
||||
|
||||
const auto& state = builder.Build();
|
||||
|
||||
auto& entry = m_cache[hash];
|
||||
auto& slot = m_cache[hash];
|
||||
if (!slot) {
|
||||
slot = MakeUnique<BackendVertexInputState>();
|
||||
}
|
||||
BackendVertexInputState& entry = *slot;
|
||||
entry.hash = hash;
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
entry.bindingDivisors = Move(bindingDivisors);
|
||||
entry.bindings = builder.GetBindings();
|
||||
entry.attributes = builder.GetAttributes();
|
||||
// See the layoutHash declaration: hash only the resolved layout, never
|
||||
// buffer identities, so identical layouts across VAOs/buffers agree.
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, 0));
|
||||
for (const auto& binding : entry.bindings) {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &binding.binding, sizeof(binding.binding)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &binding.stride, sizeof(binding.stride)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &binding.inputRate, sizeof(binding.inputRate)));
|
||||
}
|
||||
for (const auto& attribute : entry.attributes) {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.location, sizeof(attribute.location)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.binding, sizeof(attribute.binding)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.format, sizeof(attribute.format)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &attribute.offset, sizeof(attribute.offset)));
|
||||
}
|
||||
for (const auto& divisor : entry.bindingDivisors) {
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &divisor.binding, sizeof(divisor.binding)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &divisor.divisor, sizeof(divisor.divisor)));
|
||||
}
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &unsupportedAttribMask, sizeof(unsupportedAttribMask)));
|
||||
entry.layoutHash = XXH64_digest(m_hashState);
|
||||
entry.attributeLocationMask = 0;
|
||||
for (const auto& attribute : entry.attributes) {
|
||||
if (attribute.location < 32u) {
|
||||
entry.attributeLocationMask |= (1u << attribute.location);
|
||||
}
|
||||
}
|
||||
entry.bindingBufferKeys = std::move(bindingBufferKeys);
|
||||
entry.bindingBaseOffsets = std::move(bindingBaseOffsets);
|
||||
entry.bindingAttributeLocations = std::move(bindingAttributeLocations);
|
||||
entry.bindingUsesClientMemory = std::move(bindingUsesClientMemory);
|
||||
entry.bindingConversions = std::move(bindingConversions);
|
||||
entry.unsupportedAttribMask = unsupportedAttribMask;
|
||||
entry.state = state;
|
||||
entry.state.pVertexBindingDescriptions = entry.bindings.empty() ? nullptr : entry.bindings.data();
|
||||
entry.state.pVertexAttributeDescriptions = entry.attributes.empty() ? nullptr : entry.attributes.data();
|
||||
if (!entry.bindingDivisors.empty()) {
|
||||
entry.divisorState.vertexBindingDivisorCount = static_cast<Uint32>(entry.bindingDivisors.size());
|
||||
entry.divisorState.pVertexBindingDivisors = entry.bindingDivisors.data();
|
||||
entry.state.pNext = &entry.divisorState;
|
||||
} else {
|
||||
entry.state.pNext = nullptr;
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
VkFormat VertexInputStateFactory::ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger) {
|
||||
void VertexInputStateFactory::OnFrameBoundary() {
|
||||
++m_frameBoundaryCounter;
|
||||
|
||||
// Sweep occasionally; evict entries whose last hit is far in the past.
|
||||
// Erasure happens only here, never mid-frame: the draw path holds a
|
||||
// reference into the current entry across its setup, and unordered_map
|
||||
// erase would invalidate it. Entries are CPU-side only, so no GPU-idle
|
||||
// proof is needed; an evicted entry that is used again is simply rebuilt
|
||||
// from the VAO state (same hash, same content).
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
if ((m_frameBoundaryCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_cache.begin(); it != m_cache.end();) {
|
||||
if (m_frameBoundaryCounter - it->second->lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
it = m_cache.erase(it);
|
||||
// Invalidate every VAO's state-pointer memo: the erased node's
|
||||
// address may be reused by a future insert.
|
||||
++m_evictionEpoch;
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VertexInputStateFactory::ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger,
|
||||
Bool isBgra, Bool isLong) {
|
||||
if (isBgra) {
|
||||
// GL_BGRA: four reversed-order components, always normalized (enforced at validation), only
|
||||
// legal with GL_UNSIGNED_BYTE or a 2_10_10_10 type. The reversed VkFormats put the
|
||||
// components back into R,G,B,A order for the shader.
|
||||
switch (type) {
|
||||
case DataType::Uint8:
|
||||
return VK_FORMAT_B8G8R8A8_UNORM;
|
||||
case DataType::Uint2101010Rev:
|
||||
return VK_FORMAT_A2R10G10B10_UNORM_PACK32;
|
||||
case DataType::Int2101010Rev:
|
||||
return VK_FORMAT_A2R10G10B10_SNORM_PACK32;
|
||||
default:
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
switch (type) {
|
||||
case DataType::Uint2101010Rev:
|
||||
// Packed 2_10_10_10 travels the float-normalizing path only; size is always 4. SNORM/UNORM
|
||||
// normalize, SSCALED/USCALED cast the packed field to float.
|
||||
if (isInteger || size != 4) return VK_FORMAT_UNDEFINED;
|
||||
return normalized ? VK_FORMAT_A2B10G10R10_UNORM_PACK32 : VK_FORMAT_A2B10G10R10_USCALED_PACK32;
|
||||
case DataType::Int2101010Rev:
|
||||
if (isInteger || size != 4) return VK_FORMAT_UNDEFINED;
|
||||
return normalized ? VK_FORMAT_A2B10G10R10_SNORM_PACK32 : VK_FORMAT_A2B10G10R10_SSCALED_PACK32;
|
||||
case DataType::Float64:
|
||||
// A 64-bit attribute is fetched as its 32-bit word pair and bitcast back to double in the
|
||||
// shader (PackDoubleVertexInputsPass does the shader half). That is bit-exact and, unlike
|
||||
// VK_FORMAT_R64*_SFLOAT, needs no format capability: lavapipe reports bufferFeatures = 0
|
||||
// for every R64 float format, so a native 64-bit vertex fetch is simply unavailable there
|
||||
// while shaderFloat64 is not. Both halves key off nothing but the attribute being long,
|
||||
// so they always agree without extra plumbing.
|
||||
if (!isLong || isInteger || normalized) return VK_FORMAT_UNDEFINED;
|
||||
switch (size) {
|
||||
case 1: return VK_FORMAT_R32G32_UINT;
|
||||
case 2: return VK_FORMAT_R32G32B32A32_UINT;
|
||||
// A dvec3/dvec4 input is 6/8 uint32 components: no single VkFormat, and GL spreads it
|
||||
// over two attribute locations, which the location-per-VAO-index model here does not
|
||||
// express. Declined rather than fetched wrong.
|
||||
default: return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
case DataType::Float32:
|
||||
switch (size) {
|
||||
case 1: return VK_FORMAT_R32_SFLOAT;
|
||||
@@ -127,6 +345,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case 4: return VK_FORMAT_R32G32B32A32_SFLOAT;
|
||||
default: return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
case DataType::Float16:
|
||||
// GL_HALF_FLOAT is a floating-point array type: it is never an integer attribute, and
|
||||
// GL_TRUE for `normalized` is ignored for float types rather than selecting a *NORM format.
|
||||
if (isInteger) return VK_FORMAT_UNDEFINED;
|
||||
switch (size) {
|
||||
case 1: return VK_FORMAT_R16_SFLOAT;
|
||||
case 2: return VK_FORMAT_R16G16_SFLOAT;
|
||||
case 3: return VK_FORMAT_R16G16B16_SFLOAT;
|
||||
case 4: return VK_FORMAT_R16G16B16A16_SFLOAT;
|
||||
default: return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
case DataType::Int32:
|
||||
if (!isInteger || normalized) return VK_FORMAT_UNDEFINED;
|
||||
switch (size) {
|
||||
@@ -148,8 +377,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case DataType::Int16:
|
||||
switch (size) {
|
||||
case 1:
|
||||
return isInteger ? VK_FORMAT_R16_SINT
|
||||
: (normalized ? VK_FORMAT_R16_SNORM : VK_FORMAT_R16_SSCALED);
|
||||
return isInteger ? VK_FORMAT_R16_SINT : (normalized ? VK_FORMAT_R16_SNORM : VK_FORMAT_R16_SSCALED);
|
||||
case 2:
|
||||
return isInteger ? VK_FORMAT_R16G16_SINT
|
||||
: (normalized ? VK_FORMAT_R16G16_SNORM : VK_FORMAT_R16G16_SSCALED);
|
||||
@@ -164,8 +392,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case DataType::Uint16:
|
||||
switch (size) {
|
||||
case 1:
|
||||
return isInteger ? VK_FORMAT_R16_UINT
|
||||
: (normalized ? VK_FORMAT_R16_UNORM : VK_FORMAT_R16_USCALED);
|
||||
return isInteger ? VK_FORMAT_R16_UINT : (normalized ? VK_FORMAT_R16_UNORM : VK_FORMAT_R16_USCALED);
|
||||
case 2:
|
||||
return isInteger ? VK_FORMAT_R16G16_UINT
|
||||
: (normalized ? VK_FORMAT_R16G16_UNORM : VK_FORMAT_R16G16_USCALED);
|
||||
@@ -180,8 +407,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case DataType::Int8:
|
||||
switch (size) {
|
||||
case 1:
|
||||
return isInteger ? VK_FORMAT_R8_SINT
|
||||
: (normalized ? VK_FORMAT_R8_SNORM : VK_FORMAT_R8_SSCALED);
|
||||
return isInteger ? VK_FORMAT_R8_SINT : (normalized ? VK_FORMAT_R8_SNORM : VK_FORMAT_R8_SSCALED);
|
||||
case 2:
|
||||
return isInteger ? VK_FORMAT_R8G8_SINT
|
||||
: (normalized ? VK_FORMAT_R8G8_SNORM : VK_FORMAT_R8G8_SSCALED);
|
||||
@@ -196,8 +422,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case DataType::Uint8:
|
||||
switch (size) {
|
||||
case 1:
|
||||
return isInteger ? VK_FORMAT_R8_UINT
|
||||
: (normalized ? VK_FORMAT_R8_UNORM : VK_FORMAT_R8_USCALED);
|
||||
return isInteger ? VK_FORMAT_R8_UINT : (normalized ? VK_FORMAT_R8_UNORM : VK_FORMAT_R8_USCALED);
|
||||
case 2:
|
||||
return isInteger ? VK_FORMAT_R8G8_UINT
|
||||
: (normalized ? VK_FORMAT_R8G8_UNORM : VK_FORMAT_R8G8_USCALED);
|
||||
@@ -234,4 +459,57 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
SizeT VertexInputStateFactory::GetAttributeByteSize(DataType type, Int size, Bool isBgra) {
|
||||
// The packed 2_10_10_10 types are a single 32-bit word for all 4 components; GL_BGRA is always
|
||||
// 4 components (GL_UNSIGNED_BYTE x4 = 4 bytes, or a packed word = 4 bytes) -- both are 4 bytes.
|
||||
if (type == DataType::Int2101010Rev || type == DataType::Uint2101010Rev || isBgra) {
|
||||
return 4;
|
||||
}
|
||||
const SizeT componentSize = GetComponentSize(type);
|
||||
return componentSize == 0 ? 0 : componentSize * static_cast<SizeT>(size);
|
||||
}
|
||||
|
||||
Bool VertexInputStateFactory::IsScaledIntegerVertexFormat(VkFormat format) {
|
||||
switch (format) {
|
||||
case VK_FORMAT_R8_USCALED:
|
||||
case VK_FORMAT_R8_SSCALED:
|
||||
case VK_FORMAT_R8G8_USCALED:
|
||||
case VK_FORMAT_R8G8_SSCALED:
|
||||
case VK_FORMAT_R8G8B8_USCALED:
|
||||
case VK_FORMAT_R8G8B8_SSCALED:
|
||||
case VK_FORMAT_R8G8B8A8_USCALED:
|
||||
case VK_FORMAT_R8G8B8A8_SSCALED:
|
||||
case VK_FORMAT_R16_USCALED:
|
||||
case VK_FORMAT_R16_SSCALED:
|
||||
case VK_FORMAT_R16G16_USCALED:
|
||||
case VK_FORMAT_R16G16_SSCALED:
|
||||
case VK_FORMAT_R16G16B16_USCALED:
|
||||
case VK_FORMAT_R16G16B16_SSCALED:
|
||||
case VK_FORMAT_R16G16B16A16_USCALED:
|
||||
case VK_FORMAT_R16G16B16A16_SSCALED:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VertexInputStateFactory::ToFloat32VertexFormat(Int componentCount) {
|
||||
switch (componentCount) {
|
||||
case 1: return VK_FORMAT_R32_SFLOAT;
|
||||
case 2: return VK_FORMAT_R32G32_SFLOAT;
|
||||
case 3: return VK_FORMAT_R32G32B32_SFLOAT;
|
||||
case 4: return VK_FORMAT_R32G32B32A32_SFLOAT;
|
||||
default: return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
Bool VertexInputStateFactory::SupportsVertexBufferFormat(VkFormat format) const {
|
||||
if (m_physicalDevice == VK_NULL_HANDLE || format == VK_FORMAT_UNDEFINED) {
|
||||
return false;
|
||||
}
|
||||
VkFormatProperties properties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &properties);
|
||||
return (properties.bufferFeatures & VK_FORMAT_FEATURE_VERTEX_BUFFER_BIT) != 0;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -19,30 +19,113 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
public:
|
||||
using HashType = Uint64;
|
||||
|
||||
enum class VertexStreamConversion : Uint8 {
|
||||
None = 0,
|
||||
Repack,
|
||||
ScaledIntegerToFloat32,
|
||||
};
|
||||
|
||||
struct BackendVertexInputState {
|
||||
HashType hash = 0;
|
||||
// Hash of the resolved Vulkan vertex layout only (bindings, attributes,
|
||||
// unsupported mask) - NO buffer identities. `hash` mixes each bound
|
||||
// buffer's never-reused LIFETIME ID, so per-chunk VBOs mint a fresh
|
||||
// identity per buffer; keying pipelines on that minted one VkPipeline per
|
||||
// chunk section for an identical layout, defeating pipeline reuse and the
|
||||
// per-draw memo. Pipelines depend only on the layout, so they key on this
|
||||
// instead.
|
||||
HashType layoutHash = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age are evicted (CPU heap only).
|
||||
// Mutable: the VAO's state-pointer memo fast path stamps it through
|
||||
// a const entry reference.
|
||||
mutable Uint64 lastUsedFrameBoundary = 0;
|
||||
Vector<VkVertexInputBindingDescription> bindings;
|
||||
Vector<VkVertexInputAttributeDescription> attributes;
|
||||
Vector<SizeT> bindingBufferKeys;
|
||||
Vector<SizeT> bindingBaseOffsets;
|
||||
Vector<Uint32> bindingAttributeLocations;
|
||||
Vector<Bool> bindingUsesClientMemory;
|
||||
Vector<VertexStreamConversion> bindingConversions;
|
||||
// Locations whose array is ENABLED but whose GL format has no VkFormat mapping. They are
|
||||
// absent from `attributes`, so without this mask the draw path cannot tell them apart from
|
||||
// a genuinely disabled array and would silently feed the shader the current attribute value.
|
||||
Uint32 unsupportedAttribMask = 0;
|
||||
// Bitmask of `attributes[i].location` - the draw path needs it up to
|
||||
// three times per draw, so it is baked once at build time.
|
||||
Uint32 attributeLocationMask = 0;
|
||||
// Per-binding glVertexAttribDivisor values other than 1. Vulkan's instance input
|
||||
// rate advances once per instance and nothing else, so anything else has to be
|
||||
// stated through VK_EXT_vertex_attribute_divisor. Empty when every instanced
|
||||
// binding uses divisor 1, which is what the plain input rate already means.
|
||||
Vector<VkVertexInputBindingDivisorDescriptionEXT> bindingDivisors;
|
||||
VkPipelineVertexInputDivisorStateCreateInfoEXT divisorState{
|
||||
VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_DIVISOR_STATE_CREATE_INFO_EXT
|
||||
};
|
||||
VkPipelineVertexInputStateCreateInfo state{
|
||||
VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO
|
||||
};
|
||||
};
|
||||
|
||||
explicit VertexInputStateFactory(const VulkanRendererConfig& config):
|
||||
m_config(config) {}
|
||||
VertexInputStateFactory(const VulkanRendererConfig& config, VkPhysicalDevice physicalDevice):
|
||||
m_config(config), m_physicalDevice(physicalDevice) {}
|
||||
~VertexInputStateFactory() = default;
|
||||
VertexInputStateFactory(const VertexInputStateFactory&) = delete;
|
||||
|
||||
// The VAO aux-memo payload GetOrCreateVertexInputState(vao) stamps: aux0 is the
|
||||
// entry's layoutHash, aux1 packs (unsupportedAttribMask << 32) | attributeLocationMask.
|
||||
// Readers that find the aux memo valid can use these without resolving the entry.
|
||||
static Uint64 PackVertexInputAuxMasks(Uint32 unsupportedAttribMask, Uint32 attributeLocationMask) {
|
||||
return (static_cast<Uint64>(unsupportedAttribMask) << 32) | attributeLocationMask;
|
||||
}
|
||||
|
||||
HashType ComputeHash(const MG_State::GLState::VertexArrayObject& vao) const;
|
||||
// Memoized ComputeHash: reuses the VAO's cached hash while its config version
|
||||
// is unchanged. Use this on per-draw paths.
|
||||
HashType GetOrComputeHash(const MG_State::GLState::VertexArrayObject& vao) const;
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(
|
||||
const MG_State::GLState::VertexArrayObject& vao, HashType hash);
|
||||
const BackendVertexInputState& GetOrCreateVertexInputState(const MG_State::GLState::VertexArrayObject& vao);
|
||||
// Frame boundary hook: ages the cache and evicts entries not hit for many
|
||||
// frames. The key mixes each bound buffer's never-reused lifetime id, so
|
||||
// buffer/VAO churn keeps minting fresh keys - and does so by construction,
|
||||
// not by luck: a recreated buffer can no longer land back on its dead
|
||||
// predecessor's key. Without eviction the map grows for the whole session.
|
||||
// Entries hold no Vulkan handles (pipeline creation copies the descriptions)
|
||||
// and the draw path's entry reference never spans a frame boundary, so
|
||||
// eviction here needs no GPU-idle proof. Self-gated: one counter bump and
|
||||
// compare except on sweep boundaries.
|
||||
void OnFrameBoundary();
|
||||
static SizeT GetComponentSize(DataType type);
|
||||
// Tightly-packed byte size of one vertex element for this attribute: componentSize * size for
|
||||
// normal types, and 4 (one packed word) for the 2_10_10_10 types and GL_BGRA. Returns 0 for
|
||||
// an unknown/unsupported type.
|
||||
static SizeT GetAttributeByteSize(DataType type, Int size, Bool isBgra);
|
||||
|
||||
private:
|
||||
static VkFormat ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger);
|
||||
static SizeT GetComponentSize(DataType type);
|
||||
static VkFormat ToVkVertexFormat(DataType type, Int size, Bool normalized, Bool isInteger, Bool isBgra = false,
|
||||
Bool isLong = false);
|
||||
static Bool IsScaledIntegerVertexFormat(VkFormat format);
|
||||
static VkFormat ToFloat32VertexFormat(Int componentCount);
|
||||
Bool SupportsVertexBufferFormat(VkFormat format) const;
|
||||
|
||||
const VulkanRendererConfig& m_config;
|
||||
UnorderedMap<HashType, BackendVertexInputState> m_cache;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
// Values are heap-allocated: UnorderedMap is open-addressing, so INSERT
|
||||
// invalidates references to stored values - and so does ERASE, which shifts
|
||||
// the rest of the probe cluster into the hole and therefore moves entries
|
||||
// other than the erased one. The draw path (and the VAOs' state-pointer
|
||||
// memos) hold entry pointers across both; only the unique_ptr cell moves,
|
||||
// never the pointee.
|
||||
UnorderedMap<HashType, UniquePtr<BackendVertexInputState>> m_cache;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
// Bumped whenever any cache entry is erased. VAOs memo a raw pointer to
|
||||
// their heap-allocated entry (stable across map insert/rehash by
|
||||
// construction); a memo is honored only while its recorded epoch
|
||||
// matches, so an evicted entry can never be dereferenced through a
|
||||
// stale memo.
|
||||
Uint64 m_evictionEpoch = 1;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -0,0 +1,753 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkBufferManager.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "VkBufferManager.h"
|
||||
#include "../DirectVulkan.h"
|
||||
#include "VulkanRenderer.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
constexpr VmaAllocationCreateFlags kResidentBufferAllocationFlags =
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT;
|
||||
constexpr SizeT kLiveResourcePruneThreshold = 256;
|
||||
|
||||
// A zero-copy persistent buffer is created once and never recreated (the app holds
|
||||
// its mapped pointer), and may be bound to any role, so it carries every usage.
|
||||
// TRANSFER_DST is added by CreateResidentStorage.
|
||||
constexpr VkBufferUsageFlags kPersistentBackedUsage =
|
||||
VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT | VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT |
|
||||
// "Every usage" has to mean every usage: a buffer texture reached through an IMAGE
|
||||
// unit takes a VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER descriptor, and the write is
|
||||
// invalid unless the buffer was created with this bit. Nothing asked for it until
|
||||
// imageBuffer support existed, so the omission was invisible.
|
||||
VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
|
||||
// Appended to kPersistentBackedUsage when VK_EXT_transform_feedback is enabled
|
||||
// (see VkBufferManagerInitInfo::transformFeedbackUsageEnabled).
|
||||
constexpr VkBufferUsageFlags kTransformFeedbackUsage =
|
||||
VK_BUFFER_USAGE_TRANSFORM_FEEDBACK_BUFFER_BIT_EXT;
|
||||
// The app writes into the persistent map with no explicit flush, so its memory must
|
||||
// be host-coherent (Adreno host-visible memory is; requiring it keeps us portable).
|
||||
constexpr VkMemoryPropertyFlags kPersistentBackedRequiredFlags =
|
||||
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
|
||||
using MG_State::GLState::BackendBufferResource;
|
||||
using MG_State::GLState::BufferBackendOps;
|
||||
using MG_State::GLState::BufferObject;
|
||||
|
||||
// The manager owned by the active VulkanRenderer; immediate ops route here.
|
||||
VkBufferManager* g_activeBufferManager = nullptr;
|
||||
|
||||
void Ops_Respecify(BufferObject& bufferObject) {
|
||||
if (g_activeBufferManager) {
|
||||
g_activeBufferManager->OnRespecify(bufferObject);
|
||||
}
|
||||
}
|
||||
|
||||
void Ops_SubData(BufferObject& bufferObject, SizeT offset, SizeT size) {
|
||||
if (g_activeBufferManager) {
|
||||
g_activeBufferManager->OnSubData(bufferObject, offset, size);
|
||||
}
|
||||
}
|
||||
|
||||
void Ops_FlushMappedRange(BufferObject& bufferObject, Range1D range,
|
||||
Flags<BufferMappingAccessBit> appAccess) {
|
||||
if (g_activeBufferManager) {
|
||||
g_activeBufferManager->OnFlushMappedRange(bufferObject, range, appAccess);
|
||||
}
|
||||
}
|
||||
|
||||
// The CPU is about to read a buffer a shader wrote. Its bytes live in coherent
|
||||
// host-visible GPU storage (EnsureGpuResidentStorage adopts it when the buffer is
|
||||
// bound as a shader storage buffer), so nothing needs copying - but coherence only
|
||||
// says the writes are visible once they have happened, so the work has to retire
|
||||
// first.
|
||||
void Ops_ReadbackFromGpu(BufferObject& bufferObject) {
|
||||
(void)bufferObject;
|
||||
if (pVulkanRenderer) {
|
||||
pVulkanRenderer->FinishPendingGpuWork();
|
||||
}
|
||||
}
|
||||
|
||||
void* Ops_AcquirePersistentMap(BufferObject& bufferObject) {
|
||||
if (g_activeBufferManager) {
|
||||
return g_activeBufferManager->AcquirePersistentMap(bufferObject);
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void Ops_OnDestroy(SharedPtr<BackendBufferResource>&& resource) {
|
||||
if (g_activeBufferManager) {
|
||||
g_activeBufferManager->OnResourceDestroyed(std::move(resource));
|
||||
}
|
||||
// No active manager: the device/allocator is gone or going away and
|
||||
// Shutdown() already destroyed the storage; dropping the handle here
|
||||
// must not touch Vulkan. VkBufferResource's dtor destroys via VMA only
|
||||
// when the allocation is still valid, which Shutdown() cleared.
|
||||
}
|
||||
|
||||
const BufferBackendOps g_vulkanBufferBackendOps = {
|
||||
.Respecify = Ops_Respecify,
|
||||
.SubData = Ops_SubData,
|
||||
.FlushMappedRange = Ops_FlushMappedRange,
|
||||
.OnDestroy = Ops_OnDestroy,
|
||||
.AcquirePersistentMap = Ops_AcquirePersistentMap,
|
||||
.ReadbackFromGpu = Ops_ReadbackFromGpu,
|
||||
};
|
||||
} // namespace
|
||||
|
||||
Bool VkBufferManager::Initialize(const VkBufferManagerInitInfo& initInfo) {
|
||||
Shutdown();
|
||||
|
||||
MOBILEGL_ASSERT(initInfo.allocator != nullptr, "VkBufferManager::Initialize requires valid allocator");
|
||||
MOBILEGL_ASSERT(initInfo.frameCount > 0, "VkBufferManager::Initialize requires non-zero frame count");
|
||||
|
||||
m_initInfo = initInfo;
|
||||
m_deferredBufferReleases.resize(initInfo.frameCount);
|
||||
m_deferredResourceReleases.resize(initInfo.frameCount);
|
||||
m_currentFrameIndex = 0;
|
||||
m_frameSerial = 1;
|
||||
m_completedSerialFloor = 0;
|
||||
if (!InitializeTransientArenas()) {
|
||||
return false;
|
||||
}
|
||||
g_activeBufferManager = this;
|
||||
MG_State::GLState::SetBufferBackendOps(&g_vulkanBufferBackendOps);
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkBufferManager::Shutdown() {
|
||||
if (g_activeBufferManager == this) {
|
||||
g_activeBufferManager = nullptr;
|
||||
if (MG_State::GLState::GetBufferBackendOps() == &g_vulkanBufferBackendOps) {
|
||||
MG_State::GLState::SetBufferBackendOps(nullptr);
|
||||
}
|
||||
}
|
||||
m_transientUploadArena.Shutdown();
|
||||
DestroyAllDeferredReleases();
|
||||
ReleaseAllLiveResources();
|
||||
m_copyProvider = nullptr;
|
||||
m_initInfo = {};
|
||||
m_currentFrameIndex = 0;
|
||||
m_frameSerial = 1;
|
||||
m_completedSerialFloor = 0;
|
||||
}
|
||||
|
||||
Bool VkBufferManager::RecreateTransientArenas(Uint32 frameCount) {
|
||||
MOBILEGL_ASSERT(m_initInfo.allocator != nullptr,
|
||||
"VkBufferManager::RecreateTransientArenas requires initialized manager");
|
||||
MOBILEGL_ASSERT(frameCount > 0, "VkBufferManager::RecreateTransientArenas requires non-zero frame count");
|
||||
|
||||
// Callers guarantee the device is idle around arena recreation.
|
||||
NotifyDeviceIdle();
|
||||
m_transientUploadArena.Shutdown();
|
||||
m_initInfo.frameCount = frameCount;
|
||||
DestroyAllDeferredReleases();
|
||||
m_deferredBufferReleases.resize(frameCount);
|
||||
m_deferredResourceReleases.resize(frameCount);
|
||||
m_currentFrameIndex = 0;
|
||||
return InitializeTransientArenas();
|
||||
}
|
||||
|
||||
void VkBufferManager::BeginFrame(Uint32 frameIndex) {
|
||||
MOBILEGL_ASSERT(frameIndex < m_deferredBufferReleases.size(),
|
||||
"VkBufferManager::BeginFrame frame index out of range");
|
||||
m_currentFrameIndex = frameIndex;
|
||||
++m_frameSerial;
|
||||
CollectDeferredReleases(frameIndex);
|
||||
m_transientUploadArena.BeginFrame(frameIndex);
|
||||
}
|
||||
|
||||
void VkBufferManager::CollectAllDeferredReleases() {
|
||||
// Per-resource releases only. Every one of them was deferred behind a BumpSliceEpoch,
|
||||
// so no memo can still name the handle, and the caller has proved the GPU is idle.
|
||||
//
|
||||
// The transient arena's releases are deliberately NOT collected here. A buffer lands
|
||||
// there when the arena outgrows it mid-frame (BufferArena::EnsureCapacity), and at
|
||||
// that moment every slice already handed out from this frame's arena still names it -
|
||||
// VkBufferResource::transientSlice above all, which AcquireStreamedSlice keeps
|
||||
// serving for the whole frame serial on the strength of transientFrameSerial alone.
|
||||
// Nothing bumps the slice epoch for those other resources, so freeing the buffer
|
||||
// here left the streamed memo handing a destroyed VkBuffer to vkCmdBindIndexBuffer
|
||||
// (llvmpipe then faulted inside the draw; the Create/Flywheel indirect retrace died
|
||||
// exactly this way). Mid-frame drains do not advance m_frameSerial, so they must not
|
||||
// free arena storage either: the arena's own ResetFrame/BeginFrame is the point where
|
||||
// the slot's slices stop being reachable, and that is where these releases land.
|
||||
for (Uint32 frameIndex = 0; frameIndex < m_deferredBufferReleases.size(); ++frameIndex) {
|
||||
CollectDeferredReleases(frameIndex);
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::NotifyDeviceIdle() {
|
||||
// Everything submitted so far has completed. Work recorded for the
|
||||
// current frame has not been submitted yet, so the current serial
|
||||
// remains busy.
|
||||
if (m_frameSerial > 0) {
|
||||
m_completedSerialFloor = m_frameSerial - 1;
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::NotifyFrameSerialComplete(Uint64 serial) {
|
||||
// The current serial's work is still being recorded; a completion
|
||||
// report for it (or beyond) can only come from a stale caller.
|
||||
if (serial >= m_frameSerial) {
|
||||
return;
|
||||
}
|
||||
m_completedSerialFloor = std::max(m_completedSerialFloor, serial);
|
||||
}
|
||||
|
||||
void VkBufferManager::SetCopyCommandProvider(IBufferCopyCommandProvider* provider) {
|
||||
m_copyProvider = provider;
|
||||
}
|
||||
|
||||
Uint64 VkBufferManager::GetCompletedSerial() const {
|
||||
const Uint64 frameCount = m_initInfo.frameCount > 0 ? m_initInfo.frameCount : 1;
|
||||
const Uint64 completed = m_frameSerial > frameCount ? m_frameSerial - frameCount : 0;
|
||||
return std::max(completed, m_completedSerialFloor);
|
||||
}
|
||||
|
||||
Bool VkBufferManager::IsResourceBusy(const VkBufferResource& resource) const {
|
||||
return resource.lastUseSerial > GetCompletedSerial();
|
||||
}
|
||||
|
||||
Bool VkBufferManager::UploadTransient(BufferKind kind, Uint32 frameIndex, const void* data,
|
||||
VkDeviceSize size, VkDeviceSize alignment, BufferSlice& outSlice) {
|
||||
(void)kind;
|
||||
return m_transientUploadArena.Upload(frameIndex, data, size, alignment, outSlice);
|
||||
}
|
||||
|
||||
Bool VkBufferManager::InitializeTransientArenas() {
|
||||
return m_transientUploadArena.Initialize({
|
||||
.allocator = m_initInfo.allocator,
|
||||
.frameCount = m_initInfo.frameCount,
|
||||
.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_SRC_BIT,
|
||||
.memoryUsage = m_initInfo.transientMemoryUsage,
|
||||
.allocationFlags = m_initInfo.transientAllocationFlags,
|
||||
.minBufferSize = m_initInfo.minUploadBytes,
|
||||
.persistentlyMapped = m_initInfo.transientPersistentMapping,
|
||||
});
|
||||
}
|
||||
|
||||
VkBufferResource* VkBufferManager::ResourceOf(MG_State::GLState::BufferObject& bufferObject) {
|
||||
return static_cast<VkBufferResource*>(bufferObject.GetBackendResource().get());
|
||||
}
|
||||
|
||||
VkBufferResource* VkBufferManager::GetOrCreateResource(
|
||||
const SharedPtr<MG_State::GLState::BufferObject>& bufferObject) {
|
||||
// Return by raw pointer: the resource is owned for its whole lifetime by the BufferObject's
|
||||
// backend-resource SharedPtr (already set, or set below), so callers that only dereference
|
||||
// it avoid a static_pointer_cast + SharedPtr refcount inc/dec on every per-draw buffer bind.
|
||||
const auto& existing = bufferObject->GetBackendResource();
|
||||
if (existing) {
|
||||
return static_cast<VkBufferResource*>(existing.get());
|
||||
}
|
||||
auto resource = MakeShared<VkBufferResource>();
|
||||
VkBufferResource* raw = resource.get();
|
||||
bufferObject->SetBackendResource(resource);
|
||||
TrackLiveResource(resource);
|
||||
return raw;
|
||||
}
|
||||
|
||||
void VkBufferManager::TrackLiveResource(const SharedPtr<VkBufferResource>& resource) {
|
||||
// Sweep on a doubling watermark rather than on every insert past the threshold. The old
|
||||
// form walked the whole vector for each new buffer once the list passed 256, and when the
|
||||
// buffers are all live the walk removes nothing and the list grows by one - so creating N
|
||||
// live buffers cost ~N^2/2 expired() checks. Reclamation semantics are unchanged: the sweep
|
||||
// still removes exactly the expired entries, just less often and with the same bound on how
|
||||
// much dead weight can accumulate (at most as many entries as were live at the last sweep).
|
||||
if (m_liveResources.size() >= std::max<SizeT>(kLiveResourcePruneThreshold, 2 * m_liveResourcesLastPruned)) {
|
||||
std::erase_if(m_liveResources, [](const WeakPtr<VkBufferResource>& weak) { return weak.expired(); });
|
||||
m_liveResourcesLastPruned = m_liveResources.size();
|
||||
}
|
||||
m_liveResources.push_back(resource);
|
||||
}
|
||||
|
||||
void VkBufferManager::ReleaseAllLiveResources() {
|
||||
for (auto& weak : m_liveResources) {
|
||||
if (auto resource = weak.lock()) {
|
||||
BumpSliceEpoch(*resource);
|
||||
resource->buffer.Destroy();
|
||||
resource->storageSize = 0;
|
||||
resource->usageFlags = 0;
|
||||
resource->lastUseSerial = 0;
|
||||
resource->pendingFullUpload = true;
|
||||
resource->transientSlice = {};
|
||||
resource->transientFrameSerial = 0;
|
||||
}
|
||||
}
|
||||
m_liveResources.clear();
|
||||
}
|
||||
|
||||
Bool VkBufferManager::CreateResidentStorage(VkBufferResource& resource, VkDeviceSize size,
|
||||
VkBufferUsageFlags usage, VkMemoryPropertyFlags requiredFlags) {
|
||||
// The only place a resident VkBuffer handle is minted, so every resident slice
|
||||
// change funnels through here (callers release the old handle first).
|
||||
BumpSliceEpoch(resource);
|
||||
// Staged range copies write resident storage with vkCmdCopyBuffer.
|
||||
usage |= VK_BUFFER_USAGE_TRANSFER_DST_BIT;
|
||||
const Bool created = resource.buffer.Create({
|
||||
.allocator = m_initInfo.allocator,
|
||||
.size = size,
|
||||
.usage = usage,
|
||||
.memoryUsage = VMA_MEMORY_USAGE_AUTO,
|
||||
.allocationFlags = kResidentBufferAllocationFlags,
|
||||
.requiredFlags = requiredFlags,
|
||||
});
|
||||
if (!created || resource.buffer.Map() == nullptr) {
|
||||
MGLOG_E_ONCE("VkBufferManager::CreateResidentStorage failed (size=%llu)",
|
||||
static_cast<unsigned long long>(size));
|
||||
resource.buffer.Destroy();
|
||||
resource.storageSize = 0;
|
||||
resource.usageFlags = 0;
|
||||
return false;
|
||||
}
|
||||
resource.storageSize = size;
|
||||
resource.usageFlags = usage;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkBufferManager::SwapStorageAndUploadAll(VkBufferResource& resource,
|
||||
MG_State::GLState::BufferObject& bufferObject) {
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject.GetSize());
|
||||
const VkBufferUsageFlags usage = resource.usageFlags;
|
||||
DeferRelease(std::move(resource.buffer));
|
||||
if (!CreateResidentStorage(resource, size, usage)) {
|
||||
resource.pendingFullUpload = true;
|
||||
return false;
|
||||
}
|
||||
if (!resource.buffer.Upload(bufferObject.MappedData(), size, 0)) {
|
||||
MGLOG_E_ONCE("VkBufferManager::SwapStorageAndUploadAll: upload failed");
|
||||
resource.pendingFullUpload = true;
|
||||
return false;
|
||||
}
|
||||
resource.pendingFullUpload = false;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkBufferManager::StagedRangeCopy(VkBufferResource& resource, MG_State::GLState::BufferObject& bufferObject,
|
||||
SizeT offset, SizeT size) {
|
||||
if (!m_copyProvider) {
|
||||
return false;
|
||||
}
|
||||
BufferSlice staging{};
|
||||
if (!m_transientUploadArena.Upload(m_currentFrameIndex, bufferObject.MappedData() + offset,
|
||||
static_cast<VkDeviceSize>(size), 16, staging)) {
|
||||
return false;
|
||||
}
|
||||
VkCommandBuffer commandBuffer = m_copyProvider->AcquireBufferCopyCommandBuffer();
|
||||
if (commandBuffer == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Order the copy after every prior read/write of this buffer, both from
|
||||
// in-flight frames (submission order) and from commands already recorded
|
||||
// in this frame's command buffer.
|
||||
VkMemoryBarrier beforeBarrier{VK_STRUCTURE_TYPE_MEMORY_BARRIER};
|
||||
beforeBarrier.srcAccessMask = VK_ACCESS_MEMORY_READ_BIT | VK_ACCESS_MEMORY_WRITE_BIT;
|
||||
beforeBarrier.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 1,
|
||||
&beforeBarrier, 0, nullptr, 0, nullptr);
|
||||
|
||||
VkBufferCopy region{};
|
||||
region.srcOffset = staging.offset;
|
||||
region.dstOffset = static_cast<VkDeviceSize>(offset);
|
||||
region.size = static_cast<VkDeviceSize>(size);
|
||||
vkCmdCopyBuffer(commandBuffer, staging.buffer, resource.buffer.GetHandle(), 1, ®ion);
|
||||
|
||||
VkMemoryBarrier afterBarrier{VK_STRUCTURE_TYPE_MEMORY_BARRIER};
|
||||
afterBarrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
afterBarrier.dstAccessMask = VK_ACCESS_MEMORY_READ_BIT | VK_ACCESS_MEMORY_WRITE_BIT;
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 1,
|
||||
&afterBarrier, 0, nullptr, 0, nullptr);
|
||||
|
||||
resource.lastUseSerial = m_frameSerial;
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkBufferManager::OnRespecify(MG_State::GLState::BufferObject& bufferObject) {
|
||||
auto* resource = ResourceOf(bufferObject);
|
||||
if (!resource) {
|
||||
return; // lazy: AcquireResidentSlice performs a full upload on creation
|
||||
}
|
||||
// A respecify can change the size, the usage hint (so the resident/streamed
|
||||
// route), and the contents at once; retire every memo before deciding what to
|
||||
// do about the storage.
|
||||
BumpSliceEpoch(*resource);
|
||||
// Any cached streaming slice refers to the previous contents.
|
||||
resource->transientFrameSerial = 0;
|
||||
// Redefining the store hands any adopted mapping back to the CPU shadow
|
||||
// (BufferObject::RedefineStorage), so a buffer that reaches here persistent-mapped
|
||||
// is an ordinary resident one again: it needs the busy-tracking and conditional
|
||||
// orphan below, and the next AcquirePersistentMap has to mint storage for the new
|
||||
// store rather than hand back a mapping of the old one.
|
||||
resource->persistentMapped = false;
|
||||
if (!resource->buffer.IsValid()) {
|
||||
return; // streaming-only resource: shadow + serial are enough
|
||||
}
|
||||
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject.GetSize());
|
||||
if (size == 0) {
|
||||
DeferRelease(std::move(resource->buffer));
|
||||
resource->storageSize = 0;
|
||||
resource->pendingFullUpload = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if (size != resource->storageSize || IsResourceBusy(*resource)) {
|
||||
// Conditional orphan: only swap the storage when the old one is
|
||||
// still referenced by the GPU (or no longer fits).
|
||||
SwapStorageAndUploadAll(*resource, bufferObject);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData(), size, 0)) {
|
||||
MGLOG_E_ONCE("VkBufferManager::OnRespecify: in-place upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::OnSubData(MG_State::GLState::BufferObject& bufferObject, SizeT offset, SizeT size) {
|
||||
auto* resource = ResourceOf(bufferObject);
|
||||
if (!resource) {
|
||||
return;
|
||||
}
|
||||
// Drops the streaming memo below and may end in a storage swap or a deferred
|
||||
// full re-upload, so no memoised slice survives this.
|
||||
BumpSliceEpoch(*resource);
|
||||
resource->transientFrameSerial = 0;
|
||||
if (!resource->buffer.IsValid() || resource->pendingFullUpload) {
|
||||
return;
|
||||
}
|
||||
if (static_cast<VkDeviceSize>(bufferObject.GetSize()) != resource->storageSize) {
|
||||
resource->pendingFullUpload = true;
|
||||
return;
|
||||
}
|
||||
|
||||
if (!IsResourceBusy(*resource)) {
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData() + offset,
|
||||
static_cast<VkDeviceSize>(size), static_cast<VkDeviceSize>(offset))) {
|
||||
MGLOG_E_ONCE("VkBufferManager::OnSubData: host upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Busy partial write: stage + GPU copy preserves GL ordering within the
|
||||
// frame and leaves bytes outside the range (possibly GPU-written, e.g.
|
||||
// SSBO) intact. Fall back to a storage swap if staging is unavailable.
|
||||
if (!StagedRangeCopy(*resource, bufferObject, offset, size)) {
|
||||
SwapStorageAndUploadAll(*resource, bufferObject);
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::OnFlushMappedRange(MG_State::GLState::BufferObject& bufferObject, Range1D range,
|
||||
Flags<BufferMappingAccessBit> appAccess) {
|
||||
auto* resource = ResourceOf(bufferObject);
|
||||
if (!resource) {
|
||||
return;
|
||||
}
|
||||
BumpSliceEpoch(*resource);
|
||||
resource->transientFrameSerial = 0;
|
||||
if (!resource->buffer.IsValid() || resource->pendingFullUpload) {
|
||||
return;
|
||||
}
|
||||
if (static_cast<VkDeviceSize>(bufferObject.GetSize()) != resource->storageSize) {
|
||||
resource->pendingFullUpload = true;
|
||||
return;
|
||||
}
|
||||
|
||||
const SizeT offset = range.start;
|
||||
const SizeT size = range.end - range.start;
|
||||
// GL_MAP_UNSYNCHRONIZED_BIT: the app guarantees it does not overwrite
|
||||
// data the GPU is still reading; honour it with a direct host write.
|
||||
if ((appAccess & BufferMappingAccessBit::Unsynchronized) || !IsResourceBusy(*resource)) {
|
||||
if (!resource->buffer.Upload(bufferObject.MappedData() + offset,
|
||||
static_cast<VkDeviceSize>(size), static_cast<VkDeviceSize>(offset))) {
|
||||
MGLOG_E_ONCE("VkBufferManager::OnFlushMappedRange: host upload failed");
|
||||
resource->pendingFullUpload = true;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (!StagedRangeCopy(*resource, bufferObject, offset, size)) {
|
||||
SwapStorageAndUploadAll(*resource, bufferObject);
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::OnResourceDestroyed(SharedPtr<MG_State::GLState::BackendBufferResource>&& resource) {
|
||||
if (!resource) {
|
||||
return;
|
||||
}
|
||||
auto vkResource = std::static_pointer_cast<VkBufferResource>(std::move(resource));
|
||||
if (!vkResource->buffer.IsValid()) {
|
||||
return;
|
||||
}
|
||||
if (m_deferredResourceReleases.empty()) {
|
||||
vkResource->buffer.Destroy();
|
||||
return;
|
||||
}
|
||||
MOBILEGL_ASSERT(m_currentFrameIndex < m_deferredResourceReleases.size(),
|
||||
"VkBufferManager::OnResourceDestroyed current frame index out of range");
|
||||
// Keep the whole resource alive until this frame slot's fence has been
|
||||
// waited, then the storage is destroyed with it.
|
||||
m_deferredResourceReleases[m_currentFrameIndex].push_back(std::move(vkResource));
|
||||
}
|
||||
|
||||
void* VkBufferManager::AcquirePersistentMap(MG_State::GLState::BufferObject& bufferObject) {
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject.GetSize());
|
||||
if (size == 0) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto resource = std::static_pointer_cast<VkBufferResource>(bufferObject.GetBackendResource());
|
||||
if (!resource) {
|
||||
resource = MakeShared<VkBufferResource>();
|
||||
bufferObject.SetBackendResource(resource);
|
||||
TrackLiveResource(resource);
|
||||
}
|
||||
|
||||
// Bumped for the request, not just for the storage it may create. This is the
|
||||
// one call the frontend makes when a buffer becomes persistently mapped for
|
||||
// writing (BufferObject::AcquireMemoryRange), and a map the backend declines
|
||||
// keeps mutating its shadow with no further API call - so it is what lets
|
||||
// GetSliceEpochCounter stand for "no buffer needs a persistent-map range push".
|
||||
BumpSliceEpoch(*resource);
|
||||
|
||||
// Idempotent: an already-backed buffer returns the same mapped base.
|
||||
if (resource->persistentMapped && resource->buffer.IsValid() && resource->storageSize == size) {
|
||||
return resource->buffer.GetMappedData();
|
||||
}
|
||||
|
||||
// One-time creation of HOST_VISIBLE + HOST_COHERENT, persistently mapped storage
|
||||
// carrying every usage (never recreated, so the app's pointer never dangles). Seed
|
||||
// it from the current shadow - MappedData() is still the shadow here because the
|
||||
// frontend adopts (and drops) the shadow only after this returns.
|
||||
DeferRelease(std::move(resource->buffer));
|
||||
const VkBufferUsageFlags persistentUsage =
|
||||
kPersistentBackedUsage |
|
||||
(m_initInfo.transformFeedbackUsageEnabled ? kTransformFeedbackUsage : 0);
|
||||
if (!CreateResidentStorage(*resource, size, persistentUsage, kPersistentBackedRequiredFlags)) {
|
||||
resource->persistentMapped = false;
|
||||
resource->storageSize = 0;
|
||||
resource->usageFlags = 0;
|
||||
return nullptr;
|
||||
}
|
||||
const Uint8* seed = bufferObject.MappedData();
|
||||
if (seed != nullptr) {
|
||||
resource->buffer.Upload(seed, size, 0);
|
||||
}
|
||||
resource->persistentMapped = true;
|
||||
resource->pendingFullUpload = false;
|
||||
resource->storageSize = size;
|
||||
resource->lastUseSerial = 0;
|
||||
return resource->buffer.GetMappedData();
|
||||
}
|
||||
|
||||
Bool VkBufferManager::AcquireResidentSlice(BufferKind kind,
|
||||
const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
BufferSlice& outSlice) {
|
||||
const VkBufferUsageFlags requiredUsage = GetVkBufferUsage(kind);
|
||||
MOBILEGL_ASSERT(requiredUsage != 0, "VkBufferManager::AcquireResidentSlice unsupported buffer kind");
|
||||
MOBILEGL_ASSERT(bufferObject != nullptr, "VkBufferManager::AcquireResidentSlice requires valid buffer object");
|
||||
|
||||
auto resource = GetOrCreateResource(bufferObject);
|
||||
bufferObject->SyncPersistentMappedRange();
|
||||
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject->GetSize());
|
||||
if (size == 0) {
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireResidentSlice failed: buffer size is zero");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Zero-copy persistent buffers already hold the app's live coherent writes in
|
||||
// host-visible storage carrying every usage; bind directly, no re-upload/staging.
|
||||
if (resource->persistentMapped && resource->buffer.IsValid() && resource->storageSize == size) {
|
||||
resource->lastUseSerial = m_frameSerial;
|
||||
outSlice = resource->buffer.GetSlice(0, size);
|
||||
return outSlice.IsValid();
|
||||
}
|
||||
|
||||
const Bool needsRecreate = !resource->buffer.IsValid() || resource->storageSize != size ||
|
||||
((resource->usageFlags & requiredUsage) != requiredUsage) ||
|
||||
resource->pendingFullUpload;
|
||||
if (needsRecreate) {
|
||||
const VkBufferUsageFlags usage = resource->usageFlags | requiredUsage;
|
||||
DeferRelease(std::move(resource->buffer));
|
||||
if (!CreateResidentStorage(*resource, size, usage)) {
|
||||
return false;
|
||||
}
|
||||
if (!resource->buffer.Upload(bufferObject->MappedData(), size, 0)) {
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireResidentSlice failed: initial upload failed");
|
||||
resource->buffer.Destroy();
|
||||
resource->storageSize = 0;
|
||||
resource->usageFlags = 0;
|
||||
return false;
|
||||
}
|
||||
resource->pendingFullUpload = false;
|
||||
}
|
||||
|
||||
resource->lastUseSerial = m_frameSerial;
|
||||
outSlice = resource->buffer.GetSlice(0, size);
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkBufferManager::AcquireStreamedSlice(BufferKind kind,
|
||||
const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
BufferSlice& outSlice) {
|
||||
(void)kind;
|
||||
MOBILEGL_ASSERT(bufferObject != nullptr, "VkBufferManager::AcquireStreamedSlice requires valid buffer object");
|
||||
|
||||
auto resource = GetOrCreateResource(bufferObject);
|
||||
bufferObject->SyncPersistentMappedRange();
|
||||
|
||||
// A persistently mapped resource's storage IS the application's copy of the bytes -
|
||||
// the frontend adopted it in place of the shadow and hands out pointers into it, and
|
||||
// a shader can have written bytes the shadow never saw (a transform feedback
|
||||
// capture). Streaming a second copy would feed this draw the stale shadow, and the
|
||||
// downgrade below would release the storage the application still points at,
|
||||
// breaking the "never recreated" promise AcquirePersistentMap makes.
|
||||
if (resource->persistentMapped) {
|
||||
return AcquireResidentSlice(kind, bufferObject, outSlice);
|
||||
}
|
||||
|
||||
const VkDeviceSize size = static_cast<VkDeviceSize>(bufferObject->GetSize());
|
||||
if (size == 0) {
|
||||
MGLOG_E_ONCE("VkBufferManager::AcquireStreamedSlice failed: buffer size is zero");
|
||||
return false;
|
||||
}
|
||||
|
||||
const Uint64 changeSerial = bufferObject->GetChangeSerial();
|
||||
if (resource->transientFrameSerial == m_frameSerial && resource->transientChangeSerial == changeSerial &&
|
||||
resource->transientSize == size && resource->transientSlice.IsValid()) {
|
||||
outSlice = resource->transientSlice;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Idle-content promotion: see the field comments in VkBufferResource. The
|
||||
// streak counts frame BOUNDARIES survived unchanged (the same-frame memo
|
||||
// above swallows repeat draws), so a promotion needs the content stable
|
||||
// for kStreamedPromotionStreak whole frames - one no-op frame does not
|
||||
// trigger the resident round-trip, whose creation upload is itself a
|
||||
// staged copy worth avoiding for content that is about to change again.
|
||||
constexpr Uint32 kStreamedPromotionStreak = 2;
|
||||
if (resource->promotedResident) {
|
||||
if (resource->promotedChangeSerial == changeSerial &&
|
||||
static_cast<VkDeviceSize>(bufferObject->GetSize()) == size) {
|
||||
return AcquireResidentSlice(kind, bufferObject, outSlice);
|
||||
}
|
||||
resource->promotedResident = false;
|
||||
resource->unchangedStreak = 0;
|
||||
} else if (resource->transientChangeSerial == changeSerial && resource->transientSize == size &&
|
||||
resource->transientFrameSerial != 0) {
|
||||
if (++resource->unchangedStreak >= kStreamedPromotionStreak) {
|
||||
// Promotion moves the buffer off the arena and onto resident storage.
|
||||
resource->promotedResident = true;
|
||||
resource->promotedChangeSerial = changeSerial;
|
||||
BumpSliceEpoch(*resource);
|
||||
if (AcquireResidentSlice(kind, bufferObject, outSlice)) {
|
||||
return true;
|
||||
}
|
||||
resource->promotedResident = false; // resident creation failed: stream as before
|
||||
}
|
||||
} else {
|
||||
resource->unchangedStreak = 0;
|
||||
}
|
||||
|
||||
// A fresh arena allocation: a different slice than the last call handed back,
|
||||
// and (below) the point where a promoted buffer's resident storage is released.
|
||||
// The stable-promotion exit above returns before this, so a buffer the app has
|
||||
// stopped touching keeps one slice for as long as it keeps its resident storage.
|
||||
BumpSliceEpoch(*resource);
|
||||
if (!m_transientUploadArena.Upload(m_currentFrameIndex, bufferObject->MappedData(), size, 16,
|
||||
outSlice)) {
|
||||
return false;
|
||||
}
|
||||
resource->transientSlice = outSlice;
|
||||
resource->transientFrameSerial = m_frameSerial;
|
||||
resource->transientChangeSerial = changeSerial;
|
||||
resource->transientSize = size;
|
||||
|
||||
// Streaming path is authoritative now; release resident storage so we do
|
||||
// not keep a second, stale copy alive (downgrade).
|
||||
if (resource->buffer.IsValid()) {
|
||||
DeferRelease(std::move(resource->buffer));
|
||||
resource->storageSize = 0;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkBufferManager::DeferRelease(VkBufferObject&& buffer) {
|
||||
if (!buffer.IsValid()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (m_deferredBufferReleases.empty()) {
|
||||
buffer.Destroy();
|
||||
return;
|
||||
}
|
||||
|
||||
MOBILEGL_ASSERT(m_currentFrameIndex < m_deferredBufferReleases.size(),
|
||||
"VkBufferManager::DeferRelease current frame index out of range");
|
||||
m_deferredBufferReleases[m_currentFrameIndex].push_back(std::move(buffer));
|
||||
}
|
||||
|
||||
void VkBufferManager::CollectDeferredReleases(Uint32 frameIndex) {
|
||||
MOBILEGL_ASSERT(frameIndex < m_deferredBufferReleases.size(),
|
||||
"VkBufferManager::CollectDeferredReleases frame index out of range");
|
||||
m_deferredBufferReleases[frameIndex].clear();
|
||||
m_deferredResourceReleases[frameIndex].clear();
|
||||
}
|
||||
|
||||
VkBufferUsageFlags VkBufferManager::GetVkBufferUsage(BufferKind kind) {
|
||||
switch (kind) {
|
||||
case BufferKind::Vertex:
|
||||
case BufferKind::Index:
|
||||
// A GL buffer can be rebound between ARRAY_BUFFER and ELEMENT_ARRAY_BUFFER,
|
||||
// and may even be used as both within the same draw setup. Keep resident
|
||||
// vertex/index buffers compatible with both roles from the start so we
|
||||
// never need to recreate a buffer after it has already been bound.
|
||||
return VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT;
|
||||
case BufferKind::Uniform:
|
||||
return VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT;
|
||||
case BufferKind::TextureBuffer:
|
||||
// Both texel roles, for the same reason vertex/index carry both bits: one GL buffer
|
||||
// texture can be read as a samplerBuffer and written as an imageBuffer, and which of
|
||||
// the two it is only becomes known when a shader that uses it is bound - long after
|
||||
// the resident buffer was created. A VkBufferView for a storage-texel descriptor is
|
||||
// invalid unless the buffer was created with the storage bit, so a buffer that
|
||||
// acquired only the uniform bit could never be given one.
|
||||
return VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT;
|
||||
case BufferKind::ShaderStorage:
|
||||
return VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT;
|
||||
case BufferKind::Indirect:
|
||||
return VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
void VkBufferManager::DestroyAllDeferredReleases() {
|
||||
for (auto& releases : m_deferredBufferReleases) {
|
||||
for (auto& buffer : releases) {
|
||||
buffer.Destroy();
|
||||
}
|
||||
releases.clear();
|
||||
}
|
||||
m_deferredBufferReleases.clear();
|
||||
for (auto& releases : m_deferredResourceReleases) {
|
||||
for (auto& resource : releases) {
|
||||
resource->buffer.Destroy();
|
||||
}
|
||||
releases.clear();
|
||||
}
|
||||
m_deferredResourceReleases.clear();
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,197 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkBufferManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "BufferArena.h"
|
||||
#include "MG_State/GLState/BufferState/BufferObject.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class BufferKind : Uint8 {
|
||||
Vertex,
|
||||
Index,
|
||||
Uniform,
|
||||
TextureBuffer,
|
||||
ShaderStorage,
|
||||
Indirect,
|
||||
};
|
||||
|
||||
struct VkBufferManagerInitInfo {
|
||||
VmaAllocator allocator = nullptr;
|
||||
Uint32 frameCount = 0;
|
||||
VkDeviceSize minUploadBytes = 4 * 1024 * 1024;
|
||||
VmaMemoryUsage transientMemoryUsage = VMA_MEMORY_USAGE_AUTO;
|
||||
VmaAllocationCreateFlags transientAllocationFlags = VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT;
|
||||
Bool transientPersistentMapping = false;
|
||||
// VK_EXT_transform_feedback is enabled: persistent-map storage additionally
|
||||
// carries the transform feedback usage so capture targets can bind directly.
|
||||
Bool transformFeedbackUsageEnabled = false;
|
||||
};
|
||||
|
||||
// The DirectVulkan storage behind one frontend buffer (pipe_resource analogue).
|
||||
// Owned (refcounted) by the frontend BufferObject; the manager holds only weak
|
||||
// references (for shutdown) plus strong references on deferred-release lists.
|
||||
class VkBufferResource : public MG_State::GLState::BackendBufferResource {
|
||||
public:
|
||||
~VkBufferResource() override = default;
|
||||
|
||||
// Resident storage (may be invalid for streaming-only buffers).
|
||||
VkBufferObject buffer;
|
||||
VkDeviceSize storageSize = 0;
|
||||
VkBufferUsageFlags usageFlags = 0;
|
||||
// Frame serial of the last GPU reference; drives busy tracking.
|
||||
Uint64 lastUseSerial = 0;
|
||||
// Set when an immediate op could not be applied; forces a full re-upload
|
||||
// on the next AcquireResidentSlice.
|
||||
Bool pendingFullUpload = false;
|
||||
// Backs a zero-copy coherent persistent map (PipeResource GPU residency): the
|
||||
// buffer is HOST_VISIBLE+COHERENT, persistently mapped, carries every usage and is
|
||||
// never orphaned or recreated. Draw-time acquire binds it directly, no re-upload.
|
||||
Bool persistentMapped = false;
|
||||
|
||||
// Bumped from a manager-wide counter every time anything that decides which
|
||||
// BufferSlice an Acquire*Slice call hands back changes: storage created or
|
||||
// released, a full re-upload becoming due, a promotion/demotion between
|
||||
// resident and streamed storage, or a new per-frame arena slice. Callers that
|
||||
// memoise a resolved slice compare this to prove the memo still describes the
|
||||
// buffer. The counter is manager-wide (never per-resource) so a freshly
|
||||
// created resource - including one that replaces a destroyed resource at the
|
||||
// same address - can never reproduce a value some memo already holds. 0 means
|
||||
// "no slice has ever been handed out", which no memo can match.
|
||||
Uint64 sliceEpoch = 0;
|
||||
|
||||
// Cached transient (streaming) slice for the current frame.
|
||||
BufferSlice transientSlice{};
|
||||
Uint64 transientFrameSerial = 0;
|
||||
Uint64 transientChangeSerial = 0;
|
||||
VkDeviceSize transientSize = 0;
|
||||
|
||||
// Streaming re-copies the whole store into the per-frame arena on every
|
||||
// frame, which is right for genuinely per-frame data but pure waste for a
|
||||
// Dynamic-hinted buffer the app stopped touching. After the content
|
||||
// survives kStreamedPromotionStreak frame boundaries unchanged it is
|
||||
// promoted to resident storage (one final upload, then zero per-frame
|
||||
// cost); the first content change demotes it back to streaming, and the
|
||||
// streaming path's existing downgrade releases the resident store.
|
||||
Uint32 unchangedStreak = 0;
|
||||
Bool promotedResident = false;
|
||||
Uint64 promotedChangeSerial = 0;
|
||||
};
|
||||
|
||||
// Supplies a command buffer that is recording and outside any render pass,
|
||||
// for staged buffer-range copies. Implemented by VulkanRenderer.
|
||||
class IBufferCopyCommandProvider {
|
||||
public:
|
||||
virtual ~IBufferCopyCommandProvider() = default;
|
||||
virtual VkCommandBuffer AcquireBufferCopyCommandBuffer() = 0;
|
||||
};
|
||||
|
||||
class VkBufferManager {
|
||||
public:
|
||||
Bool Initialize(const VkBufferManagerInitInfo& initInfo);
|
||||
void Shutdown();
|
||||
|
||||
// Recreate all per-frame transient arenas
|
||||
Bool RecreateTransientArenas(Uint32 frameCount);
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Drains every frame slot's deferred buffer/resource releases. Only valid when
|
||||
// the caller has proven every queue submission complete; used by the present-less
|
||||
// frame-boundary drain. Deliberately does NOT touch the transient arena's parked
|
||||
// superseded blocks: those are still named by this frame's slices (see the
|
||||
// definition), and only a frame rewind retires them.
|
||||
void CollectAllDeferredReleases();
|
||||
// All previously submitted GPU work has completed (vkDeviceWaitIdle).
|
||||
void NotifyDeviceIdle();
|
||||
// A frame slot's submission fence has been waited: every serial up to
|
||||
// and including `serial` is complete. Raises the completed floor so
|
||||
// GetCompletedSerial reflects real fence progress instead of only the
|
||||
// frameSerial-minus-frameCount inference.
|
||||
void NotifyFrameSerialComplete(Uint64 serial);
|
||||
void SetCopyCommandProvider(IBufferCopyCommandProvider* provider);
|
||||
|
||||
Bool UploadTransient(BufferKind kind, Uint32 frameIndex, const void* data, VkDeviceSize size,
|
||||
VkDeviceSize alignment, BufferSlice& outSlice);
|
||||
|
||||
// Draw-time acquire for resident (device-storage) buffers: ensures the
|
||||
// resource exists and is fully uploaded, marks it used this frame.
|
||||
Bool AcquireResidentSlice(BufferKind kind, const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
BufferSlice& outSlice);
|
||||
// Draw-time acquire for streamed buffers: uploads the whole shadow into
|
||||
// the per-frame arena (cached by change serial), releasing any resident
|
||||
// storage the buffer may still own.
|
||||
Bool AcquireStreamedSlice(BufferKind kind, const SharedPtr<MG_State::GLState::BufferObject>& bufferObject,
|
||||
BufferSlice& outSlice);
|
||||
|
||||
// Zero-copy persistent map (PipeResource GPU residency): create (once) a
|
||||
// HOST_VISIBLE+COHERENT, persistently mapped resident buffer carrying every usage,
|
||||
// seed it from the shadow, and return its mapped base for the app to write into
|
||||
// directly. Idempotent. Returns nullptr on failure (frontend keeps its shadow).
|
||||
void* AcquirePersistentMap(MG_State::GLState::BufferObject& bufferObject);
|
||||
|
||||
// Immediate ops, dispatched from the frontend BufferBackendOps table.
|
||||
void OnRespecify(MG_State::GLState::BufferObject& bufferObject);
|
||||
void OnSubData(MG_State::GLState::BufferObject& bufferObject, SizeT offset, SizeT size);
|
||||
void OnFlushMappedRange(MG_State::GLState::BufferObject& bufferObject, Range1D range,
|
||||
Flags<BufferMappingAccessBit> appAccess);
|
||||
void OnResourceDestroyed(SharedPtr<MG_State::GLState::BackendBufferResource>&& resource);
|
||||
|
||||
Uint64 GetFrameSerial() const { return m_frameSerial; }
|
||||
// Highest value handed to any VkBufferResource::sliceEpoch. Unchanged since a
|
||||
// memo was taken means no buffer this manager owns changed which slice it hands
|
||||
// back, and none was persistently mapped, in between - so a memo of resolved
|
||||
// slices needs no per-buffer re-check. See AcquirePersistentMap for the mapping half.
|
||||
Uint64 GetSliceEpochCounter() const { return m_sliceEpochCounter; }
|
||||
// Highest frame serial whose GPU work is known complete; serials at or
|
||||
// below it may be considered signaled. Drives IsResourceBusy and the
|
||||
// backend GL fence objects.
|
||||
Uint64 GetCompletedSerial() const;
|
||||
// Busy = potentially referenced by GPU work that has not been fenced yet
|
||||
// (including commands recorded for the current, unsubmitted frame).
|
||||
Bool IsResourceBusy(const VkBufferResource& resource) const;
|
||||
|
||||
private:
|
||||
Bool InitializeTransientArenas();
|
||||
static VkBufferUsageFlags GetVkBufferUsage(BufferKind kind);
|
||||
VkBufferResource* GetOrCreateResource(const SharedPtr<MG_State::GLState::BufferObject>& bufferObject);
|
||||
static VkBufferResource* ResourceOf(MG_State::GLState::BufferObject& bufferObject);
|
||||
Bool CreateResidentStorage(VkBufferResource& resource, VkDeviceSize size, VkBufferUsageFlags usage,
|
||||
VkMemoryPropertyFlags requiredFlags = 0);
|
||||
// Swap storage (conditional orphan) and refill it from the shadow copy.
|
||||
Bool SwapStorageAndUploadAll(VkBufferResource& resource, MG_State::GLState::BufferObject& bufferObject);
|
||||
// Record a staging-slice copy into the resident storage, ordered against
|
||||
// in-flight and already-recorded GPU work.
|
||||
Bool StagedRangeCopy(VkBufferResource& resource, MG_State::GLState::BufferObject& bufferObject,
|
||||
SizeT offset, SizeT size);
|
||||
void DeferRelease(VkBufferObject&& buffer);
|
||||
void CollectDeferredReleases(Uint32 frameIndex);
|
||||
void DestroyAllDeferredReleases();
|
||||
void TrackLiveResource(const SharedPtr<VkBufferResource>& resource);
|
||||
void ReleaseAllLiveResources();
|
||||
// See VkBufferResource::sliceEpoch.
|
||||
void BumpSliceEpoch(VkBufferResource& resource) { resource.sliceEpoch = ++m_sliceEpochCounter; }
|
||||
|
||||
VkBufferManagerInitInfo m_initInfo{};
|
||||
BufferArena m_transientUploadArena;
|
||||
IBufferCopyCommandProvider* m_copyProvider = nullptr;
|
||||
Vector<Vector<VkBufferObject>> m_deferredBufferReleases;
|
||||
Vector<Vector<SharedPtr<VkBufferResource>>> m_deferredResourceReleases;
|
||||
Vector<WeakPtr<VkBufferResource>> m_liveResources;
|
||||
// Size m_liveResources had just after the last sweep; the next sweep waits for it to double.
|
||||
SizeT m_liveResourcesLastPruned = 0;
|
||||
Uint32 m_currentFrameIndex = 0;
|
||||
Uint64 m_frameSerial = 1;
|
||||
Uint64 m_completedSerialFloor = 0;
|
||||
// Never reset (not even by Shutdown): a value handed to a resource must stay
|
||||
// unique for the process, or a memo taken before a re-initialize could match
|
||||
// a different resource's state after it.
|
||||
Uint64 m_sliceEpochCounter = 0;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -13,11 +13,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_allocator = other.m_allocator;
|
||||
m_buffer = other.m_buffer;
|
||||
m_allocation = other.m_allocation;
|
||||
m_mappedData = other.m_mappedData;
|
||||
m_size = other.m_size;
|
||||
|
||||
other.m_allocator = nullptr;
|
||||
other.m_buffer = VK_NULL_HANDLE;
|
||||
other.m_allocation = nullptr;
|
||||
other.m_mappedData = nullptr;
|
||||
other.m_size = 0;
|
||||
}
|
||||
|
||||
@@ -31,11 +33,13 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_allocator = other.m_allocator;
|
||||
m_buffer = other.m_buffer;
|
||||
m_allocation = other.m_allocation;
|
||||
m_mappedData = other.m_mappedData;
|
||||
m_size = other.m_size;
|
||||
|
||||
other.m_allocator = nullptr;
|
||||
other.m_buffer = VK_NULL_HANDLE;
|
||||
other.m_allocation = nullptr;
|
||||
other.m_mappedData = nullptr;
|
||||
other.m_size = 0;
|
||||
return *this;
|
||||
}
|
||||
@@ -44,8 +48,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Destroy();
|
||||
}
|
||||
|
||||
Bool VkBufferObject::Create(const VkBufferObjectDesc& desc) {
|
||||
return Create(desc.allocator, desc.size, desc.usage, desc.memoryUsage, desc.allocationFlags,
|
||||
desc.requiredFlags);
|
||||
}
|
||||
|
||||
Bool VkBufferObject::Create(VmaAllocator allocator, VkDeviceSize size, VkBufferUsageFlags usage,
|
||||
VmaMemoryUsage memoryUsage, VmaAllocationCreateFlags allocationFlags) {
|
||||
VmaMemoryUsage memoryUsage, VmaAllocationCreateFlags allocationFlags,
|
||||
VkMemoryPropertyFlags requiredFlags) {
|
||||
MOBILEGL_ASSERT(allocator != nullptr, "VkBufferObject::Create requires valid VMA allocator");
|
||||
MOBILEGL_ASSERT(size > 0, "VkBufferObject::Create requires non-zero size");
|
||||
|
||||
@@ -61,11 +71,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VmaAllocationCreateInfo allocationInfo{};
|
||||
allocationInfo.usage = memoryUsage;
|
||||
allocationInfo.flags = allocationFlags;
|
||||
allocationInfo.requiredFlags = requiredFlags;
|
||||
|
||||
const VkResult result =
|
||||
vmaCreateBuffer(m_allocator, &bufferInfo, &allocationInfo, &m_buffer, &m_allocation, nullptr);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("VkBufferObject::Create failed: vmaCreateBuffer returned %d", result);
|
||||
MGLOG_E_ONCE("VkBufferObject::Create failed: vmaCreateBuffer returned %d", result);
|
||||
m_allocator = nullptr;
|
||||
m_buffer = VK_NULL_HANDLE;
|
||||
m_allocation = nullptr;
|
||||
@@ -78,6 +89,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
void VkBufferObject::Destroy() {
|
||||
Unmap();
|
||||
if (m_allocator != nullptr && m_buffer != VK_NULL_HANDLE && m_allocation != nullptr) {
|
||||
vmaDestroyBuffer(m_allocator, m_buffer, m_allocation);
|
||||
}
|
||||
@@ -87,6 +99,33 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_size = 0;
|
||||
}
|
||||
|
||||
void* VkBufferObject::Map() {
|
||||
MOBILEGL_ASSERT(IsValid(), "VkBufferObject::Map called on invalid buffer");
|
||||
|
||||
if (m_mappedData != nullptr) {
|
||||
return m_mappedData;
|
||||
}
|
||||
|
||||
const VkResult mapResult = vmaMapMemory(m_allocator, m_allocation, &m_mappedData);
|
||||
if (mapResult != VK_SUCCESS || m_mappedData == nullptr) {
|
||||
MGLOG_E_ONCE("VkBufferObject::Map failed: vmaMapMemory returned %d", mapResult);
|
||||
m_mappedData = nullptr;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
return m_mappedData;
|
||||
}
|
||||
|
||||
void VkBufferObject::Unmap() {
|
||||
if (!IsValid() || m_mappedData == nullptr) {
|
||||
m_mappedData = nullptr;
|
||||
return;
|
||||
}
|
||||
|
||||
vmaUnmapMemory(m_allocator, m_allocation);
|
||||
m_mappedData = nullptr;
|
||||
}
|
||||
|
||||
Bool VkBufferObject::Upload(const void* data, VkDeviceSize size, VkDeviceSize offset) {
|
||||
MOBILEGL_ASSERT(IsValid(), "VkBufferObject::Upload called on invalid buffer");
|
||||
MOBILEGL_ASSERT(data != nullptr || size == 0, "VkBufferObject::Upload data pointer is null");
|
||||
@@ -96,15 +135,45 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
void* mapped = nullptr;
|
||||
const VkResult mapResult = vmaMapMemory(m_allocator, m_allocation, &mapped);
|
||||
if (mapResult != VK_SUCCESS || mapped == nullptr) {
|
||||
MGLOG_E("VkBufferObject::Upload failed: vmaMapMemory returned %d", mapResult);
|
||||
const Bool wasMapped = IsMapped();
|
||||
void* mapped = wasMapped ? m_mappedData : Map();
|
||||
if (mapped == nullptr) {
|
||||
MGLOG_E_ONCE("VkBufferObject::Upload failed: unable to map buffer");
|
||||
return false;
|
||||
}
|
||||
|
||||
Memcpy(static_cast<Uint8*>(mapped) + offset, data, static_cast<SizeT>(size));
|
||||
vmaUnmapMemory(m_allocator, m_allocation);
|
||||
const VkResult flushResult = vmaFlushAllocation(m_allocator, m_allocation, offset, size);
|
||||
if (flushResult != VK_SUCCESS) {
|
||||
MGLOG_E_ONCE("VkBufferObject::Upload failed: vmaFlushAllocation returned %d", flushResult);
|
||||
if (!wasMapped) {
|
||||
Unmap();
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (!wasMapped) {
|
||||
Unmap();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkBufferObject::Invalidate(VkDeviceSize size, VkDeviceSize offset) {
|
||||
MOBILEGL_ASSERT(IsValid(), "VkBufferObject::Invalidate called on invalid buffer");
|
||||
MOBILEGL_ASSERT(IsMapped(), "VkBufferObject::Invalidate requires mapped memory");
|
||||
MOBILEGL_ASSERT(offset <= m_size, "VkBufferObject::Invalidate offset out of range");
|
||||
|
||||
const VkDeviceSize resolvedSize = size == VK_WHOLE_SIZE ? m_size - offset : size;
|
||||
MOBILEGL_ASSERT(offset + resolvedSize <= m_size, "VkBufferObject::Invalidate range out of bounds");
|
||||
if (resolvedSize == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const VkResult result = vmaInvalidateAllocation(m_allocator, m_allocation, offset, resolvedSize);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E_ONCE("VkBufferObject::Invalidate failed: vmaInvalidateAllocation returned %d", result);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -8,11 +8,23 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "BufferSlice.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
struct VkBufferObjectDesc {
|
||||
VmaAllocator allocator = nullptr;
|
||||
VkDeviceSize size = 0;
|
||||
VkBufferUsageFlags usage = 0;
|
||||
VmaMemoryUsage memoryUsage = VMA_MEMORY_USAGE_AUTO;
|
||||
VmaAllocationCreateFlags allocationFlags = 0;
|
||||
// Memory property bits the allocation MUST satisfy (e.g. HOST_VISIBLE|HOST_COHERENT
|
||||
// for a persistently-mapped buffer the app writes into without explicit flushes).
|
||||
VkMemoryPropertyFlags requiredFlags = 0;
|
||||
};
|
||||
|
||||
class VkBufferObject {
|
||||
public:
|
||||
VkBufferObject() = default;
|
||||
@@ -23,20 +35,42 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkBufferObject(VkBufferObject&& other) noexcept;
|
||||
VkBufferObject& operator=(VkBufferObject&& other) noexcept;
|
||||
|
||||
Bool Create(const VkBufferObjectDesc& desc);
|
||||
Bool Create(VmaAllocator allocator, VkDeviceSize size, VkBufferUsageFlags usage,
|
||||
VmaMemoryUsage memoryUsage, VmaAllocationCreateFlags allocationFlags = 0);
|
||||
VmaMemoryUsage memoryUsage, VmaAllocationCreateFlags allocationFlags = 0,
|
||||
VkMemoryPropertyFlags requiredFlags = 0);
|
||||
void Destroy();
|
||||
|
||||
void* Map();
|
||||
void Unmap();
|
||||
Bool Upload(const void* data, VkDeviceSize size, VkDeviceSize offset = 0);
|
||||
Bool Invalidate(VkDeviceSize size = VK_WHOLE_SIZE, VkDeviceSize offset = 0);
|
||||
|
||||
VkBuffer GetHandle() const { return m_buffer; }
|
||||
VkDeviceSize GetSize() const { return m_size; }
|
||||
// Inline: runs on the per-draw acquire path (a resident buffer bind is a
|
||||
// GetSlice per binding), where an out-of-line call was measurable.
|
||||
BufferSlice GetSlice(VkDeviceSize offset = 0, VkDeviceSize size = VK_WHOLE_SIZE) const {
|
||||
MOBILEGL_ASSERT(offset <= m_size, "VkBufferObject::GetSlice offset out of range");
|
||||
const VkDeviceSize resolvedSize = (size == VK_WHOLE_SIZE) ? (m_size - offset) : size;
|
||||
MOBILEGL_ASSERT(offset + resolvedSize <= m_size, "VkBufferObject::GetSlice range out of bounds");
|
||||
|
||||
BufferSlice slice{};
|
||||
slice.buffer = m_buffer;
|
||||
slice.offset = offset;
|
||||
slice.size = resolvedSize;
|
||||
slice.mapped = (m_mappedData != nullptr) ? static_cast<Uint8*>(m_mappedData) + offset : nullptr;
|
||||
return slice;
|
||||
}
|
||||
void* GetMappedData() const { return m_mappedData; }
|
||||
Bool IsMapped() const { return m_mappedData != nullptr; }
|
||||
Bool IsValid() const { return m_allocator != nullptr && m_buffer != VK_NULL_HANDLE && m_allocation != nullptr; }
|
||||
|
||||
private:
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
VkBuffer m_buffer = VK_NULL_HANDLE;
|
||||
VmaAllocation m_allocation = nullptr;
|
||||
void* m_mappedData = nullptr;
|
||||
VkDeviceSize m_size = 0;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -0,0 +1,483 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkClearManager.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "VkClearManager.h"
|
||||
|
||||
#include "MG_State/GLState/Core.h"
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/TextureEnumConverter.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static Bool IsCubeMapFaceUploadTarget(TextureUploadTarget target) {
|
||||
return target >= TextureUploadTarget::CubeMapPositiveX &&
|
||||
target <= TextureUploadTarget::CubeMapNegativeZ;
|
||||
}
|
||||
|
||||
VkClearColorValue MakeVkClearColorValue(const ClearAttachmentPayload& payload, Bool formatLacksAlpha) {
|
||||
VkClearColorValue clearValue{};
|
||||
switch (payload.colorEncoding) {
|
||||
case ClearColorEncoding::Int:
|
||||
clearValue.int32[0] = payload.colorInt.x();
|
||||
clearValue.int32[1] = payload.colorInt.y();
|
||||
clearValue.int32[2] = payload.colorInt.z();
|
||||
clearValue.int32[3] = formatLacksAlpha ? 1 : payload.colorInt.w();
|
||||
break;
|
||||
case ClearColorEncoding::Uint:
|
||||
clearValue.uint32[0] = payload.colorUint.x();
|
||||
clearValue.uint32[1] = payload.colorUint.y();
|
||||
clearValue.uint32[2] = payload.colorUint.z();
|
||||
clearValue.uint32[3] = formatLacksAlpha ? 1u : payload.colorUint.w();
|
||||
break;
|
||||
case ClearColorEncoding::Float:
|
||||
clearValue.float32[0] = payload.color.x();
|
||||
clearValue.float32[1] = payload.color.y();
|
||||
clearValue.float32[2] = payload.color.z();
|
||||
clearValue.float32[3] = formatLacksAlpha ? 1.0f : payload.color.w();
|
||||
break;
|
||||
}
|
||||
return clearValue;
|
||||
}
|
||||
|
||||
void PreCompensateSrgbClearColor(ClearAttachmentPayload& payload, VkFormat destinationFormat) {
|
||||
if (payload.colorEncoding != ClearColorEncoding::Float) return;
|
||||
// With GL_FRAMEBUFFER_SRGB enabled GL performs the encoding itself, so the driver doing it
|
||||
// is exactly right and there is nothing to undo.
|
||||
if (MG_State::pGLContext->IsCapabilityEnabled(MobileGL::CapabilityInput::FramebufferSrgb)) return;
|
||||
if (ResolveSrgbAttachmentWriteFormat(destinationFormat, false) == destinationFormat) return;
|
||||
|
||||
// sRGB -> linear (GL 4.6 core 8.24), applied to the colour channels only: alpha is stored
|
||||
// linearly in an sRGB format and must pass through untouched.
|
||||
const auto toLinear = [](Float encoded) {
|
||||
const Float value = std::clamp(encoded, 0.0f, 1.0f);
|
||||
return value <= 0.04045f ? value / 12.92f : std::pow((value + 0.055f) / 1.055f, 2.4f);
|
||||
};
|
||||
payload.color = FloatVec4(toLinear(payload.color.x()), toLinear(payload.color.y()),
|
||||
toLinear(payload.color.z()), payload.color.w());
|
||||
}
|
||||
|
||||
void ForceOpaqueClearAlpha(ClearAttachmentPayload& payload) {
|
||||
switch (payload.colorEncoding) {
|
||||
case ClearColorEncoding::Int:
|
||||
payload.colorInt = IntVec4(payload.colorInt.x(), payload.colorInt.y(), payload.colorInt.z(), 1);
|
||||
break;
|
||||
case ClearColorEncoding::Uint:
|
||||
payload.colorUint = UintVec4(payload.colorUint.x(), payload.colorUint.y(), payload.colorUint.z(), 1u);
|
||||
break;
|
||||
case ClearColorEncoding::Float:
|
||||
payload.color = FloatVec4(payload.color.x(), payload.color.y(), payload.color.z(), 1.0f);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static Bool PendingClearMatchesTextureIdentity(const PendingClearKey& key, const TextureIdentity& identity) {
|
||||
return key.texture == identity.texture && key.textureLifetimeId == identity.lifetimeId;
|
||||
}
|
||||
|
||||
static Uint32 ResolveAttachmentBaseArrayLayer(TextureUploadTarget target) {
|
||||
if (!IsCubeMapFaceUploadTarget(target)) {
|
||||
return 0;
|
||||
}
|
||||
return static_cast<Uint32>(target) - static_cast<Uint32>(TextureUploadTarget::CubeMapPositiveX);
|
||||
}
|
||||
|
||||
static Uint32 ResolveAttachmentBaseArrayLayer(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (attachment.IsLayered()) {
|
||||
return 0;
|
||||
}
|
||||
const TextureUploadTarget uploadTarget = attachment.GetTextureUploadTarget();
|
||||
if (!IsCubeMapFaceUploadTarget(uploadTarget)) {
|
||||
return static_cast<Uint32>(std::max(attachment.GetTextureLayer(), 0));
|
||||
}
|
||||
return ResolveAttachmentBaseArrayLayer(uploadTarget);
|
||||
}
|
||||
|
||||
static Uint32 ResolveAttachmentLayerCount(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (attachment.IsLayered()) {
|
||||
return static_cast<Uint32>(std::max(attachment.GetSize().z(), 1));
|
||||
}
|
||||
return 1u;
|
||||
}
|
||||
|
||||
static const MG_State::GLState::FramebufferAttachmentObject* GetClearableAttachment(
|
||||
const MG_State::GLState::FramebufferObject& drawFbo, FramebufferAttachmentType attachmentType) {
|
||||
if (attachmentType == FramebufferAttachmentType::None) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const auto& attachment = drawFbo.GetAttachment(attachmentType);
|
||||
if (!attachment.IsTexture() || attachment.IsRenderbuffer()) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
return &attachment;
|
||||
}
|
||||
|
||||
PendingClearKey VkClearManager::MakePendingClearKey(MG_State::GLState::ITextureObject* texture, Uint32 mipLevel,
|
||||
Uint32 baseArrayLayer, Uint32 layerCount) {
|
||||
return PendingClearKey {
|
||||
.texture = texture,
|
||||
.textureLifetimeId = texture ? texture->GetLifetimeId() : 0,
|
||||
.mipLevel = mipLevel,
|
||||
.baseArrayLayer = baseArrayLayer,
|
||||
.layerCount = layerCount,
|
||||
};
|
||||
}
|
||||
|
||||
PendingClearKey VkClearManager::MakePendingClearKey(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
MOBILEGL_ASSERT(attachment.IsTexture() && !attachment.IsRenderbuffer(),
|
||||
"MakePendingClearKey requires a texture framebuffer attachment");
|
||||
auto* texture = attachment.GetTexture().get();
|
||||
MOBILEGL_ASSERT(texture != nullptr, "MakePendingClearKey: texture attachment resolved to null");
|
||||
const Uint32 mipLevel = static_cast<Uint32>(std::max(attachment.GetTextureLevel(), 0));
|
||||
const Uint32 baseArrayLayer = ResolveAttachmentBaseArrayLayer(attachment);
|
||||
const Uint32 layerCount = ResolveAttachmentLayerCount(attachment);
|
||||
return MakePendingClearKey(texture, mipLevel, baseArrayLayer, layerCount);
|
||||
}
|
||||
|
||||
Bool VkClearManager::Initialize() {
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkClearManager::Shutdown() {
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_pendingClears.clear();
|
||||
m_aliveObjects.clear();
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
TextureIdentity VkClearManager::MakeTextureIdentity(MG_State::GLState::ITextureObject* texture) {
|
||||
return TextureIdentity {
|
||||
.texture = texture,
|
||||
.lifetimeId = texture ? texture->GetLifetimeId() : 0,
|
||||
};
|
||||
}
|
||||
|
||||
void VkClearManager::MergeClearPayload(ClearAttachmentPayload& dst, const ClearAttachmentPayload& src) {
|
||||
dst.mask |= src.mask;
|
||||
if ((src.mask & GL_COLOR_BUFFER_BIT) != 0) {
|
||||
dst.color = src.color;
|
||||
}
|
||||
if ((src.mask & GL_DEPTH_BUFFER_BIT) != 0) {
|
||||
dst.depth = src.depth;
|
||||
}
|
||||
if ((src.mask & GL_STENCIL_BUFFER_BIT) != 0) {
|
||||
dst.stencil = src.stencil;
|
||||
}
|
||||
}
|
||||
|
||||
void VkClearManager::ErasePendingClearsForTextureLocked(const TextureIdentity& identity) {
|
||||
Vector<PendingClearKey> keysToErase;
|
||||
keysToErase.reserve(m_pendingClears.size());
|
||||
for (auto it = m_pendingClears.begin(); it != m_pendingClears.end(); ++it) {
|
||||
if (PendingClearMatchesTextureIdentity(it->first, identity)) {
|
||||
keysToErase.emplace_back(it->first);
|
||||
}
|
||||
}
|
||||
for (const auto& key : keysToErase) {
|
||||
m_pendingClears.erase(key);
|
||||
}
|
||||
m_aliveObjects.erase(identity);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
Bool VkClearManager::LockTextureIdentityLocked(const TextureIdentity& identity,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture) {
|
||||
outTexture.reset();
|
||||
if (identity.texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto aliveIt = m_aliveObjects.find(identity);
|
||||
if (aliveIt == m_aliveObjects.end()) {
|
||||
ErasePendingClearsForTextureLocked(identity);
|
||||
return false;
|
||||
}
|
||||
|
||||
outTexture = aliveIt->second.lock();
|
||||
if (!outTexture || outTexture.get() != identity.texture || outTexture->GetLifetimeId() != identity.lifetimeId) {
|
||||
ErasePendingClearsForTextureLocked(identity);
|
||||
outTexture.reset();
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkClearManager::LockTextureLocked(const PendingClearKey& key,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture) {
|
||||
return LockTextureIdentityLocked(TextureIdentity{
|
||||
.texture = key.texture,
|
||||
.lifetimeId = key.textureLifetimeId,
|
||||
}, outTexture);
|
||||
}
|
||||
|
||||
void VkClearManager::QueueClear(GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo) {
|
||||
if (mask & GL_COLOR_BUFFER_BIT) {
|
||||
auto& drawbufs = drawFbo.GetDrawBuffers();
|
||||
// This should automatically work on default & offscreen FBO
|
||||
for (auto drawbuf: drawbufs) {
|
||||
const auto* attachment = GetClearableAttachment(drawFbo, drawbuf);
|
||||
if (!attachment) {
|
||||
continue;
|
||||
}
|
||||
|
||||
QueueClear({
|
||||
.mask = GL_COLOR_BUFFER_BIT,
|
||||
.color = clearPayload.color
|
||||
}, *attachment);
|
||||
|
||||
MGLOG_D("%s: %s (texture %d) - color = (%.2f, %.2f, %.2f, %.2f)", __func__,
|
||||
MG_Util::ConvertFramebufferAttachmentTypeToString(drawbuf).c_str(),
|
||||
attachment->GetTexture()->GetExternalIndex(),
|
||||
clearPayload.color[0], clearPayload.color[1], clearPayload.color[2], clearPayload.color[3]);
|
||||
}
|
||||
}
|
||||
|
||||
if (mask & GL_DEPTH_BUFFER_BIT) {
|
||||
const auto* attachment = GetClearableAttachment(drawFbo, FramebufferAttachmentType::Depth);
|
||||
if (attachment) {
|
||||
QueueClear({
|
||||
.mask = GL_DEPTH_BUFFER_BIT,
|
||||
.depth = clearPayload.depth,
|
||||
}, *attachment);
|
||||
|
||||
MGLOG_D("%s: Depth (texture %d) - depth = (%.2f)", __func__,
|
||||
attachment->GetTexture()->GetExternalIndex(), clearPayload.depth);
|
||||
}
|
||||
}
|
||||
|
||||
if (mask & GL_STENCIL_BUFFER_BIT) {
|
||||
const auto* attachment = GetClearableAttachment(drawFbo, FramebufferAttachmentType::Stencil);
|
||||
if (attachment) {
|
||||
QueueClear({
|
||||
.mask = GL_STENCIL_BUFFER_BIT,
|
||||
.stencil = clearPayload.stencil,
|
||||
}, *attachment);
|
||||
|
||||
MGLOG_D("%s: Stencil (texture %d) - stencil = (%u)", __func__,
|
||||
attachment->GetTexture()->GetExternalIndex(), clearPayload.stencil);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void VkClearManager::QueueClear(const ClearAttachmentPayload& clearPayload,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& texture) {
|
||||
if (clearPayload.mask == 0 || !texture) {
|
||||
return;
|
||||
}
|
||||
|
||||
const PendingClearKey key = MakePendingClearKey(texture.get());
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_aliveObjects[MakeTextureIdentity(texture.get())] = texture;
|
||||
auto& pending = m_pendingClears[key];
|
||||
MergeClearPayload(pending, clearPayload);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
void VkClearManager::QueueClear(const ClearAttachmentPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (clearPayload.mask == 0 || !attachment.IsTexture() || attachment.IsRenderbuffer()) {
|
||||
return;
|
||||
}
|
||||
const auto texture = attachment.GetTexture();
|
||||
if (!texture) {
|
||||
return;
|
||||
}
|
||||
|
||||
const PendingClearKey key = MakePendingClearKey(attachment);
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_aliveObjects[MakeTextureIdentity(texture.get())] = texture;
|
||||
auto& pending = m_pendingClears[key];
|
||||
MergeClearPayload(pending, clearPayload);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
Bool VkClearManager::HasPendingClear(MG_State::GLState::ITextureObject* texture) {
|
||||
if (texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const Uint64 lifetimeId = texture->GetLifetimeId();
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
for (auto it = m_pendingClears.begin(); it != m_pendingClears.end(); ++it) {
|
||||
if (it->first.texture == texture && it->first.textureLifetimeId == lifetimeId) {
|
||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||
return LockTextureLocked(it->first, liveTexture);
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool VkClearManager::HasPendingClear(const PendingClearKey& key) {
|
||||
if (key.texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
if (m_pendingClears.find(key) == m_pendingClears.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||
return LockTextureLocked(key, liveTexture);
|
||||
}
|
||||
|
||||
Bool VkClearManager::HasPendingClear(const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (!attachment.IsTexture() || attachment.IsRenderbuffer() || !attachment.GetTexture()) {
|
||||
return false;
|
||||
}
|
||||
return HasPendingClear(MakePendingClearKey(attachment));
|
||||
}
|
||||
|
||||
Bool VkClearManager::GetPendingClear(const PendingClearKey& key, ClearAttachmentPayload& outPayload) {
|
||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||
return GetPendingClear(key, outPayload, liveTexture);
|
||||
}
|
||||
|
||||
Bool VkClearManager::GetPendingClear(const PendingClearKey& key, ClearAttachmentPayload& outPayload,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture) {
|
||||
if (key.texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
if (!LockTextureLocked(key, outTexture)) {
|
||||
return false;
|
||||
}
|
||||
auto it = m_pendingClears.find(key);
|
||||
if (it == m_pendingClears.end()) {
|
||||
outTexture.reset();
|
||||
return false;
|
||||
}
|
||||
|
||||
outPayload = it->second;
|
||||
MGLOG_D("%s: Got pending clear for texture@%p lifetime=%llu, mip=%u layer=%u count=%u mask=0x%x clear value: color = (%.2f, %.2f, %.2f, %.2f), depth = (%.2f), stencil = (%u)", __func__,
|
||||
static_cast<void*>(key.texture),
|
||||
static_cast<unsigned long long>(key.textureLifetimeId),
|
||||
key.mipLevel, key.baseArrayLayer, key.layerCount,
|
||||
static_cast<Uint32>(outPayload.mask),
|
||||
outPayload.color[0], outPayload.color[1], outPayload.color[2], outPayload.color[3],
|
||||
outPayload.depth,
|
||||
outPayload.stencil);
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkClearManager::GetPendingClear(const MG_State::GLState::FramebufferAttachmentObject& attachment,
|
||||
ClearAttachmentPayload& outPayload) {
|
||||
if (!attachment.IsTexture() || attachment.IsRenderbuffer() || !attachment.GetTexture()) {
|
||||
MGLOG_D("%s: Failed getting pending clear for non-texture framebuffer attachment", __func__);
|
||||
return false;
|
||||
}
|
||||
return GetPendingClear(MakePendingClearKey(attachment), outPayload);
|
||||
}
|
||||
|
||||
Bool VkClearManager::GetPendingClears(MG_State::GLState::ITextureObject* texture,
|
||||
Vector<PendingClearEntry>& outEntries) {
|
||||
outEntries.clear();
|
||||
if (texture == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return false; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
|
||||
const Uint64 lifetimeId = texture->GetLifetimeId();
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
SharedPtr<MG_State::GLState::ITextureObject> liveTexture;
|
||||
if (!LockTextureIdentityLocked(MakeTextureIdentity(texture), liveTexture)) {
|
||||
return false;
|
||||
}
|
||||
for (auto it = m_pendingClears.begin(); it != m_pendingClears.end(); ++it) {
|
||||
if (it->first.texture == texture && it->first.textureLifetimeId == lifetimeId) {
|
||||
outEntries.emplace_back(PendingClearEntry{.key = it->first, .payload = it->second});
|
||||
}
|
||||
}
|
||||
return !outEntries.empty();
|
||||
}
|
||||
|
||||
void VkClearManager::PopPendingClear(MG_State::GLState::ITextureObject* texture) {
|
||||
if (texture == nullptr) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (m_pendingCount.load(std::memory_order_relaxed) == 0) {
|
||||
return; // per-draw hot path: nothing pending anywhere
|
||||
}
|
||||
const TextureIdentity identity = MakeTextureIdentity(texture);
|
||||
MGLOG_D("%s: Pop all pending clears for texture %d", __func__, texture->GetExternalIndex());
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
ErasePendingClearsForTextureLocked(identity);
|
||||
}
|
||||
|
||||
void VkClearManager::PopPendingClear(const PendingClearKey& key) {
|
||||
if (key.texture == nullptr) {
|
||||
return;
|
||||
}
|
||||
|
||||
{
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
auto it = m_pendingClears.find(key);
|
||||
if (it != m_pendingClears.end()) {
|
||||
m_pendingClears.erase(it);
|
||||
m_pendingCount.store(static_cast<Uint32>(m_pendingClears.size()), std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
MGLOG_D("%s: Pop pending clear for texture@%p lifetime=%llu mip=%u layer=%u count=%u", __func__,
|
||||
static_cast<void*>(key.texture), static_cast<unsigned long long>(key.textureLifetimeId),
|
||||
key.mipLevel, key.baseArrayLayer, key.layerCount);
|
||||
}
|
||||
|
||||
void VkClearManager::PopPendingClear(const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
if (!attachment.IsTexture() || attachment.IsRenderbuffer() || !attachment.GetTexture()) {
|
||||
return;
|
||||
}
|
||||
PopPendingClear(MakePendingClearKey(attachment));
|
||||
}
|
||||
|
||||
SizeT VkClearManager::CollectGarbage() {
|
||||
const std::lock_guard<std::mutex> lock(m_mutex);
|
||||
m_gcCounter++;
|
||||
if (m_gcCounter != 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Vector<TextureIdentity> expiredTextures;
|
||||
expiredTextures.reserve(m_aliveObjects.size());
|
||||
for (auto it = m_aliveObjects.begin(); it != m_aliveObjects.end(); ++it) {
|
||||
if (it->second.expired()) {
|
||||
expiredTextures.emplace_back(it->first);
|
||||
}
|
||||
}
|
||||
if (expiredTextures.empty()) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
for (const auto& identity : expiredTextures) {
|
||||
ErasePendingClearsForTextureLocked(identity);
|
||||
}
|
||||
return expiredTextures.size();
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,168 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkClearManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include "../VulkanRendererConfig.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
#include "MG_Util/Math/VectorTypes.h"
|
||||
|
||||
#include <Includes.h>
|
||||
#include <atomic>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
struct ClearFramebufferPayload {
|
||||
FloatVec4 color;
|
||||
Float depth{};
|
||||
Uint32 stencil{};
|
||||
};
|
||||
|
||||
// A colour clear reaches us from one of glClear/ClearBufferfv, ClearBufferiv or
|
||||
// ClearBufferuiv, and Vulkan reads VkClearColorValue's union according to the destination
|
||||
// image's format rather than converting between the members - a float written where an
|
||||
// integer format is expected is reinterpreted bit for bit, not rounded. Remember which entry
|
||||
// point supplied the value so the member written when the clear is materialized matches.
|
||||
enum class ClearColorEncoding : Uint8 { Float, Int, Uint };
|
||||
|
||||
struct ClearAttachmentPayload {
|
||||
GLbitfield mask = 0;
|
||||
FloatVec4 color = FloatVec4(0.0f, 0.0f, 0.0f, 0.0f);
|
||||
ClearColorEncoding colorEncoding = ClearColorEncoding::Float;
|
||||
IntVec4 colorInt = IntVec4(0, 0, 0, 0);
|
||||
UintVec4 colorUint = UintVec4(0u, 0u, 0u, 0u);
|
||||
Float depth = 1.0f;
|
||||
Uint32 stencil = 0;
|
||||
};
|
||||
|
||||
// Builds the clear value for `payload` in the union member its encoding calls for.
|
||||
// `formatLacksAlpha` applies GL's rule that a format without an alpha channel reads as one,
|
||||
// expressed in whichever type matches (GL 4.6 core 15.2.3).
|
||||
VkClearColorValue MakeVkClearColorValue(const ClearAttachmentPayload& payload, Bool formatLacksAlpha);
|
||||
|
||||
// Applies that same rule in place, for the paths that have to bake it into the payload before
|
||||
// the destination is known.
|
||||
void ForceOpaqueClearAlpha(ClearAttachmentPayload& payload);
|
||||
|
||||
// vkCmdClearColorImage names the image, so the driver applies the destination format's transfer
|
||||
// function to whatever value it is handed. Every other write path in this backend goes through
|
||||
// the UNORM twin view while GL_FRAMEBUFFER_SRGB is off (ResolveSrgbAttachmentWriteFormat) and
|
||||
// therefore stores the raw value GL asked for. Rewrites `payload` to the linear colour whose
|
||||
// encoding is that raw value, so a direct image clear of an sRGB destination agrees with them.
|
||||
// A no-op for every other format, for integer clear encodings, and when GL is doing the
|
||||
// encoding itself.
|
||||
void PreCompensateSrgbClearColor(ClearAttachmentPayload& payload, VkFormat destinationFormat);
|
||||
|
||||
struct PendingClearKey {
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
Uint64 textureLifetimeId = 0;
|
||||
Uint32 mipLevel = 0;
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = 1;
|
||||
|
||||
Bool operator==(const PendingClearKey& other) const {
|
||||
return texture == other.texture && textureLifetimeId == other.textureLifetimeId &&
|
||||
mipLevel == other.mipLevel &&
|
||||
baseArrayLayer == other.baseArrayLayer && layerCount == other.layerCount;
|
||||
}
|
||||
};
|
||||
|
||||
struct TextureIdentity {
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
Uint64 lifetimeId = 0;
|
||||
|
||||
Bool operator==(const TextureIdentity& other) const {
|
||||
return texture == other.texture && lifetimeId == other.lifetimeId;
|
||||
}
|
||||
};
|
||||
|
||||
struct PendingClearEntry {
|
||||
PendingClearKey key{};
|
||||
ClearAttachmentPayload payload{};
|
||||
};
|
||||
|
||||
struct PendingClearKeyHash {
|
||||
SizeT operator()(const PendingClearKey& key) const {
|
||||
const SizeT textureHash = std::hash<MG_State::GLState::ITextureObject*>{}(key.texture);
|
||||
const SizeT textureLifetimeHash = std::hash<Uint64>{}(key.textureLifetimeId);
|
||||
const SizeT mipHash = std::hash<Uint32>{}(key.mipLevel);
|
||||
const SizeT layerHash = std::hash<Uint32>{}(key.baseArrayLayer);
|
||||
const SizeT layerCountHash = std::hash<Uint32>{}(key.layerCount);
|
||||
SizeT hash = textureHash;
|
||||
hash ^= textureLifetimeHash + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= mipHash + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= layerHash + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= layerCountHash + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
struct TextureIdentityHash {
|
||||
SizeT operator()(const TextureIdentity& key) const {
|
||||
SizeT hash = std::hash<MG_State::GLState::ITextureObject*>{}(key.texture);
|
||||
hash ^= std::hash<Uint64>{}(key.lifetimeId) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
class VkClearManager {
|
||||
public:
|
||||
static PendingClearKey MakePendingClearKey(const MG_State::GLState::FramebufferAttachmentObject& attachment);
|
||||
static PendingClearKey MakePendingClearKey(MG_State::GLState::ITextureObject* texture, Uint32 mipLevel = 0,
|
||||
Uint32 baseArrayLayer = 0, Uint32 layerCount = 1);
|
||||
|
||||
Bool Initialize();
|
||||
void Shutdown();
|
||||
|
||||
void QueueClear(GLbitfield mask, const ClearFramebufferPayload& clearPayload, const MG_State::GLState::FramebufferObject& drawFbo);
|
||||
void QueueClear(
|
||||
const ClearAttachmentPayload& clearPayload,
|
||||
const SharedPtr<MG_State::GLState::ITextureObject>& texture);
|
||||
void QueueClear(const ClearAttachmentPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment);
|
||||
Bool HasPendingClear(MG_State::GLState::ITextureObject* texture);
|
||||
Bool HasPendingClear(const PendingClearKey& key);
|
||||
Bool HasPendingClear(const MG_State::GLState::FramebufferAttachmentObject& attachment);
|
||||
Bool GetPendingClear(const PendingClearKey& key, ClearAttachmentPayload& outPayload);
|
||||
Bool GetPendingClear(const PendingClearKey& key, ClearAttachmentPayload& outPayload,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture);
|
||||
Bool GetPendingClear(const MG_State::GLState::FramebufferAttachmentObject& attachment,
|
||||
ClearAttachmentPayload& outPayload);
|
||||
Bool GetPendingClears(MG_State::GLState::ITextureObject* texture, Vector<PendingClearEntry>& outEntries);
|
||||
void PopPendingClear(MG_State::GLState::ITextureObject* texture);
|
||||
void PopPendingClear(const PendingClearKey& key);
|
||||
void PopPendingClear(const MG_State::GLState::FramebufferAttachmentObject& attachment);
|
||||
SizeT CollectGarbage();
|
||||
private:
|
||||
static TextureIdentity MakeTextureIdentity(MG_State::GLState::ITextureObject* texture);
|
||||
static void MergeClearPayload(ClearAttachmentPayload& dst, const ClearAttachmentPayload& src);
|
||||
void ErasePendingClearsForTextureLocked(const TextureIdentity& identity);
|
||||
Bool LockTextureIdentityLocked(const TextureIdentity& identity,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture);
|
||||
Bool LockTextureLocked(const PendingClearKey& key,
|
||||
SharedPtr<MG_State::GLState::ITextureObject>& outTexture);
|
||||
|
||||
Uint8 m_gcCounter = 0;
|
||||
public:
|
||||
// Lock-free probe for the consecutive-draw fast path: any pending clear
|
||||
// forces the full SetupDraw path (which materializes/consumes it).
|
||||
Bool HasAnyPendingClears() const { return m_pendingCount.load(std::memory_order_relaxed) != 0; }
|
||||
|
||||
private:
|
||||
mutable std::mutex m_mutex;
|
||||
// Lock-free mirror of m_pendingClears.size(), maintained under m_mutex
|
||||
// by every mutation. The per-draw probes (HasPendingClear/GetPending*)
|
||||
// read it before taking the lock: during draw batches the pending set
|
||||
// is almost always empty, so this turns several locked map probes per
|
||||
// draw into one relaxed load.
|
||||
std::atomic<Uint32> m_pendingCount{0};
|
||||
std::unordered_map<PendingClearKey, ClearAttachmentPayload, PendingClearKeyHash> m_pendingClears;
|
||||
std::unordered_map<TextureIdentity, WeakPtr<MG_State::GLState::ITextureObject>, TextureIdentityHash> m_aliveObjects;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -1,547 +0,0 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkFramebufferManager.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "VkFramebufferManager.h"
|
||||
|
||||
#include <MG_State/GLState/RenderbufferState/RenderbufferObject.h>
|
||||
#include <MG_State/GLState/TextureState/TextureEnum.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool VkFramebufferManager::Initialize(const InitInfo& initInfo) {
|
||||
m_device = initInfo.device;
|
||||
m_physicalDevice = initInfo.physicalDevice;
|
||||
return m_device != VK_NULL_HANDLE && m_physicalDevice != VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
void VkFramebufferManager::Shutdown() {
|
||||
for (auto& [_, target] : m_offscreenColorTargets) {
|
||||
DestroyOffscreenColorTarget(target);
|
||||
}
|
||||
m_offscreenColorTargets.clear();
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_physicalDevice = VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::EnsureOffscreenColorTarget(Uint glFboExternalIndex,
|
||||
const MG_State::GLState::FramebufferObject& glFbo) {
|
||||
const auto& colorAttachment = glFbo.GetAttachment(FramebufferAttachmentType::Color0);
|
||||
if (!colorAttachment.IsValid() || colorAttachment.IsEmpty()) {
|
||||
MGLOG_W("VkFramebufferManager: FBO %u has no valid COLOR0 attachment", glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto objectVersion = glFbo.GetObjectVersion();
|
||||
auto& target = m_offscreenColorTargets[glFboExternalIndex];
|
||||
if (target.image != VK_NULL_HANDLE && target.glObjectVersion == objectVersion) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return RecreateOffscreenColorTarget(target, glFbo, colorAttachment, objectVersion);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenColorToAttachment(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenColorToAttachment skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
if (!TransitionImageLayout(commandBuffer, target.image, target.layout, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL,
|
||||
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, 0,
|
||||
VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT,
|
||||
VK_IMAGE_ASPECT_COLOR_BIT)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (target.depthStencilImage == VK_NULL_HANDLE) {
|
||||
return true;
|
||||
}
|
||||
|
||||
VkImageAspectFlags aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
||||
if (target.depthStencilFormat == VK_FORMAT_D24_UNORM_S8_UINT ||
|
||||
target.depthStencilFormat == VK_FORMAT_D32_SFLOAT_S8_UINT) {
|
||||
aspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
|
||||
}
|
||||
|
||||
return TransitionImageLayout(
|
||||
commandBuffer, target.depthStencilImage, target.depthStencilLayout,
|
||||
VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
|
||||
VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT, 0,
|
||||
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT, aspectMask);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenColorToTransferSrc(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenColorToTransferSrc skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
return TransitionImageLayout(commandBuffer, target.image, target.layout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
|
||||
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0,
|
||||
VK_ACCESS_TRANSFER_READ_BIT, VK_IMAGE_ASPECT_COLOR_BIT);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenColorToTransferDst(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenColorToTransferDst skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
return TransitionImageLayout(commandBuffer, target.image, target.layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
|
||||
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0,
|
||||
VK_ACCESS_TRANSFER_WRITE_BIT, VK_IMAGE_ASPECT_COLOR_BIT);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenColorToGeneral(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenColorToGeneral skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
return TransitionImageLayout(commandBuffer, target.image, target.layout, VK_IMAGE_LAYOUT_GENERAL,
|
||||
VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0,
|
||||
VK_ACCESS_TRANSFER_READ_BIT | VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
VK_IMAGE_ASPECT_COLOR_BIT);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenDepthStencilToTransferSrc(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenDepthStencilToTransferSrc skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
if (target.depthStencilImage == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
VkImageAspectFlags aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
||||
if (target.depthStencilFormat == VK_FORMAT_D24_UNORM_S8_UINT ||
|
||||
target.depthStencilFormat == VK_FORMAT_D32_SFLOAT_S8_UINT) {
|
||||
aspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
|
||||
}
|
||||
return TransitionImageLayout(commandBuffer, target.depthStencilImage, target.depthStencilLayout,
|
||||
VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT, 0, VK_ACCESS_TRANSFER_READ_BIT, aspectMask);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenDepthStencilToTransferDst(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenDepthStencilToTransferDst skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
if (target.depthStencilImage == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
VkImageAspectFlags aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
||||
if (target.depthStencilFormat == VK_FORMAT_D24_UNORM_S8_UINT ||
|
||||
target.depthStencilFormat == VK_FORMAT_D32_SFLOAT_S8_UINT) {
|
||||
aspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
|
||||
}
|
||||
return TransitionImageLayout(commandBuffer, target.depthStencilImage, target.depthStencilLayout,
|
||||
VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT, 0, VK_ACCESS_TRANSFER_WRITE_BIT, aspectMask);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenDepthStencilToGeneral(VkCommandBuffer commandBuffer,
|
||||
Uint glFboExternalIndex) {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end()) {
|
||||
MGLOG_W("VkFramebufferManager::TransitionOffscreenDepthStencilToGeneral skipped: FBO %u not found",
|
||||
glFboExternalIndex);
|
||||
return false;
|
||||
}
|
||||
auto& target = it->second;
|
||||
if (target.depthStencilImage == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
VkImageAspectFlags aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
||||
if (target.depthStencilFormat == VK_FORMAT_D24_UNORM_S8_UINT ||
|
||||
target.depthStencilFormat == VK_FORMAT_D32_SFLOAT_S8_UINT) {
|
||||
aspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
|
||||
}
|
||||
return TransitionImageLayout(commandBuffer, target.depthStencilImage, target.depthStencilLayout,
|
||||
VK_IMAGE_LAYOUT_GENERAL, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT, 0,
|
||||
VK_ACCESS_TRANSFER_READ_BIT | VK_ACCESS_TRANSFER_WRITE_BIT, aspectMask);
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionOffscreenColorTextureToShaderRead(VkCommandBuffer commandBuffer,
|
||||
Uint textureExternalIndex) {
|
||||
for (auto& [_, target] : m_offscreenColorTargets) {
|
||||
if (target.colorTextureExternalIndex != textureExternalIndex || target.image == VK_NULL_HANDLE) {
|
||||
continue;
|
||||
}
|
||||
const Bool fromUndefined = (target.layout == VK_IMAGE_LAYOUT_UNDEFINED);
|
||||
return TransitionImageLayout(
|
||||
commandBuffer, target.image, target.layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL,
|
||||
fromUndefined ? VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT
|
||||
: (VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT),
|
||||
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT,
|
||||
fromUndefined ? 0 : (VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT),
|
||||
VK_ACCESS_SHADER_READ_BIT, VK_IMAGE_ASPECT_COLOR_BIT);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::GetOffscreenColorImage(Uint glFboExternalIndex, VkImage& outImage,
|
||||
VkExtent2D& outExtent) const {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end() || it->second.image == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
outImage = it->second.image;
|
||||
outExtent = it->second.extent;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::GetOffscreenDepthStencilImage(Uint glFboExternalIndex, VkImage& outImage,
|
||||
VkExtent2D& outExtent, VkFormat& outFormat) const {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end() || it->second.depthStencilImage == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
outImage = it->second.depthStencilImage;
|
||||
outExtent = it->second.extent;
|
||||
outFormat = it->second.depthStencilFormat;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::GetOffscreenColorViewByTexture(Uint textureExternalIndex,
|
||||
VkImageView& outImageView) const {
|
||||
for (const auto& [_, target] : m_offscreenColorTargets) {
|
||||
if (target.colorTextureExternalIndex != textureExternalIndex || target.imageView == VK_NULL_HANDLE) {
|
||||
continue;
|
||||
}
|
||||
outImageView = target.imageView;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::GetOffscreenRenderSurface(Uint glFboExternalIndex, VkImageView& outColorView,
|
||||
VkFormat& outColorFormat, VkImageView& outDepthStencilView,
|
||||
VkFormat& outDepthStencilFormat, VkExtent2D& outExtent) const {
|
||||
auto it = m_offscreenColorTargets.find(glFboExternalIndex);
|
||||
if (it == m_offscreenColorTargets.end() || it->second.imageView == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
outColorView = it->second.imageView;
|
||||
outColorFormat = it->second.format;
|
||||
outDepthStencilView = it->second.depthStencilImageView;
|
||||
outExtent = it->second.extent;
|
||||
outDepthStencilFormat = it->second.depthStencilFormat;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::RecreateOffscreenColorTarget(
|
||||
OffscreenColorTarget& target, const MG_State::GLState::FramebufferObject& glFbo,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& colorAttachment, Uint16 glObjectVersion) {
|
||||
DestroyOffscreenColorTarget(target);
|
||||
|
||||
const auto size = colorAttachment.GetSize();
|
||||
if (size.x() <= 0 || size.y() <= 0) {
|
||||
MGLOG_W("VkFramebufferManager: COLOR0 attachment size is invalid (%d, %d)", size.x(), size.y());
|
||||
return false;
|
||||
}
|
||||
|
||||
const VkFormat format = ResolveColorFormat(colorAttachment);
|
||||
if (format == VK_FORMAT_UNDEFINED) {
|
||||
MGLOG_W("VkFramebufferManager: COLOR0 attachment format is unsupported for Vulkan clear");
|
||||
return false;
|
||||
}
|
||||
|
||||
VkImageCreateInfo imageInfo{};
|
||||
imageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
|
||||
imageInfo.imageType = VK_IMAGE_TYPE_2D;
|
||||
imageInfo.extent.width = static_cast<Uint32>(size.x());
|
||||
imageInfo.extent.height = static_cast<Uint32>(size.y());
|
||||
imageInfo.extent.depth = 1;
|
||||
imageInfo.mipLevels = 1;
|
||||
imageInfo.arrayLayers = 1;
|
||||
imageInfo.format = format;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT |
|
||||
VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_SAMPLED_BIT;
|
||||
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateImage(m_device, &imageInfo, nullptr, &target.image), "vkCreateImage(offscreen color)");
|
||||
|
||||
VkMemoryRequirements memoryRequirements{};
|
||||
vkGetImageMemoryRequirements(m_device, target.image, &memoryRequirements);
|
||||
|
||||
VkMemoryAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
allocInfo.allocationSize = memoryRequirements.size;
|
||||
allocInfo.memoryTypeIndex =
|
||||
FindMemoryType(memoryRequirements.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(m_device, &allocInfo, nullptr, &target.memory), "vkAllocateMemory(offscreen color)");
|
||||
VK_VERIFY(vkBindImageMemory(m_device, target.image, target.memory, 0), "vkBindImageMemory(offscreen color)");
|
||||
|
||||
VkImageViewCreateInfo viewInfo{};
|
||||
viewInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
|
||||
viewInfo.image = target.image;
|
||||
viewInfo.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
viewInfo.format = format;
|
||||
viewInfo.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
viewInfo.subresourceRange.baseMipLevel = 0;
|
||||
viewInfo.subresourceRange.levelCount = 1;
|
||||
viewInfo.subresourceRange.baseArrayLayer = 0;
|
||||
viewInfo.subresourceRange.layerCount = 1;
|
||||
VK_VERIFY(vkCreateImageView(m_device, &viewInfo, nullptr, &target.imageView),
|
||||
"vkCreateImageView(offscreen color)");
|
||||
|
||||
const auto& depthAttachment = glFbo.GetAttachment(FramebufferAttachmentType::Depth);
|
||||
const auto& stencilAttachment = glFbo.GetAttachment(FramebufferAttachmentType::Stencil);
|
||||
const Bool requestedDepthStencil = (depthAttachment.IsValid() && !depthAttachment.IsEmpty()) ||
|
||||
(stencilAttachment.IsValid() && !stencilAttachment.IsEmpty());
|
||||
VkFormat depthStencilFormat = ResolveDepthStencilFormat(depthAttachment, stencilAttachment);
|
||||
if (depthStencilFormat == VK_FORMAT_D24_UNORM_S8_UINT) {
|
||||
depthStencilFormat =
|
||||
FindSupportedDepthStencilFormat({VK_FORMAT_D24_UNORM_S8_UINT, VK_FORMAT_D32_SFLOAT_S8_UINT});
|
||||
} else if (depthStencilFormat == VK_FORMAT_D32_SFLOAT) {
|
||||
depthStencilFormat = FindSupportedDepthStencilFormat({VK_FORMAT_D32_SFLOAT, VK_FORMAT_D16_UNORM});
|
||||
}
|
||||
const Bool hasDepthStencil = (depthStencilFormat != VK_FORMAT_UNDEFINED);
|
||||
if (requestedDepthStencil && !hasDepthStencil) {
|
||||
MGLOG_W("VkFramebufferManager: FBO %u depth/stencil attachment exists but format is unsupported",
|
||||
glFbo.GetExternalIndex());
|
||||
}
|
||||
if (hasDepthStencil) {
|
||||
VkImageCreateInfo depthImageInfo{};
|
||||
depthImageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
|
||||
depthImageInfo.imageType = VK_IMAGE_TYPE_2D;
|
||||
depthImageInfo.extent.width = static_cast<Uint32>(size.x());
|
||||
depthImageInfo.extent.height = static_cast<Uint32>(size.y());
|
||||
depthImageInfo.extent.depth = 1;
|
||||
depthImageInfo.mipLevels = 1;
|
||||
depthImageInfo.arrayLayers = 1;
|
||||
depthImageInfo.format = depthStencilFormat;
|
||||
depthImageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
depthImageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
depthImageInfo.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
|
||||
depthImageInfo.samples = VK_SAMPLE_COUNT_1_BIT;
|
||||
depthImageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateImage(m_device, &depthImageInfo, nullptr, &target.depthStencilImage),
|
||||
"vkCreateImage(offscreen depth/stencil)");
|
||||
|
||||
VkMemoryRequirements depthMemoryRequirements{};
|
||||
vkGetImageMemoryRequirements(m_device, target.depthStencilImage, &depthMemoryRequirements);
|
||||
VkMemoryAllocateInfo depthAllocInfo{};
|
||||
depthAllocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
depthAllocInfo.allocationSize = depthMemoryRequirements.size;
|
||||
depthAllocInfo.memoryTypeIndex =
|
||||
FindMemoryType(depthMemoryRequirements.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(m_device, &depthAllocInfo, nullptr, &target.depthStencilMemory),
|
||||
"vkAllocateMemory(offscreen depth/stencil)");
|
||||
VK_VERIFY(vkBindImageMemory(m_device, target.depthStencilImage, target.depthStencilMemory, 0),
|
||||
"vkBindImageMemory(offscreen depth/stencil)");
|
||||
|
||||
VkImageAspectFlags depthAspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
||||
if (depthStencilFormat == VK_FORMAT_D24_UNORM_S8_UINT ||
|
||||
depthStencilFormat == VK_FORMAT_D32_SFLOAT_S8_UINT) {
|
||||
depthAspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
|
||||
}
|
||||
VkImageViewCreateInfo depthViewInfo{};
|
||||
depthViewInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
|
||||
depthViewInfo.image = target.depthStencilImage;
|
||||
depthViewInfo.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
depthViewInfo.format = depthStencilFormat;
|
||||
depthViewInfo.subresourceRange.aspectMask = depthAspectMask;
|
||||
depthViewInfo.subresourceRange.baseMipLevel = 0;
|
||||
depthViewInfo.subresourceRange.levelCount = 1;
|
||||
depthViewInfo.subresourceRange.baseArrayLayer = 0;
|
||||
depthViewInfo.subresourceRange.layerCount = 1;
|
||||
VK_VERIFY(vkCreateImageView(m_device, &depthViewInfo, nullptr, &target.depthStencilImageView),
|
||||
"vkCreateImageView(offscreen depth/stencil)");
|
||||
}
|
||||
|
||||
target.layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
target.extent = {static_cast<Uint32>(size.x()), static_cast<Uint32>(size.y())};
|
||||
target.format = format;
|
||||
target.depthStencilLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
target.depthStencilFormat = depthStencilFormat;
|
||||
target.glObjectVersion = glObjectVersion;
|
||||
target.colorTextureExternalIndex = (colorAttachment.IsTexture() && colorAttachment.GetTexture())
|
||||
? colorAttachment.GetTexture()->GetExternalIndex()
|
||||
: 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkFramebufferManager::DestroyOffscreenColorTarget(OffscreenColorTarget& target) {
|
||||
if (target.imageView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(m_device, target.imageView, nullptr);
|
||||
target.imageView = VK_NULL_HANDLE;
|
||||
}
|
||||
if (target.depthStencilImageView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(m_device, target.depthStencilImageView, nullptr);
|
||||
target.depthStencilImageView = VK_NULL_HANDLE;
|
||||
}
|
||||
if (target.image != VK_NULL_HANDLE) {
|
||||
vkDestroyImage(m_device, target.image, nullptr);
|
||||
target.image = VK_NULL_HANDLE;
|
||||
}
|
||||
if (target.depthStencilImage != VK_NULL_HANDLE) {
|
||||
vkDestroyImage(m_device, target.depthStencilImage, nullptr);
|
||||
target.depthStencilImage = VK_NULL_HANDLE;
|
||||
}
|
||||
if (target.memory != VK_NULL_HANDLE) {
|
||||
vkFreeMemory(m_device, target.memory, nullptr);
|
||||
target.memory = VK_NULL_HANDLE;
|
||||
}
|
||||
if (target.depthStencilMemory != VK_NULL_HANDLE) {
|
||||
vkFreeMemory(m_device, target.depthStencilMemory, nullptr);
|
||||
target.depthStencilMemory = VK_NULL_HANDLE;
|
||||
}
|
||||
target.layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
target.depthStencilLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
target.extent = {0, 0};
|
||||
target.format = VK_FORMAT_UNDEFINED;
|
||||
target.depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
target.glObjectVersion = 0;
|
||||
target.colorTextureExternalIndex = 0;
|
||||
}
|
||||
|
||||
Bool VkFramebufferManager::TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image,
|
||||
VkImageLayout& trackedLayout, VkImageLayout newLayout,
|
||||
VkPipelineStageFlags srcStageMask,
|
||||
VkPipelineStageFlags dstStageMask, VkAccessFlags srcAccessMask,
|
||||
VkAccessFlags dstAccessMask, VkImageAspectFlags aspectMask) {
|
||||
if (image == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
if (trackedLayout == newLayout) {
|
||||
return true;
|
||||
}
|
||||
|
||||
VkImageMemoryBarrier barrier{};
|
||||
barrier.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
barrier.srcAccessMask = srcAccessMask;
|
||||
barrier.dstAccessMask = dstAccessMask;
|
||||
barrier.oldLayout = trackedLayout;
|
||||
barrier.newLayout = newLayout;
|
||||
barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
barrier.image = image;
|
||||
barrier.subresourceRange.aspectMask = aspectMask;
|
||||
barrier.subresourceRange.baseMipLevel = 0;
|
||||
barrier.subresourceRange.levelCount = 1;
|
||||
barrier.subresourceRange.baseArrayLayer = 0;
|
||||
barrier.subresourceRange.layerCount = 1;
|
||||
vkCmdPipelineBarrier(commandBuffer, srcStageMask, dstStageMask, 0, 0, nullptr, 0, nullptr, 1, &barrier);
|
||||
|
||||
trackedLayout = newLayout;
|
||||
return true;
|
||||
}
|
||||
|
||||
Uint32 VkFramebufferManager::FindMemoryType(Uint32 typeFilter, VkMemoryPropertyFlags properties) const {
|
||||
VkPhysicalDeviceMemoryProperties memoryProperties{};
|
||||
vkGetPhysicalDeviceMemoryProperties(m_physicalDevice, &memoryProperties);
|
||||
for (Uint32 i = 0; i < memoryProperties.memoryTypeCount; ++i) {
|
||||
if ((typeFilter & (1U << i)) &&
|
||||
(memoryProperties.memoryTypes[i].propertyFlags & properties) == properties) {
|
||||
return i;
|
||||
}
|
||||
}
|
||||
MOBILEGL_ASSERT(false, "VkFramebufferManager::FindMemoryType failed");
|
||||
return 0;
|
||||
}
|
||||
|
||||
VkFormat VkFramebufferManager::ResolveColorFormat(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& colorAttachment) {
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::Unknown;
|
||||
if (colorAttachment.IsTexture()) {
|
||||
const auto texture = colorAttachment.GetTexture();
|
||||
internalFormat = texture ? texture->GetFormat() : TextureInternalFormat::Unknown;
|
||||
} else if (colorAttachment.IsRenderbuffer()) {
|
||||
const auto renderbuffer = colorAttachment.GetRenderbuffer();
|
||||
internalFormat = renderbuffer ? renderbuffer->GetInternalFormat() : TextureInternalFormat::Unknown;
|
||||
}
|
||||
|
||||
switch (internalFormat) {
|
||||
case TextureInternalFormat::RGBA:
|
||||
case TextureInternalFormat::RGBA8:
|
||||
case TextureInternalFormat::SRGB8Alpha8:
|
||||
return VK_FORMAT_R8G8B8A8_UNORM;
|
||||
default:
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VkFramebufferManager::ResolveDepthStencilFormat(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& depthAttachment,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& stencilAttachment) {
|
||||
const auto resolveAttachmentFormat = [](const MG_State::GLState::FramebufferAttachmentObject& attachment) {
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::Unknown;
|
||||
if (attachment.IsTexture()) {
|
||||
const auto texture = attachment.GetTexture();
|
||||
internalFormat = texture ? texture->GetFormat() : TextureInternalFormat::Unknown;
|
||||
} else if (attachment.IsRenderbuffer()) {
|
||||
const auto renderbuffer = attachment.GetRenderbuffer();
|
||||
internalFormat = renderbuffer ? renderbuffer->GetInternalFormat() : TextureInternalFormat::Unknown;
|
||||
}
|
||||
return internalFormat;
|
||||
};
|
||||
|
||||
const auto depthFormat = resolveAttachmentFormat(depthAttachment);
|
||||
const auto stencilFormat = resolveAttachmentFormat(stencilAttachment);
|
||||
|
||||
switch (depthFormat) {
|
||||
case TextureInternalFormat::Depth24Stencil8:
|
||||
case TextureInternalFormat::Depth32FStencil8:
|
||||
case TextureInternalFormat::DepthStencil:
|
||||
return VK_FORMAT_D24_UNORM_S8_UINT;
|
||||
case TextureInternalFormat::DepthComponent16:
|
||||
return VK_FORMAT_D16_UNORM;
|
||||
case TextureInternalFormat::DepthComponent24:
|
||||
case TextureInternalFormat::DepthComponent32:
|
||||
case TextureInternalFormat::DepthComponent32F:
|
||||
case TextureInternalFormat::DepthComponent:
|
||||
return VK_FORMAT_D32_SFLOAT;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
switch (stencilFormat) {
|
||||
case TextureInternalFormat::Depth24Stencil8:
|
||||
case TextureInternalFormat::Depth32FStencil8:
|
||||
case TextureInternalFormat::DepthStencil:
|
||||
return VK_FORMAT_D24_UNORM_S8_UINT;
|
||||
default:
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
VkFormat VkFramebufferManager::FindSupportedDepthStencilFormat(const Vector<VkFormat>& candidates) const {
|
||||
for (auto format : candidates) {
|
||||
VkFormatProperties properties{};
|
||||
vkGetPhysicalDeviceFormatProperties(m_physicalDevice, format, &properties);
|
||||
if ((properties.optimalTilingFeatures & VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT) != 0) {
|
||||
return format;
|
||||
}
|
||||
}
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -1,83 +0,0 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkFramebufferManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/FramebufferState/FramebufferObject.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkFramebufferManager {
|
||||
public:
|
||||
struct InitInfo {
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice physicalDevice = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
VkFramebufferManager() = default;
|
||||
~VkFramebufferManager() = default;
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
|
||||
Bool EnsureOffscreenColorTarget(Uint glFboExternalIndex, const MG_State::GLState::FramebufferObject& glFbo);
|
||||
Bool TransitionOffscreenColorToAttachment(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenColorToTransferSrc(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenColorToTransferDst(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenColorToGeneral(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenDepthStencilToTransferSrc(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenDepthStencilToTransferDst(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenDepthStencilToGeneral(VkCommandBuffer commandBuffer, Uint glFboExternalIndex);
|
||||
Bool TransitionOffscreenColorTextureToShaderRead(VkCommandBuffer commandBuffer, Uint textureExternalIndex);
|
||||
Bool GetOffscreenColorImage(Uint glFboExternalIndex, VkImage& outImage, VkExtent2D& outExtent) const;
|
||||
Bool GetOffscreenDepthStencilImage(Uint glFboExternalIndex, VkImage& outImage, VkExtent2D& outExtent,
|
||||
VkFormat& outFormat) const;
|
||||
Bool GetOffscreenColorViewByTexture(Uint textureExternalIndex, VkImageView& outImageView) const;
|
||||
Bool GetOffscreenRenderSurface(Uint glFboExternalIndex, VkImageView& outColorView, VkFormat& outColorFormat,
|
||||
VkImageView& outDepthStencilView, VkFormat& outDepthStencilFormat,
|
||||
VkExtent2D& outExtent) const;
|
||||
|
||||
private:
|
||||
struct OffscreenColorTarget {
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VkDeviceMemory memory = VK_NULL_HANDLE;
|
||||
VkImageView imageView = VK_NULL_HANDLE;
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkExtent2D extent = {0, 0};
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
VkImage depthStencilImage = VK_NULL_HANDLE;
|
||||
VkDeviceMemory depthStencilMemory = VK_NULL_HANDLE;
|
||||
VkImageView depthStencilImageView = VK_NULL_HANDLE;
|
||||
VkImageLayout depthStencilLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkFormat depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
Uint16 glObjectVersion = 0;
|
||||
Uint colorTextureExternalIndex = 0;
|
||||
};
|
||||
|
||||
Bool RecreateOffscreenColorTarget(OffscreenColorTarget& target,
|
||||
const MG_State::GLState::FramebufferObject& glFbo,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& colorAttachment,
|
||||
Uint16 glObjectVersion);
|
||||
void DestroyOffscreenColorTarget(OffscreenColorTarget& target);
|
||||
Bool TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image, VkImageLayout& trackedLayout,
|
||||
VkImageLayout newLayout, VkPipelineStageFlags srcStageMask,
|
||||
VkPipelineStageFlags dstStageMask, VkAccessFlags srcAccessMask,
|
||||
VkAccessFlags dstAccessMask, VkImageAspectFlags aspectMask);
|
||||
Uint32 FindMemoryType(Uint32 typeFilter, VkMemoryPropertyFlags properties) const;
|
||||
static VkFormat ResolveColorFormat(const MG_State::GLState::FramebufferAttachmentObject& colorAttachment);
|
||||
static VkFormat ResolveDepthStencilFormat(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& depthAttachment,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& stencilAttachment);
|
||||
VkFormat FindSupportedDepthStencilFormat(const Vector<VkFormat>& candidates) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
UnorderedMap<Uint, OffscreenColorTarget> m_offscreenColorTargets;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
File diff suppressed because it is too large
Load Diff
@@ -8,79 +8,395 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "SwapchainObject.h"
|
||||
#include "VkClearManager.h"
|
||||
#include "VkTextureManager.h"
|
||||
#include "../VkIncludes.h"
|
||||
#include "../VulkanRendererConfig.h"
|
||||
#include "MG_State/GLState/FramebufferState/FramebufferObject.h"
|
||||
|
||||
#include <Includes.h>
|
||||
#include <unordered_map>
|
||||
#include <vk_mem_alloc.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class TrackedAttachmentTarget : Uint8 {
|
||||
Texture,
|
||||
Renderbuffer,
|
||||
SwapchainColor,
|
||||
SwapchainDepthStencil
|
||||
};
|
||||
|
||||
struct PendingClearAttachmentInfo {
|
||||
// Index into the render pass attachment descriptions (VkRenderPassBeginInfo::pClearValues space).
|
||||
Uint32 attachmentIndex = 0;
|
||||
// Index into the subpass pColorAttachments (VkClearAttachment::colorAttachment space) — the GL
|
||||
// draw-buffer slot. Differs from attachmentIndex when earlier slots are GL_NONE/incomplete.
|
||||
// Only meaningful for color clears.
|
||||
Uint32 colorAttachmentSlot = 0;
|
||||
PendingClearKey key{};
|
||||
MG_State::GLState::RenderbufferObject* renderbuffer = nullptr;
|
||||
Bool hasInlinePayload = false;
|
||||
ClearAttachmentPayload inlinePayload{};
|
||||
};
|
||||
|
||||
struct TrackedAttachmentLayoutInfo {
|
||||
TrackedAttachmentTarget target = TrackedAttachmentTarget::Texture;
|
||||
WeakPtr<MG_State::GLState::ITextureObject> texture;
|
||||
// Identity-compare shortcut for the per-draw "does the active pass use
|
||||
// this sampled texture" probe: comparing this against a LIVE texture's
|
||||
// address needs no weak_ptr::lock (two refcount atomics per probe).
|
||||
// May dangle once the texture dies - compare only, never dereference.
|
||||
MG_State::GLState::ITextureObject* textureRaw = nullptr;
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
Uint32 textureMipLevel = 0;
|
||||
Uint32 swapchainImageIndex = 0;
|
||||
VkImageLayout finalLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
};
|
||||
|
||||
struct DepthStencilAttachmentLoadInfo {
|
||||
VkAttachmentLoadOp depthLoadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
|
||||
VkAttachmentLoadOp stencilLoadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
|
||||
VkImageLayout initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
};
|
||||
|
||||
DepthStencilAttachmentLoadInfo ResolveDepthStencilAttachmentLoadInfo(
|
||||
VkImageLayout trackedLayout, Bool clearDepth, Bool clearStencil);
|
||||
IntVec2 ResolveRenderPassFramebufferExtent(Bool isDefaultFbo, const TextureSize& attachmentExtent,
|
||||
VkExtent2D swapchainExtent);
|
||||
|
||||
struct RenderPassEntry {
|
||||
static inline VkDevice s_device;
|
||||
static inline Vector<VkTextureManager::TextureResource*> s_textureResourcesScratch;
|
||||
Uint64 hash = 0;
|
||||
VkRenderPass renderPass = VK_NULL_HANDLE;
|
||||
VkFramebuffer framebuffer = VK_NULL_HANDLE;
|
||||
Uint64 compatibilityHash = 0;
|
||||
Vector<PendingClearAttachmentInfo> pendingClearAttachments;
|
||||
Vector<TrackedAttachmentLayoutInfo> trackedAttachmentLayouts;
|
||||
Uint32 attachmentCount = 0;
|
||||
Uint32 colorAttachmentCount = 0;
|
||||
Bool hasDepthStencilAttachment = false;
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
IntVec2 extent = {0, 0};
|
||||
// VkFramebufferCreateInfo::layers of the entry's framebuffer (>1 for layered GL attachments).
|
||||
Uint32 layers = 1;
|
||||
// Frame counter value of the last GetOrCreateRenderPass hit; drives cache eviction.
|
||||
Uint64 lastUsedFrame = 0;
|
||||
|
||||
RenderPassEntry() = default;
|
||||
RenderPassEntry(const RenderPassEntry&) = delete;
|
||||
RenderPassEntry(RenderPassEntry&& that) noexcept {
|
||||
std::swap(hash, that.hash);
|
||||
std::swap(renderPass, that.renderPass);
|
||||
std::swap(framebuffer, that.framebuffer);
|
||||
std::swap(compatibilityHash, that.compatibilityHash);
|
||||
std::swap(pendingClearAttachments, that.pendingClearAttachments);
|
||||
std::swap(trackedAttachmentLayouts, that.trackedAttachmentLayouts);
|
||||
std::swap(attachmentCount, that.attachmentCount);
|
||||
std::swap(colorAttachmentCount, that.colorAttachmentCount);
|
||||
std::swap(hasDepthStencilAttachment, that.hasDepthStencilAttachment);
|
||||
std::swap(sampleCount, that.sampleCount);
|
||||
std::swap(extent, that.extent);
|
||||
std::swap(layers, that.layers);
|
||||
std::swap(lastUsedFrame, that.lastUsedFrame);
|
||||
}
|
||||
// Move ASSIGNMENT, not just construction. The move constructor above and the
|
||||
// destructor below each independently suppress the implicit one, which left the
|
||||
// type move-constructible but not move-assignable - and therefore not swappable,
|
||||
// which std::swap(pair&, pair&) requires. That was invisible while UnorderedMap
|
||||
// only ever move-CONSTRUCTED an element into a fresh slot. ska::flat_hash_map
|
||||
// probes robin-hood: inserting swaps the entry being placed against the one
|
||||
// already sitting in the slot whenever it has travelled further from its desired
|
||||
// position, so the mapped type has to be swappable or the table fails to
|
||||
// instantiate at all.
|
||||
//
|
||||
// SWAP SEMANTICS, exactly like the move constructor: this does not release the
|
||||
// destination's handles, it parks them in `that`, which destroys them when it
|
||||
// dies. That is correct for the only caller - std::swap, whose temporary expires
|
||||
// immediately - and it is what keeps the three-move sequence from destroying a
|
||||
// live render pass. It is NOT correct for a hand-written `a = std::move(b)` where
|
||||
// `a` held live handles and `b` outlives the statement: those handles would then
|
||||
// survive until `b` dies. There is no such caller; add a destroy-then-steal
|
||||
// assignment before writing one.
|
||||
RenderPassEntry& operator=(RenderPassEntry&& that) noexcept {
|
||||
if (this != &that) {
|
||||
std::swap(hash, that.hash);
|
||||
std::swap(renderPass, that.renderPass);
|
||||
std::swap(framebuffer, that.framebuffer);
|
||||
std::swap(compatibilityHash, that.compatibilityHash);
|
||||
std::swap(pendingClearAttachments, that.pendingClearAttachments);
|
||||
std::swap(trackedAttachmentLayouts, that.trackedAttachmentLayouts);
|
||||
std::swap(attachmentCount, that.attachmentCount);
|
||||
std::swap(colorAttachmentCount, that.colorAttachmentCount);
|
||||
std::swap(hasDepthStencilAttachment, that.hasDepthStencilAttachment);
|
||||
std::swap(sampleCount, that.sampleCount);
|
||||
std::swap(extent, that.extent);
|
||||
std::swap(layers, that.layers);
|
||||
std::swap(lastUsedFrame, that.lastUsedFrame);
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
RenderPassEntry(
|
||||
Uint64 hash,
|
||||
VkRenderPass renderpass,
|
||||
VkFramebuffer framebuffer,
|
||||
Uint64 compatibilityHash,
|
||||
const Vector<PendingClearAttachmentInfo>& pendingClearAttachments,
|
||||
const Vector<TrackedAttachmentLayoutInfo>& trackedAttachmentLayouts,
|
||||
Uint32 attachmentCount,
|
||||
Uint32 colorAttachmentCount,
|
||||
Bool hasDepthStencilAttachment,
|
||||
VkSampleCountFlagBits sampleCount,
|
||||
IntVec2 extent, Uint32 layers):
|
||||
hash(hash),
|
||||
renderPass(renderpass),
|
||||
framebuffer(framebuffer),
|
||||
compatibilityHash(compatibilityHash),
|
||||
pendingClearAttachments(Move(pendingClearAttachments)),
|
||||
trackedAttachmentLayouts(Move(trackedAttachmentLayouts)),
|
||||
attachmentCount(attachmentCount),
|
||||
colorAttachmentCount(colorAttachmentCount),
|
||||
hasDepthStencilAttachment(hasDepthStencilAttachment),
|
||||
sampleCount(sampleCount),
|
||||
extent(extent),
|
||||
layers(layers)
|
||||
{}
|
||||
|
||||
~RenderPassEntry() {
|
||||
if (renderPass != VK_NULL_HANDLE) {
|
||||
vkDestroyRenderPass(s_device, renderPass, nullptr);
|
||||
}
|
||||
if (framebuffer != VK_NULL_HANDLE) {
|
||||
vkDestroyFramebuffer(s_device, framebuffer, nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
Bool CompatibleWith(const RenderPassEntry& that) const {
|
||||
return this->compatibilityHash == that.compatibilityHash;
|
||||
}
|
||||
|
||||
Bool CompatibleWith(Uint64 compatibilityHash) const {
|
||||
return this->compatibilityHash == compatibilityHash;
|
||||
}
|
||||
};
|
||||
|
||||
struct ActiveRenderPassInfo {
|
||||
Uint64 hash = 0;
|
||||
Uint64 compatibilityHash = 0;
|
||||
Vector<TrackedAttachmentLayoutInfo> trackedAttachmentLayouts;
|
||||
IntVec2 extent = {0, 0};
|
||||
|
||||
Bool CompatibleWith(const RenderPassEntry& that) const {
|
||||
return compatibilityHash == that.compatibilityHash;
|
||||
}
|
||||
|
||||
Bool CompatibleWith(Uint64 thatCompatibilityHash) const {
|
||||
return compatibilityHash == thatCompatibilityHash;
|
||||
}
|
||||
};
|
||||
|
||||
class VkRenderPassManager {
|
||||
public:
|
||||
struct InitInfo {
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
VkFormat colorFormat = VK_FORMAT_UNDEFINED;
|
||||
VkFormat depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
using HashType = Uint64;
|
||||
|
||||
// Notified once per OnPresent sweep with every aged-out entry's VkRenderPass
|
||||
// value: pipelines are hashed on the raw handle, and once destroyed the value
|
||||
// may be recycled for an incompatible pass, so dependent caches must purge
|
||||
// everything keyed on them before any new pass can be created (the sweep and
|
||||
// the notification run back-to-back with no creation in between; observers
|
||||
// compare the values, never dereference them). Batched so a mass-idle cohort
|
||||
// (shader-pack switch, dimension exit) costs the observer one pipeline-cache
|
||||
// scan, not one per dying pass. The wholesale paths
|
||||
// (Shutdown/RecreateSwapchain) do not notify - their callers already drop
|
||||
// every pipeline outright.
|
||||
class IEvictionObserver {
|
||||
public:
|
||||
virtual ~IEvictionObserver() = default;
|
||||
virtual void OnRenderPassesDestroyed(const Vector<VkRenderPass>& renderPasses) = 0;
|
||||
};
|
||||
|
||||
struct OffscreenRenderTargetInfo {
|
||||
Uint targetExternalIndex = 0;
|
||||
Uint16 targetVersion = 0;
|
||||
VkImageView colorView = VK_NULL_HANDLE;
|
||||
VkFormat colorFormat = VK_FORMAT_UNDEFINED;
|
||||
VkImageView depthStencilView = VK_NULL_HANDLE;
|
||||
VkFormat depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
VkExtent2D extent = {0, 0};
|
||||
};
|
||||
VkRenderPassManager(VkDevice device,
|
||||
VkPhysicalDevice physicalDevice, VmaAllocator allocator, const VulkanRendererConfig& config,
|
||||
VkClearManager& clearManager, VkTextureManager& textureManager, SwapchainObject& swapchainObject);
|
||||
~VkRenderPassManager();
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
// Observer may be null (no notifications). Not owned.
|
||||
void SetEvictionObserver(IEvictionObserver* observer) { m_evictionObserver = observer; }
|
||||
|
||||
Bool Initialize();
|
||||
void Shutdown();
|
||||
|
||||
Bool RecreateDefaultFramebuffers(const Vector<VkImageView>& colorViews,
|
||||
const Vector<VkImageView>& depthStencilViews, VkExtent2D extent);
|
||||
Bool GetDefaultRenderTarget(Uint32 imageIndex, VkRenderPass& outRenderPass, VkFramebuffer& outFramebuffer,
|
||||
VkExtent2D& outExtent, VkFormat& outDepthStencilFormat) const;
|
||||
HashType ComputeHash(
|
||||
const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool includePendingClear = true,
|
||||
Bool includeDefaultFboDepthStencil = true);
|
||||
// drawUsesDepthStencil: whether the operation about to run inside the pass
|
||||
// reads or writes the depth/stencil buffer (depth test or stencil test
|
||||
// enabled, or a depth/stencil clear). Only consulted for the DEFAULT
|
||||
// framebuffer: EGL undefines its ancillary buffers at every swap, so a
|
||||
// default-FBO pass whose draws provably never touch depth/stencil is
|
||||
// created WITHOUT the depth attachment - on a tiler that skips the whole
|
||||
// depth tile load AND store. The flavor only escalates: once a pass with
|
||||
// depth is active, later depth-less draws keep using it, and a depth-using
|
||||
// draw against a depth-less active pass resolves to a new (incompatible)
|
||||
// entry, which the caller's compatibility check turns into a pass split;
|
||||
// the new pass's depth loads DONT_CARE (content was undefined all along).
|
||||
RenderPassEntry& GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil = true);
|
||||
void QueueRenderbufferClear(GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo);
|
||||
void QueueRenderbufferClear(const ClearAttachmentPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment);
|
||||
void PopPendingRenderbufferClear(MG_State::GLState::RenderbufferObject* renderbuffer);
|
||||
// Frame boundary hook: ages the render-pass cache and evicts long-unused
|
||||
// entries (their command buffers retired many frames ago).
|
||||
void OnPresent();
|
||||
static Bool BeginRenderPass(VkCommandBuffer commandBuffer, RenderPassEntry& renderPassEntry);
|
||||
static Bool EndRenderPass(VkCommandBuffer commandBuffer);
|
||||
static ActiveRenderPassInfo* GetActiveRenderPass();
|
||||
private:
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
const VulkanRendererConfig& m_config;
|
||||
VkClearManager& m_clearManager;
|
||||
VkTextureManager& m_textureManager;
|
||||
SwapchainObject& m_swapchainObject;
|
||||
UnorderedMap<Uint64, RenderPassEntry> m_renderPasses;
|
||||
// Monotonic frame counter (bumped in OnPresent) for render-pass cache aging.
|
||||
Uint64 m_frameCounter = 0;
|
||||
IEvictionObserver* m_evictionObserver = nullptr;
|
||||
|
||||
Bool EnsureOffscreenRenderTarget(const OffscreenRenderTargetInfo& targetInfo);
|
||||
Bool GetOffscreenRenderTarget(Uint targetExternalIndex, VkRenderPass& outRenderPass,
|
||||
VkFramebuffer& outFramebuffer, VkExtent2D& outExtent,
|
||||
VkFormat& outDepthStencilFormat) const;
|
||||
void RemoveOffscreenRenderTarget(Uint targetExternalIndex);
|
||||
// Bumped whenever a renderbuffer VkImage is (re)created; together with the texture
|
||||
// manager's image epoch this invalidates the render-pass fast path on any attachment
|
||||
// image recreation.
|
||||
Uint64 m_renderbufferImageEpoch = 1;
|
||||
|
||||
void BeginRenderPass(VkCommandBuffer commandBuffer, VkRenderPass renderPass, VkFramebuffer framebuffer,
|
||||
VkExtent2D extent) const;
|
||||
void EndRenderPass(VkCommandBuffer commandBuffer) const;
|
||||
void RecordColorClear(VkCommandBuffer commandBuffer, VkExtent2D extent,
|
||||
const VkClearColorValue& clearColor) const;
|
||||
void RecordDepthStencilClear(VkCommandBuffer commandBuffer, VkExtent2D extent, GLbitfield mask, Float depth,
|
||||
Uint32 stencil, VkFormat depthStencilFormat) const;
|
||||
|
||||
VkRenderPass GetLoadRenderPass() const;
|
||||
VkRenderPass GetClearRenderPass() const;
|
||||
public:
|
||||
// Bumped whenever a renderbuffer backing is (re)created; consecutive-draw
|
||||
// snapshots include it so an attachment respecify forces a re-resolve.
|
||||
Uint64 GetRenderbufferImageEpoch() const { return m_renderbufferImageEpoch; }
|
||||
|
||||
private:
|
||||
struct OffscreenRenderTarget {
|
||||
Uint16 targetVersion = 0;
|
||||
VkImageView colorView = VK_NULL_HANDLE;
|
||||
VkFormat colorFormat = VK_FORMAT_UNDEFINED;
|
||||
VkImageView depthStencilView = VK_NULL_HANDLE;
|
||||
VkFormat depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
|
||||
// Per-draw fast-path memo for GetOrCreateRenderPass (dirty-flag state tracking): when the
|
||||
// framebuffer state is provably unchanged since the last resolution, the active render pass
|
||||
// is reused WITHOUT recomputing the expensive per-draw hash. Invalidated by FBO switch /
|
||||
// version change, swapchain rotation, any attachment image recreation (the two epochs),
|
||||
// or a pending clear. Portable to Vulkan 1.1 (no dynamic_rendering / imageless FB needed).
|
||||
Bool m_rpFastValid = false;
|
||||
const MG_State::GLState::FramebufferObject* m_rpFastFbo = nullptr;
|
||||
Uint16 m_rpFastFboVersion = 0;
|
||||
Uint32 m_rpFastSwapchainIndex = 0;
|
||||
Uint64 m_rpFastTexEpoch = 0;
|
||||
Uint64 m_rpFastRbEpoch = 0;
|
||||
Uint64 m_rpFastRenderPassHash = 0;
|
||||
// Whether the memoized entry carries a depth/stencil attachment; a
|
||||
// default-FBO resolution whose effective depth request differs must
|
||||
// miss the memo (the depth-less/depth-full flavors hash differently).
|
||||
Bool m_rpFastHadDepthStencil = false;
|
||||
|
||||
public:
|
||||
struct RenderbufferResource {
|
||||
// deadSinceFrame sentinel: the owning weak reference has not been observed
|
||||
// expired. Dead resources age past every in-flight frame before Destroy
|
||||
// (see CollectRenderbufferGarbage); the GPU may still reference the image
|
||||
// for frames-in-flight frames after the GL object dies.
|
||||
static constexpr Uint64 kNeverObservedDead = UINT64_MAX;
|
||||
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
VkImageView view = VK_NULL_HANDLE;
|
||||
// UNORM reinterpretation of an sRGB image, used as the attachment view while
|
||||
// GL_FRAMEBUFFER_SRGB is disabled (raw writes). Null for non-sRGB formats.
|
||||
VkImageView unormTwinView = VK_NULL_HANDLE;
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
VkImageAspectFlags aspect = VK_IMAGE_ASPECT_NONE;
|
||||
VkExtent2D extent = {0, 0};
|
||||
VkRenderPass renderPassLoad = VK_NULL_HANDLE;
|
||||
VkFramebuffer framebuffer = VK_NULL_HANDLE;
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
TextureInternalFormat internalFormat = TextureInternalFormat::Unknown;
|
||||
Int samples = 0;
|
||||
// m_frameCounter value at which the weak reference was first seen expired.
|
||||
Uint64 deadSinceFrame = kNeverObservedDead;
|
||||
|
||||
void Destroy(VkDevice device, VmaAllocator allocator);
|
||||
};
|
||||
|
||||
VkRenderPass CreateDefaultRenderPass(VkAttachmentLoadOp colorLoadOp) const;
|
||||
VkRenderPass CreateRenderPass(VkFormat colorFormat, VkFormat depthStencilFormat, VkAttachmentLoadOp colorLoadOp,
|
||||
VkImageLayout colorFinalLayout) const;
|
||||
void DestroyDefaultFramebuffers();
|
||||
void DestroyOffscreenRenderTarget(OffscreenRenderTarget& target);
|
||||
static Bool HasStencilComponent(VkFormat format);
|
||||
// Public so the renderer's blit/copy/readback bindings can source renderbuffer
|
||||
// attachments the same way texture attachments go through the texture manager.
|
||||
RenderbufferResource* GetOrCreateRenderbufferResource(
|
||||
const SharedPtr<MG_State::GLState::RenderbufferObject>& renderbuffer);
|
||||
Bool GetPendingRenderbufferClear(MG_State::GLState::RenderbufferObject* renderbuffer,
|
||||
ClearAttachmentPayload& outPayload) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkFormat m_colorFormat = VK_FORMAT_UNDEFINED;
|
||||
VkFormat m_depthStencilFormat = VK_FORMAT_UNDEFINED;
|
||||
VkRenderPass m_renderPassLoad = VK_NULL_HANDLE;
|
||||
VkRenderPass m_renderPassClear = VK_NULL_HANDLE;
|
||||
Vector<VkFramebuffer> m_defaultFramebuffers;
|
||||
VkExtent2D m_defaultExtent = {0, 0};
|
||||
UnorderedMap<Uint, OffscreenRenderTarget> m_offscreenRenderTargets;
|
||||
private:
|
||||
struct PendingRenderbufferClear {
|
||||
WeakPtr<MG_State::GLState::RenderbufferObject> renderbuffer;
|
||||
ClearAttachmentPayload payload{};
|
||||
};
|
||||
|
||||
// A superseded renderbuffer backing (glRenderbufferStorage respecify) parked
|
||||
// until enough frame boundaries have passed that no in-flight command buffer
|
||||
// can still reference it; destroyed in OnPresent (see RetireAgeFrames).
|
||||
struct DeferredRenderbufferRelease {
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
VkImageView view = VK_NULL_HANDLE;
|
||||
VkImageView unormTwinView = VK_NULL_HANDLE;
|
||||
Uint64 deferredAtFrame = 0;
|
||||
};
|
||||
|
||||
// Node-based std::unordered_map, deliberately NOT the open-addressing UnorderedMap:
|
||||
// callers cache a RenderbufferResource* - or a bare &resource->layout - and then make further
|
||||
// calls that touch this map. BlitFramebuffer is the one that bit: it resolves the source and
|
||||
// destination colour bindings (ResolveColorBlitBinding caches &rbResource->layout), then
|
||||
// materializes the source's pending clear, which looks that same resource up again. Growing
|
||||
// an open-addressed table relocates every element, so the cached pointer went on to name
|
||||
// freed storage still holding the pre-clear VK_IMAGE_LAYOUT_UNDEFINED; BlitFramebuffer bailed
|
||||
// out at "source image layout is undefined", silently dropping the blit -
|
||||
// renderbuffers_storage_multisample read back zero instead of the clear colour on exactly the
|
||||
// iterations that grew the table.
|
||||
//
|
||||
// Reordering the materialize ahead of the resolves - the fix ReadPixels got - does not cover
|
||||
// this: the destination resolve still runs after the source pointer is taken. The depth blit,
|
||||
// GetOrCreateRenderPass's depthRenderbufferResource and ReadDepthStencilPixels cache the same
|
||||
// kind of pointer, so the invariant belongs in the container rather than in a per-call-site
|
||||
// ordering rule. m_textureResources is node-based for the same reason.
|
||||
//
|
||||
// The case for keeping this node-based got STRONGER with ska::flat_hash_map, so do not read
|
||||
// the paragraph above as merely historical: ska erases by shifting the rest of the probe
|
||||
// cluster backwards into the hole, so erasing one renderbuffer relocates OTHER renderbuffers'
|
||||
// entries - a cached pointer can now be invalidated by a key it has nothing to do with, which
|
||||
// no call-site ordering rule can defend against. (What did change: ska's operator[] returns on
|
||||
// a hit before it runs its grow check, so a plain lookup of a PRESENT key no longer relocates.
|
||||
// That narrows the insert hazard; it does not touch the erase one.)
|
||||
std::unordered_map<MG_State::GLState::RenderbufferObject*, RenderbufferResource> m_renderbufferResources;
|
||||
UnorderedMap<MG_State::GLState::RenderbufferObject*, PendingRenderbufferClear> m_pendingRenderbufferClears;
|
||||
Vector<DeferredRenderbufferRelease> m_deferredRenderbufferReleases;
|
||||
// Supported sample counts per attachment format, so per-draw resource lookups
|
||||
// do not repeat vkGetPhysicalDeviceImageFormatProperties.
|
||||
UnorderedMap<VkFormat, VkSampleCountFlags> m_attachmentSampleCountsByFormat;
|
||||
|
||||
Bool HasPendingRenderbufferClear(
|
||||
const MG_State::GLState::FramebufferAttachmentObject& attachment) const;
|
||||
void CollectRenderbufferGarbage();
|
||||
// Frame-boundary margin after which a resource last referenced by a retired
|
||||
// GL object (or superseded backing) is provably past every in-flight frame.
|
||||
Uint64 RetireAgeFrames() const;
|
||||
void DeferRenderbufferBackingRelease(RenderbufferResource& resource);
|
||||
void CollectDeferredRenderbufferReleases(Bool destroyAll);
|
||||
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
static inline ActiveRenderPassInfo s_activeRenderPass{};
|
||||
static inline Bool s_hasActiveRenderPass = false;
|
||||
static inline VkClearManager* s_clearManager = nullptr;
|
||||
static inline VkTextureManager* s_textureManager = nullptr;
|
||||
static inline SwapchainObject* s_swapchainObject = nullptr;
|
||||
static inline VkRenderPassManager* s_renderPassManager = nullptr;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -0,0 +1,319 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkSamplerManager.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "VkSamplerManager.h"
|
||||
|
||||
#include "MG_State/GLState/Core.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
Bool UsesBorderColor(const MG_State::GLState::SamplerObject& sampler) {
|
||||
return sampler.GetWrapS() == SamplerWrapMode::ClampToBorder ||
|
||||
sampler.GetWrapT() == SamplerWrapMode::ClampToBorder ||
|
||||
sampler.GetWrapR() == SamplerWrapMode::ClampToBorder;
|
||||
}
|
||||
|
||||
Bool IsDepthTextureFormat(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::DepthComponent:
|
||||
case TextureInternalFormat::DepthComponent16:
|
||||
case TextureInternalFormat::DepthComponent24:
|
||||
case TextureInternalFormat::DepthComponent32:
|
||||
case TextureInternalFormat::DepthComponent32F:
|
||||
case TextureInternalFormat::Depth24Stencil8:
|
||||
case TextureInternalFormat::Depth32FStencil8:
|
||||
case TextureInternalFormat::DepthStencil:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
Bool NearlyEqual(Float lhs, Float rhs) {
|
||||
return std::fabs(lhs - rhs) <= 1e-6f;
|
||||
}
|
||||
|
||||
Float ResolveEffectiveMaxLod(const MG_State::GLState::SamplerObject& sampler) {
|
||||
if (sampler.GetMipmapMode() == SamplerMipmapMode::None) {
|
||||
return 0.0f;
|
||||
}
|
||||
return sampler.GetMaxLod();
|
||||
}
|
||||
|
||||
Float ResolveEffectiveMinLod(const MG_State::GLState::SamplerObject& sampler, Float effectiveMaxLod) {
|
||||
return std::min(sampler.GetMinLod(), effectiveMaxLod);
|
||||
}
|
||||
|
||||
// A single-level view can only ever deliver the base level, but the LOD clamp must not be
|
||||
// collapsed to exactly 0: both GL and Vulkan pick magFilter over minFilter from the
|
||||
// *clamped* lambda, so maxLod = 0 would make every fragment magnify and quietly retire the
|
||||
// min filter. 0.25 is the value VkSamplerCreateInfo's own note prescribes for emulating
|
||||
// GL's non-mipmapped minification - large enough for lambda to stay positive, small enough
|
||||
// that a NEAREST mip mode still rounds down to level 0. Clamped rather than assigned, so a
|
||||
// texture whose GL_TEXTURE_MAX_LOD really is 0 keeps magnifying as GL says it must.
|
||||
Float ResolveSingleLevelMaxLod(const MG_State::GLState::SamplerObject& sampler, Bool singleLevelView) {
|
||||
const Float maxLod = ResolveEffectiveMaxLod(sampler);
|
||||
return singleLevelView ? std::min(maxLod, 0.25f) : maxLod;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool VkSamplerManager::Initialize(const InitInfo& initInfo) {
|
||||
Shutdown();
|
||||
|
||||
m_device = initInfo.device;
|
||||
m_config = initInfo.config;
|
||||
m_samplerAnisotropySupported = initInfo.samplerAnisotropySupported;
|
||||
m_maxSamplerAnisotropy = std::max(initInfo.maxSamplerAnisotropy, 1.0f);
|
||||
MOBILEGL_ASSERT(m_device != VK_NULL_HANDLE && m_config != nullptr,
|
||||
"VkSamplerManager::Initialize failed: invalid initialization info");
|
||||
return true;
|
||||
}
|
||||
|
||||
Float VkSamplerManager::ResolveEffectiveMaxAnisotropy(const MG_State::GLState::SamplerObject& sampler,
|
||||
Bool forceNearestFiltering) const {
|
||||
if (!m_samplerAnisotropySupported) return 1.0f;
|
||||
if (forceNearestFiltering) return 1.0f;
|
||||
// VUID-VkSamplerCreateInfo-anisotropyEnable-01071/01072: anisotropy requires both filters to
|
||||
// be LINEAR and the value to sit within [1, limits.maxSamplerAnisotropy].
|
||||
if (sampler.GetMinFilter() != SamplerFilterMode::Linear ||
|
||||
sampler.GetMagFilter() != SamplerFilterMode::Linear) {
|
||||
return 1.0f;
|
||||
}
|
||||
return std::clamp(sampler.GetMaxAnisotropy(), 1.0f, m_maxSamplerAnisotropy);
|
||||
}
|
||||
|
||||
void VkSamplerManager::Shutdown() {
|
||||
for (auto& [_, sampler] : m_samplers) {
|
||||
if (m_device != VK_NULL_HANDLE && sampler.handle != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, sampler.handle, nullptr);
|
||||
}
|
||||
sampler.handle = VK_NULL_HANDLE;
|
||||
}
|
||||
m_samplers.clear();
|
||||
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_config = nullptr;
|
||||
m_frameBoundaryCounter = 0;
|
||||
}
|
||||
|
||||
void VkSamplerManager::OnFrameBoundary() {
|
||||
++m_frameBoundaryCounter;
|
||||
|
||||
// Sweep occasionally; destroy samplers whose last use is far past every
|
||||
// in-flight frame. Destroy and erase must stay atomic, or Shutdown would
|
||||
// double-free the handle; an evicted key that recurs simply re-creates
|
||||
// its sampler on the next miss.
|
||||
constexpr Uint64 kSweepInterval = 256;
|
||||
constexpr Uint64 kRetireAgeBoundaries = 1024;
|
||||
if ((m_frameBoundaryCounter % kSweepInterval) != 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (auto it = m_samplers.begin(); it != m_samplers.end();) {
|
||||
auto& entry = it->second;
|
||||
if (m_frameBoundaryCounter - entry.lastUsedFrameBoundary > kRetireAgeBoundaries) {
|
||||
if (m_device != VK_NULL_HANDLE && entry.handle != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, entry.handle, nullptr);
|
||||
}
|
||||
it = m_samplers.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Uint64 VkSamplerManager::BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const {
|
||||
MOBILEGL_ASSERT(m_config != nullptr, "VkSamplerManager::BuildSamplerKey: m_config is null");
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, m_config->CacheVersion));
|
||||
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &forceNearestFiltering, sizeof(forceNearestFiltering)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &singleLevelView, sizeof(singleLevelView)));
|
||||
|
||||
const auto minFilter = sampler.GetMinFilter();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &minFilter, sizeof(minFilter)));
|
||||
const auto magFilter = sampler.GetMagFilter();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &magFilter, sizeof(magFilter)));
|
||||
const auto mipmapMode = sampler.GetMipmapMode();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &mipmapMode, sizeof(mipmapMode)));
|
||||
const auto wrapS = sampler.GetWrapS();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapS, sizeof(wrapS)));
|
||||
const auto wrapT = sampler.GetWrapT();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapT, sizeof(wrapT)));
|
||||
const auto wrapR = sampler.GetWrapR();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &wrapR, sizeof(wrapR)));
|
||||
const auto maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
const auto minLod = ResolveEffectiveMinLod(sampler, maxLod);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &minLod, sizeof(minLod)));
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &maxLod, sizeof(maxLod)));
|
||||
const auto lodBias = sampler.GetLodBias();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &lodBias, sizeof(lodBias)));
|
||||
// The RESOLVED value, not the GL request: samplers that only differ in an anisotropy Vulkan
|
||||
// will not apply (NEAREST filtering, or requests past the device limit) must still share one
|
||||
// VkSampler, while two samplers that really do differ must not collide onto the first one's.
|
||||
const auto maxAnisotropy = ResolveEffectiveMaxAnisotropy(sampler, forceNearestFiltering);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &maxAnisotropy, sizeof(maxAnisotropy)));
|
||||
const auto compareMode = sampler.GetCompareMode();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &compareMode, sizeof(compareMode)));
|
||||
const auto compareFunc = sampler.GetSamplerCompareFunc();
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &compareFunc, sizeof(compareFunc)));
|
||||
const auto borderColor = ResolveVkBorderColor(sampler, texture);
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &borderColor, sizeof(borderColor)));
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
VkSampler VkSamplerManager::GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Uint32 viewLevelCount) {
|
||||
// A view that exposes a single mip level has no second level to blend with, so GL's
|
||||
// *_MIPMAP_* minification filters degenerate to plain filtering on the base level -
|
||||
// sampling is unchanged by pinning the Vulkan sampler to NEAREST mip mode at LOD 0.
|
||||
// It is not cosmetic: MobileGL backs such a view with a fully allocated mip chain whose
|
||||
// tail is never written, and a LINEAR mip mode lets the texture unit issue the level+1
|
||||
// fetch anyway. On Adreno that fetch lands in uninitialized UBWC pages (or past the
|
||||
// allocation for a genuinely single-level image) and faults the GPU - the same failure
|
||||
// the default-framebuffer blit shader had to work around with an explicit-LOD sample.
|
||||
const Bool singleLevelView = viewLevelCount == 1;
|
||||
const Uint64 key = BuildSamplerKey(sampler, texture, forceNearestFiltering, singleLevelView);
|
||||
auto it = m_samplers.find(key);
|
||||
if (it != m_samplers.end()) {
|
||||
it->second.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
return it->second.handle;
|
||||
}
|
||||
|
||||
VkSamplerCreateInfo samplerInfo{};
|
||||
samplerInfo.sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO;
|
||||
samplerInfo.magFilter = forceNearestFiltering ? VK_FILTER_NEAREST : ToVkFilter(sampler.GetMagFilter());
|
||||
samplerInfo.minFilter = forceNearestFiltering ? VK_FILTER_NEAREST : ToVkFilter(sampler.GetMinFilter());
|
||||
samplerInfo.mipmapMode = (forceNearestFiltering || singleLevelView)
|
||||
? VK_SAMPLER_MIPMAP_MODE_NEAREST
|
||||
: ToVkMipmapMode(sampler.GetMipmapMode());
|
||||
samplerInfo.addressModeU = ToVkAddressMode(sampler.GetWrapS());
|
||||
samplerInfo.addressModeV = ToVkAddressMode(sampler.GetWrapT());
|
||||
samplerInfo.addressModeW = ToVkAddressMode(sampler.GetWrapR());
|
||||
samplerInfo.mipLodBias = sampler.GetLodBias();
|
||||
// Must use the same resolver as BuildSamplerKey - a divergence would either collide two
|
||||
// different samplers or silently create duplicates.
|
||||
const Float maxAnisotropy = ResolveEffectiveMaxAnisotropy(sampler, forceNearestFiltering);
|
||||
samplerInfo.anisotropyEnable = maxAnisotropy > 1.0f ? VK_TRUE : VK_FALSE;
|
||||
samplerInfo.maxAnisotropy = maxAnisotropy;
|
||||
samplerInfo.compareEnable = sampler.GetCompareMode() == SamplerCompareMode::CompareToTexture ? VK_TRUE : VK_FALSE;
|
||||
samplerInfo.compareOp = ToVkCompareOp(sampler.GetSamplerCompareFunc());
|
||||
// Must match BuildSamplerKey's resolution exactly.
|
||||
samplerInfo.maxLod = ResolveSingleLevelMaxLod(sampler, singleLevelView);
|
||||
samplerInfo.minLod = ResolveEffectiveMinLod(sampler, samplerInfo.maxLod);
|
||||
samplerInfo.borderColor = ResolveVkBorderColor(sampler, texture);
|
||||
samplerInfo.unnormalizedCoordinates = VK_FALSE;
|
||||
|
||||
VkSampler vkSampler = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateSampler(m_device, &samplerInfo, nullptr, &vkSampler), "vkCreateSampler(texture)");
|
||||
|
||||
SamplerCacheEntry entry{};
|
||||
entry.handle = vkSampler;
|
||||
entry.externalIndex = sampler.GetExternalIndex();
|
||||
entry.version = sampler.GetVersion();
|
||||
entry.lastUsedFrameBoundary = m_frameBoundaryCounter;
|
||||
m_samplers[key] = entry;
|
||||
return vkSampler;
|
||||
}
|
||||
|
||||
VkFilter VkSamplerManager::ToVkFilter(SamplerFilterMode mode) {
|
||||
return mode == SamplerFilterMode::Nearest ? VK_FILTER_NEAREST : VK_FILTER_LINEAR;
|
||||
}
|
||||
|
||||
VkSamplerMipmapMode VkSamplerManager::ToVkMipmapMode(SamplerMipmapMode mode) {
|
||||
switch (mode) {
|
||||
case SamplerMipmapMode::Nearest:
|
||||
return VK_SAMPLER_MIPMAP_MODE_NEAREST;
|
||||
case SamplerMipmapMode::Linear:
|
||||
return VK_SAMPLER_MIPMAP_MODE_LINEAR;
|
||||
case SamplerMipmapMode::None:
|
||||
default:
|
||||
return VK_SAMPLER_MIPMAP_MODE_NEAREST;
|
||||
}
|
||||
}
|
||||
|
||||
VkSamplerAddressMode VkSamplerManager::ToVkAddressMode(SamplerWrapMode mode) {
|
||||
switch (mode) {
|
||||
case SamplerWrapMode::ClampToEdge:
|
||||
return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
||||
case SamplerWrapMode::MirroredRepeat:
|
||||
return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT;
|
||||
case SamplerWrapMode::Repeat:
|
||||
return VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
||||
case SamplerWrapMode::ClampToBorder:
|
||||
return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER;
|
||||
case SamplerWrapMode::MirrorClampToEdge:
|
||||
return VK_SAMPLER_ADDRESS_MODE_MIRROR_CLAMP_TO_EDGE;
|
||||
default:
|
||||
return VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
||||
}
|
||||
}
|
||||
|
||||
VkCompareOp VkSamplerManager::ToVkCompareOp(SamplerCompareFunc func) {
|
||||
switch (func) {
|
||||
case SamplerCompareFunc::Never:
|
||||
return VK_COMPARE_OP_NEVER;
|
||||
case SamplerCompareFunc::Less:
|
||||
return VK_COMPARE_OP_LESS;
|
||||
case SamplerCompareFunc::Equal:
|
||||
return VK_COMPARE_OP_EQUAL;
|
||||
case SamplerCompareFunc::LessEqual:
|
||||
return VK_COMPARE_OP_LESS_OR_EQUAL;
|
||||
case SamplerCompareFunc::Greater:
|
||||
return VK_COMPARE_OP_GREATER;
|
||||
case SamplerCompareFunc::NotEqual:
|
||||
return VK_COMPARE_OP_NOT_EQUAL;
|
||||
case SamplerCompareFunc::GreaterEqual:
|
||||
return VK_COMPARE_OP_GREATER_OR_EQUAL;
|
||||
case SamplerCompareFunc::Always:
|
||||
default:
|
||||
return VK_COMPARE_OP_ALWAYS;
|
||||
}
|
||||
}
|
||||
|
||||
VkBorderColor VkSamplerManager::ResolveVkBorderColor(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture) {
|
||||
if (!UsesBorderColor(sampler)) {
|
||||
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
}
|
||||
|
||||
// Border colour is sampler state: a bound sampler object supplies its own, and a texture
|
||||
// with none reaches the very same value through the sampler object it owns.
|
||||
const auto& borderColor = sampler.GetBorderColor();
|
||||
const Bool isDepthTexture = IsDepthTextureFormat(texture.GetFormat());
|
||||
|
||||
if (isDepthTexture) {
|
||||
if (NearlyEqual(borderColor.x(), 1.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
||||
}
|
||||
if (NearlyEqual(borderColor.x(), 0.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK;
|
||||
}
|
||||
}
|
||||
|
||||
const Bool rgbZero = NearlyEqual(borderColor.x(), 0.0f) && NearlyEqual(borderColor.y(), 0.0f) &&
|
||||
NearlyEqual(borderColor.z(), 0.0f);
|
||||
if (rgbZero && NearlyEqual(borderColor.w(), 0.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
}
|
||||
if (rgbZero && NearlyEqual(borderColor.w(), 1.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_BLACK;
|
||||
}
|
||||
if (NearlyEqual(borderColor.x(), 1.0f) && NearlyEqual(borderColor.y(), 1.0f) &&
|
||||
NearlyEqual(borderColor.z(), 1.0f) && NearlyEqual(borderColor.w(), 1.0f)) {
|
||||
return VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
||||
}
|
||||
|
||||
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,90 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkSamplerManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include "../VulkanRendererConfig.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/SamplerState/SamplerObject.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class SamplerObject;
|
||||
class ITextureObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkSamplerManager {
|
||||
public:
|
||||
struct InitInfo {
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
const VulkanRendererConfig* config = nullptr;
|
||||
// The samplerAnisotropy device feature was requested and granted at vkCreateDevice.
|
||||
Bool samplerAnisotropySupported = false;
|
||||
// VkPhysicalDeviceLimits::maxSamplerAnisotropy.
|
||||
Float maxSamplerAnisotropy = 1.0f;
|
||||
};
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
|
||||
// viewLevelCount is the mip-level count of the image view this sampler will be paired
|
||||
// with; 0 means "unknown, do not narrow". See GetOrCreateSampler for why it matters.
|
||||
VkSampler GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering = false,
|
||||
Uint32 viewLevelCount = 0);
|
||||
// Frame boundary hook: ages the sampler cache and destroys samplers not used
|
||||
// for many frames. The key hashes continuous float state (lodBias, LOD clamps,
|
||||
// anisotropy), so an app animating those would otherwise mint an unbounded
|
||||
// stream of never-destroyed VkSamplers and eventually exhaust the device's
|
||||
// maxSamplerAllocationCount. A sampler idle for over a thousand frame
|
||||
// boundaries cannot be referenced by any in-flight command buffer (frames in
|
||||
// flight are single digits), and every descriptor set the GPU consumes is
|
||||
// written that same frame with live handles (the per-binding resolve memo and
|
||||
// descriptor-set reuse are both frame-reset), so destruction here needs no
|
||||
// fence wait. Self-gated: one counter bump and compare except on sweep
|
||||
// boundaries.
|
||||
void OnFrameBoundary();
|
||||
|
||||
private:
|
||||
struct SamplerCacheEntry {
|
||||
VkSampler handle = VK_NULL_HANDLE;
|
||||
Uint externalIndex = 0;
|
||||
Uint16 version = 0;
|
||||
// Frame boundary of the last cache hit; entries idle past the
|
||||
// OnFrameBoundary retirement age have their VkSampler destroyed.
|
||||
Uint64 lastUsedFrameBoundary = 0;
|
||||
};
|
||||
|
||||
Uint64 BuildSamplerKey(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
Bool forceNearestFiltering, Bool singleLevelView) const;
|
||||
static VkFilter ToVkFilter(SamplerFilterMode mode);
|
||||
static VkSamplerMipmapMode ToVkMipmapMode(SamplerMipmapMode mode);
|
||||
static VkSamplerAddressMode ToVkAddressMode(SamplerWrapMode mode);
|
||||
static VkCompareOp ToVkCompareOp(SamplerCompareFunc func);
|
||||
static VkBorderColor ResolveVkBorderColor(const MG_State::GLState::SamplerObject& sampler,
|
||||
const MG_State::GLState::ITextureObject& texture);
|
||||
// The anisotropy Vulkan will actually apply: 1.0 (i.e. disabled) unless the feature is on and
|
||||
// the sampler filters linearly both ways, otherwise the GL request clamped to the device limit.
|
||||
// GL happily carries GL_TEXTURE_MAX_ANISOTROPY on a NEAREST sampler (Blaze3D's blocks do exactly
|
||||
// that) while Vulkan forbids anisotropyEnable there, so the GL value must never be forwarded raw.
|
||||
Float ResolveEffectiveMaxAnisotropy(const MG_State::GLState::SamplerObject& sampler,
|
||||
Bool forceNearestFiltering) const;
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
const VulkanRendererConfig* m_config = nullptr;
|
||||
Bool m_samplerAnisotropySupported = false;
|
||||
Float m_maxSamplerAnisotropy = 1.0f;
|
||||
UnorderedMap<Uint64, SamplerCacheEntry> m_samplers;
|
||||
// Monotonic frame-boundary counter (bumped in OnFrameBoundary) for cache aging.
|
||||
Uint64 m_frameBoundaryCounter = 0;
|
||||
static inline XXH64_state_t* m_hashState = XXH64_createState();
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,596 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkTextureManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/TextureState/TextureObject.h>
|
||||
#include <vk_mem_alloc.h>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
enum class SamplerNumericDomain : Uint8;
|
||||
|
||||
class VkTextureManager {
|
||||
public:
|
||||
// Monotonic epoch bumped whenever a texture VkImage is (re)created. The render-pass
|
||||
// manager keys its per-draw fast path on this so an attachment's image recreation
|
||||
// invalidates the cached render pass (dirty-flag tracking; portable to Vulkan 1.1).
|
||||
Uint64 GetTextureImageEpoch() const { return m_textureImageEpoch; }
|
||||
// Bumped whenever any tracked texture resource is erased; cached
|
||||
// TextureResource pointers are valid only while this is unchanged.
|
||||
Uint64 GetResourceEraseEpoch() const { return m_resourceEraseEpoch; }
|
||||
|
||||
struct TextureIdentity {
|
||||
MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
Uint64 lifetimeId = 0;
|
||||
|
||||
Bool operator==(const TextureIdentity& other) const {
|
||||
return texture == other.texture && lifetimeId == other.lifetimeId;
|
||||
}
|
||||
};
|
||||
|
||||
struct TextureIdentityHash {
|
||||
SizeT operator()(const TextureIdentity& key) const {
|
||||
SizeT hash = std::hash<MG_State::GLState::ITextureObject*>{}(key.texture);
|
||||
hash ^= std::hash<Uint64>{}(key.lifetimeId) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
struct InitInfo {
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice physicalDevice = VK_NULL_HANDLE;
|
||||
VmaAllocator allocator = nullptr;
|
||||
VkCommandPool commandPool = VK_NULL_HANDLE;
|
||||
VkQueue graphicsQueue = VK_NULL_HANDLE;
|
||||
Uint32 frameCount = 0;
|
||||
// VK_KHR_image_format_list is enabled: MUTABLE_FORMAT images can name the exact set of
|
||||
// formats they will be viewed as, which is what lets a tiler keep them compressed.
|
||||
Bool imageFormatListSupported = false;
|
||||
// Union of shader stages sampled-read barriers may name on this device; the renderer
|
||||
// builds it from the enabled features because geometry/tessellation stage bits are
|
||||
// invalid in a barrier when their feature is off.
|
||||
VkPipelineStageFlags sampledReadStageMask = VK_PIPELINE_STAGE_VERTEX_SHADER_BIT |
|
||||
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
|
||||
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
|
||||
// Family of `graphicsQueue`; the manager creates its own command pool
|
||||
// on it for the recycled upload-batch command buffers, so their parked
|
||||
// allocations never sit in (and fragment) the renderer's shared pool
|
||||
// that frame command buffers churn through every frame.
|
||||
Uint32 graphicsQueueFamilyIndex = 0;
|
||||
};
|
||||
|
||||
struct TextureResource {
|
||||
struct AttachmentViewKey {
|
||||
Uint32 mipLevel = 0;
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = 1;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
// May differ from the image format: sRGB images attach through their UNORM
|
||||
// twin while GL_FRAMEBUFFER_SRGB is disabled.
|
||||
VkFormat viewFormat = VK_FORMAT_UNDEFINED;
|
||||
|
||||
Bool operator==(const AttachmentViewKey& other) const {
|
||||
return mipLevel == other.mipLevel &&
|
||||
baseArrayLayer == other.baseArrayLayer &&
|
||||
layerCount == other.layerCount &&
|
||||
viewType == other.viewType &&
|
||||
viewFormat == other.viewFormat;
|
||||
}
|
||||
};
|
||||
|
||||
struct AttachmentViewKeyHash {
|
||||
SizeT operator()(const AttachmentViewKey& key) const {
|
||||
SizeT hash = std::hash<Uint32>{}(key.mipLevel);
|
||||
hash ^= std::hash<Uint32>{}(key.baseArrayLayer) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(key.layerCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewType)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewFormat)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
struct StorageImageViewKey {
|
||||
Uint32 mipLevel = 0;
|
||||
Uint32 baseArrayLayer = 0;
|
||||
Uint32 layerCount = 1;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
|
||||
Bool operator==(const StorageImageViewKey& other) const {
|
||||
return mipLevel == other.mipLevel &&
|
||||
baseArrayLayer == other.baseArrayLayer &&
|
||||
layerCount == other.layerCount &&
|
||||
viewType == other.viewType &&
|
||||
format == other.format;
|
||||
}
|
||||
};
|
||||
|
||||
struct SampledImageViewKey {
|
||||
Uint32 baseMipLevel = 0;
|
||||
Uint32 levelCount = 1;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
|
||||
Bool operator==(const SampledImageViewKey& other) const {
|
||||
return baseMipLevel == other.baseMipLevel &&
|
||||
levelCount == other.levelCount &&
|
||||
viewType == other.viewType &&
|
||||
format == other.format;
|
||||
}
|
||||
};
|
||||
|
||||
struct SampledImageViewKeyHash {
|
||||
SizeT operator()(const SampledImageViewKey& key) const {
|
||||
SizeT hash = std::hash<Uint32>{}(key.baseMipLevel);
|
||||
hash ^= std::hash<Uint32>{}(key.levelCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewType)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.format)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
struct StorageImageViewKeyHash {
|
||||
SizeT operator()(const StorageImageViewKey& key) const {
|
||||
SizeT hash = std::hash<Uint32>{}(key.mipLevel);
|
||||
hash ^= std::hash<Uint32>{}(key.baseArrayLayer) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(key.layerCount) + 0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.viewType)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
hash ^= std::hash<Uint32>{}(static_cast<Uint32>(key.format)) +
|
||||
0x9e3779b9u + (hash << 6) + (hash >> 2);
|
||||
return hash;
|
||||
}
|
||||
};
|
||||
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
VkImageView fullView = VK_NULL_HANDLE;
|
||||
VkImageView sampledView = VK_NULL_HANDLE;
|
||||
Vector<VkImageView> perMipViews;
|
||||
Vector<VkImageView> perMipSampledViews;
|
||||
UnorderedMap<AttachmentViewKey, VkImageView, AttachmentViewKeyHash> attachmentViews;
|
||||
UnorderedMap<SampledImageViewKey, VkImageView, SampledImageViewKeyHash> alternateSampledViews;
|
||||
UnorderedMap<StorageImageViewKey, VkImageView, StorageImageViewKeyHash> storageImageViews;
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkExtent2D extent = {0, 0};
|
||||
Uint32 depth = 1;
|
||||
Uint32 arrayLayers = 1;
|
||||
Uint32 mipLevels = 1;
|
||||
Uint32 sampledBaseMipLevel = 0;
|
||||
Uint32 sampledLevelCount = 1;
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
VkImageAspectFlags aspect = VK_IMAGE_ASPECT_NONE;
|
||||
VkImageViewType viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
VkSampleCountFlagBits sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
VkImageCreateFlags imageCreateFlags = 0;
|
||||
// Usage the live image was created with. STORAGE is only requested for textures that
|
||||
// have actually been bound to a GL image unit, because on Adreno a storage-capable
|
||||
// image loses UBWC bandwidth compression; a later image binding upgrades the usage
|
||||
// and recreates the image, so the resolved usage has to be part of the compatibility
|
||||
// check that decides whether the existing image can be kept.
|
||||
VkImageUsageFlags usageFlags = 0;
|
||||
// True once this image was (re)resolved while the texture was already marked as an
|
||||
// image-unit texture. Distinguishes "not upgraded yet" from "cannot be upgraded"
|
||||
// (a format whose optimalTilingFeatures lack STORAGE_IMAGE never gains the bit), so
|
||||
// NeedsStorageImagePreparation cannot ask for a recreate that will never happen.
|
||||
Bool storageUsageResolved = false;
|
||||
Uint16 syncedTextureParamsVersion = 0;
|
||||
// Recording generation (VkTextureManager::GetRecordingGeneration) of the last
|
||||
// command referencing this image that was recorded into the CURRENT frame
|
||||
// command buffer. An image untouched by the open recording may have its
|
||||
// out-of-pass work (deferred clears, sampled-layout transitions) recorded
|
||||
// into the frame's PRE command buffer - which executes strictly before the
|
||||
// frame's commands - instead of splitting the active render pass.
|
||||
Uint64 lastRecordingGeneration = 0;
|
||||
// Snapshot of ITextureObject::GetContentVersion() at the last successful sync;
|
||||
// lets SyncTexture skip the whole re-check/re-upload when content is unchanged.
|
||||
Uint64 syncedContentVersion = 0;
|
||||
// Snapshot of the defined mip-level count at the last sync. Folded into the early-out key
|
||||
// as defense-in-depth: any path that grows the level set (which resizes the sampled view)
|
||||
// busts the skip even if it failed to bump the content version.
|
||||
Uint32 syncedMipLevelCount = 0;
|
||||
|
||||
TextureResource() = default;
|
||||
TextureResource(const TextureResource&) = delete;
|
||||
TextureResource(TextureResource&& that) noexcept {
|
||||
std::swap(this->image, that.image);
|
||||
std::swap(this->allocation, that.allocation);
|
||||
std::swap(this->fullView, that.fullView);
|
||||
std::swap(this->sampledView, that.sampledView);
|
||||
std::swap(this->perMipViews, that.perMipViews);
|
||||
std::swap(this->perMipSampledViews, that.perMipSampledViews);
|
||||
std::swap(this->attachmentViews, that.attachmentViews);
|
||||
std::swap(this->alternateSampledViews, that.alternateSampledViews);
|
||||
std::swap(this->storageImageViews, that.storageImageViews);
|
||||
std::swap(this->layout, that.layout);
|
||||
std::swap(this->extent, that.extent);
|
||||
std::swap(this->depth, that.depth);
|
||||
std::swap(this->arrayLayers, that.arrayLayers);
|
||||
std::swap(this->mipLevels, that.mipLevels);
|
||||
std::swap(this->sampledBaseMipLevel, that.sampledBaseMipLevel);
|
||||
std::swap(this->sampledLevelCount, that.sampledLevelCount);
|
||||
std::swap(this->format, that.format);
|
||||
std::swap(this->aspect, that.aspect);
|
||||
std::swap(this->viewType, that.viewType);
|
||||
std::swap(this->sampleCount, that.sampleCount);
|
||||
std::swap(this->imageCreateFlags, that.imageCreateFlags);
|
||||
std::swap(this->usageFlags, that.usageFlags);
|
||||
std::swap(this->storageUsageResolved, that.storageUsageResolved);
|
||||
std::swap(this->syncedTextureParamsVersion, that.syncedTextureParamsVersion);
|
||||
std::swap(this->lastRecordingGeneration, that.lastRecordingGeneration);
|
||||
std::swap(this->syncedContentVersion, that.syncedContentVersion);
|
||||
std::swap(this->syncedMipLevelCount, that.syncedMipLevelCount);
|
||||
}
|
||||
|
||||
void Reset() {
|
||||
if (fullView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, fullView, nullptr);
|
||||
}
|
||||
if (sampledView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, sampledView, nullptr);
|
||||
}
|
||||
for (const auto attachmentView : perMipViews) {
|
||||
if (attachmentView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, attachmentView, nullptr);
|
||||
}
|
||||
}
|
||||
for (const auto sampledView : perMipSampledViews) {
|
||||
if (sampledView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, sampledView, nullptr);
|
||||
}
|
||||
}
|
||||
for (const auto& [_, attachmentView] : attachmentViews) {
|
||||
if (attachmentView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, attachmentView, nullptr);
|
||||
}
|
||||
}
|
||||
for (const auto& [_, sampledView] : alternateSampledViews) {
|
||||
if (sampledView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, sampledView, nullptr);
|
||||
}
|
||||
}
|
||||
for (const auto& [_, storageImageView] : storageImageViews) {
|
||||
if (storageImageView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(s_device, storageImageView, nullptr);
|
||||
}
|
||||
}
|
||||
if (image != VK_NULL_HANDLE && allocation != nullptr) {
|
||||
vmaDestroyImage(s_allocator, image, allocation);
|
||||
}
|
||||
fullView = VK_NULL_HANDLE;
|
||||
sampledView = VK_NULL_HANDLE;
|
||||
perMipViews.clear();
|
||||
perMipSampledViews.clear();
|
||||
attachmentViews.clear();
|
||||
alternateSampledViews.clear();
|
||||
storageImageViews.clear();
|
||||
image = VK_NULL_HANDLE;
|
||||
allocation = nullptr;
|
||||
layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
extent = {0, 0};
|
||||
depth = 1;
|
||||
arrayLayers = 1;
|
||||
mipLevels = 1;
|
||||
sampledBaseMipLevel = 0;
|
||||
sampledLevelCount = 1;
|
||||
format = VK_FORMAT_UNDEFINED;
|
||||
aspect = VK_IMAGE_ASPECT_NONE;
|
||||
viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
sampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageCreateFlags = 0;
|
||||
usageFlags = 0;
|
||||
storageUsageResolved = false;
|
||||
syncedTextureParamsVersion = 0;
|
||||
syncedContentVersion = 0;
|
||||
syncedMipLevelCount = 0;
|
||||
}
|
||||
|
||||
~TextureResource() {
|
||||
Reset();
|
||||
}
|
||||
|
||||
static inline VkDevice s_device = VK_NULL_HANDLE;
|
||||
static inline VmaAllocator s_allocator = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
void BeginFrame(Uint32 frameIndex);
|
||||
// Submits the accumulated texture-upload batch (one command buffer, one
|
||||
// vkQueueSubmit, one pooled fence) if any uploads are pending. MUST run
|
||||
// before any other vkQueueSubmit on the shared graphics queue whose
|
||||
// commands may consume an image the batch writes - the frame command
|
||||
// buffer submit (mid-frame flush, readback, Present) and the
|
||||
// preserve-on-recreate copy are the existing callers. No-op when the
|
||||
// batch is empty.
|
||||
void FlushPendingUploads();
|
||||
// Drains every frame slot's deferred image/view releases. Only valid when
|
||||
// the caller has proven every queue submission complete; used by the
|
||||
// present-less frame-boundary drain.
|
||||
void CollectAllDeferredReleases();
|
||||
|
||||
TextureResource* SyncTextureAndGetDescriptor(
|
||||
MG_State::GLState::ITextureObject& texture);
|
||||
VkImageView GetOrCreateViewAtMipLevel(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel);
|
||||
VkImageView GetOrCreateAttachmentViewAtMipLevel(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel,
|
||||
Uint32 baseArrayLayer, Uint32 layerCount,
|
||||
VkImageViewType viewType);
|
||||
VkImageView GetOrCreateSampledViewAtMipLevel(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel);
|
||||
VkImageView GetOrCreateSampledImageView(MG_State::GLState::ITextureObject& texture, VkFormat format);
|
||||
VkImageView GetOrCreateStorageImageView(MG_State::GLState::ITextureObject& texture, Uint32 mipLevel,
|
||||
VkFormat format, Bool layered, Int32 layer);
|
||||
void UpdateTrackedImageLayout(MG_State::GLState::ITextureObject* texture, VkImageLayout newLayout);
|
||||
void UpdateTrackedImageLayoutAfterAttachmentWrite(VkCommandBuffer commandBuffer,
|
||||
MG_State::GLState::ITextureObject* texture,
|
||||
Uint32 writtenMipLevel,
|
||||
VkImageLayout newLayout);
|
||||
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
|
||||
// Recording-generation bookkeeping for the pre-pass command stream. The
|
||||
// generation advances every time the frame command buffer (re)begins
|
||||
// recording; a resource whose stamp does not match was not referenced by
|
||||
// any command in the open recording, so its out-of-pass work may safely
|
||||
// execute ahead of the whole recording (in the pre command buffer).
|
||||
void AdvanceRecordingGeneration() { ++m_recordingGeneration; }
|
||||
void StampResourceRecordingUse(TextureResource& resource) const {
|
||||
resource.lastRecordingGeneration = m_recordingGeneration;
|
||||
}
|
||||
// Map-lookup variant for callers that only hold the GL texture object.
|
||||
void StampTextureRecordingUse(MG_State::GLState::ITextureObject* texture);
|
||||
Bool WasTouchedThisRecording(const TextureResource& resource) const {
|
||||
return resource.lastRecordingGeneration == m_recordingGeneration;
|
||||
}
|
||||
// Records that this texture is bound to a GL image unit, so its image must carry
|
||||
// VK_IMAGE_USAGE_STORAGE_BIT. Must be called before NeedsStorageImagePreparation, and
|
||||
// therefore before the render pass is committed: an image that has to be upgraded is
|
||||
// recreated, which is illegal inside a render pass. Sticky for the texture's lifetime -
|
||||
// GL lets an image binding come and go, and re-creating the image every time it does
|
||||
// would cost far more than the compression it wins back.
|
||||
void MarkStorageImageTexture(MG_State::GLState::ITextureObject& texture);
|
||||
// True when this texture is marked but its live image predates the mark, i.e. the next sync
|
||||
// will recreate it with STORAGE usage and copy the old contents forward. Callers use this to
|
||||
// submit their pending recording first, so that copy cannot read pre-flush content.
|
||||
Bool NeedsStorageUsageUpgrade(MG_State::GLState::ITextureObject& texture) const;
|
||||
// The same ordering question for the other recreate-and-preserve trigger: true when this
|
||||
// texture's live image carries a shorter mip chain than a full one, so defining the missing
|
||||
// levels recreates it and copies the old contents forward.
|
||||
Bool NeedsMipChainGrowth(MG_State::GLState::ITextureObject& texture) const;
|
||||
// Non-mutating probe for the per-draw storage-image fast path: true when preparing this
|
||||
// texture as a storage image may need work that is illegal inside a render pass (resource
|
||||
// creation, dirty-content upload, or a layout transition to GENERAL). Unknown state reports
|
||||
// true - a false positive merely ends the render pass, a false negative would skip a barrier.
|
||||
Bool NeedsStorageImagePreparation(MG_State::GLState::ITextureObject& texture) const;
|
||||
|
||||
// `depthStencilTextureMode` is the texture's GL_DEPTH_STENCIL_TEXTURE_MODE; it only decides
|
||||
// anything for an image that carries both aspects. Defaulted so the call sites that have no
|
||||
// texture in hand keep the depth-aspect answer they have always given.
|
||||
static VkImageAspectFlags ResolveSampledImageViewAspectMask(VkImageAspectFlags imageAspect,
|
||||
GLenum depthStencilTextureMode = GL_DEPTH_COMPONENT);
|
||||
static VkFormat ResolveSampledImageViewFormat(VkFormat imageFormat, SamplerNumericDomain numericDomain);
|
||||
static Bool AreSampledImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
|
||||
static Bool AreStorageImageViewFormatsCompatible(VkFormat imageFormat, VkFormat viewFormat);
|
||||
|
||||
// Moves `image` to `newLayout` and writes the new layout back through `trackedLayout`.
|
||||
//
|
||||
// The barrier covers EVERY array layer of the image, and there is deliberately no layer
|
||||
// parameter to say otherwise: layout here is tracked per IMAGE (one `TextureResource::layout`,
|
||||
// or one caller-owned variable), so a barrier narrower than the image would leave the layers it
|
||||
// skipped in the old layout while the tracker claims they moved. Every transfer against a
|
||||
// framebuffer attachment above layer 0 - glReadPixels, glBlitFramebuffer, glCopyTexSubImage,
|
||||
// glCopyImageSubData - then ran its copy on a layer no barrier had transitioned.
|
||||
//
|
||||
// The mip range IS a parameter, because mip levels really are transitioned piecewise (see
|
||||
// UpdateTrackedImageLayoutAfterAttachmentWrite and the mipmap generation loops): those callers
|
||||
// move the complement of the level they wrote so the whole image converges on one layout again.
|
||||
// Nothing does, or can, do that per layer.
|
||||
static Bool TransitionImageLayout(VkCommandBuffer commandBuffer, VkImage image, VkImageLayout& trackedLayout,
|
||||
VkImageLayout newLayout, VkPipelineStageFlags srcStageMask,
|
||||
VkPipelineStageFlags dstStageMask, VkAccessFlags srcAccessMask,
|
||||
VkAccessFlags dstAccessMask, VkImageAspectFlags aspectMask,
|
||||
Uint32 baseMipLevel = 0, Uint32 levelCount = 1);
|
||||
|
||||
SizeT CollectGarbage();
|
||||
|
||||
// Per-draw sync memo. Within a single SetupDraw the same sampled texture is
|
||||
// resolved ~3x (SetupDraw's layout-probe loop, its post-transition loop, and
|
||||
// again inside ResolveSamplerDescriptor). No GL texture mutation can happen
|
||||
// mid-SetupDraw, and layout is tracked on the TextureResource independently of
|
||||
// SyncTexture, so after the first successful sync of a texture in a draw the
|
||||
// heavy SyncTexture work (mip-completeness/resource/view resync + dirty scan)
|
||||
// is pure redundancy. BeginDrawSyncScope opens a window in which repeat
|
||||
// SyncTextureAndGetDescriptor calls short-circuit to the already-synced
|
||||
// resource; EndDrawSyncScope closes it. Use the RAII DrawSyncScope guard.
|
||||
void BeginDrawSyncScope();
|
||||
void EndDrawSyncScope();
|
||||
|
||||
// RAII guard that opens/closes a per-draw sync memo window (see above).
|
||||
class DrawSyncScope {
|
||||
public:
|
||||
explicit DrawSyncScope(VkTextureManager& manager) : m_manager(manager) { m_manager.BeginDrawSyncScope(); }
|
||||
~DrawSyncScope() { m_manager.EndDrawSyncScope(); }
|
||||
DrawSyncScope(const DrawSyncScope&) = delete;
|
||||
DrawSyncScope& operator=(const DrawSyncScope&) = delete;
|
||||
private:
|
||||
VkTextureManager& m_manager;
|
||||
};
|
||||
|
||||
private:
|
||||
// Bumped in SyncTextureResource right after vmaCreateImage(texture). See GetTextureImageEpoch().
|
||||
Uint64 m_textureImageEpoch = 1;
|
||||
// See AdvanceRecordingGeneration. Starts above every resource's default
|
||||
// stamp of 0 so a fresh resource counts as untouched.
|
||||
Uint64 m_recordingGeneration = 1;
|
||||
|
||||
Bool SyncTexture(MG_State::GLState::ITextureObject &texture,
|
||||
TextureResource &outResource);
|
||||
Bool SyncTextureResource(const MG_State::GLState::ITextureObject &texture,
|
||||
TextureUploadTarget uploadTarget,
|
||||
const IntVec3 &texelSize, SizeT byteSize, Uint32 mipLevels,
|
||||
TextureResource &resource);
|
||||
Bool SyncTextureViews(const MG_State::GLState::ITextureObject& texture, TextureResource& resource);
|
||||
VkImageView CreateImageView(VkImage image, VkFormat format, VkImageAspectFlags aspect,
|
||||
VkImageViewType viewType, Uint32 baseMipLevel, Uint32 levelCount,
|
||||
Uint32 baseArrayLayer,
|
||||
Uint32 layerCount,
|
||||
const VkComponentMapping* components = nullptr,
|
||||
VkImageUsageFlags viewUsage = 0) const;
|
||||
Bool UploadDirtyMipLevels(MG_State::GLState::TextureObjectMipmap &mipmapTexture,
|
||||
TextureUploadTarget uploadTarget,
|
||||
TextureResource &outResource);
|
||||
static Bool CheckMipmapCompleteness(const MG_State::GLState::ITextureObject& texture,
|
||||
TextureUploadTarget& outTarget,
|
||||
IntVec3& outTexelSize,
|
||||
SizeT& outByteSize,
|
||||
Uint32& outMipLevelCount);
|
||||
static Uint32 GetUploadMipLevelCount(const MG_State::GLState::TextureObjectMipmap& texture, TextureUploadTarget target);
|
||||
static void ResolveViewMipRange(const MG_State::GLState::ITextureObject& texture, Uint32 mipLevels,
|
||||
Uint32& outBaseMipLevel, Uint32& outLevelCount);
|
||||
static VkImageAspectFlags GetAspectMaskForFormat(VkFormat format);
|
||||
void DeferResourceRelease(TextureResource&& resource);
|
||||
void DeferViewRelease(VkImageView view);
|
||||
void CollectDeferredReleases(Uint32 frameIndex);
|
||||
void DestroyDeferredReleases();
|
||||
// Frees the fence/command buffer/staging buffer of every in-flight texture
|
||||
// upload whose fence has signaled (submission order = completion order on
|
||||
// the single queue, so the scan stops at the first still-pending entry).
|
||||
// waitAll blocks on every entry - Shutdown's drain.
|
||||
void ReclaimCompletedUploads(Bool waitAll = false);
|
||||
static TextureIdentity MakeTextureIdentity(MG_State::GLState::ITextureObject* texture);
|
||||
void EraseTrackedTexture(const TextureIdentity& identity);
|
||||
void PruneStaleTextureAliases(MG_State::GLState::ITextureObject* texture);
|
||||
SizeT PruneDeadTextures();
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VmaAllocator m_allocator = nullptr;
|
||||
VkCommandPool m_commandPool = VK_NULL_HANDLE;
|
||||
// Dedicated pool for the recycled upload-batch command buffers (see
|
||||
// InitInfo::graphicsQueueFamilyIndex).
|
||||
VkCommandPool m_uploadCommandPool = VK_NULL_HANDLE;
|
||||
VkQueue m_graphicsQueue = VK_NULL_HANDLE;
|
||||
Bool m_imageFormatListSupported = false;
|
||||
Uint32 m_currentFrameIndex = 0;
|
||||
|
||||
Uint8 m_gcCounter = 0;
|
||||
// Frame-boundary GC gate: counts BeginFrame calls, not draws, so texture churn
|
||||
// through non-draw paths (FBO clears, readbacks) still reaches the prune.
|
||||
Uint32 m_gcFrameCounter = 0;
|
||||
// Active only between BeginDrawSyncScope/EndDrawSyncScope; identities of
|
||||
// textures already fully synced in the current draw (small N -> flat scan).
|
||||
Bool m_drawSyncScopeActive = false;
|
||||
// Per-draw sync memo: the identity plus the resolved resource pointer. The pointer is stable
|
||||
// across rehash in the node-based m_textureResources and stays valid for the draw (a texture
|
||||
// synced this draw is alive and is not erased mid-draw), so a repeat sync of the same texture
|
||||
// returns the resource without re-hashing the identity into m_textureResources.
|
||||
struct DrawSyncedTexture {
|
||||
TextureIdentity identity;
|
||||
TextureResource* resource = nullptr;
|
||||
};
|
||||
Vector<DrawSyncedTexture> m_drawSyncedThisDraw;
|
||||
// Cross-draw sampled-texture memo: the same few textures (atlas, lightmap)
|
||||
// are resolved on every draw, so cache their resource pointers and skip the
|
||||
// alive/resource map lookups. Node-based std::unordered_map keeps the
|
||||
// pointees stable across inserts; erases bump m_resourceEraseEpoch, which
|
||||
// every memo entry must match. SyncTexture still runs on memo hits, so
|
||||
// content/param freshness is unaffected. A dead-then-reused texture address
|
||||
// cannot false-hit: the new object carries a new lifetime id.
|
||||
struct SyncedTextureMemoEntry {
|
||||
const MG_State::GLState::ITextureObject* texture = nullptr;
|
||||
Uint64 lifetimeId = 0;
|
||||
Uint64 eraseEpoch = 0;
|
||||
TextureResource* resource = nullptr;
|
||||
};
|
||||
static constexpr Uint32 kSyncedTextureMemoSize = 8;
|
||||
SyncedTextureMemoEntry m_syncedTextureMemo[kSyncedTextureMemoSize];
|
||||
Uint32 m_syncedTextureMemoNext = 0;
|
||||
Uint64 m_resourceEraseEpoch = 1;
|
||||
// Formats whose mutable-image probe failed on this device; their images are created
|
||||
// without MUTABLE_FORMAT_BIT so repeat syncs neither re-probe nor flag-mismatch.
|
||||
std::unordered_set<VkFormat> m_mutableFormatUnsupported;
|
||||
// Formats whose 3D images refused VK_IMAGE_CREATE_2D_ARRAY_COMPATIBLE_BIT. Per format+usage,
|
||||
// exactly like the mutable-format verdict above, so it is answered at image creation and
|
||||
// remembered rather than probed once globally.
|
||||
std::unordered_set<VkFormat> m_2dArrayCompatibleUnsupported;
|
||||
std::unordered_map<TextureIdentity, WeakPtr<MG_State::GLState::ITextureObject>, TextureIdentityHash> m_aliveObjects;
|
||||
std::unordered_map<TextureIdentity, TextureResource, TextureIdentityHash> m_textureResources;
|
||||
// Textures that have been bound to a GL image unit (see MarkStorageImageTexture).
|
||||
std::unordered_set<TextureIdentity, TextureIdentityHash> m_storageImageTextures;
|
||||
// Supported multisample counts per format, so repeat texture syncs do not
|
||||
// re-query vkGetPhysicalDeviceImageFormatProperties.
|
||||
std::unordered_map<VkFormat, VkSampleCountFlags> m_multisampleCountsByFormat;
|
||||
Vector<Vector<TextureResource>> m_deferredReleases;
|
||||
Vector<Vector<VkImageView>> m_deferredViewReleases;
|
||||
|
||||
// --- Batched upload machinery ---
|
||||
// Uploads within a frame are recorded into ONE shared command buffer and
|
||||
// submitted with ONE vkQueueSubmit at FlushPendingUploads (the renderer
|
||||
// flushes before every frame-command-buffer submit). Staging memory comes
|
||||
// from a pool of persistently-mapped, reusable blocks instead of a
|
||||
// vmaCreateBuffer per upload.
|
||||
struct UploadStagingBlock {
|
||||
VkBuffer buffer = VK_NULL_HANDLE;
|
||||
VmaAllocation allocation = nullptr;
|
||||
Uint8* mapped = nullptr; // persistently mapped for the block's lifetime
|
||||
VkDeviceSize capacity = 0;
|
||||
VkDeviceSize cursor = 0; // bump cursor while the block backs the open batch
|
||||
};
|
||||
// Opens the batch command buffer lazily (allocates/reuses + begins recording).
|
||||
VkCommandBuffer EnsureUploadBatchOpen();
|
||||
// Bump-allocates `size` staging bytes for the open batch, growing onto a
|
||||
// new/pooled block when the current one cannot fit. Returns the write
|
||||
// pointer; outBuffer/outBaseOffset locate the space for copy commands.
|
||||
Uint8* AcquireUploadStagingSpace(VkDeviceSize size, VkBuffer& outBuffer, VkDeviceSize& outBaseOffset);
|
||||
void RecycleUploadStagingBlock(UploadStagingBlock&& block);
|
||||
// Drops a recorded-but-unsubmitted batch on the floor. Shutdown only: the
|
||||
// device is being torn down, so the lost texel data is unobservable.
|
||||
void DiscardPendingUploadBatch();
|
||||
void DestroyUploadPools();
|
||||
|
||||
Vector<UploadStagingBlock> m_freeUploadStagingBlocks;
|
||||
VkDeviceSize m_freeUploadStagingBytes = 0;
|
||||
Vector<VkCommandBuffer> m_freeUploadCommandBuffers;
|
||||
Vector<VkFence> m_freeUploadFences;
|
||||
Bool m_uploadBatchOpen = false;
|
||||
VkCommandBuffer m_uploadBatchCommandBuffer = VK_NULL_HANDLE;
|
||||
// Blocks whose staging bytes the open batch's copies reference (last =
|
||||
// the block the bump cursor is currently allocating from).
|
||||
Vector<UploadStagingBlock> m_uploadBatchBlocks;
|
||||
// Images the open batch writes; consulted for the rare re-upload-after-
|
||||
// draw flush and by DeferResourceRelease (an unsubmitted command buffer
|
||||
// referencing a deferred-released image would escape every fence-based
|
||||
// destruction proof, so the batch is flushed before the image is parked).
|
||||
Vector<VkImage> m_uploadBatchImages;
|
||||
VkDeviceSize m_uploadBatchStagingBytes = 0;
|
||||
|
||||
// Texture uploads are submitted out-of-band but NOT waited on (waiting
|
||||
// behind the queue serialized the CPU against the previous frame's GPU
|
||||
// work every time an animated atlas re-uploaded). Each flushed batch's
|
||||
// transients are parked here and RECYCLED (fence reset to the fence pool,
|
||||
// command buffer reset to the CB pool, staging blocks back to the block
|
||||
// pool) once the batch fence signals.
|
||||
struct PendingUploadReclaim {
|
||||
VkFence fence = VK_NULL_HANDLE;
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
Vector<UploadStagingBlock> stagingBlocks;
|
||||
};
|
||||
Vector<PendingUploadReclaim> m_pendingUploadReclaims;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -1,644 +0,0 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkTextureSamplerManager.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "VkTextureSamplerManager.h"
|
||||
|
||||
#include "MG_State/GLState/Core.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
namespace {
|
||||
constexpr Uint64 BuildSamplerKey(Uint externalIndex, Uint16 version) {
|
||||
return (static_cast<Uint64>(externalIndex) << 16) | static_cast<Uint64>(version);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool VkTextureSamplerManager::Initialize(const InitInfo& initInfo) {
|
||||
Shutdown();
|
||||
|
||||
m_device = initInfo.device;
|
||||
m_physicalDevice = initInfo.physicalDevice;
|
||||
m_commandPool = initInfo.commandPool;
|
||||
m_graphicsQueue = initInfo.graphicsQueue;
|
||||
|
||||
if (m_device == VK_NULL_HANDLE || m_physicalDevice == VK_NULL_HANDLE || m_commandPool == VK_NULL_HANDLE ||
|
||||
m_graphicsQueue == VK_NULL_HANDLE) {
|
||||
MGLOG_E("VkTextureSamplerManager::Initialize failed: invalid Vulkan handles");
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!UploadFallbackTexture()) {
|
||||
MGLOG_E("VkTextureSamplerManager::Initialize failed: fallback texture creation failed");
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkTextureSamplerManager::Shutdown() {
|
||||
for (auto& [_, resource] : m_textureResources) {
|
||||
DestroyTextureResource(resource);
|
||||
}
|
||||
m_textureResources.clear();
|
||||
|
||||
for (auto& [_, sampler] : m_samplers) {
|
||||
if (m_device != VK_NULL_HANDLE && sampler.handle != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, sampler.handle, nullptr);
|
||||
}
|
||||
sampler.handle = VK_NULL_HANDLE;
|
||||
}
|
||||
m_samplers.clear();
|
||||
|
||||
if (m_device != VK_NULL_HANDLE && m_fallbackSampler != VK_NULL_HANDLE) {
|
||||
vkDestroySampler(m_device, m_fallbackSampler, nullptr);
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && m_fallbackImageView != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(m_device, m_fallbackImageView, nullptr);
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && m_fallbackImage != VK_NULL_HANDLE) {
|
||||
vkDestroyImage(m_device, m_fallbackImage, nullptr);
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && m_fallbackImageMemory != VK_NULL_HANDLE) {
|
||||
vkFreeMemory(m_device, m_fallbackImageMemory, nullptr);
|
||||
}
|
||||
m_fallbackSampler = VK_NULL_HANDLE;
|
||||
m_fallbackImageView = VK_NULL_HANDLE;
|
||||
m_fallbackImage = VK_NULL_HANDLE;
|
||||
m_fallbackImageMemory = VK_NULL_HANDLE;
|
||||
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_physicalDevice = VK_NULL_HANDLE;
|
||||
m_commandPool = VK_NULL_HANDLE;
|
||||
m_graphicsQueue = VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::GetFallbackDescriptor(VkDescriptorImageInfo& outImageInfo) const {
|
||||
if (m_fallbackSampler == VK_NULL_HANDLE || m_fallbackImageView == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
outImageInfo.sampler = m_fallbackSampler;
|
||||
outImageInfo.imageView = m_fallbackImageView;
|
||||
outImageInfo.imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::SyncTextureAndGetDescriptor(const MG_State::GLState::ITextureObject& texture,
|
||||
const MG_State::GLState::SamplerObject* samplerOverride,
|
||||
VkDescriptorImageInfo& outImageInfo) {
|
||||
if (m_device == VK_NULL_HANDLE) {
|
||||
return GetFallbackDescriptor(outImageInfo);
|
||||
}
|
||||
|
||||
auto it = m_textureResources.find(texture.GetExternalIndex());
|
||||
if (it == m_textureResources.end()) {
|
||||
TextureResource initial{};
|
||||
initial.textureExternalIndex = texture.GetExternalIndex();
|
||||
auto [insertIt, _] = m_textureResources.emplace(texture.GetExternalIndex(), initial);
|
||||
it = insertIt;
|
||||
}
|
||||
|
||||
if (!EnsureTextureSynced(it->second, texture)) {
|
||||
return GetFallbackDescriptor(outImageInfo);
|
||||
}
|
||||
|
||||
const MG_State::GLState::SamplerObject* samplerToUse = samplerOverride;
|
||||
if (!samplerToUse) {
|
||||
auto textureSampler = texture.GetSamplerObject();
|
||||
if (textureSampler) {
|
||||
samplerToUse = textureSampler.get();
|
||||
}
|
||||
}
|
||||
|
||||
VkSampler sampler = m_fallbackSampler;
|
||||
if (samplerToUse) {
|
||||
sampler = GetOrCreateSampler(*samplerToUse);
|
||||
}
|
||||
if (sampler == VK_NULL_HANDLE) {
|
||||
sampler = m_fallbackSampler;
|
||||
}
|
||||
|
||||
if (it->second.view == VK_NULL_HANDLE || sampler == VK_NULL_HANDLE) {
|
||||
return GetFallbackDescriptor(outImageInfo);
|
||||
}
|
||||
|
||||
outImageInfo.sampler = sampler;
|
||||
outImageInfo.imageView = it->second.view;
|
||||
outImageInfo.imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::EnsureTextureSynced(TextureResource& resource,
|
||||
const MG_State::GLState::ITextureObject& texture) {
|
||||
TextureUploadTarget level0Target = TextureUploadTarget::Unknown;
|
||||
IntVec3 texelSize{0, 0, 0};
|
||||
SizeT byteSize = 0;
|
||||
if (!ResolveLevel0(texture, level0Target, texelSize, byteSize)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!EnsureTextureResource(resource, texture, level0Target, texelSize, byteSize)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto* mipTexture = dynamic_cast<const MG_State::GLState::TextureObjectMipmap*>(&texture);
|
||||
if (!mipTexture) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!mipTexture->IsStorageDirty(level0Target, 0)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (!UploadLevel0(resource, *mipTexture, level0Target, byteSize)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto& mutableTexture = const_cast<MG_State::GLState::TextureObjectMipmap&>(*mipTexture);
|
||||
mutableTexture.MarkStorageDirty(level0Target, 0, false);
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::EnsureTextureResource(TextureResource& resource,
|
||||
const MG_State::GLState::ITextureObject& texture,
|
||||
TextureUploadTarget level0Target, const IntVec3& texelSize,
|
||||
SizeT byteSize) {
|
||||
const VkFormat format = ResolveTextureFormat(texture.GetFormat());
|
||||
if (format == VK_FORMAT_UNDEFINED) {
|
||||
return false;
|
||||
}
|
||||
if (texelSize.x() <= 0 || texelSize.y() <= 0 || byteSize == 0) {
|
||||
return false;
|
||||
}
|
||||
if (level0Target != TextureUploadTarget::Texture2D) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const Bool compatible = resource.image != VK_NULL_HANDLE && resource.format == format &&
|
||||
resource.extent.width == static_cast<Uint32>(texelSize.x()) &&
|
||||
resource.extent.height == static_cast<Uint32>(texelSize.y());
|
||||
if (compatible) {
|
||||
return true;
|
||||
}
|
||||
|
||||
DestroyTextureResource(resource);
|
||||
|
||||
VkImageCreateInfo imageInfo{};
|
||||
imageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
|
||||
imageInfo.imageType = VK_IMAGE_TYPE_2D;
|
||||
imageInfo.extent.width = static_cast<Uint32>(texelSize.x());
|
||||
imageInfo.extent.height = static_cast<Uint32>(texelSize.y());
|
||||
imageInfo.extent.depth = 1;
|
||||
imageInfo.mipLevels = 1;
|
||||
imageInfo.arrayLayers = 1;
|
||||
imageInfo.format = format;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
|
||||
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateImage(m_device, &imageInfo, nullptr, &resource.image), "vkCreateImage(texture)");
|
||||
|
||||
VkMemoryRequirements requirements{};
|
||||
vkGetImageMemoryRequirements(m_device, resource.image, &requirements);
|
||||
|
||||
VkMemoryAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
allocInfo.allocationSize = requirements.size;
|
||||
allocInfo.memoryTypeIndex = FindMemoryType(requirements.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(m_device, &allocInfo, nullptr, &resource.memory), "vkAllocateMemory(texture)");
|
||||
VK_VERIFY(vkBindImageMemory(m_device, resource.image, resource.memory, 0), "vkBindImageMemory(texture)");
|
||||
|
||||
VkImageViewCreateInfo viewInfo{};
|
||||
viewInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
|
||||
viewInfo.image = resource.image;
|
||||
viewInfo.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
viewInfo.format = format;
|
||||
viewInfo.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
viewInfo.subresourceRange.baseMipLevel = 0;
|
||||
viewInfo.subresourceRange.levelCount = 1;
|
||||
viewInfo.subresourceRange.baseArrayLayer = 0;
|
||||
viewInfo.subresourceRange.layerCount = 1;
|
||||
VK_VERIFY(vkCreateImageView(m_device, &viewInfo, nullptr, &resource.view), "vkCreateImageView(texture)");
|
||||
|
||||
resource.layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
resource.extent = {static_cast<Uint32>(texelSize.x()), static_cast<Uint32>(texelSize.y())};
|
||||
resource.format = format;
|
||||
resource.textureExternalIndex = texture.GetExternalIndex();
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::UploadLevel0(TextureResource& resource,
|
||||
const MG_State::GLState::TextureObjectMipmap& mipmapTexture,
|
||||
TextureUploadTarget level0Target, SizeT byteSize) {
|
||||
auto& mutableTexture = const_cast<MG_State::GLState::TextureObjectMipmap&>(mipmapTexture);
|
||||
const void* source = mutableTexture.MapMipmapData(level0Target, 0);
|
||||
if (source == nullptr || byteSize == 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
VkBuffer stagingBuffer = VK_NULL_HANDLE;
|
||||
VkDeviceMemory stagingMemory = VK_NULL_HANDLE;
|
||||
|
||||
VkBufferCreateInfo bufferInfo{};
|
||||
bufferInfo.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO;
|
||||
bufferInfo.size = byteSize;
|
||||
bufferInfo.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
|
||||
bufferInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateBuffer(m_device, &bufferInfo, nullptr, &stagingBuffer), "vkCreateBuffer(staging texture)");
|
||||
|
||||
VkMemoryRequirements requirements{};
|
||||
vkGetBufferMemoryRequirements(m_device, stagingBuffer, &requirements);
|
||||
|
||||
VkMemoryAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
allocInfo.allocationSize = requirements.size;
|
||||
allocInfo.memoryTypeIndex =
|
||||
FindMemoryType(requirements.memoryTypeBits,
|
||||
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(m_device, &allocInfo, nullptr, &stagingMemory), "vkAllocateMemory(staging texture)");
|
||||
VK_VERIFY(vkBindBufferMemory(m_device, stagingBuffer, stagingMemory, 0), "vkBindBufferMemory(staging texture)");
|
||||
|
||||
void* mapped = nullptr;
|
||||
VK_VERIFY(vkMapMemory(m_device, stagingMemory, 0, byteSize, 0, &mapped), "vkMapMemory(staging texture)");
|
||||
std::memcpy(mapped, source, byteSize);
|
||||
vkUnmapMemory(m_device, stagingMemory);
|
||||
|
||||
const Bool ok = ExecuteImmediate([&](VkCommandBuffer commandBuffer) {
|
||||
VkImageMemoryBarrier toTransferDst{};
|
||||
toTransferDst.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
toTransferDst.srcAccessMask = 0;
|
||||
toTransferDst.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
toTransferDst.oldLayout = resource.layout;
|
||||
toTransferDst.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
||||
toTransferDst.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toTransferDst.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toTransferDst.image = resource.image;
|
||||
toTransferDst.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
toTransferDst.subresourceRange.baseMipLevel = 0;
|
||||
toTransferDst.subresourceRange.levelCount = 1;
|
||||
toTransferDst.subresourceRange.baseArrayLayer = 0;
|
||||
toTransferDst.subresourceRange.layerCount = 1;
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0,
|
||||
nullptr, 0, nullptr, 1, &toTransferDst);
|
||||
|
||||
VkBufferImageCopy copy{};
|
||||
copy.bufferOffset = 0;
|
||||
copy.bufferRowLength = 0;
|
||||
copy.bufferImageHeight = 0;
|
||||
copy.imageSubresource.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
copy.imageSubresource.mipLevel = 0;
|
||||
copy.imageSubresource.baseArrayLayer = 0;
|
||||
copy.imageSubresource.layerCount = 1;
|
||||
copy.imageOffset = {0, 0, 0};
|
||||
copy.imageExtent = {resource.extent.width, resource.extent.height, 1};
|
||||
vkCmdCopyBufferToImage(commandBuffer, stagingBuffer, resource.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1,
|
||||
©);
|
||||
|
||||
VkImageMemoryBarrier toSampled{};
|
||||
toSampled.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
toSampled.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
toSampled.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
|
||||
toSampled.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
||||
toSampled.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
||||
toSampled.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toSampled.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toSampled.image = resource.image;
|
||||
toSampled.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
toSampled.subresourceRange.baseMipLevel = 0;
|
||||
toSampled.subresourceRange.levelCount = 1;
|
||||
toSampled.subresourceRange.baseArrayLayer = 0;
|
||||
toSampled.subresourceRange.layerCount = 1;
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, 0,
|
||||
0, nullptr, 0, nullptr, 1, &toSampled);
|
||||
});
|
||||
|
||||
vkDestroyBuffer(m_device, stagingBuffer, nullptr);
|
||||
vkFreeMemory(m_device, stagingMemory, nullptr);
|
||||
|
||||
if (!ok) {
|
||||
return false;
|
||||
}
|
||||
resource.layout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::ExecuteImmediate(const std::function<void(VkCommandBuffer)>& recorder) const {
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
allocInfo.commandPool = m_commandPool;
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = 1;
|
||||
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkAllocateCommandBuffers(m_device, &allocInfo, &commandBuffer), "vkAllocateCommandBuffers(texture)");
|
||||
|
||||
VkCommandBufferBeginInfo beginInfo{};
|
||||
beginInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO;
|
||||
beginInfo.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
|
||||
VK_VERIFY(vkBeginCommandBuffer(commandBuffer, &beginInfo), "vkBeginCommandBuffer(texture)");
|
||||
|
||||
recorder(commandBuffer);
|
||||
|
||||
VK_VERIFY(vkEndCommandBuffer(commandBuffer), "vkEndCommandBuffer(texture)");
|
||||
|
||||
VkSubmitInfo submitInfo{};
|
||||
submitInfo.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO;
|
||||
submitInfo.commandBufferCount = 1;
|
||||
submitInfo.pCommandBuffers = &commandBuffer;
|
||||
VK_VERIFY(vkQueueSubmit(m_graphicsQueue, 1, &submitInfo, VK_NULL_HANDLE), "vkQueueSubmit(texture)");
|
||||
VK_VERIFY(vkQueueWaitIdle(m_graphicsQueue), "vkQueueWaitIdle(texture)");
|
||||
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &commandBuffer);
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkTextureSamplerManager::DestroyTextureResource(TextureResource& resource) const {
|
||||
if (m_device != VK_NULL_HANDLE && resource.view != VK_NULL_HANDLE) {
|
||||
vkDestroyImageView(m_device, resource.view, nullptr);
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && resource.image != VK_NULL_HANDLE) {
|
||||
vkDestroyImage(m_device, resource.image, nullptr);
|
||||
}
|
||||
if (m_device != VK_NULL_HANDLE && resource.memory != VK_NULL_HANDLE) {
|
||||
vkFreeMemory(m_device, resource.memory, nullptr);
|
||||
}
|
||||
resource.view = VK_NULL_HANDLE;
|
||||
resource.image = VK_NULL_HANDLE;
|
||||
resource.memory = VK_NULL_HANDLE;
|
||||
resource.layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
resource.extent = {0, 0};
|
||||
resource.format = VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::ResolveLevel0(const MG_State::GLState::ITextureObject& texture,
|
||||
TextureUploadTarget& outTarget, IntVec3& outTexelSize,
|
||||
SizeT& outByteSize) {
|
||||
const auto* mipTexture = dynamic_cast<const MG_State::GLState::TextureObjectMipmap*>(&texture);
|
||||
if (!mipTexture) {
|
||||
return false;
|
||||
}
|
||||
const auto& targets = texture.GetUploadTargets();
|
||||
if (targets.empty()) {
|
||||
return false;
|
||||
}
|
||||
outTarget = targets.front();
|
||||
outTexelSize = mipTexture->GetMipmapTexelSize(outTarget, 0);
|
||||
outByteSize = mipTexture->GetMipmapByteSize(outTarget, 0);
|
||||
return outTexelSize.x() > 0 && outTexelSize.y() > 0 && outByteSize > 0;
|
||||
}
|
||||
|
||||
VkFormat VkTextureSamplerManager::ResolveTextureFormat(TextureInternalFormat format) {
|
||||
switch (format) {
|
||||
case TextureInternalFormat::RGBA:
|
||||
case TextureInternalFormat::RGBA8:
|
||||
return VK_FORMAT_R8G8B8A8_UNORM;
|
||||
case TextureInternalFormat::SRGB8Alpha8:
|
||||
return VK_FORMAT_R8G8B8A8_SRGB;
|
||||
default:
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
Uint32 VkTextureSamplerManager::FindMemoryType(Uint32 typeFilter, VkMemoryPropertyFlags properties) const {
|
||||
VkPhysicalDeviceMemoryProperties memProperties{};
|
||||
vkGetPhysicalDeviceMemoryProperties(m_physicalDevice, &memProperties);
|
||||
for (Uint32 i = 0; i < memProperties.memoryTypeCount; ++i) {
|
||||
if ((typeFilter & (1u << i)) != 0 &&
|
||||
(memProperties.memoryTypes[i].propertyFlags & properties) == properties) {
|
||||
return i;
|
||||
}
|
||||
}
|
||||
MOBILEGL_ASSERT(false, "VkTextureSamplerManager::FindMemoryType failed");
|
||||
return 0;
|
||||
}
|
||||
|
||||
VkSampler VkTextureSamplerManager::GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler) {
|
||||
const Uint64 key = BuildSamplerKey(sampler.GetExternalIndex(), sampler.GetVersion());
|
||||
auto it = m_samplers.find(key);
|
||||
if (it != m_samplers.end()) {
|
||||
return it->second.handle;
|
||||
}
|
||||
|
||||
VkSamplerCreateInfo samplerInfo{};
|
||||
samplerInfo.sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO;
|
||||
samplerInfo.magFilter = ToVkFilter(sampler.GetMagFilter());
|
||||
samplerInfo.minFilter = ToVkFilter(sampler.GetMinFilter());
|
||||
samplerInfo.mipmapMode = ToVkMipmapMode(sampler.GetMipmapMode());
|
||||
samplerInfo.addressModeU = ToVkAddressMode(sampler.GetWrapS());
|
||||
samplerInfo.addressModeV = ToVkAddressMode(sampler.GetWrapT());
|
||||
samplerInfo.addressModeW = ToVkAddressMode(sampler.GetWrapR());
|
||||
samplerInfo.mipLodBias = sampler.GetLodBias();
|
||||
samplerInfo.anisotropyEnable = VK_FALSE;
|
||||
samplerInfo.maxAnisotropy = 1.0f;
|
||||
samplerInfo.compareEnable = sampler.GetCompareMode() == SamplerCompareMode::CompareToTexture ? VK_TRUE : VK_FALSE;
|
||||
samplerInfo.compareOp = ToVkCompareOp(sampler.GetSamplerCompareFunc());
|
||||
samplerInfo.minLod = sampler.GetMinLod();
|
||||
samplerInfo.maxLod = sampler.GetMaxLod();
|
||||
samplerInfo.borderColor = VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
|
||||
samplerInfo.unnormalizedCoordinates = VK_FALSE;
|
||||
|
||||
VkSampler vkSampler = VK_NULL_HANDLE;
|
||||
VK_VERIFY(vkCreateSampler(m_device, &samplerInfo, nullptr, &vkSampler), "vkCreateSampler(texture)");
|
||||
|
||||
SamplerCacheEntry entry{};
|
||||
entry.handle = vkSampler;
|
||||
entry.externalIndex = sampler.GetExternalIndex();
|
||||
entry.version = sampler.GetVersion();
|
||||
m_samplers[key] = entry;
|
||||
return vkSampler;
|
||||
}
|
||||
|
||||
VkFilter VkTextureSamplerManager::ToVkFilter(SamplerFilterMode mode) {
|
||||
return mode == SamplerFilterMode::Nearest ? VK_FILTER_NEAREST : VK_FILTER_LINEAR;
|
||||
}
|
||||
|
||||
VkSamplerMipmapMode VkTextureSamplerManager::ToVkMipmapMode(SamplerMipmapMode mode) {
|
||||
switch (mode) {
|
||||
case SamplerMipmapMode::Nearest:
|
||||
return VK_SAMPLER_MIPMAP_MODE_NEAREST;
|
||||
case SamplerMipmapMode::Linear:
|
||||
return VK_SAMPLER_MIPMAP_MODE_LINEAR;
|
||||
case SamplerMipmapMode::None:
|
||||
default:
|
||||
return VK_SAMPLER_MIPMAP_MODE_NEAREST;
|
||||
}
|
||||
}
|
||||
|
||||
VkSamplerAddressMode VkTextureSamplerManager::ToVkAddressMode(SamplerWrapMode mode) {
|
||||
switch (mode) {
|
||||
case SamplerWrapMode::ClampToEdge:
|
||||
return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
||||
case SamplerWrapMode::MirroredRepeat:
|
||||
return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT;
|
||||
case SamplerWrapMode::Repeat:
|
||||
return VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
||||
case SamplerWrapMode::ClampToBorder:
|
||||
return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER;
|
||||
case SamplerWrapMode::MirrorClampToEdge:
|
||||
return VK_SAMPLER_ADDRESS_MODE_MIRROR_CLAMP_TO_EDGE;
|
||||
default:
|
||||
return VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
||||
}
|
||||
}
|
||||
|
||||
VkCompareOp VkTextureSamplerManager::ToVkCompareOp(SamplerCompareFunc func) {
|
||||
switch (func) {
|
||||
case SamplerCompareFunc::Never:
|
||||
return VK_COMPARE_OP_NEVER;
|
||||
case SamplerCompareFunc::Less:
|
||||
return VK_COMPARE_OP_LESS;
|
||||
case SamplerCompareFunc::Equal:
|
||||
return VK_COMPARE_OP_EQUAL;
|
||||
case SamplerCompareFunc::LessEqual:
|
||||
return VK_COMPARE_OP_LESS_OR_EQUAL;
|
||||
case SamplerCompareFunc::Greater:
|
||||
return VK_COMPARE_OP_GREATER;
|
||||
case SamplerCompareFunc::NotEqual:
|
||||
return VK_COMPARE_OP_NOT_EQUAL;
|
||||
case SamplerCompareFunc::GreaterEqual:
|
||||
return VK_COMPARE_OP_GREATER_OR_EQUAL;
|
||||
case SamplerCompareFunc::Always:
|
||||
default:
|
||||
return VK_COMPARE_OP_ALWAYS;
|
||||
}
|
||||
}
|
||||
|
||||
Bool VkTextureSamplerManager::UploadFallbackTexture() {
|
||||
const Uint32 rgba = 0xFFFFFFFFu;
|
||||
|
||||
VkImageCreateInfo imageInfo{};
|
||||
imageInfo.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
|
||||
imageInfo.imageType = VK_IMAGE_TYPE_2D;
|
||||
imageInfo.extent = {1, 1, 1};
|
||||
imageInfo.mipLevels = 1;
|
||||
imageInfo.arrayLayers = 1;
|
||||
imageInfo.format = VK_FORMAT_R8G8B8A8_UNORM;
|
||||
imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT;
|
||||
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT;
|
||||
imageInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateImage(m_device, &imageInfo, nullptr, &m_fallbackImage), "vkCreateImage(fallback)");
|
||||
|
||||
VkMemoryRequirements imageMemReq{};
|
||||
vkGetImageMemoryRequirements(m_device, m_fallbackImage, &imageMemReq);
|
||||
|
||||
VkMemoryAllocateInfo imageAllocInfo{};
|
||||
imageAllocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
imageAllocInfo.allocationSize = imageMemReq.size;
|
||||
imageAllocInfo.memoryTypeIndex = FindMemoryType(imageMemReq.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(m_device, &imageAllocInfo, nullptr, &m_fallbackImageMemory),
|
||||
"vkAllocateMemory(fallback)");
|
||||
VK_VERIFY(vkBindImageMemory(m_device, m_fallbackImage, m_fallbackImageMemory, 0), "vkBindImageMemory(fallback)");
|
||||
|
||||
VkBuffer stagingBuffer = VK_NULL_HANDLE;
|
||||
VkDeviceMemory stagingMemory = VK_NULL_HANDLE;
|
||||
|
||||
VkBufferCreateInfo bufferInfo{};
|
||||
bufferInfo.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO;
|
||||
bufferInfo.size = sizeof(rgba);
|
||||
bufferInfo.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
|
||||
bufferInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
||||
VK_VERIFY(vkCreateBuffer(m_device, &bufferInfo, nullptr, &stagingBuffer), "vkCreateBuffer(fallback)");
|
||||
|
||||
VkMemoryRequirements stagingMemReq{};
|
||||
vkGetBufferMemoryRequirements(m_device, stagingBuffer, &stagingMemReq);
|
||||
|
||||
VkMemoryAllocateInfo stagingAllocInfo{};
|
||||
stagingAllocInfo.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO;
|
||||
stagingAllocInfo.allocationSize = stagingMemReq.size;
|
||||
stagingAllocInfo.memoryTypeIndex =
|
||||
FindMemoryType(stagingMemReq.memoryTypeBits,
|
||||
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
|
||||
VK_VERIFY(vkAllocateMemory(m_device, &stagingAllocInfo, nullptr, &stagingMemory), "vkAllocateMemory(fallback)");
|
||||
VK_VERIFY(vkBindBufferMemory(m_device, stagingBuffer, stagingMemory, 0), "vkBindBufferMemory(fallback)");
|
||||
|
||||
void* mapped = nullptr;
|
||||
VK_VERIFY(vkMapMemory(m_device, stagingMemory, 0, sizeof(rgba), 0, &mapped), "vkMapMemory(fallback)");
|
||||
std::memcpy(mapped, &rgba, sizeof(rgba));
|
||||
vkUnmapMemory(m_device, stagingMemory);
|
||||
|
||||
const Bool uploadOk = ExecuteImmediate([&](VkCommandBuffer commandBuffer) {
|
||||
VkImageMemoryBarrier toTransferDst{};
|
||||
toTransferDst.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
toTransferDst.srcAccessMask = 0;
|
||||
toTransferDst.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
toTransferDst.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
toTransferDst.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
||||
toTransferDst.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toTransferDst.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toTransferDst.image = m_fallbackImage;
|
||||
toTransferDst.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
toTransferDst.subresourceRange.baseMipLevel = 0;
|
||||
toTransferDst.subresourceRange.levelCount = 1;
|
||||
toTransferDst.subresourceRange.baseArrayLayer = 0;
|
||||
toTransferDst.subresourceRange.layerCount = 1;
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0,
|
||||
nullptr, 0, nullptr, 1, &toTransferDst);
|
||||
|
||||
VkBufferImageCopy copy{};
|
||||
copy.imageSubresource.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
copy.imageSubresource.mipLevel = 0;
|
||||
copy.imageSubresource.baseArrayLayer = 0;
|
||||
copy.imageSubresource.layerCount = 1;
|
||||
copy.imageExtent = {1, 1, 1};
|
||||
vkCmdCopyBufferToImage(commandBuffer, stagingBuffer, m_fallbackImage, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1,
|
||||
©);
|
||||
|
||||
VkImageMemoryBarrier toSampled{};
|
||||
toSampled.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
||||
toSampled.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
toSampled.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
|
||||
toSampled.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
||||
toSampled.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
||||
toSampled.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toSampled.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
toSampled.image = m_fallbackImage;
|
||||
toSampled.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
toSampled.subresourceRange.baseMipLevel = 0;
|
||||
toSampled.subresourceRange.levelCount = 1;
|
||||
toSampled.subresourceRange.baseArrayLayer = 0;
|
||||
toSampled.subresourceRange.layerCount = 1;
|
||||
vkCmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, 0,
|
||||
0, nullptr, 0, nullptr, 1, &toSampled);
|
||||
});
|
||||
|
||||
vkDestroyBuffer(m_device, stagingBuffer, nullptr);
|
||||
vkFreeMemory(m_device, stagingMemory, nullptr);
|
||||
if (!uploadOk) {
|
||||
return false;
|
||||
}
|
||||
|
||||
VkImageViewCreateInfo viewInfo{};
|
||||
viewInfo.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
|
||||
viewInfo.image = m_fallbackImage;
|
||||
viewInfo.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
||||
viewInfo.format = VK_FORMAT_R8G8B8A8_UNORM;
|
||||
viewInfo.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
viewInfo.subresourceRange.baseMipLevel = 0;
|
||||
viewInfo.subresourceRange.levelCount = 1;
|
||||
viewInfo.subresourceRange.baseArrayLayer = 0;
|
||||
viewInfo.subresourceRange.layerCount = 1;
|
||||
VK_VERIFY(vkCreateImageView(m_device, &viewInfo, nullptr, &m_fallbackImageView), "vkCreateImageView(fallback)");
|
||||
|
||||
VkSamplerCreateInfo samplerInfo{};
|
||||
samplerInfo.sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO;
|
||||
samplerInfo.magFilter = VK_FILTER_NEAREST;
|
||||
samplerInfo.minFilter = VK_FILTER_NEAREST;
|
||||
samplerInfo.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST;
|
||||
samplerInfo.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
||||
samplerInfo.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
||||
samplerInfo.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
||||
samplerInfo.compareEnable = VK_FALSE;
|
||||
samplerInfo.minLod = 0.0f;
|
||||
samplerInfo.maxLod = 0.0f;
|
||||
samplerInfo.maxAnisotropy = 1.0f;
|
||||
VK_VERIFY(vkCreateSampler(m_device, &samplerInfo, nullptr, &m_fallbackSampler), "vkCreateSampler(fallback)");
|
||||
return true;
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -1,88 +0,0 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkTextureSamplerManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
#include <MG_State/GLState/SamplerState/SamplerObject.h>
|
||||
#include <MG_State/GLState/TextureState/TextureObject.h>
|
||||
|
||||
namespace MobileGL::MG_State::GLState {
|
||||
class ITextureObject;
|
||||
class SamplerObject;
|
||||
}
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
class VkTextureSamplerManager {
|
||||
public:
|
||||
struct InitInfo {
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice physicalDevice = VK_NULL_HANDLE;
|
||||
VkCommandPool commandPool = VK_NULL_HANDLE;
|
||||
VkQueue graphicsQueue = VK_NULL_HANDLE;
|
||||
};
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
void Shutdown();
|
||||
|
||||
Bool GetFallbackDescriptor(VkDescriptorImageInfo& outImageInfo) const;
|
||||
Bool SyncTextureAndGetDescriptor(const MG_State::GLState::ITextureObject& texture,
|
||||
const MG_State::GLState::SamplerObject* samplerOverride,
|
||||
VkDescriptorImageInfo& outImageInfo);
|
||||
|
||||
private:
|
||||
struct TextureResource {
|
||||
VkImage image = VK_NULL_HANDLE;
|
||||
VkDeviceMemory memory = VK_NULL_HANDLE;
|
||||
VkImageView view = VK_NULL_HANDLE;
|
||||
VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkExtent2D extent = {0, 0};
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
Uint textureExternalIndex = 0;
|
||||
};
|
||||
|
||||
struct SamplerCacheEntry {
|
||||
VkSampler handle = VK_NULL_HANDLE;
|
||||
Uint externalIndex = 0;
|
||||
Uint16 version = 0;
|
||||
};
|
||||
|
||||
Bool EnsureTextureSynced(TextureResource& resource, const MG_State::GLState::ITextureObject& texture);
|
||||
Bool EnsureTextureResource(TextureResource& resource, const MG_State::GLState::ITextureObject& texture,
|
||||
TextureUploadTarget level0Target, const IntVec3& texelSize, SizeT byteSize);
|
||||
Bool UploadLevel0(TextureResource& resource, const MG_State::GLState::TextureObjectMipmap& mipmapTexture,
|
||||
TextureUploadTarget level0Target, SizeT byteSize);
|
||||
Bool ExecuteImmediate(const std::function<void(VkCommandBuffer)>& recorder) const;
|
||||
void DestroyTextureResource(TextureResource& resource) const;
|
||||
static Bool ResolveLevel0(const MG_State::GLState::ITextureObject& texture, TextureUploadTarget& outTarget,
|
||||
IntVec3& outTexelSize, SizeT& outByteSize);
|
||||
static VkFormat ResolveTextureFormat(TextureInternalFormat format);
|
||||
Uint32 FindMemoryType(Uint32 typeFilter, VkMemoryPropertyFlags properties) const;
|
||||
|
||||
VkSampler GetOrCreateSampler(const MG_State::GLState::SamplerObject& sampler);
|
||||
static VkFilter ToVkFilter(SamplerFilterMode mode);
|
||||
static VkSamplerMipmapMode ToVkMipmapMode(SamplerMipmapMode mode);
|
||||
static VkSamplerAddressMode ToVkAddressMode(SamplerWrapMode mode);
|
||||
static VkCompareOp ToVkCompareOp(SamplerCompareFunc func);
|
||||
Bool UploadFallbackTexture();
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
VkPhysicalDevice m_physicalDevice = VK_NULL_HANDLE;
|
||||
VkCommandPool m_commandPool = VK_NULL_HANDLE;
|
||||
VkQueue m_graphicsQueue = VK_NULL_HANDLE;
|
||||
|
||||
UnorderedMap<Uint, TextureResource> m_textureResources;
|
||||
UnorderedMap<Uint64, SamplerCacheEntry> m_samplers;
|
||||
|
||||
VkImage m_fallbackImage = VK_NULL_HANDLE;
|
||||
VkDeviceMemory m_fallbackImageMemory = VK_NULL_HANDLE;
|
||||
VkImageView m_fallbackImageView = VK_NULL_HANDLE;
|
||||
VkSampler m_fallbackSampler = VK_NULL_HANDLE;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,179 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkTimerQueryManager.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "VkTimerQueryManager.h"
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Bool VkTimerQueryManager::Initialize(const InitInfo& initInfo) {
|
||||
Shutdown();
|
||||
|
||||
MOBILEGL_ASSERT(initInfo.device != VK_NULL_HANDLE, "VkTimerQueryManager::Initialize requires valid VkDevice");
|
||||
MOBILEGL_ASSERT(initInfo.frameCount > 0, "VkTimerQueryManager::Initialize requires non-zero frame count");
|
||||
if (initInfo.timestampValidBits == 0 || initInfo.timestampPeriodNs <= 0.0f || initInfo.slotsPerPool == 0) {
|
||||
MGLOG_W_ONCE("VkTimerQueryManager: timestamps unsupported (validBits=%u, period=%f, slots=%u)",
|
||||
initInfo.timestampValidBits, initInfo.timestampPeriodNs, initInfo.slotsPerPool);
|
||||
return false;
|
||||
}
|
||||
|
||||
m_device = initInfo.device;
|
||||
m_timestampPeriodNs = initInfo.timestampPeriodNs;
|
||||
m_validBitsMask = initInfo.timestampValidBits >= 64
|
||||
? ~0ull
|
||||
: ((1ull << initInfo.timestampValidBits) - 1ull);
|
||||
m_slotsPerPool = initInfo.slotsPerPool;
|
||||
m_pools.resize(initInfo.frameCount);
|
||||
|
||||
VkQueryPoolCreateInfo poolInfo{};
|
||||
poolInfo.sType = VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO;
|
||||
poolInfo.queryType = VK_QUERY_TYPE_TIMESTAMP;
|
||||
poolInfo.queryCount = m_slotsPerPool;
|
||||
for (auto& poolState : m_pools) {
|
||||
const VkResult result = vkCreateQueryPool(m_device, &poolInfo, nullptr, &poolState.pool);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E_ONCE("VkTimerQueryManager: vkCreateQueryPool failed with %s", VkResultToString(result));
|
||||
Shutdown();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkTimerQueryManager::Shutdown() {
|
||||
if (m_device != VK_NULL_HANDLE) {
|
||||
for (auto& poolState : m_pools) {
|
||||
if (poolState.pool != VK_NULL_HANDLE) {
|
||||
vkDestroyQueryPool(m_device, poolState.pool, nullptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
// Records the frontend still holds simply stay unharvested; their
|
||||
// results read back as 0.
|
||||
m_pools.clear();
|
||||
m_device = VK_NULL_HANDLE;
|
||||
m_timestampPeriodNs = 0.0f;
|
||||
m_validBitsMask = 0;
|
||||
m_slotsPerPool = 0;
|
||||
}
|
||||
|
||||
void VkTimerQueryManager::OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer, Uint32 frameIndex,
|
||||
Uint64 frameSerial) {
|
||||
MOBILEGL_ASSERT(frameIndex < m_pools.size(), "VkTimerQueryManager frame index out of range");
|
||||
auto& poolState = m_pools[frameIndex];
|
||||
if (poolState.preparedFrameSerial == frameSerial) {
|
||||
// Recording re-began within the same frame (mid-frame readback
|
||||
// submit or the Present layout transition); the pool was already
|
||||
// harvested and reset for this cycle, and resetting again would
|
||||
// clobber timestamps written earlier in the frame.
|
||||
return;
|
||||
}
|
||||
|
||||
// Harvest what the pool's previous cycle left behind. The frame slot's
|
||||
// fence was waited before re-recording, so every executed query is
|
||||
// already available and the reads return immediately.
|
||||
DrainPoolPending(poolState);
|
||||
|
||||
vkCmdResetQueryPool(commandBuffer, poolState.pool, 0, m_slotsPerPool);
|
||||
poolState.cursor = 0;
|
||||
poolState.exhaustionWarned = false;
|
||||
poolState.preparedFrameSerial = frameSerial;
|
||||
}
|
||||
|
||||
SharedPtr<VkTimerQueryManager::TimestampRecord> VkTimerQueryManager::WriteTimestamp(VkCommandBuffer commandBuffer,
|
||||
Uint32 frameIndex,
|
||||
Uint64 frameSerial) {
|
||||
MOBILEGL_ASSERT(frameIndex < m_pools.size(), "VkTimerQueryManager frame index out of range");
|
||||
auto& poolState = m_pools[frameIndex];
|
||||
if (poolState.cursor >= m_slotsPerPool) {
|
||||
if (!poolState.exhaustionWarned) {
|
||||
MGLOG_W_ONCE("VkTimerQueryManager: frame %u timestamp pool exhausted (%u slots); further timer queries "
|
||||
"this frame fall back to the frontend path",
|
||||
frameIndex, m_slotsPerPool);
|
||||
poolState.exhaustionWarned = true;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto record = MakeShared<TimestampRecord>();
|
||||
record->poolIndex = frameIndex;
|
||||
record->slot = poolState.cursor++;
|
||||
record->frameSerial = frameSerial;
|
||||
vkCmdWriteTimestamp(commandBuffer, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, poolState.pool, record->slot);
|
||||
poolState.pendingRecords.push_back(record);
|
||||
return record;
|
||||
}
|
||||
|
||||
Bool VkTimerQueryManager::TryHarvest(TimestampRecord& record) {
|
||||
if (record.harvested) {
|
||||
return true;
|
||||
}
|
||||
if (m_device == VK_NULL_HANDLE || record.poolIndex >= m_pools.size()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Uint64 resultWithAvailability[2] = {0, 0};
|
||||
const VkResult result = vkGetQueryPoolResults(
|
||||
m_device, m_pools[record.poolIndex].pool, record.slot, 1, sizeof(resultWithAvailability),
|
||||
resultWithAvailability, sizeof(Uint64), VK_QUERY_RESULT_64_BIT | VK_QUERY_RESULT_WITH_AVAILABILITY_BIT);
|
||||
if (result != VK_SUCCESS && result != VK_NOT_READY) {
|
||||
MGLOG_E_ONCE("VkTimerQueryManager: vkGetQueryPoolResults failed with %s", VkResultToString(result));
|
||||
return false;
|
||||
}
|
||||
if (resultWithAvailability[1] == 0) {
|
||||
return false;
|
||||
}
|
||||
record.rawTicks = resultWithAvailability[0];
|
||||
record.harvested = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
void VkTimerQueryManager::InvalidatePendingRecords() {
|
||||
for (auto& poolState : m_pools) {
|
||||
DrainPoolPending(poolState);
|
||||
// Force a harvest-free reset cycle the next time this pool's frame
|
||||
// begins recording.
|
||||
poolState.preparedFrameSerial = 0;
|
||||
}
|
||||
}
|
||||
|
||||
void VkTimerQueryManager::DrainPoolPending(PoolState& poolState) {
|
||||
for (auto& record : poolState.pendingRecords) {
|
||||
if (record->harvested) {
|
||||
continue;
|
||||
}
|
||||
if (!TryHarvest(*record)) {
|
||||
// The commands carrying this timestamp never executed (they
|
||||
// were dropped, e.g. by a swapchain recreation mid-frame).
|
||||
// Mark the record resolved-as-invalid so waits on it cannot
|
||||
// hang; its result reads back as 0.
|
||||
record->harvested = true;
|
||||
record->valid = false;
|
||||
}
|
||||
}
|
||||
poolState.pendingRecords.clear();
|
||||
}
|
||||
|
||||
Uint64 VkTimerQueryManager::MaskToValidBits(Uint64 ticks) const {
|
||||
return ticks & m_validBitsMask;
|
||||
}
|
||||
|
||||
Uint64 VkTimerQueryManager::ElapsedNs(const TimestampRecord& begin, const TimestampRecord& end) const {
|
||||
if (!begin.valid || !end.valid) {
|
||||
return 0;
|
||||
}
|
||||
const Uint64 deltaTicks = MaskToValidBits(end.rawTicks - begin.rawTicks);
|
||||
return static_cast<Uint64>(static_cast<double>(deltaTicks) * static_cast<double>(m_timestampPeriodNs));
|
||||
}
|
||||
|
||||
Uint64 VkTimerQueryManager::TimestampNs(const TimestampRecord& record) const {
|
||||
if (!record.valid) {
|
||||
return 0;
|
||||
}
|
||||
return static_cast<Uint64>(static_cast<double>(MaskToValidBits(record.rawTicks)) *
|
||||
static_cast<double>(m_timestampPeriodNs));
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
@@ -0,0 +1,111 @@
|
||||
// MobileGL - MobileGL/MG_Backend/DirectVulkan/Renderer/VkTimerQueryManager.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../VkIncludes.h"
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// GPU timestamp storage backing the GL timer-query frontend (GL_TIME_ELAPSED
|
||||
// spans and GL_TIMESTAMP one-shots): one VkQueryPool of timestamp slots per
|
||||
// frame in flight.
|
||||
//
|
||||
// Per-frame lifecycle: right after a frame slot's command buffer begins
|
||||
// recording (and before any render pass, since vkCmdResetQueryPool must be
|
||||
// recorded outside one), OnFrameCommandRecordingBegan harvests every
|
||||
// not-yet-read slot of the pool about to be reused (the slot's frame fence
|
||||
// was waited before re-recording, so the results are already available),
|
||||
// records a reset of the whole pool, and rewinds the allocation cursor.
|
||||
class VkTimerQueryManager {
|
||||
public:
|
||||
// One vkCmdWriteTimestamp landing spot. Shared (via SharedPtr) between
|
||||
// the frontend-held query object and the owning pool's pending list, so
|
||||
// deleting a query while its result is still in flight never leaves the
|
||||
// pool with a dangling record.
|
||||
struct TimestampRecord {
|
||||
Uint32 poolIndex = 0;
|
||||
Uint32 slot = 0;
|
||||
// VkBufferManager frame serial current when the timestamp was
|
||||
// recorded; result availability is bounded by its completion.
|
||||
Uint64 frameSerial = 0;
|
||||
Bool harvested = false;
|
||||
// Cleared when the recorded commands were dropped before they could
|
||||
// execute (swapchain recreation abandons the in-progress command
|
||||
// buffer); the result then reads back as 0.
|
||||
Bool valid = true;
|
||||
Uint64 rawTicks = 0;
|
||||
};
|
||||
|
||||
struct InitInfo {
|
||||
VkDevice device = VK_NULL_HANDLE;
|
||||
Uint32 frameCount = 0;
|
||||
Uint32 timestampValidBits = 0;
|
||||
Float timestampPeriodNs = 0.0f; // nanoseconds per timestamp tick
|
||||
Uint32 slotsPerPool = 128;
|
||||
};
|
||||
|
||||
Bool Initialize(const InitInfo& initInfo);
|
||||
// The caller guarantees the device is idle (same contract as the other
|
||||
// DirectVulkan managers' Shutdown paths).
|
||||
void Shutdown();
|
||||
|
||||
// The per-frame hook described in the class comment. Re-begins within
|
||||
// the same frame serial (mid-frame readback submits, the Present layout
|
||||
// transition) are skipped so already-written slots survive.
|
||||
void OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer, Uint32 frameIndex, Uint64 frameSerial);
|
||||
|
||||
// Allocates a slot from the frame's pool and records a bottom-of-pipe
|
||||
// vkCmdWriteTimestamp (valid both inside and outside a render pass).
|
||||
// Returns null on pool exhaustion, with one warning per pool cycle; the
|
||||
// frontend falls back gracefully on a null handle.
|
||||
SharedPtr<TimestampRecord> WriteTimestamp(VkCommandBuffer commandBuffer, Uint32 frameIndex,
|
||||
Uint64 frameSerial);
|
||||
|
||||
// Non-blocking single-slot read (WITH_AVAILABILITY, no WAIT). Returns
|
||||
// true once the record holds its raw ticks. Callers gate this on the
|
||||
// record's frame serial being complete.
|
||||
Bool TryHarvest(TimestampRecord& record);
|
||||
|
||||
// Reads every pending result that is available (the caller guarantees
|
||||
// the device is idle) and marks the rest invalid. Called when recorded
|
||||
// but unsubmitted commands are dropped (swapchain recreation), which
|
||||
// would otherwise leave slots that never become available. Each pool is
|
||||
// reset lazily on its next OnFrameCommandRecordingBegan.
|
||||
void InvalidatePendingRecords();
|
||||
|
||||
// end - begin using unsigned wrap arithmetic masked to the queue's
|
||||
// timestampValidBits, converted to nanoseconds. 0 if either record was
|
||||
// invalidated.
|
||||
Uint64 ElapsedNs(const TimestampRecord& begin, const TimestampRecord& end) const;
|
||||
// Raw GPU timestamp converted to nanoseconds. 0 if invalidated.
|
||||
Uint64 TimestampNs(const TimestampRecord& record) const;
|
||||
|
||||
private:
|
||||
struct PoolState {
|
||||
VkQueryPool pool = VK_NULL_HANDLE;
|
||||
Uint32 cursor = 0;
|
||||
// Frame serial the pool was last harvested + reset for; guards
|
||||
// against double resets when recording re-begins mid-frame.
|
||||
Uint64 preparedFrameSerial = 0;
|
||||
Bool exhaustionWarned = false;
|
||||
Vector<SharedPtr<TimestampRecord>> pendingRecords;
|
||||
};
|
||||
|
||||
Uint64 MaskToValidBits(Uint64 ticks) const;
|
||||
// Harvest (or invalidate, when the result never became available)
|
||||
// every pending record of a pool and clear its pending list.
|
||||
void DrainPoolPending(PoolState& pool);
|
||||
|
||||
VkDevice m_device = VK_NULL_HANDLE;
|
||||
Float m_timestampPeriodNs = 0.0f;
|
||||
Uint64 m_validBitsMask = 0;
|
||||
Uint32 m_slotsPerPool = 0;
|
||||
Vector<PoolState> m_pools;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -10,16 +10,102 @@
|
||||
|
||||
#include "VulkanRendererConfig.h"
|
||||
|
||||
#define ENUM_STR_CASE(c) case c: return #c;
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
inline const char* VkResultToString(VkResult result) {
|
||||
switch (result) {
|
||||
ENUM_STR_CASE(VK_SUCCESS)
|
||||
ENUM_STR_CASE(VK_NOT_READY)
|
||||
ENUM_STR_CASE(VK_TIMEOUT)
|
||||
ENUM_STR_CASE(VK_EVENT_SET)
|
||||
ENUM_STR_CASE(VK_EVENT_RESET)
|
||||
ENUM_STR_CASE(VK_INCOMPLETE)
|
||||
ENUM_STR_CASE(VK_ERROR_OUT_OF_HOST_MEMORY)
|
||||
ENUM_STR_CASE(VK_ERROR_OUT_OF_DEVICE_MEMORY)
|
||||
ENUM_STR_CASE(VK_ERROR_INITIALIZATION_FAILED)
|
||||
ENUM_STR_CASE(VK_ERROR_DEVICE_LOST)
|
||||
ENUM_STR_CASE(VK_ERROR_MEMORY_MAP_FAILED)
|
||||
ENUM_STR_CASE(VK_ERROR_LAYER_NOT_PRESENT)
|
||||
ENUM_STR_CASE(VK_ERROR_EXTENSION_NOT_PRESENT)
|
||||
ENUM_STR_CASE(VK_ERROR_FEATURE_NOT_PRESENT)
|
||||
ENUM_STR_CASE(VK_ERROR_INCOMPATIBLE_DRIVER)
|
||||
ENUM_STR_CASE(VK_ERROR_TOO_MANY_OBJECTS)
|
||||
ENUM_STR_CASE(VK_ERROR_FORMAT_NOT_SUPPORTED)
|
||||
ENUM_STR_CASE(VK_ERROR_FRAGMENTED_POOL)
|
||||
ENUM_STR_CASE(VK_ERROR_UNKNOWN)
|
||||
ENUM_STR_CASE(VK_ERROR_OUT_OF_POOL_MEMORY)
|
||||
ENUM_STR_CASE(VK_ERROR_INVALID_EXTERNAL_HANDLE)
|
||||
ENUM_STR_CASE(VK_ERROR_FRAGMENTATION)
|
||||
ENUM_STR_CASE(VK_ERROR_INVALID_OPAQUE_CAPTURE_ADDRESS)
|
||||
ENUM_STR_CASE(VK_PIPELINE_COMPILE_REQUIRED)
|
||||
ENUM_STR_CASE(VK_ERROR_SURFACE_LOST_KHR)
|
||||
ENUM_STR_CASE(VK_ERROR_NATIVE_WINDOW_IN_USE_KHR)
|
||||
ENUM_STR_CASE(VK_SUBOPTIMAL_KHR)
|
||||
ENUM_STR_CASE(VK_ERROR_OUT_OF_DATE_KHR)
|
||||
ENUM_STR_CASE(VK_ERROR_INCOMPATIBLE_DISPLAY_KHR)
|
||||
ENUM_STR_CASE(VK_ERROR_VALIDATION_FAILED_EXT)
|
||||
ENUM_STR_CASE(VK_ERROR_INVALID_SHADER_NV)
|
||||
default:
|
||||
return "VK_RESULT_UNKNOWN";
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// GL renders into sRGB color attachments RAW while GL_FRAMEBUFFER_SRGB is disabled
|
||||
// (the core-profile default); Vulkan sRGB attachments always encode on write. The
|
||||
// attachment view (and render pass format) therefore drops to the UNORM twin
|
||||
// whenever the capability is off. Sampled views keep the sRGB format (decode on
|
||||
// sample is unconditional in GL).
|
||||
inline VkFormat ResolveSrgbAttachmentWriteFormat(VkFormat format, bool framebufferSrgbEnabled) {
|
||||
if (framebufferSrgbEnabled) return format;
|
||||
switch (format) {
|
||||
case VK_FORMAT_R8G8B8A8_SRGB:
|
||||
return VK_FORMAT_R8G8B8A8_UNORM;
|
||||
case VK_FORMAT_B8G8R8A8_SRGB:
|
||||
return VK_FORMAT_B8G8R8A8_UNORM;
|
||||
default:
|
||||
return format;
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
// The context line (__VA_ARGS__ = its own format string + args) must be a SEPARATE log
|
||||
// call: appending its format to the base format while its arguments precede the base
|
||||
// arguments makes every conversion read the wrong slot (a %s pulling an int crashes).
|
||||
//
|
||||
// MGLOG_F and deliberately NOT latched. VK_VERIFY is the invariant-check macro: a Vulkan call
|
||||
// MobileGL believes it has already made legal came back non-success, which is a
|
||||
// should-never-happen state, not an expected failure mode a user hits. Those fast-fail loudly
|
||||
// and keep saying so - the log-quietness rules that latch W/E cover expected failures (driver
|
||||
// capability gaps, app misuse), not broken internal invariants. MOBILEGL_ASSERT below traps in
|
||||
// a DEBUG build; MGLOG_F is what makes the same condition visible in an INFO test run, where
|
||||
// the assert is compiled out by contract.
|
||||
//
|
||||
// A soft, recoverable failure must therefore NOT be routed through VK_VERIFY. Check the
|
||||
// VkResult directly and report it with MGLOG_E_ONCE - see VkTextureManager::SyncTextureResource,
|
||||
// where a driver legitimately refuses an image the format pre-check accepted.
|
||||
#define VK_VERIFY(expr, ...) \
|
||||
do { \
|
||||
VkResult _vk_verify_result = (expr); \
|
||||
MOBILEGL_ASSERT(_vk_verify_result == VK_SUCCESS, "Vulkan error %d at %s:%d" __VA_OPT__(" - ") __VA_ARGS__, _vk_verify_result, __FILE__, __LINE__); \
|
||||
if (_vk_verify_result != VK_SUCCESS) { \
|
||||
__VA_OPT__(MGLOG_F(__VA_ARGS__);) \
|
||||
MGLOG_F("Vulkan error %s (%d) at %s:%d", \
|
||||
MobileGL::MG_Backend::DirectVulkan::VkResultToString(_vk_verify_result), \
|
||||
_vk_verify_result, __FILE__, __LINE__); \
|
||||
} \
|
||||
MOBILEGL_ASSERT(_vk_verify_result == VK_SUCCESS, "Vulkan error %s (%d) at %s:%d", \
|
||||
MobileGL::MG_Backend::DirectVulkan::VkResultToString(_vk_verify_result), \
|
||||
_vk_verify_result, __FILE__, __LINE__); \
|
||||
} while (0)
|
||||
|
||||
#define ENUM_STR_CASE(c) case c: return #c;
|
||||
|
||||
#define XXHASH_VERIFY(expr, ...) \
|
||||
do { \
|
||||
XXH_errorcode _xxh_verify_result = (expr); \
|
||||
MOBILEGL_ASSERT(_xxh_verify_result == XXH_OK, "XXHash error %d at %s:%d" __VA_OPT__(" - ") __VA_ARGS__, _xxh_verify_result, __FILE__, __LINE__); \
|
||||
} while (0)
|
||||
if (_xxh_verify_result != XXH_OK) { \
|
||||
__VA_OPT__(MGLOG_F(__VA_ARGS__);) \
|
||||
} \
|
||||
MOBILEGL_ASSERT(_xxh_verify_result == XXH_OK, "XXHash error %d at %s:%d", _xxh_verify_result, __FILE__, \
|
||||
__LINE__); \
|
||||
} while (0)
|
||||
|
||||
@@ -11,10 +11,22 @@
|
||||
|
||||
namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
struct VulkanRendererConfig {
|
||||
Uint32 MaxFramesInFlight = 2;
|
||||
// Fallback CPU pipeline depth used when the MOBILEGL_MAGMA_FRAMESINFLIGHT env var is
|
||||
// unset/invalid. A deeper pipeline lets the CPU run further ahead of the GPU, hiding
|
||||
// per-frame GPU-completion latency. Whatever value is chosen (env or this fallback) is
|
||||
// only a request: VulkanRenderer::Initialize clamps it down to the surface's maxImageCount
|
||||
// (and never below 2), since not every driver allows that many swapchain images.
|
||||
Uint32 MaxFramesInFlight = 3;
|
||||
String AppName = "MobileGL-VulkanRenderer";
|
||||
Version Version = MG_Config::CoreVersion;
|
||||
MobileGL::Version Version = MG_Config::CoreVersion;
|
||||
Uint64 CacheVersion = MG_Config::CacheVersion;
|
||||
Uint32 SurfaceWidth = 1;
|
||||
Uint32 SurfaceHeight = 1;
|
||||
Bool DisablePipelineCache = false;
|
||||
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG
|
||||
Bool EnableValidationLayers = true;
|
||||
#else
|
||||
Bool EnableValidationLayers = false;
|
||||
#endif
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -16,4 +16,5 @@ target_link_libraries(
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
add_test(NAME BufferBench COMMAND BufferBench --benchmark_counters_tabular=true)
|
||||
add_test(NAME BufferBench COMMAND BufferBench --benchmark_counters_tabular=true)
|
||||
set_tests_properties(BufferBench PROPERTIES LABELS benchmark)
|
||||
@@ -38,6 +38,9 @@ target_link_libraries(
|
||||
)
|
||||
|
||||
add_test(NAME SanityBench COMMAND SanityBench --benchmark_counters_tabular=true)
|
||||
set_tests_properties(SanityBench PROPERTIES LABELS benchmark)
|
||||
|
||||
add_subdirectory(Program)
|
||||
add_subdirectory(Buffer)
|
||||
add_subdirectory(Buffer)
|
||||
add_subdirectory(Driver)
|
||||
add_subdirectory(Container)
|
||||
@@ -0,0 +1,20 @@
|
||||
cmake_minimum_required(VERSION 3.24)
|
||||
|
||||
add_executable(
|
||||
UnorderedMapBench
|
||||
UnorderedMapBench.cpp
|
||||
)
|
||||
|
||||
target_include_directories(UnorderedMapBench PRIVATE
|
||||
${MGL_ROOT}/include
|
||||
${MGL_ROOT}/MobileGL
|
||||
)
|
||||
|
||||
target_link_libraries(
|
||||
UnorderedMapBench PRIVATE
|
||||
benchmark::benchmark
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
add_test(NAME UnorderedMapBench COMMAND UnorderedMapBench --benchmark_counters_tabular=true)
|
||||
set_tests_properties(UnorderedMapBench PROPERTIES LABELS benchmark)
|
||||
@@ -0,0 +1,248 @@
|
||||
// MobileGL - MobileGL/MG_Benchmark/Container/UnorderedMapBench.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// The standing performance observatory for MobileGL::UnorderedMap.
|
||||
//
|
||||
// This benchmarks the ALIAS, never a concrete table, so whatever UnorderedMap
|
||||
// names today is what gets measured - swap the container in MG_Util/Types.h and
|
||||
// re-run this same binary to get a directly comparable set of numbers. That is
|
||||
// the point of it: the container sits on per-draw paths, so a change to it needs
|
||||
// evidence, and the evidence should be produced the same way every time.
|
||||
//
|
||||
// The workloads are the shapes the tree actually exercises, not generic hash-map
|
||||
// microbenchmarks. Four key shapes, because they stress a hash function very
|
||||
// differently:
|
||||
// * SEQUENTIAL dense small integers - GL object names from the index generator
|
||||
// (buffer/texture/framebuffer/sampler registries).
|
||||
// * POINTER real heap addresses - StateBackendObjectRegistry keys on
|
||||
// StateObject*. These are aligned, so their low bits are the
|
||||
// least random part of the key; a table that indexes on raw low
|
||||
// bits clusters badly here and one that mixes first does not.
|
||||
// Taken from the real allocator rather than a synthetic stride,
|
||||
// which would flatter whichever table mixes its bits.
|
||||
// * DIGEST already well-mixed 64-bit values - the XXH64 pipeline,
|
||||
// vertex-input-state and program memos.
|
||||
// * NAME short strings - uniform/attribute name to location maps.
|
||||
//
|
||||
// Sizes sweep from 8 upward because the per-draw memos are usually SMALL; a table
|
||||
// that only wins at 4096 entries has not won anything that matters here.
|
||||
//
|
||||
// Run: build-linux/MobileGL/MG_Benchmark/Container/UnorderedMapBench
|
||||
// or: ctest -R UnorderedMapBench (label: benchmark)
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include "MG_Util/Types.h"
|
||||
|
||||
using namespace MobileGL;
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr Int64 kMinSize = 8;
|
||||
constexpr Int64 kMaxSize = 4096;
|
||||
|
||||
// Keep the real allocations alive for the whole process: the POINTER shape is
|
||||
// only honest if the keys are addresses the allocator actually handed out, and
|
||||
// they have to stay unique (a freed address can be handed out twice).
|
||||
std::vector<std::unique_ptr<char[]>>& PointerKeyStorage() {
|
||||
static std::vector<std::unique_ptr<char[]>> storage;
|
||||
return storage;
|
||||
}
|
||||
|
||||
Vector<Uint64> SequentialKeys(SizeT n) {
|
||||
Vector<Uint64> keys;
|
||||
keys.reserve(n);
|
||||
for (SizeT i = 0; i < n; ++i) keys.push_back(static_cast<Uint64>(i) + 1);
|
||||
return keys;
|
||||
}
|
||||
|
||||
Vector<Uint64> PointerKeys(SizeT n) {
|
||||
auto& storage = PointerKeyStorage();
|
||||
Vector<Uint64> keys;
|
||||
keys.reserve(n);
|
||||
std::mt19937_64 rng(0xBEEF);
|
||||
std::vector<std::unique_ptr<char[]>> churn;
|
||||
for (SizeT i = 0; i < n; ++i) {
|
||||
// State objects are not all one size, and the allocator sees other
|
||||
// traffic between them - a single uniform stride is not what this
|
||||
// registry ever sees.
|
||||
const SizeT sz = 96 + (rng() % 192);
|
||||
auto p = std::make_unique<char[]>(sz);
|
||||
keys.push_back(reinterpret_cast<Uint64>(p.get()));
|
||||
storage.push_back(std::move(p));
|
||||
if ((rng() & 3) == 0) churn.push_back(std::make_unique<char[]>(32 + (rng() % 128)));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
Vector<Uint64> DigestKeys(SizeT n) {
|
||||
Vector<Uint64> keys;
|
||||
keys.reserve(n);
|
||||
std::mt19937_64 rng(0xC0FFEE);
|
||||
for (SizeT i = 0; i < n; ++i) keys.push_back(rng());
|
||||
return keys;
|
||||
}
|
||||
|
||||
Vector<String> NameKeys(SizeT n) {
|
||||
static const char* kPrefixes[] = {"u_", "a_", "mc_", "iris_", "gl_", "v_"};
|
||||
Vector<String> keys;
|
||||
keys.reserve(n);
|
||||
for (SizeT i = 0; i < n; ++i) {
|
||||
keys.push_back(String(kPrefixes[i % 6]) + "Uniform" + std::to_string(i) + "_xyz");
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
// Key sets are built once per size and shared: generating them inside the timed
|
||||
// loop would measure the generator (and, for POINTER, the allocator) instead of
|
||||
// the table.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
const KeyVec& CachedKeys(SizeT n) {
|
||||
static UnorderedMap<SizeT, KeyVec> cache;
|
||||
auto it = cache.find(n);
|
||||
if (it != cache.end()) return it->second;
|
||||
return cache.emplace(n, Make(n)).first->second;
|
||||
}
|
||||
|
||||
template <typename Key>
|
||||
UnorderedMap<Key, Uint64> Populated(const Vector<Key>& keys) {
|
||||
UnorderedMap<Key, Uint64> map;
|
||||
for (SizeT i = 0; i < keys.size(); ++i) map[keys[i]] = i;
|
||||
return map;
|
||||
}
|
||||
|
||||
// ---- the workloads ----------------------------------------------------
|
||||
|
||||
// The dominant per-draw operation by a wide margin: a populated cache that is
|
||||
// read far more often than it is written.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void LookupHit(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
auto map = Populated(keys);
|
||||
for (auto _ : state) {
|
||||
for (const auto& k : keys) {
|
||||
auto it = map.find(k);
|
||||
benchmark::DoNotOptimize(it->second);
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// "Is this resource cached yet?" answered NO - the probe length on a miss is a
|
||||
// different cost from a hit, and resource caches ask this constantly.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void LookupMiss(benchmark::State& state) {
|
||||
const SizeT n = static_cast<SizeT>(state.range(0));
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(n);
|
||||
auto map = Populated(keys);
|
||||
const KeyVec absent = Make(n); // same shape, never inserted
|
||||
for (auto _ : state) {
|
||||
for (const auto& k : absent) {
|
||||
benchmark::DoNotOptimize(map.find(k) != map.end());
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(absent.size()));
|
||||
}
|
||||
|
||||
// Building a cache from empty, rehashes included.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void InsertGrow(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
UnorderedMap<typename KeyVec::value_type, Uint64> map;
|
||||
for (SizeT i = 0; i < keys.size(); ++i) map[keys[i]] = i;
|
||||
benchmark::DoNotOptimize(map.size());
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// Cache eviction and refill: erase half by key, put them back. This is the
|
||||
// aged-out-entry sweep the pipeline and vertex-input caches do.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void EraseChurn(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
state.PauseTiming();
|
||||
auto map = Populated(keys);
|
||||
state.ResumeTiming();
|
||||
for (SizeT i = 0; i < keys.size(); i += 2) benchmark::DoNotOptimize(map.erase(keys[i]));
|
||||
for (SizeT i = 0; i < keys.size(); i += 2) map[keys[i]] = i;
|
||||
benchmark::DoNotOptimize(map.size());
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// Mass eviction: erase-while-iterating across the whole table. This is the loop
|
||||
// shape that a container's erase()-return contract can get wrong, and the one
|
||||
// that fed garbage handles to vkDestroyPipeline when it was wrong before.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void EraseSweep(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
for (auto _ : state) {
|
||||
state.PauseTiming();
|
||||
auto map = Populated(keys);
|
||||
state.ResumeTiming();
|
||||
for (auto it = map.begin(); it != map.end();) it = map.erase(it);
|
||||
benchmark::DoNotOptimize(map.size());
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
// Whole-table walks: the per-frame sweeps that age entries out, and the
|
||||
// teardown loops that destroy every Vulkan object a cache owns.
|
||||
template <typename KeyVec, KeyVec (*Make)(SizeT)>
|
||||
void Iterate(benchmark::State& state) {
|
||||
const auto& keys = CachedKeys<KeyVec, Make>(static_cast<SizeT>(state.range(0)));
|
||||
auto map = Populated(keys);
|
||||
for (auto _ : state) {
|
||||
Uint64 acc = 0;
|
||||
for (const auto& entry : map) acc += entry.second;
|
||||
benchmark::DoNotOptimize(acc);
|
||||
}
|
||||
state.SetItemsProcessed(state.iterations() * static_cast<Int64>(keys.size()));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
#define MGL_MAP_BENCH(WORKLOAD, SHAPE, VEC, MAKER) \
|
||||
BENCHMARK_TEMPLATE(WORKLOAD, VEC, MAKER) \
|
||||
->Name(#WORKLOAD "/" #SHAPE) \
|
||||
->RangeMultiplier(8) \
|
||||
->Range(kMinSize, kMaxSize)
|
||||
|
||||
MGL_MAP_BENCH(LookupHit, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(LookupHit, pointer, Vector<Uint64>, PointerKeys);
|
||||
MGL_MAP_BENCH(LookupHit, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(LookupHit, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(LookupMiss, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(LookupMiss, pointer, Vector<Uint64>, PointerKeys);
|
||||
MGL_MAP_BENCH(LookupMiss, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(LookupMiss, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(InsertGrow, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(InsertGrow, pointer, Vector<Uint64>, PointerKeys);
|
||||
MGL_MAP_BENCH(InsertGrow, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(InsertGrow, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(EraseChurn, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(EraseChurn, digest, Vector<Uint64>, DigestKeys);
|
||||
MGL_MAP_BENCH(EraseChurn, name, Vector<String>, NameKeys);
|
||||
|
||||
MGL_MAP_BENCH(EraseSweep, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(EraseSweep, digest, Vector<Uint64>, DigestKeys);
|
||||
|
||||
MGL_MAP_BENCH(Iterate, sequential, Vector<Uint64>, SequentialKeys);
|
||||
MGL_MAP_BENCH(Iterate, digest, Vector<Uint64>, DigestKeys);
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,15 @@
|
||||
cmake_minimum_required(VERSION 3.24)
|
||||
|
||||
# A real, headless EGL client, deliberately NOT linked against MobileGL: it
|
||||
# dlopens one EGL provider at runtime ($DRIVERBENCH_EGL_LIB - the system
|
||||
# libEGL.so.1 for the native driver, or a libMobileGL.so path for either
|
||||
# MobileGL backend), so the same binary measures all three stacks.
|
||||
if (NOT UNIX OR APPLE OR ANDROID)
|
||||
return()
|
||||
endif()
|
||||
|
||||
add_executable(DriverBench DriverBench.c)
|
||||
target_link_libraries(DriverBench PRIVATE dl)
|
||||
|
||||
add_test(NAME DriverBench COMMAND DriverBench draw_tiny)
|
||||
set_tests_properties(DriverBench PROPERTIES LABELS benchmark)
|
||||
@@ -0,0 +1,501 @@
|
||||
/* MobileGL - MobileGL/MG_Benchmark/Driver/DriverBench.c
|
||||
* Copyright (c) 2025-2026 MobileGL-Dev
|
||||
* Licensed under the GNU Lesser General Public License v3.0:
|
||||
* https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
* https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
* SPDX-License-Identifier: LGPL-3.0-only
|
||||
* End of Source File Header
|
||||
*
|
||||
* Headless, EGL-based driver benchmark shaped like Minecraft's GL usage.
|
||||
* Unlike the MobileGL_s microbenches next door this exercises a full GL
|
||||
* stack: it dlopens ONE EGL provider ($DRIVERBENCH_EGL_LIB - the system
|
||||
* libEGL.so.1 for the native driver, or a libMobileGL.so path for either
|
||||
* MobileGL backend selected with MOBILEGL_BACKEND_TYPE), creates a desktop-GL
|
||||
* context on a small pbuffer, renders into its own FBO and paces frames with
|
||||
* glFinish. No window system is required: the default display is tried first
|
||||
* so a desktop run reaches the real driver, and a headless box (CI, a build
|
||||
* server) falls back to EGL_MESA_platform_surfaceless - see
|
||||
* run_driver_bench.sh.
|
||||
*
|
||||
* Every case models one hot pattern from captured Minecraft traces:
|
||||
* draw_tiny back-to-back glDrawElements, shared state (chunk batch)
|
||||
* draw_uniform per-draw vec3 offset uniform + draw (chunk sections)
|
||||
* draw_multi_vao per-draw VAO/VBO switch + draw (per-section buffers)
|
||||
* tex_pingpong per-draw texture bind churn on one unit
|
||||
* program_pingpong alternate two programs + mat4 upload (chunk<->entity)
|
||||
* chunk_upload glBufferData(NULL) orphan + glBufferSubData + draw
|
||||
* atlas_sprite N 16x16 glTexSubImage2D into a 1024x512 atlas + draw
|
||||
* lightmap full 16x16 lightmap respecify per frame + draw
|
||||
* scene_mix composite frame built from the knobs below
|
||||
*
|
||||
* Output: one CSV line per case:
|
||||
* case,frames,ops_per_frame,median_frame_ms,ns_per_op,fps
|
||||
*/
|
||||
#include <dlfcn.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
|
||||
/* ---- EGL constants ---- */
|
||||
typedef void* EGLDisplay;
|
||||
typedef void* EGLConfig;
|
||||
typedef void* EGLContext;
|
||||
typedef void* EGLSurface;
|
||||
typedef int EGLint;
|
||||
typedef unsigned int EGLBoolean;
|
||||
typedef unsigned int EGLenum;
|
||||
#define EGL_DEFAULT_DISPLAY ((void*)0)
|
||||
#define EGL_NO_CONTEXT ((EGLContext)0)
|
||||
#define EGL_NO_SURFACE ((EGLSurface)0)
|
||||
#define EGL_FALSE 0
|
||||
#define EGL_SURFACE_TYPE 0x3033
|
||||
#define EGL_PBUFFER_BIT 0x0001
|
||||
#define EGL_RENDERABLE_TYPE 0x3040
|
||||
#define EGL_OPENGL_BIT 0x0008
|
||||
#define EGL_RED_SIZE 0x3024
|
||||
#define EGL_GREEN_SIZE 0x3023
|
||||
#define EGL_BLUE_SIZE 0x3022
|
||||
#define EGL_DEPTH_SIZE 0x3025
|
||||
#define EGL_WIDTH 0x3057
|
||||
#define EGL_HEIGHT 0x3056
|
||||
#define EGL_NONE 0x3038
|
||||
#define EGL_OPENGL_API 0x30A2
|
||||
#define EGL_OPENGL_ES_API 0x30A0
|
||||
#define EGL_OPENGL_ES3_BIT 0x0040
|
||||
#define EGL_CONTEXT_CLIENT_VERSION 0x3098
|
||||
#define EGL_CONTEXT_MAJOR_VERSION 0x3098
|
||||
#define EGL_CONTEXT_MINOR_VERSION 0x30FB
|
||||
#define EGL_CONTEXT_OPENGL_PROFILE_MASK 0x30FD
|
||||
#define EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT 0x00000001
|
||||
#define EGL_PLATFORM_SURFACELESS_MESA 0x31DD
|
||||
|
||||
/* ---- GL constants ---- */
|
||||
#define GL_COLOR_BUFFER_BIT 0x00004000
|
||||
#define GL_DEPTH_BUFFER_BIT 0x00000100
|
||||
#define GL_TRIANGLES 0x0004
|
||||
#define GL_UNSIGNED_INT 0x1405
|
||||
#define GL_SHORT 0x1402
|
||||
#define GL_FLOAT 0x1406
|
||||
#define GL_UNSIGNED_BYTE 0x1401
|
||||
#define GL_ARRAY_BUFFER 0x8892
|
||||
#define GL_ELEMENT_ARRAY_BUFFER 0x8893
|
||||
#define GL_STATIC_DRAW 0x88E4
|
||||
#define GL_TEXTURE_2D 0x0DE1
|
||||
#define GL_TEXTURE0 0x84C0
|
||||
#define GL_RGBA 0x1908
|
||||
#define GL_RGBA8 0x8058
|
||||
#define GL_DEPTH_COMPONENT24 0x81A6
|
||||
#define GL_TEXTURE_MIN_FILTER 0x2801
|
||||
#define GL_TEXTURE_MAG_FILTER 0x2800
|
||||
#define GL_NEAREST 0x2600
|
||||
#define GL_NEAREST_MIPMAP_LINEAR 0x2702
|
||||
#define GL_DEPTH_TEST 0x0B71
|
||||
#define GL_BLEND 0x0BE2
|
||||
#define GL_SRC_ALPHA 0x0302
|
||||
#define GL_ONE_MINUS_SRC_ALPHA 0x0303
|
||||
#define GL_ONE 1
|
||||
#define GL_ZERO 0
|
||||
#define GL_VERTEX_SHADER 0x8B31
|
||||
#define GL_FRAGMENT_SHADER 0x8B30
|
||||
#define GL_COMPILE_STATUS 0x8B81
|
||||
#define GL_LINK_STATUS 0x8B82
|
||||
#define GL_VERSION 0x1F02
|
||||
#define GL_RENDERER 0x1F01
|
||||
#define GL_NO_ERROR 0
|
||||
#define GL_FRAMEBUFFER 0x8D40
|
||||
#define GL_RENDERBUFFER 0x8D41
|
||||
#define GL_COLOR_ATTACHMENT0 0x8CE0
|
||||
#define GL_DEPTH_ATTACHMENT 0x8D00
|
||||
#define GL_FRAMEBUFFER_COMPLETE 0x8CD5
|
||||
#define GL_SYNC_GPU_COMMANDS_COMPLETE 0x9117
|
||||
#define GL_SYNC_FLUSH_COMMANDS_BIT 0x00000001
|
||||
#define GL_UNIFORM_BUFFER 0x8A11
|
||||
#define GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT 0x8A34
|
||||
#define GL_DYNAMIC_DRAW 0x88E8
|
||||
#define GL_STREAM_DRAW 0x88E0
|
||||
#define GL_UNPACK_ALIGNMENT 0x0CF5
|
||||
#define GL_UNPACK_ROW_LENGTH 0x0CF2
|
||||
#define GL_UNPACK_SKIP_ROWS 0x0CF3
|
||||
#define GL_UNPACK_SKIP_PIXELS 0x0CF4
|
||||
#define GL_TEXTURE_WRAP_S 0x2802
|
||||
#define GL_TEXTURE_WRAP_T 0x2803
|
||||
#define GL_CLAMP_TO_EDGE 0x812F
|
||||
#define GL_REPEAT 0x2901
|
||||
|
||||
typedef unsigned int GLuint;
|
||||
typedef int GLint;
|
||||
typedef int GLsizei;
|
||||
typedef unsigned int GLenum;
|
||||
typedef char GLchar;
|
||||
typedef unsigned char GLboolean;
|
||||
typedef long GLsizeiptr;
|
||||
typedef long GLintptr;
|
||||
|
||||
/* ---- resolved entry points ---- */
|
||||
static void* (*g_eglGetProcAddress)(const char*);
|
||||
static void* g_provider;
|
||||
|
||||
#define GLF(ret, name, args) static ret(*name) args;
|
||||
GLF(void, glClear, (unsigned))
|
||||
GLF(void, glClearColor, (float, float, float, float))
|
||||
GLF(void, glEnable, (GLenum))
|
||||
GLF(void, glDisable, (GLenum))
|
||||
GLF(void, glBlendFuncSeparate, (GLenum, GLenum, GLenum, GLenum))
|
||||
GLF(void, glDrawBuffers, (GLsizei, const GLenum*))
|
||||
GLF(void, glViewport, (GLint, GLint, GLsizei, GLsizei))
|
||||
GLF(const unsigned char*, glGetString, (GLenum))
|
||||
GLF(GLenum, glGetError, (void))
|
||||
GLF(void, glFinish, (void))
|
||||
GLF(void, glFlush, (void))
|
||||
GLF(void, glGenBuffers, (GLsizei, GLuint*))
|
||||
GLF(void, glBindBuffer, (GLenum, GLuint))
|
||||
GLF(void, glBufferData, (GLenum, GLsizeiptr, const void*, GLenum))
|
||||
GLF(void, glBufferSubData, (GLenum, GLintptr, GLsizeiptr, const void*))
|
||||
GLF(void, glGenVertexArrays, (GLsizei, GLuint*))
|
||||
GLF(void, glBindVertexArray, (GLuint))
|
||||
GLF(void, glEnableVertexAttribArray, (GLuint))
|
||||
GLF(void, glVertexAttribPointer, (GLuint, GLint, GLenum, GLboolean, GLsizei, const void*))
|
||||
GLF(void, glGenTextures, (GLsizei, GLuint*))
|
||||
GLF(void, glBindTexture, (GLenum, GLuint))
|
||||
GLF(void, glActiveTexture, (GLenum))
|
||||
GLF(void, glTexImage2D, (GLenum, GLint, GLint, GLsizei, GLsizei, GLint, GLenum, GLenum, const void*))
|
||||
GLF(void, glTexSubImage2D, (GLenum, GLint, GLint, GLint, GLsizei, GLsizei, GLenum, GLenum, const void*))
|
||||
GLF(void, glTexParameteri, (GLenum, GLenum, GLint))
|
||||
GLF(void, glPixelStorei, (GLenum, GLint))
|
||||
GLF(void, glGetIntegerv, (GLenum, GLint*))
|
||||
GLF(void, glGenerateMipmap, (GLenum))
|
||||
GLF(GLuint, glCreateShader, (GLenum))
|
||||
GLF(void, glShaderSource, (GLuint, GLsizei, const GLchar* const*, const GLint*))
|
||||
GLF(void, glCompileShader, (GLuint))
|
||||
GLF(void, glGetShaderiv, (GLuint, GLenum, GLint*))
|
||||
GLF(void, glGetShaderInfoLog, (GLuint, GLsizei, GLsizei*, GLchar*))
|
||||
GLF(GLuint, glCreateProgram, (void))
|
||||
GLF(void, glAttachShader, (GLuint, GLuint))
|
||||
GLF(void, glLinkProgram, (GLuint))
|
||||
GLF(void, glGetProgramiv, (GLuint, GLenum, GLint*))
|
||||
GLF(void, glUseProgram, (GLuint))
|
||||
GLF(GLint, glGetUniformLocation, (GLuint, const GLchar*))
|
||||
GLF(void, glUniform1i, (GLint, GLint))
|
||||
GLF(void, glUniform3f, (GLint, float, float, float))
|
||||
GLF(void, glUniformMatrix4fv, (GLint, GLsizei, GLboolean, const float*))
|
||||
GLF(void, glDrawElements, (GLenum, GLsizei, GLenum, const void*))
|
||||
GLF(void, glBindAttribLocation, (GLuint, GLuint, const GLchar*))
|
||||
GLF(void, glUniform3fv, (GLint, GLsizei, const float*))
|
||||
GLF(void, glDrawArrays, (GLenum, GLint, GLsizei))
|
||||
GLF(void, glDrawElementsBaseVertex, (GLenum, GLsizei, GLenum, const void*, GLint))
|
||||
GLF(void, glMultiDrawElementsBaseVertex,
|
||||
(GLenum, const GLsizei*, GLenum, const void* const*, GLsizei, const GLint*))
|
||||
GLF(void, glBindBufferRange, (GLenum, GLuint, GLuint, GLintptr, GLsizeiptr))
|
||||
GLF(void, glBindBufferBase, (GLenum, GLuint, GLuint))
|
||||
GLF(GLuint, glGetUniformBlockIndex, (GLuint, const GLchar*))
|
||||
GLF(void, glUniformBlockBinding, (GLuint, GLuint, GLuint))
|
||||
GLF(void, glGenSamplers, (GLsizei, GLuint*))
|
||||
GLF(void, glBindSampler, (GLuint, GLuint))
|
||||
GLF(void, glSamplerParameteri, (GLuint, GLenum, GLint))
|
||||
GLF(void, glGenFramebuffers, (GLsizei, GLuint*))
|
||||
GLF(void, glBindFramebuffer, (GLenum, GLuint))
|
||||
GLF(void, glGenRenderbuffers, (GLsizei, GLuint*))
|
||||
GLF(void, glBindRenderbuffer, (GLenum, GLuint))
|
||||
GLF(void, glRenderbufferStorage, (GLenum, GLenum, GLsizei, GLsizei))
|
||||
GLF(void, glFramebufferRenderbuffer, (GLenum, GLenum, GLenum, GLuint))
|
||||
GLF(GLenum, glCheckFramebufferStatus, (GLenum))
|
||||
GLF(void*, glFenceSync, (GLenum, unsigned))
|
||||
GLF(GLenum, glClientWaitSync, (void*, unsigned, unsigned long long))
|
||||
GLF(void, glDeleteSync, (void*))
|
||||
|
||||
static uint64_t now_ns(void) {
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return (uint64_t)ts.tv_sec * 1000000000ull + (uint64_t)ts.tv_nsec;
|
||||
}
|
||||
|
||||
static int cmp_u64(const void* a, const void* b) {
|
||||
uint64_t x = *(const uint64_t*)a, y = *(const uint64_t*)b;
|
||||
return x < y ? -1 : x > y;
|
||||
}
|
||||
|
||||
|
||||
/* Scene, cases and the case table live next door so the Android plugin's
|
||||
* in-process benchmark runs byte-identical bodies. */
|
||||
static void bench_gl_failed(const char* what, const char* detail) {
|
||||
fprintf(stderr, "FAIL: %s %s\n", what, detail ? detail : "");
|
||||
exit(1);
|
||||
}
|
||||
|
||||
/* GLES has glDrawElementsBaseVertex (3.2 core) but no multi-draw form of it, so
|
||||
* against a native mobile driver the multi-draw case issues the same sub-draws
|
||||
* one at a time - which is what the extension folds up, and what an application
|
||||
* without it would have to write. Desktop GL and MobileGL take the real call. */
|
||||
static void bench_multi_draw_elements_base_vertex(GLenum mode, const GLsizei* counts, GLenum type,
|
||||
const void* const* offsets, GLsizei drawCount,
|
||||
const GLint* baseVertices) {
|
||||
if (glMultiDrawElementsBaseVertex) {
|
||||
glMultiDrawElementsBaseVertex(mode, counts, type, offsets, drawCount, baseVertices);
|
||||
return;
|
||||
}
|
||||
for (GLsizei i = 0; i < drawCount; ++i) {
|
||||
glDrawElementsBaseVertex(mode, counts[i], type, offsets[i], baseVertices[i]);
|
||||
}
|
||||
}
|
||||
|
||||
#include "DriverBenchCases.inc"
|
||||
|
||||
/* ---- bench driver: fence-paced frames on the offscreen FBO ----------------
|
||||
* Frames are closed with a real fence wait, not glFinish: MobileGL implements
|
||||
* glFinish and glFlush as no-ops (MG_Impl/GLImpl/Exporting/Definitions.cpp),
|
||||
* so a glFinish-paced loop would time only the CPU-side submit on a MobileGL
|
||||
* backend while timing submit-plus-GPU on the native driver - the two numbers
|
||||
* would not describe the same work. A sync object is honoured by every stack
|
||||
* measured here.
|
||||
*/
|
||||
typedef void (*case_fn)(int frame, long a, long b);
|
||||
static int g_warmup = 30, g_frames = 120;
|
||||
|
||||
static void end_frame_wait(void) {
|
||||
if (glFenceSync && glClientWaitSync && glDeleteSync) {
|
||||
void* sync = glFenceSync(GL_SYNC_GPU_COMMANDS_COMPLETE, 0);
|
||||
if (sync) {
|
||||
glClientWaitSync(sync, GL_SYNC_FLUSH_COMMANDS_BIT, 1000000000ull);
|
||||
glDeleteSync(sync);
|
||||
return;
|
||||
}
|
||||
}
|
||||
glFinish();
|
||||
}
|
||||
|
||||
static void run_case(const char* name, case_fn body, long a, long b, long opsPerFrame) {
|
||||
static uint64_t samples[4096];
|
||||
if (g_frames > 4096) g_frames = 4096;
|
||||
end_frame_wait();
|
||||
for (int i = 0; i < g_warmup; ++i) {
|
||||
glClear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT);
|
||||
body(i, a, b);
|
||||
end_frame_wait();
|
||||
}
|
||||
for (int i = 0; i < g_frames; ++i) {
|
||||
uint64_t t0 = now_ns();
|
||||
glClear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT);
|
||||
body(i, a, b);
|
||||
end_frame_wait();
|
||||
samples[i] = now_ns() - t0;
|
||||
}
|
||||
qsort(samples, g_frames, sizeof(uint64_t), cmp_u64);
|
||||
uint64_t med = samples[g_frames / 2];
|
||||
double frameMs = med / 1e6;
|
||||
double nsPerOp = opsPerFrame > 0 ? (double)med / (double)opsPerFrame : 0.0;
|
||||
printf("%s,%d,%ld,%.3f,%.1f,%.1f\n", name, g_frames, opsPerFrame, frameMs, nsPerOp,
|
||||
1e9 / (double)med);
|
||||
fflush(stdout);
|
||||
if (glGetError() != GL_NO_ERROR) fprintf(stderr, "WARN: GL error after %s\n", name);
|
||||
}
|
||||
|
||||
/* A display that needs no window system. eglGetPlatformDisplay is EGL 1.5
|
||||
* core and eglGetPlatformDisplayEXT is the EGL_EXT_platform_base spelling
|
||||
* older loaders ship; both are client entry points, so they resolve before
|
||||
* any display exists. Only the attribute-list types differ between the two
|
||||
* and this passes none, so one cast covers both. */
|
||||
static EGLDisplay surfaceless_display(void) {
|
||||
void* fn = dlsym(g_provider, "eglGetPlatformDisplay");
|
||||
if (!fn) fn = g_eglGetProcAddress("eglGetPlatformDisplay");
|
||||
if (!fn) fn = dlsym(g_provider, "eglGetPlatformDisplayEXT");
|
||||
if (!fn) fn = g_eglGetProcAddress("eglGetPlatformDisplayEXT");
|
||||
if (!fn) return NULL;
|
||||
return ((EGLDisplay(*)(EGLenum, void*, const void*))fn)(EGL_PLATFORM_SURFACELESS_MESA,
|
||||
EGL_DEFAULT_DISPLAY, NULL);
|
||||
}
|
||||
|
||||
/* ---- EGL bootstrap: one provider library, pbuffer, desktop-GL context ---- */
|
||||
static int boot_egl(void) {
|
||||
const char* libpath = getenv("DRIVERBENCH_EGL_LIB");
|
||||
if (!libpath) libpath = "libEGL.so.1";
|
||||
g_provider = dlopen(libpath, RTLD_LAZY | RTLD_LOCAL);
|
||||
if (!g_provider) {
|
||||
fprintf(stderr, "FAIL: dlopen %s: %s\n", libpath, dlerror());
|
||||
return 1;
|
||||
}
|
||||
#define ESYM(name) \
|
||||
void* p_##name = dlsym(g_provider, #name); \
|
||||
if (!p_##name) { fprintf(stderr, "FAIL: dlsym %s\n", #name); return 1; }
|
||||
ESYM(eglGetDisplay)
|
||||
ESYM(eglInitialize)
|
||||
ESYM(eglChooseConfig)
|
||||
ESYM(eglBindAPI)
|
||||
ESYM(eglCreateContext)
|
||||
ESYM(eglCreatePbufferSurface)
|
||||
ESYM(eglMakeCurrent)
|
||||
ESYM(eglGetProcAddress)
|
||||
ESYM(eglGetError)
|
||||
g_eglGetProcAddress = (void* (*)(const char*))p_eglGetProcAddress;
|
||||
|
||||
EGLint (*getError)(void) = (EGLint(*)(void))p_eglGetError;
|
||||
EGLBoolean (*initialize)(EGLDisplay, EGLint*, EGLint*) =
|
||||
(EGLBoolean(*)(EGLDisplay, EGLint*, EGLint*))p_eglInitialize;
|
||||
|
||||
/* The default display first: it is the one a windowed app would get, and
|
||||
* on a desktop it is the one that reaches the real GPU - which is the
|
||||
* driver this bench exists to measure. It does need a window system,
|
||||
* though; Mesa's default platform is X11, so with no $DISPLAY (CI, a
|
||||
* build server, ssh without forwarding) eglInitialize fails. Fall back to
|
||||
* EGL_MESA_platform_surfaceless rather than give up: every case draws into
|
||||
* the FBO built by build_resources(), so no window is needed for any of
|
||||
* the work being timed. */
|
||||
EGLint maj = 0, min = 0;
|
||||
const char* how = "default display";
|
||||
EGLDisplay dpy = ((EGLDisplay(*)(void*))p_eglGetDisplay)(EGL_DEFAULT_DISPLAY);
|
||||
if (!dpy || !initialize(dpy, &maj, &min)) {
|
||||
dpy = surfaceless_display();
|
||||
how = "surfaceless display";
|
||||
if (!dpy || !initialize(dpy, &maj, &min)) {
|
||||
fprintf(stderr, "FAIL: eglInitialize (0x%x)\n", getError());
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
fprintf(stderr, "EGL %d.%d via %s (%s)\n", maj, min, libpath, how);
|
||||
|
||||
// Desktop GL first (that is what MobileGL exposes and what the cases are
|
||||
// written against), GLES 3 second so the same binary can measure a device's
|
||||
// native driver as the baseline. The .inc picks ESSL shader sources when the
|
||||
// context turns out to be ES.
|
||||
EGLBoolean (*chooseConfig)(EGLDisplay, const EGLint*, EGLConfig*, EGLint, EGLint*) =
|
||||
(EGLBoolean(*)(EGLDisplay, const EGLint*, EGLConfig*, EGLint, EGLint*))p_eglChooseConfig;
|
||||
EGLContext (*createContext)(EGLDisplay, EGLConfig, EGLContext, const EGLint*) =
|
||||
(EGLContext(*)(EGLDisplay, EGLConfig, EGLContext, const EGLint*))p_eglCreateContext;
|
||||
EGLBoolean (*bindApi)(EGLenum) = (EGLBoolean(*)(EGLenum))p_eglBindAPI;
|
||||
|
||||
EGLConfig cfg = NULL;
|
||||
EGLint ncfg = 0;
|
||||
EGLContext ctx = EGL_NO_CONTEXT;
|
||||
|
||||
if (bindApi(EGL_OPENGL_API)) {
|
||||
const EGLint cfgAttribs[] = {EGL_SURFACE_TYPE, EGL_PBUFFER_BIT, EGL_RED_SIZE, 8,
|
||||
EGL_DEPTH_SIZE, 24, EGL_RENDERABLE_TYPE, EGL_OPENGL_BIT, EGL_NONE};
|
||||
if (chooseConfig(dpy, cfgAttribs, &cfg, 1, &ncfg) && ncfg >= 1) {
|
||||
const EGLint ctxAttribs[] = {EGL_CONTEXT_MAJOR_VERSION, 3, EGL_CONTEXT_MINOR_VERSION, 2,
|
||||
EGL_CONTEXT_OPENGL_PROFILE_MASK,
|
||||
EGL_CONTEXT_OPENGL_CORE_PROFILE_BIT, EGL_NONE};
|
||||
ctx = createContext(dpy, cfg, EGL_NO_CONTEXT, ctxAttribs);
|
||||
if (ctx == EGL_NO_CONTEXT) ctx = createContext(dpy, cfg, EGL_NO_CONTEXT, NULL);
|
||||
}
|
||||
}
|
||||
if (ctx == EGL_NO_CONTEXT) {
|
||||
if (!bindApi(EGL_OPENGL_ES_API)) {
|
||||
fprintf(stderr, "FAIL: neither OpenGL nor OpenGL ES is bindable on this provider\n");
|
||||
return 1;
|
||||
}
|
||||
const EGLint esCfgAttribs[] = {EGL_SURFACE_TYPE, EGL_PBUFFER_BIT, EGL_RED_SIZE, 8,
|
||||
EGL_GREEN_SIZE, 8, EGL_BLUE_SIZE, 8, EGL_DEPTH_SIZE, 24,
|
||||
EGL_RENDERABLE_TYPE, EGL_OPENGL_ES3_BIT, EGL_NONE};
|
||||
ncfg = 0;
|
||||
if (!chooseConfig(dpy, esCfgAttribs, &cfg, 1, &ncfg) || ncfg < 1) {
|
||||
// EGL_SURFACE_TYPE 0 matches any config: a stack that offers no
|
||||
// pbuffer at all is still usable through the surfaceless context
|
||||
// path below.
|
||||
const EGLint relaxed[] = {EGL_SURFACE_TYPE, 0, EGL_RED_SIZE, 8, EGL_NONE};
|
||||
if (!chooseConfig(dpy, relaxed, &cfg, 1, &ncfg) || ncfg < 1) {
|
||||
fprintf(stderr, "FAIL: eglChooseConfig\n");
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
const EGLint esCtxAttribs[] = {EGL_CONTEXT_CLIENT_VERSION, 3, EGL_NONE};
|
||||
ctx = createContext(dpy, cfg, EGL_NO_CONTEXT, esCtxAttribs);
|
||||
}
|
||||
if (ctx == EGL_NO_CONTEXT) {
|
||||
fprintf(stderr, "FAIL: eglCreateContext (0x%x)\n", getError());
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* The pbuffer only exists to have something to make current - nothing is
|
||||
* ever drawn to it. Where there is no pbuffer config, EGL_NO_SURFACE is
|
||||
* exactly what EGL_KHR_surfaceless_context takes, so the same call covers
|
||||
* both. */
|
||||
const EGLint pbAttribs[] = {EGL_WIDTH, 64, EGL_HEIGHT, 64, EGL_NONE};
|
||||
EGLSurface surf = ((EGLSurface(*)(EGLDisplay, EGLConfig, const EGLint*))p_eglCreatePbufferSurface)(
|
||||
dpy, cfg, pbAttribs);
|
||||
if (surf == EGL_NO_SURFACE)
|
||||
fprintf(stderr, "no pbuffer (0x%x), using a surfaceless context\n", getError());
|
||||
if (!((EGLBoolean(*)(EGLDisplay, EGLSurface, EGLSurface, EGLContext))p_eglMakeCurrent)(dpy, surf,
|
||||
surf, ctx)) {
|
||||
fprintf(stderr, "FAIL: eglMakeCurrent (0x%x)\n", getError());
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Core GL entry points: eglGetProcAddress first (EGL 1.5 serves core
|
||||
* functions), provider dlsym as fallback (both glvnd and MobileGL export
|
||||
* the gl* symbols directly). */
|
||||
#define RESOLVE(name) \
|
||||
do { \
|
||||
*(void**)&name = g_eglGetProcAddress(#name); \
|
||||
if (!name) *(void**)&name = dlsym(g_provider, #name); \
|
||||
if (!name) { fprintf(stderr, "FAIL: resolve %s\n", #name); return 1; } \
|
||||
} while (0)
|
||||
RESOLVE(glClear); RESOLVE(glClearColor); RESOLVE(glEnable); RESOLVE(glViewport);
|
||||
RESOLVE(glDisable); RESOLVE(glBlendFuncSeparate); RESOLVE(glDrawBuffers);
|
||||
RESOLVE(glGetString); RESOLVE(glGetError); RESOLVE(glFinish); RESOLVE(glFlush);
|
||||
RESOLVE(glGenBuffers); RESOLVE(glBindBuffer); RESOLVE(glBufferData); RESOLVE(glBufferSubData);
|
||||
RESOLVE(glGenVertexArrays); RESOLVE(glBindVertexArray); RESOLVE(glEnableVertexAttribArray);
|
||||
RESOLVE(glVertexAttribPointer); RESOLVE(glGenTextures); RESOLVE(glBindTexture);
|
||||
RESOLVE(glActiveTexture); RESOLVE(glTexImage2D); RESOLVE(glTexSubImage2D);
|
||||
RESOLVE(glTexParameteri); RESOLVE(glGenerateMipmap); RESOLVE(glCreateShader);
|
||||
RESOLVE(glPixelStorei); RESOLVE(glGetIntegerv);
|
||||
RESOLVE(glShaderSource); RESOLVE(glCompileShader); RESOLVE(glGetShaderiv);
|
||||
RESOLVE(glGetShaderInfoLog); RESOLVE(glCreateProgram); RESOLVE(glAttachShader);
|
||||
RESOLVE(glLinkProgram); RESOLVE(glGetProgramiv); RESOLVE(glUseProgram);
|
||||
RESOLVE(glGetUniformLocation); RESOLVE(glUniform1i); RESOLVE(glUniform3f);
|
||||
RESOLVE(glUniformMatrix4fv); RESOLVE(glDrawElements); RESOLVE(glBindAttribLocation);
|
||||
RESOLVE(glUniform3fv); RESOLVE(glDrawArrays); RESOLVE(glDrawElementsBaseVertex);
|
||||
RESOLVE(glBindBufferRange); RESOLVE(glBindBufferBase);
|
||||
RESOLVE(glGetUniformBlockIndex); RESOLVE(glUniformBlockBinding);
|
||||
RESOLVE(glGenSamplers); RESOLVE(glBindSampler); RESOLVE(glSamplerParameteri);
|
||||
RESOLVE(glGenFramebuffers); RESOLVE(glBindFramebuffer); RESOLVE(glGenRenderbuffers);
|
||||
RESOLVE(glBindRenderbuffer); RESOLVE(glRenderbufferStorage); RESOLVE(glFramebufferRenderbuffer);
|
||||
RESOLVE(glCheckFramebufferStatus);
|
||||
// Optional: end_frame_wait() falls back to glFinish when a stack has no
|
||||
// sync objects, so resolve without failing the run.
|
||||
*(void**)&glFenceSync = g_eglGetProcAddress("glFenceSync");
|
||||
if (!glFenceSync) *(void**)&glFenceSync = dlsym(g_provider, "glFenceSync");
|
||||
*(void**)&glClientWaitSync = g_eglGetProcAddress("glClientWaitSync");
|
||||
if (!glClientWaitSync) *(void**)&glClientWaitSync = dlsym(g_provider, "glClientWaitSync");
|
||||
*(void**)&glDeleteSync = g_eglGetProcAddress("glDeleteSync");
|
||||
if (!glDeleteSync) *(void**)&glDeleteSync = dlsym(g_provider, "glDeleteSync");
|
||||
// Desktop-only: GLES 3.2 has DrawElementsBaseVertex but no multi-draw form,
|
||||
// so bench_multi_draw_elements_base_vertex() emulates it when this is null.
|
||||
*(void**)&glMultiDrawElementsBaseVertex = g_eglGetProcAddress("glMultiDrawElementsBaseVertex");
|
||||
if (!glMultiDrawElementsBaseVertex)
|
||||
*(void**)&glMultiDrawElementsBaseVertex = dlsym(g_provider, "glMultiDrawElementsBaseVertex");
|
||||
|
||||
fprintf(stderr, "renderer: %s\n", glGetString(GL_RENDERER));
|
||||
fprintf(stderr, "version: %s\n", glGetString(GL_VERSION));
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
long draws = 2048;
|
||||
if (getenv("DRIVERBENCH_DRAWS")) draws = atol(getenv("DRIVERBENCH_DRAWS"));
|
||||
if (getenv("DRIVERBENCH_FRAMES")) g_frames = atoi(getenv("DRIVERBENCH_FRAMES"));
|
||||
if (getenv("DRIVERBENCH_SPRITES")) g_mixSprites = atol(getenv("DRIVERBENCH_SPRITES"));
|
||||
|
||||
if (boot_egl()) return 1;
|
||||
build_resources();
|
||||
|
||||
printf("case,frames,ops_per_frame,median_frame_ms,ns_per_op,fps\n");
|
||||
for (int i = 0; i < kBenchCaseCount; ++i) {
|
||||
const BenchCaseDesc* c = &kBenchCases[i];
|
||||
if (argc > 1) {
|
||||
int wanted = 0;
|
||||
for (int j = 1; j < argc; ++j)
|
||||
if (strcmp(argv[j], c->name) == 0) wanted = 1;
|
||||
if (!wanted) continue;
|
||||
}
|
||||
// The generic cases scale with DRIVERBENCH_DRAWS; the mc_* rates are
|
||||
// measured and must not move, or the numbers stop being comparable.
|
||||
long a = c->a, ops = c->opsPerFrame;
|
||||
if (strncmp(c->name, "mc_", 3) != 0 && a > 100) {
|
||||
a = draws * a / 2048;
|
||||
ops = c->opsPerFrame * draws / 2048;
|
||||
}
|
||||
run_case(c->name, c->fn, a, c->b, ops);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,640 @@
|
||||
/* MobileGL - MobileGL/MG_Benchmark/Driver/DriverBenchCases.inc
|
||||
* Copyright (c) 2025-2026 MobileGL-Dev
|
||||
* Licensed under the GNU Lesser General Public License v3.0:
|
||||
* https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
* https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
* SPDX-License-Identifier: LGPL-3.0-only
|
||||
* End of Source File Header
|
||||
*
|
||||
* The benchmark scene and its cases, with no harness and no GL loader: the
|
||||
* includer supplies both. DriverBench.c drives it through function pointers
|
||||
* resolved from one EGL provider; MG_Util/SelfTest/DriverBenchJni.cpp drives
|
||||
* it through MobileGL's own frontend entry points inside the Android plugin.
|
||||
* Sharing the bodies is the point - a number from the phone and a number from
|
||||
* the desktop have to describe the same work.
|
||||
*
|
||||
* The includer must have declared, before including this file: the GL types
|
||||
* and enums used below, and callable gl* entry points with the standard
|
||||
* signatures. bench_gl_failed() is called (and must be defined) when shader
|
||||
* compilation or linking fails, so a caller can report the failure instead of
|
||||
* dying inside a benchmark.
|
||||
*/
|
||||
|
||||
/* ---- shared scene resources (Minecraft-shaped) ---- */
|
||||
#define MAX_SECTIONS 512
|
||||
static GLuint g_progChunk, g_progEntity;
|
||||
static GLint g_uOffsetChunk, g_uMvpChunk, g_uMvpEntity;
|
||||
static GLuint g_vao[MAX_SECTIONS], g_vbo[MAX_SECTIONS];
|
||||
static GLuint g_sharedIbo;
|
||||
static GLuint g_texAtlas, g_texLight, g_texEntity;
|
||||
static int g_quadsPerSection = 128; /* 128 quads = 512 verts, 768 indices */
|
||||
static unsigned char* g_scratch;
|
||||
/* Uniform ring + sampler for the 26.2-shaped cases (see the case block below). */
|
||||
static GLuint g_uboRing;
|
||||
static GLint g_uboAlign = 256;
|
||||
static size_t g_uboSlot = 256;
|
||||
static GLuint g_sampler;
|
||||
/* Two small offscreen targets for the 26.2-style render-pass churn case. */
|
||||
static GLuint g_passFbo[2];
|
||||
static GLuint g_passColor[2];
|
||||
static float g_mvp[16] = {0.002f, 0, 0, 0, 0, 0.002f, 0, 0, 0, 0, -0.001f, 0, -1.f, -1.f, 0.f, 1.f};
|
||||
|
||||
/* Minecraft chunk vertex: pos 3f, color 4ub, uv 2f, packed light 2s -> 32 B */
|
||||
#define VERT_STRIDE 32
|
||||
static void fill_section_vertices(unsigned char* dst, int quads, unsigned seed) {
|
||||
for (int q = 0; q < quads * 4; ++q) {
|
||||
float* f = (float*)(dst + q * VERT_STRIDE);
|
||||
unsigned r = seed = seed * 1664525u + 1013904223u;
|
||||
f[0] = (float)(q & 31) * 8.0f + (float)(r & 7);
|
||||
f[1] = (float)((q >> 5) & 31) * 8.0f;
|
||||
f[2] = (float)(q % 7) * 0.1f;
|
||||
dst[q * VERT_STRIDE + 12] = (unsigned char)r;
|
||||
dst[q * VERT_STRIDE + 13] = (unsigned char)(r >> 8);
|
||||
dst[q * VERT_STRIDE + 14] = (unsigned char)(r >> 16);
|
||||
dst[q * VERT_STRIDE + 15] = 255;
|
||||
f[4] = (float)(r & 1023) / 1024.0f;
|
||||
f[5] = (float)((r >> 10) & 511) / 512.0f;
|
||||
((short*)(dst + q * VERT_STRIDE + 24))[0] = 15 << 4;
|
||||
((short*)(dst + q * VERT_STRIDE + 24))[1] = 15 << 4;
|
||||
}
|
||||
}
|
||||
|
||||
static GLuint make_shader(GLenum kind, const char* src) {
|
||||
GLuint sh = glCreateShader(kind);
|
||||
glShaderSource(sh, 1, &src, NULL);
|
||||
glCompileShader(sh);
|
||||
GLint ok = 0;
|
||||
glGetShaderiv(sh, GL_COMPILE_STATUS, &ok);
|
||||
if (!ok) {
|
||||
char log[1024];
|
||||
glGetShaderInfoLog(sh, sizeof log, NULL, log);
|
||||
bench_gl_failed("shader compile", log);
|
||||
return 0;
|
||||
}
|
||||
return sh;
|
||||
}
|
||||
|
||||
static GLuint make_program(const char* vs_src, const char* fs_src) {
|
||||
GLuint prog = glCreateProgram();
|
||||
glAttachShader(prog, make_shader(GL_VERTEX_SHADER, vs_src));
|
||||
glAttachShader(prog, make_shader(GL_FRAGMENT_SHADER, fs_src));
|
||||
glBindAttribLocation(prog, 0, "aPos");
|
||||
glBindAttribLocation(prog, 1, "aColor");
|
||||
glBindAttribLocation(prog, 2, "aUv");
|
||||
glBindAttribLocation(prog, 3, "aLight");
|
||||
glLinkProgram(prog);
|
||||
GLint ok = 0;
|
||||
glGetProgramiv(prog, GL_LINK_STATUS, &ok);
|
||||
if (!ok) {
|
||||
bench_gl_failed("program link", "");
|
||||
return 0;
|
||||
}
|
||||
return prog;
|
||||
}
|
||||
|
||||
static const char* kChunkVs =
|
||||
"#version 150 core\n"
|
||||
"in vec3 aPos; in vec4 aColor; in vec2 aUv; in vec2 aLight;\n"
|
||||
"uniform mat4 uMvp; uniform vec3 uOffset;\n"
|
||||
"out vec4 vColor; out vec2 vUv; out vec2 vLight;\n"
|
||||
"void main(){ gl_Position = uMvp * vec4(aPos + uOffset, 1.0);\n"
|
||||
" vColor = aColor; vUv = aUv; vLight = aLight * (1.0/256.0); }\n";
|
||||
static const char* kChunkFs =
|
||||
"#version 150 core\n"
|
||||
"in vec4 vColor; in vec2 vUv; in vec2 vLight; out vec4 o;\n"
|
||||
"uniform sampler2D uAtlas; uniform sampler2D uLight;\n"
|
||||
"void main(){ o = texture(uAtlas, vUv) * vColor * texture(uLight, vLight); }\n";
|
||||
static const char* kEntityVs =
|
||||
"#version 150 core\n"
|
||||
"in vec3 aPos; in vec4 aColor; in vec2 aUv; in vec2 aLight;\n"
|
||||
"uniform mat4 uMvp; uniform mat4 uModel;\n"
|
||||
"out vec4 vColor; out vec2 vUv;\n"
|
||||
"void main(){ gl_Position = uMvp * uModel * vec4(aPos, 1.0); vColor = aColor; vUv = aUv; }\n";
|
||||
static const char* kEntityFs =
|
||||
"#version 150 core\n"
|
||||
"in vec4 vColor; in vec2 vUv; out vec4 o; uniform sampler2D uTex;\n"
|
||||
"void main(){ o = texture(uTex, vUv) * vColor; }\n";
|
||||
|
||||
// ESSL 3.20 twins of the four shaders above. The bodies are identical; only the
|
||||
// version line and the precision qualifiers differ, so the two paths compile the
|
||||
// same work. Needed because this bench also runs against a device's native GLES
|
||||
// driver as the baseline MobileGL is measured against, and that driver rejects
|
||||
// desktop GLSL - while MobileGL is fed desktop GLSL on purpose, since translating
|
||||
// it is the thing under test.
|
||||
static const char* kChunkVsEs =
|
||||
"#version 320 es\n"
|
||||
"precision highp float;\n"
|
||||
"in vec3 aPos; in vec4 aColor; in vec2 aUv; in vec2 aLight;\n"
|
||||
"uniform mat4 uMvp; uniform vec3 uOffset;\n"
|
||||
"out vec4 vColor; out vec2 vUv; out vec2 vLight;\n"
|
||||
"void main(){ gl_Position = uMvp * vec4(aPos + uOffset, 1.0);\n"
|
||||
" vColor = aColor; vUv = aUv; vLight = aLight * (1.0/256.0); }\n";
|
||||
static const char* kChunkFsEs =
|
||||
"#version 320 es\n"
|
||||
"precision mediump float;\n"
|
||||
"in vec4 vColor; in vec2 vUv; in vec2 vLight; out vec4 o;\n"
|
||||
"uniform sampler2D uAtlas; uniform sampler2D uLight;\n"
|
||||
"void main(){ o = texture(uAtlas, vUv) * vColor * texture(uLight, vLight); }\n";
|
||||
static const char* kEntityVsEs =
|
||||
"#version 320 es\n"
|
||||
"precision highp float;\n"
|
||||
"in vec3 aPos; in vec4 aColor; in vec2 aUv; in vec2 aLight;\n"
|
||||
"uniform mat4 uMvp; uniform mat4 uModel;\n"
|
||||
"out vec4 vColor; out vec2 vUv;\n"
|
||||
"void main(){ gl_Position = uMvp * uModel * vec4(aPos, 1.0); vColor = aColor; vUv = aUv; }\n";
|
||||
static const char* kEntityFsEs =
|
||||
"#version 320 es\n"
|
||||
"precision mediump float;\n"
|
||||
"in vec4 vColor; in vec2 vUv; out vec4 o; uniform sampler2D uTex;\n"
|
||||
"void main(){ o = texture(uTex, vUv) * vColor; }\n";
|
||||
|
||||
// True once build_resources() has seen a GL_VERSION beginning with "OpenGL ES".
|
||||
static int g_isGlesContext = 0;
|
||||
|
||||
static void setup_vao(GLuint vao, GLuint vbo, GLuint ibo) {
|
||||
glBindVertexArray(vao);
|
||||
glBindBuffer(GL_ARRAY_BUFFER, vbo);
|
||||
glEnableVertexAttribArray(0);
|
||||
glEnableVertexAttribArray(1);
|
||||
glEnableVertexAttribArray(2);
|
||||
glEnableVertexAttribArray(3);
|
||||
glVertexAttribPointer(0, 3, GL_FLOAT, 0, VERT_STRIDE, (void*)0);
|
||||
glVertexAttribPointer(1, 4, GL_UNSIGNED_BYTE, 1, VERT_STRIDE, (void*)12);
|
||||
glVertexAttribPointer(2, 2, GL_FLOAT, 0, VERT_STRIDE, (void*)16);
|
||||
glVertexAttribPointer(3, 2, GL_SHORT, 0, VERT_STRIDE, (void*)24);
|
||||
glBindBuffer(GL_ELEMENT_ARRAY_BUFFER, ibo);
|
||||
}
|
||||
|
||||
static GLuint g_mainFbo;
|
||||
|
||||
static void build_resources(void) {
|
||||
/* offscreen render target: 1280x720 RBO FBO, like CTS fbo surface mode */
|
||||
GLuint fbo, rboColor, rboDepth;
|
||||
glGenFramebuffers(1, &fbo);
|
||||
g_mainFbo = fbo;
|
||||
glBindFramebuffer(GL_FRAMEBUFFER, fbo);
|
||||
glGenRenderbuffers(1, &rboColor);
|
||||
glBindRenderbuffer(GL_RENDERBUFFER, rboColor);
|
||||
glRenderbufferStorage(GL_RENDERBUFFER, GL_RGBA8, 1280, 720);
|
||||
glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, rboColor);
|
||||
glGenRenderbuffers(1, &rboDepth);
|
||||
glBindRenderbuffer(GL_RENDERBUFFER, rboDepth);
|
||||
glRenderbufferStorage(GL_RENDERBUFFER, GL_DEPTH_COMPONENT24, 1280, 720);
|
||||
glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_DEPTH_ATTACHMENT, GL_RENDERBUFFER, rboDepth);
|
||||
if (glCheckFramebufferStatus(GL_FRAMEBUFFER) != GL_FRAMEBUFFER_COMPLETE) {
|
||||
bench_gl_failed("FBO incomplete", "");
|
||||
return;
|
||||
}
|
||||
|
||||
const char* versionString = (const char*)glGetString(GL_VERSION);
|
||||
g_isGlesContext = versionString != NULL && strncmp(versionString, "OpenGL ES", 9) == 0;
|
||||
g_progChunk = g_isGlesContext ? make_program(kChunkVsEs, kChunkFsEs) : make_program(kChunkVs, kChunkFs);
|
||||
g_progEntity = g_isGlesContext ? make_program(kEntityVsEs, kEntityFsEs) : make_program(kEntityVs, kEntityFs);
|
||||
glUseProgram(g_progChunk);
|
||||
g_uMvpChunk = glGetUniformLocation(g_progChunk, "uMvp");
|
||||
g_uOffsetChunk = glGetUniformLocation(g_progChunk, "uOffset");
|
||||
glUniform1i(glGetUniformLocation(g_progChunk, "uAtlas"), 0);
|
||||
glUniform1i(glGetUniformLocation(g_progChunk, "uLight"), 2);
|
||||
glUniformMatrix4fv(g_uMvpChunk, 1, 0, g_mvp);
|
||||
glUseProgram(g_progEntity);
|
||||
g_uMvpEntity = glGetUniformLocation(g_progEntity, "uMvp");
|
||||
glUniform1i(glGetUniformLocation(g_progEntity, "uTex"), 0);
|
||||
glUniformMatrix4fv(g_uMvpEntity, 1, 0, g_mvp);
|
||||
glUseProgram(g_progChunk);
|
||||
|
||||
/* shared quad index buffer, like Blaze3D's RenderSystem shared sequences */
|
||||
int maxQuads = 4096;
|
||||
unsigned* idx = (unsigned*)malloc((size_t)maxQuads * 6 * 4);
|
||||
for (int q = 0; q < maxQuads; ++q) {
|
||||
unsigned base = q * 4;
|
||||
unsigned* p = idx + q * 6;
|
||||
p[0] = base; p[1] = base + 1; p[2] = base + 2;
|
||||
p[3] = base + 2; p[4] = base + 3; p[5] = base;
|
||||
}
|
||||
glGenBuffers(1, &g_sharedIbo);
|
||||
glBindBuffer(GL_ELEMENT_ARRAY_BUFFER, g_sharedIbo);
|
||||
glBufferData(GL_ELEMENT_ARRAY_BUFFER, maxQuads * 6 * 4, idx, GL_STATIC_DRAW);
|
||||
free(idx);
|
||||
|
||||
g_scratch = (unsigned char*)malloc(4 * 1024 * 1024);
|
||||
memset(g_scratch, 0x5a, 4 * 1024 * 1024);
|
||||
|
||||
glGenVertexArrays(MAX_SECTIONS, g_vao);
|
||||
glGenBuffers(MAX_SECTIONS, g_vbo);
|
||||
int bytes = g_quadsPerSection * 4 * VERT_STRIDE;
|
||||
for (int i = 0; i < MAX_SECTIONS; ++i) {
|
||||
fill_section_vertices(g_scratch, g_quadsPerSection, i * 7919u + 1);
|
||||
glBindBuffer(GL_ARRAY_BUFFER, g_vbo[i]);
|
||||
glBufferData(GL_ARRAY_BUFFER, bytes, g_scratch, GL_STATIC_DRAW);
|
||||
setup_vao(g_vao[i], g_vbo[i], g_sharedIbo);
|
||||
}
|
||||
|
||||
glGenTextures(1, &g_texAtlas);
|
||||
glActiveTexture(GL_TEXTURE0);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
glTexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 1024, 512, 0, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST_MIPMAP_LINEAR);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
glGenerateMipmap(GL_TEXTURE_2D);
|
||||
|
||||
glGenTextures(1, &g_texLight);
|
||||
glActiveTexture(GL_TEXTURE0 + 2);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texLight);
|
||||
glTexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 16, 16, 0, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
|
||||
glGenTextures(1, &g_texEntity);
|
||||
glActiveTexture(GL_TEXTURE0);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texEntity);
|
||||
glTexImage2D(GL_TEXTURE_2D, 0, GL_RGBA8, 64, 64, 0, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
|
||||
// Uniform ring the 26.2-style case sub-ranges into, sized like a real
|
||||
// frame's worth of per-draw uniform slots.
|
||||
GLint align = 256;
|
||||
glGetIntegerv(GL_UNIFORM_BUFFER_OFFSET_ALIGNMENT, &align);
|
||||
g_uboAlign = align > 0 ? align : 256;
|
||||
g_uboSlot = (size_t)g_uboAlign;
|
||||
glGenBuffers(1, &g_uboRing);
|
||||
glBindBuffer(GL_UNIFORM_BUFFER, g_uboRing);
|
||||
glBufferData(GL_UNIFORM_BUFFER, 4 * 1024 * 1024, g_scratch, GL_DYNAMIC_DRAW);
|
||||
glBindBuffer(GL_UNIFORM_BUFFER, 0);
|
||||
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
glGenFramebuffers(1, &g_passFbo[i]);
|
||||
glBindFramebuffer(GL_FRAMEBUFFER, g_passFbo[i]);
|
||||
glGenRenderbuffers(1, &g_passColor[i]);
|
||||
glBindRenderbuffer(GL_RENDERBUFFER, g_passColor[i]);
|
||||
glRenderbufferStorage(GL_RENDERBUFFER, GL_RGBA8, 256, 256);
|
||||
glFramebufferRenderbuffer(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_RENDERBUFFER, g_passColor[i]);
|
||||
if (glCheckFramebufferStatus(GL_FRAMEBUFFER) != GL_FRAMEBUFFER_COMPLETE) {
|
||||
bench_gl_failed("pass FBO incomplete", "");
|
||||
return;
|
||||
}
|
||||
}
|
||||
/* back to the main offscreen target the harness set up */
|
||||
glBindFramebuffer(GL_FRAMEBUFFER, g_mainFbo);
|
||||
|
||||
glGenSamplers(1, &g_sampler);
|
||||
glSamplerParameteri(g_sampler, GL_TEXTURE_MIN_FILTER, GL_NEAREST);
|
||||
glSamplerParameteri(g_sampler, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
|
||||
glEnable(GL_DEPTH_TEST);
|
||||
glClearColor(0.3f, 0.5f, 0.9f, 1.0f);
|
||||
glViewport(0, 0, 1280, 720);
|
||||
const GLenum setupError = glGetError();
|
||||
if (setupError != GL_NO_ERROR) {
|
||||
char message[64];
|
||||
snprintf(message, sizeof message, "0x%04x", setupError);
|
||||
bench_gl_failed("GL error during resource setup", message);
|
||||
}
|
||||
}
|
||||
|
||||
static void case_draw_tiny(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
|
||||
static void case_draw_uniform(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glUniform3f(g_uOffsetChunk, (float)(i & 15), (float)((i >> 4) & 15), 0.0f);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
|
||||
static void case_draw_multi_vao(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glBindVertexArray(g_vao[i % MAX_SECTIONS]);
|
||||
glUniform3f(g_uOffsetChunk, (float)(i & 15), (float)((i >> 4) & 15), 0.0f);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
|
||||
static void case_tex_pingpong(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glBindTexture(GL_TEXTURE_2D, (i & 1) ? g_texEntity : g_texAtlas);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
}
|
||||
|
||||
static void case_program_pingpong(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
if (i & 1) {
|
||||
glUseProgram(g_progEntity);
|
||||
glUniformMatrix4fv(g_uMvpEntity, 1, 0, g_mvp);
|
||||
} else {
|
||||
glUseProgram(g_progChunk);
|
||||
glUniform3f(g_uOffsetChunk, (float)(i & 15), 0.0f, 0.0f);
|
||||
}
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
glUseProgram(g_progChunk);
|
||||
}
|
||||
|
||||
/* a = uploads per frame, b = bytes per upload (0 => section size) */
|
||||
static void case_chunk_upload(int frame, long a, long b) {
|
||||
if (b <= 0) b = g_quadsPerSection * 4 * VERT_STRIDE;
|
||||
if (b > 4 * 1024 * 1024) b = 4 * 1024 * 1024;
|
||||
for (long i = 0; i < a; ++i) {
|
||||
int slot = (int)(((long)frame * a + i) % MAX_SECTIONS);
|
||||
glBindBuffer(GL_ARRAY_BUFFER, g_vbo[slot]);
|
||||
glBufferData(GL_ARRAY_BUFFER, b, NULL, GL_STATIC_DRAW); /* orphan */
|
||||
glBufferSubData(GL_ARRAY_BUFFER, 0, b, g_scratch);
|
||||
glBindVertexArray(g_vao[slot]);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/* a = sprite updates per frame */
|
||||
static void case_atlas_sprite(int frame, long a, long b) {
|
||||
(void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
int x = (int)((frame * 13 + i * 17) % (1024 - 16));
|
||||
int y = (int)((frame * 7 + i * 29) % (512 - 16));
|
||||
glTexSubImage2D(GL_TEXTURE_2D, 0, x, y, 16, 16, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
}
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
|
||||
/* a = lightmap updates (+draw) per frame */
|
||||
static void case_lightmap(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glActiveTexture(GL_TEXTURE0 + 2);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texLight);
|
||||
glTexSubImage2D(GL_TEXTURE_2D, 0, 0, 0, 16, 16, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
glActiveTexture(GL_TEXTURE0);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/* Composite: a = total draws, b = uploads per frame. Mix modeled on trace
|
||||
* analysis: chunk draws with per-draw offset uniform across sections, 10%
|
||||
* entity-style program flips, per-frame lightmap + sprite updates, b chunk
|
||||
* re-uploads. */
|
||||
static long g_mixSprites = 8;
|
||||
static void case_scene_mix(int frame, long a, long b) {
|
||||
glActiveTexture(GL_TEXTURE0 + 2);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texLight);
|
||||
glTexSubImage2D(GL_TEXTURE_2D, 0, 0, 0, 16, 16, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
glActiveTexture(GL_TEXTURE0);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
for (long i = 0; i < g_mixSprites; ++i) {
|
||||
int x = (int)((frame * 13 + i * 17) % (1024 - 16));
|
||||
int y = (int)((frame * 7 + i * 29) % (512 - 16));
|
||||
glTexSubImage2D(GL_TEXTURE_2D, 0, x, y, 16, 16, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
}
|
||||
for (long i = 0; i < b; ++i) {
|
||||
int slot = (int)(((long)frame * b + i) % MAX_SECTIONS);
|
||||
long bytes = g_quadsPerSection * 4 * VERT_STRIDE;
|
||||
glBindBuffer(GL_ARRAY_BUFFER, g_vbo[slot]);
|
||||
glBufferData(GL_ARRAY_BUFFER, bytes, NULL, GL_STATIC_DRAW);
|
||||
glBufferSubData(GL_ARRAY_BUFFER, 0, bytes, g_scratch);
|
||||
}
|
||||
long entityEvery = 10;
|
||||
for (long i = 0; i < a; ++i) {
|
||||
if (i % entityEvery == entityEvery - 1) {
|
||||
glUseProgram(g_progEntity);
|
||||
glUniformMatrix4fv(g_uMvpEntity, 1, 0, g_mvp);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texEntity);
|
||||
glBindVertexArray(g_vao[i % MAX_SECTIONS]);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
glUseProgram(g_progChunk);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
} else {
|
||||
glBindVertexArray(g_vao[i % MAX_SECTIONS]);
|
||||
glUniform3f(g_uOffsetChunk, (float)(i & 15), (float)((i >> 4) & 15), 0.0f);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- Trace-derived cases -------------------------------------------------
|
||||
* Per-frame call mixes measured from the three captured Minecraft traces
|
||||
* (render distance 32, 1280x720, hovering in-world). Each case reproduces one
|
||||
* renderer's dominant per-draw sequence at its measured rate, so the number a
|
||||
* backend posts here is directly comparable to what that game version asks of
|
||||
* the driver every frame.
|
||||
*
|
||||
* vanilla 1.21.1 : 5495 glDrawElements, 5490 glBindVertexArray,
|
||||
* 5487 glUniform3fv, 95 glTexSubImage2D (+382 glPixelStorei,
|
||||
* 247 glTexParameteri), 23 glBufferData per frame
|
||||
* fabric+sodium : 132 glMultiDrawElementsBaseVertex, 279 glBindVertexArray,
|
||||
* 132 glUniform3f, 32 glBufferData per frame
|
||||
* 26.2 snapshot : 3401 glDrawElementsBaseVertex, each preceded by
|
||||
* glBindBufferRange + glBindBuffer (3639/3412 per frame)
|
||||
*/
|
||||
/* vanilla: bind VAO, push the chunk offset, draw. a = draws per frame. */
|
||||
static void case_mc_vanilla_draw(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
float offset[3];
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glBindVertexArray(g_vao[i % MAX_SECTIONS]);
|
||||
offset[0] = (float)(i & 15);
|
||||
offset[1] = (float)((i >> 4) & 15);
|
||||
offset[2] = 0.0f;
|
||||
glUniform3fv(g_uOffsetChunk, 1, offset);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/* sodium: one multi-draw covers many chunk sections out of a shared buffer.
|
||||
* a = multi-draws per frame, b = sub-draws inside each. */
|
||||
static void case_mc_sodium_multidraw(int frame, long a, long b) {
|
||||
(void)frame;
|
||||
enum { kMaxSub = 64 };
|
||||
if (b <= 0 || b > kMaxSub) b = 32;
|
||||
GLsizei counts[kMaxSub];
|
||||
const void* offsets[kMaxSub];
|
||||
GLint baseVertices[kMaxSub];
|
||||
for (long s = 0; s < b; ++s) {
|
||||
counts[s] = (GLsizei)(g_quadsPerSection * 6 / b);
|
||||
offsets[s] = (const void*)(uintptr_t)(s * (g_quadsPerSection * 6 / b) * 4);
|
||||
baseVertices[s] = 0;
|
||||
}
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glBindVertexArray(g_vao[i % MAX_SECTIONS]);
|
||||
glBindVertexArray(g_vao[i % MAX_SECTIONS]); /* sodium rebinds ~2x per draw */
|
||||
glUniform3f(g_uOffsetChunk, (float)(i & 15), (float)((i >> 4) & 15), 0.0f);
|
||||
// Routed through the includer: GLES has no multi-draw-with-base-vertex, so
|
||||
// a native-driver harness emulates it with the loop the extension folds up.
|
||||
bench_multi_draw_elements_base_vertex(GL_TRIANGLES, counts, GL_UNSIGNED_INT, offsets,
|
||||
(GLsizei)b, baseVertices);
|
||||
}
|
||||
}
|
||||
|
||||
/* 26.2: every draw rebinds a fresh uniform-buffer range out of a ring.
|
||||
* a = draws per frame. */
|
||||
static void case_mc_ubo_range(int frame, long a, long b) {
|
||||
(void)b;
|
||||
const size_t slots = (4u * 1024u * 1024u) / g_uboSlot;
|
||||
for (long i = 0; i < a; ++i) {
|
||||
const size_t slot = (size_t)(((long)frame * a + i) % (long)slots);
|
||||
glBindBufferRange(GL_UNIFORM_BUFFER, 0, g_uboRing, (GLintptr)(slot * g_uboSlot),
|
||||
(GLsizeiptr)g_uboSlot);
|
||||
glBindBuffer(GL_UNIFORM_BUFFER, g_uboRing);
|
||||
glDrawElementsBaseVertex(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/* vanilla's animated-sprite path: every upload is wrapped in the pixel-store
|
||||
* and filter state Blaze3D re-sets around it. a = uploads per frame. */
|
||||
static void case_mc_tex_stream(int frame, long a, long b) {
|
||||
(void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glPixelStorei(GL_UNPACK_ALIGNMENT, 4);
|
||||
glPixelStorei(GL_UNPACK_ROW_LENGTH, 0);
|
||||
glPixelStorei(GL_UNPACK_SKIP_ROWS, 0);
|
||||
glPixelStorei(GL_UNPACK_SKIP_PIXELS, 0);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_EDGE);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE);
|
||||
int x = (int)((frame * 13 + i * 17) % (1024 - 16));
|
||||
int y = (int)((frame * 7 + i * 29) % (512 - 16));
|
||||
glTexSubImage2D(GL_TEXTURE_2D, 0, x, y, 16, 16, GL_RGBA, GL_UNSIGNED_BYTE, g_scratch);
|
||||
}
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
|
||||
/* Blaze3D re-resolves uniform locations by name every frame. a = lookups. */
|
||||
static void case_mc_uniform_lookup(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
static const char* names[4] = {"uMvp", "uOffset", "uAtlas", "uLight"};
|
||||
volatile GLint sink = 0;
|
||||
for (long i = 0; i < a; ++i) sink += glGetUniformLocation(g_progChunk, names[i & 3]);
|
||||
(void)sink;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
|
||||
/* 26.2 rebinds a sampler object per texture unit switch. a = switches. */
|
||||
static void case_mc_sampler_churn(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glActiveTexture(GL_TEXTURE0 + (GLenum)(i & 3));
|
||||
glBindTexture(GL_TEXTURE_2D, (i & 1) ? g_texEntity : g_texAtlas);
|
||||
glBindSampler((GLuint)(i & 3), g_sampler);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
glActiveTexture(GL_TEXTURE0);
|
||||
}
|
||||
|
||||
|
||||
/* 26.2 switches render targets constantly: 132 glBindFramebuffer and 198
|
||||
* glDrawBuffers per frame. Pass switching is where a Vulkan backend pays for
|
||||
* render-pass breaks, so this case is the one to watch on Magma. a = passes. */
|
||||
static void case_mc_pass_switch(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
static const GLenum kColor0[1] = {GL_COLOR_ATTACHMENT0};
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glBindFramebuffer(GL_FRAMEBUFFER, g_passFbo[i & 1]);
|
||||
glDrawBuffers(1, kColor0);
|
||||
glViewport(0, 0, 256, 256);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
glBindFramebuffer(GL_FRAMEBUFFER, g_mainFbo);
|
||||
glViewport(0, 0, 1280, 720);
|
||||
}
|
||||
|
||||
/* Blaze3D toggles blend around batches: 46 glEnable/glDisable pairs and 28
|
||||
* glBlendFuncSeparate per vanilla frame. a = toggle pairs. */
|
||||
static void case_mc_state_toggle(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
glEnable(GL_BLEND);
|
||||
glBlendFuncSeparate(GL_SRC_ALPHA, GL_ONE_MINUS_SRC_ALPHA, GL_ONE, GL_ZERO);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
glDisable(GL_BLEND);
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/* 26.2 re-sets texture parameters relentlessly - 612 glTexParameteri per frame,
|
||||
* almost always to the value already in place. Measures redundant-param
|
||||
* filtering. a = parameter writes. */
|
||||
static void case_mc_tex_param(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
glBindTexture(GL_TEXTURE_2D, g_texAtlas);
|
||||
for (long i = 0; i < a; i += 4) {
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_EDGE);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST_MIPMAP_LINEAR);
|
||||
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST);
|
||||
}
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
|
||||
/* Sodium switches programs mid-frame far more than vanilla: 62 glUseProgram and
|
||||
* 60 mat4 uploads per frame. a = program switches. */
|
||||
static void case_mc_use_program(int frame, long a, long b) {
|
||||
(void)frame; (void)b;
|
||||
glBindVertexArray(g_vao[0]);
|
||||
for (long i = 0; i < a; ++i) {
|
||||
if (i & 1) {
|
||||
glUseProgram(g_progEntity);
|
||||
glUniformMatrix4fv(g_uMvpEntity, 1, 0, g_mvp);
|
||||
} else {
|
||||
glUseProgram(g_progChunk);
|
||||
glUniformMatrix4fv(g_uMvpChunk, 1, 0, g_mvp);
|
||||
}
|
||||
glDrawElements(GL_TRIANGLES, g_quadsPerSection * 6, GL_UNSIGNED_INT, 0);
|
||||
}
|
||||
glUseProgram(g_progChunk);
|
||||
}
|
||||
|
||||
/* ---- the case table both harnesses iterate --------------------------------
|
||||
* a/b are the case's own knobs; opsPerFrame is what one bench frame is
|
||||
* normalised by, so ns_per_op compares across renderers. The mc_* rates are
|
||||
* the per-frame call counts measured from the captured traces.
|
||||
*/
|
||||
typedef void (*bench_case_fn)(int frame, long a, long b);
|
||||
|
||||
typedef struct {
|
||||
const char* name;
|
||||
bench_case_fn fn;
|
||||
long a, b, opsPerFrame;
|
||||
} BenchCaseDesc;
|
||||
|
||||
static const BenchCaseDesc kBenchCases[] = {
|
||||
{"mc_vanilla_draw", case_mc_vanilla_draw, 5495, 0, 5495},
|
||||
{"mc_sodium_multidraw", case_mc_sodium_multidraw, 132, 32, 132},
|
||||
{"mc_ubo_range", case_mc_ubo_range, 3401, 0, 3401},
|
||||
{"mc_tex_stream", case_mc_tex_stream, 95, 0, 95},
|
||||
{"mc_uniform_lookup", case_mc_uniform_lookup, 41, 0, 41},
|
||||
{"mc_sampler_churn", case_mc_sampler_churn, 306, 0, 306},
|
||||
{"mc_pass_switch", case_mc_pass_switch, 132, 0, 132},
|
||||
{"mc_state_toggle", case_mc_state_toggle, 46, 0, 46},
|
||||
{"mc_tex_param", case_mc_tex_param, 612, 0, 612},
|
||||
{"mc_use_program", case_mc_use_program, 62, 0, 62},
|
||||
{"draw_tiny", case_draw_tiny, 2048, 0, 2048},
|
||||
{"draw_uniform", case_draw_uniform, 2048, 0, 2048},
|
||||
{"draw_multi_vao", case_draw_multi_vao, 2048, 0, 2048},
|
||||
{"tex_pingpong", case_tex_pingpong, 1024, 0, 1024},
|
||||
{"program_pingpong", case_program_pingpong, 512, 0, 512},
|
||||
{"chunk_upload", case_chunk_upload, 24, 0, 24},
|
||||
{"atlas_sprite", case_atlas_sprite, 32, 0, 32},
|
||||
{"lightmap", case_lightmap, 4, 0, 4},
|
||||
{"scene_mix", case_scene_mix, 2048, 12, 2048},
|
||||
};
|
||||
static const int kBenchCaseCount = (int)(sizeof kBenchCases / sizeof kBenchCases[0]);
|
||||
@@ -0,0 +1,41 @@
|
||||
#!/bin/bash
|
||||
# Run the headless EGL DriverBench on one renderer:
|
||||
# ./run_driver_bench.sh native [bench args...]
|
||||
# ./run_driver_bench.sh espryt <libMobileGL.so> [bench args...]
|
||||
# ./run_driver_bench.sh magma <libMobileGL.so> [bench args...]
|
||||
# The bench dlopens exactly one EGL provider (DRIVERBENCH_EGL_LIB): the system
|
||||
# libEGL.so.1 for native, or the given libMobileGL.so for a MobileGL backend -
|
||||
# no LD_LIBRARY_PATH shadowing, so MobileGL's own loader still finds the real
|
||||
# driver underneath.
|
||||
#
|
||||
# Pin the vendor libraries explicitly. A bare libEGL.so.1 on a glvnd system
|
||||
# picks whatever vendor eglGetDisplay(EGL_DEFAULT_DISPLAY) resolves first,
|
||||
# which is Mesa/llvmpipe here - a software rasteriser silently replacing the
|
||||
# GPU under a benchmark. Override MGL_EGL_VENDOR / MGL_VK_ICD to test another
|
||||
# driver.
|
||||
set -eu
|
||||
HERE=$(cd "$(dirname "$0")" && pwd)
|
||||
BENCH=${DRIVERBENCH_BIN:-$HERE/DriverBench}
|
||||
EGL_VENDOR=${MGL_EGL_VENDOR:-/usr/share/glvnd/egl_vendor.d/10_nvidia.json}
|
||||
VK_ICD=${MGL_VK_ICD:-/usr/share/vulkan/icd.d/nvidia_icd.x86_64.json}
|
||||
MODE=$1; shift
|
||||
|
||||
export __EGL_VENDOR_LIBRARY_FILENAMES=$EGL_VENDOR
|
||||
export EGL_PLATFORM=${EGL_PLATFORM:-x11}
|
||||
|
||||
case "$MODE" in
|
||||
native)
|
||||
export DRIVERBENCH_EGL_LIB=${DRIVERBENCH_EGL_LIB:-libEGL.so.1}
|
||||
;;
|
||||
espryt)
|
||||
export DRIVERBENCH_EGL_LIB=$(readlink -f "$1"); shift
|
||||
export MOBILEGL_BACKEND_TYPE=DirectGLES
|
||||
;;
|
||||
magma)
|
||||
export DRIVERBENCH_EGL_LIB=$(readlink -f "$1"); shift
|
||||
export MOBILEGL_BACKEND_TYPE=DirectVulkan
|
||||
export VK_ICD_FILENAMES=$VK_ICD
|
||||
;;
|
||||
*) echo "unknown mode: $MODE (native|espryt|magma)"; exit 1 ;;
|
||||
esac
|
||||
exec "$BENCH" "$@"
|
||||
@@ -16,4 +16,5 @@ target_link_libraries(
|
||||
${LINK_LIBRARIES}
|
||||
)
|
||||
|
||||
add_test(NAME ProgramBench COMMAND ProgramBench --benchmark_counters_tabular=true)
|
||||
add_test(NAME ProgramBench COMMAND ProgramBench --benchmark_counters_tabular=true)
|
||||
set_tests_properties(ProgramBench PROPERTIES LABELS benchmark)
|
||||
@@ -0,0 +1,733 @@
|
||||
// MobileGL - MobileGL/MG_Impl/CGLImpl/CGLImpl.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "CGLImpl.h"
|
||||
|
||||
#if defined(__APPLE__)
|
||||
#include "../EGLImpl/EGLImpl.h"
|
||||
|
||||
namespace MobileGL::MG_Impl::CGLImpl {
|
||||
namespace {
|
||||
struct PixelFormatObject {
|
||||
Uint32 RetainCount = 1;
|
||||
Bool DoubleBuffer = true;
|
||||
GLint ColorSize = 24;
|
||||
GLint AlphaSize = 8;
|
||||
GLint DepthSize = 24;
|
||||
GLint StencilSize = 8;
|
||||
GLint SampleBuffers = 0;
|
||||
GLint Samples = 0;
|
||||
GLint Profile = kCGLOGLPVersion_3_2_Core;
|
||||
GLint RendererId = 0x4d474c;
|
||||
GLint DisplayMask = 0;
|
||||
};
|
||||
|
||||
struct ContextObject {
|
||||
Uint32 RetainCount = 1;
|
||||
CGLPixelFormatObj PixelFormat = nullptr;
|
||||
CGLContextObj Share = nullptr;
|
||||
EGLDisplay Display = EGL_NO_DISPLAY;
|
||||
EGLConfig Config = nullptr;
|
||||
EGLContext Context = EGL_NO_CONTEXT;
|
||||
EGLSurface Surface = EGL_NO_SURFACE;
|
||||
void* NSObject = nullptr;
|
||||
void* View = nullptr;
|
||||
void* MetalLayer = nullptr;
|
||||
GLint SwapInterval = 1;
|
||||
GLint VirtualScreen = 0;
|
||||
GLint SurfaceBackingSize[2] = {0, 0};
|
||||
Bool HasDrawable = false;
|
||||
Bool Locked = false;
|
||||
};
|
||||
|
||||
std::recursive_mutex& RegistryMutex() {
|
||||
static auto* mutex = new std::recursive_mutex();
|
||||
return *mutex;
|
||||
}
|
||||
|
||||
Uint64& NextPixelFormatHandle() {
|
||||
static auto* handle = new Uint64(1);
|
||||
return *handle;
|
||||
}
|
||||
|
||||
Uint64& NextContextHandle() {
|
||||
static auto* handle = new Uint64(1);
|
||||
return *handle;
|
||||
}
|
||||
|
||||
UnorderedMap<CGLPixelFormatObj, PixelFormatObject>& PixelFormats() {
|
||||
static auto* formats = new UnorderedMap<CGLPixelFormatObj, PixelFormatObject>();
|
||||
return *formats;
|
||||
}
|
||||
|
||||
UnorderedMap<CGLContextObj, ContextObject>& Contexts() {
|
||||
static auto* contexts = new UnorderedMap<CGLContextObj, ContextObject>();
|
||||
return *contexts;
|
||||
}
|
||||
|
||||
UnorderedMap<std::thread::id, CGLContextObj>& CurrentContexts() {
|
||||
static auto* contexts = new UnorderedMap<std::thread::id, CGLContextObj>();
|
||||
return *contexts;
|
||||
}
|
||||
|
||||
CGLPixelFormatObj EncodePixelFormat(Uint64 handle) {
|
||||
return reinterpret_cast<CGLPixelFormatObj>(static_cast<SizeT>(handle));
|
||||
}
|
||||
|
||||
CGLContextObj EncodeContext(Uint64 handle) {
|
||||
return reinterpret_cast<CGLContextObj>(static_cast<SizeT>(handle));
|
||||
}
|
||||
|
||||
std::thread::id CurrentThreadKey() {
|
||||
return std::this_thread::get_id();
|
||||
}
|
||||
|
||||
Bool AttributeHasValue(CGLPixelFormatAttribute attrib) {
|
||||
switch (attrib) {
|
||||
case kCGLPFAColorSize:
|
||||
case kCGLPFAAlphaSize:
|
||||
case kCGLPFADepthSize:
|
||||
case kCGLPFAStencilSize:
|
||||
case kCGLPFASampleBuffers:
|
||||
case kCGLPFASamples:
|
||||
case kCGLPFARendererID:
|
||||
case kCGLPFADisplayMask:
|
||||
case kCGLPFAOpenGLProfile:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
void ApplyPixelFormatAttribute(PixelFormatObject& pixelFormat,
|
||||
CGLPixelFormatAttribute attrib,
|
||||
GLint value) {
|
||||
switch (attrib) {
|
||||
case kCGLPFADoubleBuffer:
|
||||
pixelFormat.DoubleBuffer = true;
|
||||
break;
|
||||
case kCGLPFAColorSize:
|
||||
pixelFormat.ColorSize = value;
|
||||
break;
|
||||
case kCGLPFAAlphaSize:
|
||||
pixelFormat.AlphaSize = value;
|
||||
break;
|
||||
case kCGLPFADepthSize:
|
||||
pixelFormat.DepthSize = value;
|
||||
break;
|
||||
case kCGLPFAStencilSize:
|
||||
pixelFormat.StencilSize = value;
|
||||
break;
|
||||
case kCGLPFASampleBuffers:
|
||||
pixelFormat.SampleBuffers = value;
|
||||
break;
|
||||
case kCGLPFASamples:
|
||||
pixelFormat.Samples = value;
|
||||
break;
|
||||
case kCGLPFAOpenGLProfile:
|
||||
pixelFormat.Profile = value;
|
||||
break;
|
||||
case kCGLPFARendererID:
|
||||
pixelFormat.RendererId = value;
|
||||
break;
|
||||
case kCGLPFADisplayMask:
|
||||
pixelFormat.DisplayMask = value;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
Bool InitEGLContext(ContextObject& object, CGLPixelFormatObj pix, CGLContextObj share) {
|
||||
auto* pixelFormat = [&]() -> PixelFormatObject* {
|
||||
auto& pixelFormats = PixelFormats();
|
||||
auto it = pixelFormats.find(pix);
|
||||
return it == pixelFormats.end() ? nullptr : &it->second;
|
||||
}();
|
||||
if (!pixelFormat) {
|
||||
return false;
|
||||
}
|
||||
|
||||
EGLDisplay display = EGLImpl::GetDisplay(EGL_DEFAULT_DISPLAY);
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
return false;
|
||||
}
|
||||
if (!EGLImpl::Initialize(display, nullptr, nullptr)) {
|
||||
return false;
|
||||
}
|
||||
EGLImpl::BindAPI(EGL_OPENGL_API);
|
||||
|
||||
const EGLint attribs[] = {
|
||||
EGL_RED_SIZE, 8,
|
||||
EGL_GREEN_SIZE, 8,
|
||||
EGL_BLUE_SIZE, 8,
|
||||
EGL_ALPHA_SIZE, std::max(pixelFormat->AlphaSize, 0),
|
||||
EGL_DEPTH_SIZE, std::max(pixelFormat->DepthSize, 0),
|
||||
EGL_STENCIL_SIZE, std::max(pixelFormat->StencilSize, 0),
|
||||
EGL_SURFACE_TYPE, EGL_WINDOW_BIT | EGL_PBUFFER_BIT,
|
||||
EGL_RENDERABLE_TYPE, EGL_OPENGL_BIT,
|
||||
EGL_NONE,
|
||||
};
|
||||
|
||||
EGLConfig config = nullptr;
|
||||
EGLint count = 0;
|
||||
if (!EGLImpl::ChooseConfig(display, attribs, &config, 1, &count) || count <= 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
EGLContext shareContext = EGL_NO_CONTEXT;
|
||||
if (share != nullptr) {
|
||||
auto& contexts = Contexts();
|
||||
auto shareIt = contexts.find(share);
|
||||
if (shareIt == contexts.end()) {
|
||||
return false;
|
||||
}
|
||||
shareContext = shareIt->second.Context;
|
||||
}
|
||||
|
||||
const EGLint contextAttribs[] = {
|
||||
EGL_CONTEXT_MAJOR_VERSION, 3,
|
||||
EGL_CONTEXT_MINOR_VERSION, 3,
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLContext eglContext = EGLImpl::CreateContext(display, config, shareContext, contextAttribs);
|
||||
if (eglContext == EGL_NO_CONTEXT) {
|
||||
return false;
|
||||
}
|
||||
|
||||
object.Display = display;
|
||||
object.Config = config;
|
||||
object.Context = eglContext;
|
||||
object.PixelFormat = pix;
|
||||
object.Share = share;
|
||||
return true;
|
||||
}
|
||||
|
||||
ContextObject* TryGetContext(CGLContextObj ctx) {
|
||||
auto& contexts = Contexts();
|
||||
auto it = contexts.find(ctx);
|
||||
return it == contexts.end() ? nullptr : &it->second;
|
||||
}
|
||||
|
||||
const ContextObject* TryGetContext(CGLContextObj ctx, const std::lock_guard<std::recursive_mutex>&) {
|
||||
auto& contexts = Contexts();
|
||||
auto it = contexts.find(ctx);
|
||||
return it == contexts.end() ? nullptr : &it->second;
|
||||
}
|
||||
|
||||
PixelFormatObject* TryGetPixelFormat(CGLPixelFormatObj pix) {
|
||||
auto& pixelFormats = PixelFormats();
|
||||
auto it = pixelFormats.find(pix);
|
||||
return it == pixelFormats.end() ? nullptr : &it->second;
|
||||
}
|
||||
|
||||
CGLError MakeCurrentLocked(CGLContextObj ctx, ContextObject& object) {
|
||||
CurrentContexts()[CurrentThreadKey()] = ctx;
|
||||
if (!object.HasDrawable || object.Surface == EGL_NO_SURFACE) {
|
||||
return kCGLNoError;
|
||||
}
|
||||
if (!EGLImpl::MakeCurrent(object.Display, object.Surface, object.Surface, object.Context)) {
|
||||
return kCGLBadState;
|
||||
}
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError RecreateSurfaceLocked(CGLContextObj ctx, ContextObject& object) {
|
||||
if (!object.MetalLayer) {
|
||||
return kCGLBadDrawable;
|
||||
}
|
||||
if (object.Surface != EGL_NO_SURFACE) {
|
||||
EGLImpl::DestroySurface(object.Display, object.Surface);
|
||||
object.Surface = EGL_NO_SURFACE;
|
||||
}
|
||||
const EGLAttrib attribs[] = {
|
||||
EGL_WIDTH, std::max<GLint>(object.SurfaceBackingSize[0], 1),
|
||||
EGL_HEIGHT, std::max<GLint>(object.SurfaceBackingSize[1], 1),
|
||||
EGL_NONE,
|
||||
};
|
||||
EGLSurface surface = EGLImpl::CreatePlatformWindowSurface(object.Display, object.Config,
|
||||
object.MetalLayer, attribs);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
object.HasDrawable = false;
|
||||
return kCGLBadDrawable;
|
||||
}
|
||||
object.Surface = surface;
|
||||
object.HasDrawable = true;
|
||||
return GetCurrentContext() == ctx ? MakeCurrentLocked(ctx, object) : kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError ResizeSurfaceLocked(ContextObject& object) {
|
||||
if (object.Surface == EGL_NO_SURFACE) {
|
||||
return kCGLBadDrawable;
|
||||
}
|
||||
return EGLImpl::ResizePlatformWindowSurface(
|
||||
object.Display, object.Surface,
|
||||
std::max<GLint>(object.SurfaceBackingSize[0], 1),
|
||||
std::max<GLint>(object.SurfaceBackingSize[1], 1))
|
||||
? kCGLNoError
|
||||
: kCGLBadDrawable;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
CGLError ChoosePixelFormat(const CGLPixelFormatAttribute* attribs, CGLPixelFormatObj* pix, GLint* npix) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
if (!pix || !npix) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
|
||||
PixelFormatObject object;
|
||||
if (attribs) {
|
||||
for (SizeT i = 0; attribs[i] != static_cast<CGLPixelFormatAttribute>(0); ++i) {
|
||||
const auto attrib = attribs[i];
|
||||
GLint value = 1;
|
||||
if (AttributeHasValue(attrib)) {
|
||||
value = static_cast<GLint>(attribs[++i]);
|
||||
}
|
||||
ApplyPixelFormatAttribute(object, attrib, value);
|
||||
}
|
||||
}
|
||||
|
||||
const auto handle = EncodePixelFormat(NextPixelFormatHandle()++);
|
||||
PixelFormats()[handle] = object;
|
||||
*pix = handle;
|
||||
*npix = 1;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError DestroyPixelFormat(CGLPixelFormatObj pix) {
|
||||
ReleasePixelFormat(pix);
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError DescribePixelFormat(CGLPixelFormatObj pix, GLint pixNum, CGLPixelFormatAttribute attrib, GLint* value) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
if (!value) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
if (pixNum != 0 && pixNum != 1) {
|
||||
return kCGLBadValue;
|
||||
}
|
||||
auto* pixelFormat = TryGetPixelFormat(pix);
|
||||
if (!pixelFormat) {
|
||||
return kCGLBadPixelFormat;
|
||||
}
|
||||
|
||||
switch (attrib) {
|
||||
case kCGLPFADoubleBuffer:
|
||||
*value = pixelFormat->DoubleBuffer ? 1 : 0;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAAccelerated:
|
||||
case kCGLPFAAcceleratedCompute:
|
||||
case kCGLPFASupportsAutomaticGraphicsSwitching:
|
||||
*value = 1;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAColorSize:
|
||||
*value = pixelFormat->ColorSize;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAAlphaSize:
|
||||
*value = pixelFormat->AlphaSize;
|
||||
return kCGLNoError;
|
||||
case kCGLPFADepthSize:
|
||||
*value = pixelFormat->DepthSize;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAStencilSize:
|
||||
*value = pixelFormat->StencilSize;
|
||||
return kCGLNoError;
|
||||
case kCGLPFASampleBuffers:
|
||||
*value = pixelFormat->SampleBuffers;
|
||||
return kCGLNoError;
|
||||
case kCGLPFASamples:
|
||||
*value = pixelFormat->Samples;
|
||||
return kCGLNoError;
|
||||
case kCGLPFARendererID:
|
||||
*value = pixelFormat->RendererId;
|
||||
return kCGLNoError;
|
||||
case kCGLPFADisplayMask:
|
||||
*value = pixelFormat->DisplayMask;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAOpenGLProfile:
|
||||
*value = pixelFormat->Profile;
|
||||
return kCGLNoError;
|
||||
case kCGLPFAVirtualScreenCount:
|
||||
*value = 1;
|
||||
return kCGLNoError;
|
||||
default:
|
||||
*value = 0;
|
||||
return kCGLNoError;
|
||||
}
|
||||
}
|
||||
|
||||
void ReleasePixelFormat(CGLPixelFormatObj pix) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* pixelFormat = TryGetPixelFormat(pix);
|
||||
if (!pixelFormat) {
|
||||
return;
|
||||
}
|
||||
if (pixelFormat->RetainCount > 1) {
|
||||
--pixelFormat->RetainCount;
|
||||
return;
|
||||
}
|
||||
PixelFormats().erase(pix);
|
||||
}
|
||||
|
||||
CGLPixelFormatObj RetainPixelFormat(CGLPixelFormatObj pix) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* pixelFormat = TryGetPixelFormat(pix);
|
||||
if (pixelFormat) {
|
||||
++pixelFormat->RetainCount;
|
||||
}
|
||||
return pix;
|
||||
}
|
||||
|
||||
GLuint GetPixelFormatRetainCount(CGLPixelFormatObj pix) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* pixelFormat = TryGetPixelFormat(pix);
|
||||
return pixelFormat ? pixelFormat->RetainCount : 0;
|
||||
}
|
||||
|
||||
CGLError CreateContext(CGLPixelFormatObj pix, CGLContextObj share, CGLContextObj* ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
if (!ctx) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
if (!TryGetPixelFormat(pix)) {
|
||||
return kCGLBadPixelFormat;
|
||||
}
|
||||
if (share && !TryGetContext(share)) {
|
||||
return kCGLBadMatch;
|
||||
}
|
||||
|
||||
ContextObject object;
|
||||
if (!InitEGLContext(object, pix, share)) {
|
||||
return kCGLBadAlloc;
|
||||
}
|
||||
RetainPixelFormat(pix);
|
||||
const auto handle = EncodeContext(NextContextHandle()++);
|
||||
Contexts()[handle] = object;
|
||||
*ctx = handle;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError DestroyContext(CGLContextObj ctx) {
|
||||
ReleaseContext(ctx);
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLContextObj RetainContext(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (object) {
|
||||
++object->RetainCount;
|
||||
}
|
||||
return ctx;
|
||||
}
|
||||
|
||||
void ReleaseContext(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return;
|
||||
}
|
||||
if (object->RetainCount > 1) {
|
||||
--object->RetainCount;
|
||||
return;
|
||||
}
|
||||
if (object->Surface != EGL_NO_SURFACE) {
|
||||
EGLImpl::DestroySurface(object->Display, object->Surface);
|
||||
}
|
||||
if (object->Context != EGL_NO_CONTEXT) {
|
||||
EGLImpl::DestroyContext(object->Display, object->Context);
|
||||
}
|
||||
ReleasePixelFormat(object->PixelFormat);
|
||||
auto& currentContexts = CurrentContexts();
|
||||
for (auto it = currentContexts.begin(); it != currentContexts.end();) {
|
||||
if (it->second == ctx) {
|
||||
it = currentContexts.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
Contexts().erase(ctx);
|
||||
}
|
||||
|
||||
GLuint GetContextRetainCount(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
return object ? object->RetainCount : 0;
|
||||
}
|
||||
|
||||
CGLPixelFormatObj GetPixelFormat(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
return object ? object->PixelFormat : nullptr;
|
||||
}
|
||||
|
||||
CGLError SetCurrentContext(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
if (!ctx) {
|
||||
CurrentContexts().erase(CurrentThreadKey());
|
||||
EGLImpl::MakeCurrent(EGL_NO_DISPLAY, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT);
|
||||
return kCGLNoError;
|
||||
}
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
return MakeCurrentLocked(ctx, *object);
|
||||
}
|
||||
|
||||
CGLContextObj GetCurrentContext() {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto& currentContexts = CurrentContexts();
|
||||
auto it = currentContexts.find(CurrentThreadKey());
|
||||
return it == currentContexts.end() ? nullptr : it->second;
|
||||
}
|
||||
|
||||
CGLError SetVirtualScreen(CGLContextObj ctx, GLint screen) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (screen != 0) {
|
||||
return kCGLBadValue;
|
||||
}
|
||||
object->VirtualScreen = screen;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError GetVirtualScreen(CGLContextObj ctx, GLint* screen) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!screen) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
*screen = object->VirtualScreen;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError SetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!params && pname != kCGLCPReclaimResources) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
switch (pname) {
|
||||
case kCGLCPSwapInterval:
|
||||
object->SwapInterval = params[0];
|
||||
EGLImpl::SwapInterval(object->Display, object->SwapInterval);
|
||||
return kCGLNoError;
|
||||
case kCGLCPSurfaceBackingSize:
|
||||
{
|
||||
const GLint width = std::max<GLint>(params[0], 1);
|
||||
const GLint height = std::max<GLint>(params[1], 1);
|
||||
if (object->SurfaceBackingSize[0] == width && object->SurfaceBackingSize[1] == height) {
|
||||
return kCGLNoError;
|
||||
}
|
||||
object->SurfaceBackingSize[0] = width;
|
||||
object->SurfaceBackingSize[1] = height;
|
||||
if (object->MetalLayer && object->Surface != EGL_NO_SURFACE) {
|
||||
return ResizeSurfaceLocked(*object);
|
||||
}
|
||||
return kCGLNoError;
|
||||
}
|
||||
case kCGLCPSurfaceOpacity:
|
||||
case kCGLCPSurfaceOrder:
|
||||
case kCGLCPMPSwapsInFlight:
|
||||
case kCGLCPReclaimResources:
|
||||
return kCGLNoError;
|
||||
default:
|
||||
return kCGLNoError;
|
||||
}
|
||||
}
|
||||
|
||||
CGLError GetParameter(CGLContextObj ctx, CGLContextParameter pname, GLint* params) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!params) {
|
||||
return kCGLBadAddress;
|
||||
}
|
||||
switch (pname) {
|
||||
case kCGLCPSwapInterval:
|
||||
params[0] = object->SwapInterval;
|
||||
return kCGLNoError;
|
||||
case kCGLCPSurfaceBackingSize:
|
||||
params[0] = object->SurfaceBackingSize[0];
|
||||
params[1] = object->SurfaceBackingSize[1];
|
||||
return kCGLNoError;
|
||||
case kCGLCPCurrentRendererID:
|
||||
params[0] = 0x4d474c;
|
||||
return kCGLNoError;
|
||||
case kCGLCPGPUVertexProcessing:
|
||||
case kCGLCPGPUFragmentProcessing:
|
||||
case kCGLCPHasDrawable:
|
||||
params[0] = object->HasDrawable ? 1 : 0;
|
||||
return kCGLNoError;
|
||||
case kCGLCPMPSwapsInFlight:
|
||||
params[0] = 1;
|
||||
return kCGLNoError;
|
||||
default:
|
||||
params[0] = 0;
|
||||
return kCGLNoError;
|
||||
}
|
||||
}
|
||||
|
||||
CGLError UpdateContext(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
return TryGetContext(ctx) ? kCGLNoError : kCGLBadContext;
|
||||
}
|
||||
|
||||
CGLError ClearDrawable(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (object->Surface != EGL_NO_SURFACE) {
|
||||
EGLImpl::DestroySurface(object->Display, object->Surface);
|
||||
}
|
||||
object->Surface = EGL_NO_SURFACE;
|
||||
object->View = nullptr;
|
||||
object->MetalLayer = nullptr;
|
||||
object->HasDrawable = false;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError FlushDrawable(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!object->HasDrawable || object->Surface == EGL_NO_SURFACE) {
|
||||
return kCGLBadDrawable;
|
||||
}
|
||||
const auto currentError = MakeCurrentLocked(ctx, *object);
|
||||
if (currentError != kCGLNoError) {
|
||||
return currentError;
|
||||
}
|
||||
return EGLImpl::SwapBuffers(object->Display, object->Surface) ? kCGLNoError : kCGLBadDrawable;
|
||||
}
|
||||
|
||||
CGLError LockContext(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
object->Locked = true;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
CGLError UnlockContext(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
object->Locked = false;
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
void GetVersion(GLint* majorvers, GLint* minorvers) {
|
||||
if (majorvers) {
|
||||
*majorvers = 1;
|
||||
}
|
||||
if (minorvers) {
|
||||
*minorvers = 0;
|
||||
}
|
||||
}
|
||||
|
||||
const char* ErrorString(CGLError error) {
|
||||
switch (error) {
|
||||
case kCGLNoError:
|
||||
return "no error";
|
||||
case kCGLBadAttribute:
|
||||
return "invalid pixel format attribute";
|
||||
case kCGLBadPixelFormat:
|
||||
return "invalid pixel format";
|
||||
case kCGLBadContext:
|
||||
return "invalid context";
|
||||
case kCGLBadDrawable:
|
||||
return "invalid drawable";
|
||||
case kCGLBadState:
|
||||
return "invalid context state";
|
||||
case kCGLBadValue:
|
||||
return "invalid numerical value";
|
||||
case kCGLBadMatch:
|
||||
return "invalid share context";
|
||||
case kCGLBadAddress:
|
||||
return "invalid pointer";
|
||||
case kCGLBadAlloc:
|
||||
return "invalid memory allocation";
|
||||
default:
|
||||
return "unknown CGL error";
|
||||
}
|
||||
}
|
||||
|
||||
CGLError AttachDrawable(CGLContextObj ctx, void* nsView, void* metalLayer, GLint width, GLint height) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (!object) {
|
||||
return kCGLBadContext;
|
||||
}
|
||||
if (!metalLayer) {
|
||||
return kCGLBadDrawable;
|
||||
}
|
||||
width = std::max<GLint>(width, 1);
|
||||
height = std::max<GLint>(height, 1);
|
||||
const Bool sameSize = object->SurfaceBackingSize[0] == width && object->SurfaceBackingSize[1] == height;
|
||||
object->SurfaceBackingSize[0] = width;
|
||||
object->SurfaceBackingSize[1] = height;
|
||||
if (object->Surface != EGL_NO_SURFACE && object->MetalLayer == metalLayer && sameSize) {
|
||||
object->View = nsView;
|
||||
object->HasDrawable = true;
|
||||
return kCGLNoError;
|
||||
}
|
||||
if (object->Surface != EGL_NO_SURFACE && object->MetalLayer == metalLayer) {
|
||||
object->View = nsView;
|
||||
object->HasDrawable = true;
|
||||
return ResizeSurfaceLocked(*object);
|
||||
}
|
||||
if (object->Surface != EGL_NO_SURFACE) {
|
||||
EGLImpl::DestroySurface(object->Display, object->Surface);
|
||||
object->Surface = EGL_NO_SURFACE;
|
||||
}
|
||||
object->View = nsView;
|
||||
object->MetalLayer = metalLayer;
|
||||
const auto recreateError = RecreateSurfaceLocked(ctx, *object);
|
||||
if (recreateError != kCGLNoError) {
|
||||
return recreateError;
|
||||
}
|
||||
return kCGLNoError;
|
||||
}
|
||||
|
||||
void* GetContextNSObject(CGLContextObj ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
return object ? object->NSObject : nullptr;
|
||||
}
|
||||
|
||||
void SetContextNSObject(CGLContextObj ctx, void* nsObject) {
|
||||
const std::lock_guard<std::recursive_mutex> lock(RegistryMutex());
|
||||
auto* object = TryGetContext(ctx);
|
||||
if (object) {
|
||||
object->NSObject = nsObject;
|
||||
}
|
||||
}
|
||||
} // namespace MobileGL::MG_Impl::CGLImpl
|
||||
#endif
|
||||
@@ -0,0 +1,51 @@
|
||||
// MobileGL - MobileGL/MG_Impl/CGLImpl/CGLImpl.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include <Includes.h>
|
||||
|
||||
#if defined(__APPLE__)
|
||||
#ifndef GL_SILENCE_DEPRECATION
|
||||
#define GL_SILENCE_DEPRECATION
|
||||
#endif
|
||||
#include <OpenGL/OpenGL.h>
|
||||
|
||||
namespace MobileGL::MG_Impl::CGLImpl {
|
||||
CGLError ChoosePixelFormat(const CGLPixelFormatAttribute* attribs, CGLPixelFormatObj* pix, GLint* npix);
|
||||
CGLError DestroyPixelFormat(CGLPixelFormatObj pix);
|
||||
CGLError DescribePixelFormat(CGLPixelFormatObj pix, GLint pixNum, CGLPixelFormatAttribute attrib, GLint* value);
|
||||
void ReleasePixelFormat(CGLPixelFormatObj pix);
|
||||
CGLPixelFormatObj RetainPixelFormat(CGLPixelFormatObj pix);
|
||||
GLuint GetPixelFormatRetainCount(CGLPixelFormatObj pix);
|
||||
|
||||
CGLError CreateContext(CGLPixelFormatObj pix, CGLContextObj share, CGLContextObj* ctx);
|
||||
CGLError DestroyContext(CGLContextObj ctx);
|
||||
CGLContextObj RetainContext(CGLContextObj ctx);
|
||||
void ReleaseContext(CGLContextObj ctx);
|
||||
GLuint GetContextRetainCount(CGLContextObj ctx);
|
||||
CGLPixelFormatObj GetPixelFormat(CGLContextObj ctx);
|
||||
|
||||
CGLError SetCurrentContext(CGLContextObj ctx);
|
||||
CGLContextObj GetCurrentContext();
|
||||
CGLError SetVirtualScreen(CGLContextObj ctx, GLint screen);
|
||||
CGLError GetVirtualScreen(CGLContextObj ctx, GLint* screen);
|
||||
CGLError SetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params);
|
||||
CGLError GetParameter(CGLContextObj ctx, CGLContextParameter pname, GLint* params);
|
||||
CGLError UpdateContext(CGLContextObj ctx);
|
||||
CGLError ClearDrawable(CGLContextObj ctx);
|
||||
CGLError FlushDrawable(CGLContextObj ctx);
|
||||
CGLError LockContext(CGLContextObj ctx);
|
||||
CGLError UnlockContext(CGLContextObj ctx);
|
||||
void GetVersion(GLint* majorvers, GLint* minorvers);
|
||||
const char* ErrorString(CGLError error);
|
||||
|
||||
CGLError AttachDrawable(CGLContextObj ctx, void* nsView, void* metalLayer, GLint width, GLint height);
|
||||
void* GetContextNSObject(CGLContextObj ctx);
|
||||
void SetContextNSObject(CGLContextObj ctx, void* nsObject);
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,118 @@
|
||||
// MobileGL - MobileGL/MG_Impl/CGLImpl/Exporting/Definitions.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "../CGLImpl.h"
|
||||
|
||||
#if defined(__APPLE__)
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLChoosePixelFormat(const CGLPixelFormatAttribute* attribs,
|
||||
CGLPixelFormatObj* pix,
|
||||
GLint* npix) {
|
||||
return MobileGL::MG_Impl::CGLImpl::ChoosePixelFormat(attribs, pix, npix);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLDestroyPixelFormat(CGLPixelFormatObj pix) {
|
||||
return MobileGL::MG_Impl::CGLImpl::DestroyPixelFormat(pix);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLDescribePixelFormat(CGLPixelFormatObj pix,
|
||||
GLint pix_num,
|
||||
CGLPixelFormatAttribute attrib,
|
||||
GLint* value) {
|
||||
return MobileGL::MG_Impl::CGLImpl::DescribePixelFormat(pix, pix_num, attrib, value);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API void CGLReleasePixelFormat(CGLPixelFormatObj pix) {
|
||||
MobileGL::MG_Impl::CGLImpl::ReleasePixelFormat(pix);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLPixelFormatObj CGLRetainPixelFormat(CGLPixelFormatObj pix) {
|
||||
return MobileGL::MG_Impl::CGLImpl::RetainPixelFormat(pix);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API GLuint CGLGetPixelFormatRetainCount(CGLPixelFormatObj pix) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetPixelFormatRetainCount(pix);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLCreateContext(CGLPixelFormatObj pix, CGLContextObj share, CGLContextObj* ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::CreateContext(pix, share, ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLDestroyContext(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::DestroyContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLContextObj CGLRetainContext(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::RetainContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API void CGLReleaseContext(CGLContextObj ctx) {
|
||||
MobileGL::MG_Impl::CGLImpl::ReleaseContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API GLuint CGLGetContextRetainCount(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetContextRetainCount(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLPixelFormatObj CGLGetPixelFormat(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetPixelFormat(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetCurrentContext(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetCurrentContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLContextObj CGLGetCurrentContext(void) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetCurrentContext();
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetVirtualScreen(CGLContextObj ctx, GLint screen) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetVirtualScreen(ctx, screen);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLGetVirtualScreen(CGLContextObj ctx, GLint* screen) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetVirtualScreen(ctx, screen);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLSetParameter(CGLContextObj ctx, CGLContextParameter pname, const GLint* params) {
|
||||
return MobileGL::MG_Impl::CGLImpl::SetParameter(ctx, pname, params);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLGetParameter(CGLContextObj ctx, CGLContextParameter pname, GLint* params) {
|
||||
return MobileGL::MG_Impl::CGLImpl::GetParameter(ctx, pname, params);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLUpdateContext(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::UpdateContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLClearDrawable(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::ClearDrawable(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLFlushDrawable(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::FlushDrawable(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLLockContext(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::LockContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API CGLError CGLUnlockContext(CGLContextObj ctx) {
|
||||
return MobileGL::MG_Impl::CGLImpl::UnlockContext(ctx);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API void CGLGetVersion(GLint* majorvers, GLint* minorvers) {
|
||||
MobileGL::MG_Impl::CGLImpl::GetVersion(majorvers, minorvers);
|
||||
}
|
||||
|
||||
MOBILEGL_CGL_API const char* CGLErrorString(CGLError error) {
|
||||
return MobileGL::MG_Impl::CGLImpl::ErrorString(error);
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,102 @@
|
||||
// MobileGL - MobileGL/MG_Impl/DyldInterpose/DyldInterpose.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
#if defined(__APPLE__)
|
||||
|
||||
#include "MG_Impl/CGLImpl/CGLImpl.h"
|
||||
#include "MG_Impl/GetProcAddress.h"
|
||||
|
||||
#include <CoreGraphics/CoreGraphics.h>
|
||||
#include <CoreVideo/CVDisplayLink.h>
|
||||
#include <cstdint>
|
||||
#include <dlfcn.h>
|
||||
|
||||
namespace {
|
||||
struct DyldInterposeEntry {
|
||||
const void* Replacement;
|
||||
const void* Replacee;
|
||||
};
|
||||
|
||||
bool IsGLProcName(const char* name) {
|
||||
if (name == nullptr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (strncmp(name, "CGL", 3) == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (strncmp(name, "gl", 2) != 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Avoid stealing glfw*/glib*/glX*/global application symbols.
|
||||
return name[2] >= 'A' && name[2] <= 'Z' && name[2] != 'X';
|
||||
}
|
||||
|
||||
void* MobileGLDlsym(void* handle, const char* symbol) {
|
||||
if (IsGLProcName(symbol)) {
|
||||
if (void* proc = MobileGL::MG_Impl::GetProcAddress(symbol)) {
|
||||
return proc;
|
||||
}
|
||||
}
|
||||
|
||||
return dlsym(handle, symbol);
|
||||
}
|
||||
|
||||
CGDirectDisplayID DisplayForMask(GLint displayMask) {
|
||||
constexpr std::uint32_t MaxDisplays = sizeof(CGOpenGLDisplayMask) * 8;
|
||||
CGDirectDisplayID displays[MaxDisplays] = {};
|
||||
std::uint32_t displayCount = 0;
|
||||
if (displayMask != 0 &&
|
||||
CGGetActiveDisplayList(MaxDisplays, displays, &displayCount) == kCGErrorSuccess) {
|
||||
const auto mask = static_cast<CGOpenGLDisplayMask>(displayMask);
|
||||
for (std::uint32_t i = 0; i < displayCount; ++i) {
|
||||
if ((CGDisplayIDToOpenGLDisplayMask(displays[i]) & mask) != 0) {
|
||||
return displays[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
return CGMainDisplayID();
|
||||
}
|
||||
|
||||
#pragma clang diagnostic push
|
||||
#pragma clang diagnostic ignored "-Wdeprecated-declarations"
|
||||
CVReturn MobileGLCVDisplayLinkSetCurrentCGDisplayFromOpenGLContext(
|
||||
CVDisplayLinkRef displayLink,
|
||||
CGLContextObj context,
|
||||
CGLPixelFormatObj pixelFormat) {
|
||||
GLint virtualScreen = 0;
|
||||
if (MobileGL::MG_Impl::CGLImpl::GetVirtualScreen(context, &virtualScreen) == kCGLNoError) {
|
||||
GLint displayMask = 0;
|
||||
if (!displayLink ||
|
||||
MobileGL::MG_Impl::CGLImpl::DescribePixelFormat(
|
||||
pixelFormat, virtualScreen, kCGLPFADisplayMask, &displayMask) != kCGLNoError) {
|
||||
return kCVReturnInvalidArgument;
|
||||
}
|
||||
return CVDisplayLinkSetCurrentCGDisplay(displayLink, DisplayForMask(displayMask));
|
||||
}
|
||||
|
||||
using OriginalFunction = CVReturn (*)(CVDisplayLinkRef, CGLContextObj, CGLPixelFormatObj);
|
||||
static const auto original = reinterpret_cast<OriginalFunction>(
|
||||
dlsym(RTLD_NEXT, "CVDisplayLinkSetCurrentCGDisplayFromOpenGLContext"));
|
||||
return original ? original(displayLink, context, pixelFormat) : kCVReturnError;
|
||||
}
|
||||
|
||||
__attribute__((used)) static const DyldInterposeEntry kMobileGLDyldInterpose[]
|
||||
__attribute__((section("__DATA,__interpose"))) = {
|
||||
{reinterpret_cast<const void*>(MobileGLDlsym), reinterpret_cast<const void*>(dlsym)},
|
||||
{reinterpret_cast<const void*>(MobileGLCVDisplayLinkSetCurrentCGDisplayFromOpenGLContext),
|
||||
reinterpret_cast<const void*>(CVDisplayLinkSetCurrentCGDisplayFromOpenGLContext)},
|
||||
};
|
||||
#pragma clang diagnostic pop
|
||||
} // namespace
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,10 @@
|
||||
# Public CGL entry points.
|
||||
_CGL*
|
||||
|
||||
# Public EGL entry points.
|
||||
_egl*
|
||||
|
||||
# Public OpenGL and GLX entry points. OpenGL function names always use an
|
||||
# uppercase letter or digit after the "gl" prefix; excluding lowercase here
|
||||
# deliberately prevents glslang_* from matching this pattern.
|
||||
_gl[A-Z0-9]*
|
||||
@@ -8,8 +8,11 @@
|
||||
|
||||
#include "EGLImpl.h"
|
||||
#include "../GetProcAddress.h"
|
||||
#include <Init.h>
|
||||
#include <MG_Backend/BackendObjects.h>
|
||||
#include <MG_State/EGLState/Core.h>
|
||||
#include <mutex>
|
||||
#include <sstream>
|
||||
#include <type_traits>
|
||||
|
||||
namespace MobileGL::MG_Impl::EGLImpl {
|
||||
@@ -18,9 +21,20 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
EGLStateContext* GetState() {
|
||||
if (!MG_State::pEGLContext) {
|
||||
MGLOG_E("pEGLContext is null. MG_State may not be initialized.");
|
||||
MGLOG_E_ONCE("pEGLContext is null. MG_State may not be initialized.");
|
||||
}
|
||||
return MG_State::pEGLContext;
|
||||
return MG_State::pEGLContext.get();
|
||||
}
|
||||
|
||||
// Entry points that can legitimately be an application's FIRST EGL
|
||||
// call (display/proc-address/string queries) lazily bring MobileGL
|
||||
// up here, so the library needs no static constructor and can
|
||||
// re-initialize after the last eglTerminate tore everything down.
|
||||
// Teardown-ish entry points keep using GetState() and fail benignly
|
||||
// when MobileGL is not initialized.
|
||||
EGLStateContext* GetStateEnsureInitialized() {
|
||||
MobileGL::EnsureInitialized();
|
||||
return GetState();
|
||||
}
|
||||
|
||||
MG_Backend::BackendObject* GetBackendObject(EGLStateContext* state) {
|
||||
@@ -31,14 +45,55 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return backendObject;
|
||||
}
|
||||
|
||||
std::recursive_mutex& EGLOperationMutex() {
|
||||
static std::recursive_mutex mutex;
|
||||
return mutex;
|
||||
}
|
||||
|
||||
String CurrentThreadIdString() {
|
||||
std::ostringstream stream;
|
||||
stream << std::this_thread::get_id();
|
||||
return stream.str();
|
||||
}
|
||||
|
||||
MG_Backend::WindowBackend DetectWindowBackend() {
|
||||
#if defined(ANDROID) || defined(__ANDROID__)
|
||||
return MG_Backend::WindowBackend::Android;
|
||||
#elif defined(__APPLE__)
|
||||
return MG_Backend::WindowBackend::MetalLayer;
|
||||
#elif defined(_WIN32)
|
||||
return MG_Backend::WindowBackend::Win32;
|
||||
#elif defined(__linux__)
|
||||
return MG_Backend::WindowBackend::X11;
|
||||
#else
|
||||
return MG_Backend::WindowBackend::Unknown;
|
||||
#endif
|
||||
}
|
||||
|
||||
EGLint GetAttribValue(const EGLint* attribList, EGLint attrib, EGLint defaultValue) {
|
||||
if (!attribList) {
|
||||
return defaultValue;
|
||||
}
|
||||
for (SizeT i = 0; attribList[i] != EGL_NONE; i += 2) {
|
||||
if (attribList[i] == attrib) {
|
||||
return attribList[i + 1];
|
||||
}
|
||||
}
|
||||
return defaultValue;
|
||||
}
|
||||
|
||||
EGLint GetAttribValueAttrib(const EGLAttrib* attribList, EGLint attrib, EGLint defaultValue) {
|
||||
if (!attribList) {
|
||||
return defaultValue;
|
||||
}
|
||||
for (SizeT i = 0; attribList[i] != EGL_NONE; i += 2) {
|
||||
if (attribList[i] == attrib) {
|
||||
return static_cast<EGLint>(attribList[i + 1]);
|
||||
}
|
||||
}
|
||||
return defaultValue;
|
||||
}
|
||||
|
||||
template <typename NativeType>
|
||||
Bool IsNullNativeHandle(NativeType nativeHandle) {
|
||||
if constexpr (std::is_pointer_v<NativeType>) {
|
||||
@@ -77,25 +132,35 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
const MG_Backend::WindowHandle windowHandle = {
|
||||
.Backend = DetectWindowBackend(),
|
||||
.Handle = ToVoidHandle(window),
|
||||
.Width = static_cast<Uint32>(std::max<EGLint>(GetAttribValue(attrib_list, EGL_WIDTH, 0), 0)),
|
||||
.Height = static_cast<Uint32>(std::max<EGLint>(GetAttribValue(attrib_list, EGL_HEIGHT, 0), 0)),
|
||||
};
|
||||
if (!backendObject->CreateEGLWindowSurface(windowHandle)) {
|
||||
|
||||
EGLSurface surface = state->CreateWindowSurface(dpy, config, window, attrib_list);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
state->DestroySurface(dpy, surface);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
if (!backendObject->CreateEGLWindowSurface(surface, windowHandle)) {
|
||||
state->DestroySurface(dpy, surface);
|
||||
state->SetError(EGL_BAD_NATIVE_WINDOW);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
return state->CreateWindowSurface(dpy, config, window, attrib_list);
|
||||
return surface;
|
||||
}
|
||||
|
||||
EGLBoolean SwapBuffers(EGLDisplay dpy, EGLSurface draw) {
|
||||
const std::lock_guard<std::recursive_mutex> operationLock(EGLOperationMutex());
|
||||
auto* state = GetState();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
@@ -107,10 +172,11 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!backendObject->SwapEGLBuffers(dpy, draw)) {
|
||||
MGLOG_E_ONCE("eglSwapBuffers failed on thread=%s dpy=%p draw=%p", CurrentThreadIdString().c_str(), dpy, draw);
|
||||
state->SetError(EGL_BAD_SURFACE);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -135,7 +201,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean Initialize(EGLDisplay dpy, EGLint* major, EGLint* minor) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -145,7 +211,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!backendObject->InitializeEGLDisplay(dpy, major, minor)) {
|
||||
@@ -156,7 +222,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLDisplay GetDisplay(NativeDisplayType display) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
@@ -172,6 +238,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean MakeCurrent(EGLDisplay dpy, EGLSurface draw, EGLSurface read, EGLContext ctx) {
|
||||
const std::lock_guard<std::recursive_mutex> operationLock(EGLOperationMutex());
|
||||
auto* state = GetState();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
@@ -181,31 +248,48 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
const auto oldDraw = state->GetCurrentSurface(EGL_DRAW);
|
||||
const auto oldRead = state->GetCurrentSurface(EGL_READ);
|
||||
const auto oldContext = state->GetCurrentContext();
|
||||
const String threadId = CurrentThreadIdString();
|
||||
|
||||
MGLOG_D("eglMakeCurrent begin thread=%s dpy=%p draw=%p read=%p ctx=%p oldDpy=%p oldDraw=%p oldRead=%p oldCtx=%p",
|
||||
threadId.c_str(), dpy, draw, read, ctx, oldDisplay, oldDraw, oldRead, oldContext);
|
||||
|
||||
if (!state->MakeCurrent(dpy, draw, read, ctx)) {
|
||||
const EGLint error = state->ConsumeError();
|
||||
MGLOG_D("eglMakeCurrent rejected by EGLState thread=%s error=0x%04x", threadId.c_str(), error);
|
||||
state->SetError(error);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
|
||||
const Bool releaseCurrentRequest =
|
||||
dpy == EGL_NO_DISPLAY && draw == EGL_NO_SURFACE && read == EGL_NO_SURFACE && ctx == EGL_NO_CONTEXT;
|
||||
draw == EGL_NO_SURFACE && read == EGL_NO_SURFACE && ctx == EGL_NO_CONTEXT;
|
||||
if (releaseCurrentRequest) {
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
(void)backendObject->MakeEGLCurrent(dpy, draw, read, ctx);
|
||||
if (!backendObject->MakeEGLCurrent(dpy, draw, read, ctx)) {
|
||||
MGLOG_E_ONCE("eglMakeCurrent release failed in backend thread=%s", threadId.c_str());
|
||||
state->MakeCurrent(oldDisplay, oldDraw, oldRead, oldContext);
|
||||
state->SetError(EGL_BAD_ACCESS);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
}
|
||||
MGLOG_D("eglMakeCurrent release succeeded thread=%s", threadId.c_str());
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
state->MakeCurrent(oldDisplay, oldDraw, oldRead, oldContext);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!backendObject->MakeEGLCurrent(dpy, draw, read, ctx)) {
|
||||
MGLOG_E_ONCE("eglMakeCurrent backend attach failed thread=%s dpy=%p draw=%p read=%p ctx=%p", threadId.c_str(),
|
||||
dpy, draw, read, ctx);
|
||||
state->SetError(EGL_BAD_ACCESS);
|
||||
state->MakeCurrent(oldDisplay, oldDraw, oldRead, oldContext);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
MGLOG_D("eglMakeCurrent attach succeeded thread=%s dpy=%p draw=%p read=%p ctx=%p", threadId.c_str(), dpy, draw,
|
||||
read, ctx);
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
@@ -218,11 +302,18 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean DestroySurface(EGLDisplay dpy, EGLSurface surface) {
|
||||
const std::lock_guard<std::recursive_mutex> operationLock(EGLOperationMutex());
|
||||
auto* state = GetState();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
return state->DestroySurface(dpy, surface) ? EGL_TRUE : EGL_FALSE;
|
||||
if (!state->DestroySurface(dpy, surface)) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
backendObject->ReleaseEGLSurface(surface);
|
||||
}
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
EGLBoolean Terminate(EGLDisplay dpy) {
|
||||
@@ -230,7 +321,21 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
return state->TerminateDisplay(dpy) ? EGL_TRUE : EGL_FALSE;
|
||||
if (!state->TerminateDisplay(dpy)) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
backendObject->ReleaseEGLResources();
|
||||
}
|
||||
// The last initialized display is gone and nothing is current on any
|
||||
// thread: tear the whole library down deterministically inside the
|
||||
// EGL lifecycle (backend, GL/EGL state, glslang). A later EGL call
|
||||
// re-initializes lazily via GetStateEnsureInitialized(); process exit
|
||||
// then has nothing left to destroy.
|
||||
if (!state->HasAnyInitializedDisplay() && !state->HasAnyCurrentContext()) {
|
||||
MobileGL::Destroy();
|
||||
}
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
EGLBoolean ReleaseThread() {
|
||||
@@ -238,6 +343,9 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (auto* backendObject = MG_Backend::pActiveBackendObject.get()) {
|
||||
(void)backendObject->MakeEGLCurrent(EGL_NO_DISPLAY, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT);
|
||||
}
|
||||
state->ReleaseThread();
|
||||
return EGL_TRUE;
|
||||
}
|
||||
@@ -259,7 +367,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
EGLBoolean BindAPI(EGLenum api) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
@@ -292,7 +400,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
}
|
||||
|
||||
char const* QueryString(EGLDisplay display, EGLint name) {
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return nullptr;
|
||||
}
|
||||
@@ -310,7 +418,14 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
case EGL_CLIENT_APIS:
|
||||
return "OpenGL OpenGL_ES";
|
||||
case EGL_EXTENSIONS:
|
||||
return "";
|
||||
if (display == EGL_NO_DISPLAY) {
|
||||
return "EGL_EXT_client_extensions "
|
||||
"EGL_EXT_platform_base "
|
||||
"EGL_KHR_platform_base "
|
||||
"EGL_MESA_platform_surfaceless";
|
||||
}
|
||||
return "EGL_KHR_create_context "
|
||||
"EGL_MESA_platform_surfaceless";
|
||||
default:
|
||||
state->SetError(EGL_BAD_PARAMETER);
|
||||
return nullptr;
|
||||
@@ -322,7 +437,17 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
return state->SwapInterval(dpy, interval) ? EGL_TRUE : EGL_FALSE;
|
||||
if (!state->SwapInterval(dpy, interval)) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
// Forward the request to the backend's native presentation path; without this
|
||||
// the app's vsync setting only ever reaches MobileGL's shadow state and the
|
||||
// native surface stays at the driver default (interval 1 = always vsynced).
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (backendObject) {
|
||||
backendObject->SetEGLSwapInterval(static_cast<Int>(interval));
|
||||
}
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
EGLSurface CreatePbufferSurface(EGLDisplay dpy, EGLConfig config, const EGLint* attrib_list) {
|
||||
@@ -330,7 +455,24 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!state) {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
return state->CreatePbufferSurface(dpy, config, attrib_list);
|
||||
const EGLint width = GetAttribValue(attrib_list, EGL_WIDTH, 1);
|
||||
const EGLint height = GetAttribValue(attrib_list, EGL_HEIGHT, 1);
|
||||
EGLSurface surface = state->CreatePbufferSurface(dpy, config, attrib_list);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
if (!backendObject->CreateEGLPbufferSurface(surface, width, height)) {
|
||||
state->DestroySurface(dpy, surface);
|
||||
state->SetError(EGL_BAD_ALLOC);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
return surface;
|
||||
}
|
||||
|
||||
EGLBoolean BindTexImage(EGLDisplay dpy, EGLSurface surface, EGLint buffer) {
|
||||
@@ -521,7 +663,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
EGLDisplay GetPlatformDisplay(EGLenum platform, void* native_display, const EGLAttrib* attrib_list) {
|
||||
(void)attrib_list;
|
||||
|
||||
auto* state = GetState();
|
||||
auto* state = GetStateEnsureInitialized();
|
||||
if (!state) {
|
||||
return EGL_NO_DISPLAY;
|
||||
}
|
||||
@@ -547,22 +689,53 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E("activeBackendObject not initialized!");
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
const MG_Backend::WindowHandle windowHandle = {
|
||||
.Backend = DetectWindowBackend(),
|
||||
.Handle = native_window,
|
||||
.Width = static_cast<Uint32>(std::max<EGLint>(GetAttribValueAttrib(attrib_list, EGL_WIDTH, 0), 0)),
|
||||
.Height = static_cast<Uint32>(std::max<EGLint>(GetAttribValueAttrib(attrib_list, EGL_HEIGHT, 0), 0)),
|
||||
};
|
||||
if (!backendObject->CreateEGLWindowSurface(windowHandle)) {
|
||||
|
||||
EGLSurface surface = state->CreatePlatformWindowSurface(dpy, config, native_window, attrib_list);
|
||||
if (surface == EGL_NO_SURFACE) {
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
state->DestroySurface(dpy, surface);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
if (!backendObject->CreateEGLWindowSurface(surface, windowHandle)) {
|
||||
state->DestroySurface(dpy, surface);
|
||||
state->SetError(EGL_BAD_NATIVE_WINDOW);
|
||||
return EGL_NO_SURFACE;
|
||||
}
|
||||
|
||||
return state->CreatePlatformWindowSurface(dpy, config, native_window, attrib_list);
|
||||
return surface;
|
||||
}
|
||||
|
||||
EGLBoolean ResizePlatformWindowSurface(EGLDisplay dpy, EGLSurface surface, EGLint width, EGLint height) {
|
||||
auto* state = GetState();
|
||||
if (!state) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
if (!state->ResizeSurface(dpy, surface, width, height)) {
|
||||
return EGL_FALSE;
|
||||
}
|
||||
auto* backendObject = GetBackendObject(state);
|
||||
if (!backendObject) {
|
||||
MGLOG_E_ONCE("activeBackendObject not initialized!");
|
||||
return EGL_FALSE;
|
||||
}
|
||||
width = std::max<EGLint>(width, 1);
|
||||
height = std::max<EGLint>(height, 1);
|
||||
if (!backendObject->ResizeEGLWindowSurface(surface, static_cast<Uint32>(width), static_cast<Uint32>(height))) {
|
||||
state->SetError(EGL_BAD_NATIVE_WINDOW);
|
||||
return EGL_FALSE;
|
||||
}
|
||||
return EGL_TRUE;
|
||||
}
|
||||
|
||||
EGLSurface CreatePlatformPixmapSurface(EGLDisplay dpy, EGLConfig config, void* native_pixmap,
|
||||
@@ -586,11 +759,12 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
if (!name) {
|
||||
return nullptr;
|
||||
}
|
||||
MobileGL::EnsureInitialized();
|
||||
|
||||
MGLOG_D("eglGetProcAddress(%s)", name);
|
||||
void* proc = MG_Impl::GetProcAddress(name);
|
||||
if (!proc) {
|
||||
MGLOG_W("Failed to get function: %s", name);
|
||||
MGLOG_D("Failed to get function: %s", name);
|
||||
return nullptr;
|
||||
}
|
||||
return (__eglMustCastToProperFunctionPointerType)proc;
|
||||
|
||||
@@ -57,6 +57,7 @@ namespace MobileGL::MG_Impl::EGLImpl {
|
||||
EGLDisplay GetPlatformDisplay(EGLenum platform, void* native_display, const EGLAttrib* attrib_list);
|
||||
EGLSurface CreatePlatformWindowSurface(EGLDisplay dpy, EGLConfig config, void* native_window,
|
||||
const EGLAttrib* attrib_list);
|
||||
EGLBoolean ResizePlatformWindowSurface(EGLDisplay dpy, EGLSurface surface, EGLint width, EGLint height);
|
||||
EGLSurface CreatePlatformPixmapSurface(EGLDisplay dpy, EGLConfig config, void* native_pixmap,
|
||||
const EGLAttrib* attrib_list);
|
||||
EGLBoolean WaitSync(EGLDisplay dpy, EGLSync sync, EGLint flags);
|
||||
|
||||
@@ -235,6 +235,14 @@ MOBILEGL_EGL_API EGLDisplay eglGetPlatformDisplay(EGLenum platform, void* native
|
||||
return MobileGL::MG_Impl::EGLImpl::GetPlatformDisplay(platform, native_display, attrib_list);
|
||||
}
|
||||
|
||||
MOBILEGL_EGL_API EGLDisplay eglGetPlatformDisplayEXT(EGLenum platform, void* native_display,
|
||||
const EGLint* attrib_list) {
|
||||
MGLOG_D("eglGetPlatformDisplayEXT(platform=%u, native_display=%p, attrib_list=%p)", platform, native_display,
|
||||
attrib_list);
|
||||
return MobileGL::MG_Impl::EGLImpl::GetPlatformDisplay(
|
||||
platform, native_display, reinterpret_cast<const EGLAttrib*>(attrib_list));
|
||||
}
|
||||
|
||||
MOBILEGL_EGL_API EGLSurface eglCreatePlatformWindowSurface(EGLDisplay dpy, EGLConfig config, void* native_window,
|
||||
const EGLAttrib* attrib_list) {
|
||||
MGLOG_D("eglCreatePlatformWindowSurface(dpy=%p, config=%p, native_window=%p, attrib_list=%p)", dpy, config,
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user