mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-08 04:08:32 +09:00
Merge branch 'feat/cts-draw-parameters' into dev
This commit is contained in:
@@ -279,6 +279,7 @@ set(SOURCE_FILES
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LowerDrawParametersPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PackDoubleVertexInputsPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/RebaseInstanceIndexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/NormalizeRectCoordinatesPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/PrivateToEntryLocalPass.cpp
|
||||
MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/StripUniformLocationsPass.cpp
|
||||
|
||||
@@ -1002,6 +1002,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
funcsTable.GL.MultiDrawElementsIndirect = MultiDrawElementsIndirect;
|
||||
funcsTable.GL.MultiDrawElementsIndirectCount = MultiDrawElementsIndirectCount;
|
||||
funcsTable.GL.MultiDrawArraysIndirect = MultiDrawArraysIndirect;
|
||||
funcsTable.GL.MultiDrawArraysIndirectCount = MultiDrawArraysIndirectCount;
|
||||
funcsTable.GL.DrawRangeElementsBaseVertex = DrawRangeElementsBaseVertex;
|
||||
funcsTable.GL.DrawRangeElements = DrawRangeElements;
|
||||
funcsTable.GL.DrawElementsInstancedBaseVertexBaseInstance = DrawElementsInstancedBaseVertexBaseInstance;
|
||||
|
||||
@@ -2905,11 +2905,40 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
}
|
||||
|
||||
void SetCurrentBaseVertex(Int32 baseVertex) {
|
||||
if (const auto program = GetCurrentBackendProgram()) {
|
||||
program->SetBaseVertex(baseVertex);
|
||||
}
|
||||
}
|
||||
|
||||
Bool CurrentProgramReadsDrawID() {
|
||||
const auto program = GetCurrentBackendProgram();
|
||||
return program != nullptr && program->ReadsDrawID();
|
||||
}
|
||||
|
||||
Bool CurrentProgramReadsBaseVertex() {
|
||||
const auto program = GetCurrentBackendProgram();
|
||||
return program != nullptr && program->ReadsBaseVertex();
|
||||
}
|
||||
|
||||
// The two questions above, asked from BEFORE PrepareForDraw - where neither can be
|
||||
// answered honestly. GetCurrentBackendProgram only sees a twin that a previous draw
|
||||
// already synced, and a twin from before a relink still carries the previous link's
|
||||
// uniform locations, so "no" there means "not known yet" at least as often as it
|
||||
// means no. The multi-draw compute tier has to decide whether to flatten a batch
|
||||
// before PrepareForDraw runs (its dispatch cannot come after the draw state), and
|
||||
// flattening a batch that turns out to need per-sub-draw values is unrecoverable -
|
||||
// so an unanswerable program counts as needing them.
|
||||
Bool CurrentProgramMayNeedPerSubDrawBuiltins(Bool batchCarriesBaseVertices) {
|
||||
const auto& currentProgram = MG_State::pGLContext->GetProgramForDraw();
|
||||
const auto program = GetCurrentBackendProgram();
|
||||
if (!currentProgram || program == nullptr ||
|
||||
program->GetSyncedLinkVersion() != currentProgram->GetLinkVersion()) {
|
||||
return true;
|
||||
}
|
||||
return program->ReadsDrawID() || (batchCarriesBaseVertices && program->ReadsBaseVertex());
|
||||
}
|
||||
|
||||
static Bool SupportsNativeIndirectDraws() {
|
||||
const auto& version = g_GLESCapabilities.GLESVersion;
|
||||
const Bool esVersionOk = version.Major > 3 || (version.Major == 3 && version.Minor >= 1);
|
||||
@@ -2947,16 +2976,28 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
resource->id);
|
||||
}
|
||||
}
|
||||
// gl_BaseVertex has no SSBO view of its own: the command's baseVertex word is read
|
||||
// from the CPU shadow, so a command whose baseVertex a compute shader wrote this
|
||||
// frame is not observable here (baseInstance is, through the view above). Feeding
|
||||
// the stale-but-usually-correct shadow beats leaving the uniform at the previous
|
||||
// draw's value, which is what a program reading gl_BaseVertex saw before.
|
||||
const Bool feedBaseVertex = CurrentProgramReadsBaseVertex();
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
const SizeT cmdByteOffset = commandOffset + static_cast<SizeT>(i) * stride;
|
||||
SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
if (paramsBinding >= 0 && backendProgram) {
|
||||
// baseInstance is the 5th word of DrawElementsIndirectCommand.
|
||||
backendProgram->SetBaseInstanceWordIndex(static_cast<Int32>((cmdByteOffset + 16) / 4));
|
||||
if (feedBaseVertex) {
|
||||
DrawElementsIndirectCommand cmd{};
|
||||
std::memcpy(&cmd, commandBytes + static_cast<SizeT>(i) * stride, sizeof(cmd));
|
||||
SetCurrentBaseVertex(cmd.baseVertex);
|
||||
}
|
||||
} else {
|
||||
DrawElementsIndirectCommand cmd{};
|
||||
std::memcpy(&cmd, commandBytes + static_cast<SizeT>(i) * stride, sizeof(cmd));
|
||||
SetCurrentBaseInstance(cmd.baseInstance);
|
||||
SetCurrentBaseVertex(cmd.baseVertex);
|
||||
}
|
||||
g_GLESFuncs.glDrawElementsIndirect(mode, type, reinterpret_cast<const void*>(cmdByteOffset));
|
||||
}
|
||||
@@ -2969,6 +3010,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
SetCurrentBaseInstance(cmd.baseInstance);
|
||||
SetCurrentBaseVertex(cmd.baseVertex);
|
||||
const auto indexByteOffset = static_cast<SizeT>(cmd.firstIndex) * indexSize;
|
||||
g_GLESFuncs.glDrawElementsInstancedBaseVertex(
|
||||
mode, static_cast<GLsizei>(cmd.count), type, reinterpret_cast<const GLvoid*>(indexByteOffset),
|
||||
@@ -2977,12 +3019,18 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
}
|
||||
SetCurrentDrawID(0);
|
||||
SetCurrentBaseInstance(0);
|
||||
SetCurrentBaseVertex(0);
|
||||
}
|
||||
|
||||
static void ExecuteArraysIndirectCommands(GLenum mode, const Uint8* commandBytes, SizeT commandOffset,
|
||||
const SharedPtr<MG_State::GLState::BufferObject>& drawIndirectBuffer,
|
||||
GLsizei drawcount, GLsizei stride, const char* label) {
|
||||
(void)label;
|
||||
// DrawArraysIndirectCommand has no baseVertex word, so gl_BaseVertex is zero for every
|
||||
// command here. Written BEFORE the draws, not merely restored after them: the previous
|
||||
// draw is what leaves a stale value, and restoring afterwards would only protect the
|
||||
// NEXT draw while these commands ran with the stale one.
|
||||
SetCurrentBaseVertex(0);
|
||||
const Bool useNative = drawIndirectBuffer != nullptr && SupportsNativeIndirectDraws();
|
||||
if (useNative) {
|
||||
const auto backendProgram = GetCurrentBackendProgram();
|
||||
@@ -3269,7 +3317,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
DrawSyncFlags syncBit = DrawSyncBit::IndexBuffer;
|
||||
PrepareForDraw(syncBit);
|
||||
CheckPrimitiveRestartSupported(type);
|
||||
SetCurrentBaseVertex(basevertex);
|
||||
g_GLESFuncs.glDrawElementsBaseVertex(mode, count, type, indices, basevertex);
|
||||
SetCurrentBaseVertex(0);
|
||||
}
|
||||
|
||||
void MultiDrawArrays(GLenum mode, const GLint* first, const GLsizei* count, GLsizei drawcount) {
|
||||
@@ -3279,6 +3329,11 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
DrawSyncFlags syncBit = DrawSyncBit::None;
|
||||
PrepareForDraw(syncBit);
|
||||
|
||||
// This loop IS the emulation - there is no batched tier for the non-indexed form -
|
||||
// so each sub-draw has to be given its own gl_DrawID here, exactly as the indexed
|
||||
// ladder and the indirect executors do. Without it every sub-draw of a
|
||||
// glMultiDrawArrays read draw index 0.
|
||||
const Bool feedDrawID = CurrentProgramReadsDrawID();
|
||||
const auto& currentVAO = MG_State::pGLContext->GetBoundVertexArray();
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
// Client-side arrays are uploaded per sub-draw range, like the single DrawArrays path.
|
||||
@@ -3288,8 +3343,10 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
(*backendVAOSlot)->SyncClientSideAttributesForDrawArrays(currentVAO, first[i], count[i]);
|
||||
}
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
g_GLESFuncs.glDrawArrays(mode, first[i], count[i]);
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
}
|
||||
|
||||
// Both glMultiDrawElements entry points are emulated - ES has neither in core - by the
|
||||
@@ -3403,6 +3460,15 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
return;
|
||||
}
|
||||
|
||||
// Both counts are read from the CPU shadow, which a buffer with no shadow does not
|
||||
// have - MappedData() is null there and the reads below would be a null dereference,
|
||||
// not a wrong picture. The DirectVulkan twin declines the same way.
|
||||
if (parameterBuffer->MappedData() == nullptr || drawBuffer->MappedData() == nullptr) {
|
||||
MGLOG_E("MultiDrawElementsIndirectCount skipped: CPU fallback cannot read the parameter or "
|
||||
"draw-indirect buffer");
|
||||
return;
|
||||
}
|
||||
|
||||
Uint32 actualDrawCount = 0;
|
||||
std::memcpy(&actualDrawCount, parameterBuffer->MappedData() + drawcount, sizeof(actualDrawCount));
|
||||
actualDrawCount = std::min<Uint32>(actualDrawCount, static_cast<Uint32>(maxdrawcount));
|
||||
@@ -3444,11 +3510,79 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
drawcount, stride, "MultiDrawArraysIndirect");
|
||||
}
|
||||
|
||||
// The non-indexed twin of MultiDrawElementsIndirectCount, and structurally identical to it:
|
||||
// ES has no GL_PARAMETER_BUFFER at all, so the draw count is read from the CPU shadow of the
|
||||
// bound one and the batch degenerates into an ordinary indirect multi-draw of that many
|
||||
// commands. Missing from the backend table until now, which made every
|
||||
// glMultiDrawArraysIndirectCount an INVALID_OPERATION ("backend does not support
|
||||
// indirect-parameter array draws") on DirectGLES while the extension was advertised.
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount, GLsizei maxdrawcount,
|
||||
GLsizei stride) {
|
||||
#if MOBILEGL_LOG_ACTIVE_LEVEL <= MOBILEGL_LOG_LEVEL_DEBUG && MOBILEGL_ENABLE_SCOPE_MARKER
|
||||
DebugImpl::OpenGLScopeMarker marker(__func__);
|
||||
#endif
|
||||
if (maxdrawcount <= 0) {
|
||||
return;
|
||||
}
|
||||
if (stride == 0) {
|
||||
stride = sizeof(DrawArraysIndirectCommand);
|
||||
}
|
||||
if (stride < static_cast<GLsizei>(sizeof(DrawArraysIndirectCommand))) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: stride %d is smaller than command size %zu",
|
||||
stride, sizeof(DrawArraysIndirectCommand));
|
||||
return;
|
||||
}
|
||||
|
||||
DrawSyncFlags syncBit = DrawSyncBit::IndirectBuffer | DrawSyncBit::Instancing;
|
||||
PrepareForDraw(syncBit);
|
||||
|
||||
auto drawBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::DrawIndirect).GetBoundObject();
|
||||
auto parameterBuffer = MG_State::pGLContext->GetBufferBindingSlot(BufferTarget::Parameter).GetBoundObject();
|
||||
if (!drawBuffer) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: no GL_DRAW_INDIRECT_BUFFER is bound");
|
||||
return;
|
||||
}
|
||||
if (!parameterBuffer) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: no GL_PARAMETER_BUFFER is bound");
|
||||
return;
|
||||
}
|
||||
|
||||
drawBuffer->SyncPersistentMappedRange();
|
||||
parameterBuffer->SyncPersistentMappedRange();
|
||||
|
||||
const SizeT commandOffset = reinterpret_cast<SizeT>(indirect);
|
||||
const SizeT commandBytes = commandOffset + static_cast<SizeT>(stride) * static_cast<SizeT>(maxdrawcount - 1) +
|
||||
sizeof(DrawArraysIndirectCommand);
|
||||
if (commandBytes > drawBuffer->GetSize()) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: invalid GL_DRAW_INDIRECT_BUFFER binding or range");
|
||||
return;
|
||||
}
|
||||
if (drawcount < 0 || static_cast<SizeT>(drawcount) + sizeof(Uint32) > parameterBuffer->GetSize()) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: invalid GL_PARAMETER_BUFFER binding or range");
|
||||
return;
|
||||
}
|
||||
|
||||
// See the indexed twin: no CPU shadow means no count to read, not a wrong one.
|
||||
if (parameterBuffer->MappedData() == nullptr || drawBuffer->MappedData() == nullptr) {
|
||||
MGLOG_E("MultiDrawArraysIndirectCount skipped: CPU fallback cannot read the parameter or "
|
||||
"draw-indirect buffer");
|
||||
return;
|
||||
}
|
||||
|
||||
Uint32 actualDrawCount = 0;
|
||||
std::memcpy(&actualDrawCount, parameterBuffer->MappedData() + drawcount, sizeof(actualDrawCount));
|
||||
actualDrawCount = std::min<Uint32>(actualDrawCount, static_cast<Uint32>(maxdrawcount));
|
||||
ExecuteArraysIndirectCommands(mode, drawBuffer->MappedData() + commandOffset, commandOffset, drawBuffer,
|
||||
static_cast<GLsizei>(actualDrawCount), stride, "MultiDrawArraysIndirectCount");
|
||||
}
|
||||
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex) {
|
||||
DrawSyncFlags syncBit = DrawSyncBit::IndexBuffer;
|
||||
PrepareForDraw(syncBit);
|
||||
SetCurrentBaseVertex(basevertex);
|
||||
g_GLESFuncs.glDrawRangeElementsBaseVertex(mode, start, end, count, type, indices, basevertex);
|
||||
SetCurrentBaseVertex(0);
|
||||
}
|
||||
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices) {
|
||||
@@ -3462,7 +3596,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
DrawSyncFlags syncBit = DrawSyncBit::IndexBuffer | DrawSyncBit::Instancing;
|
||||
PrepareForDraw(syncBit);
|
||||
SetCurrentBaseInstance(baseinstance);
|
||||
SetCurrentBaseVertex(basevertex);
|
||||
g_GLESFuncs.glDrawElementsInstancedBaseVertex(mode, count, type, indices, instancecount, basevertex);
|
||||
SetCurrentBaseVertex(0);
|
||||
SetCurrentBaseInstance(0);
|
||||
}
|
||||
|
||||
@@ -3470,7 +3606,9 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
GLsizei instancecount, GLint basevertex) {
|
||||
DrawSyncFlags syncBit = DrawSyncBit::IndexBuffer | DrawSyncBit::Instancing;
|
||||
PrepareForDraw(syncBit);
|
||||
SetCurrentBaseVertex(basevertex);
|
||||
g_GLESFuncs.glDrawElementsInstancedBaseVertex(mode, count, type, indices, instancecount, basevertex);
|
||||
SetCurrentBaseVertex(0);
|
||||
}
|
||||
|
||||
void DrawElementsInstancedBaseInstance(GLenum mode, GLsizei count, GLenum type, const void* indices,
|
||||
|
||||
@@ -40,6 +40,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void MultiDrawElementsIndirectCount(GLenum mode, GLenum type, const void* indirect, GLintptr drawcount,
|
||||
GLsizei maxdrawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirect(GLenum mode, const void* indirect, GLsizei drawcount, GLsizei stride);
|
||||
void MultiDrawArraysIndirectCount(GLenum mode, const void* indirect, GLintptr drawcount, GLsizei maxdrawcount,
|
||||
GLsizei stride);
|
||||
void DrawRangeElementsBaseVertex(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type,
|
||||
const void* indices, GLint basevertex);
|
||||
void DrawRangeElements(GLenum mode, GLuint start, GLuint end, GLsizei count, GLenum type, const void* indices);
|
||||
|
||||
@@ -155,6 +155,14 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// (possibly GPU-written) indirect command buffer, so its declaration expands into a
|
||||
// std430 SSBO view of that buffer indexed by a CPU-computed word index, with the plain
|
||||
// mg_BaseInstance uniform as the fallback for non-indirect draws.
|
||||
//
|
||||
// The word index is stored ONE-BASED, so that zero - the value every GLSL uniform starts
|
||||
// at - is the "not an indirect draw" sentinel. Nothing seeds this uniform before a
|
||||
// program's first draw, and the non-indirect draw entry points never write it at all, so a
|
||||
// zero-based index with a negative sentinel would leave every such draw reading
|
||||
// mg_indirectWords[0] out of a storage buffer no one bound. That is not a silent zero on a
|
||||
// real driver: it returned garbage on Adreno, and a garbage gl_BaseInstance pushed the CTS
|
||||
// shader_draw_parameters geometry clean off screen.
|
||||
String PromoteDrawParameterGlobalsToUniforms(String source, GLenum shaderType) {
|
||||
if (shaderType != GL_VERTEX_SHADER) {
|
||||
return source;
|
||||
@@ -208,12 +216,12 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
" { highp uint mg_indirectWords[]; };\n";
|
||||
if (rebaseInstanceId) {
|
||||
machinery += String("#define ") + ZERO_BASED_INSTANCE_ID_NAME + " (gl_InstanceID - ((" +
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + " >= 0) ? int(mg_indirectWords[uint(" +
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + ")]) : 0))\n";
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + " > 0) ? int(mg_indirectWords[uint(" +
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + " - 1)]) : 0))\n";
|
||||
}
|
||||
machinery += String("#define ") + BASE_INSTANCE_LOWERED_NAME + " ((" +
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + " >= 0) ? int(mg_indirectWords[uint(" +
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + ")]) : " + BASE_INSTANCE_UNIFORM_NAME + ")";
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + " > 0) ? int(mg_indirectWords[uint(" +
|
||||
BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME + " - 1)]) : " + BASE_INSTANCE_UNIFORM_NAME + ")";
|
||||
source.replace(pos, declaration.size(), machinery);
|
||||
break;
|
||||
}
|
||||
@@ -4594,6 +4602,8 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
m_baseInstanceUniformLocation = g_GLESFuncs.glGetUniformLocation(m_backendProgramId,
|
||||
BASE_INSTANCE_UNIFORM_NAME);
|
||||
m_drawIdUniformLocation = g_GLESFuncs.glGetUniformLocation(m_backendProgramId, DRAW_ID_UNIFORM_NAME);
|
||||
m_baseVertexUniformLocation = g_GLESFuncs.glGetUniformLocation(m_backendProgramId,
|
||||
BASE_VERTEX_UNIFORM_NAME);
|
||||
m_baseInstanceWordIndexUniformLocation =
|
||||
g_GLESFuncs.glGetUniformLocation(m_backendProgramId, BASE_INSTANCE_WORD_INDEX_UNIFORM_NAME);
|
||||
// The mg_IndirectParams block binding is baked into the ESSL (ES cannot rebind
|
||||
@@ -4772,14 +4782,21 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
g_GLESFuncs.glUniform1i(m_baseInstanceUniformLocation, static_cast<GLint>(baseInstance));
|
||||
}
|
||||
// A direct value disables the indirect-command-buffer read.
|
||||
SetBaseInstanceWordIndex(-1);
|
||||
}
|
||||
|
||||
// The uniform is written one-based so that its GLSL initial value, zero, already reads
|
||||
// as "no indirect command" - see PromoteDrawParameterGlobalsToUniforms.
|
||||
void BackendProgramObjectImpl::SetBaseInstanceWordIndex(Int32 wordIndex) const {
|
||||
if (m_baseInstanceWordIndexUniformLocation >= 0) {
|
||||
g_GLESFuncs.glUniform1i(m_baseInstanceWordIndexUniformLocation, -1);
|
||||
g_GLESFuncs.glUniform1i(m_baseInstanceWordIndexUniformLocation,
|
||||
wordIndex < 0 ? 0 : wordIndex + 1);
|
||||
}
|
||||
}
|
||||
|
||||
void BackendProgramObjectImpl::SetBaseInstanceWordIndex(Int32 wordIndex) const {
|
||||
if (m_baseInstanceWordIndexUniformLocation >= 0) {
|
||||
g_GLESFuncs.glUniform1i(m_baseInstanceWordIndexUniformLocation, wordIndex);
|
||||
void BackendProgramObjectImpl::SetBaseVertex(Int32 baseVertex) const {
|
||||
if (m_baseVertexUniformLocation >= 0) {
|
||||
g_GLESFuncs.glUniform1i(m_baseVertexUniformLocation, baseVertex);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -82,14 +82,26 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
// GLES core supports only GL_PRIMITIVE_RESTART_FIXED_INDEX. Throws when the app enabled
|
||||
// the arbitrary GL_PRIMITIVE_RESTART with a non-fixed index for this index type.
|
||||
void CheckPrimitiveRestartSupported(GLenum indexType);
|
||||
// Feed the current program's gl_BaseInstance / gl_DrawID emulation uniforms. Both are
|
||||
// no-ops when the program does not read the corresponding builtin.
|
||||
// Feed the current program's gl_BaseInstance / gl_DrawID / gl_BaseVertex emulation
|
||||
// uniforms. All are no-ops when the program does not read the corresponding builtin.
|
||||
void SetCurrentBaseInstance(Uint32 baseInstance);
|
||||
void SetCurrentDrawID(Uint32 drawId);
|
||||
// GL's gl_BaseVertex is the base-vertex parameter of an indexed draw and zero for every
|
||||
// command that has none - including all the DrawArrays forms - so every draw path that
|
||||
// does not carry one must leave this at zero rather than inherit the last draw's value.
|
||||
void SetCurrentBaseVertex(Int32 baseVertex);
|
||||
// True when the current program actually reads gl_DrawID, i.e. when a batched
|
||||
// (single driver call) multi-draw tier would have to feed it one value for the whole
|
||||
// batch and would therefore be wrong.
|
||||
Bool CurrentProgramReadsDrawID();
|
||||
// Same question for gl_BaseVertex: a batched multi-draw tier cannot give each sub-draw
|
||||
// its own base vertex through a uniform either.
|
||||
Bool CurrentProgramReadsBaseVertex();
|
||||
// Both of the above, conservatively, for a caller that must decide BEFORE PrepareForDraw
|
||||
// has synced the program - where "does not read it" is indistinguishable from "cannot be
|
||||
// asked yet". Answers true whenever the backend twin is missing or predates the current
|
||||
// link.
|
||||
Bool CurrentProgramMayNeedPerSubDrawBuiltins(Bool batchCarriesBaseVertices);
|
||||
|
||||
template <typename StateObject, typename BackendObject>
|
||||
class StateBackendObjectRegistry {
|
||||
@@ -1025,9 +1037,13 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
void SetBaseInstance(Uint32 baseInstance) const;
|
||||
void SetBaseInstanceWordIndex(Int32 wordIndex) const;
|
||||
void SetDrawID(Uint32 drawId) const;
|
||||
void SetBaseVertex(Int32 baseVertex) const;
|
||||
// True when the transpiled program kept a gl_DrawID uniform, i.e. SetDrawID
|
||||
// actually reaches a shader read rather than being discarded.
|
||||
Bool ReadsDrawID() const { return m_drawIdUniformLocation >= 0; }
|
||||
// Same for gl_BaseVertex: only a program that reads it pays for the per-draw
|
||||
// uniform write, and only such a program needs the reset after one.
|
||||
Bool ReadsBaseVertex() const { return m_baseVertexUniformLocation >= 0; }
|
||||
Int GetIndirectParamsBinding() const { return m_indirectParamsBinding; }
|
||||
Uint GetBackendProgramId() const { return m_backendProgramId; }
|
||||
// False when the last SyncToBackend could not produce a usable program (a
|
||||
@@ -1077,6 +1093,7 @@ namespace MobileGL::MG_Backend::DirectGLES {
|
||||
Uint m_backendGlobalUBOId = 0;
|
||||
Int m_baseInstanceUniformLocation = -1;
|
||||
Int m_drawIdUniformLocation = -1;
|
||||
Int m_baseVertexUniformLocation = -1;
|
||||
Int m_baseInstanceWordIndexUniformLocation = -1;
|
||||
Int m_indirectParamsBinding = -1;
|
||||
Uint32 m_snormFallbackClampOutputMask = 0;
|
||||
|
||||
@@ -274,17 +274,22 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// the batch's own shape - not the driver - rules it out; the compute tier keeps
|
||||
// its remaining feasibility checks inside its implementation, where the data it
|
||||
// has to walk is already in hand.
|
||||
GLESMultiDrawMode ResolveTierForBatch(Bool programReadsDrawID, Bool hasIndexBuffer) {
|
||||
GLESMultiDrawMode ResolveTierForBatch(Bool programReadsDrawID, Bool perSubDrawBaseVertex,
|
||||
Bool hasIndexBuffer) {
|
||||
ResolveTierOnce();
|
||||
GLESMultiDrawMode tier = g_resolvedTier;
|
||||
|
||||
// Batched tiers issue one driver entry for the whole batch, so the emulated
|
||||
// gl_DrawID uniform can only hold one value across every sub-draw. A program
|
||||
// that reads gl_DrawID gets an unrolled tier, which feeds each sub-draw its
|
||||
// own index (the spec's value); nothing else observes the difference.
|
||||
// own index (the spec's value); nothing else observes the difference. The
|
||||
// emulated gl_BaseVertex is one uniform for the same reason, so a batch whose
|
||||
// sub-draws carry their own base vertices unrolls too - even the Ext tier,
|
||||
// which hands the driver the whole basevertex array, can only leave ONE value
|
||||
// in the uniform the shader reads.
|
||||
const Bool batched = tier == GLESMultiDrawMode::Ext || tier == GLESMultiDrawMode::MultiIndirect ||
|
||||
tier == GLESMultiDrawMode::Compute;
|
||||
if (batched && programReadsDrawID) {
|
||||
if (batched && (programReadsDrawID || perSubDrawBaseVertex)) {
|
||||
tier = SupportsTier(GLESMultiDrawMode::BaseVertex) ? GLESMultiDrawMode::BaseVertex
|
||||
: GLESMultiDrawMode::DrawElements;
|
||||
}
|
||||
@@ -371,7 +376,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunIndirect(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool batched, Bool feedDrawID) {
|
||||
GLsizei drawcount, const GLint* basevertex, Bool batched, Bool feedDrawID,
|
||||
Bool feedBaseVertex) {
|
||||
if (!SupportsTier(batched ? GLESMultiDrawMode::MultiIndirect : GLESMultiDrawMode::Indirect)) return false;
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return false;
|
||||
@@ -413,10 +419,12 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
} else {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
const SizeT commandOffset = commandBase + static_cast<SizeT>(i) * sizeof(DrawElementsIndirectCommand);
|
||||
g_GLESFuncs.glDrawElementsIndirect(mode, type, reinterpret_cast<const void*>(commandOffset));
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
}
|
||||
BufferImpl::BindBufferId(GL_DRAW_INDIRECT_BUFFER, previousIndirectBinding);
|
||||
NoteTierExecuted(batched ? GLESMultiDrawMode::MultiIndirect : GLESMultiDrawMode::Indirect);
|
||||
@@ -428,15 +436,17 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunBaseVertexLoop(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID) {
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID, Bool feedBaseVertex) {
|
||||
if (!SupportsTier(GLESMultiDrawMode::BaseVertex)) return false;
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
g_GLESFuncs.glDrawElementsBaseVertex(mode, count[i], type, indices[i],
|
||||
basevertex ? basevertex[i] : 0);
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
NoteTierExecuted(GLESMultiDrawMode::BaseVertex);
|
||||
return true;
|
||||
}
|
||||
@@ -446,7 +456,8 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
Bool RunRebasedDrawElements(GLenum mode, const GLsizei* count, GLenum type, const GLvoid* const* indices,
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID) {
|
||||
GLsizei drawcount, const GLint* basevertex, Bool feedDrawID,
|
||||
Bool feedBaseVertex) {
|
||||
const SizeT indexSize = IndexTypeSize(type);
|
||||
if (indexSize == 0) return false;
|
||||
|
||||
@@ -500,11 +511,16 @@ namespace MobileGL::MG_Backend::DirectGLES::MultiDrawImpl {
|
||||
for (GLsizei i = 0; i < drawcount; ++i) {
|
||||
if (count[i] <= 0) continue;
|
||||
if (feedDrawID) SetCurrentDrawID(static_cast<Uint32>(i));
|
||||
// The base vertex is folded into the rewritten index stream here, so the
|
||||
// driver sees none - but gl_BaseVertex still has to report the value the
|
||||
// application passed for this sub-draw.
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(basevertex ? basevertex[i] : 0);
|
||||
g_GLESFuncs.glDrawElements(mode, count[i], GL_UNSIGNED_INT,
|
||||
reinterpret_cast<const void*>(indexBase + cursor * sizeof(Uint32)));
|
||||
cursor += static_cast<SizeT>(count[i]);
|
||||
}
|
||||
if (feedDrawID) SetCurrentDrawID(0);
|
||||
if (feedBaseVertex) SetCurrentBaseVertex(0);
|
||||
BufferImpl::BindBufferId(GL_ELEMENT_ARRAY_BUFFER, previousIndexBinding);
|
||||
NoteTierExecuted(GLESMultiDrawMode::DrawElements);
|
||||
return true;
|
||||
@@ -837,8 +853,15 @@ void main() {
|
||||
// afterwards would mean unpicking the program, SSBO and index bindings
|
||||
// PrepareForDraw just made, and a dispatch inside an open transform feedback
|
||||
// span is not legal at all. On success it hands back a flattened index stream.
|
||||
// A batch whose sub-draws carry their own base vertices cannot be flattened either
|
||||
// when the program reads gl_BaseVertex: one draw call leaves one uniform value.
|
||||
// Asked conservatively because this decision precedes PrepareForDraw - see
|
||||
// CurrentProgramMayNeedPerSubDrawBuiltins. Flattening is the irreversible half:
|
||||
// once the batch is one draw the values are gone, whereas declining to flatten only
|
||||
// costs the unrolled tier.
|
||||
FlattenedStream flattened;
|
||||
if (ResolvedTier() == GLESMultiDrawMode::Compute && !CurrentProgramReadsDrawID()) {
|
||||
if (ResolvedTier() == GLESMultiDrawMode::Compute &&
|
||||
!CurrentProgramMayNeedPerSubDrawBuiltins(basevertex != nullptr)) {
|
||||
FlattenWithCompute(mode, count, type, indices, drawcount, basevertex, flattened);
|
||||
}
|
||||
|
||||
@@ -852,8 +875,11 @@ void main() {
|
||||
return;
|
||||
}
|
||||
|
||||
// Now that PrepareForDraw has synced the program, both questions have real answers;
|
||||
// the tier choice and the per-sub-draw feeds use those, not the guess above.
|
||||
const Bool feedDrawID = CurrentProgramReadsDrawID();
|
||||
const GLESMultiDrawMode tier = ResolveTierForBatch(feedDrawID, hasIndexBuffer);
|
||||
const Bool feedBaseVertex = basevertex != nullptr && CurrentProgramReadsBaseVertex();
|
||||
const GLESMultiDrawMode tier = ResolveTierForBatch(feedDrawID, feedBaseVertex, hasIndexBuffer);
|
||||
|
||||
Bool drawn = false;
|
||||
switch (tier) {
|
||||
@@ -861,16 +887,19 @@ void main() {
|
||||
drawn = RunExt(mode, count, type, indices, drawcount, basevertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::MultiIndirect:
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/true, feedDrawID);
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/true, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::Indirect:
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/false, feedDrawID);
|
||||
drawn = RunIndirect(mode, count, type, indices, drawcount, basevertex, /*batched=*/false, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::BaseVertex:
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::DrawElements:
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID,
|
||||
feedBaseVertex);
|
||||
break;
|
||||
case GLESMultiDrawMode::Compute:
|
||||
// Its pre-pass ran above; reaching here means it declined this batch's shape.
|
||||
@@ -883,8 +912,13 @@ void main() {
|
||||
// below are the floor: a base-vertex replay where the driver has one, and the
|
||||
// rewritten index stream where it does not. Both are safe for any batch these
|
||||
// entry points can receive.
|
||||
if (!drawn) drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
if (!drawn) drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID);
|
||||
if (!drawn) {
|
||||
drawn = RunBaseVertexLoop(mode, count, type, indices, drawcount, basevertex, feedDrawID, feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
drawn = RunRebasedDrawElements(mode, count, type, indices, drawcount, basevertex, feedDrawID,
|
||||
feedBaseVertex);
|
||||
}
|
||||
if (!drawn) {
|
||||
MGLOG_E("DirectGLES multi-draw: no usable tier for a %d sub-draw batch (mode 0x%x, type 0x%x); "
|
||||
"the batch was dropped",
|
||||
|
||||
@@ -1979,13 +1979,27 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// cannot be corrected and instanced draws with a non-zero baseInstance misrender; this
|
||||
// detects the case so the user gets one warning instead of silent corruption.
|
||||
Bool ProgramFactory::ReflectedReadsInstanceIndexBuiltin(const SpvReflectShaderModule& reflectModule) {
|
||||
return ReflectedDeclaresInputBuiltin(reflectModule, SpvBuiltInInstanceIndex);
|
||||
}
|
||||
|
||||
// GL's gl_BaseVertex and Vulkan's BaseVertex agree for indexed draws and disagree for every
|
||||
// other command, so a program declaring the builtin needs the ZeroBaseVertex variant when a
|
||||
// non-indexed draw uses it (see CompileOptionBit::ZeroBaseVertex). "Declares" rather than
|
||||
// "reads" is the honest word and the useful one: the zeroing pass keeps the variable, so
|
||||
// both variants of a program answer this question identically.
|
||||
Bool ProgramFactory::ReflectedReadsBaseVertexBuiltin(const SpvReflectShaderModule& reflectModule) {
|
||||
return ReflectedDeclaresInputBuiltin(reflectModule, SpvBuiltInBaseVertex);
|
||||
}
|
||||
|
||||
Bool ProgramFactory::ReflectedDeclaresInputBuiltin(const SpvReflectShaderModule& reflectModule,
|
||||
SpvBuiltIn builtin) {
|
||||
for (Uint32 entryIndex = 0; entryIndex < reflectModule.entry_point_count; ++entryIndex) {
|
||||
const SpvReflectEntryPoint& entryPoint = reflectModule.entry_points[entryIndex];
|
||||
for (Uint32 variableIndex = 0; variableIndex < entryPoint.input_variable_count; ++variableIndex) {
|
||||
const SpvReflectInterfaceVariable* variable = entryPoint.input_variables[variableIndex];
|
||||
if (variable != nullptr &&
|
||||
(variable->decoration_flags & SPV_REFLECT_DECORATION_BUILT_IN) != 0 &&
|
||||
variable->built_in == SpvBuiltInInstanceIndex) {
|
||||
variable->built_in == builtin) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -2244,6 +2258,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkProgramObject& entry) const {
|
||||
entry.activeVertexInputLocationMask = 0;
|
||||
entry.vertexInputTypes.fill(0);
|
||||
entry.readsBaseVertexBuiltin = false;
|
||||
|
||||
for (SizeT moduleIndex = 0; moduleIndex < shaders.size() && moduleIndex < spirv.size(); ++moduleIndex) {
|
||||
if (!shaders[moduleIndex] || shaders[moduleIndex]->GetShaderStage() != ShaderStage::Vertex) {
|
||||
@@ -2265,6 +2280,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
continue;
|
||||
}
|
||||
|
||||
entry.readsBaseVertexBuiltin = ReflectedReadsBaseVertexBuiltin(reflectModule);
|
||||
|
||||
if (!m_shaderDrawParametersEnabled && ReflectedReadsInstanceIndexBuiltin(reflectModule)) {
|
||||
static Bool s_warnedInstanceIndexUnsupported = false;
|
||||
if (!s_warnedInstanceIndexUnsupported) {
|
||||
@@ -2909,6 +2926,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// a FragCoordYFlip variant also depends on the baked default-framebuffer height, so
|
||||
// that height rides in the free high half of the key. Flags occupy the low bits, and a
|
||||
// height cannot exceed the 16 bits a swapchain extent fits in.
|
||||
//
|
||||
// "The low bits" is load-bearing and was until now only a comment: a flag that reached
|
||||
// bit 16 would alias the height and two different variants would share one memo slot.
|
||||
static_assert(static_cast<Uint>(CompileOptionBit::ZeroBaseVertex) < (1u << 16),
|
||||
"CompileOptionBit values must stay below bit 16: GetOrCreateProgram packs the "
|
||||
"default-framebuffer height into the high half of the same memo key");
|
||||
const Uint memoKey = (flags & CompileOptionBit::FragCoordYFlip)
|
||||
? (flags.GetRaw() | (m_defaultFramebufferHeight << 16))
|
||||
: flags.GetRaw();
|
||||
@@ -3027,6 +3050,26 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
// The non-indexed variant of a vertex stage that reads gl_BaseVertex: GL wants zero
|
||||
// there, Vulkan's builtin would hand it the draw's firstVertex. Requested per draw
|
||||
// through CompileOptionBit::ZeroBaseVertex, so the indexed variant of the same
|
||||
// program keeps the native builtin and stays correct for glDrawElementsBaseVertex
|
||||
// and for the baseVertex word of an indexed indirect command.
|
||||
if (shaders[i] && shaders[i]->GetShaderStage() == ShaderStage::Vertex &&
|
||||
(flags & CompileOptionBit::ZeroBaseVertex)) {
|
||||
Vector<Uint> zeroedSpirv;
|
||||
if (MG_Util::ShaderTranspiler::ShaderCompiler::ZeroBaseVertexForVulkan(moduleSpirvs[i],
|
||||
zeroedSpirv)) {
|
||||
moduleSpirvs[i] = std::move(zeroedSpirv);
|
||||
} else {
|
||||
// Failing open keeps the native builtin, which is the pre-fix behavior:
|
||||
// gl_BaseVertex reads firstVertex on a DrawArrays instead of zero.
|
||||
MGLOG_E("ProgramFactory: failed to zero gl_BaseVertex for program %u; non-indexed "
|
||||
"draws will read the draw's first vertex from it instead of zero",
|
||||
program.GetExternalIndex());
|
||||
}
|
||||
}
|
||||
|
||||
// A 64-bit vertex input has to arrive as its 32-bit word pair: VK_FORMAT_R64*_SFLOAT is
|
||||
// optional and lavapipe advertises none of them at all. The pass is unconditional so it
|
||||
// always agrees with the Float64 case in VertexInputStateFactory::ToVkVertexFormat, and
|
||||
|
||||
@@ -60,6 +60,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// PositionYFlip (the two are the same fact about the same draws) except under a
|
||||
// quarter turn, which this renderer does not convert rectangles for either.
|
||||
FragCoordYFlip = 1 << 7,
|
||||
// Replaces the vertex stage's gl_BaseVertex reads with zero. GL defines the builtin
|
||||
// as zero for every drawing command that has no baseVertex parameter - all the
|
||||
// DrawArrays forms - while Vulkan's BaseVertex reports firstVertex there. Set only
|
||||
// for a non-indexed draw whose program actually reads the builtin, so nothing else
|
||||
// acquires a second program/pipeline variant. See ZeroBaseVertexPass.
|
||||
ZeroBaseVertex = 1 << 8,
|
||||
};
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
@@ -129,6 +135,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// gl_FragDepth); shader-computed depth is immune to the cross-pipeline
|
||||
// position-invariance quirk (see PipelineFactory::ShouldSuppressDepthWrite).
|
||||
Bool fragmentReplacesDepth = false;
|
||||
// The vertex module declares the BaseVertex builtin. Selects the ZeroBaseVertex
|
||||
// program variant for non-indexed draws, and is deliberately a property of the
|
||||
// PROGRAM rather than of the variant: the zeroed variant leaves the variable
|
||||
// declared, so both variants answer the same and the draw path can ask either.
|
||||
Bool readsBaseVertexBuiltin = false;
|
||||
// Frame-boundary counter value of the last GetOrCreateProgram hit; drives
|
||||
// cache eviction (see OnFrameBoundary). Mutable: the draw snapshot's memoised
|
||||
// entry pointer re-stamps use through a const reference (StampProgramUse).
|
||||
@@ -179,6 +190,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
@@ -192,6 +204,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.readsBaseVertexBuiltin = false;
|
||||
other.lastUsedFrame = 0;
|
||||
}
|
||||
VkProgramObject& operator=(VkProgramObject&& other) noexcept {
|
||||
@@ -231,6 +244,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
producerOutputComponentCount = other.producerOutputComponentCount;
|
||||
fragmentInputComponentCount = other.fragmentInputComponentCount;
|
||||
fragmentReplacesDepth = other.fragmentReplacesDepth;
|
||||
readsBaseVertexBuiltin = other.readsBaseVertexBuiltin;
|
||||
lastUsedFrame = other.lastUsedFrame;
|
||||
other.hash = 0;
|
||||
other.descriptorSetLayout = VK_NULL_HANDLE;
|
||||
@@ -244,6 +258,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
other.producerOutputComponentCount = 0;
|
||||
other.fragmentInputComponentCount = 0;
|
||||
other.fragmentReplacesDepth = false;
|
||||
other.readsBaseVertexBuiltin = false;
|
||||
other.lastUsedFrame = 0;
|
||||
return *this;
|
||||
}
|
||||
@@ -341,6 +356,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// True when an entry point reads the InstanceIndex builtin. Only gates a diagnostic:
|
||||
// without shaderDrawParameters such a shader cannot have gl_InstanceID rebased.
|
||||
static Bool ReflectedReadsInstanceIndexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
// True when an entry point declares the BaseVertex builtin, i.e. when a non-indexed
|
||||
// draw with this program has to take the ZeroBaseVertex variant.
|
||||
static Bool ReflectedReadsBaseVertexBuiltin(const SpvReflectShaderModule& reflectModule);
|
||||
// Shared by the two above: does any entry point list an input variable decorated with
|
||||
// this builtin?
|
||||
static Bool ReflectedDeclaresInputBuiltin(const SpvReflectShaderModule& reflectModule, SpvBuiltIn builtin);
|
||||
|
||||
private:
|
||||
struct ProgramLookupCache {
|
||||
|
||||
@@ -5821,7 +5821,40 @@ void main() {
|
||||
m_lastLodParamsSum = 0; // filled below once the sampled set is known
|
||||
}
|
||||
}
|
||||
const auto& programObj = m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
// GL's gl_BaseVertex is zero for every command without a baseVertex parameter, while
|
||||
// Vulkan's builtin reports the draw's firstVertex; a non-indexed draw therefore takes
|
||||
// the zeroed program variant. The question is about the program's SPIR-V, not about
|
||||
// this draw, so it is memoized on (program lifetime, backend-state version): only the
|
||||
// very first draw of a program pays the extra lookup, and a program used exclusively
|
||||
// with non-indexed draws never resolves - never compiles, never re-stamps - the
|
||||
// variant no draw of it would use.
|
||||
const Bool nonIndexedDraw = !(aspects & DrawSetupAspect::IndexBuffer);
|
||||
const Uint64 baseVertexProgramLifetimeId = program.GetLifetimeId();
|
||||
const Uint32 baseVertexProgramVersion = program.GetBackendStateVersion();
|
||||
const Bool baseVertexQueryKnown = m_lastBaseVertexQueryValid &&
|
||||
m_lastBaseVertexProgramLifetimeId == baseVertexProgramLifetimeId &&
|
||||
m_lastBaseVertexProgramVersion == baseVertexProgramVersion;
|
||||
if (baseVertexQueryKnown && nonIndexedDraw && m_lastBaseVertexReads) {
|
||||
transformFlags |= ProgramFactory::CompileOptionBit::ZeroBaseVertex;
|
||||
}
|
||||
const ProgramFactory::VkProgramObject* resolvedProgramObj =
|
||||
&m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
if (!baseVertexQueryKnown) {
|
||||
// Read the answer out of the entry BEFORE any second lookup: that lookup may
|
||||
// insert and move every entry of the open-addressing cache, dangling the
|
||||
// reference. The zeroing pass leaves the variable declared, so the variant just
|
||||
// resolved answers the same as the base one either way.
|
||||
const Bool readsBaseVertex = resolvedProgramObj->readsBaseVertexBuiltin;
|
||||
m_lastBaseVertexQueryValid = true;
|
||||
m_lastBaseVertexProgramLifetimeId = baseVertexProgramLifetimeId;
|
||||
m_lastBaseVertexProgramVersion = baseVertexProgramVersion;
|
||||
m_lastBaseVertexReads = readsBaseVertex;
|
||||
if (nonIndexedDraw && readsBaseVertex) {
|
||||
transformFlags |= ProgramFactory::CompileOptionBit::ZeroBaseVertex;
|
||||
resolvedProgramObj = &m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
}
|
||||
}
|
||||
const auto& programObj = *resolvedProgramObj;
|
||||
// For the snapshot's memoised entry pointer: if anything below inserts into the
|
||||
// program cache (blit/aux program compiles), the epoch moves and the snapshot
|
||||
// stores no pointer for this draw - the fast path then re-looks-up once.
|
||||
|
||||
@@ -773,6 +773,19 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
ProgramFactory::CompileOptionFlags m_lastLodBaseFlags = {};
|
||||
ProgramFactory::CompileOptionFlags m_lastLodResultFlags = {};
|
||||
|
||||
// Does the current program's vertex stage declare the BaseVertex builtin? A property
|
||||
// of the program's SPIR-V, so (lifetime id, backend-state version) is the whole key.
|
||||
//
|
||||
// Memoized rather than re-asked because asking means resolving the UN-zeroed program
|
||||
// variant, and a program that only ever draws non-indexed would then compile a variant
|
||||
// no draw uses AND re-stamp its use every draw, so the idle sweep could never retire
|
||||
// it. With the memo the answer is known before the first lookup and only the variant
|
||||
// the draw actually needs is resolved.
|
||||
Bool m_lastBaseVertexQueryValid = false;
|
||||
Uint64 m_lastBaseVertexProgramLifetimeId = 0;
|
||||
Uint32 m_lastBaseVertexProgramVersion = 0;
|
||||
Bool m_lastBaseVertexReads = false;
|
||||
|
||||
// Snapshot behind TrySetupDrawFastPath. Values only: the program and
|
||||
// render-pass caches are open-addressing maps whose entries move on
|
||||
// insert, so no pointers into them are cached; the pipeline handle is
|
||||
|
||||
@@ -50,6 +50,7 @@ add_executable(MobileGLIntegrationTest
|
||||
Scenarios/CrossFrameBufferScenario.cpp
|
||||
Scenarios/ResidentIndexScenario.cpp
|
||||
Scenarios/MultiDrawScenario.cpp
|
||||
Scenarios/DrawParametersScenario.cpp
|
||||
Scenarios/AsyncCompileScenario.cpp
|
||||
Scenarios/XfbAfterClipDistanceScenario.cpp
|
||||
Scenarios/ThreeChannelAttachmentScenario.cpp
|
||||
|
||||
@@ -0,0 +1,348 @@
|
||||
// MobileGL - MobileGL/MG_IntegrationTest/Scenarios/DrawParametersScenario.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
//
|
||||
// gl_BaseVertex / gl_BaseInstance / gl_DrawID (GL_ARB_shader_draw_parameters),
|
||||
// read straight out of the shader that a draw command produced.
|
||||
//
|
||||
// Neither backend has these builtins for free, and each is wrong in its own way
|
||||
// when nobody watches:
|
||||
//
|
||||
// * DirectVulkan HAS a BaseVertex builtin, but Vulkan's carries the draw's
|
||||
// firstVertex on a NON-INDEXED draw where GL's is defined to be zero ("the
|
||||
// value passed to the baseVertex parameter, or zero for a command with no
|
||||
// such parameter"). Only the indexed meaning of the two agrees. Every
|
||||
// DrawArrays form therefore takes the ZeroBaseVertex program variant.
|
||||
// * DirectGLES has no such builtins at all: ESSL knows none of them, so the
|
||||
// transpiler demotes each one to a uniform the draw paths feed. A uniform
|
||||
// nobody writes keeps whatever the previous draw left in it - which is what
|
||||
// made gl_BaseVertex report a stale base vertex, and what made
|
||||
// gl_BaseInstance read an unbound storage buffer on a plain glDrawArrays.
|
||||
//
|
||||
// The shader paints the three values, so a draw that carries the wrong ones
|
||||
// paints the wrong colour rather than merely disagreeing with an expectation
|
||||
// somewhere. The framebuffer is cleared to WHITE and no case expects 255 in any
|
||||
// channel, so "the draw did not happen" can never be mistaken for a pass.
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "../Harness/HeadlessGL.h"
|
||||
#include "../Harness/ScenarioFixture.h"
|
||||
|
||||
#ifdef GLAPI
|
||||
#undef GLAPI
|
||||
#endif
|
||||
#define GL_GLEXT_PROTOTYPES
|
||||
#include <GL/gl.h>
|
||||
#include <GL/glext.h>
|
||||
|
||||
namespace MGITest {
|
||||
namespace {
|
||||
|
||||
// #version 450: glslang only declares the ARB builtins from 440 up.
|
||||
//
|
||||
// Each value is painted at 8 units per count, not 1: the errors these builtins
|
||||
// actually have are OFF BY ONE (a sub-draw that never got its own gl_DrawID reads
|
||||
// the previous one's, a base vertex that arrives one command late), and at one unit
|
||||
// per count no readback tolerance can tell those from rounding.
|
||||
//
|
||||
// And biased by two counts, so that ZERO is not the clamp floor. Five of these cases
|
||||
// expect zero, and an unbiased encoding would let every negative value - the shape a
|
||||
// sign or rebase mistake produces - clamp to the same black and pass.
|
||||
constexpr const char* kVertexSource = R"(#version 450 core
|
||||
#extension GL_ARB_shader_draw_parameters : require
|
||||
layout(location = 0) in vec2 aPos;
|
||||
flat out vec3 vParams;
|
||||
void main() {
|
||||
vParams = (vec3(gl_BaseVertexARB, gl_BaseInstanceARB, gl_DrawIDARB) * 8.0 + 16.0) / 255.0;
|
||||
gl_Position = vec4(aPos, 0.0, 1.0);
|
||||
}
|
||||
)";
|
||||
|
||||
constexpr const char* kFragmentSource = R"(#version 450 core
|
||||
flat in vec3 vParams;
|
||||
out vec4 oColor;
|
||||
void main() {
|
||||
oColor = vec4(vParams, 1.0);
|
||||
}
|
||||
)";
|
||||
|
||||
struct Vertex {
|
||||
float x, y;
|
||||
};
|
||||
|
||||
// 3 dummy vertices, then the left half of the viewport as two triangles,
|
||||
// then the right half. Nothing here is symmetric by accident:
|
||||
//
|
||||
// * the padding makes a draw that ignores `first` / baseVertex paint a
|
||||
// degenerate triangle (i.e. nothing) instead of the right picture;
|
||||
// * the two halves let one multi-draw show TWO different gl_DrawID
|
||||
// values in one readback.
|
||||
//
|
||||
// Indices 3..14 together cover the whole viewport, which is what the
|
||||
// single-draw cases use.
|
||||
constexpr int kPad = 3;
|
||||
constexpr int kLeftFirst = kPad; // 3
|
||||
constexpr int kRightFirst = kPad + 6; // 9
|
||||
constexpr int kHalfCount = 6;
|
||||
|
||||
std::vector<Vertex> SceneVertices() {
|
||||
std::vector<Vertex> vertices(static_cast<std::size_t>(kPad), Vertex{0.0f, 0.0f});
|
||||
const float bounds[2][2] = {{-1.0f, 0.0f}, {0.0f, 1.0f}};
|
||||
for (const auto& half : bounds) {
|
||||
const float x0 = half[0];
|
||||
const float x1 = half[1];
|
||||
vertices.push_back({x0, -1.0f});
|
||||
vertices.push_back({x1, -1.0f});
|
||||
vertices.push_back({x1, 1.0f});
|
||||
vertices.push_back({x0, -1.0f});
|
||||
vertices.push_back({x1, 1.0f});
|
||||
vertices.push_back({x0, 1.0f});
|
||||
}
|
||||
return vertices;
|
||||
}
|
||||
|
||||
// GL's DrawArraysIndirectCommand / DrawElementsIndirectCommand, spelled out
|
||||
// so a test can write one without depending on a GL header's struct.
|
||||
struct ArraysCommand {
|
||||
std::uint32_t count, instanceCount, first, baseInstance;
|
||||
};
|
||||
struct ElementsCommand {
|
||||
std::uint32_t count, instanceCount, firstIndex;
|
||||
std::int32_t baseVertex;
|
||||
std::uint32_t baseInstance;
|
||||
};
|
||||
|
||||
class DrawParametersScenario : public ScenarioTest {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
ScenarioTest::SetUp();
|
||||
if (!Ready()) return;
|
||||
std::string error;
|
||||
m_program = CompileProgram(kVertexSource, kFragmentSource, &error);
|
||||
ASSERT_NE(m_program, 0u) << error;
|
||||
|
||||
const std::vector<Vertex> vertices = SceneVertices();
|
||||
glGenVertexArrays(1, &m_vao);
|
||||
glBindVertexArray(m_vao);
|
||||
glGenBuffers(1, &m_vbo);
|
||||
glBindBuffer(GL_ARRAY_BUFFER, m_vbo);
|
||||
glBufferData(GL_ARRAY_BUFFER, static_cast<GLsizeiptr>(vertices.size() * sizeof(Vertex)),
|
||||
vertices.data(), GL_STATIC_DRAW);
|
||||
glEnableVertexAttribArray(0);
|
||||
glVertexAttribPointer(0, 2, GL_FLOAT, GL_FALSE, sizeof(Vertex), reinterpret_cast<const void*>(0));
|
||||
ASSERT_EQ(FirstGLError(), GLenum(GL_NO_ERROR)) << "scene setup left a GL error behind";
|
||||
}
|
||||
|
||||
void TearDown() override {
|
||||
if (!Ready()) return;
|
||||
for (GLuint* buffer : {&m_ebo, &m_indirect, &m_parameter, &m_vbo}) {
|
||||
if (*buffer != 0) glDeleteBuffers(1, buffer);
|
||||
*buffer = 0;
|
||||
}
|
||||
if (m_vao != 0) glDeleteVertexArrays(1, &m_vao);
|
||||
if (m_program != 0) glDeleteProgram(m_program);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void FillBuffer(GLuint& name, GLenum target, const std::vector<T>& data) {
|
||||
if (name == 0) glGenBuffers(1, &name);
|
||||
glBindBuffer(target, name);
|
||||
glBufferData(target, static_cast<GLsizeiptr>(data.size() * sizeof(T)), data.data(), GL_STATIC_DRAW);
|
||||
}
|
||||
|
||||
// Clears to white, runs `draw` and reads the frame back.
|
||||
template <typename DrawFn>
|
||||
Image Render(DrawFn&& draw) {
|
||||
BindDefaultFramebuffer();
|
||||
glViewport(0, 0, HeadlessGL::Get().Width(), HeadlessGL::Get().Height());
|
||||
ClearTo(1.0f, 1.0f, 1.0f, 1.0f);
|
||||
glUseProgram(m_program);
|
||||
glBindVertexArray(m_vao);
|
||||
draw();
|
||||
return ReadPixels(HeadlessGL::Get().Width(), HeadlessGL::Get().Height());
|
||||
}
|
||||
|
||||
// The three builtins as the shader saw them, at a point in one half of
|
||||
// the viewport. `half` is 0 for the left half and 1 for the right.
|
||||
struct DrawParams {
|
||||
int baseVertex = -1, baseInstance = -1, drawId = -1;
|
||||
};
|
||||
// Decodes the biased 8-units-per-count encoding back to the integer the
|
||||
// shader saw. Rounding to the nearest step absorbs any UNORM slop; adjacent
|
||||
// values stay eight units apart, so an off-by-one still reads as one, and a
|
||||
// negative value lands below the bias and decodes negative rather than
|
||||
// clamping into a legitimate zero.
|
||||
static DrawParams ParamsAt(const Image& image, int half) {
|
||||
const int x = image.Width() * (1 + 2 * half) / 4;
|
||||
const Rgba8 pixel = image.At(x, image.Height() / 2);
|
||||
const auto decode = [](std::uint8_t channel) {
|
||||
return (static_cast<int>(channel) - 16 + 4) / 8;
|
||||
};
|
||||
return {decode(pixel.r), decode(pixel.g), decode(pixel.b)};
|
||||
}
|
||||
|
||||
static void ExpectParams(const Image& image, int half, const DrawParams& expected,
|
||||
const std::string& what) {
|
||||
const DrawParams actual = ParamsAt(image, half);
|
||||
EXPECT_EQ(actual.baseVertex, expected.baseVertex)
|
||||
<< what << ": gl_BaseVertex (half " << half << ")";
|
||||
EXPECT_EQ(actual.baseInstance, expected.baseInstance)
|
||||
<< what << ": gl_BaseInstance (half " << half << ")";
|
||||
EXPECT_EQ(actual.drawId, expected.drawId) << what << ": gl_DrawID (half " << half << ")";
|
||||
}
|
||||
|
||||
GLuint m_program = 0;
|
||||
GLuint m_vao = 0;
|
||||
GLuint m_vbo = 0;
|
||||
GLuint m_ebo = 0;
|
||||
GLuint m_indirect = 0;
|
||||
GLuint m_parameter = 0;
|
||||
};
|
||||
|
||||
// ---- the non-indexed forms: gl_BaseVertex is zero, `first` or not ----
|
||||
|
||||
// Vulkan's BaseVertex would answer 3 here (the draw's firstVertex); GL's
|
||||
// must answer 0, because glDrawArrays has no baseVertex parameter at all.
|
||||
TEST_F(DrawParametersScenario, DrawArraysReportsAZeroBaseVertexDespiteItsFirst) {
|
||||
if (!Ready()) return;
|
||||
const Image image = Render([&] { glDrawArrays(GL_TRIANGLES, kLeftFirst, 2 * kHalfCount); });
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {0, 0, 0}, "glDrawArrays(first=3)");
|
||||
ExpectParams(image, 1, {0, 0, 0}, "glDrawArrays(first=3)");
|
||||
}
|
||||
|
||||
TEST_F(DrawParametersScenario, DrawArraysInstancedBaseInstanceReportsItsBaseInstance) {
|
||||
if (!Ready()) return;
|
||||
const Image image = Render([&] {
|
||||
glDrawArraysInstancedBaseInstance(GL_TRIANGLES, kLeftFirst, 2 * kHalfCount, 1, 5);
|
||||
});
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {0, 5, 0}, "glDrawArraysInstancedBaseInstance(baseInstance=5)");
|
||||
}
|
||||
|
||||
// The base instance of one draw must not survive into the next one. This is
|
||||
// the shape that broke on DirectGLES: the emulation uniform is per-program
|
||||
// state, so a draw that never writes it inherits the last writer's value.
|
||||
TEST_F(DrawParametersScenario, APlainDrawAfterABaseInstancedOneSeesZeroAgain) {
|
||||
if (!Ready()) return;
|
||||
const Image image = Render([&] {
|
||||
glDrawArraysInstancedBaseInstance(GL_TRIANGLES, kLeftFirst, 2 * kHalfCount, 1, 7);
|
||||
glDrawArrays(GL_TRIANGLES, kLeftFirst, 2 * kHalfCount);
|
||||
});
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {0, 0, 0}, "plain glDrawArrays after a base-instanced draw");
|
||||
}
|
||||
|
||||
// ---- the indexed forms: gl_BaseVertex IS the base vertex ----
|
||||
|
||||
TEST_F(DrawParametersScenario, DrawElementsBaseVertexReportsItsBaseVertex) {
|
||||
if (!Ready()) return;
|
||||
std::vector<std::uint32_t> indices;
|
||||
for (std::uint32_t i = 0; i < 2 * kHalfCount; ++i) indices.push_back(i);
|
||||
FillBuffer(m_ebo, GL_ELEMENT_ARRAY_BUFFER, indices);
|
||||
|
||||
const Image image = Render([&] {
|
||||
glDrawElementsBaseVertex(GL_TRIANGLES, 2 * kHalfCount, GL_UNSIGNED_INT,
|
||||
reinterpret_cast<const void*>(0), kLeftFirst);
|
||||
});
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {kLeftFirst, 0, 0}, "glDrawElementsBaseVertex(basevertex=3)");
|
||||
ExpectParams(image, 1, {kLeftFirst, 0, 0}, "glDrawElementsBaseVertex(basevertex=3)");
|
||||
}
|
||||
|
||||
// ... and is zero again for the command that has none, including after one
|
||||
// that did: the same leak the base instance has, on the other builtin. The
|
||||
// preceding draw MUST carry a non-zero base vertex or this case proves nothing -
|
||||
// one index run reaches the geometry through the base vertex, the second through
|
||||
// its own indices, so the two draws paint the same picture with different
|
||||
// gl_BaseVertex and only the second one's value survives in the framebuffer.
|
||||
TEST_F(DrawParametersScenario, DrawElementsAfterABaseVertexDrawReportsZeroAgain) {
|
||||
if (!Ready()) return;
|
||||
std::vector<std::uint32_t> indices;
|
||||
for (std::uint32_t i = 0; i < 2 * kHalfCount; ++i) indices.push_back(i);
|
||||
for (std::uint32_t i = 0; i < 2 * kHalfCount; ++i) indices.push_back(i + kLeftFirst);
|
||||
FillBuffer(m_ebo, GL_ELEMENT_ARRAY_BUFFER, indices);
|
||||
const auto rebasedRun = reinterpret_cast<const void*>(2 * kHalfCount * sizeof(std::uint32_t));
|
||||
|
||||
const Image image = Render([&] {
|
||||
glDrawElementsBaseVertex(GL_TRIANGLES, 2 * kHalfCount, GL_UNSIGNED_INT,
|
||||
reinterpret_cast<const void*>(0), kLeftFirst);
|
||||
glDrawElements(GL_TRIANGLES, 2 * kHalfCount, GL_UNSIGNED_INT, rebasedRun);
|
||||
});
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {0, 0, 0}, "glDrawElements after a base-vertex draw");
|
||||
ExpectParams(image, 1, {0, 0, 0}, "glDrawElements after a base-vertex draw");
|
||||
}
|
||||
|
||||
// ---- the multi-draw forms: one gl_DrawID per sub-draw ----
|
||||
|
||||
TEST_F(DrawParametersScenario, MultiDrawArraysNumbersItsSubDraws) {
|
||||
if (!Ready()) return;
|
||||
const GLint firsts[2] = {kLeftFirst, kRightFirst};
|
||||
const GLsizei counts[2] = {kHalfCount, kHalfCount};
|
||||
|
||||
const Image image = Render([&] { glMultiDrawArrays(GL_TRIANGLES, firsts, counts, 2); });
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {0, 0, 0}, "glMultiDrawArrays sub-draw 0");
|
||||
ExpectParams(image, 1, {0, 0, 1}, "glMultiDrawArrays sub-draw 1");
|
||||
}
|
||||
|
||||
// Every field of an indexed indirect command at once: its own gl_DrawID, the
|
||||
// baseVertex word (which the CPU reads out of the command) and the
|
||||
// baseInstance word (which DirectGLES reads through a storage-buffer view of
|
||||
// the very same buffer).
|
||||
TEST_F(DrawParametersScenario, MultiDrawElementsIndirectCarriesEveryCommandsParameters) {
|
||||
if (!Ready()) return;
|
||||
std::vector<std::uint32_t> indices;
|
||||
for (std::uint32_t i = 0; i < kHalfCount; ++i) indices.push_back(i);
|
||||
FillBuffer(m_ebo, GL_ELEMENT_ARRAY_BUFFER, indices);
|
||||
|
||||
const std::vector<ElementsCommand> commands = {
|
||||
{kHalfCount, 1, 0, kLeftFirst, 0},
|
||||
{kHalfCount, 1, 0, kRightFirst, 4},
|
||||
};
|
||||
FillBuffer(m_indirect, GL_DRAW_INDIRECT_BUFFER, commands);
|
||||
|
||||
const Image image = Render([&] {
|
||||
glMultiDrawElementsIndirect(GL_TRIANGLES, GL_UNSIGNED_INT, reinterpret_cast<const void*>(0), 2,
|
||||
sizeof(ElementsCommand));
|
||||
});
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {kLeftFirst, 0, 0}, "indirect command 0");
|
||||
ExpectParams(image, 1, {kRightFirst, 4, 1}, "indirect command 1");
|
||||
}
|
||||
|
||||
// glMultiDrawArraysIndirectCount was missing from the DirectGLES backend
|
||||
// table entirely, so the frontend answered INVALID_OPERATION for every call
|
||||
// while GL_ARB_indirect_parameters was advertised. The parameter buffer here
|
||||
// holds a count SMALLER than maxdrawcount, so a path that ignores it draws a
|
||||
// third command over the top of the second and changes the right half.
|
||||
TEST_F(DrawParametersScenario, MultiDrawArraysIndirectCountObeysItsParameterBuffer) {
|
||||
if (!Ready()) return;
|
||||
const std::vector<ArraysCommand> commands = {
|
||||
{kHalfCount, 1, kLeftFirst, 0},
|
||||
{kHalfCount, 1, kRightFirst, 6},
|
||||
{kHalfCount, 1, kRightFirst, 9},
|
||||
};
|
||||
FillBuffer(m_indirect, GL_DRAW_INDIRECT_BUFFER, commands);
|
||||
const std::vector<std::uint32_t> parameters = {2};
|
||||
FillBuffer(m_parameter, GL_PARAMETER_BUFFER, parameters);
|
||||
|
||||
const Image image = Render([&] {
|
||||
glMultiDrawArraysIndirectCount(GL_TRIANGLES, reinterpret_cast<const void*>(0), 0, 3,
|
||||
sizeof(ArraysCommand));
|
||||
});
|
||||
EXPECT_EQ(FirstGLError(), GLenum(GL_NO_ERROR));
|
||||
ExpectParams(image, 0, {0, 0, 0}, "counted indirect command 0");
|
||||
ExpectParams(image, 1, {0, 6, 1}, "counted indirect command 1");
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace MGITest
|
||||
@@ -346,8 +346,10 @@ void main() {
|
||||
|
||||
EXPECT_NE(rewritten.find("int instance = mg_ZeroBasedInstanceID + mg_BaseInstanceLowered;"),
|
||||
MobileGL::String::npos);
|
||||
EXPECT_NE(rewritten.find("#define mg_ZeroBasedInstanceID (gl_InstanceID - ((mg_BaseInstanceWordIndex >= 0) ? "
|
||||
"int(mg_indirectWords[uint(mg_BaseInstanceWordIndex)]) : 0))"),
|
||||
// One-based word index: zero is the "not an indirect draw" sentinel because that is
|
||||
// the value a GLSL uniform starts at and no draw path writes it before the first draw.
|
||||
EXPECT_NE(rewritten.find("#define mg_ZeroBasedInstanceID (gl_InstanceID - ((mg_BaseInstanceWordIndex > 0) ? "
|
||||
"int(mg_indirectWords[uint(mg_BaseInstanceWordIndex - 1)]) : 0))"),
|
||||
MobileGL::String::npos);
|
||||
EXPECT_NE(rewritten.find(
|
||||
"layout(std430, binding = 12) readonly buffer mg_IndirectParams { highp uint mg_indirectWords[]; };"),
|
||||
@@ -356,6 +358,38 @@ void main() {
|
||||
EXPECT_EQ(CountOccurrences(rewritten, "gl_InstanceID"), 1u);
|
||||
}
|
||||
|
||||
// The sentinel itself, on the builtin it exists for. A zero-based index with a
|
||||
// negative "off" value made every NON-indirect draw of such a program read
|
||||
// mg_indirectWords[0] out of a storage buffer nothing had bound - the uniform starts
|
||||
// at zero and no non-indirect draw path writes it - which is where the CTS
|
||||
// shader_draw_parameters cases lost their geometry on Adreno. Pinned as text because
|
||||
// this contract lives in two places at once: the generated ESSL below and the +1 that
|
||||
// BackendProgramObjectImpl::SetBaseInstanceWordIndex applies.
|
||||
TEST(DirectGLESSanity, TheIndirectWordIndexIsOneBasedSoItsUnwrittenValueMeansNotIndirect) {
|
||||
const ScopedGLESCapabilitiesOverride capsGuard;
|
||||
auto& caps = MobileGL::MG_Backend::DirectGLES::g_GLESCapabilities;
|
||||
caps.IndirectDrawInstanceIdIncludesBaseInstance = false;
|
||||
caps.MaxShaderStorageBufferBindings = 13;
|
||||
|
||||
const MobileGL::String source = R"(#version 310 es
|
||||
highp int mg_BaseInstanceLowered;
|
||||
void main() {
|
||||
gl_Position = vec4(float(mg_BaseInstanceLowered));
|
||||
}
|
||||
)";
|
||||
|
||||
const auto rewritten = MobileGL::MG_Backend::DirectGLES::PromoteDrawParameterGlobalsToUniforms(
|
||||
source, GL_VERTEX_SHADER);
|
||||
|
||||
EXPECT_NE(rewritten.find("#define mg_BaseInstanceLowered ((mg_BaseInstanceWordIndex > 0) ? "
|
||||
"int(mg_indirectWords[uint(mg_BaseInstanceWordIndex - 1)]) : mg_BaseInstance)"),
|
||||
MobileGL::String::npos)
|
||||
<< rewritten;
|
||||
// A zero-based form would spell either of these; neither may survive.
|
||||
EXPECT_EQ(rewritten.find("mg_BaseInstanceWordIndex >= 0"), MobileGL::String::npos);
|
||||
EXPECT_EQ(rewritten.find("uint(mg_BaseInstanceWordIndex)"), MobileGL::String::npos);
|
||||
}
|
||||
|
||||
TEST(DirectGLESSanity, KeepsInstanceIdWhenIndirectDrawsAreConforming) {
|
||||
const ScopedGLESCapabilitiesOverride capsGuard;
|
||||
auto& caps = MobileGL::MG_Backend::DirectGLES::g_GLESCapabilities;
|
||||
|
||||
@@ -21,6 +21,7 @@
|
||||
#include "SpirvPasses/LowerDrawParametersPass.h"
|
||||
#include "SpirvPasses/PackDoubleVertexInputsPass.h"
|
||||
#include "SpirvPasses/RebaseInstanceIndexPass.h"
|
||||
#include "SpirvPasses/ZeroBaseVertexPass.h"
|
||||
#include "SpirvPasses/NormalizeRectCoordinatesPass.h"
|
||||
#include "SpirvPasses/PrivateToEntryLocalPass.h"
|
||||
#include "SpirvPasses/StripUniformLocationsPass.h"
|
||||
@@ -758,6 +759,15 @@ namespace MobileGL {
|
||||
outputBinary);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::ZeroBaseVertexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
|
||||
optimizer.RegisterPass(ZeroBaseVertexPass::CreateZeroBaseVertexPass());
|
||||
|
||||
return RunOptimizerChecked("ZeroBaseVertexForVulkan", optimizer, inputBinary, outputBinary);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::DecoratePositionInvariantForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
|
||||
@@ -65,6 +65,12 @@ namespace MobileGL {
|
||||
static bool LowerRectImages(const Vector<Uint32>& inputBinary, Vector<uint32_t>& outputBinary);
|
||||
static bool RebaseInstanceIndexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Builds the non-indexed-draw variant of a vertex shader: every gl_BaseVertex
|
||||
// read becomes zero, which is what GL defines for a command carrying no
|
||||
// baseVertex parameter while Vulkan's builtin would report firstVertex.
|
||||
// See ZeroBaseVertexPass.
|
||||
static bool ZeroBaseVertexForVulkan(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Re-declares 64-bit float vertex inputs as their 32-bit unsigned word pair
|
||||
// (double -> uvec2, dvec2 -> uvec4) and bitcasts them back to double at entry, so no
|
||||
// VK_FORMAT_R64*_SFLOAT is needed - lavapipe advertises none of them for vertex
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "ZeroBaseVertexPass.h"
|
||||
|
||||
#include "spirv.hpp"
|
||||
#include "source/opt/constants.h"
|
||||
#include "source/opt/def_use_manager.h"
|
||||
#include "source/opt/instruction.h"
|
||||
#include "source/opt/ir_context.h"
|
||||
#include "source/opt/module.h"
|
||||
#include "source/util/make_unique.h"
|
||||
|
||||
#include <vector>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
namespace {
|
||||
using spvtools::opt::Instruction;
|
||||
using spvtools::opt::IRContext;
|
||||
using spvtools::opt::Operand;
|
||||
|
||||
// Returns the Input OpVariable decorated with |builtin|, or nullptr if none.
|
||||
Instruction* FindBuiltinInputVariable(IRContext* context, spv::BuiltIn builtin) {
|
||||
auto* defUseMgr = context->get_def_use_mgr();
|
||||
for (auto& annotation : context->annotations()) {
|
||||
if (annotation.opcode() != spv::Op::OpDecorate || annotation.NumInOperands() < 3) {
|
||||
continue;
|
||||
}
|
||||
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(1)) !=
|
||||
spv::Decoration::BuiltIn) {
|
||||
continue;
|
||||
}
|
||||
if (static_cast<spv::BuiltIn>(annotation.GetSingleWordInOperand(2)) != builtin) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Instruction* variable = defUseMgr->GetDef(annotation.GetSingleWordInOperand(0));
|
||||
if (variable == nullptr || variable->opcode() != spv::Op::OpVariable ||
|
||||
static_cast<spv::StorageClass>(variable->GetSingleWordInOperand(0)) !=
|
||||
spv::StorageClass::Input) {
|
||||
continue;
|
||||
}
|
||||
return variable;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
spvtools::opt::Pass::Status ZeroBaseVertexPass::Process() {
|
||||
auto* irContext = context();
|
||||
auto* defUseMgr = irContext->get_def_use_mgr();
|
||||
|
||||
Instruction* baseVertexVar = FindBuiltinInputVariable(irContext, spv::BuiltIn::BaseVertex);
|
||||
if (baseVertexVar == nullptr) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
const uint32_t baseVertexVarId = baseVertexVar->result_id();
|
||||
|
||||
// Collect every load before mutating: rewriting invalidates the use list.
|
||||
//
|
||||
// Every OTHER kind of user is enumerated and refused rather than ignored. A read
|
||||
// that reaches the variable through a copied pointer or a pointer function
|
||||
// parameter would keep Vulkan's firstVertex while the pass still reported
|
||||
// success, i.e. a partial rewrite indistinguishable from a complete one. glslang
|
||||
// emits neither shape from GLSL today, so this fails closed on something that
|
||||
// cannot happen yet rather than silently half-doing it when it can.
|
||||
std::vector<Instruction*> baseVertexLoads;
|
||||
Bool sawUnexpectedUser = false;
|
||||
defUseMgr->ForEachUser(baseVertexVar, [&](Instruction* user) {
|
||||
switch (user->opcode()) {
|
||||
case spv::Op::OpLoad:
|
||||
if (user->GetSingleWordInOperand(0) == baseVertexVarId) {
|
||||
baseVertexLoads.push_back(user);
|
||||
} else {
|
||||
sawUnexpectedUser = true;
|
||||
}
|
||||
return;
|
||||
// Declarations of the variable, not reads of it.
|
||||
case spv::Op::OpDecorate:
|
||||
case spv::Op::OpDecorateId:
|
||||
case spv::Op::OpDecorateString:
|
||||
case spv::Op::OpName:
|
||||
case spv::Op::OpEntryPoint:
|
||||
return;
|
||||
default:
|
||||
sawUnexpectedUser = true;
|
||||
return;
|
||||
}
|
||||
});
|
||||
|
||||
if (sawUnexpectedUser) {
|
||||
return Status::Failure;
|
||||
}
|
||||
if (baseVertexLoads.empty()) {
|
||||
// Declared but never read - the variant is already the shader itself.
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
|
||||
// Materialize every zero constant BEFORE touching a single instruction, so the
|
||||
// constant/type managers are never consulted against a module this pass has
|
||||
// already half-rewritten - and so the rewrite loop below cannot fail partway
|
||||
// and leave one behind.
|
||||
auto* constantMgr = irContext->get_constant_mgr();
|
||||
auto* typeMgr = irContext->get_type_mgr();
|
||||
std::vector<uint32_t> zeroIds(baseVertexLoads.size(), 0);
|
||||
for (size_t i = 0; i < baseVertexLoads.size(); ++i) {
|
||||
// The zero is built from the LOAD's own type, because a shader may declare
|
||||
// the builtin as either int or uint.
|
||||
const uint32_t typeId = baseVertexLoads[i]->type_id();
|
||||
const spvtools::opt::analysis::Type* type = typeMgr->GetType(typeId);
|
||||
if (type == nullptr) {
|
||||
return Status::Failure;
|
||||
}
|
||||
const spvtools::opt::analysis::Constant* zero = constantMgr->GetConstant(type, {0u});
|
||||
if (zero == nullptr) {
|
||||
return Status::Failure;
|
||||
}
|
||||
const Instruction* zeroInst = constantMgr->GetDefiningInstruction(zero, typeId);
|
||||
if (zeroInst == nullptr) {
|
||||
return Status::Failure;
|
||||
}
|
||||
zeroIds[i] = zeroInst->result_id();
|
||||
}
|
||||
|
||||
// `OpLoad %ty %res %baseVertex` becomes `OpCopyObject %ty %res %zero`. Keeping
|
||||
// %res makes every downstream use pick the zero up with no further rewriting.
|
||||
for (size_t i = 0; i < baseVertexLoads.size(); ++i) {
|
||||
baseVertexLoads[i]->SetOpcode(spv::Op::OpCopyObject);
|
||||
baseVertexLoads[i]->SetInOperands(Instruction::OperandList{
|
||||
{SPV_OPERAND_TYPE_ID, {zeroIds[i]}}});
|
||||
}
|
||||
|
||||
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
spvtools::Optimizer::PassToken ZeroBaseVertexPass::CreateZeroBaseVertexPass() {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<ZeroBaseVertexPass>());
|
||||
}
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
@@ -0,0 +1,40 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/ZeroBaseVertexPass.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include "source/opt/pass.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
#include <Includes.h>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
// GL and Vulkan disagree about gl_BaseVertex on NON-INDEXED draws: GL defines it as
|
||||
// "the value passed to the baseVertex parameter, or zero for a command that has
|
||||
// none", so every DrawArrays form reads zero, while Vulkan's BaseVertex builtin
|
||||
// carries the draw's firstVertex there. (For indexed draws both mean the same thing,
|
||||
// GL's basevertex / Vulkan's vertexOffset, so those must keep the native builtin.)
|
||||
//
|
||||
// This pass produces the non-indexed variant of a vertex shader by replacing every
|
||||
// read of the BaseVertex builtin with a constant zero. The variable itself is left
|
||||
// declared - removing it would also have to reason about the DrawParameters
|
||||
// capability that a BaseInstance read in the same module still needs.
|
||||
//
|
||||
// Vulkan backend only, and only for the ZeroBaseVertex program variant: the
|
||||
// DirectGLES path has no BaseVertex builtin at all (see LowerDrawParametersPass).
|
||||
class ZeroBaseVertexPass : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "zero-base-vertex"; }
|
||||
Status Process() override;
|
||||
|
||||
static spvtools::Optimizer::PassToken CreateZeroBaseVertexPass();
|
||||
};
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
Reference in New Issue
Block a user