[Test] (MG_Util, MG_State): remove the TEMP stage-timing probes - the async-compile measurement campaign is done

This commit is contained in:
2026-08-10 09:56:05 -04:00
parent c3412ca394
commit d868e1c476
6 changed files with 8 additions and 228 deletions
@@ -13,7 +13,6 @@
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
#include <MG_Util/Converters/MGToGL/ProgramEnumConverter.h>
#include <MG_Util/Converters/SPIRVCrossToGL/SpvcTypeConverter.h>
#include <MG_Util/Debug/TempStageProbe.h> // TEMP-STAGE-PROBE
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/ShaderSourceProcessor.h>
#include <MG_Util/ShaderTranspiler/Types.h>
@@ -294,15 +293,6 @@ namespace MobileGL::MG_State::GLState {
const GlslangThreadAllocatorGuard glslangGuard;
using namespace MG_Util::ShaderTranspiler;
// TEMP-STAGE-PROBE: "linktask-total" (whole link body, superset of glslang-link /
// spirv-gen / spirv-opt / reflection / spvc-routing / the ConsumeShaders re-parse)
// plus the every-25-links running-total dump of every stage.
// Declared FIRST so it is destroyed LAST: the dump then already includes this link's
// own contribution to every stage, including "linktask-total".
const MG_Util::Debug::TempStageProbeLinkTick tempStageProbeLinkTick;
const MG_Util::Debug::TempStageProbeScope tempStageProbeLinkTask(
MG_Util::Debug::kTempStageProbeLinkTaskTotal);
MOBILEGL_ASSERT(in.env != nullptr, "ProgramLinkTask: the CompileEnv snapshot is missing");
const CompileEnv& env = *in.env;
@@ -342,12 +332,7 @@ namespace MobileGL::MG_State::GLState {
.explicitOpaqueUniformBindings = &artifacts.explicitOpaqueUniformBindings};
MGLOG_D("ProgramObject %u: Calling ShaderCompiler::LinkProgram", in.externalIndex);
// TEMP-STAGE-PROBE: "glslang-link" - TProgram::link + mapIO.
auto result = [&] {
const MG_Util::Debug::TempStageProbeScope tempStageProbeGlslangLink(
MG_Util::Debug::kTempStageProbeGlslangLink);
return ShaderCompiler::LinkProgram(attrib);
}();
auto result = ShaderCompiler::LinkProgram(attrib);
if (result) {
artifacts.linkStatus = true;
artifacts.program = result.value();
@@ -416,13 +401,7 @@ namespace MobileGL::MG_State::GLState {
// AND read the OPTIMIZED SPIR-V, so BuildGlobalUboRouting still runs strictly after
// both DoReflection and GenerateSpirv.
MGLOG_D("ProgramObject %u: Starting reflection", in.externalIndex);
// TEMP-STAGE-PROBE: "reflection" - buildReflection + the GL location assignment.
const bool tempStageProbeReflectionOk = [&] {
const MG_Util::Debug::TempStageProbeScope tempStageProbeReflection(
MG_Util::Debug::kTempStageProbeReflection);
return static_cast<bool>(DoReflection(env));
}();
if (!tempStageProbeReflectionOk) {
if (!DoReflection(env)) {
DeferLog(std::format("ProgramObject {}: Link failed during reflection: {}", in.externalIndex,
artifacts.infoLog));
return;
@@ -10,7 +10,6 @@
#include <MG_State/GLState/ProgramState/ShaderCompileTask.h> // GlslangThreadAllocatorGuard
#include <MG_Util/Async/ShaderCompilePool.h>
#include <MG_Util/Debug/TempStageProbe.h> // TEMP-STAGE-PROBE
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/SpvcSession.h>
#include <MG_Util/ShaderTranspiler/Types.h>
@@ -102,11 +101,6 @@ namespace MobileGL::MG_State::GLState {
return;
}
// TEMP-STAGE-PROBE: "spirvtask-total" (whole phase-B body, superset of spirv-gen /
// spirv-null / spirv-opt / spvc-routing).
const MG_Util::Debug::TempStageProbeScope tempStageProbeSpirvTask(
MG_Util::Debug::kTempStageProbeSpirvTaskTotal);
MGLOG_D("ProgramObject %u: Starting SPIR-V generation", externalIndex);
GenerateSpirv(handoff, externalIndex);
// GlslangToSpv was the only consumer of the parsed ASTs; everything after this point
@@ -138,12 +132,7 @@ namespace MobileGL::MG_State::GLState {
handoff.shaders.clear();
MGLOG_D("ProgramObject %u: Building global-UBO routing tables", externalIndex);
{
// TEMP-STAGE-PROBE: "spvc-routing" - the SPIRV-Cross session per SPIR-V module.
const MG_Util::Debug::TempStageProbeScope tempStageProbeSpvcRouting(
MG_Util::Debug::kTempStageProbeSpvcRouting);
BuildGlobalUboRouting(handoff, externalIndex);
}
BuildGlobalUboRouting(handoff, externalIndex);
MGLOG_D("ProgramObject %u: Binary generation finished (generatedSpirv size=%zu)", externalIndex,
artifacts.generatedSpirv.size());
}
@@ -166,12 +155,7 @@ namespace MobileGL::MG_State::GLState {
.program = *handoff.reflection.program,
};
MGLOG_D("ProgramObject %u: GenerateSpirv - requesting SPIR-V binary from program", externalIndex);
// TEMP-STAGE-PROBE: "spirv-gen" - GlslangToSpv for every stage of this program.
auto binaryResult = [&] {
const MG_Util::Debug::TempStageProbeScope tempStageProbeSpirvGen(
MG_Util::Debug::kTempStageProbeSpirvGen);
return ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
}();
auto binaryResult = ShaderCompiler::GetSpirvBinaryFromProgram(binaryAttrib);
if (!binaryResult) {
DeferLog(std::format("ProgramObject {}: GenerateSpirv - GetSpirvBinaryFromProgram failed", externalIndex));
MOBILEGL_ASSERT(binaryResult, "GetSpirvBinaryFromProgram failed");
@@ -182,24 +166,8 @@ namespace MobileGL::MG_State::GLState {
artifacts.generatedSpirv.size());
// Linked SPIR-V generated, sanitize and optimize it
{
// TEMP-STAGE-PROBE: "spirv-null" - the same Optimizer::Run with ZERO passes,
// on the pre-optimize binary: pure BuildModule + serialize + IRContext
// teardown. Its device/desktop share ratio against "spirv-opt" is the
// allocator-pathology discriminator. Costs one extra plumbing round per
// module; diagnostic build only.
const MG_Util::Debug::TempStageProbeScope tempStageProbeSpirvNull(
MG_Util::Debug::kTempStageProbeSpirvNull);
for (auto& spv : artifacts.generatedSpirv) {
Vector<uint32_t> nullOut;
(void)ShaderCompiler::TempProbeNullOptimizeBinary(spv, nullOut);
}
}
Bool allOptimized = true;
{
// TEMP-STAGE-PROBE: "spirv-opt" - the spirv-tools optimizer run over every module.
const MG_Util::Debug::TempStageProbeScope tempStageProbeSpirvOpt(
MG_Util::Debug::kTempStageProbeSpirvOpt);
for (auto& spv : artifacts.generatedSpirv) {
auto success = ShaderCompiler::SanitizeAndOptimizeBinary(spv, spv);
if (!success) {
@@ -9,7 +9,6 @@
#include "ShaderCompileTask.h"
#include <MG_Util/Converters/MGToGL/ProgramEnumConverter.h>
#include <MG_Util/Debug/TempStageProbe.h> // TEMP-STAGE-PROBE
#include <MG_Util/ShaderTranspiler/ShaderCompiler.h>
#include <MG_Util/ShaderTranspiler/ShaderSourceProcessor.h>
#include <MG_Util/ShaderTranspiler/Types.h>
@@ -156,13 +155,7 @@ namespace {
MobileGL::MG_State::GLState::ShaderPreprocessResult result;
result.preprocessedSource = source;
{
// TEMP-STAGE-PROBE: "preprocess". Only reached on a ShaderPreprocessCache miss,
// so its call count is the miss count, not the glCompileShader count.
const MobileGL::MG_Util::Debug::TempStageProbeScope tempStageProbePreprocess(
MobileGL::MG_Util::Debug::kTempStageProbePreprocess);
PreprocessShaderSource(stage, result.preprocessedSource, env);
}
PreprocessShaderSource(stage, result.preprocessedSource, env);
if (stage == ShaderStage::Compute) {
if (const std::optional<String> localSizeError =
@@ -214,10 +207,6 @@ namespace MobileGL::MG_State::GLState {
// status + log, so turn a throw into exactly that - a completed job whose result
// is "this shader did not compile", with a log the application can read.
// (JobNode still catches: it is the last resort for anything below.)
// TEMP-STAGE-PROBE: "compiletask-total" - whole ShaderCompileTask body (superset of
// "preprocess" + "parse"); the difference is the task's own bookkeeping.
const MG_Util::Debug::TempStageProbeScope tempStageProbeCompileTask(
MG_Util::Debug::kTempStageProbeCompileTaskTotal);
try {
RunCompilePipeline();
} catch (const std::exception& e) {
@@ -338,14 +327,7 @@ namespace MobileGL::MG_State::GLState {
// Re-parse against the SAME environment the original parse used,
// not against whatever the backend reports now.
.env = artifacts.env.get()};
// TEMP-STAGE-PROBE: "parse-reparse" - the link-time consume-once re-parse. This goes
// through ShaderCompiler::CompileShader, so it is ALSO counted in "parse"; it is
// recorded separately so the "parse" total can be split into first-parse vs re-parse.
auto result = [&] {
const MG_Util::Debug::TempStageProbeScope tempStageProbeReparse(
MG_Util::Debug::kTempStageProbeParseReparse);
return ShaderCompiler::CompileShader(attrib);
}();
auto result = ShaderCompiler::CompileShader(attrib);
if (!result) {
// Should be unreachable: the same source parsed successfully at Compile().
outReparseLog = result.error().log;
-121
View File
@@ -1,121 +0,0 @@
// MobileGL - MobileGL/MG_Util/Debug/TempStageProbe.h
// Copyright (c) 2025-2026 MobileGL-Dev
// Licensed under the GNU Lesser General Public License v3.0:
// https://www.gnu.org/licenses/gpl-3.0.txt
// https://www.gnu.org/licenses/lgpl-3.0.txt
// SPDX-License-Identifier: LGPL-3.0-only
// End of Source File Header
//
// TEMP-STAGE-PROBE: TEMPORARY instrumentation. Measures the per-stage wall time of the
// shader frontend (preprocess / glslang parse / glslang link+mapIO / SPIR-V generation /
// spirv-tools optimize / reflection / SPIRV-Cross global-UBO routing) so a shaderpack load
// can be attributed stage by stage. Delete this file together with every call site marked
// TEMP-STAGE-PROBE when the measurement is finished. Every identifier is spelled
// tempStageProbe* / TempStageProbe* / kTempStageProbe* so both spellings are greppable.
#pragma once
#include <MG_Util/Debug/Log.h>
#include <atomic>
#include <chrono>
namespace MobileGL::MG_Util::Debug {
// TEMP-STAGE-PROBE: stage ids. The two *Total entries are SUPERSETS of the stages above
// them (they time the whole task body), kept so the residue inside a task body is
// visible; kTempStageProbeParseReparse is a SUBSET of kTempStageProbeParse (the link's
// consume-once re-parse goes through the same ShaderCompiler::CompileShader).
enum TempStageProbeStageId : int {
kTempStageProbePreprocess = 0,
kTempStageProbeParse,
kTempStageProbeParseReparse,
kTempStageProbeGlslangLink,
kTempStageProbeSpirvGen,
kTempStageProbeSpirvNull,
kTempStageProbeSpirvOpt,
kTempStageProbeReflection,
kTempStageProbeSpvcRouting,
kTempStageProbeCompileTaskTotal,
kTempStageProbeLinkTaskTotal,
kTempStageProbeSpirvTaskTotal,
kTempStageProbeStageCount
};
inline const char* const kTempStageProbeStageNames[kTempStageProbeStageCount] = {
"preprocess",
"parse",
"parse-reparse[subset-of-parse]",
"glslang-link",
"spirv-gen",
"spirv-null[plumbing-only]",
"spirv-opt",
"reflection",
"spvc-routing",
"compiletask-total[superset]",
"linktask-total[superset]",
"spirvtask-total[superset]",
};
inline std::atomic<unsigned long long> tempStageProbeMicros[kTempStageProbeStageCount] = {};
inline std::atomic<unsigned long long> tempStageProbeCalls[kTempStageProbeStageCount] = {};
inline std::atomic<unsigned long long> tempStageProbeLinkCount{0};
inline std::atomic<long long> tempStageProbeLastDumpNanos{0};
// TEMP-STAGE-PROBE: scoped accumulator. Relaxed atomics only - nothing here orders any
// other memory, and the probe must not perturb what it measures.
class TempStageProbeScope {
public:
explicit TempStageProbeScope(int tempStageProbeStageId)
: tempStageProbeStage(tempStageProbeStageId),
tempStageProbeStart(std::chrono::steady_clock::now()) {}
~TempStageProbeScope() {
const auto tempStageProbeElapsedUs = std::chrono::duration_cast<std::chrono::microseconds>(
std::chrono::steady_clock::now() - tempStageProbeStart)
.count();
tempStageProbeMicros[tempStageProbeStage].fetch_add(
static_cast<unsigned long long>(tempStageProbeElapsedUs), std::memory_order_relaxed);
tempStageProbeCalls[tempStageProbeStage].fetch_add(1ull, std::memory_order_relaxed);
}
TempStageProbeScope(const TempStageProbeScope&) = delete;
TempStageProbeScope& operator=(const TempStageProbeScope&) = delete;
private:
int tempStageProbeStage;
std::chrono::steady_clock::time_point tempStageProbeStart;
};
// TEMP-STAGE-PROBE: one MGLOG_I line per stage carrying the running cumulative total.
inline void TempStageProbeDumpTotals(unsigned long long tempStageProbeLinkIndex) {
for (int tempStageProbeIdx = 0; tempStageProbeIdx < kTempStageProbeStageCount; ++tempStageProbeIdx) {
const unsigned long long tempStageProbeUs =
tempStageProbeMicros[tempStageProbeIdx].load(std::memory_order_relaxed);
const unsigned long long tempStageProbeN =
tempStageProbeCalls[tempStageProbeIdx].load(std::memory_order_relaxed);
MGLOG_I("TEMP-STAGE-PROBE: %s total %llu.%03llu ms over %llu calls (after %llu links)",
kTempStageProbeStageNames[tempStageProbeIdx], tempStageProbeUs / 1000ull,
tempStageProbeUs % 1000ull, tempStageProbeN, tempStageProbeLinkIndex);
}
}
// TEMP-STAGE-PROBE: ticks the program-link counter on scope exit (so a link that returns
// early still counts) and dumps every stage's running total every 25 links. The 3-second
// fallback exists so the LAST link of a load also publishes totals - the process is
// force-stopped afterwards, so there is no exit hook to rely on.
class TempStageProbeLinkTick {
public:
~TempStageProbeLinkTick() {
const unsigned long long tempStageProbeIndex =
tempStageProbeLinkCount.fetch_add(1ull, std::memory_order_relaxed) + 1ull;
const long long tempStageProbeNowNanos =
std::chrono::duration_cast<std::chrono::nanoseconds>(
std::chrono::steady_clock::now().time_since_epoch())
.count();
const long long tempStageProbeLastNanos = tempStageProbeLastDumpNanos.load(std::memory_order_relaxed);
const bool tempStageProbeDueByCount = (tempStageProbeIndex % 25ull) == 0ull;
const bool tempStageProbeDueByTime = tempStageProbeNowNanos - tempStageProbeLastNanos > 3000000000LL;
if (!tempStageProbeDueByCount && !tempStageProbeDueByTime) return;
tempStageProbeLastDumpNanos.store(tempStageProbeNowNanos, std::memory_order_relaxed);
TempStageProbeDumpTotals(tempStageProbeIndex);
}
};
} // namespace MobileGL::MG_Util::Debug
@@ -32,7 +32,6 @@
#include <MG_Backend/BackendObjects.h>
#include <MG_Util/Converters/GLToStr/GLEnumConverter.h>
#include <MG_Util/Converters/GLToGlslang/ProgramEnumConverter.h>
#include <MG_Util/Debug/TempStageProbe.h> // TEMP-STAGE-PROBE
#include <cstdlib>
namespace MobileGL {
@@ -225,18 +224,8 @@ namespace MobileGL {
return std::unexpected(r);
}
// TEMP-STAGE-PROBE: "parse" - one glslang TShader::parse. Both the first
// attempt and the legacy-#version retry below go through here, and so does
// ProgramLinkTask's consume-once re-parse (separately counted there as
// "parse-reparse") and ShaderCompiler::PrewarmBuiltins' three warm-up parses.
const auto tempStageProbeParseShaderSource = [&](const String& tempStageProbeSource) {
const MobileGL::MG_Util::Debug::TempStageProbeScope tempStageProbeParse(
MobileGL::MG_Util::Debug::kTempStageProbeParse);
return ParseShaderSource(lang, shaderType, tempStageProbeSource, attrib.flags, attrib.env);
};
const String source(attrib.sourceStr);
auto result = tempStageProbeParseShaderSource(source);
auto result = ParseShaderSource(lang, shaderType, source, attrib.flags, attrib.env);
if (result) return result;
// Legacy desktop sources are normalized to "#version 330 core" (with a marker on the
@@ -252,7 +241,7 @@ namespace MobileGL {
return result;
}
auto retryResult = tempStageProbeParseShaderSource(retrySource); // TEMP-STAGE-PROBE
auto retryResult = ParseShaderSource(lang, shaderType, retrySource, attrib.flags, attrib.env);
if (!retryResult) return result;
MGLOG_D("CompileShader: %s only parsed after retargeting its legacy #version to 460",
@@ -385,19 +374,6 @@ namespace MobileGL {
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
}
// TEMP-STAGE-PROBE tempStageProbeNullOptimize: SanitizeAndOptimizeBinary minus
// every pass - measures the pure IR plumbing (BuildModule + serialize +
// IRContext teardown) so a device run can separate allocator-bound plumbing
// from transformation work. Remove with the probes.
bool ShaderCompiler::TempProbeNullOptimizeBinary(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
using namespace spvtools;
OptimizerOptions options;
options.set_run_validator(false);
Optimizer optimizer(SPV_ENV_VULKAN_1_1);
return optimizer.Run(inputBinary.data(), inputBinary.size(), &outputBinary, options);
}
bool ShaderCompiler::LowerDrawParametersForEssl(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary) {
using namespace spvtools;
@@ -22,10 +22,6 @@ namespace MobileGL {
static Result<Vector<Vector<unsigned>>> GetSpirvBinaryFromProgram(const ProgramBinaryAttrib& attrib);
static bool SanitizeAndOptimizeBinary(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
// TEMP-STAGE-PROBE: SanitizeAndOptimizeBinary with zero passes - pure IR
// plumbing, for the allocator-pathology discriminator. Remove with the probes.
static bool TempProbeNullOptimizeBinary(const Vector<Uint32>& inputBinary,
Vector<uint32_t>& outputBinary);
// Demotes DrawIndex/BaseInstance/BaseVertex builtins to plain Private globals
// (mg_DrawID/mg_BaseInstance/mg_BaseVertex) so SPIRV-Cross can emit ESSL.
// Only for backends without native draw-parameter support (DirectGLES).