[Feat] (Pipe): content-address sampler states, mint one sampler view per texture and push the three unit sets behind their own content hashes

This commit is contained in:
2026-09-08 15:20:36 -04:00
parent 712c946744
commit 1534cf3784
3 changed files with 805 additions and 11 deletions
+116 -4
View File
@@ -33,23 +33,135 @@
// THIS FILE IS CREATED BY THE CONTRACT COMMIT AND FILLED BY THE PACKAGE THAT OWNS IT - see
// FramebufferEmit.h for why, in full.
#if MOBILEGL_PIPE_PUSH
#include <MG_Impl/Pipe/SamplerEmit.h>
#include <MG_Impl/Pipe/SetHashSuppressor.h>
#include <MG_Impl/Pipe/SlotAllocator.h>
#include <MG_Pipe/MGPipe.h>
#include <MG_Pipe/PipeApply.h>
#include <MG_State/GLState/Core.h>
#include <MG_State/GLState/TextureState/TextureState.h>
#include <MG_Util/Metrics/PipeStats.h>
#include <xxhash.h>
namespace MobileGL::MG_Pipe {
// STUB AT THE CONTRACT COMMIT: emits nothing, returns 0 payload bytes.
// D-G3. Over the tail with Start and Count mixed in, the same shape the two sampler sets
// use - and it covers InternalFormat and Access because those are live glBindImageTexture
// state that the format-less image bake keys on, not decoration.
inline Uint64 MGPipeShaderImageSetContentHash(const MGPImageView* entries, Uint32 start, Uint32 count) {
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPImageView), 0);
hash = MGPipeMixShutter(hash, start);
hash = MGPipeMixShutter(hash, count);
return hash;
}
class MGPipeImageEmitter {
public:
using GLContext = MG_State::GLState::GLContext;
// set_shader_images. Start is 0 and Count is the image-unit window described below.
//
// WHERE THE HIGH-WATER MARK COMES FROM, because the frontend has none and this is the
// one place a reader will look for it. DirectGLES keeps g_imageUnitHighWaterMark, but
// that is written from inside its own per-unit sync and lives on the far side of the
// boundary; TextureState::NoteUnitTouched is the TEXTURE-unit path and
// glBindImageTexture does not reach it. Adding a counter to TextureState would edit
// another package's file and resize the pull build's object, which G1 forbids outright.
//
// So the window is derived instead, from the one thing that decides whether an image
// unit can matter at all: the highest image unit the CURRENT PROGRAM names, memoised
// per program state in SamplerEmit.h's shared inversion, UNIONED with a sticky mark of
// every unit this emitter has already described. A program with no image uniforms
// gives MaxImageUnit == -1 and, with nothing sticky yet, a window of 0 - which is the
// zero early-out, taken BEFORE any hash and before any 192-entry walk, exactly as
// property 1 requires. The mark is sticky so that a program which stops naming a unit
// does not silently stop describing it: the window only grows, and shrinking it is how
// a stale binding would become invisible to the server.
Uint64 EmitShaderImages(GLContext& ctx) {
(void)ctx;
return 0;
const auto& program = ctx.GetProgramForDraw();
const auto& resolution = MGPipeProgramOpaqueUnitsShared().For(program.get());
const Uint32 programWindow =
resolution.MaxImageUnit < 0 ? 0u : static_cast<Uint32>(resolution.MaxImageUnit) + 1u;
if (programWindow > m_window) m_window = programWindow;
const Uint32 count = m_window < kMGPipeMaxImageUnits ? m_window : kMGPipeMaxImageUnits;
// PROPERTY 1, and it is one integer test on every draw of every application that
// never binds an image.
if (count == 0) return 0;
for (Uint32 unit = 0; unit < count; ++unit) {
const auto& binding = ctx.GetImageTextureBinding(static_cast<Int>(unit));
MGPImageView& entry = m_entries[unit];
entry = MGPImageView{};
entry.Unit = unit;
entry.Res = binding.Texture ? MGPipeSlots().Acquire(MGPipeKind::Texture,
binding.Texture->GetLifetimeId())
: kMGPipeNullHandle;
// THE APPLICATION's format and access, verbatim. The bind-format recast and the
// buffer-texture split view are server-side and stay there; so does
// SupportsLayeredImageBinding's rule, which asks the BACKEND target after
// MapToBackendTextureTarget and forces layer to 0 for a non-layerable one -
// Adreno took a stray layer index literally. A client that pre-applied any of
// that would be answering a driver question from the wrong side.
entry.InternalFormat = static_cast<Uint32>(binding.Format);
entry.Layer = static_cast<Uint32>(binding.Layer);
entry.Level = static_cast<Uint16>(binding.Level);
entry.Layered = binding.Layered != GL_FALSE ? 1 : 0;
entry.Access = static_cast<Uint8>(MGPipeEncodeImageAccess(binding.Access));
}
const Uint64 hash = MGPipeShaderImageSetContentHash(m_entries.data(), 0, count);
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetShaderImages, hash)) {
return 0;
}
m_lastImages = MGPShaderImages{};
m_lastImages.Start = 0;
m_lastImages.Count = count;
m_lastImages.ContentHash = hash;
MGPipeApplySetShaderImages(m_lastImages, m_entries.data());
++m_imageSets;
if (MG_Util::PipeStats::Enabled()) {
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::ShaderImageEmissions, 1);
}
return sizeof(MGPShaderImages) + static_cast<Uint64>(count) * sizeof(MGPImageView);
}
void Reset() {}
// The validate point's FreshlyPrimed arm. A fresh context is a fresh set of image
// bindings, so the sticky window starts over; the suppressor slot this set latches is
// invalidated beside this call. There is no record half here at all - set_shader_images
// is pure working state and mints no object of its own.
void Reset() { m_window = 0; }
void ResetCounters() { m_imageSets = 0; }
const MGPShaderImages& LastShaderImages() const { return m_lastImages; }
const Array<MGPImageView, kMGPipeMaxImageUnits>& LastImageViews() const { return m_entries; }
Uint64 ImageSetCount() const { return m_imageSets; }
Uint32 Window() const { return m_window; }
private:
// GL_READ_ONLY / GL_WRITE_ONLY / GL_READ_WRITE folded into the one byte the wire
// carries. A value the enum does not name would otherwise truncate silently into a
// Uint8, which is the class of bug the descriptors exist to close.
static Uint32 MGPipeEncodeImageAccess(GLenum access) {
switch (access) {
case GL_READ_ONLY:
return 0;
case GL_WRITE_ONLY:
return 1;
case GL_READ_WRITE:
return 2;
default:
MOBILEGL_ASSERT(false, "glBindImageTexture access 0x%x is not one of the three GL names",
static_cast<Uint>(access));
return 0;
}
}
Array<MGPImageView, kMGPipeMaxImageUnits> m_entries{};
MGPShaderImages m_lastImages{};
Uint32 m_window = 0;
Uint64 m_imageSets = 0;
};
inline MGPipeImageEmitter& MGPipeImageEmitterInstance() {
+673 -6
View File
@@ -29,39 +29,706 @@
// one subsystem bit, because an operator switching samplers off has to get the whole family's
// legacy arm rather than two thirds of it.
//
// WHAT THIS FILE IS THE CLIENT HALF OF, named so a reader can check it against the oracle:
// DirectGLES' ResolveAndBindUnitTextures (the per-unit sampler-view resolution),
// BindCurrentUnitSamplers (the per-unit sampler-object walk) and the program pass' sampler
// override. TWO BACKEND POST-PROCESSINGS DELIBERATELY STAY ON THE SERVER and act on the
// RESOLVED set: Espryt's raw-depth-fetch sampler substitution and Magma's feedback-loop
// detection. Neither is reproduced here and neither may be.
//
// HEADER-ONLY, for the ownership reason Tracker.h and ResourceTracker.h both state.
#if MOBILEGL_PIPE_PUSH
#include <MG_Impl/Pipe/SetHashSuppressor.h>
#include <MG_Impl/Pipe/SlotAllocator.h>
#include <MG_Impl/Pipe/Tracker.h>
#include <MG_Pipe/MGPipe.h>
#include <MG_Pipe/MGPipeHostSpan.h>
#include <MG_Pipe/PipeApply.h>
#include <MG_State/GLState/Core.h>
#include <MG_State/GLState/ProgramState/ProgramObject.h>
#include <MG_State/GLState/SamplerState/SamplerObject.h>
#include <MG_State/GLState/TextureState/TextureObject.h>
#include <MG_State/GLState/TextureState/TextureUnit.h>
#include <MG_Util/Metrics/PipeStats.h>
#include <xxhash.h>
#include <cstring>
namespace MobileGL::MG_Pipe {
// 0 until the emitters below and in ImageEmit.h have bodies; see FramebufferEmit.h's note.
inline constexpr Uint64 kMGPipeWiredSamplerSubsystem = 0;
// STUB AT THE CONTRACT COMMIT: emits nothing, returns 0 payload bytes.
// ---------------------------------------------------------------------------------
// D-F1: the canonical SamplerParameters copy, and why it is not a memcpy
// ---------------------------------------------------------------------------------
//
// sizeof(SamplerParameters) == 100 and its members occupy 97 of those bytes: six 4-byte
// enums, four Floats, two more 4-byte enums, three 16-byte colour vectors and the one-byte
// borderColorForm. Bytes 97, 98 and 99 are PADDING and no writer ever touches them.
//
// A cache that hashed or memcmp'd the object's own bytes would therefore read
// uninitialised memory. In practice SamplerObject value-initialises its member and never
// rewrites it, so in practice those bytes are stable - and "in practice" is not a
// contract. The whole point of a content-addressed cache is that a false MISS mints a
// fresh CSO per call: a 256-entry cache with a hit rate of zero, on a path nobody looks
// at, because the pixels are right either way.
//
// So every hash and every confirm runs over a copy that is memset to zero FIRST and then
// assigned field by field, which makes the padding deterministically zero on both sides of
// the comparison. The field list is MGP_FIELDS_SamplerParameters', in its order, and the
// verify build compares the same sixteen fields one at a time - without that
// PipeFields.def row the blob would be compared as bytes and G4 would be a coin flip.
//
// ONE COPY PER MINT ATTEMPT, never per draw: the version-first skip in the two emitters
// below decides whether to come here at all.
inline SamplerParameters MGPipeCanonicalSamplerParameters(const SamplerParameters& src) {
static_assert(std::is_trivially_copyable_v<SamplerParameters>,
"the canonical copy is memset and then assigned field by field");
SamplerParameters canon;
std::memset(static_cast<void*>(&canon), 0, sizeof(canon));
canon.wrapS = src.wrapS;
canon.wrapT = src.wrapT;
canon.wrapR = src.wrapR;
canon.minFilter = src.minFilter;
canon.magFilter = src.magFilter;
canon.mipmapMode = src.mipmapMode;
canon.minLod = src.minLod;
canon.maxLod = src.maxLod;
canon.lodBias = src.lodBias;
canon.maxAnisotropy = src.maxAnisotropy;
canon.compareFunc = src.compareFunc;
canon.compareMode = src.compareMode;
canon.borderColor = src.borderColor;
canon.borderColorI = src.borderColorI;
canon.borderColorUI = src.borderColorUI;
// ALL FOUR BORDER-COLOUR MEMBERS CROSS, the form included. All three representations
// are always numerically populated, so the value alone cannot say which driver entry
// point applies (glSamplerParameterIiv vs fv, or which VkBorderColor family), and both
// of Espryt's redundancy filters compare all four. Dropping this one line is G7's
// scripted negative control and SamplerEmit's suite must go red naming it.
canon.borderColorForm = src.borderColorForm;
return canon;
}
inline Uint64 MGPipeHashSamplerParameters(const SamplerParameters& canon) {
return XXH64(&canon, sizeof(canon), 0);
}
// ---------------------------------------------------------------------------------
// D-F1: the content-addressed sampler CSO cache, capacity 256
// ---------------------------------------------------------------------------------
//
// Unlike P3a's vertex-elements CSO, content addressing is RIGHT here: a SamplerObject is a
// pure 100-byte value with no driver-side per-object binding state, two identical samplers
// can share one CSO with no extra work on either side, and Espryt's BackendSamplerObject
// is a driver name plus a parameter shadow - nothing a second frontend object would have
// to re-establish.
//
// THE LOOKUP IS CsoCache.h's, one for one: hash, probe, and CONFIRM WITH A MEMCMP before
// reusing a handle, because a bare 64-bit equality would alias two different sampler
// states onto one CSO and that is silent wrong filtering with no gate that can see it.
//
// CAPACITY 256 IS ALSO A SAFETY BOUND, not only a size. One emission pass acquires at most
// kMGPipeMaxTextureUnits == 192 CSOs and touches every one of them, so LRU can only ever
// evict an entry from an EARLIER pass - a handle named in the tail of the record being
// built right now can never be the victim. The static_assert below is what keeps that true
// if either number is ever retuned.
//
// THE SLOT IS ALLOCATED WITHOUT A LIFETIME ID, deliberately: a content-addressed CSO
// belongs to a VALUE and not to a frontend object, so ~SamplerObject must not free it -
// another live SamplerObject may hold the same value. MGPipeEmitSamplerCsoDestroyAndFree
// resolves nothing for such an id and correctly frees nothing; the only death path for
// these slots is the LRU eviction below, which is client-side and therefore
// backend-neutral on day one.
inline constexpr SizeT kMGPipeSamplerCsoCacheCapacity = 256;
static_assert(kMGPipeSamplerCsoCacheCapacity > kMGPipeMaxTextureUnits,
"one emission pass touches every unit's CSO, so the cache must be able to "
"hold a whole pass without evicting a handle that pass is about to name");
class MGPipeSamplerCsoCache {
public:
struct Counters {
Uint64 Mints = 0; // create_sampler_state emissions
Uint64 Acquisitions = 0;
Uint64 Hits = 0; // a probe that found a live entry and passed the memcmp
Uint64 Collisions = 0; // a hash hit the memcmp REJECTED - the reason it exists
Uint64 Evictions = 0; // LRU evictions, each one a delete_sampler_state
};
// The handle for `params`' value. Mints and emits create_sampler_state on a miss and
// emits delete_sampler_state for whatever it evicts to make room. `payloadBytes`
// accumulates what went on the wire.
//
// THIS IS ALSO PACKAGE B's SEAM. MGPTextureParams::BuiltinSampler names the CSO that
// carries the SamplerParameters of the SamplerObject every ITextureObject owns, and a
// null handle there is Fatal{ProtocolCorruption} rather than "no sampler". TextureEmit.h
// acquires it from here, so the texture's built-in sampler and a glBindSampler'd
// sampler object with the same value share one CSO and one server-side twin - which is
// exactly the sharing that makes content addressing the right answer for this kind.
MGPipeHandle Acquire(const SamplerParameters& params, Uint64& payloadBytes) {
++m_counters.Acquisitions;
const SamplerParameters canon = MGPipeCanonicalSamplerParameters(params);
const Uint64 hash = MGPipeHashSamplerParameters(canon);
for (SizeT i = 0; i < m_entries.size(); ++i) {
if (m_entries[i].Hash != hash) continue;
if (std::memcmp(&m_entries[i].Params, &canon, sizeof(canon)) != 0) {
// A 64-bit collision between two DIFFERENT sampler states. Reusing the
// handle would filter one state with the other's parameters, so the entry
// is dropped and the caller mints - correctness first, and the counter
// says how often it happened.
++m_counters.Collisions;
Evict(i);
break;
}
m_entries[i].LastUsed = ++m_clock;
++m_counters.Hits;
return m_entries[i].Cso;
}
return Mint(hash, canon, payloadBytes);
}
// "Does the applier hold a create_sampler_state record for exactly this handle?" A
// slot is not evidence of a record - a backend twin table mints one through
// MGPipeSlots().Acquire whether or not this client was ever asked to emit - so the
// death paths ask this rather than guessing from the slot.
Bool RecordIsPublished(MGPipeHandle handle) const {
if (MGPipeHandleIsNull(handle)) return false;
for (const Entry& entry : m_entries) {
if (entry.Cso == handle) return true;
}
return false;
}
// A unit test's fixture, and nothing else. NOT called from the validate point's
// FreshlyPrimed arm: MGPipeApplierReset is a make-current and does NOT drop object
// records, so a sampler CSO the applier holds outlives a context switch. Dropping the
// cache there would leak the applier's record and re-mint a value it already has.
void ResetForTest() {
for (const Entry& entry : m_entries) MGPipeSlots().Free(MGPipeKind::SamplerCso, entry.Cso);
m_entries.clear();
m_clock = 0;
}
void ResetCounters() { m_counters = Counters{}; }
SizeT Size() const { return m_entries.size(); }
const Counters& GetCounters() const { return m_counters; }
private:
struct Entry {
Uint64 Hash = 0;
Uint64 LastUsed = 0;
MGPipeHandle Cso = kMGPipeNullHandle;
// THE CANONICAL BYTES, not the caller's. This is the memcmp's other operand and it
// has to have the same deterministic padding the probe's copy has.
SamplerParameters Params{};
};
MGPipeHandle Mint(Uint64 hash, const SamplerParameters& canon, Uint64& payloadBytes) {
if (m_entries.size() >= kMGPipeSamplerCsoCacheCapacity) {
SizeT victim = 0;
for (SizeT i = 1; i < m_entries.size(); ++i) {
if (m_entries[i].LastUsed < m_entries[victim].LastUsed) victim = i;
}
Evict(victim);
}
const MGPipeHandle cso = MGPipeSlots().Allocate(MGPipeKind::SamplerCso);
MGPSamplerDesc desc{};
desc.Cso = cso;
// THE ONE BLOB RULE (MGPipeTypes.h): Size 0 means "this record does not declare its
// blob", which is what a monolith emission is - the parameters ride beside the
// record through the entry point's companion pointer and the applier stores them by
// value. Splitting this for a transport is P5's problem, not this emitter's.
desc.Parameters.Seg = kMGHostSpanSegNone;
desc.Parameters.Offset = 0;
desc.Parameters.Size = 0;
Entry entry;
entry.Hash = hash;
entry.LastUsed = ++m_clock;
entry.Cso = cso;
entry.Params = canon;
m_entries.push_back(entry);
// The applier is handed the CACHE's copy, so the pointer stays valid for the whole
// call and the bytes it stores are provably the bytes the memcmp will confirm
// against later.
MGPipeApplyCreateSamplerState(desc, &m_entries.back().Params);
++m_counters.Mints;
payloadBytes += sizeof(MGPSamplerDesc) + sizeof(SamplerParameters);
if (MG_Util::PipeStats::Enabled()) {
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::CsoBlobBytes,
sizeof(SamplerParameters));
}
return cso;
}
void Evict(SizeT index) {
MGPHandleOnly handle{};
handle.Handle = m_entries[index].Cso;
handle.Kind = static_cast<Uint32>(MGPipeKind::SamplerCso);
// THE THREE-STEP ORDER, the same one PipeMutation.h fixes for the death helpers:
// the wire delete drops the applier's record while the record still exists, and
// only then does the slot go back. There is no NotifyStateObjectDestroyed step
// here - a content-addressed CSO has no frontend object whose death is being
// announced, which is precisely why this eviction is the only death path it has.
MGPipeApplyDeleteSamplerState(handle);
MGPipeSlots().Free(MGPipeKind::SamplerCso, m_entries[index].Cso);
m_entries[index] = m_entries.back();
m_entries.pop_back();
++m_counters.Evictions;
}
Vector<Entry> m_entries;
Uint64 m_clock = 0;
Counters m_counters;
};
inline MGPipeSamplerCsoCache& MGPipeSamplerCsoCacheInstance() {
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason, and this one is on a death
// path: TextureEmit.h asks it for every texture's built-in sampler CSO, so an exit
// handler running a frontend destructor must not find it freed.
static MGPipeSamplerCsoCache* cache = new MGPipeSamplerCsoCache();
return *cache;
}
// ---------------------------------------------------------------------------------
// D-F3: the client-side sampling resolution
// ---------------------------------------------------------------------------------
//
// GL binds one texture per unit PER TARGET; which one the shader actually samples depends
// on the sampler uniform's TYPE. Gallium's one-view-per-slot is the resolved form, so the
// resolution moves to the client and the record carries the answer rather than the inputs.
//
// This is DirectGLES' SamplerUniformTextureTarget, moved to the side of the boundary that
// now owns the question. Targets with no sampler spelling map to Unknown, and a unit whose
// uniform type resolves to Unknown carries no view - which is exactly "the program does not
// sample this unit".
inline MobileGL::TextureTarget MGPipeSamplerUniformTextureTarget(GLenum uniformType) {
switch (uniformType) {
case GL_SAMPLER_1D:
case GL_INT_SAMPLER_1D:
case GL_UNSIGNED_INT_SAMPLER_1D:
case GL_SAMPLER_1D_SHADOW:
return TextureTarget::Texture1D;
case GL_SAMPLER_2D:
case GL_INT_SAMPLER_2D:
case GL_UNSIGNED_INT_SAMPLER_2D:
case GL_SAMPLER_2D_SHADOW:
return TextureTarget::Texture2D;
case GL_SAMPLER_3D:
case GL_INT_SAMPLER_3D:
case GL_UNSIGNED_INT_SAMPLER_3D:
return TextureTarget::Texture3D;
case GL_SAMPLER_CUBE:
case GL_INT_SAMPLER_CUBE:
case GL_UNSIGNED_INT_SAMPLER_CUBE:
case GL_SAMPLER_CUBE_SHADOW:
return TextureTarget::TextureCubeMap;
case GL_SAMPLER_1D_ARRAY:
case GL_INT_SAMPLER_1D_ARRAY:
case GL_UNSIGNED_INT_SAMPLER_1D_ARRAY:
case GL_SAMPLER_1D_ARRAY_SHADOW:
return TextureTarget::Texture1DArray;
case GL_SAMPLER_2D_ARRAY:
case GL_INT_SAMPLER_2D_ARRAY:
case GL_UNSIGNED_INT_SAMPLER_2D_ARRAY:
case GL_SAMPLER_2D_ARRAY_SHADOW:
return TextureTarget::Texture2DArray;
case GL_SAMPLER_CUBE_MAP_ARRAY:
case GL_INT_SAMPLER_CUBE_MAP_ARRAY:
case GL_UNSIGNED_INT_SAMPLER_CUBE_MAP_ARRAY:
case GL_SAMPLER_CUBE_MAP_ARRAY_SHADOW:
return TextureTarget::TextureCubeMapArray;
case GL_SAMPLER_2D_RECT:
case GL_INT_SAMPLER_2D_RECT:
case GL_UNSIGNED_INT_SAMPLER_2D_RECT:
case GL_SAMPLER_2D_RECT_SHADOW:
return TextureTarget::TextureRectangle;
case GL_SAMPLER_2D_MULTISAMPLE:
case GL_INT_SAMPLER_2D_MULTISAMPLE:
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE:
return TextureTarget::Texture2DMultisample;
case GL_SAMPLER_2D_MULTISAMPLE_ARRAY:
case GL_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
return TextureTarget::Texture2DMultisampleArray;
case GL_SAMPLER_BUFFER:
case GL_INT_SAMPLER_BUFFER:
case GL_UNSIGNED_INT_SAMPLER_BUFFER:
return TextureTarget::TextureBuffer;
default:
return TextureTarget::Unknown;
}
}
// Unit -> sampled target, and the highest image unit the program names, both INVERTED ONCE
// per program state rather than searched per unit.
//
// DirectGLES asks the question the other way round - for one unit it walks every uniform
// location - and gets away with it because it only asks on an aliasing conflict. A client
// that asked per unit would be O(units x locations) at every verb the sampler bit fires
// on, which is exactly the per-draw cost the phase's budget forbids. So the walk runs once
// and is memoised on (lifetime id, link version, backend state version): the third is the
// counter SetUniformSamplerOrImageUnitIndex bumps, i.e. the one thing that can move a
// uniform's unit without relinking.
//
// IT IS SHARED WITH ImageEmit.h on purpose. The sampler and image halves come out of one
// walk over one array, they ride one subsystem bit, and computing them separately would
// walk the same locations twice per program change.
class MGPipeProgramOpaqueUnits {
public:
using ProgramObject = MG_State::GLState::ProgramObject;
struct Resolution {
// TextureTarget + 1 per unit, 0 meaning "this program samples nothing here". A
// Uint8 because TextureTargetCount is 11 and this array is 192 entries long on a
// path that wants to stay in cache.
Array<Uint8, kMGPipeMaxTextureUnits> SamplerTarget{};
// Highest image unit the program names, or -1 for a program with no image
// uniforms - which is what gives ImageEmit.h its zero early-out for free.
Int32 MaxImageUnit = -1;
};
const Resolution& For(const ProgramObject* program) {
const Uint64 lifetimeId = program != nullptr ? program->GetLifetimeId() : 0;
const Uint32 linkVersion = program != nullptr ? program->GetLinkVersion() : 0;
const Uint32 stateVersion = program != nullptr ? program->GetBackendStateVersion() : 0;
if (m_valid && m_lifetimeId == lifetimeId && m_linkVersion == linkVersion &&
m_stateVersion == stateVersion) {
return m_resolution;
}
m_resolution = Resolution{};
// GetLinkStatus is the same guard DirectGLES uses before it trusts the reflection:
// a program that did not link has no usable uniform table, and every unit resolves
// to "not sampled".
if (program != nullptr && program->GetLinkStatus()) {
const Uint maxLocation = program->GetMaxUniformLocation();
for (Uint location = 0; location <= maxLocation; ++location) {
const Int unit = program->GetUniformSamplerOrImageUnitIndex(location);
if (unit < 0 || unit >= static_cast<Int>(kMGPipeMaxTextureUnits)) continue;
if (program->GetUniformTypeFacts(location).isImage) {
if (unit > m_resolution.MaxImageUnit) m_resolution.MaxImageUnit = unit;
continue;
}
const TextureTarget target =
MGPipeSamplerUniformTextureTarget(program->GetUniformType(location));
if (target == TextureTarget::Unknown) continue;
// FIRST WRITER WINS, which is DirectGLES' arbitration read forwards: when
// two sampler uniforms of different types share a unit the binding placed
// first stands rather than being silently overwritten by whichever location
// comes last.
Uint8& slot = m_resolution.SamplerTarget[static_cast<SizeT>(unit)];
if (slot == 0) slot = static_cast<Uint8>(static_cast<Int>(target) + 1);
}
}
m_lifetimeId = lifetimeId;
m_linkVersion = linkVersion;
m_stateVersion = stateVersion;
m_valid = true;
return m_resolution;
}
void Invalidate() { m_valid = false; }
private:
Resolution m_resolution;
Uint64 m_lifetimeId = 0;
Uint32 m_linkVersion = 0;
Uint32 m_stateVersion = 0;
Bool m_valid = false;
};
// ONE inversion for the whole family, shared by MGPipeSamplerEmitter and
// MGPipeImageEmitter. Two memos would walk the same uniform table twice per program
// change, and - worse - could disagree about which locations they saw, which is how a
// sampler unit and an image unit come to be resolved against two different readings of one
// program. Never destroyed, like every other MGPipe process singleton.
inline MGPipeProgramOpaqueUnits& MGPipeProgramOpaqueUnitsShared() {
static MGPipeProgramOpaqueUnits* units = new MGPipeProgramOpaqueUnits();
return *units;
}
// ---------------------------------------------------------------------------------
// D-G3: the two unit sets' content hashes
// ---------------------------------------------------------------------------------
//
// XXH64 over the tail entries, with Start and Count mixed in through MGPipeMixShutter -
// the VertexInputEmit shape, and for its reason: the hash must cover EVERY input the
// record carries, or a record whose one changed field is outside the tail gets suppressed.
// Both tails are built into zero-initialised staging arrays, so no padding byte enters
// either hash.
inline Uint64 MGPipeSamplerViewSetContentHash(const MGPBoundView* entries, Uint32 start, Uint32 count) {
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPBoundView), 0);
hash = MGPipeMixShutter(hash, start);
hash = MGPipeMixShutter(hash, count);
return hash;
}
inline Uint64 MGPipeSamplerStateSetContentHash(const MGPipeHandle* entries, Uint32 start, Uint32 count) {
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPipeHandle), 0);
hash = MGPipeMixShutter(hash, start);
hash = MGPipeMixShutter(hash, count);
return hash;
}
// ---------------------------------------------------------------------------------
// The emitter
// ---------------------------------------------------------------------------------
class MGPipeSamplerEmitter {
public:
using GLContext = MG_State::GLState::GLContext;
using ITextureObject = MG_State::GLState::ITextureObject;
using SamplerObject = MG_State::GLState::SamplerObject;
// set_sampler_views: the PROGRAM-RESOLVED set only, one entry per unit, no stage
// dimension. Start is 0 and Count is GetMaxTouchedTextureUnit() + 1 clamped to the
// wire bound - the high-water mark is directly the count argument and is not
// re-derived.
//
// WHAT A UNIT'S ENTRY MEANS, and all three cases are legal rather than holes:
// the program resolves no target here -> View and Texture both null
// it resolves one, nothing is bound -> View and Texture both null
// it resolves one and a texture is bound-> Texture is that object's handle and View
// is its sampler-view CSO
// An UNDEFINED DEFAULT texture (name 0 with no image) and a texture that
// SAMPLES AS INCOMPLETE are both dropped to null here, which is the resolution
// DirectGLES performs by leaving the native target unbound: an incomplete texture
// samples (0,0,0,1) and the driver cannot work that out for itself, because the
// backend storage is immutable and never saw the level the application redefined at
// the wrong size.
Uint64 EmitSamplerViews(GLContext& ctx) {
(void)ctx;
return 0;
const Int maxTouched = ctx.GetMaxTouchedTextureUnit();
const Uint32 count =
maxTouched < 0 ? 0u
: Min(static_cast<Uint32>(maxTouched) + 1u, kMGPipeMaxTextureUnits);
// The join is the EMITTER's, deliberately, and it is the same GetProgramForDraw()
// the verb is about to make anyway: the tracker's own shutters read
// GetCurrentProgram() precisely so that answering "did the shader move" never
// forces a compile. In the ladder EmitShaderState runs before this, so in the
// steady state the program is already joined by the time this line runs.
const auto& program = ctx.GetProgramForDraw();
const auto& resolution = MGPipeProgramOpaqueUnitsShared().For(program.get());
Uint64 bytes = 0;
for (Uint32 unit = 0; unit < count; ++unit) {
MGPBoundView& entry = m_views[unit];
entry = MGPBoundView{};
entry.Unit = unit;
entry.View = kMGPipeNullHandle;
entry.Texture = kMGPipeNullHandle;
const Uint8 encoded = resolution.SamplerTarget[unit];
if (encoded == 0) continue;
const auto target = static_cast<TextureTarget>(static_cast<Int>(encoded) - 1);
auto& textureUnit = ctx.GetTextureUnitObject(static_cast<Int>(unit));
const auto& texture = textureUnit.GetBindingSlot(target).GetBoundObject();
if (!texture) continue;
if (MG_State::GLState::IsUndefinedDefaultTexture(texture.get())) continue;
// The unit's sampler object overrides the texture's own, exactly as in GL, and
// the completeness answer depends on which one applies - a mipmap mode of None
// makes a single-level texture complete that would otherwise sample black.
const auto& unitSampler = textureUnit.GetSamplerObject();
const SamplerObject* effective =
unitSampler ? unitSampler.get() : texture->GetSamplerObject().get();
if (MG_State::GLState::SamplesAsIncompleteTexture(texture.get(), effective)) continue;
entry.Texture = MGPipeSlots().Acquire(MGPipeKind::Texture, texture->GetLifetimeId());
entry.View = AcquireSamplerView(*texture, entry.Texture, bytes);
}
const Uint64 hash = MGPipeSamplerViewSetContentHash(m_views.data(), 0, count);
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetSamplerViews, hash)) {
return bytes;
}
m_lastViews = MGPSamplerViews{};
m_lastViews.Start = 0;
m_lastViews.Count = count;
m_lastViews.ContentHash = hash;
MGPipeApplySetSamplerViews(m_lastViews, m_views.data());
++m_viewSets;
if (MG_Util::PipeStats::Enabled()) {
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::SamplerViewEmissions, 1);
}
return bytes + sizeof(MGPSamplerViews) + static_cast<Uint64>(count) * sizeof(MGPBoundView);
}
// bind_sampler_states: the unit's sampler CSO, or the null handle when the unit has no
// sampler object - the texture's built-in sampler then applies, exactly as today.
//
// NOT program-resolved, and that asymmetry with the view set above is deliberate: a
// sampler object is bound to a UNIT and applies to whatever the unit holds, so its set
// is the plain per-unit walk BindCurrentUnitSamplers already does. A redundant re-bind
// of the same sampler object emits nothing at all, which is the whole point of the
// suppressor slot: GetTextureBindGeneration() bumps on a redundant re-bind - 26.2
// rebinds the same sampler at every texture-unit switch - so without it this would be
// a several-hundred-byte variable-length record per batch.
Uint64 EmitSamplerStates(GLContext& ctx) {
(void)ctx;
return 0;
const Int maxTouched = ctx.GetMaxTouchedTextureUnit();
const Uint32 count =
maxTouched < 0 ? 0u
: Min(static_cast<Uint32>(maxTouched) + 1u, kMGPipeMaxTextureUnits);
Uint64 bytes = 0;
for (Uint32 unit = 0; unit < count; ++unit) {
const auto& sampler = ctx.GetTextureUnitObject(static_cast<Int>(unit)).GetSamplerObject();
m_states[unit] = sampler ? MGPipeSamplerCsoCacheInstance().Acquire(
sampler->GetAllSamplerParameters(), bytes)
: kMGPipeNullHandle;
}
const Uint64 hash = MGPipeSamplerStateSetContentHash(m_states.data(), 0, count);
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::BindSamplerStates, hash)) {
return bytes;
}
m_lastStates = MGPSamplerStates{};
m_lastStates.Start = 0;
m_lastStates.Count = count;
m_lastStates.ContentHash = hash;
MGPipeApplyBindSamplerStates(m_lastStates, m_states.data());
++m_stateSets;
if (MG_Util::PipeStats::Enabled()) {
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::SamplerStateEmissions, 1);
}
return bytes + sizeof(MGPSamplerStates) + static_cast<Uint64>(count) * sizeof(MGPipeHandle);
}
void Reset() {}
// D-F2: ONE SamplerViewCso PER ITextureObject, minted off its lifetime id, and
// create_sampler_view RE-ISSUED ON THE SAME HANDLE whenever the restrictions move -
// legal because Gen increments only on slot reuse and never on a respecify.
//
// A DEVIATION FROM THE CONTENT-ADDRESSED 4096-ENTRY CACHE the design gives this kind,
// and it is P3a's vertex-elements deviation for the same reason: MobileGL has no
// frontend sampler-view object at all, so EVERY sampled texture needs one minted, and
// a content-addressed mint per texture per verb is exactly the per-draw cost the
// budget forbids. The content-addressed cache is right when Magma's image-view factory
// takes the CSO over and its VkImageView create-info is what is being addressed.
//
// THE VERSION-FIRST SKIP is the shape version, which is what BumpShapeVersion moves on
// every shape and format change - i.e. on exactly the inputs MGPSamplerView carries.
// The params version rides beside it belt-and-braces: no field of the record depends on
// it, so a wrap of that Uint16 can only cost a skipped re-issue of an identical record.
MGPipeHandle AcquireSamplerView(const ITextureObject& texture, MGPipeHandle textureHandle,
Uint64& payloadBytes) {
const MGPipeHandle handle =
MGPipeSlots().Acquire(MGPipeKind::SamplerViewCso, texture.GetLifetimeId());
const SizeT slot = handle.Slot;
if (slot >= m_viewLatch.size()) m_viewLatch.resize(slot + 1);
ViewLatch& latch = m_viewLatch[slot];
const Uint64 shapeVersion = texture.GetShapeVersion();
const Uint16 paramsVersion = texture.GetTextureParamsVersion();
if (latch.RecordLive && latch.RecordGen == handle.Gen && latch.ShapeVersion == shapeVersion &&
latch.ParamsVersion == paramsVersion && latch.Texture == textureHandle) {
return handle;
}
m_lastView = MGPSamplerView{};
m_lastView.Cso = handle;
m_lastView.Texture = textureHandle;
// THE ALIASING FORMAT. For a glTextureView this is the view's own internal format
// and not the storage owner's, which is the whole point of the call; for an
// ordinary texture it is simply its format.
m_lastView.InternalFormat = static_cast<Uint32>(texture.GetFormat());
m_lastView.Target = static_cast<Uint8>(texture.GetTarget());
// THE FOUR RESTRICTIONS COME FROM ONE PLACE. TextureObjectBase leaves all four at
// 0 for an ordinary texture and glTextureView writes them for a view, and a view
// always has NumLevels >= 1 - so a zero here unambiguously means "no restriction,
// the whole storage" and there is no second spelling of the unrestricted case that
// could disagree with the first.
m_lastView.MinLevel = static_cast<Uint16>(texture.GetViewMinLevel());
m_lastView.NumLevels = static_cast<Uint16>(texture.GetViewNumLevels());
m_lastView.MinLayer = static_cast<Uint16>(texture.GetViewMinLayer());
m_lastView.NumLayers = static_cast<Uint16>(texture.GetViewNumLayers());
m_lastView.Samples = static_cast<Uint16>(texture.GetSamples() < 0 ? 0 : texture.GetSamples());
m_lastView.FixedSampleLocations = texture.HasFixedSampleLocations() ? 1 : 0;
MGPipeApplyCreateSamplerView(m_lastView);
++m_viewCreates;
payloadBytes += sizeof(MGPSamplerView);
latch.RecordLive = true;
latch.RecordGen = handle.Gen;
latch.ShapeVersion = shapeVersion;
latch.ParamsVersion = paramsVersion;
latch.Texture = textureHandle;
return handle;
}
// C-1's question for this kind, and the answer the texture's death helper needs:
// MGPipeEmitSamplerViewCsoDestroyAndFree must not emit delete_sampler_view for a slot
// a backend twin table minted through MGPipeSlots().Acquire while this client was
// never asked to publish anything - which is exactly what a push lane with the sampler
// bit clear runs, and what the applier's resolver counts as a refusal.
Bool RecordIsPublished(MGPipeHandle handle) const {
if (MGPipeHandleIsNull(handle)) return false;
const SizeT slot = handle.Slot;
if (slot >= m_viewLatch.size()) return false;
const ViewLatch& latch = m_viewLatch[slot];
return latch.RecordLive && latch.RecordGen == handle.Gen;
}
void NoteRecordDestroyed(MGPipeHandle handle) {
if (MGPipeHandleIsNull(handle)) return;
const SizeT slot = handle.Slot;
if (slot < m_viewLatch.size() && m_viewLatch[slot].RecordGen == handle.Gen) {
m_viewLatch[slot] = ViewLatch{};
}
}
// The validate point's FreshlyPrimed arm. It clears the PER-CONTEXT memo and NOTHING
// ELSE, and the absence is the rule rather than an oversight (D-J4):
// MGPipeApplierReset is a make-current, so it clears the three unit-set windows - whose
// mirrors are the suppressor slots the validate point invalidates beside this call -
// and it deliberately does NOT drop the object records. The sampler CSOs and the
// sampler views are object records, so re-publishing them here would move their Serial
// for nothing, and no P4a emitter may have a re-publication path.
void Reset() { MGPipeProgramOpaqueUnitsShared().Invalidate(); }
void ResetCounters() { m_viewSets = m_stateSets = m_viewCreates = 0; }
// ---- what a unit case reads. No copy: the emitter builds INTO these and hands the
// applier the same pointers. ----
const MGPSamplerViews& LastSamplerViews() const { return m_lastViews; }
const Array<MGPBoundView, kMGPipeMaxTextureUnits>& LastBoundViews() const { return m_views; }
const MGPSamplerStates& LastSamplerStates() const { return m_lastStates; }
const Array<MGPipeHandle, kMGPipeMaxTextureUnits>& LastSamplerStateHandles() const {
return m_states;
}
const MGPSamplerView& LastCreatedView() const { return m_lastView; }
Uint64 ViewSetCount() const { return m_viewSets; }
Uint64 StateSetCount() const { return m_stateSets; }
Uint64 ViewCreateCount() const { return m_viewCreates; }
private:
struct ViewLatch {
// "Does the applier hold a create_sampler_view record at this slot, for this
// generation, describing this shape". Lives exactly as long as the record does -
// see Reset() for why it is not cleared at a make-current.
Bool RecordLive = false;
Uint32 RecordGen = 0;
Uint64 ShapeVersion = 0;
Uint16 ParamsVersion = 0;
MGPipeHandle Texture = kMGPipeNullHandle;
};
static constexpr Uint32 Min(Uint32 a, Uint32 b) { return a < b ? a : b; }
Array<MGPBoundView, kMGPipeMaxTextureUnits> m_views{};
Array<MGPipeHandle, kMGPipeMaxTextureUnits> m_states{};
MGPSamplerViews m_lastViews{};
MGPSamplerStates m_lastStates{};
MGPSamplerView m_lastView{};
Vector<ViewLatch> m_viewLatch;
Uint64 m_viewSets = 0;
Uint64 m_stateSets = 0;
Uint64 m_viewCreates = 0;
};
inline MGPipeSamplerEmitter& MGPipeSamplerEmitterInstance() {
@@ -36,7 +36,22 @@ namespace MobileGL {
// because the object no longer exists to be passed, and because the lifetime id
// is what the client slot allocator resolves the handle from. No-op unless a
// backend registered the ops (a pull build declares none at all).
NotifyStateObjectDestroyed(MG_Pipe::MGPipeKind::SamplerCso, m_lifetimeId);
//
// P4a D-I1: the notice is no longer raised directly - it is step 2 of the ONE
// client-side death helper for this kind, which emits delete_sampler_state
// first, raises the notice second and frees the slot last. Making the client
// the only death path is what stops a slot leaking under a backend that
// installs no death-ops table at all, and the backend's own notice becomes a
// redundant, idempotent second path rather than the only one.
//
// FOR A CONTENT-ADDRESSED SAMPLER CSO THIS HELPER CORRECTLY FREES NOTHING, and
// that is the design rather than a gap: the CSO belongs to a VALUE, not to this
// object (two identical SamplerObjects share one), so it is allocated with no
// lifetime id, the helper resolves nothing for this one, and the only death
// path for that slot is the CSO cache's LRU eviction - which is client-side and
// therefore backend-neutral on day one. What still goes out, unconditionally
// and exactly as before, is the notice.
MG_Pipe::MGPipeEmitSamplerCsoDestroyAndFree(m_lifetimeId);
}
#endif