mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-12 06:08:30 +09:00
[Feat] (Pipe): content-address sampler states, mint one sampler view per texture and push the three unit sets behind their own content hashes
This commit is contained in:
@@ -33,23 +33,135 @@
|
|||||||
// THIS FILE IS CREATED BY THE CONTRACT COMMIT AND FILLED BY THE PACKAGE THAT OWNS IT - see
|
// THIS FILE IS CREATED BY THE CONTRACT COMMIT AND FILLED BY THE PACKAGE THAT OWNS IT - see
|
||||||
// FramebufferEmit.h for why, in full.
|
// FramebufferEmit.h for why, in full.
|
||||||
#if MOBILEGL_PIPE_PUSH
|
#if MOBILEGL_PIPE_PUSH
|
||||||
|
#include <MG_Impl/Pipe/SamplerEmit.h>
|
||||||
|
#include <MG_Impl/Pipe/SetHashSuppressor.h>
|
||||||
|
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||||
#include <MG_Pipe/MGPipe.h>
|
#include <MG_Pipe/MGPipe.h>
|
||||||
#include <MG_Pipe/PipeApply.h>
|
#include <MG_Pipe/PipeApply.h>
|
||||||
#include <MG_State/GLState/Core.h>
|
#include <MG_State/GLState/Core.h>
|
||||||
|
#include <MG_State/GLState/TextureState/TextureState.h>
|
||||||
|
#include <MG_Util/Metrics/PipeStats.h>
|
||||||
|
|
||||||
|
#include <xxhash.h>
|
||||||
|
|
||||||
namespace MobileGL::MG_Pipe {
|
namespace MobileGL::MG_Pipe {
|
||||||
|
|
||||||
// STUB AT THE CONTRACT COMMIT: emits nothing, returns 0 payload bytes.
|
// D-G3. Over the tail with Start and Count mixed in, the same shape the two sampler sets
|
||||||
|
// use - and it covers InternalFormat and Access because those are live glBindImageTexture
|
||||||
|
// state that the format-less image bake keys on, not decoration.
|
||||||
|
inline Uint64 MGPipeShaderImageSetContentHash(const MGPImageView* entries, Uint32 start, Uint32 count) {
|
||||||
|
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPImageView), 0);
|
||||||
|
hash = MGPipeMixShutter(hash, start);
|
||||||
|
hash = MGPipeMixShutter(hash, count);
|
||||||
|
return hash;
|
||||||
|
}
|
||||||
|
|
||||||
class MGPipeImageEmitter {
|
class MGPipeImageEmitter {
|
||||||
public:
|
public:
|
||||||
using GLContext = MG_State::GLState::GLContext;
|
using GLContext = MG_State::GLState::GLContext;
|
||||||
|
|
||||||
|
// set_shader_images. Start is 0 and Count is the image-unit window described below.
|
||||||
|
//
|
||||||
|
// WHERE THE HIGH-WATER MARK COMES FROM, because the frontend has none and this is the
|
||||||
|
// one place a reader will look for it. DirectGLES keeps g_imageUnitHighWaterMark, but
|
||||||
|
// that is written from inside its own per-unit sync and lives on the far side of the
|
||||||
|
// boundary; TextureState::NoteUnitTouched is the TEXTURE-unit path and
|
||||||
|
// glBindImageTexture does not reach it. Adding a counter to TextureState would edit
|
||||||
|
// another package's file and resize the pull build's object, which G1 forbids outright.
|
||||||
|
//
|
||||||
|
// So the window is derived instead, from the one thing that decides whether an image
|
||||||
|
// unit can matter at all: the highest image unit the CURRENT PROGRAM names, memoised
|
||||||
|
// per program state in SamplerEmit.h's shared inversion, UNIONED with a sticky mark of
|
||||||
|
// every unit this emitter has already described. A program with no image uniforms
|
||||||
|
// gives MaxImageUnit == -1 and, with nothing sticky yet, a window of 0 - which is the
|
||||||
|
// zero early-out, taken BEFORE any hash and before any 192-entry walk, exactly as
|
||||||
|
// property 1 requires. The mark is sticky so that a program which stops naming a unit
|
||||||
|
// does not silently stop describing it: the window only grows, and shrinking it is how
|
||||||
|
// a stale binding would become invisible to the server.
|
||||||
Uint64 EmitShaderImages(GLContext& ctx) {
|
Uint64 EmitShaderImages(GLContext& ctx) {
|
||||||
(void)ctx;
|
const auto& program = ctx.GetProgramForDraw();
|
||||||
return 0;
|
const auto& resolution = MGPipeProgramOpaqueUnitsShared().For(program.get());
|
||||||
|
const Uint32 programWindow =
|
||||||
|
resolution.MaxImageUnit < 0 ? 0u : static_cast<Uint32>(resolution.MaxImageUnit) + 1u;
|
||||||
|
if (programWindow > m_window) m_window = programWindow;
|
||||||
|
const Uint32 count = m_window < kMGPipeMaxImageUnits ? m_window : kMGPipeMaxImageUnits;
|
||||||
|
// PROPERTY 1, and it is one integer test on every draw of every application that
|
||||||
|
// never binds an image.
|
||||||
|
if (count == 0) return 0;
|
||||||
|
|
||||||
|
for (Uint32 unit = 0; unit < count; ++unit) {
|
||||||
|
const auto& binding = ctx.GetImageTextureBinding(static_cast<Int>(unit));
|
||||||
|
MGPImageView& entry = m_entries[unit];
|
||||||
|
entry = MGPImageView{};
|
||||||
|
entry.Unit = unit;
|
||||||
|
entry.Res = binding.Texture ? MGPipeSlots().Acquire(MGPipeKind::Texture,
|
||||||
|
binding.Texture->GetLifetimeId())
|
||||||
|
: kMGPipeNullHandle;
|
||||||
|
// THE APPLICATION's format and access, verbatim. The bind-format recast and the
|
||||||
|
// buffer-texture split view are server-side and stay there; so does
|
||||||
|
// SupportsLayeredImageBinding's rule, which asks the BACKEND target after
|
||||||
|
// MapToBackendTextureTarget and forces layer to 0 for a non-layerable one -
|
||||||
|
// Adreno took a stray layer index literally. A client that pre-applied any of
|
||||||
|
// that would be answering a driver question from the wrong side.
|
||||||
|
entry.InternalFormat = static_cast<Uint32>(binding.Format);
|
||||||
|
entry.Layer = static_cast<Uint32>(binding.Layer);
|
||||||
|
entry.Level = static_cast<Uint16>(binding.Level);
|
||||||
|
entry.Layered = binding.Layered != GL_FALSE ? 1 : 0;
|
||||||
|
entry.Access = static_cast<Uint8>(MGPipeEncodeImageAccess(binding.Access));
|
||||||
|
}
|
||||||
|
|
||||||
|
const Uint64 hash = MGPipeShaderImageSetContentHash(m_entries.data(), 0, count);
|
||||||
|
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetShaderImages, hash)) {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
m_lastImages = MGPShaderImages{};
|
||||||
|
m_lastImages.Start = 0;
|
||||||
|
m_lastImages.Count = count;
|
||||||
|
m_lastImages.ContentHash = hash;
|
||||||
|
MGPipeApplySetShaderImages(m_lastImages, m_entries.data());
|
||||||
|
++m_imageSets;
|
||||||
|
if (MG_Util::PipeStats::Enabled()) {
|
||||||
|
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::ShaderImageEmissions, 1);
|
||||||
|
}
|
||||||
|
return sizeof(MGPShaderImages) + static_cast<Uint64>(count) * sizeof(MGPImageView);
|
||||||
}
|
}
|
||||||
|
|
||||||
void Reset() {}
|
// The validate point's FreshlyPrimed arm. A fresh context is a fresh set of image
|
||||||
|
// bindings, so the sticky window starts over; the suppressor slot this set latches is
|
||||||
|
// invalidated beside this call. There is no record half here at all - set_shader_images
|
||||||
|
// is pure working state and mints no object of its own.
|
||||||
|
void Reset() { m_window = 0; }
|
||||||
|
|
||||||
|
void ResetCounters() { m_imageSets = 0; }
|
||||||
|
|
||||||
|
const MGPShaderImages& LastShaderImages() const { return m_lastImages; }
|
||||||
|
const Array<MGPImageView, kMGPipeMaxImageUnits>& LastImageViews() const { return m_entries; }
|
||||||
|
Uint64 ImageSetCount() const { return m_imageSets; }
|
||||||
|
Uint32 Window() const { return m_window; }
|
||||||
|
|
||||||
|
private:
|
||||||
|
// GL_READ_ONLY / GL_WRITE_ONLY / GL_READ_WRITE folded into the one byte the wire
|
||||||
|
// carries. A value the enum does not name would otherwise truncate silently into a
|
||||||
|
// Uint8, which is the class of bug the descriptors exist to close.
|
||||||
|
static Uint32 MGPipeEncodeImageAccess(GLenum access) {
|
||||||
|
switch (access) {
|
||||||
|
case GL_READ_ONLY:
|
||||||
|
return 0;
|
||||||
|
case GL_WRITE_ONLY:
|
||||||
|
return 1;
|
||||||
|
case GL_READ_WRITE:
|
||||||
|
return 2;
|
||||||
|
default:
|
||||||
|
MOBILEGL_ASSERT(false, "glBindImageTexture access 0x%x is not one of the three GL names",
|
||||||
|
static_cast<Uint>(access));
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Array<MGPImageView, kMGPipeMaxImageUnits> m_entries{};
|
||||||
|
MGPShaderImages m_lastImages{};
|
||||||
|
Uint32 m_window = 0;
|
||||||
|
Uint64 m_imageSets = 0;
|
||||||
};
|
};
|
||||||
|
|
||||||
inline MGPipeImageEmitter& MGPipeImageEmitterInstance() {
|
inline MGPipeImageEmitter& MGPipeImageEmitterInstance() {
|
||||||
|
|||||||
@@ -29,39 +29,706 @@
|
|||||||
// one subsystem bit, because an operator switching samplers off has to get the whole family's
|
// one subsystem bit, because an operator switching samplers off has to get the whole family's
|
||||||
// legacy arm rather than two thirds of it.
|
// legacy arm rather than two thirds of it.
|
||||||
//
|
//
|
||||||
|
// WHAT THIS FILE IS THE CLIENT HALF OF, named so a reader can check it against the oracle:
|
||||||
|
// DirectGLES' ResolveAndBindUnitTextures (the per-unit sampler-view resolution),
|
||||||
|
// BindCurrentUnitSamplers (the per-unit sampler-object walk) and the program pass' sampler
|
||||||
|
// override. TWO BACKEND POST-PROCESSINGS DELIBERATELY STAY ON THE SERVER and act on the
|
||||||
|
// RESOLVED set: Espryt's raw-depth-fetch sampler substitution and Magma's feedback-loop
|
||||||
|
// detection. Neither is reproduced here and neither may be.
|
||||||
|
//
|
||||||
// HEADER-ONLY, for the ownership reason Tracker.h and ResourceTracker.h both state.
|
// HEADER-ONLY, for the ownership reason Tracker.h and ResourceTracker.h both state.
|
||||||
#if MOBILEGL_PIPE_PUSH
|
#if MOBILEGL_PIPE_PUSH
|
||||||
|
#include <MG_Impl/Pipe/SetHashSuppressor.h>
|
||||||
|
#include <MG_Impl/Pipe/SlotAllocator.h>
|
||||||
|
#include <MG_Impl/Pipe/Tracker.h>
|
||||||
#include <MG_Pipe/MGPipe.h>
|
#include <MG_Pipe/MGPipe.h>
|
||||||
|
#include <MG_Pipe/MGPipeHostSpan.h>
|
||||||
#include <MG_Pipe/PipeApply.h>
|
#include <MG_Pipe/PipeApply.h>
|
||||||
#include <MG_State/GLState/Core.h>
|
#include <MG_State/GLState/Core.h>
|
||||||
|
#include <MG_State/GLState/ProgramState/ProgramObject.h>
|
||||||
|
#include <MG_State/GLState/SamplerState/SamplerObject.h>
|
||||||
|
#include <MG_State/GLState/TextureState/TextureObject.h>
|
||||||
|
#include <MG_State/GLState/TextureState/TextureUnit.h>
|
||||||
|
#include <MG_Util/Metrics/PipeStats.h>
|
||||||
|
|
||||||
|
#include <xxhash.h>
|
||||||
|
|
||||||
|
#include <cstring>
|
||||||
|
|
||||||
namespace MobileGL::MG_Pipe {
|
namespace MobileGL::MG_Pipe {
|
||||||
|
|
||||||
// 0 until the emitters below and in ImageEmit.h have bodies; see FramebufferEmit.h's note.
|
// 0 until the emitters below and in ImageEmit.h have bodies; see FramebufferEmit.h's note.
|
||||||
inline constexpr Uint64 kMGPipeWiredSamplerSubsystem = 0;
|
inline constexpr Uint64 kMGPipeWiredSamplerSubsystem = 0;
|
||||||
|
|
||||||
// STUB AT THE CONTRACT COMMIT: emits nothing, returns 0 payload bytes.
|
// ---------------------------------------------------------------------------------
|
||||||
|
// D-F1: the canonical SamplerParameters copy, and why it is not a memcpy
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// sizeof(SamplerParameters) == 100 and its members occupy 97 of those bytes: six 4-byte
|
||||||
|
// enums, four Floats, two more 4-byte enums, three 16-byte colour vectors and the one-byte
|
||||||
|
// borderColorForm. Bytes 97, 98 and 99 are PADDING and no writer ever touches them.
|
||||||
|
//
|
||||||
|
// A cache that hashed or memcmp'd the object's own bytes would therefore read
|
||||||
|
// uninitialised memory. In practice SamplerObject value-initialises its member and never
|
||||||
|
// rewrites it, so in practice those bytes are stable - and "in practice" is not a
|
||||||
|
// contract. The whole point of a content-addressed cache is that a false MISS mints a
|
||||||
|
// fresh CSO per call: a 256-entry cache with a hit rate of zero, on a path nobody looks
|
||||||
|
// at, because the pixels are right either way.
|
||||||
|
//
|
||||||
|
// So every hash and every confirm runs over a copy that is memset to zero FIRST and then
|
||||||
|
// assigned field by field, which makes the padding deterministically zero on both sides of
|
||||||
|
// the comparison. The field list is MGP_FIELDS_SamplerParameters', in its order, and the
|
||||||
|
// verify build compares the same sixteen fields one at a time - without that
|
||||||
|
// PipeFields.def row the blob would be compared as bytes and G4 would be a coin flip.
|
||||||
|
//
|
||||||
|
// ONE COPY PER MINT ATTEMPT, never per draw: the version-first skip in the two emitters
|
||||||
|
// below decides whether to come here at all.
|
||||||
|
inline SamplerParameters MGPipeCanonicalSamplerParameters(const SamplerParameters& src) {
|
||||||
|
static_assert(std::is_trivially_copyable_v<SamplerParameters>,
|
||||||
|
"the canonical copy is memset and then assigned field by field");
|
||||||
|
SamplerParameters canon;
|
||||||
|
std::memset(static_cast<void*>(&canon), 0, sizeof(canon));
|
||||||
|
canon.wrapS = src.wrapS;
|
||||||
|
canon.wrapT = src.wrapT;
|
||||||
|
canon.wrapR = src.wrapR;
|
||||||
|
canon.minFilter = src.minFilter;
|
||||||
|
canon.magFilter = src.magFilter;
|
||||||
|
canon.mipmapMode = src.mipmapMode;
|
||||||
|
canon.minLod = src.minLod;
|
||||||
|
canon.maxLod = src.maxLod;
|
||||||
|
canon.lodBias = src.lodBias;
|
||||||
|
canon.maxAnisotropy = src.maxAnisotropy;
|
||||||
|
canon.compareFunc = src.compareFunc;
|
||||||
|
canon.compareMode = src.compareMode;
|
||||||
|
canon.borderColor = src.borderColor;
|
||||||
|
canon.borderColorI = src.borderColorI;
|
||||||
|
canon.borderColorUI = src.borderColorUI;
|
||||||
|
// ALL FOUR BORDER-COLOUR MEMBERS CROSS, the form included. All three representations
|
||||||
|
// are always numerically populated, so the value alone cannot say which driver entry
|
||||||
|
// point applies (glSamplerParameterIiv vs fv, or which VkBorderColor family), and both
|
||||||
|
// of Espryt's redundancy filters compare all four. Dropping this one line is G7's
|
||||||
|
// scripted negative control and SamplerEmit's suite must go red naming it.
|
||||||
|
canon.borderColorForm = src.borderColorForm;
|
||||||
|
return canon;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline Uint64 MGPipeHashSamplerParameters(const SamplerParameters& canon) {
|
||||||
|
return XXH64(&canon, sizeof(canon), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
// D-F1: the content-addressed sampler CSO cache, capacity 256
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// Unlike P3a's vertex-elements CSO, content addressing is RIGHT here: a SamplerObject is a
|
||||||
|
// pure 100-byte value with no driver-side per-object binding state, two identical samplers
|
||||||
|
// can share one CSO with no extra work on either side, and Espryt's BackendSamplerObject
|
||||||
|
// is a driver name plus a parameter shadow - nothing a second frontend object would have
|
||||||
|
// to re-establish.
|
||||||
|
//
|
||||||
|
// THE LOOKUP IS CsoCache.h's, one for one: hash, probe, and CONFIRM WITH A MEMCMP before
|
||||||
|
// reusing a handle, because a bare 64-bit equality would alias two different sampler
|
||||||
|
// states onto one CSO and that is silent wrong filtering with no gate that can see it.
|
||||||
|
//
|
||||||
|
// CAPACITY 256 IS ALSO A SAFETY BOUND, not only a size. One emission pass acquires at most
|
||||||
|
// kMGPipeMaxTextureUnits == 192 CSOs and touches every one of them, so LRU can only ever
|
||||||
|
// evict an entry from an EARLIER pass - a handle named in the tail of the record being
|
||||||
|
// built right now can never be the victim. The static_assert below is what keeps that true
|
||||||
|
// if either number is ever retuned.
|
||||||
|
//
|
||||||
|
// THE SLOT IS ALLOCATED WITHOUT A LIFETIME ID, deliberately: a content-addressed CSO
|
||||||
|
// belongs to a VALUE and not to a frontend object, so ~SamplerObject must not free it -
|
||||||
|
// another live SamplerObject may hold the same value. MGPipeEmitSamplerCsoDestroyAndFree
|
||||||
|
// resolves nothing for such an id and correctly frees nothing; the only death path for
|
||||||
|
// these slots is the LRU eviction below, which is client-side and therefore
|
||||||
|
// backend-neutral on day one.
|
||||||
|
inline constexpr SizeT kMGPipeSamplerCsoCacheCapacity = 256;
|
||||||
|
static_assert(kMGPipeSamplerCsoCacheCapacity > kMGPipeMaxTextureUnits,
|
||||||
|
"one emission pass touches every unit's CSO, so the cache must be able to "
|
||||||
|
"hold a whole pass without evicting a handle that pass is about to name");
|
||||||
|
|
||||||
|
class MGPipeSamplerCsoCache {
|
||||||
|
public:
|
||||||
|
struct Counters {
|
||||||
|
Uint64 Mints = 0; // create_sampler_state emissions
|
||||||
|
Uint64 Acquisitions = 0;
|
||||||
|
Uint64 Hits = 0; // a probe that found a live entry and passed the memcmp
|
||||||
|
Uint64 Collisions = 0; // a hash hit the memcmp REJECTED - the reason it exists
|
||||||
|
Uint64 Evictions = 0; // LRU evictions, each one a delete_sampler_state
|
||||||
|
};
|
||||||
|
|
||||||
|
// The handle for `params`' value. Mints and emits create_sampler_state on a miss and
|
||||||
|
// emits delete_sampler_state for whatever it evicts to make room. `payloadBytes`
|
||||||
|
// accumulates what went on the wire.
|
||||||
|
//
|
||||||
|
// THIS IS ALSO PACKAGE B's SEAM. MGPTextureParams::BuiltinSampler names the CSO that
|
||||||
|
// carries the SamplerParameters of the SamplerObject every ITextureObject owns, and a
|
||||||
|
// null handle there is Fatal{ProtocolCorruption} rather than "no sampler". TextureEmit.h
|
||||||
|
// acquires it from here, so the texture's built-in sampler and a glBindSampler'd
|
||||||
|
// sampler object with the same value share one CSO and one server-side twin - which is
|
||||||
|
// exactly the sharing that makes content addressing the right answer for this kind.
|
||||||
|
MGPipeHandle Acquire(const SamplerParameters& params, Uint64& payloadBytes) {
|
||||||
|
++m_counters.Acquisitions;
|
||||||
|
const SamplerParameters canon = MGPipeCanonicalSamplerParameters(params);
|
||||||
|
const Uint64 hash = MGPipeHashSamplerParameters(canon);
|
||||||
|
for (SizeT i = 0; i < m_entries.size(); ++i) {
|
||||||
|
if (m_entries[i].Hash != hash) continue;
|
||||||
|
if (std::memcmp(&m_entries[i].Params, &canon, sizeof(canon)) != 0) {
|
||||||
|
// A 64-bit collision between two DIFFERENT sampler states. Reusing the
|
||||||
|
// handle would filter one state with the other's parameters, so the entry
|
||||||
|
// is dropped and the caller mints - correctness first, and the counter
|
||||||
|
// says how often it happened.
|
||||||
|
++m_counters.Collisions;
|
||||||
|
Evict(i);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
m_entries[i].LastUsed = ++m_clock;
|
||||||
|
++m_counters.Hits;
|
||||||
|
return m_entries[i].Cso;
|
||||||
|
}
|
||||||
|
return Mint(hash, canon, payloadBytes);
|
||||||
|
}
|
||||||
|
|
||||||
|
// "Does the applier hold a create_sampler_state record for exactly this handle?" A
|
||||||
|
// slot is not evidence of a record - a backend twin table mints one through
|
||||||
|
// MGPipeSlots().Acquire whether or not this client was ever asked to emit - so the
|
||||||
|
// death paths ask this rather than guessing from the slot.
|
||||||
|
Bool RecordIsPublished(MGPipeHandle handle) const {
|
||||||
|
if (MGPipeHandleIsNull(handle)) return false;
|
||||||
|
for (const Entry& entry : m_entries) {
|
||||||
|
if (entry.Cso == handle) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// A unit test's fixture, and nothing else. NOT called from the validate point's
|
||||||
|
// FreshlyPrimed arm: MGPipeApplierReset is a make-current and does NOT drop object
|
||||||
|
// records, so a sampler CSO the applier holds outlives a context switch. Dropping the
|
||||||
|
// cache there would leak the applier's record and re-mint a value it already has.
|
||||||
|
void ResetForTest() {
|
||||||
|
for (const Entry& entry : m_entries) MGPipeSlots().Free(MGPipeKind::SamplerCso, entry.Cso);
|
||||||
|
m_entries.clear();
|
||||||
|
m_clock = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
void ResetCounters() { m_counters = Counters{}; }
|
||||||
|
SizeT Size() const { return m_entries.size(); }
|
||||||
|
const Counters& GetCounters() const { return m_counters; }
|
||||||
|
|
||||||
|
private:
|
||||||
|
struct Entry {
|
||||||
|
Uint64 Hash = 0;
|
||||||
|
Uint64 LastUsed = 0;
|
||||||
|
MGPipeHandle Cso = kMGPipeNullHandle;
|
||||||
|
// THE CANONICAL BYTES, not the caller's. This is the memcmp's other operand and it
|
||||||
|
// has to have the same deterministic padding the probe's copy has.
|
||||||
|
SamplerParameters Params{};
|
||||||
|
};
|
||||||
|
|
||||||
|
MGPipeHandle Mint(Uint64 hash, const SamplerParameters& canon, Uint64& payloadBytes) {
|
||||||
|
if (m_entries.size() >= kMGPipeSamplerCsoCacheCapacity) {
|
||||||
|
SizeT victim = 0;
|
||||||
|
for (SizeT i = 1; i < m_entries.size(); ++i) {
|
||||||
|
if (m_entries[i].LastUsed < m_entries[victim].LastUsed) victim = i;
|
||||||
|
}
|
||||||
|
Evict(victim);
|
||||||
|
}
|
||||||
|
|
||||||
|
const MGPipeHandle cso = MGPipeSlots().Allocate(MGPipeKind::SamplerCso);
|
||||||
|
MGPSamplerDesc desc{};
|
||||||
|
desc.Cso = cso;
|
||||||
|
// THE ONE BLOB RULE (MGPipeTypes.h): Size 0 means "this record does not declare its
|
||||||
|
// blob", which is what a monolith emission is - the parameters ride beside the
|
||||||
|
// record through the entry point's companion pointer and the applier stores them by
|
||||||
|
// value. Splitting this for a transport is P5's problem, not this emitter's.
|
||||||
|
desc.Parameters.Seg = kMGHostSpanSegNone;
|
||||||
|
desc.Parameters.Offset = 0;
|
||||||
|
desc.Parameters.Size = 0;
|
||||||
|
|
||||||
|
Entry entry;
|
||||||
|
entry.Hash = hash;
|
||||||
|
entry.LastUsed = ++m_clock;
|
||||||
|
entry.Cso = cso;
|
||||||
|
entry.Params = canon;
|
||||||
|
m_entries.push_back(entry);
|
||||||
|
// The applier is handed the CACHE's copy, so the pointer stays valid for the whole
|
||||||
|
// call and the bytes it stores are provably the bytes the memcmp will confirm
|
||||||
|
// against later.
|
||||||
|
MGPipeApplyCreateSamplerState(desc, &m_entries.back().Params);
|
||||||
|
|
||||||
|
++m_counters.Mints;
|
||||||
|
payloadBytes += sizeof(MGPSamplerDesc) + sizeof(SamplerParameters);
|
||||||
|
if (MG_Util::PipeStats::Enabled()) {
|
||||||
|
MG_Util::PipeStats::AddBytes(MG_Util::PipeStats::ByteClass::CsoBlobBytes,
|
||||||
|
sizeof(SamplerParameters));
|
||||||
|
}
|
||||||
|
return cso;
|
||||||
|
}
|
||||||
|
|
||||||
|
void Evict(SizeT index) {
|
||||||
|
MGPHandleOnly handle{};
|
||||||
|
handle.Handle = m_entries[index].Cso;
|
||||||
|
handle.Kind = static_cast<Uint32>(MGPipeKind::SamplerCso);
|
||||||
|
// THE THREE-STEP ORDER, the same one PipeMutation.h fixes for the death helpers:
|
||||||
|
// the wire delete drops the applier's record while the record still exists, and
|
||||||
|
// only then does the slot go back. There is no NotifyStateObjectDestroyed step
|
||||||
|
// here - a content-addressed CSO has no frontend object whose death is being
|
||||||
|
// announced, which is precisely why this eviction is the only death path it has.
|
||||||
|
MGPipeApplyDeleteSamplerState(handle);
|
||||||
|
MGPipeSlots().Free(MGPipeKind::SamplerCso, m_entries[index].Cso);
|
||||||
|
m_entries[index] = m_entries.back();
|
||||||
|
m_entries.pop_back();
|
||||||
|
++m_counters.Evictions;
|
||||||
|
}
|
||||||
|
|
||||||
|
Vector<Entry> m_entries;
|
||||||
|
Uint64 m_clock = 0;
|
||||||
|
Counters m_counters;
|
||||||
|
};
|
||||||
|
|
||||||
|
inline MGPipeSamplerCsoCache& MGPipeSamplerCsoCacheInstance() {
|
||||||
|
// NEVER DESTROYED, for MGPipeTrackerInstance()' reason, and this one is on a death
|
||||||
|
// path: TextureEmit.h asks it for every texture's built-in sampler CSO, so an exit
|
||||||
|
// handler running a frontend destructor must not find it freed.
|
||||||
|
static MGPipeSamplerCsoCache* cache = new MGPipeSamplerCsoCache();
|
||||||
|
return *cache;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
// D-F3: the client-side sampling resolution
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// GL binds one texture per unit PER TARGET; which one the shader actually samples depends
|
||||||
|
// on the sampler uniform's TYPE. Gallium's one-view-per-slot is the resolved form, so the
|
||||||
|
// resolution moves to the client and the record carries the answer rather than the inputs.
|
||||||
|
//
|
||||||
|
// This is DirectGLES' SamplerUniformTextureTarget, moved to the side of the boundary that
|
||||||
|
// now owns the question. Targets with no sampler spelling map to Unknown, and a unit whose
|
||||||
|
// uniform type resolves to Unknown carries no view - which is exactly "the program does not
|
||||||
|
// sample this unit".
|
||||||
|
inline MobileGL::TextureTarget MGPipeSamplerUniformTextureTarget(GLenum uniformType) {
|
||||||
|
switch (uniformType) {
|
||||||
|
case GL_SAMPLER_1D:
|
||||||
|
case GL_INT_SAMPLER_1D:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_1D:
|
||||||
|
case GL_SAMPLER_1D_SHADOW:
|
||||||
|
return TextureTarget::Texture1D;
|
||||||
|
case GL_SAMPLER_2D:
|
||||||
|
case GL_INT_SAMPLER_2D:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_2D:
|
||||||
|
case GL_SAMPLER_2D_SHADOW:
|
||||||
|
return TextureTarget::Texture2D;
|
||||||
|
case GL_SAMPLER_3D:
|
||||||
|
case GL_INT_SAMPLER_3D:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_3D:
|
||||||
|
return TextureTarget::Texture3D;
|
||||||
|
case GL_SAMPLER_CUBE:
|
||||||
|
case GL_INT_SAMPLER_CUBE:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_CUBE:
|
||||||
|
case GL_SAMPLER_CUBE_SHADOW:
|
||||||
|
return TextureTarget::TextureCubeMap;
|
||||||
|
case GL_SAMPLER_1D_ARRAY:
|
||||||
|
case GL_INT_SAMPLER_1D_ARRAY:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_1D_ARRAY:
|
||||||
|
case GL_SAMPLER_1D_ARRAY_SHADOW:
|
||||||
|
return TextureTarget::Texture1DArray;
|
||||||
|
case GL_SAMPLER_2D_ARRAY:
|
||||||
|
case GL_INT_SAMPLER_2D_ARRAY:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_2D_ARRAY:
|
||||||
|
case GL_SAMPLER_2D_ARRAY_SHADOW:
|
||||||
|
return TextureTarget::Texture2DArray;
|
||||||
|
case GL_SAMPLER_CUBE_MAP_ARRAY:
|
||||||
|
case GL_INT_SAMPLER_CUBE_MAP_ARRAY:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_CUBE_MAP_ARRAY:
|
||||||
|
case GL_SAMPLER_CUBE_MAP_ARRAY_SHADOW:
|
||||||
|
return TextureTarget::TextureCubeMapArray;
|
||||||
|
case GL_SAMPLER_2D_RECT:
|
||||||
|
case GL_INT_SAMPLER_2D_RECT:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_2D_RECT:
|
||||||
|
case GL_SAMPLER_2D_RECT_SHADOW:
|
||||||
|
return TextureTarget::TextureRectangle;
|
||||||
|
case GL_SAMPLER_2D_MULTISAMPLE:
|
||||||
|
case GL_INT_SAMPLER_2D_MULTISAMPLE:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE:
|
||||||
|
return TextureTarget::Texture2DMultisample;
|
||||||
|
case GL_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||||
|
case GL_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_2D_MULTISAMPLE_ARRAY:
|
||||||
|
return TextureTarget::Texture2DMultisampleArray;
|
||||||
|
case GL_SAMPLER_BUFFER:
|
||||||
|
case GL_INT_SAMPLER_BUFFER:
|
||||||
|
case GL_UNSIGNED_INT_SAMPLER_BUFFER:
|
||||||
|
return TextureTarget::TextureBuffer;
|
||||||
|
default:
|
||||||
|
return TextureTarget::Unknown;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unit -> sampled target, and the highest image unit the program names, both INVERTED ONCE
|
||||||
|
// per program state rather than searched per unit.
|
||||||
|
//
|
||||||
|
// DirectGLES asks the question the other way round - for one unit it walks every uniform
|
||||||
|
// location - and gets away with it because it only asks on an aliasing conflict. A client
|
||||||
|
// that asked per unit would be O(units x locations) at every verb the sampler bit fires
|
||||||
|
// on, which is exactly the per-draw cost the phase's budget forbids. So the walk runs once
|
||||||
|
// and is memoised on (lifetime id, link version, backend state version): the third is the
|
||||||
|
// counter SetUniformSamplerOrImageUnitIndex bumps, i.e. the one thing that can move a
|
||||||
|
// uniform's unit without relinking.
|
||||||
|
//
|
||||||
|
// IT IS SHARED WITH ImageEmit.h on purpose. The sampler and image halves come out of one
|
||||||
|
// walk over one array, they ride one subsystem bit, and computing them separately would
|
||||||
|
// walk the same locations twice per program change.
|
||||||
|
class MGPipeProgramOpaqueUnits {
|
||||||
|
public:
|
||||||
|
using ProgramObject = MG_State::GLState::ProgramObject;
|
||||||
|
|
||||||
|
struct Resolution {
|
||||||
|
// TextureTarget + 1 per unit, 0 meaning "this program samples nothing here". A
|
||||||
|
// Uint8 because TextureTargetCount is 11 and this array is 192 entries long on a
|
||||||
|
// path that wants to stay in cache.
|
||||||
|
Array<Uint8, kMGPipeMaxTextureUnits> SamplerTarget{};
|
||||||
|
// Highest image unit the program names, or -1 for a program with no image
|
||||||
|
// uniforms - which is what gives ImageEmit.h its zero early-out for free.
|
||||||
|
Int32 MaxImageUnit = -1;
|
||||||
|
};
|
||||||
|
|
||||||
|
const Resolution& For(const ProgramObject* program) {
|
||||||
|
const Uint64 lifetimeId = program != nullptr ? program->GetLifetimeId() : 0;
|
||||||
|
const Uint32 linkVersion = program != nullptr ? program->GetLinkVersion() : 0;
|
||||||
|
const Uint32 stateVersion = program != nullptr ? program->GetBackendStateVersion() : 0;
|
||||||
|
if (m_valid && m_lifetimeId == lifetimeId && m_linkVersion == linkVersion &&
|
||||||
|
m_stateVersion == stateVersion) {
|
||||||
|
return m_resolution;
|
||||||
|
}
|
||||||
|
m_resolution = Resolution{};
|
||||||
|
// GetLinkStatus is the same guard DirectGLES uses before it trusts the reflection:
|
||||||
|
// a program that did not link has no usable uniform table, and every unit resolves
|
||||||
|
// to "not sampled".
|
||||||
|
if (program != nullptr && program->GetLinkStatus()) {
|
||||||
|
const Uint maxLocation = program->GetMaxUniformLocation();
|
||||||
|
for (Uint location = 0; location <= maxLocation; ++location) {
|
||||||
|
const Int unit = program->GetUniformSamplerOrImageUnitIndex(location);
|
||||||
|
if (unit < 0 || unit >= static_cast<Int>(kMGPipeMaxTextureUnits)) continue;
|
||||||
|
if (program->GetUniformTypeFacts(location).isImage) {
|
||||||
|
if (unit > m_resolution.MaxImageUnit) m_resolution.MaxImageUnit = unit;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const TextureTarget target =
|
||||||
|
MGPipeSamplerUniformTextureTarget(program->GetUniformType(location));
|
||||||
|
if (target == TextureTarget::Unknown) continue;
|
||||||
|
// FIRST WRITER WINS, which is DirectGLES' arbitration read forwards: when
|
||||||
|
// two sampler uniforms of different types share a unit the binding placed
|
||||||
|
// first stands rather than being silently overwritten by whichever location
|
||||||
|
// comes last.
|
||||||
|
Uint8& slot = m_resolution.SamplerTarget[static_cast<SizeT>(unit)];
|
||||||
|
if (slot == 0) slot = static_cast<Uint8>(static_cast<Int>(target) + 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
m_lifetimeId = lifetimeId;
|
||||||
|
m_linkVersion = linkVersion;
|
||||||
|
m_stateVersion = stateVersion;
|
||||||
|
m_valid = true;
|
||||||
|
return m_resolution;
|
||||||
|
}
|
||||||
|
|
||||||
|
void Invalidate() { m_valid = false; }
|
||||||
|
|
||||||
|
private:
|
||||||
|
Resolution m_resolution;
|
||||||
|
Uint64 m_lifetimeId = 0;
|
||||||
|
Uint32 m_linkVersion = 0;
|
||||||
|
Uint32 m_stateVersion = 0;
|
||||||
|
Bool m_valid = false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// ONE inversion for the whole family, shared by MGPipeSamplerEmitter and
|
||||||
|
// MGPipeImageEmitter. Two memos would walk the same uniform table twice per program
|
||||||
|
// change, and - worse - could disagree about which locations they saw, which is how a
|
||||||
|
// sampler unit and an image unit come to be resolved against two different readings of one
|
||||||
|
// program. Never destroyed, like every other MGPipe process singleton.
|
||||||
|
inline MGPipeProgramOpaqueUnits& MGPipeProgramOpaqueUnitsShared() {
|
||||||
|
static MGPipeProgramOpaqueUnits* units = new MGPipeProgramOpaqueUnits();
|
||||||
|
return *units;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
// D-G3: the two unit sets' content hashes
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// XXH64 over the tail entries, with Start and Count mixed in through MGPipeMixShutter -
|
||||||
|
// the VertexInputEmit shape, and for its reason: the hash must cover EVERY input the
|
||||||
|
// record carries, or a record whose one changed field is outside the tail gets suppressed.
|
||||||
|
// Both tails are built into zero-initialised staging arrays, so no padding byte enters
|
||||||
|
// either hash.
|
||||||
|
inline Uint64 MGPipeSamplerViewSetContentHash(const MGPBoundView* entries, Uint32 start, Uint32 count) {
|
||||||
|
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPBoundView), 0);
|
||||||
|
hash = MGPipeMixShutter(hash, start);
|
||||||
|
hash = MGPipeMixShutter(hash, count);
|
||||||
|
return hash;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline Uint64 MGPipeSamplerStateSetContentHash(const MGPipeHandle* entries, Uint32 start, Uint32 count) {
|
||||||
|
Uint64 hash = XXH64(entries, static_cast<SizeT>(count) * sizeof(MGPipeHandle), 0);
|
||||||
|
hash = MGPipeMixShutter(hash, start);
|
||||||
|
hash = MGPipeMixShutter(hash, count);
|
||||||
|
return hash;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
// The emitter
|
||||||
|
// ---------------------------------------------------------------------------------
|
||||||
|
|
||||||
class MGPipeSamplerEmitter {
|
class MGPipeSamplerEmitter {
|
||||||
public:
|
public:
|
||||||
using GLContext = MG_State::GLState::GLContext;
|
using GLContext = MG_State::GLState::GLContext;
|
||||||
|
using ITextureObject = MG_State::GLState::ITextureObject;
|
||||||
|
using SamplerObject = MG_State::GLState::SamplerObject;
|
||||||
|
|
||||||
// set_sampler_views: the PROGRAM-RESOLVED set only, one entry per unit, no stage
|
// set_sampler_views: the PROGRAM-RESOLVED set only, one entry per unit, no stage
|
||||||
// dimension. Start is 0 and Count is GetMaxTouchedTextureUnit() + 1 clamped to the
|
// dimension. Start is 0 and Count is GetMaxTouchedTextureUnit() + 1 clamped to the
|
||||||
// wire bound - the high-water mark is directly the count argument and is not
|
// wire bound - the high-water mark is directly the count argument and is not
|
||||||
// re-derived.
|
// re-derived.
|
||||||
|
//
|
||||||
|
// WHAT A UNIT'S ENTRY MEANS, and all three cases are legal rather than holes:
|
||||||
|
// the program resolves no target here -> View and Texture both null
|
||||||
|
// it resolves one, nothing is bound -> View and Texture both null
|
||||||
|
// it resolves one and a texture is bound-> Texture is that object's handle and View
|
||||||
|
// is its sampler-view CSO
|
||||||
|
// An UNDEFINED DEFAULT texture (name 0 with no image) and a texture that
|
||||||
|
// SAMPLES AS INCOMPLETE are both dropped to null here, which is the resolution
|
||||||
|
// DirectGLES performs by leaving the native target unbound: an incomplete texture
|
||||||
|
// samples (0,0,0,1) and the driver cannot work that out for itself, because the
|
||||||
|
// backend storage is immutable and never saw the level the application redefined at
|
||||||
|
// the wrong size.
|
||||||
Uint64 EmitSamplerViews(GLContext& ctx) {
|
Uint64 EmitSamplerViews(GLContext& ctx) {
|
||||||
(void)ctx;
|
const Int maxTouched = ctx.GetMaxTouchedTextureUnit();
|
||||||
return 0;
|
const Uint32 count =
|
||||||
|
maxTouched < 0 ? 0u
|
||||||
|
: Min(static_cast<Uint32>(maxTouched) + 1u, kMGPipeMaxTextureUnits);
|
||||||
|
|
||||||
|
// The join is the EMITTER's, deliberately, and it is the same GetProgramForDraw()
|
||||||
|
// the verb is about to make anyway: the tracker's own shutters read
|
||||||
|
// GetCurrentProgram() precisely so that answering "did the shader move" never
|
||||||
|
// forces a compile. In the ladder EmitShaderState runs before this, so in the
|
||||||
|
// steady state the program is already joined by the time this line runs.
|
||||||
|
const auto& program = ctx.GetProgramForDraw();
|
||||||
|
const auto& resolution = MGPipeProgramOpaqueUnitsShared().For(program.get());
|
||||||
|
|
||||||
|
Uint64 bytes = 0;
|
||||||
|
for (Uint32 unit = 0; unit < count; ++unit) {
|
||||||
|
MGPBoundView& entry = m_views[unit];
|
||||||
|
entry = MGPBoundView{};
|
||||||
|
entry.Unit = unit;
|
||||||
|
entry.View = kMGPipeNullHandle;
|
||||||
|
entry.Texture = kMGPipeNullHandle;
|
||||||
|
|
||||||
|
const Uint8 encoded = resolution.SamplerTarget[unit];
|
||||||
|
if (encoded == 0) continue;
|
||||||
|
const auto target = static_cast<TextureTarget>(static_cast<Int>(encoded) - 1);
|
||||||
|
|
||||||
|
auto& textureUnit = ctx.GetTextureUnitObject(static_cast<Int>(unit));
|
||||||
|
const auto& texture = textureUnit.GetBindingSlot(target).GetBoundObject();
|
||||||
|
if (!texture) continue;
|
||||||
|
if (MG_State::GLState::IsUndefinedDefaultTexture(texture.get())) continue;
|
||||||
|
// The unit's sampler object overrides the texture's own, exactly as in GL, and
|
||||||
|
// the completeness answer depends on which one applies - a mipmap mode of None
|
||||||
|
// makes a single-level texture complete that would otherwise sample black.
|
||||||
|
const auto& unitSampler = textureUnit.GetSamplerObject();
|
||||||
|
const SamplerObject* effective =
|
||||||
|
unitSampler ? unitSampler.get() : texture->GetSamplerObject().get();
|
||||||
|
if (MG_State::GLState::SamplesAsIncompleteTexture(texture.get(), effective)) continue;
|
||||||
|
|
||||||
|
entry.Texture = MGPipeSlots().Acquire(MGPipeKind::Texture, texture->GetLifetimeId());
|
||||||
|
entry.View = AcquireSamplerView(*texture, entry.Texture, bytes);
|
||||||
|
}
|
||||||
|
|
||||||
|
const Uint64 hash = MGPipeSamplerViewSetContentHash(m_views.data(), 0, count);
|
||||||
|
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::SetSamplerViews, hash)) {
|
||||||
|
return bytes;
|
||||||
|
}
|
||||||
|
m_lastViews = MGPSamplerViews{};
|
||||||
|
m_lastViews.Start = 0;
|
||||||
|
m_lastViews.Count = count;
|
||||||
|
m_lastViews.ContentHash = hash;
|
||||||
|
MGPipeApplySetSamplerViews(m_lastViews, m_views.data());
|
||||||
|
++m_viewSets;
|
||||||
|
if (MG_Util::PipeStats::Enabled()) {
|
||||||
|
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::SamplerViewEmissions, 1);
|
||||||
|
}
|
||||||
|
return bytes + sizeof(MGPSamplerViews) + static_cast<Uint64>(count) * sizeof(MGPBoundView);
|
||||||
}
|
}
|
||||||
|
|
||||||
// bind_sampler_states: the unit's sampler CSO, or the null handle when the unit has no
|
// bind_sampler_states: the unit's sampler CSO, or the null handle when the unit has no
|
||||||
// sampler object - the texture's built-in sampler then applies, exactly as today.
|
// sampler object - the texture's built-in sampler then applies, exactly as today.
|
||||||
|
//
|
||||||
|
// NOT program-resolved, and that asymmetry with the view set above is deliberate: a
|
||||||
|
// sampler object is bound to a UNIT and applies to whatever the unit holds, so its set
|
||||||
|
// is the plain per-unit walk BindCurrentUnitSamplers already does. A redundant re-bind
|
||||||
|
// of the same sampler object emits nothing at all, which is the whole point of the
|
||||||
|
// suppressor slot: GetTextureBindGeneration() bumps on a redundant re-bind - 26.2
|
||||||
|
// rebinds the same sampler at every texture-unit switch - so without it this would be
|
||||||
|
// a several-hundred-byte variable-length record per batch.
|
||||||
Uint64 EmitSamplerStates(GLContext& ctx) {
|
Uint64 EmitSamplerStates(GLContext& ctx) {
|
||||||
(void)ctx;
|
const Int maxTouched = ctx.GetMaxTouchedTextureUnit();
|
||||||
return 0;
|
const Uint32 count =
|
||||||
|
maxTouched < 0 ? 0u
|
||||||
|
: Min(static_cast<Uint32>(maxTouched) + 1u, kMGPipeMaxTextureUnits);
|
||||||
|
|
||||||
|
Uint64 bytes = 0;
|
||||||
|
for (Uint32 unit = 0; unit < count; ++unit) {
|
||||||
|
const auto& sampler = ctx.GetTextureUnitObject(static_cast<Int>(unit)).GetSamplerObject();
|
||||||
|
m_states[unit] = sampler ? MGPipeSamplerCsoCacheInstance().Acquire(
|
||||||
|
sampler->GetAllSamplerParameters(), bytes)
|
||||||
|
: kMGPipeNullHandle;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Uint64 hash = MGPipeSamplerStateSetContentHash(m_states.data(), 0, count);
|
||||||
|
if (!MGPipeSetHashSuppressorInstance().ShouldEmit(MGPipeSuppressorSlot::BindSamplerStates, hash)) {
|
||||||
|
return bytes;
|
||||||
|
}
|
||||||
|
m_lastStates = MGPSamplerStates{};
|
||||||
|
m_lastStates.Start = 0;
|
||||||
|
m_lastStates.Count = count;
|
||||||
|
m_lastStates.ContentHash = hash;
|
||||||
|
MGPipeApplyBindSamplerStates(m_lastStates, m_states.data());
|
||||||
|
++m_stateSets;
|
||||||
|
if (MG_Util::PipeStats::Enabled()) {
|
||||||
|
MG_Util::PipeStats::AddCalls(MG_Util::PipeStats::CallClass::SamplerStateEmissions, 1);
|
||||||
|
}
|
||||||
|
return bytes + sizeof(MGPSamplerStates) + static_cast<Uint64>(count) * sizeof(MGPipeHandle);
|
||||||
}
|
}
|
||||||
|
|
||||||
void Reset() {}
|
// D-F2: ONE SamplerViewCso PER ITextureObject, minted off its lifetime id, and
|
||||||
|
// create_sampler_view RE-ISSUED ON THE SAME HANDLE whenever the restrictions move -
|
||||||
|
// legal because Gen increments only on slot reuse and never on a respecify.
|
||||||
|
//
|
||||||
|
// A DEVIATION FROM THE CONTENT-ADDRESSED 4096-ENTRY CACHE the design gives this kind,
|
||||||
|
// and it is P3a's vertex-elements deviation for the same reason: MobileGL has no
|
||||||
|
// frontend sampler-view object at all, so EVERY sampled texture needs one minted, and
|
||||||
|
// a content-addressed mint per texture per verb is exactly the per-draw cost the
|
||||||
|
// budget forbids. The content-addressed cache is right when Magma's image-view factory
|
||||||
|
// takes the CSO over and its VkImageView create-info is what is being addressed.
|
||||||
|
//
|
||||||
|
// THE VERSION-FIRST SKIP is the shape version, which is what BumpShapeVersion moves on
|
||||||
|
// every shape and format change - i.e. on exactly the inputs MGPSamplerView carries.
|
||||||
|
// The params version rides beside it belt-and-braces: no field of the record depends on
|
||||||
|
// it, so a wrap of that Uint16 can only cost a skipped re-issue of an identical record.
|
||||||
|
MGPipeHandle AcquireSamplerView(const ITextureObject& texture, MGPipeHandle textureHandle,
|
||||||
|
Uint64& payloadBytes) {
|
||||||
|
const MGPipeHandle handle =
|
||||||
|
MGPipeSlots().Acquire(MGPipeKind::SamplerViewCso, texture.GetLifetimeId());
|
||||||
|
const SizeT slot = handle.Slot;
|
||||||
|
if (slot >= m_viewLatch.size()) m_viewLatch.resize(slot + 1);
|
||||||
|
ViewLatch& latch = m_viewLatch[slot];
|
||||||
|
|
||||||
|
const Uint64 shapeVersion = texture.GetShapeVersion();
|
||||||
|
const Uint16 paramsVersion = texture.GetTextureParamsVersion();
|
||||||
|
if (latch.RecordLive && latch.RecordGen == handle.Gen && latch.ShapeVersion == shapeVersion &&
|
||||||
|
latch.ParamsVersion == paramsVersion && latch.Texture == textureHandle) {
|
||||||
|
return handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
m_lastView = MGPSamplerView{};
|
||||||
|
m_lastView.Cso = handle;
|
||||||
|
m_lastView.Texture = textureHandle;
|
||||||
|
// THE ALIASING FORMAT. For a glTextureView this is the view's own internal format
|
||||||
|
// and not the storage owner's, which is the whole point of the call; for an
|
||||||
|
// ordinary texture it is simply its format.
|
||||||
|
m_lastView.InternalFormat = static_cast<Uint32>(texture.GetFormat());
|
||||||
|
m_lastView.Target = static_cast<Uint8>(texture.GetTarget());
|
||||||
|
// THE FOUR RESTRICTIONS COME FROM ONE PLACE. TextureObjectBase leaves all four at
|
||||||
|
// 0 for an ordinary texture and glTextureView writes them for a view, and a view
|
||||||
|
// always has NumLevels >= 1 - so a zero here unambiguously means "no restriction,
|
||||||
|
// the whole storage" and there is no second spelling of the unrestricted case that
|
||||||
|
// could disagree with the first.
|
||||||
|
m_lastView.MinLevel = static_cast<Uint16>(texture.GetViewMinLevel());
|
||||||
|
m_lastView.NumLevels = static_cast<Uint16>(texture.GetViewNumLevels());
|
||||||
|
m_lastView.MinLayer = static_cast<Uint16>(texture.GetViewMinLayer());
|
||||||
|
m_lastView.NumLayers = static_cast<Uint16>(texture.GetViewNumLayers());
|
||||||
|
m_lastView.Samples = static_cast<Uint16>(texture.GetSamples() < 0 ? 0 : texture.GetSamples());
|
||||||
|
m_lastView.FixedSampleLocations = texture.HasFixedSampleLocations() ? 1 : 0;
|
||||||
|
MGPipeApplyCreateSamplerView(m_lastView);
|
||||||
|
++m_viewCreates;
|
||||||
|
payloadBytes += sizeof(MGPSamplerView);
|
||||||
|
|
||||||
|
latch.RecordLive = true;
|
||||||
|
latch.RecordGen = handle.Gen;
|
||||||
|
latch.ShapeVersion = shapeVersion;
|
||||||
|
latch.ParamsVersion = paramsVersion;
|
||||||
|
latch.Texture = textureHandle;
|
||||||
|
return handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
// C-1's question for this kind, and the answer the texture's death helper needs:
|
||||||
|
// MGPipeEmitSamplerViewCsoDestroyAndFree must not emit delete_sampler_view for a slot
|
||||||
|
// a backend twin table minted through MGPipeSlots().Acquire while this client was
|
||||||
|
// never asked to publish anything - which is exactly what a push lane with the sampler
|
||||||
|
// bit clear runs, and what the applier's resolver counts as a refusal.
|
||||||
|
Bool RecordIsPublished(MGPipeHandle handle) const {
|
||||||
|
if (MGPipeHandleIsNull(handle)) return false;
|
||||||
|
const SizeT slot = handle.Slot;
|
||||||
|
if (slot >= m_viewLatch.size()) return false;
|
||||||
|
const ViewLatch& latch = m_viewLatch[slot];
|
||||||
|
return latch.RecordLive && latch.RecordGen == handle.Gen;
|
||||||
|
}
|
||||||
|
|
||||||
|
void NoteRecordDestroyed(MGPipeHandle handle) {
|
||||||
|
if (MGPipeHandleIsNull(handle)) return;
|
||||||
|
const SizeT slot = handle.Slot;
|
||||||
|
if (slot < m_viewLatch.size() && m_viewLatch[slot].RecordGen == handle.Gen) {
|
||||||
|
m_viewLatch[slot] = ViewLatch{};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The validate point's FreshlyPrimed arm. It clears the PER-CONTEXT memo and NOTHING
|
||||||
|
// ELSE, and the absence is the rule rather than an oversight (D-J4):
|
||||||
|
// MGPipeApplierReset is a make-current, so it clears the three unit-set windows - whose
|
||||||
|
// mirrors are the suppressor slots the validate point invalidates beside this call -
|
||||||
|
// and it deliberately does NOT drop the object records. The sampler CSOs and the
|
||||||
|
// sampler views are object records, so re-publishing them here would move their Serial
|
||||||
|
// for nothing, and no P4a emitter may have a re-publication path.
|
||||||
|
void Reset() { MGPipeProgramOpaqueUnitsShared().Invalidate(); }
|
||||||
|
|
||||||
|
void ResetCounters() { m_viewSets = m_stateSets = m_viewCreates = 0; }
|
||||||
|
|
||||||
|
// ---- what a unit case reads. No copy: the emitter builds INTO these and hands the
|
||||||
|
// applier the same pointers. ----
|
||||||
|
const MGPSamplerViews& LastSamplerViews() const { return m_lastViews; }
|
||||||
|
const Array<MGPBoundView, kMGPipeMaxTextureUnits>& LastBoundViews() const { return m_views; }
|
||||||
|
const MGPSamplerStates& LastSamplerStates() const { return m_lastStates; }
|
||||||
|
const Array<MGPipeHandle, kMGPipeMaxTextureUnits>& LastSamplerStateHandles() const {
|
||||||
|
return m_states;
|
||||||
|
}
|
||||||
|
const MGPSamplerView& LastCreatedView() const { return m_lastView; }
|
||||||
|
Uint64 ViewSetCount() const { return m_viewSets; }
|
||||||
|
Uint64 StateSetCount() const { return m_stateSets; }
|
||||||
|
Uint64 ViewCreateCount() const { return m_viewCreates; }
|
||||||
|
|
||||||
|
private:
|
||||||
|
struct ViewLatch {
|
||||||
|
// "Does the applier hold a create_sampler_view record at this slot, for this
|
||||||
|
// generation, describing this shape". Lives exactly as long as the record does -
|
||||||
|
// see Reset() for why it is not cleared at a make-current.
|
||||||
|
Bool RecordLive = false;
|
||||||
|
Uint32 RecordGen = 0;
|
||||||
|
Uint64 ShapeVersion = 0;
|
||||||
|
Uint16 ParamsVersion = 0;
|
||||||
|
MGPipeHandle Texture = kMGPipeNullHandle;
|
||||||
|
};
|
||||||
|
|
||||||
|
static constexpr Uint32 Min(Uint32 a, Uint32 b) { return a < b ? a : b; }
|
||||||
|
|
||||||
|
Array<MGPBoundView, kMGPipeMaxTextureUnits> m_views{};
|
||||||
|
Array<MGPipeHandle, kMGPipeMaxTextureUnits> m_states{};
|
||||||
|
MGPSamplerViews m_lastViews{};
|
||||||
|
MGPSamplerStates m_lastStates{};
|
||||||
|
MGPSamplerView m_lastView{};
|
||||||
|
|
||||||
|
Vector<ViewLatch> m_viewLatch;
|
||||||
|
|
||||||
|
Uint64 m_viewSets = 0;
|
||||||
|
Uint64 m_stateSets = 0;
|
||||||
|
Uint64 m_viewCreates = 0;
|
||||||
};
|
};
|
||||||
|
|
||||||
inline MGPipeSamplerEmitter& MGPipeSamplerEmitterInstance() {
|
inline MGPipeSamplerEmitter& MGPipeSamplerEmitterInstance() {
|
||||||
|
|||||||
@@ -36,7 +36,22 @@ namespace MobileGL {
|
|||||||
// because the object no longer exists to be passed, and because the lifetime id
|
// because the object no longer exists to be passed, and because the lifetime id
|
||||||
// is what the client slot allocator resolves the handle from. No-op unless a
|
// is what the client slot allocator resolves the handle from. No-op unless a
|
||||||
// backend registered the ops (a pull build declares none at all).
|
// backend registered the ops (a pull build declares none at all).
|
||||||
NotifyStateObjectDestroyed(MG_Pipe::MGPipeKind::SamplerCso, m_lifetimeId);
|
//
|
||||||
|
// P4a D-I1: the notice is no longer raised directly - it is step 2 of the ONE
|
||||||
|
// client-side death helper for this kind, which emits delete_sampler_state
|
||||||
|
// first, raises the notice second and frees the slot last. Making the client
|
||||||
|
// the only death path is what stops a slot leaking under a backend that
|
||||||
|
// installs no death-ops table at all, and the backend's own notice becomes a
|
||||||
|
// redundant, idempotent second path rather than the only one.
|
||||||
|
//
|
||||||
|
// FOR A CONTENT-ADDRESSED SAMPLER CSO THIS HELPER CORRECTLY FREES NOTHING, and
|
||||||
|
// that is the design rather than a gap: the CSO belongs to a VALUE, not to this
|
||||||
|
// object (two identical SamplerObjects share one), so it is allocated with no
|
||||||
|
// lifetime id, the helper resolves nothing for this one, and the only death
|
||||||
|
// path for that slot is the CSO cache's LRU eviction - which is client-side and
|
||||||
|
// therefore backend-neutral on day one. What still goes out, unconditionally
|
||||||
|
// and exactly as before, is the notice.
|
||||||
|
MG_Pipe::MGPipeEmitSamplerCsoDestroyAndFree(m_lifetimeId);
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user