mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-11 21:58:31 +09:00
[Fix, Test] (ShaderTranspiler, DirectGLES): fold or lower dynamically indexed fragment outputs before ESSL emission - GLSL ES requires constant integral indices, so the OIT coefficient shader linked nothing on ANGLE and every translucent draw was a silent no-op
This commit is contained in:
@@ -27,6 +27,7 @@
|
||||
#include "SpirvPasses/StripUboMemberRelaxedPrecisionPass.h"
|
||||
#include "SpirvPasses/StripNoPerspectivePass.h"
|
||||
#include "SpirvPasses/EmulateNoPerspectivePass.h"
|
||||
#include "SpirvPasses/LegalizeFragmentOutputIndexPass.h"
|
||||
#include "spirv-tools/libspirv.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
@@ -631,6 +632,75 @@ namespace MobileGL {
|
||||
outputBinary);
|
||||
}
|
||||
|
||||
bool ShaderCompiler::LegalizeFragmentOutputIndexingForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
|
||||
// Detection gates everything: a module with no dynamically indexed fragment
|
||||
// output - every shader but a handful - pays one BuildModule and is handed
|
||||
// back byte for byte, so the folding chain can never perturb a shader that
|
||||
// did not need it.
|
||||
if (!LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(inputBinary)) {
|
||||
outputBinary = inputBinary;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Stock passes do the real work. The only bespoke member is the loop-control
|
||||
// hint the stock unroller demands (see the pass header); with it set, an index
|
||||
// derived from a loop counter - the shape of the Minecraft 26.3 OIT
|
||||
// coefficient shader and of most real ones - folds to a literal here, and the
|
||||
// fallback below never runs.
|
||||
Optimizer folder(SPV_ENV_VULKAN_1_1);
|
||||
// First, because both the unroller and the marking pass below read the
|
||||
// induction variable as an OpPhi, and glslang emits it as loads and stores of
|
||||
// a Function variable.
|
||||
folder.RegisterPass(CreateLocalMultiStoreElimPass());
|
||||
folder.RegisterPass(LegalizeFragmentOutputIndexPass::CreateMarkLoopsForUnrollPass());
|
||||
folder.RegisterPass(CreateLoopUnrollPass(true));
|
||||
// Fold the unrolled induction values into the access chains, then clear out
|
||||
// what constant conditions leave behind.
|
||||
folder.RegisterPass(CreateCCPPass());
|
||||
folder.RegisterPass(CreateSimplificationPass());
|
||||
folder.RegisterPass(CreateDeadBranchElimPass());
|
||||
folder.RegisterPass(CreateBlockMergePass());
|
||||
|
||||
Vector<uint32_t> folded;
|
||||
if (!RunOptimizerChecked("LegalizeFragmentOutputIndexingForEssl.fold", folder, inputBinary,
|
||||
folded) ||
|
||||
folded.empty()) {
|
||||
// Fail open onto the fallback rather than onto the illegal module.
|
||||
folded = inputBinary;
|
||||
}
|
||||
|
||||
if (!LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(folded)) {
|
||||
outputBinary = folded;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Genuinely dynamic (uniform-derived, non-constant trip count, ...): lower it.
|
||||
Optimizer lowerer(SPV_ENV_VULKAN_1_1);
|
||||
lowerer.RegisterPass(LegalizeFragmentOutputIndexPass::CreateLowerToConstantSwitchPass());
|
||||
// The chains the lowering replaced are dead now; remove_outputs must stay
|
||||
// false here for the same reason it does in SanitizeAndOptimizeBinary.
|
||||
lowerer.RegisterPass(CreateAggressiveDCEPass(false));
|
||||
|
||||
if (!RunOptimizerChecked("LegalizeFragmentOutputIndexingForEssl.lower", lowerer, folded,
|
||||
outputBinary) ||
|
||||
outputBinary.empty()) {
|
||||
outputBinary = folded;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(outputBinary)) {
|
||||
// MGLOG_I, deliberately: MGLOG_E/W are compiled out at the INFO level every
|
||||
// CI and retrace build uses, and this is precisely the diagnostic that has
|
||||
// to survive to explain a shader the driver is about to reject.
|
||||
MGLOG_I("[spirv] LegalizeFragmentOutputIndexingForEssl: a fragment output is still "
|
||||
"indexed dynamically; a strict ES driver will reject this shader");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ShaderCompiler::LowerRectImages(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary) {
|
||||
using namespace spvtools;
|
||||
|
||||
@@ -43,6 +43,17 @@ namespace MobileGL {
|
||||
// devices lacking GL_NV_shader_noperspective_interpolation. See EmulateNoPerspectivePass.
|
||||
static bool EmulateNoPerspectiveForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Makes every index into a fragment-output array a constant integral
|
||||
// expression, which is what GLSL ES requires and SPIR-V does not. Runs the
|
||||
// stock folding chain first (loop unrolling folds the loop-derived indices
|
||||
// real shaders use), and lowers whatever is left - a genuinely dynamic index -
|
||||
// to a switch over the array's range. DirectGLES transpile path only: the
|
||||
// original module is legal for Vulkan, and no other stage is constrained this
|
||||
// way. Copies the input through untouched when no fragment output is indexed
|
||||
// dynamically, which is every shader but a handful.
|
||||
// See LegalizeFragmentOutputIndexPass.
|
||||
static bool LegalizeFragmentOutputIndexingForEssl(const Vector<Uint32>& inputBinary,
|
||||
Vector<uint32_t>& outputBinary);
|
||||
// Rebases loads of the InstanceIndex builtin to (InstanceIndex - BaseInstance) so
|
||||
// shaders see GL's zero-based gl_InstanceID. Vertex shaders only; DirectVulkan
|
||||
// backend only (glslang's relaxed mode aliases gl_InstanceID to gl_InstanceIndex,
|
||||
|
||||
@@ -0,0 +1,580 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeFragmentOutputIndexPass.cpp
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#include "LegalizeFragmentOutputIndexPass.h"
|
||||
|
||||
#include "spirv.hpp"
|
||||
#include "source/opt/basic_block.h"
|
||||
#include "source/opt/build_module.h"
|
||||
#include "source/opt/constants.h"
|
||||
#include "source/opt/def_use_manager.h"
|
||||
#include "source/opt/function.h"
|
||||
#include "source/opt/instruction.h"
|
||||
#include "source/opt/ir_builder.h"
|
||||
#include "source/opt/ir_context.h"
|
||||
#include "source/opt/loop_descriptor.h"
|
||||
#include "source/opt/module.h"
|
||||
#include "source/opt/type_manager.h"
|
||||
#include "source/util/make_unique.h"
|
||||
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
#include <vector>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
namespace {
|
||||
using spvtools::MakeUnique;
|
||||
using spvtools::opt::BasicBlock;
|
||||
using spvtools::opt::Function;
|
||||
using spvtools::opt::Instruction;
|
||||
using spvtools::opt::InstructionBuilder;
|
||||
using spvtools::opt::IRContext;
|
||||
using spvtools::opt::Operand;
|
||||
|
||||
// A fragment output array is at most GL_MAX_DRAW_BUFFERS elements (8 on ES
|
||||
// 3.0, 16 in practice) and each lowered element costs one basic block, so a
|
||||
// module claiming more than this is refused rather than exploded.
|
||||
constexpr uint32_t kMaxLoweredArrayLength = 32;
|
||||
// One CFG-changing rewrite per round (analyses are dropped after each), so
|
||||
// the round budget bounds the work on a pathological module.
|
||||
constexpr int kMaxLoweringRounds = 256;
|
||||
// Full unrolling copies the body once per iteration, and nothing in the stock
|
||||
// unroller bounds that. A shader whose output index comes from a 4096-trip
|
||||
// loop would be legalized into a module orders of magnitude larger and slower
|
||||
// to compile - so past this count the loop is left alone and the switch
|
||||
// lowering, whose cost is the array length rather than the trip count, takes
|
||||
// it instead. Real shaders of this shape (Minecraft 26.3's OIT coefficient
|
||||
// writer included) iterate a handful of times.
|
||||
constexpr size_t kMaxUnrolledIterations = 64;
|
||||
|
||||
struct DynamicIndexUse {
|
||||
Instruction* accessChain = nullptr;
|
||||
uint32_t arrayLength = 0;
|
||||
};
|
||||
|
||||
bool HasFragmentEntryPoint(IRContext* context) {
|
||||
for (const Instruction& entryPoint : context->module()->entry_points()) {
|
||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) ==
|
||||
spv::ExecutionModel::Fragment) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Every Output-storage variable whose pointee is an array, mapped to that
|
||||
// array's length. A length that is not a plain OpConstant (a spec constant)
|
||||
// maps to 0: still detected as illegal ESSL, never lowered.
|
||||
std::unordered_map<uint32_t, uint32_t> CollectOutputArrays(IRContext* context) {
|
||||
std::unordered_map<uint32_t, uint32_t> outputArrays;
|
||||
auto* defUseMgr = context->get_def_use_mgr();
|
||||
auto* constantMgr = context->get_constant_mgr();
|
||||
|
||||
for (Instruction& inst : context->module()->types_values()) {
|
||||
if (inst.opcode() != spv::Op::OpVariable ||
|
||||
static_cast<spv::StorageClass>(inst.GetSingleWordInOperand(0)) !=
|
||||
spv::StorageClass::Output) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Instruction* pointerType = defUseMgr->GetDef(inst.type_id());
|
||||
if (pointerType == nullptr || pointerType->opcode() != spv::Op::OpTypePointer) {
|
||||
continue;
|
||||
}
|
||||
Instruction* pointeeType = defUseMgr->GetDef(pointerType->GetSingleWordInOperand(1));
|
||||
if (pointeeType == nullptr || pointeeType->opcode() != spv::Op::OpTypeArray) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t arrayLength = 0;
|
||||
const spvtools::opt::analysis::Constant* lengthConstant =
|
||||
constantMgr->FindDeclaredConstant(pointeeType->GetSingleWordInOperand(1));
|
||||
if (lengthConstant != nullptr && lengthConstant->AsIntConstant() != nullptr) {
|
||||
arrayLength = lengthConstant->AsIntConstant()->GetU32BitValue();
|
||||
}
|
||||
outputArrays.emplace(inst.result_id(), arrayLength);
|
||||
}
|
||||
return outputArrays;
|
||||
}
|
||||
|
||||
// "Constant integral expression" in the ESSL sense: an OpConstant (or the
|
||||
// zero an OpConstantNull stands for). A spec constant is deliberately NOT
|
||||
// one - SPIRV-Cross prints it as an identifier, which is exactly what the
|
||||
// driver rejects.
|
||||
bool IsConstantIndex(IRContext* context, uint32_t indexId) {
|
||||
Instruction* def = context->get_def_use_mgr()->GetDef(indexId);
|
||||
return def != nullptr && (def->opcode() == spv::Op::OpConstant ||
|
||||
def->opcode() == spv::Op::OpConstantNull);
|
||||
}
|
||||
|
||||
// Access chains that index a fragment output array with a non-constant.
|
||||
// Only the FIRST index is considered: it is the one that selects the array
|
||||
// element, and it is the only one ESSL constrains. Chains rooted at another
|
||||
// access chain (a component of an element) are indexing inside the element
|
||||
// and are legal however they are computed.
|
||||
std::vector<DynamicIndexUse> CollectDynamicIndexUses(IRContext* context) {
|
||||
std::vector<DynamicIndexUse> uses;
|
||||
if (!HasFragmentEntryPoint(context)) {
|
||||
return uses;
|
||||
}
|
||||
|
||||
const std::unordered_map<uint32_t, uint32_t> outputArrays = CollectOutputArrays(context);
|
||||
if (outputArrays.empty()) {
|
||||
return uses;
|
||||
}
|
||||
|
||||
for (Function& function : *context->module()) {
|
||||
for (BasicBlock& block : function) {
|
||||
for (Instruction& inst : block) {
|
||||
if (inst.opcode() != spv::Op::OpAccessChain &&
|
||||
inst.opcode() != spv::Op::OpInBoundsAccessChain) {
|
||||
continue;
|
||||
}
|
||||
if (inst.NumInOperands() < 2) {
|
||||
continue;
|
||||
}
|
||||
const auto arrayIt = outputArrays.find(inst.GetSingleWordInOperand(0));
|
||||
if (arrayIt == outputArrays.end()) {
|
||||
continue;
|
||||
}
|
||||
if (IsConstantIndex(context, inst.GetSingleWordInOperand(1))) {
|
||||
continue;
|
||||
}
|
||||
uses.push_back({&inst, arrayIt->second});
|
||||
}
|
||||
}
|
||||
}
|
||||
return uses;
|
||||
}
|
||||
|
||||
// The array index operand of |accessChain| replaced by the constant |element|,
|
||||
// built at the builder's insertion point. Every later index is copied through
|
||||
// unchanged: `coeff[idx][i]` keeps its (legal) dynamic component index.
|
||||
Instruction* CloneChainWithConstantIndex(InstructionBuilder& builder, IRContext* context,
|
||||
Instruction* accessChain, uint32_t constantIndexId) {
|
||||
std::vector<Operand> operands;
|
||||
operands.reserve(accessChain->NumInOperands());
|
||||
for (uint32_t i = 0; i < accessChain->NumInOperands(); ++i) {
|
||||
if (i == 1) {
|
||||
operands.push_back({SPV_OPERAND_TYPE_ID, {constantIndexId}});
|
||||
} else {
|
||||
operands.push_back(accessChain->GetInOperand(i));
|
||||
}
|
||||
}
|
||||
return builder.AddInstruction(MakeUnique<Instruction>(context, accessChain->opcode(),
|
||||
accessChain->type_id(),
|
||||
context->TakeNextId(), operands));
|
||||
}
|
||||
|
||||
// The id of |element| as a constant of the same integer type as |indexId|.
|
||||
uint32_t ConstantLikeIndex(IRContext* context, uint32_t indexId, uint32_t element) {
|
||||
Instruction* indexDef = context->get_def_use_mgr()->GetDef(indexId);
|
||||
const spvtools::opt::analysis::Type* indexType =
|
||||
context->get_type_mgr()->GetType(indexDef->type_id());
|
||||
const spvtools::opt::analysis::Constant* constant =
|
||||
context->get_constant_mgr()->GetConstant(indexType, {element});
|
||||
return context->get_constant_mgr()->GetDefiningInstruction(constant)->result_id();
|
||||
}
|
||||
|
||||
// A 32-bit integer is the only index this pass lowers: OpSwitch matches its
|
||||
// literals against the selector's width, and every ESSL fragment-output index
|
||||
// is an int or uint.
|
||||
bool IsLowerableIndexType(IRContext* context, uint32_t indexId) {
|
||||
Instruction* indexDef = context->get_def_use_mgr()->GetDef(indexId);
|
||||
if (indexDef == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const spvtools::opt::analysis::Type* type =
|
||||
context->get_type_mgr()->GetType(indexDef->type_id());
|
||||
const spvtools::opt::analysis::Integer* integer =
|
||||
type != nullptr ? type->AsInteger() : nullptr;
|
||||
return integer != nullptr && integer->width() == 32;
|
||||
}
|
||||
|
||||
// The condition type OpSelect needs for |resultTypeId|. Before SPIR-V 1.4 a
|
||||
// scalar bool may not select between vectors, so a vector result needs a bool
|
||||
// vector of the same width - built by broadcasting the scalar comparison.
|
||||
// Anything that is neither scalar nor vector (a matrix or struct element) is
|
||||
// refused: pre-1.4 OpSelect cannot express it either.
|
||||
bool TryGetSelectConditionType(IRContext* context, uint32_t resultTypeId,
|
||||
uint32_t* conditionTypeId, uint32_t* dimension) {
|
||||
auto* typeMgr = context->get_type_mgr();
|
||||
const spvtools::opt::analysis::Type* resultType = typeMgr->GetType(resultTypeId);
|
||||
if (resultType == nullptr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
spvtools::opt::analysis::Bool boolType;
|
||||
if (resultType->AsVector() != nullptr) {
|
||||
const uint32_t count = resultType->AsVector()->element_count();
|
||||
spvtools::opt::analysis::Vector boolVector(&boolType, count);
|
||||
*conditionTypeId = typeMgr->GetTypeInstruction(&boolVector);
|
||||
*dimension = count;
|
||||
return *conditionTypeId != 0;
|
||||
}
|
||||
if (resultType->AsInteger() != nullptr || resultType->AsFloat() != nullptr ||
|
||||
resultType->AsBool() != nullptr) {
|
||||
*conditionTypeId = typeMgr->GetTypeInstruction(&boolType);
|
||||
*dimension = 1;
|
||||
return *conditionTypeId != 0;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Whether fully unrolling |loop| is bounded work. The trip count is read the
|
||||
// same way the stock unroller reads it, so a loop this declines to measure is
|
||||
// one CanPerformUnroll would refuse anyway - the hint would be inert on it,
|
||||
// and the fallback lowering is what handles it. Requires the induction
|
||||
// variable to already be an OpPhi, which is why this runs after ssa-rewrite.
|
||||
bool IsBoundedUnrollCandidate(spvtools::opt::Loop* loop) {
|
||||
const spvtools::opt::BasicBlock* condition = loop->FindConditionBlock();
|
||||
if (condition == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const Instruction* induction = loop->FindConditionVariable(condition);
|
||||
if (induction == nullptr || induction->opcode() != spv::Op::OpPhi) {
|
||||
return false;
|
||||
}
|
||||
size_t iterations = 0;
|
||||
if (!loop->FindNumberOfIterations(induction, &*condition->ctail(), &iterations)) {
|
||||
return false;
|
||||
}
|
||||
return iterations <= kMaxUnrolledIterations;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool LegalizeFragmentOutputIndexPass::BinaryHasDynamicOutputIndexing(
|
||||
const std::vector<uint32_t>& binary) {
|
||||
if (binary.empty()) {
|
||||
return false;
|
||||
}
|
||||
std::unique_ptr<IRContext> context = spvtools::BuildModule(
|
||||
SPV_ENV_VULKAN_1_1,
|
||||
[](spv_message_level_t, const char*, const spv_position_t&, const char*) {},
|
||||
binary.data(), binary.size());
|
||||
if (!context) {
|
||||
return false;
|
||||
}
|
||||
return !CollectDynamicIndexUses(context.get()).empty();
|
||||
}
|
||||
|
||||
spvtools::opt::Pass::Status LegalizeFragmentOutputIndexPass::Process() {
|
||||
return m_mode == Mode::MarkLoopsForUnroll ? MarkLoopsForUnroll() : LowerToConstantSwitch();
|
||||
}
|
||||
|
||||
spvtools::opt::Pass::Status LegalizeFragmentOutputIndexPass::MarkLoopsForUnroll() {
|
||||
auto* irContext = context();
|
||||
const std::vector<DynamicIndexUse> uses = CollectDynamicIndexUses(irContext);
|
||||
if (uses.empty()) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
|
||||
bool modified = false;
|
||||
for (const DynamicIndexUse& use : uses) {
|
||||
BasicBlock* block = irContext->get_instr_block(use.accessChain);
|
||||
if (block == nullptr) {
|
||||
continue;
|
||||
}
|
||||
Function* function = block->GetParent();
|
||||
if (function == nullptr) {
|
||||
continue;
|
||||
}
|
||||
|
||||
spvtools::opt::LoopDescriptor* loops = irContext->GetLoopDescriptor(function);
|
||||
for (spvtools::opt::Loop* loop = (*loops)[block->id()]; loop != nullptr;
|
||||
loop = loop->GetParent()) {
|
||||
if (!IsBoundedUnrollCandidate(loop)) {
|
||||
continue;
|
||||
}
|
||||
Instruction* mergeInst = loop->GetHeaderBlock()->GetLoopMergeInst();
|
||||
// Only a bare `None` control is promoted, and only when no extra
|
||||
// literal (PartialCount, PeelCount, ...) follows it: the unroller
|
||||
// tests the control word for equality with Unroll, so ORing the bit
|
||||
// into a control that already carries something - DontUnroll above
|
||||
// all - would neither unroll nor mean what it says.
|
||||
if (mergeInst == nullptr || mergeInst->NumOperands() != 3 ||
|
||||
mergeInst->GetSingleWordOperand(2) !=
|
||||
static_cast<uint32_t>(spv::LoopControlMask::MaskNone)) {
|
||||
continue;
|
||||
}
|
||||
mergeInst->SetOperand(
|
||||
2, {static_cast<uint32_t>(spv::LoopControlMask::Unroll)});
|
||||
modified = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (!modified) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
MGLOG_D("[spirv] fragment-output index: marked enclosing loops for full unrolling");
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
spvtools::opt::Pass::Status LegalizeFragmentOutputIndexPass::LowerToConstantSwitch() {
|
||||
auto* irContext = context();
|
||||
if (!HasFragmentEntryPoint(irContext)) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
|
||||
bool modified = false;
|
||||
// Access chains this pass has already refused, so a shape it cannot rewrite
|
||||
// exactly cannot spin the round loop.
|
||||
std::unordered_set<uint32_t> declined;
|
||||
|
||||
for (int round = 0; round < kMaxLoweringRounds; ++round) {
|
||||
const std::vector<DynamicIndexUse> uses = CollectDynamicIndexUses(irContext);
|
||||
bool progressed = false;
|
||||
|
||||
for (const DynamicIndexUse& use : uses) {
|
||||
if (declined.count(use.accessChain->result_id()) != 0) {
|
||||
continue;
|
||||
}
|
||||
const LoweringOutcome outcome = LowerOneChain(use.accessChain, use.arrayLength);
|
||||
if (outcome == LoweringOutcome::Declined) {
|
||||
declined.insert(use.accessChain->result_id());
|
||||
continue;
|
||||
}
|
||||
if (outcome == LoweringOutcome::Changed) {
|
||||
modified = true;
|
||||
progressed = true;
|
||||
// A store rewrite splits the block it sat in; every cached
|
||||
// analysis (and the instruction list this loop is walking) is
|
||||
// stale from here on. Recollect from scratch.
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!progressed) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!modified) {
|
||||
return Status::SuccessWithoutChange;
|
||||
}
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
LegalizeFragmentOutputIndexPass::LoweringOutcome LegalizeFragmentOutputIndexPass::LowerOneChain(
|
||||
Instruction* accessChain, uint32_t arrayLength) {
|
||||
auto* irContext = context();
|
||||
if (arrayLength == 0 || arrayLength > kMaxLoweredArrayLength) {
|
||||
MGLOG_D("[spirv] fragment-output index: array length %u is not lowerable", arrayLength);
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
if (!IsLowerableIndexType(irContext, accessChain->GetSingleWordInOperand(1))) {
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
|
||||
std::vector<Instruction*> stores;
|
||||
std::vector<Instruction*> loads;
|
||||
bool unsupportedUse = false;
|
||||
irContext->get_def_use_mgr()->ForEachUser(accessChain, [&](Instruction* user) {
|
||||
switch (user->opcode()) {
|
||||
case spv::Op::OpName:
|
||||
case spv::Op::OpDecorate:
|
||||
case spv::Op::OpDecorateId:
|
||||
return;
|
||||
case spv::Op::OpStore:
|
||||
// Only as the pointer. A pointer stored as a *value* is not a
|
||||
// fragment-output write and cannot be redirected element-wise.
|
||||
if (user->GetSingleWordInOperand(0) == accessChain->result_id()) {
|
||||
stores.push_back(user);
|
||||
} else {
|
||||
unsupportedUse = true;
|
||||
}
|
||||
return;
|
||||
case spv::Op::OpLoad:
|
||||
// Memory operands (Volatile, Aligned, ...) would be dropped by the
|
||||
// per-element rebuild, so a load carrying any is refused instead.
|
||||
if (user->NumInOperands() == 1) {
|
||||
loads.push_back(user);
|
||||
} else {
|
||||
unsupportedUse = true;
|
||||
}
|
||||
return;
|
||||
default:
|
||||
// A pointer passed to a function, copied, or chained further cannot
|
||||
// be resolved to one element here.
|
||||
unsupportedUse = true;
|
||||
return;
|
||||
}
|
||||
});
|
||||
|
||||
if (unsupportedUse) {
|
||||
MGLOG_D("[spirv] fragment-output index: chain %%%u has a use this pass cannot rewrite",
|
||||
accessChain->result_id());
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
|
||||
if (!loads.empty()) {
|
||||
return LowerLoad(accessChain, arrayLength, loads.front());
|
||||
}
|
||||
if (!stores.empty()) {
|
||||
return LowerStore(accessChain, arrayLength, stores.front());
|
||||
}
|
||||
|
||||
// No uses left: the chain itself is what detection is still seeing.
|
||||
irContext->KillInst(accessChain);
|
||||
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||
return LoweringOutcome::Changed;
|
||||
}
|
||||
|
||||
// switch (idx) { case 0: o[0] = v; break; case 1: o[1] = v; break; ... }
|
||||
//
|
||||
// The block holding the store is split at the store, and the tail becomes the
|
||||
// switch's merge block, so whatever followed the store still runs exactly once
|
||||
// on every path. An index outside [0, length) reaches the default target, which
|
||||
// is the merge block: nothing is stored, which is what an out-of-range write to
|
||||
// an output array already meant.
|
||||
LegalizeFragmentOutputIndexPass::LoweringOutcome LegalizeFragmentOutputIndexPass::LowerStore(
|
||||
Instruction* accessChain, uint32_t arrayLength, Instruction* store) {
|
||||
auto* irContext = context();
|
||||
BasicBlock* block = irContext->get_instr_block(store);
|
||||
if (block == nullptr) {
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
// Splitting a loop header keeps the label - and so the back edge's target -
|
||||
// on the first half while the OpLoopMerge moves to the second, which is not
|
||||
// a loop any more. Refuse instead of producing that.
|
||||
if (block->GetLoopMergeInst() != nullptr) {
|
||||
MGLOG_D("[spirv] fragment-output index: store sits in a loop header, declining");
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
Function* function = block->GetParent();
|
||||
if (function == nullptr) {
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
|
||||
const uint32_t indexId = accessChain->GetSingleWordInOperand(1);
|
||||
const uint32_t valueId = store->GetSingleWordInOperand(1);
|
||||
std::vector<Operand> memoryOperands;
|
||||
for (uint32_t i = 2; i < store->NumInOperands(); ++i) {
|
||||
memoryOperands.push_back(store->GetInOperand(i));
|
||||
}
|
||||
|
||||
const uint32_t mergeLabelId = irContext->TakeNextId();
|
||||
block->SplitBasicBlock(irContext, mergeLabelId, BasicBlock::iterator(store));
|
||||
// |store| now heads the merge block; the per-element stores replace it.
|
||||
irContext->KillInst(store);
|
||||
|
||||
std::vector<std::pair<Operand::OperandData, uint32_t>> targets;
|
||||
targets.reserve(arrayLength);
|
||||
BasicBlock* insertAfter = block;
|
||||
for (uint32_t element = 0; element < arrayLength; ++element) {
|
||||
const uint32_t caseLabelId = irContext->TakeNextId();
|
||||
auto caseBlock = MakeUnique<BasicBlock>(MakeUnique<Instruction>(
|
||||
irContext, spv::Op::OpLabel, 0, caseLabelId, std::initializer_list<Operand>{}));
|
||||
caseBlock->SetParent(function);
|
||||
BasicBlock* casePtr = function->InsertBasicBlockAfter(std::move(caseBlock), insertAfter);
|
||||
// The builders below register what they add, but this label was built by
|
||||
// hand: without this the OpSwitch would name a target the def-use manager
|
||||
// has never seen, which a consistency-checking build calls out.
|
||||
irContext->AnalyzeDefUse(casePtr->GetLabelInst());
|
||||
irContext->set_instr_block(casePtr->GetLabelInst(), casePtr);
|
||||
|
||||
InstructionBuilder caseBuilder(
|
||||
irContext, casePtr,
|
||||
IRContext::kAnalysisDefUse | IRContext::kAnalysisInstrToBlockMapping);
|
||||
const uint32_t constantId = ConstantLikeIndex(irContext, indexId, element);
|
||||
Instruction* elementChain =
|
||||
CloneChainWithConstantIndex(caseBuilder, irContext, accessChain, constantId);
|
||||
|
||||
std::vector<Operand> storeOperands;
|
||||
storeOperands.push_back({SPV_OPERAND_TYPE_ID, {elementChain->result_id()}});
|
||||
storeOperands.push_back({SPV_OPERAND_TYPE_ID, {valueId}});
|
||||
for (const Operand& memoryOperand : memoryOperands) {
|
||||
storeOperands.push_back(memoryOperand);
|
||||
}
|
||||
caseBuilder.AddInstruction(
|
||||
MakeUnique<Instruction>(irContext, spv::Op::OpStore, 0, 0, storeOperands));
|
||||
caseBuilder.AddBranch(mergeLabelId);
|
||||
|
||||
targets.push_back({Operand::OperandData{element}, caseLabelId});
|
||||
insertAfter = casePtr;
|
||||
}
|
||||
|
||||
InstructionBuilder switchBuilder(
|
||||
irContext, block, IRContext::kAnalysisDefUse | IRContext::kAnalysisInstrToBlockMapping);
|
||||
switchBuilder.AddSwitch(indexId, mergeLabelId, targets, mergeLabelId);
|
||||
|
||||
if (irContext->get_def_use_mgr()->NumUsers(accessChain) == 0) {
|
||||
irContext->KillInst(accessChain);
|
||||
}
|
||||
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||
MGLOG_D("[spirv] fragment-output index: lowered a dynamic write to a %u-way switch",
|
||||
arrayLength);
|
||||
return LoweringOutcome::Changed;
|
||||
}
|
||||
|
||||
// A read needs no control flow: load every element through a constant index and
|
||||
// pick with OpSelect. Reading an output array is rare, but it is legal SPIR-V and
|
||||
// legal ESSL, and the elements this adds reads of were already readable here.
|
||||
LegalizeFragmentOutputIndexPass::LoweringOutcome LegalizeFragmentOutputIndexPass::LowerLoad(
|
||||
Instruction* accessChain, uint32_t arrayLength, Instruction* load) {
|
||||
auto* irContext = context();
|
||||
uint32_t conditionTypeId = 0;
|
||||
uint32_t dimension = 0;
|
||||
if (!TryGetSelectConditionType(irContext, load->type_id(), &conditionTypeId, &dimension)) {
|
||||
MGLOG_D("[spirv] fragment-output index: element type is not selectable, declining");
|
||||
return LoweringOutcome::Declined;
|
||||
}
|
||||
const uint32_t boolTypeId = irContext->get_type_mgr()->GetBoolTypeId();
|
||||
const uint32_t indexId = accessChain->GetSingleWordInOperand(1);
|
||||
|
||||
InstructionBuilder builder(
|
||||
irContext, load, IRContext::kAnalysisDefUse | IRContext::kAnalysisInstrToBlockMapping);
|
||||
|
||||
uint32_t selectedId = 0;
|
||||
for (uint32_t element = 0; element < arrayLength; ++element) {
|
||||
const uint32_t constantId = ConstantLikeIndex(irContext, indexId, element);
|
||||
Instruction* elementChain =
|
||||
CloneChainWithConstantIndex(builder, irContext, accessChain, constantId);
|
||||
Instruction* elementLoad = builder.AddLoad(load->type_id(), elementChain->result_id());
|
||||
if (element == 0) {
|
||||
// Element 0 is the else-arm of the whole ladder, so an out-of-range
|
||||
// index reads it - an undefined element for an undefined index.
|
||||
selectedId = elementLoad->result_id();
|
||||
continue;
|
||||
}
|
||||
|
||||
Instruction* isElement =
|
||||
builder.AddBinaryOp(boolTypeId, spv::Op::OpIEqual, indexId, constantId);
|
||||
uint32_t conditionId = isElement->result_id();
|
||||
if (dimension > 1) {
|
||||
std::vector<uint32_t> components(dimension, conditionId);
|
||||
conditionId = builder.AddCompositeConstruct(conditionTypeId, components)->result_id();
|
||||
}
|
||||
selectedId = builder
|
||||
.AddSelect(load->type_id(), conditionId, elementLoad->result_id(),
|
||||
selectedId)
|
||||
->result_id();
|
||||
}
|
||||
|
||||
irContext->ReplaceAllUsesWith(load->result_id(), selectedId);
|
||||
irContext->KillInst(load);
|
||||
irContext->InvalidateAnalysesExceptFor(IRContext::kAnalysisNone);
|
||||
MGLOG_D("[spirv] fragment-output index: lowered a dynamic read to %u constant-indexed loads",
|
||||
arrayLength);
|
||||
return LoweringOutcome::Changed;
|
||||
}
|
||||
|
||||
spvtools::Optimizer::PassToken LegalizeFragmentOutputIndexPass::CreateMarkLoopsForUnrollPass() {
|
||||
return spvtools::Optimizer::PassToken(
|
||||
MakeUnique<LegalizeFragmentOutputIndexPass>(Mode::MarkLoopsForUnroll));
|
||||
}
|
||||
|
||||
spvtools::Optimizer::PassToken LegalizeFragmentOutputIndexPass::CreateLowerToConstantSwitchPass() {
|
||||
return spvtools::Optimizer::PassToken(
|
||||
MakeUnique<LegalizeFragmentOutputIndexPass>(Mode::LowerToConstantSwitch));
|
||||
}
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
@@ -0,0 +1,111 @@
|
||||
// MobileGL - MobileGL/MG_Util/ShaderTranspiler/SpirvPasses/LegalizeFragmentOutputIndexPass.h
|
||||
// Copyright (c) 2025-2026 MobileGL-Dev
|
||||
// Licensed under the GNU Lesser General Public License v3.0:
|
||||
// https://www.gnu.org/licenses/gpl-3.0.txt
|
||||
// https://www.gnu.org/licenses/lgpl-3.0.txt
|
||||
// SPDX-License-Identifier: LGPL-3.0-only
|
||||
// End of Source File Header
|
||||
|
||||
#pragma once
|
||||
#include "source/opt/pass.h"
|
||||
#include "spirv-tools/optimizer.hpp"
|
||||
|
||||
#include <Includes.h>
|
||||
#include <vector>
|
||||
|
||||
namespace MobileGL {
|
||||
namespace MG_Util {
|
||||
namespace ShaderTranspiler {
|
||||
// GLSL ES requires a *constant integral expression* to index a fragment output
|
||||
// array (GLSL ES 3.00 4.3.6 / 3.20 4.4.2); SPIR-V has no such rule, so a shader
|
||||
// that writes `coeff[i]` from a loop reaches SPIRV-Cross intact and comes out as
|
||||
// ESSL a strict driver rejects outright:
|
||||
//
|
||||
// '[' : array indexes for fragment outputs must be constant integral expressions
|
||||
//
|
||||
// The program then links nothing and every draw that uses it is a silent no-op.
|
||||
// Mesa accepts the same source, which is why this only ever showed on the ANGLE
|
||||
// lane (see tools/trace_replay/README.md, improved-transparency-minecraft-26.3:
|
||||
// the whole translucent layer disappears because the OIT coefficient shader is
|
||||
// exactly this shape).
|
||||
//
|
||||
// Two modes, used as two halves of one legalization in
|
||||
// ShaderCompiler::LegalizeFragmentOutputIndexingForEssl:
|
||||
//
|
||||
// MarkLoopsForUnroll - the companion the stock unroller needs. spirv-opt's
|
||||
// CreateLoopUnrollPass only touches loops whose OpLoopMerge carries the
|
||||
// Unroll loop control (LoopUtils::HasUnrollLoopControl), which glslang emits
|
||||
// only for an explicit [[unroll]]. This mode sets that hint on the loops that
|
||||
// actually enclose an offending access chain - and only those, so an
|
||||
// unrelated long loop elsewhere in the same shader is never unrolled - and
|
||||
// only when their trip count is known and small, so legalizing a shader can
|
||||
// never explode it. With the hint set, the stock chain (ssa-rewrite,
|
||||
// loop-unroll, ccp, simplification, dead-branch-elim) folds a loop-derived
|
||||
// index to a literal, which is what the real-world shaders (the OIT one
|
||||
// included) need. Must run AFTER ssa-rewrite: both the trip-count check and
|
||||
// the unroller itself need the induction variable as an OpPhi.
|
||||
//
|
||||
// LowerToConstantSwitch - the fallback for an index that is *genuinely*
|
||||
// dynamic (uniform-derived, a non-constant trip count, vertex data). It
|
||||
// rewrites each write through such an access chain into an OpSwitch over the
|
||||
// array's range with one constant-indexed store per case - the SPIR-V of
|
||||
// `switch (i) { case 0: o[0] = v; break; case 1: o[1] = v; break; }` - and
|
||||
// each read into per-element constant-indexed loads combined with OpSelect.
|
||||
// An out-of-range index stores nothing, which is what indexing an output
|
||||
// array out of range already meant.
|
||||
//
|
||||
// Fragment stage only: every other stage may index an output array dynamically
|
||||
// in ESSL, and on DirectVulkan the original SPIR-V is legal as-is. The pass
|
||||
// declines (leaving the module untouched) rather than half-transforming whenever
|
||||
// it meets a shape it cannot rewrite exactly - a pointer handed to a function, a
|
||||
// spec-constant array length, an index type that is not a 32-bit integer, or a
|
||||
// store sitting in a loop header block, where splitting would move the
|
||||
// OpLoopMerge away from the back edge's target.
|
||||
class LegalizeFragmentOutputIndexPass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
enum class Mode {
|
||||
MarkLoopsForUnroll,
|
||||
LowerToConstantSwitch,
|
||||
};
|
||||
|
||||
explicit LegalizeFragmentOutputIndexPass(Mode mode) : m_mode(mode) {}
|
||||
|
||||
const char* name() const override {
|
||||
return m_mode == Mode::MarkLoopsForUnroll ? "mobilegl-mark-fragment-output-index-loops"
|
||||
: "mobilegl-lower-fragment-output-index";
|
||||
}
|
||||
|
||||
Status Process() override;
|
||||
|
||||
static spvtools::Optimizer::PassToken CreateMarkLoopsForUnrollPass();
|
||||
static spvtools::Optimizer::PassToken CreateLowerToConstantSwitchPass();
|
||||
|
||||
// The detection half, on a serialized module: true when a fragment entry
|
||||
// point indexes an Output-storage array with anything but an OpConstant.
|
||||
// Cheap enough to gate the whole legalization on (one BuildModule, no
|
||||
// serialization) and used again after the folding chain to decide whether
|
||||
// the fallback has to run at all.
|
||||
static bool BinaryHasDynamicOutputIndexing(const std::vector<uint32_t>& binary);
|
||||
|
||||
private:
|
||||
enum class LoweringOutcome {
|
||||
// The shape is not one this pass can rewrite exactly; the module keeps
|
||||
// the illegal chain rather than a half-transform of it.
|
||||
Declined,
|
||||
Changed,
|
||||
};
|
||||
|
||||
Status MarkLoopsForUnroll();
|
||||
Status LowerToConstantSwitch();
|
||||
|
||||
LoweringOutcome LowerOneChain(spvtools::opt::Instruction* accessChain, uint32_t arrayLength);
|
||||
LoweringOutcome LowerStore(spvtools::opt::Instruction* accessChain, uint32_t arrayLength,
|
||||
spvtools::opt::Instruction* store);
|
||||
LoweringOutcome LowerLoad(spvtools::opt::Instruction* accessChain, uint32_t arrayLength,
|
||||
spvtools::opt::Instruction* load);
|
||||
|
||||
Mode m_mode;
|
||||
};
|
||||
} // namespace ShaderTranspiler
|
||||
} // namespace MG_Util
|
||||
} // namespace MobileGL
|
||||
Reference in New Issue
Block a user