mirror of
https://github.com/MobileGL-Dev/MobileGL
synced 2026-09-12 06:08:30 +09:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3223ecb14e |
@@ -16,19 +16,18 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_device = device;
|
||||
m_commandPool = commandPool;
|
||||
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount * 2, VK_NULL_HANDLE);
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount, VK_NULL_HANDLE);
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
allocInfo.commandPool = commandPool;
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = frameCount * 2;
|
||||
allocInfo.commandBufferCount = frameCount;
|
||||
VkResult result = vkAllocateCommandBuffers(device, &allocInfo, commandBuffers.data());
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
m_frames[i].commandBuffer = commandBuffers[i];
|
||||
m_frames[i].preCommandBuffer = commandBuffers[frameCount + i];
|
||||
}
|
||||
|
||||
VkSemaphoreCreateInfo semaphoreInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
|
||||
@@ -48,10 +47,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
void FrameContext::Destroy(VkDevice device, VkCommandPool commandPool) {
|
||||
const Uint32 frameCount = static_cast<Uint32>(m_frames.size());
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount * 2, VK_NULL_HANDLE);
|
||||
Vector<VkCommandBuffer> commandBuffers(frameCount, VK_NULL_HANDLE);
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
commandBuffers[i] = m_frames[i].commandBuffer;
|
||||
commandBuffers[frameCount + i] = m_frames[i].preCommandBuffer;
|
||||
}
|
||||
|
||||
for (Uint32 i = 0; i < frameCount; ++i) {
|
||||
@@ -62,7 +60,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
for (auto& frame : m_frames) {
|
||||
FreeRetiredCommandBuffers(frame);
|
||||
}
|
||||
vkFreeCommandBuffers(device, commandPool, frameCount * 2, commandBuffers.data());
|
||||
vkFreeCommandBuffers(device, commandPool, frameCount, commandBuffers.data());
|
||||
}
|
||||
m_frames.clear();
|
||||
currentFrameIndex = 0;
|
||||
@@ -89,8 +87,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
currentFrameIndex = (currentFrameIndex + 1) % static_cast<Uint32>(m_frames.size());
|
||||
GetCurrent().isCommandRecording = false;
|
||||
GetCurrent().hasCommandBufferRecorded = false;
|
||||
GetCurrent().isPreCommandRecording = false;
|
||||
GetCurrent().hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
|
||||
VkCommandBuffer& FrameContext::BeginCommandRecording(VkCommandBufferUsageFlags flags,
|
||||
@@ -122,41 +118,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
}
|
||||
|
||||
VkCommandBuffer FrameContext::BeginPreCommandRecording() {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.isPreCommandRecording) {
|
||||
return frame.preCommandBuffer;
|
||||
}
|
||||
MOBILEGL_ASSERT(!frame.hasPreCommandBufferRecorded,
|
||||
"BeginPreCommandRecording: a recorded pre stream is still awaiting submission");
|
||||
VK_VERIFY(vkResetCommandBuffer(frame.preCommandBuffer, 0), "BeginPreCommandRecording, vkResetCommandBuffer");
|
||||
VkCommandBufferBeginInfo beginInfo{};
|
||||
beginInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO;
|
||||
VK_VERIFY(vkBeginCommandBuffer(frame.preCommandBuffer, &beginInfo),
|
||||
"BeginPreCommandRecording, vkBeginCommandBuffer");
|
||||
frame.isPreCommandRecording = true;
|
||||
return frame.preCommandBuffer;
|
||||
}
|
||||
|
||||
void FrameContext::EndPreCommandRecordingIfOpen() {
|
||||
auto& frame = GetCurrent();
|
||||
if (!frame.isPreCommandRecording) {
|
||||
return;
|
||||
}
|
||||
VK_VERIFY(vkEndCommandBuffer(frame.preCommandBuffer), "EndPreCommandRecordingIfOpen, vkEndCommandBuffer");
|
||||
frame.isPreCommandRecording = false;
|
||||
frame.hasPreCommandBufferRecorded = true;
|
||||
}
|
||||
|
||||
void FrameContext::AbandonPreCommandRecording() {
|
||||
auto& frame = GetCurrent();
|
||||
if (frame.isPreCommandRecording) {
|
||||
VK_VERIFY(vkEndCommandBuffer(frame.preCommandBuffer), "AbandonPreCommandRecording, vkEndCommandBuffer");
|
||||
}
|
||||
frame.isPreCommandRecording = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
|
||||
VkResult FrameContext::InitializeSwapchainSemaphores(VkDevice device, Uint32 swapchainImageCount) {
|
||||
DestroySwapchainSemaphores(device);
|
||||
if (swapchainImageCount == 0) {
|
||||
@@ -241,27 +202,17 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint32 swapchainImageIndex) const {
|
||||
const auto& frame = GetCurrent();
|
||||
MOBILEGL_ASSERT(!frame.isCommandRecording, "GetSubmitInfo called while command buffer recording is still active");
|
||||
MOBILEGL_ASSERT(!frame.isPreCommandRecording,
|
||||
"GetSubmitInfo called while the pre-pass stream is still recording");
|
||||
AssertValidSwapchainImageIndex(swapchainImageIndex);
|
||||
SubmitInfoPacket packet{};
|
||||
packet.waitSemaphore = frame.imageAvailableSemaphore;
|
||||
packet.signalSemaphore = m_swapchainImageRenderFinishedSemaphores[swapchainImageIndex];
|
||||
|
||||
Uint32 commandBufferCount = 0;
|
||||
// The pre-pass stream executes strictly before the frame's commands.
|
||||
if (frame.hasPreCommandBufferRecorded) {
|
||||
packet.commandBuffers[commandBufferCount++] = frame.preCommandBuffer;
|
||||
}
|
||||
if (shouldSubmitCommandBuffer) {
|
||||
packet.commandBuffers[commandBufferCount++] = frame.commandBuffer;
|
||||
}
|
||||
packet.commandBuffer = frame.commandBuffer;
|
||||
|
||||
packet.submitInfo.waitSemaphoreCount = frame.imageAvailableSemaphoreConsumed ? 0U : 1U;
|
||||
packet.submitInfo.pWaitSemaphores = frame.imageAvailableSemaphoreConsumed ? nullptr : &packet.waitSemaphore;
|
||||
packet.submitInfo.pWaitDstStageMask = frame.imageAvailableSemaphoreConsumed ? nullptr : &packet.waitDstStageMask;
|
||||
packet.submitInfo.commandBufferCount = commandBufferCount;
|
||||
packet.submitInfo.pCommandBuffers = commandBufferCount > 0 ? packet.commandBuffers : nullptr;
|
||||
packet.submitInfo.commandBufferCount = shouldSubmitCommandBuffer ? 1U : 0U;
|
||||
packet.submitInfo.pCommandBuffers = shouldSubmitCommandBuffer ? &packet.commandBuffer : nullptr;
|
||||
packet.submitInfo.signalSemaphoreCount = 1;
|
||||
packet.submitInfo.pSignalSemaphores = &packet.signalSemaphore;
|
||||
return packet;
|
||||
@@ -325,14 +276,12 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_recordingObserver = observer;
|
||||
}
|
||||
|
||||
VkResult FrameContext::RetireCurrentCommandBuffer(Bool retirePreCommandBuffer) {
|
||||
VkResult FrameContext::RetireCurrentCommandBuffer() {
|
||||
MOBILEGL_ASSERT(m_device != VK_NULL_HANDLE && m_commandPool != VK_NULL_HANDLE,
|
||||
"RetireCurrentCommandBuffer requires an initialized FrameContext");
|
||||
auto& frame = GetCurrent();
|
||||
MOBILEGL_ASSERT(!frame.isCommandRecording,
|
||||
"RetireCurrentCommandBuffer called while the command buffer is still recording");
|
||||
MOBILEGL_ASSERT(!frame.isPreCommandRecording,
|
||||
"RetireCurrentCommandBuffer called while the pre-pass stream is still recording");
|
||||
|
||||
VkCommandBufferAllocateInfo allocInfo{};
|
||||
allocInfo.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO;
|
||||
@@ -340,20 +289,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
allocInfo.commandBufferCount = 1;
|
||||
VkCommandBuffer replacement = VK_NULL_HANDLE;
|
||||
VkResult result = vkAllocateCommandBuffers(m_device, &allocInfo, &replacement);
|
||||
const VkResult result = vkAllocateCommandBuffers(m_device, &allocInfo, &replacement);
|
||||
if (result != VK_SUCCESS) {
|
||||
return result;
|
||||
}
|
||||
if (retirePreCommandBuffer) {
|
||||
VkCommandBuffer preReplacement = VK_NULL_HANDLE;
|
||||
result = vkAllocateCommandBuffers(m_device, &allocInfo, &preReplacement);
|
||||
if (result != VK_SUCCESS) {
|
||||
vkFreeCommandBuffers(m_device, m_commandPool, 1, &replacement);
|
||||
return result;
|
||||
}
|
||||
frame.retiredCommandBuffers.push_back({frame.preCommandBuffer, frame.lastSubmitIndex});
|
||||
frame.preCommandBuffer = preReplacement;
|
||||
}
|
||||
// lastSubmitIndex was just written by the renderer for the submission
|
||||
// that carried this command buffer.
|
||||
frame.retiredCommandBuffers.push_back({frame.commandBuffer, frame.lastSubmitIndex});
|
||||
|
||||
@@ -29,9 +29,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkPipelineStageFlags waitDstStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
VkSemaphore waitSemaphore = VK_NULL_HANDLE;
|
||||
VkSemaphore signalSemaphore = VK_NULL_HANDLE;
|
||||
// [0] = pre-pass command buffer (when recorded), then the frame
|
||||
// command buffer; submitInfo.pCommandBuffers points here.
|
||||
VkCommandBuffer commandBuffers[2] = {VK_NULL_HANDLE, VK_NULL_HANDLE};
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
VkSubmitInfo submitInfo{VK_STRUCTURE_TYPE_SUBMIT_INFO};
|
||||
};
|
||||
|
||||
@@ -54,18 +52,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
struct FrameData {
|
||||
VkCommandBuffer commandBuffer = VK_NULL_HANDLE;
|
||||
// Pre-pass work stream: out-of-pass commands (deferred clear
|
||||
// materialization, sampled-layout transitions) for resources the
|
||||
// frame's recording has not touched yet. Submitted immediately
|
||||
// BEFORE commandBuffer in the same vkQueueSubmit, so recording
|
||||
// into it never has to split the frame's active render pass.
|
||||
VkCommandBuffer preCommandBuffer = VK_NULL_HANDLE;
|
||||
VkSemaphore imageAvailableSemaphore = VK_NULL_HANDLE;
|
||||
VkFence imageInFlightFence = VK_NULL_HANDLE;
|
||||
Bool isCommandRecording = false;
|
||||
Bool hasCommandBufferRecorded = false;
|
||||
Bool isPreCommandRecording = false;
|
||||
Bool hasPreCommandBufferRecorded = false;
|
||||
Bool imageAvailableSemaphoreConsumed = false;
|
||||
// Command buffers submitted mid-frame (FlushPendingCommands),
|
||||
// appended in submit order; freed once their submission is known
|
||||
@@ -87,14 +77,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkCommandBuffer& BeginCommandRecording(VkCommandBufferUsageFlags flags = 0,
|
||||
const VkCommandBufferInheritanceInfo* pInheritanceInfo = nullptr);
|
||||
void EndCommandRecording();
|
||||
// Lazily opens the pre-pass work stream (see FrameData::preCommandBuffer).
|
||||
VkCommandBuffer BeginPreCommandRecording();
|
||||
// Closes the pre stream if open, marking it for submission ahead of the
|
||||
// frame command buffer. Safe to call when it never opened.
|
||||
void EndPreCommandRecordingIfOpen();
|
||||
// Drops an in-progress or recorded-but-unsubmitted pre stream (dropped
|
||||
// frame recordings, swapchain recreation).
|
||||
void AbandonPreCommandRecording();
|
||||
VkResult InitializeSwapchainSemaphores(VkDevice device, Uint32 swapchainImageCount);
|
||||
void DestroySwapchainSemaphores(VkDevice device);
|
||||
Bool TransitionToPresent(VkImage image, VkImageLayout oldLayout,
|
||||
@@ -109,7 +91,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// can restart while the submitted buffer is still executing. Retired
|
||||
// buffers are freed after the slot's fence is next waited, or as soon
|
||||
// as their submission is observed complete.
|
||||
VkResult RetireCurrentCommandBuffer(Bool retirePreCommandBuffer = false);
|
||||
VkResult RetireCurrentCommandBuffer();
|
||||
|
||||
// Frees every retired command buffer whose tagged submission index is
|
||||
// known complete. Driven by the renderer's submit tracker on completion
|
||||
|
||||
@@ -12,7 +12,10 @@
|
||||
#include "MG_Util/ShaderTranspiler/ShaderCompiler.h"
|
||||
#include "MG_Util/ShaderTranspiler/SpvcSession.h"
|
||||
#include "MG_Util/ShaderTranspiler/Types.h"
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <unordered_set>
|
||||
#include <spirv-tools/libspirv.h>
|
||||
#include <spirv-tools/optimizer.hpp>
|
||||
#include <source/opt/build_module.h>
|
||||
@@ -1099,6 +1102,374 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<ForceExplicitLod0SamplePass>());
|
||||
}
|
||||
|
||||
// TEMP-PERFDIAG: measure what fragment-stage fp32 costs on this GPU. Desktop GLSL carries
|
||||
// no precision qualifiers, so everything reaches the driver as full fp32 while Adreno runs
|
||||
// fp16 at twice the rate. Decorating every float-typed result in a fragment entry point
|
||||
// with RelaxedPrecision is the blunt "all mediump" upper bound - it changes results, so it
|
||||
// is a probe, not a shipping transform. Toggled by /sdcard/MG/exp_relaxed_precision.
|
||||
class RelaxedPrecisionProbePass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "relaxed-precision-probe"; }
|
||||
|
||||
Status Process() override {
|
||||
Bool isFragment = false;
|
||||
for (auto& entryPoint : get_module()->entry_points()) {
|
||||
if (entryPoint.opcode() != spv::Op::OpEntryPoint) continue;
|
||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) ==
|
||||
spv::ExecutionModel::Fragment) {
|
||||
isFragment = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!isFragment) return Status::SuccessWithoutChange;
|
||||
|
||||
// Every 32-bit-float scalar/vector/matrix type in the module. Anything wider (f64)
|
||||
// or narrower is left alone: RelaxedPrecision only has meaning for 32-bit floats.
|
||||
std::unordered_set<Uint32> relaxableTypes;
|
||||
for (auto& type : get_module()->types_values()) {
|
||||
const Uint32 typeId = type.result_id();
|
||||
if (typeId == 0) continue;
|
||||
switch (type.opcode()) {
|
||||
case spv::Op::OpTypeFloat:
|
||||
if (type.GetSingleWordInOperand(0) == 32) relaxableTypes.insert(typeId);
|
||||
break;
|
||||
case spv::Op::OpTypeVector:
|
||||
case spv::Op::OpTypeMatrix:
|
||||
if (relaxableTypes.count(type.GetSingleWordInOperand(0)) != 0) {
|
||||
relaxableTypes.insert(typeId);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (relaxableTypes.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
Vector<Uint32> targets;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0) continue;
|
||||
if (relaxableTypes.count(inst.type_id()) == 0) continue;
|
||||
targets.push_back(resultId);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (targets.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
for (const Uint32 id : targets) {
|
||||
context()->get_decoration_mgr()->AddDecoration(
|
||||
id, static_cast<Uint32>(spv::Decoration::RelaxedPrecision));
|
||||
}
|
||||
context()->InvalidateAnalysesExceptFor(spvtools::opt::IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
};
|
||||
|
||||
// Relax fragment-stage arithmetic that provably came out of a texture read. Desktop GLSL
|
||||
// has no precision qualifiers, so every fragment value reaches the driver as fp32 while
|
||||
// Adreno runs fp16 at twice the rate - and a texel is at most 8 bits per channel, which
|
||||
// fp16's 11-bit mantissa carries exactly. Seeding at image reads and propagating only
|
||||
// through operations whose every input is already relaxed keeps everything the shader
|
||||
// computes from other sources (screen coordinates, depth, wide-range uniforms) at full
|
||||
// precision, which is where fp16 would actually go wrong: fp16 cannot even represent a
|
||||
// 3044-pixel gl_FragCoord.x exactly.
|
||||
class RelaxTextureDerivedPrecisionPass final : public spvtools::opt::Pass {
|
||||
public:
|
||||
const char* name() const override { return "relax-texture-derived-precision"; }
|
||||
|
||||
Status Process() override {
|
||||
if (!IsFragmentEntryPoint()) return Status::SuccessWithoutChange;
|
||||
// A shader that drives depth or coverage itself is out of scope: those values must
|
||||
// stay exact, and proving which computations feed them is not worth it here.
|
||||
if (WritesDepthOrSampleMask()) return Status::SuccessWithoutChange;
|
||||
|
||||
CollectRelaxableFloatTypes();
|
||||
if (m_relaxableTypes.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
// Whitelisting from texture reads captures nothing in practice: MC's fragment
|
||||
// shaders multiply every texel by an interpolated colour and a UBO value, so one
|
||||
// un-relaxed operand vetoes the whole expression (measured: no fps change).
|
||||
// Taint the few genuinely precision-critical sources instead and relax the rest.
|
||||
std::unordered_set<Uint32> tainted;
|
||||
CollectPrecisionCriticalSeeds(tainted);
|
||||
Bool grew = true;
|
||||
while (grew) {
|
||||
grew = false;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0 || tainted.count(resultId) != 0) continue;
|
||||
if (!AnyOperandTainted(inst, tainted)) continue;
|
||||
tainted.insert(resultId);
|
||||
grew = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unordered_set<Uint32> relaxed;
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0 || tainted.count(resultId) != 0) continue;
|
||||
if (m_relaxableTypes.count(inst.type_id()) == 0) continue;
|
||||
relaxed.insert(resultId);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (relaxed.empty()) return Status::SuccessWithoutChange;
|
||||
|
||||
for (const Uint32 id : relaxed) {
|
||||
context()->get_decoration_mgr()->AddDecoration(
|
||||
id, static_cast<Uint32>(spv::Decoration::RelaxedPrecision));
|
||||
}
|
||||
context()->InvalidateAnalysesExceptFor(spvtools::opt::IRContext::kAnalysisNone);
|
||||
return Status::SuccessWithChange;
|
||||
}
|
||||
|
||||
private:
|
||||
std::unordered_set<Uint32> m_relaxableTypes;
|
||||
|
||||
Bool IsFragmentEntryPoint() const {
|
||||
for (auto& entryPoint : get_module()->entry_points()) {
|
||||
if (entryPoint.opcode() != spv::Op::OpEntryPoint) continue;
|
||||
if (static_cast<spv::ExecutionModel>(entryPoint.GetSingleWordInOperand(0)) ==
|
||||
spv::ExecutionModel::Fragment) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool WritesDepthOrSampleMask() const {
|
||||
for (auto& annotation : get_module()->annotations()) {
|
||||
if (annotation.opcode() != spv::Op::OpDecorate) continue;
|
||||
if (static_cast<spv::Decoration>(annotation.GetSingleWordInOperand(1)) !=
|
||||
spv::Decoration::BuiltIn) {
|
||||
continue;
|
||||
}
|
||||
const auto builtIn = static_cast<spv::BuiltIn>(annotation.GetSingleWordInOperand(2));
|
||||
if (builtIn == spv::BuiltIn::FragDepth || builtIn == spv::BuiltIn::SampleMask) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void CollectRelaxableFloatTypes() {
|
||||
m_relaxableTypes.clear();
|
||||
for (auto& type : get_module()->types_values()) {
|
||||
const Uint32 typeId = type.result_id();
|
||||
if (typeId == 0) continue;
|
||||
switch (type.opcode()) {
|
||||
case spv::Op::OpTypeFloat:
|
||||
if (type.GetSingleWordInOperand(0) == 32) m_relaxableTypes.insert(typeId);
|
||||
break;
|
||||
case spv::Op::OpTypeVector:
|
||||
if (m_relaxableTypes.count(type.GetSingleWordInOperand(0)) != 0) {
|
||||
m_relaxableTypes.insert(typeId);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void CollectImageReadSeeds(std::unordered_set<Uint32>& relaxed) const {
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
const Uint32 resultId = inst.result_id();
|
||||
if (resultId == 0 || m_relaxableTypes.count(inst.type_id()) == 0) continue;
|
||||
// Interpolated user varyings seed too, or propagation dies at the
|
||||
// first `texel * vertexColour`: the load of an Input can never be
|
||||
// relaxed by the rule below (its operand is a pointer), so a single
|
||||
// varying vetoes every downstream operation. This is what ESSL's
|
||||
// mediump varyings already mean. Built-ins are excluded - gl_FragCoord
|
||||
// carries pixel coordinates that fp16 cannot represent exactly.
|
||||
if (inst.opcode() == spv::Op::OpLoad && IsNonBuiltInFragmentInput(inst)) {
|
||||
relaxed.insert(resultId);
|
||||
continue;
|
||||
}
|
||||
switch (inst.opcode()) {
|
||||
case spv::Op::OpImageSampleImplicitLod:
|
||||
case spv::Op::OpImageSampleExplicitLod:
|
||||
case spv::Op::OpImageSampleProjImplicitLod:
|
||||
case spv::Op::OpImageSampleProjExplicitLod:
|
||||
case spv::Op::OpImageSampleDrefImplicitLod:
|
||||
case spv::Op::OpImageSampleDrefExplicitLod:
|
||||
case spv::Op::OpImageFetch:
|
||||
case spv::Op::OpImageRead:
|
||||
case spv::Op::OpImageGather:
|
||||
relaxed.insert(resultId);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// OpLoad straight out of a fragment Input variable that carries no BuiltIn decoration.
|
||||
// Only a direct load counts: a load through an access chain could be indexing a
|
||||
// structure whose other members are not interpolated colour data.
|
||||
Bool IsNonBuiltInFragmentInput(const spvtools::opt::Instruction& load) const {
|
||||
const Uint32 pointerId = load.GetSingleWordInOperand(0);
|
||||
const auto* pointer = context()->get_def_use_mgr()->GetDef(pointerId);
|
||||
if (pointer == nullptr || pointer->opcode() != spv::Op::OpVariable) return false;
|
||||
if (static_cast<spv::StorageClass>(pointer->GetSingleWordInOperand(0)) !=
|
||||
spv::StorageClass::Input) {
|
||||
return false;
|
||||
}
|
||||
Bool isBuiltIn = false;
|
||||
context()->get_decoration_mgr()->ForEachDecoration(
|
||||
pointerId, static_cast<Uint32>(spv::Decoration::BuiltIn),
|
||||
[&isBuiltIn](const spvtools::opt::Instruction&) { isBuiltIn = true; });
|
||||
return !isBuiltIn;
|
||||
}
|
||||
|
||||
// A float constant small enough that fp16 represents it without surprise. Colour math
|
||||
// constants (0, 1, 0.5, 255, gamma exponents) all live here; anything larger is
|
||||
// treated as unknown so it stops propagation.
|
||||
Bool IsBoundedFloatConstant(Uint32 id) const {
|
||||
const auto* constant = context()->get_constant_mgr()->FindDeclaredConstant(id);
|
||||
if (constant == nullptr) return false;
|
||||
if (const auto* scalar = constant->AsFloatConstant()) {
|
||||
const float value = scalar->GetFloat();
|
||||
return std::isfinite(value) && std::fabs(value) <= 1024.0f;
|
||||
}
|
||||
if (const auto* composite = constant->AsVectorConstant()) {
|
||||
for (const auto* component : composite->GetComponents()) {
|
||||
const auto* scalar = component->AsFloatConstant();
|
||||
if (scalar == nullptr) return false;
|
||||
const float value = scalar->GetFloat();
|
||||
if (!std::isfinite(value) || std::fabs(value) > 1024.0f) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Precision-critical sources: a built-in fragment input. gl_FragCoord is the one that
|
||||
// matters - fp16 cannot represent a 3044-pixel x coordinate exactly, and anything
|
||||
// derived from it (screen-space effects, manual depth reconstruction) would visibly
|
||||
// quantise. Everything else a fragment shader reads is colour-range data.
|
||||
void CollectPrecisionCriticalSeeds(std::unordered_set<Uint32>& tainted) const {
|
||||
for (auto& function : *get_module()) {
|
||||
for (auto& block : function) {
|
||||
for (auto& inst : block) {
|
||||
if (inst.opcode() != spv::Op::OpLoad || inst.result_id() == 0) continue;
|
||||
if (IsBuiltInInputLoad(inst)) tainted.insert(inst.result_id());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Bool IsBuiltInInputLoad(const spvtools::opt::Instruction& load) const {
|
||||
const Uint32 pointerId = load.GetSingleWordInOperand(0);
|
||||
const auto* pointer = context()->get_def_use_mgr()->GetDef(pointerId);
|
||||
if (pointer == nullptr || pointer->opcode() != spv::Op::OpVariable) return false;
|
||||
if (static_cast<spv::StorageClass>(pointer->GetSingleWordInOperand(0)) !=
|
||||
spv::StorageClass::Input) {
|
||||
return false;
|
||||
}
|
||||
Bool isBuiltIn = false;
|
||||
context()->get_decoration_mgr()->ForEachDecoration(
|
||||
pointerId, static_cast<Uint32>(spv::Decoration::BuiltIn),
|
||||
[&isBuiltIn](const spvtools::opt::Instruction&) { isBuiltIn = true; });
|
||||
return isBuiltIn;
|
||||
}
|
||||
|
||||
Bool AnyOperandTainted(const spvtools::opt::Instruction& inst,
|
||||
const std::unordered_set<Uint32>& tainted) const {
|
||||
const Uint32 operandCount = inst.NumInOperands();
|
||||
for (Uint32 i = 0; i < operandCount; ++i) {
|
||||
const auto& operand = inst.GetInOperand(i);
|
||||
if (!spvIsIdType(operand.type)) continue;
|
||||
if (IsNonNumericOperand(inst, i)) continue;
|
||||
if (tainted.count(operand.words[0]) != 0) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Bool AllValueOperandsRelaxed(const spvtools::opt::Instruction& inst,
|
||||
const std::unordered_set<Uint32>& relaxed) const {
|
||||
switch (inst.opcode()) {
|
||||
// Pointer-typed plumbing: relaxing the loaded value would say nothing about the
|
||||
// memory it came from, and the pointer operand can never be in the set.
|
||||
case spv::Op::OpLoad:
|
||||
case spv::Op::OpStore:
|
||||
case spv::Op::OpAccessChain:
|
||||
case spv::Op::OpInBoundsAccessChain:
|
||||
case spv::Op::OpFunctionCall:
|
||||
return false;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
Bool sawValueOperand = false;
|
||||
Bool allRelaxed = true;
|
||||
const Uint32 operandCount = inst.NumInOperands();
|
||||
for (Uint32 i = 0; i < operandCount; ++i) {
|
||||
const auto& operand = inst.GetInOperand(i);
|
||||
if (!spvIsIdType(operand.type)) continue; // literals: selectors, swizzle indices
|
||||
const Uint32 id = operand.words[0];
|
||||
// OpPhi's block labels, OpSelect's condition and OpExtInst's instruction-set id
|
||||
// are ids that carry no numeric precision; skip them rather than let them veto.
|
||||
if (IsNonNumericOperand(inst, i)) continue;
|
||||
sawValueOperand = true;
|
||||
if (relaxed.count(id) != 0) continue;
|
||||
if (IsBoundedFloatConstant(id)) continue;
|
||||
allRelaxed = false;
|
||||
break;
|
||||
}
|
||||
return sawValueOperand && allRelaxed;
|
||||
}
|
||||
|
||||
static Bool IsNonNumericOperand(const spvtools::opt::Instruction& inst, Uint32 index) {
|
||||
switch (inst.opcode()) {
|
||||
case spv::Op::OpPhi:
|
||||
return (index % 2) == 1; // parent block labels
|
||||
case spv::Op::OpSelect:
|
||||
return index == 0; // condition
|
||||
case spv::Op::OpExtInst:
|
||||
return index == 0; // extended instruction set
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// TEMP-PERFDIAG: A/B switch between the scoped transform and the all-float upper bound.
|
||||
Bool PerfDiagRelaxAllPrecision() {
|
||||
static const Bool enabled = [] {
|
||||
std::FILE* probe = std::fopen("/sdcard/MG/exp_relaxed_precision_all", "rb");
|
||||
if (probe == nullptr) return false;
|
||||
std::fclose(probe);
|
||||
MGLOG_I("[PERFDIAG] fragment RelaxedPrecision: ALL floats (upper-bound probe)");
|
||||
return true;
|
||||
}();
|
||||
return enabled;
|
||||
}
|
||||
|
||||
// TEMP-PERFDIAG: lets a run turn the transform off entirely for an A/B baseline.
|
||||
Bool PerfDiagRelaxedPrecisionEnabled() {
|
||||
static const Bool disabled = [] {
|
||||
std::FILE* probe = std::fopen("/sdcard/MG/exp_no_relaxed_precision", "rb");
|
||||
if (probe == nullptr) return false;
|
||||
std::fclose(probe);
|
||||
MGLOG_I("[PERFDIAG] fragment RelaxedPrecision DISABLED");
|
||||
return true;
|
||||
}();
|
||||
return !disabled;
|
||||
}
|
||||
|
||||
Bool TransformSpirvForExplicitLod0Sampling(const Vector<Uint>& input, Vector<Uint>& output) {
|
||||
if (input.empty()) {
|
||||
output.clear();
|
||||
@@ -1128,6 +1499,37 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return spvtools::Optimizer::PassToken(MakeUnique<GlToVulkanPositionFixPass>(transformFlags));
|
||||
}
|
||||
|
||||
// TEMP-PERFDIAG
|
||||
Bool TransformSpirvForRelaxedPrecisionProbe(const Vector<Uint>& input, Vector<Uint>& output) {
|
||||
if (input.empty()) {
|
||||
output.clear();
|
||||
return true;
|
||||
}
|
||||
spvtools::Optimizer optimizer(SPV_ENV_VULKAN_1_3);
|
||||
spvtools::OptimizerOptions options;
|
||||
options.set_run_validator(false);
|
||||
optimizer.SetMessageConsumer([](spv_message_level_t, const char*, const spv_position_t&,
|
||||
const char* message) {
|
||||
MGLOG_E("Vulkan: relaxed-precision probe: %s", message != nullptr ? message : "");
|
||||
});
|
||||
// SSA promotion first: glslang emits function-local variables with stores and loads,
|
||||
// and a load can never be relaxed (its operand is a pointer), so without this the
|
||||
// propagation below dies at the first temporary.
|
||||
optimizer.RegisterPass(spvtools::CreateLocalMultiStoreElimPass());
|
||||
if (PerfDiagRelaxAllPrecision()) {
|
||||
optimizer.RegisterPass(spvtools::Optimizer::PassToken(MakeUnique<RelaxedPrecisionProbePass>()));
|
||||
} else {
|
||||
optimizer.RegisterPass(
|
||||
spvtools::Optimizer::PassToken(MakeUnique<RelaxTextureDerivedPrecisionPass>()));
|
||||
}
|
||||
const Bool success = optimizer.Run(input.data(), input.size(), &output, options);
|
||||
if (!success) {
|
||||
MGLOG_E("Vulkan: relaxed-precision probe failed; keeping the original module");
|
||||
output = input;
|
||||
}
|
||||
return success;
|
||||
}
|
||||
|
||||
Bool TransformSpirvForVulkanPositionFix(const Vector<Uint>& input, Vector<Uint>& output,
|
||||
ProgramFactory::CompileOptionFlags transformFlags) {
|
||||
if (input.empty()) {
|
||||
@@ -2186,6 +2588,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
if ((flags & ProgramFactory::CompileOptionBit::RelaxedFragmentPrecision) &&
|
||||
PerfDiagRelaxedPrecisionEnabled() && shaders[i] &&
|
||||
shaders[i]->GetShaderStage() == ShaderStage::Fragment) {
|
||||
Vector<Uint> relaxedSpirv;
|
||||
if (TransformSpirvForRelaxedPrecisionProbe(moduleSpirvs[i], relaxedSpirv)) {
|
||||
moduleSpirvs[i] = Move(relaxedSpirv);
|
||||
}
|
||||
}
|
||||
|
||||
// GL apps depend on cross-program position invariance for multi-pass equality
|
||||
// depth tests (MC 26.3's OIT re-draws the cloud geometry with GEQUAL against the
|
||||
// depth its own first pass wrote); decorate Position outputs Invariant so
|
||||
|
||||
@@ -47,6 +47,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// level, which makes the two forms produce identical texels (the implicit lambda is
|
||||
// clamped into [minLod, maxLod] = [0, 0] regardless of derivatives or bias).
|
||||
ExplicitLod0Sampling = 1 << 5,
|
||||
// Fragment arithmetic may run at relaxed (fp16) precision. Only requested for draws
|
||||
// where every sampled texture and every colour attachment is an 8-bit-or-less
|
||||
// normalized format, so nothing the shader reads or writes carries more precision
|
||||
// than fp16 already represents exactly.
|
||||
RelaxedFragmentPrecision = 1 << 6,
|
||||
};
|
||||
using CompileOptionFlags = Flags<CompileOptionBit>;
|
||||
using HashType = Uint64;
|
||||
|
||||
@@ -262,9 +262,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_images.resize(imageCount, VK_NULL_HANDLE);
|
||||
VK_VERIFY(vkGetSwapchainImagesKHR(device, m_swapchain, &imageCount, m_images.data()));
|
||||
m_imageLayouts.assign(imageCount, VK_IMAGE_LAYOUT_UNDEFINED);
|
||||
// Fresh swapchain images hold garbage until a render pass stores into them.
|
||||
m_imageContentDefined.assign(imageCount, false);
|
||||
m_depthStencilContentDefined.assign(imageCount, false);
|
||||
|
||||
CreateImageViews(device);
|
||||
CreateDepthStencilResources(device, physicalDevice);
|
||||
@@ -436,39 +433,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
|
||||
m_images.clear();
|
||||
m_imageLayouts.clear();
|
||||
m_imageContentDefined.clear();
|
||||
m_depthStencilContentDefined.clear();
|
||||
m_preTransform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
|
||||
}
|
||||
|
||||
Bool SwapchainObject::IsImageContentDefined(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_imageContentDefined.size(), "Swapchain image content index out of range");
|
||||
return m_imageContentDefined[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetImageContentDefined(Uint32 index, Bool defined) {
|
||||
MOBILEGL_ASSERT(index < m_imageContentDefined.size(), "Swapchain image content index out of range");
|
||||
m_imageContentDefined[index] = defined;
|
||||
}
|
||||
|
||||
Bool SwapchainObject::IsDepthStencilContentDefined(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilContentDefined.size(),
|
||||
"Swapchain depth/stencil content index out of range");
|
||||
return m_depthStencilContentDefined[index];
|
||||
}
|
||||
|
||||
void SwapchainObject::SetDepthStencilContentDefined(Uint32 index, Bool defined) {
|
||||
MOBILEGL_ASSERT(index < m_depthStencilContentDefined.size(),
|
||||
"Swapchain depth/stencil content index out of range");
|
||||
m_depthStencilContentDefined[index] = defined;
|
||||
}
|
||||
|
||||
void SwapchainObject::SetAllDepthStencilContentUndefined() {
|
||||
for (SizeT i = 0; i < m_depthStencilContentDefined.size(); ++i) {
|
||||
m_depthStencilContentDefined[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
VkImage SwapchainObject::GetImage(Uint32 index) const {
|
||||
MOBILEGL_ASSERT(index < m_images.size(), "Swapchain image index out of range");
|
||||
return m_images[index];
|
||||
|
||||
@@ -52,21 +52,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
void SetImageLayout(Uint32 index, VkImageLayout layout);
|
||||
SizeT GetImageCount() const { return m_images.size(); }
|
||||
|
||||
// EGL content-validity tracking for the default framebuffer. A color
|
||||
// buffer's content is undefined once its image has been presented
|
||||
// (EGL_BUFFER_DESTROYED swap behaviour, the implementation default),
|
||||
// and every ancillary (depth/stencil) buffer's content is undefined
|
||||
// after ANY swap regardless of swap behaviour (EGL 1.5 §3.10.1). The
|
||||
// render-pass manager turns an undefined attachment's tile load into
|
||||
// LOAD_OP_DONT_CARE. Flags start false (a fresh swapchain image holds
|
||||
// garbage) and a render pass storing into an attachment sets it back
|
||||
// to defined.
|
||||
Bool IsImageContentDefined(Uint32 index) const;
|
||||
void SetImageContentDefined(Uint32 index, Bool defined);
|
||||
Bool IsDepthStencilContentDefined(Uint32 index) const;
|
||||
void SetDepthStencilContentDefined(Uint32 index, Bool defined);
|
||||
void SetAllDepthStencilContentUndefined();
|
||||
|
||||
private:
|
||||
void CreateImageViews(VkDevice device);
|
||||
void CreateDepthStencilResources(VkDevice device, VkPhysicalDevice physicalDevice);
|
||||
@@ -92,7 +77,5 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Vector<VkDeviceMemory> m_depthStencilImageMemories;
|
||||
Vector<VkImageView> m_depthStencilImageViews;
|
||||
Vector<VkImageLayout> m_depthStencilImageLayouts;
|
||||
Vector<Bool> m_imageContentDefined;
|
||||
Vector<Bool> m_depthStencilContentDefined;
|
||||
};
|
||||
} // namespace MobileGL::MG_Backend::DirectVulkan
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
#include "MG_Util/Converters/GLToMG/TextureEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToStr/FramebufferEnumConverter.h"
|
||||
#include "MG_Util/Converters/MGToVk/TextureEnumConverter.h"
|
||||
#include <vulkan/utility/vk_format_utils.h>
|
||||
#include "MG_Util/Metrics/TextureMetrics.h"
|
||||
#include <Config.h>
|
||||
#include <cstdio>
|
||||
@@ -446,6 +447,64 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return outImageInfo.sampler != VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
namespace {
|
||||
// fp16 carries an 11-bit mantissa, so an 8-bit normalized channel round-trips exactly.
|
||||
// Anything wider - 16-bit normalized, half float, full float, and every packed HDR
|
||||
// encoding - holds precision or range that relaxing the arithmetic would throw away.
|
||||
Bool IsLowPrecisionNormalizedFormat(VkFormat format) {
|
||||
if (format == VK_FORMAT_UNDEFINED) return false;
|
||||
if (!vkuFormatIsUNORM(format) && !vkuFormatIsSNORM(format) && !vkuFormatIsSRGB(format)) {
|
||||
return false;
|
||||
}
|
||||
const struct VKU_FORMAT_INFO info = vkuGetFormatInfo(format);
|
||||
for (Uint32 i = 0; i < info.component_count; ++i) {
|
||||
if (info.components[i].size > 8) return false;
|
||||
}
|
||||
return info.component_count > 0;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Bool UniformManager::DrawTargetIsLowPrecision(const MG_State::GLState::FramebufferObject* drawFramebuffer) {
|
||||
// Default framebuffer: the swapchain is an 8-bit normalized surface.
|
||||
if (drawFramebuffer == nullptr) return true;
|
||||
|
||||
Bool sawColour = false;
|
||||
for (Int i = static_cast<Int>(FramebufferAttachmentType::Color0);
|
||||
i < static_cast<Int>(FramebufferAttachmentType::FramebufferAttachmentTypeCount);
|
||||
++i) {
|
||||
const auto& attachment =
|
||||
drawFramebuffer->GetAttachment(static_cast<FramebufferAttachmentType>(i));
|
||||
VkFormat format = VK_FORMAT_UNDEFINED;
|
||||
if (const auto& texture = attachment.GetTexture()) {
|
||||
format = MG_Util::ConvertTextureInternalFormatToVkEnum(texture->GetFormat());
|
||||
} else if (const auto& renderbuffer = attachment.GetRenderbuffer()) {
|
||||
format = MG_Util::ConvertTextureInternalFormatToVkEnum(
|
||||
renderbuffer->GetInternalFormat());
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
if (!IsLowPrecisionNormalizedFormat(format)) return false;
|
||||
sawColour = true;
|
||||
}
|
||||
return sawColour;
|
||||
}
|
||||
|
||||
Bool UniformManager::ProgramSamplesOnlyLowPrecisionTextures(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj) {
|
||||
for (Uint32 binding = 0; binding < programObj.bindingKinds.size(); ++binding) {
|
||||
if (programObj.bindingKinds[binding] != ProgramFactory::DescriptorBindingKind::CombinedImageSampler) {
|
||||
continue;
|
||||
}
|
||||
const auto* texture = ResolveSamplerTextureRaw(program, programObj, binding);
|
||||
// An unresolvable binding is unknown territory, not licence to relax.
|
||||
if (texture == nullptr) return false;
|
||||
const VkFormat format =
|
||||
MG_Util::ConvertTextureInternalFormatToVkEnum(texture->GetFormat());
|
||||
if (!IsLowPrecisionNormalizedFormat(format)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Bool UniformManager::ProgramSamplesOnlySingleLevelTextures(
|
||||
const MG_State::GLState::ProgramObject& program, const ProgramFactory::VkProgramObject& programObj) {
|
||||
Bool sawSampler = false;
|
||||
|
||||
@@ -76,6 +76,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
// ExplicitLod0Sampling SPIR-V rewrite safe to request. Deliberately conservative: it reads
|
||||
// only GL state, so a texture that ends up single-level for another reason (one uploaded
|
||||
// level under a wide level range) merely misses the rewrite.
|
||||
// True when every texture this program samples is an 8-bit-or-less normalized format, so
|
||||
// relaxing the fragment stage to fp16 cannot lose a bit the texel ever carried. Says
|
||||
// nothing about the render target - the caller must check that too.
|
||||
static Bool ProgramSamplesOnlyLowPrecisionTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
// True when every colour attachment the draw writes is an 8-bit-or-less normalized
|
||||
// format (nullptr = default framebuffer, which is). Blending happens at attachment
|
||||
// precision, so a wider target must keep the fragment stage at full precision.
|
||||
static Bool DrawTargetIsLowPrecision(const MG_State::GLState::FramebufferObject* drawFramebuffer);
|
||||
static Bool ProgramSamplesOnlySingleLevelTextures(const MG_State::GLState::ProgramObject& program,
|
||||
const ProgramFactory::VkProgramObject& programObj);
|
||||
|
||||
|
||||
@@ -481,8 +481,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
VkRenderPassManager::HashType VkRenderPassManager::ComputeHash(
|
||||
const MG_State::GLState::FramebufferObject& fbo, Uint32 swapchainImageIndex, Bool includePendingClear,
|
||||
Bool includeDefaultFboDepthStencil) {
|
||||
const MG_State::GLState::FramebufferObject& fbo, Uint32 swapchainImageIndex, Bool includePendingClear) {
|
||||
XXHASH_VERIFY(XXH64_reset(m_hashState, m_config.CacheVersion));
|
||||
const Bool isDefaultFbo = fbo.IsDefaultFramebuffer();
|
||||
if (isDefaultFbo) {
|
||||
@@ -561,17 +560,9 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
attachment <= FramebufferAttachmentType::BackRight);
|
||||
if (isDefaultColorAttachment) {
|
||||
currentLayout = m_swapchainObject.GetImageLayout(swapchainImageIndex);
|
||||
// Content validity feeds the attachment's loadOp (see the
|
||||
// creation path), so it must key the cache as well.
|
||||
if (!m_swapchainObject.IsImageContentDefined(swapchainImageIndex)) {
|
||||
currentLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
} else if (attachment == FramebufferAttachmentType::Depth ||
|
||||
attachment == FramebufferAttachmentType::Stencil) {
|
||||
currentLayout = m_swapchainObject.GetDepthStencilImageLayout(swapchainImageIndex);
|
||||
if (!m_swapchainObject.IsDepthStencilContentDefined(swapchainImageIndex)) {
|
||||
currentLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
auto* textureResource = m_textureManager.SyncTextureAndGetDescriptor(*texture);
|
||||
@@ -626,49 +617,14 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
combineFramebufferAttachmentObjHash(drawbuf);
|
||||
}
|
||||
|
||||
// The depth-less default-FBO flavor omits the depth/stencil attachment
|
||||
// entirely, so it must hash differently from the depth-full flavor.
|
||||
const Bool depthStencilIncluded = !isDefaultFbo || includeDefaultFboDepthStencil;
|
||||
XXHASH_VERIFY(XXH64_update(m_hashState, &depthStencilIncluded, sizeof(depthStencilIncluded)));
|
||||
if (depthStencilIncluded) {
|
||||
combineFramebufferAttachmentObjHash(FramebufferAttachmentType::Depth);
|
||||
combineFramebufferAttachmentObjHash(FramebufferAttachmentType::Stencil);
|
||||
}
|
||||
combineFramebufferAttachmentObjHash(FramebufferAttachmentType::Depth);
|
||||
combineFramebufferAttachmentObjHash(FramebufferAttachmentType::Stencil);
|
||||
|
||||
return XXH64_digest(m_hashState);
|
||||
}
|
||||
|
||||
RenderPassEntry& VkRenderPassManager::GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil) {
|
||||
// Resolve the default-FBO depth flavor (see the header comment): keep the
|
||||
// depth attachment when the caller needs it, when a depth/stencil clear is
|
||||
// pending, or when the active pass already carries it (escalate-only, so
|
||||
// alternating depth-less draws never split an established depth pass).
|
||||
Bool includeDefaultFboDepthStencil = true;
|
||||
if (fbo.IsDefaultFramebuffer()) {
|
||||
Bool activeDefaultHasDepthStencil = false;
|
||||
if (const auto* active = GetActiveRenderPass()) {
|
||||
Bool activeIsSwapchainPass = false;
|
||||
Bool activeHasSwapchainDepthStencil = false;
|
||||
for (const auto& tracked : active->trackedAttachmentLayouts) {
|
||||
activeIsSwapchainPass |= tracked.target == TrackedAttachmentTarget::SwapchainColor;
|
||||
activeHasSwapchainDepthStencil |=
|
||||
tracked.target == TrackedAttachmentTarget::SwapchainDepthStencil;
|
||||
}
|
||||
activeDefaultHasDepthStencil = activeIsSwapchainPass && activeHasSwapchainDepthStencil;
|
||||
}
|
||||
const auto& defaultDepthAtt = fbo.GetAttachment(FramebufferAttachmentType::Depth);
|
||||
const auto& defaultStencilAtt = fbo.GetAttachment(FramebufferAttachmentType::Stencil);
|
||||
const Bool pendingDepthStencilClear =
|
||||
(defaultDepthAtt.IsTexture() && m_clearManager.HasPendingClear(defaultDepthAtt)) ||
|
||||
HasPendingRenderbufferClear(defaultDepthAtt) ||
|
||||
(defaultStencilAtt.IsTexture() && m_clearManager.HasPendingClear(defaultStencilAtt)) ||
|
||||
HasPendingRenderbufferClear(defaultStencilAtt);
|
||||
includeDefaultFboDepthStencil =
|
||||
drawUsesDepthStencil || activeDefaultHasDepthStencil || pendingDepthStencilClear;
|
||||
}
|
||||
|
||||
Uint32 swapchainImageIndex) {
|
||||
auto hasPendingClearOnFramebuffer = [&]() -> Bool {
|
||||
const auto& drawBuffers = fbo.GetDrawBuffers();
|
||||
for (auto attachment : drawBuffers) {
|
||||
@@ -718,7 +674,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastFboVersion == fbo.GetObjectVersion() && m_rpFastSwapchainIndex == swapchainImageIndex &&
|
||||
m_rpFastTexEpoch == m_textureManager.GetTextureImageEpoch() &&
|
||||
m_rpFastRbEpoch == m_renderbufferImageEpoch &&
|
||||
(!fbo.IsDefaultFramebuffer() || m_rpFastHadDepthStencil == includeDefaultFboDepthStencil) &&
|
||||
m_rpFastRenderPassHash == activeRenderPass->hash && !hasPendingClearOnFramebuffer()) {
|
||||
auto activeIt = m_renderPasses.find(activeRenderPass->hash);
|
||||
if (activeIt != m_renderPasses.end()) {
|
||||
@@ -727,7 +682,7 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
}
|
||||
|
||||
auto compatibilityHash = ComputeHash(fbo, swapchainImageIndex, false, includeDefaultFboDepthStencil);
|
||||
auto compatibilityHash = ComputeHash(fbo, swapchainImageIndex, false);
|
||||
if (activeRenderPass != nullptr &&
|
||||
activeRenderPass->CompatibleWith(compatibilityHash) &&
|
||||
!hasPendingClearOnFramebuffer()) {
|
||||
@@ -744,11 +699,10 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
m_rpFastTexEpoch = m_textureManager.GetTextureImageEpoch();
|
||||
m_rpFastRbEpoch = m_renderbufferImageEpoch;
|
||||
m_rpFastRenderPassHash = activeRenderPass->hash;
|
||||
m_rpFastHadDepthStencil = activeIt->second.hasDepthStencilAttachment;
|
||||
activeIt->second.lastUsedFrame = m_frameCounter;
|
||||
return activeIt->second;
|
||||
}
|
||||
auto hash = ComputeHash(fbo, swapchainImageIndex, true, includeDefaultFboDepthStencil);
|
||||
auto hash = ComputeHash(fbo, swapchainImageIndex, true);
|
||||
auto it = m_renderPasses.find(hash);
|
||||
if (it != m_renderPasses.end()) {
|
||||
it->second.lastUsedFrame = m_frameCounter;
|
||||
@@ -940,13 +894,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(swapchainImageIndex < swapchainViews.size(),
|
||||
"GetOrCreateRenderPass: swapchain image index out of range");
|
||||
trackedColorLayout = m_swapchainObject.GetImageLayout(swapchainImageIndex);
|
||||
// EGL: a presented color buffer's content is undefined when its
|
||||
// image comes back around (EGL_BUFFER_DESTROYED, the default
|
||||
// swap behaviour) - skip the tile load instead of reloading
|
||||
// stale pixels nobody may rely on.
|
||||
if (!hasClear && !m_swapchainObject.IsImageContentDefined(swapchainImageIndex)) {
|
||||
trackedColorLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
trackedAttachmentLayouts.emplace_back(TrackedAttachmentLayoutInfo {
|
||||
.target = TrackedAttachmentTarget::SwapchainColor,
|
||||
.swapchainImageIndex = swapchainImageIndex,
|
||||
@@ -1029,12 +976,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
};
|
||||
const auto* selectedDepthStencilAttachment = isUsableDepthStencilAttachment(depthAtt) ? &depthAtt :
|
||||
(isUsableDepthStencilAttachment(stencilAtt) ? &stencilAtt : nullptr);
|
||||
// Depth-less default-FBO flavor: nothing in this pass touches depth/stencil
|
||||
// and their content is undefined anyway (EGL swap), so drop the attachment
|
||||
// and its whole tile load + store.
|
||||
if (isDefaultFbo && !includeDefaultFboDepthStencil) {
|
||||
selectedDepthStencilAttachment = nullptr;
|
||||
}
|
||||
const Bool hasDistinctDepthAndStencilAttachments =
|
||||
isUsableDepthStencilAttachment(depthAtt) && isUsableDepthStencilAttachment(stencilAtt) &&
|
||||
!sameDepthStencilAttachmentObject(depthAtt, stencilAtt);
|
||||
@@ -1053,12 +994,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VkImageLayout trackedDepthLayout = isDefaultFbo ?
|
||||
m_swapchainObject.GetDepthStencilImageLayout(swapchainImageIndex) :
|
||||
VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
||||
// EGL 1.5 §3.10.1: every ancillary (depth/stencil) buffer's content is
|
||||
// undefined after a swap, so the first default-FBO pass of a frame can
|
||||
// skip the depth/stencil tile load outright.
|
||||
if (isDefaultFbo && !m_swapchainObject.IsDepthStencilContentDefined(swapchainImageIndex)) {
|
||||
trackedDepthLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
}
|
||||
depthAttachmentDescription.flags = 0;
|
||||
VkSampleCountFlagBits depthAttachmentSampleCount = VK_SAMPLE_COUNT_1_BIT;
|
||||
Int depthAttachmentId = 0;
|
||||
@@ -1187,22 +1122,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
const Bool hasDepthStencilAttachment = depthAttachmentRef.attachment != VK_ATTACHMENT_UNUSED;
|
||||
|
||||
// Declare only the used colour-reference span. The GL draw-buffer array
|
||||
// always spans 8 slots, so passes used to declare colorAttachmentCount=8
|
||||
// with trailing VK_ATTACHMENT_UNUSED holes - and Adreno configures its
|
||||
// per-pixel render-backend/export path from the DECLARED count, so every
|
||||
// fragment of every pass paid the 8-target export cost (measured on
|
||||
// Adreno 650 / MC 26.2: 11.9 -> 7.5 ms of GPU time per frame, with the
|
||||
// single-quad swapchain blit pass alone dropping 1.26 -> 0.40 ms).
|
||||
// Interior GL_NONE holes keep their slots so fragment-output locations
|
||||
// still line up; a fragment output at a location past the trimmed count
|
||||
// is discarded, which is exactly GL's semantic for writing to a draw
|
||||
// buffer set to GL_NONE.
|
||||
while (!colorAttachmentRefs.empty() &&
|
||||
colorAttachmentRefs.back().attachment == VK_ATTACHMENT_UNUSED) {
|
||||
colorAttachmentRefs.pop_back();
|
||||
}
|
||||
|
||||
// Subpass
|
||||
VkSubpassDescription subpassDesc;
|
||||
subpassDesc.flags = 0;
|
||||
@@ -1411,17 +1330,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
renderPassBeginInfo.pClearValues = clearValues.data();
|
||||
|
||||
vkCmdBeginRenderPass(commandBuffer, &renderPassBeginInfo, VK_SUBPASS_CONTENTS_INLINE);
|
||||
// Pre-pass stream bookkeeping: this pass's attachment images are now
|
||||
// referenced by the open frame recording.
|
||||
if (s_textureManager != nullptr) {
|
||||
for (const auto& tracked : renderPassEntry.trackedAttachmentLayouts) {
|
||||
if (tracked.target == TrackedAttachmentTarget::Texture) {
|
||||
if (const auto texture = tracked.texture.lock()) {
|
||||
s_textureManager->StampTextureRecordingUse(texture.get());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const auto& pending: renderPassEntry.pendingClearAttachments) {
|
||||
if (pending.hasInlinePayload) {
|
||||
if (s_renderPassManager != nullptr) {
|
||||
@@ -1474,15 +1382,11 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
case TrackedAttachmentTarget::SwapchainColor:
|
||||
MOBILEGL_ASSERT(s_swapchainObject != nullptr, "EndRenderPass: swapchain object is null");
|
||||
s_swapchainObject->SetImageLayout(trackedAttachment.swapchainImageIndex, trackedAttachment.finalLayout);
|
||||
// The pass stored into the attachment: its content is defined
|
||||
// until the image is next presented.
|
||||
s_swapchainObject->SetImageContentDefined(trackedAttachment.swapchainImageIndex, true);
|
||||
break;
|
||||
case TrackedAttachmentTarget::SwapchainDepthStencil:
|
||||
MOBILEGL_ASSERT(s_swapchainObject != nullptr, "EndRenderPass: swapchain object is null");
|
||||
s_swapchainObject->SetDepthStencilImageLayout(trackedAttachment.swapchainImageIndex,
|
||||
trackedAttachment.finalLayout);
|
||||
s_swapchainObject->SetDepthStencilContentDefined(trackedAttachment.swapchainImageIndex, true);
|
||||
break;
|
||||
default:
|
||||
MOBILEGL_ASSERT(false, "EndRenderPass: unsupported tracked attachment target=%d",
|
||||
|
||||
@@ -188,22 +188,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
HashType ComputeHash(
|
||||
const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool includePendingClear = true,
|
||||
Bool includeDefaultFboDepthStencil = true);
|
||||
// drawUsesDepthStencil: whether the operation about to run inside the pass
|
||||
// reads or writes the depth/stencil buffer (depth test or stencil test
|
||||
// enabled, or a depth/stencil clear). Only consulted for the DEFAULT
|
||||
// framebuffer: EGL undefines its ancillary buffers at every swap, so a
|
||||
// default-FBO pass whose draws provably never touch depth/stencil is
|
||||
// created WITHOUT the depth attachment - on a tiler that skips the whole
|
||||
// depth tile load AND store. The flavor only escalates: once a pass with
|
||||
// depth is active, later depth-less draws keep using it, and a depth-using
|
||||
// draw against a depth-less active pass resolves to a new (incompatible)
|
||||
// entry, which the caller's compatibility check turns into a pass split;
|
||||
// the new pass's depth loads DONT_CARE (content was undefined all along).
|
||||
RenderPassEntry& GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo,
|
||||
Uint32 swapchainImageIndex,
|
||||
Bool drawUsesDepthStencil = true);
|
||||
Bool includePendingClear = true);
|
||||
RenderPassEntry& GetOrCreateRenderPass(const MG_State::GLState::FramebufferObject& fbo, Uint32 swapchainImageIndex);
|
||||
void QueueRenderbufferClear(GLbitfield mask, const ClearFramebufferPayload& clearPayload,
|
||||
const MG_State::GLState::FramebufferObject& drawFbo);
|
||||
void QueueRenderbufferClear(const ClearAttachmentPayload& clearPayload,
|
||||
@@ -245,10 +231,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
Uint64 m_rpFastTexEpoch = 0;
|
||||
Uint64 m_rpFastRbEpoch = 0;
|
||||
Uint64 m_rpFastRenderPassHash = 0;
|
||||
// Whether the memoized entry carries a depth/stencil attachment; a
|
||||
// default-FBO resolution whose effective depth request differs must
|
||||
// miss the memo (the depth-less/depth-full flavors hash differently).
|
||||
Bool m_rpFastHadDepthStencil = false;
|
||||
|
||||
public:
|
||||
struct RenderbufferResource {
|
||||
|
||||
@@ -1049,16 +1049,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return view;
|
||||
}
|
||||
|
||||
void VkTextureManager::StampTextureRecordingUse(MG_State::GLState::ITextureObject* texture) {
|
||||
if (texture == nullptr) {
|
||||
return;
|
||||
}
|
||||
auto it = m_textureResources.find(MakeTextureIdentity(texture));
|
||||
if (it != m_textureResources.end()) {
|
||||
it->second.lastRecordingGeneration = m_recordingGeneration;
|
||||
}
|
||||
}
|
||||
|
||||
void VkTextureManager::UpdateTrackedImageLayout(MG_State::GLState::ITextureObject* texture, VkImageLayout newLayout) {
|
||||
MOBILEGL_ASSERT(texture != nullptr, "UpdateTrackedImageLayout: texture is null");
|
||||
auto it = m_textureResources.find(MakeTextureIdentity(texture));
|
||||
@@ -1086,8 +1076,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
MOBILEGL_ASSERT(writtenMipLevel < resource.mipLevels,
|
||||
"UpdateTrackedImageLayoutAfterAttachmentWrite: textureId=%d mipLevel=%u out of range %u",
|
||||
texture->GetExternalIndex(), writtenMipLevel, resource.mipLevels);
|
||||
// Pre-pass stream bookkeeping: the render pass that just ended wrote this image.
|
||||
StampResourceRecordingUse(resource);
|
||||
|
||||
if (resource.layout != newLayout && resource.mipLevels > 1) {
|
||||
VkPipelineStageFlags srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
@@ -1172,8 +1160,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
VK_ACCESS_SHADER_READ_BIT, resource->aspect, 0, resource->mipLevels,
|
||||
resource->arrayLayers);
|
||||
MOBILEGL_ASSERT(ok, "TransitionTextureForSampling: transition failed for textureId=%d", texture.GetExternalIndex());
|
||||
// Pre-pass stream bookkeeping: a command referencing the image was recorded.
|
||||
StampResourceRecordingUse(*resource);
|
||||
return ok;
|
||||
}
|
||||
|
||||
@@ -1203,8 +1189,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
resource->aspect, 0, resource->mipLevels, resource->arrayLayers);
|
||||
MOBILEGL_ASSERT(ok, "TransitionTextureForStorageImage: transition failed for textureId=%d",
|
||||
texture.GetExternalIndex());
|
||||
// Pre-pass stream bookkeeping: a command referencing the image was recorded.
|
||||
StampResourceRecordingUse(*resource);
|
||||
return ok;
|
||||
}
|
||||
|
||||
@@ -1436,17 +1420,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return false;
|
||||
}
|
||||
const Bool isMultisampleTexture = IsMultisampleTextureUploadTarget(uploadTarget);
|
||||
// A texture that has only ever defined level 0 gets a single-level backing
|
||||
// (ANGLE's model). Preallocating the full chain put every render target
|
||||
// onto Adreno's multi-mip image layout and grew each texture by a third
|
||||
// for levels most textures never define. Once a second level is defined
|
||||
// the backing is recreated ONE time with the full chain (the
|
||||
// preserve-copy path below carries the pixels over), so sequentially-
|
||||
// defined atlas mips do not recreate per level, and glGenerateMipmap -
|
||||
// which defines every level before syncing - works unchanged.
|
||||
const Uint32 backingMipLevels =
|
||||
isMultisampleTexture ? 1u
|
||||
: (mipLevels > 1 ? std::max(mipLevels, ComputeFullMipLevelCount(texelSize)) : 1u);
|
||||
isMultisampleTexture ? 1u : std::max(mipLevels, ComputeFullMipLevelCount(texelSize));
|
||||
TextureShapeInfo shapeInfo{};
|
||||
const Bool supportedShape = TryResolveTextureShapeInfo(texture, uploadTarget, texelSize, shapeInfo);
|
||||
MOBILEGL_ASSERT(supportedShape,
|
||||
|
||||
@@ -172,13 +172,6 @@ public:
|
||||
// NeedsStorageImagePreparation cannot ask for a recreate that will never happen.
|
||||
Bool storageUsageResolved = false;
|
||||
Uint16 syncedTextureParamsVersion = 0;
|
||||
// Recording generation (VkTextureManager::GetRecordingGeneration) of the last
|
||||
// command referencing this image that was recorded into the CURRENT frame
|
||||
// command buffer. An image untouched by the open recording may have its
|
||||
// out-of-pass work (deferred clears, sampled-layout transitions) recorded
|
||||
// into the frame's PRE command buffer - which executes strictly before the
|
||||
// frame's commands - instead of splitting the active render pass.
|
||||
Uint64 lastRecordingGeneration = 0;
|
||||
// Snapshot of ITextureObject::GetContentVersion() at the last successful sync;
|
||||
// lets SyncTexture skip the whole re-check/re-upload when content is unchanged.
|
||||
Uint64 syncedContentVersion = 0;
|
||||
@@ -214,7 +207,6 @@ public:
|
||||
std::swap(this->usageFlags, that.usageFlags);
|
||||
std::swap(this->storageUsageResolved, that.storageUsageResolved);
|
||||
std::swap(this->syncedTextureParamsVersion, that.syncedTextureParamsVersion);
|
||||
std::swap(this->lastRecordingGeneration, that.lastRecordingGeneration);
|
||||
std::swap(this->syncedContentVersion, that.syncedContentVersion);
|
||||
std::swap(this->syncedMipLevelCount, that.syncedMipLevelCount);
|
||||
}
|
||||
@@ -315,21 +307,6 @@ public:
|
||||
VkImageLayout newLayout);
|
||||
Bool TransitionTextureForSampling(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
Bool TransitionTextureForStorageImage(VkCommandBuffer commandBuffer, MG_State::GLState::ITextureObject& texture);
|
||||
|
||||
// Recording-generation bookkeeping for the pre-pass command stream. The
|
||||
// generation advances every time the frame command buffer (re)begins
|
||||
// recording; a resource whose stamp does not match was not referenced by
|
||||
// any command in the open recording, so its out-of-pass work may safely
|
||||
// execute ahead of the whole recording (in the pre command buffer).
|
||||
void AdvanceRecordingGeneration() { ++m_recordingGeneration; }
|
||||
void StampResourceRecordingUse(TextureResource& resource) const {
|
||||
resource.lastRecordingGeneration = m_recordingGeneration;
|
||||
}
|
||||
// Map-lookup variant for callers that only hold the GL texture object.
|
||||
void StampTextureRecordingUse(MG_State::GLState::ITextureObject* texture);
|
||||
Bool WasTouchedThisRecording(const TextureResource& resource) const {
|
||||
return resource.lastRecordingGeneration == m_recordingGeneration;
|
||||
}
|
||||
// Records that this texture is bound to a GL image unit, so its image must carry
|
||||
// VK_IMAGE_USAGE_STORAGE_BIT. Must be called before NeedsStorageImagePreparation, and
|
||||
// therefore before the render pass is committed: an image that has to be upgraded is
|
||||
@@ -387,9 +364,6 @@ public:
|
||||
private:
|
||||
// Bumped in SyncTextureResource right after vmaCreateImage(texture). See GetTextureImageEpoch().
|
||||
Uint64 m_textureImageEpoch = 1;
|
||||
// See AdvanceRecordingGeneration. Starts above every resource's default
|
||||
// stamp of 0 so a fresh resource counts as untouched.
|
||||
Uint64 m_recordingGeneration = 1;
|
||||
|
||||
Bool SyncTexture(MG_State::GLState::ITextureObject &texture,
|
||||
TextureResource &outResource);
|
||||
|
||||
@@ -217,53 +217,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
return static_cast<Int>((static_cast<Int64>(value) * toExtent + fromExtent / 2) / fromExtent);
|
||||
}
|
||||
|
||||
// Redundant dynamic-state elimination for the per-draw hot path: within one
|
||||
// command-buffer recording, a vkCmdSet* whose values already match what the
|
||||
// command buffer holds is skipped. Valid because every PipelineFactory
|
||||
// pipeline declares the same eight dynamic states, so the values persist
|
||||
// across those pipeline binds; the shadow resets whenever a recording
|
||||
// (re)begins, and whenever an auxiliary pipeline with a narrower dynamic
|
||||
// set (blit, depth-mipmap) binds - their static state makes the
|
||||
// corresponding dynamic values undefined per the spec.
|
||||
struct DynamicStateShadow {
|
||||
Bool viewportValid = false;
|
||||
VkViewport viewport{};
|
||||
Bool scissorValid = false;
|
||||
VkRect2D scissor{};
|
||||
Bool blendConstantsValid = false;
|
||||
Float blendConstants[4] = {0.0f, 0.0f, 0.0f, 0.0f};
|
||||
Bool depthBiasValid = false;
|
||||
Float depthBiasConstantFactor = 0.0f;
|
||||
Float depthBiasSlopeFactor = 0.0f;
|
||||
Bool lineWidthValid = false;
|
||||
Float lineWidth = 0.0f;
|
||||
Bool stencilValid = false;
|
||||
Uint32 stencilFrontCompareMask = 0;
|
||||
Uint32 stencilBackCompareMask = 0;
|
||||
Uint32 stencilFrontWriteMask = 0;
|
||||
Uint32 stencilBackWriteMask = 0;
|
||||
Uint32 stencilFrontReference = 0;
|
||||
Uint32 stencilBackReference = 0;
|
||||
};
|
||||
static DynamicStateShadow g_dynamicStateShadow;
|
||||
|
||||
static void ResetDynamicStateShadow() {
|
||||
g_dynamicStateShadow = {};
|
||||
}
|
||||
|
||||
static void ShadowedSetScissor(VkCommandBuffer commandBuffer, const VkRect2D& scissor) {
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.scissorValid && shadow.scissor.offset.x == scissor.offset.x &&
|
||||
shadow.scissor.offset.y == scissor.offset.y &&
|
||||
shadow.scissor.extent.width == scissor.extent.width &&
|
||||
shadow.scissor.extent.height == scissor.extent.height) {
|
||||
return;
|
||||
}
|
||||
shadow.scissorValid = true;
|
||||
shadow.scissor = scissor;
|
||||
vkCmdSetScissor(commandBuffer, 0, 1, &scissor);
|
||||
}
|
||||
|
||||
static void ApplyGLViewportState(VkCommandBuffer commandBuffer,
|
||||
const IntVec2& framebufferExtent,
|
||||
VkSurfaceTransformFlagBitsKHR preTransform,
|
||||
@@ -293,14 +246,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
viewport.height = static_cast<float>(viewportHeight);
|
||||
viewport.minDepth = depthRange.x();
|
||||
viewport.maxDepth = depthRange.y();
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.viewportValid && shadow.viewport.x == viewport.x && shadow.viewport.y == viewport.y &&
|
||||
shadow.viewport.width == viewport.width && shadow.viewport.height == viewport.height &&
|
||||
shadow.viewport.minDepth == viewport.minDepth && shadow.viewport.maxDepth == viewport.maxDepth) {
|
||||
return;
|
||||
}
|
||||
shadow.viewportValid = true;
|
||||
shadow.viewport = viewport;
|
||||
vkCmdSetViewport(commandBuffer, 0, 1, &viewport);
|
||||
}
|
||||
|
||||
@@ -312,17 +257,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
blendColor.z(),
|
||||
blendColor.w(),
|
||||
};
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.blendConstantsValid && shadow.blendConstants[0] == blendConstants[0] &&
|
||||
shadow.blendConstants[1] == blendConstants[1] && shadow.blendConstants[2] == blendConstants[2] &&
|
||||
shadow.blendConstants[3] == blendConstants[3]) {
|
||||
return;
|
||||
}
|
||||
shadow.blendConstantsValid = true;
|
||||
shadow.blendConstants[0] = blendConstants[0];
|
||||
shadow.blendConstants[1] = blendConstants[1];
|
||||
shadow.blendConstants[2] = blendConstants[2];
|
||||
shadow.blendConstants[3] = blendConstants[3];
|
||||
vkCmdSetBlendConstants(commandBuffer, blendConstants);
|
||||
}
|
||||
|
||||
@@ -338,17 +272,8 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
}
|
||||
|
||||
static void ApplyPolygonOffsetState(VkCommandBuffer commandBuffer) {
|
||||
const Float constantFactor = MG_State::pGLContext->GetPolygonOffsetUnits();
|
||||
const Float slopeFactor = MG_State::pGLContext->GetPolygonOffsetFactor();
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.depthBiasValid && shadow.depthBiasConstantFactor == constantFactor &&
|
||||
shadow.depthBiasSlopeFactor == slopeFactor) {
|
||||
return;
|
||||
}
|
||||
shadow.depthBiasValid = true;
|
||||
shadow.depthBiasConstantFactor = constantFactor;
|
||||
shadow.depthBiasSlopeFactor = slopeFactor;
|
||||
vkCmdSetDepthBias(commandBuffer, constantFactor, 0.0f, slopeFactor);
|
||||
vkCmdSetDepthBias(commandBuffer, MG_State::pGLContext->GetPolygonOffsetUnits(), 0.0f,
|
||||
MG_State::pGLContext->GetPolygonOffsetFactor());
|
||||
}
|
||||
|
||||
static void ApplyLineWidthState(VkCommandBuffer commandBuffer) {
|
||||
@@ -363,12 +288,6 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
lineWidth = maxLineWidth;
|
||||
}
|
||||
}
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.lineWidthValid && shadow.lineWidth == lineWidth) {
|
||||
return;
|
||||
}
|
||||
shadow.lineWidthValid = true;
|
||||
shadow.lineWidth = lineWidth;
|
||||
vkCmdSetLineWidth(commandBuffer, lineWidth);
|
||||
}
|
||||
|
||||
@@ -417,31 +336,15 @@ namespace MobileGL::MG_Backend::DirectVulkan {
|
||||
static void ApplyStencilState(VkCommandBuffer commandBuffer) {
|
||||
const StencilFaceState& frontStencil = MG_State::pGLContext->GetStencilState(StencilFace::Front);
|
||||
const StencilFaceState& backStencil = MG_State::pGLContext->GetStencilState(StencilFace::Back);
|
||||
const Uint32 frontReference = static_cast<Uint32>(std::max(frontStencil.Ref, 0));
|
||||
const Uint32 backReference = static_cast<Uint32>(std::max(backStencil.Ref, 0));
|
||||
|
||||
auto& shadow = g_dynamicStateShadow;
|
||||
if (shadow.stencilValid && shadow.stencilFrontCompareMask == frontStencil.ValueMask &&
|
||||
shadow.stencilBackCompareMask == backStencil.ValueMask &&
|
||||
shadow.stencilFrontWriteMask == frontStencil.WriteMask &&
|
||||
shadow.stencilBackWriteMask == backStencil.WriteMask &&
|
||||
shadow.stencilFrontReference == frontReference && shadow.stencilBackReference == backReference) {
|
||||
return;
|
||||
}
|
||||
shadow.stencilValid = true;
|
||||
shadow.stencilFrontCompareMask = frontStencil.ValueMask;
|
||||
shadow.stencilBackCompareMask = backStencil.ValueMask;
|
||||
shadow.stencilFrontWriteMask = frontStencil.WriteMask;
|
||||
shadow.stencilBackWriteMask = backStencil.WriteMask;
|
||||
shadow.stencilFrontReference = frontReference;
|
||||
shadow.stencilBackReference = backReference;
|
||||
|
||||
vkCmdSetStencilCompareMask(commandBuffer, VK_STENCIL_FACE_FRONT_BIT, frontStencil.ValueMask);
|
||||
vkCmdSetStencilCompareMask(commandBuffer, VK_STENCIL_FACE_BACK_BIT, backStencil.ValueMask);
|
||||
vkCmdSetStencilWriteMask(commandBuffer, VK_STENCIL_FACE_FRONT_BIT, frontStencil.WriteMask);
|
||||
vkCmdSetStencilWriteMask(commandBuffer, VK_STENCIL_FACE_BACK_BIT, backStencil.WriteMask);
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_FRONT_BIT, frontReference);
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_BACK_BIT, backReference);
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_FRONT_BIT,
|
||||
static_cast<Uint32>(std::max(frontStencil.Ref, 0)));
|
||||
vkCmdSetStencilReference(commandBuffer, VK_STENCIL_FACE_BACK_BIT,
|
||||
static_cast<Uint32>(std::max(backStencil.Ref, 0)));
|
||||
}
|
||||
|
||||
enum class NumericDomain {
|
||||
@@ -3750,10 +3653,6 @@ void main() {
|
||||
vkCmdSetScissor(frame.commandBuffer, 0, 1, &scissor);
|
||||
|
||||
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
|
||||
// The depth-mipmap pipeline's narrower dynamic set (viewport/scissor
|
||||
// only) leaves the other dynamic states undefined; its raw scissor
|
||||
// and viewport writes also bypass the shadow.
|
||||
ResetDynamicStateShadow();
|
||||
|
||||
std::fill(depthProgramData,
|
||||
depthProgramData + m_depthMipmapResources.program->GetUBOSize(),
|
||||
@@ -4041,15 +3940,12 @@ void main() {
|
||||
payload.backStencilCompareOp = VK_COMPARE_OP_ALWAYS;
|
||||
}
|
||||
const Uint32 fragmentOutputMask = programObj.activeFragmentOutputLocationMask;
|
||||
// Outputs at locations past the render pass's trimmed colour span are
|
||||
// simply discarded - GL's semantic for a fragment output whose draw
|
||||
// buffer is GL_NONE (the trailing UNUSED slots no longer occupy
|
||||
// references, see GetOrCreateRenderPass).
|
||||
if ((fragmentOutputMask >> payload.colorAttachmentCount) != 0) {
|
||||
MGLOG_D("GetOrCreatePipeline: fragmentOutputMask=0x%x exceeds colorAttachmentCount=%u for program=%u; "
|
||||
"outputs past the span are discarded",
|
||||
fragmentOutputMask, payload.colorAttachmentCount, program.GetExternalIndex());
|
||||
}
|
||||
MOBILEGL_ASSERT(
|
||||
(fragmentOutputMask >> payload.colorAttachmentCount) == 0,
|
||||
"GetOrCreatePipeline: fragmentOutputMask=0x%x exceeds colorAttachmentCount=%u for program=%u",
|
||||
fragmentOutputMask,
|
||||
payload.colorAttachmentCount,
|
||||
program.GetExternalIndex());
|
||||
MOBILEGL_ASSERT(payload.colorAttachmentCount <= PipelineFactory::PipelineCreatePayload::kMaxColorAttachments,
|
||||
"GetOrCreatePipeline: colorAttachmentCount=%u exceeds payload capacity",
|
||||
payload.colorAttachmentCount);
|
||||
@@ -4419,6 +4315,15 @@ void main() {
|
||||
transformFlags |= ProgramFactory::CompileOptionBit::ExplicitLod0Sampling;
|
||||
programObjPtr = &m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
}
|
||||
// fp16 fragment arithmetic is only sound when nothing this draw reads or writes carries
|
||||
// more than 8 normalized bits per channel. A shaderpack's HDR gbuffer, or a data texture
|
||||
// holding positions, must keep full precision - and SPIR-V cannot tell, since sampler2D
|
||||
// yields vec4 whatever the bound format is, so the decision has to be made here.
|
||||
if (UniformManager::ProgramSamplesOnlyLowPrecisionTextures(program, *programObjPtr) &&
|
||||
UniformManager::DrawTargetIsLowPrecision(drawFbo.get())) {
|
||||
transformFlags |= ProgramFactory::CompileOptionBit::RelaxedFragmentPrecision;
|
||||
programObjPtr = &m_programFactory->GetOrCreateProgram(program, transformFlags);
|
||||
}
|
||||
const auto& programObj = *programObjPtr;
|
||||
|
||||
// Begin command recording if not yet
|
||||
@@ -4501,30 +4406,8 @@ void main() {
|
||||
static_cast<Int>(textureResource->layout));
|
||||
if (m_clearManager->HasPendingClear(sampledTexture) ||
|
||||
!IsValidSampledImageLayout(textureResource->layout)) {
|
||||
// Out-of-pass work is needed (deferred clear materialization or
|
||||
// a sampled-layout transition). When the open frame recording
|
||||
// has not referenced this image yet, that work can execute
|
||||
// ahead of the WHOLE recording - record it into the pre-pass
|
||||
// stream instead of splitting the active render pass (ANGLE's
|
||||
// outside-render-pass command stream, restricted to the
|
||||
// provably reorderable case).
|
||||
if (activeRenderPass != nullptr &&
|
||||
!m_frameContext.GetCurrent().hasPreCommandBufferRecorded &&
|
||||
!m_textureManager->WasTouchedThisRecording(*textureResource)) {
|
||||
VkCommandBuffer preCommandBuffer = m_frameContext.BeginPreCommandRecording();
|
||||
const Bool preClearReady =
|
||||
MaterializePendingClearForTexture(preCommandBuffer, *sampledTexture);
|
||||
MOBILEGL_ASSERT(preClearReady,
|
||||
"%s: pre-pass MaterializePendingClearForTexture failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
const Bool preTransitionReady =
|
||||
m_textureManager->TransitionTextureForSampling(preCommandBuffer, *sampledTexture);
|
||||
MOBILEGL_ASSERT(preTransitionReady,
|
||||
"%s: pre-pass TransitionTextureForSampling failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
continue;
|
||||
}
|
||||
needSampledTextureTransitions = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4548,28 +4431,16 @@ void main() {
|
||||
MOBILEGL_ASSERT(transitionedResource != nullptr,
|
||||
"%s: post-transition SyncTextureAndGetDescriptor failed for textureId=%d",
|
||||
__func__, sampledTexture->GetExternalIndex());
|
||||
// Pre-pass stream bookkeeping: the draw about to be recorded reads
|
||||
// this image, so later out-of-pass work on it can no longer jump
|
||||
// ahead of the recording.
|
||||
m_textureManager->StampResourceRecordingUse(*transitionedResource);
|
||||
MGLOG_D("SetupDraw: sampled textureId=%d layout(after)=%s(%d)",
|
||||
sampledTexture->GetExternalIndex(), VkImageLayoutToString(transitionedResource->layout),
|
||||
static_cast<Int>(transitionedResource->layout));
|
||||
}
|
||||
|
||||
// Depth/stencil participation of THIS draw, for the default-FBO depth-less
|
||||
// pass flavor (GL: a disabled depth/stencil test neither reads nor writes
|
||||
// its buffer).
|
||||
const Bool drawUsesDepthStencil =
|
||||
MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::DepthTest) ||
|
||||
MG_State::pGLContext->IsCapabilityEnabled(CapabilityInput::StencilTest);
|
||||
auto* renderPassEntry =
|
||||
&m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired, drawUsesDepthStencil);
|
||||
auto* renderPassEntry = &m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired);
|
||||
if (activeRenderPass && !activeRenderPass->CompatibleWith(*renderPassEntry)) {
|
||||
VkRenderPassManager::EndRenderPass(frame.commandBuffer);
|
||||
activeRenderPass = nullptr;
|
||||
renderPassEntry =
|
||||
&m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired, drawUsesDepthStencil);
|
||||
renderPassEntry = &m_renderPassManager->GetOrCreateRenderPass(*drawFbo, m_imageIndexAcquired);
|
||||
}
|
||||
if (renderPassEntry->attachmentCount == 0 || renderPassEntry->extent.x() <= 0 || renderPassEntry->extent.y() <= 0) {
|
||||
MGLOG_D("SetupDraw skipped: drawFbo=%u resolved to an empty render pass (attachmentCount=%u extent=%dx%d)",
|
||||
@@ -4669,7 +4540,7 @@ void main() {
|
||||
scissor.offset = {0, 0};
|
||||
scissor.extent = { (Uint)renderPassEntry->extent.x(), (Uint)renderPassEntry->extent.y() };
|
||||
}
|
||||
ShadowedSetScissor(frame.commandBuffer, scissor);
|
||||
vkCmdSetScissor(frame.commandBuffer, 0, 1, &scissor);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -5360,12 +5231,8 @@ void main() {
|
||||
if (!m_clearManager->GetPendingClears(&texture, pendingClears)) {
|
||||
return true;
|
||||
}
|
||||
// A pass may stay open on the FRAME command buffer while this clear is
|
||||
// recorded into the pre-pass stream (a different command buffer that
|
||||
// executes strictly before the frame's commands).
|
||||
MOBILEGL_ASSERT(VkRenderPassManager::GetActiveRenderPass() == nullptr ||
|
||||
commandBuffer != m_frameContext.GetCurrent().commandBuffer,
|
||||
"MaterializePendingClearForTexture requires no active render pass on the target buffer");
|
||||
MOBILEGL_ASSERT(VkRenderPassManager::GetActiveRenderPass() == nullptr,
|
||||
"MaterializePendingClearForTexture requires no active render pass");
|
||||
|
||||
auto* resource = m_textureManager->SyncTextureAndGetDescriptor(texture);
|
||||
MOBILEGL_ASSERT(resource != nullptr,
|
||||
@@ -5595,10 +5462,7 @@ void main() {
|
||||
"TryBlitToDefaultFramebufferWithShader: failed to create sampled view for textureId=%d mip=%u",
|
||||
sourceTexture->GetExternalIndex(), srcBinding.mipLevel);
|
||||
|
||||
// A color-only blit never touches depth/stencil: let the default-FBO pass
|
||||
// it opens skip the depth attachment (depth-less flavor).
|
||||
auto& renderPassEntry =
|
||||
m_renderPassManager->GetOrCreateRenderPass(drawFbo, m_imageIndexAcquired, /*drawUsesDepthStencil=*/false);
|
||||
auto& renderPassEntry = m_renderPassManager->GetOrCreateRenderPass(drawFbo, m_imageIndexAcquired);
|
||||
const Bool ok = VkRenderPassManager::BeginRenderPass(frame.commandBuffer, renderPassEntry);
|
||||
MOBILEGL_ASSERT(ok, "%s: BeginRenderPass failed", __func__);
|
||||
|
||||
@@ -5614,10 +5478,6 @@ void main() {
|
||||
const VkPipeline pipeline = GetOrCreateBlitPipeline(renderPassEntry);
|
||||
MOBILEGL_ASSERT(pipeline != VK_NULL_HANDLE, "TryBlitToDefaultFramebufferWithShader: blit pipeline is null");
|
||||
vkCmdBindPipeline(frame.commandBuffer, VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
|
||||
// The blit pipeline's narrower dynamic set (viewport/scissor only)
|
||||
// leaves the other dynamic states undefined; its raw viewport/scissor
|
||||
// writes also bypass the shadow.
|
||||
ResetDynamicStateShadow();
|
||||
|
||||
auto* blitProgramData = static_cast<Uint8*>(m_blitResources.program->MapUBO());
|
||||
MOBILEGL_ASSERT(blitProgramData != nullptr, "TryBlitToDefaultFramebufferWithShader: blit UBO is null");
|
||||
@@ -6414,11 +6274,7 @@ void main() {
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
m_lastPipelineValid = false; // command-buffer boundary: drop the pipeline memo
|
||||
}
|
||||
// The pre-pass stream must never be submitted later than the recording
|
||||
// it was paired with (frame commands recorded after a pre-pass move
|
||||
// rely on the moved work having executed first).
|
||||
m_frameContext.EndPreCommandRecordingIfOpen();
|
||||
if (!frame.hasCommandBufferRecorded && !frame.hasPreCommandBufferRecorded) {
|
||||
if (!frame.hasCommandBufferRecorded) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -7553,18 +7409,8 @@ void main() {
|
||||
submitInfo.pWaitSemaphores = &waitSemaphore;
|
||||
submitInfo.pWaitDstStageMask = &waitDstStageMask;
|
||||
}
|
||||
// The pre-pass stream, when recorded, executes strictly before the
|
||||
// frame's commands within the same submission.
|
||||
VkCommandBuffer commandBuffers[2] = {VK_NULL_HANDLE, VK_NULL_HANDLE};
|
||||
Uint32 commandBufferCount = 0;
|
||||
if (frame.hasPreCommandBufferRecorded) {
|
||||
commandBuffers[commandBufferCount++] = frame.preCommandBuffer;
|
||||
}
|
||||
if (frame.hasCommandBufferRecorded) {
|
||||
commandBuffers[commandBufferCount++] = frame.commandBuffer;
|
||||
}
|
||||
submitInfo.commandBufferCount = commandBufferCount;
|
||||
submitInfo.pCommandBuffers = commandBuffers;
|
||||
submitInfo.commandBufferCount = 1;
|
||||
submitInfo.pCommandBuffers = &frame.commandBuffer;
|
||||
const VkResult result = vkQueueSubmit(m_graphicsQueue, 1, &submitInfo, fence);
|
||||
if (result != VK_SUCCESS) {
|
||||
MGLOG_E("SubmitPendingCommandBuffer: vkQueueSubmit returned %d", result);
|
||||
@@ -7572,7 +7418,6 @@ void main() {
|
||||
}
|
||||
frame.imageAvailableSemaphoreConsumed = true;
|
||||
frame.hasCommandBufferRecorded = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
RegisterSubmit(fence, pooledFence);
|
||||
frame.lastSubmitIndex = m_submitCounter;
|
||||
return true;
|
||||
@@ -7603,8 +7448,6 @@ void main() {
|
||||
}
|
||||
m_frameContext.EndCommandRecording();
|
||||
}
|
||||
m_frameContext.EndPreCommandRecordingIfOpen();
|
||||
const Bool submittingPreCommandBuffer = frame.hasPreCommandBufferRecorded;
|
||||
if (!SubmitPendingCommandBuffer(frame, fence, /*pooledFence=*/true)) {
|
||||
// Submit failure (device loss regime): the ended command buffer
|
||||
// stays marked recorded so Present can still try to submit it.
|
||||
@@ -7623,7 +7466,7 @@ void main() {
|
||||
// The submitted command buffer may still be executing; recording must
|
||||
// restart on a fresh one. If none can be allocated, fall back to
|
||||
// draining this submission so reusing the buffer stays legal.
|
||||
const VkResult retireResult = m_frameContext.RetireCurrentCommandBuffer(submittingPreCommandBuffer);
|
||||
const VkResult retireResult = m_frameContext.RetireCurrentCommandBuffer();
|
||||
if (retireResult != VK_SUCCESS) {
|
||||
MGLOG_E("FlushPendingCommands: RetireCurrentCommandBuffer returned %d; draining submission", retireResult);
|
||||
if (vkWaitForFences(m_device, 1, &fence, VK_TRUE, UINT64_MAX) == VK_SUCCESS) {
|
||||
@@ -7689,13 +7532,6 @@ void main() {
|
||||
}
|
||||
|
||||
void VulkanRenderer::OnFrameCommandRecordingBegan(VkCommandBuffer commandBuffer) {
|
||||
// Dynamic state does not survive a command-buffer boundary.
|
||||
ResetDynamicStateShadow();
|
||||
// Pre-pass stream bookkeeping: a fresh frame recording references no
|
||||
// textures yet.
|
||||
if (m_textureManager) {
|
||||
m_textureManager->AdvanceRecordingGeneration();
|
||||
}
|
||||
if (m_timerQueryManager) {
|
||||
m_timerQueryManager->OnFrameCommandRecordingBegan(commandBuffer, m_frameContext.GetCurrentFrameIndex(),
|
||||
m_bufferManager.GetFrameSerial());
|
||||
@@ -7768,7 +7604,6 @@ void main() {
|
||||
if (suspendedFrame.isCommandRecording) {
|
||||
m_frameContext.EndCommandRecording();
|
||||
}
|
||||
m_frameContext.AbandonPreCommandRecording();
|
||||
suspendedFrame.isCommandRecording = false;
|
||||
suspendedFrame.hasCommandBufferRecorded = false;
|
||||
m_lastPipelineValid = false;
|
||||
@@ -7833,19 +7668,16 @@ void main() {
|
||||
frame.hasCommandBufferRecorded = true;
|
||||
m_lastPipelineValid = false; // command-buffer boundary: drop the pipeline memo
|
||||
}
|
||||
m_frameContext.EndPreCommandRecordingIfOpen();
|
||||
|
||||
const Bool shouldSubmitCommandBuffer = frame.hasCommandBufferRecorded;
|
||||
|
||||
// 1) Submit current frame work (the pre-pass stream, when recorded,
|
||||
// rides the same submission strictly ahead of the frame commands).
|
||||
// 1) Submit current frame work.
|
||||
auto submitPacket = m_frameContext.GetSubmitInfo(shouldSubmitCommandBuffer, m_imageIndexAcquired);
|
||||
VK_VERIFY(vkQueueSubmit(m_graphicsQueue, 1, &submitPacket.submitInfo, frame.imageInFlightFence));
|
||||
RegisterSubmit(frame.imageInFlightFence, /*pooledFence=*/false);
|
||||
frame.lastSubmitIndex = m_submitCounter;
|
||||
frame.isCommandRecording = false;
|
||||
frame.hasCommandBufferRecorded = false;
|
||||
frame.hasPreCommandBufferRecorded = false;
|
||||
m_swapchainObject.SetImageLayout(m_imageIndexAcquired, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR);
|
||||
|
||||
// 2) Present current frame.
|
||||
@@ -7873,13 +7705,6 @@ void main() {
|
||||
result = VK_SUCCESS;
|
||||
}
|
||||
VK_VERIFY(result, "Present, vkQueuePresentKHR");
|
||||
// EGL swap semantics: the presented color buffer's content is undefined the
|
||||
// next time this image is acquired (EGL_BUFFER_DESTROYED, the default swap
|
||||
// behaviour), and EVERY ancillary depth/stencil buffer's content is
|
||||
// undefined after any swap. The render-pass manager turns the undefined
|
||||
// attachments' next tile loads into LOAD_OP_DONT_CARE.
|
||||
m_swapchainObject.SetImageContentDefined(m_imageIndexAcquired, false);
|
||||
m_swapchainObject.SetAllDepthStencilContentUndefined();
|
||||
// The authoritative check, done here - after the frame is presented, before the next
|
||||
// acquire. This is what makes a launcher-side resolution change take effect: shrinking
|
||||
// the window's buffer (SurfaceHolder.setFixedSize) moves currentExtent, the swapchain
|
||||
@@ -8839,10 +8664,6 @@ void main() {
|
||||
if (m_frameContext.GetFrameCount() > 0) {
|
||||
m_frameContext.GetCurrent().isCommandRecording = false;
|
||||
m_frameContext.GetCurrent().hasCommandBufferRecorded = false;
|
||||
// The pre-pass stream paired with the abandoned recording is
|
||||
// dropped with it (its next Begin resets the buffer).
|
||||
m_frameContext.GetCurrent().isPreCommandRecording = false;
|
||||
m_frameContext.GetCurrent().hasPreCommandBufferRecorded = false;
|
||||
}
|
||||
const Bool okArena = m_bufferManager.RecreateTransientArenas(m_frameContext.GetFrameCount());
|
||||
MOBILEGL_ASSERT(okArena, "RecreateSwapchain: buffer manager transient arena initialization failed");
|
||||
|
||||
Reference in New Issue
Block a user