Re-do geometry compaction pass

This commit is contained in:
CamilleLaVey
2026-09-26 00:42:58 -04:00
committed by PavelBARABANOV
parent 350cc1c9a7
commit 03268427db
6 changed files with 134 additions and 0 deletions
+1
View File
@@ -172,6 +172,7 @@ add_library(shader_recompiler STATIC
ir_opt/constant_propagation_pass.cpp ir_opt/constant_propagation_pass.cpp
ir_opt/dead_code_elimination_pass.cpp ir_opt/dead_code_elimination_pass.cpp
ir_opt/dual_vertex_pass.cpp ir_opt/dual_vertex_pass.cpp
ir_opt/geometry_compaction_pass.cpp
ir_opt/global_memory_to_storage_buffer_pass.cpp ir_opt/global_memory_to_storage_buffer_pass.cpp
ir_opt/identity_removal_pass.cpp ir_opt/identity_removal_pass.cpp
ir_opt/layer_pass.cpp ir_opt/layer_pass.cpp
@@ -393,6 +393,7 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
Optimization::PositionPass(env, program); Optimization::PositionPass(env, program);
Optimization::GlobalMemoryToStorageBufferPass(program, normalized_host_info); Optimization::GlobalMemoryToStorageBufferPass(program, normalized_host_info);
Optimization::GeometryCompactionPass(program, normalized_host_info);
Optimization::TexturePass(env, program, normalized_host_info); Optimization::TexturePass(env, program, normalized_host_info);
if (Settings::values.resolution_info.active || Settings::values.rescale_hack.GetValue()) { if (Settings::values.resolution_info.active || Settings::values.rescale_hack.GetValue()) {
@@ -39,6 +39,7 @@ struct HostTranslateInfo {
///< passthrough shaders ///< passthrough shaders
bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional
///< control flow ///< control flow
bool single_lane_geometry_subgroups{};
void ApplyDescriptorLimitPolicy() noexcept { void ApplyDescriptorLimitPolicy() noexcept {
if (min_ssbo_alignment == 0) { if (min_ssbo_alignment == 0) {
@@ -0,0 +1,129 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#include <boost/container/small_vector.hpp>
#include "shader_recompiler/frontend/ir/ir_emitter.h"
#include "shader_recompiler/host_translate_info.h"
#include "shader_recompiler/ir_opt/passes.h"
namespace Shader::Optimization {
namespace {
constexpr int MAX_BALLOT_DEPTH = 8;
struct SlotAtomic {
IR::Block* block;
IR::Inst* atomic;
IR::Inst* ballot;
};
IR::Inst* FindBallot(const IR::Value& value, int depth) {
if (depth > MAX_BALLOT_DEPTH || value.IsImmediate()) {
return nullptr;
}
IR::Inst* const inst{value.InstRecursive()};
if (inst->GetOpcode() == IR::Opcode::SubgroupBallot) {
return inst;
}
if (inst->GetOpcode() == IR::Opcode::Phi) {
return nullptr;
}
for (size_t index = 0; index < inst->NumArgs(); ++index) {
if (IR::Inst* const ballot{FindBallot(inst->Arg(index), depth + 1)}) {
return ballot;
}
}
return nullptr;
}
bool IsShuffle(IR::Opcode opcode) {
switch (opcode) {
case IR::Opcode::ShuffleIndex:
case IR::Opcode::ShuffleUp:
case IR::Opcode::ShuffleDown:
case IR::Opcode::ShuffleButterfly:
return true;
default:
return false;
}
}
IR::U32 PrimitiveSlot(IR::IREmitter& ir, u32 invocations, const IR::U32& amount) {
IR::U32 key{ir.GetAttributeU32(IR::Attribute::PrimitiveId)};
if (invocations > 1) {
key = ir.IAdd(ir.IMul(key, ir.Imm32(invocations)), ir.InvocationId());
}
return ir.IMul(key, amount);
}
bool MergesAtomic(const IR::Inst& phi, const IR::Block& block, const IR::Inst& atomic) {
for (size_t index = 0; index < phi.NumArgs(); ++index) {
const IR::Value arg{phi.Arg(index)};
if (phi.PhiBlock(index) == &block && !arg.IsImmediate() &&
arg.InstRecursive() == &atomic) {
return true;
}
}
return false;
}
void RewriteMergedSlots(IR::Block& block, IR::Inst& atomic, u32 invocations) {
for (IR::Block* const successor : block.ImmSuccessors()) {
for (IR::Inst& phi : successor->Instructions()) {
if (phi.GetOpcode() != IR::Opcode::Phi || !MergesAtomic(phi, block, atomic)) {
continue;
}
for (size_t index = 0; index < phi.NumArgs(); ++index) {
const IR::Value arg{phi.Arg(index)};
if (arg.IsImmediate() || !IsShuffle(arg.InstRecursive()->GetOpcode())) {
continue;
}
IR::IREmitter ir{*phi.PhiBlock(index)};
const IR::U32 amount{arg.InstRecursive()->Arg(0)};
phi.SetArg(index, PrimitiveSlot(ir, invocations, amount));
}
}
}
}
void RewriteAtomic(IR::Block& block, IR::Inst& atomic, u32 invocations) {
const auto insert_point{IR::Block::InstructionList::s_iterator_to(atomic)};
IR::IREmitter ir{block, insert_point};
const IR::U32 amount{atomic.Arg(2)};
const IR::U32 slot{PrimitiveSlot(ir, invocations, amount)};
block.PrependNewInst(insert_point, IR::Opcode::StorageAtomicUMax32,
{atomic.Arg(0), atomic.Arg(1), ir.IAdd(slot, amount)});
atomic.ReplaceUsesWith(slot);
}
void NeutralizePredicate(IR::Inst& ballot) {
const IR::Value pred{ballot.Arg(0)};
if (!pred.IsImmediate()) {
pred.InstRecursive()->ReplaceUsesWith(IR::Value{true});
}
}
}
void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info) {
if (program.stage != Stage::Geometry || !host_info.single_lane_geometry_subgroups) {
return;
}
boost::container::small_vector<SlotAtomic, 4> slot_atomics;
for (IR::Block* const block : program.post_order_blocks) {
for (IR::Inst& inst : block->Instructions()) {
if (inst.GetOpcode() != IR::Opcode::StorageAtomicIAdd32) {
continue;
}
if (IR::Inst* const ballot{FindBallot(inst.Arg(2), 0)}) {
slot_atomics.push_back({block, &inst, ballot});
}
}
}
for (const SlotAtomic& slot_atomic : slot_atomics) {
RewriteMergedSlots(*slot_atomic.block, *slot_atomic.atomic, program.invocations);
RewriteAtomic(*slot_atomic.block, *slot_atomic.atomic, program.invocations);
NeutralizePredicate(*slot_atomic.ballot);
}
}
}
+1
View File
@@ -16,6 +16,7 @@ void CollectShaderInfoPass(Environment& env, IR::Program& program);
void ConditionalBarrierPass(IR::Program& program); void ConditionalBarrierPass(IR::Program& program);
void ConstantPropagationPass(Environment& env, IR::Program& program); void ConstantPropagationPass(Environment& env, IR::Program& program);
void DeadCodeEliminationPass(IR::Program& program); void DeadCodeEliminationPass(IR::Program& program);
void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info);
void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateInfo& host_info); void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateInfo& host_info);
void IdentityRemovalPass(IR::Program& program); void IdentityRemovalPass(IR::Program& program);
void LowerFp64ToFp32(IR::Program& program); void LowerFp64ToFp32(IR::Program& program);
@@ -489,6 +489,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
.support_viewport_mask = device.IsNvViewportArray2Supported(), .support_viewport_mask = device.IsNvViewportArray2Supported(),
.support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(), .support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(),
.support_conditional_barrier = device.SupportsConditionalBarriers(), .support_conditional_barrier = device.SupportsConditionalBarriers(),
.single_lane_geometry_subgroups = !profile.SupportsSubgroupStage(Shader::Stage::Geometry),
}; };
host_info.ApplyDescriptorLimitPolicy(); host_info.ApplyDescriptorLimitPolicy();