diff --git a/src/shader_recompiler/CMakeLists.txt b/src/shader_recompiler/CMakeLists.txt index 957ed5fb55..cc9ce95d92 100644 --- a/src/shader_recompiler/CMakeLists.txt +++ b/src/shader_recompiler/CMakeLists.txt @@ -172,6 +172,7 @@ add_library(shader_recompiler STATIC ir_opt/constant_propagation_pass.cpp ir_opt/dead_code_elimination_pass.cpp ir_opt/dual_vertex_pass.cpp + ir_opt/geometry_compaction_pass.cpp ir_opt/global_memory_to_storage_buffer_pass.cpp ir_opt/identity_removal_pass.cpp ir_opt/layer_pass.cpp diff --git a/src/shader_recompiler/frontend/maxwell/translate_program.cpp b/src/shader_recompiler/frontend/maxwell/translate_program.cpp index ebc5a825dd..e7d7e5a6b3 100644 --- a/src/shader_recompiler/frontend/maxwell/translate_program.cpp +++ b/src/shader_recompiler/frontend/maxwell/translate_program.cpp @@ -299,6 +299,7 @@ IR::Program TranslateProgram(ObjectPool& inst_pool, ObjectPool + +#include "shader_recompiler/frontend/ir/ir_emitter.h" +#include "shader_recompiler/host_translate_info.h" +#include "shader_recompiler/ir_opt/passes.h" + +namespace Shader::Optimization { +namespace { +constexpr int MAX_BALLOT_DEPTH = 8; + +struct SlotAtomic { + IR::Block* block; + IR::Inst* atomic; + IR::Inst* ballot; +}; + +IR::Inst* FindBallot(const IR::Value& value, int depth) { + if (depth > MAX_BALLOT_DEPTH || value.IsImmediate()) { + return nullptr; + } + IR::Inst* const inst{value.InstRecursive()}; + if (inst->GetOpcode() == IR::Opcode::SubgroupBallot) { + return inst; + } + if (inst->GetOpcode() == IR::Opcode::Phi) { + return nullptr; + } + for (size_t index = 0; index < inst->NumArgs(); ++index) { + if (IR::Inst* const ballot{FindBallot(inst->Arg(index), depth + 1)}) { + return ballot; + } + } + return nullptr; +} + +bool IsShuffle(IR::Opcode opcode) { + switch (opcode) { + case IR::Opcode::ShuffleIndex: + case IR::Opcode::ShuffleUp: + case IR::Opcode::ShuffleDown: + case IR::Opcode::ShuffleButterfly: + return true; + default: + return false; + } +} + +IR::U32 PrimitiveSlot(IR::IREmitter& ir, u32 invocations, const IR::U32& amount) { + IR::U32 key{ir.GetAttributeU32(IR::Attribute::PrimitiveId)}; + if (invocations > 1) { + key = ir.IAdd(ir.IMul(key, ir.Imm32(invocations)), ir.InvocationId()); + } + return ir.IMul(key, amount); +} + +bool MergesAtomic(const IR::Inst& phi, const IR::Block& block, const IR::Inst& atomic) { + for (size_t index = 0; index < phi.NumArgs(); ++index) { + const IR::Value arg{phi.Arg(index)}; + if (phi.PhiBlock(index) == &block && !arg.IsImmediate() && + arg.InstRecursive() == &atomic) { + return true; + } + } + return false; +} + +void RewriteMergedSlots(IR::Block& block, IR::Inst& atomic, u32 invocations) { + for (IR::Block* const successor : block.ImmSuccessors()) { + for (IR::Inst& phi : successor->Instructions()) { + if (phi.GetOpcode() != IR::Opcode::Phi || !MergesAtomic(phi, block, atomic)) { + continue; + } + for (size_t index = 0; index < phi.NumArgs(); ++index) { + const IR::Value arg{phi.Arg(index)}; + if (arg.IsImmediate() || !IsShuffle(arg.InstRecursive()->GetOpcode())) { + continue; + } + IR::IREmitter ir{*phi.PhiBlock(index)}; + const IR::U32 amount{arg.InstRecursive()->Arg(0)}; + phi.SetArg(index, PrimitiveSlot(ir, invocations, amount)); + } + } + } +} + +void RewriteAtomic(IR::Block& block, IR::Inst& atomic, u32 invocations) { + const auto insert_point{IR::Block::InstructionList::s_iterator_to(atomic)}; + IR::IREmitter ir{block, insert_point}; + const IR::U32 amount{atomic.Arg(2)}; + const IR::U32 slot{PrimitiveSlot(ir, invocations, amount)}; + block.PrependNewInst(insert_point, IR::Opcode::StorageAtomicUMax32, + {atomic.Arg(0), atomic.Arg(1), ir.IAdd(slot, amount)}); + atomic.ReplaceUsesWith(slot); +} + +void NeutralizePredicate(IR::Inst& ballot) { + const IR::Value pred{ballot.Arg(0)}; + if (!pred.IsImmediate()) { + pred.InstRecursive()->ReplaceUsesWith(IR::Value{true}); + } +} +} + +void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info) { + if (program.stage != Stage::Geometry || !host_info.single_lane_geometry_subgroups) { + return; + } + boost::container::small_vector slot_atomics; + for (IR::Block* const block : program.post_order_blocks) { + for (IR::Inst& inst : block->Instructions()) { + if (inst.GetOpcode() != IR::Opcode::StorageAtomicIAdd32) { + continue; + } + if (IR::Inst* const ballot{FindBallot(inst.Arg(2), 0)}) { + slot_atomics.push_back({block, &inst, ballot}); + } + } + } + for (const SlotAtomic& slot_atomic : slot_atomics) { + RewriteMergedSlots(*slot_atomic.block, *slot_atomic.atomic, program.invocations); + RewriteAtomic(*slot_atomic.block, *slot_atomic.atomic, program.invocations); + NeutralizePredicate(*slot_atomic.ballot); + } +} + +} diff --git a/src/shader_recompiler/ir_opt/passes.h b/src/shader_recompiler/ir_opt/passes.h index 1e637cb23c..0c2b44e022 100644 --- a/src/shader_recompiler/ir_opt/passes.h +++ b/src/shader_recompiler/ir_opt/passes.h @@ -16,6 +16,7 @@ void CollectShaderInfoPass(Environment& env, IR::Program& program); void ConditionalBarrierPass(IR::Program& program); void ConstantPropagationPass(Environment& env, IR::Program& program); void DeadCodeEliminationPass(IR::Program& program); +void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info); void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateInfo& host_info); void IdentityRemovalPass(IR::Program& program); void LowerFp64ToFp32(IR::Program& program); diff --git a/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp b/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp index 7c530e1cfb..d04d36046e 100644 --- a/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp +++ b/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp @@ -488,6 +488,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_, .support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(), .support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(), .support_conditional_barrier = device.SupportsConditionalBarriers(), + .single_lane_geometry_subgroups = !profile.SupportsSubgroupStage(Shader::Stage::Geometry), }; host_info.ApplyDescriptorLimitPolicy();