diff --git a/src/shader_recompiler/frontend/maxwell/translate_program.cpp b/src/shader_recompiler/frontend/maxwell/translate_program.cpp index e7d7e5a6b3..a7b0ad4cba 100644 --- a/src/shader_recompiler/frontend/maxwell/translate_program.cpp +++ b/src/shader_recompiler/frontend/maxwell/translate_program.cpp @@ -5,7 +5,10 @@ // SPDX-License-Identifier: GPL-2.0-or-later #include +#include +#include #include +#include #include #include @@ -172,7 +175,17 @@ std::map GenerateLegacyToGenericMappings( void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program, const Shader::VaryingState& passthrough_mask, bool passthrough_position, - std::optional passthrough_layer_attr) { + std::optional passthrough_layer_attr, + std::optional viewport_mask_reg) { + constexpr std::array CULLED_POSITION{2.0f, 2.0f, 2.0f, 1.0f}; + IR::U1 culled{ir.Imm1(false)}; + IR::U32 viewport{ir.Imm32(0)}; + if (viewport_mask_reg) { + const IR::U32 mask{ir.GetReg(*viewport_mask_reg)}; + const IR::U32 lowest_bit{ir.BitwiseAnd(mask, IR::U32{ir.INeg(mask)})}; + culled = ir.IEqual(mask, ir.Imm32(0)); + viewport = IR::U32{ir.Select(culled, ir.Imm32(0), ir.FindUMsb(lowest_bit))}; + } for (u32 i = 0; i < program.output_vertices; i++) { // Assign generics from input for (u32 j = 0; j < 32; j++) { @@ -190,10 +203,19 @@ void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program, if (passthrough_position) { // Assign position from input const IR::Attribute attr = IR::Attribute::PositionX; - ir.SetAttribute(attr + 0, ir.GetAttribute(attr + 0, ir.Imm32(i)), ir.Imm32(0)); - ir.SetAttribute(attr + 1, ir.GetAttribute(attr + 1, ir.Imm32(i)), ir.Imm32(0)); - ir.SetAttribute(attr + 2, ir.GetAttribute(attr + 2, ir.Imm32(i)), ir.Imm32(0)); - ir.SetAttribute(attr + 3, ir.GetAttribute(attr + 3, ir.Imm32(i)), ir.Imm32(0)); + for (u32 component = 0; component < 4; ++component) { + IR::F32 value{ir.GetAttribute(attr + component, ir.Imm32(i))}; + if (viewport_mask_reg) { + value = IR::F32{ + ir.Select(culled, ir.Imm32(CULLED_POSITION[component]), value)}; + } + ir.SetAttribute(attr + component, value, ir.Imm32(0)); + } + } + + if (viewport_mask_reg) { + ir.SetAttribute(IR::Attribute::ViewportIndex, ir.BitCast(viewport), + ir.Imm32(0)); } if (passthrough_layer_attr) { @@ -219,19 +241,91 @@ u32 GetOutputTopologyVertices(OutputTopology output_topology) { } } +std::optional FindFreeRegister(const IR::Program& program) { + std::bitset used; + for (IR::Block* const block : program.blocks) { + for (const IR::Inst& inst : block->Instructions()) { + const IR::Opcode opcode{inst.GetOpcode()}; + if (opcode == IR::Opcode::GetRegister || opcode == IR::Opcode::SetRegister) { + used.set(IR::RegIndex(inst.Arg(0).Reg())); + } + } + } + for (size_t index = IR::NUM_USER_REGS; index-- > 0;) { + if (!used.test(index)) { + return static_cast(index); + } + } + return std::nullopt; +} + +std::optional LowerViewportMask(const IR::Program& program) { + const std::optional reg{FindFreeRegister(program)}; + if (!reg || program.blocks.empty()) { + return std::nullopt; + } + bool stores_mask{}; + for (IR::Block* const block : program.blocks) { + for (IR::Inst& inst : block->Instructions()) { + if (inst.GetOpcode() != IR::Opcode::SetAttribute || + inst.Arg(0).Attribute() != IR::Attribute::ViewportMask) { + continue; + } + IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)}; + ir.SetReg(*reg, ir.BitCast(IR::F32{inst.Arg(1)})); + inst.Invalidate(); + stores_mask = true; + } + } + if (!stores_mask) { + return std::nullopt; + } + IR::Block& entry{*program.blocks.front()}; + IR::IREmitter ir{entry, entry.begin()}; + ir.SetReg(*reg, ir.Imm32(1)); + return reg; +} + void LowerGeometryPassthrough(const IR::Program& program, const HostTranslateInfo& host_info) { + std::optional viewport_mask_reg; + if (!host_info.support_viewport_mask) { + viewport_mask_reg = LowerViewportMask(program); + } for (IR::Block* const block : program.blocks) { for (IR::Inst& inst : block->Instructions()) { if (inst.GetOpcode() == IR::Opcode::Epilogue) { IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)}; EmitGeometryPassthrough( ir, program, program.info.passthrough, - program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {}); + program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {}, + viewport_mask_reg); } } } } +void TightenOutputVertices(IR::Program& program) { + if (program.stage != Stage::Geometry || program.is_geometry_passthrough) { + return; + } + const bool has_loops = std::ranges::any_of(program.syntax_list, [](const auto& node) { + return node.type == IR::AbstractSyntaxNode::Type::Loop; + }); + if (has_loops) { + return; + } + u32 num_emits = 0; + for (IR::Block* const block : program.blocks) { + for (const IR::Inst& inst : block->Instructions()) { + if (inst.GetOpcode() == IR::Opcode::EmitVertex) { + ++num_emits; + } + } + } + if (num_emits != 0) { + program.output_vertices = (std::min)(program.output_vertices, num_emits); + } +} } // Anonymous namespace IR::Program TranslateProgram(ObjectPool& inst_pool, ObjectPool& block_pool, @@ -306,6 +400,7 @@ IR::Program TranslateProgram(ObjectPool& inst_pool, ObjectPool& inst_pool, IR::IREmitter ir{*current_block}; EmitGeometryPassthrough(ir, program, program.info.stores, true, - source_program.info.emulated_layer); + source_program.info.emulated_layer, std::nullopt); IR::Block* return_block{block_pool.Create(inst_pool)}; IR::IREmitter{*return_block}.Epilogue(); diff --git a/src/shader_recompiler/host_translate_info.h b/src/shader_recompiler/host_translate_info.h index 0128ee07eb..44df85ccb2 100644 --- a/src/shader_recompiler/host_translate_info.h +++ b/src/shader_recompiler/host_translate_info.h @@ -34,6 +34,7 @@ struct HostTranslateInfo { bool needs_demote_reorder{}; ///< True when the device needs DemoteToHelperInvocation reordered bool support_snorm_render_buffer{}; ///< True when the device supports SNORM render buffers bool support_viewport_index_layer{}; ///< True when the device supports gl_Layer in VS + bool support_viewport_mask{}; bool support_geometry_shader_passthrough{}; ///< True when the device supports geometry ///< passthrough shaders bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional diff --git a/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp b/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp index 14428c20c2..4f42474a05 100644 --- a/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp +++ b/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp @@ -485,6 +485,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_, driver_id == VK_DRIVER_ID_SAMSUNG_PROPRIETARY, .support_snorm_render_buffer = true, .support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(), + .support_viewport_mask = device.IsNvViewportArray2Supported(), .support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(), .support_conditional_barrier = device.SupportsConditionalBarriers(), .single_lane_geometry_subgroups = !profile.SupportsSubgroupStage(Shader::Stage::Geometry),