mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-06 14:09:58 +00:00
More work on GS
This commit is contained in:
@@ -5,7 +5,10 @@
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
#include <queue>
|
||||
|
||||
@@ -172,7 +175,17 @@ std::map<IR::Attribute, IR::Attribute> GenerateLegacyToGenericMappings(
|
||||
void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program,
|
||||
const Shader::VaryingState& passthrough_mask,
|
||||
bool passthrough_position,
|
||||
std::optional<IR::Attribute> passthrough_layer_attr) {
|
||||
std::optional<IR::Attribute> passthrough_layer_attr,
|
||||
std::optional<IR::Reg> viewport_mask_reg) {
|
||||
constexpr std::array CULLED_POSITION{2.0f, 2.0f, 2.0f, 1.0f};
|
||||
IR::U1 culled{ir.Imm1(false)};
|
||||
IR::U32 viewport{ir.Imm32(0)};
|
||||
if (viewport_mask_reg) {
|
||||
const IR::U32 mask{ir.GetReg(*viewport_mask_reg)};
|
||||
const IR::U32 lowest_bit{ir.BitwiseAnd(mask, IR::U32{ir.INeg(mask)})};
|
||||
culled = ir.IEqual(mask, ir.Imm32(0));
|
||||
viewport = IR::U32{ir.Select(culled, ir.Imm32(0), ir.FindUMsb(lowest_bit))};
|
||||
}
|
||||
for (u32 i = 0; i < program.output_vertices; i++) {
|
||||
// Assign generics from input
|
||||
for (u32 j = 0; j < 32; j++) {
|
||||
@@ -190,10 +203,19 @@ void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program,
|
||||
if (passthrough_position) {
|
||||
// Assign position from input
|
||||
const IR::Attribute attr = IR::Attribute::PositionX;
|
||||
ir.SetAttribute(attr + 0, ir.GetAttribute(attr + 0, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 1, ir.GetAttribute(attr + 1, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 2, ir.GetAttribute(attr + 2, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 3, ir.GetAttribute(attr + 3, ir.Imm32(i)), ir.Imm32(0));
|
||||
for (u32 component = 0; component < 4; ++component) {
|
||||
IR::F32 value{ir.GetAttribute(attr + component, ir.Imm32(i))};
|
||||
if (viewport_mask_reg) {
|
||||
value = IR::F32{
|
||||
ir.Select(culled, ir.Imm32(CULLED_POSITION[component]), value)};
|
||||
}
|
||||
ir.SetAttribute(attr + component, value, ir.Imm32(0));
|
||||
}
|
||||
}
|
||||
|
||||
if (viewport_mask_reg) {
|
||||
ir.SetAttribute(IR::Attribute::ViewportIndex, ir.BitCast<IR::F32>(viewport),
|
||||
ir.Imm32(0));
|
||||
}
|
||||
|
||||
if (passthrough_layer_attr) {
|
||||
@@ -219,19 +241,91 @@ u32 GetOutputTopologyVertices(OutputTopology output_topology) {
|
||||
}
|
||||
}
|
||||
|
||||
std::optional<IR::Reg> FindFreeRegister(const IR::Program& program) {
|
||||
std::bitset<IR::NUM_USER_REGS> used;
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
const IR::Opcode opcode{inst.GetOpcode()};
|
||||
if (opcode == IR::Opcode::GetRegister || opcode == IR::Opcode::SetRegister) {
|
||||
used.set(IR::RegIndex(inst.Arg(0).Reg()));
|
||||
}
|
||||
}
|
||||
}
|
||||
for (size_t index = IR::NUM_USER_REGS; index-- > 0;) {
|
||||
if (!used.test(index)) {
|
||||
return static_cast<IR::Reg>(index);
|
||||
}
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
std::optional<IR::Reg> LowerViewportMask(const IR::Program& program) {
|
||||
const std::optional<IR::Reg> reg{FindFreeRegister(program)};
|
||||
if (!reg || program.blocks.empty()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
bool stores_mask{};
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() != IR::Opcode::SetAttribute ||
|
||||
inst.Arg(0).Attribute() != IR::Attribute::ViewportMask) {
|
||||
continue;
|
||||
}
|
||||
IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
ir.SetReg(*reg, ir.BitCast<IR::U32>(IR::F32{inst.Arg(1)}));
|
||||
inst.Invalidate();
|
||||
stores_mask = true;
|
||||
}
|
||||
}
|
||||
if (!stores_mask) {
|
||||
return std::nullopt;
|
||||
}
|
||||
IR::Block& entry{*program.blocks.front()};
|
||||
IR::IREmitter ir{entry, entry.begin()};
|
||||
ir.SetReg(*reg, ir.Imm32(1));
|
||||
return reg;
|
||||
}
|
||||
|
||||
void LowerGeometryPassthrough(const IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
std::optional<IR::Reg> viewport_mask_reg;
|
||||
if (!host_info.support_viewport_mask) {
|
||||
viewport_mask_reg = LowerViewportMask(program);
|
||||
}
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() == IR::Opcode::Epilogue) {
|
||||
IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
EmitGeometryPassthrough(
|
||||
ir, program, program.info.passthrough,
|
||||
program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {});
|
||||
program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {},
|
||||
viewport_mask_reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void TightenOutputVertices(IR::Program& program) {
|
||||
if (program.stage != Stage::Geometry || program.is_geometry_passthrough) {
|
||||
return;
|
||||
}
|
||||
const bool has_loops = std::ranges::any_of(program.syntax_list, [](const auto& node) {
|
||||
return node.type == IR::AbstractSyntaxNode::Type::Loop;
|
||||
});
|
||||
if (has_loops) {
|
||||
return;
|
||||
}
|
||||
u32 num_emits = 0;
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() == IR::Opcode::EmitVertex) {
|
||||
++num_emits;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (num_emits != 0) {
|
||||
program.output_vertices = (std::min)(program.output_vertices, num_emits);
|
||||
}
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Block>& block_pool,
|
||||
@@ -306,6 +400,7 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
Optimization::RescalingPass(program);
|
||||
}
|
||||
Optimization::DeadCodeEliminationPass(program);
|
||||
TightenOutputVertices(program);
|
||||
if (Settings::values.renderer_debug) {
|
||||
Optimization::VerificationPass(program);
|
||||
}
|
||||
@@ -434,7 +529,7 @@ IR::Program GenerateGeometryPassthrough(ObjectPool<IR::Inst>& inst_pool,
|
||||
|
||||
IR::IREmitter ir{*current_block};
|
||||
EmitGeometryPassthrough(ir, program, program.info.stores, true,
|
||||
source_program.info.emulated_layer);
|
||||
source_program.info.emulated_layer, std::nullopt);
|
||||
|
||||
IR::Block* return_block{block_pool.Create(inst_pool)};
|
||||
IR::IREmitter{*return_block}.Epilogue();
|
||||
|
||||
@@ -34,6 +34,7 @@ struct HostTranslateInfo {
|
||||
bool needs_demote_reorder{}; ///< True when the device needs DemoteToHelperInvocation reordered
|
||||
bool support_snorm_render_buffer{}; ///< True when the device supports SNORM render buffers
|
||||
bool support_viewport_index_layer{}; ///< True when the device supports gl_Layer in VS
|
||||
bool support_viewport_mask{};
|
||||
bool support_geometry_shader_passthrough{}; ///< True when the device supports geometry
|
||||
///< passthrough shaders
|
||||
bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional
|
||||
|
||||
@@ -485,6 +485,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
driver_id == VK_DRIVER_ID_SAMSUNG_PROPRIETARY,
|
||||
.support_snorm_render_buffer = true,
|
||||
.support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(),
|
||||
.support_viewport_mask = device.IsNvViewportArray2Supported(),
|
||||
.support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(),
|
||||
.support_conditional_barrier = device.SupportsConditionalBarriers(),
|
||||
.single_lane_geometry_subgroups = !profile.SupportsSubgroupStage(Shader::Stage::Geometry),
|
||||
|
||||
Reference in New Issue
Block a user