mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-09-05 20:04:13 +00:00
Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4a93006764 |
Vendored
-1
@@ -11,7 +11,6 @@
|
||||
#include <limits>
|
||||
#include <span>
|
||||
#include <array>
|
||||
#include <algorithm>
|
||||
#include <time.h>
|
||||
|
||||
namespace Tz {
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <string>
|
||||
#include <cstdlib>
|
||||
#include <string_view>
|
||||
#ifdef _WIN32
|
||||
#include <llvm/Demangle/Demangle.h>
|
||||
|
||||
@@ -117,7 +117,6 @@ template <typename T>
|
||||
static void GetFuncAddress(Common::DynamicLibrary& dll, const char* name, T& pfn) {
|
||||
if (!dll.GetSymbol(name, &pfn)) {
|
||||
LOG_CRITICAL(HW_Memory, "Failed to load {}", name);
|
||||
throw std::bad_alloc{};
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -492,9 +492,6 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
|
||||
if (ctx.uses_nonuniform_storage_texel_buffer) {
|
||||
ctx.AddCapability(spv::Capability::StorageTexelBufferArrayNonUniformIndexing);
|
||||
}
|
||||
if (ctx.uses_nonuniform_storage_buffer) {
|
||||
ctx.AddCapability(spv::Capability::StorageBufferArrayNonUniformIndexing);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <bit>
|
||||
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv.h"
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv_instructions.h"
|
||||
#include "shader_recompiler/backend/spirv/spirv_emit_context.h"
|
||||
@@ -21,13 +23,29 @@ Id SharedPointer(EmitContext& ctx, Id offset, u32 index_offset = 0) {
|
||||
: ctx.OpAccessChain(ctx.shared_u32, ctx.shared_memory_u32, index);
|
||||
}
|
||||
|
||||
Id StorageIndex(EmitContext& ctx, const IR::Value& offset, size_t element_size) {
|
||||
if (offset.IsImmediate()) {
|
||||
const u32 imm_offset{static_cast<u32>(offset.U32() / element_size)};
|
||||
return ctx.Const(imm_offset);
|
||||
}
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(element_size))};
|
||||
const Id index{ctx.Def(offset)};
|
||||
if (shift == 0) {
|
||||
return index;
|
||||
}
|
||||
const Id shift_id{ctx.Const(shift)};
|
||||
return ctx.OpShiftRightLogical(ctx.U32[1], index, shift_id);
|
||||
}
|
||||
|
||||
Id StoragePointer(EmitContext& ctx, const StorageTypeDefinition& type_def,
|
||||
Id StorageDefinitions::*member_ptr, const IR::Value& binding,
|
||||
const IR::Value& offset, size_t element_size) {
|
||||
if (!binding.IsImmediate()) {
|
||||
throw NotImplementedException("Dynamic storage buffer indexing");
|
||||
}
|
||||
return ctx.StoragePointer(binding.U32(), ctx.Def(offset), type_def, static_cast<u32>(element_size), member_ptr);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].*member_ptr};
|
||||
const Id index{StorageIndex(ctx, offset, element_size)};
|
||||
return ctx.OpAccessChain(type_def.element, ssbo, ctx.u32_zero_value, index);
|
||||
}
|
||||
|
||||
std::pair<Id, Id> AtomicArgs(EmitContext& ctx) {
|
||||
@@ -198,14 +216,16 @@ Id EmitStorageAtomicUMax32(EmitContext& ctx, const IR::Value& binding, const IR:
|
||||
|
||||
Id EmitStorageAtomicInc32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.increment_cas_ssbo, pointer, value);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.increment_cas_ssbo, base_index, value, ssbo);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicDec32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.decrement_cas_ssbo, pointer, value);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.decrement_cas_ssbo, base_index, value, ssbo);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicAnd32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
@@ -344,49 +364,56 @@ Id EmitStorageAtomicExchange32x2(EmitContext& ctx, const IR::Value& binding,
|
||||
|
||||
Id EmitStorageAtomicAddF32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.F32[1], ctx.f32_add_cas, pointer, value);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.F32[1], ctx.f32_add_cas, base_index, value, ssbo);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicAddF16x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_add_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_add_cas, base_index, value, ssbo)};
|
||||
return ctx.OpBitcast(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicAddF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_add_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_add_cas, base_index, value, ssbo)};
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMinF16x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_min_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_min_cas, base_index, value, ssbo)};
|
||||
return ctx.OpBitcast(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMinF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_min_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_min_cas, base_index, value, ssbo)};
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMaxF16x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_max_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_max_cas, base_index, value, ssbo)};
|
||||
return ctx.OpBitcast(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMaxF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_max_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_max_cas, base_index, value, ssbo)};
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
|
||||
@@ -4,33 +4,46 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <bit>
|
||||
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv.h"
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv_instructions.h"
|
||||
#include "shader_recompiler/backend/spirv/spirv_emit_context.h"
|
||||
|
||||
namespace Shader::Backend::SPIRV {
|
||||
namespace {
|
||||
Id StorageByteOffset(EmitContext& ctx, const IR::Value& offset, size_t element_size, u32 index_offset) {
|
||||
Id byte_offset{ctx.Def(offset)};
|
||||
if (index_offset != 0) {
|
||||
byte_offset = ctx.OpIAdd(ctx.U32[1], byte_offset, ctx.Const(static_cast<u32>(index_offset * element_size)));
|
||||
Id StorageIndex(EmitContext& ctx, const IR::Value& offset, size_t element_size,
|
||||
u32 index_offset = 0) {
|
||||
if (offset.IsImmediate()) {
|
||||
const u32 imm_offset{static_cast<u32>(offset.U32() / element_size) + index_offset};
|
||||
return ctx.Const(imm_offset);
|
||||
}
|
||||
return byte_offset;
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(element_size))};
|
||||
Id index{ctx.Def(offset)};
|
||||
if (shift != 0) {
|
||||
const Id shift_id{ctx.Const(shift)};
|
||||
index = ctx.OpShiftRightLogical(ctx.U32[1], index, shift_id);
|
||||
}
|
||||
if (index_offset != 0) {
|
||||
index = ctx.OpIAdd(ctx.U32[1], index, ctx.Const(index_offset));
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
Id StoragePointer(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
const StorageTypeDefinition& type_def, size_t element_size,
|
||||
Id StorageDefinitions::* member_ptr, u32 index_offset = 0) {
|
||||
Id StorageDefinitions::*member_ptr, u32 index_offset = 0) {
|
||||
if (!binding.IsImmediate()) {
|
||||
throw NotImplementedException("Dynamic storage buffer indexing");
|
||||
}
|
||||
const Id byte_offset{StorageByteOffset(ctx, offset, element_size, index_offset)};
|
||||
return ctx.StoragePointer(binding.U32(), byte_offset, type_def, static_cast<u32>(element_size), member_ptr);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].*member_ptr};
|
||||
const Id index{StorageIndex(ctx, offset, element_size, index_offset)};
|
||||
return ctx.OpAccessChain(type_def.element, ssbo, ctx.u32_zero_value, index);
|
||||
}
|
||||
|
||||
Id LoadStorage(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset, Id result_type,
|
||||
const StorageTypeDefinition& type_def, size_t element_size,
|
||||
Id StorageDefinitions::* member_ptr, u32 index_offset = 0) {
|
||||
Id StorageDefinitions::*member_ptr, u32 index_offset = 0) {
|
||||
const Id pointer{
|
||||
StoragePointer(ctx, binding, offset, type_def, element_size, member_ptr, index_offset)};
|
||||
return ctx.OpLoad(result_type, pointer);
|
||||
@@ -44,7 +57,7 @@ Id LoadStorage32(EmitContext& ctx, const IR::Value& binding, const IR::Value& of
|
||||
|
||||
void WriteStorage(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset, Id value,
|
||||
const StorageTypeDefinition& type_def, size_t element_size,
|
||||
Id StorageDefinitions::* member_ptr, u32 index_offset = 0) {
|
||||
Id StorageDefinitions::*member_ptr, u32 index_offset = 0) {
|
||||
const Id pointer{
|
||||
StoragePointer(ctx, binding, offset, type_def, element_size, member_ptr, index_offset)};
|
||||
ctx.OpStore(pointer, value);
|
||||
|
||||
@@ -299,7 +299,7 @@ void DefineConstBuffers(EmitContext& ctx, const Info& info, Id UniformDefinition
|
||||
}
|
||||
|
||||
void DefineSsbos(EmitContext& ctx, StorageTypeDefinition& type_def,
|
||||
Id StorageDefinitions::* member_type, const Info& info, u32 binding, Id type,
|
||||
Id StorageDefinitions::*member_type, const Info& info, u32 binding, Id type,
|
||||
u32 stride) {
|
||||
const Id array_type{ctx.TypeRuntimeArray(type)};
|
||||
ctx.Decorate(array_type, spv::Decoration::ArrayStride, stride);
|
||||
@@ -309,27 +309,23 @@ void DefineSsbos(EmitContext& ctx, StorageTypeDefinition& type_def,
|
||||
ctx.MemberDecorate(struct_type, 0, spv::Decoration::Offset, 0U);
|
||||
|
||||
const Id struct_pointer{ctx.TypePointer(spv::StorageClass::StorageBuffer, struct_type)};
|
||||
type_def.array = struct_pointer;
|
||||
type_def.element = ctx.TypePointer(spv::StorageClass::StorageBuffer, type);
|
||||
|
||||
u32 index{};
|
||||
for (const StorageBufferDescriptor& desc : info.storage_buffers_descriptors) {
|
||||
const Id variable_type{[&] {
|
||||
if (desc.count == 1) {
|
||||
return struct_pointer;
|
||||
}
|
||||
const Id descriptor_array{ctx.TypeArray(struct_type, ctx.Const(desc.count))};
|
||||
return ctx.TypePointer(spv::StorageClass::StorageBuffer, descriptor_array);
|
||||
}()};
|
||||
const Id id{ctx.AddGlobalVariable(variable_type, spv::StorageClass::StorageBuffer)};
|
||||
const Id id{ctx.AddGlobalVariable(struct_pointer, spv::StorageClass::StorageBuffer)};
|
||||
ctx.Decorate(id, spv::Decoration::Binding, binding);
|
||||
ctx.Decorate(id, spv::Decoration::DescriptorSet, 0U);
|
||||
ctx.Name(id, fmt::format("ssbo{}", index));
|
||||
if (ctx.profile.supported_spirv >= 0x00010400) {
|
||||
ctx.interfaces.push_back(id);
|
||||
}
|
||||
ctx.ssbos[index].*member_type = id;
|
||||
++index;
|
||||
++binding;
|
||||
for (size_t i = 0; i < desc.count; ++i) {
|
||||
ctx.ssbos[index + i].*member_type = id;
|
||||
}
|
||||
index += desc.count;
|
||||
binding += desc.count;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -372,7 +368,8 @@ Id CasFunction(EmitContext& ctx, Operation operation, Id value_type) {
|
||||
return func;
|
||||
}
|
||||
|
||||
Id CasLoop(EmitContext& ctx, Operation operation, Id element_pointer, Id value_type, Id memory_type, spv::Scope scope) {
|
||||
Id CasLoop(EmitContext& ctx, Operation operation, Id array_pointer, Id element_pointer,
|
||||
Id value_type, Id memory_type, spv::Scope scope) {
|
||||
const bool is_shared{scope == spv::Scope::Workgroup};
|
||||
const bool is_struct{!is_shared || ctx.uses_explicit_workgroup_layout};
|
||||
const Id cas_func{CasFunction(ctx, operation, value_type)};
|
||||
@@ -382,12 +379,14 @@ Id CasLoop(EmitContext& ctx, Operation operation, Id element_pointer, Id value_t
|
||||
const Id loop_header{ctx.OpLabel()};
|
||||
const Id continue_block{ctx.OpLabel()};
|
||||
const Id merge_block{ctx.OpLabel()};
|
||||
const Id func_type{is_shared ? ctx.TypeFunction(value_type, ctx.U32[1], value_type)
|
||||
: ctx.TypeFunction(value_type, element_pointer, value_type)};
|
||||
const Id func_type{is_shared
|
||||
? ctx.TypeFunction(value_type, ctx.U32[1], value_type)
|
||||
: ctx.TypeFunction(value_type, ctx.U32[1], value_type, array_pointer)};
|
||||
|
||||
const Id func{ctx.OpFunction(value_type, spv::FunctionControlMask::MaskNone, func_type)};
|
||||
const Id address{ctx.OpFunctionParameter(is_shared ? ctx.U32[1] : element_pointer)};
|
||||
const Id index{ctx.OpFunctionParameter(ctx.U32[1])};
|
||||
const Id op_b{ctx.OpFunctionParameter(value_type)};
|
||||
const Id base{is_shared ? ctx.shared_memory_u32 : ctx.OpFunctionParameter(array_pointer)};
|
||||
ctx.AddLabel();
|
||||
ctx.OpBranch(loop_header);
|
||||
ctx.AddLabel(loop_header);
|
||||
@@ -396,13 +395,8 @@ Id CasLoop(EmitContext& ctx, Operation operation, Id element_pointer, Id value_t
|
||||
ctx.OpBranch(continue_block);
|
||||
|
||||
ctx.AddLabel(continue_block);
|
||||
const Id word_pointer{[&] {
|
||||
if (!is_shared) {
|
||||
return address;
|
||||
}
|
||||
return is_struct ? ctx.OpAccessChain(element_pointer, ctx.shared_memory_u32, zero, address)
|
||||
: ctx.OpAccessChain(element_pointer, ctx.shared_memory_u32, address);
|
||||
}()};
|
||||
const Id word_pointer{is_struct ? ctx.OpAccessChain(element_pointer, base, zero, index)
|
||||
: ctx.OpAccessChain(element_pointer, base, index)};
|
||||
if (value_type.value == ctx.F32[2].value) {
|
||||
const Id u32_value{ctx.OpLoad(ctx.U32[1], word_pointer)};
|
||||
const Id value{ctx.OpUnpackHalf2x16(ctx.F32[2], u32_value)};
|
||||
@@ -486,7 +480,6 @@ EmitContext::EmitContext(const Profile& profile_, const RuntimeInfo& runtime_inf
|
||||
DefineSharedMemoryFunctions(program);
|
||||
DefineConstantBuffers(program.info, uniform_binding);
|
||||
DefineConstantBufferIndirectFunctions(program.info);
|
||||
DefineStorageBufferMappings(program.info, storage_binding);
|
||||
DefineStorageBuffers(program.info, storage_binding);
|
||||
DefineTextureBuffers(program.info, texture_binding);
|
||||
DefineImageBuffers(program.info, image_binding);
|
||||
@@ -539,36 +532,6 @@ Id EmitContext::BitOffset16(const IR::Value& offset) {
|
||||
return OpBitwiseAnd(U32[1], OpShiftLeftLogical(U32[1], Def(offset), Const(3u)), Const(16u));
|
||||
}
|
||||
|
||||
Id EmitContext::StoragePointer(u32 binding, Id byte_offset, const StorageTypeDefinition& type_def,
|
||||
u32 element_size, Id StorageDefinitions::* member_ptr) {
|
||||
const Id ssbo{ssbos[binding].*member_ptr};
|
||||
const u32 segment_count{storage_buffer_mapping_counts[binding]};
|
||||
if (segment_count <= 1) {
|
||||
const Id index{
|
||||
element_size == 1
|
||||
? byte_offset
|
||||
: OpShiftRightLogical(U32[1], byte_offset, Const(static_cast<u32>(std::countr_zero(element_size))))};
|
||||
return OpAccessChain(type_def.element, ssbo, u32_zero_value, index);
|
||||
}
|
||||
|
||||
const Id mapped{OpFunctionCall(U32[2], storage_buffer_map_func,
|
||||
Const(storage_buffer_mapping_bases[binding]),
|
||||
Const(segment_count), byte_offset)};
|
||||
const Id segment{OpCompositeExtract(U32[1], mapped, 0U)};
|
||||
const Id local_offset{OpCompositeExtract(U32[1], mapped, 1U)};
|
||||
const Id index{
|
||||
element_size == 1
|
||||
? local_offset
|
||||
: OpShiftRightLogical(U32[1], local_offset, Const(static_cast<u32>(std::countr_zero(element_size))))};
|
||||
Decorate(segment, spv::Decoration::NonUniform);
|
||||
non_uniform_ids.insert(segment.value);
|
||||
const Id pointer{OpAccessChain(type_def.element, ssbo, segment, u32_zero_value, index)};
|
||||
Decorate(pointer, spv::Decoration::NonUniform);
|
||||
non_uniform_ids.insert(pointer.value);
|
||||
uses_nonuniform_storage_buffer = true;
|
||||
return pointer;
|
||||
}
|
||||
|
||||
void EmitContext::DefineCommonTypes(const Info& info) {
|
||||
void_id = TypeVoid();
|
||||
|
||||
@@ -733,10 +696,12 @@ void EmitContext::DefineSharedMemory(const IR::Program& program) {
|
||||
|
||||
void EmitContext::DefineSharedMemoryFunctions(const IR::Program& program) {
|
||||
if (program.info.uses_shared_increment) {
|
||||
increment_cas_shared = CasLoop(*this, Operation::Increment, shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
increment_cas_shared = CasLoop(*this, Operation::Increment, shared_memory_u32_type,
|
||||
shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
}
|
||||
if (program.info.uses_shared_decrement) {
|
||||
decrement_cas_shared = CasLoop(*this, Operation::Decrement, shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
decrement_cas_shared = CasLoop(*this, Operation::Decrement, shared_memory_u32_type,
|
||||
shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -980,7 +945,8 @@ void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
}
|
||||
using DefPtr = Id StorageDefinitions::*;
|
||||
const Id zero{u32_zero_value};
|
||||
const auto define_body{[&](DefPtr ssbo_member, Id addr, const StorageTypeDefinition& type_def, u32 shift, auto&& callback) {
|
||||
const auto define_body{[&](DefPtr ssbo_member, Id addr, Id element_pointer, u32 shift,
|
||||
auto&& callback) {
|
||||
AddLabel();
|
||||
const size_t num_buffers{info.storage_buffers_descriptors.size()};
|
||||
for (size_t index = 0; index < num_buffers; ++index) {
|
||||
@@ -1007,28 +973,30 @@ void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
OpSelectionMerge(else_label, spv::SelectionControlMask::MaskNone);
|
||||
OpBranchConditional(cond, then_label, else_label);
|
||||
AddLabel(then_label);
|
||||
const Id ssbo_id{ssbos[index].*ssbo_member};
|
||||
const Id ssbo_offset{OpUConvert(U32[1], OpISub(U64, addr, ssbo_addr))};
|
||||
const Id ssbo_pointer{StoragePointer(static_cast<u32>(index), ssbo_offset, type_def, 1U << shift, ssbo_member)};
|
||||
const Id ssbo_index{OpShiftRightLogical(U32[1], ssbo_offset, Const(shift))};
|
||||
const Id ssbo_pointer{OpAccessChain(element_pointer, ssbo_id, zero, ssbo_index)};
|
||||
callback(ssbo_pointer);
|
||||
AddLabel(else_label);
|
||||
}
|
||||
}};
|
||||
const auto define_load{
|
||||
[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(type, U64)};
|
||||
const Id func_id{OpFunction(type, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
define_body(ssbo_member, addr, type_def, shift, [&](Id ssbo_pointer) { OpReturnValue(OpLoad(type, ssbo_pointer)); });
|
||||
OpReturnValue(ConstantNull(type));
|
||||
OpFunctionEnd();
|
||||
return func_id;
|
||||
}};
|
||||
const auto define_write{[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, u32 shift) {
|
||||
const auto define_load{[&](DefPtr ssbo_member, Id element_pointer, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(type, U64)};
|
||||
const Id func_id{OpFunction(type, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
define_body(ssbo_member, addr, element_pointer, shift,
|
||||
[&](Id ssbo_pointer) { OpReturnValue(OpLoad(type, ssbo_pointer)); });
|
||||
OpReturnValue(ConstantNull(type));
|
||||
OpFunctionEnd();
|
||||
return func_id;
|
||||
}};
|
||||
const auto define_write{[&](DefPtr ssbo_member, Id element_pointer, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(void_id, U64, type)};
|
||||
const Id func_id{OpFunction(void_id, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
const Id data{OpFunctionParameter(type)};
|
||||
define_body(ssbo_member, addr, type_def, shift, [&](Id ssbo_pointer) {
|
||||
define_body(ssbo_member, addr, element_pointer, shift, [&](Id ssbo_pointer) {
|
||||
OpStore(ssbo_pointer, data);
|
||||
OpReturn();
|
||||
});
|
||||
@@ -1038,9 +1006,10 @@ void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
}};
|
||||
const auto define{
|
||||
[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, size_t size) {
|
||||
const Id element_type{type_def.element};
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(size))};
|
||||
const Id load_func{define_load(ssbo_member, type_def, type, shift)};
|
||||
const Id write_func{define_write(ssbo_member, type_def, type, shift)};
|
||||
const Id load_func{define_load(ssbo_member, element_type, type, shift)};
|
||||
const Id write_func{define_write(ssbo_member, element_type, type, shift)};
|
||||
return std::make_pair(load_func, write_func);
|
||||
}};
|
||||
std::tie(load_global_func_u32, write_global_func_u32) =
|
||||
@@ -1259,79 +1228,6 @@ void EmitContext::DefineConstantBufferIndirectFunctions(const Info& info) {
|
||||
}
|
||||
}
|
||||
|
||||
void EmitContext::DefineStorageBufferMappings(const Info& info, u32& binding) {
|
||||
if (!UsesStorageBufferMappings(info)) {
|
||||
return;
|
||||
}
|
||||
ASSERT(profile.support_storage_buffer_array_nonuniform_indexing);
|
||||
AddExtension("SPV_KHR_storage_buffer_storage_class");
|
||||
|
||||
const u32 num_entries{NumDescriptors(info.storage_buffers_descriptors)};
|
||||
const Id array_type{TypeArray(U32[1], Const(num_entries))};
|
||||
Decorate(array_type, spv::Decoration::ArrayStride, sizeof(u32));
|
||||
const Id struct_type{TypeStruct(array_type)};
|
||||
Decorate(struct_type, spv::Decoration::Block);
|
||||
MemberName(struct_type, 0, "segment_sizes");
|
||||
MemberDecorate(struct_type, 0, spv::Decoration::Offset, 0U);
|
||||
const Id pointer_type{TypePointer(spv::StorageClass::StorageBuffer, struct_type)};
|
||||
storage_buffer_mapping_u32 = TypePointer(spv::StorageClass::StorageBuffer, U32[1]);
|
||||
storage_buffer_mapping = AddGlobalVariable(pointer_type, spv::StorageClass::StorageBuffer);
|
||||
Decorate(storage_buffer_mapping, spv::Decoration::Binding, binding++);
|
||||
Decorate(storage_buffer_mapping, spv::Decoration::DescriptorSet, 0U);
|
||||
Name(storage_buffer_mapping, "storage_buffer_mapping");
|
||||
//Starting with version 1.4... (https://registry.khronos.org/SPIR-V/specs/unified1/SPIRV.html)
|
||||
if (profile.supported_spirv >= 0x00010400) {
|
||||
interfaces.push_back(storage_buffer_mapping);
|
||||
}
|
||||
|
||||
u32 mapping_base{};
|
||||
for (u32 index = 0; index < info.storage_buffers_descriptors.size(); ++index) {
|
||||
const StorageBufferDescriptor& desc = info.storage_buffers_descriptors[index];
|
||||
storage_buffer_mapping_bases[index] = mapping_base;
|
||||
storage_buffer_mapping_counts[index] = desc.count;
|
||||
mapping_base += desc.count;
|
||||
}
|
||||
|
||||
const Id function_type{TypeFunction(U32[2], U32[1], U32[1], U32[1])};
|
||||
storage_buffer_map_func = OpFunction(U32[2], spv::FunctionControlMask::MaskNone, function_type);
|
||||
const Id base{OpFunctionParameter(U32[1])};
|
||||
const Id count{OpFunctionParameter(U32[1])};
|
||||
const Id byte_offset{OpFunctionParameter(U32[1])};
|
||||
const Id index_pointer_type{TypePointer(spv::StorageClass::Function, U32[1])};
|
||||
const Id loop_header{OpLabel()};
|
||||
const Id continue_block{OpLabel()};
|
||||
const Id merge_block{OpLabel()};
|
||||
AddLabel();
|
||||
const Id segment_var{AddLocalVariable(index_pointer_type, spv::StorageClass::Function)};
|
||||
const Id offset_var{AddLocalVariable(index_pointer_type, spv::StorageClass::Function)};
|
||||
OpStore(segment_var, u32_zero_value);
|
||||
OpStore(offset_var, byte_offset);
|
||||
OpBranch(loop_header);
|
||||
|
||||
AddLabel(loop_header);
|
||||
const Id segment{OpLoad(U32[1], segment_var)};
|
||||
const Id local_offset{OpLoad(U32[1], offset_var)};
|
||||
const Id mapping_index{OpIAdd(U32[1], base, segment)};
|
||||
const Id size_pointer{OpAccessChain(storage_buffer_mapping_u32, storage_buffer_mapping, u32_zero_value, mapping_index)};
|
||||
const Id segment_size{OpLoad(U32[1], size_pointer)};
|
||||
const Id fits{OpULessThan(U1, local_offset, segment_size)};
|
||||
const Id last_segment{OpISub(U32[1], count, Const(1U))};
|
||||
const Id is_last{OpIEqual(U1, segment, last_segment)};
|
||||
const Id found{OpLogicalOr(U1, fits, is_last)};
|
||||
OpLoopMerge(merge_block, continue_block, spv::LoopControlMask::MaskNone);
|
||||
OpBranchConditional(found, merge_block, continue_block);
|
||||
|
||||
AddLabel(continue_block);
|
||||
OpStore(offset_var, OpISub(U32[1], local_offset, segment_size));
|
||||
OpStore(segment_var, OpIAdd(U32[1], segment, Const(1U)));
|
||||
OpBranch(loop_header);
|
||||
|
||||
AddLabel(merge_block);
|
||||
OpReturnValue(OpCompositeConstruct(U32[2], segment, local_offset));
|
||||
OpFunctionEnd();
|
||||
Name(storage_buffer_map_func, "map_storage_buffer");
|
||||
}
|
||||
|
||||
void EmitContext::DefineStorageBuffers(const Info& info, u32& binding) {
|
||||
if (info.storage_buffers_descriptors.empty()) {
|
||||
return;
|
||||
@@ -1375,7 +1271,9 @@ void EmitContext::DefineStorageBuffers(const Info& info, u32& binding) {
|
||||
DefineSsbos(*this, storage_types.U32x4, &StorageDefinitions::U32x4, info, binding, U32[4],
|
||||
sizeof(u32[4]));
|
||||
}
|
||||
binding += static_cast<u32>(info.storage_buffers_descriptors.size());
|
||||
for (const StorageBufferDescriptor& desc : info.storage_buffers_descriptors) {
|
||||
binding += desc.count;
|
||||
}
|
||||
const bool needs_function{
|
||||
info.uses_global_increment || info.uses_global_decrement || info.uses_atomic_f32_add ||
|
||||
info.uses_atomic_f16x2_add || info.uses_atomic_f16x2_min || info.uses_atomic_f16x2_max ||
|
||||
@@ -1384,31 +1282,40 @@ void EmitContext::DefineStorageBuffers(const Info& info, u32& binding) {
|
||||
AddCapability(spv::Capability::VariablePointersStorageBuffer);
|
||||
}
|
||||
if (info.uses_global_increment) {
|
||||
increment_cas_ssbo = CasLoop(*this, Operation::Increment, storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
increment_cas_ssbo = CasLoop(*this, Operation::Increment, storage_types.U32.array,
|
||||
storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_global_decrement) {
|
||||
decrement_cas_ssbo = CasLoop(*this, Operation::Decrement, storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
decrement_cas_ssbo = CasLoop(*this, Operation::Decrement, storage_types.U32.array,
|
||||
storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32_add) {
|
||||
f32_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.element, F32[1], U32[1], spv::Scope::Device);
|
||||
f32_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[1], U32[1], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f16x2_add) {
|
||||
f16x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
f16x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.array,
|
||||
storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f16x2_min) {
|
||||
f16x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
f16x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.array,
|
||||
storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f16x2_max) {
|
||||
f16x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
f16x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.array,
|
||||
storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32x2_add) {
|
||||
f32x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
f32x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32x2_min) {
|
||||
f32x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
f32x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32x2_max) {
|
||||
f32x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
f32x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -114,6 +114,7 @@ struct UniformDefinitions {
|
||||
};
|
||||
|
||||
struct StorageTypeDefinition {
|
||||
Id array{};
|
||||
Id element{};
|
||||
};
|
||||
|
||||
@@ -172,8 +173,6 @@ public:
|
||||
[[nodiscard]] Id BitOffset8(const IR::Value& offset);
|
||||
[[nodiscard]] Id BitOffset16(const IR::Value& offset);
|
||||
|
||||
[[nodiscard]] Id StoragePointer(u32 binding, Id byte_offset, const StorageTypeDefinition& type_def, u32 element_size, Id StorageDefinitions::*member_ptr);
|
||||
|
||||
Id Const(u32 value) {
|
||||
return Constant(U32[1], value);
|
||||
}
|
||||
@@ -258,11 +257,6 @@ public:
|
||||
|
||||
std::array<UniformDefinitions, Info::MAX_CBUFS> cbufs{};
|
||||
std::array<StorageDefinitions, Info::MAX_SSBOS> ssbos{};
|
||||
std::array<u32, Info::MAX_SSBOS> storage_buffer_mapping_bases{};
|
||||
std::array<u32, Info::MAX_SSBOS> storage_buffer_mapping_counts{};
|
||||
Id storage_buffer_mapping{};
|
||||
Id storage_buffer_mapping_u32{};
|
||||
Id storage_buffer_map_func{};
|
||||
std::vector<TextureBufferDefinition> texture_buffers;
|
||||
std::vector<ImageBufferDefinition> image_buffers;
|
||||
std::vector<TextureDefinition> textures;
|
||||
@@ -382,7 +376,6 @@ public:
|
||||
bool uses_nonuniform_storage_image{};
|
||||
bool uses_nonuniform_uniform_texel_buffer{};
|
||||
bool uses_nonuniform_storage_texel_buffer{};
|
||||
bool uses_nonuniform_storage_buffer{};
|
||||
|
||||
private:
|
||||
void DefineCommonTypes(const Info& info);
|
||||
@@ -393,7 +386,6 @@ private:
|
||||
void DefineSharedMemoryFunctions(const IR::Program& program);
|
||||
void DefineConstantBuffers(const Info& info, u32& binding);
|
||||
void DefineConstantBufferIndirectFunctions(const Info& info);
|
||||
void DefineStorageBufferMappings(const Info& info, u32& binding);
|
||||
void DefineStorageBuffers(const Info& info, u32& binding);
|
||||
void DefineTextureBuffers(const Info& info, u32& binding);
|
||||
void DefineImageBuffers(const Info& info, u32& binding);
|
||||
|
||||
@@ -132,14 +132,6 @@ void AddNVNStorageBuffers(IR::Program& program) {
|
||||
}
|
||||
}
|
||||
|
||||
void ConfigureStorageBufferMappings(IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
// https://docs.vulkan.org/guide/latest/descriptor_arrays.html
|
||||
// will be needed: descriptor array elements represent physical spans of one guest virtual buffer.
|
||||
for (StorageBufferDescriptor& desc : program.info.storage_buffers_descriptors) {
|
||||
desc.count = host_info.storage_buffer_segment_count;
|
||||
}
|
||||
}
|
||||
|
||||
using IR::IsLegacyAttribute; //rescoped to attribute.h to make it visible in load_store_attribute.cpp IPA
|
||||
|
||||
std::map<IR::Attribute, IR::Attribute> GenerateLegacyToGenericMappings(
|
||||
@@ -307,7 +299,6 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
Optimization::PositionPass(env, program);
|
||||
|
||||
Optimization::GlobalMemoryToStorageBufferPass(program, normalized_host_info);
|
||||
ConfigureStorageBufferMappings(program, normalized_host_info);
|
||||
Optimization::TexturePass(env, program, normalized_host_info);
|
||||
|
||||
if (Settings::values.resolution_info.active || Settings::values.rescale_hack.GetValue()) {
|
||||
@@ -323,7 +314,6 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
|
||||
CollectInterpolationInfo(env, program);
|
||||
AddNVNStorageBuffers(program);
|
||||
ConfigureStorageBufferMappings(program, normalized_host_info);
|
||||
return program;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include "common/common_types.h"
|
||||
|
||||
namespace Shader {
|
||||
@@ -20,7 +19,6 @@ struct HostTranslateInfo {
|
||||
|
||||
u64 min_ssbo_alignment{}; ///< Minimum alignment supported by the device for SSBOs
|
||||
u32 max_per_stage_descriptor_sampled_images{}; ///< maximum sampled descriptors per stage
|
||||
u32 max_per_stage_descriptor_storage_buffers{}; ///< maximum storage descriptors per stage
|
||||
u32 max_per_stage_resources{}; ///< maximum resources per stage
|
||||
u32 max_descriptor_set_samplers{};
|
||||
u32 max_descriptor_set_uniform_buffers{};
|
||||
@@ -40,14 +38,12 @@ struct HostTranslateInfo {
|
||||
///< passthrough shaders
|
||||
bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional
|
||||
///< control flow
|
||||
u32 storage_buffer_segment_count{1}; ///< Physical ranges available to each guest SSBO
|
||||
|
||||
void ApplyDescriptorLimitPolicy() noexcept {
|
||||
if (min_ssbo_alignment == 0) {
|
||||
min_ssbo_alignment = 1;
|
||||
}
|
||||
ApplyDescriptorLimitFallback(max_per_stage_descriptor_sampled_images);
|
||||
ApplyDescriptorLimitFallback(max_per_stage_descriptor_storage_buffers);
|
||||
ApplyDescriptorLimitFallback(max_per_stage_resources);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_samplers);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_uniform_buffers);
|
||||
@@ -57,7 +53,6 @@ struct HostTranslateInfo {
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_sampled_images);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_storage_images);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_input_attachements);
|
||||
storage_buffer_segment_count = (std::max)(storage_buffer_segment_count, 1U);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
@@ -64,7 +64,6 @@ struct Profile {
|
||||
bool support_storage_image_array_nonuniform_indexing{};
|
||||
bool support_uniform_texel_buffer_array_nonuniform_indexing{};
|
||||
bool support_storage_texel_buffer_array_nonuniform_indexing{};
|
||||
bool support_storage_buffer_array_nonuniform_indexing{};
|
||||
|
||||
bool warp_size_potentially_larger_than_guest{};
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <map>
|
||||
@@ -341,10 +340,6 @@ struct Info {
|
||||
ImageDescriptors image_descriptors;
|
||||
};
|
||||
|
||||
[[nodiscard]] inline bool UsesStorageBufferMappings(const Info& info) noexcept {
|
||||
return std::ranges::any_of(info.storage_buffers_descriptors, [](const auto& desc) { return desc.count > 1; });
|
||||
}
|
||||
|
||||
template <typename Descriptors>
|
||||
u32 NumDescriptors(const Descriptors& descriptors) {
|
||||
u32 num{};
|
||||
|
||||
@@ -7,8 +7,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <memory>
|
||||
#include <numeric>
|
||||
|
||||
@@ -425,9 +423,20 @@ void BufferCache<P>::UnbindGraphicsStorageBuffers(size_t stage) {
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::BindGraphicsStorageBuffer(size_t stage, size_t ssbo_index, u32 cbuf_index,
|
||||
u32 cbuf_offset, bool is_written, u32 descriptor_count) {
|
||||
u32 cbuf_offset, bool is_written) {
|
||||
const bool already_enabled =
|
||||
((channel_state->enabled_storage_buffers[stage] >> ssbo_index) & 1U) != 0;
|
||||
if constexpr (requires { runtime.ShouldLimitDynamicStorageBuffers(); }) {
|
||||
if (runtime.ShouldLimitDynamicStorageBuffers() && !already_enabled) {
|
||||
const u32 max_bindings = runtime.GetMaxDynamicStorageBuffers();
|
||||
if (channel_state->total_graphics_storage_buffers >= max_bindings) {
|
||||
LOG_WARNING(HW_GPU,
|
||||
"Skipping graphics storage buffer {} due to driver limit {}",
|
||||
ssbo_index, max_bindings);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
channel_state->enabled_storage_buffers[stage] |= 1U << ssbo_index;
|
||||
channel_state->written_storage_buffers[stage] |= (is_written ? 1U : 0U) << ssbo_index;
|
||||
if constexpr (requires { runtime.ShouldLimitDynamicStorageBuffers(); }) {
|
||||
@@ -438,18 +447,9 @@ bool BufferCache<P>::BindGraphicsStorageBuffer(size_t stage, size_t ssbo_index,
|
||||
|
||||
const auto& cbufs = maxwell3d->state.shader_stages[stage];
|
||||
const GPUVAddr ssbo_addr = cbufs.const_buffers[cbuf_index].address + cbuf_offset;
|
||||
StorageBufferBindingInfo& slot = channel_state->storage_buffers[stage][ssbo_index];
|
||||
const StorageBufferBindingInfo binding =
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written, descriptor_count);
|
||||
if (slot.gpu_addr != binding.gpu_addr || slot.size != binding.size ||
|
||||
slot.descriptor_count != binding.descriptor_count) {
|
||||
slot.gpu_addr = binding.gpu_addr;
|
||||
slot.size = binding.size;
|
||||
slot.descriptor_count = binding.descriptor_count;
|
||||
slot.mapping_generation = 0;
|
||||
slot.segments.clear();
|
||||
}
|
||||
return slot.gpu_addr != 0;
|
||||
channel_state->storage_buffers[stage][ssbo_index] =
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written);
|
||||
return (channel_state->storage_buffers[stage][ssbo_index].buffer_id != NULL_BUFFER_ID);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -487,7 +487,7 @@ void BufferCache<P>::UnbindComputeStorageBuffers() {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index, u32 cbuf_offset,
|
||||
bool is_written, u32 descriptor_count) {
|
||||
bool is_written) {
|
||||
if (ssbo_index >= channel_state->compute_storage_buffers.size()) [[unlikely]] {
|
||||
LOG_ERROR(HW_GPU, "Storage buffer index {} exceeds maximum storage buffer count",
|
||||
ssbo_index);
|
||||
@@ -495,6 +495,17 @@ void BufferCache<P>::BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index,
|
||||
}
|
||||
const bool already_enabled =
|
||||
((channel_state->enabled_compute_storage_buffers >> ssbo_index) & 1U) != 0;
|
||||
if constexpr (requires { runtime.ShouldLimitDynamicStorageBuffers(); }) {
|
||||
if (runtime.ShouldLimitDynamicStorageBuffers() && !already_enabled) {
|
||||
const u32 max_bindings = runtime.GetMaxDynamicStorageBuffers();
|
||||
if (channel_state->total_compute_storage_buffers >= max_bindings) {
|
||||
LOG_WARNING(HW_GPU,
|
||||
"Skipping compute storage buffer {} due to driver limit {}",
|
||||
ssbo_index, max_bindings);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
channel_state->enabled_compute_storage_buffers |= 1U << ssbo_index;
|
||||
channel_state->written_compute_storage_buffers |= (is_written ? 1U : 0U) << ssbo_index;
|
||||
if constexpr (requires { runtime.ShouldLimitDynamicStorageBuffers(); }) {
|
||||
@@ -512,17 +523,8 @@ void BufferCache<P>::BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index,
|
||||
|
||||
const auto& cbufs = launch_desc.const_buffer_config;
|
||||
const GPUVAddr ssbo_addr = cbufs[cbuf_index].Address() + cbuf_offset;
|
||||
StorageBufferBindingInfo& slot = channel_state->compute_storage_buffers[ssbo_index];
|
||||
const StorageBufferBindingInfo binding =
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written, descriptor_count);
|
||||
if (slot.gpu_addr != binding.gpu_addr || slot.size != binding.size ||
|
||||
slot.descriptor_count != binding.descriptor_count) {
|
||||
slot.gpu_addr = binding.gpu_addr;
|
||||
slot.size = binding.size;
|
||||
slot.descriptor_count = binding.descriptor_count;
|
||||
slot.mapping_generation = 0;
|
||||
slot.segments.clear();
|
||||
}
|
||||
channel_state->compute_storage_buffers[ssbo_index] =
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -998,52 +1000,27 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
boost::container::small_vector<u32, NUM_STORAGE_BUFFERS * NUM_STORAGE_BUFFER_SEGMENTS> segment_sizes;
|
||||
bool uses_mapping{};
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
const StorageBufferBindingInfo& binding = channel_state->storage_buffers[stage][index];
|
||||
uses_mapping |= binding.descriptor_count > 1;
|
||||
for (u32 segment = 0; segment < binding.descriptor_count; ++segment) {
|
||||
segment_sizes.push_back(segment < binding.segments.size() ? binding.segments[segment].size : 0);
|
||||
}
|
||||
});
|
||||
if (uses_mapping) {
|
||||
const u32 mapping_size = static_cast<u32>(segment_sizes.size() * sizeof(u32));
|
||||
if constexpr (!IS_OPENGL) {
|
||||
const std::span<u8> mapped = runtime.BindMappedStorageBuffer(mapping_size);
|
||||
std::memcpy(mapped.data(), segment_sizes.data(), mapping_size);
|
||||
}
|
||||
}
|
||||
|
||||
u32 binding_index = 0;
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
const StorageBufferBindingInfo& storage = channel_state->storage_buffers[stage][index];
|
||||
const Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
|
||||
const u32 offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
for (u32 segment = 0; segment < storage.descriptor_count; ++segment) {
|
||||
Buffer* buffer = &slot_buffers[NULL_BUFFER_ID];
|
||||
u32 offset{};
|
||||
u32 size{IS_OPENGL ? 0U : static_cast<u32>(sizeof(u32))};
|
||||
const bool is_actual_segment = segment < storage.segments.size();
|
||||
// shall be safe enough if the segment is not actual, use the last available segment or nullptr if none exist.
|
||||
const Binding* binding = is_actual_segment ? &storage.segments[segment] : storage.segments.empty() ? nullptr : &storage.segments.back();
|
||||
if (binding) {
|
||||
buffer = &slot_buffers[binding->buffer_id];
|
||||
size = binding->size;
|
||||
offset = buffer->Offset(binding->device_addr);
|
||||
if (is_actual_segment) {
|
||||
TouchBuffer(*buffer, binding->buffer_id);
|
||||
SynchronizeBuffer(*buffer, binding->device_addr, size);
|
||||
buffer->MarkUsage(offset, size);
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding->buffer_id, binding->device_addr, size);
|
||||
}
|
||||
}
|
||||
}
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindStorageBuffer(stage, binding_index++, *buffer, offset, size, is_written);
|
||||
} else {
|
||||
runtime.BindStorageBuffer(*buffer, offset, size, is_written);
|
||||
}
|
||||
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
}
|
||||
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindStorageBuffer(stage, binding_index, buffer, offset, size, is_written);
|
||||
++binding_index;
|
||||
} else {
|
||||
runtime.BindStorageBuffer(buffer, offset, size, is_written);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1159,53 +1136,28 @@ void BufferCache<P>::BindHostComputeUniformBuffers() {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostComputeStorageBuffers() {
|
||||
boost::container::small_vector<u32, NUM_STORAGE_BUFFERS * NUM_STORAGE_BUFFER_SEGMENTS> segment_sizes;
|
||||
bool uses_mapping{};
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
const StorageBufferBindingInfo& binding = channel_state->compute_storage_buffers[index];
|
||||
uses_mapping |= binding.descriptor_count > 1;
|
||||
for (u32 segment = 0; segment < binding.descriptor_count; ++segment) {
|
||||
segment_sizes.push_back(segment < binding.segments.size() ? binding.segments[segment].size : 0);
|
||||
}
|
||||
});
|
||||
if (uses_mapping) {
|
||||
const u32 mapping_size = static_cast<u32>(segment_sizes.size() * sizeof(u32));
|
||||
if constexpr (!IS_OPENGL) {
|
||||
const std::span<u8> mapped = runtime.BindMappedStorageBuffer(mapping_size);
|
||||
std::memcpy(mapped.data(), segment_sizes.data(), mapping_size);
|
||||
}
|
||||
}
|
||||
|
||||
u32 binding_index = 0;
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
const StorageBufferBindingInfo& storage = channel_state->compute_storage_buffers[index];
|
||||
const Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
|
||||
const u32 offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
for (u32 segment = 0; segment < storage.descriptor_count; ++segment) {
|
||||
Buffer* buffer = &slot_buffers[NULL_BUFFER_ID];
|
||||
u32 offset{};
|
||||
u32 size{IS_OPENGL ? 0U : static_cast<u32>(sizeof(u32))};
|
||||
const bool is_actual_segment = segment < storage.segments.size();
|
||||
//same fallback logic
|
||||
const Binding* binding = is_actual_segment ? &storage.segments[segment] : storage.segments.empty() ? nullptr : &storage.segments.back();
|
||||
if (binding) {
|
||||
buffer = &slot_buffers[binding->buffer_id];
|
||||
size = binding->size;
|
||||
offset = buffer->Offset(binding->device_addr);
|
||||
if (is_actual_segment) {
|
||||
TouchBuffer(*buffer, binding->buffer_id);
|
||||
SynchronizeBuffer(*buffer, binding->device_addr, size);
|
||||
buffer->MarkUsage(offset, size);
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding->buffer_id, binding->device_addr, size);
|
||||
}
|
||||
}
|
||||
}
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindComputeStorageBuffer(binding_index++, *buffer, offset, size, is_written);
|
||||
} else {
|
||||
runtime.BindStorageBuffer(*buffer, offset, size, is_written);
|
||||
}
|
||||
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
}
|
||||
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindComputeStorageBuffer(binding_index, buffer, offset, size, is_written);
|
||||
++binding_index;
|
||||
} else {
|
||||
runtime.BindStorageBuffer(buffer, offset, size, is_written);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1398,7 +1350,10 @@ void BufferCache<P>::UpdateUniformBuffers(size_t stage) {
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateStorageBuffers(size_t stage) {
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
UpdateStorageBuffer(channel_state->storage_buffers[stage][index]);
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = buffer_id;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1459,58 +1414,12 @@ void BufferCache<P>::UpdateComputeUniformBuffers() {
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateComputeStorageBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
UpdateStorageBuffer(channel_state->compute_storage_buffers[index]);
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
});
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateStorageBuffer(StorageBufferBindingInfo& binding) {
|
||||
if (binding.gpu_addr == 0 || binding.size == 0) {
|
||||
binding.segments.clear();
|
||||
return;
|
||||
}
|
||||
if (binding.descriptor_count == 1) {
|
||||
binding.segments.clear();
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(binding.gpu_addr);
|
||||
if (device_addr) {
|
||||
binding.segments.push_back(Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = binding.size,
|
||||
.buffer_id = FindBuffer(*device_addr, binding.size),
|
||||
});
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const u64 generation = gpu_memory->MappingGeneration();
|
||||
if (binding.mapping_generation != generation || binding.segments.empty()) {
|
||||
binding.mapping_generation = generation;
|
||||
binding.segments.clear();
|
||||
const auto ranges = gpu_memory->GetSubmappedRange(binding.gpu_addr, binding.size);
|
||||
for (const auto& [gpu_addr, size] : ranges) {
|
||||
if (binding.segments.size() >= binding.descriptor_count) {
|
||||
break;
|
||||
}
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(gpu_addr);
|
||||
if (!device_addr) {
|
||||
break;
|
||||
}
|
||||
u32 segment_size = (std::numeric_limits<u32>::max)();
|
||||
if (size < static_cast<size_t>(segment_size)) {
|
||||
segment_size = static_cast<u32>(size);
|
||||
}
|
||||
binding.segments.push_back(Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = segment_size,
|
||||
.buffer_id = BufferId{},
|
||||
});
|
||||
}
|
||||
}
|
||||
for (Binding& segment : binding.segments) {
|
||||
segment.buffer_id = FindBuffer(segment.device_addr, segment.size);
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateComputeTextureBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_texture_buffers, [&](u32 index) {
|
||||
@@ -1932,11 +1841,6 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
const auto replace = [scalar_replace](std::span<Binding> bindings) {
|
||||
std::ranges::for_each(bindings, scalar_replace);
|
||||
};
|
||||
const auto storage_replace = [scalar_replace](std::span<StorageBufferBindingInfo> bindings) {
|
||||
for (StorageBufferBindingInfo& binding : bindings) {
|
||||
std::ranges::for_each(binding.segments, scalar_replace);
|
||||
}
|
||||
};
|
||||
|
||||
if (channel_state->index_buffer.buffer_id == buffer_id) {
|
||||
channel_state->index_buffer.buffer_id = BufferId{};
|
||||
@@ -1952,10 +1856,10 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
}
|
||||
}
|
||||
std::ranges::for_each(channel_state->uniform_buffers, replace);
|
||||
std::ranges::for_each(channel_state->storage_buffers, storage_replace);
|
||||
std::ranges::for_each(channel_state->storage_buffers, replace);
|
||||
replace(channel_state->transform_feedback_buffers);
|
||||
replace(channel_state->compute_uniform_buffers);
|
||||
storage_replace(channel_state->compute_storage_buffers);
|
||||
replace(channel_state->compute_storage_buffers);
|
||||
|
||||
// Mark the whole buffer as CPU written to stop tracking CPU writes
|
||||
if (!do_not_mark) {
|
||||
@@ -1992,14 +1896,12 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
}
|
||||
|
||||
template <class P>
|
||||
StorageBufferBindingInfo BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written, u32 descriptor_count) const {
|
||||
// time to get rid of these null bindings
|
||||
ASSERT(descriptor_count > 0); // shant happen
|
||||
Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written) const {
|
||||
const GPUVAddr gpu_addr = gpu_memory->Read<u64>(ssbo_addr);
|
||||
|
||||
if (gpu_addr == 0) {
|
||||
return {.descriptor_count = descriptor_count};
|
||||
return NULL_BINDING;
|
||||
}
|
||||
|
||||
const auto size = [&]() {
|
||||
@@ -2021,17 +1923,21 @@ StorageBufferBindingInfo BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr
|
||||
const GPUVAddr aligned_gpu_addr = Common::AlignDown(gpu_addr, alignment);
|
||||
const u32 aligned_size = static_cast<u32>(gpu_addr - aligned_gpu_addr) + size;
|
||||
|
||||
if (!gpu_memory->GpuToCpuAddress(aligned_gpu_addr) || size == 0) {
|
||||
const std::optional<DAddr> aligned_device_addr = gpu_memory->GpuToCpuAddress(aligned_gpu_addr);
|
||||
if (!aligned_device_addr || size == 0) {
|
||||
LOG_DEBUG(HW_GPU, "Failed to find storage buffer for cbuf index {}", cbuf_index);
|
||||
return {.descriptor_count = descriptor_count};
|
||||
return NULL_BINDING;
|
||||
}
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(gpu_addr);
|
||||
ASSERT_MSG(device_addr, "Unaligned storage buffer address not found for cbuf index {}",
|
||||
cbuf_index);
|
||||
// The end address used for size calculation does not need to be aligned
|
||||
const GPUVAddr gpu_end = Common::AlignUp(gpu_addr + size, Core::DEVICE_PAGESIZE);
|
||||
const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE);
|
||||
|
||||
const StorageBufferBindingInfo binding{
|
||||
.gpu_addr = aligned_gpu_addr,
|
||||
.size = is_written ? aligned_size : static_cast<u32>(gpu_end - aligned_gpu_addr),
|
||||
.descriptor_count = descriptor_count,
|
||||
const Binding binding{
|
||||
.device_addr = *aligned_device_addr,
|
||||
.size = is_written ? aligned_size : static_cast<u32>(cpu_end - *aligned_device_addr),
|
||||
.buffer_id = BufferId{},
|
||||
};
|
||||
return binding;
|
||||
}
|
||||
|
||||
@@ -54,8 +54,7 @@ constexpr u32 NUM_VERTEX_BUFFERS = 32;
|
||||
constexpr u32 NUM_TRANSFORM_FEEDBACK_BUFFERS = 4;
|
||||
constexpr u32 NUM_GRAPHICS_UNIFORM_BUFFERS = 18;
|
||||
constexpr u32 NUM_COMPUTE_UNIFORM_BUFFERS = 8;
|
||||
constexpr u32 NUM_STORAGE_BUFFERS = 32;
|
||||
constexpr u32 NUM_STORAGE_BUFFER_SEGMENTS = 8;
|
||||
constexpr u32 NUM_STORAGE_BUFFERS = 16;
|
||||
constexpr u32 NUM_TEXTURE_BUFFERS = 32;
|
||||
constexpr u32 NUM_STAGES = 5;
|
||||
|
||||
@@ -90,14 +89,6 @@ struct TextureBufferBinding : Binding {
|
||||
PixelFormat format;
|
||||
};
|
||||
|
||||
struct StorageBufferBindingInfo {
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 size{};
|
||||
u32 descriptor_count{1};
|
||||
u64 mapping_generation{};
|
||||
boost::container::small_vector<Binding, 1> segments;
|
||||
};
|
||||
|
||||
static constexpr Binding NULL_BINDING{
|
||||
.device_addr = 0,
|
||||
.size = 0,
|
||||
@@ -124,15 +115,14 @@ public:
|
||||
Binding index_buffer;
|
||||
std::array<Binding, NUM_VERTEX_BUFFERS> vertex_buffers;
|
||||
std::array<std::array<Binding, NUM_GRAPHICS_UNIFORM_BUFFERS>, NUM_STAGES> uniform_buffers;
|
||||
std::array<std::array<StorageBufferBindingInfo, NUM_STORAGE_BUFFERS>, NUM_STAGES>
|
||||
storage_buffers;
|
||||
std::array<std::array<Binding, NUM_STORAGE_BUFFERS>, NUM_STAGES> storage_buffers;
|
||||
std::array<std::array<TextureBufferBinding, NUM_TEXTURE_BUFFERS>, NUM_STAGES> texture_buffers;
|
||||
std::array<Binding, NUM_TRANSFORM_FEEDBACK_BUFFERS> transform_feedback_buffers;
|
||||
Binding count_buffer_binding;
|
||||
Binding indirect_buffer_binding;
|
||||
|
||||
std::array<Binding, NUM_COMPUTE_UNIFORM_BUFFERS> compute_uniform_buffers;
|
||||
std::array<StorageBufferBindingInfo, NUM_STORAGE_BUFFERS> compute_storage_buffers;
|
||||
std::array<Binding, NUM_STORAGE_BUFFERS> compute_storage_buffers;
|
||||
std::array<TextureBufferBinding, NUM_TEXTURE_BUFFERS> compute_texture_buffers;
|
||||
|
||||
std::array<u32, NUM_STAGES> enabled_uniform_buffer_masks{};
|
||||
@@ -259,7 +249,7 @@ public:
|
||||
void UnbindGraphicsStorageBuffers(size_t stage);
|
||||
|
||||
bool BindGraphicsStorageBuffer(size_t stage, size_t ssbo_index, u32 cbuf_index, u32 cbuf_offset,
|
||||
bool is_written, u32 descriptor_count = 1);
|
||||
bool is_written);
|
||||
|
||||
void UnbindGraphicsTextureBuffers(size_t stage);
|
||||
|
||||
@@ -269,7 +259,7 @@ public:
|
||||
void UnbindComputeStorageBuffers();
|
||||
|
||||
void BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index, u32 cbuf_offset,
|
||||
bool is_written, u32 descriptor_count = 1);
|
||||
bool is_written);
|
||||
|
||||
void UnbindComputeTextureBuffers();
|
||||
|
||||
@@ -410,8 +400,6 @@ private:
|
||||
|
||||
void UpdateStorageBuffers(size_t stage);
|
||||
|
||||
void UpdateStorageBuffer(StorageBufferBindingInfo& binding);
|
||||
|
||||
void UpdateTextureBuffers(size_t stage);
|
||||
|
||||
void UpdateTransformFeedbackBuffers();
|
||||
@@ -461,9 +449,8 @@ private:
|
||||
|
||||
void DeleteBuffer(BufferId buffer_id, bool do_not_mark = false);
|
||||
|
||||
[[nodiscard]] StorageBufferBindingInfo StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written,
|
||||
u32 descriptor_count) const;
|
||||
[[nodiscard]] Binding StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written) const;
|
||||
|
||||
[[nodiscard]] TextureBufferBinding GetTextureBufferBinding(GPUVAddr gpu_addr, u32 size,
|
||||
PixelFormat format);
|
||||
|
||||
@@ -176,14 +176,12 @@ void MemoryManager::BindRasterizer(VideoCore::RasterizerInterface* rasterizer_)
|
||||
}
|
||||
|
||||
GPUVAddr MemoryManager::Map(GPUVAddr gpu_addr, DAddr dev_addr, std::size_t size, PTEKind kind, bool is_big_pages) {
|
||||
mapping_generation.fetch_add(1, std::memory_order_release);
|
||||
if (is_big_pages)
|
||||
return BigPageTableOp(gpu_addr, dev_addr, size, kind, EntryType::Mapped);
|
||||
return PageTableOp(gpu_addr, dev_addr, size, kind, EntryType::Mapped);
|
||||
}
|
||||
|
||||
GPUVAddr MemoryManager::MapSparse(GPUVAddr gpu_addr, std::size_t size, bool is_big_pages) {
|
||||
mapping_generation.fetch_add(1, std::memory_order_release);
|
||||
if (is_big_pages)
|
||||
return BigPageTableOp(gpu_addr, 0, size, PTEKind::INVALID, EntryType::Reserved);
|
||||
return PageTableOp(gpu_addr, 0, size, PTEKind::INVALID, EntryType::Reserved);
|
||||
@@ -193,7 +191,6 @@ void MemoryManager::Unmap(GPUVAddr gpu_addr, std::size_t size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
}
|
||||
mapping_generation.fetch_add(1, std::memory_order_release);
|
||||
GetSubmappedRangeImpl<false>(gpu_addr, size, page_stash);
|
||||
|
||||
for (const auto& [map_addr, map_size] : page_stash) {
|
||||
|
||||
@@ -145,10 +145,6 @@ public:
|
||||
return gpu_addr < address_space_size;
|
||||
}
|
||||
|
||||
u64 MappingGeneration() const noexcept {
|
||||
return mapping_generation.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
PTEKind GetPageKind(GPUVAddr gpu_addr) const;
|
||||
|
||||
size_t GetMemoryLayoutSize(GPUVAddr gpu_addr,
|
||||
@@ -201,8 +197,6 @@ private:
|
||||
|
||||
VideoCore::RasterizerInterface* rasterizer = nullptr;
|
||||
|
||||
std::atomic<u64> mapping_generation{1};
|
||||
|
||||
enum class EntryType : u64 {
|
||||
Free = 0,
|
||||
Reserved = 1,
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <optional>
|
||||
|
||||
@@ -137,7 +136,6 @@ inline void WriteDescriptorBuffer(const Device& device, const DescriptorBufferLa
|
||||
|
||||
[[nodiscard]] inline u32 NumDescriptorEntries(const Shader::Info& info) {
|
||||
return Shader::NumDescriptors(info.constant_buffer_descriptors) +
|
||||
static_cast<u32>(Shader::UsesStorageBufferMappings(info)) +
|
||||
Shader::NumDescriptors(info.storage_buffers_descriptors) +
|
||||
Shader::NumDescriptors(info.texture_buffer_descriptors) +
|
||||
Shader::NumDescriptors(info.image_buffer_descriptors) +
|
||||
@@ -270,12 +268,6 @@ public:
|
||||
is_compute |= (stage & VK_SHADER_STAGE_COMPUTE_BIT) != 0;
|
||||
|
||||
Add(VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER, stage, info.constant_buffer_descriptors);
|
||||
// for extra implicit storage-buffer binding required by mapped-storage-buffer support
|
||||
if (Shader::UsesStorageBufferMappings(info)) {
|
||||
struct Descriptor { u32 count; };
|
||||
const std::array descriptors{Descriptor{1}};
|
||||
Add(VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, stage, descriptors);
|
||||
}
|
||||
Add(VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, stage, info.storage_buffers_descriptors);
|
||||
Add(VK_DESCRIPTOR_TYPE_UNIFORM_TEXEL_BUFFER, stage, info.texture_buffer_descriptors);
|
||||
Add(VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER, stage, info.image_buffer_descriptors);
|
||||
|
||||
@@ -695,7 +695,7 @@ vk::Buffer BufferCacheRuntime::CreateNullBuffer() {
|
||||
.flags = 0,
|
||||
.size = 4,
|
||||
.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
|
||||
@@ -155,13 +155,6 @@ public:
|
||||
return ref.mapped_span;
|
||||
}
|
||||
|
||||
// compute/graphics new binding
|
||||
std::span<u8> BindMappedStorageBuffer(u32 size) {
|
||||
const StagingBufferRef ref = staging_pool.Request(size, MemoryUsage::Upload);
|
||||
guest_descriptor_queue.AddBuffer(ref.buffer, ref.device_address, static_cast<u32>(ref.offset), size);
|
||||
return ref.mapped_span;
|
||||
}
|
||||
|
||||
void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) {
|
||||
BindBuffer(buffer, offset, size);
|
||||
}
|
||||
|
||||
@@ -157,7 +157,9 @@ bool ComputePipeline::Configure(Tegra::Engines::KeplerCompute& kepler_compute,
|
||||
buffer_cache.UnbindComputeStorageBuffers();
|
||||
size_t ssbo_index{};
|
||||
for (const auto& desc : info.storage_buffers_descriptors) {
|
||||
buffer_cache.BindComputeStorageBuffer(ssbo_index, desc.cbuf_index, desc.cbuf_offset, desc.is_written, desc.count);
|
||||
ASSERT(desc.count == 1);
|
||||
buffer_cache.BindComputeStorageBuffer(ssbo_index, desc.cbuf_index, desc.cbuf_offset,
|
||||
desc.is_written);
|
||||
++ssbo_index;
|
||||
}
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ class Scheduler;
|
||||
|
||||
class DescriptorBufferRing final {
|
||||
static constexpr size_t FRAMES_IN_FLIGHT = 8;
|
||||
static constexpr VkDeviceSize TILER_FRAME_SIZE = 8 * 1024 * 1024;
|
||||
static constexpr VkDeviceSize TILER_FRAME_SIZE = 2 * 1024 * 1024;
|
||||
static constexpr VkDeviceSize DESKTOP_FRAME_SIZE = 4 * 1024 * 1024;
|
||||
|
||||
public:
|
||||
|
||||
@@ -47,7 +47,7 @@ static DescriptorBankInfo MakeBankInfo(std::span<const Shader::Info> infos) {
|
||||
DescriptorBankInfo bank;
|
||||
for (const Shader::Info& info : infos) {
|
||||
bank.uniform_buffers += Accumulate(info.constant_buffer_descriptors);
|
||||
bank.storage_buffers += Accumulate(info.storage_buffers_descriptors) + static_cast<u32>(Shader::UsesStorageBufferMappings(info));
|
||||
bank.storage_buffers += Accumulate(info.storage_buffers_descriptors);
|
||||
bank.texture_buffers += Accumulate(info.texture_buffer_descriptors);
|
||||
bank.image_buffers += Accumulate(info.image_buffer_descriptors);
|
||||
bank.textures += Accumulate(info.texture_descriptors);
|
||||
|
||||
@@ -315,7 +315,7 @@ GraphicsPipeline::GraphicsPipeline(
|
||||
try {
|
||||
MakePipeline(render_pass);
|
||||
} catch (const vk::Exception& exception) {
|
||||
LOG_DEBUG(Render_Vulkan, "Graphics pipeline build failed: {}", exception.what());
|
||||
LOG_CRITICAL(Render_Vulkan, "Graphics pipeline build failed: {}", exception.what());
|
||||
std::scoped_lock lock{build_mutex};
|
||||
is_built = true;
|
||||
build_condvar.notify_one();
|
||||
@@ -367,8 +367,9 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
|
||||
if constexpr (Spec::has_storage_buffers) {
|
||||
size_t ssbo_index{};
|
||||
for (const auto& desc : info.storage_buffers_descriptors) {
|
||||
ASSERT(desc.count == 1);
|
||||
buffer_cache.BindGraphicsStorageBuffer(stage, ssbo_index, desc.cbuf_index,
|
||||
desc.cbuf_offset, desc.is_written, desc.count);
|
||||
desc.cbuf_offset, desc.is_written);
|
||||
++ssbo_index;
|
||||
}
|
||||
}
|
||||
@@ -581,6 +582,7 @@ bool GraphicsPipeline::ConfigureDraw(const RescalingPushConstant& rescaling,
|
||||
const DescriptorBufferRing::Allocation alloc{
|
||||
descriptor_buffer_ring.Allocate(scheduler, descriptor_buffer_layout.size)};
|
||||
if (!alloc.host) {
|
||||
LOG_DEBUG(Render_Vulkan, "Failed to reserve descriptor memory, skipping draw");
|
||||
return false;
|
||||
}
|
||||
WriteDescriptorBuffer(device, descriptor_buffer_layout, entries, alloc.host);
|
||||
|
||||
@@ -62,9 +62,7 @@ using VideoCommon::FileEnvironment;
|
||||
using VideoCommon::GenericEnvironment;
|
||||
using VideoCommon::GraphicsEnvironment;
|
||||
|
||||
constexpr u32 MAX_MAPPED_STORAGE_BUFFER_DESCRIPTORS = 8;
|
||||
|
||||
constexpr u32 CACHE_VERSION = 19;
|
||||
constexpr u32 CACHE_VERSION = 18;
|
||||
constexpr size_t VULKAN_CACHE_FLUSH_PIPELINES = 128;
|
||||
constexpr size_t VULKAN_CACHE_FLUSH_MIN_SECONDS = 30;
|
||||
constexpr std::array<char, 8> VULKAN_CACHE_MAGIC_NUMBER{'y', 'u', 'z', 'u', 'v', 'k', 'c', 'h'};
|
||||
@@ -434,8 +432,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
device.IsUniformTexelBufferArrayNonUniformIndexingSupported(),
|
||||
.support_storage_texel_buffer_array_nonuniform_indexing =
|
||||
device.IsStorageTexelBufferArrayNonUniformIndexingSupported(),
|
||||
.support_storage_buffer_array_nonuniform_indexing =
|
||||
device.IsStorageBufferArrayNonUniformIndexingSupported(),
|
||||
|
||||
.warp_size_potentially_larger_than_guest = device.IsWarpSizePotentiallyBiggerThanGuest(),
|
||||
|
||||
@@ -464,7 +460,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
host_info = Shader::HostTranslateInfo{
|
||||
.min_ssbo_alignment = device.GetStorageBufferAlignment(),
|
||||
.max_per_stage_descriptor_sampled_images = device.GetMaxPerStageDescriptorSampledImages(),
|
||||
.max_per_stage_descriptor_storage_buffers = device.GetMaxPerStageDescriptorStorageBuffers(),
|
||||
.max_per_stage_resources = device.GetMaxPerStageResources(),
|
||||
.max_descriptor_set_samplers = device.GetMaxDescriptorSetSamplers(),
|
||||
.max_descriptor_set_uniform_buffers = device.GetMaxDescriptorSetUniformBuffers(),
|
||||
@@ -484,20 +479,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
.support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(),
|
||||
.support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(),
|
||||
.support_conditional_barrier = device.SupportsConditionalBarriers(),
|
||||
.storage_buffer_segment_count = [&] {
|
||||
if (!device.IsStorageBufferArrayNonUniformIndexingSupported()) {
|
||||
return 1U;
|
||||
}
|
||||
constexpr u32 MaxGraphicsStages = static_cast<u32>(Maxwell::MaxShaderStage);
|
||||
const auto reserve = [](u32 limit, u32 count) {
|
||||
return limit > count ? limit - count : 0U;
|
||||
};
|
||||
const u32 per_stage = reserve(device.GetMaxPerStageDescriptorStorageBuffers(), 1) / static_cast<u32>(Shader::Info::MAX_SSBOS);
|
||||
const u32 per_set = reserve(device.GetMaxDescriptorSetStorageBuffers(), MaxGraphicsStages) / (static_cast<u32>(Shader::Info::MAX_SSBOS) * MaxGraphicsStages);
|
||||
const u32 resources = reserve(device.GetMaxPerStageResources(), 1) / (static_cast<u32>(Shader::Info::MAX_SSBOS) * 2);
|
||||
// ensure at least one storage buffer segment is available per stage. max is still arbitrary
|
||||
return (std::max)(1U, (std::min)({MAX_MAPPED_STORAGE_BUFFER_DESCRIPTORS, per_stage, per_set, resources}));
|
||||
}(),
|
||||
};
|
||||
host_info.ApplyDescriptorLimitPolicy();
|
||||
|
||||
|
||||
@@ -708,6 +708,7 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
|
||||
descriptor_indexing.shaderUniformTexelBufferArrayDynamicIndexing = false;
|
||||
descriptor_indexing.shaderStorageTexelBufferArrayDynamicIndexing = false;
|
||||
descriptor_indexing.shaderUniformBufferArrayNonUniformIndexing = false;
|
||||
descriptor_indexing.shaderStorageBufferArrayNonUniformIndexing = false;
|
||||
descriptor_indexing.shaderInputAttachmentArrayNonUniformIndexing = false;
|
||||
descriptor_indexing.descriptorBindingUniformBufferUpdateAfterBind = false;
|
||||
descriptor_indexing.descriptorBindingSampledImageUpdateAfterBind = false;
|
||||
|
||||
@@ -360,7 +360,6 @@ public:
|
||||
#define FN_MAX_LIMIT_LIST \
|
||||
FN_MAX_LIMIT_ELEM(ComputeSharedMemorySize) \
|
||||
FN_MAX_LIMIT_ELEM(PerStageDescriptorSampledImages) \
|
||||
FN_MAX_LIMIT_ELEM(PerStageDescriptorStorageBuffers) \
|
||||
FN_MAX_LIMIT_ELEM(PerStageResources) \
|
||||
FN_MAX_LIMIT_ELEM(DescriptorSetSamplers) \
|
||||
FN_MAX_LIMIT_ELEM(DescriptorSetUniformBuffers) \
|
||||
@@ -416,10 +415,6 @@ FN_MAX_LIMIT_LIST
|
||||
return features.descriptor_indexing.shaderStorageTexelBufferArrayNonUniformIndexing;
|
||||
}
|
||||
|
||||
bool IsStorageBufferArrayNonUniformIndexingSupported() const {
|
||||
return features.descriptor_indexing.shaderStorageBufferArrayNonUniformIndexing;
|
||||
}
|
||||
|
||||
/// Returns true if the device supports float64 natively.
|
||||
bool IsFloat64Supported() const {
|
||||
return features.features.shaderFloat64;
|
||||
|
||||
Reference in New Issue
Block a user