mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-09-08 21:15:58 +00:00
[shader_recompiler, vulkan] Virtual buffer pages with multi-range + sparse buffers binding (#4362)
Storage buffers can span GPU pages that aren't contiguous in host memory, but the buffer cache assumed one contiguous buffer per binding, so anything crossing a mapping boundary read the wrong bytes. Multi-range binding resolves the real segments and presents them to the shader as one buffer, aliasing their memory into a sparse VkBuffer or gathering them into a copy, all of this was mostly fixed by making a path to use sparse on buffers (the same way as texture cache has it) and retrieve properly the mapping ranges to the virtual pages, including the actual structure on the use of binding sparse. Meanwhile this fixes Monster Hunter Sunbreak z-fighting, bad performance and mostly vertex explosions (coming from the contiguous buffer read from the shader's game) it's only the basis for a further investigation towards how are we gonna treat phi calculations, optimized them and mostly eradicate the currently excess of exposure on the textures where the lightning trace should not being reflected. Reviewed-on: https://git.eden-emu.dev/eden-emu/eden/pulls/4362 Reviewed-by: lizzie <lizzie@eden-emu.dev> Reviewed-by: Maufeat <sahyno1996@gmail.com>
This commit is contained in:
@@ -20,6 +20,7 @@ add_library(video_core STATIC
|
||||
buffer_cache/buffer_cache.h
|
||||
buffer_cache/memory_tracker_base.h
|
||||
buffer_cache/usage_tracker.h
|
||||
buffer_cache/virtual_range_cache.h
|
||||
buffer_cache/word_manager.h
|
||||
cache_types.h
|
||||
capture.h
|
||||
@@ -166,6 +167,8 @@ add_library(video_core STATIC
|
||||
renderer_vulkan/vk_fence_manager.h
|
||||
renderer_vulkan/vk_graphics_pipeline.cpp
|
||||
renderer_vulkan/vk_graphics_pipeline.h
|
||||
renderer_vulkan/vk_multi_range_buffer.cpp
|
||||
renderer_vulkan/vk_multi_range_buffer.h
|
||||
renderer_vulkan/vk_master_semaphore.cpp
|
||||
renderer_vulkan/vk_master_semaphore.h
|
||||
renderer_vulkan/vk_pipeline_cache.cpp
|
||||
|
||||
@@ -112,6 +112,13 @@ void BufferCache<P>::TickFrame() {
|
||||
async_buffers_death_ring.clear();
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size) {
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
|
||||
virtual_ranges.Unmap(as_id, gpu_addr, size);
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) {
|
||||
if (memory_tracker.IsRegionGpuModified(device_addr, size)) {
|
||||
@@ -208,8 +215,8 @@ bool BufferCache<P>::DMACopy(GPUVAddr src_address, GPUVAddr dest_address, u64 am
|
||||
BufferId buffer_b;
|
||||
do {
|
||||
channel_state->has_deleted_buffers = false;
|
||||
buffer_a = FindBuffer(*cpu_src_address, static_cast<u32>(amount));
|
||||
buffer_b = FindBuffer(*cpu_dest_address, static_cast<u32>(amount));
|
||||
buffer_a = FindBuffer(*cpu_src_address, static_cast<u32>(amount), false);
|
||||
buffer_b = FindBuffer(*cpu_dest_address, static_cast<u32>(amount), false);
|
||||
} while (channel_state->has_deleted_buffers);
|
||||
auto& src_buffer = slot_buffers[buffer_a];
|
||||
auto& dest_buffer = slot_buffers[buffer_b];
|
||||
@@ -265,7 +272,7 @@ bool BufferCache<P>::DMAClear(GPUVAddr dst_address, u64 amount, u32 value) {
|
||||
ClearDownload(*cpu_dst_address, size);
|
||||
gpu_modified_ranges.Subtract(*cpu_dst_address, size);
|
||||
|
||||
const BufferId buffer = FindBuffer(*cpu_dst_address, static_cast<u32>(size));
|
||||
const BufferId buffer = FindBuffer(*cpu_dst_address, static_cast<u32>(size), false);
|
||||
Buffer& dest_buffer = slot_buffers[buffer];
|
||||
const u32 offset = dest_buffer.Offset(*cpu_dst_address);
|
||||
runtime.ClearBuffer(dest_buffer, offset, size, value);
|
||||
@@ -287,7 +294,7 @@ std::pair<typename P::Buffer*, u32> BufferCache<P>::ObtainBuffer(GPUVAddr gpu_ad
|
||||
template <class P>
|
||||
std::pair<typename P::Buffer*, u32> BufferCache<P>::ObtainCPUBuffer(
|
||||
DAddr device_addr, u32 size, ObtainBufferSynchronize sync_info, ObtainBufferOperation post_op) {
|
||||
const BufferId buffer_id = FindBuffer(device_addr, size);
|
||||
const BufferId buffer_id = FindBuffer(device_addr, size, false);
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
|
||||
// synchronize op
|
||||
@@ -998,11 +1005,85 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
|
||||
channel_state->fast_bound_uniform_buffers[stage] &= ~(1u << binding_index);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool) {
|
||||
binding.segment_first = 0;
|
||||
binding.segment_count = 0;
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
|
||||
if (binding.gpu_addr == 0 || binding.size == 0) {
|
||||
return;
|
||||
}
|
||||
if (is_written && !runtime.PrefersSparseSources()) {
|
||||
return;
|
||||
}
|
||||
const VirtualSegments* found =
|
||||
virtual_ranges.Query(*gpu_memory, binding.gpu_addr, binding.size);
|
||||
if (!found || found->size() < 2) {
|
||||
return;
|
||||
}
|
||||
const VirtualSegments segments = *found;
|
||||
const u32 first = static_cast<u32>(pool.size());
|
||||
const bool prefer_sparse = runtime.PrefersSparseSources();
|
||||
for (const VirtualSegment& segment : segments) {
|
||||
const BufferId buffer_id =
|
||||
FindBuffer(segment.device_addr, segment.size, prefer_sparse);
|
||||
if (!buffer_id) {
|
||||
pool.resize(first);
|
||||
return;
|
||||
}
|
||||
pool.push_back(MultiRangeSegment{
|
||||
.buffer_id = buffer_id,
|
||||
.device_addr = segment.device_addr,
|
||||
.size = segment.size,
|
||||
});
|
||||
}
|
||||
binding.segment_first = first;
|
||||
binding.segment_count = static_cast<u32>(segments.size());
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::BindMultiRangeStorage(const Binding& binding, bool is_written,
|
||||
std::span<const MultiRangeSegment> pool) {
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
|
||||
if (binding.segment_count < 2) {
|
||||
return false;
|
||||
}
|
||||
if (binding.segment_first + binding.segment_count > pool.size()) {
|
||||
return false;
|
||||
}
|
||||
const u64 key = (static_cast<u64>(gpu_memory->GetID()) << 48) ^ binding.gpu_addr;
|
||||
runtime.ResetMultiRange();
|
||||
for (u32 index = 0; index < binding.segment_count; ++index) {
|
||||
const MultiRangeSegment& segment = pool[binding.segment_first + index];
|
||||
Buffer& buffer = slot_buffers[segment.buffer_id];
|
||||
TouchBuffer(buffer, segment.buffer_id);
|
||||
if (SynchronizeBuffer(buffer, segment.device_addr, segment.size)) {
|
||||
runtime.InvalidateMultiRange(key);
|
||||
}
|
||||
const u32 offset = buffer.Offset(segment.device_addr);
|
||||
buffer.MarkUsage(offset, segment.size);
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(segment.buffer_id, segment.device_addr, segment.size);
|
||||
}
|
||||
runtime.PushMultiRangeSource(buffer, offset, segment.size);
|
||||
}
|
||||
return runtime.BindMultiRangeStorageBuffer(key, is_written);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
u32 binding_index = 0;
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
const Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
if (BindMultiRangeStorage(binding, is_written, graphics_segments)) {
|
||||
return;
|
||||
}
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
@@ -1010,7 +1091,6 @@ void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
|
||||
const u32 offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
@@ -1139,6 +1219,11 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
|
||||
u32 binding_index = 0;
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
const Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
if (BindMultiRangeStorage(binding, is_written, compute_segments)) {
|
||||
return;
|
||||
}
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
@@ -1146,8 +1231,6 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
|
||||
|
||||
const u32 offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
@@ -1193,6 +1276,7 @@ void BufferCache<P>::BindHostComputeTextureBuffers() {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) {
|
||||
graphics_segments.clear();
|
||||
BufferOperations([&]() {
|
||||
if (is_indexed) {
|
||||
UpdateIndexBuffer();
|
||||
@@ -1212,6 +1296,7 @@ void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::DoUpdateComputeBuffers() {
|
||||
compute_segments.clear();
|
||||
BufferOperations([&]() {
|
||||
UpdateComputeUniformBuffers();
|
||||
UpdateComputeStorageBuffers();
|
||||
@@ -1234,11 +1319,11 @@ void BufferCache<P>::UpdateIndexBuffer() {
|
||||
auto inline_index_size = static_cast<u32>(draw_state.inline_index_draw_indexes.size());
|
||||
u32 buffer_size = Common::AlignUp(inline_index_size, CACHING_PAGESIZE);
|
||||
if (inline_buffer_id == NULL_BUFFER_ID) [[unlikely]] {
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size);
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size, false);
|
||||
}
|
||||
if (slot_buffers[inline_buffer_id].SizeBytes() < buffer_size) [[unlikely]] {
|
||||
slot_buffers.erase(inline_buffer_id);
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size);
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size, false);
|
||||
}
|
||||
channel_state->index_buffer = Binding{
|
||||
.device_addr = 0,
|
||||
@@ -1261,7 +1346,7 @@ void BufferCache<P>::UpdateIndexBuffer() {
|
||||
channel_state->index_buffer = Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = size,
|
||||
.buffer_id = FindBuffer(*device_addr, size),
|
||||
.buffer_id = FindBuffer(*device_addr, size, false),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1298,7 +1383,7 @@ void BufferCache<P>::UpdateVertexBuffer(u32 index) {
|
||||
if (!gpu_memory->IsWithinGPUAddressRange(gpu_addr_end) || size >= 64_MiB) {
|
||||
size = static_cast<u32>(gpu_memory->MaxContinuousRange(gpu_addr_begin, size));
|
||||
}
|
||||
const BufferId buffer_id = FindBuffer(*device_addr, size);
|
||||
const BufferId buffer_id = FindBuffer(*device_addr, size, false);
|
||||
const Binding binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = size,
|
||||
@@ -1319,7 +1404,7 @@ void BufferCache<P>::UpdateDrawIndirect() {
|
||||
binding = Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = static_cast<u32>(size),
|
||||
.buffer_id = FindBuffer(*device_addr, static_cast<u32>(size)),
|
||||
.buffer_id = FindBuffer(*device_addr, static_cast<u32>(size), false),
|
||||
};
|
||||
};
|
||||
if (current_draw_indirect->include_count) {
|
||||
@@ -1343,7 +1428,7 @@ void BufferCache<P>::UpdateUniformBuffers(size_t stage) {
|
||||
channel_state->dirty_uniform_buffers[stage] |= 1U << index;
|
||||
}
|
||||
// Resolve buffer
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1352,8 +1437,10 @@ void BufferCache<P>::UpdateStorageBuffers(size_t stage) {
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
binding.buffer_id = buffer_id;
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, is_written, graphics_segments);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1361,7 +1448,7 @@ template <class P>
|
||||
void BufferCache<P>::UpdateTextureBuffers(size_t stage) {
|
||||
ForEachEnabledBit(channel_state->enabled_texture_buffers[stage], [&](u32 index) {
|
||||
Binding& binding = channel_state->texture_buffers[stage][index];
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1385,7 +1472,7 @@ void BufferCache<P>::UpdateTransformFeedbackBuffer(u32 index) {
|
||||
channel_state->transform_feedback_buffers[index] = NULL_BINDING;
|
||||
return;
|
||||
}
|
||||
const BufferId buffer_id = FindBuffer(*device_addr, size);
|
||||
const BufferId buffer_id = FindBuffer(*device_addr, size, false);
|
||||
channel_state->transform_feedback_buffers[index] = Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = size,
|
||||
@@ -1407,7 +1494,7 @@ void BufferCache<P>::UpdateComputeUniformBuffers() {
|
||||
binding.size = cbuf.size;
|
||||
}
|
||||
}
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1416,7 +1503,10 @@ void BufferCache<P>::UpdateComputeStorageBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, is_written, compute_segments);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1424,7 +1514,7 @@ template <class P>
|
||||
void BufferCache<P>::UpdateComputeTextureBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_texture_buffers, [&](u32 index) {
|
||||
Binding& binding = channel_state->compute_texture_buffers[index];
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1440,7 +1530,7 @@ void BufferCache<P>::MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u3
|
||||
}
|
||||
|
||||
template <class P>
|
||||
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
|
||||
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible) {
|
||||
if (device_addr == 0) {
|
||||
return NULL_BUFFER_ID;
|
||||
}
|
||||
@@ -1450,10 +1540,18 @@ BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
WaitForGpuFenceIfNeeded(buffer);
|
||||
if (buffer.IsInBounds(device_addr, size)) {
|
||||
return buffer_id;
|
||||
bool usable = true;
|
||||
if constexpr (requires { buffer.IsSparseCompatible(); }) {
|
||||
if (sparse_compatible && !buffer.IsSparseCompatible()) {
|
||||
usable = false;
|
||||
}
|
||||
}
|
||||
if (usable) {
|
||||
return buffer_id;
|
||||
}
|
||||
}
|
||||
}
|
||||
return CreateBuffer(device_addr, size);
|
||||
return CreateBuffer(device_addr, size, sparse_compatible);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -1575,13 +1673,15 @@ void BufferCache<P>::JoinOverlap(BufferId new_buffer_id, BufferId overlap_id,
|
||||
}
|
||||
|
||||
template <class P>
|
||||
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size) {
|
||||
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||
bool sparse_compatible) {
|
||||
DAddr device_addr_end = Common::AlignUp(device_addr + wanted_size, CACHING_PAGESIZE);
|
||||
device_addr = Common::AlignDown(device_addr, CACHING_PAGESIZE);
|
||||
wanted_size = static_cast<u32>(device_addr_end - device_addr);
|
||||
const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size);
|
||||
const u32 size = static_cast<u32>(overlap.end - overlap.begin);
|
||||
const BufferId new_buffer_id = slot_buffers.insert(runtime, overlap.begin, size);
|
||||
const BufferId new_buffer_id =
|
||||
slot_buffers.insert(runtime, overlap.begin, size, sparse_compatible);
|
||||
auto& new_buffer = slot_buffers[new_buffer_id];
|
||||
const size_t size_bytes = new_buffer.SizeBytes();
|
||||
runtime.ClearBuffer(new_buffer, 0, size_bytes, 0);
|
||||
@@ -1745,7 +1845,7 @@ void BufferCache<P>::InlineMemoryImplementation(DAddr dest_address, size_t copy_
|
||||
ClearDownload(dest_address, copy_size);
|
||||
gpu_modified_ranges.Subtract(dest_address, copy_size);
|
||||
|
||||
BufferId buffer_id = FindBuffer(dest_address, static_cast<u32>(copy_size));
|
||||
BufferId buffer_id = FindBuffer(dest_address, static_cast<u32>(copy_size), false);
|
||||
auto& buffer = slot_buffers[buffer_id];
|
||||
SynchronizeBuffer(buffer, dest_address, static_cast<u32>(copy_size));
|
||||
|
||||
@@ -1831,6 +1931,9 @@ void BufferCache<P>::DownloadBufferMemory(Buffer& buffer, DAddr device_addr, u64
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
if constexpr (requires { runtime.OnBufferDeleted(slot_buffers[buffer_id]); }) {
|
||||
runtime.OnBufferDeleted(slot_buffers[buffer_id]);
|
||||
}
|
||||
bool dirty_index{false};
|
||||
boost::container::small_vector<u64, NUM_VERTEX_BUFFERS> dirty_vertex_buffers;
|
||||
const auto scalar_replace = [buffer_id](Binding& binding) {
|
||||
@@ -1934,9 +2037,14 @@ Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
// The end address used for size calculation does not need to be aligned
|
||||
const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE);
|
||||
|
||||
u32 binding_size = static_cast<u32>(cpu_end - *aligned_device_addr);
|
||||
if (is_written) {
|
||||
binding_size = aligned_size;
|
||||
}
|
||||
const Binding binding{
|
||||
.device_addr = *aligned_device_addr,
|
||||
.size = is_written ? aligned_size : static_cast<u32>(cpu_end - *aligned_device_addr),
|
||||
.gpu_addr = aligned_gpu_addr,
|
||||
.size = binding_size,
|
||||
.buffer_id = BufferId{},
|
||||
};
|
||||
return binding;
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
#include "common/settings.h"
|
||||
#include "common/slot_vector.h"
|
||||
#include "video_core/buffer_cache/buffer_base.h"
|
||||
#include "video_core/buffer_cache/virtual_range_cache.h"
|
||||
#include "video_core/control/channel_state_cache.h"
|
||||
#include "video_core/delayed_destruction_ring.h"
|
||||
#include "video_core/dirty_flags.h"
|
||||
@@ -81,8 +82,17 @@ static constexpr u32 DEFAULT_SKIP_CACHE_SIZE = static_cast<u32>(4_KiB);
|
||||
|
||||
struct Binding {
|
||||
DAddr device_addr{};
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 size{};
|
||||
BufferId buffer_id;
|
||||
u32 segment_first{};
|
||||
u32 segment_count{};
|
||||
};
|
||||
|
||||
struct MultiRangeSegment {
|
||||
BufferId buffer_id;
|
||||
DAddr device_addr{};
|
||||
u32 size{};
|
||||
};
|
||||
|
||||
struct TextureBufferBinding : Binding {
|
||||
@@ -215,6 +225,14 @@ public:
|
||||
|
||||
void TickFrame();
|
||||
|
||||
bool BindMultiRangeStorage(const Binding& binding, bool is_written,
|
||||
std::span<const MultiRangeSegment> pool);
|
||||
|
||||
void ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool);
|
||||
|
||||
void UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size);
|
||||
|
||||
void WriteMemory(DAddr device_addr, u64 size);
|
||||
|
||||
void CachedWriteMemory(DAddr device_addr, u64 size);
|
||||
@@ -414,7 +432,7 @@ private:
|
||||
|
||||
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size);
|
||||
|
||||
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size);
|
||||
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible);
|
||||
|
||||
void WaitForGpuFenceIfNeeded(Buffer& buffer);
|
||||
|
||||
@@ -422,7 +440,8 @@ private:
|
||||
|
||||
void JoinOverlap(BufferId new_buffer_id, BufferId overlap_id, bool accumulate_stream_score);
|
||||
|
||||
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size);
|
||||
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||
bool sparse_compatible);
|
||||
|
||||
void Register(BufferId buffer_id);
|
||||
|
||||
@@ -513,6 +532,9 @@ private:
|
||||
using TickType = u64;
|
||||
};
|
||||
Common::LeastRecentlyUsedCache<LRUItemParams> lru_cache;
|
||||
VirtualRangeCache virtual_ranges;
|
||||
std::vector<MultiRangeSegment> graphics_segments;
|
||||
std::vector<MultiRangeSegment> compute_segments;
|
||||
u64 frame_tick = 0;
|
||||
u64 total_used_memory = 0;
|
||||
u64 minimum_memory = 0;
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <limits>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "common/common_types.h"
|
||||
#include "common/container/unordered_map.h"
|
||||
#include "video_core/memory_manager.h"
|
||||
|
||||
namespace VideoCommon {
|
||||
|
||||
struct VirtualSegment {
|
||||
GPUVAddr gpu_addr;
|
||||
DAddr device_addr;
|
||||
u32 size;
|
||||
};
|
||||
|
||||
using VirtualSegments = boost::container::small_vector<VirtualSegment, 8>;
|
||||
|
||||
class VirtualRangeCache {
|
||||
public:
|
||||
static constexpr size_t MAX_ENTRIES = 8192;
|
||||
static constexpr size_t MAX_DEFERRED = 4096;
|
||||
|
||||
const VirtualSegments* Query(Tegra::MemoryManager& memory, GPUVAddr gpu_addr, u32 size) {
|
||||
if (has_deferred.load(std::memory_order_acquire)) {
|
||||
ApplyDeferred();
|
||||
}
|
||||
if (entries.size() > MAX_ENTRIES) {
|
||||
entries.clear();
|
||||
}
|
||||
const size_t as_id = memory.GetID();
|
||||
const u64 key = MakeKey(as_id, gpu_addr);
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end() && it->second.as_id == as_id &&
|
||||
it->second.gpu_addr == gpu_addr && it->second.size == size) {
|
||||
return &it->second.segments;
|
||||
}
|
||||
Entry entry{};
|
||||
entry.as_id = as_id;
|
||||
entry.gpu_addr = gpu_addr;
|
||||
entry.size = size;
|
||||
const auto ranges = memory.GetSubmappedRange(gpu_addr, size);
|
||||
GPUVAddr expected = gpu_addr;
|
||||
bool contiguous = true;
|
||||
for (const auto& [range_addr, range_size] : ranges) {
|
||||
if (range_addr != expected || range_size == 0) {
|
||||
contiguous = false;
|
||||
break;
|
||||
}
|
||||
const std::optional<DAddr> device_addr = memory.GpuToCpuAddress(range_addr);
|
||||
if (!device_addr || *device_addr == 0) {
|
||||
contiguous = false;
|
||||
break;
|
||||
}
|
||||
if (range_size > static_cast<size_t>((std::numeric_limits<u32>::max)())) {
|
||||
contiguous = false;
|
||||
break;
|
||||
}
|
||||
entry.segments.push_back(VirtualSegment{
|
||||
.gpu_addr = range_addr,
|
||||
.device_addr = *device_addr,
|
||||
.size = static_cast<u32>(range_size),
|
||||
});
|
||||
expected += range_size;
|
||||
}
|
||||
if (!contiguous || expected != gpu_addr + size) {
|
||||
entry.segments.clear();
|
||||
}
|
||||
const auto result = entries.insert_or_assign(key, std::move(entry));
|
||||
return &result.first->second.segments;
|
||||
}
|
||||
|
||||
void Unmap(size_t as_id, GPUVAddr gpu_addr, u64 size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
}
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
if (!deferred.empty()) {
|
||||
DeferredUnmap& last = deferred.back();
|
||||
if (last.as_id == as_id && last.gpu_addr + last.size == gpu_addr) {
|
||||
last.size += size;
|
||||
has_deferred.store(true, std::memory_order_release);
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (deferred.size() >= MAX_DEFERRED) {
|
||||
deferred.clear();
|
||||
deferred_overflow = true;
|
||||
} else {
|
||||
deferred.push_back(DeferredUnmap{
|
||||
.as_id = as_id,
|
||||
.gpu_addr = gpu_addr,
|
||||
.size = size,
|
||||
});
|
||||
}
|
||||
}
|
||||
has_deferred.store(true, std::memory_order_release);
|
||||
}
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
VirtualSegments segments;
|
||||
size_t as_id{};
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 size{};
|
||||
};
|
||||
|
||||
struct DeferredUnmap {
|
||||
size_t as_id;
|
||||
GPUVAddr gpu_addr;
|
||||
u64 size;
|
||||
};
|
||||
|
||||
static u64 MakeKey(size_t as_id, GPUVAddr gpu_addr) {
|
||||
return (static_cast<u64>(as_id) << 48) ^ gpu_addr;
|
||||
}
|
||||
|
||||
void ApplyDeferred() {
|
||||
std::vector<DeferredUnmap> pending;
|
||||
bool overflow = false;
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
has_deferred.store(false, std::memory_order_release);
|
||||
pending.swap(deferred);
|
||||
overflow = deferred_overflow;
|
||||
deferred_overflow = false;
|
||||
}
|
||||
if (overflow) {
|
||||
entries.clear();
|
||||
return;
|
||||
}
|
||||
if (pending.empty() || entries.empty()) {
|
||||
return;
|
||||
}
|
||||
for (auto it = entries.begin(); it != entries.end();) {
|
||||
const Entry& entry = it->second;
|
||||
const GPUVAddr entry_end = entry.gpu_addr + entry.size;
|
||||
bool overlaps = false;
|
||||
for (const DeferredUnmap& unmap : pending) {
|
||||
if (unmap.as_id != entry.as_id) {
|
||||
continue;
|
||||
}
|
||||
if (entry.gpu_addr < unmap.gpu_addr + unmap.size && unmap.gpu_addr < entry_end) {
|
||||
overlaps = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (overlaps) {
|
||||
it = entries.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
::Common::unordered_map<u64, Entry> entries;
|
||||
std::vector<DeferredUnmap> deferred;
|
||||
std::mutex deferred_mutex;
|
||||
std::atomic<bool> has_deferred{false};
|
||||
bool deferred_overflow{};
|
||||
};
|
||||
|
||||
} // namespace VideoCommon
|
||||
@@ -52,7 +52,7 @@ constexpr std::array PROGRAM_LUT{
|
||||
Buffer::Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params)
|
||||
: VideoCommon::BufferBase(null_params) {}
|
||||
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_, bool)
|
||||
: VideoCommon::BufferBase(cpu_addr_, size_bytes_) {
|
||||
buffer.Create();
|
||||
if (runtime.device.HasDebuggingToolAttached()) {
|
||||
|
||||
@@ -23,7 +23,8 @@ class BufferCacheRuntime;
|
||||
|
||||
class Buffer : public VideoCommon::BufferBase {
|
||||
public:
|
||||
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes);
|
||||
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes,
|
||||
bool sparse_compatible);
|
||||
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams);
|
||||
|
||||
void ImmediateUpload(size_t offset, std::span<const u8> data) noexcept;
|
||||
|
||||
@@ -56,7 +56,8 @@ size_t BytesPerIndex(VkIndexType index_type) {
|
||||
}
|
||||
}
|
||||
|
||||
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size) {
|
||||
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size,
|
||||
VkDeviceSize sparse_alignment) {
|
||||
VkBufferUsageFlags flags =
|
||||
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||
VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
|
||||
@@ -82,6 +83,9 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
if (sparse_alignment > 1) {
|
||||
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal, sparse_alignment);
|
||||
}
|
||||
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal);
|
||||
}
|
||||
} // Anonymous namespace
|
||||
@@ -99,10 +103,14 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
|
||||
}
|
||||
}
|
||||
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_,
|
||||
bool sparse_compatible_)
|
||||
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
|
||||
scheduler{&runtime.scheduler},
|
||||
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
|
||||
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes(),
|
||||
runtime.SparseAlignmentFor(sparse_compatible_))},
|
||||
tracker{SizeBytes()} {
|
||||
sparse_compatible = sparse_compatible_;
|
||||
if (runtime.device.HasDebuggingToolAttached()) {
|
||||
buffer.SetObjectNameEXT(fmt::format("Buffer {:#x}", CpuAddr()).c_str());
|
||||
}
|
||||
@@ -348,7 +356,8 @@ BufferCacheRuntime::BufferCacheRuntime(const Device& device_, MemoryAllocator& m
|
||||
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_},
|
||||
staging_pool{staging_pool_}, guest_descriptor_queue{guest_descriptor_queue_},
|
||||
quad_index_pass(device, scheduler, descriptor_pool, staging_pool,
|
||||
compute_pass_descriptor_queue) {
|
||||
compute_pass_descriptor_queue),
|
||||
multi_range_buffers(device_) {
|
||||
const VkDriverIdKHR driver_id = device.GetDriverID();
|
||||
limit_dynamic_storage_buffers = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY ||
|
||||
driver_id == VK_DRIVER_ID_ARM_PROPRIETARY;
|
||||
@@ -536,6 +545,37 @@ void BufferCacheRuntime::ClearBuffer(VkBuffer dest_buffer, u32 offset, size_t si
|
||||
});
|
||||
}
|
||||
|
||||
bool BufferCacheRuntime::BindMultiRangeStorageBuffer(u64 key, bool is_written) {
|
||||
if (multi_range_sources.empty() || multi_range_total == 0) {
|
||||
return false;
|
||||
}
|
||||
const MultiRangeRef ref = multi_range_buffers.Get(device, scheduler, memory_allocator, key,
|
||||
multi_range_sources, multi_range_total);
|
||||
if (ref.handle == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
if (is_written && !ref.sparse) {
|
||||
return false;
|
||||
}
|
||||
if (ref.needs_gather) {
|
||||
PreCopyBarrier();
|
||||
VkDeviceSize dst_offset = 0;
|
||||
for (const MultiRangeSource& source : multi_range_sources) {
|
||||
const std::array<VideoCommon::BufferCopy, 1> copy{VideoCommon::BufferCopy{
|
||||
.src_offset = u64(source.offset),
|
||||
.dst_offset = u64(dst_offset),
|
||||
.size = size_t(source.size),
|
||||
}};
|
||||
CopyBuffer(ref.handle, source.handle, copy, false);
|
||||
dst_offset += source.size;
|
||||
}
|
||||
PostCopyBarrier();
|
||||
multi_range_buffers.MarkGathered(key);
|
||||
}
|
||||
guest_descriptor_queue.AddBuffer(ref.handle, ref.address, 0, ref.size);
|
||||
return true;
|
||||
}
|
||||
|
||||
void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format,
|
||||
u32 base_vertex, u32 num_indices, VkBuffer buffer,
|
||||
u32 offset, [[maybe_unused]] u32 size) {
|
||||
|
||||
@@ -8,11 +8,14 @@
|
||||
|
||||
#include <limits>
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "video_core/buffer_cache/buffer_cache_base.h"
|
||||
#include "video_core/buffer_cache/memory_tracker_base.h"
|
||||
#include "video_core/buffer_cache/usage_tracker.h"
|
||||
#include "video_core/engines/maxwell_3d.h"
|
||||
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
||||
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
|
||||
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
|
||||
#include "video_core/renderer_vulkan/vk_update_descriptor.h"
|
||||
#include "video_core/surface.h"
|
||||
@@ -31,7 +34,8 @@ class BufferCacheRuntime;
|
||||
class Buffer : public VideoCommon::BufferBase {
|
||||
public:
|
||||
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
|
||||
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_);
|
||||
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_,
|
||||
bool sparse_compatible_);
|
||||
|
||||
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
|
||||
|
||||
@@ -43,6 +47,14 @@ public:
|
||||
return device_address;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsSparseCompatible() const noexcept {
|
||||
return sparse_compatible;
|
||||
}
|
||||
|
||||
[[nodiscard]] vk::MemoryLocation Location() const noexcept {
|
||||
return buffer.Location();
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsRegionUsed(u64 offset, u64 size) const noexcept {
|
||||
return tracker.IsUsed(offset, size);
|
||||
}
|
||||
@@ -77,6 +89,7 @@ private:
|
||||
VkDeviceAddress device_address{};
|
||||
u64 last_usage_tick{};
|
||||
bool is_null{};
|
||||
bool sparse_compatible{};
|
||||
};
|
||||
|
||||
class QuadArrayIndexBuffer;
|
||||
@@ -125,7 +138,7 @@ public:
|
||||
|
||||
void PreCopyBarrier();
|
||||
|
||||
void CopyBuffer(VkBuffer src_buffer, VkBuffer dst_buffer,
|
||||
void CopyBuffer(VkBuffer dst_buffer, VkBuffer src_buffer,
|
||||
std::span<const VideoCommon::BufferCopy> copies, bool barrier,
|
||||
bool can_reorder_upload = false);
|
||||
|
||||
@@ -155,6 +168,46 @@ public:
|
||||
return ref.mapped_span;
|
||||
}
|
||||
|
||||
[[nodiscard]] VkDeviceSize SparseAlignmentFor(bool sparse_compatible) const noexcept {
|
||||
if (!sparse_compatible || !multi_range_buffers.use_sparse) {
|
||||
return 0;
|
||||
}
|
||||
return multi_range_buffers.block_size;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool PrefersSparseSources() const noexcept {
|
||||
return multi_range_buffers.use_sparse;
|
||||
}
|
||||
|
||||
void ResetMultiRange() noexcept {
|
||||
multi_range_sources.clear();
|
||||
multi_range_total = 0;
|
||||
}
|
||||
|
||||
void PushMultiRangeSource(const Buffer& buffer, u32 offset, u32 size) {
|
||||
const vk::MemoryLocation location = buffer.Location();
|
||||
multi_range_sources.push_back(MultiRangeSource{
|
||||
.handle = buffer.Handle(),
|
||||
.memory = location.memory,
|
||||
.memory_offset = location.offset,
|
||||
.offset = offset,
|
||||
.size = size,
|
||||
.write_tick = buffer.getWriteTick(),
|
||||
.memory_type = location.memory_type,
|
||||
});
|
||||
multi_range_total += size;
|
||||
}
|
||||
|
||||
bool BindMultiRangeStorageBuffer(u64 key, bool is_written);
|
||||
|
||||
void InvalidateMultiRange(u64 key) {
|
||||
multi_range_buffers.Invalidate(key);
|
||||
}
|
||||
|
||||
void OnBufferDeleted(const Buffer& buffer) {
|
||||
multi_range_buffers.DropOwner(scheduler, buffer.Handle());
|
||||
}
|
||||
|
||||
void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) {
|
||||
BindBuffer(buffer, offset, size);
|
||||
}
|
||||
@@ -208,6 +261,10 @@ private:
|
||||
std::unique_ptr<Uint8Pass> uint8_pass;
|
||||
QuadIndexedPass quad_index_pass;
|
||||
|
||||
MultiRangeBufferCache multi_range_buffers;
|
||||
boost::container::small_vector<MultiRangeSource, 16> multi_range_sources;
|
||||
VkDeviceSize multi_range_total{};
|
||||
|
||||
bool limit_dynamic_storage_buffers = false;
|
||||
u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)();
|
||||
};
|
||||
|
||||
@@ -0,0 +1,338 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <mutex>
|
||||
#include <utility>
|
||||
|
||||
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
|
||||
#include "video_core/renderer_vulkan/vk_scheduler.h"
|
||||
#include "video_core/vulkan_common/vulkan_device.h"
|
||||
|
||||
namespace Vulkan {
|
||||
|
||||
MultiRangeBufferCache::MultiRangeBufferCache(const Device& device) {
|
||||
sparse_usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||
if (device.IsBufferDeviceAddressSupported()) {
|
||||
sparse_usage |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
|
||||
}
|
||||
if (!device.IsSparseBindingSupported()) {
|
||||
return;
|
||||
}
|
||||
u32 memory_type_bits = 0;
|
||||
const VkDeviceSize queried = QueryBlockSize(device, memory_type_bits);
|
||||
if (queried == 0 || memory_type_bits == 0) {
|
||||
return;
|
||||
}
|
||||
block_size = queried;
|
||||
sparse_memory_type_bits = memory_type_bits;
|
||||
use_sparse = true;
|
||||
}
|
||||
|
||||
VkDeviceSize MultiRangeBufferCache::QueryBlockSize(const Device& device,
|
||||
u32& memory_type_bits) const {
|
||||
const VkDevice logical = *device.GetLogical();
|
||||
const auto& dld = device.GetDispatchLoader();
|
||||
const VkBufferCreateInfo probe_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
|
||||
.size = DEFAULT_BLOCK_SIZE,
|
||||
.usage = sparse_usage,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
VkBuffer probe{};
|
||||
if (dld.vkCreateBuffer(logical, &probe_ci, nullptr, &probe) != VK_SUCCESS) {
|
||||
return 0;
|
||||
}
|
||||
const SparseBuffer owned{probe, logical, dld};
|
||||
const VkBufferMemoryRequirementsInfo2 reqs_info{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_REQUIREMENTS_INFO_2,
|
||||
.pNext = nullptr,
|
||||
.buffer = probe,
|
||||
};
|
||||
VkMemoryRequirements2 reqs2{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_REQUIREMENTS_2,
|
||||
.pNext = nullptr,
|
||||
.memoryRequirements = {},
|
||||
};
|
||||
dld.vkGetBufferMemoryRequirements2(logical, &reqs_info, &reqs2);
|
||||
memory_type_bits = reqs2.memoryRequirements.memoryTypeBits;
|
||||
return reqs2.memoryRequirements.alignment;
|
||||
}
|
||||
|
||||
u64 MultiRangeBufferCache::HashSources(std::span<const MultiRangeSource> sources) const {
|
||||
u64 hash = 0xcbf29ce484222325ULL;
|
||||
const auto mix = [&hash](u64 value) {
|
||||
hash ^= value;
|
||||
hash *= 0x100000001b3ULL;
|
||||
};
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
mix(u64(source.handle));
|
||||
mix(u64(source.offset));
|
||||
mix(u64(source.size));
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
u64 MultiRangeBufferCache::HashContent(std::span<const MultiRangeSource> sources) const {
|
||||
u64 hash = 0xcbf29ce484222325ULL;
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
hash ^= source.write_tick;
|
||||
hash *= 0x100000001b3ULL;
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
bool MultiRangeBufferCache::CanBindSparse(std::span<const MultiRangeSource> sources) const {
|
||||
return use_sparse &&
|
||||
std::none_of(sources.begin(), sources.end(),
|
||||
[block = block_size, bits = sparse_memory_type_bits](auto const& e) {
|
||||
const VkDeviceSize memory_offset = e.memory_offset + e.offset;
|
||||
return e.memory == VK_NULL_HANDLE || e.memory_type >= 32 ||
|
||||
((bits >> e.memory_type) & 1) == 0 ||
|
||||
(memory_offset % block) != 0 || (e.size % block) != 0;
|
||||
});
|
||||
}
|
||||
|
||||
SparseBuffer MultiRangeBufferCache::CreateSparse(const Device& device, Scheduler& scheduler,
|
||||
std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total) {
|
||||
const VkDevice logical = *device.GetLogical();
|
||||
const auto& dld = device.GetDispatchLoader();
|
||||
const VkBufferCreateInfo buffer_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
|
||||
.size = total,
|
||||
.usage = sparse_usage,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
VkBuffer raw{};
|
||||
if (dld.vkCreateBuffer(logical, &buffer_ci, nullptr, &raw) != VK_SUCCESS) {
|
||||
return SparseBuffer{};
|
||||
}
|
||||
SparseBuffer handle{raw, logical, dld};
|
||||
std::vector<VkSparseMemoryBind> binds;
|
||||
binds.reserve(sources.size());
|
||||
VkDeviceSize resource_offset = 0;
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
binds.push_back(VkSparseMemoryBind{
|
||||
.resourceOffset = resource_offset,
|
||||
.size = source.size,
|
||||
.memory = source.memory,
|
||||
.memoryOffset = source.memory_offset + source.offset,
|
||||
.flags = 0,
|
||||
});
|
||||
resource_offset += source.size;
|
||||
}
|
||||
const VkSparseBufferMemoryBindInfo buffer_bind{
|
||||
.buffer = raw,
|
||||
.bindCount = static_cast<u32>(binds.size()),
|
||||
.pBinds = binds.data(),
|
||||
};
|
||||
const VkBindSparseInfo bind_info{
|
||||
.sType = VK_STRUCTURE_TYPE_BIND_SPARSE_INFO,
|
||||
.pNext = nullptr,
|
||||
.waitSemaphoreCount = 0,
|
||||
.pWaitSemaphores = nullptr,
|
||||
.bufferBindCount = 1,
|
||||
.pBufferBinds = &buffer_bind,
|
||||
.imageOpaqueBindCount = 0,
|
||||
.pImageOpaqueBinds = nullptr,
|
||||
.imageBindCount = 0,
|
||||
.pImageBinds = nullptr,
|
||||
.signalSemaphoreCount = 0,
|
||||
.pSignalSemaphores = nullptr,
|
||||
};
|
||||
const VkFenceCreateInfo fence_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = 0,
|
||||
};
|
||||
vk::Fence fence = device.GetLogical().CreateFence(fence_ci);
|
||||
VkResult bind_result = VK_ERROR_UNKNOWN;
|
||||
{
|
||||
std::scoped_lock lock{scheduler.submit_mutex};
|
||||
bind_result = device.GetGraphicsQueue().BindSparse(bind_info, *fence);
|
||||
}
|
||||
if (bind_result != VK_SUCCESS) {
|
||||
return SparseBuffer{};
|
||||
}
|
||||
fence.Wait();
|
||||
return handle;
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::RetireEntry(Scheduler& scheduler, Entry& entry) {
|
||||
if (!entry.sparse_handle && !entry.gathered) {
|
||||
return;
|
||||
}
|
||||
if (retired.size() == retired.capacity()) {
|
||||
DrainRetired(scheduler);
|
||||
}
|
||||
if (retired.size() == retired.capacity()) {
|
||||
u64 oldest = retired.front().tick;
|
||||
for (const Retired& item : retired) {
|
||||
if (item.tick < oldest) {
|
||||
oldest = item.tick;
|
||||
}
|
||||
}
|
||||
scheduler.Wait(oldest);
|
||||
DrainRetired(scheduler);
|
||||
}
|
||||
retired.push_back(Retired{
|
||||
.handle = std::move(entry.sparse_handle),
|
||||
.gathered = std::move(entry.gathered),
|
||||
.tick = scheduler.CurrentTick(),
|
||||
});
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::DrainRetired(Scheduler& scheduler) {
|
||||
size_t index = 0;
|
||||
while (index < retired.size()) {
|
||||
if (scheduler.IsFree(retired[index].tick)) {
|
||||
if (index + 1 != retired.size()) {
|
||||
retired[index] = std::move(retired.back());
|
||||
}
|
||||
retired.pop_back();
|
||||
} else {
|
||||
++index;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MultiRangeRef MultiRangeBufferCache::Get(const Device& device, Scheduler& scheduler,
|
||||
MemoryAllocator& memory_allocator, u64 key,
|
||||
std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total) {
|
||||
if (sources.empty() || total == 0) {
|
||||
return MultiRangeRef{};
|
||||
}
|
||||
if (!retired.empty()) {
|
||||
DrainRetired(scheduler);
|
||||
}
|
||||
const u64 geometry = HashSources(sources);
|
||||
const u64 content = HashContent(sources);
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end() && it->second.geometry == geometry && it->second.size == total) {
|
||||
Entry& entry = it->second;
|
||||
if (entry.content != content) {
|
||||
entry.content = content;
|
||||
entry.dirty = true;
|
||||
}
|
||||
MultiRangeRef ref{
|
||||
.handle = *entry.sparse_handle,
|
||||
.address = entry.address,
|
||||
.size = entry.size,
|
||||
.sparse = true,
|
||||
.needs_gather = false,
|
||||
};
|
||||
if (!entry.sparse_handle) {
|
||||
ref.handle = *entry.gathered;
|
||||
ref.sparse = false;
|
||||
ref.needs_gather = entry.dirty;
|
||||
}
|
||||
return ref;
|
||||
}
|
||||
if (it != entries.end()) {
|
||||
RetireEntry(scheduler, it->second);
|
||||
entries.erase(it);
|
||||
}
|
||||
|
||||
Entry entry{};
|
||||
entry.geometry = geometry;
|
||||
entry.content = content;
|
||||
entry.size = total;
|
||||
if (CanBindSparse(sources)) {
|
||||
entry.sparse_handle = CreateSparse(device, scheduler, sources, total);
|
||||
if (entry.sparse_handle) {
|
||||
entry.owners.reserve(sources.size());
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
entry.owners.push_back(source.handle);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!entry.sparse_handle) {
|
||||
VkBufferUsageFlags flags = VK_BUFFER_USAGE_TRANSFER_SRC_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||
if (device.IsBufferDeviceAddressSupported()) {
|
||||
flags |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
|
||||
}
|
||||
const VkBufferCreateInfo gather_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = 0,
|
||||
.size = total,
|
||||
.usage = flags,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
entry.gathered = memory_allocator.CreateBuffer(gather_ci, MemoryUsage::DeviceLocal);
|
||||
entry.dirty = true;
|
||||
}
|
||||
if (device.IsBufferDeviceAddressSupported()) {
|
||||
VkBuffer address_handle = *entry.sparse_handle;
|
||||
if (!entry.sparse_handle) {
|
||||
address_handle = *entry.gathered;
|
||||
}
|
||||
entry.address = device.GetLogical().GetBufferDeviceAddress(address_handle);
|
||||
}
|
||||
|
||||
MultiRangeRef ref{
|
||||
.handle = *entry.sparse_handle,
|
||||
.address = entry.address,
|
||||
.size = entry.size,
|
||||
.sparse = true,
|
||||
.needs_gather = false,
|
||||
};
|
||||
if (!entry.sparse_handle) {
|
||||
ref.handle = *entry.gathered;
|
||||
ref.sparse = false;
|
||||
ref.needs_gather = true;
|
||||
}
|
||||
entries.emplace(key, std::move(entry));
|
||||
return ref;
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::MarkGathered(u64 key) {
|
||||
if (auto const it = entries.find(key); it != entries.end()) {
|
||||
it->second.dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::DropOwner(Scheduler& scheduler, VkBuffer owner) {
|
||||
if (owner == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto it = entries.begin(); it != entries.end();) {
|
||||
Entry& entry = it->second;
|
||||
bool owned = false;
|
||||
for (const VkBuffer handle : entry.owners) {
|
||||
if (handle == owner) {
|
||||
owned = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!owned) {
|
||||
++it;
|
||||
continue;
|
||||
}
|
||||
RetireEntry(scheduler, entry);
|
||||
it = entries.erase(it);
|
||||
}
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::Invalidate(u64 key) {
|
||||
if (auto const it = entries.find(key); it != entries.end()) {
|
||||
it->second.dirty = true;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace Vulkan
|
||||
@@ -0,0 +1,105 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <span>
|
||||
#include <vector>
|
||||
|
||||
#include <boost/container/static_vector.hpp>
|
||||
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
#include "common/container/unordered_map.h"
|
||||
#include "video_core/vulkan_common/vulkan_memory_allocator.h"
|
||||
#include "video_core/vulkan_common/vulkan_wrapper.h"
|
||||
|
||||
namespace Vulkan {
|
||||
|
||||
using SparseBuffer = vk::Handle<VkBuffer, VkDevice, vk::DeviceDispatch>;
|
||||
|
||||
class Device;
|
||||
class Scheduler;
|
||||
|
||||
struct MultiRangeSource {
|
||||
VkBuffer handle{};
|
||||
VkDeviceMemory memory{};
|
||||
VkDeviceSize memory_offset{};
|
||||
VkDeviceSize offset{};
|
||||
VkDeviceSize size{};
|
||||
u64 write_tick{};
|
||||
u32 memory_type{};
|
||||
};
|
||||
|
||||
struct MultiRangeRef {
|
||||
VkBuffer handle{};
|
||||
VkDeviceAddress address{};
|
||||
VkDeviceSize size{};
|
||||
bool sparse{};
|
||||
bool needs_gather{};
|
||||
};
|
||||
|
||||
class MultiRangeBufferCache final {
|
||||
public:
|
||||
static constexpr VkDeviceSize DEFAULT_BLOCK_SIZE = 64 * 1024;
|
||||
static constexpr size_t MAX_RETIRED = 256;
|
||||
|
||||
explicit MultiRangeBufferCache(const Device& device);
|
||||
|
||||
YUZU_NON_COPYABLE(MultiRangeBufferCache);
|
||||
|
||||
[[nodiscard]] MultiRangeRef Get(const Device& device, Scheduler& scheduler,
|
||||
MemoryAllocator& memory_allocator, u64 key,
|
||||
std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total);
|
||||
|
||||
void MarkGathered(u64 key);
|
||||
|
||||
void Invalidate(u64 key);
|
||||
|
||||
void DropOwner(Scheduler& scheduler, VkBuffer owner);
|
||||
|
||||
VkDeviceSize block_size{DEFAULT_BLOCK_SIZE};
|
||||
bool use_sparse{};
|
||||
|
||||
private:
|
||||
struct Retired {
|
||||
SparseBuffer handle;
|
||||
vk::Buffer gathered;
|
||||
u64 tick{};
|
||||
};
|
||||
|
||||
struct Entry {
|
||||
vk::Buffer gathered;
|
||||
SparseBuffer sparse_handle;
|
||||
std::vector<VkBuffer> owners;
|
||||
VkDeviceAddress address{};
|
||||
VkDeviceSize size{};
|
||||
u64 geometry{};
|
||||
u64 content{};
|
||||
bool dirty{true};
|
||||
};
|
||||
|
||||
[[nodiscard]] u64 HashSources(std::span<const MultiRangeSource> sources) const;
|
||||
|
||||
[[nodiscard]] u64 HashContent(std::span<const MultiRangeSource> sources) const;
|
||||
|
||||
[[nodiscard]] bool CanBindSparse(std::span<const MultiRangeSource> sources) const;
|
||||
|
||||
[[nodiscard]] SparseBuffer CreateSparse(const Device& device, Scheduler& scheduler,
|
||||
std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total);
|
||||
|
||||
[[nodiscard]] VkDeviceSize QueryBlockSize(const Device& device, u32& memory_type_bits) const;
|
||||
|
||||
void RetireEntry(Scheduler& scheduler, Entry& entry);
|
||||
|
||||
void DrainRetired(Scheduler& scheduler);
|
||||
|
||||
::Common::unordered_map<u64, Entry> entries;
|
||||
boost::container::static_vector<Retired, MAX_RETIRED> retired;
|
||||
u32 sparse_memory_type_bits{};
|
||||
VkBufferUsageFlags sparse_usage{};
|
||||
};
|
||||
|
||||
} // namespace Vulkan
|
||||
@@ -819,6 +819,7 @@ void RasterizerVulkan::ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) {
|
||||
std::scoped_lock lock{texture_cache.mutex};
|
||||
texture_cache.UnmapGPUMemory(as_id, addr, size);
|
||||
}
|
||||
buffer_cache.UnmapGPUMemory(as_id, addr, size);
|
||||
}
|
||||
|
||||
void RasterizerVulkan::SignalFence(std::function<void()>&& func) {
|
||||
|
||||
@@ -1570,6 +1570,8 @@ void Device::SetupFamilies(VkSurfaceKHR surface) {
|
||||
}
|
||||
if (graphics) {
|
||||
graphics_family = *graphics;
|
||||
graphics_family_sparse_binding =
|
||||
(queue_family_properties[*graphics].queueFlags & VK_QUEUE_SPARSE_BINDING_BIT) != 0;
|
||||
}
|
||||
if (present) {
|
||||
present_family = *present;
|
||||
|
||||
@@ -317,6 +317,10 @@ public:
|
||||
return properties.driver.driverID;
|
||||
}
|
||||
|
||||
bool IsSparseBindingSupported() const {
|
||||
return features.features.sparseBinding && graphics_family_sparse_binding;
|
||||
}
|
||||
|
||||
/// Returns true for tile-based deferred renderers.
|
||||
bool IsTiler() const {
|
||||
switch (GetDriverID()) {
|
||||
@@ -1147,6 +1151,7 @@ private:
|
||||
u32 instance_version{}; ///< Vulkan instance version.
|
||||
u32 graphics_family{}; ///< Main graphics queue family index.
|
||||
u32 present_family{}; ///< Main present queue family index.
|
||||
bool graphics_family_sparse_binding{};
|
||||
|
||||
struct Extensions {
|
||||
#define EXTENSION(prefix, macro_name, var_name) bool var_name{};
|
||||
|
||||
@@ -275,9 +275,63 @@ vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsa
|
||||
const std::span<u8> mapped_data = data ? std::span<u8>{data, ci.size} : std::span<u8>{};
|
||||
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
|
||||
|
||||
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data,
|
||||
is_coherent,
|
||||
device.GetDispatchLoader());
|
||||
const vk::MemoryLocation location{
|
||||
.memory = alloc_info.deviceMemory,
|
||||
.offset = alloc_info.offset,
|
||||
.memory_type = alloc_info.memoryType,
|
||||
};
|
||||
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
|
||||
location, device.GetDispatchLoader());
|
||||
}
|
||||
|
||||
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
|
||||
VkDeviceSize min_alignment) const {
|
||||
if (min_alignment <= 1) {
|
||||
return CreateBuffer(ci, usage);
|
||||
}
|
||||
VkMemoryPropertyFlags anv_flags = 0;
|
||||
if (usage == MemoryUsage::Stream &&
|
||||
device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA) {
|
||||
anv_flags = VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
|
||||
}
|
||||
u32 memory_type_bits = valid_memory_types;
|
||||
if (usage == MemoryUsage::Stream) {
|
||||
memory_type_bits = 0u;
|
||||
}
|
||||
const VmaAllocationCreateInfo alloc_ci = {
|
||||
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
|
||||
.usage = MemoryUsageVma(usage),
|
||||
.requiredFlags = 0,
|
||||
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
|
||||
.memoryTypeBits = memory_type_bits,
|
||||
.pool = VK_NULL_HANDLE,
|
||||
.pUserData = nullptr,
|
||||
.priority = 0.f,
|
||||
};
|
||||
|
||||
VkBuffer handle{};
|
||||
VmaAllocationInfo alloc_info{};
|
||||
VmaAllocation allocation{};
|
||||
VkMemoryPropertyFlags property_flags{};
|
||||
|
||||
vk::Check(vmaCreateBufferWithAlignment(allocator, &ci, &alloc_ci, min_alignment, &handle,
|
||||
&allocation, &alloc_info));
|
||||
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
|
||||
|
||||
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
|
||||
std::span<u8> mapped_data{};
|
||||
if (data) {
|
||||
mapped_data = std::span<u8>{data, ci.size};
|
||||
}
|
||||
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
|
||||
|
||||
const vk::MemoryLocation location{
|
||||
.memory = alloc_info.deviceMemory,
|
||||
.offset = alloc_info.offset,
|
||||
.memory_type = alloc_info.memoryType,
|
||||
};
|
||||
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
|
||||
location, device.GetDispatchLoader());
|
||||
}
|
||||
|
||||
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
|
||||
@@ -107,6 +107,9 @@ namespace Vulkan {
|
||||
|
||||
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const;
|
||||
|
||||
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
|
||||
VkDeviceSize min_alignment) const;
|
||||
|
||||
/**
|
||||
* Commits a memory with the specified requirements.
|
||||
*
|
||||
|
||||
@@ -229,6 +229,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
|
||||
X(vkGetPipelineExecutableStatisticsKHR);
|
||||
X(vkGetSemaphoreCounterValue);
|
||||
X(vkMapMemory);
|
||||
X(vkQueueBindSparse);
|
||||
X(vkQueueSubmit);
|
||||
X(vkQueueSubmit2);
|
||||
X(vkResetFences);
|
||||
|
||||
@@ -345,6 +345,7 @@ struct DeviceDispatch : InstanceDispatch {
|
||||
PFN_vkGetQueryPoolResults vkGetQueryPoolResults{};
|
||||
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
|
||||
PFN_vkMapMemory vkMapMemory{};
|
||||
PFN_vkQueueBindSparse vkQueueBindSparse{};
|
||||
PFN_vkQueueSubmit vkQueueSubmit{};
|
||||
PFN_vkQueueSubmit2 vkQueueSubmit2{};
|
||||
PFN_vkResetFences vkResetFences{};
|
||||
@@ -740,13 +741,20 @@ private:
|
||||
const DeviceDispatch* dld = nullptr;
|
||||
};
|
||||
|
||||
struct MemoryLocation {
|
||||
VkDeviceMemory memory{};
|
||||
VkDeviceSize offset{};
|
||||
u32 memory_type{};
|
||||
};
|
||||
|
||||
class Buffer {
|
||||
public:
|
||||
explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_,
|
||||
VmaAllocation allocation_, std::span<u8> mapped_, bool is_coherent_,
|
||||
const DeviceDispatch& dld_) noexcept
|
||||
MemoryLocation location_, const DeviceDispatch& dld_) noexcept
|
||||
: handle{handle_}, owner{owner_}, allocator{allocator_},
|
||||
allocation{allocation_}, mapped{mapped_}, is_coherent{is_coherent_}, dld{&dld_} {}
|
||||
allocation{allocation_}, mapped{mapped_}, location{location_},
|
||||
is_coherent{is_coherent_}, dld{&dld_} {}
|
||||
Buffer() = default;
|
||||
|
||||
Buffer(const Buffer&) = delete;
|
||||
@@ -754,7 +762,7 @@ public:
|
||||
|
||||
Buffer(Buffer&& rhs) noexcept
|
||||
: handle{std::exchange(rhs.handle, VkBuffer{})}, owner{rhs.owner}, allocator{rhs.allocator},
|
||||
allocation{rhs.allocation}, mapped{rhs.mapped},
|
||||
allocation{rhs.allocation}, mapped{rhs.mapped}, location{rhs.location},
|
||||
is_coherent{rhs.is_coherent}, dld{rhs.dld} {}
|
||||
|
||||
Buffer& operator=(Buffer&& rhs) noexcept {
|
||||
@@ -764,6 +772,7 @@ public:
|
||||
allocator = rhs.allocator;
|
||||
allocation = rhs.allocation;
|
||||
mapped = rhs.mapped;
|
||||
location = rhs.location;
|
||||
is_coherent = rhs.is_coherent;
|
||||
dld = rhs.dld;
|
||||
return *this;
|
||||
@@ -811,6 +820,10 @@ public:
|
||||
|
||||
void SetObjectNameEXT(const char* name) const;
|
||||
|
||||
MemoryLocation Location() const noexcept {
|
||||
return location;
|
||||
}
|
||||
|
||||
private:
|
||||
void Release() const noexcept;
|
||||
|
||||
@@ -819,6 +832,7 @@ private:
|
||||
VmaAllocator allocator = nullptr;
|
||||
VmaAllocation allocation = nullptr;
|
||||
std::span<u8> mapped = {};
|
||||
MemoryLocation location{};
|
||||
bool is_coherent = false;
|
||||
const DeviceDispatch* dld = nullptr;
|
||||
};
|
||||
@@ -843,6 +857,11 @@ public:
|
||||
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
|
||||
}
|
||||
|
||||
VkResult BindSparse(Span<VkBindSparseInfo> bind_infos,
|
||||
VkFence fence = VK_NULL_HANDLE) const noexcept {
|
||||
return dld->vkQueueBindSparse(queue, bind_infos.size(), bind_infos.data(), fence);
|
||||
}
|
||||
|
||||
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
|
||||
return dld->vkQueuePresentKHR(queue, &present_info);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user