mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-09-06 04:06:58 +00:00
Compare commits
3 Commits
master
...
sparse-buffers
| Author | SHA1 | Date | |
|---|---|---|---|
| 49e96351b1 | |||
| 29b2b04a2d | |||
| 76b7a561ba |
@@ -312,10 +312,66 @@ static inline std::optional<ConstBufferAddr> TrackCached(const IR::Value& v, Env
|
|||||||
|
|
||||||
std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environment& env, const HostTranslateInfo& host_info);
|
std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environment& env, const HostTranslateInfo& host_info);
|
||||||
|
|
||||||
|
bool IsSameConstBufferAddr(const ConstBufferAddr& lhs, const ConstBufferAddr& rhs) {
|
||||||
|
return lhs.index == rhs.index && lhs.offset == rhs.offset &&
|
||||||
|
lhs.shift_left == rhs.shift_left && lhs.secondary_index == rhs.secondary_index &&
|
||||||
|
lhs.secondary_offset == rhs.secondary_offset &&
|
||||||
|
lhs.secondary_shift_left == rhs.secondary_shift_left && lhs.count == rhs.count &&
|
||||||
|
lhs.has_secondary == rhs.has_secondary && lhs.dynamic_offset == rhs.dynamic_offset;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::optional<ConstBufferAddr> TrackPhi(const IR::Inst* phi, Environment& env,
|
||||||
|
const HostTranslateInfo& host_info, bool& ambiguous) {
|
||||||
|
std::optional<ConstBufferAddr> agreed;
|
||||||
|
const size_t num_args{phi->NumArgs()};
|
||||||
|
for (size_t index = 0; index < num_args; ++index) {
|
||||||
|
const IR::Value arg{phi->Arg(index).Resolve()};
|
||||||
|
if (arg.IsImmediate()) {
|
||||||
|
ambiguous = true;
|
||||||
|
return std::nullopt;
|
||||||
|
}
|
||||||
|
const IR::Inst* arg_inst{arg.InstRecursive()};
|
||||||
|
if (arg_inst == phi) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (arg_inst->GetOpcode() == IR::Opcode::Phi) {
|
||||||
|
ambiguous = true;
|
||||||
|
return std::nullopt;
|
||||||
|
}
|
||||||
|
const std::optional<ConstBufferAddr> operand{TrackCached(arg, env, host_info)};
|
||||||
|
if (!operand) {
|
||||||
|
ambiguous = true;
|
||||||
|
return std::nullopt;
|
||||||
|
}
|
||||||
|
if (!agreed) {
|
||||||
|
agreed = operand;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!IsSameConstBufferAddr(*agreed, *operand)) {
|
||||||
|
ambiguous = true;
|
||||||
|
return std::nullopt;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!agreed) {
|
||||||
|
ambiguous = true;
|
||||||
|
}
|
||||||
|
return agreed;
|
||||||
|
}
|
||||||
|
|
||||||
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env, const HostTranslateInfo& host_info) {
|
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env, const HostTranslateInfo& host_info) {
|
||||||
return IR::BreadthFirstSearch(value, [&env, &host_info](const IR::Inst* inst) {
|
bool ambiguous = false;
|
||||||
|
const std::optional<ConstBufferAddr> result{IR::BreadthFirstSearch(
|
||||||
|
value, [&env, &host_info, &ambiguous](const IR::Inst* inst)
|
||||||
|
-> std::optional<ConstBufferAddr> {
|
||||||
|
if (inst->GetOpcode() == IR::Opcode::Phi) {
|
||||||
|
return TrackPhi(inst, env, host_info, ambiguous);
|
||||||
|
}
|
||||||
return TryGetConstBuffer(inst, env, host_info);
|
return TryGetConstBuffer(inst, env, host_info);
|
||||||
});
|
})};
|
||||||
|
if (ambiguous) {
|
||||||
|
return std::nullopt;
|
||||||
|
}
|
||||||
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::optional<u32> TryGetConstant(IR::Value& value, Environment& env) {
|
std::optional<u32> TryGetConstant(IR::Value& value, Environment& env) {
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ add_library(video_core STATIC
|
|||||||
buffer_cache/buffer_cache.h
|
buffer_cache/buffer_cache.h
|
||||||
buffer_cache/memory_tracker_base.h
|
buffer_cache/memory_tracker_base.h
|
||||||
buffer_cache/usage_tracker.h
|
buffer_cache/usage_tracker.h
|
||||||
|
buffer_cache/virtual_range_cache.h
|
||||||
buffer_cache/word_manager.h
|
buffer_cache/word_manager.h
|
||||||
cache_types.h
|
cache_types.h
|
||||||
capture.h
|
capture.h
|
||||||
@@ -166,6 +167,8 @@ add_library(video_core STATIC
|
|||||||
renderer_vulkan/vk_fence_manager.h
|
renderer_vulkan/vk_fence_manager.h
|
||||||
renderer_vulkan/vk_graphics_pipeline.cpp
|
renderer_vulkan/vk_graphics_pipeline.cpp
|
||||||
renderer_vulkan/vk_graphics_pipeline.h
|
renderer_vulkan/vk_graphics_pipeline.h
|
||||||
|
renderer_vulkan/vk_multi_range_buffer.cpp
|
||||||
|
renderer_vulkan/vk_multi_range_buffer.h
|
||||||
renderer_vulkan/vk_master_semaphore.cpp
|
renderer_vulkan/vk_master_semaphore.cpp
|
||||||
renderer_vulkan/vk_master_semaphore.h
|
renderer_vulkan/vk_master_semaphore.h
|
||||||
renderer_vulkan/vk_pipeline_cache.cpp
|
renderer_vulkan/vk_pipeline_cache.cpp
|
||||||
|
|||||||
@@ -112,6 +112,11 @@ void BufferCache<P>::TickFrame() {
|
|||||||
async_buffers_death_ring.clear();
|
async_buffers_death_ring.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
template <class P>
|
||||||
|
void BufferCache<P>::UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size) {
|
||||||
|
virtual_ranges.Unmap(as_id, gpu_addr, size);
|
||||||
|
}
|
||||||
|
|
||||||
template <class P>
|
template <class P>
|
||||||
void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) {
|
void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) {
|
||||||
if (memory_tracker.IsRegionGpuModified(device_addr, size)) {
|
if (memory_tracker.IsRegionGpuModified(device_addr, size)) {
|
||||||
@@ -998,11 +1003,56 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
|
|||||||
channel_state->fast_bound_uniform_buffers[stage] &= ~(1u << binding_index);
|
channel_state->fast_bound_uniform_buffers[stage] &= ~(1u << binding_index);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
template <class P>
|
||||||
|
bool BufferCache<P>::BindMultiRangeStorage(const Binding& binding, bool is_written) {
|
||||||
|
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}); }) {
|
||||||
|
if (binding.gpu_addr == 0 || binding.size == 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (is_written && !runtime.PrefersSparseSources()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const VirtualSegments* segments =
|
||||||
|
virtual_ranges.Query(*gpu_memory, binding.gpu_addr, binding.size);
|
||||||
|
if (!segments || segments->size() < 2) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const u64 key = (static_cast<u64>(gpu_memory->GetID()) << 48) ^ binding.gpu_addr;
|
||||||
|
const bool prefer_sparse = runtime.PrefersSparseSources();
|
||||||
|
runtime.ResetMultiRange();
|
||||||
|
for (const VirtualSegment& segment : *segments) {
|
||||||
|
const BufferId buffer_id =
|
||||||
|
FindBuffer(segment.device_addr, segment.size, prefer_sparse);
|
||||||
|
if (!buffer_id) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
Buffer& buffer = slot_buffers[buffer_id];
|
||||||
|
TouchBuffer(buffer, buffer_id);
|
||||||
|
if (SynchronizeBuffer(buffer, segment.device_addr, segment.size)) {
|
||||||
|
runtime.InvalidateMultiRange(key);
|
||||||
|
}
|
||||||
|
const u32 offset = buffer.Offset(segment.device_addr);
|
||||||
|
buffer.MarkUsage(offset, segment.size);
|
||||||
|
if (is_written) {
|
||||||
|
MarkWrittenBuffer(buffer_id, segment.device_addr, segment.size);
|
||||||
|
}
|
||||||
|
runtime.PushMultiRangeSource(buffer, offset, segment.size);
|
||||||
|
}
|
||||||
|
return runtime.BindMultiRangeStorageBuffer(key);
|
||||||
|
} else {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
template <class P>
|
template <class P>
|
||||||
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||||
u32 binding_index = 0;
|
u32 binding_index = 0;
|
||||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||||
const Binding& binding = channel_state->storage_buffers[stage][index];
|
const Binding& binding = channel_state->storage_buffers[stage][index];
|
||||||
|
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||||
|
if (BindMultiRangeStorage(binding, is_written)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||||
TouchBuffer(buffer, binding.buffer_id);
|
TouchBuffer(buffer, binding.buffer_id);
|
||||||
const u32 size = binding.size;
|
const u32 size = binding.size;
|
||||||
@@ -1010,7 +1060,6 @@ void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
|||||||
|
|
||||||
const u32 offset = buffer.Offset(binding.device_addr);
|
const u32 offset = buffer.Offset(binding.device_addr);
|
||||||
buffer.MarkUsage(offset, size);
|
buffer.MarkUsage(offset, size);
|
||||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
|
||||||
|
|
||||||
if (is_written) {
|
if (is_written) {
|
||||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||||
@@ -1139,6 +1188,11 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
|
|||||||
u32 binding_index = 0;
|
u32 binding_index = 0;
|
||||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||||
const Binding& binding = channel_state->compute_storage_buffers[index];
|
const Binding& binding = channel_state->compute_storage_buffers[index];
|
||||||
|
const bool is_written =
|
||||||
|
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||||
|
if (BindMultiRangeStorage(binding, is_written)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||||
TouchBuffer(buffer, binding.buffer_id);
|
TouchBuffer(buffer, binding.buffer_id);
|
||||||
const u32 size = binding.size;
|
const u32 size = binding.size;
|
||||||
@@ -1146,8 +1200,6 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
|
|||||||
|
|
||||||
const u32 offset = buffer.Offset(binding.device_addr);
|
const u32 offset = buffer.Offset(binding.device_addr);
|
||||||
buffer.MarkUsage(offset, size);
|
buffer.MarkUsage(offset, size);
|
||||||
const bool is_written =
|
|
||||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
|
||||||
|
|
||||||
if (is_written) {
|
if (is_written) {
|
||||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||||
@@ -1440,7 +1492,7 @@ void BufferCache<P>::MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u3
|
|||||||
}
|
}
|
||||||
|
|
||||||
template <class P>
|
template <class P>
|
||||||
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
|
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible) {
|
||||||
if (device_addr == 0) {
|
if (device_addr == 0) {
|
||||||
return NULL_BUFFER_ID;
|
return NULL_BUFFER_ID;
|
||||||
}
|
}
|
||||||
@@ -1450,10 +1502,18 @@ BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
|
|||||||
Buffer& buffer = slot_buffers[buffer_id];
|
Buffer& buffer = slot_buffers[buffer_id];
|
||||||
WaitForGpuFenceIfNeeded(buffer);
|
WaitForGpuFenceIfNeeded(buffer);
|
||||||
if (buffer.IsInBounds(device_addr, size)) {
|
if (buffer.IsInBounds(device_addr, size)) {
|
||||||
|
bool usable = true;
|
||||||
|
if constexpr (requires { buffer.IsSparseCompatible(); }) {
|
||||||
|
if (sparse_compatible && !buffer.IsSparseCompatible()) {
|
||||||
|
usable = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (usable) {
|
||||||
return buffer_id;
|
return buffer_id;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return CreateBuffer(device_addr, size);
|
}
|
||||||
|
return CreateBuffer(device_addr, size, sparse_compatible);
|
||||||
}
|
}
|
||||||
|
|
||||||
template <class P>
|
template <class P>
|
||||||
@@ -1575,13 +1635,15 @@ void BufferCache<P>::JoinOverlap(BufferId new_buffer_id, BufferId overlap_id,
|
|||||||
}
|
}
|
||||||
|
|
||||||
template <class P>
|
template <class P>
|
||||||
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size) {
|
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||||
|
bool sparse_compatible) {
|
||||||
DAddr device_addr_end = Common::AlignUp(device_addr + wanted_size, CACHING_PAGESIZE);
|
DAddr device_addr_end = Common::AlignUp(device_addr + wanted_size, CACHING_PAGESIZE);
|
||||||
device_addr = Common::AlignDown(device_addr, CACHING_PAGESIZE);
|
device_addr = Common::AlignDown(device_addr, CACHING_PAGESIZE);
|
||||||
wanted_size = static_cast<u32>(device_addr_end - device_addr);
|
wanted_size = static_cast<u32>(device_addr_end - device_addr);
|
||||||
const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size);
|
const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size);
|
||||||
const u32 size = static_cast<u32>(overlap.end - overlap.begin);
|
const u32 size = static_cast<u32>(overlap.end - overlap.begin);
|
||||||
const BufferId new_buffer_id = slot_buffers.insert(runtime, overlap.begin, size);
|
const BufferId new_buffer_id =
|
||||||
|
slot_buffers.insert(runtime, overlap.begin, size, sparse_compatible);
|
||||||
auto& new_buffer = slot_buffers[new_buffer_id];
|
auto& new_buffer = slot_buffers[new_buffer_id];
|
||||||
const size_t size_bytes = new_buffer.SizeBytes();
|
const size_t size_bytes = new_buffer.SizeBytes();
|
||||||
runtime.ClearBuffer(new_buffer, 0, size_bytes, 0);
|
runtime.ClearBuffer(new_buffer, 0, size_bytes, 0);
|
||||||
@@ -1934,10 +1996,15 @@ Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
|||||||
// The end address used for size calculation does not need to be aligned
|
// The end address used for size calculation does not need to be aligned
|
||||||
const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE);
|
const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE);
|
||||||
|
|
||||||
|
u32 binding_size = static_cast<u32>(cpu_end - *aligned_device_addr);
|
||||||
|
if (is_written) {
|
||||||
|
binding_size = aligned_size;
|
||||||
|
}
|
||||||
const Binding binding{
|
const Binding binding{
|
||||||
.device_addr = *aligned_device_addr,
|
.device_addr = *aligned_device_addr,
|
||||||
.size = is_written ? aligned_size : static_cast<u32>(cpu_end - *aligned_device_addr),
|
.size = binding_size,
|
||||||
.buffer_id = BufferId{},
|
.buffer_id = BufferId{},
|
||||||
|
.gpu_addr = aligned_gpu_addr,
|
||||||
};
|
};
|
||||||
return binding;
|
return binding;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -29,6 +29,7 @@
|
|||||||
#include "common/settings.h"
|
#include "common/settings.h"
|
||||||
#include "common/slot_vector.h"
|
#include "common/slot_vector.h"
|
||||||
#include "video_core/buffer_cache/buffer_base.h"
|
#include "video_core/buffer_cache/buffer_base.h"
|
||||||
|
#include "video_core/buffer_cache/virtual_range_cache.h"
|
||||||
#include "video_core/control/channel_state_cache.h"
|
#include "video_core/control/channel_state_cache.h"
|
||||||
#include "video_core/delayed_destruction_ring.h"
|
#include "video_core/delayed_destruction_ring.h"
|
||||||
#include "video_core/dirty_flags.h"
|
#include "video_core/dirty_flags.h"
|
||||||
@@ -83,6 +84,7 @@ struct Binding {
|
|||||||
DAddr device_addr{};
|
DAddr device_addr{};
|
||||||
u32 size{};
|
u32 size{};
|
||||||
BufferId buffer_id;
|
BufferId buffer_id;
|
||||||
|
GPUVAddr gpu_addr{};
|
||||||
};
|
};
|
||||||
|
|
||||||
struct TextureBufferBinding : Binding {
|
struct TextureBufferBinding : Binding {
|
||||||
@@ -215,6 +217,10 @@ public:
|
|||||||
|
|
||||||
void TickFrame();
|
void TickFrame();
|
||||||
|
|
||||||
|
bool BindMultiRangeStorage(const Binding& binding, bool is_written);
|
||||||
|
|
||||||
|
void UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size);
|
||||||
|
|
||||||
void WriteMemory(DAddr device_addr, u64 size);
|
void WriteMemory(DAddr device_addr, u64 size);
|
||||||
|
|
||||||
void CachedWriteMemory(DAddr device_addr, u64 size);
|
void CachedWriteMemory(DAddr device_addr, u64 size);
|
||||||
@@ -414,7 +420,8 @@ private:
|
|||||||
|
|
||||||
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size);
|
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size);
|
||||||
|
|
||||||
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size);
|
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size,
|
||||||
|
bool sparse_compatible = false);
|
||||||
|
|
||||||
void WaitForGpuFenceIfNeeded(Buffer& buffer);
|
void WaitForGpuFenceIfNeeded(Buffer& buffer);
|
||||||
|
|
||||||
@@ -422,7 +429,8 @@ private:
|
|||||||
|
|
||||||
void JoinOverlap(BufferId new_buffer_id, BufferId overlap_id, bool accumulate_stream_score);
|
void JoinOverlap(BufferId new_buffer_id, BufferId overlap_id, bool accumulate_stream_score);
|
||||||
|
|
||||||
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size);
|
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||||
|
bool sparse_compatible = false);
|
||||||
|
|
||||||
void Register(BufferId buffer_id);
|
void Register(BufferId buffer_id);
|
||||||
|
|
||||||
@@ -514,6 +522,7 @@ private:
|
|||||||
};
|
};
|
||||||
Common::LeastRecentlyUsedCache<LRUItemParams> lru_cache;
|
Common::LeastRecentlyUsedCache<LRUItemParams> lru_cache;
|
||||||
u64 frame_tick = 0;
|
u64 frame_tick = 0;
|
||||||
|
VirtualRangeCache virtual_ranges;
|
||||||
u64 total_used_memory = 0;
|
u64 total_used_memory = 0;
|
||||||
u64 minimum_memory = 0;
|
u64 minimum_memory = 0;
|
||||||
u64 critical_memory = 0;
|
u64 critical_memory = 0;
|
||||||
|
|||||||
@@ -0,0 +1,152 @@
|
|||||||
|
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <atomic>
|
||||||
|
#include <limits>
|
||||||
|
#include <mutex>
|
||||||
|
#include <optional>
|
||||||
|
#include <unordered_map>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
#include <boost/container/small_vector.hpp>
|
||||||
|
|
||||||
|
#include "common/common_types.h"
|
||||||
|
#include "video_core/memory_manager.h"
|
||||||
|
|
||||||
|
namespace VideoCommon {
|
||||||
|
|
||||||
|
struct VirtualSegment {
|
||||||
|
GPUVAddr gpu_addr;
|
||||||
|
DAddr device_addr;
|
||||||
|
u32 size;
|
||||||
|
};
|
||||||
|
|
||||||
|
using VirtualSegments = boost::container::small_vector<VirtualSegment, 8>;
|
||||||
|
|
||||||
|
class VirtualRangeCache {
|
||||||
|
public:
|
||||||
|
const VirtualSegments* Query(Tegra::MemoryManager& memory, GPUVAddr gpu_addr, u32 size) {
|
||||||
|
if (has_deferred.load(std::memory_order_acquire)) {
|
||||||
|
ApplyDeferred();
|
||||||
|
}
|
||||||
|
const size_t as_id = memory.GetID();
|
||||||
|
const u64 key = MakeKey(as_id, gpu_addr);
|
||||||
|
const auto it = entries.find(key);
|
||||||
|
if (it != entries.end() && it->second.as_id == as_id &&
|
||||||
|
it->second.gpu_addr == gpu_addr && it->second.size == size) {
|
||||||
|
return &it->second.segments;
|
||||||
|
}
|
||||||
|
Entry entry;
|
||||||
|
entry.as_id = as_id;
|
||||||
|
entry.gpu_addr = gpu_addr;
|
||||||
|
entry.size = size;
|
||||||
|
const auto ranges = memory.GetSubmappedRange(gpu_addr, size);
|
||||||
|
for (const auto& [range_addr, range_size] : ranges) {
|
||||||
|
const std::optional<DAddr> device_addr = memory.GpuToCpuAddress(range_addr);
|
||||||
|
if (!device_addr) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
u32 segment_size = (std::numeric_limits<u32>::max)();
|
||||||
|
if (range_size < static_cast<size_t>(segment_size)) {
|
||||||
|
segment_size = static_cast<u32>(range_size);
|
||||||
|
}
|
||||||
|
entry.segments.push_back(VirtualSegment{
|
||||||
|
.gpu_addr = range_addr,
|
||||||
|
.device_addr = *device_addr,
|
||||||
|
.size = segment_size,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
const auto result = entries.insert_or_assign(key, std::move(entry));
|
||||||
|
return &result.first->second.segments;
|
||||||
|
}
|
||||||
|
|
||||||
|
void Unmap(size_t as_id, GPUVAddr gpu_addr, u64 size) {
|
||||||
|
if (size == 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
{
|
||||||
|
std::scoped_lock lock{deferred_mutex};
|
||||||
|
if (!deferred.empty()) {
|
||||||
|
DeferredUnmap& last = deferred.back();
|
||||||
|
if (last.as_id == as_id && last.gpu_addr + last.size == gpu_addr) {
|
||||||
|
last.size += size;
|
||||||
|
has_deferred.store(true, std::memory_order_release);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
deferred.push_back(DeferredUnmap{
|
||||||
|
.as_id = as_id,
|
||||||
|
.gpu_addr = gpu_addr,
|
||||||
|
.size = size,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
has_deferred.store(true, std::memory_order_release);
|
||||||
|
}
|
||||||
|
|
||||||
|
void Clear() {
|
||||||
|
{
|
||||||
|
std::scoped_lock lock{deferred_mutex};
|
||||||
|
deferred.clear();
|
||||||
|
}
|
||||||
|
has_deferred.store(false, std::memory_order_release);
|
||||||
|
entries.clear();
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
struct Entry {
|
||||||
|
size_t as_id{};
|
||||||
|
GPUVAddr gpu_addr{};
|
||||||
|
u32 size{};
|
||||||
|
VirtualSegments segments;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct DeferredUnmap {
|
||||||
|
size_t as_id;
|
||||||
|
GPUVAddr gpu_addr;
|
||||||
|
u64 size;
|
||||||
|
};
|
||||||
|
|
||||||
|
static u64 MakeKey(size_t as_id, GPUVAddr gpu_addr) {
|
||||||
|
return (static_cast<u64>(as_id) << 48) ^ gpu_addr;
|
||||||
|
}
|
||||||
|
|
||||||
|
void ApplyDeferred() {
|
||||||
|
std::vector<DeferredUnmap> pending;
|
||||||
|
{
|
||||||
|
std::scoped_lock lock{deferred_mutex};
|
||||||
|
has_deferred.store(false, std::memory_order_release);
|
||||||
|
pending.swap(deferred);
|
||||||
|
}
|
||||||
|
if (pending.empty() || entries.empty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
for (auto it = entries.begin(); it != entries.end();) {
|
||||||
|
const Entry& entry = it->second;
|
||||||
|
const GPUVAddr entry_end = entry.gpu_addr + entry.size;
|
||||||
|
bool overlaps = false;
|
||||||
|
for (const DeferredUnmap& unmap : pending) {
|
||||||
|
if (unmap.as_id != entry.as_id) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (entry.gpu_addr < unmap.gpu_addr + unmap.size && unmap.gpu_addr < entry_end) {
|
||||||
|
overlaps = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (overlaps) {
|
||||||
|
it = entries.erase(it);
|
||||||
|
} else {
|
||||||
|
++it;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
std::unordered_map<u64, Entry> entries;
|
||||||
|
std::vector<DeferredUnmap> deferred;
|
||||||
|
std::mutex deferred_mutex;
|
||||||
|
std::atomic<bool> has_deferred{false};
|
||||||
|
};
|
||||||
|
|
||||||
|
} // namespace VideoCommon
|
||||||
@@ -52,7 +52,7 @@ constexpr std::array PROGRAM_LUT{
|
|||||||
Buffer::Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params)
|
Buffer::Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params)
|
||||||
: VideoCommon::BufferBase(null_params) {}
|
: VideoCommon::BufferBase(null_params) {}
|
||||||
|
|
||||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
|
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_, bool)
|
||||||
: VideoCommon::BufferBase(cpu_addr_, size_bytes_) {
|
: VideoCommon::BufferBase(cpu_addr_, size_bytes_) {
|
||||||
buffer.Create();
|
buffer.Create();
|
||||||
if (runtime.device.HasDebuggingToolAttached()) {
|
if (runtime.device.HasDebuggingToolAttached()) {
|
||||||
|
|||||||
@@ -23,7 +23,8 @@ class BufferCacheRuntime;
|
|||||||
|
|
||||||
class Buffer : public VideoCommon::BufferBase {
|
class Buffer : public VideoCommon::BufferBase {
|
||||||
public:
|
public:
|
||||||
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes);
|
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes,
|
||||||
|
bool sparse_compatible = false);
|
||||||
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams);
|
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams);
|
||||||
|
|
||||||
void ImmediateUpload(size_t offset, std::span<const u8> data) noexcept;
|
void ImmediateUpload(size_t offset, std::span<const u8> data) noexcept;
|
||||||
|
|||||||
@@ -56,7 +56,8 @@ size_t BytesPerIndex(VkIndexType index_type) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size) {
|
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size,
|
||||||
|
VkDeviceSize sparse_alignment) {
|
||||||
VkBufferUsageFlags flags =
|
VkBufferUsageFlags flags =
|
||||||
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||||
VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
|
VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
|
||||||
@@ -82,6 +83,9 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
|
|||||||
.queueFamilyIndexCount = 0,
|
.queueFamilyIndexCount = 0,
|
||||||
.pQueueFamilyIndices = nullptr,
|
.pQueueFamilyIndices = nullptr,
|
||||||
};
|
};
|
||||||
|
if (sparse_alignment > 1) {
|
||||||
|
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal, sparse_alignment);
|
||||||
|
}
|
||||||
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal);
|
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal);
|
||||||
}
|
}
|
||||||
} // Anonymous namespace
|
} // Anonymous namespace
|
||||||
@@ -99,10 +103,14 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
|
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_,
|
||||||
|
bool sparse_compatible_)
|
||||||
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
|
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
|
||||||
scheduler{&runtime.scheduler},
|
scheduler{&runtime.scheduler},
|
||||||
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
|
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes(),
|
||||||
|
runtime.SparseAlignmentFor(sparse_compatible_))},
|
||||||
|
tracker{SizeBytes()} {
|
||||||
|
sparse_compatible = sparse_compatible_;
|
||||||
if (runtime.device.HasDebuggingToolAttached()) {
|
if (runtime.device.HasDebuggingToolAttached()) {
|
||||||
buffer.SetObjectNameEXT(fmt::format("Buffer {:#x}", CpuAddr()).c_str());
|
buffer.SetObjectNameEXT(fmt::format("Buffer {:#x}", CpuAddr()).c_str());
|
||||||
}
|
}
|
||||||
@@ -348,7 +356,8 @@ BufferCacheRuntime::BufferCacheRuntime(const Device& device_, MemoryAllocator& m
|
|||||||
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_},
|
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_},
|
||||||
staging_pool{staging_pool_}, guest_descriptor_queue{guest_descriptor_queue_},
|
staging_pool{staging_pool_}, guest_descriptor_queue{guest_descriptor_queue_},
|
||||||
quad_index_pass(device, scheduler, descriptor_pool, staging_pool,
|
quad_index_pass(device, scheduler, descriptor_pool, staging_pool,
|
||||||
compute_pass_descriptor_queue) {
|
compute_pass_descriptor_queue),
|
||||||
|
multi_range_buffers(device_, memory_allocator_, scheduler_) {
|
||||||
const VkDriverIdKHR driver_id = device.GetDriverID();
|
const VkDriverIdKHR driver_id = device.GetDriverID();
|
||||||
limit_dynamic_storage_buffers = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY ||
|
limit_dynamic_storage_buffers = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY ||
|
||||||
driver_id == VK_DRIVER_ID_ARM_PROPRIETARY;
|
driver_id == VK_DRIVER_ID_ARM_PROPRIETARY;
|
||||||
@@ -536,6 +545,33 @@ void BufferCacheRuntime::ClearBuffer(VkBuffer dest_buffer, u32 offset, size_t si
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool BufferCacheRuntime::BindMultiRangeStorageBuffer(u64 key) {
|
||||||
|
if (multi_range_sources.empty() || multi_range_total == 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const MultiRangeRef ref = multi_range_buffers.Get(key, multi_range_sources, multi_range_total);
|
||||||
|
if (ref.handle == VK_NULL_HANDLE) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (ref.needs_gather) {
|
||||||
|
PreCopyBarrier();
|
||||||
|
VkDeviceSize dst_offset = 0;
|
||||||
|
for (const MultiRangeSource& source : multi_range_sources) {
|
||||||
|
const std::array<VideoCommon::BufferCopy, 1> copy{VideoCommon::BufferCopy{
|
||||||
|
.src_offset = static_cast<u64>(source.offset),
|
||||||
|
.dst_offset = static_cast<u64>(dst_offset),
|
||||||
|
.size = static_cast<size_t>(source.size),
|
||||||
|
}};
|
||||||
|
CopyBuffer(ref.handle, source.handle, copy, false);
|
||||||
|
dst_offset += source.size;
|
||||||
|
}
|
||||||
|
PostCopyBarrier();
|
||||||
|
multi_range_buffers.MarkGathered(key);
|
||||||
|
}
|
||||||
|
guest_descriptor_queue.AddBuffer(ref.handle, ref.address, 0, ref.size);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format,
|
void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format,
|
||||||
u32 base_vertex, u32 num_indices, VkBuffer buffer,
|
u32 base_vertex, u32 num_indices, VkBuffer buffer,
|
||||||
u32 offset, [[maybe_unused]] u32 size) {
|
u32 offset, [[maybe_unused]] u32 size) {
|
||||||
|
|||||||
@@ -8,11 +8,14 @@
|
|||||||
|
|
||||||
#include <limits>
|
#include <limits>
|
||||||
|
|
||||||
|
#include <boost/container/small_vector.hpp>
|
||||||
|
|
||||||
#include "video_core/buffer_cache/buffer_cache_base.h"
|
#include "video_core/buffer_cache/buffer_cache_base.h"
|
||||||
#include "video_core/buffer_cache/memory_tracker_base.h"
|
#include "video_core/buffer_cache/memory_tracker_base.h"
|
||||||
#include "video_core/buffer_cache/usage_tracker.h"
|
#include "video_core/buffer_cache/usage_tracker.h"
|
||||||
#include "video_core/engines/maxwell_3d.h"
|
#include "video_core/engines/maxwell_3d.h"
|
||||||
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
||||||
|
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
|
||||||
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
|
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
|
||||||
#include "video_core/renderer_vulkan/vk_update_descriptor.h"
|
#include "video_core/renderer_vulkan/vk_update_descriptor.h"
|
||||||
#include "video_core/surface.h"
|
#include "video_core/surface.h"
|
||||||
@@ -31,7 +34,8 @@ class BufferCacheRuntime;
|
|||||||
class Buffer : public VideoCommon::BufferBase {
|
class Buffer : public VideoCommon::BufferBase {
|
||||||
public:
|
public:
|
||||||
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
|
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
|
||||||
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_);
|
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_,
|
||||||
|
bool sparse_compatible_ = false);
|
||||||
|
|
||||||
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
|
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
|
||||||
|
|
||||||
@@ -43,6 +47,14 @@ public:
|
|||||||
return device_address;
|
return device_address;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] bool IsSparseCompatible() const noexcept {
|
||||||
|
return sparse_compatible;
|
||||||
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] vk::MemoryLocation Location() const noexcept {
|
||||||
|
return buffer.Location();
|
||||||
|
}
|
||||||
|
|
||||||
[[nodiscard]] bool IsRegionUsed(u64 offset, u64 size) const noexcept {
|
[[nodiscard]] bool IsRegionUsed(u64 offset, u64 size) const noexcept {
|
||||||
return tracker.IsUsed(offset, size);
|
return tracker.IsUsed(offset, size);
|
||||||
}
|
}
|
||||||
@@ -77,6 +89,7 @@ private:
|
|||||||
VkDeviceAddress device_address{};
|
VkDeviceAddress device_address{};
|
||||||
u64 last_usage_tick{};
|
u64 last_usage_tick{};
|
||||||
bool is_null{};
|
bool is_null{};
|
||||||
|
bool sparse_compatible{};
|
||||||
};
|
};
|
||||||
|
|
||||||
class QuadArrayIndexBuffer;
|
class QuadArrayIndexBuffer;
|
||||||
@@ -125,7 +138,7 @@ public:
|
|||||||
|
|
||||||
void PreCopyBarrier();
|
void PreCopyBarrier();
|
||||||
|
|
||||||
void CopyBuffer(VkBuffer src_buffer, VkBuffer dst_buffer,
|
void CopyBuffer(VkBuffer dst_buffer, VkBuffer src_buffer,
|
||||||
std::span<const VideoCommon::BufferCopy> copies, bool barrier,
|
std::span<const VideoCommon::BufferCopy> copies, bool barrier,
|
||||||
bool can_reorder_upload = false);
|
bool can_reorder_upload = false);
|
||||||
|
|
||||||
@@ -155,6 +168,40 @@ public:
|
|||||||
return ref.mapped_span;
|
return ref.mapped_span;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] VkDeviceSize SparseAlignmentFor(bool sparse_compatible) const noexcept {
|
||||||
|
if (!sparse_compatible || !multi_range_buffers.UsesSparse()) {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
return multi_range_buffers.BlockSize();
|
||||||
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] bool PrefersSparseSources() const noexcept {
|
||||||
|
return multi_range_buffers.UsesSparse();
|
||||||
|
}
|
||||||
|
|
||||||
|
void ResetMultiRange() noexcept {
|
||||||
|
multi_range_sources.clear();
|
||||||
|
multi_range_total = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
void PushMultiRangeSource(const Buffer& buffer, u32 offset, u32 size) {
|
||||||
|
const vk::MemoryLocation location = buffer.Location();
|
||||||
|
multi_range_sources.push_back(MultiRangeSource{
|
||||||
|
.handle = buffer.Handle(),
|
||||||
|
.memory = location.memory,
|
||||||
|
.memory_offset = location.offset,
|
||||||
|
.offset = offset,
|
||||||
|
.size = size,
|
||||||
|
});
|
||||||
|
multi_range_total += size;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool BindMultiRangeStorageBuffer(u64 key);
|
||||||
|
|
||||||
|
void InvalidateMultiRange(u64 key) {
|
||||||
|
multi_range_buffers.Invalidate(key);
|
||||||
|
}
|
||||||
|
|
||||||
void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) {
|
void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) {
|
||||||
BindBuffer(buffer, offset, size);
|
BindBuffer(buffer, offset, size);
|
||||||
}
|
}
|
||||||
@@ -208,6 +255,10 @@ private:
|
|||||||
std::unique_ptr<Uint8Pass> uint8_pass;
|
std::unique_ptr<Uint8Pass> uint8_pass;
|
||||||
QuadIndexedPass quad_index_pass;
|
QuadIndexedPass quad_index_pass;
|
||||||
|
|
||||||
|
MultiRangeBufferCache multi_range_buffers;
|
||||||
|
boost::container::small_vector<MultiRangeSource, 16> multi_range_sources;
|
||||||
|
VkDeviceSize multi_range_total{};
|
||||||
|
|
||||||
bool limit_dynamic_storage_buffers = false;
|
bool limit_dynamic_storage_buffers = false;
|
||||||
u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)();
|
u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)();
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -0,0 +1,304 @@
|
|||||||
|
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
#include <mutex>
|
||||||
|
|
||||||
|
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
|
||||||
|
#include "video_core/renderer_vulkan/vk_scheduler.h"
|
||||||
|
#include "video_core/vulkan_common/vulkan_device.h"
|
||||||
|
|
||||||
|
namespace Vulkan {
|
||||||
|
|
||||||
|
MultiRangeBufferCache::MultiRangeBufferCache(const Device& device_,
|
||||||
|
MemoryAllocator& memory_allocator_,
|
||||||
|
Scheduler& scheduler_)
|
||||||
|
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_} {
|
||||||
|
sparse_usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||||
|
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||||
|
if (device.IsBufferDeviceAddressSupported()) {
|
||||||
|
sparse_usage |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
|
||||||
|
}
|
||||||
|
if (!device.IsSparseBindingSupported()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const VkDeviceSize queried = QueryBlockSize();
|
||||||
|
if (queried == 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
block_size = queried;
|
||||||
|
use_sparse = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
MultiRangeBufferCache::~MultiRangeBufferCache() {
|
||||||
|
const VkDevice logical = *device.GetLogical();
|
||||||
|
const auto& dld = device.GetDispatchLoader();
|
||||||
|
for (auto& [key, entry] : entries) {
|
||||||
|
if (entry.sparse_handle != VK_NULL_HANDLE) {
|
||||||
|
dld.vkDestroyBuffer(logical, entry.sparse_handle, nullptr);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
entries.clear();
|
||||||
|
for (const Retired& item : retired) {
|
||||||
|
dld.vkDestroyBuffer(logical, item.handle, nullptr);
|
||||||
|
}
|
||||||
|
retired.clear();
|
||||||
|
}
|
||||||
|
|
||||||
|
VkDeviceSize MultiRangeBufferCache::QueryBlockSize() const {
|
||||||
|
const VkDevice logical = *device.GetLogical();
|
||||||
|
const auto& dld = device.GetDispatchLoader();
|
||||||
|
const VkBufferCreateInfo probe_ci{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
|
||||||
|
.size = DEFAULT_BLOCK_SIZE,
|
||||||
|
.usage = sparse_usage,
|
||||||
|
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||||
|
.queueFamilyIndexCount = 0,
|
||||||
|
.pQueueFamilyIndices = nullptr,
|
||||||
|
};
|
||||||
|
VkBuffer probe{};
|
||||||
|
if (dld.vkCreateBuffer(logical, &probe_ci, nullptr, &probe) != VK_SUCCESS) {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
const VkBufferMemoryRequirementsInfo2 reqs_info{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_REQUIREMENTS_INFO_2,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.buffer = probe,
|
||||||
|
};
|
||||||
|
VkMemoryRequirements2 reqs2{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_MEMORY_REQUIREMENTS_2,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.memoryRequirements = {},
|
||||||
|
};
|
||||||
|
dld.vkGetBufferMemoryRequirements2(logical, &reqs_info, &reqs2);
|
||||||
|
dld.vkDestroyBuffer(logical, probe, nullptr);
|
||||||
|
return reqs2.memoryRequirements.alignment;
|
||||||
|
}
|
||||||
|
|
||||||
|
u64 MultiRangeBufferCache::HashSources(std::span<const MultiRangeSource> sources) const {
|
||||||
|
u64 hash = 0xcbf29ce484222325ULL;
|
||||||
|
const auto mix = [&hash](u64 value) {
|
||||||
|
hash ^= value;
|
||||||
|
hash *= 0x100000001b3ULL;
|
||||||
|
};
|
||||||
|
for (const MultiRangeSource& source : sources) {
|
||||||
|
mix(reinterpret_cast<u64>(source.handle));
|
||||||
|
mix(static_cast<u64>(source.offset));
|
||||||
|
mix(static_cast<u64>(source.size));
|
||||||
|
}
|
||||||
|
return hash;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool MultiRangeBufferCache::CanBindSparse(std::span<const MultiRangeSource> sources) const {
|
||||||
|
if (!use_sparse) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
for (const MultiRangeSource& source : sources) {
|
||||||
|
if (source.memory == VK_NULL_HANDLE) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const VkDeviceSize memory_offset = source.memory_offset + source.offset;
|
||||||
|
if ((memory_offset % block_size) != 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if ((source.size % block_size) != 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
VkBuffer MultiRangeBufferCache::CreateSparse(std::span<const MultiRangeSource> sources,
|
||||||
|
VkDeviceSize total) {
|
||||||
|
const VkDevice logical = *device.GetLogical();
|
||||||
|
const auto& dld = device.GetDispatchLoader();
|
||||||
|
const VkBufferCreateInfo buffer_ci{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
|
||||||
|
.size = total,
|
||||||
|
.usage = sparse_usage,
|
||||||
|
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||||
|
.queueFamilyIndexCount = 0,
|
||||||
|
.pQueueFamilyIndices = nullptr,
|
||||||
|
};
|
||||||
|
VkBuffer handle{};
|
||||||
|
if (dld.vkCreateBuffer(logical, &buffer_ci, nullptr, &handle) != VK_SUCCESS) {
|
||||||
|
return VK_NULL_HANDLE;
|
||||||
|
}
|
||||||
|
std::vector<VkSparseMemoryBind> binds;
|
||||||
|
binds.reserve(sources.size());
|
||||||
|
VkDeviceSize resource_offset = 0;
|
||||||
|
for (const MultiRangeSource& source : sources) {
|
||||||
|
binds.push_back(VkSparseMemoryBind{
|
||||||
|
.resourceOffset = resource_offset,
|
||||||
|
.size = source.size,
|
||||||
|
.memory = source.memory,
|
||||||
|
.memoryOffset = source.memory_offset + source.offset,
|
||||||
|
.flags = 0,
|
||||||
|
});
|
||||||
|
resource_offset += source.size;
|
||||||
|
}
|
||||||
|
const VkSparseBufferMemoryBindInfo buffer_bind{
|
||||||
|
.buffer = handle,
|
||||||
|
.bindCount = static_cast<u32>(binds.size()),
|
||||||
|
.pBinds = binds.data(),
|
||||||
|
};
|
||||||
|
const VkBindSparseInfo bind_info{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_BIND_SPARSE_INFO,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.waitSemaphoreCount = 0,
|
||||||
|
.pWaitSemaphores = nullptr,
|
||||||
|
.bufferBindCount = 1,
|
||||||
|
.pBufferBinds = &buffer_bind,
|
||||||
|
.imageOpaqueBindCount = 0,
|
||||||
|
.pImageOpaqueBinds = nullptr,
|
||||||
|
.imageBindCount = 0,
|
||||||
|
.pImageBinds = nullptr,
|
||||||
|
.signalSemaphoreCount = 0,
|
||||||
|
.pSignalSemaphores = nullptr,
|
||||||
|
};
|
||||||
|
const VkFenceCreateInfo fence_ci{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.flags = 0,
|
||||||
|
};
|
||||||
|
vk::Fence fence = device.GetLogical().CreateFence(fence_ci);
|
||||||
|
VkResult bind_result = VK_ERROR_UNKNOWN;
|
||||||
|
{
|
||||||
|
std::scoped_lock lock{scheduler.submit_mutex};
|
||||||
|
bind_result = device.GetGraphicsQueue().BindSparse(bind_info, *fence);
|
||||||
|
}
|
||||||
|
if (bind_result != VK_SUCCESS) {
|
||||||
|
dld.vkDestroyBuffer(logical, handle, nullptr);
|
||||||
|
return VK_NULL_HANDLE;
|
||||||
|
}
|
||||||
|
fence.Wait();
|
||||||
|
return handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
void MultiRangeBufferCache::DestroySparse(VkBuffer handle) {
|
||||||
|
if (handle == VK_NULL_HANDLE) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
retired.push_back(Retired{
|
||||||
|
.handle = handle,
|
||||||
|
.tick = scheduler.CurrentTick(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
void MultiRangeBufferCache::DrainRetired() {
|
||||||
|
const VkDevice logical = *device.GetLogical();
|
||||||
|
const auto& dld = device.GetDispatchLoader();
|
||||||
|
size_t index = 0;
|
||||||
|
while (index < retired.size()) {
|
||||||
|
if (scheduler.IsFree(retired[index].tick)) {
|
||||||
|
dld.vkDestroyBuffer(logical, retired[index].handle, nullptr);
|
||||||
|
retired[index] = retired.back();
|
||||||
|
retired.pop_back();
|
||||||
|
} else {
|
||||||
|
++index;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
MultiRangeRef MultiRangeBufferCache::Get(u64 key, std::span<const MultiRangeSource> sources,
|
||||||
|
VkDeviceSize total) {
|
||||||
|
if (sources.empty() || total == 0) {
|
||||||
|
return MultiRangeRef{};
|
||||||
|
}
|
||||||
|
if (!retired.empty()) {
|
||||||
|
DrainRetired();
|
||||||
|
}
|
||||||
|
const u64 geometry = HashSources(sources);
|
||||||
|
const auto it = entries.find(key);
|
||||||
|
if (it != entries.end() && it->second.geometry == geometry && it->second.size == total) {
|
||||||
|
Entry& entry = it->second;
|
||||||
|
MultiRangeRef ref{
|
||||||
|
.handle = entry.sparse_handle,
|
||||||
|
.address = entry.address,
|
||||||
|
.size = entry.size,
|
||||||
|
.needs_gather = false,
|
||||||
|
};
|
||||||
|
if (entry.sparse_handle == VK_NULL_HANDLE) {
|
||||||
|
ref.handle = *entry.gathered;
|
||||||
|
ref.needs_gather = entry.dirty;
|
||||||
|
}
|
||||||
|
return ref;
|
||||||
|
}
|
||||||
|
if (it != entries.end()) {
|
||||||
|
DestroySparse(it->second.sparse_handle);
|
||||||
|
entries.erase(it);
|
||||||
|
}
|
||||||
|
|
||||||
|
Entry entry;
|
||||||
|
entry.geometry = geometry;
|
||||||
|
entry.size = total;
|
||||||
|
if (CanBindSparse(sources)) {
|
||||||
|
entry.sparse_handle = CreateSparse(sources, total);
|
||||||
|
}
|
||||||
|
if (entry.sparse_handle == VK_NULL_HANDLE) {
|
||||||
|
VkBufferUsageFlags flags = VK_BUFFER_USAGE_TRANSFER_SRC_BIT |
|
||||||
|
VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||||
|
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||||
|
if (device.IsBufferDeviceAddressSupported()) {
|
||||||
|
flags |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
|
||||||
|
}
|
||||||
|
const VkBufferCreateInfo gather_ci{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.flags = 0,
|
||||||
|
.size = total,
|
||||||
|
.usage = flags,
|
||||||
|
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||||
|
.queueFamilyIndexCount = 0,
|
||||||
|
.pQueueFamilyIndices = nullptr,
|
||||||
|
};
|
||||||
|
entry.gathered = memory_allocator.CreateBuffer(gather_ci, MemoryUsage::DeviceLocal);
|
||||||
|
entry.dirty = true;
|
||||||
|
}
|
||||||
|
if (device.IsBufferDeviceAddressSupported()) {
|
||||||
|
VkBuffer address_handle = entry.sparse_handle;
|
||||||
|
if (address_handle == VK_NULL_HANDLE) {
|
||||||
|
address_handle = *entry.gathered;
|
||||||
|
}
|
||||||
|
entry.address = device.GetLogical().GetBufferDeviceAddress(address_handle);
|
||||||
|
}
|
||||||
|
|
||||||
|
MultiRangeRef ref{
|
||||||
|
.handle = entry.sparse_handle,
|
||||||
|
.address = entry.address,
|
||||||
|
.size = entry.size,
|
||||||
|
.needs_gather = false,
|
||||||
|
};
|
||||||
|
if (entry.sparse_handle == VK_NULL_HANDLE) {
|
||||||
|
ref.handle = *entry.gathered;
|
||||||
|
ref.needs_gather = true;
|
||||||
|
}
|
||||||
|
entries.emplace(key, std::move(entry));
|
||||||
|
return ref;
|
||||||
|
}
|
||||||
|
|
||||||
|
void MultiRangeBufferCache::MarkGathered(u64 key) {
|
||||||
|
const auto it = entries.find(key);
|
||||||
|
if (it != entries.end()) {
|
||||||
|
it->second.dirty = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void MultiRangeBufferCache::Invalidate(u64 key) {
|
||||||
|
const auto it = entries.find(key);
|
||||||
|
if (it != entries.end()) {
|
||||||
|
it->second.dirty = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void MultiRangeBufferCache::Clear() {
|
||||||
|
for (auto& [key, entry] : entries) {
|
||||||
|
DestroySparse(entry.sparse_handle);
|
||||||
|
}
|
||||||
|
entries.clear();
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace Vulkan
|
||||||
@@ -0,0 +1,100 @@
|
|||||||
|
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||||
|
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <span>
|
||||||
|
#include <unordered_map>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
#include "common/common_types.h"
|
||||||
|
#include "video_core/vulkan_common/vulkan_memory_allocator.h"
|
||||||
|
#include "video_core/vulkan_common/vulkan_wrapper.h"
|
||||||
|
|
||||||
|
namespace Vulkan {
|
||||||
|
|
||||||
|
class Device;
|
||||||
|
class Scheduler;
|
||||||
|
|
||||||
|
struct MultiRangeSource {
|
||||||
|
VkBuffer handle{};
|
||||||
|
VkDeviceMemory memory{};
|
||||||
|
VkDeviceSize memory_offset{};
|
||||||
|
VkDeviceSize offset{};
|
||||||
|
VkDeviceSize size{};
|
||||||
|
};
|
||||||
|
|
||||||
|
struct MultiRangeRef {
|
||||||
|
VkBuffer handle{};
|
||||||
|
VkDeviceAddress address{};
|
||||||
|
VkDeviceSize size{};
|
||||||
|
bool needs_gather{};
|
||||||
|
};
|
||||||
|
|
||||||
|
class MultiRangeBufferCache final {
|
||||||
|
public:
|
||||||
|
static constexpr VkDeviceSize DEFAULT_BLOCK_SIZE = 64 * 1024;
|
||||||
|
|
||||||
|
explicit MultiRangeBufferCache(const Device& device_, MemoryAllocator& memory_allocator_,
|
||||||
|
Scheduler& scheduler_);
|
||||||
|
~MultiRangeBufferCache();
|
||||||
|
|
||||||
|
MultiRangeBufferCache(const MultiRangeBufferCache&) = delete;
|
||||||
|
MultiRangeBufferCache& operator=(const MultiRangeBufferCache&) = delete;
|
||||||
|
|
||||||
|
[[nodiscard]] bool UsesSparse() const noexcept {
|
||||||
|
return use_sparse;
|
||||||
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] VkDeviceSize BlockSize() const noexcept {
|
||||||
|
return block_size;
|
||||||
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] MultiRangeRef Get(u64 key, std::span<const MultiRangeSource> sources,
|
||||||
|
VkDeviceSize total);
|
||||||
|
|
||||||
|
void MarkGathered(u64 key);
|
||||||
|
|
||||||
|
void Invalidate(u64 key);
|
||||||
|
|
||||||
|
void Clear();
|
||||||
|
|
||||||
|
private:
|
||||||
|
struct Retired {
|
||||||
|
VkBuffer handle{};
|
||||||
|
u64 tick{};
|
||||||
|
};
|
||||||
|
|
||||||
|
struct Entry {
|
||||||
|
vk::Buffer gathered;
|
||||||
|
VkBuffer sparse_handle{};
|
||||||
|
VkDeviceAddress address{};
|
||||||
|
VkDeviceSize size{};
|
||||||
|
u64 geometry{};
|
||||||
|
bool dirty{true};
|
||||||
|
};
|
||||||
|
|
||||||
|
[[nodiscard]] u64 HashSources(std::span<const MultiRangeSource> sources) const;
|
||||||
|
|
||||||
|
[[nodiscard]] bool CanBindSparse(std::span<const MultiRangeSource> sources) const;
|
||||||
|
|
||||||
|
[[nodiscard]] VkBuffer CreateSparse(std::span<const MultiRangeSource> sources,
|
||||||
|
VkDeviceSize total);
|
||||||
|
|
||||||
|
[[nodiscard]] VkDeviceSize QueryBlockSize() const;
|
||||||
|
|
||||||
|
void DestroySparse(VkBuffer handle);
|
||||||
|
|
||||||
|
void DrainRetired();
|
||||||
|
|
||||||
|
const Device& device;
|
||||||
|
MemoryAllocator& memory_allocator;
|
||||||
|
Scheduler& scheduler;
|
||||||
|
bool use_sparse{};
|
||||||
|
VkDeviceSize block_size{DEFAULT_BLOCK_SIZE};
|
||||||
|
VkBufferUsageFlags sparse_usage{};
|
||||||
|
std::unordered_map<u64, Entry> entries;
|
||||||
|
std::vector<Retired> retired;
|
||||||
|
};
|
||||||
|
|
||||||
|
} // namespace Vulkan
|
||||||
@@ -819,6 +819,7 @@ void RasterizerVulkan::ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) {
|
|||||||
std::scoped_lock lock{texture_cache.mutex};
|
std::scoped_lock lock{texture_cache.mutex};
|
||||||
texture_cache.UnmapGPUMemory(as_id, addr, size);
|
texture_cache.UnmapGPUMemory(as_id, addr, size);
|
||||||
}
|
}
|
||||||
|
buffer_cache.UnmapGPUMemory(as_id, addr, size);
|
||||||
}
|
}
|
||||||
|
|
||||||
void RasterizerVulkan::SignalFence(std::function<void()>&& func) {
|
void RasterizerVulkan::SignalFence(std::function<void()>&& func) {
|
||||||
|
|||||||
@@ -1570,6 +1570,8 @@ void Device::SetupFamilies(VkSurfaceKHR surface) {
|
|||||||
}
|
}
|
||||||
if (graphics) {
|
if (graphics) {
|
||||||
graphics_family = *graphics;
|
graphics_family = *graphics;
|
||||||
|
graphics_family_sparse_binding =
|
||||||
|
(queue_family_properties[*graphics].queueFlags & VK_QUEUE_SPARSE_BINDING_BIT) != 0;
|
||||||
}
|
}
|
||||||
if (present) {
|
if (present) {
|
||||||
present_family = *present;
|
present_family = *present;
|
||||||
|
|||||||
@@ -317,6 +317,10 @@ public:
|
|||||||
return properties.driver.driverID;
|
return properties.driver.driverID;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool IsSparseBindingSupported() const {
|
||||||
|
return features.features.sparseBinding && graphics_family_sparse_binding;
|
||||||
|
}
|
||||||
|
|
||||||
/// Returns true for tile-based deferred renderers.
|
/// Returns true for tile-based deferred renderers.
|
||||||
bool IsTiler() const {
|
bool IsTiler() const {
|
||||||
switch (GetDriverID()) {
|
switch (GetDriverID()) {
|
||||||
@@ -1146,6 +1150,7 @@ private:
|
|||||||
bool owns_static_pipeline_cache{};
|
bool owns_static_pipeline_cache{};
|
||||||
u32 instance_version{}; ///< Vulkan instance version.
|
u32 instance_version{}; ///< Vulkan instance version.
|
||||||
u32 graphics_family{}; ///< Main graphics queue family index.
|
u32 graphics_family{}; ///< Main graphics queue family index.
|
||||||
|
bool graphics_family_sparse_binding{};
|
||||||
u32 present_family{}; ///< Main present queue family index.
|
u32 present_family{}; ///< Main present queue family index.
|
||||||
|
|
||||||
struct Extensions {
|
struct Extensions {
|
||||||
|
|||||||
@@ -280,6 +280,85 @@ vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsa
|
|||||||
device.GetDispatchLoader());
|
device.GetDispatchLoader());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
|
||||||
|
VkDeviceSize min_alignment) const {
|
||||||
|
if (min_alignment <= 1) {
|
||||||
|
return CreateBuffer(ci, usage);
|
||||||
|
}
|
||||||
|
VkMemoryPropertyFlags anv_flags = 0;
|
||||||
|
if (usage == MemoryUsage::Stream &&
|
||||||
|
device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA) {
|
||||||
|
anv_flags = VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
|
||||||
|
}
|
||||||
|
u32 memory_type_bits = valid_memory_types;
|
||||||
|
if (usage == MemoryUsage::Stream) {
|
||||||
|
memory_type_bits = 0u;
|
||||||
|
}
|
||||||
|
const VmaAllocationCreateInfo alloc_ci = {
|
||||||
|
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
|
||||||
|
.usage = MemoryUsageVma(usage),
|
||||||
|
.requiredFlags = 0,
|
||||||
|
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
|
||||||
|
.memoryTypeBits = memory_type_bits,
|
||||||
|
.pool = VK_NULL_HANDLE,
|
||||||
|
.pUserData = nullptr,
|
||||||
|
.priority = 0.f,
|
||||||
|
};
|
||||||
|
|
||||||
|
const VkDevice logical = *device.GetLogical();
|
||||||
|
const auto &dld = device.GetDispatchLoader();
|
||||||
|
|
||||||
|
VkBuffer handle{};
|
||||||
|
vk::Check(dld.vkCreateBuffer(logical, &ci, nullptr, &handle));
|
||||||
|
|
||||||
|
const VkBufferMemoryRequirementsInfo2 reqs_info{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_REQUIREMENTS_INFO_2,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.buffer = handle,
|
||||||
|
};
|
||||||
|
VkMemoryRequirements2 reqs2{
|
||||||
|
.sType = VK_STRUCTURE_TYPE_MEMORY_REQUIREMENTS_2,
|
||||||
|
.pNext = nullptr,
|
||||||
|
.memoryRequirements = {},
|
||||||
|
};
|
||||||
|
dld.vkGetBufferMemoryRequirements2(logical, &reqs_info, &reqs2);
|
||||||
|
|
||||||
|
VkMemoryRequirements reqs = reqs2.memoryRequirements;
|
||||||
|
reqs.alignment = (std::max)(reqs.alignment, min_alignment);
|
||||||
|
reqs.memoryTypeBits &= alloc_ci.memoryTypeBits;
|
||||||
|
|
||||||
|
VmaAllocation allocation{};
|
||||||
|
VmaAllocationInfo alloc_info{};
|
||||||
|
VkResult res = vmaAllocateMemory(allocator, &reqs, &alloc_ci, &allocation, &alloc_info);
|
||||||
|
if (res != VK_SUCCESS) {
|
||||||
|
auto relaxed = alloc_ci;
|
||||||
|
relaxed.flags &= ~VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT;
|
||||||
|
res = vmaAllocateMemory(allocator, &reqs, &relaxed, &allocation, &alloc_info);
|
||||||
|
}
|
||||||
|
if (res != VK_SUCCESS) {
|
||||||
|
dld.vkDestroyBuffer(logical, handle, nullptr);
|
||||||
|
vk::Check(res);
|
||||||
|
}
|
||||||
|
const VkResult bind_res = vmaBindBufferMemory(allocator, allocation, handle);
|
||||||
|
if (bind_res != VK_SUCCESS) {
|
||||||
|
vmaFreeMemory(allocator, allocation);
|
||||||
|
dld.vkDestroyBuffer(logical, handle, nullptr);
|
||||||
|
vk::Check(bind_res);
|
||||||
|
}
|
||||||
|
|
||||||
|
VkMemoryPropertyFlags property_flags{};
|
||||||
|
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
|
||||||
|
|
||||||
|
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
|
||||||
|
std::span<u8> mapped_data{};
|
||||||
|
if (data) {
|
||||||
|
mapped_data = std::span<u8>{data, ci.size};
|
||||||
|
}
|
||||||
|
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
|
||||||
|
|
||||||
|
return vk::Buffer(handle, logical, allocator, allocation, mapped_data, is_coherent, dld);
|
||||||
|
}
|
||||||
|
|
||||||
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
|
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
|
||||||
{
|
{
|
||||||
const auto vma_usage = MemoryUsageVma(usage);
|
const auto vma_usage = MemoryUsageVma(usage);
|
||||||
|
|||||||
@@ -107,6 +107,9 @@ namespace Vulkan {
|
|||||||
|
|
||||||
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const;
|
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const;
|
||||||
|
|
||||||
|
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
|
||||||
|
VkDeviceSize min_alignment) const;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Commits a memory with the specified requirements.
|
* Commits a memory with the specified requirements.
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -229,6 +229,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
|
|||||||
X(vkGetPipelineExecutableStatisticsKHR);
|
X(vkGetPipelineExecutableStatisticsKHR);
|
||||||
X(vkGetSemaphoreCounterValue);
|
X(vkGetSemaphoreCounterValue);
|
||||||
X(vkMapMemory);
|
X(vkMapMemory);
|
||||||
|
X(vkQueueBindSparse);
|
||||||
X(vkQueueSubmit);
|
X(vkQueueSubmit);
|
||||||
X(vkQueueSubmit2);
|
X(vkQueueSubmit2);
|
||||||
X(vkResetFences);
|
X(vkResetFences);
|
||||||
@@ -539,6 +540,18 @@ void Buffer::SetObjectNameEXT(const char* name) const {
|
|||||||
SetObjectName(dld, owner, handle, VK_OBJECT_TYPE_BUFFER, name);
|
SetObjectName(dld, owner, handle, VK_OBJECT_TYPE_BUFFER, name);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
MemoryLocation Buffer::Location() const noexcept {
|
||||||
|
if (!allocation) {
|
||||||
|
return MemoryLocation{};
|
||||||
|
}
|
||||||
|
VmaAllocationInfo info{};
|
||||||
|
vmaGetAllocationInfo(allocator, allocation, &info);
|
||||||
|
return MemoryLocation{
|
||||||
|
.memory = info.deviceMemory,
|
||||||
|
.offset = info.offset,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
void Buffer::Release() const noexcept {
|
void Buffer::Release() const noexcept {
|
||||||
if (handle) {
|
if (handle) {
|
||||||
vmaDestroyBuffer(allocator, handle, allocation);
|
vmaDestroyBuffer(allocator, handle, allocation);
|
||||||
|
|||||||
@@ -345,6 +345,7 @@ struct DeviceDispatch : InstanceDispatch {
|
|||||||
PFN_vkGetQueryPoolResults vkGetQueryPoolResults{};
|
PFN_vkGetQueryPoolResults vkGetQueryPoolResults{};
|
||||||
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
|
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
|
||||||
PFN_vkMapMemory vkMapMemory{};
|
PFN_vkMapMemory vkMapMemory{};
|
||||||
|
PFN_vkQueueBindSparse vkQueueBindSparse{};
|
||||||
PFN_vkQueueSubmit vkQueueSubmit{};
|
PFN_vkQueueSubmit vkQueueSubmit{};
|
||||||
PFN_vkQueueSubmit2 vkQueueSubmit2{};
|
PFN_vkQueueSubmit2 vkQueueSubmit2{};
|
||||||
PFN_vkResetFences vkResetFences{};
|
PFN_vkResetFences vkResetFences{};
|
||||||
@@ -740,6 +741,11 @@ private:
|
|||||||
const DeviceDispatch* dld = nullptr;
|
const DeviceDispatch* dld = nullptr;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
struct MemoryLocation {
|
||||||
|
VkDeviceMemory memory{};
|
||||||
|
VkDeviceSize offset{};
|
||||||
|
};
|
||||||
|
|
||||||
class Buffer {
|
class Buffer {
|
||||||
public:
|
public:
|
||||||
explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_,
|
explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_,
|
||||||
@@ -811,6 +817,8 @@ public:
|
|||||||
|
|
||||||
void SetObjectNameEXT(const char* name) const;
|
void SetObjectNameEXT(const char* name) const;
|
||||||
|
|
||||||
|
MemoryLocation Location() const noexcept;
|
||||||
|
|
||||||
private:
|
private:
|
||||||
void Release() const noexcept;
|
void Release() const noexcept;
|
||||||
|
|
||||||
@@ -843,6 +851,11 @@ public:
|
|||||||
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
|
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
VkResult BindSparse(Span<VkBindSparseInfo> bind_infos,
|
||||||
|
VkFence fence = VK_NULL_HANDLE) const noexcept {
|
||||||
|
return dld->vkQueueBindSparse(queue, bind_infos.size(), bind_infos.data(), fence);
|
||||||
|
}
|
||||||
|
|
||||||
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
|
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
|
||||||
return dld->vkQueuePresentKHR(queue, &present_info);
|
return dld->vkQueuePresentKHR(queue, &present_info);
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user