[shader_recompiler, vulkan] Virtual buffer pages with multi-range + sparse buffers binding (#4362)

Storage buffers can span GPU pages that aren't contiguous in host memory, but the buffer cache assumed one contiguous buffer per binding, so anything crossing a mapping boundary read the wrong bytes. Multi-range binding resolves the real segments and presents them to the shader as one buffer, aliasing their memory into a sparse VkBuffer or gathering them into a copy, all of this was mostly fixed by making a path to use sparse on buffers (the same way as texture cache has it) and retrieve properly the mapping ranges to the virtual pages, including the actual structure on the use of binding sparse.

Meanwhile this fixes Monster Hunter Sunbreak z-fighting, bad performance and mostly vertex explosions (coming from the contiguous buffer read from the shader's game) it's only the basis for a further investigation towards how are we gonna treat phi calculations, optimized them and mostly eradicate the currently excess of exposure on the textures where the lightning trace should not being reflected.

Reviewed-on: https://git.eden-emu.dev/eden-emu/eden/pulls/4362
Reviewed-by: lizzie <lizzie@eden-emu.dev>
Reviewed-by: Maufeat <sahyno1996@gmail.com>
This commit is contained in:
CamilleLaVey
2026-09-08 20:42:20 +02:00
committed by crueter
parent ce202292cf
commit a538cd9aff
17 changed files with 975 additions and 43 deletions
@@ -1570,6 +1570,8 @@ void Device::SetupFamilies(VkSurfaceKHR surface) {
}
if (graphics) {
graphics_family = *graphics;
graphics_family_sparse_binding =
(queue_family_properties[*graphics].queueFlags & VK_QUEUE_SPARSE_BINDING_BIT) != 0;
}
if (present) {
present_family = *present;
@@ -317,6 +317,10 @@ public:
return properties.driver.driverID;
}
bool IsSparseBindingSupported() const {
return features.features.sparseBinding && graphics_family_sparse_binding;
}
/// Returns true for tile-based deferred renderers.
bool IsTiler() const {
switch (GetDriverID()) {
@@ -1147,6 +1151,7 @@ private:
u32 instance_version{}; ///< Vulkan instance version.
u32 graphics_family{}; ///< Main graphics queue family index.
u32 present_family{}; ///< Main present queue family index.
bool graphics_family_sparse_binding{};
struct Extensions {
#define EXTENSION(prefix, macro_name, var_name) bool var_name{};
@@ -275,9 +275,63 @@ vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsa
const std::span<u8> mapped_data = data ? std::span<u8>{data, ci.size} : std::span<u8>{};
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data,
is_coherent,
device.GetDispatchLoader());
const vk::MemoryLocation location{
.memory = alloc_info.deviceMemory,
.offset = alloc_info.offset,
.memory_type = alloc_info.memoryType,
};
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
location, device.GetDispatchLoader());
}
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
VkDeviceSize min_alignment) const {
if (min_alignment <= 1) {
return CreateBuffer(ci, usage);
}
VkMemoryPropertyFlags anv_flags = 0;
if (usage == MemoryUsage::Stream &&
device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA) {
anv_flags = VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
}
u32 memory_type_bits = valid_memory_types;
if (usage == MemoryUsage::Stream) {
memory_type_bits = 0u;
}
const VmaAllocationCreateInfo alloc_ci = {
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
.usage = MemoryUsageVma(usage),
.requiredFlags = 0,
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
.memoryTypeBits = memory_type_bits,
.pool = VK_NULL_HANDLE,
.pUserData = nullptr,
.priority = 0.f,
};
VkBuffer handle{};
VmaAllocationInfo alloc_info{};
VmaAllocation allocation{};
VkMemoryPropertyFlags property_flags{};
vk::Check(vmaCreateBufferWithAlignment(allocator, &ci, &alloc_ci, min_alignment, &handle,
&allocation, &alloc_info));
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
std::span<u8> mapped_data{};
if (data) {
mapped_data = std::span<u8>{data, ci.size};
}
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
const vk::MemoryLocation location{
.memory = alloc_info.deviceMemory,
.offset = alloc_info.offset,
.memory_type = alloc_info.memoryType,
};
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
location, device.GetDispatchLoader());
}
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
@@ -107,6 +107,9 @@ namespace Vulkan {
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const;
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
VkDeviceSize min_alignment) const;
/**
* Commits a memory with the specified requirements.
*
@@ -229,6 +229,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkGetPipelineExecutableStatisticsKHR);
X(vkGetSemaphoreCounterValue);
X(vkMapMemory);
X(vkQueueBindSparse);
X(vkQueueSubmit);
X(vkQueueSubmit2);
X(vkResetFences);
+22 -3
View File
@@ -345,6 +345,7 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkGetQueryPoolResults vkGetQueryPoolResults{};
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
PFN_vkMapMemory vkMapMemory{};
PFN_vkQueueBindSparse vkQueueBindSparse{};
PFN_vkQueueSubmit vkQueueSubmit{};
PFN_vkQueueSubmit2 vkQueueSubmit2{};
PFN_vkResetFences vkResetFences{};
@@ -740,13 +741,20 @@ private:
const DeviceDispatch* dld = nullptr;
};
struct MemoryLocation {
VkDeviceMemory memory{};
VkDeviceSize offset{};
u32 memory_type{};
};
class Buffer {
public:
explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_,
VmaAllocation allocation_, std::span<u8> mapped_, bool is_coherent_,
const DeviceDispatch& dld_) noexcept
MemoryLocation location_, const DeviceDispatch& dld_) noexcept
: handle{handle_}, owner{owner_}, allocator{allocator_},
allocation{allocation_}, mapped{mapped_}, is_coherent{is_coherent_}, dld{&dld_} {}
allocation{allocation_}, mapped{mapped_}, location{location_},
is_coherent{is_coherent_}, dld{&dld_} {}
Buffer() = default;
Buffer(const Buffer&) = delete;
@@ -754,7 +762,7 @@ public:
Buffer(Buffer&& rhs) noexcept
: handle{std::exchange(rhs.handle, VkBuffer{})}, owner{rhs.owner}, allocator{rhs.allocator},
allocation{rhs.allocation}, mapped{rhs.mapped},
allocation{rhs.allocation}, mapped{rhs.mapped}, location{rhs.location},
is_coherent{rhs.is_coherent}, dld{rhs.dld} {}
Buffer& operator=(Buffer&& rhs) noexcept {
@@ -764,6 +772,7 @@ public:
allocator = rhs.allocator;
allocation = rhs.allocation;
mapped = rhs.mapped;
location = rhs.location;
is_coherent = rhs.is_coherent;
dld = rhs.dld;
return *this;
@@ -811,6 +820,10 @@ public:
void SetObjectNameEXT(const char* name) const;
MemoryLocation Location() const noexcept {
return location;
}
private:
void Release() const noexcept;
@@ -819,6 +832,7 @@ private:
VmaAllocator allocator = nullptr;
VmaAllocation allocation = nullptr;
std::span<u8> mapped = {};
MemoryLocation location{};
bool is_coherent = false;
const DeviceDispatch* dld = nullptr;
};
@@ -843,6 +857,11 @@ public:
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
}
VkResult BindSparse(Span<VkBindSparseInfo> bind_infos,
VkFence fence = VK_NULL_HANDLE) const noexcept {
return dld->vkQueueBindSparse(queue, bind_infos.size(), bind_infos.data(), fence);
}
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
return dld->vkQueuePresentKHR(queue, &present_info);
}