mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-09-06 12:13:57 +00:00
Compare commits
10 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 89c781ae83 | |||
| b858e52f0a | |||
| 6c87c031a2 | |||
| a77c18fbbf | |||
| 203ac3cb51 | |||
| cf4e6ea901 | |||
| 49e96351b1 | |||
| 29b2b04a2d | |||
| 76b7a561ba | |||
| 11de264541 |
@@ -0,0 +1,112 @@
|
||||
diff --git a/include/vk_mem_alloc.h b/include/vk_mem_alloc.h
|
||||
index 8df0364..4856064 100644
|
||||
--- a/include/vk_mem_alloc.h
|
||||
+++ b/include/vk_mem_alloc.h
|
||||
@@ -3017,7 +3017,7 @@ remove them if not needed.
|
||||
|
||||
#if defined(__ANDROID_API__) && (__ANDROID_API__ < 16)
|
||||
#include <cstdlib>
|
||||
-static void* vma_aligned_alloc(size_t alignment, size_t size)
|
||||
+static inline void* vma_aligned_alloc(size_t alignment, size_t size)
|
||||
{
|
||||
// alignment must be >= sizeof(void*)
|
||||
if(alignment < sizeof(void*))
|
||||
@@ -3860,7 +3860,7 @@ Returned value is the found element, if present in the collection or place where
|
||||
new element with value (key) should be inserted.
|
||||
*/
|
||||
template <typename CmpLess, typename IterT, typename KeyT>
|
||||
-static IterT VmaBinaryFindFirstNotLess(IterT beg, IterT end, const KeyT& key, const CmpLess& cmp)
|
||||
+static inline IterT VmaBinaryFindFirstNotLess(IterT beg, IterT end, const KeyT& key, const CmpLess& cmp)
|
||||
{
|
||||
size_t down = 0;
|
||||
size_t up = size_t(end - beg);
|
||||
@@ -3898,7 +3898,7 @@ Warning! O(n^2) complexity. Use only inside VMA_HEAVY_ASSERT.
|
||||
T must be pointer type, e.g. VmaAllocation, VmaPool.
|
||||
*/
|
||||
template<typename T>
|
||||
-static bool VmaValidatePointerArray(uint32_t count, const T* arr)
|
||||
+static inline bool VmaValidatePointerArray(uint32_t count, const T* arr)
|
||||
{
|
||||
for (uint32_t i = 0; i < count; ++i)
|
||||
{
|
||||
@@ -4188,13 +4188,13 @@ static void VmaFree(const VkAllocationCallbacks* pAllocationCallbacks, void* ptr
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static T* VmaAllocate(const VkAllocationCallbacks* pAllocationCallbacks)
|
||||
+static inline T* VmaAllocate(const VkAllocationCallbacks* pAllocationCallbacks)
|
||||
{
|
||||
return (T*)VmaMalloc(pAllocationCallbacks, sizeof(T), VMA_ALIGN_OF(T));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static T* VmaAllocateArray(const VkAllocationCallbacks* pAllocationCallbacks, size_t count)
|
||||
+static inline T* VmaAllocateArray(const VkAllocationCallbacks* pAllocationCallbacks, size_t count)
|
||||
{
|
||||
return (T*)VmaMalloc(pAllocationCallbacks, sizeof(T) * count, VMA_ALIGN_OF(T));
|
||||
}
|
||||
@@ -4204,14 +4204,14 @@ static T* VmaAllocateArray(const VkAllocationCallbacks* pAllocationCallbacks, si
|
||||
#define vma_new_array(allocator, type, count) new(VmaAllocateArray<type>((allocator), (count)))(type)
|
||||
|
||||
template<typename T>
|
||||
-static void vma_delete(const VkAllocationCallbacks* pAllocationCallbacks, T* ptr)
|
||||
+static inline void vma_delete(const VkAllocationCallbacks* pAllocationCallbacks, T* ptr)
|
||||
{
|
||||
ptr->~T();
|
||||
VmaFree(pAllocationCallbacks, ptr);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static void vma_delete_array(const VkAllocationCallbacks* pAllocationCallbacks, T* ptr, size_t count)
|
||||
+static inline void vma_delete_array(const VkAllocationCallbacks* pAllocationCallbacks, T* ptr, size_t count)
|
||||
{
|
||||
if (ptr != VMA_NULL)
|
||||
{
|
||||
@@ -4658,13 +4658,13 @@ void VmaVector<T, AllocatorT>::remove(size_t index)
|
||||
#endif // _VMA_VECTOR_FUNCTIONS
|
||||
|
||||
template<typename T, typename allocatorT>
|
||||
-static void VmaVectorInsert(VmaVector<T, allocatorT>& vec, size_t index, const T& item)
|
||||
+static inline void VmaVectorInsert(VmaVector<T, allocatorT>& vec, size_t index, const T& item)
|
||||
{
|
||||
vec.insert(index, item);
|
||||
}
|
||||
|
||||
template<typename T, typename allocatorT>
|
||||
-static void VmaVectorRemove(VmaVector<T, allocatorT>& vec, size_t index)
|
||||
+static inline void VmaVectorRemove(VmaVector<T, allocatorT>& vec, size_t index)
|
||||
{
|
||||
vec.remove(index);
|
||||
}
|
||||
@@ -10620,19 +10620,19 @@ static void VmaFree(VmaAllocator hAllocator, void* ptr)
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static T* VmaAllocate(VmaAllocator hAllocator)
|
||||
+static inline T* VmaAllocate(VmaAllocator hAllocator)
|
||||
{
|
||||
return (T*)VmaMalloc(hAllocator, sizeof(T), VMA_ALIGN_OF(T));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static T* VmaAllocateArray(VmaAllocator hAllocator, size_t count)
|
||||
+static inline T* VmaAllocateArray(VmaAllocator hAllocator, size_t count)
|
||||
{
|
||||
return (T*)VmaMalloc(hAllocator, sizeof(T) * count, VMA_ALIGN_OF(T));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static void vma_delete(VmaAllocator hAllocator, T* ptr)
|
||||
+static inline void vma_delete(VmaAllocator hAllocator, T* ptr)
|
||||
{
|
||||
if(ptr != VMA_NULL)
|
||||
{
|
||||
@@ -10642,7 +10642,7 @@ static void vma_delete(VmaAllocator hAllocator, T* ptr)
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
-static void vma_delete_array(VmaAllocator hAllocator, T* ptr, size_t count)
|
||||
+static inline void vma_delete_array(VmaAllocator hAllocator, T* ptr, size_t count)
|
||||
{
|
||||
if(ptr != VMA_NULL)
|
||||
{
|
||||
+2
-1
@@ -3,6 +3,7 @@
|
||||
|
||||
cmake_minimum_required(VERSION 3.31)
|
||||
|
||||
set(CMAKE_OSX_DEPLOYMENT_TARGET "15.0" CACHE STRING "macOS deployment target")
|
||||
project(yuzu)
|
||||
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/CMakeModules")
|
||||
@@ -512,7 +513,7 @@ endfunction()
|
||||
# =============================================
|
||||
|
||||
if (APPLE)
|
||||
foreach(fw Carbon Metal Cocoa IOKit CoreVideo CoreMedia Security UniformTypeIdentifiers)
|
||||
foreach(fw Carbon Metal Cocoa IOKit CoreVideo CoreMedia Security UniformTypeIdentifiers Foundation)
|
||||
find_library(${fw}_LIBRARY ${fw} REQUIRED)
|
||||
list(APPEND PLATFORM_LIBRARIES ${${fw}_LIBRARY})
|
||||
endforeach()
|
||||
|
||||
+5
-10
@@ -1,12 +1,4 @@
|
||||
{
|
||||
"": {
|
||||
"ci": true,
|
||||
"hash": "9f50d993c39529e022ad456163de91ac5934e16faf8fc348f305355fc0a643c534f87ded707e11bd03bcbc0a1760bd20852b6533a67ac0558a078385181cb184",
|
||||
"name": "SDL3",
|
||||
"package": "SDL3",
|
||||
"repo": "crueter-ci/SDL3",
|
||||
"version": "3.4.14-1788231389-147a8ee32d"
|
||||
},
|
||||
"biscuit": {
|
||||
"hash": "1229f345b014f7ca544dedb4edb3311e41ba736f9aa9a67f88b5f26f3c983288c6bb6cdedcfb0b8a02c63088a37e6a0d7ba97d9c2a4d721b213916327cffe28a",
|
||||
"min_version": "0.9.1",
|
||||
@@ -68,10 +60,10 @@
|
||||
},
|
||||
"discord-rpc": {
|
||||
"find_args": "MODULE",
|
||||
"hash": "8213c43dcb0f7d479f5861091d111ed12fbdec1e62e6d729d65a4bc181d82f48a35d5fd3cd5c291f2393ac7c9681eabc6b76609755f55376284c8a8d67e148f3",
|
||||
"hash": "8d680b3a16d6f6bf292ad823cf8635595ff986f2a49f758e3009f505b540a9d75194a8cf44cddea75736a8c3e3b5160166e716a0939ce3f18cc463cf648753fe",
|
||||
"package": "DiscordRPC",
|
||||
"repo": "eden-emulator/discord-rpc",
|
||||
"version": "0d8b2d6a37"
|
||||
"version": "76616d8675"
|
||||
},
|
||||
"enet": {
|
||||
"find_args": "MODULE",
|
||||
@@ -311,6 +303,9 @@
|
||||
"find_args": "CONFIG",
|
||||
"hash": "deb5902ef8db0e329fbd5f3f4385eb0e26bdd9f14f3a2334823fb3fe18f36bc5d235d620d6e5f6fe3551ec3ea7038638899db8778c09f6d5c278f5ff95c3344b",
|
||||
"package": "VulkanMemoryAllocator",
|
||||
"patches": [
|
||||
"0001-macos-clang.patch"
|
||||
],
|
||||
"repo": "GPUOpen-LibrariesAndSDKs/VulkanMemoryAllocator",
|
||||
"version": "v3.3.0"
|
||||
},
|
||||
|
||||
+1
@@ -30,6 +30,7 @@ enum class BooleanSetting(override val key: String) : AbstractBooleanSetting {
|
||||
RENDERER_REACTIVE_FLUSHING("use_reactive_flushing"),
|
||||
ENABLE_BUFFER_HISTORY("enable_buffer_history"),
|
||||
USE_OPTIMIZED_VERTEX_BUFFERS("use_optimized_vertex_buffers"),
|
||||
ENABLE_SHADER_PHI_TRACKING("enable_shader_phi_tracking"),
|
||||
ENABLE_GPU_BUFFER_READBACK("enable_gpu_buffer_readback"),
|
||||
SYNC_MEMORY_OPERATIONS("sync_memory_operations"),
|
||||
BUFFER_REORDER_DISABLE("disable_buffer_reorder"),
|
||||
|
||||
+7
@@ -927,6 +927,13 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.use_optimized_vertex_buffers_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.ENABLE_SHADER_PHI_TRACKING,
|
||||
titleId = R.string.enable_shader_phi_tracking,
|
||||
descriptionId = R.string.enable_shader_phi_tracking_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.SYNC_MEMORY_OPERATIONS,
|
||||
|
||||
+1
@@ -338,6 +338,7 @@ class SettingsFragmentPresenter(
|
||||
add(BooleanSetting.ENABLE_BUFFER_HISTORY.key)
|
||||
add(BooleanSetting.ENABLE_GPU_BUFFER_READBACK.key)
|
||||
add(BooleanSetting.USE_OPTIMIZED_VERTEX_BUFFERS.key)
|
||||
add(BooleanSetting.ENABLE_SHADER_PHI_TRACKING.key)
|
||||
|
||||
add(HeaderSetting(R.string.hacks))
|
||||
|
||||
|
||||
@@ -570,6 +570,8 @@
|
||||
<string name="enable_gpu_buffer_readback_description">Preserves GPU-modified buffer data by reading it back before uploads. Some games require this to render certain effects properly. May cause issues if the hardware cannot handle the additional workload.</string>
|
||||
<string name="use_optimized_vertex_buffers">Optimized Vertex Buffers</string>
|
||||
<string name="use_optimized_vertex_buffers_description">Enables optimized vertex buffer binding for improved performance. Requires Mesa 26.0+ Turnip drivers/ QCOM drivers. Will crash on older Turnip drivers (25.3 and below).</string>
|
||||
<string name="enable_shader_phi_tracking">Shader Phi Tracking</string>
|
||||
<string name="enable_shader_phi_tracking_description">toggle for test on shader phi tracking.</string>
|
||||
|
||||
<string name="hacks">Hacks</string>
|
||||
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <iterator>
|
||||
#include <cstring>
|
||||
|
||||
#include "common/make_unique_for_overwrite.h"
|
||||
|
||||
@@ -61,7 +65,7 @@ public:
|
||||
void resize(size_type size) {
|
||||
if (size > buffer_capacity) {
|
||||
auto new_buffer = Common::make_unique_for_overwrite<T[]>(size);
|
||||
std::move(buffer.get(), buffer.get() + buffer_capacity, new_buffer.get());
|
||||
std::memcpy(new_buffer.get(), buffer.get(), buffer_capacity * sizeof(T));
|
||||
buffer = std::move(new_buffer);
|
||||
buffer_capacity = size;
|
||||
}
|
||||
|
||||
@@ -592,6 +592,14 @@ struct Values {
|
||||
true,
|
||||
true};
|
||||
|
||||
SwitchableSetting<bool> enable_shader_phi_tracking{linkage,
|
||||
true,
|
||||
"enable_shader_phi_tracking",
|
||||
Category::RendererAdvanced,
|
||||
Specialization::Default,
|
||||
true,
|
||||
true};
|
||||
|
||||
#ifdef __ANDROID__
|
||||
SwitchableSetting<bool> use_optimized_vertex_buffers{linkage,
|
||||
false,
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
#include <functional>
|
||||
#include <span>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/common_types.h"
|
||||
|
||||
|
||||
@@ -6,13 +6,13 @@
|
||||
|
||||
#include <mutex>
|
||||
#include <utility>
|
||||
#include <type_traits>
|
||||
|
||||
#include <boost/asio.hpp>
|
||||
#include <boost/version.hpp>
|
||||
|
||||
#if BOOST_VERSION > 108400 && (!defined(_WINDOWS) && !defined(__ANDROID__)) || defined(YUZU_BOOST_v1)
|
||||
#define USE_BOOST_v1
|
||||
#endif
|
||||
|
||||
#ifdef USE_BOOST_v1
|
||||
#include <boost/process/v1/async_pipe.hpp>
|
||||
#else
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
namespace FileSys {
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <optional>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/literals.h"
|
||||
#include "core/file_sys/fssystem/fs_i_storage.h"
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <type_traits>
|
||||
#include "core/file_sys/errors.h"
|
||||
#include "core/file_sys/fssystem/fssystem_bucket_tree.h"
|
||||
#include "core/file_sys/fssystem/fssystem_bucket_tree_utils.h"
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <mutex>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/alignment.h"
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include "core/file_sys/errors.h"
|
||||
#include "core/file_sys/fssystem/fssystem_bucket_tree.h"
|
||||
#include "core/file_sys/fssystem/fssystem_bucket_tree_utils.h"
|
||||
|
||||
@@ -6,6 +6,8 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include <cstddef>
|
||||
#include "common/literals.h"
|
||||
|
||||
#include "core/file_sys/errors.h"
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include "common/alignment.h"
|
||||
#include "core/file_sys/fssystem/fs_i_storage.h"
|
||||
#include "core/file_sys/fssystem/fs_types.h"
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include "core/file_sys/errors.h"
|
||||
#include "core/file_sys/fssystem/fs_i_storage.h"
|
||||
#include "core/file_sys/fssystem/fssystem_bucket_tree.h"
|
||||
|
||||
@@ -1,9 +1,15 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <optional>
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <type_traits>
|
||||
|
||||
#include "core/file_sys/fssystem/fs_i_storage.h"
|
||||
#include "core/file_sys/fssystem/fs_types.h"
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
@@ -6,6 +6,8 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include <cstddef>
|
||||
#include "core/file_sys/fssystem/fssystem_compression_common.h"
|
||||
#include "core/file_sys/fssystem/fssystem_nca_header.h"
|
||||
#include "core/file_sys/vfs/vfs.h"
|
||||
|
||||
@@ -1,8 +1,14 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
#include "common/literals.h"
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/page_table.h"
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include "common/assert.h"
|
||||
#include "common/bit_field.h"
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2024 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <type_traits>
|
||||
#include <functional>
|
||||
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
@@ -175,19 +175,10 @@ Result AlbumManager::LoadAlbumScreenShotImage(LoadAlbumScreenShotImageOutput& ou
|
||||
return ResultIsNotMounted;
|
||||
}
|
||||
|
||||
out_image_output = {
|
||||
.width = 1280,
|
||||
.height = 720,
|
||||
.attribute =
|
||||
{
|
||||
.unknown_0{},
|
||||
.orientation = AlbumImageOrientation::None,
|
||||
.unknown_1{},
|
||||
.unknown_2{},
|
||||
.pad163{},
|
||||
},
|
||||
.pad179{},
|
||||
};
|
||||
out_image_output = {};
|
||||
out_image_output.width = 1280;
|
||||
out_image_output.height = 720;
|
||||
out_image_output.attribute.orientation = AlbumImageOrientation::None;
|
||||
|
||||
std::filesystem::path path;
|
||||
const auto result = GetFile(path, file_id);
|
||||
@@ -211,19 +202,10 @@ Result AlbumManager::LoadAlbumScreenShotThumbnail(
|
||||
return ResultIsNotMounted;
|
||||
}
|
||||
|
||||
out_image_output = {
|
||||
.width = 320,
|
||||
.height = 180,
|
||||
.attribute =
|
||||
{
|
||||
.unknown_0{},
|
||||
.orientation = AlbumImageOrientation::None,
|
||||
.unknown_1{},
|
||||
.unknown_2{},
|
||||
.pad163{},
|
||||
},
|
||||
.pad179{},
|
||||
};
|
||||
out_image_output = {};
|
||||
out_image_output.width = 320;
|
||||
out_image_output.height = 180;
|
||||
out_image_output.attribute.orientation = AlbumImageOrientation::None;
|
||||
|
||||
std::filesystem::path path;
|
||||
const auto result = GetFile(path, file_id);
|
||||
|
||||
@@ -73,13 +73,8 @@ void IScreenShotApplicationService::CaptureAndSaveScreenshot(AlbumReportOption r
|
||||
Layout::FramebufferLayout layout =
|
||||
Layout::DefaultFrameLayout(screenshot_width, screenshot_height);
|
||||
|
||||
const Capture::ScreenShotAttribute attribute{
|
||||
.unknown_0{},
|
||||
.orientation = Capture::AlbumImageOrientation::None,
|
||||
.unknown_1{},
|
||||
.unknown_2{},
|
||||
.pad163{},
|
||||
};
|
||||
Capture::ScreenShotAttribute attribute{};
|
||||
attribute.orientation = Capture::AlbumImageOrientation::None;
|
||||
|
||||
renderer.RequestScreenshot(
|
||||
image_data.data(),
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <type_traits>
|
||||
#include <fmt/ranges.h>
|
||||
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include <array>
|
||||
#include <chrono>
|
||||
#include <type_traits>
|
||||
#include <fmt/ranges.h>
|
||||
|
||||
#include "common/common_types.h"
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
|
||||
#include <type_traits>
|
||||
#include "common/common_types.h"
|
||||
#include "core/hle/service/psc/time/common.h"
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/bit_field.h"
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
@@ -1,8 +1,15 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <array>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/bit_field.h"
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
|
||||
@@ -267,6 +267,8 @@ std::unique_ptr<TranslationMap> InitializeTranslations(QObject* parent) {
|
||||
INSERT(Settings, enable_buffer_history, tr("Enable buffer history"),
|
||||
tr("Enables access to previous buffer states.\nThis option may improve rendering "
|
||||
"quality and performance consistency in some games."));
|
||||
INSERT(Settings, enable_shader_phi_tracking, tr("Shader phi tracking"),
|
||||
tr("toggle for test on shader phi tracking."));
|
||||
INSERT(Settings, fix_bloom_effects, tr("Fix bloom effects"), tr("Removes bloom in Burnout."));
|
||||
|
||||
INSERT(Settings, rescale_hack, tr("Enable Legacy Rescale Pass"),
|
||||
|
||||
@@ -6,55 +6,6 @@
|
||||
|
||||
add_library(shader_recompiler STATIC
|
||||
backend/bindings.h
|
||||
backend/glasm/emit_glasm.cpp
|
||||
backend/glasm/emit_glasm.h
|
||||
backend/glasm/emit_glasm_barriers.cpp
|
||||
backend/glasm/emit_glasm_bitwise_conversion.cpp
|
||||
backend/glasm/emit_glasm_composite.cpp
|
||||
backend/glasm/emit_glasm_context_get_set.cpp
|
||||
backend/glasm/emit_glasm_control_flow.cpp
|
||||
backend/glasm/emit_glasm_convert.cpp
|
||||
backend/glasm/emit_glasm_floating_point.cpp
|
||||
backend/glasm/emit_glasm_image.cpp
|
||||
backend/glasm/emit_glasm_instructions.h
|
||||
backend/glasm/emit_glasm_integer.cpp
|
||||
backend/glasm/emit_glasm_logical.cpp
|
||||
backend/glasm/emit_glasm_memory.cpp
|
||||
backend/glasm/emit_glasm_not_implemented.cpp
|
||||
backend/glasm/emit_glasm_select.cpp
|
||||
backend/glasm/emit_glasm_shared_memory.cpp
|
||||
backend/glasm/emit_glasm_special.cpp
|
||||
backend/glasm/emit_glasm_undefined.cpp
|
||||
backend/glasm/emit_glasm_warp.cpp
|
||||
backend/glasm/glasm_emit_context.cpp
|
||||
backend/glasm/glasm_emit_context.h
|
||||
backend/glasm/reg_alloc.cpp
|
||||
backend/glasm/reg_alloc.h
|
||||
backend/glsl/emit_glsl.cpp
|
||||
backend/glsl/emit_glsl.h
|
||||
backend/glsl/emit_glsl_atomic.cpp
|
||||
backend/glsl/emit_glsl_barriers.cpp
|
||||
backend/glsl/emit_glsl_bitwise_conversion.cpp
|
||||
backend/glsl/emit_glsl_composite.cpp
|
||||
backend/glsl/emit_glsl_context_get_set.cpp
|
||||
backend/glsl/emit_glsl_control_flow.cpp
|
||||
backend/glsl/emit_glsl_convert.cpp
|
||||
backend/glsl/emit_glsl_floating_point.cpp
|
||||
backend/glsl/emit_glsl_image.cpp
|
||||
backend/glsl/emit_glsl_instructions.h
|
||||
backend/glsl/emit_glsl_integer.cpp
|
||||
backend/glsl/emit_glsl_logical.cpp
|
||||
backend/glsl/emit_glsl_memory.cpp
|
||||
backend/glsl/emit_glsl_not_implemented.cpp
|
||||
backend/glsl/emit_glsl_select.cpp
|
||||
backend/glsl/emit_glsl_shared_memory.cpp
|
||||
backend/glsl/emit_glsl_special.cpp
|
||||
backend/glsl/emit_glsl_undefined.cpp
|
||||
backend/glsl/emit_glsl_warp.cpp
|
||||
backend/glsl/glsl_emit_context.cpp
|
||||
backend/glsl/glsl_emit_context.h
|
||||
backend/glsl/var_alloc.cpp
|
||||
backend/glsl/var_alloc.h
|
||||
backend/spirv/emit_spirv.cpp
|
||||
backend/spirv/emit_spirv.h
|
||||
backend/spirv/emit_spirv_atomic.cpp
|
||||
@@ -239,9 +190,60 @@ add_library(shader_recompiler STATIC
|
||||
program_header.h
|
||||
runtime_info.h
|
||||
shader_info.h
|
||||
varying_state.h
|
||||
varying_state.h)
|
||||
|
||||
)
|
||||
if (ENABLE_OPENGL)
|
||||
target_sources(shader_recompiler PRIVATE
|
||||
backend/glasm/emit_glasm.cpp
|
||||
backend/glasm/emit_glasm.h
|
||||
backend/glasm/emit_glasm_barriers.cpp
|
||||
backend/glasm/emit_glasm_bitwise_conversion.cpp
|
||||
backend/glasm/emit_glasm_composite.cpp
|
||||
backend/glasm/emit_glasm_context_get_set.cpp
|
||||
backend/glasm/emit_glasm_control_flow.cpp
|
||||
backend/glasm/emit_glasm_convert.cpp
|
||||
backend/glasm/emit_glasm_floating_point.cpp
|
||||
backend/glasm/emit_glasm_image.cpp
|
||||
backend/glasm/emit_glasm_instructions.h
|
||||
backend/glasm/emit_glasm_integer.cpp
|
||||
backend/glasm/emit_glasm_logical.cpp
|
||||
backend/glasm/emit_glasm_memory.cpp
|
||||
backend/glasm/emit_glasm_not_implemented.cpp
|
||||
backend/glasm/emit_glasm_select.cpp
|
||||
backend/glasm/emit_glasm_shared_memory.cpp
|
||||
backend/glasm/emit_glasm_special.cpp
|
||||
backend/glasm/emit_glasm_undefined.cpp
|
||||
backend/glasm/emit_glasm_warp.cpp
|
||||
backend/glasm/glasm_emit_context.cpp
|
||||
backend/glasm/glasm_emit_context.h
|
||||
backend/glasm/reg_alloc.cpp
|
||||
backend/glasm/reg_alloc.h
|
||||
backend/glsl/emit_glsl.cpp
|
||||
backend/glsl/emit_glsl.h
|
||||
backend/glsl/emit_glsl_atomic.cpp
|
||||
backend/glsl/emit_glsl_barriers.cpp
|
||||
backend/glsl/emit_glsl_bitwise_conversion.cpp
|
||||
backend/glsl/emit_glsl_composite.cpp
|
||||
backend/glsl/emit_glsl_context_get_set.cpp
|
||||
backend/glsl/emit_glsl_control_flow.cpp
|
||||
backend/glsl/emit_glsl_convert.cpp
|
||||
backend/glsl/emit_glsl_floating_point.cpp
|
||||
backend/glsl/emit_glsl_image.cpp
|
||||
backend/glsl/emit_glsl_instructions.h
|
||||
backend/glsl/emit_glsl_integer.cpp
|
||||
backend/glsl/emit_glsl_logical.cpp
|
||||
backend/glsl/emit_glsl_memory.cpp
|
||||
backend/glsl/emit_glsl_not_implemented.cpp
|
||||
backend/glsl/emit_glsl_select.cpp
|
||||
backend/glsl/emit_glsl_shared_memory.cpp
|
||||
backend/glsl/emit_glsl_special.cpp
|
||||
backend/glsl/emit_glsl_undefined.cpp
|
||||
backend/glsl/emit_glsl_warp.cpp
|
||||
backend/glsl/glsl_emit_context.cpp
|
||||
backend/glsl/glsl_emit_context.h
|
||||
backend/glsl/var_alloc.cpp
|
||||
backend/glsl/var_alloc.h)
|
||||
endif()
|
||||
|
||||
target_link_libraries(shader_recompiler PUBLIC common fmt::fmt sirit::sirit)
|
||||
|
||||
|
||||
@@ -492,9 +492,6 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
|
||||
if (ctx.uses_nonuniform_storage_texel_buffer) {
|
||||
ctx.AddCapability(spv::Capability::StorageTexelBufferArrayNonUniformIndexing);
|
||||
}
|
||||
if (ctx.uses_nonuniform_storage_buffer) {
|
||||
ctx.AddCapability(spv::Capability::StorageBufferArrayNonUniformIndexing);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <bit>
|
||||
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv.h"
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv_instructions.h"
|
||||
#include "shader_recompiler/backend/spirv/spirv_emit_context.h"
|
||||
@@ -21,13 +23,29 @@ Id SharedPointer(EmitContext& ctx, Id offset, u32 index_offset = 0) {
|
||||
: ctx.OpAccessChain(ctx.shared_u32, ctx.shared_memory_u32, index);
|
||||
}
|
||||
|
||||
Id StorageIndex(EmitContext& ctx, const IR::Value& offset, size_t element_size) {
|
||||
if (offset.IsImmediate()) {
|
||||
const u32 imm_offset{static_cast<u32>(offset.U32() / element_size)};
|
||||
return ctx.Const(imm_offset);
|
||||
}
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(element_size))};
|
||||
const Id index{ctx.Def(offset)};
|
||||
if (shift == 0) {
|
||||
return index;
|
||||
}
|
||||
const Id shift_id{ctx.Const(shift)};
|
||||
return ctx.OpShiftRightLogical(ctx.U32[1], index, shift_id);
|
||||
}
|
||||
|
||||
Id StoragePointer(EmitContext& ctx, const StorageTypeDefinition& type_def,
|
||||
Id StorageDefinitions::*member_ptr, const IR::Value& binding,
|
||||
const IR::Value& offset, size_t element_size) {
|
||||
if (!binding.IsImmediate()) {
|
||||
throw NotImplementedException("Dynamic storage buffer indexing");
|
||||
}
|
||||
return ctx.StoragePointer(binding.U32(), ctx.Def(offset), type_def, static_cast<u32>(element_size), member_ptr);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].*member_ptr};
|
||||
const Id index{StorageIndex(ctx, offset, element_size)};
|
||||
return ctx.OpAccessChain(type_def.element, ssbo, ctx.u32_zero_value, index);
|
||||
}
|
||||
|
||||
std::pair<Id, Id> AtomicArgs(EmitContext& ctx) {
|
||||
@@ -198,14 +216,16 @@ Id EmitStorageAtomicUMax32(EmitContext& ctx, const IR::Value& binding, const IR:
|
||||
|
||||
Id EmitStorageAtomicInc32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.increment_cas_ssbo, pointer, value);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.increment_cas_ssbo, base_index, value, ssbo);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicDec32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.decrement_cas_ssbo, pointer, value);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.decrement_cas_ssbo, base_index, value, ssbo);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicAnd32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
@@ -344,49 +364,56 @@ Id EmitStorageAtomicExchange32x2(EmitContext& ctx, const IR::Value& binding,
|
||||
|
||||
Id EmitStorageAtomicAddF32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.F32[1], ctx.f32_add_cas, pointer, value);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
return ctx.OpFunctionCall(ctx.F32[1], ctx.f32_add_cas, base_index, value, ssbo);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicAddF16x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_add_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_add_cas, base_index, value, ssbo)};
|
||||
return ctx.OpBitcast(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicAddF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_add_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_add_cas, base_index, value, ssbo)};
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMinF16x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_min_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_min_cas, base_index, value, ssbo)};
|
||||
return ctx.OpBitcast(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMinF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_min_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_min_cas, base_index, value, ssbo)};
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMaxF16x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_max_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F16[2], ctx.f16x2_max_cas, base_index, value, ssbo)};
|
||||
return ctx.OpBitcast(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitStorageAtomicMaxF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value) {
|
||||
const Id pointer{StoragePointer(ctx, ctx.storage_types.U32, &StorageDefinitions::U32, binding, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_max_cas, pointer, value)};
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].U32};
|
||||
const Id base_index{StorageIndex(ctx, offset, sizeof(u32))};
|
||||
const Id result{ctx.OpFunctionCall(ctx.F32[2], ctx.f32x2_max_cas, base_index, value, ssbo)};
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
|
||||
@@ -4,33 +4,46 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <bit>
|
||||
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv.h"
|
||||
#include "shader_recompiler/backend/spirv/emit_spirv_instructions.h"
|
||||
#include "shader_recompiler/backend/spirv/spirv_emit_context.h"
|
||||
|
||||
namespace Shader::Backend::SPIRV {
|
||||
namespace {
|
||||
Id StorageByteOffset(EmitContext& ctx, const IR::Value& offset, size_t element_size, u32 index_offset) {
|
||||
Id byte_offset{ctx.Def(offset)};
|
||||
if (index_offset != 0) {
|
||||
byte_offset = ctx.OpIAdd(ctx.U32[1], byte_offset, ctx.Const(static_cast<u32>(index_offset * element_size)));
|
||||
Id StorageIndex(EmitContext& ctx, const IR::Value& offset, size_t element_size,
|
||||
u32 index_offset = 0) {
|
||||
if (offset.IsImmediate()) {
|
||||
const u32 imm_offset{static_cast<u32>(offset.U32() / element_size) + index_offset};
|
||||
return ctx.Const(imm_offset);
|
||||
}
|
||||
return byte_offset;
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(element_size))};
|
||||
Id index{ctx.Def(offset)};
|
||||
if (shift != 0) {
|
||||
const Id shift_id{ctx.Const(shift)};
|
||||
index = ctx.OpShiftRightLogical(ctx.U32[1], index, shift_id);
|
||||
}
|
||||
if (index_offset != 0) {
|
||||
index = ctx.OpIAdd(ctx.U32[1], index, ctx.Const(index_offset));
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
Id StoragePointer(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
const StorageTypeDefinition& type_def, size_t element_size,
|
||||
Id StorageDefinitions::* member_ptr, u32 index_offset = 0) {
|
||||
Id StorageDefinitions::*member_ptr, u32 index_offset = 0) {
|
||||
if (!binding.IsImmediate()) {
|
||||
throw NotImplementedException("Dynamic storage buffer indexing");
|
||||
}
|
||||
const Id byte_offset{StorageByteOffset(ctx, offset, element_size, index_offset)};
|
||||
return ctx.StoragePointer(binding.U32(), byte_offset, type_def, static_cast<u32>(element_size), member_ptr);
|
||||
const Id ssbo{ctx.ssbos[binding.U32()].*member_ptr};
|
||||
const Id index{StorageIndex(ctx, offset, element_size, index_offset)};
|
||||
return ctx.OpAccessChain(type_def.element, ssbo, ctx.u32_zero_value, index);
|
||||
}
|
||||
|
||||
Id LoadStorage(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset, Id result_type,
|
||||
const StorageTypeDefinition& type_def, size_t element_size,
|
||||
Id StorageDefinitions::* member_ptr, u32 index_offset = 0) {
|
||||
Id StorageDefinitions::*member_ptr, u32 index_offset = 0) {
|
||||
const Id pointer{
|
||||
StoragePointer(ctx, binding, offset, type_def, element_size, member_ptr, index_offset)};
|
||||
return ctx.OpLoad(result_type, pointer);
|
||||
@@ -44,7 +57,7 @@ Id LoadStorage32(EmitContext& ctx, const IR::Value& binding, const IR::Value& of
|
||||
|
||||
void WriteStorage(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset, Id value,
|
||||
const StorageTypeDefinition& type_def, size_t element_size,
|
||||
Id StorageDefinitions::* member_ptr, u32 index_offset = 0) {
|
||||
Id StorageDefinitions::*member_ptr, u32 index_offset = 0) {
|
||||
const Id pointer{
|
||||
StoragePointer(ctx, binding, offset, type_def, element_size, member_ptr, index_offset)};
|
||||
ctx.OpStore(pointer, value);
|
||||
|
||||
@@ -299,7 +299,7 @@ void DefineConstBuffers(EmitContext& ctx, const Info& info, Id UniformDefinition
|
||||
}
|
||||
|
||||
void DefineSsbos(EmitContext& ctx, StorageTypeDefinition& type_def,
|
||||
Id StorageDefinitions::* member_type, const Info& info, u32 binding, Id type,
|
||||
Id StorageDefinitions::*member_type, const Info& info, u32 binding, Id type,
|
||||
u32 stride) {
|
||||
const Id array_type{ctx.TypeRuntimeArray(type)};
|
||||
ctx.Decorate(array_type, spv::Decoration::ArrayStride, stride);
|
||||
@@ -309,27 +309,23 @@ void DefineSsbos(EmitContext& ctx, StorageTypeDefinition& type_def,
|
||||
ctx.MemberDecorate(struct_type, 0, spv::Decoration::Offset, 0U);
|
||||
|
||||
const Id struct_pointer{ctx.TypePointer(spv::StorageClass::StorageBuffer, struct_type)};
|
||||
type_def.array = struct_pointer;
|
||||
type_def.element = ctx.TypePointer(spv::StorageClass::StorageBuffer, type);
|
||||
|
||||
u32 index{};
|
||||
for (const StorageBufferDescriptor& desc : info.storage_buffers_descriptors) {
|
||||
const Id variable_type{[&] {
|
||||
if (desc.count == 1) {
|
||||
return struct_pointer;
|
||||
}
|
||||
const Id descriptor_array{ctx.TypeArray(struct_type, ctx.Const(desc.count))};
|
||||
return ctx.TypePointer(spv::StorageClass::StorageBuffer, descriptor_array);
|
||||
}()};
|
||||
const Id id{ctx.AddGlobalVariable(variable_type, spv::StorageClass::StorageBuffer)};
|
||||
const Id id{ctx.AddGlobalVariable(struct_pointer, spv::StorageClass::StorageBuffer)};
|
||||
ctx.Decorate(id, spv::Decoration::Binding, binding);
|
||||
ctx.Decorate(id, spv::Decoration::DescriptorSet, 0U);
|
||||
ctx.Name(id, fmt::format("ssbo{}", index));
|
||||
if (ctx.profile.supported_spirv >= 0x00010400) {
|
||||
ctx.interfaces.push_back(id);
|
||||
}
|
||||
ctx.ssbos[index].*member_type = id;
|
||||
++index;
|
||||
++binding;
|
||||
for (size_t i = 0; i < desc.count; ++i) {
|
||||
ctx.ssbos[index + i].*member_type = id;
|
||||
}
|
||||
index += desc.count;
|
||||
binding += desc.count;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -372,7 +368,8 @@ Id CasFunction(EmitContext& ctx, Operation operation, Id value_type) {
|
||||
return func;
|
||||
}
|
||||
|
||||
Id CasLoop(EmitContext& ctx, Operation operation, Id element_pointer, Id value_type, Id memory_type, spv::Scope scope) {
|
||||
Id CasLoop(EmitContext& ctx, Operation operation, Id array_pointer, Id element_pointer,
|
||||
Id value_type, Id memory_type, spv::Scope scope) {
|
||||
const bool is_shared{scope == spv::Scope::Workgroup};
|
||||
const bool is_struct{!is_shared || ctx.uses_explicit_workgroup_layout};
|
||||
const Id cas_func{CasFunction(ctx, operation, value_type)};
|
||||
@@ -382,12 +379,14 @@ Id CasLoop(EmitContext& ctx, Operation operation, Id element_pointer, Id value_t
|
||||
const Id loop_header{ctx.OpLabel()};
|
||||
const Id continue_block{ctx.OpLabel()};
|
||||
const Id merge_block{ctx.OpLabel()};
|
||||
const Id func_type{is_shared ? ctx.TypeFunction(value_type, ctx.U32[1], value_type)
|
||||
: ctx.TypeFunction(value_type, element_pointer, value_type)};
|
||||
const Id func_type{is_shared
|
||||
? ctx.TypeFunction(value_type, ctx.U32[1], value_type)
|
||||
: ctx.TypeFunction(value_type, ctx.U32[1], value_type, array_pointer)};
|
||||
|
||||
const Id func{ctx.OpFunction(value_type, spv::FunctionControlMask::MaskNone, func_type)};
|
||||
const Id address{ctx.OpFunctionParameter(is_shared ? ctx.U32[1] : element_pointer)};
|
||||
const Id index{ctx.OpFunctionParameter(ctx.U32[1])};
|
||||
const Id op_b{ctx.OpFunctionParameter(value_type)};
|
||||
const Id base{is_shared ? ctx.shared_memory_u32 : ctx.OpFunctionParameter(array_pointer)};
|
||||
ctx.AddLabel();
|
||||
ctx.OpBranch(loop_header);
|
||||
ctx.AddLabel(loop_header);
|
||||
@@ -396,13 +395,8 @@ Id CasLoop(EmitContext& ctx, Operation operation, Id element_pointer, Id value_t
|
||||
ctx.OpBranch(continue_block);
|
||||
|
||||
ctx.AddLabel(continue_block);
|
||||
const Id word_pointer{[&] {
|
||||
if (!is_shared) {
|
||||
return address;
|
||||
}
|
||||
return is_struct ? ctx.OpAccessChain(element_pointer, ctx.shared_memory_u32, zero, address)
|
||||
: ctx.OpAccessChain(element_pointer, ctx.shared_memory_u32, address);
|
||||
}()};
|
||||
const Id word_pointer{is_struct ? ctx.OpAccessChain(element_pointer, base, zero, index)
|
||||
: ctx.OpAccessChain(element_pointer, base, index)};
|
||||
if (value_type.value == ctx.F32[2].value) {
|
||||
const Id u32_value{ctx.OpLoad(ctx.U32[1], word_pointer)};
|
||||
const Id value{ctx.OpUnpackHalf2x16(ctx.F32[2], u32_value)};
|
||||
@@ -486,7 +480,6 @@ EmitContext::EmitContext(const Profile& profile_, const RuntimeInfo& runtime_inf
|
||||
DefineSharedMemoryFunctions(program);
|
||||
DefineConstantBuffers(program.info, uniform_binding);
|
||||
DefineConstantBufferIndirectFunctions(program.info);
|
||||
DefineStorageBufferMappings(program.info, storage_binding);
|
||||
DefineStorageBuffers(program.info, storage_binding);
|
||||
DefineTextureBuffers(program.info, texture_binding);
|
||||
DefineImageBuffers(program.info, image_binding);
|
||||
@@ -539,36 +532,6 @@ Id EmitContext::BitOffset16(const IR::Value& offset) {
|
||||
return OpBitwiseAnd(U32[1], OpShiftLeftLogical(U32[1], Def(offset), Const(3u)), Const(16u));
|
||||
}
|
||||
|
||||
Id EmitContext::StoragePointer(u32 binding, Id byte_offset, const StorageTypeDefinition& type_def,
|
||||
u32 element_size, Id StorageDefinitions::* member_ptr) {
|
||||
const Id ssbo{ssbos[binding].*member_ptr};
|
||||
const u32 segment_count{storage_buffer_mapping_counts[binding]};
|
||||
if (segment_count <= 1) {
|
||||
const Id index{
|
||||
element_size == 1
|
||||
? byte_offset
|
||||
: OpShiftRightLogical(U32[1], byte_offset, Const(static_cast<u32>(std::countr_zero(element_size))))};
|
||||
return OpAccessChain(type_def.element, ssbo, u32_zero_value, index);
|
||||
}
|
||||
|
||||
const Id mapped{OpFunctionCall(U32[2], storage_buffer_map_func,
|
||||
Const(storage_buffer_mapping_bases[binding]),
|
||||
Const(segment_count), byte_offset)};
|
||||
const Id segment{OpCompositeExtract(U32[1], mapped, 0U)};
|
||||
const Id local_offset{OpCompositeExtract(U32[1], mapped, 1U)};
|
||||
const Id index{
|
||||
element_size == 1
|
||||
? local_offset
|
||||
: OpShiftRightLogical(U32[1], local_offset, Const(static_cast<u32>(std::countr_zero(element_size))))};
|
||||
Decorate(segment, spv::Decoration::NonUniform);
|
||||
non_uniform_ids.insert(segment.value);
|
||||
const Id pointer{OpAccessChain(type_def.element, ssbo, segment, u32_zero_value, index)};
|
||||
Decorate(pointer, spv::Decoration::NonUniform);
|
||||
non_uniform_ids.insert(pointer.value);
|
||||
uses_nonuniform_storage_buffer = true;
|
||||
return pointer;
|
||||
}
|
||||
|
||||
void EmitContext::DefineCommonTypes(const Info& info) {
|
||||
void_id = TypeVoid();
|
||||
|
||||
@@ -733,10 +696,12 @@ void EmitContext::DefineSharedMemory(const IR::Program& program) {
|
||||
|
||||
void EmitContext::DefineSharedMemoryFunctions(const IR::Program& program) {
|
||||
if (program.info.uses_shared_increment) {
|
||||
increment_cas_shared = CasLoop(*this, Operation::Increment, shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
increment_cas_shared = CasLoop(*this, Operation::Increment, shared_memory_u32_type,
|
||||
shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
}
|
||||
if (program.info.uses_shared_decrement) {
|
||||
decrement_cas_shared = CasLoop(*this, Operation::Decrement, shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
decrement_cas_shared = CasLoop(*this, Operation::Decrement, shared_memory_u32_type,
|
||||
shared_u32, U32[1], U32[1], spv::Scope::Workgroup);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -980,7 +945,8 @@ void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
}
|
||||
using DefPtr = Id StorageDefinitions::*;
|
||||
const Id zero{u32_zero_value};
|
||||
const auto define_body{[&](DefPtr ssbo_member, Id addr, const StorageTypeDefinition& type_def, u32 shift, auto&& callback) {
|
||||
const auto define_body{[&](DefPtr ssbo_member, Id addr, Id element_pointer, u32 shift,
|
||||
auto&& callback) {
|
||||
AddLabel();
|
||||
const size_t num_buffers{info.storage_buffers_descriptors.size()};
|
||||
for (size_t index = 0; index < num_buffers; ++index) {
|
||||
@@ -1007,28 +973,30 @@ void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
OpSelectionMerge(else_label, spv::SelectionControlMask::MaskNone);
|
||||
OpBranchConditional(cond, then_label, else_label);
|
||||
AddLabel(then_label);
|
||||
const Id ssbo_id{ssbos[index].*ssbo_member};
|
||||
const Id ssbo_offset{OpUConvert(U32[1], OpISub(U64, addr, ssbo_addr))};
|
||||
const Id ssbo_pointer{StoragePointer(static_cast<u32>(index), ssbo_offset, type_def, 1U << shift, ssbo_member)};
|
||||
const Id ssbo_index{OpShiftRightLogical(U32[1], ssbo_offset, Const(shift))};
|
||||
const Id ssbo_pointer{OpAccessChain(element_pointer, ssbo_id, zero, ssbo_index)};
|
||||
callback(ssbo_pointer);
|
||||
AddLabel(else_label);
|
||||
}
|
||||
}};
|
||||
const auto define_load{
|
||||
[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(type, U64)};
|
||||
const Id func_id{OpFunction(type, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
define_body(ssbo_member, addr, type_def, shift, [&](Id ssbo_pointer) { OpReturnValue(OpLoad(type, ssbo_pointer)); });
|
||||
OpReturnValue(ConstantNull(type));
|
||||
OpFunctionEnd();
|
||||
return func_id;
|
||||
}};
|
||||
const auto define_write{[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, u32 shift) {
|
||||
const auto define_load{[&](DefPtr ssbo_member, Id element_pointer, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(type, U64)};
|
||||
const Id func_id{OpFunction(type, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
define_body(ssbo_member, addr, element_pointer, shift,
|
||||
[&](Id ssbo_pointer) { OpReturnValue(OpLoad(type, ssbo_pointer)); });
|
||||
OpReturnValue(ConstantNull(type));
|
||||
OpFunctionEnd();
|
||||
return func_id;
|
||||
}};
|
||||
const auto define_write{[&](DefPtr ssbo_member, Id element_pointer, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(void_id, U64, type)};
|
||||
const Id func_id{OpFunction(void_id, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
const Id data{OpFunctionParameter(type)};
|
||||
define_body(ssbo_member, addr, type_def, shift, [&](Id ssbo_pointer) {
|
||||
define_body(ssbo_member, addr, element_pointer, shift, [&](Id ssbo_pointer) {
|
||||
OpStore(ssbo_pointer, data);
|
||||
OpReturn();
|
||||
});
|
||||
@@ -1038,9 +1006,10 @@ void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
}};
|
||||
const auto define{
|
||||
[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, size_t size) {
|
||||
const Id element_type{type_def.element};
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(size))};
|
||||
const Id load_func{define_load(ssbo_member, type_def, type, shift)};
|
||||
const Id write_func{define_write(ssbo_member, type_def, type, shift)};
|
||||
const Id load_func{define_load(ssbo_member, element_type, type, shift)};
|
||||
const Id write_func{define_write(ssbo_member, element_type, type, shift)};
|
||||
return std::make_pair(load_func, write_func);
|
||||
}};
|
||||
std::tie(load_global_func_u32, write_global_func_u32) =
|
||||
@@ -1259,79 +1228,6 @@ void EmitContext::DefineConstantBufferIndirectFunctions(const Info& info) {
|
||||
}
|
||||
}
|
||||
|
||||
void EmitContext::DefineStorageBufferMappings(const Info& info, u32& binding) {
|
||||
if (!UsesStorageBufferMappings(info)) {
|
||||
return;
|
||||
}
|
||||
ASSERT(profile.support_storage_buffer_array_nonuniform_indexing);
|
||||
AddExtension("SPV_KHR_storage_buffer_storage_class");
|
||||
|
||||
const u32 num_entries{NumDescriptors(info.storage_buffers_descriptors)};
|
||||
const Id array_type{TypeArray(U32[1], Const(num_entries))};
|
||||
Decorate(array_type, spv::Decoration::ArrayStride, sizeof(u32));
|
||||
const Id struct_type{TypeStruct(array_type)};
|
||||
Decorate(struct_type, spv::Decoration::Block);
|
||||
MemberName(struct_type, 0, "segment_sizes");
|
||||
MemberDecorate(struct_type, 0, spv::Decoration::Offset, 0U);
|
||||
const Id pointer_type{TypePointer(spv::StorageClass::StorageBuffer, struct_type)};
|
||||
storage_buffer_mapping_u32 = TypePointer(spv::StorageClass::StorageBuffer, U32[1]);
|
||||
storage_buffer_mapping = AddGlobalVariable(pointer_type, spv::StorageClass::StorageBuffer);
|
||||
Decorate(storage_buffer_mapping, spv::Decoration::Binding, binding++);
|
||||
Decorate(storage_buffer_mapping, spv::Decoration::DescriptorSet, 0U);
|
||||
Name(storage_buffer_mapping, "storage_buffer_mapping");
|
||||
//Starting with version 1.4... (https://registry.khronos.org/SPIR-V/specs/unified1/SPIRV.html)
|
||||
if (profile.supported_spirv >= 0x00010400) {
|
||||
interfaces.push_back(storage_buffer_mapping);
|
||||
}
|
||||
|
||||
u32 mapping_base{};
|
||||
for (u32 index = 0; index < info.storage_buffers_descriptors.size(); ++index) {
|
||||
const StorageBufferDescriptor& desc = info.storage_buffers_descriptors[index];
|
||||
storage_buffer_mapping_bases[index] = mapping_base;
|
||||
storage_buffer_mapping_counts[index] = desc.count;
|
||||
mapping_base += desc.count;
|
||||
}
|
||||
|
||||
const Id function_type{TypeFunction(U32[2], U32[1], U32[1], U32[1])};
|
||||
storage_buffer_map_func = OpFunction(U32[2], spv::FunctionControlMask::MaskNone, function_type);
|
||||
const Id base{OpFunctionParameter(U32[1])};
|
||||
const Id count{OpFunctionParameter(U32[1])};
|
||||
const Id byte_offset{OpFunctionParameter(U32[1])};
|
||||
const Id index_pointer_type{TypePointer(spv::StorageClass::Function, U32[1])};
|
||||
const Id loop_header{OpLabel()};
|
||||
const Id continue_block{OpLabel()};
|
||||
const Id merge_block{OpLabel()};
|
||||
AddLabel();
|
||||
const Id segment_var{AddLocalVariable(index_pointer_type, spv::StorageClass::Function)};
|
||||
const Id offset_var{AddLocalVariable(index_pointer_type, spv::StorageClass::Function)};
|
||||
OpStore(segment_var, u32_zero_value);
|
||||
OpStore(offset_var, byte_offset);
|
||||
OpBranch(loop_header);
|
||||
|
||||
AddLabel(loop_header);
|
||||
const Id segment{OpLoad(U32[1], segment_var)};
|
||||
const Id local_offset{OpLoad(U32[1], offset_var)};
|
||||
const Id mapping_index{OpIAdd(U32[1], base, segment)};
|
||||
const Id size_pointer{OpAccessChain(storage_buffer_mapping_u32, storage_buffer_mapping, u32_zero_value, mapping_index)};
|
||||
const Id segment_size{OpLoad(U32[1], size_pointer)};
|
||||
const Id fits{OpULessThan(U1, local_offset, segment_size)};
|
||||
const Id last_segment{OpISub(U32[1], count, Const(1U))};
|
||||
const Id is_last{OpIEqual(U1, segment, last_segment)};
|
||||
const Id found{OpLogicalOr(U1, fits, is_last)};
|
||||
OpLoopMerge(merge_block, continue_block, spv::LoopControlMask::MaskNone);
|
||||
OpBranchConditional(found, merge_block, continue_block);
|
||||
|
||||
AddLabel(continue_block);
|
||||
OpStore(offset_var, OpISub(U32[1], local_offset, segment_size));
|
||||
OpStore(segment_var, OpIAdd(U32[1], segment, Const(1U)));
|
||||
OpBranch(loop_header);
|
||||
|
||||
AddLabel(merge_block);
|
||||
OpReturnValue(OpCompositeConstruct(U32[2], segment, local_offset));
|
||||
OpFunctionEnd();
|
||||
Name(storage_buffer_map_func, "map_storage_buffer");
|
||||
}
|
||||
|
||||
void EmitContext::DefineStorageBuffers(const Info& info, u32& binding) {
|
||||
if (info.storage_buffers_descriptors.empty()) {
|
||||
return;
|
||||
@@ -1375,7 +1271,9 @@ void EmitContext::DefineStorageBuffers(const Info& info, u32& binding) {
|
||||
DefineSsbos(*this, storage_types.U32x4, &StorageDefinitions::U32x4, info, binding, U32[4],
|
||||
sizeof(u32[4]));
|
||||
}
|
||||
binding += static_cast<u32>(info.storage_buffers_descriptors.size());
|
||||
for (const StorageBufferDescriptor& desc : info.storage_buffers_descriptors) {
|
||||
binding += desc.count;
|
||||
}
|
||||
const bool needs_function{
|
||||
info.uses_global_increment || info.uses_global_decrement || info.uses_atomic_f32_add ||
|
||||
info.uses_atomic_f16x2_add || info.uses_atomic_f16x2_min || info.uses_atomic_f16x2_max ||
|
||||
@@ -1384,31 +1282,40 @@ void EmitContext::DefineStorageBuffers(const Info& info, u32& binding) {
|
||||
AddCapability(spv::Capability::VariablePointersStorageBuffer);
|
||||
}
|
||||
if (info.uses_global_increment) {
|
||||
increment_cas_ssbo = CasLoop(*this, Operation::Increment, storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
increment_cas_ssbo = CasLoop(*this, Operation::Increment, storage_types.U32.array,
|
||||
storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_global_decrement) {
|
||||
decrement_cas_ssbo = CasLoop(*this, Operation::Decrement, storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
decrement_cas_ssbo = CasLoop(*this, Operation::Decrement, storage_types.U32.array,
|
||||
storage_types.U32.element, U32[1], U32[1], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32_add) {
|
||||
f32_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.element, F32[1], U32[1], spv::Scope::Device);
|
||||
f32_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[1], U32[1], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f16x2_add) {
|
||||
f16x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
f16x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.array,
|
||||
storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f16x2_min) {
|
||||
f16x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
f16x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.array,
|
||||
storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f16x2_max) {
|
||||
f16x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
f16x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.array,
|
||||
storage_types.U32.element, F16[2], F16[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32x2_add) {
|
||||
f32x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
f32x2_add_cas = CasLoop(*this, Operation::FPAdd, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32x2_min) {
|
||||
f32x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
f32x2_min_cas = CasLoop(*this, Operation::FPMin, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
}
|
||||
if (info.uses_atomic_f32x2_max) {
|
||||
f32x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
f32x2_max_cas = CasLoop(*this, Operation::FPMax, storage_types.U32.array,
|
||||
storage_types.U32.element, F32[2], F32[2], spv::Scope::Device);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -114,6 +114,7 @@ struct UniformDefinitions {
|
||||
};
|
||||
|
||||
struct StorageTypeDefinition {
|
||||
Id array{};
|
||||
Id element{};
|
||||
};
|
||||
|
||||
@@ -172,8 +173,6 @@ public:
|
||||
[[nodiscard]] Id BitOffset8(const IR::Value& offset);
|
||||
[[nodiscard]] Id BitOffset16(const IR::Value& offset);
|
||||
|
||||
[[nodiscard]] Id StoragePointer(u32 binding, Id byte_offset, const StorageTypeDefinition& type_def, u32 element_size, Id StorageDefinitions::*member_ptr);
|
||||
|
||||
Id Const(u32 value) {
|
||||
return Constant(U32[1], value);
|
||||
}
|
||||
@@ -258,11 +257,6 @@ public:
|
||||
|
||||
std::array<UniformDefinitions, Info::MAX_CBUFS> cbufs{};
|
||||
std::array<StorageDefinitions, Info::MAX_SSBOS> ssbos{};
|
||||
std::array<u32, Info::MAX_SSBOS> storage_buffer_mapping_bases{};
|
||||
std::array<u32, Info::MAX_SSBOS> storage_buffer_mapping_counts{};
|
||||
Id storage_buffer_mapping{};
|
||||
Id storage_buffer_mapping_u32{};
|
||||
Id storage_buffer_map_func{};
|
||||
std::vector<TextureBufferDefinition> texture_buffers;
|
||||
std::vector<ImageBufferDefinition> image_buffers;
|
||||
std::vector<TextureDefinition> textures;
|
||||
@@ -382,7 +376,6 @@ public:
|
||||
bool uses_nonuniform_storage_image{};
|
||||
bool uses_nonuniform_uniform_texel_buffer{};
|
||||
bool uses_nonuniform_storage_texel_buffer{};
|
||||
bool uses_nonuniform_storage_buffer{};
|
||||
|
||||
private:
|
||||
void DefineCommonTypes(const Info& info);
|
||||
@@ -393,7 +386,6 @@ private:
|
||||
void DefineSharedMemoryFunctions(const IR::Program& program);
|
||||
void DefineConstantBuffers(const Info& info, u32& binding);
|
||||
void DefineConstantBufferIndirectFunctions(const Info& info);
|
||||
void DefineStorageBufferMappings(const Info& info, u32& binding);
|
||||
void DefineStorageBuffers(const Info& info, u32& binding);
|
||||
void DefineTextureBuffers(const Info& info, u32& binding);
|
||||
void DefineImageBuffers(const Info& info, u32& binding);
|
||||
|
||||
@@ -132,14 +132,6 @@ void AddNVNStorageBuffers(IR::Program& program) {
|
||||
}
|
||||
}
|
||||
|
||||
void ConfigureStorageBufferMappings(IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
// https://docs.vulkan.org/guide/latest/descriptor_arrays.html
|
||||
// will be needed: descriptor array elements represent physical spans of one guest virtual buffer.
|
||||
for (StorageBufferDescriptor& desc : program.info.storage_buffers_descriptors) {
|
||||
desc.count = host_info.storage_buffer_segment_count;
|
||||
}
|
||||
}
|
||||
|
||||
using IR::IsLegacyAttribute; //rescoped to attribute.h to make it visible in load_store_attribute.cpp IPA
|
||||
|
||||
std::map<IR::Attribute, IR::Attribute> GenerateLegacyToGenericMappings(
|
||||
@@ -307,7 +299,6 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
Optimization::PositionPass(env, program);
|
||||
|
||||
Optimization::GlobalMemoryToStorageBufferPass(program, normalized_host_info);
|
||||
ConfigureStorageBufferMappings(program, normalized_host_info);
|
||||
Optimization::TexturePass(env, program, normalized_host_info);
|
||||
|
||||
if (Settings::values.resolution_info.active || Settings::values.rescale_hack.GetValue()) {
|
||||
@@ -323,7 +314,6 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
|
||||
CollectInterpolationInfo(env, program);
|
||||
AddNVNStorageBuffers(program);
|
||||
ConfigureStorageBufferMappings(program, normalized_host_info);
|
||||
return program;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include "common/common_types.h"
|
||||
|
||||
namespace Shader {
|
||||
@@ -20,7 +19,6 @@ struct HostTranslateInfo {
|
||||
|
||||
u64 min_ssbo_alignment{}; ///< Minimum alignment supported by the device for SSBOs
|
||||
u32 max_per_stage_descriptor_sampled_images{}; ///< maximum sampled descriptors per stage
|
||||
u32 max_per_stage_descriptor_storage_buffers{}; ///< maximum storage descriptors per stage
|
||||
u32 max_per_stage_resources{}; ///< maximum resources per stage
|
||||
u32 max_descriptor_set_samplers{};
|
||||
u32 max_descriptor_set_uniform_buffers{};
|
||||
@@ -40,14 +38,12 @@ struct HostTranslateInfo {
|
||||
///< passthrough shaders
|
||||
bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional
|
||||
///< control flow
|
||||
u32 storage_buffer_segment_count{1}; ///< Physical ranges available to each guest SSBO
|
||||
|
||||
void ApplyDescriptorLimitPolicy() noexcept {
|
||||
if (min_ssbo_alignment == 0) {
|
||||
min_ssbo_alignment = 1;
|
||||
}
|
||||
ApplyDescriptorLimitFallback(max_per_stage_descriptor_sampled_images);
|
||||
ApplyDescriptorLimitFallback(max_per_stage_descriptor_storage_buffers);
|
||||
ApplyDescriptorLimitFallback(max_per_stage_resources);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_samplers);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_uniform_buffers);
|
||||
@@ -57,7 +53,6 @@ struct HostTranslateInfo {
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_sampled_images);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_storage_images);
|
||||
ApplyDescriptorLimitFallback(max_descriptor_set_input_attachements);
|
||||
storage_buffer_segment_count = (std::max)(storage_buffer_segment_count, 1U);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
@@ -553,17 +553,6 @@ void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateIn
|
||||
}
|
||||
}
|
||||
|
||||
template <typename Descriptors, typename Descriptor, typename Func>
|
||||
static u32 Add(Descriptors& descriptors, const Descriptor& desc, Func&& pred) {
|
||||
// TODO: Handle arrays
|
||||
const auto it{std::ranges::find_if(descriptors, pred)};
|
||||
if (it != descriptors.end()) {
|
||||
return static_cast<u32>(std::distance(descriptors.begin(), it));
|
||||
}
|
||||
descriptors.push_back(desc);
|
||||
return static_cast<u32>(descriptors.size()) - 1;
|
||||
}
|
||||
|
||||
void JoinStorageInfo(Info& base, Info& source) {
|
||||
auto& descriptors = base.storage_buffers_descriptors;
|
||||
for (auto& desc : source.storage_buffers_descriptors) {
|
||||
|
||||
@@ -299,23 +299,128 @@ static inline bool IsTexturePixelFormatIntegerCached(Environment& env,
|
||||
}
|
||||
|
||||
|
||||
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env, const HostTranslateInfo& host_info);
|
||||
static inline std::optional<ConstBufferAddr> TrackCached(const IR::Value& v, Environment& env, const HostTranslateInfo& host_info) {
|
||||
constexpr size_t PHI_TRACK_MAX_DEPTH = 3;
|
||||
|
||||
struct PhiTrackState {
|
||||
boost::container::small_vector<const IR::Inst*, 8> active;
|
||||
size_t depth{};
|
||||
};
|
||||
|
||||
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env,
|
||||
const HostTranslateInfo& host_info, PhiTrackState& state);
|
||||
static inline std::optional<ConstBufferAddr> TrackCached(const IR::Value& v, Environment& env,
|
||||
const HostTranslateInfo& host_info,
|
||||
PhiTrackState& state) {
|
||||
if (const IR::Inst* key = v.InstRecursive()) {
|
||||
if (auto it = env.track_cache.find(key); it != env.track_cache.end()) return it->second;
|
||||
auto found = Track(v, env, host_info);
|
||||
auto found = Track(v, env, host_info, state);
|
||||
if (found) env.track_cache.emplace(key, *found);
|
||||
return found;
|
||||
}
|
||||
return Track(v, env, host_info);
|
||||
return Track(v, env, host_info, state);
|
||||
}
|
||||
|
||||
std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environment& env, const HostTranslateInfo& host_info);
|
||||
std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environment& env,
|
||||
const HostTranslateInfo& host_info,
|
||||
PhiTrackState& state);
|
||||
|
||||
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env, const HostTranslateInfo& host_info) {
|
||||
return IR::BreadthFirstSearch(value, [&env, &host_info](const IR::Inst* inst) {
|
||||
return TryGetConstBuffer(inst, env, host_info);
|
||||
});
|
||||
bool IsSameConstBufferAddr(const ConstBufferAddr& lhs, const ConstBufferAddr& rhs) {
|
||||
return lhs.index == rhs.index && lhs.offset == rhs.offset &&
|
||||
lhs.shift_left == rhs.shift_left && lhs.secondary_index == rhs.secondary_index &&
|
||||
lhs.secondary_offset == rhs.secondary_offset &&
|
||||
lhs.secondary_shift_left == rhs.secondary_shift_left && lhs.count == rhs.count &&
|
||||
lhs.has_secondary == rhs.has_secondary && lhs.dynamic_offset == rhs.dynamic_offset;
|
||||
}
|
||||
|
||||
std::optional<ConstBufferAddr> TrackUncached(const IR::Value& value, Environment& env,
|
||||
const HostTranslateInfo& host_info,
|
||||
PhiTrackState& state, bool& ambiguous);
|
||||
|
||||
std::optional<ConstBufferAddr> TrackPhi(const IR::Inst* phi, Environment& env,
|
||||
const HostTranslateInfo& host_info, PhiTrackState& state,
|
||||
bool& ambiguous) {
|
||||
if (state.depth >= PHI_TRACK_MAX_DEPTH) {
|
||||
ambiguous = true;
|
||||
return std::nullopt;
|
||||
}
|
||||
if (std::ranges::find(state.active, phi) != state.active.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
state.active.push_back(phi);
|
||||
++state.depth;
|
||||
|
||||
std::optional<ConstBufferAddr> agreed;
|
||||
bool failed = false;
|
||||
const size_t num_args{phi->NumArgs()};
|
||||
for (size_t index = 0; index < num_args; ++index) {
|
||||
const IR::Value arg{phi->Arg(index).Resolve()};
|
||||
if (arg.IsImmediate()) {
|
||||
failed = true;
|
||||
break;
|
||||
}
|
||||
const IR::Inst* arg_inst{arg.InstRecursive()};
|
||||
if (arg_inst == phi) {
|
||||
continue;
|
||||
}
|
||||
if (std::ranges::find(state.active, arg_inst) != state.active.end()) {
|
||||
continue;
|
||||
}
|
||||
bool operand_ambiguous = false;
|
||||
const std::optional<ConstBufferAddr> operand{
|
||||
TrackUncached(arg, env, host_info, state, operand_ambiguous)};
|
||||
if (!operand || operand_ambiguous) {
|
||||
failed = true;
|
||||
break;
|
||||
}
|
||||
if (!agreed) {
|
||||
agreed = operand;
|
||||
continue;
|
||||
}
|
||||
if (!IsSameConstBufferAddr(*agreed, *operand)) {
|
||||
failed = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
--state.depth;
|
||||
state.active.pop_back();
|
||||
|
||||
if (failed || !agreed) {
|
||||
ambiguous = true;
|
||||
return std::nullopt;
|
||||
}
|
||||
return agreed;
|
||||
}
|
||||
|
||||
std::optional<ConstBufferAddr> TrackUncached(const IR::Value& value, Environment& env,
|
||||
const HostTranslateInfo& host_info,
|
||||
PhiTrackState& state, bool& ambiguous) {
|
||||
return IR::BreadthFirstSearch(
|
||||
value, [&env, &host_info, &state, &ambiguous](
|
||||
const IR::Inst* inst) -> std::optional<ConstBufferAddr> {
|
||||
if (inst->GetOpcode() == IR::Opcode::Phi) {
|
||||
return TrackPhi(inst, env, host_info, state, ambiguous);
|
||||
}
|
||||
return TryGetConstBuffer(inst, env, host_info, state);
|
||||
});
|
||||
}
|
||||
|
||||
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env,
|
||||
const HostTranslateInfo& host_info, PhiTrackState& state) {
|
||||
if (!Settings::values.enable_shader_phi_tracking.GetValue()) {
|
||||
return IR::BreadthFirstSearch(
|
||||
value, [&env, &host_info, &state](
|
||||
const IR::Inst* inst) -> std::optional<ConstBufferAddr> {
|
||||
return TryGetConstBuffer(inst, env, host_info, state);
|
||||
});
|
||||
}
|
||||
bool ambiguous = false;
|
||||
const std::optional<ConstBufferAddr> result{
|
||||
TrackUncached(value, env, host_info, state, ambiguous)};
|
||||
if (ambiguous) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
std::optional<u32> TryGetConstant(IR::Value& value, Environment& env) {
|
||||
@@ -339,13 +444,15 @@ std::optional<u32> TryGetConstant(IR::Value& value, Environment& env) {
|
||||
return ReadCbufCached(env, index_number, offset_number);
|
||||
}
|
||||
|
||||
std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environment& env, const HostTranslateInfo& host_info) {
|
||||
std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environment& env,
|
||||
const HostTranslateInfo& host_info,
|
||||
PhiTrackState& state) {
|
||||
switch (inst->GetOpcode()) {
|
||||
default:
|
||||
return std::nullopt;
|
||||
case IR::Opcode::BitwiseOr32: {
|
||||
std::optional lhs{TrackCached(inst->Arg(0), env, host_info)};
|
||||
std::optional rhs{TrackCached(inst->Arg(1), env, host_info)};
|
||||
std::optional lhs{TrackCached(inst->Arg(0), env, host_info, state)};
|
||||
std::optional rhs{TrackCached(inst->Arg(1), env, host_info, state)};
|
||||
if (!lhs || !rhs) {
|
||||
return std::nullopt;
|
||||
}
|
||||
@@ -375,7 +482,7 @@ std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environme
|
||||
if (!shift.IsImmediate()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
std::optional lhs{TrackCached(inst->Arg(0), env, host_info)};
|
||||
std::optional lhs{TrackCached(inst->Arg(0), env, host_info, state)};
|
||||
if (lhs) {
|
||||
lhs->shift_left = shift.U32();
|
||||
}
|
||||
@@ -403,7 +510,7 @@ std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environme
|
||||
return std::nullopt;
|
||||
} while (false);
|
||||
}
|
||||
std::optional lhs{TrackCached(op1, env, host_info)};
|
||||
std::optional lhs{TrackCached(op1, env, host_info, state)};
|
||||
if (lhs) {
|
||||
lhs->shift_left = static_cast<u32>(std::countr_zero(op2.U32()));
|
||||
}
|
||||
@@ -469,7 +576,9 @@ std::optional<ConstBufferAddr> TryGetConstBuffer(const IR::Inst* inst, Environme
|
||||
TextureInst MakeInst(Environment& env, IR::Block* block, IR::Inst& inst, const HostTranslateInfo& host_info) {
|
||||
ConstBufferAddr addr;
|
||||
if (IsBindless(inst)) {
|
||||
const std::optional<ConstBufferAddr> track_addr{TrackCached(inst.Arg(0), env, host_info)};
|
||||
PhiTrackState state;
|
||||
const std::optional<ConstBufferAddr> track_addr{
|
||||
TrackCached(inst.Arg(0), env, host_info, state)};
|
||||
|
||||
if (!track_addr) {
|
||||
throw NotImplementedException("Failed to track bindless texture constant buffer");
|
||||
|
||||
@@ -64,7 +64,6 @@ struct Profile {
|
||||
bool support_storage_image_array_nonuniform_indexing{};
|
||||
bool support_uniform_texel_buffer_array_nonuniform_indexing{};
|
||||
bool support_storage_texel_buffer_array_nonuniform_indexing{};
|
||||
bool support_storage_buffer_array_nonuniform_indexing{};
|
||||
|
||||
bool warp_size_potentially_larger_than_guest{};
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <map>
|
||||
@@ -341,10 +340,6 @@ struct Info {
|
||||
ImageDescriptors image_descriptors;
|
||||
};
|
||||
|
||||
[[nodiscard]] inline bool UsesStorageBufferMappings(const Info& info) noexcept {
|
||||
return std::ranges::any_of(info.storage_buffers_descriptors, [](const auto& desc) { return desc.count > 1; });
|
||||
}
|
||||
|
||||
template <typename Descriptors>
|
||||
u32 NumDescriptors(const Descriptors& descriptors) {
|
||||
u32 num{};
|
||||
|
||||
@@ -20,6 +20,7 @@ add_library(video_core STATIC
|
||||
buffer_cache/buffer_cache.h
|
||||
buffer_cache/memory_tracker_base.h
|
||||
buffer_cache/usage_tracker.h
|
||||
buffer_cache/virtual_range_cache.h
|
||||
buffer_cache/word_manager.h
|
||||
cache_types.h
|
||||
capture.h
|
||||
@@ -166,6 +167,8 @@ add_library(video_core STATIC
|
||||
renderer_vulkan/vk_fence_manager.h
|
||||
renderer_vulkan/vk_graphics_pipeline.cpp
|
||||
renderer_vulkan/vk_graphics_pipeline.h
|
||||
renderer_vulkan/vk_multi_range_buffer.cpp
|
||||
renderer_vulkan/vk_multi_range_buffer.h
|
||||
renderer_vulkan/vk_master_semaphore.cpp
|
||||
renderer_vulkan/vk_master_semaphore.h
|
||||
renderer_vulkan/vk_pipeline_cache.cpp
|
||||
|
||||
@@ -7,8 +7,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <memory>
|
||||
#include <numeric>
|
||||
|
||||
@@ -114,6 +112,13 @@ void BufferCache<P>::TickFrame() {
|
||||
async_buffers_death_ring.clear();
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size) {
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}); }) {
|
||||
virtual_ranges.Unmap(as_id, gpu_addr, size);
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) {
|
||||
if (memory_tracker.IsRegionGpuModified(device_addr, size)) {
|
||||
@@ -425,7 +430,7 @@ void BufferCache<P>::UnbindGraphicsStorageBuffers(size_t stage) {
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::BindGraphicsStorageBuffer(size_t stage, size_t ssbo_index, u32 cbuf_index,
|
||||
u32 cbuf_offset, bool is_written, u32 descriptor_count) {
|
||||
u32 cbuf_offset, bool is_written) {
|
||||
const bool already_enabled =
|
||||
((channel_state->enabled_storage_buffers[stage] >> ssbo_index) & 1U) != 0;
|
||||
if constexpr (requires { runtime.ShouldLimitDynamicStorageBuffers(); }) {
|
||||
@@ -450,8 +455,8 @@ bool BufferCache<P>::BindGraphicsStorageBuffer(size_t stage, size_t ssbo_index,
|
||||
const auto& cbufs = maxwell3d->state.shader_stages[stage];
|
||||
const GPUVAddr ssbo_addr = cbufs.const_buffers[cbuf_index].address + cbuf_offset;
|
||||
channel_state->storage_buffers[stage][ssbo_index] =
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written, descriptor_count);
|
||||
return channel_state->storage_buffers[stage][ssbo_index].gpu_addr != 0;
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written);
|
||||
return (channel_state->storage_buffers[stage][ssbo_index].buffer_id != NULL_BUFFER_ID);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -489,7 +494,7 @@ void BufferCache<P>::UnbindComputeStorageBuffers() {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index, u32 cbuf_offset,
|
||||
bool is_written, u32 descriptor_count) {
|
||||
bool is_written) {
|
||||
if (ssbo_index >= channel_state->compute_storage_buffers.size()) [[unlikely]] {
|
||||
LOG_ERROR(HW_GPU, "Storage buffer index {} exceeds maximum storage buffer count",
|
||||
ssbo_index);
|
||||
@@ -526,7 +531,7 @@ void BufferCache<P>::BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index,
|
||||
const auto& cbufs = launch_desc.const_buffer_config;
|
||||
const GPUVAddr ssbo_addr = cbufs[cbuf_index].Address() + cbuf_offset;
|
||||
channel_state->compute_storage_buffers[ssbo_index] =
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written, descriptor_count);
|
||||
StorageBufferBinding(ssbo_addr, cbuf_index, is_written);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -1001,53 +1006,101 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
boost::container::small_vector<u32, NUM_STORAGE_BUFFERS> segment_sizes;
|
||||
bool uses_mapping{};
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
const StorageBufferBindingInfo& binding = channel_state->storage_buffers[stage][index];
|
||||
uses_mapping |= binding.descriptor_count > 1;
|
||||
for (u32 segment = 0; segment < binding.descriptor_count; ++segment) {
|
||||
segment_sizes.push_back(segment < binding.segments.size() ? binding.segments[segment].size : 0);
|
||||
void BufferCache<P>::ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool) {
|
||||
binding.segment_first = 0;
|
||||
binding.segment_count = 0;
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}); }) {
|
||||
if (binding.gpu_addr == 0 || binding.size == 0) {
|
||||
return;
|
||||
}
|
||||
});
|
||||
if (uses_mapping) {
|
||||
const u32 mapping_size = static_cast<u32>(segment_sizes.size() * sizeof(u32));
|
||||
if constexpr (!IS_OPENGL) {
|
||||
const std::span<u8> mapped = runtime.BindMappedStorageBuffer(mapping_size);
|
||||
std::memcpy(mapped.data(), segment_sizes.data(), mapping_size);
|
||||
if (is_written && !runtime.PrefersSparseSources()) {
|
||||
return;
|
||||
}
|
||||
const VirtualSegments* found =
|
||||
virtual_ranges.Query(*gpu_memory, binding.gpu_addr, binding.size);
|
||||
if (!found || found->size() < 2) {
|
||||
return;
|
||||
}
|
||||
const VirtualSegments segments = *found;
|
||||
const u32 first = static_cast<u32>(pool.size());
|
||||
const bool prefer_sparse = runtime.PrefersSparseSources();
|
||||
for (const VirtualSegment& segment : segments) {
|
||||
const BufferId buffer_id =
|
||||
FindBuffer(segment.device_addr, segment.size, prefer_sparse);
|
||||
if (!buffer_id) {
|
||||
pool.resize(first);
|
||||
return;
|
||||
}
|
||||
pool.push_back(MultiRangeSegment{
|
||||
.buffer_id = buffer_id,
|
||||
.device_addr = segment.device_addr,
|
||||
.size = segment.size,
|
||||
});
|
||||
}
|
||||
binding.segment_first = first;
|
||||
binding.segment_count = static_cast<u32>(segments.size());
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::BindMultiRangeStorage(const Binding& binding, bool is_written,
|
||||
std::span<const MultiRangeSegment> pool) {
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}); }) {
|
||||
if (binding.segment_count < 2) {
|
||||
return false;
|
||||
}
|
||||
if (binding.segment_first + binding.segment_count > pool.size()) {
|
||||
return false;
|
||||
}
|
||||
const u64 key = (static_cast<u64>(gpu_memory->GetID()) << 48) ^ binding.gpu_addr;
|
||||
runtime.ResetMultiRange();
|
||||
for (u32 index = 0; index < binding.segment_count; ++index) {
|
||||
const MultiRangeSegment& segment = pool[binding.segment_first + index];
|
||||
Buffer& buffer = slot_buffers[segment.buffer_id];
|
||||
TouchBuffer(buffer, segment.buffer_id);
|
||||
if (SynchronizeBuffer(buffer, segment.device_addr, segment.size)) {
|
||||
runtime.InvalidateMultiRange(key);
|
||||
}
|
||||
const u32 offset = buffer.Offset(segment.device_addr);
|
||||
buffer.MarkUsage(offset, segment.size);
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(segment.buffer_id, segment.device_addr, segment.size);
|
||||
}
|
||||
runtime.PushMultiRangeSource(buffer, offset, segment.size);
|
||||
}
|
||||
return runtime.BindMultiRangeStorageBuffer(key);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
u32 binding_index = 0;
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
const StorageBufferBindingInfo& storage = channel_state->storage_buffers[stage][index];
|
||||
const Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
for (u32 segment = 0; segment < storage.descriptor_count; ++segment) {
|
||||
Buffer* buffer = &slot_buffers[NULL_BUFFER_ID];
|
||||
u32 offset{};
|
||||
u32 size{IS_OPENGL ? 0U : static_cast<u32>(sizeof(u32))};
|
||||
const bool is_actual_segment = segment < storage.segments.size();
|
||||
// shall be safe enough if the segment is not actual, use the last available segment or nullptr if none exist.
|
||||
const Binding* binding = is_actual_segment ? &storage.segments[segment] : storage.segments.empty() ? nullptr : &storage.segments.back();
|
||||
if (binding) {
|
||||
buffer = &slot_buffers[binding->buffer_id];
|
||||
size = binding->size;
|
||||
offset = buffer->Offset(binding->device_addr);
|
||||
if (is_actual_segment) {
|
||||
TouchBuffer(*buffer, binding->buffer_id);
|
||||
SynchronizeBuffer(*buffer, binding->device_addr, size);
|
||||
buffer->MarkUsage(offset, size);
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding->buffer_id, binding->device_addr, size);
|
||||
}
|
||||
}
|
||||
}
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindStorageBuffer(stage, binding_index++, *buffer, offset, size, is_written);
|
||||
} else {
|
||||
runtime.BindStorageBuffer(*buffer, offset, size, is_written);
|
||||
}
|
||||
if (BindMultiRangeStorage(binding, is_written, graphics_segments)) {
|
||||
return;
|
||||
}
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
|
||||
const u32 offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
}
|
||||
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindStorageBuffer(stage, binding_index, buffer, offset, size, is_written);
|
||||
++binding_index;
|
||||
} else {
|
||||
runtime.BindStorageBuffer(buffer, offset, size, is_written);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1163,53 +1216,31 @@ void BufferCache<P>::BindHostComputeUniformBuffers() {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostComputeStorageBuffers() {
|
||||
boost::container::small_vector<u32, NUM_STORAGE_BUFFERS> segment_sizes;
|
||||
bool uses_mapping{};
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
const StorageBufferBindingInfo& binding = channel_state->compute_storage_buffers[index];
|
||||
uses_mapping |= binding.descriptor_count > 1;
|
||||
for (u32 segment = 0; segment < binding.descriptor_count; ++segment) {
|
||||
segment_sizes.push_back(segment < binding.segments.size() ? binding.segments[segment].size : 0);
|
||||
}
|
||||
});
|
||||
if (uses_mapping) {
|
||||
const u32 mapping_size = static_cast<u32>(segment_sizes.size() * sizeof(u32));
|
||||
if constexpr (!IS_OPENGL) {
|
||||
const std::span<u8> mapped = runtime.BindMappedStorageBuffer(mapping_size);
|
||||
std::memcpy(mapped.data(), segment_sizes.data(), mapping_size);
|
||||
}
|
||||
}
|
||||
|
||||
u32 binding_index = 0;
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
const StorageBufferBindingInfo& storage = channel_state->compute_storage_buffers[index];
|
||||
const Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
for (u32 segment = 0; segment < storage.descriptor_count; ++segment) {
|
||||
Buffer* buffer = &slot_buffers[NULL_BUFFER_ID];
|
||||
u32 offset{};
|
||||
u32 size{IS_OPENGL ? 0U : static_cast<u32>(sizeof(u32))};
|
||||
const bool is_actual_segment = segment < storage.segments.size();
|
||||
//same fallback logic
|
||||
const Binding* binding = is_actual_segment ? &storage.segments[segment] : storage.segments.empty() ? nullptr : &storage.segments.back();
|
||||
if (binding) {
|
||||
buffer = &slot_buffers[binding->buffer_id];
|
||||
size = binding->size;
|
||||
offset = buffer->Offset(binding->device_addr);
|
||||
if (is_actual_segment) {
|
||||
TouchBuffer(*buffer, binding->buffer_id);
|
||||
SynchronizeBuffer(*buffer, binding->device_addr, size);
|
||||
buffer->MarkUsage(offset, size);
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding->buffer_id, binding->device_addr, size);
|
||||
}
|
||||
}
|
||||
}
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindComputeStorageBuffer(binding_index++, *buffer, offset, size, is_written);
|
||||
} else {
|
||||
runtime.BindStorageBuffer(*buffer, offset, size, is_written);
|
||||
}
|
||||
if (BindMultiRangeStorage(binding, is_written, compute_segments)) {
|
||||
return;
|
||||
}
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
|
||||
const u32 offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
|
||||
if (is_written) {
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
}
|
||||
|
||||
if constexpr (NEEDS_BIND_STORAGE_INDEX) {
|
||||
runtime.BindComputeStorageBuffer(binding_index, buffer, offset, size, is_written);
|
||||
++binding_index;
|
||||
} else {
|
||||
runtime.BindStorageBuffer(buffer, offset, size, is_written);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1245,6 +1276,7 @@ void BufferCache<P>::BindHostComputeTextureBuffers() {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) {
|
||||
graphics_segments.clear();
|
||||
BufferOperations([&]() {
|
||||
if (is_indexed) {
|
||||
UpdateIndexBuffer();
|
||||
@@ -1264,6 +1296,7 @@ void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::DoUpdateComputeBuffers() {
|
||||
compute_segments.clear();
|
||||
BufferOperations([&]() {
|
||||
UpdateComputeUniformBuffers();
|
||||
UpdateComputeStorageBuffers();
|
||||
@@ -1402,7 +1435,12 @@ void BufferCache<P>::UpdateUniformBuffers(size_t stage) {
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateStorageBuffers(size_t stage) {
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
UpdateStorageBuffer(channel_state->storage_buffers[stage][index]);
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
binding.buffer_id = buffer_id;
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, is_written, graphics_segments);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1463,54 +1501,15 @@ void BufferCache<P>::UpdateComputeUniformBuffers() {
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateComputeStorageBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
UpdateStorageBuffer(channel_state->compute_storage_buffers[index]);
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, is_written, compute_segments);
|
||||
});
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateStorageBuffer(StorageBufferBindingInfo& binding) {
|
||||
binding.segments.clear();
|
||||
if (binding.gpu_addr == 0 || binding.size == 0) { return;}
|
||||
if (binding.descriptor_count == 1) {
|
||||
// for safety gotta preserve the legacy path on possible hosts without storage-buffer descriptor indexing.
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(binding.gpu_addr);
|
||||
if (device_addr) {
|
||||
binding.segments.push_back(Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = binding.size,
|
||||
.buffer_id = FindBuffer(*device_addr, binding.size),
|
||||
});
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const auto ranges = gpu_memory->GetSubmappedRange(binding.gpu_addr, binding.size);
|
||||
const size_t mapped_size =
|
||||
std::accumulate(ranges.begin(), ranges.end(), size_t{}, [](size_t total, const auto& range) { return total + range.second; });
|
||||
if (mapped_size != binding.size) {
|
||||
LOG_ERROR(HW_GPU, "Storage buffer range {:#x}+{:#x} is not fully mapped", binding.gpu_addr, binding.size);
|
||||
return;
|
||||
}
|
||||
if (ranges.size() > binding.descriptor_count) {
|
||||
LOG_ERROR(HW_GPU, "Storage buffer range {:#x}+{:#x} has {} physical segments, exceeding host capacity {}",
|
||||
binding.gpu_addr, binding.size, ranges.size(), binding.descriptor_count);
|
||||
return;
|
||||
}
|
||||
for (const auto& [gpu_addr, size] : ranges) {
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(gpu_addr);
|
||||
if (!device_addr || size > (std::numeric_limits<u32>::max)()) {
|
||||
binding.segments.clear();
|
||||
return;
|
||||
}
|
||||
const u32 segment_size = static_cast<u32>(size);
|
||||
binding.segments.push_back(Binding{
|
||||
.device_addr = *device_addr,
|
||||
.size = segment_size,
|
||||
.buffer_id = FindBuffer(*device_addr, segment_size),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateComputeTextureBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_texture_buffers, [&](u32 index) {
|
||||
@@ -1531,7 +1530,7 @@ void BufferCache<P>::MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u3
|
||||
}
|
||||
|
||||
template <class P>
|
||||
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
|
||||
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible) {
|
||||
if (device_addr == 0) {
|
||||
return NULL_BUFFER_ID;
|
||||
}
|
||||
@@ -1541,10 +1540,18 @@ BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
WaitForGpuFenceIfNeeded(buffer);
|
||||
if (buffer.IsInBounds(device_addr, size)) {
|
||||
return buffer_id;
|
||||
bool usable = true;
|
||||
if constexpr (requires { buffer.IsSparseCompatible(); }) {
|
||||
if (sparse_compatible && !buffer.IsSparseCompatible()) {
|
||||
usable = false;
|
||||
}
|
||||
}
|
||||
if (usable) {
|
||||
return buffer_id;
|
||||
}
|
||||
}
|
||||
}
|
||||
return CreateBuffer(device_addr, size);
|
||||
return CreateBuffer(device_addr, size, sparse_compatible);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -1666,13 +1673,15 @@ void BufferCache<P>::JoinOverlap(BufferId new_buffer_id, BufferId overlap_id,
|
||||
}
|
||||
|
||||
template <class P>
|
||||
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size) {
|
||||
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||
bool sparse_compatible) {
|
||||
DAddr device_addr_end = Common::AlignUp(device_addr + wanted_size, CACHING_PAGESIZE);
|
||||
device_addr = Common::AlignDown(device_addr, CACHING_PAGESIZE);
|
||||
wanted_size = static_cast<u32>(device_addr_end - device_addr);
|
||||
const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size);
|
||||
const u32 size = static_cast<u32>(overlap.end - overlap.begin);
|
||||
const BufferId new_buffer_id = slot_buffers.insert(runtime, overlap.begin, size);
|
||||
const BufferId new_buffer_id =
|
||||
slot_buffers.insert(runtime, overlap.begin, size, sparse_compatible);
|
||||
auto& new_buffer = slot_buffers[new_buffer_id];
|
||||
const size_t size_bytes = new_buffer.SizeBytes();
|
||||
runtime.ClearBuffer(new_buffer, 0, size_bytes, 0);
|
||||
@@ -1922,6 +1931,9 @@ void BufferCache<P>::DownloadBufferMemory(Buffer& buffer, DAddr device_addr, u64
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
if constexpr (requires { runtime.OnBufferDeleted(slot_buffers[buffer_id]); }) {
|
||||
runtime.OnBufferDeleted(slot_buffers[buffer_id]);
|
||||
}
|
||||
bool dirty_index{false};
|
||||
boost::container::small_vector<u64, NUM_VERTEX_BUFFERS> dirty_vertex_buffers;
|
||||
const auto scalar_replace = [buffer_id](Binding& binding) {
|
||||
@@ -1932,11 +1944,6 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
const auto replace = [scalar_replace](std::span<Binding> bindings) {
|
||||
std::ranges::for_each(bindings, scalar_replace);
|
||||
};
|
||||
const auto storage_replace = [scalar_replace](std::span<StorageBufferBindingInfo> bindings) {
|
||||
for (StorageBufferBindingInfo& binding : bindings) {
|
||||
std::ranges::for_each(binding.segments, scalar_replace);
|
||||
}
|
||||
};
|
||||
|
||||
if (channel_state->index_buffer.buffer_id == buffer_id) {
|
||||
channel_state->index_buffer.buffer_id = BufferId{};
|
||||
@@ -1952,10 +1959,10 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
}
|
||||
}
|
||||
std::ranges::for_each(channel_state->uniform_buffers, replace);
|
||||
std::ranges::for_each(channel_state->storage_buffers, storage_replace);
|
||||
std::ranges::for_each(channel_state->storage_buffers, replace);
|
||||
replace(channel_state->transform_feedback_buffers);
|
||||
replace(channel_state->compute_uniform_buffers);
|
||||
storage_replace(channel_state->compute_storage_buffers);
|
||||
replace(channel_state->compute_storage_buffers);
|
||||
|
||||
// Mark the whole buffer as CPU written to stop tracking CPU writes
|
||||
if (!do_not_mark) {
|
||||
@@ -1992,14 +1999,12 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
}
|
||||
|
||||
template <class P>
|
||||
StorageBufferBindingInfo BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written, u32 descriptor_count) const {
|
||||
// time to get rid of these null bindings
|
||||
ASSERT(descriptor_count > 0); // shant happen
|
||||
Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written) const {
|
||||
const GPUVAddr gpu_addr = gpu_memory->Read<u64>(ssbo_addr);
|
||||
|
||||
if (gpu_addr == 0) {
|
||||
return {.descriptor_count = descriptor_count};
|
||||
return NULL_BINDING;
|
||||
}
|
||||
|
||||
const auto size = [&]() {
|
||||
@@ -2021,17 +2026,26 @@ StorageBufferBindingInfo BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr
|
||||
const GPUVAddr aligned_gpu_addr = Common::AlignDown(gpu_addr, alignment);
|
||||
const u32 aligned_size = static_cast<u32>(gpu_addr - aligned_gpu_addr) + size;
|
||||
|
||||
if (!gpu_memory->GpuToCpuAddress(aligned_gpu_addr) || size == 0) {
|
||||
const std::optional<DAddr> aligned_device_addr = gpu_memory->GpuToCpuAddress(aligned_gpu_addr);
|
||||
if (!aligned_device_addr || size == 0) {
|
||||
LOG_DEBUG(HW_GPU, "Failed to find storage buffer for cbuf index {}", cbuf_index);
|
||||
return {.descriptor_count = descriptor_count};
|
||||
return NULL_BINDING;
|
||||
}
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(gpu_addr);
|
||||
ASSERT_MSG(device_addr, "Unaligned storage buffer address not found for cbuf index {}",
|
||||
cbuf_index);
|
||||
// The end address used for size calculation does not need to be aligned
|
||||
const GPUVAddr gpu_end = Common::AlignUp(gpu_addr + size, Core::DEVICE_PAGESIZE);
|
||||
const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE);
|
||||
|
||||
const StorageBufferBindingInfo binding{
|
||||
u32 binding_size = static_cast<u32>(cpu_end - *aligned_device_addr);
|
||||
if (is_written) {
|
||||
binding_size = aligned_size;
|
||||
}
|
||||
const Binding binding{
|
||||
.device_addr = *aligned_device_addr,
|
||||
.size = binding_size,
|
||||
.buffer_id = BufferId{},
|
||||
.gpu_addr = aligned_gpu_addr,
|
||||
.size = is_written ? aligned_size : static_cast<u32>(gpu_end - aligned_gpu_addr),
|
||||
.descriptor_count = descriptor_count,
|
||||
};
|
||||
return binding;
|
||||
}
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
#include "common/settings.h"
|
||||
#include "common/slot_vector.h"
|
||||
#include "video_core/buffer_cache/buffer_base.h"
|
||||
#include "video_core/buffer_cache/virtual_range_cache.h"
|
||||
#include "video_core/control/channel_state_cache.h"
|
||||
#include "video_core/delayed_destruction_ring.h"
|
||||
#include "video_core/dirty_flags.h"
|
||||
@@ -83,21 +84,21 @@ struct Binding {
|
||||
DAddr device_addr{};
|
||||
u32 size{};
|
||||
BufferId buffer_id;
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 segment_first{};
|
||||
u32 segment_count{};
|
||||
};
|
||||
|
||||
struct MultiRangeSegment {
|
||||
BufferId buffer_id;
|
||||
DAddr device_addr{};
|
||||
u32 size{};
|
||||
};
|
||||
|
||||
struct TextureBufferBinding : Binding {
|
||||
PixelFormat format;
|
||||
};
|
||||
|
||||
struct StorageBufferBindingInfo {
|
||||
// another good one: guest SSBO is a virtual interval and may span discontiguous device-memory ranges.
|
||||
// exact case of missing character frames (high sample lane)
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 size{};
|
||||
u32 descriptor_count{1};
|
||||
boost::container::small_vector<Binding, 1> segments;
|
||||
};
|
||||
|
||||
static constexpr Binding NULL_BINDING{
|
||||
.device_addr = 0,
|
||||
.size = 0,
|
||||
@@ -124,15 +125,14 @@ public:
|
||||
Binding index_buffer;
|
||||
std::array<Binding, NUM_VERTEX_BUFFERS> vertex_buffers;
|
||||
std::array<std::array<Binding, NUM_GRAPHICS_UNIFORM_BUFFERS>, NUM_STAGES> uniform_buffers;
|
||||
std::array<std::array<StorageBufferBindingInfo, NUM_STORAGE_BUFFERS>, NUM_STAGES>
|
||||
storage_buffers;
|
||||
std::array<std::array<Binding, NUM_STORAGE_BUFFERS>, NUM_STAGES> storage_buffers;
|
||||
std::array<std::array<TextureBufferBinding, NUM_TEXTURE_BUFFERS>, NUM_STAGES> texture_buffers;
|
||||
std::array<Binding, NUM_TRANSFORM_FEEDBACK_BUFFERS> transform_feedback_buffers;
|
||||
Binding count_buffer_binding;
|
||||
Binding indirect_buffer_binding;
|
||||
|
||||
std::array<Binding, NUM_COMPUTE_UNIFORM_BUFFERS> compute_uniform_buffers;
|
||||
std::array<StorageBufferBindingInfo, NUM_STORAGE_BUFFERS> compute_storage_buffers;
|
||||
std::array<Binding, NUM_STORAGE_BUFFERS> compute_storage_buffers;
|
||||
std::array<TextureBufferBinding, NUM_TEXTURE_BUFFERS> compute_texture_buffers;
|
||||
|
||||
std::array<u32, NUM_STAGES> enabled_uniform_buffer_masks{};
|
||||
@@ -225,6 +225,14 @@ public:
|
||||
|
||||
void TickFrame();
|
||||
|
||||
bool BindMultiRangeStorage(const Binding& binding, bool is_written,
|
||||
std::span<const MultiRangeSegment> pool);
|
||||
|
||||
void ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool);
|
||||
|
||||
void UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size);
|
||||
|
||||
void WriteMemory(DAddr device_addr, u64 size);
|
||||
|
||||
void CachedWriteMemory(DAddr device_addr, u64 size);
|
||||
@@ -259,7 +267,7 @@ public:
|
||||
void UnbindGraphicsStorageBuffers(size_t stage);
|
||||
|
||||
bool BindGraphicsStorageBuffer(size_t stage, size_t ssbo_index, u32 cbuf_index, u32 cbuf_offset,
|
||||
bool is_written, u32 descriptor_count = 1);
|
||||
bool is_written);
|
||||
|
||||
void UnbindGraphicsTextureBuffers(size_t stage);
|
||||
|
||||
@@ -269,7 +277,7 @@ public:
|
||||
void UnbindComputeStorageBuffers();
|
||||
|
||||
void BindComputeStorageBuffer(size_t ssbo_index, u32 cbuf_index, u32 cbuf_offset,
|
||||
bool is_written, u32 descriptor_count = 1);
|
||||
bool is_written);
|
||||
|
||||
void UnbindComputeTextureBuffers();
|
||||
|
||||
@@ -410,8 +418,6 @@ private:
|
||||
|
||||
void UpdateStorageBuffers(size_t stage);
|
||||
|
||||
void UpdateStorageBuffer(StorageBufferBindingInfo& binding);
|
||||
|
||||
void UpdateTextureBuffers(size_t stage);
|
||||
|
||||
void UpdateTransformFeedbackBuffers();
|
||||
@@ -426,7 +432,8 @@ private:
|
||||
|
||||
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size);
|
||||
|
||||
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size);
|
||||
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size,
|
||||
bool sparse_compatible = false);
|
||||
|
||||
void WaitForGpuFenceIfNeeded(Buffer& buffer);
|
||||
|
||||
@@ -434,7 +441,8 @@ private:
|
||||
|
||||
void JoinOverlap(BufferId new_buffer_id, BufferId overlap_id, bool accumulate_stream_score);
|
||||
|
||||
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size);
|
||||
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||
bool sparse_compatible = false);
|
||||
|
||||
void Register(BufferId buffer_id);
|
||||
|
||||
@@ -461,9 +469,8 @@ private:
|
||||
|
||||
void DeleteBuffer(BufferId buffer_id, bool do_not_mark = false);
|
||||
|
||||
[[nodiscard]] StorageBufferBindingInfo StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written,
|
||||
u32 descriptor_count) const;
|
||||
[[nodiscard]] Binding StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
|
||||
bool is_written) const;
|
||||
|
||||
[[nodiscard]] TextureBufferBinding GetTextureBufferBinding(GPUVAddr gpu_addr, u32 size,
|
||||
PixelFormat format);
|
||||
@@ -527,6 +534,9 @@ private:
|
||||
};
|
||||
Common::LeastRecentlyUsedCache<LRUItemParams> lru_cache;
|
||||
u64 frame_tick = 0;
|
||||
VirtualRangeCache virtual_ranges;
|
||||
std::vector<MultiRangeSegment> graphics_segments;
|
||||
std::vector<MultiRangeSegment> compute_segments;
|
||||
u64 total_used_memory = 0;
|
||||
u64 minimum_memory = 0;
|
||||
u64 critical_memory = 0;
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <limits>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "common/common_types.h"
|
||||
#include "video_core/memory_manager.h"
|
||||
|
||||
namespace VideoCommon {
|
||||
|
||||
struct VirtualSegment {
|
||||
GPUVAddr gpu_addr;
|
||||
DAddr device_addr;
|
||||
u32 size;
|
||||
};
|
||||
|
||||
using VirtualSegments = boost::container::small_vector<VirtualSegment, 8>;
|
||||
|
||||
class VirtualRangeCache {
|
||||
public:
|
||||
const VirtualSegments* Query(Tegra::MemoryManager& memory, GPUVAddr gpu_addr, u32 size) {
|
||||
if (has_deferred.load(std::memory_order_acquire)) {
|
||||
ApplyDeferred();
|
||||
}
|
||||
const size_t as_id = memory.GetID();
|
||||
const u64 key = MakeKey(as_id, gpu_addr);
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end() && it->second.as_id == as_id &&
|
||||
it->second.gpu_addr == gpu_addr && it->second.size == size) {
|
||||
return &it->second.segments;
|
||||
}
|
||||
Entry entry;
|
||||
entry.as_id = as_id;
|
||||
entry.gpu_addr = gpu_addr;
|
||||
entry.size = size;
|
||||
const auto ranges = memory.GetSubmappedRange(gpu_addr, size);
|
||||
GPUVAddr expected = gpu_addr;
|
||||
bool contiguous = true;
|
||||
for (const auto& [range_addr, range_size] : ranges) {
|
||||
if (range_addr != expected || range_size == 0) {
|
||||
contiguous = false;
|
||||
break;
|
||||
}
|
||||
const std::optional<DAddr> device_addr = memory.GpuToCpuAddress(range_addr);
|
||||
if (!device_addr || *device_addr == 0) {
|
||||
contiguous = false;
|
||||
break;
|
||||
}
|
||||
if (range_size > static_cast<size_t>((std::numeric_limits<u32>::max)())) {
|
||||
contiguous = false;
|
||||
break;
|
||||
}
|
||||
entry.segments.push_back(VirtualSegment{
|
||||
.gpu_addr = range_addr,
|
||||
.device_addr = *device_addr,
|
||||
.size = static_cast<u32>(range_size),
|
||||
});
|
||||
expected += range_size;
|
||||
}
|
||||
if (!contiguous || expected != gpu_addr + size) {
|
||||
entry.segments.clear();
|
||||
}
|
||||
const auto result = entries.insert_or_assign(key, std::move(entry));
|
||||
return &result.first->second.segments;
|
||||
}
|
||||
|
||||
void Unmap(size_t as_id, GPUVAddr gpu_addr, u64 size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
}
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
if (!deferred.empty()) {
|
||||
DeferredUnmap& last = deferred.back();
|
||||
if (last.as_id == as_id && last.gpu_addr + last.size == gpu_addr) {
|
||||
last.size += size;
|
||||
has_deferred.store(true, std::memory_order_release);
|
||||
return;
|
||||
}
|
||||
}
|
||||
deferred.push_back(DeferredUnmap{
|
||||
.as_id = as_id,
|
||||
.gpu_addr = gpu_addr,
|
||||
.size = size,
|
||||
});
|
||||
}
|
||||
has_deferred.store(true, std::memory_order_release);
|
||||
}
|
||||
|
||||
void Clear() {
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
deferred.clear();
|
||||
}
|
||||
has_deferred.store(false, std::memory_order_release);
|
||||
entries.clear();
|
||||
}
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
size_t as_id{};
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 size{};
|
||||
VirtualSegments segments;
|
||||
};
|
||||
|
||||
struct DeferredUnmap {
|
||||
size_t as_id;
|
||||
GPUVAddr gpu_addr;
|
||||
u64 size;
|
||||
};
|
||||
|
||||
static u64 MakeKey(size_t as_id, GPUVAddr gpu_addr) {
|
||||
return (static_cast<u64>(as_id) << 48) ^ gpu_addr;
|
||||
}
|
||||
|
||||
void ApplyDeferred() {
|
||||
std::vector<DeferredUnmap> pending;
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
has_deferred.store(false, std::memory_order_release);
|
||||
pending.swap(deferred);
|
||||
}
|
||||
if (pending.empty() || entries.empty()) {
|
||||
return;
|
||||
}
|
||||
for (auto it = entries.begin(); it != entries.end();) {
|
||||
const Entry& entry = it->second;
|
||||
const GPUVAddr entry_end = entry.gpu_addr + entry.size;
|
||||
bool overlaps = false;
|
||||
for (const DeferredUnmap& unmap : pending) {
|
||||
if (unmap.as_id != entry.as_id) {
|
||||
continue;
|
||||
}
|
||||
if (entry.gpu_addr < unmap.gpu_addr + unmap.size && unmap.gpu_addr < entry_end) {
|
||||
overlaps = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (overlaps) {
|
||||
it = entries.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unordered_map<u64, Entry> entries;
|
||||
std::vector<DeferredUnmap> deferred;
|
||||
std::mutex deferred_mutex;
|
||||
std::atomic<bool> has_deferred{false};
|
||||
};
|
||||
|
||||
} // namespace VideoCommon
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include <array>
|
||||
#include <vector>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/bit_field.h"
|
||||
#include "common/common_funcs.h"
|
||||
|
||||
@@ -1,8 +1,15 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/bit_field.h"
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
#include <type_traits>
|
||||
|
||||
#include "common/common_types.h"
|
||||
#include "common/scratch_buffer.h"
|
||||
|
||||
@@ -52,7 +52,7 @@ constexpr std::array PROGRAM_LUT{
|
||||
Buffer::Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params)
|
||||
: VideoCommon::BufferBase(null_params) {}
|
||||
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_, bool)
|
||||
: VideoCommon::BufferBase(cpu_addr_, size_bytes_) {
|
||||
buffer.Create();
|
||||
if (runtime.device.HasDebuggingToolAttached()) {
|
||||
|
||||
@@ -23,7 +23,8 @@ class BufferCacheRuntime;
|
||||
|
||||
class Buffer : public VideoCommon::BufferBase {
|
||||
public:
|
||||
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes);
|
||||
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes,
|
||||
bool sparse_compatible = false);
|
||||
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams);
|
||||
|
||||
void ImmediateUpload(size_t offset, std::span<const u8> data) noexcept;
|
||||
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <optional>
|
||||
|
||||
@@ -137,7 +136,6 @@ inline void WriteDescriptorBuffer(const Device& device, const DescriptorBufferLa
|
||||
|
||||
[[nodiscard]] inline u32 NumDescriptorEntries(const Shader::Info& info) {
|
||||
return Shader::NumDescriptors(info.constant_buffer_descriptors) +
|
||||
static_cast<u32>(Shader::UsesStorageBufferMappings(info)) +
|
||||
Shader::NumDescriptors(info.storage_buffers_descriptors) +
|
||||
Shader::NumDescriptors(info.texture_buffer_descriptors) +
|
||||
Shader::NumDescriptors(info.image_buffer_descriptors) +
|
||||
@@ -270,12 +268,6 @@ public:
|
||||
is_compute |= (stage & VK_SHADER_STAGE_COMPUTE_BIT) != 0;
|
||||
|
||||
Add(VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER, stage, info.constant_buffer_descriptors);
|
||||
// for extra implicit storage-buffer binding required by mapped-storage-buffer support
|
||||
if (Shader::UsesStorageBufferMappings(info)) {
|
||||
struct Descriptor { u32 count; };
|
||||
const std::array descriptors{Descriptor{1}};
|
||||
Add(VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, stage, descriptors);
|
||||
}
|
||||
Add(VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, stage, info.storage_buffers_descriptors);
|
||||
Add(VK_DESCRIPTOR_TYPE_UNIFORM_TEXEL_BUFFER, stage, info.texture_buffer_descriptors);
|
||||
Add(VK_DESCRIPTOR_TYPE_STORAGE_TEXEL_BUFFER, stage, info.image_buffer_descriptors);
|
||||
|
||||
@@ -56,7 +56,8 @@ size_t BytesPerIndex(VkIndexType index_type) {
|
||||
}
|
||||
}
|
||||
|
||||
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size) {
|
||||
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size,
|
||||
VkDeviceSize sparse_alignment) {
|
||||
VkBufferUsageFlags flags =
|
||||
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||
VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
|
||||
@@ -82,6 +83,9 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
if (sparse_alignment > 1) {
|
||||
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal, sparse_alignment);
|
||||
}
|
||||
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal);
|
||||
}
|
||||
} // Anonymous namespace
|
||||
@@ -99,10 +103,14 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
|
||||
}
|
||||
}
|
||||
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
|
||||
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_,
|
||||
bool sparse_compatible_)
|
||||
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
|
||||
scheduler{&runtime.scheduler},
|
||||
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
|
||||
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes(),
|
||||
runtime.SparseAlignmentFor(sparse_compatible_))},
|
||||
tracker{SizeBytes()} {
|
||||
sparse_compatible = sparse_compatible_;
|
||||
if (runtime.device.HasDebuggingToolAttached()) {
|
||||
buffer.SetObjectNameEXT(fmt::format("Buffer {:#x}", CpuAddr()).c_str());
|
||||
}
|
||||
@@ -348,7 +356,8 @@ BufferCacheRuntime::BufferCacheRuntime(const Device& device_, MemoryAllocator& m
|
||||
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_},
|
||||
staging_pool{staging_pool_}, guest_descriptor_queue{guest_descriptor_queue_},
|
||||
quad_index_pass(device, scheduler, descriptor_pool, staging_pool,
|
||||
compute_pass_descriptor_queue) {
|
||||
compute_pass_descriptor_queue),
|
||||
multi_range_buffers(device_, memory_allocator_, scheduler_) {
|
||||
const VkDriverIdKHR driver_id = device.GetDriverID();
|
||||
limit_dynamic_storage_buffers = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY ||
|
||||
driver_id == VK_DRIVER_ID_ARM_PROPRIETARY;
|
||||
@@ -536,6 +545,33 @@ void BufferCacheRuntime::ClearBuffer(VkBuffer dest_buffer, u32 offset, size_t si
|
||||
});
|
||||
}
|
||||
|
||||
bool BufferCacheRuntime::BindMultiRangeStorageBuffer(u64 key) {
|
||||
if (multi_range_sources.empty() || multi_range_total == 0) {
|
||||
return false;
|
||||
}
|
||||
const MultiRangeRef ref = multi_range_buffers.Get(key, multi_range_sources, multi_range_total);
|
||||
if (ref.handle == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
if (ref.needs_gather) {
|
||||
PreCopyBarrier();
|
||||
VkDeviceSize dst_offset = 0;
|
||||
for (const MultiRangeSource& source : multi_range_sources) {
|
||||
const std::array<VideoCommon::BufferCopy, 1> copy{VideoCommon::BufferCopy{
|
||||
.src_offset = static_cast<u64>(source.offset),
|
||||
.dst_offset = static_cast<u64>(dst_offset),
|
||||
.size = static_cast<size_t>(source.size),
|
||||
}};
|
||||
CopyBuffer(ref.handle, source.handle, copy, false);
|
||||
dst_offset += source.size;
|
||||
}
|
||||
PostCopyBarrier();
|
||||
multi_range_buffers.MarkGathered(key);
|
||||
}
|
||||
guest_descriptor_queue.AddBuffer(ref.handle, ref.address, 0, ref.size);
|
||||
return true;
|
||||
}
|
||||
|
||||
void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format,
|
||||
u32 base_vertex, u32 num_indices, VkBuffer buffer,
|
||||
u32 offset, [[maybe_unused]] u32 size) {
|
||||
@@ -695,7 +731,7 @@ vk::Buffer BufferCacheRuntime::CreateNullBuffer() {
|
||||
.flags = 0,
|
||||
.size = 4,
|
||||
.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
|
||||
@@ -8,11 +8,14 @@
|
||||
|
||||
#include <limits>
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "video_core/buffer_cache/buffer_cache_base.h"
|
||||
#include "video_core/buffer_cache/memory_tracker_base.h"
|
||||
#include "video_core/buffer_cache/usage_tracker.h"
|
||||
#include "video_core/engines/maxwell_3d.h"
|
||||
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
||||
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
|
||||
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
|
||||
#include "video_core/renderer_vulkan/vk_update_descriptor.h"
|
||||
#include "video_core/surface.h"
|
||||
@@ -31,7 +34,8 @@ class BufferCacheRuntime;
|
||||
class Buffer : public VideoCommon::BufferBase {
|
||||
public:
|
||||
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
|
||||
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_);
|
||||
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_,
|
||||
bool sparse_compatible_ = false);
|
||||
|
||||
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
|
||||
|
||||
@@ -43,6 +47,14 @@ public:
|
||||
return device_address;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsSparseCompatible() const noexcept {
|
||||
return sparse_compatible;
|
||||
}
|
||||
|
||||
[[nodiscard]] vk::MemoryLocation Location() const noexcept {
|
||||
return buffer.Location();
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsRegionUsed(u64 offset, u64 size) const noexcept {
|
||||
return tracker.IsUsed(offset, size);
|
||||
}
|
||||
@@ -77,6 +89,7 @@ private:
|
||||
VkDeviceAddress device_address{};
|
||||
u64 last_usage_tick{};
|
||||
bool is_null{};
|
||||
bool sparse_compatible{};
|
||||
};
|
||||
|
||||
class QuadArrayIndexBuffer;
|
||||
@@ -125,7 +138,7 @@ public:
|
||||
|
||||
void PreCopyBarrier();
|
||||
|
||||
void CopyBuffer(VkBuffer src_buffer, VkBuffer dst_buffer,
|
||||
void CopyBuffer(VkBuffer dst_buffer, VkBuffer src_buffer,
|
||||
std::span<const VideoCommon::BufferCopy> copies, bool barrier,
|
||||
bool can_reorder_upload = false);
|
||||
|
||||
@@ -155,11 +168,43 @@ public:
|
||||
return ref.mapped_span;
|
||||
}
|
||||
|
||||
// compute/graphics new binding
|
||||
std::span<u8> BindMappedStorageBuffer(u32 size) {
|
||||
const StagingBufferRef ref = staging_pool.Request(size, MemoryUsage::Upload);
|
||||
guest_descriptor_queue.AddBuffer(ref.buffer, ref.device_address, static_cast<u32>(ref.offset), size);
|
||||
return ref.mapped_span;
|
||||
[[nodiscard]] VkDeviceSize SparseAlignmentFor(bool sparse_compatible) const noexcept {
|
||||
if (!sparse_compatible || !multi_range_buffers.UsesSparse()) {
|
||||
return 0;
|
||||
}
|
||||
return multi_range_buffers.BlockSize();
|
||||
}
|
||||
|
||||
[[nodiscard]] bool PrefersSparseSources() const noexcept {
|
||||
return multi_range_buffers.UsesSparse();
|
||||
}
|
||||
|
||||
void ResetMultiRange() noexcept {
|
||||
multi_range_sources.clear();
|
||||
multi_range_total = 0;
|
||||
}
|
||||
|
||||
void PushMultiRangeSource(const Buffer& buffer, u32 offset, u32 size) {
|
||||
const vk::MemoryLocation location = buffer.Location();
|
||||
multi_range_sources.push_back(MultiRangeSource{
|
||||
.handle = buffer.Handle(),
|
||||
.memory = location.memory,
|
||||
.memory_offset = location.offset,
|
||||
.offset = offset,
|
||||
.size = size,
|
||||
.memory_type = location.memory_type,
|
||||
});
|
||||
multi_range_total += size;
|
||||
}
|
||||
|
||||
bool BindMultiRangeStorageBuffer(u64 key);
|
||||
|
||||
void InvalidateMultiRange(u64 key) {
|
||||
multi_range_buffers.Invalidate(key);
|
||||
}
|
||||
|
||||
void OnBufferDeleted(const Buffer& buffer) {
|
||||
multi_range_buffers.DropOwner(buffer.Handle());
|
||||
}
|
||||
|
||||
void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) {
|
||||
@@ -215,6 +260,10 @@ private:
|
||||
std::unique_ptr<Uint8Pass> uint8_pass;
|
||||
QuadIndexedPass quad_index_pass;
|
||||
|
||||
MultiRangeBufferCache multi_range_buffers;
|
||||
boost::container::small_vector<MultiRangeSource, 16> multi_range_sources;
|
||||
VkDeviceSize multi_range_total{};
|
||||
|
||||
bool limit_dynamic_storage_buffers = false;
|
||||
u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)();
|
||||
};
|
||||
|
||||
@@ -157,7 +157,9 @@ bool ComputePipeline::Configure(Tegra::Engines::KeplerCompute& kepler_compute,
|
||||
buffer_cache.UnbindComputeStorageBuffers();
|
||||
size_t ssbo_index{};
|
||||
for (const auto& desc : info.storage_buffers_descriptors) {
|
||||
buffer_cache.BindComputeStorageBuffer(ssbo_index, desc.cbuf_index, desc.cbuf_offset, desc.is_written, desc.count);
|
||||
ASSERT(desc.count == 1);
|
||||
buffer_cache.BindComputeStorageBuffer(ssbo_index, desc.cbuf_index, desc.cbuf_offset,
|
||||
desc.is_written);
|
||||
++ssbo_index;
|
||||
}
|
||||
|
||||
|
||||
@@ -47,7 +47,7 @@ static DescriptorBankInfo MakeBankInfo(std::span<const Shader::Info> infos) {
|
||||
DescriptorBankInfo bank;
|
||||
for (const Shader::Info& info : infos) {
|
||||
bank.uniform_buffers += Accumulate(info.constant_buffer_descriptors);
|
||||
bank.storage_buffers += Accumulate(info.storage_buffers_descriptors) + static_cast<u32>(Shader::UsesStorageBufferMappings(info));
|
||||
bank.storage_buffers += Accumulate(info.storage_buffers_descriptors);
|
||||
bank.texture_buffers += Accumulate(info.texture_buffer_descriptors);
|
||||
bank.image_buffers += Accumulate(info.image_buffer_descriptors);
|
||||
bank.textures += Accumulate(info.texture_descriptors);
|
||||
|
||||
@@ -367,8 +367,9 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
|
||||
if constexpr (Spec::has_storage_buffers) {
|
||||
size_t ssbo_index{};
|
||||
for (const auto& desc : info.storage_buffers_descriptors) {
|
||||
ASSERT(desc.count == 1);
|
||||
buffer_cache.BindGraphicsStorageBuffer(stage, ssbo_index, desc.cbuf_index,
|
||||
desc.cbuf_offset, desc.is_written, desc.count);
|
||||
desc.cbuf_offset, desc.is_written);
|
||||
++ssbo_index;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,341 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include <mutex>
|
||||
|
||||
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
|
||||
#include "video_core/renderer_vulkan/vk_scheduler.h"
|
||||
#include "video_core/vulkan_common/vulkan_device.h"
|
||||
|
||||
namespace Vulkan {
|
||||
|
||||
MultiRangeBufferCache::MultiRangeBufferCache(const Device& device_,
|
||||
MemoryAllocator& memory_allocator_,
|
||||
Scheduler& scheduler_)
|
||||
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_} {
|
||||
sparse_usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||
if (device.IsBufferDeviceAddressSupported()) {
|
||||
sparse_usage |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
|
||||
}
|
||||
if (!device.IsSparseBindingSupported()) {
|
||||
return;
|
||||
}
|
||||
u32 memory_type_bits = 0;
|
||||
const VkDeviceSize queried = QueryBlockSize(memory_type_bits);
|
||||
if (queried == 0 || memory_type_bits == 0) {
|
||||
return;
|
||||
}
|
||||
block_size = queried;
|
||||
sparse_memory_type_bits = memory_type_bits;
|
||||
use_sparse = true;
|
||||
}
|
||||
|
||||
MultiRangeBufferCache::~MultiRangeBufferCache() {
|
||||
const VkDevice logical = *device.GetLogical();
|
||||
const auto& dld = device.GetDispatchLoader();
|
||||
for (auto& [key, entry] : entries) {
|
||||
if (entry.sparse_handle != VK_NULL_HANDLE) {
|
||||
dld.vkDestroyBuffer(logical, entry.sparse_handle, nullptr);
|
||||
}
|
||||
}
|
||||
entries.clear();
|
||||
for (const Retired& item : retired) {
|
||||
dld.vkDestroyBuffer(logical, item.handle, nullptr);
|
||||
}
|
||||
retired.clear();
|
||||
}
|
||||
|
||||
VkDeviceSize MultiRangeBufferCache::QueryBlockSize(u32& memory_type_bits) const {
|
||||
const VkDevice logical = *device.GetLogical();
|
||||
const auto& dld = device.GetDispatchLoader();
|
||||
const VkBufferCreateInfo probe_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
|
||||
.size = DEFAULT_BLOCK_SIZE,
|
||||
.usage = sparse_usage,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
VkBuffer probe{};
|
||||
if (dld.vkCreateBuffer(logical, &probe_ci, nullptr, &probe) != VK_SUCCESS) {
|
||||
return 0;
|
||||
}
|
||||
const VkBufferMemoryRequirementsInfo2 reqs_info{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_REQUIREMENTS_INFO_2,
|
||||
.pNext = nullptr,
|
||||
.buffer = probe,
|
||||
};
|
||||
VkMemoryRequirements2 reqs2{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_REQUIREMENTS_2,
|
||||
.pNext = nullptr,
|
||||
.memoryRequirements = {},
|
||||
};
|
||||
dld.vkGetBufferMemoryRequirements2(logical, &reqs_info, &reqs2);
|
||||
dld.vkDestroyBuffer(logical, probe, nullptr);
|
||||
memory_type_bits = reqs2.memoryRequirements.memoryTypeBits;
|
||||
return reqs2.memoryRequirements.alignment;
|
||||
}
|
||||
|
||||
u64 MultiRangeBufferCache::HashSources(std::span<const MultiRangeSource> sources) const {
|
||||
u64 hash = 0xcbf29ce484222325ULL;
|
||||
const auto mix = [&hash](u64 value) {
|
||||
hash ^= value;
|
||||
hash *= 0x100000001b3ULL;
|
||||
};
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
mix(reinterpret_cast<u64>(source.handle));
|
||||
mix(static_cast<u64>(source.offset));
|
||||
mix(static_cast<u64>(source.size));
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
bool MultiRangeBufferCache::CanBindSparse(std::span<const MultiRangeSource> sources) const {
|
||||
if (!UsesSparse()) {
|
||||
return false;
|
||||
}
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
if (source.memory == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
if (source.memory_type >= 32) {
|
||||
return false;
|
||||
}
|
||||
if (((sparse_memory_type_bits >> source.memory_type) & 1) == 0) {
|
||||
return false;
|
||||
}
|
||||
const VkDeviceSize memory_offset = source.memory_offset + source.offset;
|
||||
if ((memory_offset % block_size) != 0) {
|
||||
return false;
|
||||
}
|
||||
if ((source.size % block_size) != 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
VkBuffer MultiRangeBufferCache::CreateSparse(std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total) {
|
||||
const VkDevice logical = *device.GetLogical();
|
||||
const auto& dld = device.GetDispatchLoader();
|
||||
const VkBufferCreateInfo buffer_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
|
||||
.size = total,
|
||||
.usage = sparse_usage,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
VkBuffer handle{};
|
||||
if (dld.vkCreateBuffer(logical, &buffer_ci, nullptr, &handle) != VK_SUCCESS) {
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
std::vector<VkSparseMemoryBind> binds;
|
||||
binds.reserve(sources.size());
|
||||
VkDeviceSize resource_offset = 0;
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
binds.push_back(VkSparseMemoryBind{
|
||||
.resourceOffset = resource_offset,
|
||||
.size = source.size,
|
||||
.memory = source.memory,
|
||||
.memoryOffset = source.memory_offset + source.offset,
|
||||
.flags = 0,
|
||||
});
|
||||
resource_offset += source.size;
|
||||
}
|
||||
const VkSparseBufferMemoryBindInfo buffer_bind{
|
||||
.buffer = handle,
|
||||
.bindCount = static_cast<u32>(binds.size()),
|
||||
.pBinds = binds.data(),
|
||||
};
|
||||
const VkBindSparseInfo bind_info{
|
||||
.sType = VK_STRUCTURE_TYPE_BIND_SPARSE_INFO,
|
||||
.pNext = nullptr,
|
||||
.waitSemaphoreCount = 0,
|
||||
.pWaitSemaphores = nullptr,
|
||||
.bufferBindCount = 1,
|
||||
.pBufferBinds = &buffer_bind,
|
||||
.imageOpaqueBindCount = 0,
|
||||
.pImageOpaqueBinds = nullptr,
|
||||
.imageBindCount = 0,
|
||||
.pImageBinds = nullptr,
|
||||
.signalSemaphoreCount = 0,
|
||||
.pSignalSemaphores = nullptr,
|
||||
};
|
||||
const VkFenceCreateInfo fence_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = 0,
|
||||
};
|
||||
vk::Fence fence = device.GetLogical().CreateFence(fence_ci);
|
||||
VkResult bind_result = VK_ERROR_UNKNOWN;
|
||||
{
|
||||
std::scoped_lock lock{scheduler.submit_mutex};
|
||||
bind_result = device.GetGraphicsQueue().BindSparse(bind_info, *fence);
|
||||
}
|
||||
if (bind_result != VK_SUCCESS) {
|
||||
dld.vkDestroyBuffer(logical, handle, nullptr);
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
fence.Wait();
|
||||
return handle;
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::DestroySparse(VkBuffer handle) {
|
||||
if (handle == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
retired.push_back(Retired{
|
||||
.handle = handle,
|
||||
.tick = scheduler.CurrentTick(),
|
||||
});
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::DrainRetired() {
|
||||
const VkDevice logical = *device.GetLogical();
|
||||
const auto& dld = device.GetDispatchLoader();
|
||||
size_t index = 0;
|
||||
while (index < retired.size()) {
|
||||
if (scheduler.IsFree(retired[index].tick)) {
|
||||
dld.vkDestroyBuffer(logical, retired[index].handle, nullptr);
|
||||
retired[index] = retired.back();
|
||||
retired.pop_back();
|
||||
} else {
|
||||
++index;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MultiRangeRef MultiRangeBufferCache::Get(u64 key, std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total) {
|
||||
if (sources.empty() || total == 0) {
|
||||
return MultiRangeRef{};
|
||||
}
|
||||
if (!retired.empty()) {
|
||||
DrainRetired();
|
||||
}
|
||||
const u64 geometry = HashSources(sources);
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end() && it->second.geometry == geometry && it->second.size == total) {
|
||||
Entry& entry = it->second;
|
||||
MultiRangeRef ref{
|
||||
.handle = entry.sparse_handle,
|
||||
.address = entry.address,
|
||||
.size = entry.size,
|
||||
.needs_gather = false,
|
||||
};
|
||||
if (entry.sparse_handle == VK_NULL_HANDLE) {
|
||||
ref.handle = *entry.gathered;
|
||||
ref.needs_gather = entry.dirty;
|
||||
}
|
||||
return ref;
|
||||
}
|
||||
if (it != entries.end()) {
|
||||
DestroySparse(it->second.sparse_handle);
|
||||
entries.erase(it);
|
||||
}
|
||||
|
||||
Entry entry;
|
||||
entry.geometry = geometry;
|
||||
entry.size = total;
|
||||
if (CanBindSparse(sources)) {
|
||||
entry.sparse_handle = CreateSparse(sources, total);
|
||||
if (entry.sparse_handle != VK_NULL_HANDLE) {
|
||||
entry.owners.reserve(sources.size());
|
||||
for (const MultiRangeSource& source : sources) {
|
||||
entry.owners.push_back(source.handle);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (entry.sparse_handle == VK_NULL_HANDLE) {
|
||||
VkBufferUsageFlags flags = VK_BUFFER_USAGE_TRANSFER_SRC_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT |
|
||||
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
|
||||
if (device.IsBufferDeviceAddressSupported()) {
|
||||
flags |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
|
||||
}
|
||||
const VkBufferCreateInfo gather_ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = 0,
|
||||
.size = total,
|
||||
.usage = flags,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
entry.gathered = memory_allocator.CreateBuffer(gather_ci, MemoryUsage::DeviceLocal);
|
||||
entry.dirty = true;
|
||||
}
|
||||
if (device.IsBufferDeviceAddressSupported()) {
|
||||
VkBuffer address_handle = entry.sparse_handle;
|
||||
if (address_handle == VK_NULL_HANDLE) {
|
||||
address_handle = *entry.gathered;
|
||||
}
|
||||
entry.address = device.GetLogical().GetBufferDeviceAddress(address_handle);
|
||||
}
|
||||
|
||||
MultiRangeRef ref{
|
||||
.handle = entry.sparse_handle,
|
||||
.address = entry.address,
|
||||
.size = entry.size,
|
||||
.needs_gather = false,
|
||||
};
|
||||
if (entry.sparse_handle == VK_NULL_HANDLE) {
|
||||
ref.handle = *entry.gathered;
|
||||
ref.needs_gather = true;
|
||||
}
|
||||
entries.emplace(key, std::move(entry));
|
||||
return ref;
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::MarkGathered(u64 key) {
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end()) {
|
||||
it->second.dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::DropOwner(VkBuffer owner) {
|
||||
if (owner == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
for (auto it = entries.begin(); it != entries.end();) {
|
||||
Entry& entry = it->second;
|
||||
bool owned = false;
|
||||
for (const VkBuffer handle : entry.owners) {
|
||||
if (handle == owner) {
|
||||
owned = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (owned) {
|
||||
DestroySparse(entry.sparse_handle);
|
||||
it = entries.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::Invalidate(u64 key) {
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end()) {
|
||||
it->second.dirty = true;
|
||||
}
|
||||
}
|
||||
|
||||
void MultiRangeBufferCache::Clear() {
|
||||
for (auto& [key, entry] : entries) {
|
||||
DestroySparse(entry.sparse_handle);
|
||||
}
|
||||
entries.clear();
|
||||
}
|
||||
|
||||
} // namespace Vulkan
|
||||
@@ -0,0 +1,105 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <span>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
#include "common/common_types.h"
|
||||
#include "video_core/vulkan_common/vulkan_memory_allocator.h"
|
||||
#include "video_core/vulkan_common/vulkan_wrapper.h"
|
||||
|
||||
namespace Vulkan {
|
||||
|
||||
class Device;
|
||||
class Scheduler;
|
||||
|
||||
struct MultiRangeSource {
|
||||
VkBuffer handle{};
|
||||
VkDeviceMemory memory{};
|
||||
VkDeviceSize memory_offset{};
|
||||
VkDeviceSize offset{};
|
||||
VkDeviceSize size{};
|
||||
u32 memory_type{};
|
||||
};
|
||||
|
||||
struct MultiRangeRef {
|
||||
VkBuffer handle{};
|
||||
VkDeviceAddress address{};
|
||||
VkDeviceSize size{};
|
||||
bool needs_gather{};
|
||||
};
|
||||
|
||||
class MultiRangeBufferCache final {
|
||||
public:
|
||||
static constexpr VkDeviceSize DEFAULT_BLOCK_SIZE = 64 * 1024;
|
||||
|
||||
explicit MultiRangeBufferCache(const Device& device_, MemoryAllocator& memory_allocator_,
|
||||
Scheduler& scheduler_);
|
||||
~MultiRangeBufferCache();
|
||||
|
||||
MultiRangeBufferCache(const MultiRangeBufferCache&) = delete;
|
||||
MultiRangeBufferCache& operator=(const MultiRangeBufferCache&) = delete;
|
||||
|
||||
[[nodiscard]] bool UsesSparse() const noexcept {
|
||||
return use_sparse;
|
||||
}
|
||||
|
||||
[[nodiscard]] VkDeviceSize BlockSize() const noexcept {
|
||||
return block_size;
|
||||
}
|
||||
|
||||
[[nodiscard]] MultiRangeRef Get(u64 key, std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total);
|
||||
|
||||
void MarkGathered(u64 key);
|
||||
|
||||
void Invalidate(u64 key);
|
||||
|
||||
void DropOwner(VkBuffer owner);
|
||||
|
||||
void Clear();
|
||||
|
||||
private:
|
||||
struct Retired {
|
||||
VkBuffer handle{};
|
||||
u64 tick{};
|
||||
};
|
||||
|
||||
struct Entry {
|
||||
vk::Buffer gathered;
|
||||
VkBuffer sparse_handle{};
|
||||
VkDeviceAddress address{};
|
||||
VkDeviceSize size{};
|
||||
u64 geometry{};
|
||||
bool dirty{true};
|
||||
std::vector<VkBuffer> owners;
|
||||
};
|
||||
|
||||
[[nodiscard]] u64 HashSources(std::span<const MultiRangeSource> sources) const;
|
||||
|
||||
[[nodiscard]] bool CanBindSparse(std::span<const MultiRangeSource> sources) const;
|
||||
|
||||
[[nodiscard]] VkBuffer CreateSparse(std::span<const MultiRangeSource> sources,
|
||||
VkDeviceSize total);
|
||||
|
||||
[[nodiscard]] VkDeviceSize QueryBlockSize(u32& memory_type_bits) const;
|
||||
|
||||
void DestroySparse(VkBuffer handle);
|
||||
|
||||
void DrainRetired();
|
||||
|
||||
const Device& device;
|
||||
MemoryAllocator& memory_allocator;
|
||||
Scheduler& scheduler;
|
||||
bool use_sparse{};
|
||||
VkDeviceSize block_size{DEFAULT_BLOCK_SIZE};
|
||||
u32 sparse_memory_type_bits{};
|
||||
VkBufferUsageFlags sparse_usage{};
|
||||
std::unordered_map<u64, Entry> entries;
|
||||
std::vector<Retired> retired;
|
||||
};
|
||||
|
||||
} // namespace Vulkan
|
||||
@@ -62,13 +62,7 @@ using VideoCommon::FileEnvironment;
|
||||
using VideoCommon::GenericEnvironment;
|
||||
using VideoCommon::GraphicsEnvironment;
|
||||
|
||||
// SPIR-V descriptor arrays require a fixed pipeline-layout count.
|
||||
// Exploration ceiling; buffer-cache telemetry records the actual physical-range demand.
|
||||
// Keep this modest because every mapped SSBO binds the full fixed array on each update.
|
||||
const u32 MAX_MAPPED_STORAGE_BUFFER_DESCRIPTORS =
|
||||
(std::max)(6u, static_cast<u32>(Settings::values.debug_knobs.GetValue()));
|
||||
|
||||
constexpr u32 CACHE_VERSION = 19;
|
||||
constexpr u32 CACHE_VERSION = 18;
|
||||
constexpr size_t VULKAN_CACHE_FLUSH_PIPELINES = 128;
|
||||
constexpr size_t VULKAN_CACHE_FLUSH_MIN_SECONDS = 30;
|
||||
constexpr std::array<char, 8> VULKAN_CACHE_MAGIC_NUMBER{'y', 'u', 'z', 'u', 'v', 'k', 'c', 'h'};
|
||||
@@ -438,8 +432,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
device.IsUniformTexelBufferArrayNonUniformIndexingSupported(),
|
||||
.support_storage_texel_buffer_array_nonuniform_indexing =
|
||||
device.IsStorageTexelBufferArrayNonUniformIndexingSupported(),
|
||||
.support_storage_buffer_array_nonuniform_indexing =
|
||||
device.IsStorageBufferArrayNonUniformIndexingSupported(),
|
||||
|
||||
.warp_size_potentially_larger_than_guest = device.IsWarpSizePotentiallyBiggerThanGuest(),
|
||||
|
||||
@@ -468,7 +460,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
host_info = Shader::HostTranslateInfo{
|
||||
.min_ssbo_alignment = device.GetStorageBufferAlignment(),
|
||||
.max_per_stage_descriptor_sampled_images = device.GetMaxPerStageDescriptorSampledImages(),
|
||||
.max_per_stage_descriptor_storage_buffers = device.GetMaxPerStageDescriptorStorageBuffers(),
|
||||
.max_per_stage_resources = device.GetMaxPerStageResources(),
|
||||
.max_descriptor_set_samplers = device.GetMaxDescriptorSetSamplers(),
|
||||
.max_descriptor_set_uniform_buffers = device.GetMaxDescriptorSetUniformBuffers(),
|
||||
@@ -488,20 +479,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
.support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(),
|
||||
.support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(),
|
||||
.support_conditional_barrier = device.SupportsConditionalBarriers(),
|
||||
.storage_buffer_segment_count = [&] {
|
||||
if (!device.IsStorageBufferArrayNonUniformIndexingSupported()) {
|
||||
return 1U;
|
||||
}
|
||||
constexpr u32 MaxGraphicsStages = static_cast<u32>(Maxwell::MaxShaderStage);
|
||||
const auto reserve = [](u32 limit, u32 count) {
|
||||
return limit > count ? limit - count : 0U;
|
||||
};
|
||||
const u32 per_stage = reserve(device.GetMaxPerStageDescriptorStorageBuffers(), 1) / static_cast<u32>(Shader::Info::MAX_SSBOS);
|
||||
const u32 per_set = reserve(device.GetMaxDescriptorSetStorageBuffers(), MaxGraphicsStages) / (static_cast<u32>(Shader::Info::MAX_SSBOS) * MaxGraphicsStages);
|
||||
const u32 resources = reserve(device.GetMaxPerStageResources(), 1) / (static_cast<u32>(Shader::Info::MAX_SSBOS) * 2);
|
||||
// ensure at least one storage buffer segment is available per stage. max is still arbitrary
|
||||
return (std::max)(1U, (std::min)({MAX_MAPPED_STORAGE_BUFFER_DESCRIPTORS, per_stage, per_set, resources}));
|
||||
}(),
|
||||
};
|
||||
host_info.ApplyDescriptorLimitPolicy();
|
||||
|
||||
|
||||
@@ -338,7 +338,7 @@ void PresentManager::PresentThread(std::stop_token token) {
|
||||
// By exchanging the lock ownership we take the swapchain lock
|
||||
// before the queue lock goes out of scope. This way the swapchain
|
||||
// lock in WaitPresent is guaranteed to occur after here.
|
||||
std::exchange(lock, std::unique_lock{swapchain_mutex});
|
||||
void(std::exchange(lock, std::unique_lock{swapchain_mutex}));
|
||||
CopyToSwapchain(frame);
|
||||
|
||||
// Free the frame for reuse
|
||||
|
||||
@@ -819,6 +819,7 @@ void RasterizerVulkan::ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) {
|
||||
std::scoped_lock lock{texture_cache.mutex};
|
||||
texture_cache.UnmapGPUMemory(as_id, addr, size);
|
||||
}
|
||||
buffer_cache.UnmapGPUMemory(as_id, addr, size);
|
||||
}
|
||||
|
||||
void RasterizerVulkan::SignalFence(std::function<void()>&& func) {
|
||||
|
||||
@@ -300,7 +300,7 @@ void Scheduler::WorkerThread(std::stop_token stop_token) {
|
||||
// Exchange lock ownership so that we take the execution lock before
|
||||
// the queue lock goes out of scope. This allows us to force execution
|
||||
// to complete in the next step.
|
||||
std::exchange(lk, std::unique_lock{execution_mutex});
|
||||
void(std::exchange(lk, std::unique_lock{execution_mutex}));
|
||||
|
||||
// Perform the work, tracking whether the chunk was a submission
|
||||
// before executing.
|
||||
|
||||
@@ -708,6 +708,7 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
|
||||
descriptor_indexing.shaderUniformTexelBufferArrayDynamicIndexing = false;
|
||||
descriptor_indexing.shaderStorageTexelBufferArrayDynamicIndexing = false;
|
||||
descriptor_indexing.shaderUniformBufferArrayNonUniformIndexing = false;
|
||||
descriptor_indexing.shaderStorageBufferArrayNonUniformIndexing = false;
|
||||
descriptor_indexing.shaderInputAttachmentArrayNonUniformIndexing = false;
|
||||
descriptor_indexing.descriptorBindingUniformBufferUpdateAfterBind = false;
|
||||
descriptor_indexing.descriptorBindingSampledImageUpdateAfterBind = false;
|
||||
@@ -1569,6 +1570,8 @@ void Device::SetupFamilies(VkSurfaceKHR surface) {
|
||||
}
|
||||
if (graphics) {
|
||||
graphics_family = *graphics;
|
||||
graphics_family_sparse_binding =
|
||||
(queue_family_properties[*graphics].queueFlags & VK_QUEUE_SPARSE_BINDING_BIT) != 0;
|
||||
}
|
||||
if (present) {
|
||||
present_family = *present;
|
||||
|
||||
@@ -317,6 +317,10 @@ public:
|
||||
return properties.driver.driverID;
|
||||
}
|
||||
|
||||
bool IsSparseBindingSupported() const {
|
||||
return features.features.sparseBinding && graphics_family_sparse_binding;
|
||||
}
|
||||
|
||||
/// Returns true for tile-based deferred renderers.
|
||||
bool IsTiler() const {
|
||||
switch (GetDriverID()) {
|
||||
@@ -360,7 +364,6 @@ public:
|
||||
#define FN_MAX_LIMIT_LIST \
|
||||
FN_MAX_LIMIT_ELEM(ComputeSharedMemorySize) \
|
||||
FN_MAX_LIMIT_ELEM(PerStageDescriptorSampledImages) \
|
||||
FN_MAX_LIMIT_ELEM(PerStageDescriptorStorageBuffers) \
|
||||
FN_MAX_LIMIT_ELEM(PerStageResources) \
|
||||
FN_MAX_LIMIT_ELEM(DescriptorSetSamplers) \
|
||||
FN_MAX_LIMIT_ELEM(DescriptorSetUniformBuffers) \
|
||||
@@ -416,10 +419,6 @@ FN_MAX_LIMIT_LIST
|
||||
return features.descriptor_indexing.shaderStorageTexelBufferArrayNonUniformIndexing;
|
||||
}
|
||||
|
||||
bool IsStorageBufferArrayNonUniformIndexingSupported() const {
|
||||
return features.descriptor_indexing.shaderStorageBufferArrayNonUniformIndexing;
|
||||
}
|
||||
|
||||
/// Returns true if the device supports float64 natively.
|
||||
bool IsFloat64Supported() const {
|
||||
return features.features.shaderFloat64;
|
||||
@@ -1151,6 +1150,7 @@ private:
|
||||
bool owns_static_pipeline_cache{};
|
||||
u32 instance_version{}; ///< Vulkan instance version.
|
||||
u32 graphics_family{}; ///< Main graphics queue family index.
|
||||
bool graphics_family_sparse_binding{};
|
||||
u32 present_family{}; ///< Main present queue family index.
|
||||
|
||||
struct Extensions {
|
||||
|
||||
@@ -26,350 +26,374 @@
|
||||
#include "common/settings.h"
|
||||
|
||||
namespace Vulkan {
|
||||
namespace {
|
||||
namespace {
|
||||
|
||||
// Helpers translating MemoryUsage to flags/usage
|
||||
|
||||
[[maybe_unused]] VkMemoryPropertyFlags MemoryUsagePropertyFlags(MemoryUsage usage) {
|
||||
switch (usage) {
|
||||
case MemoryUsage::DeviceLocal:
|
||||
return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
|
||||
case MemoryUsage::Upload:
|
||||
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
case MemoryUsage::Download:
|
||||
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
|
||||
case MemoryUsage::Stream:
|
||||
return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
}
|
||||
ASSERT_MSG(false, "Invalid memory usage={}", usage);
|
||||
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
[[maybe_unused]] VkMemoryPropertyFlags MemoryUsagePropertyFlags(MemoryUsage usage) {
|
||||
switch (usage) {
|
||||
case MemoryUsage::DeviceLocal:
|
||||
return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
|
||||
case MemoryUsage::Upload:
|
||||
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
case MemoryUsage::Download:
|
||||
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
|
||||
case MemoryUsage::Stream:
|
||||
return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
|
||||
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
}
|
||||
ASSERT_MSG(false, "Invalid memory usage={}", usage);
|
||||
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
}
|
||||
|
||||
[[nodiscard]] VkMemoryPropertyFlags MemoryUsagePreferredVmaFlags(MemoryUsage usage) {
|
||||
if (usage == MemoryUsage::Download) {
|
||||
return VK_MEMORY_PROPERTY_HOST_CACHED_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
}
|
||||
return usage != MemoryUsage::DeviceLocal ? VK_MEMORY_PROPERTY_HOST_COHERENT_BIT
|
||||
: VkMemoryPropertyFlagBits{};
|
||||
[[nodiscard]] VkMemoryPropertyFlags MemoryUsagePreferredVmaFlags(MemoryUsage usage) {
|
||||
if (usage == MemoryUsage::Download) {
|
||||
return VK_MEMORY_PROPERTY_HOST_CACHED_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
|
||||
}
|
||||
return usage != MemoryUsage::DeviceLocal ? VK_MEMORY_PROPERTY_HOST_COHERENT_BIT
|
||||
: VkMemoryPropertyFlagBits{};
|
||||
}
|
||||
|
||||
[[nodiscard]] VmaAllocationCreateFlags MemoryUsageVmaFlags(MemoryUsage usage) {
|
||||
switch (usage) {
|
||||
case MemoryUsage::Upload:
|
||||
case MemoryUsage::Stream:
|
||||
return VMA_ALLOCATION_CREATE_MAPPED_BIT |
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT;
|
||||
case MemoryUsage::Download:
|
||||
return VMA_ALLOCATION_CREATE_MAPPED_BIT |
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_RANDOM_BIT;
|
||||
case MemoryUsage::DeviceLocal:
|
||||
return {};
|
||||
}
|
||||
return {};
|
||||
[[nodiscard]] VmaAllocationCreateFlags MemoryUsageVmaFlags(MemoryUsage usage) {
|
||||
switch (usage) {
|
||||
case MemoryUsage::Upload:
|
||||
case MemoryUsage::Stream:
|
||||
return VMA_ALLOCATION_CREATE_MAPPED_BIT |
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_SEQUENTIAL_WRITE_BIT;
|
||||
case MemoryUsage::Download:
|
||||
return VMA_ALLOCATION_CREATE_MAPPED_BIT |
|
||||
VMA_ALLOCATION_CREATE_HOST_ACCESS_RANDOM_BIT;
|
||||
case MemoryUsage::DeviceLocal:
|
||||
return {};
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
[[nodiscard]] VmaMemoryUsage MemoryUsageVma(MemoryUsage usage) {
|
||||
switch (usage) {
|
||||
case MemoryUsage::DeviceLocal:
|
||||
case MemoryUsage::Stream:
|
||||
return VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE;
|
||||
case MemoryUsage::Upload:
|
||||
case MemoryUsage::Download:
|
||||
return VMA_MEMORY_USAGE_AUTO_PREFER_HOST;
|
||||
}
|
||||
return VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE;
|
||||
[[nodiscard]] VmaMemoryUsage MemoryUsageVma(MemoryUsage usage) {
|
||||
switch (usage) {
|
||||
case MemoryUsage::DeviceLocal:
|
||||
case MemoryUsage::Stream:
|
||||
return VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE;
|
||||
case MemoryUsage::Upload:
|
||||
case MemoryUsage::Download:
|
||||
return VMA_MEMORY_USAGE_AUTO_PREFER_HOST;
|
||||
}
|
||||
|
||||
|
||||
// This avoids calling vkGetBufferMemoryRequirements* directly.
|
||||
template<typename T>
|
||||
static VkBuffer GetVkHandleFromBuffer(const T &buf) {
|
||||
if constexpr (requires { static_cast<VkBuffer>(buf); }) {
|
||||
return static_cast<VkBuffer>(buf);
|
||||
} else if constexpr (requires {{ buf.GetHandle() } -> std::convertible_to<VkBuffer>; }) {
|
||||
return buf.GetHandle();
|
||||
} else if constexpr (requires {{ buf.Handle() } -> std::convertible_to<VkBuffer>; }) {
|
||||
return buf.Handle();
|
||||
} else if constexpr (requires {{ buf.vk_handle() } -> std::convertible_to<VkBuffer>; }) {
|
||||
return buf.vk_handle();
|
||||
} else {
|
||||
static_assert(sizeof(T) == 0, "Cannot extract VkBuffer handle from vk::Buffer");
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
return VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
//MemoryCommit is now VMA-backed
|
||||
MemoryCommit::MemoryCommit(VmaAllocator alloc, VmaAllocation a,
|
||||
const VmaAllocationInfo &info) noexcept
|
||||
: allocator{alloc}, allocation{a}, memory{info.deviceMemory},
|
||||
offset{info.offset}, size{info.size}, mapped_ptr{info.pMappedData} {
|
||||
// Log GPU memory allocation
|
||||
MemoryCommit::MemoryCommit(VmaAllocator alloc, VmaAllocation a,
|
||||
const VmaAllocationInfo &info) noexcept
|
||||
: allocator{alloc}, allocation{a}, memory{info.deviceMemory},
|
||||
offset{info.offset}, size{info.size}, mapped_ptr{info.pMappedData} {
|
||||
// Log GPU memory allocation
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue()) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryAllocation(
|
||||
reinterpret_cast<uintptr_t>(memory),
|
||||
static_cast<u64>(size),
|
||||
0 // Memory property flags (not easily available from VMA)
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
MemoryCommit::~MemoryCommit() { Release(); }
|
||||
|
||||
MemoryCommit::MemoryCommit(MemoryCommit &&rhs) noexcept
|
||||
: allocator{std::exchange(rhs.allocator, nullptr)},
|
||||
allocation{std::exchange(rhs.allocation, nullptr)},
|
||||
memory{std::exchange(rhs.memory, VK_NULL_HANDLE)},
|
||||
offset{std::exchange(rhs.offset, 0)},
|
||||
size{std::exchange(rhs.size, 0)},
|
||||
mapped_ptr{std::exchange(rhs.mapped_ptr, nullptr)} {}
|
||||
|
||||
MemoryCommit &MemoryCommit::operator=(MemoryCommit &&rhs) noexcept {
|
||||
if (this != &rhs) {
|
||||
Release();
|
||||
allocator = std::exchange(rhs.allocator, nullptr);
|
||||
allocation = std::exchange(rhs.allocation, nullptr);
|
||||
memory = std::exchange(rhs.memory, VK_NULL_HANDLE);
|
||||
offset = std::exchange(rhs.offset, 0);
|
||||
size = std::exchange(rhs.size, 0);
|
||||
mapped_ptr = std::exchange(rhs.mapped_ptr, nullptr);
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
|
||||
std::span<u8> MemoryCommit::Map()
|
||||
{
|
||||
if (!allocation) return {};
|
||||
if (!mapped_ptr) {
|
||||
if (vmaMapMemory(allocator, allocation, &mapped_ptr) != VK_SUCCESS) return {};
|
||||
}
|
||||
const size_t n = static_cast<size_t>(std::min<VkDeviceSize>(size,
|
||||
(std::numeric_limits<size_t>::max)()));
|
||||
return std::span<u8>{static_cast<u8 *>(mapped_ptr), n};
|
||||
}
|
||||
|
||||
std::span<const u8> MemoryCommit::Map() const
|
||||
{
|
||||
if (!allocation) return {};
|
||||
if (!mapped_ptr) {
|
||||
void *p = nullptr;
|
||||
if (vmaMapMemory(allocator, allocation, &p) != VK_SUCCESS) return {};
|
||||
const_cast<MemoryCommit *>(this)->mapped_ptr = p;
|
||||
}
|
||||
const size_t n = static_cast<size_t>(std::min<VkDeviceSize>(size,
|
||||
(std::numeric_limits<size_t>::max)()));
|
||||
return std::span<const u8>{static_cast<const u8 *>(mapped_ptr), n};
|
||||
}
|
||||
|
||||
void MemoryCommit::Unmap()
|
||||
{
|
||||
if (allocation && mapped_ptr) {
|
||||
vmaUnmapMemory(allocator, allocation);
|
||||
mapped_ptr = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
void MemoryCommit::Release() {
|
||||
if (allocation && allocator) {
|
||||
// Log GPU memory deallocation
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue()) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryAllocation(
|
||||
reinterpret_cast<uintptr_t>(memory),
|
||||
static_cast<u64>(size),
|
||||
0 // Memory property flags (not easily available from VMA)
|
||||
Settings::values.gpu_log_memory_tracking.GetValue() &&
|
||||
memory != VK_NULL_HANDLE) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryDeallocation(
|
||||
reinterpret_cast<uintptr_t>(memory)
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
MemoryCommit::~MemoryCommit() { Release(); }
|
||||
|
||||
MemoryCommit::MemoryCommit(MemoryCommit &&rhs) noexcept
|
||||
: allocator{std::exchange(rhs.allocator, nullptr)},
|
||||
allocation{std::exchange(rhs.allocation, nullptr)},
|
||||
memory{std::exchange(rhs.memory, VK_NULL_HANDLE)},
|
||||
offset{std::exchange(rhs.offset, 0)},
|
||||
size{std::exchange(rhs.size, 0)},
|
||||
mapped_ptr{std::exchange(rhs.mapped_ptr, nullptr)} {}
|
||||
|
||||
MemoryCommit &MemoryCommit::operator=(MemoryCommit &&rhs) noexcept {
|
||||
if (this != &rhs) {
|
||||
Release();
|
||||
allocator = std::exchange(rhs.allocator, nullptr);
|
||||
allocation = std::exchange(rhs.allocation, nullptr);
|
||||
memory = std::exchange(rhs.memory, VK_NULL_HANDLE);
|
||||
offset = std::exchange(rhs.offset, 0);
|
||||
size = std::exchange(rhs.size, 0);
|
||||
mapped_ptr = std::exchange(rhs.mapped_ptr, nullptr);
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
|
||||
std::span<u8> MemoryCommit::Map()
|
||||
{
|
||||
if (!allocation) return {};
|
||||
if (!mapped_ptr) {
|
||||
if (vmaMapMemory(allocator, allocation, &mapped_ptr) != VK_SUCCESS) return {};
|
||||
}
|
||||
const size_t n = static_cast<size_t>(std::min<VkDeviceSize>(size,
|
||||
(std::numeric_limits<size_t>::max)()));
|
||||
return std::span<u8>{static_cast<u8 *>(mapped_ptr), n};
|
||||
}
|
||||
|
||||
std::span<const u8> MemoryCommit::Map() const
|
||||
{
|
||||
if (!allocation) return {};
|
||||
if (!mapped_ptr) {
|
||||
void *p = nullptr;
|
||||
if (vmaMapMemory(allocator, allocation, &p) != VK_SUCCESS) return {};
|
||||
const_cast<MemoryCommit *>(this)->mapped_ptr = p;
|
||||
}
|
||||
const size_t n = static_cast<size_t>(std::min<VkDeviceSize>(size,
|
||||
(std::numeric_limits<size_t>::max)()));
|
||||
return std::span<const u8>{static_cast<const u8 *>(mapped_ptr), n};
|
||||
}
|
||||
|
||||
void MemoryCommit::Unmap()
|
||||
{
|
||||
if (allocation && mapped_ptr) {
|
||||
if (mapped_ptr) {
|
||||
vmaUnmapMemory(allocator, allocation);
|
||||
mapped_ptr = nullptr;
|
||||
}
|
||||
vmaFreeMemory(allocator, allocation);
|
||||
}
|
||||
allocation = nullptr;
|
||||
allocator = nullptr;
|
||||
memory = VK_NULL_HANDLE;
|
||||
offset = 0;
|
||||
size = 0;
|
||||
}
|
||||
|
||||
void MemoryCommit::Release() {
|
||||
if (allocation && allocator) {
|
||||
// Log GPU memory deallocation
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue() &&
|
||||
memory != VK_NULL_HANDLE) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryDeallocation(
|
||||
reinterpret_cast<uintptr_t>(memory)
|
||||
);
|
||||
}
|
||||
MemoryAllocator::MemoryAllocator(const Device &device_)
|
||||
: device{device_}, allocator{device.GetAllocator()},
|
||||
properties{device_.GetPhysical().GetMemoryProperties().memoryProperties},
|
||||
buffer_image_granularity{
|
||||
device_.GetPhysical().GetProperties().limits.bufferImageGranularity} {
|
||||
|
||||
if (mapped_ptr) {
|
||||
vmaUnmapMemory(allocator, allocation);
|
||||
mapped_ptr = nullptr;
|
||||
}
|
||||
vmaFreeMemory(allocator, allocation);
|
||||
}
|
||||
allocation = nullptr;
|
||||
allocator = nullptr;
|
||||
memory = VK_NULL_HANDLE;
|
||||
offset = 0;
|
||||
size = 0;
|
||||
}
|
||||
|
||||
MemoryAllocator::MemoryAllocator(const Device &device_)
|
||||
: device{device_}, allocator{device.GetAllocator()},
|
||||
properties{device_.GetPhysical().GetMemoryProperties().memoryProperties},
|
||||
buffer_image_granularity{
|
||||
device_.GetPhysical().GetProperties().limits.bufferImageGranularity} {
|
||||
|
||||
// Preserve the previous "RenderDoc small heap" trimming behavior that we had in original vma minus the heap bug
|
||||
if (device.HasDebuggingToolAttached())
|
||||
{
|
||||
using namespace Common::Literals;
|
||||
ForEachDeviceLocalHostVisibleHeap(device, [this](size_t heap_idx, VkMemoryHeap &heap) {
|
||||
if (heap.size <= 256_MiB) {
|
||||
for (u32 t = 0; t < properties.memoryTypeCount; ++t) {
|
||||
if (properties.memoryTypes[t].heapIndex == heap_idx) {
|
||||
valid_memory_types &= ~(1u << t);
|
||||
}
|
||||
// Preserve the previous "RenderDoc small heap" trimming behavior that we had in original vma minus the heap bug
|
||||
if (device.HasDebuggingToolAttached())
|
||||
{
|
||||
using namespace Common::Literals;
|
||||
ForEachDeviceLocalHostVisibleHeap(device, [this](size_t heap_idx, VkMemoryHeap &heap) {
|
||||
if (heap.size <= 256_MiB) {
|
||||
for (u32 t = 0; t < properties.memoryTypeCount; ++t) {
|
||||
if (properties.memoryTypes[t].heapIndex == heap_idx) {
|
||||
valid_memory_types &= ~(1u << t);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
MemoryAllocator::~MemoryAllocator() = default;
|
||||
MemoryAllocator::~MemoryAllocator() = default;
|
||||
|
||||
vk::Image MemoryAllocator::CreateImage(const VkImageCreateInfo &ci) const
|
||||
{
|
||||
const VmaAllocationCreateInfo alloc_ci = {
|
||||
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT,
|
||||
.usage = VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE,
|
||||
.requiredFlags = 0,
|
||||
.preferredFlags = VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT,
|
||||
.memoryTypeBits = 0,
|
||||
.pool = VK_NULL_HANDLE,
|
||||
.pUserData = nullptr,
|
||||
.priority = 0.f,
|
||||
};
|
||||
|
||||
VkImage handle{};
|
||||
VmaAllocation allocation{};
|
||||
VmaAllocationInfo alloc_info{};
|
||||
vk::Check(vmaCreateImage(allocator, &ci, &alloc_ci, &handle, &allocation, &alloc_info));
|
||||
|
||||
// Log GPU memory allocation for images
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue()) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryAllocation(
|
||||
reinterpret_cast<uintptr_t>(alloc_info.deviceMemory),
|
||||
static_cast<u64>(alloc_info.size),
|
||||
VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT
|
||||
);
|
||||
}
|
||||
|
||||
return vk::Image(handle, ci.usage, *device.GetLogical(), allocator, allocation,
|
||||
device.GetDispatchLoader());
|
||||
}
|
||||
|
||||
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const {
|
||||
// MESA will do memcpy() if not marked as host cached, so just force mark it for most buffers
|
||||
auto const anv_flags = (usage == MemoryUsage::Stream
|
||||
&& device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA)
|
||||
? VK_MEMORY_PROPERTY_HOST_CACHED_BIT : 0;
|
||||
const VmaAllocationCreateInfo alloc_ci = {
|
||||
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
|
||||
.usage = MemoryUsageVma(usage),
|
||||
vk::Image MemoryAllocator::CreateImage(const VkImageCreateInfo &ci) const
|
||||
{
|
||||
const VmaAllocationCreateInfo alloc_ci = {
|
||||
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT,
|
||||
.usage = VMA_MEMORY_USAGE_AUTO_PREFER_DEVICE,
|
||||
.requiredFlags = 0,
|
||||
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
|
||||
.memoryTypeBits = usage == MemoryUsage::Stream ? 0u : valid_memory_types,
|
||||
.preferredFlags = VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT,
|
||||
.memoryTypeBits = 0,
|
||||
.pool = VK_NULL_HANDLE,
|
||||
.pUserData = nullptr,
|
||||
.priority = 0.f,
|
||||
};
|
||||
};
|
||||
|
||||
VkBuffer handle{};
|
||||
VmaAllocationInfo alloc_info{};
|
||||
VmaAllocation allocation{};
|
||||
VkMemoryPropertyFlags property_flags{};
|
||||
VkImage handle{};
|
||||
VmaAllocation allocation{};
|
||||
VmaAllocationInfo alloc_info{};
|
||||
vk::Check(vmaCreateImage(allocator, &ci, &alloc_ci, &handle, &allocation, &alloc_info));
|
||||
|
||||
vk::Check(vmaCreateBuffer(allocator, &ci, &alloc_ci, &handle, &allocation, &alloc_info));
|
||||
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
|
||||
|
||||
// Log GPU memory allocation for buffers
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue()) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryAllocation(
|
||||
reinterpret_cast<uintptr_t>(alloc_info.deviceMemory),
|
||||
static_cast<u64>(alloc_info.size),
|
||||
property_flags
|
||||
);
|
||||
}
|
||||
|
||||
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
|
||||
const std::span<u8> mapped_data = data ? std::span<u8>{data, ci.size} : std::span<u8>{};
|
||||
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
|
||||
|
||||
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data,
|
||||
is_coherent,
|
||||
device.GetDispatchLoader());
|
||||
// Log GPU memory allocation for images
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue()) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryAllocation(
|
||||
reinterpret_cast<uintptr_t>(alloc_info.deviceMemory),
|
||||
static_cast<u64>(alloc_info.size),
|
||||
VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT
|
||||
);
|
||||
}
|
||||
|
||||
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
|
||||
{
|
||||
const auto vma_usage = MemoryUsageVma(usage);
|
||||
VmaAllocationCreateInfo ci{};
|
||||
ci.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage);
|
||||
ci.usage = vma_usage;
|
||||
ci.memoryTypeBits = reqs.memoryTypeBits & valid_memory_types;
|
||||
ci.requiredFlags = 0;
|
||||
ci.preferredFlags = MemoryUsagePreferredVmaFlags(usage);
|
||||
return vk::Image(handle, ci.usage, *device.GetLogical(), allocator, allocation,
|
||||
device.GetDispatchLoader());
|
||||
}
|
||||
|
||||
VmaAllocation a{};
|
||||
VmaAllocationInfo info{};
|
||||
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const {
|
||||
// MESA will do memcpy() if not marked as host cached, so just force mark it for most buffers
|
||||
auto const anv_flags = (usage == MemoryUsage::Stream
|
||||
&& device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA)
|
||||
? VK_MEMORY_PROPERTY_HOST_CACHED_BIT : 0;
|
||||
const VmaAllocationCreateInfo alloc_ci = {
|
||||
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
|
||||
.usage = MemoryUsageVma(usage),
|
||||
.requiredFlags = 0,
|
||||
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
|
||||
.memoryTypeBits = usage == MemoryUsage::Stream ? 0u : valid_memory_types,
|
||||
.pool = VK_NULL_HANDLE,
|
||||
.pUserData = nullptr,
|
||||
.priority = 0.f,
|
||||
};
|
||||
|
||||
VkResult res = vmaAllocateMemory(allocator, &reqs, &ci, &a, &info);
|
||||
VkBuffer handle{};
|
||||
VmaAllocationInfo alloc_info{};
|
||||
VmaAllocation allocation{};
|
||||
VkMemoryPropertyFlags property_flags{};
|
||||
|
||||
if (res != VK_SUCCESS) {
|
||||
// Relax 1: drop budget constraint
|
||||
auto ci2 = ci;
|
||||
ci2.flags &= ~VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT;
|
||||
res = vmaAllocateMemory(allocator, &reqs, &ci2, &a, &info);
|
||||
vk::Check(vmaCreateBuffer(allocator, &ci, &alloc_ci, &handle, &allocation, &alloc_info));
|
||||
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
|
||||
|
||||
// Relax 2: if we preferred DEVICE_LOCAL, drop that preference
|
||||
if (res != VK_SUCCESS && (ci.preferredFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT)) {
|
||||
auto ci3 = ci2;
|
||||
ci3.preferredFlags &= ~VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
|
||||
res = vmaAllocateMemory(allocator, &reqs, &ci3, &a, &info);
|
||||
}
|
||||
}
|
||||
|
||||
vk::Check(res);
|
||||
return MemoryCommit(allocator, a, info);
|
||||
// Log GPU memory allocation for buffers
|
||||
if (GPU::Logging::IsActive() &&
|
||||
Settings::values.gpu_log_memory_tracking.GetValue()) {
|
||||
GPU::Logging::GPULogger::GetInstance().LogMemoryAllocation(
|
||||
reinterpret_cast<uintptr_t>(alloc_info.deviceMemory),
|
||||
static_cast<u64>(alloc_info.size),
|
||||
property_flags
|
||||
);
|
||||
}
|
||||
|
||||
MemoryCommit MemoryAllocator::Commit(const vk::Buffer &buffer, MemoryUsage usage) {
|
||||
// Allocate memory appropriate for this buffer automatically
|
||||
const auto vma_usage = MemoryUsageVma(usage);
|
||||
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
|
||||
const std::span<u8> mapped_data = data ? std::span<u8>{data, ci.size} : std::span<u8>{};
|
||||
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
|
||||
|
||||
VmaAllocationCreateInfo ci{};
|
||||
ci.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage);
|
||||
ci.usage = vma_usage;
|
||||
ci.requiredFlags = 0;
|
||||
ci.preferredFlags = MemoryUsagePreferredVmaFlags(usage);
|
||||
ci.pool = VK_NULL_HANDLE;
|
||||
ci.pUserData = nullptr;
|
||||
ci.priority = 0.0f;
|
||||
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data,
|
||||
is_coherent,
|
||||
device.GetDispatchLoader());
|
||||
}
|
||||
|
||||
const VkBuffer raw = *buffer;
|
||||
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
|
||||
VkDeviceSize min_alignment) const {
|
||||
if (min_alignment <= 1) {
|
||||
return CreateBuffer(ci, usage);
|
||||
}
|
||||
VkMemoryPropertyFlags anv_flags = 0;
|
||||
if (usage == MemoryUsage::Stream &&
|
||||
device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA) {
|
||||
anv_flags = VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
|
||||
}
|
||||
u32 memory_type_bits = valid_memory_types;
|
||||
if (usage == MemoryUsage::Stream) {
|
||||
memory_type_bits = 0u;
|
||||
}
|
||||
const VmaAllocationCreateInfo alloc_ci = {
|
||||
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
|
||||
.usage = MemoryUsageVma(usage),
|
||||
.requiredFlags = 0,
|
||||
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
|
||||
.memoryTypeBits = memory_type_bits,
|
||||
.pool = VK_NULL_HANDLE,
|
||||
.pUserData = nullptr,
|
||||
.priority = 0.f,
|
||||
};
|
||||
|
||||
VmaAllocation a{};
|
||||
VmaAllocationInfo info{};
|
||||
VkBuffer handle{};
|
||||
VmaAllocationInfo alloc_info{};
|
||||
VmaAllocation allocation{};
|
||||
VkMemoryPropertyFlags property_flags{};
|
||||
|
||||
// Let VMA infer memory requirements from the buffer
|
||||
VkResult res = vmaAllocateMemoryForBuffer(allocator, raw, &ci, &a, &info);
|
||||
vk::Check(vmaCreateBufferWithAlignment(allocator, &ci, &alloc_ci, min_alignment, &handle,
|
||||
&allocation, &alloc_info));
|
||||
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
|
||||
|
||||
if (res != VK_SUCCESS) {
|
||||
auto ci2 = ci;
|
||||
ci2.flags &= ~VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT;
|
||||
res = vmaAllocateMemoryForBuffer(allocator, raw, &ci2, &a, &info);
|
||||
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
|
||||
std::span<u8> mapped_data{};
|
||||
if (data) {
|
||||
mapped_data = std::span<u8>{data, ci.size};
|
||||
}
|
||||
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
|
||||
|
||||
if (res != VK_SUCCESS && (ci.preferredFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT)) {
|
||||
auto ci3 = ci2;
|
||||
ci3.preferredFlags &= ~VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
|
||||
res = vmaAllocateMemoryForBuffer(allocator, raw, &ci3, &a, &info);
|
||||
}
|
||||
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
|
||||
device.GetDispatchLoader());
|
||||
}
|
||||
|
||||
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
|
||||
{
|
||||
const auto vma_usage = MemoryUsageVma(usage);
|
||||
VmaAllocationCreateInfo ci{};
|
||||
ci.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage);
|
||||
ci.usage = vma_usage;
|
||||
ci.memoryTypeBits = reqs.memoryTypeBits & valid_memory_types;
|
||||
ci.requiredFlags = 0;
|
||||
ci.preferredFlags = MemoryUsagePreferredVmaFlags(usage);
|
||||
|
||||
VmaAllocation a{};
|
||||
VmaAllocationInfo info{};
|
||||
|
||||
VkResult res = vmaAllocateMemory(allocator, &reqs, &ci, &a, &info);
|
||||
|
||||
if (res != VK_SUCCESS) {
|
||||
// Relax 1: drop budget constraint
|
||||
auto ci2 = ci;
|
||||
ci2.flags &= ~VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT;
|
||||
res = vmaAllocateMemory(allocator, &reqs, &ci2, &a, &info);
|
||||
|
||||
// Relax 2: if we preferred DEVICE_LOCAL, drop that preference
|
||||
if (res != VK_SUCCESS && (ci.preferredFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT)) {
|
||||
auto ci3 = ci2;
|
||||
ci3.preferredFlags &= ~VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
|
||||
res = vmaAllocateMemory(allocator, &reqs, &ci3, &a, &info);
|
||||
}
|
||||
|
||||
vk::Check(res);
|
||||
vk::Check(vmaBindBufferMemory2(allocator, a, 0, raw, nullptr));
|
||||
return MemoryCommit(allocator, a, info);
|
||||
}
|
||||
|
||||
vk::Check(res);
|
||||
return MemoryCommit(allocator, a, info);
|
||||
}
|
||||
|
||||
MemoryCommit MemoryAllocator::Commit(const vk::Buffer &buffer, MemoryUsage usage) {
|
||||
// Allocate memory appropriate for this buffer automatically
|
||||
const auto vma_usage = MemoryUsageVma(usage);
|
||||
|
||||
VmaAllocationCreateInfo ci{};
|
||||
ci.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage);
|
||||
ci.usage = vma_usage;
|
||||
ci.requiredFlags = 0;
|
||||
ci.preferredFlags = MemoryUsagePreferredVmaFlags(usage);
|
||||
ci.pool = VK_NULL_HANDLE;
|
||||
ci.pUserData = nullptr;
|
||||
ci.priority = 0.0f;
|
||||
|
||||
const VkBuffer raw = *buffer;
|
||||
|
||||
VmaAllocation a{};
|
||||
VmaAllocationInfo info{};
|
||||
|
||||
// Let VMA infer memory requirements from the buffer
|
||||
VkResult res = vmaAllocateMemoryForBuffer(allocator, raw, &ci, &a, &info);
|
||||
|
||||
if (res != VK_SUCCESS) {
|
||||
auto ci2 = ci;
|
||||
ci2.flags &= ~VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT;
|
||||
res = vmaAllocateMemoryForBuffer(allocator, raw, &ci2, &a, &info);
|
||||
|
||||
if (res != VK_SUCCESS && (ci.preferredFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT)) {
|
||||
auto ci3 = ci2;
|
||||
ci3.preferredFlags &= ~VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
|
||||
res = vmaAllocateMemoryForBuffer(allocator, raw, &ci3, &a, &info);
|
||||
}
|
||||
}
|
||||
|
||||
vk::Check(res);
|
||||
vk::Check(vmaBindBufferMemory2(allocator, a, 0, raw, nullptr));
|
||||
return MemoryCommit(allocator, a, info);
|
||||
}
|
||||
|
||||
} // namespace Vulkan
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
|
||||
@@ -107,6 +107,9 @@ namespace Vulkan {
|
||||
|
||||
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const;
|
||||
|
||||
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
|
||||
VkDeviceSize min_alignment) const;
|
||||
|
||||
/**
|
||||
* Commits a memory with the specified requirements.
|
||||
*
|
||||
|
||||
@@ -229,6 +229,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
|
||||
X(vkGetPipelineExecutableStatisticsKHR);
|
||||
X(vkGetSemaphoreCounterValue);
|
||||
X(vkMapMemory);
|
||||
X(vkQueueBindSparse);
|
||||
X(vkQueueSubmit);
|
||||
X(vkQueueSubmit2);
|
||||
X(vkResetFences);
|
||||
@@ -539,6 +540,19 @@ void Buffer::SetObjectNameEXT(const char* name) const {
|
||||
SetObjectName(dld, owner, handle, VK_OBJECT_TYPE_BUFFER, name);
|
||||
}
|
||||
|
||||
MemoryLocation Buffer::Location() const noexcept {
|
||||
if (!allocation) {
|
||||
return MemoryLocation{};
|
||||
}
|
||||
VmaAllocationInfo info{};
|
||||
vmaGetAllocationInfo(allocator, allocation, &info);
|
||||
return MemoryLocation{
|
||||
.memory = info.deviceMemory,
|
||||
.offset = info.offset,
|
||||
.memory_type = info.memoryType,
|
||||
};
|
||||
}
|
||||
|
||||
void Buffer::Release() const noexcept {
|
||||
if (handle) {
|
||||
vmaDestroyBuffer(allocator, handle, allocation);
|
||||
|
||||
@@ -345,6 +345,7 @@ struct DeviceDispatch : InstanceDispatch {
|
||||
PFN_vkGetQueryPoolResults vkGetQueryPoolResults{};
|
||||
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
|
||||
PFN_vkMapMemory vkMapMemory{};
|
||||
PFN_vkQueueBindSparse vkQueueBindSparse{};
|
||||
PFN_vkQueueSubmit vkQueueSubmit{};
|
||||
PFN_vkQueueSubmit2 vkQueueSubmit2{};
|
||||
PFN_vkResetFences vkResetFences{};
|
||||
@@ -740,6 +741,12 @@ private:
|
||||
const DeviceDispatch* dld = nullptr;
|
||||
};
|
||||
|
||||
struct MemoryLocation {
|
||||
VkDeviceMemory memory{};
|
||||
VkDeviceSize offset{};
|
||||
u32 memory_type{};
|
||||
};
|
||||
|
||||
class Buffer {
|
||||
public:
|
||||
explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_,
|
||||
@@ -811,6 +818,8 @@ public:
|
||||
|
||||
void SetObjectNameEXT(const char* name) const;
|
||||
|
||||
MemoryLocation Location() const noexcept;
|
||||
|
||||
private:
|
||||
void Release() const noexcept;
|
||||
|
||||
@@ -843,6 +852,11 @@ public:
|
||||
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
|
||||
}
|
||||
|
||||
VkResult BindSparse(Span<VkBindSparseInfo> bind_infos,
|
||||
VkFence fence = VK_NULL_HANDLE) const noexcept {
|
||||
return dld->vkQueueBindSparse(queue, bind_infos.size(), bind_infos.data(), fence);
|
||||
}
|
||||
|
||||
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
|
||||
return dld->vkQueuePresentKHR(queue, &present_info);
|
||||
}
|
||||
|
||||
@@ -4796,6 +4796,6 @@ void VolumeButton::ResetMultiplier() {
|
||||
#endif
|
||||
|
||||
#if !defined(QT_STATICPLUGIN) || defined(__APPLE__)
|
||||
#define VMA_IMPLEMENTATION
|
||||
#define VMA_IMPLEMENTATION 1
|
||||
#include "video_core/vulkan_common/vma.h"
|
||||
#endif
|
||||
|
||||
@@ -65,7 +65,7 @@ else()
|
||||
endif()
|
||||
|
||||
# update cached cpmfile content
|
||||
string(JSON cpmfile SET "${cpmfile}" "${key}" "${new_object}")
|
||||
string(JSON cpmfile SET "${cpmfile}" "${KEY}" "${new_object}")
|
||||
|
||||
# write cached cpmfile
|
||||
get_cpmfile_path(file)
|
||||
|
||||
Reference in New Issue
Block a user