Compare commits

..

2 Commits

Author SHA1 Message Date
lizzie 62c413b10e 2026-09-06 21:26:09
Signed-off-by: lizzie <lizzie@eden-emu.dev>
2026-09-06 21:26:09 +00:00
lizzie 37a605ff9e 2026-09-06 20:55:53
Signed-off-by: lizzie <lizzie@eden-emu.dev>
2026-09-06 20:55:53 +00:00
28 changed files with 77 additions and 1141 deletions
+16 -69
View File
@@ -11,34 +11,15 @@ set(dynarmic_VERSION ${dynarmic_VERSION_MAJOR}.${dynarmic_VERSION_MINOR}.${dynar
project(dynarmic LANGUAGES C CXX ASM VERSION ${dynarmic_VERSION}) project(dynarmic LANGUAGES C CXX ASM VERSION ${dynarmic_VERSION})
# Determine if we're built as a subproject (using add_subdirectory)
# or if this is the master project.
set(MASTER_PROJECT OFF)
if (CMAKE_CURRENT_SOURCE_DIR STREQUAL CMAKE_SOURCE_DIR)
set(MASTER_PROJECT ON)
endif()
if (MASTER_PROJECT)
include(CTest)
endif()
# Dynarmic project options
option(DYNARMIC_ENABLE_CPU_FEATURE_DETECTION "Turning this off causes dynarmic to assume the host CPU doesn't support anything later than SSE3" ON)
if (OPENBSD OR DRAGONFLY OR NETBSD) if (OPENBSD OR DRAGONFLY OR NETBSD)
set(REQUIRE_WX ON) set(REQUIRE_WX ON)
else() else()
set(REQUIRE_WX OFF) set(REQUIRE_WX OFF)
endif() endif()
option(DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT "Enables support for systems that require W^X" ${REQUIRE_WX}) option(DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT "Enables support for systems that require W^X" ${REQUIRE_WX})
option(DYNARMIC_IGNORE_ASSERTS "Ignore asserts" ON)
option(DYNARMIC_TESTS_USE_UNICORN "Enable fuzzing tests against unicorn" OFF) option(DYNARMIC_TESTS_USE_UNICORN "Enable fuzzing tests against unicorn" OFF)
CMAKE_DEPENDENT_OPTION(DYNARMIC_USE_LLVM "Support disassembly of jitted x86_64 code using LLVM" OFF "NOT YUZU_DISABLE_LLVM" OFF) CMAKE_DEPENDENT_OPTION(DYNARMIC_USE_LLVM "Support disassembly of jitted x86_64 code using LLVM" OFF "NOT YUZU_DISABLE_LLVM" OFF)
option(DYNARMIC_INSTALL "Install dynarmic headers and CMake files" OFF)
option(DYNARMIC_USE_BUNDLED_EXTERNALS "Use all bundled externals (useful when e.g. cross-compiling)" OFF)
# Set hard requirements for C++ # Set hard requirements for C++
set(CMAKE_CXX_STANDARD 20) set(CMAKE_CXX_STANDARD 20)
set(CMAKE_CXX_STANDARD_REQUIRED ON) set(CMAKE_CXX_STANDARD_REQUIRED ON)
@@ -88,7 +69,7 @@ if (MSVC)
-Wno-missing-braces) -Wno-missing-braces)
endif() endif()
else() else()
set(DYNARMIC_CXX_FLAGS list(APPEND DYNARMIC_CXX_FLAGS
-Wall -Wall
-Wextra -Wextra
-Wcast-qual -Wcast-qual
@@ -103,37 +84,35 @@ else()
# to Address& after first checking isMEM(), and that code is inlined in a situation where # to Address& after first checking isMEM(), and that code is inlined in a situation where
# GCC knows that the variable is actually a Reg64. isMEM() will never return true for a # GCC knows that the variable is actually a Reg64. isMEM() will never return true for a
# Reg64, but GCC doesn't know that. # Reg64, but GCC doesn't know that.
list(APPEND DYNARMIC_CXX_FLAGS -Wno-array-bounds) list(APPEND DYNARMIC_CXX_FLAGS
list(APPEND DYNARMIC_CXX_FLAGS -Wstack-usage=4096) -Wno-array-bounds
-Wstack-usage=4096)
endif() endif()
if (CXX_CLANG) if (CXX_CLANG)
# Bracket depth determines maximum size of a fold expression in Clang since 9c9974c3ccb6. list(APPEND DYNARMIC_CXX_FLAGS
# And this in turns limits the size of a std::array. # Bracket depth determines maximum size of a fold expression in Clang since 9c9974c3ccb6.
list(APPEND DYNARMIC_CXX_FLAGS -fbracket-depth=1024) # And this in turns limits the size of a std::array.
# Clang mistakenly blames CMake for using unused arguments during compilation -fbracket-depth=1024
list(APPEND DYNARMIC_CXX_FLAGS -Wno-unused-command-line-argument) # Clang mistakenly blames CMake for using unused arguments during compilation
-Wno-unused-command-line-argument)
endif() endif()
endif() endif()
if (NOT Boost_FOUND) if (NOT Boost_FOUND)
find_package(Boost 1.57 REQUIRED) find_package(Boost 1.57 REQUIRED)
endif() endif()
find_package(fmt 8 CONFIG) find_package(fmt 8 CONFIG)
if ("arm64" IN_LIST ARCHITECTURE OR DYNARMIC_TESTS) if (ARCHITECTURE_arm64)
find_package(oaknut 2.0.1 CONFIG) find_package(oaknut 2.0.1 CONFIG)
endif() endif()
if (ARCHITECTURE_riscv64)
if ("riscv64" IN_LIST ARCHITECTURE)
find_package(biscuit 0.9.1 REQUIRED) find_package(biscuit 0.9.1 REQUIRED)
endif() endif()
if (ARCHITECTURE_loongarch64)
if ("loongarch64" IN_LIST ARCHITECTURE)
find_package(lagoon REQUIRED) find_package(lagoon REQUIRED)
endif() endif()
if (ARCHITECTURE_x86_64)
if ("x86_64" IN_LIST ARCHITECTURE)
find_package(xbyak 7 CONFIG) find_package(xbyak 7 CONFIG)
endif() endif()
@@ -142,44 +121,12 @@ if (DYNARMIC_USE_LLVM)
separate_arguments(LLVM_DEFINITIONS) separate_arguments(LLVM_DEFINITIONS)
endif() endif()
# Dynarmic project files
add_subdirectory(src/dynarmic)
if (DYNARMIC_TESTS) if (DYNARMIC_TESTS)
find_package(Catch2 3 CONFIG) find_package(Catch2 3 CONFIG)
if (DYNARMIC_TESTS_USE_UNICORN) if (DYNARMIC_TESTS_USE_UNICORN)
find_package(Unicorn REQUIRED) find_package(Unicorn REQUIRED)
endif() endif()
endif()
# Dynarmic project files
add_subdirectory(src/dynarmic)
if (DYNARMIC_TESTS)
add_subdirectory(tests) add_subdirectory(tests)
endif() endif()
#
# Install
#
if (DYNARMIC_INSTALL)
include(GNUInstallDirs)
include(CMakePackageConfigHelpers)
install(TARGETS dynarmic EXPORT dynarmicTargets)
install(EXPORT dynarmicTargets
NAMESPACE dynarmic::
DESTINATION "${CMAKE_INSTALL_LIBDIR}/cmake/dynarmic"
)
configure_package_config_file(CMakeModules/dynarmicConfig.cmake.in
dynarmicConfig.cmake
INSTALL_DESTINATION "${CMAKE_INSTALL_LIBDIR}/cmake/dynarmic"
)
write_basic_package_version_file(dynarmicConfigVersion.cmake
COMPATIBILITY SameMajorVersion
)
install(FILES
"${CMAKE_CURRENT_BINARY_DIR}/dynarmicConfig.cmake"
"${CMAKE_CURRENT_BINARY_DIR}/dynarmicConfigVersion.cmake"
DESTINATION "${CMAKE_INSTALL_LIBDIR}/cmake/dynarmic"
)
install(DIRECTORY src/dynarmic TYPE INCLUDE FILES_MATCHING PATTERN "*.h")
endif()
@@ -1,29 +0,0 @@
# SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
# SPDX-License-Identifier: GPL-3.0-or-later
function(target_architecture_specific_sources project arch)
if (NOT MULTIARCH_BUILD)
target_sources("${project}" PRIVATE ${ARGN})
return()
endif()
foreach(input_file IN LISTS ARGN)
if(input_file MATCHES ".cpp$")
if(NOT IS_ABSOLUTE ${input_file})
set(input_file "${CMAKE_CURRENT_SOURCE_DIR}/${input_file}")
endif()
set(output_file "${CMAKE_CURRENT_BINARY_DIR}/arch_gen/${input_file}")
add_custom_command(
OUTPUT "${output_file}"
COMMAND ${CMAKE_COMMAND} "-Darch=${arch}"
"-Dinput_file=${input_file}"
"-Doutput_file=${output_file}"
-P "${CMAKE_CURRENT_FUNCTION_LIST_DIR}/impl/TargetArchitectureSpecificSourcesWrapFile.cmake"
DEPENDS "${input_file}"
VERBATIM
)
target_sources(${project} PRIVATE "${output_file}")
endif()
endforeach()
endfunction()
@@ -1,31 +0,0 @@
@PACKAGE_INIT@
include(CMakeFindDependencyMacro)
set(ARCHITECTURE "@ARCHITECTURE@")
if (NOT @BUILD_SHARED_LIBS@)
find_dependency(Boost 1.57)
find_dependency(fmt 9)
find_dependency(mcl 0.1.12 EXACT)
if ("arm64" IN_LIST ARCHITECTURE)
find_dependency(oaknut 2.0.1)
endif()
if ("riscv" IN_LIST ARCHITECTURE)
find_dependency(biscuit 0.9.1)
endif()
if ("x86_64" IN_LIST ARCHITECTURE)
find_dependency(xbyak 7)
endif()
if (@DYNARMIC_USE_LLVM@)
find_dependency(LLVM)
endif()
endif()
include("${CMAKE_CURRENT_LIST_DIR}/@PROJECT_NAME@Targets.cmake")
check_required_components(@PROJECT_NAME@)
@@ -1,6 +0,0 @@
# SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
# SPDX-License-Identifier: GPL-3.0-or-later
string(TOUPPER "${arch}" arch)
file(READ "${input_file}" f_contents)
file(WRITE "${output_file}" "#if defined(ARCHITECTURE_${arch})\n${f_contents}\n#endif\n")
+6 -14
View File
@@ -1,7 +1,5 @@
# SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project # SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
# SPDX-License-Identifier: GPL-3.0-or-later # SPDX-License-Identifier: GPL-3.0-or-later
include(TargetArchitectureSpecificSources)
add_library(dynarmic STATIC add_library(dynarmic STATIC
mcl/bit.hpp mcl/bit.hpp
mcl/function_info.hpp mcl/function_info.hpp
@@ -135,7 +133,7 @@ add_library(dynarmic STATIC
interface/A64/config.h interface/A64/config.h
) )
if ("x86_64" IN_LIST ARCHITECTURE) if (ARCHITECTURE_x86_64)
# Newer versions of xbyak (>= 7.25.0) have stricter checks that currently # Newer versions of xbyak (>= 7.25.0) have stricter checks that currently
# fail in dynarmic # fail in dynarmic
target_compile_definitions(dynarmic PRIVATE XBYAK_STRICT_CHECK_MEM_REG_SIZE=0) target_compile_definitions(dynarmic PRIVATE XBYAK_STRICT_CHECK_MEM_REG_SIZE=0)
@@ -143,7 +141,7 @@ if ("x86_64" IN_LIST ARCHITECTURE)
target_compile_definitions(dynarmic PRIVATE XBYAK_OLD_DISP_CHECK=1) target_compile_definitions(dynarmic PRIVATE XBYAK_OLD_DISP_CHECK=1)
target_link_libraries(dynarmic PRIVATE xbyak::xbyak) target_link_libraries(dynarmic PRIVATE xbyak::xbyak)
target_architecture_specific_sources(dynarmic "x86_64" target_sources(dynarmic PRIVATE
backend/x64/abi.cpp backend/x64/abi.cpp
backend/x64/abi.h backend/x64/abi.h
backend/x64/block_of_code.cpp backend/x64/block_of_code.cpp
@@ -201,10 +199,10 @@ if ("x86_64" IN_LIST ARCHITECTURE)
) )
endif() endif()
if ("arm64" IN_LIST ARCHITECTURE) if (ARCHITECTURE_arm64)
target_link_libraries(dynarmic PRIVATE merry::oaknut) target_link_libraries(dynarmic PRIVATE merry::oaknut)
target_architecture_specific_sources(dynarmic "arm64" target_sources(dynarmic PRIVATE
backend/arm64/a32_jitstate.cpp backend/arm64/a32_jitstate.cpp
backend/arm64/a32_jitstate.h backend/arm64/a32_jitstate.h
backend/arm64/a64_jitstate.h backend/arm64/a64_jitstate.h
@@ -255,7 +253,7 @@ if ("arm64" IN_LIST ARCHITECTURE)
) )
endif() endif()
if ("riscv64" IN_LIST ARCHITECTURE) if (ARCHITECTURE_riscv64)
target_link_libraries(dynarmic PRIVATE biscuit::biscuit) target_link_libraries(dynarmic PRIVATE biscuit::biscuit)
target_sources(dynarmic PRIVATE target_sources(dynarmic PRIVATE
@@ -295,7 +293,7 @@ if ("riscv64" IN_LIST ARCHITECTURE)
message(WARNING "TODO: Incomplete frontend for this host architecture") message(WARNING "TODO: Incomplete frontend for this host architecture")
endif() endif()
if ("loongarch64" IN_LIST ARCHITECTURE) if (ARCHITECTURE_loongarch64)
target_link_libraries(dynarmic PRIVATE lagoon::lagoon) target_link_libraries(dynarmic PRIVATE lagoon::lagoon)
target_sources(dynarmic PRIVATE target_sources(dynarmic PRIVATE
@@ -400,15 +398,9 @@ if (DYNARMIC_USE_LLVM)
target_compile_definitions(dynarmic PRIVATE DYNARMIC_USE_LLVM=1 ${LLVM_DEFINITIONS}) target_compile_definitions(dynarmic PRIVATE DYNARMIC_USE_LLVM=1 ${LLVM_DEFINITIONS})
llvm_config(dynarmic USE_SHARED armdesc armdisassembler aarch64desc aarch64disassembler x86desc x86disassembler) llvm_config(dynarmic USE_SHARED armdesc armdisassembler aarch64desc aarch64disassembler x86desc x86disassembler)
endif() endif()
if (DYNARMIC_ENABLE_CPU_FEATURE_DETECTION)
target_compile_definitions(dynarmic PRIVATE DYNARMIC_ENABLE_CPU_FEATURE_DETECTION=1)
endif()
if (DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT) if (DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT)
target_compile_definitions(dynarmic PRIVATE DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT=1) target_compile_definitions(dynarmic PRIVATE DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT=1)
endif() endif()
if (DYNARMIC_IGNORE_ASSERTS)
target_compile_definitions(dynarmic PRIVATE MCL_IGNORE_ASSERTS=1)
endif()
if (CMAKE_SYSTEM_NAME STREQUAL "Windows") if (CMAKE_SYSTEM_NAME STREQUAL "Windows")
target_compile_definitions(dynarmic PRIVATE FMT_USE_WINDOWS_H=0) target_compile_definitions(dynarmic PRIVATE FMT_USE_WINDOWS_H=0)
endif() endif()
@@ -78,7 +78,6 @@ void ProtectMemory(const void* base, size_t size, bool is_executable) {
static const HostFeature features = []() { static const HostFeature features = []() {
HostFeature f{}; HostFeature f{};
#ifdef DYNARMIC_ENABLE_CPU_FEATURE_DETECTION
using Cpu = Xbyak::util::Cpu; using Cpu = Xbyak::util::Cpu;
Xbyak::util::Cpu cpu_info{}; Xbyak::util::Cpu cpu_info{};
if (cpu_info.has(Cpu::tSSSE3)) f |= HostFeature::SSSE3; if (cpu_info.has(Cpu::tSSSE3)) f |= HostFeature::SSSE3;
@@ -121,7 +120,6 @@ static const HostFeature features = []() {
} }
} }
return f; return f;
#endif
}(); }();
HostFeature GetHostFeatures() { HostFeature GetHostFeatures() {
return features; return features;
@@ -235,6 +235,7 @@ const void* EmitReadMemoryMov(BlockOfCode& code, int value_idx, const Xbyak::Reg
code.xadd(qword[addr], Xbyak::Reg64(value_idx)); code.xadd(qword[addr], Xbyak::Reg64(value_idx));
break; break;
case 128: case 128:
ASSERT(Xbyak::Xmm(value_idx) != xmm0);
code.lock(); code.lock();
code.cmpxchg16b(xword[addr]); code.cmpxchg16b(xword[addr]);
if (code.HasHostFeature(HostFeature::SSE41)) { if (code.HasHostFeature(HostFeature::SSE41)) {
@@ -6226,9 +6226,7 @@ void EmitX64::EmitVectorZeroExtend64(EmitContext& ctx, IR::Inst* inst) {
void EmitX64::EmitVectorZeroUpper(EmitContext& ctx, IR::Inst* inst) { void EmitX64::EmitVectorZeroUpper(EmitContext& ctx, IR::Inst* inst) {
auto args = ctx.reg_alloc.GetArgumentInfo(inst); auto args = ctx.reg_alloc.GetArgumentInfo(inst);
auto const a = ctx.reg_alloc.UseScratchXmm(code, args[0]); auto const a = ctx.reg_alloc.UseScratchXmm(code, args[0]);
code.movq(a, a); // TODO: !IsLastUse code.movq(a, a); // TODO: !IsLastUse
ctx.reg_alloc.DefineValue(code, inst, a); ctx.reg_alloc.DefineValue(code, inst, a);
} }
@@ -14,12 +14,6 @@
#include "dynarmic/backend/x64/hostloc.h" #include "dynarmic/backend/x64/hostloc.h"
#include "dynarmic/common/spin_lock.h" #include "dynarmic/common/spin_lock.h"
#ifdef DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT
static const auto default_cg_mode = Xbyak::DontSetProtectRWE;
#else
static const auto default_cg_mode = nullptr; //Allow RWE
#endif
namespace Dynarmic { namespace Dynarmic {
void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32 tmp, bool waitpkg) { void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32 tmp, bool waitpkg) {
@@ -78,7 +72,13 @@ namespace {
struct SpinLockImpl { struct SpinLockImpl {
void Initialize() noexcept; void Initialize() noexcept;
static void GlobalInitialize() noexcept; static void GlobalInitialize() noexcept;
Xbyak::CodeGenerator code = Xbyak::CodeGenerator(4096, default_cg_mode); Xbyak::CodeGenerator code = Xbyak::CodeGenerator(4096
#ifdef DYNARMIC_ENABLE_NO_EXECUTE_SUPPORT
, Xbyak::DontSetProtectRWE
#else
, nullptr //Allow RWE
#endif
);
void (*lock)(volatile int*) = nullptr; void (*lock)(volatile int*) = nullptr;
void (*unlock)(volatile int*) = nullptr; void (*unlock)(volatile int*) = nullptr;
}; };
@@ -91,7 +91,7 @@ struct detail {
shifts[arg_index] = bit_position; shifts[arg_index] = bit_position;
} }
} }
#if !defined(DYNARMIC_IGNORE_ASSERTS) && !defined(__ANDROID__) #if !defined(__ANDROID__)
// Avoids a MSVC ICE, and avoids Android NDK issue. // Avoids a MSVC ICE, and avoids Android NDK issue.
DEBUG_ASSERT(std::all_of(masks.begin(), masks.end(), [](auto m) { return m != 0; })); DEBUG_ASSERT(std::all_of(masks.begin(), masks.end(), [](auto m) { return m != 0; }));
#endif #endif
+3 -5
View File
@@ -1,7 +1,5 @@
# SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project # SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
# SPDX-License-Identifier: GPL-3.0-or-later # SPDX-License-Identifier: GPL-3.0-or-later
include(TargetArchitectureSpecificSources)
add_executable(dynarmic_tests add_executable(dynarmic_tests
fp/FPToFixed.cpp fp/FPToFixed.cpp
fp/FPValue.cpp fp/FPValue.cpp
@@ -44,13 +42,13 @@ if (DYNARMIC_TESTS_USE_UNICORN)
) )
endif() endif()
if ("riscv" IN_LIST ARCHITECTURE) if (ARCHITECTURE_riscv64)
target_link_libraries(dynarmic_tests PRIVATE biscuit::biscuit) target_link_libraries(dynarmic_tests PRIVATE biscuit::biscuit)
endif() endif()
if ("x86_64" IN_LIST ARCHITECTURE) if (ARCHITECTURE_x86_64)
target_link_libraries(dynarmic_tests PRIVATE xbyak::xbyak) target_link_libraries(dynarmic_tests PRIVATE xbyak::xbyak)
target_architecture_specific_sources(dynarmic_tests "x86_64" target_sources(dynarmic_tests PRIVATE
x64_cpu_info.cpp x64_cpu_info.cpp
native/preserve_xmm.cpp native/preserve_xmm.cpp
) )
-3
View File
@@ -20,7 +20,6 @@ add_library(video_core STATIC
buffer_cache/buffer_cache.h buffer_cache/buffer_cache.h
buffer_cache/memory_tracker_base.h buffer_cache/memory_tracker_base.h
buffer_cache/usage_tracker.h buffer_cache/usage_tracker.h
buffer_cache/virtual_range_cache.h
buffer_cache/word_manager.h buffer_cache/word_manager.h
cache_types.h cache_types.h
capture.h capture.h
@@ -167,8 +166,6 @@ add_library(video_core STATIC
renderer_vulkan/vk_fence_manager.h renderer_vulkan/vk_fence_manager.h
renderer_vulkan/vk_graphics_pipeline.cpp renderer_vulkan/vk_graphics_pipeline.cpp
renderer_vulkan/vk_graphics_pipeline.h renderer_vulkan/vk_graphics_pipeline.h
renderer_vulkan/vk_multi_range_buffer.cpp
renderer_vulkan/vk_multi_range_buffer.h
renderer_vulkan/vk_master_semaphore.cpp renderer_vulkan/vk_master_semaphore.cpp
renderer_vulkan/vk_master_semaphore.h renderer_vulkan/vk_master_semaphore.h
renderer_vulkan/vk_pipeline_cache.cpp renderer_vulkan/vk_pipeline_cache.cpp
+26 -134
View File
@@ -112,13 +112,6 @@ void BufferCache<P>::TickFrame() {
async_buffers_death_ring.clear(); async_buffers_death_ring.clear();
} }
template <class P>
void BufferCache<P>::UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size) {
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
virtual_ranges.Unmap(as_id, gpu_addr, size);
}
}
template <class P> template <class P>
void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) { void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) {
if (memory_tracker.IsRegionGpuModified(device_addr, size)) { if (memory_tracker.IsRegionGpuModified(device_addr, size)) {
@@ -215,8 +208,8 @@ bool BufferCache<P>::DMACopy(GPUVAddr src_address, GPUVAddr dest_address, u64 am
BufferId buffer_b; BufferId buffer_b;
do { do {
channel_state->has_deleted_buffers = false; channel_state->has_deleted_buffers = false;
buffer_a = FindBuffer(*cpu_src_address, static_cast<u32>(amount), false); buffer_a = FindBuffer(*cpu_src_address, static_cast<u32>(amount));
buffer_b = FindBuffer(*cpu_dest_address, static_cast<u32>(amount), false); buffer_b = FindBuffer(*cpu_dest_address, static_cast<u32>(amount));
} while (channel_state->has_deleted_buffers); } while (channel_state->has_deleted_buffers);
auto& src_buffer = slot_buffers[buffer_a]; auto& src_buffer = slot_buffers[buffer_a];
auto& dest_buffer = slot_buffers[buffer_b]; auto& dest_buffer = slot_buffers[buffer_b];
@@ -272,7 +265,7 @@ bool BufferCache<P>::DMAClear(GPUVAddr dst_address, u64 amount, u32 value) {
ClearDownload(*cpu_dst_address, size); ClearDownload(*cpu_dst_address, size);
gpu_modified_ranges.Subtract(*cpu_dst_address, size); gpu_modified_ranges.Subtract(*cpu_dst_address, size);
const BufferId buffer = FindBuffer(*cpu_dst_address, static_cast<u32>(size), false); const BufferId buffer = FindBuffer(*cpu_dst_address, static_cast<u32>(size));
Buffer& dest_buffer = slot_buffers[buffer]; Buffer& dest_buffer = slot_buffers[buffer];
const u32 offset = dest_buffer.Offset(*cpu_dst_address); const u32 offset = dest_buffer.Offset(*cpu_dst_address);
runtime.ClearBuffer(dest_buffer, offset, size, value); runtime.ClearBuffer(dest_buffer, offset, size, value);
@@ -294,7 +287,7 @@ std::pair<typename P::Buffer*, u32> BufferCache<P>::ObtainBuffer(GPUVAddr gpu_ad
template <class P> template <class P>
std::pair<typename P::Buffer*, u32> BufferCache<P>::ObtainCPUBuffer( std::pair<typename P::Buffer*, u32> BufferCache<P>::ObtainCPUBuffer(
DAddr device_addr, u32 size, ObtainBufferSynchronize sync_info, ObtainBufferOperation post_op) { DAddr device_addr, u32 size, ObtainBufferSynchronize sync_info, ObtainBufferOperation post_op) {
const BufferId buffer_id = FindBuffer(device_addr, size, false); const BufferId buffer_id = FindBuffer(device_addr, size);
Buffer& buffer = slot_buffers[buffer_id]; Buffer& buffer = slot_buffers[buffer_id];
// synchronize op // synchronize op
@@ -1005,85 +998,11 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
channel_state->fast_bound_uniform_buffers[stage] &= ~(1u << binding_index); channel_state->fast_bound_uniform_buffers[stage] &= ~(1u << binding_index);
} }
template <class P>
void BufferCache<P>::ResolveMultiRangeStorage(Binding& binding, bool is_written,
std::vector<MultiRangeSegment>& pool) {
binding.segment_first = 0;
binding.segment_count = 0;
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
if (binding.gpu_addr == 0 || binding.size == 0) {
return;
}
if (is_written && !runtime.PrefersSparseSources()) {
return;
}
const VirtualSegments* found =
virtual_ranges.Query(*gpu_memory, binding.gpu_addr, binding.size);
if (!found || found->size() < 2) {
return;
}
const VirtualSegments segments = *found;
const u32 first = static_cast<u32>(pool.size());
const bool prefer_sparse = runtime.PrefersSparseSources();
for (const VirtualSegment& segment : segments) {
const BufferId buffer_id =
FindBuffer(segment.device_addr, segment.size, prefer_sparse);
if (!buffer_id) {
pool.resize(first);
return;
}
pool.push_back(MultiRangeSegment{
.buffer_id = buffer_id,
.device_addr = segment.device_addr,
.size = segment.size,
});
}
binding.segment_first = first;
binding.segment_count = static_cast<u32>(segments.size());
}
}
template <class P>
bool BufferCache<P>::BindMultiRangeStorage(const Binding& binding, bool is_written,
std::span<const MultiRangeSegment> pool) {
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
if (binding.segment_count < 2) {
return false;
}
if (binding.segment_first + binding.segment_count > pool.size()) {
return false;
}
const u64 key = (static_cast<u64>(gpu_memory->GetID()) << 48) ^ binding.gpu_addr;
runtime.ResetMultiRange();
for (u32 index = 0; index < binding.segment_count; ++index) {
const MultiRangeSegment& segment = pool[binding.segment_first + index];
Buffer& buffer = slot_buffers[segment.buffer_id];
TouchBuffer(buffer, segment.buffer_id);
if (SynchronizeBuffer(buffer, segment.device_addr, segment.size)) {
runtime.InvalidateMultiRange(key);
}
const u32 offset = buffer.Offset(segment.device_addr);
buffer.MarkUsage(offset, segment.size);
if (is_written) {
MarkWrittenBuffer(segment.buffer_id, segment.device_addr, segment.size);
}
runtime.PushMultiRangeSource(buffer, offset, segment.size);
}
return runtime.BindMultiRangeStorageBuffer(key, is_written);
} else {
return false;
}
}
template <class P> template <class P>
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) { void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
u32 binding_index = 0; u32 binding_index = 0;
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) { ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
const Binding& binding = channel_state->storage_buffers[stage][index]; const Binding& binding = channel_state->storage_buffers[stage][index];
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
if (BindMultiRangeStorage(binding, is_written, graphics_segments)) {
return;
}
Buffer& buffer = slot_buffers[binding.buffer_id]; Buffer& buffer = slot_buffers[binding.buffer_id];
TouchBuffer(buffer, binding.buffer_id); TouchBuffer(buffer, binding.buffer_id);
const u32 size = binding.size; const u32 size = binding.size;
@@ -1091,6 +1010,7 @@ void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
const u32 offset = buffer.Offset(binding.device_addr); const u32 offset = buffer.Offset(binding.device_addr);
buffer.MarkUsage(offset, size); buffer.MarkUsage(offset, size);
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
if (is_written) { if (is_written) {
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size); MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
@@ -1219,11 +1139,6 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
u32 binding_index = 0; u32 binding_index = 0;
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) { ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
const Binding& binding = channel_state->compute_storage_buffers[index]; const Binding& binding = channel_state->compute_storage_buffers[index];
const bool is_written =
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
if (BindMultiRangeStorage(binding, is_written, compute_segments)) {
return;
}
Buffer& buffer = slot_buffers[binding.buffer_id]; Buffer& buffer = slot_buffers[binding.buffer_id];
TouchBuffer(buffer, binding.buffer_id); TouchBuffer(buffer, binding.buffer_id);
const u32 size = binding.size; const u32 size = binding.size;
@@ -1231,6 +1146,8 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
const u32 offset = buffer.Offset(binding.device_addr); const u32 offset = buffer.Offset(binding.device_addr);
buffer.MarkUsage(offset, size); buffer.MarkUsage(offset, size);
const bool is_written =
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
if (is_written) { if (is_written) {
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size); MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
@@ -1276,7 +1193,6 @@ void BufferCache<P>::BindHostComputeTextureBuffers() {
template <class P> template <class P>
void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) { void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) {
graphics_segments.clear();
BufferOperations([&]() { BufferOperations([&]() {
if (is_indexed) { if (is_indexed) {
UpdateIndexBuffer(); UpdateIndexBuffer();
@@ -1296,7 +1212,6 @@ void BufferCache<P>::DoUpdateGraphicsBuffers(bool is_indexed) {
template <class P> template <class P>
void BufferCache<P>::DoUpdateComputeBuffers() { void BufferCache<P>::DoUpdateComputeBuffers() {
compute_segments.clear();
BufferOperations([&]() { BufferOperations([&]() {
UpdateComputeUniformBuffers(); UpdateComputeUniformBuffers();
UpdateComputeStorageBuffers(); UpdateComputeStorageBuffers();
@@ -1319,11 +1234,11 @@ void BufferCache<P>::UpdateIndexBuffer() {
auto inline_index_size = static_cast<u32>(draw_state.inline_index_draw_indexes.size()); auto inline_index_size = static_cast<u32>(draw_state.inline_index_draw_indexes.size());
u32 buffer_size = Common::AlignUp(inline_index_size, CACHING_PAGESIZE); u32 buffer_size = Common::AlignUp(inline_index_size, CACHING_PAGESIZE);
if (inline_buffer_id == NULL_BUFFER_ID) [[unlikely]] { if (inline_buffer_id == NULL_BUFFER_ID) [[unlikely]] {
inline_buffer_id = CreateBuffer(0, buffer_size, false); inline_buffer_id = CreateBuffer(0, buffer_size);
} }
if (slot_buffers[inline_buffer_id].SizeBytes() < buffer_size) [[unlikely]] { if (slot_buffers[inline_buffer_id].SizeBytes() < buffer_size) [[unlikely]] {
slot_buffers.erase(inline_buffer_id); slot_buffers.erase(inline_buffer_id);
inline_buffer_id = CreateBuffer(0, buffer_size, false); inline_buffer_id = CreateBuffer(0, buffer_size);
} }
channel_state->index_buffer = Binding{ channel_state->index_buffer = Binding{
.device_addr = 0, .device_addr = 0,
@@ -1346,7 +1261,7 @@ void BufferCache<P>::UpdateIndexBuffer() {
channel_state->index_buffer = Binding{ channel_state->index_buffer = Binding{
.device_addr = *device_addr, .device_addr = *device_addr,
.size = size, .size = size,
.buffer_id = FindBuffer(*device_addr, size, false), .buffer_id = FindBuffer(*device_addr, size),
}; };
} }
@@ -1383,7 +1298,7 @@ void BufferCache<P>::UpdateVertexBuffer(u32 index) {
if (!gpu_memory->IsWithinGPUAddressRange(gpu_addr_end) || size >= 64_MiB) { if (!gpu_memory->IsWithinGPUAddressRange(gpu_addr_end) || size >= 64_MiB) {
size = static_cast<u32>(gpu_memory->MaxContinuousRange(gpu_addr_begin, size)); size = static_cast<u32>(gpu_memory->MaxContinuousRange(gpu_addr_begin, size));
} }
const BufferId buffer_id = FindBuffer(*device_addr, size, false); const BufferId buffer_id = FindBuffer(*device_addr, size);
const Binding binding{ const Binding binding{
.device_addr = *device_addr, .device_addr = *device_addr,
.size = size, .size = size,
@@ -1404,7 +1319,7 @@ void BufferCache<P>::UpdateDrawIndirect() {
binding = Binding{ binding = Binding{
.device_addr = *device_addr, .device_addr = *device_addr,
.size = static_cast<u32>(size), .size = static_cast<u32>(size),
.buffer_id = FindBuffer(*device_addr, static_cast<u32>(size), false), .buffer_id = FindBuffer(*device_addr, static_cast<u32>(size)),
}; };
}; };
if (current_draw_indirect->include_count) { if (current_draw_indirect->include_count) {
@@ -1428,7 +1343,7 @@ void BufferCache<P>::UpdateUniformBuffers(size_t stage) {
channel_state->dirty_uniform_buffers[stage] |= 1U << index; channel_state->dirty_uniform_buffers[stage] |= 1U << index;
} }
// Resolve buffer // Resolve buffer
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false); binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
}); });
} }
@@ -1437,10 +1352,8 @@ void BufferCache<P>::UpdateStorageBuffers(size_t stage) {
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) { ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
// Resolve buffer // Resolve buffer
Binding& binding = channel_state->storage_buffers[stage][index]; Binding& binding = channel_state->storage_buffers[stage][index];
const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size, false); const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size);
binding.buffer_id = buffer_id; binding.buffer_id = buffer_id;
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
ResolveMultiRangeStorage(binding, is_written, graphics_segments);
}); });
} }
@@ -1448,7 +1361,7 @@ template <class P>
void BufferCache<P>::UpdateTextureBuffers(size_t stage) { void BufferCache<P>::UpdateTextureBuffers(size_t stage) {
ForEachEnabledBit(channel_state->enabled_texture_buffers[stage], [&](u32 index) { ForEachEnabledBit(channel_state->enabled_texture_buffers[stage], [&](u32 index) {
Binding& binding = channel_state->texture_buffers[stage][index]; Binding& binding = channel_state->texture_buffers[stage][index];
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false); binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
}); });
} }
@@ -1472,7 +1385,7 @@ void BufferCache<P>::UpdateTransformFeedbackBuffer(u32 index) {
channel_state->transform_feedback_buffers[index] = NULL_BINDING; channel_state->transform_feedback_buffers[index] = NULL_BINDING;
return; return;
} }
const BufferId buffer_id = FindBuffer(*device_addr, size, false); const BufferId buffer_id = FindBuffer(*device_addr, size);
channel_state->transform_feedback_buffers[index] = Binding{ channel_state->transform_feedback_buffers[index] = Binding{
.device_addr = *device_addr, .device_addr = *device_addr,
.size = size, .size = size,
@@ -1494,7 +1407,7 @@ void BufferCache<P>::UpdateComputeUniformBuffers() {
binding.size = cbuf.size; binding.size = cbuf.size;
} }
} }
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false); binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
}); });
} }
@@ -1503,10 +1416,7 @@ void BufferCache<P>::UpdateComputeStorageBuffers() {
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) { ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
// Resolve buffer // Resolve buffer
Binding& binding = channel_state->compute_storage_buffers[index]; Binding& binding = channel_state->compute_storage_buffers[index];
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false); binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
const bool is_written =
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
ResolveMultiRangeStorage(binding, is_written, compute_segments);
}); });
} }
@@ -1514,7 +1424,7 @@ template <class P>
void BufferCache<P>::UpdateComputeTextureBuffers() { void BufferCache<P>::UpdateComputeTextureBuffers() {
ForEachEnabledBit(channel_state->enabled_compute_texture_buffers, [&](u32 index) { ForEachEnabledBit(channel_state->enabled_compute_texture_buffers, [&](u32 index) {
Binding& binding = channel_state->compute_texture_buffers[index]; Binding& binding = channel_state->compute_texture_buffers[index];
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false); binding.buffer_id = FindBuffer(binding.device_addr, binding.size);
}); });
} }
@@ -1530,7 +1440,7 @@ void BufferCache<P>::MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u3
} }
template <class P> template <class P>
BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible) { BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size) {
if (device_addr == 0) { if (device_addr == 0) {
return NULL_BUFFER_ID; return NULL_BUFFER_ID;
} }
@@ -1540,18 +1450,10 @@ BufferId BufferCache<P>::FindBuffer(DAddr device_addr, u32 size, bool sparse_com
Buffer& buffer = slot_buffers[buffer_id]; Buffer& buffer = slot_buffers[buffer_id];
WaitForGpuFenceIfNeeded(buffer); WaitForGpuFenceIfNeeded(buffer);
if (buffer.IsInBounds(device_addr, size)) { if (buffer.IsInBounds(device_addr, size)) {
bool usable = true; return buffer_id;
if constexpr (requires { buffer.IsSparseCompatible(); }) {
if (sparse_compatible && !buffer.IsSparseCompatible()) {
usable = false;
}
}
if (usable) {
return buffer_id;
}
} }
} }
return CreateBuffer(device_addr, size, sparse_compatible); return CreateBuffer(device_addr, size);
} }
template <class P> template <class P>
@@ -1673,15 +1575,13 @@ void BufferCache<P>::JoinOverlap(BufferId new_buffer_id, BufferId overlap_id,
} }
template <class P> template <class P>
BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size, BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size) {
bool sparse_compatible) {
DAddr device_addr_end = Common::AlignUp(device_addr + wanted_size, CACHING_PAGESIZE); DAddr device_addr_end = Common::AlignUp(device_addr + wanted_size, CACHING_PAGESIZE);
device_addr = Common::AlignDown(device_addr, CACHING_PAGESIZE); device_addr = Common::AlignDown(device_addr, CACHING_PAGESIZE);
wanted_size = static_cast<u32>(device_addr_end - device_addr); wanted_size = static_cast<u32>(device_addr_end - device_addr);
const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size); const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size);
const u32 size = static_cast<u32>(overlap.end - overlap.begin); const u32 size = static_cast<u32>(overlap.end - overlap.begin);
const BufferId new_buffer_id = const BufferId new_buffer_id = slot_buffers.insert(runtime, overlap.begin, size);
slot_buffers.insert(runtime, overlap.begin, size, sparse_compatible);
auto& new_buffer = slot_buffers[new_buffer_id]; auto& new_buffer = slot_buffers[new_buffer_id];
const size_t size_bytes = new_buffer.SizeBytes(); const size_t size_bytes = new_buffer.SizeBytes();
runtime.ClearBuffer(new_buffer, 0, size_bytes, 0); runtime.ClearBuffer(new_buffer, 0, size_bytes, 0);
@@ -1845,7 +1745,7 @@ void BufferCache<P>::InlineMemoryImplementation(DAddr dest_address, size_t copy_
ClearDownload(dest_address, copy_size); ClearDownload(dest_address, copy_size);
gpu_modified_ranges.Subtract(dest_address, copy_size); gpu_modified_ranges.Subtract(dest_address, copy_size);
BufferId buffer_id = FindBuffer(dest_address, static_cast<u32>(copy_size), false); BufferId buffer_id = FindBuffer(dest_address, static_cast<u32>(copy_size));
auto& buffer = slot_buffers[buffer_id]; auto& buffer = slot_buffers[buffer_id];
SynchronizeBuffer(buffer, dest_address, static_cast<u32>(copy_size)); SynchronizeBuffer(buffer, dest_address, static_cast<u32>(copy_size));
@@ -1931,9 +1831,6 @@ void BufferCache<P>::DownloadBufferMemory(Buffer& buffer, DAddr device_addr, u64
template <class P> template <class P>
void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) { void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
if constexpr (requires { runtime.OnBufferDeleted(slot_buffers[buffer_id]); }) {
runtime.OnBufferDeleted(slot_buffers[buffer_id]);
}
bool dirty_index{false}; bool dirty_index{false};
boost::container::small_vector<u64, NUM_VERTEX_BUFFERS> dirty_vertex_buffers; boost::container::small_vector<u64, NUM_VERTEX_BUFFERS> dirty_vertex_buffers;
const auto scalar_replace = [buffer_id](Binding& binding) { const auto scalar_replace = [buffer_id](Binding& binding) {
@@ -2037,14 +1934,9 @@ Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
// The end address used for size calculation does not need to be aligned // The end address used for size calculation does not need to be aligned
const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE); const DAddr cpu_end = Common::AlignUp(*device_addr + size, Core::DEVICE_PAGESIZE);
u32 binding_size = static_cast<u32>(cpu_end - *aligned_device_addr);
if (is_written) {
binding_size = aligned_size;
}
const Binding binding{ const Binding binding{
.device_addr = *aligned_device_addr, .device_addr = *aligned_device_addr,
.gpu_addr = aligned_gpu_addr, .size = is_written ? aligned_size : static_cast<u32>(cpu_end - *aligned_device_addr),
.size = binding_size,
.buffer_id = BufferId{}, .buffer_id = BufferId{},
}; };
return binding; return binding;
@@ -29,7 +29,6 @@
#include "common/settings.h" #include "common/settings.h"
#include "common/slot_vector.h" #include "common/slot_vector.h"
#include "video_core/buffer_cache/buffer_base.h" #include "video_core/buffer_cache/buffer_base.h"
#include "video_core/buffer_cache/virtual_range_cache.h"
#include "video_core/control/channel_state_cache.h" #include "video_core/control/channel_state_cache.h"
#include "video_core/delayed_destruction_ring.h" #include "video_core/delayed_destruction_ring.h"
#include "video_core/dirty_flags.h" #include "video_core/dirty_flags.h"
@@ -82,17 +81,8 @@ static constexpr u32 DEFAULT_SKIP_CACHE_SIZE = static_cast<u32>(4_KiB);
struct Binding { struct Binding {
DAddr device_addr{}; DAddr device_addr{};
GPUVAddr gpu_addr{};
u32 size{}; u32 size{};
BufferId buffer_id; BufferId buffer_id;
u32 segment_first{};
u32 segment_count{};
};
struct MultiRangeSegment {
BufferId buffer_id;
DAddr device_addr{};
u32 size{};
}; };
struct TextureBufferBinding : Binding { struct TextureBufferBinding : Binding {
@@ -225,14 +215,6 @@ public:
void TickFrame(); void TickFrame();
bool BindMultiRangeStorage(const Binding& binding, bool is_written,
std::span<const MultiRangeSegment> pool);
void ResolveMultiRangeStorage(Binding& binding, bool is_written,
std::vector<MultiRangeSegment>& pool);
void UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size);
void WriteMemory(DAddr device_addr, u64 size); void WriteMemory(DAddr device_addr, u64 size);
void CachedWriteMemory(DAddr device_addr, u64 size); void CachedWriteMemory(DAddr device_addr, u64 size);
@@ -432,7 +414,7 @@ private:
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size); void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size);
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible); [[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size);
void WaitForGpuFenceIfNeeded(Buffer& buffer); void WaitForGpuFenceIfNeeded(Buffer& buffer);
@@ -440,8 +422,7 @@ private:
void JoinOverlap(BufferId new_buffer_id, BufferId overlap_id, bool accumulate_stream_score); void JoinOverlap(BufferId new_buffer_id, BufferId overlap_id, bool accumulate_stream_score);
[[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size, [[nodiscard]] BufferId CreateBuffer(DAddr device_addr, u32 wanted_size);
bool sparse_compatible);
void Register(BufferId buffer_id); void Register(BufferId buffer_id);
@@ -532,9 +513,6 @@ private:
using TickType = u64; using TickType = u64;
}; };
Common::LeastRecentlyUsedCache<LRUItemParams> lru_cache; Common::LeastRecentlyUsedCache<LRUItemParams> lru_cache;
VirtualRangeCache virtual_ranges;
std::vector<MultiRangeSegment> graphics_segments;
std::vector<MultiRangeSegment> compute_segments;
u64 frame_tick = 0; u64 frame_tick = 0;
u64 total_used_memory = 0; u64 total_used_memory = 0;
u64 minimum_memory = 0; u64 minimum_memory = 0;
@@ -1,173 +0,0 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#pragma once
#include <atomic>
#include <limits>
#include <mutex>
#include <optional>
#include <vector>
#include <boost/container/small_vector.hpp>
#include "common/common_types.h"
#include "common/container/unordered_map.h"
#include "video_core/memory_manager.h"
namespace VideoCommon {
struct VirtualSegment {
GPUVAddr gpu_addr;
DAddr device_addr;
u32 size;
};
using VirtualSegments = boost::container::small_vector<VirtualSegment, 8>;
class VirtualRangeCache {
public:
static constexpr size_t MAX_ENTRIES = 8192;
static constexpr size_t MAX_DEFERRED = 4096;
const VirtualSegments* Query(Tegra::MemoryManager& memory, GPUVAddr gpu_addr, u32 size) {
if (has_deferred.load(std::memory_order_acquire)) {
ApplyDeferred();
}
if (entries.size() > MAX_ENTRIES) {
entries.clear();
}
const size_t as_id = memory.GetID();
const u64 key = MakeKey(as_id, gpu_addr);
const auto it = entries.find(key);
if (it != entries.end() && it->second.as_id == as_id &&
it->second.gpu_addr == gpu_addr && it->second.size == size) {
return &it->second.segments;
}
Entry entry{};
entry.as_id = as_id;
entry.gpu_addr = gpu_addr;
entry.size = size;
const auto ranges = memory.GetSubmappedRange(gpu_addr, size);
GPUVAddr expected = gpu_addr;
bool contiguous = true;
for (const auto& [range_addr, range_size] : ranges) {
if (range_addr != expected || range_size == 0) {
contiguous = false;
break;
}
const std::optional<DAddr> device_addr = memory.GpuToCpuAddress(range_addr);
if (!device_addr || *device_addr == 0) {
contiguous = false;
break;
}
if (range_size > static_cast<size_t>((std::numeric_limits<u32>::max)())) {
contiguous = false;
break;
}
entry.segments.push_back(VirtualSegment{
.gpu_addr = range_addr,
.device_addr = *device_addr,
.size = static_cast<u32>(range_size),
});
expected += range_size;
}
if (!contiguous || expected != gpu_addr + size) {
entry.segments.clear();
}
const auto result = entries.insert_or_assign(key, std::move(entry));
return &result.first->second.segments;
}
void Unmap(size_t as_id, GPUVAddr gpu_addr, u64 size) {
if (size == 0) {
return;
}
{
std::scoped_lock lock{deferred_mutex};
if (!deferred.empty()) {
DeferredUnmap& last = deferred.back();
if (last.as_id == as_id && last.gpu_addr + last.size == gpu_addr) {
last.size += size;
has_deferred.store(true, std::memory_order_release);
return;
}
}
if (deferred.size() >= MAX_DEFERRED) {
deferred.clear();
deferred_overflow = true;
} else {
deferred.push_back(DeferredUnmap{
.as_id = as_id,
.gpu_addr = gpu_addr,
.size = size,
});
}
}
has_deferred.store(true, std::memory_order_release);
}
private:
struct Entry {
VirtualSegments segments;
size_t as_id{};
GPUVAddr gpu_addr{};
u32 size{};
};
struct DeferredUnmap {
size_t as_id;
GPUVAddr gpu_addr;
u64 size;
};
static u64 MakeKey(size_t as_id, GPUVAddr gpu_addr) {
return (static_cast<u64>(as_id) << 48) ^ gpu_addr;
}
void ApplyDeferred() {
std::vector<DeferredUnmap> pending;
bool overflow = false;
{
std::scoped_lock lock{deferred_mutex};
has_deferred.store(false, std::memory_order_release);
pending.swap(deferred);
overflow = deferred_overflow;
deferred_overflow = false;
}
if (overflow) {
entries.clear();
return;
}
if (pending.empty() || entries.empty()) {
return;
}
for (auto it = entries.begin(); it != entries.end();) {
const Entry& entry = it->second;
const GPUVAddr entry_end = entry.gpu_addr + entry.size;
bool overlaps = false;
for (const DeferredUnmap& unmap : pending) {
if (unmap.as_id != entry.as_id) {
continue;
}
if (entry.gpu_addr < unmap.gpu_addr + unmap.size && unmap.gpu_addr < entry_end) {
overlaps = true;
break;
}
}
if (overlaps) {
it = entries.erase(it);
} else {
++it;
}
}
}
::Common::unordered_map<u64, Entry> entries;
std::vector<DeferredUnmap> deferred;
std::mutex deferred_mutex;
std::atomic<bool> has_deferred{false};
bool deferred_overflow{};
};
} // namespace VideoCommon
@@ -52,7 +52,7 @@ constexpr std::array PROGRAM_LUT{
Buffer::Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params) Buffer::Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params)
: VideoCommon::BufferBase(null_params) {} : VideoCommon::BufferBase(null_params) {}
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_, bool) Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
: VideoCommon::BufferBase(cpu_addr_, size_bytes_) { : VideoCommon::BufferBase(cpu_addr_, size_bytes_) {
buffer.Create(); buffer.Create();
if (runtime.device.HasDebuggingToolAttached()) { if (runtime.device.HasDebuggingToolAttached()) {
@@ -23,8 +23,7 @@ class BufferCacheRuntime;
class Buffer : public VideoCommon::BufferBase { class Buffer : public VideoCommon::BufferBase {
public: public:
explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes, explicit Buffer(BufferCacheRuntime&, DAddr cpu_addr, u64 size_bytes);
bool sparse_compatible);
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams); explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams);
void ImmediateUpload(size_t offset, std::span<const u8> data) noexcept; void ImmediateUpload(size_t offset, std::span<const u8> data) noexcept;
@@ -56,8 +56,7 @@ size_t BytesPerIndex(VkIndexType index_type) {
} }
} }
vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size, vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allocator, u64 size) {
VkDeviceSize sparse_alignment) {
VkBufferUsageFlags flags = VkBufferUsageFlags flags =
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_UNIFORM_TEXEL_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_TEXEL_BUFFER_BIT |
@@ -83,9 +82,6 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
.queueFamilyIndexCount = 0, .queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr, .pQueueFamilyIndices = nullptr,
}; };
if (sparse_alignment > 1) {
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal, sparse_alignment);
}
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal); return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal);
} }
} // Anonymous namespace } // Anonymous namespace
@@ -103,14 +99,10 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
} }
} }
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_, Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
bool sparse_compatible_)
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device}, : VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
scheduler{&runtime.scheduler}, scheduler{&runtime.scheduler},
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes(), buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
runtime.SparseAlignmentFor(sparse_compatible_))},
tracker{SizeBytes()} {
sparse_compatible = sparse_compatible_;
if (runtime.device.HasDebuggingToolAttached()) { if (runtime.device.HasDebuggingToolAttached()) {
buffer.SetObjectNameEXT(fmt::format("Buffer {:#x}", CpuAddr()).c_str()); buffer.SetObjectNameEXT(fmt::format("Buffer {:#x}", CpuAddr()).c_str());
} }
@@ -356,8 +348,7 @@ BufferCacheRuntime::BufferCacheRuntime(const Device& device_, MemoryAllocator& m
: device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_}, : device{device_}, memory_allocator{memory_allocator_}, scheduler{scheduler_},
staging_pool{staging_pool_}, guest_descriptor_queue{guest_descriptor_queue_}, staging_pool{staging_pool_}, guest_descriptor_queue{guest_descriptor_queue_},
quad_index_pass(device, scheduler, descriptor_pool, staging_pool, quad_index_pass(device, scheduler, descriptor_pool, staging_pool,
compute_pass_descriptor_queue), compute_pass_descriptor_queue) {
multi_range_buffers(device_) {
const VkDriverIdKHR driver_id = device.GetDriverID(); const VkDriverIdKHR driver_id = device.GetDriverID();
limit_dynamic_storage_buffers = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY || limit_dynamic_storage_buffers = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY ||
driver_id == VK_DRIVER_ID_ARM_PROPRIETARY; driver_id == VK_DRIVER_ID_ARM_PROPRIETARY;
@@ -545,37 +536,6 @@ void BufferCacheRuntime::ClearBuffer(VkBuffer dest_buffer, u32 offset, size_t si
}); });
} }
bool BufferCacheRuntime::BindMultiRangeStorageBuffer(u64 key, bool is_written) {
if (multi_range_sources.empty() || multi_range_total == 0) {
return false;
}
const MultiRangeRef ref = multi_range_buffers.Get(device, scheduler, memory_allocator, key,
multi_range_sources, multi_range_total);
if (ref.handle == VK_NULL_HANDLE) {
return false;
}
if (is_written && !ref.sparse) {
return false;
}
if (ref.needs_gather) {
PreCopyBarrier();
VkDeviceSize dst_offset = 0;
for (const MultiRangeSource& source : multi_range_sources) {
const std::array<VideoCommon::BufferCopy, 1> copy{VideoCommon::BufferCopy{
.src_offset = u64(source.offset),
.dst_offset = u64(dst_offset),
.size = size_t(source.size),
}};
CopyBuffer(ref.handle, source.handle, copy, false);
dst_offset += source.size;
}
PostCopyBarrier();
multi_range_buffers.MarkGathered(key);
}
guest_descriptor_queue.AddBuffer(ref.handle, ref.address, 0, ref.size);
return true;
}
void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format, void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format,
u32 base_vertex, u32 num_indices, VkBuffer buffer, u32 base_vertex, u32 num_indices, VkBuffer buffer,
u32 offset, [[maybe_unused]] u32 size) { u32 offset, [[maybe_unused]] u32 size) {
@@ -8,14 +8,11 @@
#include <limits> #include <limits>
#include <boost/container/small_vector.hpp>
#include "video_core/buffer_cache/buffer_cache_base.h" #include "video_core/buffer_cache/buffer_cache_base.h"
#include "video_core/buffer_cache/memory_tracker_base.h" #include "video_core/buffer_cache/memory_tracker_base.h"
#include "video_core/buffer_cache/usage_tracker.h" #include "video_core/buffer_cache/usage_tracker.h"
#include "video_core/engines/maxwell_3d.h" #include "video_core/engines/maxwell_3d.h"
#include "video_core/renderer_vulkan/vk_compute_pass.h" #include "video_core/renderer_vulkan/vk_compute_pass.h"
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h" #include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
#include "video_core/renderer_vulkan/vk_update_descriptor.h" #include "video_core/renderer_vulkan/vk_update_descriptor.h"
#include "video_core/surface.h" #include "video_core/surface.h"
@@ -34,8 +31,7 @@ class BufferCacheRuntime;
class Buffer : public VideoCommon::BufferBase { class Buffer : public VideoCommon::BufferBase {
public: public:
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params); explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_, explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_);
bool sparse_compatible_);
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format); [[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
@@ -47,14 +43,6 @@ public:
return device_address; return device_address;
} }
[[nodiscard]] bool IsSparseCompatible() const noexcept {
return sparse_compatible;
}
[[nodiscard]] vk::MemoryLocation Location() const noexcept {
return buffer.Location();
}
[[nodiscard]] bool IsRegionUsed(u64 offset, u64 size) const noexcept { [[nodiscard]] bool IsRegionUsed(u64 offset, u64 size) const noexcept {
return tracker.IsUsed(offset, size); return tracker.IsUsed(offset, size);
} }
@@ -89,7 +77,6 @@ private:
VkDeviceAddress device_address{}; VkDeviceAddress device_address{};
u64 last_usage_tick{}; u64 last_usage_tick{};
bool is_null{}; bool is_null{};
bool sparse_compatible{};
}; };
class QuadArrayIndexBuffer; class QuadArrayIndexBuffer;
@@ -138,7 +125,7 @@ public:
void PreCopyBarrier(); void PreCopyBarrier();
void CopyBuffer(VkBuffer dst_buffer, VkBuffer src_buffer, void CopyBuffer(VkBuffer src_buffer, VkBuffer dst_buffer,
std::span<const VideoCommon::BufferCopy> copies, bool barrier, std::span<const VideoCommon::BufferCopy> copies, bool barrier,
bool can_reorder_upload = false); bool can_reorder_upload = false);
@@ -168,46 +155,6 @@ public:
return ref.mapped_span; return ref.mapped_span;
} }
[[nodiscard]] VkDeviceSize SparseAlignmentFor(bool sparse_compatible) const noexcept {
if (!sparse_compatible || !multi_range_buffers.use_sparse) {
return 0;
}
return multi_range_buffers.block_size;
}
[[nodiscard]] bool PrefersSparseSources() const noexcept {
return multi_range_buffers.use_sparse;
}
void ResetMultiRange() noexcept {
multi_range_sources.clear();
multi_range_total = 0;
}
void PushMultiRangeSource(const Buffer& buffer, u32 offset, u32 size) {
const vk::MemoryLocation location = buffer.Location();
multi_range_sources.push_back(MultiRangeSource{
.handle = buffer.Handle(),
.memory = location.memory,
.memory_offset = location.offset,
.offset = offset,
.size = size,
.write_tick = buffer.getWriteTick(),
.memory_type = location.memory_type,
});
multi_range_total += size;
}
bool BindMultiRangeStorageBuffer(u64 key, bool is_written);
void InvalidateMultiRange(u64 key) {
multi_range_buffers.Invalidate(key);
}
void OnBufferDeleted(const Buffer& buffer) {
multi_range_buffers.DropOwner(scheduler, buffer.Handle());
}
void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) { void BindUniformBuffer(const Buffer& buffer, u32 offset, u32 size) {
BindBuffer(buffer, offset, size); BindBuffer(buffer, offset, size);
} }
@@ -261,10 +208,6 @@ private:
std::unique_ptr<Uint8Pass> uint8_pass; std::unique_ptr<Uint8Pass> uint8_pass;
QuadIndexedPass quad_index_pass; QuadIndexedPass quad_index_pass;
MultiRangeBufferCache multi_range_buffers;
boost::container::small_vector<MultiRangeSource, 16> multi_range_sources;
VkDeviceSize multi_range_total{};
bool limit_dynamic_storage_buffers = false; bool limit_dynamic_storage_buffers = false;
u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)(); u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)();
}; };
@@ -1,338 +0,0 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#include <algorithm>
#include <mutex>
#include <utility>
#include "video_core/renderer_vulkan/vk_multi_range_buffer.h"
#include "video_core/renderer_vulkan/vk_scheduler.h"
#include "video_core/vulkan_common/vulkan_device.h"
namespace Vulkan {
MultiRangeBufferCache::MultiRangeBufferCache(const Device& device) {
sparse_usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
if (device.IsBufferDeviceAddressSupported()) {
sparse_usage |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
}
if (!device.IsSparseBindingSupported()) {
return;
}
u32 memory_type_bits = 0;
const VkDeviceSize queried = QueryBlockSize(device, memory_type_bits);
if (queried == 0 || memory_type_bits == 0) {
return;
}
block_size = queried;
sparse_memory_type_bits = memory_type_bits;
use_sparse = true;
}
VkDeviceSize MultiRangeBufferCache::QueryBlockSize(const Device& device,
u32& memory_type_bits) const {
const VkDevice logical = *device.GetLogical();
const auto& dld = device.GetDispatchLoader();
const VkBufferCreateInfo probe_ci{
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
.pNext = nullptr,
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
.size = DEFAULT_BLOCK_SIZE,
.usage = sparse_usage,
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
.queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr,
};
VkBuffer probe{};
if (dld.vkCreateBuffer(logical, &probe_ci, nullptr, &probe) != VK_SUCCESS) {
return 0;
}
const SparseBuffer owned{probe, logical, dld};
const VkBufferMemoryRequirementsInfo2 reqs_info{
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_REQUIREMENTS_INFO_2,
.pNext = nullptr,
.buffer = probe,
};
VkMemoryRequirements2 reqs2{
.sType = VK_STRUCTURE_TYPE_MEMORY_REQUIREMENTS_2,
.pNext = nullptr,
.memoryRequirements = {},
};
dld.vkGetBufferMemoryRequirements2(logical, &reqs_info, &reqs2);
memory_type_bits = reqs2.memoryRequirements.memoryTypeBits;
return reqs2.memoryRequirements.alignment;
}
u64 MultiRangeBufferCache::HashSources(std::span<const MultiRangeSource> sources) const {
u64 hash = 0xcbf29ce484222325ULL;
const auto mix = [&hash](u64 value) {
hash ^= value;
hash *= 0x100000001b3ULL;
};
for (const MultiRangeSource& source : sources) {
mix(u64(source.handle));
mix(u64(source.offset));
mix(u64(source.size));
}
return hash;
}
u64 MultiRangeBufferCache::HashContent(std::span<const MultiRangeSource> sources) const {
u64 hash = 0xcbf29ce484222325ULL;
for (const MultiRangeSource& source : sources) {
hash ^= source.write_tick;
hash *= 0x100000001b3ULL;
}
return hash;
}
bool MultiRangeBufferCache::CanBindSparse(std::span<const MultiRangeSource> sources) const {
return use_sparse &&
std::none_of(sources.begin(), sources.end(),
[block = block_size, bits = sparse_memory_type_bits](auto const& e) {
const VkDeviceSize memory_offset = e.memory_offset + e.offset;
return e.memory == VK_NULL_HANDLE || e.memory_type >= 32 ||
((bits >> e.memory_type) & 1) == 0 ||
(memory_offset % block) != 0 || (e.size % block) != 0;
});
}
SparseBuffer MultiRangeBufferCache::CreateSparse(const Device& device, Scheduler& scheduler,
std::span<const MultiRangeSource> sources,
VkDeviceSize total) {
const VkDevice logical = *device.GetLogical();
const auto& dld = device.GetDispatchLoader();
const VkBufferCreateInfo buffer_ci{
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
.pNext = nullptr,
.flags = VK_BUFFER_CREATE_SPARSE_BINDING_BIT | VK_BUFFER_CREATE_SPARSE_ALIASED_BIT,
.size = total,
.usage = sparse_usage,
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
.queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr,
};
VkBuffer raw{};
if (dld.vkCreateBuffer(logical, &buffer_ci, nullptr, &raw) != VK_SUCCESS) {
return SparseBuffer{};
}
SparseBuffer handle{raw, logical, dld};
std::vector<VkSparseMemoryBind> binds;
binds.reserve(sources.size());
VkDeviceSize resource_offset = 0;
for (const MultiRangeSource& source : sources) {
binds.push_back(VkSparseMemoryBind{
.resourceOffset = resource_offset,
.size = source.size,
.memory = source.memory,
.memoryOffset = source.memory_offset + source.offset,
.flags = 0,
});
resource_offset += source.size;
}
const VkSparseBufferMemoryBindInfo buffer_bind{
.buffer = raw,
.bindCount = static_cast<u32>(binds.size()),
.pBinds = binds.data(),
};
const VkBindSparseInfo bind_info{
.sType = VK_STRUCTURE_TYPE_BIND_SPARSE_INFO,
.pNext = nullptr,
.waitSemaphoreCount = 0,
.pWaitSemaphores = nullptr,
.bufferBindCount = 1,
.pBufferBinds = &buffer_bind,
.imageOpaqueBindCount = 0,
.pImageOpaqueBinds = nullptr,
.imageBindCount = 0,
.pImageBinds = nullptr,
.signalSemaphoreCount = 0,
.pSignalSemaphores = nullptr,
};
const VkFenceCreateInfo fence_ci{
.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
};
vk::Fence fence = device.GetLogical().CreateFence(fence_ci);
VkResult bind_result = VK_ERROR_UNKNOWN;
{
std::scoped_lock lock{scheduler.submit_mutex};
bind_result = device.GetGraphicsQueue().BindSparse(bind_info, *fence);
}
if (bind_result != VK_SUCCESS) {
return SparseBuffer{};
}
fence.Wait();
return handle;
}
void MultiRangeBufferCache::RetireEntry(Scheduler& scheduler, Entry& entry) {
if (!entry.sparse_handle && !entry.gathered) {
return;
}
if (retired.size() == retired.capacity()) {
DrainRetired(scheduler);
}
if (retired.size() == retired.capacity()) {
u64 oldest = retired.front().tick;
for (const Retired& item : retired) {
if (item.tick < oldest) {
oldest = item.tick;
}
}
scheduler.Wait(oldest);
DrainRetired(scheduler);
}
retired.push_back(Retired{
.handle = std::move(entry.sparse_handle),
.gathered = std::move(entry.gathered),
.tick = scheduler.CurrentTick(),
});
}
void MultiRangeBufferCache::DrainRetired(Scheduler& scheduler) {
size_t index = 0;
while (index < retired.size()) {
if (scheduler.IsFree(retired[index].tick)) {
if (index + 1 != retired.size()) {
retired[index] = std::move(retired.back());
}
retired.pop_back();
} else {
++index;
}
}
}
MultiRangeRef MultiRangeBufferCache::Get(const Device& device, Scheduler& scheduler,
MemoryAllocator& memory_allocator, u64 key,
std::span<const MultiRangeSource> sources,
VkDeviceSize total) {
if (sources.empty() || total == 0) {
return MultiRangeRef{};
}
if (!retired.empty()) {
DrainRetired(scheduler);
}
const u64 geometry = HashSources(sources);
const u64 content = HashContent(sources);
const auto it = entries.find(key);
if (it != entries.end() && it->second.geometry == geometry && it->second.size == total) {
Entry& entry = it->second;
if (entry.content != content) {
entry.content = content;
entry.dirty = true;
}
MultiRangeRef ref{
.handle = *entry.sparse_handle,
.address = entry.address,
.size = entry.size,
.sparse = true,
.needs_gather = false,
};
if (!entry.sparse_handle) {
ref.handle = *entry.gathered;
ref.sparse = false;
ref.needs_gather = entry.dirty;
}
return ref;
}
if (it != entries.end()) {
RetireEntry(scheduler, it->second);
entries.erase(it);
}
Entry entry{};
entry.geometry = geometry;
entry.content = content;
entry.size = total;
if (CanBindSparse(sources)) {
entry.sparse_handle = CreateSparse(device, scheduler, sources, total);
if (entry.sparse_handle) {
entry.owners.reserve(sources.size());
for (const MultiRangeSource& source : sources) {
entry.owners.push_back(source.handle);
}
}
}
if (!entry.sparse_handle) {
VkBufferUsageFlags flags = VK_BUFFER_USAGE_TRANSFER_SRC_BIT |
VK_BUFFER_USAGE_TRANSFER_DST_BIT |
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT;
if (device.IsBufferDeviceAddressSupported()) {
flags |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT;
}
const VkBufferCreateInfo gather_ci{
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.size = total,
.usage = flags,
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
.queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr,
};
entry.gathered = memory_allocator.CreateBuffer(gather_ci, MemoryUsage::DeviceLocal);
entry.dirty = true;
}
if (device.IsBufferDeviceAddressSupported()) {
VkBuffer address_handle = *entry.sparse_handle;
if (!entry.sparse_handle) {
address_handle = *entry.gathered;
}
entry.address = device.GetLogical().GetBufferDeviceAddress(address_handle);
}
MultiRangeRef ref{
.handle = *entry.sparse_handle,
.address = entry.address,
.size = entry.size,
.sparse = true,
.needs_gather = false,
};
if (!entry.sparse_handle) {
ref.handle = *entry.gathered;
ref.sparse = false;
ref.needs_gather = true;
}
entries.emplace(key, std::move(entry));
return ref;
}
void MultiRangeBufferCache::MarkGathered(u64 key) {
if (auto const it = entries.find(key); it != entries.end()) {
it->second.dirty = false;
}
}
void MultiRangeBufferCache::DropOwner(Scheduler& scheduler, VkBuffer owner) {
if (owner == VK_NULL_HANDLE) {
return;
}
for (auto it = entries.begin(); it != entries.end();) {
Entry& entry = it->second;
bool owned = false;
for (const VkBuffer handle : entry.owners) {
if (handle == owner) {
owned = true;
break;
}
}
if (!owned) {
++it;
continue;
}
RetireEntry(scheduler, entry);
it = entries.erase(it);
}
}
void MultiRangeBufferCache::Invalidate(u64 key) {
if (auto const it = entries.find(key); it != entries.end()) {
it->second.dirty = true;
}
}
} // namespace Vulkan
@@ -1,105 +0,0 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#pragma once
#include <span>
#include <vector>
#include <boost/container/static_vector.hpp>
#include "common/common_funcs.h"
#include "common/common_types.h"
#include "common/container/unordered_map.h"
#include "video_core/vulkan_common/vulkan_memory_allocator.h"
#include "video_core/vulkan_common/vulkan_wrapper.h"
namespace Vulkan {
using SparseBuffer = vk::Handle<VkBuffer, VkDevice, vk::DeviceDispatch>;
class Device;
class Scheduler;
struct MultiRangeSource {
VkBuffer handle{};
VkDeviceMemory memory{};
VkDeviceSize memory_offset{};
VkDeviceSize offset{};
VkDeviceSize size{};
u64 write_tick{};
u32 memory_type{};
};
struct MultiRangeRef {
VkBuffer handle{};
VkDeviceAddress address{};
VkDeviceSize size{};
bool sparse{};
bool needs_gather{};
};
class MultiRangeBufferCache final {
public:
static constexpr VkDeviceSize DEFAULT_BLOCK_SIZE = 64 * 1024;
static constexpr size_t MAX_RETIRED = 256;
explicit MultiRangeBufferCache(const Device& device);
YUZU_NON_COPYABLE(MultiRangeBufferCache);
[[nodiscard]] MultiRangeRef Get(const Device& device, Scheduler& scheduler,
MemoryAllocator& memory_allocator, u64 key,
std::span<const MultiRangeSource> sources,
VkDeviceSize total);
void MarkGathered(u64 key);
void Invalidate(u64 key);
void DropOwner(Scheduler& scheduler, VkBuffer owner);
VkDeviceSize block_size{DEFAULT_BLOCK_SIZE};
bool use_sparse{};
private:
struct Retired {
SparseBuffer handle;
vk::Buffer gathered;
u64 tick{};
};
struct Entry {
vk::Buffer gathered;
SparseBuffer sparse_handle;
std::vector<VkBuffer> owners;
VkDeviceAddress address{};
VkDeviceSize size{};
u64 geometry{};
u64 content{};
bool dirty{true};
};
[[nodiscard]] u64 HashSources(std::span<const MultiRangeSource> sources) const;
[[nodiscard]] u64 HashContent(std::span<const MultiRangeSource> sources) const;
[[nodiscard]] bool CanBindSparse(std::span<const MultiRangeSource> sources) const;
[[nodiscard]] SparseBuffer CreateSparse(const Device& device, Scheduler& scheduler,
std::span<const MultiRangeSource> sources,
VkDeviceSize total);
[[nodiscard]] VkDeviceSize QueryBlockSize(const Device& device, u32& memory_type_bits) const;
void RetireEntry(Scheduler& scheduler, Entry& entry);
void DrainRetired(Scheduler& scheduler);
::Common::unordered_map<u64, Entry> entries;
boost::container::static_vector<Retired, MAX_RETIRED> retired;
u32 sparse_memory_type_bits{};
VkBufferUsageFlags sparse_usage{};
};
} // namespace Vulkan
@@ -819,7 +819,6 @@ void RasterizerVulkan::ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) {
std::scoped_lock lock{texture_cache.mutex}; std::scoped_lock lock{texture_cache.mutex};
texture_cache.UnmapGPUMemory(as_id, addr, size); texture_cache.UnmapGPUMemory(as_id, addr, size);
} }
buffer_cache.UnmapGPUMemory(as_id, addr, size);
} }
void RasterizerVulkan::SignalFence(std::function<void()>&& func) { void RasterizerVulkan::SignalFence(std::function<void()>&& func) {
@@ -1570,8 +1570,6 @@ void Device::SetupFamilies(VkSurfaceKHR surface) {
} }
if (graphics) { if (graphics) {
graphics_family = *graphics; graphics_family = *graphics;
graphics_family_sparse_binding =
(queue_family_properties[*graphics].queueFlags & VK_QUEUE_SPARSE_BINDING_BIT) != 0;
} }
if (present) { if (present) {
present_family = *present; present_family = *present;
@@ -317,10 +317,6 @@ public:
return properties.driver.driverID; return properties.driver.driverID;
} }
bool IsSparseBindingSupported() const {
return features.features.sparseBinding && graphics_family_sparse_binding;
}
/// Returns true for tile-based deferred renderers. /// Returns true for tile-based deferred renderers.
bool IsTiler() const { bool IsTiler() const {
switch (GetDriverID()) { switch (GetDriverID()) {
@@ -1151,7 +1147,6 @@ private:
u32 instance_version{}; ///< Vulkan instance version. u32 instance_version{}; ///< Vulkan instance version.
u32 graphics_family{}; ///< Main graphics queue family index. u32 graphics_family{}; ///< Main graphics queue family index.
u32 present_family{}; ///< Main present queue family index. u32 present_family{}; ///< Main present queue family index.
bool graphics_family_sparse_binding{};
struct Extensions { struct Extensions {
#define EXTENSION(prefix, macro_name, var_name) bool var_name{}; #define EXTENSION(prefix, macro_name, var_name) bool var_name{};
@@ -275,63 +275,9 @@ vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsa
const std::span<u8> mapped_data = data ? std::span<u8>{data, ci.size} : std::span<u8>{}; const std::span<u8> mapped_data = data ? std::span<u8>{data, ci.size} : std::span<u8>{};
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0; const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
const vk::MemoryLocation location{ return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data,
.memory = alloc_info.deviceMemory, is_coherent,
.offset = alloc_info.offset, device.GetDispatchLoader());
.memory_type = alloc_info.memoryType,
};
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
location, device.GetDispatchLoader());
}
vk::Buffer MemoryAllocator::CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
VkDeviceSize min_alignment) const {
if (min_alignment <= 1) {
return CreateBuffer(ci, usage);
}
VkMemoryPropertyFlags anv_flags = 0;
if (usage == MemoryUsage::Stream &&
device.GetDriverID() == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA) {
anv_flags = VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
}
u32 memory_type_bits = valid_memory_types;
if (usage == MemoryUsage::Stream) {
memory_type_bits = 0u;
}
const VmaAllocationCreateInfo alloc_ci = {
.flags = VMA_ALLOCATION_CREATE_WITHIN_BUDGET_BIT | MemoryUsageVmaFlags(usage),
.usage = MemoryUsageVma(usage),
.requiredFlags = 0,
.preferredFlags = MemoryUsagePreferredVmaFlags(usage) | anv_flags,
.memoryTypeBits = memory_type_bits,
.pool = VK_NULL_HANDLE,
.pUserData = nullptr,
.priority = 0.f,
};
VkBuffer handle{};
VmaAllocationInfo alloc_info{};
VmaAllocation allocation{};
VkMemoryPropertyFlags property_flags{};
vk::Check(vmaCreateBufferWithAlignment(allocator, &ci, &alloc_ci, min_alignment, &handle,
&allocation, &alloc_info));
vmaGetAllocationMemoryProperties(allocator, allocation, &property_flags);
u8 *data = reinterpret_cast<u8 *>(alloc_info.pMappedData);
std::span<u8> mapped_data{};
if (data) {
mapped_data = std::span<u8>{data, ci.size};
}
const bool is_coherent = (property_flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
const vk::MemoryLocation location{
.memory = alloc_info.deviceMemory,
.offset = alloc_info.offset,
.memory_type = alloc_info.memoryType,
};
return vk::Buffer(handle, *device.GetLogical(), allocator, allocation, mapped_data, is_coherent,
location, device.GetDispatchLoader());
} }
MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage) MemoryCommit MemoryAllocator::Commit(const VkMemoryRequirements &reqs, MemoryUsage usage)
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
@@ -107,9 +107,6 @@ namespace Vulkan {
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const; vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage) const;
vk::Buffer CreateBuffer(const VkBufferCreateInfo &ci, MemoryUsage usage,
VkDeviceSize min_alignment) const;
/** /**
* Commits a memory with the specified requirements. * Commits a memory with the specified requirements.
* *
@@ -229,7 +229,6 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkGetPipelineExecutableStatisticsKHR); X(vkGetPipelineExecutableStatisticsKHR);
X(vkGetSemaphoreCounterValue); X(vkGetSemaphoreCounterValue);
X(vkMapMemory); X(vkMapMemory);
X(vkQueueBindSparse);
X(vkQueueSubmit); X(vkQueueSubmit);
X(vkQueueSubmit2); X(vkQueueSubmit2);
X(vkResetFences); X(vkResetFences);
+3 -22
View File
@@ -345,7 +345,6 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkGetQueryPoolResults vkGetQueryPoolResults{}; PFN_vkGetQueryPoolResults vkGetQueryPoolResults{};
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{}; PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
PFN_vkMapMemory vkMapMemory{}; PFN_vkMapMemory vkMapMemory{};
PFN_vkQueueBindSparse vkQueueBindSparse{};
PFN_vkQueueSubmit vkQueueSubmit{}; PFN_vkQueueSubmit vkQueueSubmit{};
PFN_vkQueueSubmit2 vkQueueSubmit2{}; PFN_vkQueueSubmit2 vkQueueSubmit2{};
PFN_vkResetFences vkResetFences{}; PFN_vkResetFences vkResetFences{};
@@ -741,20 +740,13 @@ private:
const DeviceDispatch* dld = nullptr; const DeviceDispatch* dld = nullptr;
}; };
struct MemoryLocation {
VkDeviceMemory memory{};
VkDeviceSize offset{};
u32 memory_type{};
};
class Buffer { class Buffer {
public: public:
explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_, explicit Buffer(VkBuffer handle_, VkDevice owner_, VmaAllocator allocator_,
VmaAllocation allocation_, std::span<u8> mapped_, bool is_coherent_, VmaAllocation allocation_, std::span<u8> mapped_, bool is_coherent_,
MemoryLocation location_, const DeviceDispatch& dld_) noexcept const DeviceDispatch& dld_) noexcept
: handle{handle_}, owner{owner_}, allocator{allocator_}, : handle{handle_}, owner{owner_}, allocator{allocator_},
allocation{allocation_}, mapped{mapped_}, location{location_}, allocation{allocation_}, mapped{mapped_}, is_coherent{is_coherent_}, dld{&dld_} {}
is_coherent{is_coherent_}, dld{&dld_} {}
Buffer() = default; Buffer() = default;
Buffer(const Buffer&) = delete; Buffer(const Buffer&) = delete;
@@ -762,7 +754,7 @@ public:
Buffer(Buffer&& rhs) noexcept Buffer(Buffer&& rhs) noexcept
: handle{std::exchange(rhs.handle, VkBuffer{})}, owner{rhs.owner}, allocator{rhs.allocator}, : handle{std::exchange(rhs.handle, VkBuffer{})}, owner{rhs.owner}, allocator{rhs.allocator},
allocation{rhs.allocation}, mapped{rhs.mapped}, location{rhs.location}, allocation{rhs.allocation}, mapped{rhs.mapped},
is_coherent{rhs.is_coherent}, dld{rhs.dld} {} is_coherent{rhs.is_coherent}, dld{rhs.dld} {}
Buffer& operator=(Buffer&& rhs) noexcept { Buffer& operator=(Buffer&& rhs) noexcept {
@@ -772,7 +764,6 @@ public:
allocator = rhs.allocator; allocator = rhs.allocator;
allocation = rhs.allocation; allocation = rhs.allocation;
mapped = rhs.mapped; mapped = rhs.mapped;
location = rhs.location;
is_coherent = rhs.is_coherent; is_coherent = rhs.is_coherent;
dld = rhs.dld; dld = rhs.dld;
return *this; return *this;
@@ -820,10 +811,6 @@ public:
void SetObjectNameEXT(const char* name) const; void SetObjectNameEXT(const char* name) const;
MemoryLocation Location() const noexcept {
return location;
}
private: private:
void Release() const noexcept; void Release() const noexcept;
@@ -832,7 +819,6 @@ private:
VmaAllocator allocator = nullptr; VmaAllocator allocator = nullptr;
VmaAllocation allocation = nullptr; VmaAllocation allocation = nullptr;
std::span<u8> mapped = {}; std::span<u8> mapped = {};
MemoryLocation location{};
bool is_coherent = false; bool is_coherent = false;
const DeviceDispatch* dld = nullptr; const DeviceDispatch* dld = nullptr;
}; };
@@ -857,11 +843,6 @@ public:
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence); return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
} }
VkResult BindSparse(Span<VkBindSparseInfo> bind_infos,
VkFence fence = VK_NULL_HANDLE) const noexcept {
return dld->vkQueueBindSparse(queue, bind_infos.size(), bind_infos.data(), fence);
}
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept { VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
return dld->vkQueuePresentKHR(queue, &present_info); return dld->vkQueuePresentKHR(queue, &present_info);
} }