From 108c991431a14b584b5e496be995938b934216da Mon Sep 17 00:00:00 2001 From: CamilleLaVey Date: Thu, 24 Sep 2026 17:57:47 -0400 Subject: [PATCH] Experiment on descriptors payloads --- src/shader_recompiler/ir_opt/texture_pass.cpp | 2 +- .../renderer_opengl/gl_texture_cache.h | 1 - .../renderer_vulkan/vk_descriptor_buffer.h | 6 +-- .../renderer_vulkan/vk_texture_cache.h | 1 - .../texture_cache/descriptor_table.h | 54 +++++++++++-------- src/video_core/texture_cache/texture_cache.h | 40 +++++--------- .../texture_cache/texture_cache_base.h | 5 -- 7 files changed, 47 insertions(+), 62 deletions(-) diff --git a/src/shader_recompiler/ir_opt/texture_pass.cpp b/src/shader_recompiler/ir_opt/texture_pass.cpp index 82e1c1c39e..0228ce117d 100644 --- a/src/shader_recompiler/ir_opt/texture_pass.cpp +++ b/src/shader_recompiler/ir_opt/texture_pass.cpp @@ -33,7 +33,7 @@ using TextureInstVector = boost::container::small_vector; constexpr u32 DESCRIPTOR_SIZE = 8; constexpr u32 DESCRIPTOR_SIZE_SHIFT = u32(std::countr_zero(DESCRIPTOR_SIZE)); -constexpr u32 DESCRIPTOR_MAX_COUNT = 512; +constexpr u32 DESCRIPTOR_MAX_COUNT = 128; constexpr u32 DESCRIPTOR_CBUF_BYTES = 16 * 1024; u32 DynamicDescriptorSizeShift(const IR::U32& dynamic_offset) { diff --git a/src/video_core/renderer_opengl/gl_texture_cache.h b/src/video_core/renderer_opengl/gl_texture_cache.h index 9f3b244a8d..7c7ee14580 100644 --- a/src/video_core/renderer_opengl/gl_texture_cache.h +++ b/src/video_core/renderer_opengl/gl_texture_cache.h @@ -364,7 +364,6 @@ private: }; struct TextureCacheParams { - static constexpr bool ENABLE_VALIDATION = true; static constexpr bool FRAMEBUFFER_BLITS = true; static constexpr bool HAS_EMULATED_COPIES = true; static constexpr bool HAS_DEVICE_MEMORY_INFO = true; diff --git a/src/video_core/renderer_vulkan/vk_descriptor_buffer.h b/src/video_core/renderer_vulkan/vk_descriptor_buffer.h index 6f96bcc326..4cde5abcd4 100644 --- a/src/video_core/renderer_vulkan/vk_descriptor_buffer.h +++ b/src/video_core/renderer_vulkan/vk_descriptor_buffer.h @@ -17,9 +17,9 @@ class Device; class Scheduler; class DescriptorBufferRing final { - static constexpr size_t FRAMES_IN_FLIGHT = 8; - static constexpr VkDeviceSize TILER_FRAME_SIZE = 2 * 1024 * 1024; - static constexpr VkDeviceSize DESKTOP_FRAME_SIZE = 4 * 1024 * 1024; + static constexpr size_t FRAMES_IN_FLIGHT = 3; + static constexpr VkDeviceSize TILER_FRAME_SIZE = 5 * 1024 * 1024; + static constexpr VkDeviceSize DESKTOP_FRAME_SIZE = 8 * 1024 * 1024; public: explicit DescriptorBufferRing(const Device& device_, MemoryAllocator& memory_allocator); diff --git a/src/video_core/renderer_vulkan/vk_texture_cache.h b/src/video_core/renderer_vulkan/vk_texture_cache.h index 67cf67872b..06ce5a5c32 100644 --- a/src/video_core/renderer_vulkan/vk_texture_cache.h +++ b/src/video_core/renderer_vulkan/vk_texture_cache.h @@ -552,7 +552,6 @@ private: }; struct TextureCacheParams { - static constexpr bool ENABLE_VALIDATION = true; static constexpr bool FRAMEBUFFER_BLITS = false; static constexpr bool HAS_EMULATED_COPIES = false; static constexpr bool HAS_DEVICE_MEMORY_INFO = true; diff --git a/src/video_core/texture_cache/descriptor_table.h b/src/video_core/texture_cache/descriptor_table.h index e40c128ab5..ec310803f4 100644 --- a/src/video_core/texture_cache/descriptor_table.h +++ b/src/video_core/texture_cache/descriptor_table.h @@ -7,12 +7,16 @@ #pragma once #include +#include +#include +#include #include #include "common/alignment.h" #include "common/common_types.h" #include "common/div_ceil.h" #include "common/assert.h" +#include "common/slot_vector.h" #include "video_core/memory_manager.h" #include "video_core/rasterizer_interface.h" @@ -21,6 +25,12 @@ namespace VideoCommon { template class DescriptorTable { public: + struct Entry { + T descriptor; + Common::SlotId id; + u32 generation; + }; + [[nodiscard]] bool Synchronize(GPUVAddr gpu_addr, u32 limit) noexcept { bool ret = !(current_gpu_addr == gpu_addr && current_limit == limit); if (ret) { @@ -30,44 +40,42 @@ public: } void Invalidate() noexcept { - std::ranges::fill(read_descriptors, 0); + ++generation; } - [[nodiscard]] std::pair Read(Tegra::MemoryManager const& gpu_memory, u32 index) noexcept { + [[nodiscard]] std::pair Read(Tegra::MemoryManager const& gpu_memory, u32 index) noexcept { DEBUG_ASSERT(index <= current_limit); const GPUVAddr gpu_addr = current_gpu_addr + index * sizeof(T); - std::pair result; - gpu_memory.ReadBlockUnsafe(gpu_addr, std::addressof(result.first), sizeof(T)); - if ((read_descriptors[index / 64] & (1ULL << (index % 64))) != 0) { - result.second = result.first != descriptors[index]; - } else { - read_descriptors[index / 64] |= 1ULL << (index % 64); - result.second = true; + T value{}; + if (!aligned) { + gpu_memory.ReadBlockUnsafe(gpu_addr, std::addressof(value), sizeof(T)); + } else if (const u8* const ptr = gpu_memory.GetPointer(gpu_addr)) { + std::memcpy(std::addressof(value), ptr, sizeof(T)); } - if (result.second) { - descriptors[index] = result.first; + Entry& entry = entries[index]; + const bool is_new = entry.generation != generation || entry.descriptor != value; + if (is_new) { + entry.descriptor = value; + entry.generation = generation; } - return result; + return {entry, is_new}; } void Refresh(GPUVAddr gpu_addr, u32 limit) noexcept { current_gpu_addr = gpu_addr; current_limit = limit; - // Mario Brothership reallocates a lot of times, so use aggressive pre-alloc sizes - // std::vector by default uses quadratic growth, but that isn't even enough to satisfy brothership - const size_t num_descriptors = ((limit + 0x80000) & (~0x7ffff)) + 1; - size_t old_size = read_descriptors.size(); - read_descriptors.resize(Common::DivCeil(num_descriptors, 64U)); - old_size = (std::min)(old_size, read_descriptors.size()); - std::fill(read_descriptors.begin(), read_descriptors.begin() + old_size, 0); - // - descriptors.resize(num_descriptors); + aligned = gpu_addr % sizeof(T) == 0; + ++generation; + if (entries.size() <= limit) { + entries.resize(std::bit_ceil(size_t{limit} + 1)); + } } - std::vector read_descriptors; - std::vector descriptors; + std::vector entries; GPUVAddr current_gpu_addr{}; u32 current_limit{}; + u32 generation{1}; + bool aligned{}; }; } // namespace VideoCommon diff --git a/src/video_core/texture_cache/texture_cache.h b/src/video_core/texture_cache/texture_cache.h index 4092036b9e..2b25184d9b 100644 --- a/src/video_core/texture_cache/texture_cache.h +++ b/src/video_core/texture_cache/texture_cache.h @@ -270,14 +270,11 @@ SamplerId TextureCache

::GetSamplerId(u32 index, bool compute) { LOG_DEBUG(HW_GPU, "Invalid sampler index={}", index); return NULL_SAMPLER_ID; } - auto const map_index = index | (compute ? Common::SlotId::TAGGED_VALUE : 0); - auto const [descriptor, is_new] = table.Read(*gpu_memory, index); + auto const [entry, is_new] = table.Read(*gpu_memory, index); if (is_new) { - auto const id = FindSampler(descriptor, compute); - channel_state->sampler_ids.insert_or_assign(map_index, id); - return id; + entry.id = FindSampler(entry.descriptor, compute); } - return channel_state->sampler_ids.find(map_index)->second; + return entry.id; } template @@ -500,26 +497,19 @@ ImageViewId TextureCache

::VisitImageView(u32 index, bool compute) { LOG_DEBUG(HW_GPU, "Invalid image view index={}", index); return NULL_IMAGE_VIEW_ID; } - auto const map_index = index | (compute ? Common::SlotId::TAGGED_VALUE : 0); - // Is new (on the tegra engine side)? - auto const [descriptor, is_new] = table.Read(*gpu_memory, index); + auto const [entry, is_new] = table.Read(*gpu_memory, index); if (is_new) { - if (IsValidEntry(*gpu_memory, descriptor)) { - // Is new (registered view) on the texture cache side? - const auto [pair, is_new_tc] = channel_state->image_views.try_emplace(descriptor); + entry.id = NULL_IMAGE_VIEW_ID; + if (IsValidEntry(*gpu_memory, entry.descriptor)) { + const auto [pair, is_new_tc] = channel_state->image_views.try_emplace(entry.descriptor); if (is_new_tc) - pair->second = CreateImageView(descriptor); - PrepareImageView(pair->second, false, false); - channel_state->image_view_ids.insert_or_assign(map_index, pair->second); - return pair->second; + pair->second = CreateImageView(entry.descriptor); + entry.id = pair->second; } - channel_state->image_view_ids.insert_or_assign(map_index, NULL_IMAGE_VIEW_ID); - return NULL_IMAGE_VIEW_ID; } - auto const it = channel_state->image_view_ids.find(map_index); - if (it->second != NULL_IMAGE_VIEW_ID) - PrepareImageView(it->second, false, false); - return it->second; + if (entry.id != NULL_IMAGE_VIEW_ID) + PrepareImageView(entry.id, false, false); + return entry.id; } template @@ -1296,9 +1286,6 @@ void TextureCache

::InvalidateScale(Image& image) { for (size_t c : active_channel_ids) { auto& channel_info = channel_storage[c]; - if constexpr (ENABLE_VALIDATION) - for (auto& e : channel_info.image_view_ids) - e.second = CORRUPT_ID; channel_info.graphics_image_table.Invalidate(); channel_info.compute_image_table.Invalidate(); } @@ -2295,9 +2282,6 @@ void TextureCache

::DeleteImage(ImageId image_id, bool immediate_delete) { } for (size_t c : active_channel_ids) { auto& channel_info = channel_storage[c]; - if constexpr (ENABLE_VALIDATION) - for (auto& e : channel_info.image_view_ids) - e.second = CORRUPT_ID; channel_info.graphics_image_table.Invalidate(); channel_info.compute_image_table.Invalidate(); } diff --git a/src/video_core/texture_cache/texture_cache_base.h b/src/video_core/texture_cache/texture_cache_base.h index 7a46134490..b51f03e532 100644 --- a/src/video_core/texture_cache/texture_cache_base.h +++ b/src/video_core/texture_cache/texture_cache_base.h @@ -89,9 +89,6 @@ public: std::unordered_map image_views; std::unordered_map samplers; - ::Common::unordered_map sampler_ids; - ::Common::unordered_map image_view_ids; - TextureCacheGPUMap* gpu_page_table = nullptr; TextureCacheGPUMap* sparse_page_table = nullptr; }; @@ -101,8 +98,6 @@ class TextureCache : public VideoCommon::ChannelSetupCaches