Experiment on descriptors payloads

This commit is contained in:
CamilleLaVey
2026-09-24 17:57:47 -04:00
parent 98ad2d5e4c
commit 108c991431
7 changed files with 47 additions and 62 deletions
@@ -33,7 +33,7 @@ using TextureInstVector = boost::container::small_vector<TextureInst, 24>;
constexpr u32 DESCRIPTOR_SIZE = 8;
constexpr u32 DESCRIPTOR_SIZE_SHIFT = u32(std::countr_zero(DESCRIPTOR_SIZE));
constexpr u32 DESCRIPTOR_MAX_COUNT = 512;
constexpr u32 DESCRIPTOR_MAX_COUNT = 128;
constexpr u32 DESCRIPTOR_CBUF_BYTES = 16 * 1024;
u32 DynamicDescriptorSizeShift(const IR::U32& dynamic_offset) {
@@ -364,7 +364,6 @@ private:
};
struct TextureCacheParams {
static constexpr bool ENABLE_VALIDATION = true;
static constexpr bool FRAMEBUFFER_BLITS = true;
static constexpr bool HAS_EMULATED_COPIES = true;
static constexpr bool HAS_DEVICE_MEMORY_INFO = true;
@@ -17,9 +17,9 @@ class Device;
class Scheduler;
class DescriptorBufferRing final {
static constexpr size_t FRAMES_IN_FLIGHT = 8;
static constexpr VkDeviceSize TILER_FRAME_SIZE = 2 * 1024 * 1024;
static constexpr VkDeviceSize DESKTOP_FRAME_SIZE = 4 * 1024 * 1024;
static constexpr size_t FRAMES_IN_FLIGHT = 3;
static constexpr VkDeviceSize TILER_FRAME_SIZE = 5 * 1024 * 1024;
static constexpr VkDeviceSize DESKTOP_FRAME_SIZE = 8 * 1024 * 1024;
public:
explicit DescriptorBufferRing(const Device& device_, MemoryAllocator& memory_allocator);
@@ -552,7 +552,6 @@ private:
};
struct TextureCacheParams {
static constexpr bool ENABLE_VALIDATION = true;
static constexpr bool FRAMEBUFFER_BLITS = false;
static constexpr bool HAS_EMULATED_COPIES = false;
static constexpr bool HAS_DEVICE_MEMORY_INFO = true;
+31 -23
View File
@@ -7,12 +7,16 @@
#pragma once
#include <algorithm>
#include <bit>
#include <cstring>
#include <utility>
#include <vector>
#include "common/alignment.h"
#include "common/common_types.h"
#include "common/div_ceil.h"
#include "common/assert.h"
#include "common/slot_vector.h"
#include "video_core/memory_manager.h"
#include "video_core/rasterizer_interface.h"
@@ -21,6 +25,12 @@ namespace VideoCommon {
template <typename T>
class DescriptorTable {
public:
struct Entry {
T descriptor;
Common::SlotId id;
u32 generation;
};
[[nodiscard]] bool Synchronize(GPUVAddr gpu_addr, u32 limit) noexcept {
bool ret = !(current_gpu_addr == gpu_addr && current_limit == limit);
if (ret) {
@@ -30,44 +40,42 @@ public:
}
void Invalidate() noexcept {
std::ranges::fill(read_descriptors, 0);
++generation;
}
[[nodiscard]] std::pair<T, bool> Read(Tegra::MemoryManager const& gpu_memory, u32 index) noexcept {
[[nodiscard]] std::pair<Entry&, bool> Read(Tegra::MemoryManager const& gpu_memory, u32 index) noexcept {
DEBUG_ASSERT(index <= current_limit);
const GPUVAddr gpu_addr = current_gpu_addr + index * sizeof(T);
std::pair<T, bool> result;
gpu_memory.ReadBlockUnsafe(gpu_addr, std::addressof(result.first), sizeof(T));
if ((read_descriptors[index / 64] & (1ULL << (index % 64))) != 0) {
result.second = result.first != descriptors[index];
} else {
read_descriptors[index / 64] |= 1ULL << (index % 64);
result.second = true;
T value{};
if (!aligned) {
gpu_memory.ReadBlockUnsafe(gpu_addr, std::addressof(value), sizeof(T));
} else if (const u8* const ptr = gpu_memory.GetPointer(gpu_addr)) {
std::memcpy(std::addressof(value), ptr, sizeof(T));
}
if (result.second) {
descriptors[index] = result.first;
Entry& entry = entries[index];
const bool is_new = entry.generation != generation || entry.descriptor != value;
if (is_new) {
entry.descriptor = value;
entry.generation = generation;
}
return result;
return {entry, is_new};
}
void Refresh(GPUVAddr gpu_addr, u32 limit) noexcept {
current_gpu_addr = gpu_addr;
current_limit = limit;
// Mario Brothership reallocates a lot of times, so use aggressive pre-alloc sizes
// std::vector<T> by default uses quadratic growth, but that isn't even enough to satisfy brothership
const size_t num_descriptors = ((limit + 0x80000) & (~0x7ffff)) + 1;
size_t old_size = read_descriptors.size();
read_descriptors.resize(Common::DivCeil(num_descriptors, 64U));
old_size = (std::min)(old_size, read_descriptors.size());
std::fill(read_descriptors.begin(), read_descriptors.begin() + old_size, 0);
//
descriptors.resize(num_descriptors);
aligned = gpu_addr % sizeof(T) == 0;
++generation;
if (entries.size() <= limit) {
entries.resize(std::bit_ceil(size_t{limit} + 1));
}
}
std::vector<u64> read_descriptors;
std::vector<T> descriptors;
std::vector<Entry> entries;
GPUVAddr current_gpu_addr{};
u32 current_limit{};
u32 generation{1};
bool aligned{};
};
} // namespace VideoCommon
+12 -28
View File
@@ -270,14 +270,11 @@ SamplerId TextureCache<P>::GetSamplerId(u32 index, bool compute) {
LOG_DEBUG(HW_GPU, "Invalid sampler index={}", index);
return NULL_SAMPLER_ID;
}
auto const map_index = index | (compute ? Common::SlotId::TAGGED_VALUE : 0);
auto const [descriptor, is_new] = table.Read(*gpu_memory, index);
auto const [entry, is_new] = table.Read(*gpu_memory, index);
if (is_new) {
auto const id = FindSampler(descriptor, compute);
channel_state->sampler_ids.insert_or_assign(map_index, id);
return id;
entry.id = FindSampler(entry.descriptor, compute);
}
return channel_state->sampler_ids.find(map_index)->second;
return entry.id;
}
template <class P>
@@ -500,26 +497,19 @@ ImageViewId TextureCache<P>::VisitImageView(u32 index, bool compute) {
LOG_DEBUG(HW_GPU, "Invalid image view index={}", index);
return NULL_IMAGE_VIEW_ID;
}
auto const map_index = index | (compute ? Common::SlotId::TAGGED_VALUE : 0);
// Is new (on the tegra engine side)?
auto const [descriptor, is_new] = table.Read(*gpu_memory, index);
auto const [entry, is_new] = table.Read(*gpu_memory, index);
if (is_new) {
if (IsValidEntry(*gpu_memory, descriptor)) {
// Is new (registered view) on the texture cache side?
const auto [pair, is_new_tc] = channel_state->image_views.try_emplace(descriptor);
entry.id = NULL_IMAGE_VIEW_ID;
if (IsValidEntry(*gpu_memory, entry.descriptor)) {
const auto [pair, is_new_tc] = channel_state->image_views.try_emplace(entry.descriptor);
if (is_new_tc)
pair->second = CreateImageView(descriptor);
PrepareImageView(pair->second, false, false);
channel_state->image_view_ids.insert_or_assign(map_index, pair->second);
return pair->second;
pair->second = CreateImageView(entry.descriptor);
entry.id = pair->second;
}
channel_state->image_view_ids.insert_or_assign(map_index, NULL_IMAGE_VIEW_ID);
return NULL_IMAGE_VIEW_ID;
}
auto const it = channel_state->image_view_ids.find(map_index);
if (it->second != NULL_IMAGE_VIEW_ID)
PrepareImageView(it->second, false, false);
return it->second;
if (entry.id != NULL_IMAGE_VIEW_ID)
PrepareImageView(entry.id, false, false);
return entry.id;
}
template <class P>
@@ -1296,9 +1286,6 @@ void TextureCache<P>::InvalidateScale(Image& image) {
for (size_t c : active_channel_ids) {
auto& channel_info = channel_storage[c];
if constexpr (ENABLE_VALIDATION)
for (auto& e : channel_info.image_view_ids)
e.second = CORRUPT_ID;
channel_info.graphics_image_table.Invalidate();
channel_info.compute_image_table.Invalidate();
}
@@ -2295,9 +2282,6 @@ void TextureCache<P>::DeleteImage(ImageId image_id, bool immediate_delete) {
}
for (size_t c : active_channel_ids) {
auto& channel_info = channel_storage[c];
if constexpr (ENABLE_VALIDATION)
for (auto& e : channel_info.image_view_ids)
e.second = CORRUPT_ID;
channel_info.graphics_image_table.Invalidate();
channel_info.compute_image_table.Invalidate();
}
@@ -89,9 +89,6 @@ public:
std::unordered_map<TICEntry, ImageViewId> image_views;
std::unordered_map<TSCEntry, SamplerId> samplers;
::Common::unordered_map<u32, SamplerId> sampler_ids;
::Common::unordered_map<u32, ImageViewId> image_view_ids;
TextureCacheGPUMap* gpu_page_table = nullptr;
TextureCacheGPUMap* sparse_page_table = nullptr;
};
@@ -101,8 +98,6 @@ class TextureCache : public VideoCommon::ChannelSetupCaches<TextureCacheChannelI
/// Address shift for caching images into a hash table
static constexpr u64 YUZU_PAGEBITS = 20;
/// Enables debugging features to the texture cache
static constexpr bool ENABLE_VALIDATION = P::ENABLE_VALIDATION;
/// Implement blits as copies between framebuffers
static constexpr bool FRAMEBUFFER_BLITS = P::FRAMEBUFFER_BLITS;
/// True when some copies have to be emulated