Adjust buffer stream size + reduce descriptor loads

This commit is contained in:
CamilleLaVey
2026-09-22 16:27:47 -04:00
parent 3ab215fd30
commit fa9e069907
10 changed files with 7 additions and 27 deletions
@@ -33,8 +33,8 @@ BlitScreen::BlitScreen(Tegra::MaxwellDeviceMemoryManager& device_memory_, const
BlitScreen::~BlitScreen() = default; BlitScreen::~BlitScreen() = default;
void BlitScreen::WaitIdle(const Device& device) { void BlitScreen::WaitIdle(const Device& device) {
present_manager.WaitPresent();
scheduler.Finish(); scheduler.Finish();
present_manager.WaitPresent();
device.GetLogical().WaitIdle(); device.GetLogical().WaitIdle();
} }
@@ -125,7 +125,6 @@ vk::DescriptorSets DescriptorAllocator::AllocateDescriptors(size_t count) {
throw vk::Exception(VK_ERROR_OUT_OF_POOL_MEMORY); throw vk::Exception(VK_ERROR_OUT_OF_POOL_MEMORY);
} }
DescriptorPool::DescriptorPool(const Device& device_, Scheduler& scheduler) {}
DescriptorPool::~DescriptorPool() = default; DescriptorPool::~DescriptorPool() = default;
DescriptorAllocator DescriptorPool::Allocator(const Device& device, Scheduler& scheduler, VkDescriptorSetLayout layout, std::span<const Shader::Info> infos) { DescriptorAllocator DescriptorPool::Allocator(const Device& device, Scheduler& scheduler, VkDescriptorSetLayout layout, std::span<const Shader::Info> infos) {
@@ -65,7 +65,7 @@ private:
class DescriptorPool { class DescriptorPool {
public: public:
explicit DescriptorPool(const Device& device, Scheduler& scheduler); DescriptorPool() = default;
~DescriptorPool(); ~DescriptorPool();
DescriptorPool& operator=(const DescriptorPool&) = delete; DescriptorPool& operator=(const DescriptorPool&) = delete;
@@ -353,7 +353,7 @@ void MasterSemaphore::WaitThread(std::stop_token token) {
free_queue.push_front(std::move(fence)); free_queue.push_front(std::move(fence));
gpu_tick.store(host_tick, std::memory_order_release); gpu_tick.store(host_tick, std::memory_order_release);
} }
gpu_tick.notify_one(); gpu_tick.notify_all();
} }
} }
@@ -206,7 +206,7 @@ RasterizerVulkan::RasterizerVulkan(Core::Frontend::EmuWindow& emu_window_, Tegra
StateTracker& state_tracker_, Scheduler& scheduler_) StateTracker& state_tracker_, Scheduler& scheduler_)
: gpu{gpu_}, device_memory{device_memory_}, device{device_}, : gpu{gpu_}, device_memory{device_memory_}, device{device_},
memory_allocator{memory_allocator_}, state_tracker{state_tracker_}, scheduler{scheduler_}, memory_allocator{memory_allocator_}, state_tracker{state_tracker_}, scheduler{scheduler_},
staging_pool(device, memory_allocator, scheduler), descriptor_pool(device, scheduler), staging_pool(device, memory_allocator, scheduler),
guest_descriptor_queue(device, UpdateDescriptorQueue::GUEST_FRAME_PAYLOAD_SIZE, guest_descriptor_queue(device, UpdateDescriptorQueue::GUEST_FRAME_PAYLOAD_SIZE,
device.IsExtDescriptorBufferSupported()), device.IsExtDescriptorBufferSupported()),
compute_pass_descriptor_queue(device, UpdateDescriptorQueue::COMPUTE_FRAME_PAYLOAD_SIZE), compute_pass_descriptor_queue(device, UpdateDescriptorQueue::COMPUTE_FRAME_PAYLOAD_SIZE),
@@ -56,9 +56,7 @@ Scheduler::~Scheduler() = default;
u64 Scheduler::Flush(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore) { u64 Scheduler::Flush(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore) {
// When flushing, we only send data to the worker thread; no waiting is necessary. // When flushing, we only send data to the worker thread; no waiting is necessary.
const u64 signal_value = SubmitExecution(signal_semaphore, wait_semaphore); return SubmitExecution(signal_semaphore, wait_semaphore);
AllocateNewContext();
return signal_value;
} }
void Scheduler::Finish(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore) { void Scheduler::Finish(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore) {
@@ -66,7 +64,6 @@ void Scheduler::Finish(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore)
const u64 presubmit_tick = CurrentTick(); const u64 presubmit_tick = CurrentTick();
SubmitExecution(signal_semaphore, wait_semaphore); SubmitExecution(signal_semaphore, wait_semaphore);
Wait(presubmit_tick); Wait(presubmit_tick);
AllocateNewContext();
} }
void Scheduler::WaitWorker() { void Scheduler::WaitWorker() {
@@ -384,10 +381,6 @@ u64 Scheduler::SubmitExecution(VkSemaphore signal_semaphore, VkSemaphore wait_se
return signal_value; return signal_value;
} }
void Scheduler::AllocateNewContext() {
// Enable counters once again. These are disabled when a command buffer is finished.
}
void Scheduler::InvalidateState() { void Scheduler::InvalidateState() {
state.graphics_pipeline = nullptr; state.graphics_pipeline = nullptr;
state.rescaling_defined = false; state.rescaling_defined = false;
@@ -287,8 +287,6 @@ private:
u64 SubmitExecution(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore); u64 SubmitExecution(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore);
void AllocateNewContext();
void EndPendingOperations(); void EndPendingOperations();
void EndRenderPass(); void EndRenderPass();
@@ -28,16 +28,7 @@ using namespace Common::Literals;
// Maximum potential alignment of a Vulkan buffer // Maximum potential alignment of a Vulkan buffer
constexpr VkDeviceSize MAX_ALIGNMENT = 256; constexpr VkDeviceSize MAX_ALIGNMENT = 256;
// Stream buffer size in bytes
// *NIX drivers are more sensitive to increased buffers for streaming.
// Windows ones however, can intake bigger buffers and generally do not OOM.
// - GTX 960 on Windows will not OOM with 256mib
// - GT 1030 on ^NIX will OOM with 256mib
#if defined(__FreeBSD__)
constexpr VkDeviceSize MAX_STREAM_BUFFER_SIZE = 128_MiB; constexpr VkDeviceSize MAX_STREAM_BUFFER_SIZE = 128_MiB;
#else
constexpr VkDeviceSize MAX_STREAM_BUFFER_SIZE = 256_MiB;
#endif
size_t GetStreamBufferSize(const Device& device) { size_t GetStreamBufferSize(const Device& device) {
if (!device.HasDebuggingToolAttached()) { if (!device.HasDebuggingToolAttached()) {
@@ -40,8 +40,8 @@ class UpdateDescriptorQueue final {
static constexpr size_t FRAMES_IN_FLIGHT = 8; static constexpr size_t FRAMES_IN_FLIGHT = 8;
public: public:
static constexpr size_t GUEST_FRAME_PAYLOAD_SIZE = 0x80000; static constexpr size_t GUEST_FRAME_PAYLOAD_SIZE = 0x20000;
static constexpr size_t COMPUTE_FRAME_PAYLOAD_SIZE = 0x20000; static constexpr size_t COMPUTE_FRAME_PAYLOAD_SIZE = 0x8000;
explicit UpdateDescriptorQueue(const Device& device_, size_t frame_payload_size_, explicit UpdateDescriptorQueue(const Device& device_, size_t frame_payload_size_,
bool supports_descriptor_buffer_ = false); bool supports_descriptor_buffer_ = false);
@@ -1011,7 +1011,6 @@ FN_MAX_LIMIT_LIST
return features2.features.multiViewport; return features2.features.multiViewport;
} }
/// Returns true if the device supports VK_KHR_maintenance5.
/// Returns true if the device supports VK_KHR_maintenance4. /// Returns true if the device supports VK_KHR_maintenance4.
bool IsKhrMaintenance4Supported() const { bool IsKhrMaintenance4Supported() const {
return extensions.maintenance4; return extensions.maintenance4;