Experiments with gpu stalls per fence wait

This commit is contained in:
CamilleLaVey
2026-10-01 22:52:42 -04:00
parent 5c7520f572
commit 165a1d1da1
22 changed files with 128 additions and 66 deletions
@@ -34,6 +34,7 @@ enum class BooleanSetting(override val key: String) : AbstractBooleanSetting {
ENABLE_GPU_BUFFER_READBACK("enable_gpu_buffer_readback"),
USE_UNIFIED_MEMORY("use_unified_memory"),
SYNC_MEMORY_OPERATIONS("sync_memory_operations"),
STALL_ON_GPU_FENCE_WAIT("stall_on_gpu_fence_wait"),
BUFFER_REORDER_DISABLE("disable_buffer_reorder"),
RENDERER_DEBUG("debug"),
RENDERER_PATCH_OLD_QCOM_DRIVERS("patch_old_qcom_drivers"),
@@ -902,6 +902,13 @@ abstract class SettingsItem(
descriptionId = R.string.sync_memory_operations_description
)
)
put(
SwitchSetting(
BooleanSetting.STALL_ON_GPU_FENCE_WAIT,
titleId = R.string.stall_on_gpu_fence_wait,
descriptionId = R.string.stall_on_gpu_fence_wait_description
)
)
put(
SwitchSetting(
BooleanSetting.BUFFER_REORDER_DISABLE,
@@ -545,6 +545,7 @@ class SettingsFragmentPresenter(
add(IntSetting.RENDERER_NVDEC_EMULATION.key)
add(BooleanSetting.SYNC_MEMORY_OPERATIONS.key)
add(BooleanSetting.STALL_ON_GPU_FENCE_WAIT.key)
add(BooleanSetting.RENDERER_USE_DISK_SHADER_CACHE.key)
add(BooleanSetting.RENDERER_FORCE_MAX_CLOCK.key)
add(BooleanSetting.RENDERER_REACTIVE_FLUSHING.key)
@@ -575,6 +575,8 @@
<string name="accelerate_astc_description">Pick how ASTC-compressed textures are decoded for rendering: CPU (slow, safe), GPU (fast, recommended), or CPU Async (no stutters, may cause issues)</string>
<string name="sync_memory_operations">Sync Memory Operations</string>
<string name="stall_on_gpu_fence_wait">Pause CPU on GPU fence timeouts</string>
<string name="stall_on_gpu_fence_wait_description">When a game keeps timing out while waiting for the GPU, pauses every CPU thread until the GPU catches up. Disabling it keeps the other threads running, which can reduce stutter but may break games that rely on it.</string>
<string name="sync_memory_operations_description">Ensures data consistency between compute and memory operations. This option should fix issues in some games, but may also reduce performance in some cases. Unreal Engine 4 games often see the most significant changes thereof.</string>
<string name="use_disk_shader_cache">Disk shader cache</string>
<string name="use_disk_shader_cache_description">Reduces stuttering by locally storing and loading generated shaders.</string>
+2
View File
@@ -561,6 +561,8 @@ struct Values {
Specialization::Default,
true,
true};
SwitchableSetting<bool> stall_on_gpu_fence_wait{linkage, true, "stall_on_gpu_fence_wait",
Category::RendererAdvanced};
SwitchableSetting<bool> renderer_force_max_clock{linkage, false, "force_max_clock",
Category::RendererAdvanced};
@@ -13,6 +13,7 @@
#include "common/assert.h"
#include "common/logging.h"
#include "common/scope_exit.h"
#include "common/settings.h"
#include "core/core.h"
#include "core/hle/kernel/k_event.h"
#include "core/hle/service/nvdrv/core/container.h"
@@ -147,9 +148,12 @@ NvResult nvhost_ctrl::IocCtrlEventWait(IocCtrlEventWaitParams& params, bool is_a
const auto check_failing = [&]() {
if (events[slot].fails > 2) {
{
auto lk = system.StallApplication();
host1x_syncpoint_manager.WaitHost(fence_id, target_value);
std::unique_lock<std::mutex> stall;
if (Settings::values.stall_on_gpu_fence_wait.GetValue()) {
stall = system.StallApplication();
}
host1x_syncpoint_manager.WaitHost(fence_id, target_value);
if (stall.owns_lock()) {
system.UnstallApplication();
}
params.value.raw = target_value;
@@ -312,19 +312,17 @@ static boost::container::small_vector<Tegra::CommandHeader, 512> BuildWaitComman
static boost::container::small_vector<Tegra::CommandHeader, 512> BuildIncrementCommandList(
NvFence fence) {
boost::container::small_vector<Tegra::CommandHeader, 512> result{
const Tegra::CommandHeader increment =
BuildFenceAction(Tegra::Engines::Puller::FenceOperation::Increment, fence.id);
return {
Tegra::BuildCommandHeader(Tegra::BufferMethods::SyncpointPayload, 1,
Tegra::SubmissionMode::Increasing),
{}};
for (u32 count = 0; count < 2; ++count) {
result.push_back(Tegra::BuildCommandHeader(Tegra::BufferMethods::SyncpointOperation, 1,
Tegra::SubmissionMode::Increasing));
result.push_back(
BuildFenceAction(Tegra::Engines::Puller::FenceOperation::Increment, fence.id));
}
return result;
{},
Tegra::BuildCommandHeader(Tegra::BufferMethods::SyncpointOperation, 2,
Tegra::SubmissionMode::NonIncreasing),
increment,
increment,
};
}
static boost::container::small_vector<Tegra::CommandHeader, 512> BuildIncrementWithWfiCommandList(
@@ -370,17 +368,14 @@ NvResult nvhost_gpu::SubmitGPFIFOImpl(IoctlSubmitGpfifo& params, Tegra::CommandL
u32 increment{(flags.fence_increment.Value() != 0 ? 2 : 0) +
(flags.increment_value.Value() != 0 ? params.fence.value : 0)};
params.fence.value = syncpoint_manager.IncrementSyncpointMaxExt(channel_syncpoint, increment);
gpu.PushGPUEntries(bind_id, std::move(entries));
if (flags.fence_increment.Value()) {
if (flags.suppress_wfi.Value()) {
gpu.PushGPUEntries(bind_id,
Tegra::CommandList{BuildIncrementCommandList(params.fence)});
entries.prefetch_command_list = BuildIncrementCommandList(params.fence);
} else {
gpu.PushGPUEntries(bind_id,
Tegra::CommandList{BuildIncrementWithWfiCommandList(params.fence)});
entries.prefetch_command_list = BuildIncrementWithWfiCommandList(params.fence);
}
}
gpu.PushGPUEntries(bind_id, std::move(entries));
flags.raw = 0;
+34 -25
View File
@@ -165,13 +165,22 @@ Module::Module(Core::System& system)
Module::~Module() {}
std::shared_ptr<Devices::nvdevice> Module::FindDevice(DeviceFD fd) const {
std::scoped_lock lock(open_files_mutex);
const auto itr = open_files.find(fd);
if (itr == open_files.end()) {
return nullptr;
}
return itr->second;
}
NvResult Module::VerifyFD(DeviceFD fd) const {
if (fd < 0) {
LOG_ERROR(Service_NVDRV, "Invalid DeviceFD={}!", fd);
return NvResult::InvalidState;
}
if (open_files.find(fd) == open_files.end()) {
if (!FindDevice(fd)) {
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
return NvResult::NotImplemented;
}
@@ -187,8 +196,11 @@ DeviceFD Module::Open(const std::string& device_name, NvCore::SessionId session_
}
const DeviceFD fd = next_fd++;
auto& builder = it->second;
auto device = builder(fd)->second;
std::shared_ptr<Devices::nvdevice> device;
{
std::scoped_lock lock(open_files_mutex);
device = it->second(fd)->second;
}
device->OnOpen(session_id, fd);
@@ -202,14 +214,13 @@ NvResult Module::Ioctl1(DeviceFD fd, Ioctl command, std::span<const u8> input,
return NvResult::InvalidState;
}
const auto itr = open_files.find(fd);
if (itr == open_files.end()) {
const auto device = FindDevice(fd);
if (!device) {
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
return NvResult::NotImplemented;
}
return itr->second->Ioctl1(fd, command, input, output);
return device->Ioctl1(fd, command, input, output);
}
NvResult Module::Ioctl2(DeviceFD fd, Ioctl command, std::span<const u8> input,
@@ -219,14 +230,13 @@ NvResult Module::Ioctl2(DeviceFD fd, Ioctl command, std::span<const u8> input,
return NvResult::InvalidState;
}
const auto itr = open_files.find(fd);
if (itr == open_files.end()) {
const auto device = FindDevice(fd);
if (!device) {
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
return NvResult::NotImplemented;
}
return itr->second->Ioctl2(fd, command, input, inline_input, output);
return device->Ioctl2(fd, command, input, inline_input, output);
}
NvResult Module::Ioctl3(DeviceFD fd, Ioctl command, std::span<const u8> input, std::span<u8> output,
@@ -236,14 +246,13 @@ NvResult Module::Ioctl3(DeviceFD fd, Ioctl command, std::span<const u8> input, s
return NvResult::InvalidState;
}
const auto itr = open_files.find(fd);
if (itr == open_files.end()) {
const auto device = FindDevice(fd);
if (!device) {
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
return NvResult::NotImplemented;
}
return itr->second->Ioctl3(fd, command, input, output, inline_output);
return device->Ioctl3(fd, command, input, output, inline_output);
}
NvResult Module::Close(DeviceFD fd) {
@@ -252,16 +261,17 @@ NvResult Module::Close(DeviceFD fd) {
return NvResult::InvalidState;
}
const auto itr = open_files.find(fd);
if (itr == open_files.end()) {
const auto device = FindDevice(fd);
if (!device) {
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
return NvResult::NotImplemented;
}
itr->second->OnClose(fd);
open_files.erase(itr);
{
std::scoped_lock lock(open_files_mutex);
open_files.erase(fd);
}
device->OnClose(fd);
return NvResult::Success;
}
@@ -272,14 +282,13 @@ NvResult Module::QueryEvent(DeviceFD fd, u32 event_id, Kernel::KEvent*& event) {
return NvResult::InvalidState;
}
const auto itr = open_files.find(fd);
if (itr == open_files.end()) {
const auto device = FindDevice(fd);
if (!device) {
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
return NvResult::NotImplemented;
}
event = itr->second->QueryEvent(event_id);
event = device->QueryEvent(event_id);
if (!event) {
return NvResult::BadParameter;
}
+4 -4
View File
@@ -65,10 +65,7 @@ public:
/// Returns a pointer to one of the available devices, identified by its name.
template <typename T>
std::shared_ptr<T> GetDevice(DeviceFD fd) {
auto itr = open_files.find(fd);
if (itr == open_files.end())
return nullptr;
return std::static_pointer_cast<T>(itr->second);
return std::static_pointer_cast<T>(FindDevice(fd));
}
NvResult VerifyFD(DeviceFD fd) const;
@@ -97,6 +94,8 @@ public:
private:
friend class EventInterface;
std::shared_ptr<Devices::nvdevice> FindDevice(DeviceFD fd) const;
/// Manages syncpoints on the host
NvCore::Container container;
@@ -106,6 +105,7 @@ private:
using FilesContainerType = ::Common::unordered_map<DeviceFD, std::shared_ptr<Devices::nvdevice>>;
/// Mapping of file descriptors to the devices they reference.
FilesContainerType open_files;
mutable std::mutex open_files_mutex;
KernelHelpers::ServiceContext service_context;
@@ -222,6 +222,10 @@ std::unique_ptr<TranslationMap> InitializeTranslations(QObject* parent) {
tr("Ensures data consistency between compute and memory operations.\nThis option fixes "
"issues in games, but may degrade performance.\nUnreal Engine 4 games often see the "
"most significant changes thereof."));
INSERT(Settings, stall_on_gpu_fence_wait, tr("Pause CPU on GPU fence timeouts"),
tr("When a game keeps timing out while waiting for the GPU, pauses every CPU thread "
"until the GPU catches up.\nDisabling it keeps the other threads running, which can "
"reduce stutter but may break games that rely on it."));
INSERT(Settings, async_presentation, tr("Enable asynchronous presentation (Vulkan only)"),
tr("Slightly improves performance by moving presentation to a separate CPU thread."));
INSERT(
+2 -1
View File
@@ -56,7 +56,7 @@ bool DmaPusher::Step() {
return true;
}
if (prefetch_size > 0) {
if (command_list_size == 0) {
ProcessCommands(command_list.prefetch_command_list);
dma_pushbuffer.pop();
return true;
@@ -89,6 +89,7 @@ bool DmaPusher::Step() {
}
if (++dma_pushbuffer_subindex >= command_list_size) {
ProcessCommands(command_list.prefetch_command_list);
dma_pushbuffer.pop();
dma_pushbuffer_subindex = 0;
} else {
+1 -1
View File
@@ -588,7 +588,7 @@ void Maxwell3D::ProcessCounterReset() {
void Maxwell3D::ProcessSyncPoint() {
const u32 sync_point = regs.sync_info.sync_point.Value();
[[maybe_unused]] const u32 cache_flush = regs.sync_info.clean_l2.Value();
rasterizer->SignalSyncPoint(sync_point);
rasterizer->SignalSyncPoint(sync_point, 1);
}
void Maxwell3D::ProcessCBBind(size_t stage_index) {
+16 -1
View File
@@ -4,6 +4,8 @@
// SPDX-FileCopyrightText: 2022 yuzu Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#include <algorithm>
#include "common/assert.h"
#include "common/logging.h"
#include "common/settings.h"
@@ -71,7 +73,7 @@ void Puller::ProcessFenceActionMethod(DmaPusher& dma_pusher) {
dma_pusher.rasterizer->ReleaseFences();
break;
case Puller::FenceOperation::Increment:
dma_pusher.rasterizer->SignalSyncPoint(regs.fence_action.syncpoint_id);
dma_pusher.rasterizer->SignalSyncPoint(regs.fence_action.syncpoint_id, 1);
break;
default:
UNIMPLEMENTED_MSG("Unimplemented operation {}", regs.fence_action.op.Value());
@@ -278,6 +280,9 @@ void Puller::CallMultiMethod(DmaPusher& dma_pusher, u32 method, u32 subchannel,
ASSERT(subchannel < bound_engines.size());
if (ExecuteMethodOnEngine(dma_pusher, method)) {
CallEngineMultiMethod(dma_pusher, method, subchannel, base_start, amount, methods_pending);
} else if (IsIncrementRun(method, base_start, amount)) {
regs.reg_array[method] = base_start[0];
dma_pusher.rasterizer->SignalSyncPoint(regs.fence_action.syncpoint_id, amount);
} else {
for (u32 i = 0; i < amount; i++) {
CallPullerMethod(dma_pusher, MethodCall{
@@ -290,6 +295,16 @@ void Puller::CallMultiMethod(DmaPusher& dma_pusher, u32 method, u32 subchannel,
}
}
bool Puller::IsIncrementRun(u32 method, const u32* base_start, u32 amount) const {
if (BufferMethods(method) != BufferMethods::SyncpointOperation || amount < 2) {
return false;
}
const FenceAction action{.raw = base_start[0]};
return action.op == FenceOperation::Increment &&
std::all_of(base_start, base_start + amount,
[first = base_start[0]](u32 argument) { return argument == first; });
}
/// Determines where the method should be executed.
[[nodiscard]] bool Puller::ExecuteMethodOnEngine(DmaPusher& dma_pusher, u32 method) {
const auto buffer_method = BufferMethods(method);
+1
View File
@@ -72,6 +72,7 @@ public:
void CallMethod(DmaPusher& dma_pusher, const MethodCall& method_call);
void CallMultiMethod(DmaPusher& dma_pusher, u32 method, u32 subchannel, const u32* base_start, u32 amount, u32 methods_pending);
[[nodiscard]] bool IsIncrementRun(u32 method, const u32* base_start, u32 amount) const;
void BindRasterizer(DmaPusher& dma_pusher, VideoCore::RasterizerInterface* rasterizer);
void CallPullerMethod(DmaPusher& dma_pusher, const MethodCall& method_call);
void CallEngineMethod(DmaPusher& dma_pusher, const MethodCall& method_call);
+9 -3
View File
@@ -101,9 +101,15 @@ public:
rasterizer.InvalidateGPUCache();
}
void SignalSyncPoint(u32 value) {
syncpoint_manager.IncrementGuest(value);
std::function<void()> func([this, value] { syncpoint_manager.IncrementHost(value); });
void SignalSyncPoint(u32 value, u32 count) {
for (u32 i = 0; i < count; ++i) {
syncpoint_manager.IncrementGuest(value);
}
std::function<void()> func([this, value, count] {
for (u32 i = 0; i < count; ++i) {
syncpoint_manager.IncrementHost(value);
}
});
SignalFence(std::move(func));
}
+4 -1
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -75,7 +78,7 @@ public:
virtual void SyncOperation(std::function<void()>&& func) = 0;
/// Signal a GPU based syncpoint as a fence
virtual void SignalSyncPoint(u32 value) = 0;
virtual void SignalSyncPoint(u32 value, u32 count) = 0;
/// Signal a GPU based reference as point
virtual void SignalReference() = 0;
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -69,10 +72,12 @@ void RasterizerNull::SignalFence(std::function<void()>&& func) {
void RasterizerNull::SyncOperation(std::function<void()>&& func) {
func();
}
void RasterizerNull::SignalSyncPoint(u32 value) {
void RasterizerNull::SignalSyncPoint(u32 value, u32 count) {
auto& syncpoint_manager = m_gpu.Host1x().GetSyncpointManager();
syncpoint_manager.IncrementGuest(value);
syncpoint_manager.IncrementHost(value);
for (u32 i = 0; i < count; ++i) {
syncpoint_manager.IncrementGuest(value);
syncpoint_manager.IncrementHost(value);
}
}
void RasterizerNull::SignalReference() {}
void RasterizerNull::ReleaseFences(bool) {}
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -61,7 +64,7 @@ public:
void ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) override;
void SignalFence(std::function<void()>&& func) override;
void SyncOperation(std::function<void()>&& func) override;
void SignalSyncPoint(u32 value) override;
void SignalSyncPoint(u32 value, u32 count) override;
void SignalReference() override;
void ReleaseFences(bool force) override;
void FlushAndInvalidateRegion(
@@ -615,8 +615,8 @@ void RasterizerOpenGL::SyncOperation(std::function<void()>&& func) {
fence_manager.SyncOperation(std::move(func));
}
void RasterizerOpenGL::SignalSyncPoint(u32 value) {
fence_manager.SignalSyncPoint(value);
void RasterizerOpenGL::SignalSyncPoint(u32 value, u32 count) {
fence_manager.SignalSyncPoint(value, count);
}
void RasterizerOpenGL::SignalReference() {
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: 2015 Citra Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -106,7 +109,7 @@ public:
void ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) override;
void SignalFence(std::function<void()>&& func) override;
void SyncOperation(std::function<void()>&& func) override;
void SignalSyncPoint(u32 value) override;
void SignalSyncPoint(u32 value, u32 count) override;
void SignalReference() override;
void ReleaseFences(bool force = true) override;
void FlushAndInvalidateRegion(
@@ -862,8 +862,8 @@ void RasterizerVulkan::SyncOperation(std::function<void()>&& func) {
fence_manager.SyncOperation(std::move(func));
}
void RasterizerVulkan::SignalSyncPoint(u32 value) {
fence_manager.SignalSyncPoint(value);
void RasterizerVulkan::SignalSyncPoint(u32 value, u32 count) {
fence_manager.SignalSyncPoint(value, count);
}
void RasterizerVulkan::SignalReference() {
@@ -111,7 +111,7 @@ public:
void ModifyGPUMemory(size_t as_id, GPUVAddr addr, u64 size) override;
void SignalFence(std::function<void()>&& func) override;
void SyncOperation(std::function<void()>&& func) override;
void SignalSyncPoint(u32 value) override;
void SignalSyncPoint(u32 value, u32 count) override;
void SignalReference() override;
void ReleaseFences(bool force = true) override;
void FlushAndInvalidateRegion(